[llvm-branch-commits] [flang] [flang] Lower loops whose branching is confined to their body structurally (PR #225758)
Kareem Ergawy via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Mon Sep 28 09:59:59 PDT 2026
https://github.com/ergawy updated https://github.com/llvm/llvm-project/pull/225758
>From 31bc0984e4cc18d32022a38afe4cb76e10e6f5e3 Mon Sep 17 00:00:00 2001
From: ergawy <kareem.ergawy at gmail.com>
Date: Mon, 21 Sep 2026 06:55:52 -0700
Subject: [PATCH 1/2] [flang] Lower loops whose branching is confined to their
body structurally
Such a loop was classified separately by a previous change but still
lowered as unstructured, so its structured form was lost.
Lower it structurally instead, with its body folded into a region that can
hold the branching. The loop keeps its bounds on the op, so it remains
available to whatever transforms or parallelizes it. Only the body is
folded: the loop control statements are emitted as they are for any
structured loop, since a branch from outside may target either of them.
Loops an OpenACC or OpenMP directive owns are lowered the same way, so
they keep their form too.
---
flang/include/flang/Lower/AbstractConverter.h | 8 +
flang/include/flang/Lower/PFTBuilder.h | 16 +-
flang/lib/Lower/Bridge.cpp | 130 +++++++++++++--
flang/lib/Lower/OpenMP/OpenMP.cpp | 8 +
flang/lib/Lower/PFTBuilder.cpp | 4 +
flang/test/Lower/OpenACC/acc-cache.f90 | 19 +--
.../OpenACC/acc-unstructured-internals.f90 | 73 +++++++++
flang/test/Lower/OpenACC/acc-unstructured.f90 | 17 +-
.../Todo/metadirective-loop-unstructured.f90 | 39 -----
.../metadirective-loop-unstructured.f90 | 50 ++++++
.../OpenMP/wsloop-unstructured-cycle.f90 | 62 +++----
.../OpenMP/wsloop-unstructured-internals.f90 | 44 +++++
flang/test/Lower/always-execute-loop-body.f90 | 8 +-
.../Lower/do-loop-branch-to-loop-header.f90 | 55 +++++++
flang/test/Lower/do_loop_unstructured.f90 | 154 +++---------------
flang/test/Lower/select-case-statement.f90 | 7 +-
16 files changed, 453 insertions(+), 241 deletions(-)
create mode 100644 flang/test/Lower/OpenACC/acc-unstructured-internals.f90
delete mode 100644 flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
create mode 100644 flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
create mode 100644 flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
create mode 100644 flang/test/Lower/do-loop-branch-to-loop-header.f90
diff --git a/flang/include/flang/Lower/AbstractConverter.h b/flang/include/flang/Lower/AbstractConverter.h
index ae246d3188bd8..40d94dc9f5bfd 100644
--- a/flang/include/flang/Lower/AbstractConverter.h
+++ b/flang/include/flang/Lower/AbstractConverter.h
@@ -384,6 +384,14 @@ class AbstractConverter {
virtual void genEval(pft::Evaluation &eval,
bool unstructuredContext = true) = 0;
+ /// Emit \p loopEval's evaluations, folding the body into an
+ /// scf.execute_region when the loop's branching is confined to that body.
+ /// The loop control statements are emitted outside any wrap, exactly as they
+ /// are for a structured loop. Used by directive lowering, which consumes the
+ /// DO itself and so never reaches genFIR(DoConstruct), where a plain loop's
+ /// body is wrapped.
+ virtual void genLoopBodyEvaluations(pft::Evaluation &loopEval) = 0;
+
/// Return options controlling lowering behavior.
const Fortran::lower::LoweringOptions &getLoweringOptions() const {
return loweringOptions;
diff --git a/flang/include/flang/Lower/PFTBuilder.h b/flang/include/flang/Lower/PFTBuilder.h
index 6bf9dcb76f9e7..0f3e3d44f88f9 100644
--- a/flang/include/flang/Lower/PFTBuilder.h
+++ b/flang/include/flang/Lower/PFTBuilder.h
@@ -356,17 +356,21 @@ struct Evaluation : EvaluationVariant {
controlFlow = std::min(controlFlow, kind);
}
- /// True when control flow is not fully structured, category (c) included.
- /// Existing consumers ask this to decide whether raw blocks are needed, and
- /// a category (c) construct still needs them until its body is wrapped, so
- /// it must answer true here. Use hasUnstructuredInternals() to single out
- /// category (c) itself.
- bool isUnstructured() const { return controlFlow != ControlFlow::Structured; }
+ /// True only for fully unstructured control flow, which is lowered as raw
+ /// CFG blocks.
+ bool isUnstructured() const {
+ return controlFlow == ControlFlow::Unstructured;
+ }
bool hasUnstructuredInternals() const {
return controlFlow == ControlFlow::StructuredWithUnstructuredInternals;
}
+ /// True when this construct is lowered structurally, yet its body holds
+ /// unstructured control flow that has to be folded into an
+ /// scf.execute_region.
+ bool lowerBodyAsWrappedRegion() const;
+
bool lowerAsStructured() const;
bool lowerAsUnstructured() const;
bool forceAsUnstructured() const;
diff --git a/flang/lib/Lower/Bridge.cpp b/flang/lib/Lower/Bridge.cpp
index c7558806ee241..8bd073a479e19 100644
--- a/flang/lib/Lower/Bridge.cpp
+++ b/flang/lib/Lower/Bridge.cpp
@@ -1169,6 +1169,11 @@ class FirConverter : public Fortran::lower::AbstractConverter {
genFIR(eval, unstructuredContext);
}
+ void genLoopBodyEvaluations(
+ Fortran::lower::pft::Evaluation &loopEval) override final {
+ genLoopBodyEvaluations(loopEval, /*unstructuredContext=*/true);
+ }
+
//===--------------------------------------------------------------------===//
// Utility methods
//===--------------------------------------------------------------------===//
@@ -2686,6 +2691,97 @@ class FirConverter : public Fortran::lower::AbstractConverter {
return wrapOp;
}
+ /// Wrap a loop's *body* -- not the construct, and not the loop control -- in
+ /// an scf.execute_region. A category (c) loop keeps its structured loop op
+ /// while its self-contained raw branching lives inside the region, which may
+ /// hold as many blocks as it needs.
+ ///
+ /// The builder must already be positioned inside the loop body. On return it
+ /// is inside the region. Returns null when the loop has no such internals.
+ mlir::scf::ExecuteRegionOp
+ wrapUnstructuredBody(Fortran::lower::pft::Evaluation &eval,
+ mlir::Block *&yieldBlock,
+ mlir::Block *&savedEndDoBlock) {
+ if (!eval.lowerBodyAsWrappedRegion())
+ return nullptr;
+
+ Fortran::lower::pft::EvaluationList &list = eval.getNestedEvaluations();
+ mlir::Location loc = toLocation();
+ auto wrapOp =
+ mlir::scf::ExecuteRegionOp::create(*builder, loc, mlir::TypeRange{},
+ /*noInline=*/builder->getUnitAttr());
+ ++wrapUnstructuredCount;
+ mlir::Block *entry = builder->createBlock(&wrapOp.getRegion());
+ builder->setInsertionPointToEnd(entry);
+ // Only the body: both loop control statements live in the enclosing region
+ // and may be branched to from outside the loop, so neither may take its
+ // block from this region. Dropping only the EndDoStmt leaves the DO
+ // statement to be given a block here, which any GOTO targeting the loop
+ // head would then reference across a region boundary.
+ createEmptyBlocksIn(
+ llvm::make_range(std::next(list.begin()), std::prev(list.end())));
+ yieldBlock = builder->createBlock(&wrapOp.getRegion());
+ builder->setInsertionPointToEnd(yieldBlock);
+ mlir::scf::YieldOp::create(*builder, loc);
+
+ // A CYCLE targets the EndDoStmt, which is the boundary between the loop
+ // body and the loop control. Inside the wrap that boundary is the region's
+ // yield block, so a CYCLE leaves the region rather than branching to a
+ // block the region cannot name.
+ savedEndDoBlock = list.back().block;
+ list.back().block = yieldBlock;
+
+ builder->setInsertionPointToEnd(entry);
+ return wrapOp;
+ }
+
+ /// Emit a loop's evaluations, optionally folding the body into an
+ /// scf.execute_region. The loop control statements -- the first and last
+ /// evaluations -- are emitted exactly as they are for a structured loop:
+ /// same context flag, same region, same blocks. They may be branch targets
+ /// from outside the loop, so their blocks have to stay in the enclosing
+ /// region. Only what lies strictly between them goes inside the wrap.
+ void genLoopBodyEvaluations(Fortran::lower::pft::Evaluation &eval,
+ bool unstructuredContext) {
+ Fortran::lower::pft::EvaluationList &list = eval.getNestedEvaluations();
+ auto iter = list.begin();
+ auto end = std::prev(list.end());
+
+ // The loop control statement, outside any wrap.
+ if (iter != end) {
+ genFIR(*iter, unstructuredContext);
+ ++iter;
+ }
+
+ mlir::Block *yieldBlock = nullptr;
+ mlir::Block *savedEndDoBlock = nullptr;
+ mlir::scf::ExecuteRegionOp wrapOp =
+ wrapUnstructuredBody(eval, yieldBlock, savedEndDoBlock);
+ for (; iter != end; ++iter)
+ genFIR(*iter, unstructuredContext || wrapOp);
+ closeUnstructuredBodyWrap(wrapOp, eval, yieldBlock, savedEndDoBlock);
+ }
+
+ /// Finalize a wrap created by wrapUnstructuredBody: restore the EndDoStmt
+ /// block, fall through to the region's yield, and resume after the wrap op
+ /// so the caller emits the loop control outside it.
+ void closeUnstructuredBodyWrap(mlir::scf::ExecuteRegionOp wrapOp,
+ Fortran::lower::pft::Evaluation &eval,
+ mlir::Block *yieldBlock,
+ mlir::Block *savedEndDoBlock) {
+ if (!wrapOp)
+ return;
+
+ eval.getNestedEvaluations().back().block = savedEndDoBlock;
+
+ if (mlir::Block *current = builder->getBlock())
+ if (current->empty() ||
+ !current->back().hasTrait<mlir::OpTrait::IsTerminator>())
+ genBranch(yieldBlock);
+
+ builder->setInsertionPointAfter(wrapOp);
+ }
+
/// Finalize a wrap created by wrapUnstructuredConstruct: restore the
/// original exit block and set the insertion point after the wrap op.
void closeUnstructuredWrap(mlir::scf::ExecuteRegionOp wrapOp,
@@ -2713,9 +2809,10 @@ class FirConverter : public Fortran::lower::AbstractConverter {
// skip generating any loop — just lower the body. The IV value is
// already available from the parent acc.loop's block argument.
if (Fortran::lower::isCollapsedDoConstruct(doConstruct)) {
- auto iter = eval.getNestedEvaluations().begin();
- for (auto end = --eval.getNestedEvaluations().end(); iter != end; ++iter)
- genFIR(*iter, unstructuredContext);
+ // The parent acc.loop supplies the iteration, so no loop op is built
+ // here for a wrap to sit inside. The body may still hold self-contained
+ // raw branching, so wrap it directly.
+ genLoopBodyEvaluations(eval, unstructuredContext);
return;
}
@@ -2738,11 +2835,10 @@ class FirConverter : public Fortran::lower::AbstractConverter {
builder->getInsertionPoint()->getBlock()->getParent()) &&
"builder insertion point is not inside the newly generated loop");
- // Loop body code.
- auto iter = eval.getNestedEvaluations().begin();
- for (auto end = --eval.getNestedEvaluations().end(); iter != end;
- ++iter)
- genFIR(*iter, unstructuredContext);
+ // The acc.loop supplies the iteration, so the body is emitted here
+ // rather than through the increment-loop path below. Wrap it when it
+ // holds self-contained raw branching.
+ genLoopBodyEvaluations(eval, unstructuredContext);
builder->setInsertionPointAfter(loopOp);
return;
@@ -2891,10 +2987,12 @@ class FirConverter : public Fortran::lower::AbstractConverter {
if (!infiniteLoop && !whileCondition)
genFIRIncrementLoopBegin(incrementLoopNestInfo, doStmtEval.dirs);
+ // The loop control is structured, but the body may hold raw branching
+ // confined to it. Wrap the body, leaving the loop control outside, so the
+ // structured loop op's single-block region stays well formed.
// Loop body code.
- auto iter = eval.getNestedEvaluations().begin();
- for (auto end = --eval.getNestedEvaluations().end(); iter != end; ++iter)
- genFIR(*iter, unstructuredContext);
+ genLoopBodyEvaluations(eval, unstructuredContext);
+ auto iter = std::prev(eval.getNestedEvaluations().end());
// An EndDoStmt in unstructured code may start a new block.
Fortran::lower::pft::Evaluation &endDoEval = *iter;
@@ -6519,6 +6617,16 @@ class FirConverter : public Fortran::lower::AbstractConverter {
/// boundaries.
void createEmptyBlocks(
std::list<Fortran::lower::pft::Evaluation> &evaluationList) {
+ createEmptyBlocksIn(
+ llvm::make_range(evaluationList.begin(), evaluationList.end()));
+ }
+
+ /// createEmptyBlocks over a sub-range of an evaluation list. A body-only wrap
+ /// must not pre-create blocks for the loop control statements, which live in
+ /// the enclosing region.
+ void createEmptyBlocksIn(
+ llvm::iterator_range<Fortran::lower::pft::EvaluationList::iterator>
+ evaluationList) {
mlir::Region *region = &builder->getRegion();
for (Fortran::lower::pft::Evaluation &eval : evaluationList) {
if (eval.isNewBlock)
diff --git a/flang/lib/Lower/OpenMP/OpenMP.cpp b/flang/lib/Lower/OpenMP/OpenMP.cpp
index 48450e11fc20a..99a7882a658c8 100644
--- a/flang/lib/Lower/OpenMP/OpenMP.cpp
+++ b/flang/lib/Lower/OpenMP/OpenMP.cpp
@@ -1018,6 +1018,14 @@ static void genNestedEvaluations(lower::AbstractConverter &converter,
int collapseValue = 0) {
lower::pft::Evaluation *curEval = getCollapsedLoopEval(eval, collapseValue);
+ // The directive consumes the DO itself, so genFIR(DoConstruct) -- where a
+ // plain loop's body is wrapped -- is never reached. Go through the same
+ // helper here, so the loop control statements are emitted outside the wrap
+ // exactly as they are for a structured loop.
+ if (curEval->isA<parser::DoConstruct>() && curEval->hasNestedEvaluations()) {
+ converter.genLoopBodyEvaluations(*curEval);
+ return;
+ }
for (lower::pft::Evaluation &e : curEval->getNestedEvaluations())
converter.genEval(e);
}
diff --git a/flang/lib/Lower/PFTBuilder.cpp b/flang/lib/Lower/PFTBuilder.cpp
index 00a5495bc9753..51b8288733ea8 100644
--- a/flang/lib/Lower/PFTBuilder.cpp
+++ b/flang/lib/Lower/PFTBuilder.cpp
@@ -1735,6 +1735,10 @@ bool Fortran::lower::pft::Evaluation::lowerAsUnstructured() const {
return isUnstructured() || clDisableStructuredFir;
}
+bool Fortran::lower::pft::Evaluation::lowerBodyAsWrappedRegion() const {
+ return hasUnstructuredInternals() && !lowerAsUnstructured();
+}
+
bool Fortran::lower::pft::Evaluation::forceAsUnstructured() const {
return clDisableStructuredFir;
}
diff --git a/flang/test/Lower/OpenACC/acc-cache.f90 b/flang/test/Lower/OpenACC/acc-cache.f90
index 12bdee3833870..151ef27a96ddc 100644
--- a/flang/test/Lower/OpenACC/acc-cache.f90
+++ b/flang/test/Lower/OpenACC/acc-cache.f90
@@ -387,11 +387,11 @@ subroutine test_cache_nonunit_lb()
! For arr(10:20), startIdx = 10, element 15 has lowerbound = 15 - 10 = 5
! CHECK: %[[C10:.*]] = arith.constant 10 : index
-! Unstructured loop with SELECT CASE: acc.loop becomes unstructured
-! CHECK: acc.loop private({{.*}}) {
-! CHECK: cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK: cf.cond_br %{{.*}}, ^[[BODY:.*]], ^[[EXIT:.*]]
+! The SELECT CASE branches only within the loop body, so acc.loop keeps its
+! structured control and the raw blocks live in a wrap inside it.
+! CHECK: acc.loop private({{.*}})
+! CHECK: scf.execute_region no_inline {
+! CHECK: cf.br ^[[BODY:.*]]
! CHECK: ^[[BODY]]:
! Compute lowerbound = 15 - startIdx = 15 - 10 = 5
! CHECK: %[[C1:.*]] = arith.constant 1 : index
@@ -420,13 +420,12 @@ subroutine test_cache_nonunit_lb()
! CHECK: hlfir.designate %[[DECL]]#0
! CHECK: hlfir.assign
! CHECK: cf.br ^[[MERGE]]
-! All SELECT CASE branches converge, then loop back or exit
+! All SELECT CASE branches converge, then leave the wrap. The loop control is
+! acc.loop's own, so the region ends at its yield rather than a back edge.
! CHECK: ^[[MERGE]]:
-! CHECK: cf.br ^[[HEADER]]
+! CHECK: cf.br ^[[EXIT:.*]]
! CHECK: ^[[EXIT]]:
-! Scope termination: acc.yield marks end of cache scope
-! CHECK: acc.yield
-! CHECK-NEXT: } {{.*}}unstructured{{.*}}
+! CHECK: scf.yield
end subroutine
! CHECK-LABEL: func.func @_QPtest_cache_use_after_region()
diff --git a/flang/test/Lower/OpenACC/acc-unstructured-internals.f90 b/flang/test/Lower/OpenACC/acc-unstructured-internals.f90
new file mode 100644
index 0000000000000..f0be621384a6a
--- /dev/null
+++ b/flang/test/Lower/OpenACC/acc-unstructured-internals.f90
@@ -0,0 +1,73 @@
+! RUN: bbc -fopenacc -emit-hlfir -o - %s | FileCheck %s
+
+! Loops under an OpenACC directive whose control flow is structured on the
+! outside but whose branching is confined to the body. The directive's own
+! code-gen consumes the DO, so the body is wrapped at the directive's body
+! lowering site rather than in genFIR(DoConstruct). The loop keeps its bounds
+! on the op -- control(...) rather than a cf trip-count test -- so it is still
+! available to be parallelized.
+
+! A forward GOTO raised inside a nested IF, jumping over a whole inner DO and
+! landing on the last statement of the outer loop body. Both endpoints are
+! inside the body, so the outer loop stays an acc.loop with control(...) and
+! the two inner loops stay acc.loops of their own.
+subroutine kernels_goto_over_inner(qfx, a, its, ite, jts, jte, force, flux)
+ real :: qfx(ite,jte), a(ite,jte)
+ logical :: force
+ integer :: flux
+ !$acc kernels
+ do j = jts, jte
+ do 330 i = its, ite
+ a(i,j) = a(i,j) + 1.0
+330 continue
+ if (force) then
+ if (flux .eq. 1) goto 350
+ endif
+ do i = its, ite
+ qfx(i,j) = 0.
+ enddo
+350 continue
+ enddo
+ !$acc end kernels
+end subroutine
+
+! CHECK-LABEL: func.func @_QPkernels_goto_over_inner
+! CHECK: acc.kernels {
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
+! The inner loops are untouched by the wrap around the outer body.
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: acc.yield
+! The GOTO and the branches it feeds stay inside the region.
+! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[THEN:[0-9]+]], ^bb[[SKIP:[0-9]+]]
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: acc.yield
+! CHECK: scf.yield
+! CHECK: acc.yield
+
+! A CYCLE targets the EndDoStmt, the boundary between the loop body and the
+! loop control, so inside the wrap it leaves the region at its yield.
+subroutine parallel_loop_cycle(a, n)
+ real :: a(n)
+ !$acc parallel loop
+ do i = 1, n
+ if (a(i) > 0.0) then
+ a(i) = 1.0
+ cycle
+ end if
+ a(i) = 2.0
+ end do
+end subroutine
+
+! CHECK-LABEL: func.func @_QPparallel_loop_cycle
+! CHECK: acc.parallel combined(loop) {
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
+! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
+! CHECK: ^bb[[CYCLE]]:
+! CHECK: cf.br ^bb[[EXIT:[0-9]+]]
+! CHECK: ^bb[[BODY]]:
+! CHECK: cf.br ^bb[[EXIT]]
+! CHECK: ^bb[[EXIT]]:
+! CHECK: scf.yield
+! CHECK: acc.yield
diff --git a/flang/test/Lower/OpenACC/acc-unstructured.f90 b/flang/test/Lower/OpenACC/acc-unstructured.f90
index cbb27d74cc96c..6a95f664f78df 100644
--- a/flang/test/Lower/OpenACC/acc-unstructured.f90
+++ b/flang/test/Lower/OpenACC/acc-unstructured.f90
@@ -275,19 +275,18 @@ subroutine test_unstructured_collapse_cycle(a)
! Both induction variables (j and i) are privatized:
! CHECK: %[[PRIVJ:.*]] = acc.private varPtr(%{{.*}} : !fir.ref<i32>) recipe(@privatization_ref_i32) implicit(true) name("j") -> !fir.ref<i32>
! CHECK: %[[PRIVI:.*]] = acc.private varPtr(%{{.*}} : !fir.ref<i32>) recipe(@privatization_ref_i32) implicit(true) name("i") -> !fir.ref<i32>
-! No control(...) on acc.loop — bounds are not on the op:
! CHECK: acc.loop combined(serial) private(%[[PRIVJ]], %[[PRIVI]] : !fir.ref<i32>, !fir.ref<i32>) {
-! Outer loop trip-count test (j) emitted as cf:
-! CHECK: arith.cmpi sgt
-! CHECK: cf.cond_br
-! Inner loop trip-count test (i) emitted as cf:
-! CHECK: arith.cmpi sgt
-! CHECK: cf.cond_br
-! The if/cycle is a structured cf branch in the body:
+! The IF-guarded CYCLE branches only within the body, so both loops keep their
+! bounds on the op -- control(...) rather than a cf trip-count test -- and the
+! raw blocks are confined to a wrap inside each body.
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
+! CHECK: acc.loop private({{.*}}) control(%{{.*}} : i32) = (%{{.*}} : i32) to (%{{.*}} : i32) step (%{{.*}} : i32) {
+! CHECK: scf.execute_region no_inline {
! CHECK: arith.cmpi eq
! CHECK: cf.cond_br
+! CHECK: scf.yield
! CHECK: acc.yield
-! CHECK: }
! `acc serial loop collapse(N)` with STOP in body: wrap-in-execute-region hides
! the unstructured if/stop and the three collapsed iterators lower as a single
diff --git a/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90 b/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
deleted file mode 100644
index 08ebcc7b4c8ef..0000000000000
--- a/flang/test/Lower/OpenMP/Todo/metadirective-loop-unstructured.f90
+++ /dev/null
@@ -1,39 +0,0 @@
-! Defer unstructured associated loops until every selection path can give its
-! PFT blocks an independent mapping.
-
-! RUN: split-file %s %t
-! RUN: %not_todo_cmd %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/static.f90 2>&1 | FileCheck %s
-! RUN: %not_todo_cmd %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/runtime.f90 2>&1 | FileCheck %s
-
-! CHECK: not yet implemented: unstructured associated DO in loop-associated METADIRECTIVE variant
-
-!--- static.f90
-subroutine test_static(n, a, selector)
- integer :: n, a(n), selector, i
- !$omp metadirective &
- !$omp & when(implementation={vendor(llvm)}: do) &
- !$omp & otherwise(nothing)
- do i = 1, n
- go to (10, 20), selector
-10 a(i) = 1
- go to 30
-20 a(i) = 2
-30 continue
- end do
-end subroutine
-
-!--- runtime.f90
-subroutine test_runtime(flag, n, a, selector)
- logical :: flag
- integer :: n, a(n), selector, i
- !$omp metadirective &
- !$omp & when(user={condition(flag)}: do) &
- !$omp & otherwise(nothing)
- do i = 1, n
- go to (10, 20), selector
-10 a(i) = 1
- go to 30
-20 a(i) = 2
-30 continue
- end do
-end subroutine
diff --git a/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90 b/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
new file mode 100644
index 0000000000000..7c4197e7e92d7
--- /dev/null
+++ b/flang/test/Lower/OpenMP/metadirective-loop-unstructured.f90
@@ -0,0 +1,50 @@
+! A DO associated with a METADIRECTIVE variant whose body branches only within
+! itself. The computed GO TO and its targets are all inside the loop body, so
+! the loop keeps its structured form and the raw blocks are confined to an
+! scf.execute_region inside omp.loop_nest. This used to be rejected as not yet
+! implemented, because the body's blocks had nowhere to live.
+
+! RUN: split-file %s %t
+! RUN: %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/static.f90 | FileCheck %s
+! RUN: %flang_fc1 -emit-hlfir -fopenmp -fopenmp-version=52 -o - %t/runtime.f90 | FileCheck %s
+
+! CHECK: omp.wsloop
+! CHECK: omp.loop_nest
+! CHECK: scf.execute_region no_inline {
+! The computed GO TO and the branches it feeds stay inside the region.
+! CHECK: fir.select %{{[0-9]+}} : i32 [1, ^bb[[L10:[0-9]+]], 2, ^bb[[L20:[0-9]+]], unit, ^bb[[L10]]]
+! CHECK: ^bb[[L10]]:
+! CHECK: ^bb[[L20]]:
+! CHECK: scf.yield
+! CHECK: omp.yield
+
+!--- static.f90
+subroutine test_static(n, a, selector)
+ integer :: n, a(n), selector, i
+ !$omp metadirective &
+ !$omp & when(implementation={vendor(llvm)}: do) &
+ !$omp & otherwise(nothing)
+ do i = 1, n
+ go to (10, 20), selector
+10 a(i) = 1
+ go to 30
+20 a(i) = 2
+30 continue
+ end do
+end subroutine
+
+!--- runtime.f90
+subroutine test_runtime(flag, n, a, selector)
+ logical :: flag
+ integer :: n, a(n), selector, i
+ !$omp metadirective &
+ !$omp & when(user={condition(flag)}: do) &
+ !$omp & otherwise(nothing)
+ do i = 1, n
+ go to (10, 20), selector
+10 a(i) = 1
+ go to 30
+20 a(i) = 2
+30 continue
+ end do
+end subroutine
diff --git a/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90 b/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
index 60fe63b8d29af..5a76ee4b91a6c 100644
--- a/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
+++ b/flang/test/Lower/OpenMP/wsloop-unstructured-cycle.f90
@@ -1,13 +1,10 @@
-! RUN: bbc --wrap-unstructured-constructs-in-execute-region -emit-hlfir -fopenmp -o - %s | FileCheck %s --implicit-check-not=scf.execute_region
+! RUN: bbc --wrap-unstructured-constructs-in-execute-region -emit-hlfir -fopenmp -o - %s | FileCheck %s
! A DO associated with an OpenMP loop directive is lowered by the directive's
-! own code-gen. Such a DO must never be folded into an
-! scf.execute_region, even when wrapping is enabled and the loop is
-! unstructured -- here the IF-guarded CYCLE makes it so. The body's blocks
-! stay flat inside omp.loop_nest.
-!
-! --implicit-check-not on the RUN line asserts that no wrapping takes place
-! anywhere in the output.
+! own code-gen, which never reaches genFIR(DoConstruct) where a plain loop's
+! body is wrapped. The body is wrapped at the directive's own body-lowering
+! site instead, so a loop whose branching is confined to its body -- here an
+! IF-guarded CYCLE -- keeps its structured form inside omp.loop_nest.
subroutine repro_final(x, y, n)
implicit none
@@ -27,26 +24,29 @@ subroutine repro_final(x, y, n)
end subroutine repro_final
+! The CYCLE targets the EndDoStmt, which inside the wrap is the region's yield
+! block -- so it leaves the region instead of branching to a block outside it.
! CHECK-LABEL: func.func @_QPrepro_final(
! CHECK: omp.wsloop
! CHECK: omp.loop_nest
! CHECK: hlfir.assign
-! CHECK: cf.br ^bb[[TEST:[0-9]+]]
-! CHECK: ^bb[[TEST]]:
-! CHECK: arith.cmpf ogt
-! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
-! CHECK: ^bb[[CYCLE]]:
-! CHECK: hlfir.assign
-! CHECK: cf.br ^bb[[EXIT:[0-9]+]]
-! CHECK: ^bb[[BODY]]:
-! CHECK: hlfir.assign
-! CHECK: cf.br ^bb[[EXIT]]
-! CHECK: ^bb[[EXIT]]:
+! CHECK: scf.execute_region no_inline {
+! CHECK: ^bb[[TEST:[0-9]+]]:
+! CHECK: arith.cmpf ogt
+! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[CYCLE:[0-9]+]], ^bb[[BODY:[0-9]+]]
+! CHECK: ^bb[[CYCLE]]:
+! CHECK: hlfir.assign
+! CHECK: cf.br ^bb[[EXIT:[0-9]+]]
+! CHECK: ^bb[[BODY]]:
+! CHECK: hlfir.assign
+! CHECK: cf.br ^bb[[EXIT]]
+! CHECK: ^bb[[EXIT]]:
+! CHECK: scf.yield
! CHECK: omp.yield
-! COLLAPSE(n) and ORDERED(n) both associate n loops with the directive, and
-! the loop transforming directives (TILE, INTERCHANGE, ...) associate as many
-! as their arguments describe. None of the associated loops may be wrapped.
+! COLLAPSE(n) and ORDERED(n) both associate n loops with the directive. The
+! associated loops are collapsed into a single omp.loop_nest, and the wrap goes
+! around the innermost body it encloses.
subroutine collapse_case(x, y, n)
implicit none
@@ -68,12 +68,14 @@ subroutine collapse_case(x, y, n)
end subroutine collapse_case
-! Both loops are associated with the directive, so the body stays flat inside
-! omp.loop_nest.
+! Both loops are associated with the directive, so one wrap covers the body of
+! the collapsed nest.
! CHECK-LABEL: func.func @_QPcollapse_case(
! CHECK: omp.wsloop
! CHECK: omp.loop_nest ({{.*}}) {{.*}} collapse(2) {
-! CHECK: cf.cond_br
+! CHECK: scf.execute_region no_inline {
+! CHECK: cf.cond_br
+! CHECK: scf.yield
! CHECK: omp.yield
subroutine ordered_case(x, y, n)
@@ -96,12 +98,16 @@ subroutine ordered_case(x, y, n)
end subroutine ordered_case
-! ORDERED(2) associates the inner loop with the directive as well, so it is
-! not wrapped either.
+! ORDERED(2) keeps the inner loop as a loop of its own inside the nest, so the
+! outer body and the inner body each get a wrap.
! CHECK-LABEL: func.func @_QPordered_case(
! CHECK: omp.wsloop ordered(2)
! CHECK: omp.loop_nest
-! CHECK: cf.cond_br
+! CHECK: scf.execute_region no_inline {
+! CHECK: scf.execute_region no_inline {
+! CHECK: cf.cond_br
+! CHECK: scf.yield
+! CHECK: scf.yield
! CHECK: omp.yield
! A TILE case belongs here too, since the SIZES arguments decide how many
diff --git a/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90 b/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
new file mode 100644
index 0000000000000..5ceba4f3d3875
--- /dev/null
+++ b/flang/test/Lower/OpenMP/wsloop-unstructured-internals.f90
@@ -0,0 +1,44 @@
+! RUN: bbc -fopenmp -emit-hlfir -o - %s | FileCheck %s
+
+! A loop associated with an OpenMP loop directive whose branching is confined
+! to its body. The directive's code-gen consumes the DO, so the body is wrapped
+! at the directive's body lowering site rather than in genFIR(DoConstruct). The
+! loop stays a single omp.loop_nest -- it is not dissolved into a cf trip-count
+! loop -- so it is still available to be worksharded.
+
+! A forward GOTO raised inside a nested IF, jumping over a whole inner DO and
+! landing on the last statement of the outer loop body. Both endpoints are in
+! the body, so the associated loop keeps its structured form and the two inner
+! loops stay plain fir.do_loops inside the wrap.
+subroutine omp_goto_over_inner(qfx, a, n, force, flux)
+ real :: qfx(n,n), a(n,n)
+ logical :: force
+ integer :: flux
+ !$omp parallel do
+ do j = 1, n
+ do 330 i = 1, n
+ a(i,j) = a(i,j) + 1.0
+330 continue
+ if (force) then
+ if (flux .eq. 1) goto 350
+ endif
+ do i = 1, n
+ qfx(i,j) = 0.
+ enddo
+350 continue
+ enddo
+ !$omp end parallel do
+end subroutine
+
+! CHECK-LABEL: func.func @_QPomp_goto_over_inner
+! CHECK: omp.wsloop
+! CHECK: omp.loop_nest (%{{.*}}) : i32 = (%{{.*}}) to (%{{.*}}) inclusive step (%{{.*}}) {
+! CHECK: scf.execute_region no_inline {
+! The inner loops are untouched by the wrap around the outer body.
+! CHECK: fir.do_loop %{{.*}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
+! The GOTO and the branches it feeds stay inside the region.
+! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[THEN:[0-9]+]], ^bb[[SKIP:[0-9]+]]
+! CHECK: cf.cond_br %{{[0-9]+}}, ^bb[[GOTO:[0-9]+]], ^bb[[SKIP]]
+! CHECK: fir.do_loop %{{.*}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
+! CHECK: scf.yield
+! CHECK: omp.yield
diff --git a/flang/test/Lower/always-execute-loop-body.f90 b/flang/test/Lower/always-execute-loop-body.f90
index 13c1522cb736e..7d4e9b5e44735 100644
--- a/flang/test/Lower/always-execute-loop-body.f90
+++ b/flang/test/Lower/always-execute-loop-body.f90
@@ -2,6 +2,10 @@
! Given the flag `--always-execute-loop-body` the compiler emits an extra
! code to change the trip count, test tries to verify the extra emitted HLFIR.
+!
+! The trip count only exists on the unstructured lowering path, so the loop has
+! to stay unstructured for the flag to have anything to act on: the GOTO leaves
+! the loop, which keeps it out of the structured form.
! CHECK-LABEL: func.func @_QPsome
subroutine some()
@@ -17,7 +21,7 @@ subroutine some()
! CHECK: %[[CMP:.*]] = arith.cmpi sgt, %[[LOADED_TRIP]], %c0{{.*}} : i32
! CHECK: cf.cond_br %[[CMP]]
do i=4,1,1
- stop 2
+ goto 9
end do
- return
+9 return
end
diff --git a/flang/test/Lower/do-loop-branch-to-loop-header.f90 b/flang/test/Lower/do-loop-branch-to-loop-header.f90
new file mode 100644
index 0000000000000..d7a0e8fea430d
--- /dev/null
+++ b/flang/test/Lower/do-loop-branch-to-loop-header.f90
@@ -0,0 +1,55 @@
+! RUN: bbc -emit-fir -o - %s | FileCheck %s
+
+! A DO statement is a valid branch target (F2023 11.2.1), and branching to it
+! restarts the loop: the DO construct becomes active again, its bounds are
+! re-evaluated and the iteration count re-established (F2023 11.1.7.3 p1,
+! 11.1.7.4.1 p1).
+!
+! A loop whose body branching is self-contained keeps its structured form, with
+! the body folded into an scf.execute_region. The loop control statements are
+! not part of that body: they may be branched to from outside the loop, so
+! their blocks have to stay in the enclosing region. Emitting the DO statement
+! inside the wrap gave it a block the enclosing code could not reference.
+
+! The ASSIGN makes a body statement a new block, so the body is wrapped; the
+! GOTO targets the loop header from outside the loop.
+subroutine branch_to_header(a, b)
+ real :: a(10), b(10)
+ integer :: m
+ go to 80
+40 do 42 i = 1, 10
+ assign 41 to m
+41 a(i) = b(i)
+42 continue
+80 continue
+86 go to 40
+end subroutine
+
+! CHECK-LABEL: func.func @_QPbranch_to_header
+! The loop header is a branch target, so it starts a block in the function's
+! own region -- the same region the branch is emitted from.
+! CHECK: cf.br ^bb[[HEADER:[0-9]+]]
+! CHECK: ^bb[[HEADER]]:
+! CHECK: fir.do_loop
+! The body, and only the body, lives in the region.
+! CHECK: scf.execute_region no_inline {
+! CHECK: scf.yield
+! CHECK: }
+! CHECK: }
+
+! Same shape with the ASSIGN target outside the loop: the body needs no blocks
+! of its own, so no wrap is emitted at all and the loop stays plain.
+subroutine no_wrap_needed(a, b)
+ real :: a(10), b(10)
+ integer :: m
+ go to 80
+40 do 42 i = 1, 10
+ assign 80 to m
+42 a(i) = b(i)
+80 continue
+86 go to 40
+end subroutine
+
+! CHECK-LABEL: func.func @_QPno_wrap_needed
+! CHECK-NOT: scf.execute_region
+! CHECK: fir.do_loop
diff --git a/flang/test/Lower/do_loop_unstructured.f90 b/flang/test/Lower/do_loop_unstructured.f90
index e31c31deb433d..e38c6ceda0f2b 100644
--- a/flang/test/Lower/do_loop_unstructured.f90
+++ b/flang/test/Lower/do_loop_unstructured.f90
@@ -16,35 +16,13 @@ subroutine simple_unstructured()
404 continue
end do
end subroutine
+! The GOTO targets the statement that follows it, so nothing branches out of
+! the body and the loop keeps its structured form.
! CHECK-LABEL: simple_unstructured
-! CHECK: %[[TRIP_VAR_REF:.*]] = fir.alloca i32
! CHECK: %[[LOOP_VAR_REF:.*]] = fir.alloca i32 <{bindc_name = "i", uniq_name = "_QFsimple_unstructuredEi"}>
-! CHECK: %[[LOOP_VAR_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_REF]]
-! CHECK: %[[ONE:.*]] = arith.constant 1 : i32
-! CHECK: %[[HUNDRED:.*]] = arith.constant 100 : i32
-! CHECK: %[[STEP_ONE:.*]] = arith.constant 1 : i32
-! CHECK: %[[TMP1:.*]] = arith.subi %[[HUNDRED]], %[[ONE]] : i32
-! CHECK: %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[STEP_ONE]] : i32
-! CHECK: %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[STEP_ONE]] : i32
-! CHECK: fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: fir.store %[[ONE]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK: %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK: %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK: cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK: %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[ONE_1:.*]] = arith.constant 1 : i32
-! CHECK: %[[TRIP_VAR_NEXT:.*]] = arith.subi %[[TRIP_VAR]], %[[ONE_1]] : i32
-! CHECK: fir.store %[[TRIP_VAR_NEXT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[LOOP_VAR:.*]] = fir.load %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: %[[STEP_ONE_2:.*]] = arith.constant 1 : i32
-! CHECK: %[[LOOP_VAR_NEXT:.*]] = arith.addi %[[LOOP_VAR]], %[[STEP_ONE_2]] overflow<nsw> : i32
-! CHECK: fir.store %[[LOOP_VAR_NEXT]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK: fir.do_loop %[[IV:.*]] = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK: fir.store %[[IV]] to
+! CHECK: }
! CHECK: return
! Test an unstructured loop with a step. Mostly similar to the previous one.
@@ -56,35 +34,11 @@ subroutine simple_unstructured_with_step()
404 continue
end do
end subroutine
+! Same, with an explicit step.
! CHECK-LABEL: simple_unstructured_with_step
-! CHECK: %[[TRIP_VAR_REF:.*]] = fir.alloca i32
-! CHECK: %[[LOOP_VAR_REF:.*]] = fir.alloca i32 <{bindc_name = "i", uniq_name = "_QFsimple_unstructured_with_stepEi"}>
-! CHECK: %[[LOOP_VAR_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_REF]]
-! CHECK: %[[ONE:.*]] = arith.constant 1 : i32
-! CHECK: %[[HUNDRED:.*]] = arith.constant 100 : i32
-! CHECK: %[[STEP:.*]] = arith.constant 2 : i32
-! CHECK: %[[TMP1:.*]] = arith.subi %[[HUNDRED]], %[[ONE]] : i32
-! CHECK: %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[STEP]] : i32
-! CHECK: %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[STEP]] : i32
-! CHECK: fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: fir.store %[[ONE]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK: %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK: %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK: cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK: %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[ONE_1:.*]] = arith.constant 1 : i32
-! CHECK: %[[TRIP_VAR_NEXT:.*]] = arith.subi %[[TRIP_VAR]], %[[ONE_1]] : i32
-! CHECK: fir.store %[[TRIP_VAR_NEXT]] to %[[TRIP_VAR_REF]] : !fir.ref<i32>
-! CHECK: %[[LOOP_VAR:.*]] = fir.load %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: %[[STEP_2:.*]] = arith.constant 2 : i32
-! CHECK: %[[LOOP_VAR_NEXT:.*]] = arith.addi %[[LOOP_VAR]], %[[STEP_2]] overflow<nsw> : i32
-! CHECK: fir.store %[[LOOP_VAR_NEXT]] to %[[LOOP_VAR_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK: fir.do_loop %[[IV:.*]] = %c1{{.*}} to %c100{{.*}} step %c2{{.*}} : i32 {
+! CHECK: fir.store %[[IV]] to
+! CHECK: }
! CHECK: return
! Test a three nested unstructured loop. Three nesting is the basic case where
@@ -100,39 +54,12 @@ subroutine nested_unstructured()
end do
end do
end subroutine
-! With the wrap-unstructured-constructs-in-execute-region pass, the innermost
-! k-loop is the only one classified unstructured (the `goto 404`/`404 continue`
-! pattern). It gets wrapped in scf.execute_region, and the outer i and j
-! loops fold back to fir.do_loop.
+! The innermost GOTO stays inside its own body, so every level stays
+! structured and no wrap is needed at all.
! CHECK-LABEL: nested_unstructured
-! CHECK: %[[TRIP_VAR_K_REF:.*]] = fir.alloca i32
-! CHECK: %[[LOOP_VAR_I_REF:.*]] = fir.alloca i32 <{bindc_name = "i", uniq_name = "_QFnested_unstructuredEi"}>
-! CHECK: %[[LOOP_VAR_I_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_I_REF]]
-! CHECK: %[[LOOP_VAR_J_REF:.*]] = fir.alloca i32 <{bindc_name = "j", uniq_name = "_QFnested_unstructuredEj"}>
-! CHECK: %[[LOOP_VAR_J_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_J_REF]]
-! CHECK: %[[LOOP_VAR_K_REF:.*]] = fir.alloca i32 <{bindc_name = "k", uniq_name = "_QFnested_unstructuredEk"}>
-! CHECK: %[[LOOP_VAR_K_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_K_REF]]
-! CHECK: fir.do_loop %{{[^ ]+}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
-! CHECK: fir.do_loop %{{[^ ]+}} = %{{.*}} to %{{.*}} step %{{.*}} : i32 {
-! CHECK: scf.execute_region no_inline {
-! CHECK: cf.br ^[[HEADER_K:.*]]
-! CHECK: ^[[HEADER_K]]:
-! CHECK: %[[TRIP_COUNT_K:.*]] = arith.divsi %{{.*}}, %{{.*}} : i32
-! CHECK: fir.store %[[TRIP_COUNT_K]] to %[[TRIP_VAR_K_REF]] : !fir.ref<i32>
-! CHECK: fir.store %{{.*}} to %[[LOOP_VAR_K_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER_K_BODY:.*]]
-! CHECK: ^[[HEADER_K_BODY]]:
-! CHECK: %[[TRIP_VAR_K:.*]] = fir.load %[[TRIP_VAR_K_REF]] : !fir.ref<i32>
-! CHECK: %[[COND_K:.*]] = arith.cmpi sgt, %[[TRIP_VAR_K]], %{{.*}} : i32
-! CHECK: cf.cond_br %[[COND_K]], ^[[BODY_K:.*]], ^[[EXIT_K:.*]]
-! CHECK: ^[[BODY_K]]:
-! CHECK: cf.br ^{{.*}}
-! CHECK: cf.br ^[[HEADER_K_BODY]]
-! CHECK: ^[[EXIT_K]]:
-! CHECK: scf.yield
-! CHECK: }
-! CHECK: }
-! CHECK: }
+! CHECK: fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK: fir.do_loop %{{.*}} = %c1{{.*}} to %c200{{.*}} step %c1{{.*}} : i32 {
+! CHECK: fir.do_loop %{{.*}} = %c1{{.*}} to %c300{{.*}} step %c1{{.*}} : i32 {
! CHECK: return
! Test the existence of a structured loop inside an unstructured loop.
@@ -146,54 +73,13 @@ subroutine nested_structured_in_unstructured()
404 continue
end do
end subroutine
+! The GOTO follows an inner loop, so the outer body needs raw blocks and is
+! wrapped, while the inner loop stays a plain fir.do_loop inside the wrap.
! CHECK-LABEL: nested_structured_in_unstructured
-! CHECK: %[[TRIP_VAR_I_REF:.*]] = fir.alloca i32
-! CHECK: %[[LOOP_VAR_I_REF:.*]] = fir.alloca i32 <{bindc_name = "i", uniq_name = "_QFnested_structured_in_unstructuredEi"}>
-! CHECK: %[[LOOP_VAR_I_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_I_REF]]
-! CHECK: %[[LOOP_VAR_J_REF:.*]] = fir.alloca i32 <{bindc_name = "j", uniq_name = "_QFnested_structured_in_unstructuredEj"}>
-! CHECK: %[[LOOP_VAR_J_DECL:.*]]:2 = hlfir.declare %[[LOOP_VAR_J_REF]]
-! CHECK: %[[I_START:.*]] = arith.constant 1 : i32
-! CHECK: %[[I_END:.*]] = arith.constant 100 : i32
-! CHECK: %[[I_STEP:.*]] = arith.constant 1 : i32
-! CHECK: %[[TMP1:.*]] = arith.subi %[[I_END]], %[[I_START]] : i32
-! CHECK: %[[TMP2:.*]] = arith.addi %[[TMP1]], %[[I_STEP]] : i32
-! CHECK: %[[TRIP_COUNT:.*]] = arith.divsi %[[TMP2]], %[[I_STEP]] : i32
-! CHECK: fir.store %[[TRIP_COUNT]] to %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK: fir.store %[[I_START]] to %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER:.*]]
-! CHECK: ^[[HEADER]]:
-! CHECK: %[[TRIP_VAR:.*]] = fir.load %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK: %[[ZERO:.*]] = arith.constant 0 : i32
-! CHECK: %[[COND:.*]] = arith.cmpi sgt, %[[TRIP_VAR]], %[[ZERO]] : i32
-! CHECK: cf.cond_br %[[COND]], ^[[BODY:.*]], ^[[EXIT:.*]]
-! CHECK: ^[[BODY]]:
-! CHECK: fir.do_loop %[[J_IV:[^ ]*]] =
-! CHECK-SAME: %[[J_LB:[^ ]*]] to %[[J_UB:[^ ]*]] step %[[J_ST:[^ ]*]] : i32 {
-! CHECK: fir.store %[[J_IV]] to %[[LOOP_VAR_J_DECL]]#0 : !fir.ref<i32>
-! CHECK: }
-! CHECK: %[[J_LBIDX:.*]] = fir.convert %[[J_LB]] : (i32) -> index
-! CHECK: %[[J_UBIDX:.*]] = fir.convert %[[J_UB]] : (i32) -> index
-! CHECK: %[[J_STIDX:.*]] = fir.convert %[[J_ST]] : (i32) -> index
-! CHECK: %[[J_C0:.*]] = arith.constant 0 : index
-! CHECK: %[[J_DIFF:.*]] = arith.subi %[[J_UBIDX]], %[[J_LBIDX]] : index
-! CHECK: %[[J_ADD:.*]] = arith.addi %[[J_DIFF]], %[[J_STIDX]] : index
-! CHECK: %[[J_TRIP:.*]] = arith.divsi %[[J_ADD]], %[[J_STIDX]] : index
-! CHECK: %[[J_CMP:.*]] = arith.cmpi slt, %[[J_TRIP]], %[[J_C0]] : index
-! CHECK: %[[J_SEL:.*]] = arith.select %[[J_CMP]], %[[J_C0]], %[[J_TRIP]] : index
-! CHECK: %[[J_MUL:.*]] = arith.muli %[[J_SEL]], %[[J_STIDX]] : index
-! CHECK: %[[J_LASTIDX:.*]] = arith.addi %[[J_LBIDX]], %[[J_MUL]] : index
-! CHECK: %[[J_LAST:.*]] = fir.convert %[[J_LASTIDX]] : (index) -> i32
-! CHECK: fir.store %[[J_LAST]] to %[[LOOP_VAR_J_DECL]]#0 : !fir.ref<i32>
-! CHECK: %[[TRIP_VAR_I:.*]] = fir.load %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK: %[[C1_3:.*]] = arith.constant 1 : i32
-! CHECK: %[[TRIP_VAR_I_NEXT:.*]] = arith.subi %[[TRIP_VAR_I]], %[[C1_3]] : i32
-! CHECK: fir.store %[[TRIP_VAR_I_NEXT]] to %[[TRIP_VAR_I_REF]] : !fir.ref<i32>
-! CHECK: %[[LOOP_VAR_I:.*]] = fir.load %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK: %[[I_STEP_2:.*]] = arith.constant 1 : i32
-! CHECK: %[[LOOP_VAR_I_NEXT:.*]] = arith.addi %[[LOOP_VAR_I]], %[[I_STEP_2]] overflow<nsw> : i32
-! CHECK: fir.store %[[LOOP_VAR_I_NEXT]] to %[[LOOP_VAR_I_DECL]]#0 : !fir.ref<i32>
-! CHECK: cf.br ^[[HEADER]]
-! CHECK: ^[[EXIT]]:
+! CHECK: fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK: scf.execute_region no_inline {
+! CHECK: fir.do_loop %{{.*}} = %c1{{.*}} to %c100{{.*}} step %c1{{.*}} : i32 {
+! CHECK: scf.yield
! CHECK: return
subroutine unstructured_do_concurrent
diff --git a/flang/test/Lower/select-case-statement.f90 b/flang/test/Lower/select-case-statement.f90
index b289024da2810..c1298bf177c8e 100644
--- a/flang/test/Lower/select-case-statement.f90
+++ b/flang/test/Lower/select-case-statement.f90
@@ -348,9 +348,12 @@ subroutine sempty(n)
! select case with goto exit
subroutine sgoto
n = 0
- ! CHECK: cf.cond_br
+ ! The SELECT CASE and its GOTOs branch only within the loop body, so the
+ ! loop keeps its structured form and the raw blocks are confined to a wrap.
+ ! CHECK: fir.do_loop
+ ! CHECK: scf.execute_region no_inline {
do i=1,8
- ! CHECK: fir.select_case %8 : i32 [#fir.upper, %c2_i32, ^bb{{.*}}, #fir.lower, %c5_i32, ^bb{{.*}}, unit, ^bb{{.*}}]
+ ! CHECK: fir.select_case %{{[0-9]+}} : i32 [#fir.upper, %c2_i32, ^bb{{.*}}, #fir.lower, %c5_i32, ^bb{{.*}}, unit, ^bb{{.*}}]
select case(i)
case (:2)
! CHECK-DAG: arith.muli {{.*}}, %c10_i32 : i32
>From c94d4143492cd5b491642a51097fb6f59b5d005f Mon Sep 17 00:00:00 2001
From: ergawy <kareem.ergawy at gmail.com>
Date: Fri, 25 Sep 2026 02:28:12 -0700
Subject: [PATCH 2/2] [flang][Test] Cover the lowering of loops with a
non-terminating body
A previous change leaves such a loop unstructured. Check what that produces:
the cycle survives as a block branching to itself, no fir.do_loop is emitted
for the loop control, and a loop that only needs a block of its own still
gets the structured form with its body in a region.
---
.../Lower/do-loop-infinite-body-cycle.f90 | 22 +++++++++++++++++++
1 file changed, 22 insertions(+)
diff --git a/flang/test/Lower/do-loop-infinite-body-cycle.f90 b/flang/test/Lower/do-loop-infinite-body-cycle.f90
index 8914690084e34..205c65978b85d 100644
--- a/flang/test/Lower/do-loop-infinite-body-cycle.f90
+++ b/flang/test/Lower/do-loop-infinite-body-cycle.f90
@@ -7,6 +7,7 @@
! its body.
! RUN: %flang_fc1 -fdebug-dump-pft -o /dev/null %s 2>&1 | FileCheck %s
+! RUN: %flang_fc1 -emit-hlfir -o - %s | FileCheck %s --check-prefix=FIR
! A GO TO branching to itself never leaves the body.
subroutine self_cycle(n)
@@ -20,6 +21,14 @@ subroutine self_cycle(n)
! CHECK: Subroutine self_cycle
! CHECK: <<DoConstruct!>>
+! The cycle survives as a block branching to itself. No fir.do_loop is emitted:
+! the loop control is unstructured too.
+! FIR-LABEL: func.func @_QPself_cycle
+! FIR: cf.cond_br %{{.*}}, ^[[SELF:bb[0-9]+]], ^bb{{[0-9]+}}
+! FIR: ^[[SELF]]:
+! FIR: cf.br ^[[SELF]]
+! FIR-NOT: fir.do_loop
+
! Two GO TOs branching to each other form the same exit-free cycle.
subroutine mutual_cycle(n)
integer :: n, i
@@ -33,6 +42,14 @@ subroutine mutual_cycle(n)
! CHECK: Subroutine mutual_cycle
! CHECK: <<DoConstruct!>>
+! Two blocks branching to each other, neither leaving the cycle.
+! FIR-LABEL: func.func @_QPmutual_cycle
+! FIR: ^[[A:bb[0-9]+]]:
+! FIR: cf.br ^[[B:bb[0-9]+]]
+! FIR: ^[[B]]:
+! FIR: cf.br ^[[A]]
+! FIR-NOT: fir.do_loop
+
! Control: nothing branches here at all. The ASSIGN alone makes label 41 a
! branch target, which is what gives the body a block of its own, and with no
! branch there is nothing to trap control. The loop still qualifies.
@@ -47,3 +64,8 @@ subroutine label_target_in_body(a, b)
! CHECK: Subroutine label_target_in_body
! CHECK: <<DoConstruct~>>
+
+! The control case keeps its structured form, body folded into a region.
+! FIR-LABEL: func.func @_QPlabel_target_in_body
+! FIR: fir.do_loop
+! FIR: scf.execute_region no_inline {
More information about the llvm-branch-commits
mailing list