[flang-commits] [flang] 4be402e - [flang] disable cf canonicalization patterns at O0 to preserve line location (#227770)
via flang-commits
flang-commits at lists.llvm.org
Thu Oct 1 03:01:52 PDT 2026
Author: jeanPerier
Date: 2026-10-01T12:01:38+02:00
New Revision: 4be402e459e2fa1e98826c99e3fcef4121729e26
URL: https://github.com/llvm/llvm-project/commit/4be402e459e2fa1e98826c99e3fcef4121729e26
DIFF: https://github.com/llvm/llvm-project/commit/4be402e459e2fa1e98826c99e3fcef4121729e26.diff
LOG: [flang] disable cf canonicalization patterns at O0 to preserve line location (#227770)
This patch is a follow up to #202065 and intends to preserve at least one loop
header instructions for each expanded array expressions at O0 so that breakpoint
set on array expression statements that are expanded inline fire once (an not at
each iteration which happens when the only instruction with the statement line
location is in the loop body). CF canonicalization patterns may remove branch
instructions that are the only valid breakpoint position for an array expression
statement from the source.
To solve this, this patch adds a new canonicalization pass that allows skipping
ControlFlow dialect pattern registration at O0.
This replaces #227306 approach that was using a debug pattern filtering
approach instead of using a new pass.
Assisted-by: AI
Added:
flang/lib/Optimizer/Transforms/O0CanonicalizerPass.cpp
flang/test/Integration/array-assignment-entry-debug.f90
flang/test/Integration/array-assignment-entry-paths.f90
flang/test/Transforms/o0-canonicalize.mlir
Modified:
flang/include/flang/Optimizer/Transforms/Passes.h
flang/include/flang/Optimizer/Transforms/Passes.td
flang/lib/Optimizer/Passes/Pipelines.cpp
flang/lib/Optimizer/Transforms/CMakeLists.txt
flang/test/Driver/mlir-debug-pass-pipeline.f90
flang/test/Driver/mlir-pass-pipeline.f90
flang/test/Integration/OpenMP/parallel-private-reduction-worstcase.f90
flang/test/Integration/cold_array_repacking.f90
Removed:
################################################################################
diff --git a/flang/include/flang/Optimizer/Transforms/Passes.h b/flang/include/flang/Optimizer/Transforms/Passes.h
index 5271d17514ddc..492a3cb568d76 100644
--- a/flang/include/flang/Optimizer/Transforms/Passes.h
+++ b/flang/include/flang/Optimizer/Transforms/Passes.h
@@ -13,6 +13,7 @@
#include "mlir/Dialect/LLVMIR/LLVMAttrs.h"
#include "mlir/Pass/Pass.h"
#include "mlir/Pass/PassRegistry.h"
+#include "llvm/ADT/STLFunctionalExtras.h"
#include <memory>
#include <utility>
@@ -65,6 +66,15 @@ std::unique_ptr<mlir::Pass> createVScaleAttrPass();
std::unique_ptr<mlir::Pass>
createVScaleAttrPass(std::pair<unsigned, unsigned> vscaleAttr);
+/// Collect canonicalization patterns from loaded dialects and registered ops.
+/// When includeCFPatterns is false, omit both dialect-wide and operation-level
+/// registrations from the ControlFlow dialect to preserve statement-entry
+/// branches after CFG lowering. Other dialects' patterns and folding hooks may
+/// still modify control flow. shouldCollect can exclude additional operations.
+void populateCanonicalizationPatterns(
+ mlir::RewritePatternSet &patterns, bool includeCFPatterns,
+ llvm::function_ref<bool(mlir::RegisteredOperationName)> shouldCollect = {});
+
void populateFIRToSCFRewrites(mlir::RewritePatternSet &patterns,
bool parallelUnordered = false,
bool setNSW = true);
diff --git a/flang/include/flang/Optimizer/Transforms/Passes.td b/flang/include/flang/Optimizer/Transforms/Passes.td
index 9bd2a033e1365..f448077ead7e9 100644
--- a/flang/include/flang/Optimizer/Transforms/Passes.td
+++ b/flang/include/flang/Optimizer/Transforms/Passes.td
@@ -411,6 +411,22 @@ def AddAliasTags : Pass<"fir-add-alias-tags", "mlir::ModuleOp"> {
let dependentDialects = ["mlir::LLVM::LLVMDialect"];
}
+def O0CanonicalizerPass : Pass<"fir-o0-canonicalize"> {
+ let summary = "Simplify operations while preserving statement-entry branches";
+ let description = [{
+ Clean up IR after CFG lowering at O0, using canonicalization patterns from
+ all loaded dialects except ControlFlow. CF canonicalizations can bypass
+ loop-entry branches carrying an array assignment's source location and
+ move its PHI initializations into preceding statements, losing the
+ assignment's breakpoint location outside the loop.
+
+ Both dialect-wide and operation-level CF pattern registrations are omitted.
+ Folding and dead operation elimination remain enabled; region simplification
+ is disabled. This pass does not promise to preserve arbitrary CFG structure
+ against transformations registered by other dialects.
+ }];
+}
+
def SimplifyRegionLite : Pass<"simplify-region-lite"> {
let summary = "Region simplification";
let description = [{
diff --git a/flang/lib/Optimizer/Passes/Pipelines.cpp b/flang/lib/Optimizer/Passes/Pipelines.cpp
index ef18ff1cde945..65fed60feba62 100644
--- a/flang/lib/Optimizer/Passes/Pipelines.cpp
+++ b/flang/lib/Optimizer/Passes/Pipelines.cpp
@@ -241,7 +241,10 @@ void createDefaultFIRPostCFGOptimizerPassPipeline(
pm.addPass(mlir::createSCFToControlFlowPass());
- pm.addPass(mlir::createCanonicalizerPass(config));
+ if (pc.OptLevel == llvm::OptimizationLevel::O0)
+ pm.addPass(fir::createO0CanonicalizerPass());
+ else
+ pm.addPass(mlir::createCanonicalizerPass(config));
pm.addPass(fir::createSimplifyRegionLite());
if (!pc.SkipConvertComplexPow)
pm.addPass(fir::createConvertComplexPow());
diff --git a/flang/lib/Optimizer/Transforms/CMakeLists.txt b/flang/lib/Optimizer/Transforms/CMakeLists.txt
index 0a3d28098054b..ba63cbd564a97 100644
--- a/flang/lib/Optimizer/Transforms/CMakeLists.txt
+++ b/flang/lib/Optimizer/Transforms/CMakeLists.txt
@@ -51,6 +51,7 @@ add_flang_library(FIRTransforms
MemoryAllocation.cpp
MemoryUtils.cpp
OptimizeArrayRepacking.cpp
+ O0CanonicalizerPass.cpp
PolymorphicOpConversion.cpp
SetRuntimeCallAttributes.cpp
SimplifyFIROperations.cpp
@@ -85,6 +86,7 @@ add_flang_library(FIRTransforms
MLIR_LIBS
MLIRAffineUtils
MLIRAnalysis
+ MLIRControlFlowDialect
MLIRFuncDialect
MLIRGPUDialect
MLIRLLVMCommonConversion
diff --git a/flang/lib/Optimizer/Transforms/O0CanonicalizerPass.cpp b/flang/lib/Optimizer/Transforms/O0CanonicalizerPass.cpp
new file mode 100644
index 0000000000000..968d03863651f
--- /dev/null
+++ b/flang/lib/Optimizer/Transforms/O0CanonicalizerPass.cpp
@@ -0,0 +1,62 @@
+//===-- O0CanonicalizerPass.cpp -- Canonicalization for O0 ----------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "flang/Optimizer/Transforms/Passes.h"
+#include "mlir/Dialect/ControlFlow/IR/ControlFlow.h"
+#include "mlir/IR/PatternMatch.h"
+#include "mlir/Transforms/GreedyPatternRewriteDriver.h"
+
+namespace fir {
+#define GEN_PASS_DEF_O0CANONICALIZERPASS
+#include "flang/Optimizer/Transforms/Passes.h.inc"
+} // namespace fir
+
+// FIR version of Canonicalizer::initialize that excludes cf dialect patterns
+// under an option and also provides a way to filter patterns for given
+// operations via a callback
+void fir::populateCanonicalizationPatterns(
+ mlir::RewritePatternSet &patterns, bool includeCFPatterns,
+ llvm::function_ref<bool(mlir::RegisteredOperationName)> shouldCollect) {
+ mlir::MLIRContext *context = patterns.getContext();
+ auto includeDialect = [&](mlir::Dialect *dialect) {
+ return includeCFPatterns ||
+ !mlir::isa<mlir::cf::ControlFlowDialect>(dialect);
+ };
+ for (mlir::Dialect *dialect : context->getLoadedDialects())
+ if (includeDialect(dialect))
+ dialect->getCanonicalizationPatterns(patterns);
+ for (mlir::RegisteredOperationName op : context->getRegisteredOperations())
+ if (includeDialect(&op.getDialect()) &&
+ (!shouldCollect || shouldCollect(op)))
+ op.getCanonicalizationPatterns(patterns, context);
+}
+
+namespace {
+class O0CanonicalizerPass
+ : public fir::impl::O0CanonicalizerPassBase<O0CanonicalizerPass> {
+public:
+ mlir::LogicalResult initialize(mlir::MLIRContext *context) override {
+ mlir::RewritePatternSet owningPatterns(context);
+ fir::populateCanonicalizationPatterns(owningPatterns,
+ /*includeCFPatterns=*/false);
+ patterns = mlir::FrozenRewritePatternSet(std::move(owningPatterns));
+ return mlir::success();
+ }
+
+ void runOnOperation() override {
+ mlir::GreedyRewriteConfig config;
+ config.setRegionSimplificationLevel(
+ mlir::GreedySimplifyRegionLevel::Disabled);
+ // Like canonicalization, this cleanup is best-effort.
+ (void)mlir::applyPatternsGreedily(getOperation(), patterns, config);
+ }
+
+private:
+ mlir::FrozenRewritePatternSet patterns;
+};
+} // namespace
diff --git a/flang/test/Driver/mlir-debug-pass-pipeline.f90 b/flang/test/Driver/mlir-debug-pass-pipeline.f90
index 6db9bc666f583..1cc4fad4f76a4 100644
--- a/flang/test/Driver/mlir-debug-pass-pipeline.f90
+++ b/flang/test/Driver/mlir-debug-pass-pipeline.f90
@@ -107,7 +107,7 @@
! ALL-NEXT: StackReclaim
! ALL-NEXT: CFGConversion
! ALL-NEXT: SCFToControlFlow
-! ALL-NEXT: Canonicalizer
+! ALL-NEXT: O0CanonicalizerPass
! ALL-NEXT: SimplifyRegionLite
! ALL-NEXT: ConvertComplexPow
! ALL-NEXT: CSE
diff --git a/flang/test/Driver/mlir-pass-pipeline.f90 b/flang/test/Driver/mlir-pass-pipeline.f90
index 0c13f14f35974..f7f4a29ab2de9 100644
--- a/flang/test/Driver/mlir-pass-pipeline.f90
+++ b/flang/test/Driver/mlir-pass-pipeline.f90
@@ -166,7 +166,8 @@
! ALL-NEXT: CFGConversion
! ALL-NEXT: SCFToControlFlow
-! ALL-NEXT: Canonicalizer
+! O0-NEXT: O0CanonicalizerPass
+! O2-NEXT: Canonicalizer
! ALL-NEXT: SimplifyRegionLite
! ALL-NEXT: ConvertComplexPow
! ALL-NEXT: CSE
diff --git a/flang/test/Integration/OpenMP/parallel-private-reduction-worstcase.f90 b/flang/test/Integration/OpenMP/parallel-private-reduction-worstcase.f90
index 6d34dc8ec05a7..ea52c2a2b5bb8 100644
--- a/flang/test/Integration/OpenMP/parallel-private-reduction-worstcase.f90
+++ b/flang/test/Integration/OpenMP/parallel-private-reduction-worstcase.f90
@@ -83,72 +83,78 @@ subroutine worst_case(a, b, c, d)
! CHECK: omp.private.copy12: ; preds = %omp.private.copy
! [begin firstprivate copy for first var]
! [read the length, is it non-zero?]
-! CHECK: br i1 %{{.*}}, label %omp.private.copy13, label %omp.private.copy22
+! CHECK: br i1 %{{.*}}, label %omp.private.copy13, label %omp.private.copy26
-! CHECK: omp.private.copy22: ; preds = %omp.private.copy21, %omp.private.copy12
+! CHECK: omp.private.copy26: ; preds = %omp.private.copy25, %omp.private.copy12
! CHECK-NEXT: br label %omp.region.cont11
-! CHECK: omp.region.cont11: ; preds = %omp.private.copy22
+! CHECK: omp.region.cont11: ; preds = %omp.private.copy26
! CHECK-NEXT: %{{.*}} = phi ptr
-! CHECK-NEXT: br label %omp.private.copy24
+! CHECK-NEXT: br label %omp.private.copy28
-! CHECK: omp.private.copy24: ; preds = %omp.region.cont11
+! CHECK: omp.private.copy28: ; preds = %omp.region.cont11
! [begin firstprivate copy for second var]
! [read the length, is it non-zero?]
-! CHECK: br i1 %{{.*}}, label %omp.private.copy25, label %omp.private.copy34
+! CHECK: br i1 %{{.*}}, label %omp.private.copy29, label %omp.private.copy42
-! CHECK: omp.private.copy34: ; preds = %omp.private.copy33, %omp.private.copy24
-! CHECK-NEXT: br label %omp.region.cont23
+! CHECK: omp.private.copy42: ; preds = %omp.private.copy41, %omp.private.copy28
+! CHECK-NEXT: br label %omp.region.cont27
-! CHECK: omp.region.cont23: ; preds = %omp.private.copy34
+! CHECK: omp.region.cont27: ; preds = %omp.private.copy42
! CHECK-NEXT: %{{.*}} = phi ptr
! CHECK-NEXT: br label %omp.reduction.init
-! CHECK: omp.reduction.init: ; preds = %omp.region.cont23
+! CHECK: omp.reduction.init: ; preds = %omp.region.cont27
! [deferred stores for results of reduction alloc regions]
! CHECK: br label %[[VAL_96:.*]]
! CHECK: omp.reduction.neutral: ; preds = %omp.reduction.init
! [start of reduction initialization region]
! [null check:]
-! CHECK: br i1 %{{.*}}, label %omp.reduction.neutral36, label %omp.reduction.neutral37
+! CHECK: br i1 %{{.*}}, label %omp.reduction.neutral44, label %omp.reduction.neutral45
-! CHECK: omp.reduction.neutral37: ; preds = %omp.reduction.neutral
+! CHECK: omp.reduction.neutral45: ; preds = %omp.reduction.neutral
! [malloc and assign the default value to the reduction variable]
-! CHECK: br label %omp.reduction.neutral38
+! CHECK: br label %omp.reduction.neutral46
-! CHECK: omp.reduction.neutral38: ; preds = %omp.reduction.neutral36, %omp.reduction.neutral37
-! CHECK-NEXT: br label %omp.region.cont35
+! CHECK: omp.reduction.neutral46: ; preds = %omp.reduction.neutral44, %omp.reduction.neutral45
+! CHECK-NEXT: br label %omp.region.cont43
-! CHECK: omp.region.cont35: ; preds = %omp.reduction.neutral38
+! CHECK: omp.region.cont43: ; preds = %omp.reduction.neutral46
! CHECK-NEXT: %{{.*}} = phi ptr
-! CHECK-NEXT: br label %omp.reduction.neutral40
+! CHECK-NEXT: br label %omp.reduction.neutral48
-! CHECK: omp.reduction.neutral40: ; preds = %omp.region.cont35
+! CHECK: omp.reduction.neutral48: ; preds = %omp.region.cont43
! [start of reduction initialization region]
! [null check:]
-! CHECK: br i1 %{{.*}}, label %omp.reduction.neutral41, label %omp.reduction.neutral42
+! CHECK: br i1 %{{.*}}, label %omp.reduction.neutral49, label %omp.reduction.neutral50
-! CHECK: omp.reduction.neutral42: ; preds = %omp.reduction.neutral40
+! CHECK: omp.reduction.neutral50: ; preds = %omp.reduction.neutral48
! [malloc and assign the default value to the reduction variable]
-! CHECK: br label %omp.reduction.neutral43
+! CHECK: br label %omp.reduction.neutral51
-! CHECK: omp.reduction.neutral43: ; preds = %omp.reduction.neutral41, %omp.reduction.neutral42
-! CHECK-NEXT: br label %omp.region.cont39
+! CHECK: omp.reduction.neutral51: ; preds = %omp.reduction.neutral49, %omp.reduction.neutral50
+! CHECK-NEXT: br label %omp.region.cont47
-! CHECK: omp.region.cont39: ; preds = %omp.reduction.neutral43
+! CHECK: omp.region.cont47: ; preds = %omp.reduction.neutral51
! CHECK-NEXT: %{{.*}} = phi ptr
-! CHECK-NEXT: br label %omp.par.region45
+! CHECK-NEXT: br label %omp.par.region53
-! CHECK: omp.par.region45: ; preds = %omp.region.cont39
+! CHECK: omp.par.region53: ; preds = %omp.region.cont47
+! CHECK-NEXT: br label %omp.par.region54
+
+! CHECK: omp.par.region54: ; preds = %omp.par.region53
! [call SUM runtime function]
! [if (sum(a) == 1)]
-! CHECK: br i1 %{{.*}}, label %omp.par.region46, label %omp.par.region47
+! CHECK: br i1 %{{.*}}, label %omp.par.region55, label %omp.par.region56
+
+! CHECK: omp.par.region56: ; preds = %omp.par.region54
+! CHECK-NEXT: br label %omp.par.region57
-! CHECK: omp.par.region47: ; preds = %omp.par.region45
-! CHECK-NEXT: br label %omp.region.cont44
+! CHECK: omp.par.region57: ; preds = %omp.par.region56
+! CHECK-NEXT: br label %omp.region.cont52
-! CHECK: omp.region.cont44: ; preds = %omp.par.region47
+! CHECK: omp.region.cont52: ; preds = %omp.par.region57
! [omp parallel region done, call into the runtime to complete reduction]
! CHECK: %[[VAL_233:.*]] = call i32 @__kmpc_reduce(
! CHECK: switch i32 %[[VAL_233]], label %reduce.finalize [
@@ -156,16 +162,16 @@ subroutine worst_case(a, b, c, d)
! CHECK-NEXT: i32 2, label %reduce.switch.atomic
! CHECK-NEXT: ]
-! CHECK: reduce.switch.atomic: ; preds = %omp.region.cont44
+! CHECK: reduce.switch.atomic: ; preds = %omp.region.cont52
! CHECK-NEXT: unreachable
-! CHECK: reduce.switch.nonatomic: ; preds = %omp.region.cont44
+! CHECK: reduce.switch.nonatomic: ; preds = %omp.region.cont52
! CHECK-NEXT: %[[red_private_value_0:.*]] = load ptr, ptr %{{.*}}, align 8
! CHECK-NEXT: br label %omp.reduction.nonatomic.body
! [various blocks implementing the reduction]
-! CHECK: omp.region.cont52: ; preds =
+! CHECK: omp.region.cont62: ; preds =
! CHECK-NEXT: %{{.*}} = phi ptr
! CHECK-NEXT: call void @__kmpc_end_reduce(
! CHECK-NEXT: br label %reduce.finalize
@@ -182,38 +188,56 @@ subroutine worst_case(a, b, c, d)
! CHECK: omp.reduction.cleanup: ; preds = %.fini
! [null check]
-! CHECK: br i1 %{{.*}}, label %omp.reduction.cleanup58, label %omp.reduction.cleanup59
+! CHECK: br i1 %{{.*}}, label %omp.reduction.cleanup68, label %omp.reduction.cleanup69
-! CHECK: omp.reduction.cleanup59: ; preds = %omp.reduction.cleanup58, %omp.reduction.cleanup
-! CHECK-NEXT: br label %omp.region.cont57
+! CHECK: omp.reduction.cleanup69: ; preds = %omp.reduction.cleanup68, %omp.reduction.cleanup
+! CHECK-NEXT: br label %omp.region.cont67
-! CHECK: omp.region.cont57: ; preds = %omp.reduction.cleanup59
+! CHECK: omp.region.cont67: ; preds = %omp.reduction.cleanup69
! CHECK-NEXT: %{{.*}} = load ptr, ptr
-! CHECK-NEXT: br label %omp.reduction.cleanup61
+! CHECK-NEXT: br label %omp.reduction.cleanup71
-! CHECK: omp.reduction.cleanup61: ; preds = %omp.region.cont57
+! CHECK: omp.reduction.cleanup71: ; preds = %omp.region.cont67
! [null check]
-! CHECK: br i1 %{{.*}}, label %omp.reduction.cleanup62, label %omp.reduction.cleanup63
+! CHECK: br i1 %{{.*}}, label %omp.reduction.cleanup72, label %omp.reduction.cleanup73
-! CHECK: omp.par.region46: ; preds = %omp.par.region45
+! CHECK: omp.par.region55: ; preds = %omp.par.region54
! CHECK-NEXT: call void @_FortranAStopStatement
! CHECK-NEXT: unreachable
-! CHECK: omp.reduction.neutral41: ; preds = %omp.reduction.neutral40
+! CHECK: omp.reduction.neutral49: ; preds = %omp.reduction.neutral48
! [source length was zero: finish initializing array]
-! CHECK: br label %omp.reduction.neutral43
+! CHECK: br label %omp.reduction.neutral51
-! CHECK: omp.reduction.neutral36: ; preds = %omp.reduction.neutral
+! CHECK: omp.reduction.neutral44: ; preds = %omp.reduction.neutral
! [source length was zero: finish initializing array]
-! CHECK: br label %omp.reduction.neutral38
+! CHECK: br label %omp.reduction.neutral46
-! CHECK: omp.private.copy33: ; preds = %omp.private.copy32, %omp.private.copy29
+! CHECK: omp.private.copy41: ; preds = %omp.private.copy40, %omp.private.copy37
! [source length was non-zero: call assign runtime]
+! CHECK: br label %omp.private.copy42
+
+! CHECK: omp.private.copy32: ; preds = %omp.private.copy30
+! CHECK-NEXT: br label %omp.private.copy33
+
+! CHECK: omp.private.copy33: ; preds = %omp.private.copy31, %omp.private.copy32
! CHECK: br label %omp.private.copy34
-! CHECK: omp.private.copy21: ; preds = %omp.private.copy20, %omp.private.copy17
+! CHECK: omp.private.copy34: ; preds = %omp.private.copy33
+! CHECK-NEXT: br label %omp.private.copy36
+
+! CHECK: omp.private.copy25: ; preds = %omp.private.copy24, %omp.private.copy21
! [source length was non-zero: call assign runtime]
-! CHECK: br label %omp.private.copy22
+! CHECK: br label %omp.private.copy26
+
+! CHECK: omp.private.copy16: ; preds = %omp.private.copy14
+! CHECK-NEXT: br label %omp.private.copy17
+
+! CHECK: omp.private.copy17: ; preds = %omp.private.copy15, %omp.private.copy16
+! CHECK: br label %omp.private.copy18
+
+! CHECK: omp.private.copy18: ; preds = %omp.private.copy17
+! CHECK-NEXT: br label %omp.private.copy20
! CHECK: omp.private.init8: ; preds = %omp.private.init7
! [var extent was non-zero: malloc a private array]
@@ -223,5 +247,5 @@ subroutine worst_case(a, b, c, d)
! [var extent was non-zero: malloc a private array]
! CHECK: br label %omp.private.init5
-! CHECK: omp.par.exit.exitStub: ; preds = %omp.region.cont67
+! CHECK: omp.par.exit.exitStub: ; preds = %omp.region.cont77
! CHECK-NEXT: ret void
diff --git a/flang/test/Integration/array-assignment-entry-debug.f90 b/flang/test/Integration/array-assignment-entry-debug.f90
new file mode 100644
index 0000000000000..a2f0d905a0999
--- /dev/null
+++ b/flang/test/Integration/array-assignment-entry-debug.f90
@@ -0,0 +1,36 @@
+subroutine arrays(a, b, c)
+ integer :: a(4), b(4), c(4)
+ b = 1
+ c = 2
+ a = 11
+ a = b + c
+end subroutine
+
+! Expand the assignments explicitly, without enabling other optimizations.
+! Each following assignment must retain a separate entry branch carrying its
+! source location, outside both the preceding and following loops.
+! RUN: %flang_fc1 -emit-hlfir -mmlir --mlir-print-debuginfo -o - %s | fir-opt --inline-elementals --inline-hlfir-assign --mlir-print-debuginfo -o %t.fir && %flang_fc1 -O0 -debug-info-kind=line-tables-only -emit-llvm %t.fir -o - | FileCheck %s
+
+! CHECK-LABEL: define void @arrays_(
+! CHECK: br label %[[FIRST:[0-9]+]], !dbg ![[LOC1:[0-9]+]]
+! CHECK: [[FIRST]]:
+! CHECK: br i1 {{.*}}, label %{{[0-9]+}}, label %[[ENTRY2:[0-9]+]], !dbg ![[LOC1]]
+! CHECK: [[ENTRY2]]:
+! CHECK-NEXT: br label %[[SECOND:[0-9]+]], !dbg ![[LOC2:[0-9]+]]
+! CHECK: [[SECOND]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ {{[01]}}, %[[ENTRY2]] ]
+! CHECK: br i1 {{.*}}, label %{{[0-9]+}}, label %[[ENTRY3:[0-9]+]], !dbg ![[LOC2]]
+! CHECK: [[ENTRY3]]:
+! CHECK-NEXT: br label %[[THIRD:[0-9]+]], !dbg ![[LOC3:[0-9]+]]
+! CHECK: [[THIRD]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ {{[01]}}, %[[ENTRY3]] ]
+! CHECK: br i1 {{.*}}, label %{{[0-9]+}}, label %[[ENTRY4:[0-9]+]], !dbg ![[LOC3]]
+! CHECK: [[ENTRY4]]:
+! CHECK-NEXT: br label %[[FOURTH:[0-9]+]], !dbg ![[LOC4:[0-9]+]]
+! CHECK: [[FOURTH]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ {{[01]}}, %[[ENTRY4]] ]
+! CHECK: br i1 {{.*}}, !dbg ![[LOC4]]
+! CHECK-DAG: ![[LOC1]] = !DILocation(line: 3, column: 3,
+! CHECK-DAG: ![[LOC2]] = !DILocation(line: 4, column: 3,
+! CHECK-DAG: ![[LOC3]] = !DILocation(line: 5, column: 3,
+! CHECK-DAG: ![[LOC4]] = !DILocation(line: 6, column: 3,
diff --git a/flang/test/Integration/array-assignment-entry-paths.f90 b/flang/test/Integration/array-assignment-entry-paths.f90
new file mode 100644
index 0000000000000..10699b127358d
--- /dev/null
+++ b/flang/test/Integration/array-assignment-entry-paths.f90
@@ -0,0 +1,87 @@
+subroutine if_then(a, l)
+ integer :: a(4)
+ logical :: l
+ integer :: x
+ if (l) x = 1
+ a = 11
+end subroutine
+
+subroutine if_else(a, b, l)
+ integer :: a(4), b(4)
+ logical :: l
+ integer :: x
+ if (l) then
+ x = 1
+ else
+ x = 2
+ end if
+ a = 11
+ b = a + x
+end subroutine
+
+subroutine computed_goto(a, k)
+ integer :: a(4), k
+ goto (10,20), k
+ a = 33
+ return
+10 a = 11
+ return
+20 a = 22
+end subroutine
+
+! Expand assignments explicitly to exercise the O0 pipeline independently of
+! which HLFIR expansions are enabled by default.
+! RUN: %flang_fc1 -emit-hlfir -mmlir --mlir-print-debuginfo -o - %s | fir-opt --inline-elementals --inline-hlfir-assign --mlir-print-debuginfo -o %t.fir && %flang_fc1 -O0 -debug-info-kind=line-tables-only -emit-llvm %t.fir -o - | FileCheck %s
+
+! Every path to an expanded assignment must go through its entry branch.
+! In particular, keeping only the conditional edge is insufficient: the
+! unconditional edge from the THEN arm must also go through the preheader.
+! CHECK-LABEL: define void @if_then_(
+! CHECK: br i1 {{.*}}, label %[[THEN:[0-9]+]], label %[[ENTRY:[0-9]+]]
+! CHECK: [[THEN]]:
+! CHECK: store i32 1,
+! CHECK-NEXT: br label %[[ENTRY]],
+! CHECK: [[ENTRY]]:
+! CHECK-NEXT: br label %[[HEADER:[0-9]+]], !dbg ![[THEN_LOC:[0-9]+]]
+! CHECK: [[HEADER]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ 1, %[[ENTRY]] ]
+
+! Both arms must converge before initializing the loop PHIs. Merely finding
+! one branch with the assignment location misses a breakpoint on the other arm.
+! CHECK-LABEL: define void @if_else_(
+! CHECK: br i1 {{.*}}, label %[[THEN:[0-9]+]], label %[[ELSE:[0-9]+]]
+! CHECK: [[THEN]]:
+! CHECK: store i32 1,
+! CHECK-NEXT: br label %[[ENTRY:[0-9]+]],
+! CHECK: [[ELSE]]:
+! CHECK: store i32 2,
+! CHECK-NEXT: br label %[[ENTRY]],
+! CHECK: [[ENTRY]]:
+! CHECK-NEXT: br label %[[HEADER:[0-9]+]], !dbg ![[ELSE_LOC:[0-9]+]]
+! CHECK: [[HEADER]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ 1, %[[ENTRY]] ]
+
+! A computed GOTO lowers to cf.switch. Check both case destinations and the
+! default destination: none may skip the corresponding assignment preheader.
+! CHECK-LABEL: define void @computed_goto_(
+! CHECK: switch i64 {{.*}}, label %[[DEFAULT:[0-9]+]] [
+! CHECK-NEXT: i64 1, label %[[CASE1:[0-9]+]]
+! CHECK-NEXT: i64 2, label %[[CASE2:[0-9]+]]
+! CHECK: [[DEFAULT]]:
+! CHECK-NEXT: br label %[[HEADER:[0-9]+]], !dbg ![[DEFAULT_LOC:[0-9]+]]
+! CHECK: [[HEADER]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ 1, %[[DEFAULT]] ]
+! CHECK: [[CASE1]]:
+! CHECK-NEXT: br label %[[HEADER:[0-9]+]], !dbg ![[CASE1_LOC:[0-9]+]]
+! CHECK: [[HEADER]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ 1, %[[CASE1]] ]
+! CHECK: [[CASE2]]:
+! CHECK-NEXT: br label %[[HEADER:[0-9]+]], !dbg ![[CASE2_LOC:[0-9]+]]
+! CHECK: [[HEADER]]:
+! CHECK-NEXT: {{.*}} = phi i64 {{.*}}[ 1, %[[CASE2]] ]
+
+! CHECK-DAG: ![[THEN_LOC]] = !DILocation(line: 6, column: 3,
+! CHECK-DAG: ![[ELSE_LOC]] = !DILocation(line: 18, column: 3,
+! CHECK-DAG: ![[DEFAULT_LOC]] = !DILocation(line: 25, column: 3,
+! CHECK-DAG: ![[CASE1_LOC]] = !DILocation(line: 27, column: 1,
+! CHECK-DAG: ![[CASE2_LOC]] = !DILocation(line: 29, column: 1,
diff --git a/flang/test/Integration/cold_array_repacking.f90 b/flang/test/Integration/cold_array_repacking.f90
index 2a3769e381784..06edd38738859 100644
--- a/flang/test/Integration/cold_array_repacking.f90
+++ b/flang/test/Integration/cold_array_repacking.f90
@@ -6,22 +6,34 @@
! CHECK-SAME: ptr noalias [[TMP0:%.*]])
! CHECK: [[TMP4:%.*]] = ptrtoint ptr [[TMP0]] to i64
! CHECK: [[TMP5:%.*]] = icmp ne i64 [[TMP4]], 0
-! CHECK: br i1 [[TMP5]], label %[[BB6:.*]], label %[[BB46:.*]]
+! CHECK: br i1 [[TMP5]], label %[[BB6:.*]], label %[[ABSENT:.*]]
! CHECK: [[BB6]]:
! CHECK: [[TMP7:%.*]] = call i1 @_FortranAIsContiguous(ptr [[TMP0]])
! CHECK: [[TMP8:%.*]] = icmp eq i1 [[TMP7]], false
! CHECK: [[TMP13:%.*]] = and i1 [[TMP8]], [[TMP12:.*]]
-! CHECK: br i1 [[TMP13]], label %[[BB14:.*]], label %[[BB46]], !prof [[PROF2:![0-9]+]]
+! CHECK: br i1 [[TMP13]], label %[[BB14:.*]], label %[[NO_REPACK:.*]], !prof [[PROF2:![0-9]+]]
! CHECK: [[BB14]]:
! CHECK: call void @_FortranAShallowCopyDirect
-! CHECK: br label %[[BB46]]
+! CHECK: br label %[[BB46:.*]]
+! CHECK: [[NO_REPACK]]:
+! CHECK-NEXT: br label %[[BB46]]
! CHECK: [[BB46]]:
-! CHECK: br i1 [[TMP5]], label %[[BB48:.*]], label %[[BB57:.*]]
+! CHECK: br label %[[PRESENT:.*]]
+! CHECK: [[PRESENT]]:
+! CHECK-NEXT: br label %[[PRESENT_OR_ABSENT:.*]]
+! CHECK: [[ABSENT]]:
+! CHECK-NEXT: br label %[[PRESENT_OR_ABSENT]]
+! CHECK: [[PRESENT_OR_ABSENT]]:
+! CHECK: br label %[[COPY_BACK_TEST:.*]]
+! CHECK: [[COPY_BACK_TEST]]:
+! CHECK-NEXT: br i1 [[TMP5]], label %[[BB48:.*]], label %[[BB57:.*]]
! CHECK: [[BB48]]:
-! CHECK: br i1 [[TMP55:.*]], label %[[BB56:.*]], label %[[BB57]], !prof [[PROF2]]
+! CHECK: br i1 [[TMP55:.*]], label %[[BB56:.*]], label %[[NO_COPY_BACK:.*]], !prof [[PROF2]]
! CHECK: [[BB56]]:
! CHECK: call void @_FortranAShallowCopyDirect
-! CHECK: br label %[[BB57]]
+! CHECK: br label %[[NO_COPY_BACK]]
+! CHECK: [[NO_COPY_BACK]]:
+! CHECK-NEXT: br label %[[BB57]]
! CHECK: [[BB57]]:
! CHECK: ret void
! CHECK: [[PROF2]] = !{!"branch_weights", i32 0, i32 1}
diff --git a/flang/test/Transforms/o0-canonicalize.mlir b/flang/test/Transforms/o0-canonicalize.mlir
new file mode 100644
index 0000000000000..ea35343b2dd1b
--- /dev/null
+++ b/flang/test/Transforms/o0-canonicalize.mlir
@@ -0,0 +1,39 @@
+// RUN: fir-opt %s --fir-o0-canonicalize | FileCheck %s --check-prefix=O0 --implicit-check-not=arith.addi --implicit-check-not=arith.muli --implicit-check-not=func.call_indirect
+// RUN: fir-opt %s --canonicalize | FileCheck %s --check-prefix=CANONICAL
+
+// Other dialects' canonicalization patterns, folding, and dead operation
+// removal still run. In particular, the indirect-to-direct call rewrite is
+// implemented by CallIndirectOp::canonicalize, which TableGen registers as a
+// rewrite pattern, not a fold. CF methods and explicit patterns, including
+// constant branch simplification and single-predecessor merging, are omitted.
+// O0-LABEL: func.func @cleanup(
+// O0-SAME: %[[X:.*]]: i32)
+// O0: %[[TRUE:.*]] = arith.constant true
+// O0: %[[VALUE:.*]] = call @callee(%[[X]]) : (i32) -> i32
+// O0-NEXT: cf.cond_br %[[TRUE]], ^[[THEN:bb[0-9]+]], ^[[ELSE:bb[0-9]+]]
+// O0: ^[[THEN]]:
+// O0-NEXT: cf.br ^[[JOIN:bb[0-9]+]]
+// O0: ^[[ELSE]]:
+// O0-NEXT: cf.br ^[[JOIN]]
+// O0: ^[[JOIN]]:
+// O0-NEXT: return %[[VALUE]] : i32
+// CANONICAL-LABEL: func.func @cleanup(
+// CANONICAL-SAME: %[[X:.*]]: i32)
+// CANONICAL-NEXT: %[[VALUE:.*]] = call @callee(%[[X]]) : (i32) -> i32
+// CANONICAL-NEXT: return %[[VALUE]] : i32
+func.func @cleanup(%x: i32) -> i32 {
+ %zero = arith.constant 0 : i32
+ %true = arith.constant true
+ %sum = arith.addi %x, %zero : i32
+ %dead = arith.muli %x, %x : i32
+ %callee = func.constant @callee : (i32) -> i32
+ %value = func.call_indirect %callee(%sum) : (i32) -> i32
+ cf.cond_br %true, ^then, ^else
+^then:
+ cf.br ^join
+^else:
+ cf.br ^join
+^join:
+ return %value : i32
+}
+func.func private @callee(i32) -> i32
More information about the flang-commits
mailing list