[llvm] [ComplexDeinterleaving] Use FMF intersection instead of requiring strict equality (PR #219158)

Mattéo Rizza Murgier via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 02:44:01 PDT 2026


https://github.com/matteo-rm created https://github.com/llvm/llvm-project/pull/219158

`identifySymmetricOperation` and `identifyReassocNodes` currently require strict FMF equality between ops getting fused. Loosening the checks to only require the intersection of their flags to contain `reassoc` and only passing that intersection to the produced fused op enables further optimizations while being semantically correct.

>From 935370037640be39cd0464b7d2591c269d381654 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
 <matteo.rizza-murgier at sipearl.com>
Date: Wed, 26 Aug 2026 16:54:27 +0200
Subject: [PATCH] [ComplexDeinterleaving] Use FMF intersection instead of
 requiring strict equality

---
 .../lib/CodeGen/ComplexDeinterleavingPass.cpp |  36 +++----
 ...complex-deinterleaving-fmf-intersection.ll | 100 ++++++++++++++++++
 2 files changed, 118 insertions(+), 18 deletions(-)
 create mode 100644 llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll

diff --git a/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp b/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
index 38e1043aa9614..5dbb053be876b 100644
--- a/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
+++ b/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
@@ -937,6 +937,9 @@ ComplexDeinterleavingGraph::CompositeNode *
 ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
   auto *FirstReal = cast<Instruction>(Vals[0].Real);
   unsigned FirstOpc = FirstReal->getOpcode();
+  FastMathFlags CommonFlags;
+  if (isa<FPMathOperator>(FirstReal))
+    CommonFlags = FirstReal->getFastMathFlags();
   for (auto &V : Vals) {
     auto *Real = cast<Instruction>(V.Real);
     auto *Imag = cast<Instruction>(V.Imag);
@@ -947,10 +950,10 @@ ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
         !isInstructionPotentiallySymmetric(Imag))
       return nullptr;
 
-    if (isa<FPMathOperator>(FirstReal))
-      if (Real->getFastMathFlags() != FirstReal->getFastMathFlags() ||
-          Imag->getFastMathFlags() != FirstReal->getFastMathFlags())
-        return nullptr;
+    if (isa<FPMathOperator>(FirstReal)) {
+      CommonFlags &= Real->getFastMathFlags();
+      CommonFlags &= Imag->getFastMathFlags();
+    }
   }
 
   ComplexValues OpVals;
@@ -981,7 +984,7 @@ ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
       prepareCompositeNode(ComplexDeinterleavingOperation::Symmetric, Vals);
   Node->Opcode = FirstReal->getOpcode();
   if (isa<FPMathOperator>(FirstReal))
-    Node->Flags = FirstReal->getFastMathFlags();
+    Node->Flags = CommonFlags;
 
   Node->addOperand(Op0);
   if (FirstReal->isBinaryOp())
@@ -1224,13 +1227,7 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
 
   std::optional<FastMathFlags> Flags;
   if (isa<FPMathOperator>(Real)) {
-    if (Real->getFastMathFlags() != Imag->getFastMathFlags()) {
-      LLVM_DEBUG(dbgs() << "The flags in Real and Imaginary instructions are "
-                           "not identical\n");
-      return nullptr;
-    }
-
-    Flags = Real->getFastMathFlags();
+    Flags = Real->getFastMathFlags() & Imag->getFastMathFlags();
     if (!Flags->allowReassoc()) {
       LLVM_DEBUG(
           dbgs()
@@ -1241,7 +1238,8 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
 
   // Collect multiplications and addend instructions from the given instruction
   // while traversing it operands. Additionally, verify that all instructions
-  // have the same fast math flags.
+  // allow reassociation, and narrow \p Flags to the intersection of their
+  // flags.
   auto Collect = [&Flags](Instruction *Insn, SmallVectorImpl<Product> &Muls,
                           AddendList &Addends) -> bool {
     SmallVector<PointerIntPair<Value *, 1, bool>> Worklist = {{Insn, true}};
@@ -1334,11 +1332,13 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
         continue;
       }
 
-      if (Flags && I->getFastMathFlags() != *Flags) {
-        LLVM_DEBUG(dbgs() << "The instruction's fast math flags are "
-                             "inconsistent with the root instructions' flags: "
-                          << *I << "\n");
-        return false;
+      if (Flags) {
+        if (!I->getFastMathFlags().allowReassoc()) {
+          LLVM_DEBUG(dbgs() << "The instruction does not allow reassociation: "
+                            << *I << "\n");
+          return false;
+        }
+        *Flags &= I->getFastMathFlags();
       }
     }
     return true;
diff --git a/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll b/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll
new file mode 100644
index 0000000000000..482fb36aae254
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll
@@ -0,0 +1,100 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s --mattr=+complxnum,+neon -o - | FileCheck %s
+
+target triple = "aarch64"
+
+; Expected to be transformed (and discard extra nnan flag)
+define <4 x double> @fcmla_mixed_fmf(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_mixed_fmf:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    fcmla v4.2d, v0.2d, v2.2d, #0
+; CHECK-NEXT:    fcmla v5.2d, v1.2d, v3.2d, #0
+; CHECK-NEXT:    fcmla v4.2d, v0.2d, v2.2d, #90
+; CHECK-NEXT:    fcmla v5.2d, v1.2d, v3.2d, #90
+; CHECK-NEXT:    mov v0.16b, v4.16b
+; CHECK-NEXT:    mov v1.16b, v5.16b
+; CHECK-NEXT:    ret
+entry:
+  %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %0 = fmul reassoc contract nsz <2 x double> %b.imag, %a.real
+  %1 = fmul reassoc contract nsz <2 x double> %b.real, %a.imag
+  %2 = fadd reassoc contract nsz nnan <2 x double> %0, %1
+  %3 = fmul reassoc contract nsz <2 x double> %b.real, %a.real
+  %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %4 = fadd reassoc contract nsz <2 x double> %c.real, %3
+  %5 = fmul reassoc contract nsz <2 x double> %b.imag, %a.imag
+  %6 = fsub reassoc contract nsz <2 x double> %4, %5
+  %7 = fadd reassoc contract nsz <2 x double> %2, %c.imag
+  %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+  ret <4 x double> %interleaved.vec
+}
+
+; Expected to transform (and discard extra ninf flags)
+define <4 x double> @fcmla_mixed_root_fmf(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_mixed_root_fmf:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    fcmla v4.2d, v0.2d, v2.2d, #0
+; CHECK-NEXT:    fcmla v5.2d, v1.2d, v3.2d, #0
+; CHECK-NEXT:    fcmla v4.2d, v0.2d, v2.2d, #90
+; CHECK-NEXT:    fcmla v5.2d, v1.2d, v3.2d, #90
+; CHECK-NEXT:    mov v0.16b, v4.16b
+; CHECK-NEXT:    mov v1.16b, v5.16b
+; CHECK-NEXT:    ret
+entry:
+  %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %0 = fmul reassoc contract <2 x double> %b.imag, %a.real
+  %1 = fmul reassoc contract <2 x double> %b.real, %a.imag
+  %2 = fadd reassoc contract <2 x double> %0, %1
+  %3 = fmul reassoc contract <2 x double> %b.real, %a.real
+  %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %4 = fadd reassoc contract ninf <2 x double> %c.real, %3
+  %5 = fmul reassoc contract <2 x double> %b.imag, %a.imag
+  %6 = fsub reassoc contract ninf <2 x double> %4, %5
+  %7 = fadd reassoc contract <2 x double> %2, %c.imag
+  %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+  ret <4 x double> %interleaved.vec
+}
+
+; Expected not to transform
+define <4 x double> @fcmla_missing_reassoc(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_missing_reassoc:
+; CHECK:       // %bb.0: // %entry
+; CHECK-NEXT:    zip1 v6.2d, v2.2d, v3.2d
+; CHECK-NEXT:    zip2 v2.2d, v2.2d, v3.2d
+; CHECK-NEXT:    zip2 v3.2d, v4.2d, v5.2d
+; CHECK-NEXT:    zip1 v4.2d, v4.2d, v5.2d
+; CHECK-NEXT:    zip1 v5.2d, v0.2d, v1.2d
+; CHECK-NEXT:    zip2 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT:    fmla v3.2d, v5.2d, v2.2d
+; CHECK-NEXT:    fmla v4.2d, v5.2d, v6.2d
+; CHECK-NEXT:    fmla v3.2d, v0.2d, v6.2d
+; CHECK-NEXT:    fmls v4.2d, v0.2d, v2.2d
+; CHECK-NEXT:    zip1 v0.2d, v4.2d, v3.2d
+; CHECK-NEXT:    zip2 v1.2d, v4.2d, v3.2d
+; CHECK-NEXT:    ret
+entry:
+  %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %0 = fmul nsz <2 x double> %b.imag, %a.real
+  %1 = fmul reassoc contract <2 x double> %b.real, %a.imag
+  %2 = fadd reassoc contract <2 x double> %0, %1
+  %3 = fmul reassoc contract <2 x double> %b.real, %a.real
+  %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+  %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+  %4 = fadd reassoc contract <2 x double> %c.real, %3
+  %5 = fmul reassoc contract <2 x double> %b.imag, %a.imag
+  %6 = fsub reassoc contract <2 x double> %4, %5
+  %7 = fadd reassoc contract <2 x double> %2, %c.imag
+  %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+  ret <4 x double> %interleaved.vec
+}



More information about the llvm-commits mailing list