[llvm] [ComplexDeinterleaving] Use FMF intersection instead of requiring strict equality (PR #219158)
Mattéo Rizza Murgier via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 02:44:01 PDT 2026
https://github.com/matteo-rm created https://github.com/llvm/llvm-project/pull/219158
`identifySymmetricOperation` and `identifyReassocNodes` currently require strict FMF equality between ops getting fused. Loosening the checks to only require the intersection of their flags to contain `reassoc` and only passing that intersection to the produced fused op enables further optimizations while being semantically correct.
>From 935370037640be39cd0464b7d2591c269d381654 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Wed, 26 Aug 2026 16:54:27 +0200
Subject: [PATCH] [ComplexDeinterleaving] Use FMF intersection instead of
requiring strict equality
---
.../lib/CodeGen/ComplexDeinterleavingPass.cpp | 36 +++----
...complex-deinterleaving-fmf-intersection.ll | 100 ++++++++++++++++++
2 files changed, 118 insertions(+), 18 deletions(-)
create mode 100644 llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll
diff --git a/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp b/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
index 38e1043aa9614..5dbb053be876b 100644
--- a/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
+++ b/llvm/lib/CodeGen/ComplexDeinterleavingPass.cpp
@@ -937,6 +937,9 @@ ComplexDeinterleavingGraph::CompositeNode *
ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
auto *FirstReal = cast<Instruction>(Vals[0].Real);
unsigned FirstOpc = FirstReal->getOpcode();
+ FastMathFlags CommonFlags;
+ if (isa<FPMathOperator>(FirstReal))
+ CommonFlags = FirstReal->getFastMathFlags();
for (auto &V : Vals) {
auto *Real = cast<Instruction>(V.Real);
auto *Imag = cast<Instruction>(V.Imag);
@@ -947,10 +950,10 @@ ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
!isInstructionPotentiallySymmetric(Imag))
return nullptr;
- if (isa<FPMathOperator>(FirstReal))
- if (Real->getFastMathFlags() != FirstReal->getFastMathFlags() ||
- Imag->getFastMathFlags() != FirstReal->getFastMathFlags())
- return nullptr;
+ if (isa<FPMathOperator>(FirstReal)) {
+ CommonFlags &= Real->getFastMathFlags();
+ CommonFlags &= Imag->getFastMathFlags();
+ }
}
ComplexValues OpVals;
@@ -981,7 +984,7 @@ ComplexDeinterleavingGraph::identifySymmetricOperation(ComplexValues &Vals) {
prepareCompositeNode(ComplexDeinterleavingOperation::Symmetric, Vals);
Node->Opcode = FirstReal->getOpcode();
if (isa<FPMathOperator>(FirstReal))
- Node->Flags = FirstReal->getFastMathFlags();
+ Node->Flags = CommonFlags;
Node->addOperand(Op0);
if (FirstReal->isBinaryOp())
@@ -1224,13 +1227,7 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
std::optional<FastMathFlags> Flags;
if (isa<FPMathOperator>(Real)) {
- if (Real->getFastMathFlags() != Imag->getFastMathFlags()) {
- LLVM_DEBUG(dbgs() << "The flags in Real and Imaginary instructions are "
- "not identical\n");
- return nullptr;
- }
-
- Flags = Real->getFastMathFlags();
+ Flags = Real->getFastMathFlags() & Imag->getFastMathFlags();
if (!Flags->allowReassoc()) {
LLVM_DEBUG(
dbgs()
@@ -1241,7 +1238,8 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
// Collect multiplications and addend instructions from the given instruction
// while traversing it operands. Additionally, verify that all instructions
- // have the same fast math flags.
+ // allow reassociation, and narrow \p Flags to the intersection of their
+ // flags.
auto Collect = [&Flags](Instruction *Insn, SmallVectorImpl<Product> &Muls,
AddendList &Addends) -> bool {
SmallVector<PointerIntPair<Value *, 1, bool>> Worklist = {{Insn, true}};
@@ -1334,11 +1332,13 @@ ComplexDeinterleavingGraph::identifyReassocNodes(Instruction *Real,
continue;
}
- if (Flags && I->getFastMathFlags() != *Flags) {
- LLVM_DEBUG(dbgs() << "The instruction's fast math flags are "
- "inconsistent with the root instructions' flags: "
- << *I << "\n");
- return false;
+ if (Flags) {
+ if (!I->getFastMathFlags().allowReassoc()) {
+ LLVM_DEBUG(dbgs() << "The instruction does not allow reassociation: "
+ << *I << "\n");
+ return false;
+ }
+ *Flags &= I->getFastMathFlags();
}
}
return true;
diff --git a/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll b/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll
new file mode 100644
index 0000000000000..482fb36aae254
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/complex-deinterleaving-fmf-intersection.ll
@@ -0,0 +1,100 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc < %s --mattr=+complxnum,+neon -o - | FileCheck %s
+
+target triple = "aarch64"
+
+; Expected to be transformed (and discard extra nnan flag)
+define <4 x double> @fcmla_mixed_fmf(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_mixed_fmf:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: fcmla v4.2d, v0.2d, v2.2d, #0
+; CHECK-NEXT: fcmla v5.2d, v1.2d, v3.2d, #0
+; CHECK-NEXT: fcmla v4.2d, v0.2d, v2.2d, #90
+; CHECK-NEXT: fcmla v5.2d, v1.2d, v3.2d, #90
+; CHECK-NEXT: mov v0.16b, v4.16b
+; CHECK-NEXT: mov v1.16b, v5.16b
+; CHECK-NEXT: ret
+entry:
+ %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %0 = fmul reassoc contract nsz <2 x double> %b.imag, %a.real
+ %1 = fmul reassoc contract nsz <2 x double> %b.real, %a.imag
+ %2 = fadd reassoc contract nsz nnan <2 x double> %0, %1
+ %3 = fmul reassoc contract nsz <2 x double> %b.real, %a.real
+ %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %4 = fadd reassoc contract nsz <2 x double> %c.real, %3
+ %5 = fmul reassoc contract nsz <2 x double> %b.imag, %a.imag
+ %6 = fsub reassoc contract nsz <2 x double> %4, %5
+ %7 = fadd reassoc contract nsz <2 x double> %2, %c.imag
+ %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+ ret <4 x double> %interleaved.vec
+}
+
+; Expected to transform (and discard extra ninf flags)
+define <4 x double> @fcmla_mixed_root_fmf(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_mixed_root_fmf:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: fcmla v4.2d, v0.2d, v2.2d, #0
+; CHECK-NEXT: fcmla v5.2d, v1.2d, v3.2d, #0
+; CHECK-NEXT: fcmla v4.2d, v0.2d, v2.2d, #90
+; CHECK-NEXT: fcmla v5.2d, v1.2d, v3.2d, #90
+; CHECK-NEXT: mov v0.16b, v4.16b
+; CHECK-NEXT: mov v1.16b, v5.16b
+; CHECK-NEXT: ret
+entry:
+ %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %0 = fmul reassoc contract <2 x double> %b.imag, %a.real
+ %1 = fmul reassoc contract <2 x double> %b.real, %a.imag
+ %2 = fadd reassoc contract <2 x double> %0, %1
+ %3 = fmul reassoc contract <2 x double> %b.real, %a.real
+ %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %4 = fadd reassoc contract ninf <2 x double> %c.real, %3
+ %5 = fmul reassoc contract <2 x double> %b.imag, %a.imag
+ %6 = fsub reassoc contract ninf <2 x double> %4, %5
+ %7 = fadd reassoc contract <2 x double> %2, %c.imag
+ %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+ ret <4 x double> %interleaved.vec
+}
+
+; Expected not to transform
+define <4 x double> @fcmla_missing_reassoc(<4 x double> %a, <4 x double> %b, <4 x double> %c) {
+; CHECK-LABEL: fcmla_missing_reassoc:
+; CHECK: // %bb.0: // %entry
+; CHECK-NEXT: zip1 v6.2d, v2.2d, v3.2d
+; CHECK-NEXT: zip2 v2.2d, v2.2d, v3.2d
+; CHECK-NEXT: zip2 v3.2d, v4.2d, v5.2d
+; CHECK-NEXT: zip1 v4.2d, v4.2d, v5.2d
+; CHECK-NEXT: zip1 v5.2d, v0.2d, v1.2d
+; CHECK-NEXT: zip2 v0.2d, v0.2d, v1.2d
+; CHECK-NEXT: fmla v3.2d, v5.2d, v2.2d
+; CHECK-NEXT: fmla v4.2d, v5.2d, v6.2d
+; CHECK-NEXT: fmla v3.2d, v0.2d, v6.2d
+; CHECK-NEXT: fmls v4.2d, v0.2d, v2.2d
+; CHECK-NEXT: zip1 v0.2d, v4.2d, v3.2d
+; CHECK-NEXT: zip2 v1.2d, v4.2d, v3.2d
+; CHECK-NEXT: ret
+entry:
+ %a.real = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %a.imag = shufflevector <4 x double> %a, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %b.real = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %b.imag = shufflevector <4 x double> %b, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %0 = fmul nsz <2 x double> %b.imag, %a.real
+ %1 = fmul reassoc contract <2 x double> %b.real, %a.imag
+ %2 = fadd reassoc contract <2 x double> %0, %1
+ %3 = fmul reassoc contract <2 x double> %b.real, %a.real
+ %c.real = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 0, i32 2>
+ %c.imag = shufflevector <4 x double> %c, <4 x double> poison, <2 x i32> <i32 1, i32 3>
+ %4 = fadd reassoc contract <2 x double> %c.real, %3
+ %5 = fmul reassoc contract <2 x double> %b.imag, %a.imag
+ %6 = fsub reassoc contract <2 x double> %4, %5
+ %7 = fadd reassoc contract <2 x double> %2, %c.imag
+ %interleaved.vec = shufflevector <2 x double> %6, <2 x double> %7, <4 x i32> <i32 0, i32 2, i32 1, i32 3>
+ ret <4 x double> %interleaved.vec
+}
More information about the llvm-commits
mailing list