[llvm] [SLP] Support FAdd/FSub as interchangeable instructions (PR #208002)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Tue Jul 7 07:23:00 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/208002

>From b29c64a3216bb01e1d2d54805d14b27c7c874ca4 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Tue, 7 Jul 2026 06:49:37 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 171 +++++++++++-------
 .../AArch64/vec3-reorder-reshuffle.ll         |  24 +--
 .../RISCV/unordered-loads-operands.ll         |  26 +--
 .../X86/buildvector-shuffle-with-root.ll      |   6 +-
 .../SLPVectorizer/X86/crash_clear_undefs.ll   |   8 +-
 .../X86/gathered-loads-non-full-reg.ll        |  45 +++--
 .../SLPVectorizer/X86/reorder-vf-to-resize.ll |   7 +-
 .../X86/reorder_with_external_users.ll        |  21 +--
 .../X86/vec3-reorder-reshuffle.ll             |  15 +-
 .../X86/vect_copyable_in_binops.ll            |  34 +---
 .../X86/vectorize-widest-phis.ll              |   7 +-
 .../SLPVectorizer/jumbled_store_crash.ll      |   6 +-
 .../SLPVectorizer/semanticly-same.ll          |  34 ++++
 13 files changed, 218 insertions(+), 186 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index ea97a18403dc8..e00ccd34259b4 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -1037,8 +1037,9 @@ class BinOpSameOpcodeHelper {
   using MaskType = std::uint_fast32_t;
   /// Sort SupportedOp because it is used by binary_search.
   constexpr static unsigned SupportedOp[] = {
-      Instruction::Add,  Instruction::Sub, Instruction::Mul, Instruction::Shl,
-      Instruction::AShr, Instruction::And, Instruction::Or,  Instruction::Xor};
+      Instruction::Add, Instruction::FAdd, Instruction::Sub,  Instruction::FSub,
+      Instruction::Mul, Instruction::Shl,  Instruction::AShr, Instruction::And,
+      Instruction::Or,  Instruction::Xor};
   static_assert(llvm::is_sorted_constexpr(SupportedOp) &&
                 "SupportedOp is not sorted.");
   enum : MaskType {
@@ -1050,34 +1051,42 @@ class BinOpSameOpcodeHelper {
     AndBIT = 1 << 5,
     OrBIT = 1 << 6,
     XorBIT = 1 << 7,
-    MainOpBIT = 1 << 8,
+    FAddBIT = 1 << 8,
+    FSubBIT = 1 << 9,
+    MainOpBIT = 1 << 10,
     LLVM_MARK_AS_BITMASK_ENUM(MainOpBIT)
   };
-  /// Return a non-nullptr if either operand of I is a ConstantInt.
+  /// Return a non-nullptr if either operand of I is a ConstantInt (for the
+  /// integer opcodes) or a ConstantFP (for FAdd/FSub).
   /// The second return value represents the operand position. We check the
-  /// right-hand side first (1). If the right hand side is not a ConstantInt and
-  /// the instruction is neither Sub, Shl, nor AShr, we then check the left hand
-  /// side (0).
-  static std::pair<ConstantInt *, unsigned>
+  /// right-hand side first (1). If the right hand side is not a constant and
+  /// the instruction is neither Sub, FSub, Shl, nor AShr, we then check the
+  /// left hand side (0).
+  static std::pair<Constant *, unsigned>
   isBinOpWithConstantInt(const Instruction *I) {
     unsigned Opcode = I->getOpcode();
     assert(binary_search(SupportedOp, Opcode) && "Unsupported opcode.");
     (void)SupportedOp;
     auto *BinOp = cast<BinaryOperator>(I);
-    if (auto *CI = dyn_cast<ConstantInt>(BinOp->getOperand(1)))
-      return {CI, 1};
-    if (Opcode == Instruction::Sub || Opcode == Instruction::Shl ||
-        Opcode == Instruction::AShr)
+    auto GetConstant = [](Value *V) -> Constant * {
+      if (auto *CI = dyn_cast<ConstantInt>(V))
+        return CI;
+      return dyn_cast<ConstantFP>(V);
+    };
+    if (Constant *C = GetConstant(BinOp->getOperand(1)))
+      return {C, 1};
+    if (Opcode == Instruction::Sub || Opcode == Instruction::FSub ||
+        Opcode == Instruction::Shl || Opcode == Instruction::AShr)
       return {nullptr, 0};
-    if (auto *CI = dyn_cast<ConstantInt>(BinOp->getOperand(0)))
-      return {CI, 0};
+    if (Constant *C = GetConstant(BinOp->getOperand(0)))
+      return {C, 0};
     return {nullptr, 0};
   }
   struct InterchangeableInfo {
     const Instruction *I = nullptr;
     /// The bit it sets represents whether MainOp can be converted to.
     MaskType Mask = MainOpBIT | XorBIT | OrBIT | AndBIT | SubBIT | AddBIT |
-                    MulBIT | AShrBIT | ShlBIT;
+                    MulBIT | AShrBIT | ShlBIT | FSubBIT | FAddBIT;
     /// We cannot create an interchangeable instruction that does not exist in
     /// VL. For example, VL [x + 0, y * 1] can be converted to [x << 0, y << 0],
     /// but << does not exist in VL. In the end, we convert VL to [x * 1, y *
@@ -1112,6 +1121,10 @@ class BinOpSameOpcodeHelper {
         return Instruction::Add;
       if (Candidate & SubBIT)
         return Instruction::Sub;
+      if (Candidate & FAddBIT)
+        return Instruction::FAdd;
+      if (Candidate & FSubBIT)
+        return Instruction::FSub;
       if (Candidate & AndBIT)
         return Instruction::And;
       if (Candidate & OrBIT)
@@ -1143,9 +1156,11 @@ class BinOpSameOpcodeHelper {
         return Candidate & OrBIT;
       case Instruction::Xor:
         return Candidate & XorBIT;
-      case Instruction::LShr:
       case Instruction::FAdd:
+        return Candidate & FAddBIT;
       case Instruction::FSub:
+        return Candidate & FSubBIT;
+      case Instruction::LShr:
       case Instruction::FMul:
       case Instruction::SDiv:
       case Instruction::UDiv:
@@ -1166,59 +1181,69 @@ class BinOpSameOpcodeHelper {
       if (FromOpcode == ToOpcode)
         return SmallVector<Value *>(I->operands());
       assert(binary_search(SupportedOp, ToOpcode) && "Unsupported opcode.");
-      auto [CI, Pos] = isBinOpWithConstantInt(I);
-      const APInt &FromCIValue = CI->getValue();
-      unsigned FromCIValueBitWidth = FromCIValue.getBitWidth();
+      auto [C, Pos] = isBinOpWithConstantInt(I);
       Type *RHSType = I->getOperand(Pos)->getType();
       Constant *RHS;
-      switch (FromOpcode) {
-      case Instruction::Shl:
-        if (ToOpcode == Instruction::Add && FromCIValue.isOne())
-          return {I->getOperand(0), I->getOperand(0)};
-        if (ToOpcode == Instruction::Mul) {
-          RHS = ConstantInt::get(
-              RHSType, APInt::getOneBitSet(FromCIValueBitWidth,
-                                           FromCIValue.getZExtValue()));
-        } else {
-          assert(FromCIValue.isZero() && "Cannot convert the instruction.");
-          RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
-                                               /*AllowRHSConstant=*/true);
-        }
-        break;
-      case Instruction::Mul:
-        assert(FromCIValue.isPowerOf2() && "Cannot convert the instruction.");
-        if (ToOpcode == Instruction::Shl) {
-          RHS = ConstantInt::get(
-              RHSType, APInt(FromCIValueBitWidth, FromCIValue.logBase2()));
-        } else {
-          assert(FromCIValue.isOne() && "Cannot convert the instruction.");
+      if (auto *CFP = dyn_cast<ConstantFP>(C)) {
+        // fsub(x, c) == fadd(x, -c) for every FP constant c, since IEEE 754
+        // defines subtraction as addition of the negated operand.
+        assert(is_contained({Instruction::FAdd, Instruction::FSub}, ToOpcode) &&
+               "Cannot convert the instruction.");
+        RHS = ConstantFP::get(RHSType, -CFP->getValueAPF());
+      } else {
+        auto *CI = cast<ConstantInt>(C);
+        const APInt &FromCIValue = CI->getValue();
+        unsigned FromCIValueBitWidth = FromCIValue.getBitWidth();
+        switch (FromOpcode) {
+        case Instruction::Shl:
+          if (ToOpcode == Instruction::Add && FromCIValue.isOne())
+            return {I->getOperand(0), I->getOperand(0)};
+          if (ToOpcode == Instruction::Mul) {
+            RHS = ConstantInt::get(
+                RHSType, APInt::getOneBitSet(FromCIValueBitWidth,
+                                             FromCIValue.getZExtValue()));
+          } else {
+            assert(FromCIValue.isZero() && "Cannot convert the instruction.");
+            RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
+                                                 /*AllowRHSConstant=*/true);
+          }
+          break;
+        case Instruction::Mul:
+          assert(FromCIValue.isPowerOf2() && "Cannot convert the instruction.");
+          if (ToOpcode == Instruction::Shl) {
+            RHS = ConstantInt::get(
+                RHSType, APInt(FromCIValueBitWidth, FromCIValue.logBase2()));
+          } else {
+            assert(FromCIValue.isOne() && "Cannot convert the instruction.");
+            RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
+                                                 /*AllowRHSConstant=*/true);
+          }
+          break;
+        case Instruction::Add:
+        case Instruction::Sub:
+          if (FromCIValue.isZero()) {
+            RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
+                                                 /*AllowRHSConstant=*/true);
+          } else {
+            assert(
+                is_contained({Instruction::Add, Instruction::Sub}, ToOpcode) &&
+                "Cannot convert the instruction.");
+            APInt NegatedVal = APInt(FromCIValue);
+            NegatedVal.negate();
+            RHS = ConstantInt::get(RHSType, NegatedVal);
+          }
+          break;
+        case Instruction::And:
+          assert(FromCIValue.isAllOnes() && "Cannot convert the instruction.");
           RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
                                                /*AllowRHSConstant=*/true);
-        }
-        break;
-      case Instruction::Add:
-      case Instruction::Sub:
-        if (FromCIValue.isZero()) {
+          break;
+        default:
+          assert(FromCIValue.isZero() && "Cannot convert the instruction.");
           RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
                                                /*AllowRHSConstant=*/true);
-        } else {
-          assert(is_contained({Instruction::Add, Instruction::Sub}, ToOpcode) &&
-                 "Cannot convert the instruction.");
-          APInt NegatedVal = APInt(FromCIValue);
-          NegatedVal.negate();
-          RHS = ConstantInt::get(RHSType, NegatedVal);
+          break;
         }
-        break;
-      case Instruction::And:
-        assert(FromCIValue.isAllOnes() && "Cannot convert the instruction.");
-        RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
-                                             /*AllowRHSConstant=*/true);
-        break;
-      default:
-        assert(FromCIValue.isZero() && "Cannot convert the instruction.");
-        RHS = ConstantExpr::getBinOpIdentity(ToOpcode, RHSType,
-                                             /*AllowRHSConstant=*/true);
-        break;
       }
       Value *LHS = I->getOperand(1 - Pos);
       // If the target opcode is non-commutative (e.g., shl, sub),
@@ -1253,7 +1278,8 @@ class BinOpSameOpcodeHelper {
            "BinOpSameOpcodeHelper only accepts BinaryOperator.");
     unsigned Opcode = I->getOpcode();
     MaskType OpcodeInMaskForm;
-    // Prefer Shl, AShr, Mul, Add, Sub, And, Or and Xor over MainOp.
+    // Prefer Shl, AShr, Mul, Add, Sub, And, Or, Xor, FAdd and FSub over
+    // MainOp.
     switch (Opcode) {
     case Instruction::Shl:
       OpcodeInMaskForm = ShlBIT;
@@ -1279,13 +1305,19 @@ class BinOpSameOpcodeHelper {
     case Instruction::Xor:
       OpcodeInMaskForm = XorBIT;
       break;
+    case Instruction::FAdd:
+      OpcodeInMaskForm = FAddBIT;
+      break;
+    case Instruction::FSub:
+      OpcodeInMaskForm = FSubBIT;
+      break;
     default:
       return MainOp.equal(Opcode) ||
              (initializeAltOp(I) && AltOp.equal(Opcode));
     }
     MaskType InterchangeableMask = OpcodeInMaskForm;
-    ConstantInt *CI = isBinOpWithConstantInt(I).first;
-    if (CI) {
+    auto [C, Pos] = isBinOpWithConstantInt(I);
+    if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
       constexpr MaskType CanBeAll =
           XorBIT | OrBIT | AndBIT | SubBIT | AddBIT | MulBIT | AShrBIT | ShlBIT;
       const APInt &CIValue = CI->getValue();
@@ -1321,6 +1353,15 @@ class BinOpSameOpcodeHelper {
           InterchangeableMask = CanBeAll;
         break;
       }
+    } else if (C && Pos == 1) {
+      // FAdd/FSub with a constant RHS: negating the constant always
+      // converts one into the other, so no value check is needed. A
+      // constant LHS (Pos == 0, e.g. "0.0 - x") is excluded: unlike a
+      // constant RHS, it cannot be moved to the other opcode without also
+      // swapping the variable operand, which would misalign it against
+      // lanes that keep their native opcode (their variable operand stays
+      // on the other side).
+      InterchangeableMask = FSubBIT | FAddBIT;
     }
     return MainOp.trySet(OpcodeInMaskForm, InterchangeableMask) ||
            (initializeAltOp(I) &&
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/vec3-reorder-reshuffle.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/vec3-reorder-reshuffle.ll
index 2f59022ddae9b..f616964584d98 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/vec3-reorder-reshuffle.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/vec3-reorder-reshuffle.ll
@@ -445,13 +445,11 @@ define void @reuse_shuffle_indices_cost_crash_3(ptr %m, double %conv, double %co
 ; CHECK-LABEL: define void @reuse_shuffle_indices_cost_crash_3(
 ; CHECK-SAME: ptr [[M:%.*]], double [[CONV:%.*]], double [[CONV2:%.*]]) {
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[SUB19:%.*]] = fsub double 0.000000e+00, [[CONV2]]
-; CHECK-NEXT:    [[CONV20:%.*]] = fptrunc double [[SUB19]] to float
-; CHECK-NEXT:    store float [[CONV20]], ptr [[M]], align 4
-; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV]], 0.000000e+00
-; CHECK-NEXT:    [[CONV239:%.*]] = fptrunc double [[ADD]] to float
-; CHECK-NEXT:    [[ARRAYIDX25:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 1
-; CHECK-NEXT:    store float [[CONV239]], ptr [[ARRAYIDX25]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> <double 0.000000e+00, double poison>, double [[CONV]], i32 1
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> <double poison, double -0.000000e+00>, double [[CONV2]], i32 0
+; CHECK-NEXT:    [[TMP2:%.*]] = fsub <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = fptrunc <2 x double> [[TMP2]] to <2 x float>
+; CHECK-NEXT:    store <2 x float> [[TMP3]], ptr [[M]], align 4
 ; CHECK-NEXT:    [[ADD26:%.*]] = fsub double [[CONV]], [[CONV]]
 ; CHECK-NEXT:    [[CONV27:%.*]] = fptrunc double [[ADD26]] to float
 ; CHECK-NEXT:    [[ARRAYIDX29:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 2
@@ -524,14 +522,12 @@ define void @common_mask(ptr %m, double %conv, double %conv2) {
 ; CHECK-NEXT:    [[SUB19:%.*]] = fsub double [[CONV]], [[CONV]]
 ; CHECK-NEXT:    [[CONV20:%.*]] = fptrunc double [[SUB19]] to float
 ; CHECK-NEXT:    store float [[CONV20]], ptr [[M]], align 4
-; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV2]], 0.000000e+00
-; CHECK-NEXT:    [[CONV239:%.*]] = fptrunc double [[ADD]] to float
 ; CHECK-NEXT:    [[ARRAYIDX25:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 1
-; CHECK-NEXT:    store float [[CONV239]], ptr [[ARRAYIDX25]], align 4
-; CHECK-NEXT:    [[ADD26:%.*]] = fsub double 0.000000e+00, [[CONV]]
-; CHECK-NEXT:    [[CONV27:%.*]] = fptrunc double [[ADD26]] to float
-; CHECK-NEXT:    [[ARRAYIDX29:%.*]] = getelementptr [4 x float], ptr [[M]], i64 0, i64 2
-; CHECK-NEXT:    store float [[CONV27]], ptr [[ARRAYIDX29]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> <double poison, double 0.000000e+00>, double [[CONV2]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> <double -0.000000e+00, double poison>, double [[CONV]], i32 1
+; CHECK-NEXT:    [[TMP2:%.*]] = fsub <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = fptrunc <2 x double> [[TMP2]] to <2 x float>
+; CHECK-NEXT:    store <2 x float> [[TMP3]], ptr [[ARRAYIDX25]], align 4
 ; CHECK-NEXT:    ret void
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/unordered-loads-operands.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/unordered-loads-operands.ll
index 00f3a9e322f23..d6eb170397713 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/unordered-loads-operands.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/unordered-loads-operands.ll
@@ -8,30 +8,30 @@ define void @test(ptr %mdct_forward_x) {
 ; CHECK-NEXT:    br label %[[FOR_COND:.*]]
 ; CHECK:       [[FOR_COND]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = load ptr, ptr [[MDCT_FORWARD_X]], align 8
+; CHECK-NEXT:    [[ADD_PTR_I:%.*]] = getelementptr i8, ptr [[TMP0]], i64 24
 ; CHECK-NEXT:    [[ARRAYIDX2_I_I:%.*]] = getelementptr i8, ptr [[TMP0]], i64 32
 ; CHECK-NEXT:    [[ARRAYIDX5_I_I:%.*]] = getelementptr i8, ptr [[TMP0]], i64 40
-; CHECK-NEXT:    [[ADD_PTR_I:%.*]] = getelementptr i8, ptr [[TMP0]], i64 24
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x ptr> poison, ptr [[TMP0]], i32 0
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <4 x ptr> [[TMP1]], <4 x ptr> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr i8, <4 x ptr> [[TMP2]], <4 x i64> <i64 28, i64 36, i64 24, i64 28>
+; CHECK-NEXT:    [[ARRAYIDX10_I_I:%.*]] = getelementptr i8, ptr [[TMP0]], i64 28
 ; CHECK-NEXT:    [[TMP5:%.*]] = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 4 [[ADD_PTR_I]], <3 x i1> <i1 true, i1 false, i1 true>, <3 x float> poison)
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <3 x float> [[TMP5]], <3 x float> poison, <2 x i32> <i32 2, i32 0>
 ; CHECK-NEXT:    [[TMP6:%.*]] = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 4 [[ARRAYIDX5_I_I]], <3 x i1> <i1 true, i1 false, i1 true>, <3 x float> poison)
-; CHECK-NEXT:    [[TMP8:%.*]] = call <4 x float> @llvm.masked.gather.v4f32.v4p0(<4 x ptr> align 4 [[TMP3]], <4 x i1> splat (i1 true), <4 x float> poison)
 ; CHECK-NEXT:    [[TMP9:%.*]] = shufflevector <3 x float> [[TMP6]], <3 x float> poison, <4 x i32> <i32 2, i32 0, i32 2, i32 2>
 ; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <3 x float> [[TMP5]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x float> <float poison, float poison, float 0.000000e+00, float poison>, <4 x float> [[TMP22]], <4 x i32> <i32 poison, i32 poison, i32 2, i32 6>
+; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x float> <float poison, float poison, float -0.000000e+00, float poison>, <4 x float> [[TMP22]], <4 x i32> <i32 poison, i32 poison, i32 2, i32 6>
 ; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <4 x float> [[TMP11]], <4 x float> [[TMP10]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
 ; CHECK-NEXT:    [[TMP13:%.*]] = fsub <4 x float> [[TMP9]], [[TMP12]]
-; CHECK-NEXT:    [[TMP14:%.*]] = fadd <4 x float> [[TMP9]], [[TMP12]]
-; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <4 x float> [[TMP13]], <4 x float> [[TMP14]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
-; CHECK-NEXT:    [[TMP16:%.*]] = fsub <4 x float> zeroinitializer, [[TMP8]]
-; CHECK-NEXT:    [[TMP17:%.*]] = fadd <4 x float> zeroinitializer, [[TMP8]]
-; CHECK-NEXT:    [[TMP18:%.*]] = shufflevector <4 x float> [[TMP16]], <4 x float> [[TMP17]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
+; CHECK-NEXT:    [[TMP18:%.*]] = call <3 x float> @llvm.masked.load.v3f32.p0(ptr align 4 [[ARRAYIDX10_I_I]], <3 x i1> <i1 true, i1 false, i1 true>, <3 x float> poison)
+; CHECK-NEXT:    [[TMP23:%.*]] = shufflevector <3 x float> [[TMP18]], <3 x float> poison, <2 x i32> <i32 0, i32 2>
+; CHECK-NEXT:    [[TMP24:%.*]] = shufflevector <4 x float> <float 0.000000e+00, float 0.000000e+00, float poison, float 0.000000e+00>, <4 x float> [[TMP22]], <4 x i32> <i32 0, i32 1, i32 4, i32 3>
+; CHECK-NEXT:    [[TMP25:%.*]] = shufflevector <2 x float> [[TMP23]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP14:%.*]] = shufflevector <3 x float> [[TMP18]], <3 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 poison>
+; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <4 x float> <float poison, float poison, float -0.000000e+00, float poison>, <4 x float> [[TMP14]], <4 x i32> <i32 poison, i32 poison, i32 2, i32 4>
+; CHECK-NEXT:    [[TMP16:%.*]] = shufflevector <4 x float> [[TMP15]], <4 x float> [[TMP25]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
+; CHECK-NEXT:    [[TMP17:%.*]] = fsub <4 x float> [[TMP24]], [[TMP16]]
 ; CHECK-NEXT:    store float 0.000000e+00, ptr [[ADD_PTR_I]], align 4
-; CHECK-NEXT:    [[TMP19:%.*]] = fsub <4 x float> [[TMP15]], [[TMP18]]
-; CHECK-NEXT:    [[TMP20:%.*]] = fadd <4 x float> [[TMP15]], [[TMP18]]
+; CHECK-NEXT:    [[TMP19:%.*]] = fsub <4 x float> [[TMP13]], [[TMP17]]
+; CHECK-NEXT:    [[TMP20:%.*]] = fadd <4 x float> [[TMP13]], [[TMP17]]
 ; CHECK-NEXT:    [[TMP21:%.*]] = shufflevector <4 x float> [[TMP19]], <4 x float> [[TMP20]], <4 x i32> <i32 0, i32 5, i32 2, i32 3>
 ; CHECK-NEXT:    store <4 x float> [[TMP21]], ptr [[ARRAYIDX2_I_I]], align 4
 ; CHECK-NEXT:    br label %[[FOR_COND]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/buildvector-shuffle-with-root.ll b/llvm/test/Transforms/SLPVectorizer/X86/buildvector-shuffle-with-root.ll
index 6374cddc7346c..d9ce53a86f32d 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/buildvector-shuffle-with-root.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/buildvector-shuffle-with-root.ll
@@ -8,12 +8,10 @@ define void @test(i16 %arg) {
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i16> <i16 0, i16 poison>, i16 [[ARG]], i32 1
 ; CHECK-NEXT:    [[TMP1:%.*]] = sitofp <2 x i16> [[TMP0]] to <2 x float>
 ; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x float> [[TMP2]], <4 x float> <float 0.000000e+00, float poison, float poison, float poison>, <4 x i32> <i32 4, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x float> [[TMP2]], <4 x float> <float -0.000000e+00, float poison, float poison, float poison>, <4 x i32> <i32 4, i32 1, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <4 x float> [[TMP3]], <4 x float> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 1>
-; CHECK-NEXT:    [[TMP5:%.*]] = fadd <4 x float> zeroinitializer, [[TMP4]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = fsub <4 x float> zeroinitializer, [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <4 x float> [[TMP5]], <4 x float> [[TMP6]], <4 x i32> <i32 0, i32 5, i32 6, i32 7>
-; CHECK-NEXT:    [[TMP8:%.*]] = fsub <4 x float> [[TMP7]], [[TMP2]]
+; CHECK-NEXT:    [[TMP8:%.*]] = fsub <4 x float> [[TMP6]], [[TMP2]]
 ; CHECK-NEXT:    store <4 x float> [[TMP8]], ptr addrspace(1) getelementptr inbounds (i8, ptr addrspace(1) null, i64 20), align 4
 ; CHECK-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/crash_clear_undefs.ll b/llvm/test/Transforms/SLPVectorizer/X86/crash_clear_undefs.ll
index c2369a6a89ec1..b5478ef3fb622 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/crash_clear_undefs.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/crash_clear_undefs.ll
@@ -9,7 +9,7 @@ target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-f80:128-n8:16
 ; YAML-NEXT:  Function:        foo
 ; YAML-NEXT:  Args:
 ; YAML-NEXT:    - String:          'SLP vectorized with cost '
-; YAML-NEXT:    - Cost:            '-4'
+; YAML-NEXT:    - Cost:            '-6'
 ; YAML-NEXT:    - String:          ' and with tree size '
 ; YAML-NEXT:    - TreeSize:        '10'
 ; YAML-NEXT:  ...
@@ -23,11 +23,9 @@ define i1 @foo() {
 ; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x float> [[TMP4]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP6:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP7:%.*]] = select <4 x i1> zeroinitializer, <4 x float> [[TMP5]], <4 x float> [[TMP6]]
-; CHECK-NEXT:    [[TMP8:%.*]] = fadd <4 x float> [[TMP7]], zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = fsub <4 x float> [[TMP7]], zeroinitializer
-; CHECK-NEXT:    [[TMP10:%.*]] = shufflevector <4 x float> [[TMP8]], <4 x float> [[TMP9]], <4 x i32> <i32 0, i32 1, i32 6, i32 7>
+; CHECK-NEXT:    [[TMP10:%.*]] = fadd <4 x float> <float 0.000000e+00, float 0.000000e+00, float -0.000000e+00, float -0.000000e+00>, [[TMP7]]
 ; CHECK-NEXT:    br label [[TMP11]]
-; CHECK:       11:
+; CHECK:       9:
 ; CHECK-NEXT:    [[TMP12:%.*]] = phi <4 x float> [ [[TMP10]], [[TMP2]] ], [ zeroinitializer, [[TMP0:%.*]] ]
 ; CHECK-NEXT:    ret i1 false
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/gathered-loads-non-full-reg.ll b/llvm/test/Transforms/SLPVectorizer/X86/gathered-loads-non-full-reg.ll
index 6416d5a6062c7..113dc804090db 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/gathered-loads-non-full-reg.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/gathered-loads-non-full-reg.ll
@@ -37,39 +37,36 @@ define void @test(ptr noalias %0) {
 ; CHECK-NEXT:    [[TMP20:%.*]] = load double, ptr [[TMP12]], align 8
 ; CHECK-NEXT:    [[TMP36:%.*]] = fadd double [[TMP20]], [[TMP21]]
 ; CHECK-NEXT:    [[TMP34:%.*]] = fmul double [[TMP22]], [[DOTNEG969]]
+; CHECK-NEXT:    [[TMP38:%.*]] = load double, ptr [[TMP0]], align 8
+; CHECK-NEXT:    [[TMP40:%.*]] = load double, ptr [[TMP13]], align 8
+; CHECK-NEXT:    [[TMP33:%.*]] = load double, ptr [[TMP14]], align 8
 ; CHECK-NEXT:    [[TMP35:%.*]] = fmul double [[TMP34]], 0.000000e+00
 ; CHECK-NEXT:    [[TMP32:%.*]] = fsub double [[TMP29]], [[TMP19]]
-; CHECK-NEXT:    [[TMP37:%.*]] = fmul double [[TMP32]], [[TMP23]]
-; CHECK-NEXT:    [[TMP38:%.*]] = fmul double [[TMP37]], 0.000000e+00
-; CHECK-NEXT:    [[TMP30:%.*]] = load double, ptr [[TMP0]], align 8
-; CHECK-NEXT:    [[TMP40:%.*]] = load double, ptr [[TMP13]], align 8
 ; CHECK-NEXT:    [[TMP47:%.*]] = fmul double [[TMP40]], [[TMP35]]
-; CHECK-NEXT:    [[TMP60:%.*]] = load double, ptr [[TMP14]], align 8
-; CHECK-NEXT:    [[TMP39:%.*]] = fadd double [[TMP47]], 0.000000e+00
-; CHECK-NEXT:    [[TMP61:%.*]] = fmul double [[TMP60]], [[TMP38]]
-; CHECK-NEXT:    store double [[TMP39]], ptr getelementptr inbounds (i8, ptr @solid_, i64 408), align 8
+; CHECK-NEXT:    [[TMP37:%.*]] = fmul double [[TMP32]], [[TMP23]]
 ; CHECK-NEXT:    [[TMP41:%.*]] = fmul double [[TMP36]], [[TMP31]]
 ; CHECK-NEXT:    [[TMP42:%.*]] = fsub double 0.000000e+00, [[TMP25]]
 ; CHECK-NEXT:    [[TMP49:%.*]] = insertelement <2 x double> poison, double [[TMP42]], i32 0
 ; CHECK-NEXT:    [[TMP44:%.*]] = insertelement <2 x double> [[TMP49]], double [[TMP41]], i32 1
 ; CHECK-NEXT:    [[TMP50:%.*]] = fmul <2 x double> [[TMP44]], zeroinitializer
-; CHECK-NEXT:    [[TMP46:%.*]] = fadd <2 x double> [[TMP50]], zeroinitializer
-; CHECK-NEXT:    [[TMP45:%.*]] = fmul <2 x double> [[TMP46]], <double 0.000000e+00, double 1.000000e+00>
-; CHECK-NEXT:    [[TMP56:%.*]] = insertelement <2 x double> <double 0.000000e+00, double poison>, double [[TMP60]], i32 1
+; CHECK-NEXT:    [[TMP43:%.*]] = shufflevector <2 x double> [[TMP50]], <2 x double> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
+; CHECK-NEXT:    [[TMP51:%.*]] = insertelement <4 x double> [[TMP43]], double [[TMP37]], i32 2
+; CHECK-NEXT:    [[TMP52:%.*]] = insertelement <4 x double> [[TMP51]], double [[TMP47]], i32 3
+; CHECK-NEXT:    [[TMP46:%.*]] = fadd <4 x double> [[TMP52]], <double 0.000000e+00, double 0.000000e+00, double -0.000000e+00, double 0.000000e+00>
+; CHECK-NEXT:    [[TMP53:%.*]] = fmul <4 x double> [[TMP46]], <double 0.000000e+00, double 1.000000e+00, double 0.000000e+00, double 1.000000e+00>
+; CHECK-NEXT:    [[TMP59:%.*]] = insertelement <4 x double> <double 0.000000e+00, double poison, double poison, double 1.000000e+00>, double [[TMP33]], i32 1
+; CHECK-NEXT:    [[TMP60:%.*]] = shufflevector <4 x double> [[TMP59]], <4 x double> poison, <4 x i32> <i32 0, i32 1, i32 1, i32 3>
+; CHECK-NEXT:    [[TMP61:%.*]] = fmul <4 x double> [[TMP60]], [[TMP53]]
+; CHECK-NEXT:    store <4 x double> [[TMP61]], ptr getelementptr inbounds (i8, ptr @solid_, i64 384), align 8
+; CHECK-NEXT:    [[TMP56:%.*]] = shufflevector <4 x double> [[TMP61]], <4 x double> poison, <2 x i32> <i32 1, i32 2>
+; CHECK-NEXT:    [[TMP45:%.*]] = insertelement <2 x double> <double poison, double 0.000000e+00>, double [[TMP17]], i32 0
 ; CHECK-NEXT:    [[TMP48:%.*]] = fmul <2 x double> [[TMP56]], [[TMP45]]
-; CHECK-NEXT:    store <2 x double> [[TMP48]], ptr getelementptr inbounds (i8, ptr @solid_, i64 384), align 8
-; CHECK-NEXT:    store double [[TMP61]], ptr getelementptr inbounds (i8, ptr @solid_, i64 400), align 8
-; CHECK-NEXT:    [[TMP43:%.*]] = extractelement <2 x double> [[TMP48]], i32 1
-; CHECK-NEXT:    [[DOTNEG965:%.*]] = fmul double [[TMP43]], [[TMP17]]
-; CHECK-NEXT:    [[REASS_ADD993:%.*]] = fadd double [[DOTNEG965]], 0.000000e+00
-; CHECK-NEXT:    [[TMP58:%.*]] = fadd double [[TMP30]], [[REASS_ADD993]]
-; CHECK-NEXT:    [[TMP59:%.*]] = fsub double 0.000000e+00, [[TMP58]]
-; CHECK-NEXT:    store double [[TMP59]], ptr getelementptr inbounds (i8, ptr @solid_, i64 296), align 8
-; CHECK-NEXT:    [[DOTNEG970:%.*]] = fmul double [[TMP61]], 0.000000e+00
-; CHECK-NEXT:    [[REASS_ADD996:%.*]] = fadd double [[DOTNEG970]], 0.000000e+00
-; CHECK-NEXT:    [[TMP53:%.*]] = fadd double [[TMP60]], [[REASS_ADD996]]
-; CHECK-NEXT:    [[TMP54:%.*]] = fsub double 0.000000e+00, [[TMP53]]
-; CHECK-NEXT:    store double [[TMP54]], ptr getelementptr inbounds (i8, ptr @solid_, i64 304), align 8
+; CHECK-NEXT:    [[TMP54:%.*]] = fadd <2 x double> [[TMP48]], zeroinitializer
+; CHECK-NEXT:    [[TMP55:%.*]] = insertelement <2 x double> poison, double [[TMP38]], i32 0
+; CHECK-NEXT:    [[TMP62:%.*]] = insertelement <2 x double> [[TMP55]], double [[TMP33]], i32 1
+; CHECK-NEXT:    [[TMP57:%.*]] = fadd <2 x double> [[TMP62]], [[TMP54]]
+; CHECK-NEXT:    [[TMP58:%.*]] = fsub <2 x double> zeroinitializer, [[TMP57]]
+; CHECK-NEXT:    store <2 x double> [[TMP58]], ptr getelementptr inbounds (i8, ptr @solid_, i64 296), align 8
 ; CHECK-NEXT:    ret void
 ;
 .lr.ph1019:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
index 1805de6edf764..3f61fa3d44bc7 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder-vf-to-resize.ll
@@ -4,9 +4,10 @@
 define void @main(ptr %0) {
 ; CHECK-LABEL: @main(
 ; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x double>, ptr [[TMP0:%.*]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = fadd <2 x double> zeroinitializer, [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fsub <2 x double> zeroinitializer, [[TMP2]]
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> [[TMP4]], <4 x i32> <i32 0, i32 3, i32 0, i32 3>
+; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> <double poison, double 0.000000e+00>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> <double -0.000000e+00, double poison>, <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT:    [[TMP11:%.*]] = fsub <2 x double> [[TMP3]], [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x double> [[TMP11]], <2 x double> poison, <4 x i32> <i32 0, i32 1, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP6:%.*]] = fmul <4 x double> [[TMP5]], zeroinitializer
 ; CHECK-NEXT:    [[TMP7:%.*]] = call <4 x double> @llvm.fabs.v4f64(<4 x double> [[TMP6]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = fcmp oeq <4 x double> [[TMP7]], zeroinitializer
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/reorder_with_external_users.ll b/llvm/test/Transforms/SLPVectorizer/X86/reorder_with_external_users.ll
index aea518f337ac2..4820d2a21424a 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/reorder_with_external_users.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/reorder_with_external_users.ll
@@ -114,19 +114,16 @@ define void @addsub_and_external_users(ptr %A, ptr %ptr) {
 ; CHECK-NEXT:    [[LD:%.*]] = load double, ptr undef, align 8
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[LD]], i32 0
 ; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <2 x double> [[TMP0]], <2 x double> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = fsub <2 x double> [[TMP1]], <double 1.100000e+00, double 1.200000e+00>
-; CHECK-NEXT:    [[TMP6:%.*]] = fadd <2 x double> [[TMP1]], <double 1.100000e+00, double 1.200000e+00>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP2]], <2 x double> [[TMP6]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    [[TMP4:%.*]] = fdiv <2 x double> [[TMP3]], <double 2.100000e+00, double 2.200000e+00>
-; CHECK-NEXT:    [[TMP5:%.*]] = fmul <2 x double> [[TMP4]], <double 3.100000e+00, double 3.200000e+00>
-; CHECK-NEXT:    [[SHUFFLE1:%.*]] = shufflevector <2 x double> [[TMP5]], <2 x double> poison, <2 x i32> <i32 1, i32 0>
+; CHECK-NEXT:    [[TMP2:%.*]] = fadd <2 x double> [[TMP1]], <double 1.200000e+00, double -1.100000e+00>
+; CHECK-NEXT:    [[TMP3:%.*]] = fdiv <2 x double> [[TMP2]], <double 2.200000e+00, double 2.100000e+00>
+; CHECK-NEXT:    [[SHUFFLE1:%.*]] = fmul <2 x double> [[TMP3]], <double 3.200000e+00, double 3.100000e+00>
 ; CHECK-NEXT:    store <2 x double> [[SHUFFLE1]], ptr [[A:%.*]], align 8
 ; CHECK-NEXT:    br label [[BB2:%.*]]
 ; CHECK:       bb2:
-; CHECK-NEXT:    [[TMP7:%.*]] = fadd <2 x double> [[TMP5]], <double 4.100000e+00, double 4.200000e+00>
+; CHECK-NEXT:    [[TMP7:%.*]] = fadd <2 x double> [[SHUFFLE1]], <double 4.200000e+00, double 4.100000e+00>
 ; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP7]], i32 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP7]], i32 1
-; CHECK-NEXT:    [[SEED:%.*]] = fcmp ogt double [[TMP8]], [[TMP9]]
+; CHECK-NEXT:    [[SEED:%.*]] = fcmp ogt double [[TMP9]], [[TMP8]]
 ; CHECK-NEXT:    ret void
 ;
 bb1:
@@ -161,9 +158,7 @@ define void @subadd_and_external_users(ptr %A, ptr %ptr) {
 ; CHECK-NEXT:    [[LD:%.*]] = load double, ptr undef, align 8
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[LD]], i32 0
 ; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <2 x double> [[TMP0]], <2 x double> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = fadd <2 x double> [[TMP1]], <double 1.200000e+00, double 1.100000e+00>
-; CHECK-NEXT:    [[TMP10:%.*]] = fsub <2 x double> [[TMP1]], <double 1.200000e+00, double 1.100000e+00>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x double> [[TMP6]], <2 x double> [[TMP10]], <2 x i32> <i32 2, i32 1>
+; CHECK-NEXT:    [[TMP3:%.*]] = fsub <2 x double> [[TMP1]], <double 1.200000e+00, double -1.100000e+00>
 ; CHECK-NEXT:    [[TMP4:%.*]] = fdiv <2 x double> [[TMP3]], <double 2.200000e+00, double 2.100000e+00>
 ; CHECK-NEXT:    [[TMP5:%.*]] = fmul <2 x double> [[TMP4]], <double 3.200000e+00, double 3.100000e+00>
 ; CHECK-NEXT:    store <2 x double> [[TMP5]], ptr [[A:%.*]], align 8
@@ -206,9 +201,7 @@ define void @alt_but_not_addsub_and_external_users(ptr %A, ptr %ptr) {
 ; CHECK-NEXT:    [[LD:%.*]] = load double, ptr undef, align 8
 ; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x double> poison, double [[LD]], i32 0
 ; CHECK-NEXT:    [[SHUFFLE:%.*]] = shufflevector <4 x double> [[TMP0]], <4 x double> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP1:%.*]] = fsub <4 x double> [[SHUFFLE]], <double 1.400000e+00, double 1.300000e+00, double 1.200000e+00, double 1.100000e+00>
-; CHECK-NEXT:    [[TMP2:%.*]] = fadd <4 x double> [[SHUFFLE]], <double 1.400000e+00, double 1.300000e+00, double 1.200000e+00, double 1.100000e+00>
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <4 x double> [[TMP1]], <4 x double> [[TMP2]], <4 x i32> <i32 0, i32 5, i32 6, i32 3>
+; CHECK-NEXT:    [[TMP3:%.*]] = fadd <4 x double> [[SHUFFLE]], <double -1.400000e+00, double 1.300000e+00, double 1.200000e+00, double -1.100000e+00>
 ; CHECK-NEXT:    [[TMP4:%.*]] = fdiv <4 x double> [[TMP3]], <double 2.400000e+00, double 2.300000e+00, double 2.200000e+00, double 2.100000e+00>
 ; CHECK-NEXT:    [[TMP5:%.*]] = fmul <4 x double> [[TMP4]], <double 3.400000e+00, double 3.300000e+00, double 3.200000e+00, double 3.100000e+00>
 ; CHECK-NEXT:    store <4 x double> [[TMP5]], ptr [[A:%.*]], align 8
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/vec3-reorder-reshuffle.ll b/llvm/test/Transforms/SLPVectorizer/X86/vec3-reorder-reshuffle.ll
index 0a5550254a0dd..53ec7cb9ca676 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/vec3-reorder-reshuffle.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/vec3-reorder-reshuffle.ll
@@ -444,10 +444,9 @@ define void @reuse_shuffle_indices_cost_crash_3(ptr %m, double %conv, double %co
 ; CHECK-LABEL: define void @reuse_shuffle_indices_cost_crash_3(
 ; CHECK-SAME: ptr [[M:%.*]], double [[CONV:%.*]], double [[CONV2:%.*]]) {
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[SUB19:%.*]] = fsub double 0.000000e+00, [[CONV2]]
-; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV]], 0.000000e+00
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[SUB19]], i32 0
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[ADD]], i32 1
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> <double 0.000000e+00, double poison>, double [[CONV]], i32 1
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x double> <double poison, double -0.000000e+00>, double [[CONV2]], i32 0
+; CHECK-NEXT:    [[TMP1:%.*]] = fsub <2 x double> [[TMP0]], [[TMP3]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = fptrunc <2 x double> [[TMP1]] to <2 x float>
 ; CHECK-NEXT:    store <2 x float> [[TMP2]], ptr [[M]], align 4
 ; CHECK-NEXT:    [[ADD26:%.*]] = fsub double [[CONV]], [[CONV]]
@@ -519,10 +518,10 @@ define void @common_mask(ptr %m, double %conv, double %conv2) {
 ; CHECK-LABEL: define void @common_mask(
 ; CHECK-SAME: ptr [[M:%.*]], double [[CONV:%.*]], double [[CONV2:%.*]]) {
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[SUB19:%.*]] = fsub double [[CONV]], [[CONV]]
-; CHECK-NEXT:    [[ADD:%.*]] = fadd double [[CONV2]], 0.000000e+00
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[SUB19]], i32 0
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> [[TMP0]], double [[ADD]], i32 1
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x double> poison, double [[CONV]], i32 0
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x double> [[TMP0]], double [[CONV2]], i32 1
+; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x double> [[TMP3]], <2 x double> <double poison, double -0.000000e+00>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[TMP1:%.*]] = fsub <2 x double> [[TMP3]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = fptrunc <2 x double> [[TMP1]] to <2 x float>
 ; CHECK-NEXT:    store <2 x float> [[TMP2]], ptr [[M]], align 4
 ; CHECK-NEXT:    [[ADD26:%.*]] = fsub double 0.000000e+00, [[CONV]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/vect_copyable_in_binops.ll b/llvm/test/Transforms/SLPVectorizer/X86/vect_copyable_in_binops.ll
index f76e5049cf141..ba22132ec14dc 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/vect_copyable_in_binops.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/vect_copyable_in_binops.ll
@@ -447,20 +447,9 @@ entry:
 define void @addsub0f(ptr noalias %dst, ptr noalias %src) {
 ; CHECK-LABEL: @addsub0f(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[INCDEC_PTR:%.*]] = getelementptr inbounds float, ptr [[SRC:%.*]], i64 1
-; CHECK-NEXT:    [[TMP0:%.*]] = load float, ptr [[SRC]], align 4
-; CHECK-NEXT:    [[SUB:%.*]] = fadd fast float [[TMP0]], -1.000000e+00
-; CHECK-NEXT:    [[INCDEC_PTR1:%.*]] = getelementptr inbounds float, ptr [[DST:%.*]], i64 1
-; CHECK-NEXT:    store float [[SUB]], ptr [[DST]], align 4
-; CHECK-NEXT:    [[INCDEC_PTR2:%.*]] = getelementptr inbounds float, ptr [[SRC]], i64 2
-; CHECK-NEXT:    [[TMP1:%.*]] = load float, ptr [[INCDEC_PTR]], align 4
-; CHECK-NEXT:    [[INCDEC_PTR3:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 2
-; CHECK-NEXT:    store float [[TMP1]], ptr [[INCDEC_PTR1]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x float>, ptr [[INCDEC_PTR2]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fadd fast <2 x float> [[TMP2]], <float -2.000000e+00, float -3.000000e+00>
-; CHECK-NEXT:    [[TMP4:%.*]] = fsub fast <2 x float> [[TMP2]], <float -2.000000e+00, float -3.000000e+00>
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <2 x float> [[TMP3]], <2 x float> [[TMP4]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    store <2 x float> [[TMP5]], ptr [[INCDEC_PTR3]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[SRC:%.*]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = fadd reassoc nsz arcp contract afn <4 x float> [[TMP0]], <float -1.000000e+00, float -0.000000e+00, float -2.000000e+00, float 3.000000e+00>
+; CHECK-NEXT:    store <4 x float> [[TMP1]], ptr [[DST:%.*]], align 4
 ; CHECK-NEXT:    ret void
 ;
 entry:
@@ -487,20 +476,9 @@ entry:
 define void @addsub1f(ptr noalias %dst, ptr noalias %src) {
 ; CHECK-LABEL: @addsub1f(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[INCDEC_PTR2:%.*]] = getelementptr inbounds float, ptr [[SRC:%.*]], i64 2
-; CHECK-NEXT:    [[INCDEC_PTR3:%.*]] = getelementptr inbounds float, ptr [[DST:%.*]], i64 2
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr [[SRC]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = fadd fast <2 x float> [[TMP0]], splat (float -1.000000e+00)
-; CHECK-NEXT:    [[TMP2:%.*]] = fsub fast <2 x float> [[TMP0]], splat (float -1.000000e+00)
-; CHECK-NEXT:    [[TMP3:%.*]] = shufflevector <2 x float> [[TMP1]], <2 x float> [[TMP2]], <2 x i32> <i32 0, i32 3>
-; CHECK-NEXT:    store <2 x float> [[TMP3]], ptr [[DST]], align 4
-; CHECK-NEXT:    [[INCDEC_PTR4:%.*]] = getelementptr inbounds float, ptr [[SRC]], i64 3
-; CHECK-NEXT:    [[TMP4:%.*]] = load float, ptr [[INCDEC_PTR2]], align 4
-; CHECK-NEXT:    [[INCDEC_PTR6:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 3
-; CHECK-NEXT:    store float [[TMP4]], ptr [[INCDEC_PTR3]], align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = load float, ptr [[INCDEC_PTR4]], align 4
-; CHECK-NEXT:    [[SUB8:%.*]] = fsub fast float [[TMP5]], -3.000000e+00
-; CHECK-NEXT:    store float [[SUB8]], ptr [[INCDEC_PTR6]], align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[SRC:%.*]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = fsub reassoc nsz arcp contract afn <4 x float> [[TMP0]], <float 1.000000e+00, float -1.000000e+00, float 0.000000e+00, float -3.000000e+00>
+; CHECK-NEXT:    store <4 x float> [[TMP1]], ptr [[DST:%.*]], align 4
 ; CHECK-NEXT:    ret void
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-widest-phis.ll b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-widest-phis.ll
index 4ac93c787c4a9..03089b581f352 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/vectorize-widest-phis.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/vectorize-widest-phis.ll
@@ -18,10 +18,9 @@ define void @foo(i1 %arg) {
 ; CHECK:       bb4:
 ; CHECK-NEXT:    [[TMP4:%.*]] = fpext <4 x float> [[TMP2]] to <4 x double>
 ; CHECK-NEXT:    [[CONV2:%.*]] = uitofp i16 0 to double
-; CHECK-NEXT:    [[ADD1:%.*]] = fadd double [[TMP3]], [[CONV2]]
-; CHECK-NEXT:    [[SUB1:%.*]] = fsub double 0.000000e+00, 0.000000e+00
-; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x double> <double poison, double poison, double 0.000000e+00, double 0.000000e+00>, double [[SUB1]], i32 0
-; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x double> [[TMP5]], double [[ADD1]], i32 1
+; CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x double> <double 0.000000e+00, double poison, double 0.000000e+00, double -0.000000e+00>, double [[TMP3]], i32 1
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x double> <double -0.000000e+00, double poison, double -0.000000e+00, double 0.000000e+00>, double [[CONV2]], i32 1
+; CHECK-NEXT:    [[TMP10:%.*]] = fadd <4 x double> [[TMP5]], [[TMP6]]
 ; CHECK-NEXT:    [[TMP11:%.*]] = fcmp ogt <4 x double> [[TMP10]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP12:%.*]] = fptrunc <4 x double> [[TMP10]] to <4 x float>
 ; CHECK-NEXT:    [[TMP13:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[TMP2]], <4 x float> [[TMP12]]
diff --git a/llvm/test/Transforms/SLPVectorizer/jumbled_store_crash.ll b/llvm/test/Transforms/SLPVectorizer/jumbled_store_crash.ll
index 50cc97a529f5f..7a31474d8cb88 100644
--- a/llvm/test/Transforms/SLPVectorizer/jumbled_store_crash.ll
+++ b/llvm/test/Transforms/SLPVectorizer/jumbled_store_crash.ll
@@ -37,13 +37,11 @@ define dso_local void @j() local_unnamed_addr {
 ; CHECK-NEXT:    store float [[TMP15]], ptr @e, align 4
 ; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <4 x float> [[TMP12]], i32 1
 ; CHECK-NEXT:    store float [[TMP16]], ptr @f, align 4
-; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x float> <float poison, float -1.000000e+00, float poison, float -1.000000e+00>, float [[CONV19]], i32 0
+; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x float> <float poison, float 1.000000e+00, float poison, float 1.000000e+00>, float [[CONV19]], i32 0
 ; CHECK-NEXT:    [[TMP18:%.*]] = shufflevector <2 x float> [[TMP9]], <2 x float> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP19:%.*]] = shufflevector <4 x float> [[TMP17]], <4 x float> [[TMP18]], <4 x i32> <i32 0, i32 1, i32 5, i32 3>
 ; CHECK-NEXT:    [[TMP20:%.*]] = fsub <4 x float> [[TMP12]], [[TMP19]]
-; CHECK-NEXT:    [[TMP21:%.*]] = fadd <4 x float> [[TMP12]], [[TMP19]]
-; CHECK-NEXT:    [[TMP22:%.*]] = shufflevector <4 x float> [[TMP20]], <4 x float> [[TMP21]], <4 x i32> <i32 0, i32 5, i32 2, i32 7>
-; CHECK-NEXT:    [[TMP23:%.*]] = fptosi <4 x float> [[TMP22]] to <4 x i32>
+; CHECK-NEXT:    [[TMP23:%.*]] = fptosi <4 x float> [[TMP20]] to <4 x i32>
 ; CHECK-NEXT:    store <4 x i32> [[TMP23]], ptr [[ARRAYIDX1]], align 4
 ; CHECK-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/semanticly-same.ll b/llvm/test/Transforms/SLPVectorizer/semanticly-same.ll
index c656ddec09dae..5914c0a2f32ee 100644
--- a/llvm/test/Transforms/SLPVectorizer/semanticly-same.ll
+++ b/llvm/test/Transforms/SLPVectorizer/semanticly-same.ll
@@ -437,3 +437,37 @@ entry:
   store i16 %add3, ptr %s3
   ret void
 }
+
+define void @fadd_fsub(ptr %p, ptr %s) {
+; CHECK-LABEL: @fadd_fsub(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = fadd <4 x float> [[TMP0]], <float 3.000000e+00, float 5.000000e+00, float 2.000000e+00, float 3.000000e+00>
+; CHECK-NEXT:    store <4 x float> [[TMP1]], ptr [[S:%.*]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %p1 = getelementptr float, ptr %p, i64 1
+  %p2 = getelementptr float, ptr %p, i64 2
+  %p3 = getelementptr float, ptr %p, i64 3
+
+  %l0 = load float, ptr %p
+  %l1 = load float, ptr %p1
+  %l2 = load float, ptr %p2
+  %l3 = load float, ptr %p3
+
+  %add0 = fsub float %l0, -3.0
+  %add1 = fadd float %l1, 5.0
+  %add2 = fadd float %l2, 2.0
+  %add3 = fadd float %l3, 3.0
+
+  %s1 = getelementptr float, ptr %s, i64 1
+  %s2 = getelementptr float, ptr %s, i64 2
+  %s3 = getelementptr float, ptr %s, i64 3
+
+  store float %add0, ptr %s
+  store float %add1, ptr %s1
+  store float %add2, ptr %s2
+  store float %add3, ptr %s3
+  ret void
+}



More information about the llvm-commits mailing list