[llvm] [Hexagon] Add XQFloat extraneous conversion removal pass (PR #207236)

Fateme Hosseini via llvm-commits llvm-commits at lists.llvm.org
Thu Jul 2 10:21:27 PDT 2026


https://github.com/fhossein-quic created https://github.com/llvm/llvm-project/pull/207236

Introduce VectorConvertRemove to remove extraneous qf->sf/hf conversions after XQFloat code generation. Off by default, enabled with -enable-rem-conv.

 Co-authored-by: Santanu Das <quic_santdas at qti.qualcomm.com>

>From b0572ee5c9d2b9a3933b8c701b6ca2ccf85e955e Mon Sep 17 00:00:00 2001
From: Fateme Hosseini <fhossein at qti.qualcomm.com>
Date: Tue, 30 Jun 2026 07:27:55 -0700
Subject: [PATCH] [Hexagon] Add XQFloat extraneous conversion removal pass

Introduce VectorConvertRemove to remove extraneous qf->sf/hf
conversions after XQFloat code generation. Off by default,
enabled with -enable-rem-conv.

Co-authored-by: Santanu Das <quic_santdas at qti.qualcomm.com>
---
 .../Hexagon/HexagonXQFloatGenerator.cpp       | 500 +++++++++++++++++-
 .../CodeGen/Hexagon/autohvx/xqf-assertion1.ll |  16 +-
 .../Hexagon/autohvx/xqf-check-qf-instrs.ll    |  16 +-
 .../autohvx/xqf-compliant-ieee-mul-qf16.ll    |  86 +++
 .../autohvx/xqf-compliant-ieee-mul-qf32.ll    | 136 +++++
 .../Hexagon/autohvx/xqf-convert-elim.ll       |  77 +++
 .../Hexagon/autohvx/xqf-lossy-mul-qf16.ll     |  74 +++
 .../Hexagon/autohvx/xqf-lossy-mul-qf32.ll     | 109 ++++
 .../xqf-v81-compliant-ieee-mul-qf32.ll        | 109 ++++
 .../autohvx/xqf-v81/xqf-v81-lossy-mul-qf32.ll |  98 ++++
 .../Hexagon/autohvx/xqf-v81/xqf-v81-vsub.ll   | 164 ++++++
 llvm/test/CodeGen/Hexagon/autohvx/xqf-vsub.ll | 130 +++++
 12 files changed, 1497 insertions(+), 18 deletions(-)
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf16.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf32.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-convert-elim.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf16.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf32.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-compliant-ieee-mul-qf32.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-lossy-mul-qf32.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-vsub.ll
 create mode 100644 llvm/test/CodeGen/Hexagon/autohvx/xqf-vsub.ll

diff --git a/llvm/lib/Target/Hexagon/HexagonXQFloatGenerator.cpp b/llvm/lib/Target/Hexagon/HexagonXQFloatGenerator.cpp
index 460976cdfd7ba..8f9a4e8217ac8 100644
--- a/llvm/lib/Target/Hexagon/HexagonXQFloatGenerator.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonXQFloatGenerator.cpp
@@ -130,6 +130,7 @@
 #include "HexagonInstrInfo.h"
 #include "HexagonSubtarget.h"
 #include "HexagonTargetMachine.h"
+#include <vector>
 #include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/ADT/Statistic.h"
@@ -144,7 +145,6 @@
 #include "llvm/Support/CommandLine.h"
 #include "llvm/Support/Debug.h"
 #include "llvm/Support/raw_ostream.h"
-#include <vector>
 
 #define DEBUG_TYPE "hexagon-xqf-gen"
 
@@ -155,8 +155,16 @@ extern cl::opt<QFloatMode> QFloatModeValue;
 // Master flag to enable XQF generations
 cl::opt<bool> EnableHVXXQFloat("enable-xqf-gen", cl::init(false),
                                cl::desc("Enable XQFloat generations"));
+// Master flag to remove extraneous qf to sf/hf conversions
+cl::opt<bool>
+    EnableConversionsRemoval("enable-rem-conv", cl::init(false),
+                             cl::desc("Enable extraneous conversions removal"));
+
+// Diagnostic flags
+cl::opt<bool> PrintDebug("debug-print", cl::init(false),
+                         cl::desc("Print function mir after transformation"));
 // This vector contains the opcodes which generate qf32 from add/subtract
-static constexpr unsigned XQFPAdd32[] = {
+static const unsigned XQFPAdd32[] = {
     // vector add instructions
     Hexagon::V6_vadd_sf, Hexagon::V6_vadd_qf32, Hexagon::V6_vadd_qf32_mix,
 
@@ -165,7 +173,7 @@ static constexpr unsigned XQFPAdd32[] = {
     Hexagon::V6_vsub_sf_mix};
 
 // This vector contains the opcodes which generate qf16 from add/subtract
-static constexpr unsigned XQFPAdd16[] = {
+static const unsigned XQFPAdd16[] = {
     // vector add instructions
     Hexagon::V6_vadd_hf, Hexagon::V6_vadd_qf16, Hexagon::V6_vadd_qf16_mix,
 
@@ -174,13 +182,13 @@ static constexpr unsigned XQFPAdd16[] = {
     Hexagon::V6_vsub_hf_mix};
 
 // This vector contains the opcodes which generate qf32 from multiplication
-static constexpr unsigned XQFPMult32[] = {
+static const unsigned XQFPMult32[] = {
     Hexagon::V6_vmpy_qf32, Hexagon::V6_vmpy_qf32_qf16, Hexagon::V6_vmpy_qf32_hf,
     Hexagon::V6_vmpy_qf32_sf, Hexagon::V6_vmpy_qf32_mix_hf};
 // This vector contains the opcodes which generate qf16 from multiplication
-static constexpr unsigned XQFPMult16[] = {Hexagon::V6_vmpy_qf16,
-                                          Hexagon::V6_vmpy_qf16_hf,
-                                          Hexagon::V6_vmpy_qf16_mix_hf};
+static const unsigned XQFPMult16[] = {Hexagon::V6_vmpy_qf16,
+                                             Hexagon::V6_vmpy_qf16_hf,
+                                             Hexagon::V6_vmpy_qf16_mix_hf};
 
 namespace llvm {
 FunctionPass *createHexagonXQFloatGenerator();
@@ -253,10 +261,478 @@ struct HexagonXQFloatGenerator : public MachineFunctionPass {
       OriginalMI; // Hold the instructions to be deleted
 };
 
+// Print machine function
+static void debug_print([[maybe_unused]] MachineFunction &MF) {
+  dbgs() << "\n=== Printing function ===\n";
+#ifndef NDEBUG
+  for (MachineBasicBlock &MBB : MF)
+    MBB.dump();
+#endif // NDEBUG
+}
+
 // This class removes redundant vector convert instructions from qf to hf/sf.
 // Additionally, it relaces use of sf/hf registers with qf types.
 // The resulting code is complete without dangling instructions.
 // FIXME: Liveness is not preserved.
+class VectorConvertRemove {
+
+public:
+  VectorConvertRemove(MachineFunction &_MF, MachineRegisterInfo *_MRI,
+                      const HexagonSubtarget *_HST)
+      : MF(_MF), MRI(_MRI), HST(_HST) {
+    HII = HST->getInstrInfo();
+  }
+
+  void run();
+
+private:
+  MachineFunction &MF;
+  MachineRegisterInfo *MRI;
+  const HexagonSubtarget *HST;
+  const HexagonInstrInfo *HII;
+
+  enum Operation { Add16, Add32, Sub16, Sub32, Mul16, Mul32 };
+  // Helper functions
+  void handle_addsub_sf_sf(MachineInstr &, Register &, Register &, Register &,
+                           bool);
+  void handle_addsub_qf_sf(MachineInstr &, Register &, Register &, Register &,
+                           bool);
+  void handle_addsubmul_hf_hf(MachineInstr &, Register &, Register &,
+                              Register &, Operation);
+  void handle_addsubmul_qf_hf(MachineInstr &, Register &, Register &,
+                              Register &, Operation);
+  void handle_qf32_mul_sf_sf(MachineInstr &, Register &, Register &,
+                             Register &);
+  void handle_qf16_mul_hf_hf(MachineInstr &, Register &, Register &,
+                             Register &);
+  bool checkHVXUses32(MachineInstr *, MachineInstr *);
+  bool checkHVXUses16(MachineInstr *, MachineInstr *);
+  unsigned getOperation(Operation, bool, bool);
+
+  // List which holds conversion instructions
+  SmallPtrSet<MachineInstr *, 16> ConvInstrList;
+  // List which holds qf handling instructions
+  std::vector<MachineInstr *> SfHfInstrList;
+};
+
+// both : both operands are replaced
+unsigned VectorConvertRemove::getOperation(Operation Op, bool firstOpQf,
+                                           bool secOpQf) {
+  if (firstOpQf && secOpQf) {
+    switch (Op) {
+    case Add16:
+      return Hexagon::V6_vadd_qf16;
+    case Add32:
+      return Hexagon::V6_vadd_qf32;
+    case Sub16:
+      return Hexagon::V6_vsub_qf16;
+    case Sub32:
+      return Hexagon::V6_vsub_qf32;
+    case Mul16:
+      return Hexagon::V6_vmpy_qf16;
+    case Mul32:
+      return Hexagon::V6_vmpy_qf32_qf16;
+    }
+  } else if (firstOpQf) {
+    switch (Op) {
+    case Add16:
+      return Hexagon::V6_vadd_qf16_mix;
+    case Add32:
+      return Hexagon::V6_vadd_qf32_mix;
+    case Sub16:
+      return Hexagon::V6_vsub_qf16_mix;
+    case Sub32:
+      return Hexagon::V6_vsub_qf32_mix;
+    case Mul16:
+      return Hexagon::V6_vmpy_qf16_mix_hf;
+    case Mul32:
+      return Hexagon::V6_vmpy_qf32_mix_hf;
+    }
+  } else if (secOpQf) {
+    switch (Op) {
+    case Sub16:
+      return Hexagon::V6_vsub_hf_mix;
+    case Sub32:
+      return Hexagon::V6_vsub_sf_mix;
+    default:
+      break;
+    }
+  } else {
+  }
+  llvm_unreachable("Unknown opcode and operand combination!");
+}
+
+// Return false if there are multiple instructions where the qf32 is used
+// other than the instruction for which it is called
+bool VectorConvertRemove::checkHVXUses32(MachineInstr *MI,
+                                         MachineInstr *UseMI) {
+  Register convReg = MI->getOperand(0).getReg();
+  // Iterate over all uses of the Def we are analyzing
+  for (auto &MO : make_range(MRI->use_begin(convReg), MRI->use_end())) {
+    MachineInstr *UMI = MO.getParent();
+    if (UMI == UseMI)
+      continue;
+    // Since the convert cannot be deleted, we set the operand as NOT kill
+    MI->getOperand(1).setIsKill(false);
+    return false;
+  }
+  return true;
+}
+
+// Return false if there are multiple instructions where the qf16 is used
+// other than the instruction for which it is called
+bool VectorConvertRemove::checkHVXUses16(MachineInstr *MI,
+                                         MachineInstr *UseMI) {
+  Register convReg = MI->getOperand(0).getReg();
+  // Iterate over all uses of the Def we are analyzing
+  for (auto &MO : make_range(MRI->use_begin(convReg), MRI->use_end())) {
+    MachineInstr *UMI = MO.getParent();
+    if (UMI == UseMI)
+      continue;
+    // Since the convert cannot be deleted, we set the operand as NOT kill
+    MI->getOperand(1).setIsKill(false);
+    return false;
+  }
+  return true;
+}
+
+// Removes converts feeding to op(sf,sf), and replaces its sf operands with qf
+void VectorConvertRemove::handle_addsub_sf_sf(MachineInstr &MI, Register &Reg1,
+                                              Register &Reg2, Register &Dest,
+                                              bool isAdd) {
+
+  MachineBasicBlock &MBB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+
+  bool firstConv = false, secConv = false;
+  bool DefOp1_del = false, DefOp2_del = false;
+  Register Src1, Src2;
+
+  MachineInstr *DefOp1 = MRI->getVRegDef(Reg1);
+  MachineInstr *DefOp2 = MRI->getVRegDef(Reg2);
+  // check if the first operand is from a convert operation
+  if (DefOp1->getOpcode() == Hexagon::V6_vconv_sf_qf32) {
+    if (checkHVXUses32(DefOp1, &MI))
+      DefOp1_del = true;
+    Src1 = DefOp1->getOperand(1).getReg();
+    firstConv = true;
+  }
+
+  // check if the second operand is from a convert operation
+  if (DefOp2->getOpcode() == Hexagon::V6_vconv_sf_qf32) {
+    if (checkHVXUses32(DefOp2, &MI))
+      DefOp2_del = true;
+    Src2 = DefOp2->getOperand(1).getReg();
+    secConv = true;
+  }
+
+  if (firstConv && secConv) {
+    BuildMI(MBB, MI, DL,
+            HII->get(getOperation(isAdd ? Operation::Add32 : Operation::Sub32,
+                                  true, true)),
+            Dest)
+        .addReg(Src1)
+        .addReg(Src2);
+    SfHfInstrList.push_back(&MI);
+  } else if (firstConv) {
+    BuildMI(MBB, MI, DL,
+            HII->get(getOperation(isAdd ? Operation::Add32 : Operation::Sub32,
+                                  true, false)),
+            Dest)
+        .addReg(Src1)
+        .addReg(Reg2);
+    SfHfInstrList.push_back(&MI);
+  } else if (secConv) {
+    // For v79, there is no provision for 2nd op being qf for add/sub
+    if (HST->useHVXV81Ops()) {
+      if (isAdd)
+        BuildMI(MBB, MI, DL, HII->get(Hexagon::V6_vadd_qf32_mix), Dest)
+            .addReg(Src2)
+            .addReg(Reg1);
+      else
+        BuildMI(MBB, MI, DL, HII->get(Hexagon::V6_vsub_sf_mix), Dest)
+            .addReg(Reg1)
+            .addReg(Src2);
+      SfHfInstrList.push_back(&MI);
+      // For v79, there is no provision for 2nd op being qf for add/sub. Since
+      // add is commutative, the ops can be rotated.
+    } else if (HST->useHVXV79Ops()) {
+      // for vadd we interchange the ops, for vsub we ignore
+      if (isAdd) {
+        BuildMI(MBB, MI, DL, HII->get(Hexagon::V6_vadd_qf32_mix), Dest)
+            .addReg(Src2)
+            .addReg(Reg1);
+        SfHfInstrList.push_back(&MI);
+      } else // don't delete the convert instruction for vsub
+        DefOp2_del = false;
+    }
+  } else { // none of the operands are from convert instructions
+  }
+
+  if (DefOp1_del)
+    ConvInstrList.insert(DefOp1);
+  if (DefOp2_del)
+    ConvInstrList.insert(DefOp2);
+}
+
+// Removes converts feeding to op(hf,hf), and replaces its hf operands with qf
+void VectorConvertRemove::handle_addsubmul_hf_hf(MachineInstr &MI,
+                                                 Register &Reg1, Register &Reg2,
+                                                 Register &Dest, Operation Op) {
+
+  MachineBasicBlock &MBB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+
+  bool firstConv = false, secConv = false;
+  bool DefOp1_del = false, DefOp2_del = false;
+  bool isSub = Op == Operation::Sub16;
+  Register Src1, Src2;
+
+  MachineInstr *DefOp1 = MRI->getVRegDef(Reg1);
+  MachineInstr *DefOp2 = MRI->getVRegDef(Reg2);
+  // check if the first operand is from a convert operation
+  if (DefOp1->getOpcode() == Hexagon::V6_vconv_hf_qf16) {
+    if (checkHVXUses16(DefOp1, &MI))
+      DefOp1_del = true;
+    Src1 = DefOp1->getOperand(1).getReg();
+    firstConv = true;
+  }
+
+  // check if the second operand is from a convert operation
+  if (DefOp2->getOpcode() == Hexagon::V6_vconv_hf_qf16) {
+    if (checkHVXUses16(DefOp2, &MI))
+      DefOp2_del = true;
+    Src2 = DefOp2->getOperand(1).getReg();
+    secConv = true;
+  }
+
+  if (firstConv && secConv) {
+    BuildMI(MBB, MI, DL, HII->get(getOperation(Op, true, true)), Dest)
+        .addReg(Src1)
+        .addReg(Src2);
+    SfHfInstrList.push_back(&MI);
+  } else if (firstConv) {
+    BuildMI(MBB, MI, DL, HII->get(getOperation(Op, true, false)), Dest)
+        .addReg(Src1)
+        .addReg(Reg2);
+    SfHfInstrList.push_back(&MI);
+  } else if (secConv) {
+    // For v81, we interchange the ops for vadd/vmul
+    // for vsub we use qf as second operand
+    if (HST->useHVXV81Ops()) {
+      if (!isSub)
+        BuildMI(MBB, MI, DL, HII->get(getOperation(Op, true, false)), Dest)
+            .addReg(Src2)
+            .addReg(Reg1);
+      else
+        BuildMI(MBB, MI, DL, HII->get(getOperation(Op, false, true)), Dest)
+            .addReg(Reg1)
+            .addReg(Src2);
+      SfHfInstrList.push_back(&MI);
+    } else if (HST->useHVXV79Ops()) {
+      // for vadd/vmul we interchange the ops, for vsub we ignore
+      if (!isSub) {
+        BuildMI(MBB, MI, DL, HII->get(getOperation(Op, true, false)), Dest)
+            .addReg(Src2)
+            .addReg(Reg1);
+        SfHfInstrList.push_back(&MI);
+      } else // don't delete the convert instruction for vsub
+        DefOp2_del = false;
+    }
+  } else { // none of the operands are from convert instructions
+  }
+
+  if (DefOp1_del)
+    ConvInstrList.insert(DefOp1);
+  if (DefOp2_del)
+    ConvInstrList.insert(DefOp2);
+}
+
+// Removes converts feeding to op(qf,sf), and replaces its sf operands with qf
+void VectorConvertRemove::handle_addsub_qf_sf(MachineInstr &MI, Register &Reg1,
+                                              Register &Reg2, Register &Dest,
+                                              bool isAdd) {
+  MachineBasicBlock &MBB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+  Register Src;
+  bool conv = false;
+
+  MachineInstr *DefOp = MRI->getVRegDef(Reg2);
+  // check if the second operand is from a convert operation
+  if (DefOp->getOpcode() == Hexagon::V6_vconv_sf_qf32) {
+    if (checkHVXUses32(DefOp, &MI))
+      ConvInstrList.insert(DefOp);
+    Src = DefOp->getOperand(1).getReg();
+    conv = true;
+  }
+
+  if (conv) {
+    BuildMI(MBB, MI, DL,
+            HII->get(isAdd ? Hexagon::V6_vadd_qf32 : Hexagon::V6_vsub_qf32),
+            Dest)
+        .addReg(Reg1)
+        .addReg(Src);
+    SfHfInstrList.push_back(&MI);
+  }
+}
+
+// Removes converts feeding to op(qf,hf), and replaces its hf operands with qf
+void VectorConvertRemove::handle_addsubmul_qf_hf(MachineInstr &MI,
+                                                 Register &Reg1, Register &Reg2,
+                                                 Register &Dest, Operation Op) {
+  MachineBasicBlock &MBB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+  Register Src;
+  bool conv = false;
+
+  MachineInstr *DefOp = MRI->getVRegDef(Reg2);
+  // check if the second operand is from a convert operation
+  if (DefOp->getOpcode() == Hexagon::V6_vconv_hf_qf16) {
+    if (checkHVXUses16(DefOp, &MI))
+      ConvInstrList.insert(DefOp);
+    Src = DefOp->getOperand(1).getReg();
+    conv = true;
+  }
+
+  if (conv) {
+    BuildMI(MBB, MI, DL, HII->get(getOperation(Op, true, true)), Dest)
+        .addReg(Reg1)
+        .addReg(Src);
+    SfHfInstrList.push_back(&MI);
+  }
+}
+
+// Removes converts feeding to op(sf,sf), and replaces its sf operands with qf
+void VectorConvertRemove::handle_qf32_mul_sf_sf(MachineInstr &MI,
+                                                Register &Reg1, Register &Reg2,
+                                                Register &Dest) {
+  MachineBasicBlock &MBB = *MI.getParent();
+  const DebugLoc &DL = MI.getDebugLoc();
+  Register Src1, Src2;
+  bool firstConv = false, secConv = false;
+
+  MachineInstr *DefOp1 = MRI->getVRegDef(Reg1);
+  MachineInstr *DefOp2 = MRI->getVRegDef(Reg2);
+
+  if (DefOp1->getOpcode() == Hexagon::V6_vconv_sf_qf32 &&
+      DefOp2->getOpcode() == Hexagon::V6_vconv_sf_qf32) {
+    // If yes, we can remove the convert
+    if (checkHVXUses32(DefOp1, &MI) && checkHVXUses32(DefOp2, &MI)) {
+      ConvInstrList.insert(DefOp1);
+      ConvInstrList.insert(DefOp2);
+    }
+    Src1 = DefOp1->getOperand(1).getReg();
+    Src2 = DefOp2->getOperand(1).getReg();
+    firstConv = true;
+    secConv = true;
+  }
+
+  // If both are true, then only replace with qf32 = vmpy(qf32, qf32)
+  if (firstConv && secConv) {
+    BuildMI(MBB, MI, DL, HII->get(Hexagon::V6_vmpy_qf32), Dest)
+        .addReg(Src1)
+        .addReg(Src2);
+    SfHfInstrList.push_back(&MI);
+  }
+}
+
+void VectorConvertRemove::run() {
+  for (auto &MBB : MF) {
+    for (auto &MI : MBB) {
+      // Skip if the instruction does not have two operands,
+      // or is a bundle instruction
+      // or is a debug instruction
+      if (MI.getNumOperands() != 3 || MI.isDebugInstr())
+        continue;
+
+      auto Op1 = MI.getOperand(1);
+      if (!Op1.isReg())
+        continue;
+      auto Op2 = MI.getOperand(2);
+      if (!Op2.isReg())
+        continue;
+      auto Op0 = MI.getOperand(0);
+      if (!Op0.isReg())
+        continue;
+      Register Reg1 = Op1.getReg();
+      Register Reg2 = Op2.getReg();
+      Register Dest = Op0.getReg();
+
+      switch (MI.getOpcode()) {
+      // TODO Handle the new vsub instructions
+      // qf32 = vadd(sf, sf)
+      case Hexagon::V6_vadd_sf:
+        handle_addsub_sf_sf(MI, Reg1, Reg2, Dest, true);
+        break;
+      // qf32 = vsub(sf, sf)
+      case Hexagon::V6_vsub_sf:
+        handle_addsub_sf_sf(MI, Reg1, Reg2, Dest, false);
+        break;
+      // qf32 = vadd(qf32, sf)
+      case Hexagon::V6_vadd_qf32_mix:
+        handle_addsub_qf_sf(MI, Reg1, Reg2, Dest, true);
+        break;
+      // qf32 = vsub(qf32, sf)
+      case Hexagon::V6_vsub_qf32_mix:
+        handle_addsub_qf_sf(MI, Reg1, Reg2, Dest, false);
+        break;
+      // qf16 = vadd(hf, hf)
+      case Hexagon::V6_vadd_hf:
+        handle_addsubmul_hf_hf(MI, Reg1, Reg2, Dest, Operation::Add16);
+        break;
+      // qf16 = vsub(hf, hf)
+      case Hexagon::V6_vsub_hf:
+        handle_addsubmul_hf_hf(MI, Reg1, Reg2, Dest, Operation::Sub16);
+        break;
+      // qf16 = vadd(qf16, hf)
+      case Hexagon::V6_vadd_qf16_mix:
+        handle_addsubmul_qf_hf(MI, Reg1, Reg2, Dest, Operation::Add16);
+        break;
+      // qf16 = vsub(qf16, hf)
+      case Hexagon::V6_vsub_qf16_mix:
+        handle_addsubmul_qf_hf(MI, Reg1, Reg2, Dest, Operation::Sub16);
+        break;
+      // qf32 = vmpy(sf, sf)
+      case Hexagon::V6_vmpy_qf32_sf:
+        handle_qf32_mul_sf_sf(MI, Reg1, Reg2, Dest);
+        break;
+      // qf32 = vmpy(hf, hf)
+      case Hexagon::V6_vmpy_qf32_hf:
+        handle_addsubmul_hf_hf(MI, Reg1, Reg2, Dest, Operation::Mul32);
+        break;
+      // qf32 = vmpy(qf16, hf)
+      case Hexagon::V6_vmpy_qf32_mix_hf:
+        handle_addsubmul_qf_hf(MI, Reg1, Reg2, Dest, Operation::Mul32);
+        break;
+      // qf16 = vmpy(hf, hf)
+      case Hexagon::V6_vmpy_qf16_hf:
+        handle_addsubmul_hf_hf(MI, Reg1, Reg2, Dest, Operation::Mul16);
+        break;
+      // qf16 = vmpy(qf16, hf)
+      case Hexagon::V6_vmpy_qf16_mix_hf:
+        handle_addsubmul_qf_hf(MI, Reg1, Reg2, Dest, Operation::Mul16);
+        ;
+        break;
+      default:
+        break;
+      }
+    }
+  }
+
+  // Delete the vadd/vsub/vmpy instructions
+  for (MachineInstr *sfhfMI : SfHfInstrList) {
+    LLVM_DEBUG(dbgs() << "deleting sf/hf instruction ");
+    LLVM_DEBUG(sfhfMI->dump());
+    sfhfMI->eraseFromParent();
+  }
+  // Delete conversion instructions
+  for (MachineInstr *convMI : ConvInstrList) {
+    LLVM_DEBUG(dbgs() << "deleting conversion instruction");
+    LLVM_DEBUG(convMI->dump());
+    convMI->eraseFromParent();
+  }
+}
+
 char HexagonXQFloatGenerator::ID = 0;
 
 } // namespace
@@ -1652,6 +2128,16 @@ bool HexagonXQFloatGenerator::runOnMachineFunction(MachineFunction &MF) {
   HII = HST->getInstrInfo();
   MRI = &MF.getRegInfo();
 
+  if (EnableConversionsRemoval &&
+      !(QFloatModeValue == QFloatMode::StrictIEEE)) {
+    VectorConvertRemove VCR(MF, MRI, HST);
+    VCR.run();
+    LLVM_DEBUG(dbgs() << "\nExtraneous conversion instructions removed for "
+                      << MF.getName());
+    if (PrintDebug)
+      debug_print(MF);
+  }
+
   switch (QFloatModeValue) {
   case QFloatMode::StrictIEEE:
     LLVM_DEBUG(dbgs() << "\nGenerating code for STRICT-IEEE mode.\n");
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-assertion1.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-assertion1.ll
index f73cfc9c83729..90d0790989388 100644
--- a/llvm/test/CodeGen/Hexagon/autohvx/xqf-assertion1.ll
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-assertion1.ll
@@ -1,9 +1,8 @@
 ; On v79 and above, checks for Assertion `isImm() && "Wrong MachineOperand accessor"' failed
 
-; REQUIRES: asserts
-; RUN: llc -march=hexagon -enable-xqf-gen=true \
+; RUN: llc -march=hexagon -enable-xqf-gen=true -enable-rem-conv=true \
 ; RUN: -mattr=+hvx-ieee-fp,+hvx-length128b,+hvxv79 -o /dev/null < %s
-; RUN: llc -march=hexagon -enable-xqf-gen=true \
+; RUN: llc -march=hexagon -enable-xqf-gen=true -enable-rem-conv=true \
 ; RUN: -mattr=+hvx-ieee-fp,+hvx-length128b,+hvxv81 -o /dev/null < %s
 
 
@@ -72,3 +71,14 @@ attributes #4 = { nocallback nofree nosync nounwind willreturn memory(none) }
 attributes #5 = { nofree nounwind }
 attributes #6 = { nounwind }
 attributes #7 = { noreturn nounwind }
+
+!llvm.module.flags = !{!0, !1, !2}
+!llvm.ident = !{!6}
+
+!0 = !{i32 1, !"wchar_size", i32 4}
+!1 = !{i32 7, !"frame-pointer", i32 2}
+!2 = !{i32 5, !"CG MDInfo", !3}
+!3 = !{!4, !5}
+!4 = !{!"F", !"no_filename_available", !"", !"", i1 false, !""}
+!5 = !{!"C", !"set_double_vector_mode", !"(void)", !"(...)", i1 true, !""}
+!6 = !{!"QuIC LLVM Hexagon Clang version 8.8-alpha3 Engineering Release: hexagon-clang-88"}
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-check-qf-instrs.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-check-qf-instrs.ll
index c3b22b77d533b..96e497a571cb5 100644
--- a/llvm/test/CodeGen/Hexagon/autohvx/xqf-check-qf-instrs.ll
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-check-qf-instrs.ll
@@ -5,29 +5,29 @@
 
 ; REQUIRES: asserts
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=ieee -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=ieee -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=IEEE
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv81 -mattr=+hvxv81,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=ieee -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=ieee -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=IEEE
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=strict-ieee -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=strict-ieee -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=STRICT
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv81 -mattr=+hvxv81,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=strict-ieee -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=strict-ieee -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=STRICT
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=lossy -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=lossy -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=LOSSY
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv81 -mattr=+hvxv81,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -hexagon-qfloat-mode=lossy -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=lossy -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=LOSSY
 
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=LEGACY
 ; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv81 -mattr=+hvxv81,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
-; RUN: -debug-only=handle-qfp -o /dev/null < %s \
+; RUN: -enable-rem-conv=true -debug-only=handle-qfp -o /dev/null < %s \
 ; RUN: 2>&1 | FileCheck %s --check-prefix=LEGACY
 
 define dso_local <32 x i32> @test1(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf16.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf16.ll
new file mode 100644
index 0000000000000..ed4aaa5ccf076
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf16.ll
@@ -0,0 +1,86 @@
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv79 -force-hvx-float -enable-rem-conv=true \
+; RUN: -enable-xqf-gen=true -hexagon-qfloat-mode=ieee -mattr=+hvxv79,+hvx-length128B < %s | FileCheck %s
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv81 -force-hvx-float -enable-rem-conv=true \
+; RUN: -enable-xqf-gen=true -hexagon-qfloat-mode=ieee -mattr=+hvxv81,+hvx-length128B < %s | FileCheck %s
+
+; Test qf16 = vmpy(qf16 ,qf16) when both inputs are from vadd instruction
+define <64 x half> @mul_add_3(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vadd(v0.hf,v2.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]] = vxor([[V5]],[[V5]])
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V4]].qf16,[[V3]].qf16)
+; CHECK-DAG:     [[V6:v[0-9]+]].hf = [[V10]].qf32
+; CHECK-DAG:     qf16 = vsub([[V6]].hf,[[V5]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v1 = fadd <64 x half> %a0, %a2
+  %v3 = fmul <64 x half> %v0, %v1
+  ret <64 x half> %v3
+}
+
+; Test qf32 = vmpy(qf16 ,qf16) when both inputs are from vadd and vmul instruction
+define <64 x half> @mul_add_mul(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[V32:v[0-9]+:[0-9]+]].qf32 = vmpy(v0.hf,v2.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]] = vxor([[V5]],[[V5]])
+; CHECK-DAG:     [[V3:v[0-9]+]].hf = [[V32]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf16 = vsub([[V3]].hf,[[V5]].hf)
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V4]].qf16,[[V6]].qf16)
+; CHECK-DAG:     [[V7:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V7]].hf,[[V5]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v1 = fmul <64 x half> %a0, %a2
+  %v3 = fmul <64 x half> %v0, %v1
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(sf ,sf)
+define <64 x half> @mul_add_0(<64 x half> %a0, <64 x half> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy(v0.hf,v1.hf)
+; CHECK-DAG:     [[V2:v[0-9]+]] = vxor([[V2]],[[V2]])
+; CHECK-DAG:     [[V3:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V3]].hf,[[V2]].hf)
+label0:
+  %v3 = fmul <64 x half> %a0, %a1
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(qf16 ,qf16) when first input is from vadd instruction
+define <64 x half> @mul_add_1(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V3]].qf16,v2.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V5]].hf,[[V4]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v3 = fmul <64 x half> %v0, %a2
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(qf16 ,qf16) when second input is from vadd instruction
+define <64 x half> @mul_add_2(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_2:
+; CHECK-DAG:     [[V54:v[0-9]+:[0-9]+]].qf32 = vmpy(v0.hf,v2.hf)
+; CHECK-DAG:     [[V29:v[0-9]+:[0-9]+]].qf32 = vmpy(v1.hf,v2.hf)
+; CHECK-DAG:     [[V30:v[0-9]+]] = vxor([[V30]],[[V30]])
+; CHECK-DAG:     [[V3:v[0-9]+]].hf = [[V54]].qf32
+; CHECK-DAG:     [[V31:v[0-9]+]].hf = [[V29]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf16 = vsub([[V3]].hf,[[V30]].hf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf16 = vsub([[V31]].hf,[[V30]].hf)
+; CHECK-DAG:     [[V32:v[0-9]+:[0-9]+]].qf32 = vmpy([[V6]].qf16,[[V7]].qf16)
+; CHECK-DAG:     [[V8:v[0-9]+]].hf = [[V32]].qf32
+; CHECK:         qf16 = vsub([[V8]].hf,[[V30]].hf)
+label0:
+  %v1 = fmul <64 x half> %a0, %a2
+  %v2 = fmul <64 x half> %a1, %a2
+  %v3 = fmul <64 x half> %v1, %v2
+  ret <64 x half> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-features"="+hvx-length128b,+hvx-qfloat,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf32.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf32.ll
new file mode 100644
index 0000000000000..5db57fb7e6131
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-compliant-ieee-mul-qf32.ll
@@ -0,0 +1,136 @@
+; Tests compliant IEEE mode for XQFloat multiplication 32-bit
+
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv79 -force-hvx-float -enable-xqf-gen=true -enable-rem-conv=true \
+; RUN: -hexagon-qfloat-mode=ieee  -mattr=+hvxv79,+hvx-length128B < %s | FileCheck %s -check-prefix=CHECK
+
+; Test qf32 = vmpy(sf, sf)
+; Normalization of inputs
+define <32 x float> @mul_add_0(<32 x float> %a0, <32 x float> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V2:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V3:v[0-9]+]] = vxor([[V3]],[[V3]])
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vmpy([[V3]].sf,[[V2]].sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vadd([[V4]].qf32,v0.sf)
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = vadd([[V4]].qf32,v1.sf)
+; CHECK:         qf32 = vmpy([[V5]].qf32,[[V6]].qf32)
+label0:
+  %v3 = fmul <32 x float> %a0, %a1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(sf ,qf32) when only one input is from vadd instruction
+define <32 x float> @mul_add_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V5:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V6:v[0-9]+]].sf = [[V3]].qf32
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V5]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,v1.sf)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V6]].sf)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vadd instruction
+define <32 x float> @mul_add_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V6:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vadd(v0.sf,v1.sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V6]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V5]].qf32)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V3]].qf32)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v0 = fadd <32 x float> %a0, %a1
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when only first input is from vsub instruction
+define <32 x float> @mul_sub_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_1:
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]] = vxor([[V5]],[[V5]])
+; CHECK-DAG:     [[V6:v[0-9]+]].sf = [[V3]].qf32
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V5]].sf,[[V4]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,v1.sf)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V6]].sf)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vsub instruction
+define <32 x float> @mul_sub_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V6:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vsub(v0.sf,v1.sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V6]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V5]].qf32)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V3]].qf32)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v0 = fsub <32 x float> %a0, %a1
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when one is from adder, another from multiplier
+define <32 x float> @mul_add_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V3:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V3]].sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vadd([[V6]].qf32,v0.sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V6]].qf32,v1.sf)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V6]].qf32,[[V5]].qf32)
+; CHECK-DAG:     [[V10:v[0-9]+]].qf32 = vmpy([[V7]].qf32,[[V8]].qf32)
+; CHECK:         qf32 = vmpy([[V9]].qf32,[[V10]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+define <32 x float> @mul_mul_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+label0:
+; CHECK-LABEL: mul_mul_mul
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V3:v[0-9]+]] = vxor([[V3]],[[V3]])
+; CHECK-DAG:     [[V4:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vmpy([[V3]].sf,[[V4]].sf)
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = vadd([[V5]].qf32,v0.sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vadd([[V5]].qf32,v2.sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V5]].qf32,v1.sf)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vmpy([[V6]].qf32,[[V7]].qf32)
+; CHECK-DAG:     [[V10:v[0-9]+]].qf32 = vmpy([[V6]].qf32,[[V8]].qf32)
+; CHECK:         qf32 = vmpy([[V9]].qf32,[[V10]].qf32)
+  %v1 = fmul <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv79" "target-features"="+hvx-length128b,+hvx-qfloat,+hvxv79,+v79,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-convert-elim.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-convert-elim.ll
new file mode 100644
index 0000000000000..cd48a7008053b
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-convert-elim.ll
@@ -0,0 +1,77 @@
+; Tests if the sf/hf = qf converts have been done correctly
+
+; REQUIRES: asserts
+; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat \
+; RUN: -enable-rem-conv=true -enable-xqf-gen=true -hexagon-qfloat-mode=ieee -verify-machineinstrs \
+; RUN: -debug-print < %s 2>&1 -o /dev/null | FileCheck %s
+; RUN: llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat \
+; RUN: -enable-rem-conv=true -enable-xqf-gen=true -hexagon-qfloat-mode=lossy -verify-machineinstrs \
+; RUN: -debug-print < %s 2>&1 -o /dev/null | FileCheck %s
+
+; Single use of convert reg. The convert should be deleted.
+define dso_local <32 x i32> @conv1_qf32(<32 x i32> noundef %input1, <32 x i32> noundef %input2) local_unnamed_addr #0 {
+; CHECK: bb.0.entry
+; CHECK: [[VREG2:%[0-9]+]]:hvxvr = V6_vadd_sf [[VREG0:%[0-9]+]]:hvxvr, %1:hvxvr
+; CHECK-NOT: V6_vconv_sf_qf32 killed [[VREG2]]:hvxvr
+; CHECK: V6_vadd_qf32_mix [[VREG2]]:hvxvr, [[VREG0]]:hvxvr
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %input1, <32 x i32> %input2)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %0)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %input1, <32 x i32> %1)
+  ret <32 x i32> %2
+}
+
+; Double use of convert reg. The convert should not be deleted.
+define dso_local <32 x i32> @conv2_qf32(<32 x i32> noundef %input1, <32 x i32> noundef %input2) local_unnamed_addr #0 {
+; CHECK: bb.0.entry
+; CHECK: [[VREG2:%[0-9]+]]:hvxvr = V6_vadd_sf [[VREG0:%[0-9]+]]:hvxvr, [[VREG1:%[0-9]+]]:hvxvr
+; CHECK-NEXT: V6_vconv_sf_qf32 [[VREG2]]:hvxvr
+; CHECK-NEXT: V6_vadd_qf32_mix [[VREG2]]:hvxvr, [[VREG0]]:hvxvr
+; CHECK-NEXT: V6_vadd_qf32_mix [[VREG2]]:hvxvr, [[VREG1]]:hvxvr
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %input1, <32 x i32> %input2)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %0)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %input1, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %input2, <32 x i32> %1)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.qf32.128B(<32 x i32> %2, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; Single use of convert reg. The convert should be deleted.
+define dso_local <32 x i32> @conv1_qf16(<32 x i32> noundef %input1, <32 x i32> noundef %input2) local_unnamed_addr #0 {
+; CHECK: bb.0.entry
+; CHECK: [[VREG2:%[0-9]+]]:hvxvr = V6_vadd_hf [[VREG0:%[0-9]+]]:hvxvr, %1:hvxvr
+; CHECK-NOT: V6_vconv_hf_qf16 killed [[VREG2]]:hvxvr
+; CHECK: V6_vadd_qf16_mix [[VREG2]]:hvxvr, [[VREG0]]:hvxvr
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %input1, <32 x i32> %input2)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %0)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %input1, <32 x i32> %1)
+  ret <32 x i32> %2
+}
+
+; Double use of convert reg. The convert should not be deleted.
+define dso_local <32 x i32> @conv2_qf16(<32 x i32> noundef %input1, <32 x i32> noundef %input2) local_unnamed_addr #0 {
+; CHECK: bb.0.entry
+; CHECK: [[VREG2:%[0-9]+]]:hvxvr = V6_vadd_hf [[VREG0:%[0-9]+]]:hvxvr, [[VREG1:%[0-9]+]]:hvxvr
+; CHECK-NEXT: V6_vconv_hf_qf16 [[VREG2]]:hvxvr
+; CHECK-NEXT: V6_vadd_qf16_mix [[VREG2]]:hvxvr, [[VREG0]]:hvxvr
+; CHECK-NEXT: V6_vadd_qf16_mix [[VREG2]]:hvxvr, [[VREG1]]:hvxvr
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %input1, <32 x i32> %input2)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %0)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %input1, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %input2, <32 x i32> %1)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.qf16.128B(<32 x i32> %2, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+declare <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.qf32.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.qf16.128B(<32 x i32>, <32 x i32>) #1
+
+attributes #0 = { nounwind "frame-pointer"="all" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-features"="-long-calls,-small-data" }
+attributes #1 = { nocallback nofree nosync nounwind willreturn memory(none) }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf16.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf16.ll
new file mode 100644
index 0000000000000..5e330c7166bdb
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf16.ll
@@ -0,0 +1,74 @@
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv79 -force-hvx-float -enable-xqf-gen=true \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=lossy -mattr=+hvxv79,+hvx-length128B < %s | FileCheck %s
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv81 -force-hvx-float -enable-xqf-gen=true \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=lossy -mattr=+hvxv81,+hvx-length128B < %s | FileCheck %s
+
+; Test qf16 = vmpy(qf16 ,qf16) when both inputs are from vadd instruction
+define <64 x half> @mul_add_3(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vadd(v0.hf,v2.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]] = vxor([[V5]],[[V5]])
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V4]].qf16,[[V3]].qf16)
+; CHECK-DAG:     [[V6:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V6]].hf,[[V5]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v1 = fadd <64 x half> %a0, %a2
+  %v3 = fmul <64 x half> %v0, %v1
+  ret <64 x half> %v3
+}
+
+; Test qf32 = vmpy(qf16 ,qf16) when both inputs are from vadd and vmul instruction
+define <64 x half> @mul_add_mul(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vmpy(v0.hf,v2.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]] = vxor([[V5]],[[V5]])
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V4]].qf16,[[V3]].qf16)
+; CHECK-DAG:     [[V6:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V6]].hf,[[V5]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v1 = fmul <64 x half> %a0, %a2
+  %v3 = fmul <64 x half> %v0, %v1
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(sf ,sf)
+define <64 x half> @mul_add_0(<64 x half> %a0, <64 x half> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK:       qf16 = vmpy(v0.hf,v1.hf)
+label0:
+  %v3 = fmul <64 x half> %a0, %a1
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(qf16 ,qf16) when first input is from vadd instruction
+define <64 x half> @mul_add_1(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vadd(v0.hf,v1.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V10:v[0-9]+:[0-9]+]].qf32 = vmpy([[V3]].qf16,v2.hf)
+; CHECK-DAG:     [[V5:v[0-9]+]].hf = [[V10]].qf32
+; CHECK:         qf16 = vsub([[V5]].hf,[[V4]].hf)
+label0:
+  %v0 = fadd <64 x half> %a0, %a1
+  %v3 = fmul <64 x half> %v0, %a2
+  ret <64 x half> %v3
+}
+
+; Test qf16 = vmpy(qf16 ,qf16) when second input is from vadd instruction
+define <64 x half> @mul_add_2(<64 x half> %a0, <64 x half> %a1, <64 x half> %a2) #0 {
+; CHECK-LABEL: mul_add_2:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf16 = vmpy(v0.hf,v2.hf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf16 = vmpy(v1.hf,v2.hf)
+; CHECK-DAG:     qf16 = vmpy([[V3]].qf16,[[V4]].qf16)
+label0:
+  %v1 = fmul <64 x half> %a0, %a2
+  %v2 = fmul <64 x half> %a1, %a2
+  %v3 = fmul <64 x half> %v1, %v2
+  ret <64 x half> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-features"="+hvx-length128b,+hvx-qfloat,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf32.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf32.ll
new file mode 100644
index 0000000000000..1d9939c5ce312
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-lossy-mul-qf32.ll
@@ -0,0 +1,109 @@
+;n Tests lossy-subnormals mode for XQFloat multiplication 32-bit
+
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv79 -force-hvx-float -enable-xqf-gen=true -enable-rem-conv=true -hexagon-qfloat-mode=lossy  -mattr=+hvxv79,+hvx-length128B < %s | FileCheck %s
+
+; Test qf32 = vmpy(sf, sf)
+; Normalization of inputs
+define <32 x float> @mul_add_0(<32 x float> %a0, <32 x float> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK:     qf32 = vmpy(v0.sf,v1.sf)
+label0:
+  %v3 = fmul <32 x float> %a0, %a1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(sf ,qf32) when only one input is from vadd instruction
+define <32 x float> @mul_add_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK:     [[V4:v[0-9]+]].sf = [[V3]].qf32
+; CHECK:     qf32 = vmpy(v1.sf,[[V4]].sf)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vadd instruction
+define <32 x float> @mul_add_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V6:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vadd(v0.sf,v1.sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V6]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V5]].qf32)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V3]].qf32)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v0 = fadd <32 x float> %a0, %a1
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when only first input is from vsub instruction
+define <32 x float> @mul_sub_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_1:
+; CHECK:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK:     [[V4:v[0-9]+]].sf = [[V3]].qf32
+; CHECK:     qf32 = vmpy(v1.sf,[[V4]].sf)
+label0:
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vsub instruction
+define <32 x float> @mul_sub_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V6:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = vsub(v0.sf,v1.sf)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V6]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V5]].qf32)
+; CHECK-DAG:     [[V9:v[0-9]+]].qf32 = vadd([[V7]].qf32,[[V3]].qf32)
+; CHECK:         qf32 = vmpy([[V8]].qf32,[[V9]].qf32)
+label0:
+  %v0 = fsub <32 x float> %a0, %a1
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when one is from adder, another from multiplier
+define <32 x float> @mul_add_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy(v0.sf,v1.sf)
+; CHECK-DAG:     [[R0:r[0-9]+]] = ##2147483648
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]] = vxor([[V4]],[[V4]])
+; CHECK-DAG:     [[V5:v[0-9]+]] = vsplat([[R0]])
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = vmpy([[V4]].sf,[[V5]].sf)
+; CHECK-DAG:     [[V8:v[0-9]+]].qf32 = vadd([[V6]].qf32,[[V3]].qf32)
+; CHECK-DAG:     qf32 = vmpy([[V8]].qf32,[[V7]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when both are from multiplier
+define <32 x float> @mul_mul_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+label0:
+; CHECK-LABEL: mul_mul_mul
+; CHECK: [[V3:v[0-9]+]].qf32 = vmpy(v0.sf,v1.sf)
+; CHECK: [[V4:v[0-9]+]].qf32 = vmpy(v0.sf,v2.sf)
+; CHECK: qf32 = vmpy([[V4]].qf32,[[V3]].qf32)
+  %v1 = fmul <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv79" "target-features"="+hvx-length128b,+hvx-qfloat,+hvxv79,+v79,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-compliant-ieee-mul-qf32.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-compliant-ieee-mul-qf32.ll
new file mode 100644
index 0000000000000..ecf4ffdfc83fa
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-compliant-ieee-mul-qf32.ll
@@ -0,0 +1,109 @@
+; Tests compliant IEEE mode for XQFloat multiplication 32-bit for v81.
+
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv81 -force-hvx-float -enable-xqf-gen=true \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=ieee  -mattr=+hvxv81,+hvx-length128B \
+; RUN: < %s | FileCheck %s -check-prefix=CHECK
+
+; Test qf32 = vmpy(sf, sf)
+; Normalization of inputs
+define <32 x float> @mul_add_0(<32 x float> %a0, <32 x float> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = v0.sf
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = v1.sf
+; CHECK:         qf32 = vmpy([[V3]].qf32,[[V4]].qf32)
+label0:
+  %v3 = fmul <32 x float> %a0, %a1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(sf ,qf32) when only one input is from vadd instruction
+define <32 x float> @mul_add_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = v1.sf
+; CHECK-DAG:     [[V5:v[0-9]+]].sf = [[V3]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V5]].sf
+; CHECK:         qf32 = vmpy([[V4]].qf32,[[V6]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vadd instruction
+define <32 x float> @mul_add_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vadd(v0.sf,v1.sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = [[V3]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V4]].qf32
+; CHECK:         qf32 = vmpy([[V6]].qf32,[[V5]].qf32)
+label0:
+  %v0 = fadd <32 x float> %a0, %a1
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when only first input is from vsub instruction
+define <32 x float> @mul_sub_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_1:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = v1.sf
+; CHECK-DAG:     [[V5:v[0-9]+]].sf = [[V3]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V5]].sf
+; CHECK:         qf32 = vmpy([[V4]].qf32,[[V6]].qf32)
+label0:
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vsub instruction
+define <32 x float> @mul_sub_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vsub(v0.sf,v1.sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = [[V3]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V4]].qf32
+; CHECK:         qf32 = vmpy([[V6]].qf32,[[V5]].qf32)
+label0:
+  %v0 = fsub <32 x float> %a0, %a1
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when one is from adder, another from multiplier
+define <32 x float> @mul_add_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = v0.sf
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = v1.sf
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V3]].qf32
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V4]].qf32,[[V5]].qf32)
+; CHECK:         qf32 = vmpy([[V6]].qf32,[[V7]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+define <32 x float> @mul_mul_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+label0:
+; CHECK-LABEL: mul_mul_mul
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = v0.sf
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = v2.sf
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = v1.sf
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = vmpy([[V3]].qf32,[[V4]].qf32)
+; CHECK-DAG:     [[V7:v[0-9]+]].qf32 = vmpy([[V3]].qf32,[[V5]].qf32)
+; CHECK:         qf32 = vmpy([[V6]].qf32,[[V7]].qf32)
+  %v1 = fmul <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv81" "target-features"="+hvx-length128b,+hvx-qfloat,+hvxv81,+v81,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-lossy-mul-qf32.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-lossy-mul-qf32.ll
new file mode 100644
index 0000000000000..1fb3e9ee426ff
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-lossy-mul-qf32.ll
@@ -0,0 +1,98 @@
+; Tests lossy-subnormals mode for XQFloat multiplication 32-bit for v81.
+; The normamlization sequence is different than v79.
+
+; RUN: llc -O2 -march=hexagon -mcpu=hexagonv81 -force-hvx-float -enable-xqf-gen=true -enable-rem-conv=true \
+; RUN: -hexagon-qfloat-mode=lossy  -mattr=+hvxv81,+hvx-length128B  < %s | FileCheck %s
+
+; Test qf32 = vmpy(sf, sf)
+; Normalization of inputs
+define <32 x float> @mul_add_0(<32 x float> %a0, <32 x float> %a1) #0 {
+; CHECK-LABEL: mul_add_0:
+; CHECK:     qf32 = vmpy(v0.sf,v1.sf)
+label0:
+  %v3 = fmul <32 x float> %a0, %a1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(sf ,qf32) when only one input is from vadd instruction
+define <32 x float> @mul_add_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_1:
+; CHECK:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK:     [[V4:v[0-9]+]].sf = [[V3]].qf32
+; CHECK:     qf32 = vmpy(v1.sf,[[V4]].sf)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vadd instruction
+define <32 x float> @mul_add_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vadd(v0.sf,v1.sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = [[V4]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V3]].qf32
+; CHECK:         qf32 = vmpy([[V5]].qf32,[[V6]].qf32)
+label0:
+  %v0 = fadd <32 x float> %a0, %a1
+  %v1 = fadd <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when only first input is from vsub instruction
+define <32 x float> @mul_sub_1(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_1:
+; CHECK:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK:     [[V4:v[0-9]+]].sf = [[V3]].qf32
+; CHECK:     qf32 = vmpy(v1.sf,[[V4]].sf)
+label0:
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %a1, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32 ,qf32) when both inputs are from vsub instruction
+define <32 x float> @mul_sub_3(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_sub_3:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vsub(v0.sf,v2.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vsub(v0.sf,v1.sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = [[V4]].qf32
+; CHECK-DAG:     [[V6:v[0-9]+]].qf32 = [[V3]].qf32
+; CHECK:         qf32 = vmpy([[V5]].qf32,[[V6]].qf32)
+label0:
+  %v0 = fsub <32 x float> %a0, %a1
+  %v1 = fsub <32 x float> %a0, %a2
+  %v3 = fmul <32 x float> %v0, %v1
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when one is from adder, another from multiplier
+define <32 x float> @mul_add_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+; CHECK-LABEL: mul_add_mul:
+; CHECK-DAG:     [[V3:v[0-9]+]].qf32 = vmpy(v0.sf,v1.sf)
+; CHECK-DAG:     [[V4:v[0-9]+]].qf32 = vadd(v0.sf,v2.sf)
+; CHECK-DAG:     [[V5:v[0-9]+]].qf32 = [[V4]].qf32
+; CHECK-DAG:     qf32 = vmpy([[V5]].qf32,[[V3]].qf32)
+label0:
+  %v1 = fadd <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+; Test qf32 = vmpy(qf32, qf32) when both are from multiplier
+define <32 x float> @mul_mul_mul(<32 x float> %a0, <32 x float> %a1, <32 x float> %a2) #0 {
+label0:
+; CHECK-LABEL: mul_mul_mul
+; CHECK: [[V3:v[0-9]+]].qf32 = vmpy(v0.sf,v1.sf)
+; CHECK: [[V4:v[0-9]+]].qf32 = vmpy(v0.sf,v2.sf)
+; CHECK: qf32 = vmpy([[V4]].qf32,[[V3]].qf32)
+  %v1 = fmul <32 x float> %a0, %a2
+  %v2 = fmul <32 x float> %a0, %a1
+  %v3 = fmul <32 x float> %v1, %v2
+  ret <32 x float> %v3
+}
+
+attributes #0 = { nofree nosync nounwind "approx-func-fp-math"="true" "frame-pointer"="all" "no-infs-fp-math"="true" "no-nans-fp-math"="true" "no-signed-zeros-fp-math"="true" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv81" "target-features"="+hvx-length128b,+hvx-qfloat,+hvxv81,+v81,-long-calls" "unsafe-fp-math"="true" }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-vsub.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-vsub.ll
new file mode 100644
index 0000000000000..d18ed85713272
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-v81/xqf-v81-vsub.ll
@@ -0,0 +1,164 @@
+; For v81, tests if correct vsub instructions are generated under different conditions
+
+; RUN:   llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv81 -mattr=+hvxv81,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=ieee < %s | FileCheck %s
+
+; The convert instruction before vsub should be removed and vsub opcode changed to take in qf32 as op2
+define dso_local <32 x i32> @sub1_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub1_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vsub([[V1]].sf,[[V3]].qf32)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and vsub opcode to be changed to take in qf32 type as op1.
+define dso_local <32 x i32> @sub2_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub2_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vsub([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instruction before vsub should be removed and vsub opcode changed to take in qf16 as op2
+define dso_local <32 x i32> @sub1_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub1_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vsub([[V1]].hf,[[V3]].qf16)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and vsub opcode to be changed to take in qf16 type as op1.
+define dso_local <32 x i32> @sub2_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub2_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vsub([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vadd can be removed.
+define dso_local <32 x i32> @add1_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add1_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vadd([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vadd can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @add2_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add2_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vadd([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vadd can be removed.
+define dso_local <32 x i32> @add1_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add1_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vadd([[V3]].qf16,[[V1]].hf)
+entry:                                                                                                                                                                                                               %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)                                                                                                                                              %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)                                                                                                                                       %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4                                                                                                                                                                                                }
+
+; The convert instr before vadd can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @add2_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add2_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vadd([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vmul can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @mpy2_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: mpy2_qf32
+; CHECK: qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vmpy(v{{[0-9]+}}.qf32,v{{[0-9]+}}.qf32)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vmpy.qf32.sf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vmul can be removed.
+define dso_local <32 x i32> @mpy1_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: mpy1_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf32 = vmpy([[V3]].qf16,[[V1]].hf)
+entry:                                                                                                                                                                                                               %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)                                                                                                                                              %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)                                                                                                                                       %4 = tail call <32 x i32> @llvm.hexagon.V6.vmpy.qf16.hf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4                                                                                                                                                                                                }
+
+; The convert instr before vmul can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @mpy2_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: mpy2_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf32 = vmpy([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vmpy.qf16.hf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+
+
+declare <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32) #1
+declare <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32>, <32 x i32>) #1
+
+attributes #0 = { nounwind "frame-pointer"="all" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-features"="-long-calls,-small-data" }
+attributes #1 = { nocallback nofree nosync nounwind willreturn memory(none) }
diff --git a/llvm/test/CodeGen/Hexagon/autohvx/xqf-vsub.ll b/llvm/test/CodeGen/Hexagon/autohvx/xqf-vsub.ll
new file mode 100644
index 0000000000000..c10ff15791b23
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/autohvx/xqf-vsub.ll
@@ -0,0 +1,130 @@
+; Tests if correct vsub instructions are generated under different conditions
+
+; RUN:   llc -mtriple=hexagon-unknown-elf -mhvx -mcpu=hexagonv79 -mattr=+hvxv79,+hvx-length128b,+hvx-qfloat -enable-xqf-gen=true \
+; RUN: -enable-rem-conv=true -hexagon-qfloat-mode=ieee < %s | FileCheck %s
+
+; The convert instruction before vsub should remain as it is.
+define dso_local <32 x i32> @sub1_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub1_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: [[V4:v[0-9]+]].sf = [[V3]].qf32
+; CHECK: qf32 = vsub([[V1]].sf,[[V4]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and vsub opcode to be changed to take in qf32 type as op1.
+define dso_local <32 x i32> @sub2_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub2_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vsub([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instruction before vsub should remain as it is.
+define dso_local <32 x i32> @sub1_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub1_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: [[V4:v[0-9]+]].hf = [[V3]].qf16
+; CHECK: qf16 = vsub([[V1]].hf,[[V4]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and vsub opcode to be changed to take in qf16 type as op1.
+define dso_local <32 x i32> @sub2_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: sub2_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vsub([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vadd can be removed.
+define dso_local <32 x i32> @add1_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add1_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vadd([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @add2_qf32(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add2_qf32
+; CHECK: [[V3:v[0-9]+]].qf32 = vadd([[V1:v[0-9]+]].sf
+; CHECK: qf32 = vadd([[V3]].qf32,[[V1]].sf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vadd can be removed.
+define dso_local <32 x i32> @add1_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add1_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vadd([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %3, <32 x i32> %0)
+  ret <32 x i32> %4
+}
+
+; The convert instr before vsub can be removed and ops to last vadd can be interchanged
+define dso_local <32 x i32> @add2_qf16(i32 noundef %input1, i32 noundef %input2, i32 noundef %size) local_unnamed_addr #0 {
+; CHECK-LABEL: add2_qf16
+; CHECK: [[V3:v[0-9]+]].qf16 = vadd([[V1:v[0-9]+]].hf
+; CHECK: qf16 = vadd([[V3]].qf16,[[V1]].hf)
+entry:
+  %0 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input1)
+  %1 = tail call <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32 %input2)
+  %2 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %1)
+  %3 = tail call <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32> %2)
+  %4 = tail call <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32> %0, <32 x i32> %3)
+  ret <32 x i32> %4
+}
+
+declare <32 x i32> @llvm.hexagon.V6.vsub.sf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vsub.hf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.lvsplatw.128B(i32) #1
+declare <32 x i32> @llvm.hexagon.V6.lvsplath.128B(i32) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.sf.qf32.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vconv.hf.qf16.128B(<32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.sf.128B(<32 x i32>, <32 x i32>) #1
+declare <32 x i32> @llvm.hexagon.V6.vadd.hf.128B(<32 x i32>, <32 x i32>) #1
+
+attributes #0 = { nounwind "frame-pointer"="all" "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="hexagonv79" "target-features"="+hvx-length128b,+hvxv79,+v79,-long-calls,-small-data" }
+attributes #1 = { nocallback nofree nosync nounwind willreturn memory(none) }



More information about the llvm-commits mailing list