[llvm] [Hexagon] Fuse vminub intrinsic pair (PR #225153)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 23:27:53 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-hexagon

Author: Yashas Andaluri (yandalur)

<details>
<summary>Changes</summary>

Fuse matching A2_vminub and C2_cmpgtup intrinsics into the dual-output A6_vminub_RdP instruction in HexagonPeephole.
Added a command-line flag to toggle this behaviour. This change also removes unused generic intrinsic mapping support.

---
Full diff: https://github.com/llvm/llvm-project/pull/225153.diff


5 Files Affected:

- (modified) llvm/lib/Target/Hexagon/Hexagon.td (-8) 
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp (-13) 
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.h (+2-5) 
- (modified) llvm/lib/Target/Hexagon/HexagonPeephole.cpp (+101-7) 
- (added) llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll (+58) 


``````````diff
diff --git a/llvm/lib/Target/Hexagon/Hexagon.td b/llvm/lib/Target/Hexagon/Hexagon.td
index 7625307172abf..bd00a5e665035 100644
--- a/llvm/lib/Target/Hexagon/Hexagon.td
+++ b/llvm/lib/Target/Hexagon/Hexagon.td
@@ -215,7 +215,6 @@ class PredNewRel: PredRel;
 class NewValueRel: PredNewRel;
 class AddrModeRel: NewValueRel;
 class PostInc_BaseImm;
-class IntrinsicsRel;
 // ... through here.
 
 //===----------------------------------------------------------------------===//
@@ -397,13 +396,6 @@ def takenBranchPrediction : InstrMapping {
   let ValueCols = [["true"]];
 }
 
-def getRealHWInstr : InstrMapping {
-  let FilterClass = "IntrinsicsRel";
-  let RowFields = ["BaseOpcode"];
-  let ColFields = ["InstrType"];
-  let KeyCol = ["Pseudo"];
-  let ValueCols = [["Pseudo"], ["Real"]];
-}
 //===----------------------------------------------------------------------===//
 // Register File, Instruction Descriptions
 //===----------------------------------------------------------------------===//
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
index b31724f094ec7..00cf91ee575bc 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
@@ -3193,11 +3193,6 @@ bool HexagonInstrInfo::hasNonExtEquivalent(const MachineInstr &MI) const {
   return false;
 }
 
-bool HexagonInstrInfo::hasPseudoInstrPair(const MachineInstr &MI) const {
-  return Hexagon::getRealHWInstr(MI.getOpcode(),
-                                 Hexagon::InstrType_Pseudo) >= 0;
-}
-
 bool HexagonInstrInfo::hasUncondBranch(const MachineBasicBlock *B)
       const {
   MachineBasicBlock::const_iterator I = B->getFirstTerminator(), E = B->end();
@@ -4337,10 +4332,6 @@ HexagonII::SubInstructionGroup HexagonInstrInfo::getDuplexCandidateGroup(
   return HexagonII::HSIG_None;
 }
 
-short HexagonInstrInfo::getEquivalentHWInstr(const MachineInstr &MI) const {
-  return Hexagon::getRealHWInstr(MI.getOpcode(), Hexagon::InstrType_Real);
-}
-
 unsigned HexagonInstrInfo::getInstrTimingClassLatency(
       const InstrItineraryData *ItinData, const MachineInstr &MI) const {
   // Default to one cycle for no itinerary. However, an "empty" itinerary may
@@ -4604,10 +4595,6 @@ bool HexagonInstrInfo::getPredReg(ArrayRef<MachineOperand> Cond,
   return true;
 }
 
-short HexagonInstrInfo::getPseudoInstrPair(const MachineInstr &MI) const {
-  return Hexagon::getRealHWInstr(MI.getOpcode(), Hexagon::InstrType_Pseudo);
-}
-
 short HexagonInstrInfo::getRegForm(const MachineInstr &MI) const {
   return Hexagon::getRegForm(MI.getOpcode());
 }
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.h b/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
index 7bcd2005af10d..0c2353f282601 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
@@ -438,7 +438,6 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
   bool doesNotReturn(const MachineInstr &CallMI) const;
   bool hasEHLabel(const MachineBasicBlock *B) const;
   bool hasNonExtEquivalent(const MachineInstr &MI) const;
-  bool hasPseudoInstrPair(const MachineInstr &MI) const;
   bool hasUncondBranch(const MachineBasicBlock *B) const;
   bool mayBeCurLoad(const MachineInstr &MI) const;
   bool mayBeNewStore(const MachineInstr &MI) const;
@@ -469,9 +468,8 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
   int getDotNewPredOp(const MachineInstr &MI,
                       const MachineBranchProbabilityInfo *MBPI) const;
   int getDotOldOp(const MachineInstr &MI) const;
-  HexagonII::SubInstructionGroup getDuplexCandidateGroup(const MachineInstr &MI)
-                                                         const;
-  short getEquivalentHWInstr(const MachineInstr &MI) const;
+  HexagonII::SubInstructionGroup
+  getDuplexCandidateGroup(const MachineInstr &MI) const;
   unsigned getInstrTimingClassLatency(const InstrItineraryData *ItinData,
                                       const MachineInstr &MI) const;
   bool getInvertedPredSense(SmallVectorImpl<MachineOperand> &Cond) const;
@@ -482,7 +480,6 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
   short getNonExtOpcode(const MachineInstr &MI) const;
   bool getPredReg(ArrayRef<MachineOperand> Cond, Register &PredReg,
                   unsigned &PredRegPos, RegState &PredRegFlags) const;
-  short getPseudoInstrPair(const MachineInstr &MI) const;
   short getRegForm(const MachineInstr &MI) const;
   unsigned getSize(const MachineInstr &MI) const;
   uint64_t getType(const MachineInstr &MI) const;
diff --git a/llvm/lib/Target/Hexagon/HexagonPeephole.cpp b/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
index 67f5589137bbd..a135871e8dafb 100644
--- a/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
@@ -26,16 +26,26 @@
 //     ...
 //     JMP_cNot killed %15, <%bb.1>, implicit dead %pc;
 //
-// Note: The peephole pass makes the instrucstions like
+// 3. Fuse the A2_vminub/C2_cmpgtup intrinsic pair, which share inputs, into the
+//    dual-output A6_vminub_RdP hardware instruction.
+//    %1 = A2_vminub %a, %b
+//    %2 = C2_cmpgtup %a, %b
+// turning it into
+//    %3, %4 = A6_vminub_RdP %a, %b
+// (Hexagon has no multi-output intrinsics, so the two results are produced by
+// separate intrinsics that this pass recombines.)
+//
+// Note: The first two transformations make instructions like
 // %170 = SXTW %166 or %16 = NOT_p killed %15
-// redundant and relies on some form of dead removal instructions, like
-// DCE or DIE to actually eliminate them.
+// redundant. A dead-instruction removal pass, such as DCE or DIE, eliminates
+// them.
 
 //===----------------------------------------------------------------------===//
 
 #include "Hexagon.h"
 #include "HexagonTargetMachine.h"
 #include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/CodeGen/MachineFunction.h"
 #include "llvm/CodeGen/MachineFunctionPass.h"
@@ -68,17 +78,23 @@ static cl::opt<bool>
                       cl::init(true),
                       cl::desc("Disable Optimization of extensions to i64."));
 
+static cl::opt<bool>
+    FuseIntrinsicVMinUB("hexagon-fuse-intrinsic-vminub", cl::Hidden,
+                        cl::init(true),
+                        cl::desc("Fuse A2_vminub/C2_cmpgtup into "
+                                 "A6_vminub_RdP."));
+
 namespace {
   struct HexagonPeephole : public MachineFunctionPass {
-    const HexagonInstrInfo    *QII;
-    const HexagonRegisterInfo *QRI;
-    const MachineRegisterInfo *MRI;
+    const HexagonInstrInfo *QII;
+    MachineRegisterInfo *MRI;
 
   public:
     static char ID;
     HexagonPeephole() : MachineFunctionPass(ID) {}
 
     bool runOnMachineFunction(MachineFunction &MF) override;
+    bool fuseIntrinsicVMinUB(MachineFunction &MF);
 
     StringRef getPassName() const override {
       return "Hexagon optimize redundant zero and size extends";
@@ -99,8 +115,9 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
   if (skipFunction(MF.getFunction()))
     return false;
 
+  bool Changed = false;
+
   QII = static_cast<const HexagonInstrInfo *>(MF.getSubtarget().getInstrInfo());
-  QRI = MF.getSubtarget<HexagonSubtarget>().getRegisterInfo();
   MRI = &MF.getRegInfo();
 
   DenseMap<unsigned, unsigned> PeepholeMap;
@@ -199,6 +216,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
             // Change the 1st operand.
             MI.removeOperand(1);
             MI.addOperand(MachineOperand::CreateReg(PeepholeSrc, false));
+            Changed = true;
           } else  {
             DenseMap<unsigned, std::pair<unsigned, unsigned> >::iterator DI =
               PeepholeDoubleRegsMap.find(SrcReg);
@@ -209,6 +227,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
                   PeepholeSrc.first, false /*isDef*/, false /*isImp*/,
                   false /*isKill*/, false /*isDead*/, false /*isUndef*/,
                   false /*isEarlyClobber*/, PeepholeSrc.second));
+              Changed = true;
             }
           }
         }
@@ -232,6 +251,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
                 MRI->clearKillFlags(PeepholeSrc);
                 int NewOp = QII->getInvertedPredicatedOpcode(MI.getOpcode());
                 MI.setDesc(QII->get(NewOp));
+                Changed = true;
                 Done = true;
               }
             }
@@ -266,6 +286,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
                   .add(MI.getOperand(S1));
               MRI->clearKillFlags(POrig);
               MI.eraseFromParent();
+              Changed = true;
             }
           } // if (NewOp)
         } // if (!Done)
@@ -274,9 +295,82 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
 
     } // Instruction
   } // Basic Block
+
+  if (FuseIntrinsicVMinUB)
+    Changed |= fuseIntrinsicVMinUB(MF);
+
+  return Changed;
+}
+
+// Return true if both instructions have identical input operands in order.
+static bool hasCommonInputOps(const MachineInstr *I1, const MachineInstr *I2) {
+  if (I1->getNumOperands() != I2->getNumOperands())
+    return false;
+
+  for (unsigned i = 0, e = I1->getNumOperands(); i != e; ++i) {
+    const MachineOperand &Op1 = I1->getOperand(i);
+    if (!Op1.isDef() && !Op1.isIdenticalTo(I2->getOperand(i)))
+      return false;
+  }
   return true;
 }
 
+bool HexagonPeephole::fuseIntrinsicVMinUB(MachineFunction &MF) {
+  bool Changed = false;
+
+  for (MachineBasicBlock &MBB : MF) {
+    SmallPtrSet<MachineInstr *, 8> DeadMIs;
+
+    for (MachineInstr &MI : MBB) {
+      unsigned Opc = MI.getOpcode();
+      if ((Opc != Hexagon::A2_vminub && Opc != Hexagon::C2_cmpgtup) ||
+          DeadMIs.count(&MI))
+        continue;
+
+      unsigned SiblingOpc =
+          Opc == Hexagon::A2_vminub ? Hexagon::C2_cmpgtup : Hexagon::A2_vminub;
+      MachineInstr *Sibling = nullptr;
+      auto It = MI.getIterator();
+      for (++It; It != MBB.end(); ++It)
+        if (!DeadMIs.count(&*It) && It->getOpcode() == SiblingOpc &&
+            hasCommonInputOps(&MI, &*It)) {
+          Sibling = &*It;
+          break;
+        }
+      if (!Sibling)
+        continue;
+
+      MachineInstr *VMin = Opc == Hexagon::A2_vminub ? &MI : Sibling;
+      MachineInstr *Cmp = Opc == Hexagon::A2_vminub ? Sibling : &MI;
+      Register VMinReg =
+          MRI->createVirtualRegister(&Hexagon::DoubleRegsRegClass);
+      Register CmpReg = MRI->createVirtualRegister(&Hexagon::PredRegsRegClass);
+
+      BuildMI(MBB, MI.getIterator(), MI.getDebugLoc(),
+              QII->get(Hexagon::A6_vminub_RdP), VMinReg)
+          .addReg(CmpReg, RegState::Define)
+          .add(VMin->getOperand(1))
+          .add(VMin->getOperand(2));
+
+      Register VMinDef = VMin->getOperand(0).getReg();
+      Register CmpDef = Cmp->getOperand(0).getReg();
+      MRI->replaceRegWith(VMinDef, VMinReg);
+      VMin->getOperand(0).setReg(VMinDef);
+      MRI->replaceRegWith(CmpDef, CmpReg);
+      Cmp->getOperand(0).setReg(CmpDef);
+
+      DeadMIs.insert(&MI);
+      DeadMIs.insert(Sibling);
+      Changed = true;
+    }
+
+    for (MachineInstr *MI : DeadMIs)
+      MI->eraseFromParent();
+  }
+
+  return Changed;
+}
+
 FunctionPass *llvm::createHexagonPeephole() {
   return new HexagonPeephole();
 }
diff --git a/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll b/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll
new file mode 100644
index 0000000000000..f9b464c878a6a
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll
@@ -0,0 +1,58 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv73 -O2 -hexagon-fuse-intrinsic-vminub=true < %s | FileCheck %s --check-prefix=FUSE
+; RUN: llc -march=hexagon -mcpu=hexagonv73 -O2 -hexagon-fuse-intrinsic-vminub=false < %s | FileCheck %s --check-prefix=NOFUSE
+
+ at g = external global i64
+ at r = external global i32
+
+; FUSE-LABEL: fused:
+; FUSE-NOT: cmp.gtu
+; FUSE: [[V:r[0-9]+:[0-9]+]],[[P:p[0-3]]] = vminub(
+; FUSE-DAG: r{{[0-9]+}} = [[P]]
+; FUSE-DAG: memd(gp+#g) = [[V]]
+define i32 @fused(i64 %a, i64 %b) {
+entry:
+  %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+  %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %a, i64 %b)
+  store i64 %v, i64* @g, align 8
+  store i32 %p, i32* @r, align 4
+  %pe = zext i32 %p to i64
+  %sum = add i64 %v, %pe
+  %ret = trunc i64 %sum to i32
+  ret i32 %ret
+}
+
+; NOFUSE-LABEL: fused:
+; NOFUSE-DAG: {{p[0-3]}} = cmp.gtu(
+; NOFUSE-DAG: {{r[0-9]+:[0-9]+}} = vminub(
+; NOFUSE-NOT: {{r[0-9]+:[0-9]+}},{{p[0-3]}} = vminub(
+
+; FUSE-LABEL: only_vmin:
+; FUSE-NOT: cmp.gtu
+; FUSE: {{r[0-9]+:[0-9]+}} = vminub(
+define i64 @only_vmin(i64 %a, i64 %b) {
+  %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %a, i64 %b)
+  ret i64 %v
+}
+
+; FUSE-LABEL: only_cmp:
+; FUSE: {{p[0-3]}} = cmp.gtu(
+define i32 @only_cmp(i64 %a, i64 %b) {
+  %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+  ret i32 %p
+}
+
+; FUSE-LABEL: mismatch:
+; FUSE-NOT: {{r[0-9]+:[0-9]+}},{{p[0-3]}} = vminub(
+; FUSE-DAG: {{r[0-9]+:[0-9]+}} = vminub(
+; FUSE-DAG: {{p[0-3]}} = cmp.gtu(
+define i32 @mismatch(i64 %a, i64 %b) {
+  %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+  %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %b, i64 %a)
+  %pe = zext i32 %p to i64
+  %sum = add i64 %v, %pe
+  %ret = trunc i64 %sum to i32
+  ret i32 %ret
+}
+
+declare i64 @llvm.hexagon.A2.vminub(i64, i64)
+declare i32 @llvm.hexagon.C2.cmpgtup(i64, i64)

``````````

</details>


https://github.com/llvm/llvm-project/pull/225153


More information about the llvm-commits mailing list