[llvm] [Hexagon] Fuse vminub intrinsic pair (PR #225153)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 21 23:27:53 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-hexagon
Author: Yashas Andaluri (yandalur)
<details>
<summary>Changes</summary>
Fuse matching A2_vminub and C2_cmpgtup intrinsics into the dual-output A6_vminub_RdP instruction in HexagonPeephole.
Added a command-line flag to toggle this behaviour. This change also removes unused generic intrinsic mapping support.
---
Full diff: https://github.com/llvm/llvm-project/pull/225153.diff
5 Files Affected:
- (modified) llvm/lib/Target/Hexagon/Hexagon.td (-8)
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp (-13)
- (modified) llvm/lib/Target/Hexagon/HexagonInstrInfo.h (+2-5)
- (modified) llvm/lib/Target/Hexagon/HexagonPeephole.cpp (+101-7)
- (added) llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll (+58)
``````````diff
diff --git a/llvm/lib/Target/Hexagon/Hexagon.td b/llvm/lib/Target/Hexagon/Hexagon.td
index 7625307172abf..bd00a5e665035 100644
--- a/llvm/lib/Target/Hexagon/Hexagon.td
+++ b/llvm/lib/Target/Hexagon/Hexagon.td
@@ -215,7 +215,6 @@ class PredNewRel: PredRel;
class NewValueRel: PredNewRel;
class AddrModeRel: NewValueRel;
class PostInc_BaseImm;
-class IntrinsicsRel;
// ... through here.
//===----------------------------------------------------------------------===//
@@ -397,13 +396,6 @@ def takenBranchPrediction : InstrMapping {
let ValueCols = [["true"]];
}
-def getRealHWInstr : InstrMapping {
- let FilterClass = "IntrinsicsRel";
- let RowFields = ["BaseOpcode"];
- let ColFields = ["InstrType"];
- let KeyCol = ["Pseudo"];
- let ValueCols = [["Pseudo"], ["Real"]];
-}
//===----------------------------------------------------------------------===//
// Register File, Instruction Descriptions
//===----------------------------------------------------------------------===//
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
index b31724f094ec7..00cf91ee575bc 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.cpp
@@ -3193,11 +3193,6 @@ bool HexagonInstrInfo::hasNonExtEquivalent(const MachineInstr &MI) const {
return false;
}
-bool HexagonInstrInfo::hasPseudoInstrPair(const MachineInstr &MI) const {
- return Hexagon::getRealHWInstr(MI.getOpcode(),
- Hexagon::InstrType_Pseudo) >= 0;
-}
-
bool HexagonInstrInfo::hasUncondBranch(const MachineBasicBlock *B)
const {
MachineBasicBlock::const_iterator I = B->getFirstTerminator(), E = B->end();
@@ -4337,10 +4332,6 @@ HexagonII::SubInstructionGroup HexagonInstrInfo::getDuplexCandidateGroup(
return HexagonII::HSIG_None;
}
-short HexagonInstrInfo::getEquivalentHWInstr(const MachineInstr &MI) const {
- return Hexagon::getRealHWInstr(MI.getOpcode(), Hexagon::InstrType_Real);
-}
-
unsigned HexagonInstrInfo::getInstrTimingClassLatency(
const InstrItineraryData *ItinData, const MachineInstr &MI) const {
// Default to one cycle for no itinerary. However, an "empty" itinerary may
@@ -4604,10 +4595,6 @@ bool HexagonInstrInfo::getPredReg(ArrayRef<MachineOperand> Cond,
return true;
}
-short HexagonInstrInfo::getPseudoInstrPair(const MachineInstr &MI) const {
- return Hexagon::getRealHWInstr(MI.getOpcode(), Hexagon::InstrType_Pseudo);
-}
-
short HexagonInstrInfo::getRegForm(const MachineInstr &MI) const {
return Hexagon::getRegForm(MI.getOpcode());
}
diff --git a/llvm/lib/Target/Hexagon/HexagonInstrInfo.h b/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
index 7bcd2005af10d..0c2353f282601 100644
--- a/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
+++ b/llvm/lib/Target/Hexagon/HexagonInstrInfo.h
@@ -438,7 +438,6 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
bool doesNotReturn(const MachineInstr &CallMI) const;
bool hasEHLabel(const MachineBasicBlock *B) const;
bool hasNonExtEquivalent(const MachineInstr &MI) const;
- bool hasPseudoInstrPair(const MachineInstr &MI) const;
bool hasUncondBranch(const MachineBasicBlock *B) const;
bool mayBeCurLoad(const MachineInstr &MI) const;
bool mayBeNewStore(const MachineInstr &MI) const;
@@ -469,9 +468,8 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
int getDotNewPredOp(const MachineInstr &MI,
const MachineBranchProbabilityInfo *MBPI) const;
int getDotOldOp(const MachineInstr &MI) const;
- HexagonII::SubInstructionGroup getDuplexCandidateGroup(const MachineInstr &MI)
- const;
- short getEquivalentHWInstr(const MachineInstr &MI) const;
+ HexagonII::SubInstructionGroup
+ getDuplexCandidateGroup(const MachineInstr &MI) const;
unsigned getInstrTimingClassLatency(const InstrItineraryData *ItinData,
const MachineInstr &MI) const;
bool getInvertedPredSense(SmallVectorImpl<MachineOperand> &Cond) const;
@@ -482,7 +480,6 @@ class HexagonInstrInfo : public HexagonGenInstrInfo {
short getNonExtOpcode(const MachineInstr &MI) const;
bool getPredReg(ArrayRef<MachineOperand> Cond, Register &PredReg,
unsigned &PredRegPos, RegState &PredRegFlags) const;
- short getPseudoInstrPair(const MachineInstr &MI) const;
short getRegForm(const MachineInstr &MI) const;
unsigned getSize(const MachineInstr &MI) const;
uint64_t getType(const MachineInstr &MI) const;
diff --git a/llvm/lib/Target/Hexagon/HexagonPeephole.cpp b/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
index 67f5589137bbd..a135871e8dafb 100644
--- a/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonPeephole.cpp
@@ -26,16 +26,26 @@
// ...
// JMP_cNot killed %15, <%bb.1>, implicit dead %pc;
//
-// Note: The peephole pass makes the instrucstions like
+// 3. Fuse the A2_vminub/C2_cmpgtup intrinsic pair, which share inputs, into the
+// dual-output A6_vminub_RdP hardware instruction.
+// %1 = A2_vminub %a, %b
+// %2 = C2_cmpgtup %a, %b
+// turning it into
+// %3, %4 = A6_vminub_RdP %a, %b
+// (Hexagon has no multi-output intrinsics, so the two results are produced by
+// separate intrinsics that this pass recombines.)
+//
+// Note: The first two transformations make instructions like
// %170 = SXTW %166 or %16 = NOT_p killed %15
-// redundant and relies on some form of dead removal instructions, like
-// DCE or DIE to actually eliminate them.
+// redundant. A dead-instruction removal pass, such as DCE or DIE, eliminates
+// them.
//===----------------------------------------------------------------------===//
#include "Hexagon.h"
#include "HexagonTargetMachine.h"
#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/MachineFunction.h"
#include "llvm/CodeGen/MachineFunctionPass.h"
@@ -68,17 +78,23 @@ static cl::opt<bool>
cl::init(true),
cl::desc("Disable Optimization of extensions to i64."));
+static cl::opt<bool>
+ FuseIntrinsicVMinUB("hexagon-fuse-intrinsic-vminub", cl::Hidden,
+ cl::init(true),
+ cl::desc("Fuse A2_vminub/C2_cmpgtup into "
+ "A6_vminub_RdP."));
+
namespace {
struct HexagonPeephole : public MachineFunctionPass {
- const HexagonInstrInfo *QII;
- const HexagonRegisterInfo *QRI;
- const MachineRegisterInfo *MRI;
+ const HexagonInstrInfo *QII;
+ MachineRegisterInfo *MRI;
public:
static char ID;
HexagonPeephole() : MachineFunctionPass(ID) {}
bool runOnMachineFunction(MachineFunction &MF) override;
+ bool fuseIntrinsicVMinUB(MachineFunction &MF);
StringRef getPassName() const override {
return "Hexagon optimize redundant zero and size extends";
@@ -99,8 +115,9 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
if (skipFunction(MF.getFunction()))
return false;
+ bool Changed = false;
+
QII = static_cast<const HexagonInstrInfo *>(MF.getSubtarget().getInstrInfo());
- QRI = MF.getSubtarget<HexagonSubtarget>().getRegisterInfo();
MRI = &MF.getRegInfo();
DenseMap<unsigned, unsigned> PeepholeMap;
@@ -199,6 +216,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
// Change the 1st operand.
MI.removeOperand(1);
MI.addOperand(MachineOperand::CreateReg(PeepholeSrc, false));
+ Changed = true;
} else {
DenseMap<unsigned, std::pair<unsigned, unsigned> >::iterator DI =
PeepholeDoubleRegsMap.find(SrcReg);
@@ -209,6 +227,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
PeepholeSrc.first, false /*isDef*/, false /*isImp*/,
false /*isKill*/, false /*isDead*/, false /*isUndef*/,
false /*isEarlyClobber*/, PeepholeSrc.second));
+ Changed = true;
}
}
}
@@ -232,6 +251,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
MRI->clearKillFlags(PeepholeSrc);
int NewOp = QII->getInvertedPredicatedOpcode(MI.getOpcode());
MI.setDesc(QII->get(NewOp));
+ Changed = true;
Done = true;
}
}
@@ -266,6 +286,7 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
.add(MI.getOperand(S1));
MRI->clearKillFlags(POrig);
MI.eraseFromParent();
+ Changed = true;
}
} // if (NewOp)
} // if (!Done)
@@ -274,9 +295,82 @@ bool HexagonPeephole::runOnMachineFunction(MachineFunction &MF) {
} // Instruction
} // Basic Block
+
+ if (FuseIntrinsicVMinUB)
+ Changed |= fuseIntrinsicVMinUB(MF);
+
+ return Changed;
+}
+
+// Return true if both instructions have identical input operands in order.
+static bool hasCommonInputOps(const MachineInstr *I1, const MachineInstr *I2) {
+ if (I1->getNumOperands() != I2->getNumOperands())
+ return false;
+
+ for (unsigned i = 0, e = I1->getNumOperands(); i != e; ++i) {
+ const MachineOperand &Op1 = I1->getOperand(i);
+ if (!Op1.isDef() && !Op1.isIdenticalTo(I2->getOperand(i)))
+ return false;
+ }
return true;
}
+bool HexagonPeephole::fuseIntrinsicVMinUB(MachineFunction &MF) {
+ bool Changed = false;
+
+ for (MachineBasicBlock &MBB : MF) {
+ SmallPtrSet<MachineInstr *, 8> DeadMIs;
+
+ for (MachineInstr &MI : MBB) {
+ unsigned Opc = MI.getOpcode();
+ if ((Opc != Hexagon::A2_vminub && Opc != Hexagon::C2_cmpgtup) ||
+ DeadMIs.count(&MI))
+ continue;
+
+ unsigned SiblingOpc =
+ Opc == Hexagon::A2_vminub ? Hexagon::C2_cmpgtup : Hexagon::A2_vminub;
+ MachineInstr *Sibling = nullptr;
+ auto It = MI.getIterator();
+ for (++It; It != MBB.end(); ++It)
+ if (!DeadMIs.count(&*It) && It->getOpcode() == SiblingOpc &&
+ hasCommonInputOps(&MI, &*It)) {
+ Sibling = &*It;
+ break;
+ }
+ if (!Sibling)
+ continue;
+
+ MachineInstr *VMin = Opc == Hexagon::A2_vminub ? &MI : Sibling;
+ MachineInstr *Cmp = Opc == Hexagon::A2_vminub ? Sibling : &MI;
+ Register VMinReg =
+ MRI->createVirtualRegister(&Hexagon::DoubleRegsRegClass);
+ Register CmpReg = MRI->createVirtualRegister(&Hexagon::PredRegsRegClass);
+
+ BuildMI(MBB, MI.getIterator(), MI.getDebugLoc(),
+ QII->get(Hexagon::A6_vminub_RdP), VMinReg)
+ .addReg(CmpReg, RegState::Define)
+ .add(VMin->getOperand(1))
+ .add(VMin->getOperand(2));
+
+ Register VMinDef = VMin->getOperand(0).getReg();
+ Register CmpDef = Cmp->getOperand(0).getReg();
+ MRI->replaceRegWith(VMinDef, VMinReg);
+ VMin->getOperand(0).setReg(VMinDef);
+ MRI->replaceRegWith(CmpDef, CmpReg);
+ Cmp->getOperand(0).setReg(CmpDef);
+
+ DeadMIs.insert(&MI);
+ DeadMIs.insert(Sibling);
+ Changed = true;
+ }
+
+ for (MachineInstr *MI : DeadMIs)
+ MI->eraseFromParent();
+ }
+
+ return Changed;
+}
+
FunctionPass *llvm::createHexagonPeephole() {
return new HexagonPeephole();
}
diff --git a/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll b/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll
new file mode 100644
index 0000000000000..f9b464c878a6a
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/fuse-intrinsic-vminub.ll
@@ -0,0 +1,58 @@
+; RUN: llc -march=hexagon -mcpu=hexagonv73 -O2 -hexagon-fuse-intrinsic-vminub=true < %s | FileCheck %s --check-prefix=FUSE
+; RUN: llc -march=hexagon -mcpu=hexagonv73 -O2 -hexagon-fuse-intrinsic-vminub=false < %s | FileCheck %s --check-prefix=NOFUSE
+
+ at g = external global i64
+ at r = external global i32
+
+; FUSE-LABEL: fused:
+; FUSE-NOT: cmp.gtu
+; FUSE: [[V:r[0-9]+:[0-9]+]],[[P:p[0-3]]] = vminub(
+; FUSE-DAG: r{{[0-9]+}} = [[P]]
+; FUSE-DAG: memd(gp+#g) = [[V]]
+define i32 @fused(i64 %a, i64 %b) {
+entry:
+ %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+ %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %a, i64 %b)
+ store i64 %v, i64* @g, align 8
+ store i32 %p, i32* @r, align 4
+ %pe = zext i32 %p to i64
+ %sum = add i64 %v, %pe
+ %ret = trunc i64 %sum to i32
+ ret i32 %ret
+}
+
+; NOFUSE-LABEL: fused:
+; NOFUSE-DAG: {{p[0-3]}} = cmp.gtu(
+; NOFUSE-DAG: {{r[0-9]+:[0-9]+}} = vminub(
+; NOFUSE-NOT: {{r[0-9]+:[0-9]+}},{{p[0-3]}} = vminub(
+
+; FUSE-LABEL: only_vmin:
+; FUSE-NOT: cmp.gtu
+; FUSE: {{r[0-9]+:[0-9]+}} = vminub(
+define i64 @only_vmin(i64 %a, i64 %b) {
+ %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %a, i64 %b)
+ ret i64 %v
+}
+
+; FUSE-LABEL: only_cmp:
+; FUSE: {{p[0-3]}} = cmp.gtu(
+define i32 @only_cmp(i64 %a, i64 %b) {
+ %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+ ret i32 %p
+}
+
+; FUSE-LABEL: mismatch:
+; FUSE-NOT: {{r[0-9]+:[0-9]+}},{{p[0-3]}} = vminub(
+; FUSE-DAG: {{r[0-9]+:[0-9]+}} = vminub(
+; FUSE-DAG: {{p[0-3]}} = cmp.gtu(
+define i32 @mismatch(i64 %a, i64 %b) {
+ %p = tail call i32 @llvm.hexagon.C2.cmpgtup(i64 %a, i64 %b)
+ %v = tail call i64 @llvm.hexagon.A2.vminub(i64 %b, i64 %a)
+ %pe = zext i32 %p to i64
+ %sum = add i64 %v, %pe
+ %ret = trunc i64 %sum to i32
+ ret i32 %ret
+}
+
+declare i64 @llvm.hexagon.A2.vminub(i64, i64)
+declare i32 @llvm.hexagon.C2.cmpgtup(i64, i64)
``````````
</details>
https://github.com/llvm/llvm-project/pull/225153
More information about the llvm-commits
mailing list