[llvm] [AVX-512] make vpternlogq more aggressive for longer chains of bitmanipulations (PR #189971)
Julian Pokrovsky via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 21:36:19 PDT 2026
================
@@ -4829,118 +4848,170 @@ bool X86DAGToDAGISel::tryVPTERNLOG(SDNode *N) {
if (!(Subtarget->hasVLX() || NVT.is512BitVector()))
return false;
- auto getFoldableLogicOp = [](SDValue Op) {
- // Peek through single use bitcast.
- if (Op.getOpcode() == ISD::BITCAST && Op.hasOneUse())
- Op = Op.getOperand(0);
-
- if (!Op.hasOneUse())
- return SDValue();
-
- unsigned Opc = Op.getOpcode();
- if (Opc == ISD::AND || Opc == ISD::OR || Opc == ISD::XOR ||
- Opc == X86ISD::ANDNP)
- return Op;
-
- return SDValue();
+ auto IsLogic = [](unsigned Opc) {
+ return Opc == ISD::AND || Opc == ISD::OR || Opc == ISD::XOR ||
+ Opc == X86ISD::ANDNP;
+ };
+ auto IsNot = [](SDValue V) {
+ return V.getOpcode() == ISD::XOR &&
+ ISD::isBuildVectorAllOnes(V.getOperand(1).getNode());
+ };
+ auto PeelBitcast = [](SDValue V) {
+ if (V.getOpcode() == ISD::BITCAST && V.hasOneUse())
+ return V.getOperand(0);
+ return V;
+ };
+ auto IsLoadLike = [](SDValue V) {
+ return isa<LoadSDNode>(V.getNode()) ||
+ V.getOpcode() == X86ISD::VBROADCAST_LOAD;
};
- SDValue N0, N1, A, FoldableOp;
-
- // Identify and (optionally) peel an outer NOT that wraps a pure logic tree
- auto tryPeelOuterNotWrappingLogic = [&](SDNode *Op) {
- if (Op->getOpcode() == ISD::XOR && Op->hasOneUse() &&
- ISD::isBuildVectorAllOnes(Op->getOperand(1).getNode())) {
- SDValue InnerOp = getFoldableLogicOp(Op->getOperand(0));
-
- if (!InnerOp)
- return SDValue();
+ struct Leaf {
+ SDValue V;
+ SDNode *Parent;
+ };
- N0 = InnerOp.getOperand(0);
- N1 = InnerOp.getOperand(1);
- if ((FoldableOp = getFoldableLogicOp(N1))) {
- A = N0;
- return InnerOp;
- }
- if ((FoldableOp = getFoldableLogicOp(N0))) {
- A = N1;
- return InnerOp;
+ // Symbolically evaluate the tree rooted at Root. Opaque, if non-null, is
+ // treated as a leaf even when it is a logic op (used for cascading). On
+ // success fills Leaves (the distinct inputs in seed order, at most three),
+ // Imm and NumOps (the number of logic/NOT nodes folded - a profitability
+ // signal), and returns true. Returns false when more than three distinct
+ // leaves are required. Single-use is required to fold a node; bitcasts are
+ // peeled.
+ auto Evaluate = [&](SDValue Root, SDNode *Opaque,
+ SmallVectorImpl<Leaf> &Leaves, uint8_t &Imm,
+ unsigned &NumOps) -> bool {
+ static constexpr uint8_t Seeds[] = {0xF0, 0xCC, 0xAA};
+ NumOps = 0;
+ std::function<int(SDValue, SDNode *, bool)> Eval =
----------------
raventid wrote:
I tried this, but `Eval` is recursive, so a plain `auto` lambda can't refer to itself:
```
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp:5101:23: error: variable 'Eval' declared with deduced type 'auto' cannot appear in its own initializer
5101 | int Inner = Eval(Op.getOperand(0), Op.getNode(), false);
| ^
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp:5106:19: error: variable 'Eval' declared with deduced type 'auto' cannot appear in its own initializer
5106 | int L = Eval(Op.getOperand(0), Op.getNode(), false);
| ^
llvm/lib/Target/X86/X86ISelDAGToDAG.cpp:5107:19: error: variable 'Eval' declared with deduced type 'auto' cannot appear in its own initializer
5107 | int R = Eval(Op.getOperand(1), Op.getNode(), false);
| ^
```
In a few places we use this trick to enable auto -> https://github.com/llvm/llvm-project/blob/0e2ebe79f5c4ea36d7daecb76fb4867452d6973e/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp#L15248-L15259 I used it too here, but it looks a bit ugly to my taste. I would keep an explicit `std::function`. The change is here https://github.com/llvm/llvm-project/pull/189971/changes/fa198307d4aefc4e279dd4a7f9240d8972ce0b89
https://github.com/llvm/llvm-project/pull/189971
More information about the llvm-commits
mailing list