[llvm] [AVX-512] make vpternlogq more aggressive for longer chains of bitmanipulations (PR #189971)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 6 23:12:44 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-x86
Author: Julian Pokrovsky (raventid)
<details>
<summary>Changes</summary>
This is an initial implementation of a few add-hoc heuristics to make vpternlog more aggressive on longer bit-operation chains. Current implementation introduce few regression to existing test cases, so considered to be experimental and not ready for being merged into the upstream.
Published for initial demonstration.
Resolves https://github.com/llvm/llvm-project/issues/134768
---
Patch is 452.07 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/189971.diff
45 Files Affected:
- (modified) llvm/lib/Target/X86/X86ISelDAGToDAG.cpp (+172-101)
- (modified) llvm/test/CodeGen/X86/avx512-cvt.ll (+10-11)
- (modified) llvm/test/CodeGen/X86/avx512-logic.ll (+1-1)
- (modified) llvm/test/CodeGen/X86/avx512-mask-op.ll (+2-12)
- (modified) llvm/test/CodeGen/X86/avx512fp16-arith.ll (+14-14)
- (modified) llvm/test/CodeGen/X86/avx512vl-logic.ll (+3-7)
- (modified) llvm/test/CodeGen/X86/bitcast-vector-bool.ll (+1-1)
- (modified) llvm/test/CodeGen/X86/clmul-vector-256.ll (+42-42)
- (modified) llvm/test/CodeGen/X86/clmul-vector-512.ll (+239-239)
- (modified) llvm/test/CodeGen/X86/combine-fround.ll (+12-12)
- (modified) llvm/test/CodeGen/X86/combine-or-shuffle.ll (+4-4)
- (modified) llvm/test/CodeGen/X86/elementwise-store-of-scalar-splat.ll (+4-8)
- (modified) llvm/test/CodeGen/X86/fminimumnum-fmaximumnum.ll (+8-8)
- (modified) llvm/test/CodeGen/X86/fp-round.ll (+37-37)
- (modified) llvm/test/CodeGen/X86/fshl-splat-undef.ll (+3-3)
- (modified) llvm/test/CodeGen/X86/gfni-shifts.ll (+9-9)
- (modified) llvm/test/CodeGen/X86/icmp-pow2-diff.ll (+7-8)
- (modified) llvm/test/CodeGen/X86/min-legal-vector-width.ll (+6-6)
- (modified) llvm/test/CodeGen/X86/pmul.ll (+2-2)
- (modified) llvm/test/CodeGen/X86/prefer-avx256-wide-mul.ll (+1-1)
- (modified) llvm/test/CodeGen/X86/sat-add.ll (+23-25)
- (modified) llvm/test/CodeGen/X86/select-big-integer.ll (+4-8)
- (modified) llvm/test/CodeGen/X86/srem-seteq-vec-nonsplat.ll (+2-3)
- (modified) llvm/test/CodeGen/X86/subvectorwise-store-of-vector-splat.ll (+1160-458)
- (modified) llvm/test/CodeGen/X86/urem-seteq-vec-tautological.ll (+2-3)
- (modified) llvm/test/CodeGen/X86/vector-fshr-256.ll (+1-1)
- (modified) llvm/test/CodeGen/X86/vector-fshr-512.ll (+2-2)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-load-i16-stride-6.ll (+4-4)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-load-i8-stride-6.ll (+32-32)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-load-i8-stride-7.ll (+40-40)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-store-i16-stride-3.ll (+16-16)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-store-i16-stride-5.ll (+20-20)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-store-i16-stride-7.ll (+10-10)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-store-i8-stride-5.ll (+8-8)
- (modified) llvm/test/CodeGen/X86/vector-interleaved-store-i8-stride-7.ll (+548-554)
- (modified) llvm/test/CodeGen/X86/vector-reduce-fmaximum.ll (+21-22)
- (modified) llvm/test/CodeGen/X86/vector-reduce-fminimum.ll (+21-22)
- (modified) llvm/test/CodeGen/X86/vector-rotate-128.ll (+4-4)
- (modified) llvm/test/CodeGen/X86/vector-rotate-256.ll (+4-4)
- (modified) llvm/test/CodeGen/X86/vector-rotate-512.ll (+12-12)
- (modified) llvm/test/CodeGen/X86/vector-shift-ashr-256.ll (+2-2)
- (modified) llvm/test/CodeGen/X86/vector-shift-ashr-512.ll (+2-2)
- (modified) llvm/test/CodeGen/X86/vpternlog.ll (+168-2)
- (modified) llvm/test/CodeGen/X86/zero_extend_vector_inreg_of_broadcast.ll (+28-38)
- (modified) llvm/test/CodeGen/X86/zero_extend_vector_inreg_of_broadcast_from_memory.ll (+28-38)
``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
index b794567f6a047..457b5f8093766 100644
--- a/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
+++ b/llvm/lib/Target/X86/X86ISelDAGToDAG.cpp
@@ -31,6 +31,7 @@
#include "llvm/Support/KnownBits.h"
#include "llvm/Support/MathExtras.h"
#include <cstdint>
+#include <functional>
using namespace llvm;
@@ -4696,8 +4697,17 @@ bool X86DAGToDAGISel::matchVPTERNLOG(SDNode *Root, SDNode *ParentA,
SDNode *ParentB, SDNode *ParentC,
SDValue A, SDValue B, SDValue C,
uint8_t Imm) {
- assert(A.isOperandOf(ParentA) && B.isOperandOf(ParentB) &&
- C.isOperandOf(ParentC) && "Incorrect parent node");
+ // Unused operand slots may be padded with an IMPLICIT_DEF (e.g. when a tree
+ // folds to a VPTERNLOG with fewer than three distinct inputs); padding has no
+ // parent node.
+ auto IsPad = [](SDValue V) {
+ return V.isMachineOpcode() &&
+ V.getMachineOpcode() == TargetOpcode::IMPLICIT_DEF;
+ };
+ assert((IsPad(A) || A.isOperandOf(ParentA)) &&
+ (IsPad(B) || B.isOperandOf(ParentB)) &&
+ (IsPad(C) || C.isOperandOf(ParentC)) && "Incorrect parent node");
+ (void)IsPad;
auto tryFoldLoadOrBCast =
[this](SDNode *Root, SDNode *P, SDValue &L, SDValue &Base, SDValue &Scale,
@@ -4815,8 +4825,17 @@ bool X86DAGToDAGISel::matchVPTERNLOG(SDNode *Root, SDNode *ParentA,
return true;
}
-// Try to match two logic ops to a VPTERNLOG.
-// FIXME: Handle more complex patterns that use an operand more than once?
+// Try to match a tree of bitwise logic ops to one or more VPTERNLOG ops.
+//
+// VPTERNLOG implements an arbitrary 3-input bitwise function via an 8-bit
+// truth-table immediate. We evaluate the logic sub-DAG rooted at N
+// symbolically: each of the (up to three) distinct leaf inputs is seeded with a
+// canonical truth-table column (0xF0/0xCC/0xAA) and AND/OR/XOR/ANDNP/NOT are
+// folded bitwise over those seeds, yielding the control immediate. Operand
+// reuse, inverted inputs and arbitrary nesting depth all fall out of the
+// recursion. Trees with more than three distinct leaves are split by leaving
+// one subtree "opaque" (it selects into its own VPTERNLOG when isel reaches it)
+// and folding the rest around it.
bool X86DAGToDAGISel::tryVPTERNLOG(SDNode *N) {
MVT NVT = N->getSimpleValueType(0);
@@ -4829,118 +4848,170 @@ bool X86DAGToDAGISel::tryVPTERNLOG(SDNode *N) {
if (!(Subtarget->hasVLX() || NVT.is512BitVector()))
return false;
- auto getFoldableLogicOp = [](SDValue Op) {
- // Peek through single use bitcast.
- if (Op.getOpcode() == ISD::BITCAST && Op.hasOneUse())
- Op = Op.getOperand(0);
-
- if (!Op.hasOneUse())
- return SDValue();
-
- unsigned Opc = Op.getOpcode();
- if (Opc == ISD::AND || Opc == ISD::OR || Opc == ISD::XOR ||
- Opc == X86ISD::ANDNP)
- return Op;
-
- return SDValue();
+ auto IsLogic = [](unsigned Opc) {
+ return Opc == ISD::AND || Opc == ISD::OR || Opc == ISD::XOR ||
+ Opc == X86ISD::ANDNP;
+ };
+ auto IsNot = [](SDValue V) {
+ return V.getOpcode() == ISD::XOR &&
+ ISD::isBuildVectorAllOnes(V.getOperand(1).getNode());
+ };
+ auto PeelBitcast = [](SDValue V) {
+ if (V.getOpcode() == ISD::BITCAST && V.hasOneUse())
+ return V.getOperand(0);
+ return V;
+ };
+ auto IsLoadLike = [](SDValue V) {
+ return isa<LoadSDNode>(V.getNode()) ||
+ V.getOpcode() == X86ISD::VBROADCAST_LOAD;
};
- SDValue N0, N1, A, FoldableOp;
-
- // Identify and (optionally) peel an outer NOT that wraps a pure logic tree
- auto tryPeelOuterNotWrappingLogic = [&](SDNode *Op) {
- if (Op->getOpcode() == ISD::XOR && Op->hasOneUse() &&
- ISD::isBuildVectorAllOnes(Op->getOperand(1).getNode())) {
- SDValue InnerOp = getFoldableLogicOp(Op->getOperand(0));
-
- if (!InnerOp)
- return SDValue();
+ struct Leaf {
+ SDValue V;
+ SDNode *Parent;
+ };
- N0 = InnerOp.getOperand(0);
- N1 = InnerOp.getOperand(1);
- if ((FoldableOp = getFoldableLogicOp(N1))) {
- A = N0;
- return InnerOp;
- }
- if ((FoldableOp = getFoldableLogicOp(N0))) {
- A = N1;
- return InnerOp;
+ // Symbolically evaluate the tree rooted at Root. Opaque, if non-null, is
+ // treated as a leaf even when it is a logic op (used for cascading). On
+ // success fills Leaves (the distinct inputs in seed order, at most three),
+ // Imm and NumOps (the number of logic/NOT nodes folded - a profitability
+ // signal), and returns true. Returns false when more than three distinct
+ // leaves are required. Single-use is required to fold a node; bitcasts are
+ // peeled.
+ auto Evaluate = [&](SDValue Root, SDNode *Opaque,
+ SmallVectorImpl<Leaf> &Leaves, uint8_t &Imm,
+ unsigned &NumOps) -> bool {
+ static constexpr uint8_t Seeds[] = {0xF0, 0xCC, 0xAA};
+ NumOps = 0;
+ std::function<int(SDValue, SDNode *, bool)> Eval =
+ [&](SDValue Op, SDNode *Parent, bool IsRoot) -> int {
+ if (Op.getNode() != Opaque) {
+ if (Op.getOpcode() == ISD::BITCAST && (IsRoot || Op.hasOneUse())) {
+ Parent = Op.getNode();
+ Op = Op.getOperand(0);
+ }
+ if ((IsRoot || Op.hasOneUse()) && IsNot(Op)) {
+ ++NumOps;
+ int Inner = Eval(Op.getOperand(0), Op.getNode(), false);
+ return Inner < 0 ? -1 : (~Inner & 0xFF);
+ }
+ if ((IsRoot || Op.hasOneUse()) && IsLogic(Op.getOpcode())) {
+ ++NumOps;
+ int L = Eval(Op.getOperand(0), Op.getNode(), false);
+ int R = Eval(Op.getOperand(1), Op.getNode(), false);
+ if (L < 0 || R < 0)
+ return -1;
+ switch (Op.getOpcode()) {
+ case ISD::AND:
+ return L & R;
+ case ISD::OR:
+ return L | R;
+ case ISD::XOR:
+ return (L ^ R) & 0xFF;
+ case X86ISD::ANDNP:
+ return ~L & R;
+ default:
+ llvm_unreachable("Checked by IsLogic");
+ }
+ }
}
- }
- return SDValue();
+ // Leaf: reuse the seed of an identical input, else allocate a new one.
+ for (unsigned I = 0, E = Leaves.size(); I != E; ++I)
+ if (Leaves[I].V == Op)
+ return Seeds[I];
+ if (Leaves.size() >= 3)
+ return -1;
+ uint8_t Seed = Seeds[Leaves.size()];
+ Leaves.push_back({Op, Parent});
+ return Seed;
+ };
+ int Result = Eval(Root, Root.getNode(), /*IsRoot=*/true);
+ if (Result < 0)
+ return false;
+ Imm = Result & 0xFF;
+ return true;
};
- bool PeeledOuterNot = false;
- SDNode *OriN = N;
- if (SDValue InnerOp = tryPeelOuterNotWrappingLogic(N)) {
- PeeledOuterNot = true;
- N = InnerOp.getNode();
- } else {
- N0 = N->getOperand(0);
- N1 = N->getOperand(1);
+ // Emit one VPTERNLOG from up to three leaves. Unused operand slots are
+ // padded with undef so that, e.g., a NOT of a single memory operand folds the
+ // load with no false dependency on the other inputs.
+ auto Emit = [&](ArrayRef<Leaf> Leaves, uint8_t Imm) {
+ SDValue Undef(
+ CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, SDLoc(N), NVT), 0);
+ SDValue Ops[3] = {Undef, Undef, Undef};
+ SDNode *Parents[3] = {N, N, N};
+ for (unsigned I = 0, E = Leaves.size(); I != E; ++I) {
+ Ops[I] = Leaves[I].V;
+ Parents[I] = Leaves[I].Parent;
+ }
+ return matchVPTERNLOG(N, Parents[0], Parents[1], Parents[2], Ops[0], Ops[1],
+ Ops[2], Imm);
+ };
- if ((FoldableOp = getFoldableLogicOp(N1)))
- A = N0;
- else if ((FoldableOp = getFoldableLogicOp(N0)))
- A = N1;
- else
+ // A cover is profitable when it folds at least two logic ops into one
+ // VPTERNLOG, or when it is a NOT of a single memory operand (which saves
+ // materializing an all-ones vector). A lone binary op (and/or/xor/andn) is
+ // left to its dedicated, cheaper instruction.
+ auto Profitable = [&](ArrayRef<Leaf> Leaves, unsigned NumOps) {
+ if (NumOps >= 2)
+ return true;
+ return Leaves.size() == 1 && IsLoadLike(PeelBitcast(Leaves[0].V));
+ };
+
+ // Whole tree fits in a single VPTERNLOG (<= 3 distinct leaves).
+ SmallVector<Leaf, 3> Leaves;
+ uint8_t Imm = 0;
+ unsigned NumOps = 0;
+ if (Evaluate(SDValue(N, 0), /*Opaque=*/nullptr, Leaves, Imm, NumOps)) {
+ if (!Profitable(Leaves, NumOps))
return false;
+ return Emit(Leaves, Imm);
}
- SDValue B = FoldableOp.getOperand(0);
- SDValue C = FoldableOp.getOperand(1);
- SDNode *ParentA = N;
- SDNode *ParentB = FoldableOp.getNode();
- SDNode *ParentC = FoldableOp.getNode();
-
- // We can build the appropriate control immediate by performing the logic
- // operation we're matching using these constants for A, B, and C.
- uint8_t TernlogMagicA = 0xf0;
- uint8_t TernlogMagicB = 0xcc;
- uint8_t TernlogMagicC = 0xaa;
-
- // Some of the inputs may be inverted, peek through them and invert the
- // magic values accordingly.
- // TODO: There may be a bitcast before the xor that we should peek through.
- auto PeekThroughNot = [](SDValue &Op, SDNode *&Parent, uint8_t &Magic) {
- if (Op.getOpcode() == ISD::XOR && Op.hasOneUse() &&
- ISD::isBuildVectorAllOnes(Op.getOperand(1).getNode())) {
- Magic = ~Magic;
- Parent = Op.getNode();
- Op = Op.getOperand(0);
+ // More than three leaves: cascade. Collect the single-use logic subtrees
+ // that could be made opaque (NOTs are transparent - cutting at one would just
+ // spill it to a separate instruction).
+ SmallVector<SDNode *, 8> Candidates;
+ std::function<void(SDValue)> Collect = [&](SDValue V) {
+ V = PeelBitcast(V);
+ if (!V.hasOneUse())
+ return;
+ if (IsNot(V)) {
+ Collect(V.getOperand(0));
+ return;
+ }
+ if (IsLogic(V.getOpcode())) {
+ Candidates.push_back(V.getNode());
+ Collect(V.getOperand(0));
+ Collect(V.getOperand(1));
}
};
-
- PeekThroughNot(A, ParentA, TernlogMagicA);
- PeekThroughNot(B, ParentB, TernlogMagicB);
- PeekThroughNot(C, ParentC, TernlogMagicC);
-
- uint8_t Imm;
- switch (FoldableOp.getOpcode()) {
- default: llvm_unreachable("Unexpected opcode!");
- case ISD::AND: Imm = TernlogMagicB & TernlogMagicC; break;
- case ISD::OR: Imm = TernlogMagicB | TernlogMagicC; break;
- case ISD::XOR: Imm = TernlogMagicB ^ TernlogMagicC; break;
- case X86ISD::ANDNP: Imm = ~(TernlogMagicB) & TernlogMagicC; break;
- }
-
- switch (N->getOpcode()) {
- default: llvm_unreachable("Unexpected opcode!");
- case X86ISD::ANDNP:
- if (A == N0)
- Imm &= ~TernlogMagicA;
- else
- Imm = ~(Imm) & TernlogMagicA;
- break;
- case ISD::AND: Imm &= TernlogMagicA; break;
- case ISD::OR: Imm |= TernlogMagicA; break;
- case ISD::XOR: Imm ^= TernlogMagicA; break;
+ Collect(N->getOperand(0));
+ Collect(N->getOperand(1));
+
+ // Pick the cut that lets the root fold the most leaves (a tighter cascade),
+ // breaking ties towards folding more ops.
+ SmallVector<Leaf, 3> BestLeaves;
+ uint8_t BestImm = 0;
+ unsigned BestNumOps = 0;
+ for (SDNode *S : Candidates) {
+ SmallVector<Leaf, 3> CutLeaves;
+ uint8_t CutImm = 0;
+ unsigned CutNumOps = 0;
+ if (!Evaluate(SDValue(N, 0), S, CutLeaves, CutImm, CutNumOps))
+ continue;
+ if (CutLeaves.size() > BestLeaves.size() ||
+ (CutLeaves.size() == BestLeaves.size() && CutNumOps > BestNumOps)) {
+ BestLeaves = std::move(CutLeaves);
+ BestImm = CutImm;
+ BestNumOps = CutNumOps;
+ }
}
- if (PeeledOuterNot)
- Imm = ~Imm;
+ if (!BestLeaves.empty() && Profitable(BestLeaves, BestNumOps))
+ return Emit(BestLeaves, BestImm);
- return matchVPTERNLOG(OriN, ParentA, ParentB, ParentC, A, B, C, Imm);
+ return false;
}
/// If the high bits of an 'and' operand are known zero, try setting the
diff --git a/llvm/test/CodeGen/X86/avx512-cvt.ll b/llvm/test/CodeGen/X86/avx512-cvt.ll
index 76c87900b04d2..a8a906b69e9cb 100644
--- a/llvm/test/CodeGen/X86/avx512-cvt.ll
+++ b/llvm/test/CodeGen/X86/avx512-cvt.ll
@@ -350,8 +350,8 @@ define <4 x float> @ulto4f32_nneg(<4 x i64> %a) {
define <8 x double> @ulto8f64(<8 x i64> %a) {
; NODQ-LABEL: ulto8f64:
; NODQ: # %bb.0:
-; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm1 = [4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200]
-; NODQ-NEXT: vpternlogq {{.*#+}} zmm1 = zmm1 | (zmm0 & m64bcst)
+; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm1 = [4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295]
+; NODQ-NEXT: vpternlogq {{.*#+}} zmm1 = (zmm1 & zmm0) | m64bcst
; NODQ-NEXT: vpsrlq $32, %zmm0, %zmm0
; NODQ-NEXT: vporq {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to8}, %zmm0, %zmm0
; NODQ-NEXT: vsubpd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to8}, %zmm0, %zmm0
@@ -374,17 +374,16 @@ define <8 x double> @ulto8f64(<8 x i64> %a) {
define <16 x double> @ulto16f64(<16 x i64> %a) {
; NODQ-LABEL: ulto16f64:
; NODQ: # %bb.0:
-; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm2 = [4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295]
-; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm3 = [4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200]
-; NODQ-NEXT: vmovdqa64 %zmm3, %zmm4
-; NODQ-NEXT: vpternlogq {{.*#+}} zmm4 = zmm4 | (zmm0 & zmm2)
-; NODQ-NEXT: vpsrlq $32, %zmm0, %zmm0
+; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm2 = [4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200,4841369599423283200]
+; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm3 = [4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295,4294967295]
+; NODQ-NEXT: vpsrlq $32, %zmm0, %zmm4
+; NODQ-NEXT: vpternlogq {{.*#+}} zmm0 = (zmm0 & zmm3) | zmm2
; NODQ-NEXT: vpbroadcastq {{.*#+}} zmm5 = [4985484787499139072,4985484787499139072,4985484787499139072,4985484787499139072,4985484787499139072,4985484787499139072,4985484787499139072,4985484787499139072]
-; NODQ-NEXT: vporq %zmm5, %zmm0, %zmm0
+; NODQ-NEXT: vporq %zmm5, %zmm4, %zmm4
; NODQ-NEXT: vbroadcastsd {{.*#+}} zmm6 = [1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25,1.9342813118337666E+25]
-; NODQ-NEXT: vsubpd %zmm6, %zmm0, %zmm0
-; NODQ-NEXT: vaddpd %zmm0, %zmm4, %zmm0
-; NODQ-NEXT: vpternlogq {{.*#+}} zmm3 = zmm3 | (zmm1 & zmm2)
+; NODQ-NEXT: vsubpd %zmm6, %zmm4, %zmm4
+; NODQ-NEXT: vaddpd %zmm4, %zmm0, %zmm0
+; NODQ-NEXT: vpternlogq {{.*#+}} zmm3 = (zmm3 & zmm1) | zmm2
; NODQ-NEXT: vpsrlq $32, %zmm1, %zmm1
; NODQ-NEXT: vporq %zmm5, %zmm1, %zmm1
; NODQ-NEXT: vsubpd %zmm6, %zmm1, %zmm1
diff --git a/llvm/test/CodeGen/X86/avx512-logic.ll b/llvm/test/CodeGen/X86/avx512-logic.ll
index bdcc524545fb1..03cf284b2f26c 100644
--- a/llvm/test/CodeGen/X86/avx512-logic.ll
+++ b/llvm/test/CodeGen/X86/avx512-logic.ll
@@ -856,7 +856,7 @@ entry:
define <16 x i32> @ternlog_and_andn(<16 x i32> %x, <16 x i32> %y, <16 x i32> %z) {
; ALL-LABEL: ternlog_and_andn:
; ALL: ## %bb.0:
-; ALL-NEXT: vpternlogd {{.*#+}} zmm0 = zmm2 & zmm1 & ~zmm0
+; ALL-NEXT: vpternlogd {{.*#+}} zmm0 = zmm1 & zmm2 & ~zmm0
; ALL-NEXT: retq
%a = xor <16 x i32> %x, <i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1, i32 -1>
%b = and <16 x i32> %y, %a
diff --git a/llvm/test/CodeGen/X86/avx512-mask-op.ll b/llvm/test/CodeGen/X86/avx512-mask-op.ll
index db53dda401bc2..e21ae6e1a3347 100644
--- a/llvm/test/CodeGen/X86/avx512-mask-op.ll
+++ b/llvm/test/CodeGen/X86/avx512-mask-op.ll
@@ -3295,12 +3295,7 @@ define <16 x i1> @test_v16i1_select_chain(<16 x i1> %a0, <16 x i1> %a1, <16 x i1
;
; SKX-LABEL: test_v16i1_select_chain:
; SKX: ## %bb.0:
-; SKX-NEXT: vmovdqa %xmm0, %xmm3
-; SKX-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & ~(xmm2 & xmm1)
-; SKX-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & xmm0 & xmm1
-; SKX-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & xmm0 & xmm1
-; SKX-NEXT: vpternlogq {{.*#+}} xmm3 = xmm1 & (xmm3 ^ xmm1)
-; SKX-NEXT: vpternlogq {{.*#+}} xmm0 = xmm0 & xmm2 & xmm3
+; SKX-NEXT: vpternlogq {{.*#+}} xmm0 = xmm0 & xmm2 & xmm1
; SKX-NEXT: retq
;
; AVX512BW-LABEL: test_v16i1_select_chain:
@@ -3331,12 +3326,7 @@ define <16 x i1> @test_v16i1_select_chain(<16 x i1> %a0, <16 x i1> %a1, <16 x i1
;
; X86-LABEL: test_v16i1_select_chain:
; X86: ## %bb.0:
-; X86-NEXT: vmovdqa %xmm0, %xmm3
-; X86-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & ~(xmm2 & xmm1)
-; X86-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & xmm0 & xmm1
-; X86-NEXT: vpternlogq {{.*#+}} xmm3 = xmm3 & xmm0 & xmm1
-; X86-NEXT: vpternlogq {{.*#+}} xmm3 = xmm1 & (xmm3 ^ xmm1)
-; X86-NEXT: vpternlogq {{.*#+}} xmm0 = xmm0 & xmm2 & xmm3
+; X86-NEXT: vpternlogq {{.*#+}} xmm0 = xmm0 & xmm2 & xmm1
; X86-NEXT: retl
%i3 = select <16 x i1> %a2, <16 x i1> %a1, <16 x i1> zeroinitializer
%i4 = xor <16 x i1> %i3, splat (i1 true)
diff --git a/llvm/test/CodeGen/X86/avx512fp16-arith.ll b/llvm/test/CodeGen/X86/avx512fp16-arith.ll
index 9f74316f3c918..5114bcd98b236 100644
--- a/llvm/test/CodeGen/X86/avx512fp16-arith.ll
+++ b/llvm/test/CodeGen/X86/avx512fp16-arith.ll
@@ -339,9 +339,9 @@ declare half @llvm.copysign.f16(half, half)
define half @fround(half %x) {
; CHECK-LABEL: fround:
; CHECK: ## %bb.0:
-; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm1 = [-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0]
-; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm2 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
-; CHECK-NEXT: vpternlogq {{.*#+}} xmm2 = xmm2 | (xmm0 & xmm1)
+; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm1 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
+; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm2 = [-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0]
+; CHECK-NEXT: vpternlogq {{.*#+}} xmm2 = (xmm2 & xmm0) | xmm1
; CHECK-NEXT: vaddsh %xmm2, %xmm0, %xmm0
; CHECK-NEXT: vrndscalesh $11, %xmm0, %xmm0, %xmm0
; CHECK-NEXT: retq
@@ -394,9 +394,9 @@ declare <8 x half> @llvm.copysign.v8f16(<8 x half>, <8 x half>)
define <8 x half> @roundv8f16(<8 x half> %x) {
; CHECK-LABEL: roundv8f16:
; CHECK: ## %bb.0:
-; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm1 = [-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0]
-; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm2 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
-; CHECK-NEXT: vpternlogq {{.*#+}} xmm2 = xmm2 | (xmm0 & xmm1)
+; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm1 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
+; CHECK-NEXT: vpbroadcastw {{.*#+}} xmm2 = [-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0]
+; CHECK-NEXT: vpternlogq {{.*#+}} xmm2 = (xmm2 & xmm0) | xmm1
; CHECK-NEXT: vaddph %xmm2, %xmm0, %xmm0
; CHECK-NEXT: vrndscaleph $11, %xmm0, %xmm0
; CHECK-NEXT: retq
@@ -449,9 +449,9 @@ declare <16 x half> @llvm.copysign.v16f16(<16 x half>, <16 x half>)
define <16 x half> @roundv16f16(<16 x half> %x) {
; CHECK-LABEL: roundv16f16:
; CHECK: ## %bb.0:
-; CHECK-NEXT: vpbroadcastw {{.*#+}} ymm1 = [-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0,-0.0E+0]
-; CHECK-NEXT: vpbroadcastw {{.*#+}} ymm2 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
-; CHECK-NEXT: vpternlogq {{.*#+}} ymm2 = ymm2 | (ymm0 & ymm1)
+; CHECK-NEXT: vpbroadcastw {{.*#+}} ymm1 = [4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1,4.9976E-1]
+; CHECK-NEXT: vpbr...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/189971
More information about the llvm-commits
mailing list