[llvm] [AMDGPU] Fix wrong truth table in BitOp3_Op for shared sub-expressions. (PR #198556)
via llvm-commits
llvm-commits at lists.llvm.org
Fri Jun 5 10:56:30 PDT 2026
================
@@ -4309,16 +4246,107 @@ static std::pair<unsigned, uint8_t> BitOp3_Op(Register R,
}
// Recursion is naturally limited by the size of the operand vector.
- auto Op = BitOp3_Op(LHS, Src, MRI);
- if (Op.first) {
- NumOpcodes += Op.first;
- LHSBits = Op.second;
+ //
+ // When LHS and RHS share a common sub-expression, one side's recursion
+ // may decompose that sub-expression and replace the Src slot the other
+ // side occupies with sub-operands via the "replace parent" path in
+ // getOperandBits. The other side's cached bit-pattern then refers to a
+ // slot whose contents changed, producing a wrong truth table.
+ //
+ // We detect this in three ways:
+ // (A) If LHS recursed, its truth table is valid against the Src state
+ // when LHS recursion completed (SrcAfterLHS). If RHS recursion
+ // then mutates a Src slot that LHSBits depends on, LHSBits is
+ // stale.
+ // (B) If RHS did not recurse, RHSBits came from getOperandBits and
+ // refers to a specific Src slot. If that slot's contents changed
+ // (by either recursion), RHSBits is stale.
+ // (C) Symmetrically for LHS if it did not recurse.
+ SmallVector<Register, 3> SrcBeforeRecurse(Src.begin(), Src.end());
+ uint8_t LHSBitsOrig = LHSBits;
+ uint8_t RHSBitsOrig = RHSBits;
+
+ auto LHSOp = BitOp3_Op(LHS, Src, MRI);
+ if (LHSOp.first) {
+ NumOpcodes += LHSOp.first;
+ LHSBits = LHSOp.second;
+ }
+
+ SmallVector<Register, 3> SrcAfterLHS(Src.begin(), Src.end());
+
+ auto RHSOp = BitOp3_Op(RHS, Src, MRI);
+ if (RHSOp.first) {
+ NumOpcodes += RHSOp.first;
+ RHSBits = RHSOp.second;
+ }
+
+ // dependsOnSlot: true iff the truth table TT varies with slot Slot.
+ auto dependsOnSlot = [](uint8_t TT, int Slot) -> bool {
+ if (Slot < 0 || Slot > 2)
+ return false;
+ const uint8_t Masks[3] = {0x0f, 0x33, 0x55};
+ const int Shifts[3] = {4, 2, 1};
+ return ((TT ^ (TT >> Shifts[Slot])) & Masks[Slot]) != 0;
+ };
+
+ // findSlot: locate the Src slot a getOperandBits result depends on,
+ // including negated (NOT) patterns that getOperandBits resolves via
+ // the ~SrcBits[I] shortcut.
+ const uint8_t SrcBitsConst[3] = {0xf0, 0xcc, 0xaa};
+ auto findSlot = [&](uint8_t Bits, Register Op,
+ const SmallVectorImpl<Register> &S) -> int {
+ for (int I = 0; I < (int)S.size(); I++) {
+ if (Bits == SrcBitsConst[I] && S[I] == Op)
+ return I;
+ if (Bits == (uint8_t)~SrcBitsConst[I]) {
+ Register Inner;
+ if (mi_match(Op, MRI, m_Not(m_Reg(Inner)))) {
----------------
carlobertolli wrote:
done
https://github.com/llvm/llvm-project/pull/198556
More information about the llvm-commits
mailing list