[llvm] [LoopIdiomRecognize] Enable clmul optimization for CRC loops (PR #203405)

Sean Clarke via llvm-commits llvm-commits at lists.llvm.org
Fri Jun 26 08:54:53 PDT 2026


================
@@ -1549,7 +1551,128 @@ bool LoopIdiomRecognize::avoidLIRForMultiBlockLoop(bool IsMemset,
   return false;
 }
 
-bool LoopIdiomRecognize::optimizeCRCLoop(const PolynomialInfo &Info) {
+bool LoopIdiomRecognize::optimizeCRCLoopToClmul(const PolynomialInfo &Info) {
+  Type *CRCTy = Info.LHS->getType();
+  LLVMContext &Ctx = CRCTy->getContext();
+  unsigned CRCBW = CRCTy->getIntegerBitWidth();
+  // The TripCount determines how many bits of data are processed, regardless of
+  // whether the actual data bit width matches (if auxiliary data is even used
+  // at all).
+  unsigned EffectiveDataBW = Info.TripCount;
+  // The width used for clmul operations should be a power of 2, and should be
+  // at least CRCBW + DataBW.
+  unsigned ClmulBW = 2 * std::max(CRCBW, EffectiveDataBW);
+  Type *ClmulTy = IntegerType::get(Ctx, ClmulBW);
+
+  // For big-endian CRC loops where the auxiliary data is XORed with the CRC
+  // inside the loop, the bits won't be aligned properly if the bit widths don't
+  // match, and thus the CRC computation is incorrect, but HashRecognize will
+  // still detect the loop. Since this optimization always produces a correct
+  // CRC computation, bail in this edge case.
+  if (Info.ByteOrderSwapped && Info.LHSAux &&
+      (EffectiveDataBW != CRCBW ||
+       Info.LHSAux->getType()->getIntegerBitWidth() != CRCBW))
+    return false;
+
+  // This optimization should not be applied if there is no fast clmul operation
+  // for the required width on the target.
+  // TODO: If EffectiveDataBW > CRCBW, then the data could probably be split
+  // into multiple chunks and processed in a loop.
+  if (!TTI->haveFastClmul(ClmulTy))
+    return false;
+
+  // First, generate the constants required for GF(2) Barrett reduction.
+  CRCBarrettConstants Constants = HashRecognize::genBarrettConstants(
+      Info.RHS, EffectiveDataBW, Info.ByteOrderSwapped);
+  Value *Mu = ConstantInt::get(Ctx, Constants.Mu.zext(ClmulBW));
+  Value *FullGenPoly =
+      ConstantInt::get(Ctx, Constants.FullGenPoly.zext(ClmulBW));
+
+  IRBuilder<> Builder(CurLoop->getLoopPreheader()->getTerminator());
+
+  Value *CRCExt = Builder.CreateZExt(Info.LHS, ClmulTy, "crc.ext");
+
+  // For the big-endian case, align the leftmost bit of the CRC with the
+  // leftmost bit of the data. For the little-endian case, align the rightmost
+  // bits (nothing to do).
+  Value *CRCAlignData = CRCExt;
+  if (Info.ByteOrderSwapped) {
+    if (CRCBW > EffectiveDataBW)
+      CRCAlignData = Builder.CreateLShr(CRCAlignData, CRCBW - EffectiveDataBW,
+                                        "crc.be.lshr");
+    else if (EffectiveDataBW > CRCBW)
+      CRCAlignData = Builder.CreateShl(CRCAlignData, EffectiveDataBW - CRCBW,
+                                       "crc.be.shl");
+  }
+
+  // If auxiliary data is present, XOR it in with the CRC.
+  Value *ClmulMuInput = CRCAlignData;
+  if (Value *Data = Info.LHSAux) {
+    unsigned ActualDataBW = Data->getType()->getIntegerBitWidth();
+    if (ActualDataBW > EffectiveDataBW)
+      // Extract the useful bits of the data and discard the rest.
+      Data =
+          Info.ByteOrderSwapped
+              ? Builder.CreateLShr(Data, ActualDataBW - EffectiveDataBW,
+                                   "data.be.lshr")
+              : Builder.CreateAnd(
+                    Data,
+                    ConstantInt::get(Ctx, APInt::getLowBitsSet(
+                                              ActualDataBW, EffectiveDataBW)),
+                    "data.le.mask");
+    // This isn't necessarily a zext-- ActualDataBW could be greater than
+    // ClmulBW.
+    Value *DataExt = Builder.CreateZExtOrTrunc(Data, ClmulTy, "data.ext");
+    // For the big-endian case, ensure the data is aligned properly.
+    if (Info.ByteOrderSwapped && EffectiveDataBW > ActualDataBW)
+      DataExt = Builder.CreateShl(DataExt, EffectiveDataBW - ActualDataBW,
+                                  "data.be.shl");
+
+    ClmulMuInput = Builder.CreateXor(CRCAlignData, DataExt, "xor.crc.data");
+  }
+
+  // Perform the first clmul operation with the mu/mu' constant. Input is DataBW
+  // bits and Mu is DataBW+1 bits, so the result will be 2*DataBW bits.
+  Value *ClmulMu = Builder.CreateBinaryIntrinsic(Intrinsic::clmul, ClmulMuInput,
+                                                 Mu, {}, "clmul.mu");
+
+  // Extract the relevant DataBW bits from the result.
+  Value *ClmulGPInput =
+      Info.ByteOrderSwapped
+          ? Builder.CreateLShr(ClmulMu, EffectiveDataBW, "quot.be.lshr")
+          : Builder.CreateAnd(
+                ClmulMu,
+                ConstantInt::get(
+                    Ctx, APInt::getLowBitsSet(ClmulBW, EffectiveDataBW)),
+                "quot.le.mask");
+
+  // Perform the second clmul operation with the P(x)/P(x)' constant. Input is
+  // DataBW bits and GP is CRCBW+1 bits, so the result will be CRCBW+DataBW
+  // bits.
+  Value *ClmulGP = Builder.CreateBinaryIntrinsic(Intrinsic::clmul, ClmulGPInput,
+                                                 FullGenPoly, {}, "clmul.gp");
+
+  // For the big-endian case, align the leftmost bit of the CRC with the
+  // leftmost bit of the clmul result. For the little-endian case, align the
+  // rightmost bits (nothing to do).
+  Value *CRCAlignClmul = CRCExt;
+  if (Info.ByteOrderSwapped)
+    CRCAlignClmul = Builder.CreateShl(CRCExt, EffectiveDataBW, "crc.be.shl");
+  // Get the remainder by subtracting (XORing) the calculated multiple of
+  // GenPoly from the CRC.
+  Value *CRCNext = Builder.CreateXor(CRCAlignClmul, ClmulGP, "xor.crc.mult");
+  // For the little-endian case, the leftmost bits of the XOR are relevant.
+  if (!Info.ByteOrderSwapped)
+    CRCNext = Builder.CreateLShr(CRCNext, EffectiveDataBW, "crc.le.lshr");
+  CRCNext = Builder.CreateTrunc(CRCNext, CRCTy, "crc.next");
+
+  // Replace the result of the loop with the new computed CRC value.
+  Info.ComputedValue->replaceUsesOutsideBlock(CRCNext, CurLoop->getLoopLatch());
+
----------------
xarkenz wrote:

With a little more effort, it now reduces the loop to a single "conditional" branch that always leads to the exit block. The pass output is significantly cleaner now.

https://github.com/llvm/llvm-project/pull/203405


More information about the llvm-commits mailing list