[llvm] [X86] Implement CRC32 const folding (PR #219452)

Shreesh Adiga via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 07:35:15 PDT 2026


================
@@ -2260,6 +2260,29 @@ X86TTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
     }
     break;
 
+  case Intrinsic::x86_sse42_crc32_32_8:
+  case Intrinsic::x86_sse42_crc32_32_16:
+  case Intrinsic::x86_sse42_crc32_32_32:
+  case Intrinsic::x86_sse42_crc32_64_64: {
+    auto *CrcArg = dyn_cast<ConstantInt>(II.getArgOperand(0));
+    auto *DataArg = dyn_cast<ConstantInt>(II.getArgOperand(1));
+    if (!CrcArg || !DataArg)
+      break;
+
+    // If both operands are constant, we can completely constant fold this.
+    uint64_t Crc = CrcArg->getZExtValue() & 0xffffffff;
+    uint64_t Data = DataArg->getZExtValue();
+    unsigned BitWidth = DataArg->getBitWidth();
+    Crc ^= Data;
+    // CRC32C polynomial (iSCSI polynomial, bit-reversed)
+    const uint32_t Poly = 0x82F63B78;
+    for (unsigned Bit = 0; Bit != BitWidth; ++Bit) {
+      Crc = (Crc >> 1) ^ ((Crc & 1) ? Poly : 0);
+    }
----------------
tantei3 wrote:

With the below diff the tests are passing and core calculation is moved to CRC.h file:
```
diff --git a/clang/lib/AST/ByteCode/InterpBuiltin.cpp b/clang/lib/AST/ByteCode/InterpBuiltin.cpp
index 1a464247b5ce..255cab8aecda 100644
--- a/clang/lib/AST/ByteCode/InterpBuiltin.cpp
+++ b/clang/lib/AST/ByteCode/InterpBuiltin.cpp
@@ -21,6 +21,7 @@
 #include "clang/Basic/TargetInfo.h"
 #include "llvm/ADT/StringExtras.h"
 #include "llvm/Support/AllocToken.h"
+#include "llvm/Support/CRC.h"
 #include "llvm/Support/ErrorHandling.h"
 #include "llvm/Support/SipHash.h"

@@ -814,16 +815,9 @@ static bool interp__builtin_ia32_crc32(InterpState &S, CodePtr OpPC,
   // CRC32C polynomial (iSCSI polynomial, bit-reversed)
   static const uint32_t CRC32C_POLY = 0x82F63B78;

-  // Process each byte
+  CRCVal = llvm::calculateReflectedCrc(CRCVal, 32, DataVal, DataBytes * 8,
+                                       CRC32C_POLY);
   uint32_t Result = static_cast<uint32_t>(CRCVal);
-  for (unsigned I = 0; I != DataBytes; ++I) {
-    uint8_t Byte = static_cast<uint8_t>((DataVal >> (I * 8)) & 0xFF);
-    Result ^= Byte;
-    for (int J = 0; J != 8; ++J) {
-      Result = (Result >> 1) ^ ((Result & 1) ? CRC32C_POLY : 0);
-    }
-  }
-
   pushInteger(S, Result, Call->getType());
   return true;
 }
diff --git a/clang/lib/AST/ExprConstant.cpp b/clang/lib/AST/ExprConstant.cpp
index 9702105951b7..4bb910d290b2 100644
--- a/clang/lib/AST/ExprConstant.cpp
+++ b/clang/lib/AST/ExprConstant.cpp
@@ -61,6 +61,7 @@
 #include "llvm/ADT/SmallBitVector.h"
 #include "llvm/ADT/StringExtras.h"
 #include "llvm/Support/Casting.h"
+#include "llvm/Support/CRC.h"
 #include "llvm/Support/Debug.h"
 #include "llvm/Support/SaveAndRestore.h"
 #include "llvm/Support/SipHash.h"
@@ -16983,7 +16984,7 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,
     return Success(APValue(ResultInt), E);
   };

-  auto HandleCRC32 = [&](unsigned DataBytes) -> bool {
+  auto HandleCRC32 = [&]() -> bool {
     APSInt CRC, Data;
     if (!EvaluateInteger(E->getArg(0), CRC, Info) ||
         !EvaluateInteger(E->getArg(1), Data, Info))
@@ -16991,20 +16992,15 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,

     uint64_t CRCVal = CRC.getZExtValue();
     uint64_t DataVal = Data.getZExtValue();
+    unsigned BitWidth = Data.getBitWidth();

     // CRC32C polynomial (iSCSI polynomial, bit-reversed)
     static const uint32_t CRC32C_POLY = 0x82F63B78;

-    // Process each byte
-    uint32_t Result = static_cast<uint32_t>(CRCVal);
-    for (unsigned I = 0; I != DataBytes; ++I) {
-      uint8_t Byte = static_cast<uint8_t>((DataVal >> (I * 8)) & 0xFF);
-      Result ^= Byte;
-      for (int J = 0; J != 8; ++J) {
-        Result = (Result >> 1) ^ ((Result & 1) ? CRC32C_POLY : 0);
-      }
-    }
+    CRCVal = llvm::calculateReflectedCrc(CRCVal, 32, DataVal, BitWidth,
+                                         CRC32C_POLY);

+    uint32_t Result = static_cast<uint32_t>(CRCVal);
     return Success(Result, E);
   };

@@ -17013,13 +17009,10 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,
     return false;

   case X86::BI__builtin_ia32_crc32qi:
-    return HandleCRC32(1);
   case X86::BI__builtin_ia32_crc32hi:
-    return HandleCRC32(2);
   case X86::BI__builtin_ia32_crc32si:
-    return HandleCRC32(4);
   case X86::BI__builtin_ia32_crc32di:
-    return HandleCRC32(8);
+    return HandleCRC32();

   case Builtin::BI__builtin_dynamic_object_size:
   case Builtin::BI__builtin_object_size: {
diff --git a/llvm/include/llvm/Support/CRC.h b/llvm/include/llvm/Support/CRC.h
index a07b017d2351..06c0be8c9b83 100644
--- a/llvm/include/llvm/Support/CRC.h
+++ b/llvm/include/llvm/Support/CRC.h
@@ -26,6 +26,31 @@ LLVM_ABI uint32_t crc32(ArrayRef<uint8_t> Data);
 // checksum.
 LLVM_ABI uint32_t crc32(uint32_t CRC, ArrayRef<uint8_t> Data);

+// Calculate the bit-reflected CRC for given initial CRC, Data and Polynomial.
+// Polynomial must already be in the bit-reflected form.
+constexpr inline uint64_t calculateReflectedCrc(uint64_t Crc, unsigned CrcBits,
+                                                uint64_t Data, unsigned DataBits,
+                                                uint64_t Poly) {
+    if (CrcBits < 64) {
+        Crc &= ((1ULL << CrcBits) - 1);
+    }
+
+    if (DataBits < 64) {
+        Data &= ((1ULL << DataBits) - 1);
+    }
+
+    Crc ^= Data;
+    for (unsigned Bit = 0; Bit != DataBits; ++Bit) {
+      Crc = (Crc >> 1) ^ ((Crc & 1) ? Poly : 0);
+    }
+
+    if (CrcBits < 64) {
+        Crc &= ((1ULL << CrcBits) - 1);
+    }
+
+    return Crc;
+}
+
 // Class for computing the JamCRC.
 //
 // We will use the "Rocksoft^tm Model CRC Algorithm" to describe the properties
```

With the above change, the diff becomes:
```
diff --git a/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp b/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
index ad1c17142867..cb7250d642e2 100644
--- a/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
@@ -16,6 +16,7 @@
 #include "X86TargetTransformInfo.h"
 #include "llvm/IR/IntrinsicInst.h"
 #include "llvm/IR/IntrinsicsX86.h"
+#include "llvm/Support/CRC.h"
 #include "llvm/Support/KnownBits.h"
 #include "llvm/Transforms/InstCombine/InstCombiner.h"
 #include <optional>
@@ -2260,6 +2261,26 @@ X86TTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
     }
     break;

+  case Intrinsic::x86_sse42_crc32_32_8:
+  case Intrinsic::x86_sse42_crc32_32_16:
+  case Intrinsic::x86_sse42_crc32_32_32:
+  case Intrinsic::x86_sse42_crc32_64_64: {
+    auto *CrcArg = dyn_cast<ConstantInt>(II.getArgOperand(0));
+    auto *DataArg = dyn_cast<ConstantInt>(II.getArgOperand(1));
+    if (!CrcArg || !DataArg)
+      break;
+
+    // If both operands are constant, we can completely constant fold this.
+    uint64_t Crc = CrcArg->getZExtValue();
+    uint64_t Data = DataArg->getZExtValue();
+    unsigned BitWidth = DataArg->getBitWidth();
+    // CRC32C polynomial (iSCSI polynomial, bit-reversed)
+    const uint64_t Poly = 0x82F63B78;
+    Crc = calculateReflectedCrc(Crc, 32, Data, BitWidth, Poly);
+    uint32_t Result = static_cast<uint32_t>(Crc);
+    return IC.replaceInstUsesWith(II, ConstantInt::get(II.getType(), Result));
+  }
+
   case Intrinsic::x86_sse_cvtss2si:
   case Intrinsic::x86_sse_cvtss2si64:
   case Intrinsic::x86_sse_cvttss2si:
```

Does this look okay? If you can confirm that architecturally this looks good, I will submit a separate PR for the refactoring and rebase this PR once that is merged.

https://github.com/llvm/llvm-project/pull/219452


More information about the llvm-commits mailing list