[llvm] [X86] Implement CRC32 const folding (PR #219452)
Shreesh Adiga via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 07:35:15 PDT 2026
================
@@ -2260,6 +2260,29 @@ X86TTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
}
break;
+ case Intrinsic::x86_sse42_crc32_32_8:
+ case Intrinsic::x86_sse42_crc32_32_16:
+ case Intrinsic::x86_sse42_crc32_32_32:
+ case Intrinsic::x86_sse42_crc32_64_64: {
+ auto *CrcArg = dyn_cast<ConstantInt>(II.getArgOperand(0));
+ auto *DataArg = dyn_cast<ConstantInt>(II.getArgOperand(1));
+ if (!CrcArg || !DataArg)
+ break;
+
+ // If both operands are constant, we can completely constant fold this.
+ uint64_t Crc = CrcArg->getZExtValue() & 0xffffffff;
+ uint64_t Data = DataArg->getZExtValue();
+ unsigned BitWidth = DataArg->getBitWidth();
+ Crc ^= Data;
+ // CRC32C polynomial (iSCSI polynomial, bit-reversed)
+ const uint32_t Poly = 0x82F63B78;
+ for (unsigned Bit = 0; Bit != BitWidth; ++Bit) {
+ Crc = (Crc >> 1) ^ ((Crc & 1) ? Poly : 0);
+ }
----------------
tantei3 wrote:
With the below diff the tests are passing and core calculation is moved to CRC.h file:
```
diff --git a/clang/lib/AST/ByteCode/InterpBuiltin.cpp b/clang/lib/AST/ByteCode/InterpBuiltin.cpp
index 1a464247b5ce..255cab8aecda 100644
--- a/clang/lib/AST/ByteCode/InterpBuiltin.cpp
+++ b/clang/lib/AST/ByteCode/InterpBuiltin.cpp
@@ -21,6 +21,7 @@
#include "clang/Basic/TargetInfo.h"
#include "llvm/ADT/StringExtras.h"
#include "llvm/Support/AllocToken.h"
+#include "llvm/Support/CRC.h"
#include "llvm/Support/ErrorHandling.h"
#include "llvm/Support/SipHash.h"
@@ -814,16 +815,9 @@ static bool interp__builtin_ia32_crc32(InterpState &S, CodePtr OpPC,
// CRC32C polynomial (iSCSI polynomial, bit-reversed)
static const uint32_t CRC32C_POLY = 0x82F63B78;
- // Process each byte
+ CRCVal = llvm::calculateReflectedCrc(CRCVal, 32, DataVal, DataBytes * 8,
+ CRC32C_POLY);
uint32_t Result = static_cast<uint32_t>(CRCVal);
- for (unsigned I = 0; I != DataBytes; ++I) {
- uint8_t Byte = static_cast<uint8_t>((DataVal >> (I * 8)) & 0xFF);
- Result ^= Byte;
- for (int J = 0; J != 8; ++J) {
- Result = (Result >> 1) ^ ((Result & 1) ? CRC32C_POLY : 0);
- }
- }
-
pushInteger(S, Result, Call->getType());
return true;
}
diff --git a/clang/lib/AST/ExprConstant.cpp b/clang/lib/AST/ExprConstant.cpp
index 9702105951b7..4bb910d290b2 100644
--- a/clang/lib/AST/ExprConstant.cpp
+++ b/clang/lib/AST/ExprConstant.cpp
@@ -61,6 +61,7 @@
#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/StringExtras.h"
#include "llvm/Support/Casting.h"
+#include "llvm/Support/CRC.h"
#include "llvm/Support/Debug.h"
#include "llvm/Support/SaveAndRestore.h"
#include "llvm/Support/SipHash.h"
@@ -16983,7 +16984,7 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,
return Success(APValue(ResultInt), E);
};
- auto HandleCRC32 = [&](unsigned DataBytes) -> bool {
+ auto HandleCRC32 = [&]() -> bool {
APSInt CRC, Data;
if (!EvaluateInteger(E->getArg(0), CRC, Info) ||
!EvaluateInteger(E->getArg(1), Data, Info))
@@ -16991,20 +16992,15 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,
uint64_t CRCVal = CRC.getZExtValue();
uint64_t DataVal = Data.getZExtValue();
+ unsigned BitWidth = Data.getBitWidth();
// CRC32C polynomial (iSCSI polynomial, bit-reversed)
static const uint32_t CRC32C_POLY = 0x82F63B78;
- // Process each byte
- uint32_t Result = static_cast<uint32_t>(CRCVal);
- for (unsigned I = 0; I != DataBytes; ++I) {
- uint8_t Byte = static_cast<uint8_t>((DataVal >> (I * 8)) & 0xFF);
- Result ^= Byte;
- for (int J = 0; J != 8; ++J) {
- Result = (Result >> 1) ^ ((Result & 1) ? CRC32C_POLY : 0);
- }
- }
+ CRCVal = llvm::calculateReflectedCrc(CRCVal, 32, DataVal, BitWidth,
+ CRC32C_POLY);
+ uint32_t Result = static_cast<uint32_t>(CRCVal);
return Success(Result, E);
};
@@ -17013,13 +17009,10 @@ bool IntExprEvaluator::VisitBuiltinCallExpr(const CallExpr *E,
return false;
case X86::BI__builtin_ia32_crc32qi:
- return HandleCRC32(1);
case X86::BI__builtin_ia32_crc32hi:
- return HandleCRC32(2);
case X86::BI__builtin_ia32_crc32si:
- return HandleCRC32(4);
case X86::BI__builtin_ia32_crc32di:
- return HandleCRC32(8);
+ return HandleCRC32();
case Builtin::BI__builtin_dynamic_object_size:
case Builtin::BI__builtin_object_size: {
diff --git a/llvm/include/llvm/Support/CRC.h b/llvm/include/llvm/Support/CRC.h
index a07b017d2351..06c0be8c9b83 100644
--- a/llvm/include/llvm/Support/CRC.h
+++ b/llvm/include/llvm/Support/CRC.h
@@ -26,6 +26,31 @@ LLVM_ABI uint32_t crc32(ArrayRef<uint8_t> Data);
// checksum.
LLVM_ABI uint32_t crc32(uint32_t CRC, ArrayRef<uint8_t> Data);
+// Calculate the bit-reflected CRC for given initial CRC, Data and Polynomial.
+// Polynomial must already be in the bit-reflected form.
+constexpr inline uint64_t calculateReflectedCrc(uint64_t Crc, unsigned CrcBits,
+ uint64_t Data, unsigned DataBits,
+ uint64_t Poly) {
+ if (CrcBits < 64) {
+ Crc &= ((1ULL << CrcBits) - 1);
+ }
+
+ if (DataBits < 64) {
+ Data &= ((1ULL << DataBits) - 1);
+ }
+
+ Crc ^= Data;
+ for (unsigned Bit = 0; Bit != DataBits; ++Bit) {
+ Crc = (Crc >> 1) ^ ((Crc & 1) ? Poly : 0);
+ }
+
+ if (CrcBits < 64) {
+ Crc &= ((1ULL << CrcBits) - 1);
+ }
+
+ return Crc;
+}
+
// Class for computing the JamCRC.
//
// We will use the "Rocksoft^tm Model CRC Algorithm" to describe the properties
```
With the above change, the diff becomes:
```
diff --git a/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp b/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
index ad1c17142867..cb7250d642e2 100644
--- a/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/X86/X86InstCombineIntrinsic.cpp
@@ -16,6 +16,7 @@
#include "X86TargetTransformInfo.h"
#include "llvm/IR/IntrinsicInst.h"
#include "llvm/IR/IntrinsicsX86.h"
+#include "llvm/Support/CRC.h"
#include "llvm/Support/KnownBits.h"
#include "llvm/Transforms/InstCombine/InstCombiner.h"
#include <optional>
@@ -2260,6 +2261,26 @@ X86TTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
}
break;
+ case Intrinsic::x86_sse42_crc32_32_8:
+ case Intrinsic::x86_sse42_crc32_32_16:
+ case Intrinsic::x86_sse42_crc32_32_32:
+ case Intrinsic::x86_sse42_crc32_64_64: {
+ auto *CrcArg = dyn_cast<ConstantInt>(II.getArgOperand(0));
+ auto *DataArg = dyn_cast<ConstantInt>(II.getArgOperand(1));
+ if (!CrcArg || !DataArg)
+ break;
+
+ // If both operands are constant, we can completely constant fold this.
+ uint64_t Crc = CrcArg->getZExtValue();
+ uint64_t Data = DataArg->getZExtValue();
+ unsigned BitWidth = DataArg->getBitWidth();
+ // CRC32C polynomial (iSCSI polynomial, bit-reversed)
+ const uint64_t Poly = 0x82F63B78;
+ Crc = calculateReflectedCrc(Crc, 32, Data, BitWidth, Poly);
+ uint32_t Result = static_cast<uint32_t>(Crc);
+ return IC.replaceInstUsesWith(II, ConstantInt::get(II.getType(), Result));
+ }
+
case Intrinsic::x86_sse_cvtss2si:
case Intrinsic::x86_sse_cvtss2si64:
case Intrinsic::x86_sse_cvttss2si:
```
Does this look okay? If you can confirm that architecturally this looks good, I will submit a separate PR for the refactoring and rebase this PR once that is merged.
https://github.com/llvm/llvm-project/pull/219452
More information about the llvm-commits
mailing list