[llvm] [AggressiveInstCombine] Fold low-bits mask table loads to arithmetic (PR #222509)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 10 06:47:35 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Chris Kennelly (ckennelly)

<details>
<summary>Changes</summary>

Recognize a load from a constant "low bits mask" table

```
  static const uintN_t tbl[K] = { 0, 1, 3, 7, ... }; // tbl[j] == (1 << j) - 1
  ... x & tbl[i] ...
```

and replace the load with the equivalent arithmetic mask (1 << i) - 1.
This is the target-independent form of the X86 combineAndLoadToBZHI DAG
combine: on X86 with BMI2 the surrounding and still selects to a single
bzhi.

Assisted-by: Claude Code

---

Patch is 35.03 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/222509.diff


4 Files Affected:

- (modified) llvm/lib/Target/X86/X86ISelLowering.cpp (-105) 
- (modified) llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp (+94-11) 
- (modified) llvm/test/CodeGen/X86/replace-load-and-with-bzhi.ll (+22-144) 
- (added) llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask.ll (+237) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 5d45082a9e550..98a631c031cb7 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -52503,28 +52503,6 @@ static SDValue combineAndMaskToShift(SDNode *N, const SDLoc &DL,
   return DAG.getBitcast(N->getValueType(0), Shift);
 }
 
-// Get the index node from the lowered DAG of a GEP IR instruction with one
-// indexing dimension.
-static SDValue getIndexFromUnindexedLoad(LoadSDNode *Ld) {
-  if (Ld->isIndexed())
-    return SDValue();
-
-  SDValue Base = Ld->getBasePtr();
-  if (Base.getOpcode() != ISD::ADD)
-    return SDValue();
-
-  SDValue ShiftedIndex = Base.getOperand(0);
-  if (ShiftedIndex.getOpcode() != ISD::SHL)
-    return SDValue();
-
-  return ShiftedIndex.getOperand(0);
-}
-
-static bool hasBZHI(const X86Subtarget &Subtarget, MVT VT) {
-  return Subtarget.hasBMI2() &&
-         (VT == MVT::i32 || (VT == MVT::i64 && Subtarget.is64Bit()));
-}
-
 /// Folds (and X, (or Y, ~Z)) --> (and X, ~(and ~Y, Z))
 /// This undoes the inverse fold performed in InstCombine
 static SDValue combineAndNotOrIntoAndNotAnd(SDNode *N, const SDLoc &DL,
@@ -52584,86 +52562,6 @@ static SDValue combineMaskBitOp(SDNode *N, const SDLoc &DL, SelectionDAG &DAG) {
   return SDValue();
 }
 
-// This function recognizes cases where X86 bzhi instruction can replace and
-// 'and-load' sequence.
-// In case of loading integer value from an array of constants which is defined
-// as follows:
-//
-//   int array[SIZE] = {0x0, 0x1, 0x3, 0x7, 0xF ..., 2^(SIZE-1) - 1}
-//
-// then applying a bitwise and on the result with another input.
-// It's equivalent to performing bzhi (zero high bits) on the input, with the
-// same index of the load.
-static SDValue combineAndLoadToBZHI(SDNode *Node, SelectionDAG &DAG,
-                                    const X86Subtarget &Subtarget) {
-  MVT VT = Node->getSimpleValueType(0);
-  SDLoc dl(Node);
-
-  // Check if subtarget has BZHI instruction for the node's type
-  if (!hasBZHI(Subtarget, VT))
-    return SDValue();
-
-  // Try matching the pattern for both operands.
-  for (unsigned i = 0; i < 2; i++) {
-    // continue if the operand is not a load instruction
-    auto *Ld = dyn_cast<LoadSDNode>(Node->getOperand(i));
-    if (!Ld)
-      continue;
-    const Value *MemOp = Ld->getMemOperand()->getValue();
-    if (!MemOp)
-      continue;
-    // Get the Node which indexes into the array.
-    SDValue Index = getIndexFromUnindexedLoad(Ld);
-    if (!Index)
-      continue;
-
-    if (auto *GEP = dyn_cast<GetElementPtrInst>(MemOp)) {
-      if (auto *GV = dyn_cast<GlobalVariable>(GEP->getOperand(0))) {
-        if (GV->isConstant() && GV->hasDefinitiveInitializer()) {
-          Constant *Init = GV->getInitializer();
-          Type *Ty = Init->getType();
-          if (!isa<ConstantDataArray>(Init) ||
-              !Ty->getArrayElementType()->isIntegerTy() ||
-              Ty->getArrayElementType()->getScalarSizeInBits() !=
-                  VT.getSizeInBits() ||
-              Ty->getArrayNumElements() >
-                  Ty->getArrayElementType()->getScalarSizeInBits())
-            continue;
-
-          // Check if the array's constant elements are suitable to our case.
-          uint64_t ArrayElementCount = Init->getType()->getArrayNumElements();
-          bool ConstantsMatch = true;
-          for (uint64_t j = 0; j < ArrayElementCount; j++) {
-            auto *Elem = cast<ConstantInt>(Init->getAggregateElement(j));
-            if (Elem->getZExtValue() != (((uint64_t)1 << j) - 1)) {
-              ConstantsMatch = false;
-              break;
-            }
-          }
-          if (!ConstantsMatch)
-            continue;
-
-          // Do the transformation (For 32-bit type):
-          // -> (and (load arr[idx]), inp)
-          // <- (and (srl 0xFFFFFFFF, (sub 32, idx)))
-          //    that will be replaced with one bzhi instruction.
-          SDValue Inp = Node->getOperand(i == 0 ? 1 : 0);
-          SDValue SizeC = DAG.getConstant(VT.getSizeInBits(), dl, MVT::i32);
-
-          Index = DAG.getZExtOrTrunc(Index, dl, MVT::i32);
-          SDValue Sub = DAG.getNode(ISD::SUB, dl, MVT::i32, SizeC, Index);
-          Sub = DAG.getNode(ISD::TRUNCATE, dl, MVT::i8, Sub);
-
-          SDValue AllOnes = DAG.getAllOnesConstant(dl, VT);
-          SDValue LShr = DAG.getNode(ISD::SRL, dl, VT, AllOnes, Sub);
-          return DAG.getNode(ISD::AND, dl, VT, Inp, LShr);
-        }
-      }
-    }
-  }
-  return SDValue();
-}
-
 // Look for (and (bitcast (vXi1 (concat_vectors (vYi1 setcc), undef,))), C)
 // Where C is a mask containing the same number of bits as the setcc and
 // where the setcc will freely 0 upper bits of k-register. We can replace the
@@ -53101,9 +52999,6 @@ static SDValue combineAnd(SDNode *N, SelectionDAG &DAG,
   if (SDValue ShiftRight = combineAndMaskToShift(N, dl, DAG, Subtarget))
     return ShiftRight;
 
-  if (SDValue R = combineAndLoadToBZHI(N, DAG, Subtarget))
-    return R;
-
   if (SDValue R = combineAndNotOrIntoAndNotAnd(N, dl, DAG))
     return R;
 
diff --git a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
index f72ff61f028db..eed186488d013 100644
--- a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
+++ b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
@@ -57,6 +57,8 @@ STATISTIC(NumSelectCTTZFolded,
           "Number of select-based split cttz patterns folded");
 STATISTIC(NumSelectCTLZFolded,
           "Number of select-based split ctlz patterns folded");
+STATISTIC(NumTableBasedLowBitsMask,
+          "Number of low-bits mask table loads folded to (1 << i) - 1");
 
 static cl::opt<unsigned> MaxInstrsToScan(
     "aggressive-instcombine-max-scan-instrs", cl::init(64), cl::Hidden,
@@ -957,7 +959,7 @@ static bool isCTTZTable(Constant *Table, const APInt &Mul, const APInt &Shift,
 //
 // This shares its initial match (load from a GEP into a constant table with
 // a single variable index) with tryToRecognizeTableBasedLog2() below; see
-// tryToRecognizeTableBasedCttzOrLog2().
+// tryToRecognizeTableBasedPatterns().
 static bool tryToRecognizeTableBasedCttz(LoadInst *LI, Type *AccessType,
                                          GlobalVariable *GVTable, Value *GepIdx,
                                          const APInt &GEPScale,
@@ -1123,7 +1125,7 @@ static bool isLog2Table(Constant *Table, const APInt &Mul, const APInt &Shift,
 //
 // This shares its initial match (load from a GEP into a constant table with
 // a single variable index) with tryToRecognizeTableBasedCttz() above; see
-// tryToRecognizeTableBasedCttzOrLog2().
+// tryToRecognizeTableBasedPatterns().
 static bool tryToRecognizeTableBasedLog2(LoadInst *LI, Type *AccessType,
                                          GlobalVariable *GVTable, Value *GepIdx,
                                          const APInt &GEPScale,
@@ -1239,12 +1241,89 @@ static bool tryToRecognizeTableBasedLog2(LoadInst *LI, Type *AccessType,
   return true;
 }
 
-// Match a table-based cttz or log2 implementation. These patterns share a
-// load from a global table pattern that we match first. Then we try the
-// specific matches for the cttz and log2 patterns.
-static bool tryToRecognizeTableBasedCttzOrLog2(Instruction &I,
-                                               const DataLayout &DL,
-                                               TargetTransformInfo &TTI) {
+// Recognize a load from a "low bits mask" table, tbl[j] == (1 << j) - 1:
+//
+//   static const uintN_t tbl[K] = { 0, 1, 3, 7, ... };
+//   ... x & tbl[i] ...
+//
+// The loaded value equals the arithmetic mask (1 << i) - 1, so replace the load
+// with that expression and drop the table. On X86+BMI2 a surrounding `and` then
+// selects to a single `bzhi`; on other targets it is a shift and a decrement.
+//
+// Shares the load/GEP/offset match with the cttz and log2 recognizers above;
+// see tryToRecognizeTableBasedPatterns().
+static bool tryToRecognizeTableBasedLowBitsMask(LoadInst *LI, Type *AccessType,
+                                                GlobalVariable *GVTable,
+                                                Value *GepIdx,
+                                                const APInt &GEPScale,
+                                                const DataLayout &DL) {
+  // The value must be exactly the table element; refuse volatile/atomic loads.
+  if (!LI->isSimple())
+    return false;
+
+  // The initializer must be fixed at link time; this rules out interposable or
+  // replaceable definitions whose bytes a linker could swap for another table.
+  if (!GVTable->hasDefinitiveInitializer())
+    return false;
+
+  // The table must be [K x AccessType] of integers; a narrower load would read
+  // only part of an element.
+  auto *ArrTy = dyn_cast<ArrayType>(GVTable->getValueType());
+  if (!ArrTy || ArrTy->getElementType() != AccessType ||
+      !AccessType->isIntegerTy())
+    return false;
+
+  unsigned EltBits = AccessType->getIntegerBitWidth();
+  uint64_t EltBytes = DL.getTypeAllocSize(AccessType).getFixedValue();
+  uint64_t K = ArrTy->getNumElements();
+
+  // Only fire for an integer width the target handles natively, so the emitted
+  // shift is a cheap instruction rather than a runtime libcall (e.g. __ashlti3
+  // for i128) -- which would be worse than the load and is unavailable in
+  // freestanding environments that do not link a runtime library.
+  if (!DL.isLegalInteger(EltBits))
+    return false;
+
+  // K <= EltBits guarantees every in-bounds index i satisfies i < EltBits, so
+  // the emitted `1 << i` never shifts by >= bitwidth (no poison). A table with
+  // EltBits + 1 entries (whose last element is the all-ones mask, needing a
+  // shift by EltBits) is therefore rejected.
+  if (K == 0 || K > EltBits)
+    return false;
+
+  // The index must step by exactly one element, so the runtime index value is
+  // the shift amount; a different scale would load tbl[c * i].
+  if (GEPScale != EltBytes)
+    return false;
+
+  // Every element must be the low-bits mask for its position.
+  for (uint64_t J = 0; J < K; ++J) {
+    Constant *Elt = ConstantFoldLoadFromConst(
+        GVTable->getInitializer(), AccessType,
+        APInt(GEPScale.getBitWidth(), J) * GEPScale, DL);
+    auto *CI = dyn_cast_or_null<ConstantInt>(Elt);
+    if (!CI || CI->getValue() != APInt::getLowBitsSet(EltBits, J))
+      return false;
+  }
+
+  // Emit (1 << i) - 1 in the element type and replace the load. A later
+  // InstCombine canonicalizes this to ~(-1 << i); X86 lowers both to bzhi.
+  IRBuilder<> Builder(LI);
+  Value *Idx = Builder.CreateZExtOrTrunc(GepIdx, AccessType);
+  Value *Mask =
+      Builder.CreateSub(Builder.CreateShl(ConstantInt::get(AccessType, 1), Idx),
+                        ConstantInt::get(AccessType, 1));
+  LI->replaceAllUsesWith(Mask);
+  ++NumTableBasedLowBitsMask;
+  return true;
+}
+
+// Match a table-based cttz, log2, or low-bits-mask implementation. These
+// patterns share a load from a global table that we match first; then we
+// try the specific matches.
+static bool tryToRecognizeTableBasedPatterns(Instruction &I,
+                                             const DataLayout &DL,
+                                             TargetTransformInfo &TTI) {
   LoadInst *LI = dyn_cast<LoadInst>(&I);
   if (!LI)
     return false;
@@ -1273,8 +1352,12 @@ static bool tryToRecognizeTableBasedCttzOrLog2(Instruction &I,
                                    DL))
     return true;
 
-  return tryToRecognizeTableBasedLog2(LI, AccessType, GVTable, GepIdx, GEPScale,
-                                      DL, TTI);
+  if (tryToRecognizeTableBasedLog2(LI, AccessType, GVTable, GepIdx, GEPScale,
+                                   DL, TTI))
+    return true;
+
+  return tryToRecognizeTableBasedLowBitsMask(LI, AccessType, GVTable, GepIdx,
+                                             GEPScale, DL);
 }
 
 /// This is used by foldLoadsRecursive() to capture a Root Load node which is
@@ -2490,7 +2573,7 @@ static bool foldUnusualPatterns(Function &F, DominatorTree &DT,
       MadeChange |= tryToRecognizePopCount1(I);
       MadeChange |= tryToRecognizePopCount2n3(I);
       MadeChange |= tryToFPToSat(I, TTI);
-      MadeChange |= tryToRecognizeTableBasedCttzOrLog2(I, DL, TTI);
+      MadeChange |= tryToRecognizeTableBasedPatterns(I, DL, TTI);
       MadeChange |= foldConsecutiveLoads(I, DL, TTI, AA, DT);
       MadeChange |= foldPatternedLoads(I, DL);
       MadeChange |= foldICmpOrChain(I, DL, TTI, AA, DT);
diff --git a/llvm/test/CodeGen/X86/replace-load-and-with-bzhi.ll b/llvm/test/CodeGen/X86/replace-load-and-with-bzhi.ll
index a0bd35d5d219b..0050fb0236894 100644
--- a/llvm/test/CodeGen/X86/replace-load-and-with-bzhi.ll
+++ b/llvm/test/CodeGen/X86/replace-load-and-with-bzhi.ll
@@ -1,168 +1,46 @@
-; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+bmi2 | FileCheck %s -check-prefix=X64
-; RUN: llc < %s -mtriple=i686-unknown-unknown -mattr=+bmi2 | FileCheck %s -check-prefix=X86
+; Verify the end-to-end pipeline for the "low bits mask" table idiom
+;   x & tbl[i]     where tbl[j] == (1 << j) - 1
+; AggressiveInstCombine rewrites the load to (1 << i) - 1, InstCombine
+; canonicalizes it, and the X86 backend selects a single bzhi -- including
+; under PIC, which the removed combineAndLoadToBZHI DAG fold could not handle.
+;
+; RUN: opt < %s -passes=aggressive-instcombine,instcombine -S | llc -mtriple=x86_64-unknown-unknown -mattr=+bmi2 | FileCheck %s
+; RUN: opt < %s -passes=aggressive-instcombine,instcombine -S | llc -mtriple=x86_64-unknown-unknown -mattr=+bmi2 -relocation-model=pic | FileCheck %s
 
- at fill_table32 = internal unnamed_addr constant [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647], align 16
- at fill_table32_partial = internal unnamed_addr constant [17 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535], align 16
- at fill_table64 = internal unnamed_addr constant [64 x i64] [i64 0, i64 1, i64 3, i64 7, i64 15, i64 31, i64 63, i64 127, i64 255, i64 511, i64 1023, i64 2047, i64 4095, i64 8191, i64 16383, i64 32767, i64 65535, i64 131071, i64 262143, i64 524287, i64 1048575, i64 2097151, i64 4194303, i64 8388607, i64 16777215, i64 33554431, i64 67108863, i64 134217727, i64 268435455, i64 536870911, i64 1073741823, i64 2147483647, i64 4294967295, i64 8589934591, i64 17179869183, i64 34359738367, i64 68719476735, i64 137438953471, i64 274877906943, i64 549755813887, i64 1099511627775, i64 2199023255551, i64 4398046511103, i64 8796093022207, i64 17592186044415, i64 35184372088831, i64 70368744177663, i64 140737488355327, i64 281474976710655, i64 562949953421311, i64 1125899906842623, i64 2251799813685247, i64 4503599627370495, i64 9007199254740991, i64 18014398509481983, i64 36028797018963967, i64 72057594037927935, i64 144115188075855871, i64 288230376151711743, i64 576460752303423487, i64 1152921504606846975, i64 2305843009213693951, i64 4611686018427387903, i64 9223372036854775807], align 16
- at fill_table64_partial = internal unnamed_addr constant [51 x i64] [i64 0, i64 1, i64 3, i64 7, i64 15, i64 31, i64 63, i64 127, i64 255, i64 511, i64 1023, i64 2047, i64 4095, i64 8191, i64 16383, i64 32767, i64 65535, i64 131071, i64 262143, i64 524287, i64 1048575, i64 2097151, i64 4194303, i64 8388607, i64 16777215, i64 33554431, i64 67108863, i64 134217727, i64 268435455, i64 536870911, i64 1073741823, i64 2147483647, i64 4294967295, i64 8589934591, i64 17179869183, i64 34359738367, i64 68719476735, i64 137438953471, i64 274877906943, i64 549755813887, i64 1099511627775, i64 2199023255551, i64 4398046511103, i64 8796093022207, i64 17592186044415, i64 35184372088831, i64 70368744177663, i64 140737488355327, i64 281474976710655, i64 562949953421311, i64 1125899906842623], align 16
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
 
-define i32 @f32_bzhi(i32 %x, i32 %y) local_unnamed_addr {
-; X64-LABEL: f32_bzhi:
-; X64:       # %bb.0: # %entry
-; X64-NEXT:    bzhil %esi, %edi, %eax
-; X64-NEXT:    retq
-;
-; X86-LABEL: f32_bzhi:
-; X86:       # %bb.0: # %entry
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    bzhil %eax, {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    retl
-entry:
-  %idxprom = sext i32 %y to i64
-  %arrayidx = getelementptr inbounds [32 x i32], ptr @fill_table32, i64 0, i64 %idxprom
-  %0 = load i32, ptr %arrayidx, align 4
-  %and = and i32 %0, %x
-  ret i32 %and
-}
+ at fill_table32 = internal unnamed_addr constant [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647]
+ at fill_table64 = internal unnamed_addr constant [64 x i64] [i64 0, i64 1, i64 3, i64 7, i64 15, i64 31, i64 63, i64 127, i64 255, i64 511, i64 1023, i64 2047, i64 4095, i64 8191, i64 16383, i64 32767, i64 65535, i64 131071, i64 262143, i64 524287, i64 1048575, i64 2097151, i64 4194303, i64 8388607, i64 16777215, i64 33554431, i64 67108863, i64 134217727, i64 268435455, i64 536870911, i64 1073741823, i64 2147483647, i64 4294967295, i64 8589934591, i64 17179869183, i64 34359738367, i64 68719476735, i64 137438953471, i64 274877906943, i64 549755813887, i64 1099511627775, i64 2199023255551, i64 4398046511103, i64 8796093022207, i64 17592186044415, i64 35184372088831, i64 70368744177663, i64 140737488355327, i64 281474976710655, i64 562949953421311, i64 1125899906842623, i64 2251799813685247, i64 4503599627370495, i64 9007199254740991, i64 18014398509481983, i64 36028797018963967, i64 72057594037927935, i64 144115188075855871, i64 288230376151711743, i64 576460752303423487, i64 1152921504606846975, i64 2305843009213693951, i64 4611686018427387903, i64 9223372036854775807]
 
-define i32 @f32_bzhi_commute(i32 %x, i32 %y) local_unnamed_addr {
-; X64-LABEL: f32_bzhi_commute:
-; X64:       # %bb.0: # %entry
-; X64-NEXT:    bzhil %esi, %edi, %eax
-; X64-NEXT:    retq
-;
-; X86-LABEL: f32_bzhi_commute:
-; X86:       # %bb.0: # %entry
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    bzhil %eax, {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    retl
+define i32 @f32_bzhi(i32 %x, i32 %y) {
+; CHECK-LABEL: f32_bzhi:
+; CHECK:         bzhil
+; CHECK-NOT:     fill_table32
 entry:
   %idxprom = sext i32 %y to i64
   %arrayidx = getelementptr inbounds [32 x i32], ptr @fill_table32, i64 0, i64 %idxprom
   %0 = load i32, ptr %arrayidx, align 4
-  %and = and i32 %x, %0
-  ret i32 %and
-}
-
-define i32 @f32_bzhi_partial(i32 %x, i32 %y) local_unnamed_addr {
-; X64-LABEL: f32_bzhi_partial:
-; X64:       # %bb.0: # %entry
-; X64-NEXT:    bzhil %esi, %edi, %eax
-; X64-NEXT:    retq
-;
-; X86-LABEL: f32_bzhi_partial:
-; X86:       # %bb.0: # %entry
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    bzhil %eax, {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    retl
-entry:
-  %idxprom = sext i32 %y to i64
-  %arrayidx = getelementptr inbounds [17 x i32], ptr @fill_table32_partial, i64 0, i64 %idxprom
-  %0 = load i32, ptr %arrayidx, align 4
   %and = and i32 %0, %x
   ret i32 %and
 }
 
-define i32 @f32_bzhi_partial_commute(i32 %x, i32 %y) local_unnamed_addr {
-; X64-LABEL: f32_bzhi_partial_commute:
-; X64:       # %bb.0: # %entry
-; X64-NEXT:    bzhil %esi, %edi, %eax
-; X64-NEXT:    retq
-;
-; X86-LABEL: f32_bzhi_partial_commute:
-; X86:       # %bb.0: # %entry
-; X86-NEXT:    movl {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    bzhil %eax, {{[0-9]+}}(%esp), %eax
-; X86-NEXT:    retl
+define i32 @f32_bzhi_commute(i32 %x, i32 %y) {
+; CHECK-LABEL: f32_bzhi_commute:
+; CHECK:         bzhil
 entry:
   %idxprom = sext i32 %y to i64
-  %arrayidx = getelementptr inbou...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/222509


More information about the llvm-commits mailing list