[llvm] [AggressiveInstCombine] Fold low-bits mask table loads to arithmetic (PR #222509)
Simon Pilgrim via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 01:22:18 PDT 2026
https://github.com/RKSimon updated https://github.com/llvm/llvm-project/pull/222509
>From fb907b09e111d37f6e106526a000cb120283cf2f Mon Sep 17 00:00:00 2001
From: Chris Kennelly <ckennelly at ckennelly.com>
Date: Thu, 10 Sep 2026 04:45:30 +0000
Subject: [PATCH] [AggressiveInstCombine] Fold low-bits mask table loads to
arithmetic
Recognize a load from a constant "low bits mask" table
static const uintN_t tbl[K] = { 0, 1, 3, 7, ... }; // tbl[j] == (1 << j) - 1
... x & tbl[i] ...
and replace the load with the equivalent arithmetic mask (1 << i) - 1.
This is the target-independent form of the X86 combineAndLoadToBZHI DAG
combine: on X86 with BMI2 the surrounding and still selects to a single
bzhi. It covers cases not currently handled (PIC) and those that miscompile as
discussed in https://github.com/llvm/llvm-project/pull/221025.
Assisted-by: Claude Code
---
.../AggressiveInstCombine.cpp | 111 +++++++-
.../lower-table-based-lowbits-mask-n32.ll | 52 ++++
.../lower-table-based-lowbits-mask.ll | 253 ++++++++++++++++++
3 files changed, 405 insertions(+), 11 deletions(-)
create mode 100644 llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask-n32.ll
create mode 100644 llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask.ll
diff --git a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
index 1da09a468b583..c5548eb107a11 100644
--- a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
+++ b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombine.cpp
@@ -56,6 +56,8 @@ STATISTIC(NumSelectCTTZFolded,
STATISTIC(NumSelectCTLZFolded,
"Number of select-based split ctlz patterns folded");
STATISTIC(NumMemSetsGuarded, "Number of memsets guarded for a zero length");
+STATISTIC(NumTableBasedLowBitsMask,
+ "Number of low-bits mask table loads folded to (1 << i) - 1");
static cl::opt<unsigned> MaxInstrsToScan(
"aggressive-instcombine-max-scan-instrs", cl::init(64), cl::Hidden,
@@ -960,7 +962,7 @@ static bool isCTTZTable(Constant *Table, const APInt &Mul, const APInt &Shift,
//
// This shares its initial match (load from a GEP into a constant table with
// a single variable index) with tryToRecognizeTableBasedLog2() below; see
-// tryToRecognizeTableBasedCttzOrLog2().
+// tryToRecognizeTableBasedPatterns().
static bool tryToRecognizeTableBasedCttz(LoadInst *LI, Type *AccessType,
GlobalVariable *GVTable, Value *GepIdx,
const APInt &GEPScale,
@@ -1124,7 +1126,7 @@ static bool isLog2Table(Constant *Table, const APInt &Mul, const APInt &Shift,
//
// This shares its initial match (load from a GEP into a constant table with
// a single variable index) with tryToRecognizeTableBasedCttz() above; see
-// tryToRecognizeTableBasedCttzOrLog2().
+// tryToRecognizeTableBasedPatterns().
static bool tryToRecognizeTableBasedLog2(LoadInst *LI, Type *AccessType,
GlobalVariable *GVTable, Value *GepIdx,
const APInt &GEPScale,
@@ -1238,12 +1240,95 @@ static bool tryToRecognizeTableBasedLog2(LoadInst *LI, Type *AccessType,
return true;
}
-// Match a table-based cttz or log2 implementation. These patterns share a
-// load from a global table pattern that we match first. Then we try the
-// specific matches for the cttz and log2 patterns.
-static bool tryToRecognizeTableBasedCttzOrLog2(Instruction &I,
- const DataLayout &DL,
- TargetTransformInfo &TTI) {
+// Recognize a load from a "low bits mask" table, tbl[j] == (1 << j) - 1:
+//
+// static const uintN_t tbl[K] = { 0, 1, 3, 7, ... };
+// ... x & tbl[i] ...
+//
+// The loaded value equals the arithmetic mask (1 << i) - 1, so replace the load
+// with that expression and drop the table. On X86+BMI2 a surrounding `and` then
+// selects to a single `bzhi`; on other targets it is a shift and a decrement.
+//
+// Shares the load/GEP/offset match with the cttz and log2 recognizers above;
+// see tryToRecognizeTableBasedPatterns().
+static bool tryToRecognizeTableBasedLowBitsMask(LoadInst *LI, Type *AccessType,
+ GlobalVariable *GVTable,
+ Value *GepIdx,
+ const APInt &GEPScale,
+ const DataLayout &DL) {
+ // The value must be exactly the table element; refuse volatile/atomic loads.
+ if (!LI->isSimple())
+ return false;
+
+ // The table must be [K x AccessType] of integers; a narrower load would read
+ // only part of an element.
+ auto *ArrTy = dyn_cast<ArrayType>(GVTable->getValueType());
+ if (!ArrTy || ArrTy->getElementType() != AccessType ||
+ !AccessType->isIntegerTy())
+ return false;
+
+ unsigned EltBits = AccessType->getIntegerBitWidth();
+ uint64_t EltBytes = DL.getTypeAllocSize(AccessType).getFixedValue();
+ uint64_t K = ArrTy->getNumElements();
+
+ // Only fire when the element type fits in a legal integer, so the emitted
+ // shift is a cheap (possibly promoted) instruction rather than a runtime
+ // libcall (e.g. __ashlti3 for i128) -- which would be worse than the load
+ // and is unavailable in freestanding environments that do not link a
+ // runtime library.
+ if (!DL.fitsInLegalInteger(EltBits))
+ return false;
+
+ // The rewrite is only valid for indices inside the table. The nusw the
+ // shared matcher requires keeps the address computation from wrapping but
+ // does not keep it inside the object: a nusw-only GEP with i >= K is a
+ // well-defined pointer past the table, and a load through it (into
+ // whatever follows) is not (1 << i) - 1. With inbounds such a GEP is poison
+ // and the load is UB, so every executed load has 0 <= i < K.
+ auto *GEP = cast<GetElementPtrInst>(LI->getPointerOperand());
+ if (!GEP->isInBounds())
+ return false;
+
+ // K <= EltBits then guarantees i < EltBits, so the emitted `1 << i` never
+ // shifts by >= bitwidth (no poison). A table with EltBits + 1 entries (whose
+ // last element is the all-ones mask, needing a shift by EltBits) is
+ // therefore rejected.
+ if (K == 0 || K > EltBits)
+ return false;
+
+ // The index must step by exactly one element, so the runtime index value is
+ // the shift amount; a different scale would load tbl[c * i].
+ if (GEPScale != EltBytes)
+ return false;
+
+ // Every element must be the low-bits mask for its position.
+ for (uint64_t J = 0; J < K; ++J) {
+ Constant *Elt = ConstantFoldLoadFromConst(
+ GVTable->getInitializer(), AccessType,
+ APInt(GEPScale.getBitWidth(), J) * GEPScale, DL);
+ auto *CI = dyn_cast_or_null<ConstantInt>(Elt);
+ if (!CI || CI->getValue() != APInt::getLowBitsSet(EltBits, J))
+ return false;
+ }
+
+ // Emit (1 << i) - 1 in the element type and replace the load. A later
+ // InstCombine canonicalizes this to ~(-1 << i); X86 lowers both to bzhi.
+ IRBuilder<> Builder(LI);
+ Value *Idx = Builder.CreateZExtOrTrunc(GepIdx, AccessType);
+ Value *Mask =
+ Builder.CreateSub(Builder.CreateShl(ConstantInt::get(AccessType, 1), Idx),
+ ConstantInt::get(AccessType, 1));
+ LI->replaceAllUsesWith(Mask);
+ ++NumTableBasedLowBitsMask;
+ return true;
+}
+
+// Match a table-based cttz, log2, or low-bits-mask implementation. These
+// patterns share a load from a global table that we match first; then we
+// try the specific matches.
+static bool tryToRecognizeTableBasedPatterns(Instruction &I,
+ const DataLayout &DL,
+ TargetTransformInfo &TTI) {
LoadInst *LI = dyn_cast<LoadInst>(&I);
if (!LI)
return false;
@@ -1273,8 +1358,12 @@ static bool tryToRecognizeTableBasedCttzOrLog2(Instruction &I,
DL))
return true;
- return tryToRecognizeTableBasedLog2(LI, AccessType, GVTable, GepIdx, GEPScale,
- DL, TTI);
+ if (tryToRecognizeTableBasedLog2(LI, AccessType, GVTable, GepIdx, GEPScale,
+ DL, TTI))
+ return true;
+
+ return tryToRecognizeTableBasedLowBitsMask(LI, AccessType, GVTable, GepIdx,
+ GEPScale, DL);
}
/// This is used by foldLoadsRecursive() to capture a Root Load node which is
@@ -2553,7 +2642,7 @@ static bool foldUnusualPatterns(Function &F, DominatorTree &DT,
MadeChange |= tryToRecognizePopCount1(I);
MadeChange |= tryToRecognizePopCount2n3(I);
MadeChange |= tryToFPToSat(I, TTI);
- MadeChange |= tryToRecognizeTableBasedCttzOrLog2(I, DL, TTI);
+ MadeChange |= tryToRecognizeTableBasedPatterns(I, DL, TTI);
MadeChange |= foldConsecutiveLoads(I, DL, TTI, AA, DT);
MadeChange |= foldPatternedLoads(I, DL);
MadeChange |= foldICmpOrChain(I, DL, TTI, AA, DT);
diff --git a/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask-n32.ll b/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask-n32.ll
new file mode 100644
index 0000000000000..3eea17f403eab
--- /dev/null
+++ b/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask-n32.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=aggressive-instcombine -S < %s | FileCheck %s
+
+; i8 and i16 are not legal integers on this (AArch64-like) datalayout, but they
+; fit in one, so the fold still fires and the shift is promoted rather than
+; turned into a libcall. i128 fits in no legal integer and stays a load.
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
+
+ at masks8 = constant [8 x i8] [i8 0, i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127]
+ at masks16 = constant [16 x i16] [i16 0, i16 1, i16 3, i16 7, i16 15, i16 31, i16 63, i16 127, i16 255, i16 511, i16 1023, i16 2047, i16 4095, i16 8191, i16 16383, i16 32767]
+ at masks128 = constant [4 x i128] [i128 0, i128 1, i128 3, i128 7]
+
+define i8 @pos_i8(i8 %x, i64 %i) {
+; CHECK-LABEL: @pos_i8(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = shl i8 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i8 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i8 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i8 [[AND]]
+;
+ %p = getelementptr inbounds [8 x i8], ptr @masks8, i64 0, i64 %i
+ %m = load i8, ptr %p, align 1
+ %and = and i8 %m, %x
+ ret i8 %and
+}
+
+define i16 @pos_i16(i16 %x, i64 %i) {
+; CHECK-LABEL: @pos_i16(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i16
+; CHECK-NEXT: [[TMP2:%.*]] = shl i16 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i16 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i16 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i16 [[AND]]
+;
+ %p = getelementptr inbounds [16 x i16], ptr @masks16, i64 0, i64 %i
+ %m = load i16, ptr %p, align 2
+ %and = and i16 %m, %x
+ ret i16 %and
+}
+
+define i128 @neg_i128(i128 %x, i64 %i) {
+; CHECK-LABEL: @neg_i128(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [4 x i128], ptr @masks128, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i128, ptr [[P]], align 16
+; CHECK-NEXT: [[AND:%.*]] = and i128 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i128 [[AND]]
+;
+ %p = getelementptr inbounds [4 x i128], ptr @masks128, i64 0, i64 %i
+ %m = load i128, ptr %p, align 16
+ %and = and i128 %m, %x
+ ret i128 %and
+}
diff --git a/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask.ll b/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask.ll
new file mode 100644
index 0000000000000..0c8b6e1f1e06c
--- /dev/null
+++ b/llvm/test/Transforms/AggressiveInstCombine/lower-table-based-lowbits-mask.ll
@@ -0,0 +1,253 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=aggressive-instcombine -S < %s | FileCheck %s
+
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
+
+ at masks32 = constant [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647]
+ at masks32_16 = constant [16 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767]
+ at masks64 = constant [64 x i64] [i64 0, i64 1, i64 3, i64 7, i64 15, i64 31, i64 63, i64 127, i64 255, i64 511, i64 1023, i64 2047, i64 4095, i64 8191, i64 16383, i64 32767, i64 65535, i64 131071, i64 262143, i64 524287, i64 1048575, i64 2097151, i64 4194303, i64 8388607, i64 16777215, i64 33554431, i64 67108863, i64 134217727, i64 268435455, i64 536870911, i64 1073741823, i64 2147483647, i64 4294967295, i64 8589934591, i64 17179869183, i64 34359738367, i64 68719476735, i64 137438953471, i64 274877906943, i64 549755813887, i64 1099511627775, i64 2199023255551, i64 4398046511103, i64 8796093022207, i64 17592186044415, i64 35184372088831, i64 70368744177663, i64 140737488355327, i64 281474976710655, i64 562949953421311, i64 1125899906842623, i64 2251799813685247, i64 4503599627370495, i64 9007199254740991, i64 18014398509481983, i64 36028797018963967, i64 72057594037927935, i64 144115188075855871, i64 288230376151711743, i64 576460752303423487, i64 1152921504606846975, i64 2305843009213693951, i64 4611686018427387903, i64 9223372036854775807]
+ at masks8 = constant [8 x i8] [i8 0, i8 1, i8 3, i8 7, i8 15, i8 31, i8 63, i8 127]
+ at masks16 = constant [16 x i16] [i16 0, i16 1, i16 3, i16 7, i16 15, i16 31, i16 63, i16 127, i16 255, i16 511, i16 1023, i16 2047, i16 4095, i16 8191, i16 16383, i16 32767]
+ at masks128 = constant [4 x i128] [i128 0, i128 1, i128 3, i128 7]
+ at masks32_33 = constant [33 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647, i32 -1]
+ at masks32_bad = constant [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 42]
+ at masks32_mut = global [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647]
+ at masks32_weak = weak constant [32 x i32] [i32 0, i32 1, i32 3, i32 7, i32 15, i32 31, i32 63, i32 127, i32 255, i32 511, i32 1023, i32 2047, i32 4095, i32 8191, i32 16383, i32 32767, i32 65535, i32 131071, i32 262143, i32 524287, i32 1048575, i32 2097151, i32 4194303, i32 8388607, i32 16777215, i32 33554431, i32 67108863, i32 134217727, i32 268435455, i32 536870911, i32 1073741823, i32 2147483647]
+
+define i32 @pos_i32(i32 %x, i64 %i) {
+; CHECK-LABEL: @pos_i32(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i32
+; CHECK-NEXT: [[TMP2:%.*]] = shl i32 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i32 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i32 @pos_i32_short(i32 %x, i64 %i) {
+; CHECK-LABEL: @pos_i32_short(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i32
+; CHECK-NEXT: [[TMP2:%.*]] = shl i32 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i32 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [16 x i32], ptr @masks32_16, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i64 @pos_i64(i64 %x, i64 %i) {
+; CHECK-LABEL: @pos_i64(
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 1, [[I:%.*]]
+; CHECK-NEXT: [[TMP2:%.*]] = sub i64 [[TMP1]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i64 [[TMP2]], [[X:%.*]]
+; CHECK-NEXT: ret i64 [[AND]]
+;
+ %p = getelementptr inbounds [64 x i64], ptr @masks64, i64 0, i64 %i
+ %m = load i64, ptr %p, align 8
+ %and = and i64 %m, %x
+ ret i64 %and
+}
+
+define i32 @pos_i32_index_plus1(i32 %x, i32 %i) {
+; CHECK-LABEL: @pos_i32_index_plus1(
+; CHECK-NEXT: [[J:%.*]] = add nsw i32 [[I:%.*]], 1
+; CHECK-NEXT: [[TMP1:%.*]] = shl i32 1, [[J]]
+; CHECK-NEXT: [[TMP2:%.*]] = sub i32 [[TMP1]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[TMP2]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %j = add nsw i32 %i, 1
+ %s = sext i32 %j to i64
+ %p = getelementptr inbounds [32 x i32], ptr @masks32, i64 0, i64 %s
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+; The transform rewrites the load itself; it is not limited to `and` contexts.
+define i32 @pos_standalone(i64 %i) {
+; CHECK-LABEL: @pos_standalone(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i32
+; CHECK-NEXT: [[TMP2:%.*]] = shl i32 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i32 [[TMP2]], 1
+; CHECK-NEXT: ret i32 [[TMP3]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ ret i32 %m
+}
+
+define i8 @pos_i8(i8 %x, i64 %i) {
+; CHECK-LABEL: @pos_i8(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = shl i8 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i8 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i8 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i8 [[AND]]
+;
+ %p = getelementptr inbounds [8 x i8], ptr @masks8, i64 0, i64 %i
+ %m = load i8, ptr %p, align 1
+ %and = and i8 %m, %x
+ ret i8 %and
+}
+
+define i16 @pos_i16(i16 %x, i64 %i) {
+; CHECK-LABEL: @pos_i16(
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[I:%.*]] to i16
+; CHECK-NEXT: [[TMP2:%.*]] = shl i16 1, [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = sub i16 [[TMP2]], 1
+; CHECK-NEXT: [[AND:%.*]] = and i16 [[TMP3]], [[X:%.*]]
+; CHECK-NEXT: ret i16 [[AND]]
+;
+ %p = getelementptr inbounds [16 x i16], ptr @masks16, i64 0, i64 %i
+ %m = load i16, ptr %p, align 2
+ %and = and i16 %m, %x
+ ret i16 %and
+}
+
+define i32 @neg_bad_element(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_bad_element(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [32 x i32], ptr @masks32_bad, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32_bad, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i32 @neg_mutable(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_mutable(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [32 x i32], ptr @masks32_mut, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32_mut, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i32 @neg_weak(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_weak(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [32 x i32], ptr @masks32_weak, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32_weak, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i32 @neg_bitwidth_plus1(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_bitwidth_plus1(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [33 x i32], ptr @masks32_33, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [33 x i32], ptr @masks32_33, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+; Wide integers are rejected: a variable i128 shift would expand to a libcall.
+define i128 @neg_i128(i128 %x, i64 %i) {
+; CHECK-LABEL: @neg_i128(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [4 x i128], ptr @masks128, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i128, ptr [[P]], align 16
+; CHECK-NEXT: [[AND:%.*]] = and i128 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i128 [[AND]]
+;
+ %p = getelementptr inbounds [4 x i128], ptr @masks128, i64 0, i64 %i
+ %m = load i128, ptr %p, align 16
+ %and = and i128 %m, %x
+ ret i128 %and
+}
+
+define i32 @neg_wrong_scale(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_wrong_scale(
+; CHECK-NEXT: [[OFF:%.*]] = shl i64 [[I:%.*]], 3
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds i8, ptr @masks32, i64 [[OFF]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %off = shl i64 %i, 3
+ %p = getelementptr inbounds i8, ptr @masks32, i64 %off
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+define i32 @neg_extload(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_extload(
+; CHECK-NEXT: [[OFF:%.*]] = shl i64 [[I:%.*]], 2
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds i8, ptr @masks32, i64 [[OFF]]
+; CHECK-NEXT: [[M:%.*]] = load i16, ptr [[P]], align 4
+; CHECK-NEXT: [[Z:%.*]] = zext i16 [[M]] to i32
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[Z]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %off = shl i64 %i, 2
+ %p = getelementptr inbounds i8, ptr @masks32, i64 %off
+ %m = load i16, ptr %p, align 4
+ %z = zext i16 %m to i32
+ %and = and i32 %z, %x
+ ret i32 %and
+}
+
+define i32 @neg_volatile(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_volatile(
+; CHECK-NEXT: [[P:%.*]] = getelementptr inbounds [32 x i32], ptr @masks32, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load volatile i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr inbounds [32 x i32], ptr @masks32, i64 0, i64 %i
+ %m = load volatile i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+; No GEP flags at all: rejected by the shared table matcher.
+define i32 @neg_no_nusw(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_no_nusw(
+; CHECK-NEXT: [[P:%.*]] = getelementptr [32 x i32], ptr @masks32, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr [32 x i32], ptr @masks32, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
+
+; nusw without inbounds: the address cannot wrap, but nothing keeps it inside
+; the table, so %i may be >= 32 and the load is not (1 << %i) - 1.
+define i32 @neg_nusw_only(i32 %x, i64 %i) {
+; CHECK-LABEL: @neg_nusw_only(
+; CHECK-NEXT: [[P:%.*]] = getelementptr nusw [32 x i32], ptr @masks32, i64 0, i64 [[I:%.*]]
+; CHECK-NEXT: [[M:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT: [[AND:%.*]] = and i32 [[M]], [[X:%.*]]
+; CHECK-NEXT: ret i32 [[AND]]
+;
+ %p = getelementptr nusw [32 x i32], ptr @masks32, i64 0, i64 %i
+ %m = load i32, ptr %p, align 4
+ %and = and i32 %m, %x
+ ret i32 %and
+}
More information about the llvm-commits
mailing list