[llvm] cfc9216 - [InstCombine] Optimize GEP comparisons with constant offsets (#208547)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 22 00:56:58 PDT 2026
Author: Ömer Sinan Ağacan
Date: 2026-07-22T08:56:53+01:00
New Revision: cfc921621a05b94be5742796fdb2ba5a58979e95
URL: https://github.com/llvm/llvm-project/commit/cfc921621a05b94be5742796fdb2ba5a58979e95
DIFF: https://github.com/llvm/llvm-project/commit/cfc921621a05b94be5742796fdb2ba5a58979e95.diff
LOG: [InstCombine] Optimize GEP comparisons with constant offsets (#208547)
https://alive2.llvm.org/ce/z/dCmoVn
In a GEP comparison with the same base like
%1 = gep i8, @base, i64 a
%2 = gep i8, @base, i64 b
%cmp = icmp ... %1, %2
When we know that the offsets cross the base's alignment boundary the
same
number of times, it means that either both of them will overflow, or
none of
them will. In these cases we can turn the comparison into offset
comparison:
%cmp = icmp ... a, b
Added:
Modified:
llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
llvm/test/Transforms/InstCombine/icmp-gep.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
index 42c2983034e22..094d53363082e 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCompares.cpp
@@ -815,14 +815,35 @@ Instruction *InstCombinerImpl::foldGEPICmp(GEPOperator *GEPLHS, Value *RHS,
}
}
- if (Base.Ptr && CanFold(Base.LHSNW & Base.RHSNW) && !Base.isExpensive()) {
+ if (Base.Ptr && !Base.isExpensive()) {
// ((gep Ptr, OFFSET1) cmp (gep Ptr, OFFSET2) ---> (OFFSET1 cmp OFFSET2)
- Type *IdxTy = DL.getIndexType(GEPLHS->getType());
- Value *L =
- EmitGEPOffsets(Base.LHSGEPs, Base.LHSNW, IdxTy, /*RewriteGEP=*/true);
- Value *R =
- EmitGEPOffsets(Base.RHSGEPs, Base.RHSNW, IdxTy, /*RewriteGEP=*/true);
- return NewICmp(Base.LHSNW & Base.RHSNW, L, R);
+ bool DoFold = CanFold(Base.LHSNW & Base.RHSNW);
+
+ if (!DoFold && Base.Ptr->getType()->isPointerTy()) {
+ // Without the flags, we can still fold if the offsets are constant and
+ // they cross the base's alignment boundary the same number of times, so
+ // either both arguments will wrap, or none of them will.
+ unsigned BW = DL.getIndexTypeSizeInBits(GEPLHS->getType());
+ APInt Alignment = APInt(BW, Base.Ptr->getPointerAlignment(DL).value());
+ APInt LOff(BW, 0);
+ APInt ROff(BW, 0);
+ if (GEPLHS->stripAndAccumulateConstantOffsets(
+ DL, LOff, /*AllowNonInbounds=*/true) == Base.Ptr &&
+ RHS->stripAndAccumulateConstantOffsets(
+ DL, ROff, /*AllowNonInbounds=*/true) == Base.Ptr)
+ DoFold =
+ APIntOps::RoundingSDiv(LOff, Alignment, APInt::Rounding::DOWN) ==
+ APIntOps::RoundingSDiv(ROff, Alignment, APInt::Rounding::DOWN);
+ }
+
+ if (DoFold) {
+ Type *IdxTy = DL.getIndexType(GEPLHS->getType());
+ Value *L = EmitGEPOffsets(Base.LHSGEPs, Base.LHSNW, IdxTy,
+ /*RewriteGEP=*/true);
+ Value *R = EmitGEPOffsets(Base.RHSGEPs, Base.RHSNW, IdxTy,
+ /*RewriteGEP=*/true);
+ return NewICmp(Base.LHSNW & Base.RHSNW, L, R);
+ }
}
}
diff --git a/llvm/test/Transforms/InstCombine/icmp-gep.ll b/llvm/test/Transforms/InstCombine/icmp-gep.ll
index 048a4c4a7e5fe..abdd354a60b10 100644
--- a/llvm/test/Transforms/InstCombine/icmp-gep.ll
+++ b/llvm/test/Transforms/InstCombine/icmp-gep.ll
@@ -1124,3 +1124,143 @@ define i1 @gep_gep_multiple_ult_nuw_multi_use(ptr %base, i64 %idx1, i64 %idx2, i
%cmp = icmp ult ptr %gep2, %gep4
ret i1 %cmp
}
+
+define i1 @gep_const_same_block(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_same_block(
+; CHECK-NEXT: ret i1 true
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 4
+ %gep2 = getelementptr i8, ptr %foo, i64 8
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_same_negative_block(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_same_negative_block(
+; CHECK-NEXT: ret i1 true
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 -12
+ %gep2 = getelementptr i8, ptr %foo, i64 -8
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_
diff erent_block_1(ptr align 4 %foo) {
+; CHECK-LABEL: @gep_const_
diff erent_block_1(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[FOO:%.*]], i64 4
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, ptr [[FOO]], i64 8
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult ptr [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 4
+ %gep2 = getelementptr i8, ptr %foo, i64 8
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_
diff erent_block_2(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_
diff erent_block_2(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[FOO:%.*]], i64 -4
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, ptr [[FOO]], i64 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult ptr [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 -4
+ %gep2 = getelementptr i8, ptr %foo, i64 4
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_edge_case_1(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_edge_case_1(
+; CHECK-NEXT: ret i1 true
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 14
+ %gep2 = getelementptr i8, ptr %foo, i64 15
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_edge_case_2(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_edge_case_2(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[FOO:%.*]], i64 15
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, ptr [[FOO]], i64 16
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult ptr [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 15
+ %gep2 = getelementptr i8, ptr %foo, i64 16
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_edge_case_3(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_edge_case_3(
+; CHECK-NEXT: ret i1 true
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 16
+ %gep2 = getelementptr i8, ptr %foo, i64 17
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_edge_case_4(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_edge_case_4(
+; CHECK-NEXT: ret i1 false
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 -15
+ %gep2 = getelementptr i8, ptr %foo, i64 -16
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_const_edge_case_5(ptr align 16 %foo) {
+; CHECK-LABEL: @gep_const_edge_case_5(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[FOO:%.*]], i64 -16
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, ptr [[FOO]], i64 -17
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult ptr [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 -16
+ %gep2 = getelementptr i8, ptr %foo, i64 -17
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+define i1 @gep_variable_offsets(ptr align 16 %foo, i64 %i, i64 %j) {
+; CHECK-LABEL: @gep_variable_offsets(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, ptr [[FOO:%.*]], i64 [[I:%.*]]
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, ptr [[FOO]], i64 [[J:%.*]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult ptr [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %gep1 = getelementptr i8, ptr %foo, i64 %i
+ %gep2 = getelementptr i8, ptr %foo, i64 %j
+ %cmp = icmp ult ptr %gep1, %gep2
+ ret i1 %cmp
+}
+
+; Similar to tests above, but extracted from an actual program.
+ at g16 = global [4 x i32] zeroinitializer, align 16
+define i1 @gep_global_offsets() {
+; CHECK-LABEL: @gep_global_offsets(
+; CHECK-NEXT: ret i1 false
+;
+ %cmp = icmp ule ptr getelementptr (i8, ptr @g16, i64 20), getelementptr inbounds nuw (i8, ptr @g16, i64 16)
+ ret i1 %cmp
+}
+
+; Regression test: when checking for folding opportunities check for pointer
+; base before using trying to get the pointer alignment.
+define <2 x i1> @vec_gep_cmp(<2 x ptr> %base, <2 x i64> %i, <2 x i64> %j) {
+; CHECK-LABEL: @vec_gep_cmp(
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr i8, <2 x ptr> [[BASE:%.*]], <2 x i64> [[I:%.*]]
+; CHECK-NEXT: [[GEP2:%.*]] = getelementptr i8, <2 x ptr> [[BASE]], <2 x i64> [[J:%.*]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ult <2 x ptr> [[GEP1]], [[GEP2]]
+; CHECK-NEXT: ret <2 x i1> [[CMP]]
+;
+ %gep1 = getelementptr i8, <2 x ptr> %base, <2 x i64> %i
+ %gep2 = getelementptr i8, <2 x ptr> %base, <2 x i64> %j
+ %cmp = icmp ult <2 x ptr> %gep1, %gep2
+ ret <2 x i1> %cmp
+}
More information about the llvm-commits
mailing list