[llvm] [SeparateConstOffsetFromGEP] Stop distributing sext/zext over lossy trunc (PR #221381)
Fujun Han via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 04:37:57 PDT 2026
https://github.com/Peter9606 updated https://github.com/llvm/llvm-project/pull/221381
>From 764a49b17c459e7cbde665bb83d1fcd732bf7c1f Mon Sep 17 00:00:00 2001
From: Fujun Han <fujun.han at iluvatar.com>
Date: Sat, 5 Sep 2026 10:06:57 +0800
Subject: [PATCH 1/2] [SeparateConstOffsetFromGEP] Precommit tests for
sext/zext of trunc (NFC)
The nuw/nsw flags on an add hold at the width of the add, which says
nothing about wrapping at the width of a truncation applied to its
result. These tests capture the current behavior of distributing a
sext/zext above a truncation into the operands of the add; the lossy
cases are miscompiled today.
@sext_of_lossy_operand is the review counter-example: the sum survives
the truncation, but the remaining operand does not, so hoisting the
constant is still wrong.
Assisted-by: Cursor (Claude Fable 5)
Signed-off-by: Fujun Han <fujun.han at iluvatar.com>
Co-authored-by: Cursor <cursoragent at cursor.com>
---
.../ext-of-trunc-add-wrap.ll | 154 ++++++++++++++++++
1 file changed, 154 insertions(+)
create mode 100644 llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
new file mode 100644
index 0000000000000..242e50022f035
--- /dev/null
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
@@ -0,0 +1,154 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes='separate-const-offset-from-gep<lower-gep>' < %s | FileCheck %s
+
+; The nuw/nsw flags on an add hold at the width of the add, which says nothing
+; about wrapping at the width of a truncation applied to its result. A
+; sext/zext above such a truncation must therefore not be distributed into the
+; operands of the add.
+
+; zext i64 (trunc i8 (add nuw i32 A, B)) wraps modulo 256, so the constant 1
+; must stay inside the truncation. For A = 251 and B = 5 the index is 0, while
+; distributing the casts would give 251 + 5 = 256.
+define ptr @zext_of_lossy_trunc(ptr %p, i8 %a, i64 %iv) {
+; CHECK-LABEL: define ptr @zext_of_lossy_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i8 [[A:%.*]], i64 [[IV:%.*]]) {
+; CHECK-NEXT: [[AZ:%.*]] = zext i8 [[A]] to i32
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[AZ]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
+; CHECK-NEXT: [[TMP3:%.*]] = trunc i64 [[IV]] to i8
+; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i32
+; CHECK-NEXT: [[TMP5:%.*]] = trunc i32 [[TMP4]] to i8
+; CHECK-NEXT: [[TMP6:%.*]] = zext nneg i8 [[TMP5]] to i64
+; CHECK-NEXT: [[SUM2:%.*]] = add i64 [[TMP2]], [[TMP6]]
+; CHECK-NEXT: [[TMP7:%.*]] = shl i64 [[SUM2]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP7]]
+; CHECK-NEXT: [[UGLYGEP3:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT: ret ptr [[UGLYGEP3]]
+;
+ %iv.next = add nuw nsw i64 %iv, 1
+ %t = trunc i64 %iv.next to i8
+ %az = zext i8 %a to i32
+ %tz = zext i8 %t to i32
+ %sum = add nuw nsw i32 %az, %tz
+ %sum8 = trunc i32 %sum to i8
+ %idx = zext nneg i8 %sum8 to i64
+ %q = getelementptr inbounds i32, ptr %p, i64 %idx
+ ret ptr %q
+}
+
+; sext wraps the same way: the sum reaches 130, so sext i8 of the truncation
+; is negative and the constant must stay inside the truncation.
+define ptr @sext_of_lossy_trunc(ptr %p, i32 %x) {
+; CHECK-LABEL: define ptr @sext_of_lossy_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
+; CHECK-NEXT: [[A:%.*]] = and i32 [[X]], 127
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
+; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT: ret ptr [[UGLYGEP2]]
+;
+ %a = and i32 %x, 127
+ %sum = add nuw nsw i32 %a, 3
+ %sum8 = trunc i32 %sum to i8
+ %idx = sext i8 %sum8 to i64
+ %q = getelementptr i32, ptr %p, i64 %idx
+ ret ptr %q
+}
+
+; Even a proof that the trunc of the sum is lossless would not be enough:
+; here %sum is 124 or 127, so trunc+sext reproduces it, but the remaining
+; operand %a is 224 or 227 and wraps at i8. Distributing the sext would
+; compute sext(trunc(%a)) - 100 = -132 instead of 124. Counter-example from
+; review of llvm#221381.
+define ptr @sext_of_lossy_operand(ptr %p, i1 %c) {
+; CHECK-LABEL: define ptr @sext_of_lossy_operand(
+; CHECK-SAME: ptr [[P:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT: [[A:%.*]] = select i1 [[C]], i32 224, i32 227
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
+; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -400
+; CHECK-NEXT: ret ptr [[UGLYGEP2]]
+;
+ %a = select i1 %c, i32 224, i32 227
+ %sum = add nsw i32 %a, -100
+ %sum8 = trunc i32 %sum to i8
+ %idx = sext i8 %sum8 to i64
+ %q = getelementptr i32, ptr %p, i64 %idx
+ ret ptr %q
+}
+
+; Arithmetically this truncation loses nothing (the sum stays below 256), so
+; the constant could be hoisted, but the fix intentionally blocks all
+; non-constant truncated values to keep the guard simple. Recovering such
+; cases is left for a follow-up.
+define ptr @zext_of_lossless_trunc(ptr %p, i32 %x) {
+; CHECK-LABEL: define ptr @zext_of_lossless_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
+; CHECK-NEXT: [[A:%.*]] = and i32 [[X]], 15
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
+; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT: ret ptr [[UGLYGEP2]]
+;
+ %a = and i32 %x, 15
+ %sum = add nuw nsw i32 %a, 3
+ %sum8 = trunc i32 %sum to i8
+ %idx = zext nneg i8 %sum8 to i64
+ %q = getelementptr inbounds i32, ptr %p, i64 %idx
+ ret ptr %q
+}
+
+; A truncated constant has no variable remainder; the pending extension
+; applies to the constant itself, so extracting it is a plain constant fold.
+; 130 wraps to -126 at i8 and the offset must use the wrapped value.
+define ptr @sext_of_trunc_of_constant(ptr %p, i64 %i) {
+; CHECK-LABEL: define ptr @sext_of_trunc_of_constant(
+; CHECK-SAME: ptr [[P:%.*]], i64 [[I:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[I]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP1]]
+; CHECK-NEXT: [[UGLYGEP1:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -50400
+; CHECK-NEXT: ret ptr [[UGLYGEP1]]
+;
+ %t = trunc i64 130 to i8
+ %idx = sext i8 %t to i64
+ %q = getelementptr [100 x i32], ptr %p, i64 %idx, i64 %i
+ ret ptr %q
+}
+
+; Same with zext: trunc i8 130 is 0x82, which zext reads back as 130.
+define ptr @zext_of_trunc_of_constant(ptr %p, i64 %i) {
+; CHECK-LABEL: define ptr @zext_of_trunc_of_constant(
+; CHECK-SAME: ptr [[P:%.*]], i64 [[I:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[I]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP1]]
+; CHECK-NEXT: [[UGLYGEP1:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 52000
+; CHECK-NEXT: ret ptr [[UGLYGEP1]]
+;
+ %t = trunc i64 130 to i8
+ %idx = zext i8 %t to i64
+ %q = getelementptr [100 x i32], ptr %p, i64 %idx, i64 %i
+ ret ptr %q
+}
+
+; A truncation with no extension above it still distributes over the add,
+; because truncation is exact in modular arithmetic.
+define ptr @bare_trunc(ptr %p, i128 %i) {
+; CHECK-LABEL: define ptr @bare_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i128 [[I:%.*]]) {
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i128 [[I]] to i64
+; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[TMP1]], 2
+; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP2]]
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT: ret ptr [[UGLYGEP2]]
+;
+ %idx = add i128 %i, 1
+ %idx.conv = trunc i128 %idx to i64
+ %q = getelementptr i32, ptr %p, i64 %idx.conv
+ ret ptr %q
+}
>From 7403de275b34f03f450226f32491f09feb8d6539 Mon Sep 17 00:00:00 2001
From: Fujun Han <fujun.han at iluvatar.com>
Date: Wed, 9 Sep 2026 19:37:30 +0800
Subject: [PATCH 2/2] [SeparateConstOffsetFromGEP] Stop distributing sext/zext
over lossy trunc
find() traces through a trunc while keeping a pending sext/zext, and
canTraceInto() then justifies distributing the extension into the
operands of an add/sub by checking its nuw/nsw flags. Those flags
hold at the width of the add but say nothing about wrapping at the
truncation width, so
zext i64 (trunc i8 (add nuw i32 (zext i8 251), (zext i8 5)))
which is 0 was rebuilt as 251 + 5 = 256.
With a pending sext/zext, stop tracing through a trunc unless the
truncated value is entirely constant: in that case there is no
variable remainder and the pending casts apply to the constant
itself, so extracting it is a plain constant fold (this keeps e.g.
@trunk_explicit in NVPTX/split-gep.ll folding).
This intentionally also blocks cases where the truncation is provably
lossless. Known-bits reasoning could recover some of them, but a proof
for the truncated value alone is not enough: in the review
counter-example, add nsw (select i1 %c, i32 224, i32 227), -100 is 124
or 127 and truncates to i8 losslessly, yet the remaining operand
224/227 wraps at i8, so hoisting -100 would rebuild the index as
sext(trunc(%a)) - 100 = -132 instead of 124. Recovering the lossless
cases is left for a follow-up; this patch only fixes the miscompile.
Assisted-by: Cursor (Claude Fable 5)
Signed-off-by: Fujun Han <fujun.han at iluvatar.com>
Co-authored-by: Cursor <cursoragent at cursor.com>
---
.../Scalar/SeparateConstOffsetFromGEP.cpp | 19 +++++++++--
.../ext-of-trunc-add-wrap.ll | 34 ++++++++-----------
2 files changed, 30 insertions(+), 23 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 4870b8c888279..7af9e2c7e4db0 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -741,9 +741,22 @@ APInt ConstantOffsetExtractor::find(Value *V, GetElementPtrInst *GEP,
else if (BO->getOpcode() == Instruction::Xor)
ConstantOffset = extractDisjointBitsFromXor(BO);
} else if (isa<TruncInst>(V)) {
- ConstantOffset =
- find(U->getOperand(0), GEP, Idx, SignExtended, ZeroExtended)
- .trunc(BitWidth);
+ // With no pending extension, truncation distributes over add, sub and
+ // disjoint or in modular arithmetic, so any constant found in the wider
+ // operand stays valid after truncating it.
+ //
+ // With a pending sext/zext, distributing the extension into the operands
+ // of the truncated expression is unsound: the nuw/nsw flags checked by
+ // canTraceInto hold at the width of the add and say nothing about
+ // wrapping at the truncation width, e.g.
+ // zext i64 (trunc i8 (add nuw i32 (zext i8 251), (zext i8 5)))
+ // is 0 but would be rebuilt as 251 + 5 = 256. Only a fully constant
+ // truncated value remains exact, because then there is no remainder and
+ // the pending casts apply to the constant itself.
+ Value *TruncOp = U->getOperand(0);
+ if ((!SignExtended && !ZeroExtended) || isa<ConstantInt>(TruncOp))
+ ConstantOffset =
+ find(TruncOp, GEP, Idx, SignExtended, ZeroExtended).trunc(BitWidth);
} else if (isa<SExtInst>(V)) {
ConstantOffset =
find(U->getOperand(0), GEP, Idx, /* SignExtended */ true, ZeroExtended)
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
index 242e50022f035..670ce729b78f1 100644
--- a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
@@ -12,17 +12,14 @@
define ptr @zext_of_lossy_trunc(ptr %p, i8 %a, i64 %iv) {
; CHECK-LABEL: define ptr @zext_of_lossy_trunc(
; CHECK-SAME: ptr [[P:%.*]], i8 [[A:%.*]], i64 [[IV:%.*]]) {
+; CHECK-NEXT: [[IV_NEXT:%.*]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[TMP3:%.*]] = trunc i64 [[IV_NEXT]] to i8
; CHECK-NEXT: [[AZ:%.*]] = zext i8 [[A]] to i32
-; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[AZ]] to i8
-; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
-; CHECK-NEXT: [[TMP3:%.*]] = trunc i64 [[IV]] to i8
; CHECK-NEXT: [[TMP4:%.*]] = zext i8 [[TMP3]] to i32
-; CHECK-NEXT: [[TMP5:%.*]] = trunc i32 [[TMP4]] to i8
+; CHECK-NEXT: [[SUM:%.*]] = add nuw nsw i32 [[AZ]], [[TMP4]]
+; CHECK-NEXT: [[TMP5:%.*]] = trunc i32 [[SUM]] to i8
; CHECK-NEXT: [[TMP6:%.*]] = zext nneg i8 [[TMP5]] to i64
-; CHECK-NEXT: [[SUM2:%.*]] = add i64 [[TMP2]], [[TMP6]]
-; CHECK-NEXT: [[TMP7:%.*]] = shl i64 [[SUM2]], 2
-; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP7]]
-; CHECK-NEXT: [[UGLYGEP3:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT: [[UGLYGEP3:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[TMP6]]
; CHECK-NEXT: ret ptr [[UGLYGEP3]]
;
%iv.next = add nuw nsw i64 %iv, 1
@@ -42,11 +39,10 @@ define ptr @sext_of_lossy_trunc(ptr %p, i32 %x) {
; CHECK-LABEL: define ptr @sext_of_lossy_trunc(
; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
; CHECK-NEXT: [[A:%.*]] = and i32 [[X]], 127
-; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[SUM:%.*]] = add nuw nsw i32 [[A]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
; CHECK-NEXT: [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
-; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i32, ptr [[P]], i64 [[TMP2]]
; CHECK-NEXT: ret ptr [[UGLYGEP2]]
;
%a = and i32 %x, 127
@@ -66,11 +62,10 @@ define ptr @sext_of_lossy_operand(ptr %p, i1 %c) {
; CHECK-LABEL: define ptr @sext_of_lossy_operand(
; CHECK-SAME: ptr [[P:%.*]], i1 [[C:%.*]]) {
; CHECK-NEXT: [[A:%.*]] = select i1 [[C]], i32 224, i32 227
-; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[SUM:%.*]] = add nsw i32 [[A]], -100
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
; CHECK-NEXT: [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
-; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -400
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i32, ptr [[P]], i64 [[TMP2]]
; CHECK-NEXT: ret ptr [[UGLYGEP2]]
;
%a = select i1 %c, i32 224, i32 227
@@ -89,11 +84,10 @@ define ptr @zext_of_lossless_trunc(ptr %p, i32 %x) {
; CHECK-LABEL: define ptr @zext_of_lossless_trunc(
; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
; CHECK-NEXT: [[A:%.*]] = and i32 [[X]], 15
-; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT: [[SUM:%.*]] = add nuw nsw i32 [[A]], 3
+; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
; CHECK-NEXT: [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
-; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT: [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT: [[UGLYGEP2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[TMP2]]
; CHECK-NEXT: ret ptr [[UGLYGEP2]]
;
%a = and i32 %x, 15
More information about the llvm-commits
mailing list