[llvm] [SeparateConstOffsetFromGEP] Stop distributing sext/zext over lossy trunc (PR #221381)

Fujun Han via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 04:40:49 PDT 2026


https://github.com/Peter9606 updated https://github.com/llvm/llvm-project/pull/221381

>From 08ed8ac8b6abf54eade97b8572acac3e0c024448 Mon Sep 17 00:00:00 2001
From: Fujun Han <fujun.han at iluvatar.com>
Date: Wed, 9 Sep 2026 19:39:42 +0800
Subject: [PATCH 1/2] [SeparateConstOffsetFromGEP] Precommit tests for
 sext/zext of trunc (NFC)

The nuw/nsw flags on an add hold at the width of the add, which says
nothing about wrapping at the width of a truncation applied to its
result. These tests capture the current behavior of distributing a
sext/zext above a truncation into the operands of the add; the lossy
cases are miscompiled today.

@sext_of_lossy_operand is the review counter-example: the sum survives
the truncation, but the remaining operand does not, so hoisting the
constant is still wrong.

Assisted-by: Cursor (Claude Fable 5)
Signed-off-by: Fujun Han <fujun.han at iluvatar.com>
---
 .../ext-of-trunc-add-wrap.ll                  | 154 ++++++++++++++++++
 1 file changed, 154 insertions(+)
 create mode 100644 llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll

diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
new file mode 100644
index 0000000000000..242e50022f035
--- /dev/null
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
@@ -0,0 +1,154 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes='separate-const-offset-from-gep<lower-gep>' < %s | FileCheck %s
+
+; The nuw/nsw flags on an add hold at the width of the add, which says nothing
+; about wrapping at the width of a truncation applied to its result.  A
+; sext/zext above such a truncation must therefore not be distributed into the
+; operands of the add.
+
+; zext i64 (trunc i8 (add nuw i32 A, B)) wraps modulo 256, so the constant 1
+; must stay inside the truncation.  For A = 251 and B = 5 the index is 0, while
+; distributing the casts would give 251 + 5 = 256.
+define ptr @zext_of_lossy_trunc(ptr %p, i8 %a, i64 %iv) {
+; CHECK-LABEL: define ptr @zext_of_lossy_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i8 [[A:%.*]], i64 [[IV:%.*]]) {
+; CHECK-NEXT:    [[AZ:%.*]] = zext i8 [[A]] to i32
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[AZ]] to i8
+; CHECK-NEXT:    [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc i64 [[IV]] to i8
+; CHECK-NEXT:    [[TMP4:%.*]] = zext i8 [[TMP3]] to i32
+; CHECK-NEXT:    [[TMP5:%.*]] = trunc i32 [[TMP4]] to i8
+; CHECK-NEXT:    [[TMP6:%.*]] = zext nneg i8 [[TMP5]] to i64
+; CHECK-NEXT:    [[SUM2:%.*]] = add i64 [[TMP2]], [[TMP6]]
+; CHECK-NEXT:    [[TMP7:%.*]] = shl i64 [[SUM2]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP7]]
+; CHECK-NEXT:    [[UGLYGEP3:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT:    ret ptr [[UGLYGEP3]]
+;
+  %iv.next = add nuw nsw i64 %iv, 1
+  %t = trunc i64 %iv.next to i8
+  %az = zext i8 %a to i32
+  %tz = zext i8 %t to i32
+  %sum = add nuw nsw i32 %az, %tz
+  %sum8 = trunc i32 %sum to i8
+  %idx = zext nneg i8 %sum8 to i64
+  %q = getelementptr inbounds i32, ptr %p, i64 %idx
+  ret ptr %q
+}
+
+; sext wraps the same way: the sum reaches 130, so sext i8 of the truncation
+; is negative and the constant must stay inside the truncation.
+define ptr @sext_of_lossy_trunc(ptr %p, i32 %x) {
+; CHECK-LABEL: define ptr @sext_of_lossy_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[A:%.*]] = and i32 [[X]], 127
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
+;
+  %a = and i32 %x, 127
+  %sum = add nuw nsw i32 %a, 3
+  %sum8 = trunc i32 %sum to i8
+  %idx = sext i8 %sum8 to i64
+  %q = getelementptr i32, ptr %p, i64 %idx
+  ret ptr %q
+}
+
+; Even a proof that the trunc of the sum is lossless would not be enough:
+; here %sum is 124 or 127, so trunc+sext reproduces it, but the remaining
+; operand %a is 224 or 227 and wraps at i8.  Distributing the sext would
+; compute sext(trunc(%a)) - 100 = -132 instead of 124.  Counter-example from
+; review of llvm#221381.
+define ptr @sext_of_lossy_operand(ptr %p, i1 %c) {
+; CHECK-LABEL: define ptr @sext_of_lossy_operand(
+; CHECK-SAME: ptr [[P:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:    [[A:%.*]] = select i1 [[C]], i32 224, i32 227
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -400
+; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
+;
+  %a = select i1 %c, i32 224, i32 227
+  %sum = add nsw i32 %a, -100
+  %sum8 = trunc i32 %sum to i8
+  %idx = sext i8 %sum8 to i64
+  %q = getelementptr i32, ptr %p, i64 %idx
+  ret ptr %q
+}
+
+; Arithmetically this truncation loses nothing (the sum stays below 256), so
+; the constant could be hoisted, but the fix intentionally blocks all
+; non-constant truncated values to keep the guard simple.  Recovering such
+; cases is left for a follow-up.
+define ptr @zext_of_lossless_trunc(ptr %p, i32 %x) {
+; CHECK-LABEL: define ptr @zext_of_lossless_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[A:%.*]] = and i32 [[X]], 15
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
+;
+  %a = and i32 %x, 15
+  %sum = add nuw nsw i32 %a, 3
+  %sum8 = trunc i32 %sum to i8
+  %idx = zext nneg i8 %sum8 to i64
+  %q = getelementptr inbounds i32, ptr %p, i64 %idx
+  ret ptr %q
+}
+
+; A truncated constant has no variable remainder; the pending extension
+; applies to the constant itself, so extracting it is a plain constant fold.
+; 130 wraps to -126 at i8 and the offset must use the wrapped value.
+define ptr @sext_of_trunc_of_constant(ptr %p, i64 %i) {
+; CHECK-LABEL: define ptr @sext_of_trunc_of_constant(
+; CHECK-SAME: ptr [[P:%.*]], i64 [[I:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[I]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP1]]
+; CHECK-NEXT:    [[UGLYGEP1:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -50400
+; CHECK-NEXT:    ret ptr [[UGLYGEP1]]
+;
+  %t = trunc i64 130 to i8
+  %idx = sext i8 %t to i64
+  %q = getelementptr [100 x i32], ptr %p, i64 %idx, i64 %i
+  ret ptr %q
+}
+
+; Same with zext: trunc i8 130 is 0x82, which zext reads back as 130.
+define ptr @zext_of_trunc_of_constant(ptr %p, i64 %i) {
+; CHECK-LABEL: define ptr @zext_of_trunc_of_constant(
+; CHECK-SAME: ptr [[P:%.*]], i64 [[I:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[I]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP1]]
+; CHECK-NEXT:    [[UGLYGEP1:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 52000
+; CHECK-NEXT:    ret ptr [[UGLYGEP1]]
+;
+  %t = trunc i64 130 to i8
+  %idx = zext i8 %t to i64
+  %q = getelementptr [100 x i32], ptr %p, i64 %idx, i64 %i
+  ret ptr %q
+}
+
+; A truncation with no extension above it still distributes over the add,
+; because truncation is exact in modular arithmetic.
+define ptr @bare_trunc(ptr %p, i128 %i) {
+; CHECK-LABEL: define ptr @bare_trunc(
+; CHECK-SAME: ptr [[P:%.*]], i128 [[I:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i128 [[I]] to i64
+; CHECK-NEXT:    [[TMP2:%.*]] = shl i64 [[TMP1]], 2
+; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP2]]
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
+;
+  %idx = add i128 %i, 1
+  %idx.conv = trunc i128 %idx to i64
+  %q = getelementptr i32, ptr %p, i64 %idx.conv
+  ret ptr %q
+}

>From a3c2b02dd1c1ebccc2e6bcc60f4b0b7a052a5b02 Mon Sep 17 00:00:00 2001
From: Fujun Han <fujun.han at iluvatar.com>
Date: Wed, 9 Sep 2026 19:39:57 +0800
Subject: [PATCH 2/2] [SeparateConstOffsetFromGEP] Do not distribute sext/zext
 over lossy trunc

find() distributes a sext/zext above a truncation into the operands of
a truncated add/sub, relying on the nuw/nsw flags of that add/sub.  The
flags hold at the width of the operation, which says nothing about
wrapping at the truncation width, so the rebuilt expression can differ
from the original, e.g.

  zext i64 (trunc i8 (add nuw i32 (zext i8 251), (zext i8 5)))

is 0 but was rebuilt as 251 + 5 = 256.

Fix this conservatively: with a pending sext/zext, stop tracing at a
trunc unless the truncated value is itself a constant (then there is no
remainder and extracting the offset is a plain constant fold, which
keeps the existing NVPTX trunk_explicit test working).  Bare truncs
without a pending extension are unaffected; truncation distributes over
add/sub/or in modular arithmetic.

Cases where the truncation is provably lossless for both the sum and
the remainder could still be traced; that refinement is left for a
follow-up.

Fixes a miscompile reported in llvm#221381 review.

Assisted-by: Cursor (Claude Fable 5)
Signed-off-by: Fujun Han <fujun.han at iluvatar.com>
---
 .../Scalar/SeparateConstOffsetFromGEP.cpp     | 19 +++++++++--
 .../ext-of-trunc-add-wrap.ll                  | 34 ++++++++-----------
 2 files changed, 30 insertions(+), 23 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
index 4870b8c888279..7af9e2c7e4db0 100644
--- a/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
+++ b/llvm/lib/Transforms/Scalar/SeparateConstOffsetFromGEP.cpp
@@ -741,9 +741,22 @@ APInt ConstantOffsetExtractor::find(Value *V, GetElementPtrInst *GEP,
     else if (BO->getOpcode() == Instruction::Xor)
       ConstantOffset = extractDisjointBitsFromXor(BO);
   } else if (isa<TruncInst>(V)) {
-    ConstantOffset =
-        find(U->getOperand(0), GEP, Idx, SignExtended, ZeroExtended)
-            .trunc(BitWidth);
+    // With no pending extension, truncation distributes over add, sub and
+    // disjoint or in modular arithmetic, so any constant found in the wider
+    // operand stays valid after truncating it.
+    //
+    // With a pending sext/zext, distributing the extension into the operands
+    // of the truncated expression is unsound: the nuw/nsw flags checked by
+    // canTraceInto hold at the width of the add and say nothing about
+    // wrapping at the truncation width, e.g.
+    //   zext i64 (trunc i8 (add nuw i32 (zext i8 251), (zext i8 5)))
+    // is 0 but would be rebuilt as 251 + 5 = 256.  Only a fully constant
+    // truncated value remains exact, because then there is no remainder and
+    // the pending casts apply to the constant itself.
+    Value *TruncOp = U->getOperand(0);
+    if ((!SignExtended && !ZeroExtended) || isa<ConstantInt>(TruncOp))
+      ConstantOffset =
+          find(TruncOp, GEP, Idx, SignExtended, ZeroExtended).trunc(BitWidth);
   } else if (isa<SExtInst>(V)) {
     ConstantOffset =
         find(U->getOperand(0), GEP, Idx, /* SignExtended */ true, ZeroExtended)
diff --git a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
index 242e50022f035..670ce729b78f1 100644
--- a/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
+++ b/llvm/test/Transforms/SeparateConstOffsetFromGEP/ext-of-trunc-add-wrap.ll
@@ -12,17 +12,14 @@
 define ptr @zext_of_lossy_trunc(ptr %p, i8 %a, i64 %iv) {
 ; CHECK-LABEL: define ptr @zext_of_lossy_trunc(
 ; CHECK-SAME: ptr [[P:%.*]], i8 [[A:%.*]], i64 [[IV:%.*]]) {
+; CHECK-NEXT:    [[IV_NEXT:%.*]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc i64 [[IV_NEXT]] to i8
 ; CHECK-NEXT:    [[AZ:%.*]] = zext i8 [[A]] to i32
-; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[AZ]] to i8
-; CHECK-NEXT:    [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
-; CHECK-NEXT:    [[TMP3:%.*]] = trunc i64 [[IV]] to i8
 ; CHECK-NEXT:    [[TMP4:%.*]] = zext i8 [[TMP3]] to i32
-; CHECK-NEXT:    [[TMP5:%.*]] = trunc i32 [[TMP4]] to i8
+; CHECK-NEXT:    [[SUM:%.*]] = add nuw nsw i32 [[AZ]], [[TMP4]]
+; CHECK-NEXT:    [[TMP5:%.*]] = trunc i32 [[SUM]] to i8
 ; CHECK-NEXT:    [[TMP6:%.*]] = zext nneg i8 [[TMP5]] to i64
-; CHECK-NEXT:    [[SUM2:%.*]] = add i64 [[TMP2]], [[TMP6]]
-; CHECK-NEXT:    [[TMP7:%.*]] = shl i64 [[SUM2]], 2
-; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP7]]
-; CHECK-NEXT:    [[UGLYGEP3:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 4
+; CHECK-NEXT:    [[UGLYGEP3:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[TMP6]]
 ; CHECK-NEXT:    ret ptr [[UGLYGEP3]]
 ;
   %iv.next = add nuw nsw i64 %iv, 1
@@ -42,11 +39,10 @@ define ptr @sext_of_lossy_trunc(ptr %p, i32 %x) {
 ; CHECK-LABEL: define ptr @sext_of_lossy_trunc(
 ; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
 ; CHECK-NEXT:    [[A:%.*]] = and i32 [[X]], 127
-; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[SUM:%.*]] = add nuw nsw i32 [[A]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
 ; CHECK-NEXT:    [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i32, ptr [[P]], i64 [[TMP2]]
 ; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
 ;
   %a = and i32 %x, 127
@@ -66,11 +62,10 @@ define ptr @sext_of_lossy_operand(ptr %p, i1 %c) {
 ; CHECK-LABEL: define ptr @sext_of_lossy_operand(
 ; CHECK-SAME: ptr [[P:%.*]], i1 [[C:%.*]]) {
 ; CHECK-NEXT:    [[A:%.*]] = select i1 [[C]], i32 224, i32 227
-; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[SUM:%.*]] = add nsw i32 [[A]], -100
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
 ; CHECK-NEXT:    [[TMP2:%.*]] = sext i8 [[TMP1]] to i64
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 -400
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i32, ptr [[P]], i64 [[TMP2]]
 ; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
 ;
   %a = select i1 %c, i32 224, i32 227
@@ -89,11 +84,10 @@ define ptr @zext_of_lossless_trunc(ptr %p, i32 %x) {
 ; CHECK-LABEL: define ptr @zext_of_lossless_trunc(
 ; CHECK-SAME: ptr [[P:%.*]], i32 [[X:%.*]]) {
 ; CHECK-NEXT:    [[A:%.*]] = and i32 [[X]], 15
-; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[A]] to i8
+; CHECK-NEXT:    [[SUM:%.*]] = add nuw nsw i32 [[A]], 3
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc i32 [[SUM]] to i8
 ; CHECK-NEXT:    [[TMP2:%.*]] = zext nneg i8 [[TMP1]] to i64
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[TMP2]], 2
-; CHECK-NEXT:    [[UGLYGEP:%.*]] = getelementptr i8, ptr [[P]], i64 [[TMP3]]
-; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr i8, ptr [[UGLYGEP]], i64 12
+; CHECK-NEXT:    [[UGLYGEP2:%.*]] = getelementptr inbounds i32, ptr [[P]], i64 [[TMP2]]
 ; CHECK-NEXT:    ret ptr [[UGLYGEP2]]
 ;
   %a = and i32 %x, 15



More information about the llvm-commits mailing list