[llvm] [AggressiveInstCombine] Narrow outside-graph icmp users of TruncInstCombine (PR #204920)

Thorbjørn Ravn Andersen via llvm-commits llvm-commits at lists.llvm.org
Sat Jun 20 01:41:47 PDT 2026


https://github.com/ravn updated https://github.com/llvm/llvm-project/pull/204920

>From bf42375b5f96604f8228edf0e2eb7f3195552d3a Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Thorbj=C3=B8rn=20Ravn=20Andersen?= <tra at ravnand.dk>
Date: Sat, 20 Jun 2026 09:46:29 +0200
Subject: [PATCH] [AggressiveInstCombine] Narrow outside-graph icmp users of
 TruncInstCombine
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

TruncInstCombine's outside-user check in getBestTruncatedType bailed
on any in-graph value with a user that isn't in the expression graph
and isn't a ZExt/SExt — even when that user is an icmp whose operands
are provably narrow under KnownBits. The conservative bail forfeited
the whole rewrite, which on small-int targets (AVR, Z80, MSP430) keeps
loop bodies in expensive 16-bit arithmetic when the loop-exit `==`/`!=`
comparison could narrow alongside the rest.

Admit outside-graph ICmpInst users with equality or unsigned predicates
when KnownBits proves both operands fit in MinBitWidth. The full-value
fit check is load-bearing: a value can have its low MinBitWidth bits
match the graph's narrow rewrite while its full value exceeds
MinBitWidth, in which case the narrowed icmp would observe a different
operand value and could change the comparison result.

The check is deferred from the outside-user loop (which runs before
MinBitWidth is known) to a small validation pass after getMinBitWidth
returns. Admitted icmps are recorded on the class and rewritten by
ReduceExpressionGraph before the phi-erase loop, so phi-rooted operands
are still live when their truncated versions are formed.

Signed predicates and outside-graph `(and X, Const)` users are deliberate
follow-ups; both extend the same plumbing without changing this PR's
soundness story.

One existing CHECK line in trunc_select_cmp.ll updates because a
sub-case that previously stayed wide now narrows further (semantics
preserved; constant 32768 truncates losslessly to i16 -32768).

Fixes #204098.

Assisted-by: Claude (Anthropic, claude-opus-4-7)
---
 .../AggressiveInstCombineInternal.h           |   7 +
 .../TruncInstCombine.cpp                      |  58 +++++++
 .../AggressiveInstCombine/trunc_icmp_user.ll  | 146 ++++++++++++++++++
 .../AggressiveInstCombine/trunc_select_cmp.ll |   9 +-
 4 files changed, 215 insertions(+), 5 deletions(-)
 create mode 100644 llvm/test/Transforms/AggressiveInstCombine/trunc_icmp_user.ll

diff --git a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombineInternal.h b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombineInternal.h
index 0b8a9fa48e342..37a82f9d24ef0 100644
--- a/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombineInternal.h
+++ b/llvm/lib/Transforms/AggressiveInstCombine/AggressiveInstCombineInternal.h
@@ -43,6 +43,7 @@ class AssumptionCache;
 class DataLayout;
 class DominatorTree;
 class Function;
+class ICmpInst;
 class Instruction;
 class TargetLibraryInfo;
 class TruncInst;
@@ -76,6 +77,12 @@ class TruncInstCombine {
   /// all other instructions in the graph that uses it.
   MapVector<Instruction *, Info> InstInfoMap;
 
+  /// Outside-graph ICmp users of in-graph values that getBestTruncatedType
+  /// has validated as safe to narrow alongside the graph itself. Populated
+  /// by getBestTruncatedType and consumed (and cleared) by
+  /// ReduceExpressionGraph.
+  SmallVector<ICmpInst *, 4> NarrowedICmps;
+
 public:
   TruncInstCombine(AssumptionCache &AC, TargetLibraryInfo &TLI,
                    const DataLayout &DL, const DominatorTree &DT)
diff --git a/llvm/lib/Transforms/AggressiveInstCombine/TruncInstCombine.cpp b/llvm/lib/Transforms/AggressiveInstCombine/TruncInstCombine.cpp
index 9150b58d0acf1..db7012eff45e9 100644
--- a/llvm/lib/Transforms/AggressiveInstCombine/TruncInstCombine.cpp
+++ b/llvm/lib/Transforms/AggressiveInstCombine/TruncInstCombine.cpp
@@ -32,6 +32,7 @@
 #include "llvm/IR/Dominators.h"
 #include "llvm/IR/IRBuilder.h"
 #include "llvm/IR/Instruction.h"
+#include "llvm/IR/Instructions.h"
 #include "llvm/Support/KnownBits.h"
 
 using namespace llvm;
@@ -263,6 +264,9 @@ unsigned TruncInstCombine::getMinBitWidth() {
 }
 
 Type *TruncInstCombine::getBestTruncatedType() {
+  // Reset per-graph state from any previous run.
+  NarrowedICmps.clear();
+
   if (!buildTruncExpressionGraph())
     return nullptr;
 
@@ -271,6 +275,7 @@ Type *TruncInstCombine::getBestTruncatedType() {
   // post-dominated by the trunc instruction, i.e., were visited during the
   // expression evaluation.
   unsigned DesiredBitWidth = 0;
+  SmallVector<ICmpInst *, 4> ICmpCandidates;
   for (auto Itr : InstInfoMap) {
     Instruction *I = Itr.first;
     if (I->hasOneUse())
@@ -279,6 +284,18 @@ Type *TruncInstCombine::getBestTruncatedType() {
     for (auto *U : I->users())
       if (auto *UI = dyn_cast<Instruction>(U))
         if (UI != CurrentTruncInst && !InstInfoMap.count(UI)) {
+          // Outside-graph equality and unsigned icmp users can be narrowed
+          // alongside the graph if KnownBits proves both operands fit in the
+          // narrow type. Defer that check to after MinBitWidth is known.
+          if (auto *Cmp = dyn_cast<ICmpInst>(UI))
+            if (Cmp->isEquality() || Cmp->isUnsigned()) {
+              // The same icmp can be reached via more than one of its
+              // operands when both are in-graph; rewriting it twice would
+              // dereference a freed pointer.
+              if (!llvm::is_contained(ICmpCandidates, Cmp))
+                ICmpCandidates.push_back(Cmp);
+              continue;
+            }
           if (!IsExtInst)
             return nullptr;
           // If this is an extension from the dest type, we can eliminate it,
@@ -349,6 +366,20 @@ Type *TruncInstCombine::getBestTruncatedType() {
       (DesiredBitWidth && DesiredBitWidth != MinBitWidth))
     return nullptr;
 
+  // Validate any deferred outside-graph icmp candidates against the now-known
+  // narrow bit-width. Each operand must have KnownBits proving its full value
+  // fits in MinBitWidth — the in-graph operand's narrow form is its low
+  // MinBitWidth bits, so if the full value exceeds that the rewritten icmp
+  // would observe a different value and could change the comparison result.
+  for (ICmpInst *Cmp : ICmpCandidates) {
+    for (Value *Op : Cmp->operands()) {
+      KnownBits K = llvm::computeKnownBits(Op, DL, &AC, /*CtxI=*/Cmp, &DT);
+      if (K.getMaxValue().getActiveBits() > MinBitWidth)
+        return nullptr;
+    }
+  }
+  NarrowedICmps = std::move(ICmpCandidates);
+
   return IntegerType::get(CurrentTruncInst->getContext(), MinBitWidth);
 }
 
@@ -498,6 +529,33 @@ void TruncInstCombine::ReduceExpressionGraph(Type *SclTy) {
   // Erase old expression graph, which was replaced by the reduced expression
   // graph.
   CurrentTruncInst->eraseFromParent();
+
+  // Rewrite admitted outside-graph icmp users at the narrow type. Must
+  // happen BEFORE the phi-erase loop below: phi-valued in-graph operands
+  // are RAUW'd to poison there, and any still-wide icmp referencing them
+  // would otherwise capture poison.
+  for (ICmpInst *Cmp : NarrowedICmps) {
+    IRBuilder<> Builder(Cmp);
+    auto Narrow = [&](Value *V) -> Value * {
+      // In-graph instructions and Constants already have a narrow form
+      // produced by the main rewrite loop / getReducedOperand. Outside-graph
+      // values (e.g. an Argument or unrelated SSA value) need a fresh trunc.
+      if (isa<Constant>(V))
+        return getReducedOperand(V, SclTy);
+      if (auto *I = dyn_cast<Instruction>(V); I && InstInfoMap.count(I))
+        return getReducedOperand(V, SclTy);
+      return Builder.CreateTrunc(V, getReducedType(V, SclTy));
+    };
+    Value *L = Narrow(Cmp->getOperand(0));
+    Value *R = Narrow(Cmp->getOperand(1));
+    Value *NewCmp = Builder.CreateICmp(Cmp->getPredicate(), L, R);
+    if (auto *NewI = dyn_cast<Instruction>(NewCmp))
+      NewI->takeName(Cmp);
+    Cmp->replaceAllUsesWith(NewCmp);
+    Cmp->eraseFromParent();
+  }
+  NarrowedICmps.clear();
+
   // First, erase old phi-nodes and its uses
   for (auto &Node : OldNewPHINodes) {
     PHINode *OldPN = Node.first;
diff --git a/llvm/test/Transforms/AggressiveInstCombine/trunc_icmp_user.ll b/llvm/test/Transforms/AggressiveInstCombine/trunc_icmp_user.ll
new file mode 100644
index 0000000000000..e5cec13127349
--- /dev/null
+++ b/llvm/test/Transforms/AggressiveInstCombine/trunc_icmp_user.ll
@@ -0,0 +1,146 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt < %s -passes=aggressive-instcombine -S | FileCheck %s
+
+; Issue #204098: TruncInstCombine bailed when a non-ext in-graph value had
+; ANY outside-graph user that wasn't a ZExt/SExt — even when that user is
+; an icmp whose operands are provably narrow under KnownBits. The
+; conservative bail forfeited the whole rewrite. Admitting such icmps lets
+; the surrounding expression graph (and the icmp) narrow.
+;
+; All tests root the in-graph chain at `zext i8 %xb to i16` (an Instruction
+; leaf the existing pass accepts) so the cases stay independent of #202112
+; (which adds Argument-as-leaf).
+
+declare void @sink(i8)
+
+;; --- Positive cases (must narrow with the fix) ---
+
+; Outside-graph icmp on an in-graph `and` whose KnownBits proves it fits
+; in i8 (mask 15). The icmp compares against a small Constant that also
+; fits. Both gates pass; eq is always value-preserving at the narrow width.
+define i1 @icmp_eq_and_const(i8 %xb) {
+; CHECK-LABEL: @icmp_eq_and_const(
+; CHECK-NEXT:    [[M:%.*]] = and i8 [[XB:%.*]], 15
+; CHECK-NEXT:    [[S:%.*]] = shl i8 [[M]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[M]], 7
+; CHECK-NEXT:    call void @sink(i8 [[S]])
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %x = zext i8 %xb to i16
+  %m = and i16 %x, 15
+  %s = shl i16 %m, 1
+  %t = trunc i16 %s to i8
+  %cmp = icmp eq i16 %m, 7
+  call void @sink(i8 %t)
+  ret i1 %cmp
+}
+
+; Outside-graph icmp on TWO different in-graph values that both have
+; KnownBits proving fit in i8. eq predicate.
+define i1 @icmp_eq_two_ingraph(i8 %xb, i8 %yb) {
+; CHECK-LABEL: @icmp_eq_two_ingraph(
+; CHECK-NEXT:    [[MX:%.*]] = and i8 [[XB:%.*]], 15
+; CHECK-NEXT:    [[MY:%.*]] = and i8 [[YB:%.*]], 15
+; CHECK-NEXT:    [[ADD:%.*]] = add i8 [[MX]], [[MY]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[MX]], [[MY]]
+; CHECK-NEXT:    call void @sink(i8 [[ADD]])
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %x = zext i8 %xb to i16
+  %y = zext i8 %yb to i16
+  %mx = and i16 %x, 15
+  %my = and i16 %y, 15
+  %add = add i16 %mx, %my
+  %t = trunc i16 %add to i8
+  %cmp = icmp eq i16 %mx, %my
+  call void @sink(i8 %t)
+  ret i1 %cmp
+}
+
+; Unsigned predicate (ult): safe at the narrow width whenever both operands
+; fit in NarrowBits unsigned.
+define i1 @icmp_ult_and_const(i8 %xb) {
+; CHECK-LABEL: @icmp_ult_and_const(
+; CHECK-NEXT:    [[M:%.*]] = and i8 [[XB:%.*]], 31
+; CHECK-NEXT:    [[S:%.*]] = shl i8 [[M]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ult i8 [[M]], 16
+; CHECK-NEXT:    call void @sink(i8 [[S]])
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %x = zext i8 %xb to i16
+  %m = and i16 %x, 31
+  %s = shl i16 %m, 1
+  %t = trunc i16 %s to i8
+  %cmp = icmp ult i16 %m, 16
+  call void @sink(i8 %t)
+  ret i1 %cmp
+}
+
+;; --- Negative cases (must NOT narrow even with the fix) ---
+
+; In-graph operand `%add` reaches 256 (1 + low byte of %xb), KnownBits
+; gives 9 active bits — does NOT fit in NarrowBits=8. Narrowing the icmp
+; would compare the low byte only, which gives a different result when
+; %add == 256 (low byte 0 == 0 true; full 256 == 0 false). The KnownBits
+; gate must reject this.
+define i1 @icmp_eq_unsound_negative(i8 %xb) {
+; CHECK-LABEL: @icmp_eq_unsound_negative(
+; CHECK-NEXT:    [[X:%.*]] = zext i8 [[XB:%.*]] to i16
+; CHECK-NEXT:    [[ADD:%.*]] = add i16 [[X]], 1
+; CHECK-NEXT:    [[T:%.*]] = trunc i16 [[ADD]] to i8
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i16 [[ADD]], 0
+; CHECK-NEXT:    call void @sink(i8 [[T]])
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %x = zext i8 %xb to i16
+  %add = add i16 %x, 1
+  %t = trunc i16 %add to i8
+  %cmp = icmp eq i16 %add, 0
+  call void @sink(i8 %t)
+  ret i1 %cmp
+}
+
+; Signed predicate (slt) requires operands to fit in NarrowBits-1 unsigned
+; so the narrow-width sign bit stays clear. `%m = and i16 %x, 255` has bit
+; 7 potentially set; signedness of a comparison against 5 flips at i8.
+; Must reject.
+define i1 @icmp_slt_signed_negative(i8 %xb) {
+; CHECK-LABEL: @icmp_slt_signed_negative(
+; CHECK-NEXT:    [[X:%.*]] = zext i8 [[XB:%.*]] to i16
+; CHECK-NEXT:    [[M:%.*]] = and i16 [[X]], 255
+; CHECK-NEXT:    [[ADD:%.*]] = add i16 [[M]], [[M]]
+; CHECK-NEXT:    [[T:%.*]] = trunc i16 [[ADD]] to i8
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i16 [[M]], 5
+; CHECK-NEXT:    call void @sink(i8 [[T]])
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %x = zext i8 %xb to i16
+  %m = and i16 %x, 255
+  %add = add i16 %m, %m
+  %t = trunc i16 %add to i8
+  %cmp = icmp slt i16 %m, 5
+  call void @sink(i8 %t)
+  ret i1 %cmp
+}
+
+; Outside-graph user that is not an icmp (here: a store of the in-graph
+; AND result) must still bail — existing behavior, guarded so the
+; relaxation stays narrow.
+define void @store_user_negative(i8 %xb, ptr %p) {
+; CHECK-LABEL: @store_user_negative(
+; CHECK-NEXT:    [[X:%.*]] = zext i8 [[XB:%.*]] to i16
+; CHECK-NEXT:    [[M:%.*]] = and i16 [[X]], 15
+; CHECK-NEXT:    [[S:%.*]] = shl i16 [[M]], 1
+; CHECK-NEXT:    [[T:%.*]] = trunc i16 [[S]] to i8
+; CHECK-NEXT:    store i16 [[M]], ptr [[P:%.*]], align 2
+; CHECK-NEXT:    call void @sink(i8 [[T]])
+; CHECK-NEXT:    ret void
+;
+  %x = zext i8 %xb to i16
+  %m = and i16 %x, 15
+  %s = shl i16 %m, 1
+  %t = trunc i16 %s to i8
+  store i16 %m, ptr %p
+  call void @sink(i8 %t)
+  ret void
+}
diff --git a/llvm/test/Transforms/AggressiveInstCombine/trunc_select_cmp.ll b/llvm/test/Transforms/AggressiveInstCombine/trunc_select_cmp.ll
index 69ad6258e9cd8..09ef6f1e3158e 100644
--- a/llvm/test/Transforms/AggressiveInstCombine/trunc_select_cmp.ll
+++ b/llvm/test/Transforms/AggressiveInstCombine/trunc_select_cmp.ll
@@ -182,11 +182,10 @@ define i16 @cmp_select_signed_const_i16Const_noTransformation(i8 %a) {
 
 define i16 @cmp_select_unsigned_const_i16Const(i8 %a) {
 ; CHECK-LABEL: @cmp_select_unsigned_const_i16Const(
-; CHECK-NEXT:    [[CONV:%.*]] = zext i8 [[A:%.*]] to i32
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ult i32 [[CONV]], 32768
-; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i32 32768, i32 [[CONV]]
-; CHECK-NEXT:    [[CONV4:%.*]] = trunc i32 [[COND]] to i16
-; CHECK-NEXT:    ret i16 [[CONV4]]
+; CHECK-NEXT:    [[CONV:%.*]] = zext i8 [[A:%.*]] to i16
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ult i16 [[CONV]], -32768
+; CHECK-NEXT:    [[COND:%.*]] = select i1 [[CMP]], i16 -32768, i16 [[CONV]]
+; CHECK-NEXT:    ret i16 [[COND]]
 ;
   %conv = zext i8 %a to i32
   %cmp = icmp ult i32 %conv, 32768



More information about the llvm-commits mailing list