[llvm] [InstCombine] Recognize abs through positive-K nsw multiply. (PR #207539)

via llvm-commits llvm-commits at lists.llvm.org
Mon Jul 6 02:38:05 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-analysis

Author: Florian Hahn (fhahn)

<details>
<summary>Changes</summary>

Generalize matchSelectPattern to also consider multiplies that do not
change sign when trying to from llvm.abs.

End-to-end, this allows vectorizing with narrower element types for code
such as below. This triggers in some ffmpeg kernels on AArch64.

```
  void scaled_absdiff_u8(uint8_t * __restrict out,
                         const uint8_t * __restrict a,
                         const uint8_t * __restrict b,
                         size_t n) {
    for (size_t i = 0; i < n; ++i) {
      int32_t d = 4 * ((int32_t)a[i] - (int32_t)b[i]);
      if (d < 0) d = -d;
      if (d > 255) d = 255;
      out[i] = (uint8_t)d;
    }
  }
```

Alive2 proof: https://alive2.llvm.org/ce/z/pGYQ36

---
Full diff: https://github.com/llvm/llvm-project/pull/207539.diff


2 Files Affected:

- (modified) llvm/lib/Analysis/ValueTracking.cpp (+8-2) 
- (added) llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll (+319) 


``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 7dd23f24dfcc7..3d47ade1df20e 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -9003,9 +9003,15 @@ static SelectPatternResult matchSelectPattern(CmpInst::Predicate Pred,
 
   if (isKnownNegation(TrueVal, FalseVal)) {
     // Sign-extending LHS does not change its sign, so TrueVal/FalseVal can
-    // match against either LHS or sext(LHS).
-    auto MaybeSExtCmpLHS =
+    // match against either LHS or sign-preserving operations on LHS, like
+    // sext(LHS), or binary ops that do not wrap in signed sense.
+    auto CmpLHSOrSExt =
         m_CombineOr(m_Specific(CmpLHS), m_SExt(m_Specific(CmpLHS)));
+    auto MaybeSExtCmpLHS = m_CombineOr(
+        CmpLHSOrSExt,
+        m_CombineOr(m_CombineOr(m_NSWMul(CmpLHSOrSExt, m_StrictlyPositive()),
+                                m_NSWMul(m_StrictlyPositive(), CmpLHSOrSExt)),
+                    m_NSWShl(CmpLHSOrSExt, m_NonNegative())));
     auto ZeroOrAllOnes = m_CombineOr(m_ZeroInt(), m_AllOnes());
     auto ZeroOrOne = m_CombineOr(m_ZeroInt(), m_One());
     if (match(TrueVal, MaybeSExtCmpLHS)) {
diff --git a/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll b/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll
new file mode 100644
index 0000000000000..af92a99417a71
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll
@@ -0,0 +1,319 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=instcombine %s -S | FileCheck %s
+
+;; Tests for abs idioms with signed multiplies, like
+; `int v = (int)narrow * K; return v < 0 ? v : -v;`.
+
+define i32 @sel_mul_pos_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define i32 @sel_shl_pos_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_shl_pos_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = shl nsw i32 [[X]], 2
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = shl nsw i32 %x, 2
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+; x >= 0 ? K*x : -K*x  (arms swapped, sge predicate).
+define i32 @sel_mul_pos_x_ge_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_ge_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp sge i32 %x, 0
+  %sel = select i1 %cmp, i32 %m, i32 %neg
+  ret i32 %sel
+}
+
+; x > 0 ? K*x : -K*x
+define i32 @sel_mul_pos_x_sgt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_sgt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp sgt i32 %x, 0
+  %sel = select i1 %cmp, i32 %m, i32 %neg
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_x_sgt_minus_one(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_sgt_minus_one(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp sgt i32 %x, -1
+  %sel = select i1 %cmp, i32 %m, i32 %neg
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_x_lt_zero_lhs_const(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_lt_zero_lhs_const(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 5, %x
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define <4 x i32> @sel_mul_pos_x_lt_zero_vec(<4 x i32> %x) {
+; CHECK-LABEL: define <4 x i32> @sel_mul_pos_x_lt_zero_vec(
+; CHECK-SAME: <4 x i32> [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = shl nsw <4 x i32> [[X]], splat (i32 2)
+; CHECK-NEXT:    [[SEL:%.*]] = call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[M]], i1 true)
+; CHECK-NEXT:    ret <4 x i32> [[SEL]]
+;
+  %m = mul nsw <4 x i32> %x, splat (i32 4)
+  %neg = sub nsw <4 x i32> zeroinitializer, %m
+  %cmp = icmp slt <4 x i32> %x, zeroinitializer
+  %sel = select <4 x i1> %cmp, <4 x i32> %neg, <4 x i32> %m
+  ret <4 x i32> %sel
+}
+
+define <4 x i32> @sel_mul_pos_vec_nonsplat(<4 x i32> %x) {
+; CHECK-LABEL: define <4 x i32> @sel_mul_pos_vec_nonsplat(
+; CHECK-SAME: <4 x i32> [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw <4 x i32> [[X]], <i32 3, i32 5, i32 7, i32 11>
+; CHECK-NEXT:    [[SEL:%.*]] = call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[M]], i1 true)
+; CHECK-NEXT:    ret <4 x i32> [[SEL]]
+;
+  %m = mul nsw <4 x i32> %x, <i32 3, i32 5, i32 7, i32 11>
+  %neg = sub nsw <4 x i32> zeroinitializer, %m
+  %cmp = icmp slt <4 x i32> %x, zeroinitializer
+  %sel = select <4 x i1> %cmp, <4 x i32> %neg, <4 x i32> %m
+  ret <4 x i32> %sel
+}
+
+; (X < 0) ? K*X : -K*X --> -abs(K*X)
+define i32 @sel_mul_pos_nabs(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_nabs(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[TMP1:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 false)
+; CHECK-NEXT:    [[SEL:%.*]] = sub i32 0, [[TMP1]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %m, i32 %neg
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_neg_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_neg_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 false)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+; Constant is negative.
+define i32 @sel_mul_neg_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_neg_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], -5
+; CHECK-NEXT:    [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, -5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+; Mul is missing nsw.
+define i32 @sel_mul_pos_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul i32 [[X]], 5
+; CHECK-NEXT:    [[NEG:%.*]] = sub i32 0, [[M]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul i32 %x, 5
+  %neg = sub i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+; Shl missing nsw.
+define i32 @sel_shl_pos_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_shl_pos_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = shl i32 [[X]], 2
+; CHECK-NEXT:    [[NEG:%.*]] = sub i32 0, [[M]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = shl i32 %x, 2
+  %neg = sub i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+; Negative test: cmp operand is unrelated to the scaled value.
+define i32 @sel_unrelated_cmp(i32 %x, i32 %y) {
+; CHECK-LABEL: define i32 @sel_unrelated_cmp(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[Y:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[Y]], 0
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %y, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_cmp(i32 %x, ptr %p) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_cmp(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT:    store i1 [[CMP]], ptr [[P]], align 1
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i32 %x, 0
+  store i1 %cmp, ptr %p
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_neg(i32 %x, ptr %p) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_neg(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT:    store i32 [[NEG]], ptr [[P]], align 4
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  store i32 %neg, ptr %p
+  %cmp = icmp slt i32 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_both(i32 %x, ptr %p, ptr %q) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_both(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]], ptr [[Q:%.*]]) {
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT:    [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT:    store i32 [[NEG]], ptr [[Q]], align 4
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT:    store i1 [[CMP]], ptr [[P]], align 1
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %m = mul nsw i32 %x, 5
+  %neg = sub nsw i32 0, %m
+  store i32 %neg, ptr %q
+  %cmp = icmp slt i32 %x, 0
+  store i1 %cmp, ptr %p
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}
+
+define i16 @sel_mul_pos_sext_x_lt_zero(i8 %x) {
+; CHECK-LABEL: define i16 @sel_mul_pos_sext_x_lt_zero(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT:    [[XW:%.*]] = sext i8 [[X]] to i16
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i16 [[XW]], 3
+; CHECK-NEXT:    [[SEL:%.*]] = call i16 @llvm.abs.i16(i16 [[M]], i1 true)
+; CHECK-NEXT:    ret i16 [[SEL]]
+;
+  %xw = sext i8 %x to i16
+  %m = mul nsw i16 %xw, 3
+  %neg = sub nsw i16 0, %m
+  %cmp = icmp slt i8 %x, 0
+  %sel = select i1 %cmp, i16 %neg, i16 %m
+  ret i16 %sel
+}
+
+define i16 @sel_shl_pos_sext_x_lt_zero(i8 %x) {
+; CHECK-LABEL: define i16 @sel_shl_pos_sext_x_lt_zero(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT:    [[XW:%.*]] = sext i8 [[X]] to i16
+; CHECK-NEXT:    [[M:%.*]] = shl nsw i16 [[XW]], 2
+; CHECK-NEXT:    [[SEL:%.*]] = call i16 @llvm.abs.i16(i16 [[M]], i1 true)
+; CHECK-NEXT:    ret i16 [[SEL]]
+;
+  %xw = sext i8 %x to i16
+  %m = shl nsw i16 %xw, 2
+  %neg = sub nsw i16 0, %m
+  %cmp = icmp slt i8 %x, 0
+  %sel = select i1 %cmp, i16 %neg, i16 %m
+  ret i16 %sel
+}
+
+define i32 @sel_mul_pos_sext_i8_i32(i8 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_sext_i8_i32(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT:    [[XW:%.*]] = sext i8 [[X]] to i32
+; CHECK-NEXT:    [[M:%.*]] = mul nsw i32 [[XW]], 5
+; CHECK-NEXT:    [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT:    ret i32 [[SEL]]
+;
+  %xw = sext i8 %x to i32
+  %m = mul nsw i32 %xw, 5
+  %neg = sub nsw i32 0, %m
+  %cmp = icmp slt i8 %x, 0
+  %sel = select i1 %cmp, i32 %neg, i32 %m
+  ret i32 %sel
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/207539


More information about the llvm-commits mailing list