[llvm] [InstCombine] Recognize abs through positive-K nsw multiply. (PR #207539)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Jul 6 02:38:05 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-analysis
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
Generalize matchSelectPattern to also consider multiplies that do not
change sign when trying to from llvm.abs.
End-to-end, this allows vectorizing with narrower element types for code
such as below. This triggers in some ffmpeg kernels on AArch64.
```
void scaled_absdiff_u8(uint8_t * __restrict out,
const uint8_t * __restrict a,
const uint8_t * __restrict b,
size_t n) {
for (size_t i = 0; i < n; ++i) {
int32_t d = 4 * ((int32_t)a[i] - (int32_t)b[i]);
if (d < 0) d = -d;
if (d > 255) d = 255;
out[i] = (uint8_t)d;
}
}
```
Alive2 proof: https://alive2.llvm.org/ce/z/pGYQ36
---
Full diff: https://github.com/llvm/llvm-project/pull/207539.diff
2 Files Affected:
- (modified) llvm/lib/Analysis/ValueTracking.cpp (+8-2)
- (added) llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll (+319)
``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 7dd23f24dfcc7..3d47ade1df20e 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -9003,9 +9003,15 @@ static SelectPatternResult matchSelectPattern(CmpInst::Predicate Pred,
if (isKnownNegation(TrueVal, FalseVal)) {
// Sign-extending LHS does not change its sign, so TrueVal/FalseVal can
- // match against either LHS or sext(LHS).
- auto MaybeSExtCmpLHS =
+ // match against either LHS or sign-preserving operations on LHS, like
+ // sext(LHS), or binary ops that do not wrap in signed sense.
+ auto CmpLHSOrSExt =
m_CombineOr(m_Specific(CmpLHS), m_SExt(m_Specific(CmpLHS)));
+ auto MaybeSExtCmpLHS = m_CombineOr(
+ CmpLHSOrSExt,
+ m_CombineOr(m_CombineOr(m_NSWMul(CmpLHSOrSExt, m_StrictlyPositive()),
+ m_NSWMul(m_StrictlyPositive(), CmpLHSOrSExt)),
+ m_NSWShl(CmpLHSOrSExt, m_NonNegative())));
auto ZeroOrAllOnes = m_CombineOr(m_ZeroInt(), m_AllOnes());
auto ZeroOrOne = m_CombineOr(m_ZeroInt(), m_One());
if (match(TrueVal, MaybeSExtCmpLHS)) {
diff --git a/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll b/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll
new file mode 100644
index 0000000000000..af92a99417a71
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/select-abs-positive-mul.ll
@@ -0,0 +1,319 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=instcombine %s -S | FileCheck %s
+
+;; Tests for abs idioms with signed multiplies, like
+; `int v = (int)narrow * K; return v < 0 ? v : -v;`.
+
+define i32 @sel_mul_pos_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define i32 @sel_shl_pos_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_shl_pos_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = shl nsw i32 [[X]], 2
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = shl nsw i32 %x, 2
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+; x >= 0 ? K*x : -K*x (arms swapped, sge predicate).
+define i32 @sel_mul_pos_x_ge_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_ge_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp sge i32 %x, 0
+ %sel = select i1 %cmp, i32 %m, i32 %neg
+ ret i32 %sel
+}
+
+; x > 0 ? K*x : -K*x
+define i32 @sel_mul_pos_x_sgt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_sgt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp sgt i32 %x, 0
+ %sel = select i1 %cmp, i32 %m, i32 %neg
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_x_sgt_minus_one(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_sgt_minus_one(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp sgt i32 %x, -1
+ %sel = select i1 %cmp, i32 %m, i32 %neg
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_x_lt_zero_lhs_const(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_x_lt_zero_lhs_const(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 5, %x
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define <4 x i32> @sel_mul_pos_x_lt_zero_vec(<4 x i32> %x) {
+; CHECK-LABEL: define <4 x i32> @sel_mul_pos_x_lt_zero_vec(
+; CHECK-SAME: <4 x i32> [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = shl nsw <4 x i32> [[X]], splat (i32 2)
+; CHECK-NEXT: [[SEL:%.*]] = call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[M]], i1 true)
+; CHECK-NEXT: ret <4 x i32> [[SEL]]
+;
+ %m = mul nsw <4 x i32> %x, splat (i32 4)
+ %neg = sub nsw <4 x i32> zeroinitializer, %m
+ %cmp = icmp slt <4 x i32> %x, zeroinitializer
+ %sel = select <4 x i1> %cmp, <4 x i32> %neg, <4 x i32> %m
+ ret <4 x i32> %sel
+}
+
+define <4 x i32> @sel_mul_pos_vec_nonsplat(<4 x i32> %x) {
+; CHECK-LABEL: define <4 x i32> @sel_mul_pos_vec_nonsplat(
+; CHECK-SAME: <4 x i32> [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw <4 x i32> [[X]], <i32 3, i32 5, i32 7, i32 11>
+; CHECK-NEXT: [[SEL:%.*]] = call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[M]], i1 true)
+; CHECK-NEXT: ret <4 x i32> [[SEL]]
+;
+ %m = mul nsw <4 x i32> %x, <i32 3, i32 5, i32 7, i32 11>
+ %neg = sub nsw <4 x i32> zeroinitializer, %m
+ %cmp = icmp slt <4 x i32> %x, zeroinitializer
+ %sel = select <4 x i1> %cmp, <4 x i32> %neg, <4 x i32> %m
+ ret <4 x i32> %sel
+}
+
+; (X < 0) ? K*X : -K*X --> -abs(K*X)
+define i32 @sel_mul_pos_nabs(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_nabs(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[TMP1:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 false)
+; CHECK-NEXT: [[SEL:%.*]] = sub i32 0, [[TMP1]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %m, i32 %neg
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_neg_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_neg_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 false)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+; Constant is negative.
+define i32 @sel_mul_neg_x_lt_zero(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_neg_x_lt_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], -5
+; CHECK-NEXT: [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, -5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+; Mul is missing nsw.
+define i32 @sel_mul_pos_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul i32 [[X]], 5
+; CHECK-NEXT: [[NEG:%.*]] = sub i32 0, [[M]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul i32 %x, 5
+ %neg = sub i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+; Shl missing nsw.
+define i32 @sel_shl_pos_no_nsw(i32 %x) {
+; CHECK-LABEL: define i32 @sel_shl_pos_no_nsw(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = shl i32 [[X]], 2
+; CHECK-NEXT: [[NEG:%.*]] = sub i32 0, [[M]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = shl i32 %x, 2
+ %neg = sub i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+; Negative test: cmp operand is unrelated to the scaled value.
+define i32 @sel_unrelated_cmp(i32 %x, i32 %y) {
+; CHECK-LABEL: define i32 @sel_unrelated_cmp(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[Y:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[Y]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %y, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_cmp(i32 %x, ptr %p) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_cmp(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT: store i1 [[CMP]], ptr [[P]], align 1
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i32 %x, 0
+ store i1 %cmp, ptr %p
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_neg(i32 %x, ptr %p) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_neg(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT: store i32 [[NEG]], ptr [[P]], align 4
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ store i32 %neg, ptr %p
+ %cmp = icmp slt i32 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define i32 @sel_mul_pos_multi_use_both(i32 %x, ptr %p, ptr %q) {
+; CHECK-LABEL: define i32 @sel_mul_pos_multi_use_both(
+; CHECK-SAME: i32 [[X:%.*]], ptr [[P:%.*]], ptr [[Q:%.*]]) {
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[X]], 5
+; CHECK-NEXT: [[NEG:%.*]] = sub nsw i32 0, [[M]]
+; CHECK-NEXT: store i32 [[NEG]], ptr [[Q]], align 4
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i32 [[X]], 0
+; CHECK-NEXT: store i1 [[CMP]], ptr [[P]], align 1
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP]], i32 [[NEG]], i32 [[M]]
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %m = mul nsw i32 %x, 5
+ %neg = sub nsw i32 0, %m
+ store i32 %neg, ptr %q
+ %cmp = icmp slt i32 %x, 0
+ store i1 %cmp, ptr %p
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
+
+define i16 @sel_mul_pos_sext_x_lt_zero(i8 %x) {
+; CHECK-LABEL: define i16 @sel_mul_pos_sext_x_lt_zero(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT: [[XW:%.*]] = sext i8 [[X]] to i16
+; CHECK-NEXT: [[M:%.*]] = mul nsw i16 [[XW]], 3
+; CHECK-NEXT: [[SEL:%.*]] = call i16 @llvm.abs.i16(i16 [[M]], i1 true)
+; CHECK-NEXT: ret i16 [[SEL]]
+;
+ %xw = sext i8 %x to i16
+ %m = mul nsw i16 %xw, 3
+ %neg = sub nsw i16 0, %m
+ %cmp = icmp slt i8 %x, 0
+ %sel = select i1 %cmp, i16 %neg, i16 %m
+ ret i16 %sel
+}
+
+define i16 @sel_shl_pos_sext_x_lt_zero(i8 %x) {
+; CHECK-LABEL: define i16 @sel_shl_pos_sext_x_lt_zero(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT: [[XW:%.*]] = sext i8 [[X]] to i16
+; CHECK-NEXT: [[M:%.*]] = shl nsw i16 [[XW]], 2
+; CHECK-NEXT: [[SEL:%.*]] = call i16 @llvm.abs.i16(i16 [[M]], i1 true)
+; CHECK-NEXT: ret i16 [[SEL]]
+;
+ %xw = sext i8 %x to i16
+ %m = shl nsw i16 %xw, 2
+ %neg = sub nsw i16 0, %m
+ %cmp = icmp slt i8 %x, 0
+ %sel = select i1 %cmp, i16 %neg, i16 %m
+ ret i16 %sel
+}
+
+define i32 @sel_mul_pos_sext_i8_i32(i8 %x) {
+; CHECK-LABEL: define i32 @sel_mul_pos_sext_i8_i32(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT: [[XW:%.*]] = sext i8 [[X]] to i32
+; CHECK-NEXT: [[M:%.*]] = mul nsw i32 [[XW]], 5
+; CHECK-NEXT: [[SEL:%.*]] = call i32 @llvm.abs.i32(i32 [[M]], i1 true)
+; CHECK-NEXT: ret i32 [[SEL]]
+;
+ %xw = sext i8 %x to i32
+ %m = mul nsw i32 %xw, 5
+ %neg = sub nsw i32 0, %m
+ %cmp = icmp slt i8 %x, 0
+ %sel = select i1 %cmp, i32 %neg, i32 %m
+ ret i32 %sel
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/207539
More information about the llvm-commits
mailing list