[llvm] [SLP] Cost reduce.add(mul(ext, ext)) as a dot-product reduction (PR #224066)
Madhur Amilkanthwar via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 09:04:11 PDT 2026
https://github.com/madhur13490 updated https://github.com/llvm/llvm-project/pull/224066
>From e331efd6d01a6462600556b60a1f1f7cd8c56d2d Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Thu, 17 Sep 2026 04:13:14 -0700
Subject: [PATCH 1/4] [SLP][NFC] Add precommit test for dot-product reduction
costing
Adds a test for reduce.add(mul(ext, ext)) with and without +dotprod at a threshold between the plain and fused costs. Baseline: both stay scalar; a later change makes the +dotprod case vectorize.
---
.../AArch64/reduce-add-dotprod.ll | 273 ++++++++++++++++++
1 file changed, 273 insertions(+)
create mode 100644 llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
new file mode 100644
index 0000000000000..36d2a0f797280
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
@@ -0,0 +1,273 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=slp-vectorizer -slp-threshold=28 -S -mtriple=aarch64 < %s | FileCheck %s
+
+; The reduce.add(mul(ext, ext)) idiom lowers to a single dot-product
+; reduction (UDOT/SDOT) when the target supports it. The threshold is set
+; between the plain reduction cost and the fused dot-product cost so the
+; reduction is profitable only when the dot-product cost applies: the
+; +dotprod function vectorizes, the plain function stays scalar.
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128-Fn32"
+target triple = "aarch64"
+
+define i32 @mla_v8i8_i32_dotprod(ptr %x, ptr %y) "target-features"="+dotprod" {
+; CHECK-LABEL: @mla_v8i8_i32_dotprod(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[X:%.*]], align 1
+; CHECK-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32
+; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[Y:%.*]], align 1
+; CHECK-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32
+; CHECK-NEXT: [[MUL:%.*]] = mul nsw i32 [[CONV3]], [[CONV]]
+; CHECK-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX_1]], align 1
+; CHECK-NEXT: [[CONV_1:%.*]] = sext i8 [[TMP2]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_1:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 1
+; CHECK-NEXT: [[TMP3:%.*]] = load i8, ptr [[ARRAYIDX2_1]], align 1
+; CHECK-NEXT: [[CONV3_1:%.*]] = sext i8 [[TMP3]] to i32
+; CHECK-NEXT: [[MUL_1:%.*]] = mul nsw i32 [[CONV3_1]], [[CONV_1]]
+; CHECK-NEXT: [[ADD_1:%.*]] = add nsw i32 [[MUL_1]], [[MUL]]
+; CHECK-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 2
+; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX_2]], align 1
+; CHECK-NEXT: [[CONV_2:%.*]] = sext i8 [[TMP4]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_2:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 2
+; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX2_2]], align 1
+; CHECK-NEXT: [[CONV3_2:%.*]] = sext i8 [[TMP16]] to i32
+; CHECK-NEXT: [[MUL_2:%.*]] = mul nsw i32 [[CONV3_2]], [[CONV_2]]
+; CHECK-NEXT: [[ADD_2:%.*]] = add nsw i32 [[MUL_2]], [[ADD_1]]
+; CHECK-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 3
+; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX_3]], align 1
+; CHECK-NEXT: [[CONV_3:%.*]] = sext i8 [[TMP6]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_3:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 3
+; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX2_3]], align 1
+; CHECK-NEXT: [[CONV3_3:%.*]] = sext i8 [[TMP7]] to i32
+; CHECK-NEXT: [[MUL_3:%.*]] = mul nsw i32 [[CONV3_3]], [[CONV_3]]
+; CHECK-NEXT: [[ADD_3:%.*]] = add nsw i32 [[MUL_3]], [[ADD_2]]
+; CHECK-NEXT: [[ARRAYIDX_4:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 4
+; CHECK-NEXT: [[TMP8:%.*]] = load i8, ptr [[ARRAYIDX_4]], align 1
+; CHECK-NEXT: [[CONV_4:%.*]] = sext i8 [[TMP8]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_4:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 4
+; CHECK-NEXT: [[TMP9:%.*]] = load i8, ptr [[ARRAYIDX2_4]], align 1
+; CHECK-NEXT: [[CONV3_4:%.*]] = sext i8 [[TMP9]] to i32
+; CHECK-NEXT: [[MUL_4:%.*]] = mul nsw i32 [[CONV3_4]], [[CONV_4]]
+; CHECK-NEXT: [[ADD_4:%.*]] = add nsw i32 [[MUL_4]], [[ADD_3]]
+; CHECK-NEXT: [[ARRAYIDX_5:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 5
+; CHECK-NEXT: [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX_5]], align 1
+; CHECK-NEXT: [[CONV_5:%.*]] = sext i8 [[TMP10]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_5:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 5
+; CHECK-NEXT: [[TMP11:%.*]] = load i8, ptr [[ARRAYIDX2_5]], align 1
+; CHECK-NEXT: [[CONV3_5:%.*]] = sext i8 [[TMP11]] to i32
+; CHECK-NEXT: [[MUL_5:%.*]] = mul nsw i32 [[CONV3_5]], [[CONV_5]]
+; CHECK-NEXT: [[ADD_5:%.*]] = add nsw i32 [[MUL_5]], [[ADD_4]]
+; CHECK-NEXT: [[ARRAYIDX_6:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 6
+; CHECK-NEXT: [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX_6]], align 1
+; CHECK-NEXT: [[CONV_6:%.*]] = sext i8 [[TMP12]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_6:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 6
+; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX2_6]], align 1
+; CHECK-NEXT: [[CONV3_6:%.*]] = sext i8 [[TMP13]] to i32
+; CHECK-NEXT: [[MUL_6:%.*]] = mul nsw i32 [[CONV3_6]], [[CONV_6]]
+; CHECK-NEXT: [[ADD_6:%.*]] = add nsw i32 [[MUL_6]], [[ADD_5]]
+; CHECK-NEXT: [[ARRAYIDX_7:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 7
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX_7]], align 1
+; CHECK-NEXT: [[CONV_7:%.*]] = sext i8 [[TMP14]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_7:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 7
+; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX2_7]], align 1
+; CHECK-NEXT: [[CONV3_7:%.*]] = sext i8 [[TMP15]] to i32
+; CHECK-NEXT: [[MUL_7:%.*]] = mul nsw i32 [[CONV3_7]], [[CONV_7]]
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw i32 [[MUL_7]], [[ADD_6]]
+; CHECK-NEXT: ret i32 [[TMP5]]
+;
+entry:
+ %0 = load i8, ptr %x
+ %conv = sext i8 %0 to i32
+ %1 = load i8, ptr %y
+ %conv3 = sext i8 %1 to i32
+ %mul = mul nsw i32 %conv3, %conv
+ %arrayidx.1 = getelementptr inbounds nuw i8, ptr %x, i64 1
+ %2 = load i8, ptr %arrayidx.1
+ %conv.1 = sext i8 %2 to i32
+ %arrayidx2.1 = getelementptr inbounds nuw i8, ptr %y, i64 1
+ %3 = load i8, ptr %arrayidx2.1
+ %conv3.1 = sext i8 %3 to i32
+ %mul.1 = mul nsw i32 %conv3.1, %conv.1
+ %add.1 = add nsw i32 %mul.1, %mul
+ %arrayidx.2 = getelementptr inbounds nuw i8, ptr %x, i64 2
+ %4 = load i8, ptr %arrayidx.2
+ %conv.2 = sext i8 %4 to i32
+ %arrayidx2.2 = getelementptr inbounds nuw i8, ptr %y, i64 2
+ %5 = load i8, ptr %arrayidx2.2
+ %conv3.2 = sext i8 %5 to i32
+ %mul.2 = mul nsw i32 %conv3.2, %conv.2
+ %add.2 = add nsw i32 %mul.2, %add.1
+ %arrayidx.3 = getelementptr inbounds nuw i8, ptr %x, i64 3
+ %6 = load i8, ptr %arrayidx.3
+ %conv.3 = sext i8 %6 to i32
+ %arrayidx2.3 = getelementptr inbounds nuw i8, ptr %y, i64 3
+ %7 = load i8, ptr %arrayidx2.3
+ %conv3.3 = sext i8 %7 to i32
+ %mul.3 = mul nsw i32 %conv3.3, %conv.3
+ %add.3 = add nsw i32 %mul.3, %add.2
+ %arrayidx.4 = getelementptr inbounds nuw i8, ptr %x, i64 4
+ %8 = load i8, ptr %arrayidx.4
+ %conv.4 = sext i8 %8 to i32
+ %arrayidx2.4 = getelementptr inbounds nuw i8, ptr %y, i64 4
+ %9 = load i8, ptr %arrayidx2.4
+ %conv3.4 = sext i8 %9 to i32
+ %mul.4 = mul nsw i32 %conv3.4, %conv.4
+ %add.4 = add nsw i32 %mul.4, %add.3
+ %arrayidx.5 = getelementptr inbounds nuw i8, ptr %x, i64 5
+ %10 = load i8, ptr %arrayidx.5
+ %conv.5 = sext i8 %10 to i32
+ %arrayidx2.5 = getelementptr inbounds nuw i8, ptr %y, i64 5
+ %11 = load i8, ptr %arrayidx2.5
+ %conv3.5 = sext i8 %11 to i32
+ %mul.5 = mul nsw i32 %conv3.5, %conv.5
+ %add.5 = add nsw i32 %mul.5, %add.4
+ %arrayidx.6 = getelementptr inbounds nuw i8, ptr %x, i64 6
+ %12 = load i8, ptr %arrayidx.6
+ %conv.6 = sext i8 %12 to i32
+ %arrayidx2.6 = getelementptr inbounds nuw i8, ptr %y, i64 6
+ %13 = load i8, ptr %arrayidx2.6
+ %conv3.6 = sext i8 %13 to i32
+ %mul.6 = mul nsw i32 %conv3.6, %conv.6
+ %add.6 = add nsw i32 %mul.6, %add.5
+ %arrayidx.7 = getelementptr inbounds nuw i8, ptr %x, i64 7
+ %14 = load i8, ptr %arrayidx.7
+ %conv.7 = sext i8 %14 to i32
+ %arrayidx2.7 = getelementptr inbounds nuw i8, ptr %y, i64 7
+ %15 = load i8, ptr %arrayidx2.7
+ %conv3.7 = sext i8 %15 to i32
+ %mul.7 = mul nsw i32 %conv3.7, %conv.7
+ %add.7 = add nsw i32 %mul.7, %add.6
+ ret i32 %add.7
+}
+
+define i32 @mla_v8i8_i32_no_dotprod(ptr %x, ptr %y) {
+; CHECK-LABEL: @mla_v8i8_i32_no_dotprod(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[X:%.*]], align 1
+; CHECK-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32
+; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[Y:%.*]], align 1
+; CHECK-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32
+; CHECK-NEXT: [[MUL:%.*]] = mul nsw i32 [[CONV3]], [[CONV]]
+; CHECK-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 1
+; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX_1]], align 1
+; CHECK-NEXT: [[CONV_1:%.*]] = sext i8 [[TMP2]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_1:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 1
+; CHECK-NEXT: [[TMP3:%.*]] = load i8, ptr [[ARRAYIDX2_1]], align 1
+; CHECK-NEXT: [[CONV3_1:%.*]] = sext i8 [[TMP3]] to i32
+; CHECK-NEXT: [[MUL_1:%.*]] = mul nsw i32 [[CONV3_1]], [[CONV_1]]
+; CHECK-NEXT: [[ADD_1:%.*]] = add nsw i32 [[MUL_1]], [[MUL]]
+; CHECK-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 2
+; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX_2]], align 1
+; CHECK-NEXT: [[CONV_2:%.*]] = sext i8 [[TMP4]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_2:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 2
+; CHECK-NEXT: [[TMP5:%.*]] = load i8, ptr [[ARRAYIDX2_2]], align 1
+; CHECK-NEXT: [[CONV3_2:%.*]] = sext i8 [[TMP5]] to i32
+; CHECK-NEXT: [[MUL_2:%.*]] = mul nsw i32 [[CONV3_2]], [[CONV_2]]
+; CHECK-NEXT: [[ADD_2:%.*]] = add nsw i32 [[MUL_2]], [[ADD_1]]
+; CHECK-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 3
+; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX_3]], align 1
+; CHECK-NEXT: [[CONV_3:%.*]] = sext i8 [[TMP6]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_3:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 3
+; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX2_3]], align 1
+; CHECK-NEXT: [[CONV3_3:%.*]] = sext i8 [[TMP7]] to i32
+; CHECK-NEXT: [[MUL_3:%.*]] = mul nsw i32 [[CONV3_3]], [[CONV_3]]
+; CHECK-NEXT: [[ADD_3:%.*]] = add nsw i32 [[MUL_3]], [[ADD_2]]
+; CHECK-NEXT: [[ARRAYIDX_4:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 4
+; CHECK-NEXT: [[TMP8:%.*]] = load i8, ptr [[ARRAYIDX_4]], align 1
+; CHECK-NEXT: [[CONV_4:%.*]] = sext i8 [[TMP8]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_4:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 4
+; CHECK-NEXT: [[TMP9:%.*]] = load i8, ptr [[ARRAYIDX2_4]], align 1
+; CHECK-NEXT: [[CONV3_4:%.*]] = sext i8 [[TMP9]] to i32
+; CHECK-NEXT: [[MUL_4:%.*]] = mul nsw i32 [[CONV3_4]], [[CONV_4]]
+; CHECK-NEXT: [[ADD_4:%.*]] = add nsw i32 [[MUL_4]], [[ADD_3]]
+; CHECK-NEXT: [[ARRAYIDX_5:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 5
+; CHECK-NEXT: [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX_5]], align 1
+; CHECK-NEXT: [[CONV_5:%.*]] = sext i8 [[TMP10]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_5:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 5
+; CHECK-NEXT: [[TMP11:%.*]] = load i8, ptr [[ARRAYIDX2_5]], align 1
+; CHECK-NEXT: [[CONV3_5:%.*]] = sext i8 [[TMP11]] to i32
+; CHECK-NEXT: [[MUL_5:%.*]] = mul nsw i32 [[CONV3_5]], [[CONV_5]]
+; CHECK-NEXT: [[ADD_5:%.*]] = add nsw i32 [[MUL_5]], [[ADD_4]]
+; CHECK-NEXT: [[ARRAYIDX_6:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 6
+; CHECK-NEXT: [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX_6]], align 1
+; CHECK-NEXT: [[CONV_6:%.*]] = sext i8 [[TMP12]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_6:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 6
+; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX2_6]], align 1
+; CHECK-NEXT: [[CONV3_6:%.*]] = sext i8 [[TMP13]] to i32
+; CHECK-NEXT: [[MUL_6:%.*]] = mul nsw i32 [[CONV3_6]], [[CONV_6]]
+; CHECK-NEXT: [[ADD_6:%.*]] = add nsw i32 [[MUL_6]], [[ADD_5]]
+; CHECK-NEXT: [[ARRAYIDX_7:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 7
+; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX_7]], align 1
+; CHECK-NEXT: [[CONV_7:%.*]] = sext i8 [[TMP14]] to i32
+; CHECK-NEXT: [[ARRAYIDX2_7:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 7
+; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX2_7]], align 1
+; CHECK-NEXT: [[CONV3_7:%.*]] = sext i8 [[TMP15]] to i32
+; CHECK-NEXT: [[MUL_7:%.*]] = mul nsw i32 [[CONV3_7]], [[CONV_7]]
+; CHECK-NEXT: [[ADD_7:%.*]] = add nsw i32 [[MUL_7]], [[ADD_6]]
+; CHECK-NEXT: ret i32 [[ADD_7]]
+;
+entry:
+ %0 = load i8, ptr %x
+ %conv = sext i8 %0 to i32
+ %1 = load i8, ptr %y
+ %conv3 = sext i8 %1 to i32
+ %mul = mul nsw i32 %conv3, %conv
+ %arrayidx.1 = getelementptr inbounds nuw i8, ptr %x, i64 1
+ %2 = load i8, ptr %arrayidx.1
+ %conv.1 = sext i8 %2 to i32
+ %arrayidx2.1 = getelementptr inbounds nuw i8, ptr %y, i64 1
+ %3 = load i8, ptr %arrayidx2.1
+ %conv3.1 = sext i8 %3 to i32
+ %mul.1 = mul nsw i32 %conv3.1, %conv.1
+ %add.1 = add nsw i32 %mul.1, %mul
+ %arrayidx.2 = getelementptr inbounds nuw i8, ptr %x, i64 2
+ %4 = load i8, ptr %arrayidx.2
+ %conv.2 = sext i8 %4 to i32
+ %arrayidx2.2 = getelementptr inbounds nuw i8, ptr %y, i64 2
+ %5 = load i8, ptr %arrayidx2.2
+ %conv3.2 = sext i8 %5 to i32
+ %mul.2 = mul nsw i32 %conv3.2, %conv.2
+ %add.2 = add nsw i32 %mul.2, %add.1
+ %arrayidx.3 = getelementptr inbounds nuw i8, ptr %x, i64 3
+ %6 = load i8, ptr %arrayidx.3
+ %conv.3 = sext i8 %6 to i32
+ %arrayidx2.3 = getelementptr inbounds nuw i8, ptr %y, i64 3
+ %7 = load i8, ptr %arrayidx2.3
+ %conv3.3 = sext i8 %7 to i32
+ %mul.3 = mul nsw i32 %conv3.3, %conv.3
+ %add.3 = add nsw i32 %mul.3, %add.2
+ %arrayidx.4 = getelementptr inbounds nuw i8, ptr %x, i64 4
+ %8 = load i8, ptr %arrayidx.4
+ %conv.4 = sext i8 %8 to i32
+ %arrayidx2.4 = getelementptr inbounds nuw i8, ptr %y, i64 4
+ %9 = load i8, ptr %arrayidx2.4
+ %conv3.4 = sext i8 %9 to i32
+ %mul.4 = mul nsw i32 %conv3.4, %conv.4
+ %add.4 = add nsw i32 %mul.4, %add.3
+ %arrayidx.5 = getelementptr inbounds nuw i8, ptr %x, i64 5
+ %10 = load i8, ptr %arrayidx.5
+ %conv.5 = sext i8 %10 to i32
+ %arrayidx2.5 = getelementptr inbounds nuw i8, ptr %y, i64 5
+ %11 = load i8, ptr %arrayidx2.5
+ %conv3.5 = sext i8 %11 to i32
+ %mul.5 = mul nsw i32 %conv3.5, %conv.5
+ %add.5 = add nsw i32 %mul.5, %add.4
+ %arrayidx.6 = getelementptr inbounds nuw i8, ptr %x, i64 6
+ %12 = load i8, ptr %arrayidx.6
+ %conv.6 = sext i8 %12 to i32
+ %arrayidx2.6 = getelementptr inbounds nuw i8, ptr %y, i64 6
+ %13 = load i8, ptr %arrayidx2.6
+ %conv3.6 = sext i8 %13 to i32
+ %mul.6 = mul nsw i32 %conv3.6, %conv.6
+ %add.6 = add nsw i32 %mul.6, %add.5
+ %arrayidx.7 = getelementptr inbounds nuw i8, ptr %x, i64 7
+ %14 = load i8, ptr %arrayidx.7
+ %conv.7 = sext i8 %14 to i32
+ %arrayidx2.7 = getelementptr inbounds nuw i8, ptr %y, i64 7
+ %15 = load i8, ptr %arrayidx2.7
+ %conv3.7 = sext i8 %15 to i32
+ %mul.7 = mul nsw i32 %conv3.7, %conv.7
+ %add.7 = add nsw i32 %mul.7, %add.6
+ ret i32 %add.7
+}
>From 35363ea9367bbada39b1d1ab97861a00777f6598 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Fri, 17 Jul 2026 09:15:50 -0700
Subject: [PATCH 2/4] [SLP] Cost reduce.add(mul(ext, ext)) as a dot-product
reduction
The SLP reduction cost only used getArithmeticReductionCost /
getExtendedReductionCost, so for reduce.add(mul(ext(A), ext(B))) the
multiply and extends were costed separately and the reduction was often
rejected as unprofitable, even though it lowers to a single dot-product
reduction (e.g. UDOT/SDOT) where the target supports it.
Recognize the multiply-accumulate pattern in getReductionCost and use
getMulAccReductionCost, subtracting the multiply and extend costs that
are already counted in the tree to avoid double counting (mirroring the
existing FMA handling and LoopVectorize's getReductionPatternCost). It is
gated on the fused cost being cheaper, so it only changes the dot-product
idiom on targets where it is profitable.
This patch offers 2-3% benefit on x264 under LTO.
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 79 +++++++++++++++++++
.../AArch64/reduce-add-dotprod.ll | 67 ++--------------
.../SLPVectorizer/AArch64/vecreduceadd.ll | 4 +-
3 files changed, 87 insertions(+), 63 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index e21dbce1ea547..84918725583ec 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32305,6 +32305,85 @@ class HorizontalReduction {
cast<VectorType>(getWidenedType(RType, ReduxWidth)), FMF,
CostKind);
}
+ // reduce.add(mul(ext(A), ext(B))) lowers to a single dot-product
+ // reduction (e.g. UDOT/SDOT) where the target supports it. Prefer
+ // that fused cost when it is cheaper. The extends and the multiply
+ // are already counted in the tree cost, so subtract them here to
+ // avoid double counting (mirrors the FMA handling below).
+ if (RdxKind == RecurKind::Add && !ReducedVals.empty()) {
+ Type *SrcElemTy = nullptr;
+ bool IsZExt = true;
+ bool SameOperands = true;
+ // Match one reduced lane as mul(ext(a), ext(b)) where both
+ // factors use the same widening extend. Reports the extend
+ // signedness, the pre-extension scalar type, and whether both
+ // factors are the same extend value (a single extend column).
+ auto MatchMulAccLane = [](Value *V, bool &ZExt, Type *&SrcTy,
+ bool &SharedExt) {
+ Value *E0, *E1, *A, *B;
+ if (match(V,
+ m_Mul(m_CombineAnd(m_Value(E0), m_ZExt(m_Value(A))),
+ m_CombineAnd(m_Value(E1), m_ZExt(m_Value(B))))))
+ ZExt = true;
+ else if (match(V, m_Mul(m_CombineAnd(m_Value(E0),
+ m_SExt(m_Value(A))),
+ m_CombineAnd(m_Value(E1),
+ m_SExt(m_Value(B))))))
+ ZExt = false;
+ else
+ return false;
+ if (A->getType() != B->getType())
+ return false;
+ SrcTy = A->getType()->getScalarType();
+ SharedExt = E0 == E1;
+ return true;
+ };
+ bool IsMulAcc = all_of(ReducedVals, [&](Value *RdxVal) {
+ bool ThisZExt;
+ Type *ThisSrcTy;
+ bool SharedExt;
+ if (!MatchMulAccLane(RdxVal, ThisZExt, ThisSrcTy, SharedExt))
+ return false;
+ if (!SharedExt)
+ SameOperands = false;
+ if (!SrcElemTy) {
+ SrcElemTy = ThisSrcTy;
+ IsZExt = ThisZExt;
+ return true;
+ }
+ return SrcElemTy == ThisSrcTy && IsZExt == ThisZExt;
+ });
+ // Only fuse when the multiply factors are actually vectorized
+ // (a live non-gather tree entry); otherwise the dot-product does
+ // not form and the fused cost would not apply.
+ if (IsMulAcc && SrcElemTy &&
+ R.isVectorized(ReducedVals.front())) {
+ auto *SrcVecTy =
+ cast<VectorType>(getWidenedType(SrcElemTy, ReduxWidth));
+ InstructionCost RedCost = TTI->getMulAccReductionCost(
+ IsZExt, RdxOpcode, RedTy, SrcVecTy, CostKind);
+ if (RedCost.isValid()) {
+ // Derive operand info and cast context from the actual
+ // mul(ext, ext) so the subtracted costs match the tree.
+ auto *Mul = cast<Instruction>(ReducedVals.front());
+ auto *Ext0 = cast<Instruction>(Mul->getOperand(0));
+ auto *Ext1 = cast<Instruction>(Mul->getOperand(1));
+ InstructionCost MulCost = TTI->getArithmeticInstrCost(
+ Instruction::Mul, VectorTy, CostKind,
+ TTI::getOperandInfo(Ext0), TTI::getOperandInfo(Ext1));
+ InstructionCost TreeExtCost = TTI->getCastInstrCost(
+ Ext0->getOpcode(), VectorTy, SrcVecTy,
+ TTI::getCastContextHint(Ext0), CostKind);
+ if (!SameOperands)
+ TreeExtCost += TTI->getCastInstrCost(
+ Ext1->getOpcode(), VectorTy, SrcVecTy,
+ TTI::getCastContextHint(Ext1), CostKind);
+ InstructionCost MulAccCost = RedCost - MulCost - TreeExtCost;
+ if (MulAccCost < VectorCost)
+ VectorCost = MulAccCost;
+ }
+ }
+ }
}
} else {
Type *RedTy = VectorTy->getElementType();
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
index 36d2a0f797280..da2a427c08cfe 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/reduce-add-dotprod.ll
@@ -13,67 +13,12 @@ target triple = "aarch64"
define i32 @mla_v8i8_i32_dotprod(ptr %x, ptr %y) "target-features"="+dotprod" {
; CHECK-LABEL: @mla_v8i8_i32_dotprod(
; CHECK-NEXT: entry:
-; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[X:%.*]], align 1
-; CHECK-NEXT: [[CONV:%.*]] = sext i8 [[TMP0]] to i32
-; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[Y:%.*]], align 1
-; CHECK-NEXT: [[CONV3:%.*]] = sext i8 [[TMP1]] to i32
-; CHECK-NEXT: [[MUL:%.*]] = mul nsw i32 [[CONV3]], [[CONV]]
-; CHECK-NEXT: [[ARRAYIDX_1:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 1
-; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX_1]], align 1
-; CHECK-NEXT: [[CONV_1:%.*]] = sext i8 [[TMP2]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_1:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 1
-; CHECK-NEXT: [[TMP3:%.*]] = load i8, ptr [[ARRAYIDX2_1]], align 1
-; CHECK-NEXT: [[CONV3_1:%.*]] = sext i8 [[TMP3]] to i32
-; CHECK-NEXT: [[MUL_1:%.*]] = mul nsw i32 [[CONV3_1]], [[CONV_1]]
-; CHECK-NEXT: [[ADD_1:%.*]] = add nsw i32 [[MUL_1]], [[MUL]]
-; CHECK-NEXT: [[ARRAYIDX_2:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 2
-; CHECK-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX_2]], align 1
-; CHECK-NEXT: [[CONV_2:%.*]] = sext i8 [[TMP4]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_2:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 2
-; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX2_2]], align 1
-; CHECK-NEXT: [[CONV3_2:%.*]] = sext i8 [[TMP16]] to i32
-; CHECK-NEXT: [[MUL_2:%.*]] = mul nsw i32 [[CONV3_2]], [[CONV_2]]
-; CHECK-NEXT: [[ADD_2:%.*]] = add nsw i32 [[MUL_2]], [[ADD_1]]
-; CHECK-NEXT: [[ARRAYIDX_3:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 3
-; CHECK-NEXT: [[TMP6:%.*]] = load i8, ptr [[ARRAYIDX_3]], align 1
-; CHECK-NEXT: [[CONV_3:%.*]] = sext i8 [[TMP6]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_3:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 3
-; CHECK-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX2_3]], align 1
-; CHECK-NEXT: [[CONV3_3:%.*]] = sext i8 [[TMP7]] to i32
-; CHECK-NEXT: [[MUL_3:%.*]] = mul nsw i32 [[CONV3_3]], [[CONV_3]]
-; CHECK-NEXT: [[ADD_3:%.*]] = add nsw i32 [[MUL_3]], [[ADD_2]]
-; CHECK-NEXT: [[ARRAYIDX_4:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 4
-; CHECK-NEXT: [[TMP8:%.*]] = load i8, ptr [[ARRAYIDX_4]], align 1
-; CHECK-NEXT: [[CONV_4:%.*]] = sext i8 [[TMP8]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_4:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 4
-; CHECK-NEXT: [[TMP9:%.*]] = load i8, ptr [[ARRAYIDX2_4]], align 1
-; CHECK-NEXT: [[CONV3_4:%.*]] = sext i8 [[TMP9]] to i32
-; CHECK-NEXT: [[MUL_4:%.*]] = mul nsw i32 [[CONV3_4]], [[CONV_4]]
-; CHECK-NEXT: [[ADD_4:%.*]] = add nsw i32 [[MUL_4]], [[ADD_3]]
-; CHECK-NEXT: [[ARRAYIDX_5:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 5
-; CHECK-NEXT: [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX_5]], align 1
-; CHECK-NEXT: [[CONV_5:%.*]] = sext i8 [[TMP10]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_5:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 5
-; CHECK-NEXT: [[TMP11:%.*]] = load i8, ptr [[ARRAYIDX2_5]], align 1
-; CHECK-NEXT: [[CONV3_5:%.*]] = sext i8 [[TMP11]] to i32
-; CHECK-NEXT: [[MUL_5:%.*]] = mul nsw i32 [[CONV3_5]], [[CONV_5]]
-; CHECK-NEXT: [[ADD_5:%.*]] = add nsw i32 [[MUL_5]], [[ADD_4]]
-; CHECK-NEXT: [[ARRAYIDX_6:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 6
-; CHECK-NEXT: [[TMP12:%.*]] = load i8, ptr [[ARRAYIDX_6]], align 1
-; CHECK-NEXT: [[CONV_6:%.*]] = sext i8 [[TMP12]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_6:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 6
-; CHECK-NEXT: [[TMP13:%.*]] = load i8, ptr [[ARRAYIDX2_6]], align 1
-; CHECK-NEXT: [[CONV3_6:%.*]] = sext i8 [[TMP13]] to i32
-; CHECK-NEXT: [[MUL_6:%.*]] = mul nsw i32 [[CONV3_6]], [[CONV_6]]
-; CHECK-NEXT: [[ADD_6:%.*]] = add nsw i32 [[MUL_6]], [[ADD_5]]
-; CHECK-NEXT: [[ARRAYIDX_7:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 7
-; CHECK-NEXT: [[TMP14:%.*]] = load i8, ptr [[ARRAYIDX_7]], align 1
-; CHECK-NEXT: [[CONV_7:%.*]] = sext i8 [[TMP14]] to i32
-; CHECK-NEXT: [[ARRAYIDX2_7:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 7
-; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX2_7]], align 1
-; CHECK-NEXT: [[CONV3_7:%.*]] = sext i8 [[TMP15]] to i32
-; CHECK-NEXT: [[MUL_7:%.*]] = mul nsw i32 [[CONV3_7]], [[CONV_7]]
-; CHECK-NEXT: [[TMP5:%.*]] = add nsw i32 [[MUL_7]], [[ADD_6]]
+; CHECK-NEXT: [[TMP0:%.*]] = load <8 x i8>, ptr [[X:%.*]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = sext <8 x i8> [[TMP0]] to <8 x i32>
+; CHECK-NEXT: [[TMP2:%.*]] = load <8 x i8>, ptr [[Y:%.*]], align 1
+; CHECK-NEXT: [[TMP3:%.*]] = sext <8 x i8> [[TMP2]] to <8 x i32>
+; CHECK-NEXT: [[TMP4:%.*]] = mul nsw <8 x i32> [[TMP3]], [[TMP1]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP4]])
; CHECK-NEXT: ret i32 [[TMP5]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/vecreduceadd.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/vecreduceadd.ll
index 577efcbbac012..16c44c00d2ad9 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/vecreduceadd.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/vecreduceadd.ll
@@ -930,7 +930,7 @@ entry:
; COST-LABEL: Function: mla_v8i8_i32
-; COST: Cost: '-24'
+; COST: Cost: '-32'
define i32 @mla_v8i8_i32(ptr %x, ptr %y) "target-features"="+dotprod" {
; CHECK-LABEL: @mla_v8i8_i32(
; CHECK-NEXT: entry:
@@ -1009,7 +1009,7 @@ entry:
; COST-LABEL: Function: mla_v16i8_i32
-; COST: Cost: '-52'
+; COST: Cost: '-70'
define i32 @mla_v16i8_i32(ptr %x, ptr %y) "target-features"="+dotprod" {
; CHECK-LABEL: @mla_v16i8_i32(
; CHECK-NEXT: entry:
>From 275636d94a74ae9a797fc3e53a58e01021d49709 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Thu, 17 Sep 2026 08:47:37 -0700
Subject: [PATCH 3/4] fixup! [SLP] Cost reduce.add(mul(ext, ext)) as a
dot-product reduction
---
llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 7 ++++++-
1 file changed, 6 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 84918725583ec..3b438002a51bc 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -3385,6 +3385,7 @@ class slpvectorizer::BoUpSLP {
#endif
/// Get list of vector entries, associated with the value \p V.
+public:
ArrayRef<TreeEntry *> getTreeEntries(const Value *V) const {
assert(V && "V cannot be nullptr.");
auto It = ScalarToTreeEntries.find(V);
@@ -3393,6 +3394,7 @@ class slpvectorizer::BoUpSLP {
return It->getSecond();
}
+private:
/// Get list of split vector entries, associated with the value \p V.
ArrayRef<TreeEntry *> getSplitTreeEntries(Value *V) const {
assert(V && "V cannot be nullptr.");
@@ -32357,7 +32359,10 @@ class HorizontalReduction {
// (a live non-gather tree entry); otherwise the dot-product does
// not form and the fused cost would not apply.
if (IsMulAcc && SrcElemTy &&
- R.isVectorized(ReducedVals.front())) {
+ any_of(R.getTreeEntries(ReducedVals.front()),
+ [](const auto *TE) {
+ return TE->Idx == 0 && !TE->isGather();
+ })) {
auto *SrcVecTy =
cast<VectorType>(getWidenedType(SrcElemTy, ReduxWidth));
InstructionCost RedCost = TTI->getMulAccReductionCost(
>From 552f83eac23862689d02a2c122cd6aad37ecae51 Mon Sep 17 00:00:00 2001
From: Madhur Amilkanthwar <madhura at nvidia.com>
Date: Thu, 17 Sep 2026 08:57:17 -0700
Subject: [PATCH 4/4] fixup! [SLP] Cost reduce.add(mul(ext, ext)) as a
dot-product reduction
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 51 ++++++++++++-------
1 file changed, 32 insertions(+), 19 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3b438002a51bc..4bfc2decbd1ef 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -2449,6 +2449,7 @@ class slpvectorizer::BoUpSLP {
/// getNumScalarInsts().
uint64_t getNumVectorInsts(bool HasTreeLoop);
+public:
/// Return information about the vector formed for the specified index
/// of a vector of (the same) instruction.
TargetTransformInfo::OperandValueInfo
@@ -2461,15 +2462,16 @@ class slpvectorizer::BoUpSLP {
getOperandEntry(const_cast<const TreeEntry *>(E), Idx));
}
+ /// \returns Cast context for the given graph node.
+ TargetTransformInfo::CastContextHint
+ getCastContextHint(const TreeEntry &TE) const;
+
+private:
/// Gets the root instruction for the given node. If the node is a strided
/// load/store node with the reverse order, the root instruction is the last
/// one.
Instruction *getRootEntryInstruction(const TreeEntry &Entry) const;
- /// \returns Cast context for the given graph node.
- TargetTransformInfo::CastContextHint
- getCastContextHint(const TreeEntry &TE) const;
-
/// \returns the scale of the given tree entry to the loop iteration.
/// \p Scalar is the scalar value from the entry, if using the parent for the
/// external use.
@@ -32355,34 +32357,45 @@ class HorizontalReduction {
}
return SrcElemTy == ThisSrcTy && IsZExt == ThisZExt;
});
- // Only fuse when the multiply factors are actually vectorized
- // (a live non-gather tree entry); otherwise the dot-product does
- // not form and the fused cost would not apply.
- if (IsMulAcc && SrcElemTy &&
- any_of(R.getTreeEntries(ReducedVals.front()),
- [](const auto *TE) {
- return TE->Idx == 0 && !TE->isGather();
- })) {
+ // Only fuse when the multiply factors are vectorized as the
+ // root non-gather tree entry; otherwise the dot-product does not
+ // form and the fused cost would not apply.
+ ArrayRef MulTEs = R.getTreeEntries(ReducedVals.front());
+ auto MulTEIt = find_if(MulTEs, [](const auto *TE) {
+ return TE->Idx == 0 && !TE->isGather();
+ });
+ if (IsMulAcc && SrcElemTy && MulTEIt != MulTEs.end()) {
+ const auto *MulTE = *MulTEIt;
auto *SrcVecTy =
cast<VectorType>(getWidenedType(SrcElemTy, ReduxWidth));
InstructionCost RedCost = TTI->getMulAccReductionCost(
IsZExt, RdxOpcode, RedTy, SrcVecTy, CostKind);
if (RedCost.isValid()) {
- // Derive operand info and cast context from the actual
- // mul(ext, ext) so the subtracted costs match the tree.
auto *Mul = cast<Instruction>(ReducedVals.front());
auto *Ext0 = cast<Instruction>(Mul->getOperand(0));
auto *Ext1 = cast<Instruction>(Mul->getOperand(1));
+ // Cast context of an extend comes from its operand node when
+ // the extend is itself a single vectorized entry.
+ auto GetExtCCH = [&](Instruction *Ext) {
+ if (ArrayRef ExtTEs = R.getTreeEntries(Ext);
+ ExtTEs.size() == 1)
+ return R.getCastContextHint(
+ *R.getOperandEntry(ExtTEs.front(), 0));
+ return TTI::getCastContextHint(Ext);
+ };
+ // Operand info from the multiply's tree-entry operands so it
+ // reflects all lanes, not a single scalar.
InstructionCost MulCost = TTI->getArithmeticInstrCost(
Instruction::Mul, VectorTy, CostKind,
- TTI::getOperandInfo(Ext0), TTI::getOperandInfo(Ext1));
+ R.getOperandInfo(MulTE->getOperand(0)),
+ R.getOperandInfo(MulTE->getOperand(1)));
InstructionCost TreeExtCost = TTI->getCastInstrCost(
- Ext0->getOpcode(), VectorTy, SrcVecTy,
- TTI::getCastContextHint(Ext0), CostKind);
+ Ext0->getOpcode(), VectorTy, SrcVecTy, GetExtCCH(Ext0),
+ CostKind);
if (!SameOperands)
TreeExtCost += TTI->getCastInstrCost(
- Ext1->getOpcode(), VectorTy, SrcVecTy,
- TTI::getCastContextHint(Ext1), CostKind);
+ Ext1->getOpcode(), VectorTy, SrcVecTy, GetExtCCH(Ext1),
+ CostKind);
InstructionCost MulAccCost = RedCost - MulCost - TreeExtCost;
if (MulAccCost < VectorCost)
VectorCost = MulAccCost;
More information about the llvm-commits
mailing list