[llvm] [VectorCombine] Allow equal-cost scalarization for single-use loads (PR #218340)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 24 00:55:01 PDT 2026
https://github.com/ParkHanbum created https://github.com/llvm/llvm-project/pull/218340
Allow scalarizeLoadExtract to scalarize a single-use vector load when
the scalarized cost is equal to the original cost.
For a single extract user, this does not increase the number of memory
operations and narrows the memory access, which can expose further
optimizations such as store-to-load forwarding and DSE.
Keep requiring a strict cost improvement for loads with multiple users.
Fixes https://github.com/llvm/llvm-project/issues/217598
>From 8ebbd237888f1c54a98594c9129a3ef89e507024 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 24 Aug 2026 16:33:03 +0900
Subject: [PATCH 1/2] add testcases for upcoming patch
---
.../test/Transforms/VectorCombine/X86/load.ll | 19 +++++++++++++++++++
1 file changed, 19 insertions(+)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load.ll b/llvm/test/Transforms/VectorCombine/X86/load.ll
index 39038cab48aba..b70378710ecab 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load.ll
@@ -622,3 +622,22 @@ define <2 x i64> @PR30986(ptr %0) {
ret <2 x i64> %8
}
declare i64 @llvm.ctpop.i64(i64)
+
+define float @PR217598(ptr %p, i32 %bits) {
+; CHECK-LABEL: @PR217598(
+; CHECK-NEXT: [[A:%.*]] = alloca [20 x i8], align 16
+; CHECK-NEXT: store i32 [[BITS:%.*]], ptr [[A]], align 4
+; CHECK-NEXT: [[Q:%.*]] = getelementptr i8, ptr [[A]], i64 12
+; CHECK-NEXT: store ptr [[P:%.*]], ptr [[Q]], align 4
+; CHECK-NEXT: [[V:%.*]] = load <4 x float>, ptr [[A]], align 4
+; CHECK-NEXT: [[X:%.*]] = extractelement <4 x float> [[V]], i64 0
+; CHECK-NEXT: ret float [[X]]
+;
+ %a = alloca [20 x i8], align 16
+ store i32 %bits, ptr %a, align 4
+ %q = getelementptr i8, ptr %a, i64 12
+ store ptr %p, ptr %q, align 4
+ %v = load <4 x float>, ptr %a, align 4
+ %x = extractelement <4 x float> %v, i64 0
+ ret float %x
+}
>From c8865b2d5d1295979ab8c7fefb578577e16c7767 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 24 Aug 2026 16:35:26 +0900
Subject: [PATCH 2/2] [VectorCombine] Allow equal-cost scalarization for
single-use loads
Allow scalarizeLoadExtract to scalarize a single-use vector load when
the scalarized cost is equal to the original cost.
For a single extract user, this does not increase the number of memory
operations and narrows the memory access, which can expose further
optimizations such as store-to-load forwarding and DSE.
Keep requiring a strict cost improvement for loads with multiple users.
Fixes llvm#217598
---
.../Transforms/Vectorize/VectorCombine.cpp | 4 +-
.../Transforms/PhaseOrdering/X86/pr217598.ll | 115 ++++++++++++++++++
.../VectorCombine/X86/load-inseltpoison.ll | 12 +-
.../test/Transforms/VectorCombine/X86/load.ll | 42 +++----
4 files changed, 142 insertions(+), 31 deletions(-)
create mode 100644 llvm/test/Transforms/PhaseOrdering/X86/pr217598.ll
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index b3a545f388d17..c4dc5fade6799 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -2118,7 +2118,9 @@ bool VectorCombine::scalarizeLoadExtract(LoadInst *LI, VectorType *VecTy,
<< "\n LoadExtractCost: " << OriginalCost
<< " vs ScalarizedCost: " << ScalarizedCost << "\n");
- if (ScalarizedCost >= OriginalCost)
+ if (ScalarizedCost > OriginalCost)
+ return false;
+ if (ScalarizedCost == OriginalCost && !LI->hasOneUse())
return false;
// Ensure we add the load back to the worklist BEFORE its users so they can
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/pr217598.ll b/llvm/test/Transforms/PhaseOrdering/X86/pr217598.ll
new file mode 100644
index 0000000000000..dc3f32942ee79
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/X86/pr217598.ll
@@ -0,0 +1,115 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -S -passes='default<O2>' < %s | FileCheck %s
+
+target datalayout = "e-p:64:64"
+target triple = "x86_64-unknown-linux-gnu"
+
+; For the cases below, the X86 cost model considers the original vector load
+; plus its lane-zero extract to have the same cost as a scalar load. The other
+; store overlaps the vector access but not the scalar access. Scalarizing the
+; single-use load therefore lets later passes forward the scalar store and
+; eliminate the remaining stack traffic.
+
+; Original reproducer: forward an integer store through a floating-point load.
+define float @PR217598(ptr %p, i32 %bits) {
+; CHECK-LABEL: @PR217598(
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast i32 [[BITS:%.*]] to float
+; CHECK-NEXT: ret float [[TMP1]]
+;
+ %a = alloca [20 x i8], align 16
+ store i32 %bits, ptr %a, align 4
+ %q = getelementptr i8, ptr %a, i64 12
+ store ptr %p, ptr %q, align 4
+ %v = load <4 x float>, ptr %a, align 4
+ %x = extractelement <4 x float> %v, i64 0
+ ret float %x
+}
+
+; Forward a store with the same type as the extracted element.
+define float @forward_same_type(ptr %p, float %value) {
+; CHECK-LABEL: @forward_same_type(
+; CHECK-NEXT: ret float [[VALUE:%.*]]
+;
+ %a = alloca [16 x i8], align 16
+ store float %value, ptr %a, align 4
+ %q = getelementptr inbounds i8, ptr %a, i64 8
+ store ptr %p, ptr %q, align 8
+ %v = load <4 x float>, ptr %a, align 4
+ %x = extractelement <4 x float> %v, i64 0
+ ret float %x
+}
+
+; Forward from a vector load whose base is at a nonzero alloca offset, and
+; preserve the integer-to-double bit interpretation.
+define double @forward_from_offset(ptr %p, i64 %bits) {
+; CHECK-LABEL: @forward_from_offset(
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast i64 [[BITS:%.*]] to double
+; CHECK-NEXT: ret double [[TMP1]]
+;
+ %a = alloca [32 x i8], align 16
+ %base = getelementptr inbounds i8, ptr %a, i64 8
+ store i64 %bits, ptr %base, align 8
+ %q = getelementptr inbounds i8, ptr %base, i64 8
+ store ptr %p, ptr %q, align 8
+ %v = load <2 x double>, ptr %base, align 8
+ %x = extractelement <2 x double> %v, i64 0
+ ret double %x
+}
+
+; Promote stores from different predecessors after scalarization.
+define float @forward_conditional_store(ptr %p, i1 %cond, float %x, float %y) {
+; CHECK-LABEL: @forward_conditional_store(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[X_Y:%.*]] = select i1 [[COND:%.*]], float [[X:%.*]], float [[Y:%.*]]
+; CHECK-NEXT: ret float [[X_Y]]
+;
+entry:
+ %a = alloca [16 x i8], align 16
+ br i1 %cond, label %then, label %else
+
+then:
+ store float %x, ptr %a, align 4
+ br label %merge
+
+else:
+ store float %y, ptr %a, align 4
+ br label %merge
+
+merge:
+ %q = getelementptr inbounds i8, ptr %a, i64 8
+ store ptr %p, ptr %q, align 8
+ %v = load <4 x float>, ptr %a, align 4
+ %value = extractelement <4 x float> %v, i64 0
+ ret float %value
+}
+
+; Exercise a vector shorter than the native 128-bit vector width. The pointer
+; store only partially overlaps the original vector load.
+define float @forward_v2f32(ptr %p, i32 %bits) {
+; CHECK-LABEL: @forward_v2f32(
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast i32 [[BITS:%.*]] to float
+; CHECK-NEXT: ret float [[TMP1]]
+;
+ %a = alloca [12 x i8], align 8
+ store i32 %bits, ptr %a, align 4
+ %q = getelementptr inbounds i8, ptr %a, i64 4
+ store ptr %p, ptr %q, align 4
+ %v = load <2 x float>, ptr %a, align 4
+ %x = extractelement <2 x float> %v, i64 0
+ ret float %x
+}
+
+; Exercise a narrower floating-point element type.
+define half @forward_f16(ptr %p, i16 %bits) {
+; CHECK-LABEL: @forward_f16(
+; CHECK-NEXT: [[TMP1:%.*]] = bitcast i16 [[BITS:%.*]] to half
+; CHECK-NEXT: ret half [[TMP1]]
+;
+ %a = alloca [16 x i8], align 16
+ store i16 %bits, ptr %a, align 2
+ %q = getelementptr inbounds i8, ptr %a, i64 8
+ store ptr %p, ptr %q, align 8
+ %v = load <8 x half>, ptr %a, align 2
+ %x = extractelement <8 x half> %v, i64 0
+ ret half %x
+}
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index f35b2ea319d23..f0a5d6d84e450 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -1,6 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=sse2 | FileCheck %s
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=avx2 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
@@ -633,8 +635,7 @@ define <4 x i32> @load_i32_insert_v4i32_asan(ptr align 16 dereferenceable(16) %p
define <4 x float> @load_v2f32_extract_insert_v4f32_hwasan(ptr align 16 dereferenceable(16) %p) nofree nosync sanitize_hwaddress {
; CHECK-LABEL: @load_v2f32_extract_insert_v4f32_hwasan(
-; CHECK-NEXT: [[L:%.*]] = load <2 x float>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[S:%.*]] = extractelement <2 x float> [[L]], i32 0
+; CHECK-NEXT: [[S:%.*]] = load float, ptr [[P:%.*]], align 4
; CHECK-NEXT: [[R:%.*]] = insertelement <4 x float> poison, float [[S]], i32 0
; CHECK-NEXT: ret <4 x float> [[R]]
;
@@ -646,8 +647,7 @@ define <4 x float> @load_v2f32_extract_insert_v4f32_hwasan(ptr align 16 derefere
define <4 x float> @load_v2f32_extract_insert_v4f32_tsan(ptr align 16 dereferenceable(16) %p) nofree nosync sanitize_thread {
; CHECK-LABEL: @load_v2f32_extract_insert_v4f32_tsan(
-; CHECK-NEXT: [[L:%.*]] = load <2 x float>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[S:%.*]] = extractelement <2 x float> [[L]], i32 0
+; CHECK-NEXT: [[S:%.*]] = load float, ptr [[P:%.*]], align 4
; CHECK-NEXT: [[R:%.*]] = insertelement <4 x float> poison, float [[S]], i32 0
; CHECK-NEXT: ret <4 x float> [[R]]
;
diff --git a/llvm/test/Transforms/VectorCombine/X86/load.ll b/llvm/test/Transforms/VectorCombine/X86/load.ll
index b70378710ecab..f0daaa873c4a3 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load.ll
@@ -1,6 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=sse2 | FileCheck %s --check-prefixes=CHECK,SSE2
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=avx2 | FileCheck %s --check-prefixes=CHECK,AVX2
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v2 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v3 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mcpu=x86-64-v4 | FileCheck %s
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
@@ -540,8 +542,7 @@ define void @PR47558_multiple_use_load(ptr nocapture nonnull %resultptr, ptr noc
define <4 x float> @load_v2f32_extract_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_v2f32_extract_insert_v4f32(
-; CHECK-NEXT: [[L:%.*]] = load <2 x float>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[S:%.*]] = extractelement <2 x float> [[L]], i32 0
+; CHECK-NEXT: [[S:%.*]] = load float, ptr [[P:%.*]], align 4
; CHECK-NEXT: [[R:%.*]] = insertelement <4 x float> undef, float [[S]], i32 0
; CHECK-NEXT: ret <4 x float> [[R]]
;
@@ -552,16 +553,10 @@ define <4 x float> @load_v2f32_extract_insert_v4f32(ptr align 16 dereferenceable
}
define <4 x float> @load_v8f32_extract_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
-; SSE2-LABEL: @load_v8f32_extract_insert_v4f32(
-; SSE2-NEXT: [[S:%.*]] = load float, ptr [[TMP1:%.*]], align 4
-; SSE2-NEXT: [[R:%.*]] = insertelement <4 x float> undef, float [[S]], i32 0
-; SSE2-NEXT: ret <4 x float> [[R]]
-;
-; AVX2-LABEL: @load_v8f32_extract_insert_v4f32(
-; AVX2-NEXT: [[L:%.*]] = load <8 x float>, ptr [[P:%.*]], align 4
-; AVX2-NEXT: [[S:%.*]] = extractelement <8 x float> [[L]], i32 0
-; AVX2-NEXT: [[R:%.*]] = insertelement <4 x float> undef, float [[S]], i32 0
-; AVX2-NEXT: ret <4 x float> [[R]]
+; CHECK-LABEL: @load_v8f32_extract_insert_v4f32(
+; CHECK-NEXT: [[S:%.*]] = load float, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[R:%.*]] = insertelement <4 x float> undef, float [[S]], i32 0
+; CHECK-NEXT: ret <4 x float> [[R]]
;
%l = load <8 x float>, ptr %p, align 4
%s = extractelement <8 x float> %l, i32 0
@@ -603,14 +598,14 @@ define <8 x i16> @gep1_load_v2i16_extract_insert_v8i16(ptr align 1 dereferenceab
; PR30986 - split vector loads for scalarized operations
define <2 x i64> @PR30986(ptr %0) {
; CHECK-LABEL: @PR30986(
-; CHECK-NEXT: [[TMP3:%.*]] = load i64, ptr [[TMP2:%.*]], align 16
-; CHECK-NEXT: [[TMP4:%.*]] = tail call i64 @llvm.ctpop.i64(i64 [[TMP3]])
-; CHECK-NEXT: [[TMP5:%.*]] = insertelement <2 x i64> undef, i64 [[TMP4]], i32 0
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds <2 x i64>, ptr [[TMP2]], i64 0, i64 1
-; CHECK-NEXT: [[TMP7:%.*]] = load i64, ptr [[TMP6]], align 8
-; CHECK-NEXT: [[TMP8:%.*]] = tail call i64 @llvm.ctpop.i64(i64 [[TMP7]])
-; CHECK-NEXT: [[TMP9:%.*]] = insertelement <2 x i64> [[TMP5]], i64 [[TMP8]], i32 1
-; CHECK-NEXT: ret <2 x i64> [[TMP9]]
+; CHECK-NEXT: [[TMP2:%.*]] = load i64, ptr [[TMP0:%.*]], align 16
+; CHECK-NEXT: [[TMP3:%.*]] = tail call i64 @llvm.ctpop.i64(i64 [[TMP2]])
+; CHECK-NEXT: [[TMP4:%.*]] = insertelement <2 x i64> undef, i64 [[TMP3]], i32 0
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds <2 x i64>, ptr [[TMP0]], i64 0, i64 1
+; CHECK-NEXT: [[TMP6:%.*]] = load i64, ptr [[TMP5]], align 8
+; CHECK-NEXT: [[TMP7:%.*]] = tail call i64 @llvm.ctpop.i64(i64 [[TMP6]])
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <2 x i64> [[TMP4]], i64 [[TMP7]], i32 1
+; CHECK-NEXT: ret <2 x i64> [[TMP8]]
;
%2 = load <2 x i64>, ptr %0, align 16
%3 = extractelement <2 x i64> %2, i32 0
@@ -629,8 +624,7 @@ define float @PR217598(ptr %p, i32 %bits) {
; CHECK-NEXT: store i32 [[BITS:%.*]], ptr [[A]], align 4
; CHECK-NEXT: [[Q:%.*]] = getelementptr i8, ptr [[A]], i64 12
; CHECK-NEXT: store ptr [[P:%.*]], ptr [[Q]], align 4
-; CHECK-NEXT: [[V:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT: [[X:%.*]] = extractelement <4 x float> [[V]], i64 0
+; CHECK-NEXT: [[X:%.*]] = load float, ptr [[A]], align 4
; CHECK-NEXT: ret float [[X]]
;
%a = alloca [20 x i8], align 16
More information about the llvm-commits
mailing list