[llvm] [VectorCombine] Allow shuffling with bitcast for not multiple offset for loadsize (PR #119139)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 23 23:57:00 PDT 2026
https://github.com/ParkHanbum updated https://github.com/llvm/llvm-project/pull/119139
>From 91c1ca12b53cedd119dacb1541edcd648d0fc248 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sat, 7 Dec 2024 07:00:41 +0900
Subject: [PATCH 01/24] add test-cases
---
.../VectorCombine/X86/load-inseltpoison.ll | 82 ++++++++++++++++++-
1 file changed, 80 insertions(+), 2 deletions(-)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index 8cd99bbf31a8c..702be5a72afed 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -294,8 +294,8 @@ define <8 x i16> @gep01_load_i16_insert_v8i16_deref_minalign(ptr align 2 derefer
; must be a multiple of element size.
; TODO: Could bitcast around this limitation.
-define <4 x i32> @gep01_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(16) %p) nofree nosync {
-; CHECK-LABEL: @gep01_bitcast_load_i32_insert_v4i32(
+define <4 x i32> @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
@@ -307,6 +307,84 @@ define <4 x i32> @gep01_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceabl
ret <4 x i32> %r
}
+define <2 x i64> @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
+; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; CHECK-NEXT: ret <2 x i64> [[R]]
+;
+ %gep = getelementptr inbounds <16 x i8>, ptr %p, i64 0, i64 1
+ %s = load i64, ptr %gep, align 1
+ %r = insertelement <2 x i64> poison, i64 %s, i64 0
+ ret <2 x i64> %r
+}
+
+define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 11
+; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+ %gep = getelementptr inbounds <16 x i8>, ptr %p, i64 0, i64 11
+ %s = load i32, ptr %gep, align 1
+ %r = insertelement <4 x i32> poison, i32 %s, i64 0
+ ret <4 x i32> %r
+}
+
+define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
+; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+ %gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
+ %s = load i32, ptr %gep, align 1
+ %r = insertelement <4 x i32> poison, i32 %s, i64 0
+ ret <4 x i32> %r
+}
+
+define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
+; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; CHECK-NEXT: ret <2 x i64> [[R]]
+;
+ %gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
+ %s = load i64, ptr %gep, align 1
+ %r = insertelement <2 x i64> poison, i64 %s, i64 0
+ ret <2 x i64> %r
+}
+
+define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 5
+; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; CHECK-NEXT: ret <4 x i32> [[R]]
+;
+ %gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 5
+ %s = load i32, ptr %gep, align 1
+ %r = insertelement <4 x i32> poison, i32 %s, i64 0
+ ret <4 x i32> %r
+}
+
+define <2 x i64> @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(ptr align 1 dereferenceable(16) %p) nofree nosync {
+; CHECK-LABEL: @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[P:%.*]], i64 0, i64 1
+; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; CHECK-NEXT: ret <2 x i64> [[R]]
+;
+ %gep = getelementptr inbounds <4 x i32>, ptr %p, i64 0, i64 1
+ %s = load i64, ptr %gep, align 1
+ %r = insertelement <2 x i64> poison, i64 %s, i64 0
+ ret <2 x i64> %r
+}
+
define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
; CHECK-LABEL: @gep012_bitcast_load_i32_insert_v4i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 1
>From ad3c3c36138e419d39a4bc241285c1b0c0f38d8d Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Sat, 7 Dec 2024 06:00:08 +0900
Subject: [PATCH 02/24] [VectorCombine] Allow shuffling with bitcast for not
multiple offset for loadsize
Previously, vectorization for load-insert failed when the Offset was not
a multiple of the Load type size.
This patch allow it in two steps,
1. Vectorize it using a common multiple of Offset and LoadSize.
2. Bitcast to fit
Alive2: https://alive2.llvm.org/ce/z/Kgr9HQ
---
.../Transforms/Vectorize/VectorCombine.cpp | 74 +++++++++---
.../VectorCombine/X86/load-inseltpoison.ll | 108 ++++++++++++------
2 files changed, 129 insertions(+), 53 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index e3c45a8993819..313e4d6f5c7e9 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -258,6 +258,15 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (!canWidenLoad(Load, TTI))
return false;
+ auto MaxCommonDivisor = [](int n) {
+ if (n % 4 == 0)
+ return 4;
+ if (n % 2 == 0)
+ return 2;
+ else
+ return 1;
+ };
+
Type *ScalarTy = Scalar->getType();
uint64_t ScalarSize = ScalarTy->getPrimitiveSizeInBits();
unsigned MinVectorSize = TTI.getMinVectorRegisterBitWidth();
@@ -272,6 +281,8 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
unsigned MinVecNumElts = MinVectorSize / ScalarSize;
auto *MinVecTy = VectorType::get(ScalarTy, MinVecNumElts, false);
unsigned OffsetEltIndex = 0;
+ unsigned VectorRange = 0;
+ bool NeedCast = false;
Align Alignment = Load->getAlign();
if (!isSafeToLoadUnconditionally(SrcPtr, MinVecTy, Align(1), *DL, Load, SQ.AC,
SQ.DT)) {
@@ -288,15 +299,27 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (Offset.isNegative())
return false;
- // The offset must be a multiple of the scalar element to shuffle cleanly
- // in the element's size.
+ // If Offset is multiple of a Scalar element, it can be shuffled to the
+ // element's size; otherwise, Offset and Scalar must be shuffled to the
+ // appropriate element size for both.
uint64_t ScalarSizeInBytes = ScalarSize / 8;
- if (Offset.urem(ScalarSizeInBytes) != 0)
- return false;
+ if (auto UnalignedBytes = Offset.urem(ScalarSizeInBytes);
+ UnalignedBytes != 0) {
+ uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
+ // Assign the greatest common divisor between UnalignedBytes and Offset to
+ // ScalarSizeInBytes
+ ScalarSizeInBytes = MaxCommonDivisor(UnalignedBytes);
+ ScalarSize = ScalarSizeInBytes * 8;
+ VectorRange = OldScalarSizeInBytes / ScalarSizeInBytes;
+ MinVecNumElts = MinVectorSize / ScalarSize;
+ ScalarTy = Type::getIntNTy(I.getContext(), ScalarSize);
+ MinVecTy = VectorType::get(ScalarTy, MinVecNumElts, false);
+ NeedCast = true;
+ }
// If we load MinVecNumElts, will our target element still be loaded?
APInt OffsetEltIndexAP = Offset.udiv(ScalarSizeInBytes);
- if (OffsetEltIndexAP.uge(MinVecNumElts))
+ if ((OffsetEltIndexAP + VectorRange).uge(MinVecNumElts))
return false;
OffsetEltIndex = OffsetEltIndexAP.getZExtValue();
@@ -315,11 +338,14 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Alignment = std::max(SrcPtr->getPointerAlignment(*DL), Alignment);
Type *LoadTy = Load->getType();
unsigned AS = Load->getPointerAddressSpace();
+ auto VecTy = cast<InsertElementInst>(&I)->getType();
+
InstructionCost OldCost =
TTI.getMemoryOpCost(Instruction::Load, LoadTy, Alignment, AS, CostKind);
- APInt DemandedElts = APInt::getOneBitSet(MinVecNumElts, 0);
+ APInt DemandedElts =
+ APInt::getOneBitSet(VecTy->getElementCount().getFixedValue(), 0);
OldCost +=
- TTI.getScalarizationOverhead(MinVecTy, DemandedElts,
+ TTI.getScalarizationOverhead(VecTy, DemandedElts,
/* Insert */ true, HasExtract, CostKind);
// New pattern: load VecPtr
@@ -332,15 +358,29 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// We assume this operation has no cost in codegen if there was no offset.
// Note that we could use freeze to avoid poison problems, but then we might
// still need a shuffle to change the vector size.
- auto *Ty = cast<FixedVectorType>(I.getType());
- unsigned OutputNumElts = Ty->getNumElements();
- SmallVector<int, 16> Mask(OutputNumElts, PoisonMaskElem);
- assert(OffsetEltIndex < MinVecNumElts && "Address offset too big");
- Mask[0] = OffsetEltIndex;
+ SmallVector<int> Mask;
+ assert(OffsetEltIndex + VectorRange < MinVecNumElts &&
+ "Address offset too big");
+ if (!NeedCast) {
+ auto *Ty = cast<FixedVectorType>(I.getType());
+ unsigned OutputNumElts = Ty->getNumElements();
+ Mask.assign(OutputNumElts, PoisonMaskElem);
+ Mask[0] = OffsetEltIndex;
+ } else {
+ Mask.assign(MinVecNumElts, PoisonMaskElem);
+ for (unsigned InsertPos = 0; InsertPos < VectorRange; InsertPos++)
+ Mask[InsertPos] = OffsetEltIndex++;
+ }
+
if (OffsetEltIndex)
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy, Mask,
CostKind);
+ if (NeedCast)
+ NewCost += TTI.getCastInstrCost(Instruction::BitCast, I.getType(), MinVecTy,
+ TargetTransformInfo::CastContextHint::None,
+ CostKind);
+
// We can aggressively convert to the vector form because the backend can
// invert this transform if it does not result in a performance win.
if (OldCost < NewCost || !NewCost.isValid())
@@ -349,12 +389,16 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// It is safe and potentially profitable to load a vector directly:
// inselt undef, load Scalar, 0 --> load VecPtr
IRBuilder<> Builder(Load);
+ Value *Result;
Value *CastedPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(SrcPtr, Builder.getPtrTy(AS));
- Value *VecLd = Builder.CreateAlignedLoad(MinVecTy, CastedPtr, Alignment);
- VecLd = Builder.CreateShuffleVector(VecLd, Mask);
+ Result = Builder.CreateAlignedLoad(MinVecTy, CastedPtr, Alignment);
+ Result = Builder.CreateShuffleVector(Result, Mask);
- replaceValue(I, *VecLd);
+ if (NeedCast)
+ Result = Builder.CreateBitOrPointerCast(Result, I.getType());
+
+ replaceValue(I, *Result);
++NumVecLoad;
return true;
}
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index 702be5a72afed..3b9f9e97fe63a 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -290,16 +290,18 @@ define <8 x i16> @gep01_load_i16_insert_v8i16_deref_minalign(ptr align 2 derefer
ret <8 x i16> %r
}
-; Negative test - if we are shuffling a load from the base pointer, the address offset
-; must be a multiple of element size.
-; TODO: Could bitcast around this limitation.
-
define <4 x i32> @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
-; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; SSE2-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
+; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; SSE2-NEXT: ret <4 x i32> [[R]]
+;
+; AVX2-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <4 x i32>
+; AVX2-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <16 x i8>, ptr %p, i64 0, i64 1
%s = load i32, ptr %gep, align 1
@@ -308,11 +310,17 @@ define <4 x i32> @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 der
}
define <2 x i64> @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
-; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
-; CHECK-NEXT: ret <2 x i64> [[R]]
+; SSE2-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
+; SSE2-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; SSE2-NEXT: ret <2 x i64> [[R]]
+;
+; AVX2-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
+; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <2 x i64>
+; AVX2-NEXT: ret <2 x i64> [[R]]
;
%gep = getelementptr inbounds <16 x i8>, ptr %p, i64 0, i64 1
%s = load i64, ptr %gep, align 1
@@ -321,11 +329,17 @@ define <2 x i64> @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(ptr align 1 der
}
define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 11
-; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; SSE2-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 11
+; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; SSE2-NEXT: ret <4 x i32> [[R]]
+;
+; AVX2-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 11, i32 12, i32 13, i32 14, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <4 x i32>
+; AVX2-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <16 x i8>, ptr %p, i64 0, i64 11
%s = load i32, ptr %gep, align 1
@@ -334,11 +348,17 @@ define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 der
}
define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
-; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; SSE2-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
+; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; SSE2-NEXT: ret <4 x i32> [[R]]
+;
+; AVX2-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
+; AVX2-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
%s = load i32, ptr %gep, align 1
@@ -347,11 +367,17 @@ define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 der
}
define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
-; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
-; CHECK-NEXT: ret <2 x i64> [[R]]
+; SSE2-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
+; SSE2-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; SSE2-NEXT: ret <2 x i64> [[R]]
+;
+; AVX2-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
+; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <2 x i64>
+; AVX2-NEXT: ret <2 x i64> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
%s = load i64, ptr %gep, align 1
@@ -360,11 +386,17 @@ define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 der
}
define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 5
-; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; SSE2-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
+; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 5
+; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
+; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
+; SSE2-NEXT: ret <4 x i32> [[R]]
+;
+; AVX2-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
+; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
+; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 5, i32 6, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
+; AVX2-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 5
%s = load i32, ptr %gep, align 1
@@ -372,11 +404,11 @@ define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 der
ret <4 x i32> %r
}
-define <2 x i64> @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(ptr align 1 dereferenceable(16) %p) nofree nosync {
+define <2 x i64> @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
; CHECK-LABEL: @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <4 x i32>, ptr [[P:%.*]], i64 0, i64 1
-; CHECK-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
-; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
+; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 1, i32 2, i32 poison, i32 poison>
+; CHECK-NEXT: [[R:%.*]] = bitcast <4 x i32> [[TMP2]] to <2 x i64>
; CHECK-NEXT: ret <2 x i64> [[R]]
;
%gep = getelementptr inbounds <4 x i32>, ptr %p, i64 0, i64 1
>From ad012abd0875adff885f675164447728c94b1712 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 7 Apr 2025 05:25:40 +0900
Subject: [PATCH 03/24] BigEndian check update
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 2 ++
1 file changed, 2 insertions(+)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 313e4d6f5c7e9..08c1d6b9d6674 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -305,6 +305,8 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
uint64_t ScalarSizeInBytes = ScalarSize / 8;
if (auto UnalignedBytes = Offset.urem(ScalarSizeInBytes);
UnalignedBytes != 0) {
+ if (DL->isBigEndian())
+ return false;
uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
// Assign the greatest common divisor between UnalignedBytes and Offset to
// ScalarSizeInBytes
>From 7f2ff065e5292d335800d6f878daf39c05d7e022 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 8 Apr 2025 06:32:26 +0900
Subject: [PATCH 04/24] use std::gcd instead self code
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 11 +----------
1 file changed, 1 insertion(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 08c1d6b9d6674..42c03826ef31f 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -258,15 +258,6 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (!canWidenLoad(Load, TTI))
return false;
- auto MaxCommonDivisor = [](int n) {
- if (n % 4 == 0)
- return 4;
- if (n % 2 == 0)
- return 2;
- else
- return 1;
- };
-
Type *ScalarTy = Scalar->getType();
uint64_t ScalarSize = ScalarTy->getPrimitiveSizeInBits();
unsigned MinVectorSize = TTI.getMinVectorRegisterBitWidth();
@@ -310,7 +301,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
// Assign the greatest common divisor between UnalignedBytes and Offset to
// ScalarSizeInBytes
- ScalarSizeInBytes = MaxCommonDivisor(UnalignedBytes);
+ ScalarSizeInBytes = std::gcd(ScalarSizeInBytes, UnalignedBytes);
ScalarSize = ScalarSizeInBytes * 8;
VectorRange = OldScalarSizeInBytes / ScalarSizeInBytes;
MinVecNumElts = MinVectorSize / ScalarSize;
>From a6386b2f314adc2e53cabc61acc63fee59aca90f Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 8 Apr 2025 06:33:35 +0900
Subject: [PATCH 05/24] new created IR push to Worklist
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 7 +++++--
1 file changed, 5 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 42c03826ef31f..33322ddc01a09 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -386,10 +386,13 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Value *CastedPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(SrcPtr, Builder.getPtrTy(AS));
Result = Builder.CreateAlignedLoad(MinVecTy, CastedPtr, Alignment);
+ Worklist.pushValue(Result);
Result = Builder.CreateShuffleVector(Result, Mask);
-
- if (NeedCast)
+ Worklist.pushValue(Result);
+ if (NeedCast) {
Result = Builder.CreateBitOrPointerCast(Result, I.getType());
+ Worklist.pushValue(Result);
+ }
replaceValue(I, *Result);
++NumVecLoad;
>From 4a952a3289e7ddd111a364b843faa74cec45743c Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 29 Apr 2025 14:35:38 +0900
Subject: [PATCH 06/24] remove unnecessary checks
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 33322ddc01a09..861dedb8dd426 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -294,8 +294,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// element's size; otherwise, Offset and Scalar must be shuffled to the
// appropriate element size for both.
uint64_t ScalarSizeInBytes = ScalarSize / 8;
- if (auto UnalignedBytes = Offset.urem(ScalarSizeInBytes);
- UnalignedBytes != 0) {
+ if (auto UnalignedBytes = Offset.urem(ScalarSizeInBytes)) {
if (DL->isBigEndian())
return false;
uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
>From 3262ba0d1cc329b670187af9d1bc2d46b9ce326a Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 29 Apr 2025 14:36:41 +0900
Subject: [PATCH 07/24] replace the for statement with std::iota and relocate
the conditional statement
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 9 ++++-----
1 file changed, 4 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 861dedb8dd426..64d4364095017 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -353,15 +353,14 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
SmallVector<int> Mask;
assert(OffsetEltIndex + VectorRange < MinVecNumElts &&
"Address offset too big");
- if (!NeedCast) {
+ if (NeedCast) {
+ Mask.assign(MinVecNumElts, PoisonMaskElem);
+ std::iota(Mask.begin(), Mask.begin() + VectorRange, OffsetEltIndex);
+ } else {
auto *Ty = cast<FixedVectorType>(I.getType());
unsigned OutputNumElts = Ty->getNumElements();
Mask.assign(OutputNumElts, PoisonMaskElem);
Mask[0] = OffsetEltIndex;
- } else {
- Mask.assign(MinVecNumElts, PoisonMaskElem);
- for (unsigned InsertPos = 0; InsertPos < VectorRange; InsertPos++)
- Mask[InsertPos] = OffsetEltIndex++;
}
if (OffsetEltIndex)
>From e136c22b0cdc00fc211a0ac56090acf7b70b10ad Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Tue, 29 Apr 2025 14:37:58 +0900
Subject: [PATCH 08/24] remove duplicate pushes to Worklist
replaceValue adds new instruction to the worklist internally,
so don't need to push it to the worklist to remove it.
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 4 +---
1 file changed, 1 insertion(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 64d4364095017..9c118ca76a9c4 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -387,10 +387,8 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Worklist.pushValue(Result);
Result = Builder.CreateShuffleVector(Result, Mask);
Worklist.pushValue(Result);
- if (NeedCast) {
+ if (NeedCast)
Result = Builder.CreateBitOrPointerCast(Result, I.getType());
- Worklist.pushValue(Result);
- }
replaceValue(I, *Result);
++NumVecLoad;
>From 7d0061d6e781df65ab0233abdc96d7439fd7830f Mon Sep 17 00:00:00 2001
From: Hanbum Park <kese111 at gmail.com>
Date: Thu, 13 Nov 2025 14:31:23 +0900
Subject: [PATCH 09/24] apply the revised shuffle cost calculation
---
.../Transforms/Vectorize/VectorCombine.cpp | 12 ++-
.../VectorCombine/X86/load-inseltpoison.ll | 99 +++++++------------
2 files changed, 45 insertions(+), 66 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 9c118ca76a9c4..e535bd5f252fb 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -350,6 +350,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// We assume this operation has no cost in codegen if there was no offset.
// Note that we could use freeze to avoid poison problems, but then we might
// still need a shuffle to change the vector size.
+ auto *Ty = cast<FixedVectorType>(I.getType());
SmallVector<int> Mask;
assert(OffsetEltIndex + VectorRange < MinVecNumElts &&
"Address offset too big");
@@ -357,18 +358,21 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Mask.assign(MinVecNumElts, PoisonMaskElem);
std::iota(Mask.begin(), Mask.begin() + VectorRange, OffsetEltIndex);
} else {
- auto *Ty = cast<FixedVectorType>(I.getType());
unsigned OutputNumElts = Ty->getNumElements();
Mask.assign(OutputNumElts, PoisonMaskElem);
Mask[0] = OffsetEltIndex;
}
if (OffsetEltIndex)
- NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy, Mask,
- CostKind);
+ if (NeedCast)
+ NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
+ MinVecTy, Mask, CostKind);
+ else
+ NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy,
+ Mask, CostKind);
if (NeedCast)
- NewCost += TTI.getCastInstrCost(Instruction::BitCast, I.getType(), MinVecTy,
+ NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
TargetTransformInfo::CastContextHint::None,
CostKind);
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index 3b9f9e97fe63a..865d3c5488071 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -1,6 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=sse2 | FileCheck %s
-; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=avx2 | FileCheck %s
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=sse2 | FileCheck %s --check-prefixes=CHECK,SSE2
+; RUN: opt < %s -passes=vector-combine -S -mtriple=x86_64-- -mattr=avx2 | FileCheck %s --check-prefixes=CHECK,AVX2
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
@@ -159,8 +159,7 @@ define double @larger_fp_scalar_256bit_vec(ptr align 32 dereferenceable(32) %p)
define <4 x float> @load_f32_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_f32_insert_v4f32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%s = load float, ptr %p, align 4
%r = insertelement <4 x float> poison, float %s, i32 0
@@ -170,8 +169,7 @@ define <4 x float> @load_f32_insert_v4f32(ptr align 16 dereferenceable(16) %p) n
define <4 x float> @casted_load_f32_insert_v4f32(ptr align 4 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @casted_load_f32_insert_v4f32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%s = load float, ptr %p, align 4
%r = insertelement <4 x float> poison, float %s, i32 0
@@ -183,8 +181,7 @@ define <4 x float> @casted_load_f32_insert_v4f32(ptr align 4 dereferenceable(16)
define <4 x i32> @load_i32_insert_v4i32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_i32_insert_v4i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; CHECK-NEXT: ret <4 x i32> [[TMP1]]
;
%s = load i32, ptr %p, align 4
%r = insertelement <4 x i32> poison, i32 %s, i32 0
@@ -196,8 +193,7 @@ define <4 x i32> @load_i32_insert_v4i32(ptr align 16 dereferenceable(16) %p) nof
define <4 x i32> @casted_load_i32_insert_v4i32(ptr align 4 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @casted_load_i32_insert_v4i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x i32> [[R]]
+; CHECK-NEXT: ret <4 x i32> [[TMP1]]
;
%s = load i32, ptr %p, align 4
%r = insertelement <4 x i32> poison, i32 %s, i32 0
@@ -209,8 +205,7 @@ define <4 x i32> @casted_load_i32_insert_v4i32(ptr align 4 dereferenceable(16) %
define <4 x float> @gep00_load_f32_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @gep00_load_f32_insert_v4f32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%s = load float, ptr %p, align 16
%r = insertelement <4 x float> poison, float %s, i64 0
@@ -222,8 +217,7 @@ define <4 x float> @gep00_load_f32_insert_v4f32(ptr align 16 dereferenceable(16)
define <4 x float> @gep00_load_f32_insert_v4f32_addrspace(ptr addrspace(44) align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @gep00_load_f32_insert_v4f32_addrspace(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr addrspace(44) [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%s = load float, ptr addrspace(44) %p, align 16
%r = insertelement <4 x float> poison, float %s, i64 0
@@ -235,8 +229,8 @@ define <4 x float> @gep00_load_f32_insert_v4f32_addrspace(ptr addrspace(44) alig
define <4 x i32> @unsafe_load_i32_insert_v4i32_addrspace(ptr align 16 dereferenceable(16) %v3) {
; CHECK-LABEL: @unsafe_load_i32_insert_v4i32_addrspace(
; CHECK-NEXT: [[TMP1:%.*]] = addrspacecast ptr [[V3:%.*]] to ptr addrspace(42)
-; CHECK-NEXT: [[TMP2:%.*]] = load <4 x i32>, ptr addrspace(42) [[TMP1]], align 16
-; CHECK-NEXT: [[INSELT:%.*]] = shufflevector <4 x i32> [[TMP2]], <4 x i32> poison, <4 x i32> <i32 2, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP2:%.*]] = load <3 x i32>, ptr addrspace(42) [[TMP1]], align 16
+; CHECK-NEXT: [[INSELT:%.*]] = shufflevector <3 x i32> [[TMP2]], <3 x i32> poison, <4 x i32> <i32 2, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <4 x i32> [[INSELT]]
;
%t0 = getelementptr inbounds i32, ptr %v3, i32 1
@@ -253,8 +247,7 @@ define <8 x i16> @gep01_load_i16_insert_v8i16(ptr align 16 dereferenceable(18) %
; CHECK-LABEL: @gep01_load_i16_insert_v8i16(
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
; CHECK-NEXT: [[R:%.*]] = load <8 x i16>, ptr [[GEP]], align 2
-; CHECK-NEXT: [[R1:%.*]] = shufflevector <8 x i16> [[R]], <8 x i16> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <8 x i16> [[R1]]
+; CHECK-NEXT: ret <8 x i16> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
%s = load i16, ptr %gep, align 2
@@ -266,8 +259,8 @@ define <8 x i16> @gep01_load_i16_insert_v8i16(ptr align 16 dereferenceable(18) %
define <8 x i16> @gep01_load_i16_insert_v8i16_deref(ptr align 16 dereferenceable(17) %p) nofree nosync {
; CHECK-LABEL: @gep01_load_i16_insert_v8i16_deref(
-; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i16>, ptr [[P:%.*]], align 16
+; CHECK-NEXT: [[R:%.*]] = shufflevector <2 x i16> [[TMP1]], <2 x i16> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <8 x i16> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
@@ -280,8 +273,8 @@ define <8 x i16> @gep01_load_i16_insert_v8i16_deref(ptr align 16 dereferenceable
define <8 x i16> @gep01_load_i16_insert_v8i16_deref_minalign(ptr align 2 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @gep01_load_i16_insert_v8i16_deref_minalign(
-; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 2
-; CHECK-NEXT: [[R:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP1:%.*]] = load <2 x i16>, ptr [[P:%.*]], align 2
+; CHECK-NEXT: [[R:%.*]] = shufflevector <2 x i16> [[TMP1]], <2 x i16> poison, <8 x i32> <i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <8 x i16> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
@@ -348,17 +341,11 @@ define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 der
}
define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
-; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
-; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; SSE2-NEXT: ret <4 x i32> [[R]]
-;
-; AVX2-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
-; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
-; AVX2-NEXT: ret <4 x i32> [[R]]
+; CHECK-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
+; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
+; CHECK-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 1
%s = load i32, ptr %gep, align 1
@@ -386,17 +373,11 @@ define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 der
}
define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
-; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 5
-; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
-; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
-; SSE2-NEXT: ret <4 x i32> [[R]]
-;
-; AVX2-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
-; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
-; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 5, i32 6, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
-; AVX2-NEXT: ret <4 x i32> [[R]]
+; CHECK-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
+; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 5, i32 6, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
+; CHECK-NEXT: ret <4 x i32> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 0, i64 5
%s = load i32, ptr %gep, align 1
@@ -468,8 +449,7 @@ define <8 x i16> @gep10_load_i16_insert_v8i16(ptr align 16 dereferenceable(32) %
; CHECK-LABEL: @gep10_load_i16_insert_v8i16(
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 1, i64 0
; CHECK-NEXT: [[R:%.*]] = load <8 x i16>, ptr [[GEP]], align 16
-; CHECK-NEXT: [[R1:%.*]] = shufflevector <8 x i16> [[R]], <8 x i16> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <8 x i16> [[R1]]
+; CHECK-NEXT: ret <8 x i16> [[R]]
;
%gep = getelementptr inbounds <8 x i16>, ptr %p, i64 1, i64 0
%s = load i16, ptr %gep, align 16
@@ -571,8 +551,7 @@ define <4 x float> @load_f32_insert_v4f32_volatile(ptr align 16 dereferenceable(
define <4 x float> @load_f32_insert_v4f32_align(ptr align 1 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_f32_insert_v4f32_align(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%s = load float, ptr %p, align 4
%r = insertelement <4 x float> poison, float %s, i32 0
@@ -594,8 +573,8 @@ define <4 x float> @load_f32_insert_v4f32_deref(ptr align 4 dereferenceable(15)
define <8 x i32> @load_i32_insert_v8i32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_i32_insert_v8i32(
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP1:%.*]] = load <1 x i32>, ptr [[P:%.*]], align 16
+; CHECK-NEXT: [[R:%.*]] = shufflevector <1 x i32> [[TMP1]], <1 x i32> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <8 x i32> [[R]]
;
%s = load i32, ptr %p, align 4
@@ -605,8 +584,8 @@ define <8 x i32> @load_i32_insert_v8i32(ptr align 16 dereferenceable(16) %p) nof
define <8 x i32> @casted_load_i32_insert_v8i32(ptr align 4 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @casted_load_i32_insert_v8i32(
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 4
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP1:%.*]] = load <1 x i32>, ptr [[P:%.*]], align 4
+; CHECK-NEXT: [[R:%.*]] = shufflevector <1 x i32> [[TMP1]], <1 x i32> poison, <8 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <8 x i32> [[R]]
;
%s = load i32, ptr %p, align 4
@@ -616,8 +595,8 @@ define <8 x i32> @casted_load_i32_insert_v8i32(ptr align 4 dereferenceable(16) %
define <16 x float> @load_f32_insert_v16f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_f32_insert_v16f32(
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <16 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
+; CHECK-NEXT: [[TMP1:%.*]] = load <1 x float>, ptr [[P:%.*]], align 16
+; CHECK-NEXT: [[R:%.*]] = shufflevector <1 x float> [[TMP1]], <1 x float> poison, <16 x i32> <i32 0, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <16 x float> [[R]]
;
%s = load float, ptr %p, align 4
@@ -627,8 +606,7 @@ define <16 x float> @load_f32_insert_v16f32(ptr align 16 dereferenceable(16) %p)
define <2 x float> @load_f32_insert_v2f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_f32_insert_v2f32(
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <2 x i32> <i32 0, i32 poison>
+; CHECK-NEXT: [[R:%.*]] = load <2 x float>, ptr [[P:%.*]], align 16
; CHECK-NEXT: ret <2 x float> [[R]]
;
%s = load float, ptr %p, align 4
@@ -678,8 +656,7 @@ define void @PR47558_multiple_use_load(ptr nocapture nonnull %resultptr, ptr noc
define <4 x float> @load_v2f32_extract_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_v2f32_extract_insert_v4f32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%l = load <2 x float>, ptr %p, align 4
%s = extractelement <2 x float> %l, i32 0
@@ -690,8 +667,7 @@ define <4 x float> @load_v2f32_extract_insert_v4f32(ptr align 16 dereferenceable
define <4 x float> @load_v8f32_extract_insert_v4f32(ptr align 16 dereferenceable(16) %p) nofree nosync {
; CHECK-LABEL: @load_v8f32_extract_insert_v4f32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <4 x i32> <i32 0, i32 poison, i32 poison, i32 poison>
-; CHECK-NEXT: ret <4 x float> [[R]]
+; CHECK-NEXT: ret <4 x float> [[TMP1]]
;
%l = load <8 x float>, ptr %p, align 4
%s = extractelement <8 x float> %l, i32 0
@@ -771,8 +747,7 @@ define <4 x float> @load_v2f32_extract_insert_v4f32_tsan(ptr align 16 dereferenc
define <2 x float> @load_f32_insert_v2f32_msan(ptr align 16 dereferenceable(16) %p) nofree nosync sanitize_memory {
; CHECK-LABEL: @load_f32_insert_v2f32_msan(
-; CHECK-NEXT: [[TMP1:%.*]] = load <4 x float>, ptr [[P:%.*]], align 16
-; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x float> [[TMP1]], <4 x float> poison, <2 x i32> <i32 0, i32 poison>
+; CHECK-NEXT: [[R:%.*]] = load <2 x float>, ptr [[P:%.*]], align 16
; CHECK-NEXT: ret <2 x float> [[R]]
;
%s = load float, ptr %p, align 4
>From 02482041841743f8631763b75fc7be014b2039a0 Mon Sep 17 00:00:00 2001
From: Hanbum Park <kese111 at gmail.com>
Date: Thu, 13 Nov 2025 23:04:07 +0900
Subject: [PATCH 10/24] add braces
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index e535bd5f252fb..a76b31594db4b 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -363,13 +363,14 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Mask[0] = OffsetEltIndex;
}
- if (OffsetEltIndex)
+ if (OffsetEltIndex) {
if (NeedCast)
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
MinVecTy, Mask, CostKind);
else
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy,
Mask, CostKind);
+ }
if (NeedCast)
NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
>From b34a3cc45cd168a6c868ea3eb4e649d7cc238746 Mon Sep 17 00:00:00 2001
From: Hanbum Park <kese111 at gmail.com>
Date: Thu, 13 Nov 2025 23:30:52 +0900
Subject: [PATCH 11/24] add braces
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index a76b31594db4b..d175dc1302cd4 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -364,12 +364,13 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
}
if (OffsetEltIndex) {
- if (NeedCast)
+ if (NeedCast) {
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
MinVecTy, Mask, CostKind);
- else
+ } else {
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy,
Mask, CostKind);
+ }
}
if (NeedCast)
>From 46233cbcb8c12277db517b3d9df436626cf7ddb4 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 29 Jun 2026 19:13:03 +0900
Subject: [PATCH 12/24] Reject mismatched vector widths in load-insert
vectorization
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 8 ++++++++
1 file changed, 8 insertions(+)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index d175dc1302cd4..01b504b7dfc8b 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -307,6 +307,14 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
ScalarTy = Type::getIntNTy(I.getContext(), ScalarSize);
MinVecTy = VectorType::get(ScalarTy, MinVecNumElts, false);
NeedCast = true;
+
+ // In the NeedCast case, the shuffle result is later bitcast to the final
+ // vector type. This is only valid when the widened load vector and the
+ // final result vector have the same total bit width. Cases like <16 x i8>
+ // to <3 x i32> would require a different intermediate shuffle type, so
+ // leave them for a separate enhancement.
+ if (DL->getTypeSizeInBits(MinVecTy) != DL->getTypeSizeInBits(I.getType()))
+ return false;
}
// If we load MinVecNumElts, will our target element still be loaded?
>From dc5f8774de115620d9fa3f63c41ae8172478e3a9 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:25:39 +0900
Subject: [PATCH 13/24] Add mismatched bit-width regression test
---
.../VectorCombine/X86/load-inseltpoison.ll | 16 ++++++++++++++++
1 file changed, 16 insertions(+)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index 865d3c5488071..15508ecb89d6c 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -398,6 +398,22 @@ define <2 x i64> @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(ptr align 1 der
ret <2 x i64> %r
}
+; The widened vector and the result vector have different total bit widths,
+; so the result cannot be reconstructed with a bitcast.
+
+define <3 x i32> @load_insert_unequal_total_bitwidth(ptr align 16 dereferenceable(16) %p) {
+; CHECK-LABEL: @load_insert_unequal_total_bitwidth(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 1
+; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <3 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: ret <3 x i32> [[R]]
+;
+ %gep = getelementptr inbounds i8, ptr %p, i64 1
+ %x = load i32, ptr %gep, align 1
+ %r = insertelement <3 x i32> poison, i32 %x, i64 0
+ ret <3 x i32> %r
+}
+
define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
; CHECK-LABEL: @gep012_bitcast_load_i32_insert_v4i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 1
>From 20ba9c90119d749f11f8e967e35b9a6e918ccb1d Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:28:48 +0900
Subject: [PATCH 14/24] Add AArch64 load-insert tests
---
.../AArch64/load-inseltpoison-endian.ll | 25 ++++++++++
.../AArch64/load-inseltpoison.ll | 47 +++++++++++++++++++
2 files changed, 72 insertions(+)
create mode 100644 llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison-endian.ll
create mode 100644 llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison.ll
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison-endian.ll b/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison-endian.ll
new file mode 100644
index 0000000000000..e7e2fab452fb5
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison-endian.ll
@@ -0,0 +1,25 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=vector-combine -S -mtriple=aarch64-linux-gnu -mattr=+neon %s | FileCheck %s --check-prefix=LE
+; RUN: opt -passes=vector-combine -S -mtriple=aarch64_be-linux-gnu -mattr=+neon %s | FileCheck %s --check-prefix=BE
+
+; The unaligned load-insert folds through i16 chunks on little-endian targets,
+; but the bitcast-based fold is disabled on big-endian targets.
+
+define <2 x i32> @load_insert_unaligned_offset_i32_gcd_i16(ptr align 8 dereferenceable(8) %p) {
+; LE-LABEL: @load_insert_unaligned_offset_i32_gcd_i16(
+; LE-NEXT: [[TMP1:%.*]] = load <4 x i16>, ptr [[P:%.*]], align 8
+; LE-NEXT: [[TMP2:%.*]] = shufflevector <4 x i16> [[TMP1]], <4 x i16> poison, <4 x i32> <i32 1, i32 2, i32 poison, i32 poison>
+; LE-NEXT: [[R:%.*]] = bitcast <4 x i16> [[TMP2]] to <2 x i32>
+; LE-NEXT: ret <2 x i32> [[R]]
+;
+; BE-LABEL: @load_insert_unaligned_offset_i32_gcd_i16(
+; BE-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 2
+; BE-NEXT: [[X:%.*]] = load i32, ptr [[GEP]], align 1
+; BE-NEXT: [[R:%.*]] = insertelement <2 x i32> poison, i32 [[X]], i64 0
+; BE-NEXT: ret <2 x i32> [[R]]
+;
+ %gep = getelementptr inbounds i8, ptr %p, i64 2
+ %x = load i32, ptr %gep, align 1
+ %r = insertelement <2 x i32> poison, i32 %x, i64 0
+ ret <2 x i32> %r
+}
diff --git a/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison.ll
new file mode 100644
index 0000000000000..5a8dcb180329c
--- /dev/null
+++ b/llvm/test/Transforms/VectorCombine/AArch64/load-inseltpoison.ll
@@ -0,0 +1,47 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=vector-combine -S -mtriple=aarch64-linux-gnu -mattr=+neon %s | FileCheck %s
+
+; An i8 chunk is legal, but the AArch64 cost model rejects this transform.
+
+define <2 x i32> @load_insert_unaligned_offset_i32_gcd_i8(ptr align 8 dereferenceable(8) %p) {
+; CHECK-LABEL: @load_insert_unaligned_offset_i32_gcd_i8(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 1
+; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: ret <2 x i32> [[R]]
+;
+ %gep = getelementptr inbounds i8, ptr %p, i64 1
+ %x = load i32, ptr %gep, align 1
+ %r = insertelement <2 x i32> poison, i32 %x, i64 0
+ ret <2 x i32> %r
+}
+
+; The i64 variant with an i8 chunk is also unprofitable on AArch64.
+
+define <1 x i64> @load_insert_unaligned_offset_i64_gcd_i8(ptr align 8 dereferenceable(8) %p) {
+; CHECK-LABEL: @load_insert_unaligned_offset_i64_gcd_i8(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 1
+; CHECK-NEXT: [[X:%.*]] = load i64, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <1 x i64> poison, i64 [[X]], i64 0
+; CHECK-NEXT: ret <1 x i64> [[R]]
+;
+ %gep = getelementptr inbounds i8, ptr %p, i64 1
+ %x = load i64, ptr %gep, align 1
+ %r = insertelement <1 x i64> poison, i64 %x, i64 0
+ ret <1 x i64> %r
+}
+
+; The scalar access extends beyond the widened vector load range.
+
+define <2 x i32> @load_insert_access_exceeds_vector_range(ptr align 8 dereferenceable(9) %p) {
+; CHECK-LABEL: @load_insert_access_exceeds_vector_range(
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 5
+; CHECK-NEXT: [[X:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT: [[R:%.*]] = insertelement <2 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: ret <2 x i32> [[R]]
+;
+ %gep = getelementptr inbounds i8, ptr %p, i64 5
+ %x = load i32, ptr %gep, align 1
+ %r = insertelement <2 x i32> poison, i32 %x, i64 0
+ ret <2 x i32> %r
+}
>From 79cf54f916a18f2e8cd117a6bc57fa3ddfd3a88e Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:31:14 +0900
Subject: [PATCH 15/24] Clarify the GCD chunking comment
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 01b504b7dfc8b..65ca82d627bac 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -298,8 +298,8 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (DL->isBigEndian())
return false;
uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
- // Assign the greatest common divisor between UnalignedBytes and Offset to
- // ScalarSizeInBytes
+ // Find the largest integer element size that divides both the scalar
+ // size and the unaligned byte offset.
ScalarSizeInBytes = std::gcd(ScalarSizeInBytes, UnalignedBytes);
ScalarSize = ScalarSizeInBytes * 8;
VectorRange = OldScalarSizeInBytes / ScalarSizeInBytes;
>From f86408a0d7d23639790588c90314f05a425a3ae3 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:32:53 +0900
Subject: [PATCH 16/24] Separate scalar and chunk sizes
---
.../Transforms/Vectorize/VectorCombine.cpp | 57 ++++++++++---------
1 file changed, 31 insertions(+), 26 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 65ca82d627bac..390863fdb0727 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -258,8 +258,9 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (!canWidenLoad(Load, TTI))
return false;
- Type *ScalarTy = Scalar->getType();
- uint64_t ScalarSize = ScalarTy->getPrimitiveSizeInBits();
+ Type *OriginalScalarTy = Scalar->getType();
+ uint64_t OriginalScalarSizeInBits =
+ OriginalScalarTy->getPrimitiveSizeInBits();
unsigned MinVectorSize = TTI.getMinVectorRegisterBitWidth();
// Check safety of replacing the scalar load with a larger vector load.
@@ -269,10 +270,11 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Value *SrcPtr = Load->getPointerOperand()->stripPointerCasts();
assert(isa<PointerType>(SrcPtr->getType()) && "Expected a pointer type");
- unsigned MinVecNumElts = MinVectorSize / ScalarSize;
- auto *MinVecTy = VectorType::get(ScalarTy, MinVecNumElts, false);
+ unsigned MinVecNumElts = MinVectorSize / OriginalScalarSizeInBits;
+ auto *MinVecTy =
+ VectorType::get(OriginalScalarTy, MinVecNumElts, false);
unsigned OffsetEltIndex = 0;
- unsigned VectorRange = 0;
+ unsigned NumScalarChunks = 1;
bool NeedCast = false;
Align Alignment = Load->getAlign();
if (!isSafeToLoadUnconditionally(SrcPtr, MinVecTy, Align(1), *DL, Load, SQ.AC,
@@ -293,33 +295,36 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// If Offset is multiple of a Scalar element, it can be shuffled to the
// element's size; otherwise, Offset and Scalar must be shuffled to the
// appropriate element size for both.
- uint64_t ScalarSizeInBytes = ScalarSize / 8;
- if (auto UnalignedBytes = Offset.urem(ScalarSizeInBytes)) {
+ const uint64_t OriginalScalarSizeInBytes =
+ OriginalScalarSizeInBits / 8;
+ uint64_t ChunkSizeInBytes = OriginalScalarSizeInBytes;
+ if (uint64_t UnalignedBytes =
+ Offset.urem(OriginalScalarSizeInBytes)) {
if (DL->isBigEndian())
return false;
- uint64_t OldScalarSizeInBytes = ScalarSizeInBytes;
// Find the largest integer element size that divides both the scalar
// size and the unaligned byte offset.
- ScalarSizeInBytes = std::gcd(ScalarSizeInBytes, UnalignedBytes);
- ScalarSize = ScalarSizeInBytes * 8;
- VectorRange = OldScalarSizeInBytes / ScalarSizeInBytes;
- MinVecNumElts = MinVectorSize / ScalarSize;
- ScalarTy = Type::getIntNTy(I.getContext(), ScalarSize);
- MinVecTy = VectorType::get(ScalarTy, MinVecNumElts, false);
- NeedCast = true;
-
- // In the NeedCast case, the shuffle result is later bitcast to the final
- // vector type. This is only valid when the widened load vector and the
- // final result vector have the same total bit width. Cases like <16 x i8>
- // to <3 x i32> would require a different intermediate shuffle type, so
- // leave them for a separate enhancement.
- if (DL->getTypeSizeInBits(MinVecTy) != DL->getTypeSizeInBits(I.getType()))
+ ChunkSizeInBytes =
+ std::gcd(OriginalScalarSizeInBytes, UnalignedBytes);
+ const uint64_t ChunkSizeInBits = ChunkSizeInBytes * 8;
+ NumScalarChunks = OriginalScalarSizeInBytes / ChunkSizeInBytes;
+ MinVecNumElts = MinVectorSize / ChunkSizeInBits;
+ Type *ChunkTy = Type::getIntNTy(I.getContext(), ChunkSizeInBits);
+ auto *ChunkVecTy = VectorType::get(ChunkTy, MinVecNumElts, false);
+
+ // The chunk vector is later bitcast to the final vector type, so their
+ // total bit widths must match.
+ if (DL->getTypeSizeInBits(ChunkVecTy) !=
+ DL->getTypeSizeInBits(I.getType()))
return false;
+
+ MinVecTy = ChunkVecTy;
+ NeedCast = true;
}
// If we load MinVecNumElts, will our target element still be loaded?
- APInt OffsetEltIndexAP = Offset.udiv(ScalarSizeInBytes);
- if ((OffsetEltIndexAP + VectorRange).uge(MinVecNumElts))
+ APInt OffsetEltIndexAP = Offset.udiv(ChunkSizeInBytes);
+ if ((OffsetEltIndexAP + NumScalarChunks).ugt(MinVecNumElts))
return false;
OffsetEltIndex = OffsetEltIndexAP.getZExtValue();
@@ -360,11 +365,11 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// still need a shuffle to change the vector size.
auto *Ty = cast<FixedVectorType>(I.getType());
SmallVector<int> Mask;
- assert(OffsetEltIndex + VectorRange < MinVecNumElts &&
+ assert(OffsetEltIndex + NumScalarChunks <= MinVecNumElts &&
"Address offset too big");
if (NeedCast) {
Mask.assign(MinVecNumElts, PoisonMaskElem);
- std::iota(Mask.begin(), Mask.begin() + VectorRange, OffsetEltIndex);
+ std::iota(Mask.begin(), Mask.begin() + NumScalarChunks, OffsetEltIndex);
} else {
unsigned OutputNumElts = Ty->getNumElements();
Mask.assign(OutputNumElts, PoisonMaskElem);
>From e8eddada3fd0bad83e25b231d98cbbcfa5fee372 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:33:50 +0900
Subject: [PATCH 17/24] Avoid duplicate worklist insertion
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 390863fdb0727..ed0cb93fefca1 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -405,9 +405,10 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Result = Builder.CreateAlignedLoad(MinVecTy, CastedPtr, Alignment);
Worklist.pushValue(Result);
Result = Builder.CreateShuffleVector(Result, Mask);
- Worklist.pushValue(Result);
- if (NeedCast)
+ if (NeedCast) {
+ Worklist.pushValue(Result);
Result = Builder.CreateBitOrPointerCast(Result, I.getType());
+ }
replaceValue(I, *Result);
++NumVecLoad;
>From 9dec9670d349167652ef1f26cb462fc81fa3cb8b Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:34:29 +0900
Subject: [PATCH 18/24] Fix the out-of-range test comment
---
llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll | 4 +---
1 file changed, 1 insertion(+), 3 deletions(-)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index 15508ecb89d6c..ebe8352778251 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -426,9 +426,7 @@ define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceab
ret <4 x i32> %r
}
-; Negative test - if we are shuffling a load from the base pointer, the address offset
-; must be a multiple of element size and the offset must be low enough to fit in the vector
-; (bitcasting would not help this case).
+; The scalar access extends beyond the widened vector load range.
define <4 x i32> @gep013_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
; CHECK-LABEL: @gep013_bitcast_load_i32_insert_v4i32(
>From dfdcb8c796078d0f14d0e9aa9724d8e4246217e6 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:35:52 +0900
Subject: [PATCH 19/24] Name load-insert tests by behavior
---
.../VectorCombine/X86/load-inseltpoison.ll | 44 +++++++++----------
1 file changed, 22 insertions(+), 22 deletions(-)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
index ebe8352778251..4c31b318c546c 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load-inseltpoison.ll
@@ -283,14 +283,14 @@ define <8 x i16> @gep01_load_i16_insert_v8i16_deref_minalign(ptr align 2 derefer
ret <8 x i16> %r
}
-define <4 x i32> @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
+define <4 x i32> @load_insert_unaligned_offset1_i32_gcd_i8(ptr align 1 dereferenceable(16) %p) {
+; SSE2-LABEL: @load_insert_unaligned_offset1_i32_gcd_i8(
; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
; SSE2-NEXT: ret <4 x i32> [[R]]
;
-; AVX2-LABEL: @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(
+; AVX2-LABEL: @load_insert_unaligned_offset1_i32_gcd_i8(
; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <4 x i32>
@@ -302,14 +302,14 @@ define <4 x i32> @gep01_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 der
ret <4 x i32> %r
}
-define <2 x i64> @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
+define <2 x i64> @load_insert_unaligned_offset1_i64_gcd_i8(ptr align 1 dereferenceable(16) %p) {
+; SSE2-LABEL: @load_insert_unaligned_offset1_i64_gcd_i8(
; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 1
; SSE2-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
; SSE2-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
; SSE2-NEXT: ret <2 x i64> [[R]]
;
-; AVX2-LABEL: @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(
+; AVX2-LABEL: @load_insert_unaligned_offset1_i64_gcd_i8(
; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <2 x i64>
@@ -321,14 +321,14 @@ define <2 x i64> @gep01_bitcast_load_i64_from_v16i8_insert_v2i64(ptr align 1 der
ret <2 x i64> %r
}
-define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
+define <4 x i32> @load_insert_unaligned_offset11_i32_gcd_i8(ptr align 1 dereferenceable(16) %p) {
+; SSE2-LABEL: @load_insert_unaligned_offset11_i32_gcd_i8(
; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 11
; SSE2-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
; SSE2-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
; SSE2-NEXT: ret <4 x i32> [[R]]
;
-; AVX2-LABEL: @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(
+; AVX2-LABEL: @load_insert_unaligned_offset11_i32_gcd_i8(
; AVX2-NEXT: [[TMP1:%.*]] = load <16 x i8>, ptr [[P:%.*]], align 1
; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <16 x i8> [[TMP1]], <16 x i8> poison, <16 x i32> <i32 11, i32 12, i32 13, i32 14, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; AVX2-NEXT: [[R:%.*]] = bitcast <16 x i8> [[TMP2]] to <4 x i32>
@@ -340,8 +340,8 @@ define <4 x i32> @gep11_bitcast_load_i32_from_v16i8_insert_v4i32(ptr align 1 der
ret <4 x i32> %r
}
-define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(
+define <4 x i32> @load_insert_unaligned_offset2_i32_gcd_i16(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @load_insert_unaligned_offset2_i32_gcd_i16(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
@@ -353,14 +353,14 @@ define <4 x i32> @gep01_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 der
ret <4 x i32> %r
}
-define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
-; SSE2-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
+define <2 x i64> @load_insert_unaligned_offset2_i64_gcd_i16(ptr align 1 dereferenceable(16) %p) {
+; SSE2-LABEL: @load_insert_unaligned_offset2_i64_gcd_i16(
; SSE2-NEXT: [[GEP:%.*]] = getelementptr inbounds <8 x i16>, ptr [[P:%.*]], i64 0, i64 1
; SSE2-NEXT: [[S:%.*]] = load i64, ptr [[GEP]], align 1
; SSE2-NEXT: [[R:%.*]] = insertelement <2 x i64> poison, i64 [[S]], i64 0
; SSE2-NEXT: ret <2 x i64> [[R]]
;
-; AVX2-LABEL: @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(
+; AVX2-LABEL: @load_insert_unaligned_offset2_i64_gcd_i16(
; AVX2-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
; AVX2-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 poison, i32 poison, i32 poison, i32 poison>
; AVX2-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <2 x i64>
@@ -372,8 +372,8 @@ define <2 x i64> @gep01_bitcast_load_i64_from_v8i16_insert_v2i64(ptr align 1 der
ret <2 x i64> %r
}
-define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(
+define <4 x i32> @load_insert_unaligned_offset10_i32_gcd_i16(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @load_insert_unaligned_offset10_i32_gcd_i16(
; CHECK-NEXT: [[TMP1:%.*]] = load <8 x i16>, ptr [[P:%.*]], align 1
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <8 x i16> [[TMP1]], <8 x i16> poison, <8 x i32> <i32 5, i32 6, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: [[R:%.*]] = bitcast <8 x i16> [[TMP2]] to <4 x i32>
@@ -385,8 +385,8 @@ define <4 x i32> @gep05_bitcast_load_i32_from_v8i16_insert_v4i32(ptr align 1 der
ret <4 x i32> %r
}
-define <2 x i64> @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(ptr align 1 dereferenceable(16) %p) {
-; CHECK-LABEL: @gep01_bitcast_load_i32_from_v4i32_insert_v2i64(
+define <2 x i64> @load_insert_unaligned_offset4_i64_gcd_i32(ptr align 1 dereferenceable(16) %p) {
+; CHECK-LABEL: @load_insert_unaligned_offset4_i64_gcd_i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 1
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 1, i32 2, i32 poison, i32 poison>
; CHECK-NEXT: [[R:%.*]] = bitcast <4 x i32> [[TMP2]] to <2 x i64>
@@ -414,8 +414,8 @@ define <3 x i32> @load_insert_unequal_total_bitwidth(ptr align 16 dereferenceabl
ret <3 x i32> %r
}
-define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
-; CHECK-LABEL: @gep012_bitcast_load_i32_insert_v4i32(
+define <4 x i32> @load_insert_aligned_offset12_i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
+; CHECK-LABEL: @load_insert_aligned_offset12_i32(
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x i32>, ptr [[P:%.*]], align 1
; CHECK-NEXT: [[R:%.*]] = shufflevector <4 x i32> [[TMP1]], <4 x i32> poison, <4 x i32> <i32 3, i32 poison, i32 poison, i32 poison>
; CHECK-NEXT: ret <4 x i32> [[R]]
@@ -428,8 +428,8 @@ define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceab
; The scalar access extends beyond the widened vector load range.
-define <4 x i32> @gep013_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
-; CHECK-LABEL: @gep013_bitcast_load_i32_insert_v4i32(
+define <4 x i32> @load_insert_access_exceeds_vector_range(ptr align 1 dereferenceable(20) %p) nofree nosync {
+; CHECK-LABEL: @load_insert_access_exceeds_vector_range(
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds <16 x i8>, ptr [[P:%.*]], i64 0, i64 13
; CHECK-NEXT: [[S:%.*]] = load i32, ptr [[GEP]], align 1
; CHECK-NEXT: [[R:%.*]] = insertelement <4 x i32> poison, i32 [[S]], i64 0
>From a407e5d43a15e2cfebc8d1845b2115fbee75e251 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Mon, 20 Jul 2026 19:36:38 +0900
Subject: [PATCH 20/24] Fix stale undef test comments
---
llvm/test/Transforms/VectorCombine/X86/load.ll | 9 +++------
1 file changed, 3 insertions(+), 6 deletions(-)
diff --git a/llvm/test/Transforms/VectorCombine/X86/load.ll b/llvm/test/Transforms/VectorCombine/X86/load.ll
index 388b655641b7d..ab292841f367c 100644
--- a/llvm/test/Transforms/VectorCombine/X86/load.ll
+++ b/llvm/test/Transforms/VectorCombine/X86/load.ll
@@ -275,9 +275,7 @@ define <8 x i16> @gep01_load_i16_insert_v8i16_deref_minalign(ptr align 2 derefer
ret <8 x i16> %r
}
-; Negative test - if we are shuffling a load from the base pointer, the address offset
-; must be a multiple of element size.
-; TODO: Could bitcast around this limitation.
+; The widened load transform requires a poison vector.
define <4 x i32> @gep01_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(16) %p) {
; CHECK-LABEL: @gep01_bitcast_load_i32_insert_v4i32(
@@ -305,9 +303,8 @@ define <4 x i32> @gep012_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceab
ret <4 x i32> %r
}
-; Negative test - if we are shuffling a load from the base pointer, the address offset
-; must be a multiple of element size and the offset must be low enough to fit in the vector
-; (bitcasting would not help this case).
+; The widened load transform requires a poison vector. The scalar access also
+; extends beyond the widened vector load range.
define <4 x i32> @gep013_bitcast_load_i32_insert_v4i32(ptr align 1 dereferenceable(20) %p) nofree nosync {
; CHECK-LABEL: @gep013_bitcast_load_i32_insert_v4i32(
>From bc7bc43372f70da1f11e88f88d1c88b84dc2cdf2 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Wed, 22 Jul 2026 20:42:24 +0900
Subject: [PATCH 21/24] Document vectorizeLoadInsert flow
---
.../Transforms/Vectorize/VectorCombine.cpp | 96 +++++++++++++------
1 file changed, 68 insertions(+), 28 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index ed0cb93fefca1..5001af2646daf 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -240,15 +240,46 @@ static bool canWidenLoad(LoadInst *Load, const TargetTransformInfo &TTI) {
return true;
}
+/// Fold a scalar load inserted into lane zero into a widened vector load.
+///
+/// Input example (i32 at byte offset 4):
+/// BasePtr + 4 --> load i32 --> insert into lane 0 of <4 x i32>
+///
+/// Scalar-aligned transformation of the input example:
+/// memory: [ 0..3 ] [ 4..7 ] [ 8..11 ] [ 12..15 ]
+/// v4i32: [lane 0] [lane 1] [lane 2 ] [ lane 3 ]
+/// |
+/// v
+/// scalar bytes
+/// BasePtr --> load <4 x i32> --> shuffle [1, poison, poison, poison]
+///
+/// Unaligned example (i32 at byte offset 2, little-endian only):
+/// * ChunkSize = gcd(4, 2) = 2 bytes
+/// memory: [ 0..1 ] [ 2..3 ] [ 4..5 ] [ 6..7 ] ... [ 14..15 ]
+/// v8i16: [lane 0] [lane 1] [lane 2] [lane 3] ... [ lane 7 ]
+/// | |
+/// +----------+
+/// |
+/// v
+/// scalar bytes [ 2..5 ]
+/// BasePtr --> load <8 x i16> --> shuffle [1, 2, poison, ...]
+/// |
+/// v
+/// bitcast <4 x i32>
+///
+/// If a widened load is safe at the original load pointer, that pointer is the
+/// BasePtr above and Offset is zero. Otherwise, constant inbounds GEP offsets
+/// are stripped to find a safe BasePtr and the shuffle selects the original
+/// bytes from the widened load.
bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
- // Match insert into fixed vector of scalar value.
+ // Match a scalar inserted into lane zero, optionally through an extract from
+ // lane zero of a vector load.
// TODO: Handle non-zero insert index.
Value *Scalar;
if (!match(&I,
m_InsertElt(m_Poison(), m_OneUse(m_Value(Scalar)), m_ZeroInt())))
return false;
- // Optionally match an extract from another vector.
Value *X;
bool HasExtract = match(Scalar, m_ExtractElt(m_Value(X), m_ZeroInt()));
if (!HasExtract)
@@ -263,47 +294,49 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
OriginalScalarTy->getPrimitiveSizeInBits();
unsigned MinVectorSize = TTI.getMinVectorRegisterBitWidth();
- // Check safety of replacing the scalar load with a larger vector load.
- // We use minimal alignment (maximum flexibility) because we only care about
- // the dereferenceable region. When calculating cost and creating a new op,
- // we may use a larger value based on alignment attributes.
Value *SrcPtr = Load->getPointerOperand()->stripPointerCasts();
assert(isa<PointerType>(SrcPtr->getType()) && "Expected a pointer type");
unsigned MinVecNumElts = MinVectorSize / OriginalScalarSizeInBits;
auto *MinVecTy =
VectorType::get(OriginalScalarTy, MinVecNumElts, false);
+
unsigned OffsetEltIndex = 0;
+ // An aligned scalar occupies one vector element. Unaligned accesses below
+ // split it into smaller chunks and update this count.
unsigned NumScalarChunks = 1;
bool NeedCast = false;
+
+ // Check the widened access with minimal alignment. The actual load alignment
+ // is derived after choosing a safe pointer.
Align Alignment = Load->getAlign();
if (!isSafeToLoadUnconditionally(SrcPtr, MinVecTy, Align(1), *DL, Load, SQ.AC,
SQ.DT)) {
- // It is not safe to load directly from the pointer, but we can still peek
- // through gep offsets and check if it safe to load from a base address with
- // updated alignment. If it is, we can shuffle the element(s) into place
- // after loading.
+ // Recover a candidate load pointer from constant inbounds GEPs. Keep the
+ // original scalar's offset in the pointer index width; dynamic and
+ // non-inbounds address calculations remain in SrcPtr.
unsigned OffsetBitWidth = DL->getIndexTypeSizeInBits(SrcPtr->getType());
APInt Offset(OffsetBitWidth, 0);
SrcPtr = SrcPtr->stripAndAccumulateInBoundsConstantOffsets(*DL, Offset);
- // We want to shuffle the result down from a high element of a vector, so
- // the offset must be positive.
+ // A forward vector load cannot select bytes before its pointer.
if (Offset.isNegative())
return false;
- // If Offset is multiple of a Scalar element, it can be shuffled to the
- // element's size; otherwise, Offset and Scalar must be shuffled to the
- // appropriate element size for both.
const uint64_t OriginalScalarSizeInBytes =
OriginalScalarSizeInBits / 8;
uint64_t ChunkSizeInBytes = OriginalScalarSizeInBytes;
if (uint64_t UnalignedBytes =
Offset.urem(OriginalScalarSizeInBytes)) {
+ // Reconstruct an unaligned scalar from smaller integer chunks.
+ // Consecutive low-address chunks appear in increasing vector lanes on a
+ // little-endian target. Big-endian reconstruction would require a
+ // different lane order and is deliberately left unsupported.
if (DL->isBigEndian())
return false;
- // Find the largest integer element size that divides both the scalar
- // size and the unaligned byte offset.
+
+ // The GCD gives the largest chunk that exactly divides both the scalar
+ // width and byte offset. This minimizes the number of shuffle lanes.
ChunkSizeInBytes =
std::gcd(OriginalScalarSizeInBytes, UnalignedBytes);
const uint64_t ChunkSizeInBits = ChunkSizeInBytes * 8;
@@ -312,8 +345,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Type *ChunkTy = Type::getIntNTy(I.getContext(), ChunkSizeInBits);
auto *ChunkVecTy = VectorType::get(ChunkTy, MinVecNumElts, false);
- // The chunk vector is later bitcast to the final vector type, so their
- // total bit widths must match.
+ // The chunk vector is later bitcast, so its total width must not change.
if (DL->getTypeSizeInBits(ChunkVecTy) !=
DL->getTypeSizeInBits(I.getType()))
return false;
@@ -322,12 +354,12 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
NeedCast = true;
}
- // If we load MinVecNumElts, will our target element still be loaded?
APInt OffsetEltIndexAP = Offset.udiv(ChunkSizeInBytes);
if ((OffsetEltIndexAP + NumScalarChunks).ugt(MinVecNumElts))
return false;
OffsetEltIndex = OffsetEltIndexAP.getZExtValue();
+ // The complete widened access at the recovered pointer must not trap.
if (!isSafeToLoadUnconditionally(SrcPtr, MinVecTy, Align(1), *DL, Load,
SQ.AC, SQ.DT))
return false;
@@ -356,40 +388,48 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// New pattern: load VecPtr
InstructionCost NewCost =
TTI.getMemoryOpCost(Instruction::Load, MinVecTy, Alignment, AS, CostKind);
- // Optionally, we are shuffling the loaded vector element(s) into place.
- // For the mask set everything but element 0 to undef to prevent poison from
- // propagating from the extra loaded memory. This will also optionally
- // shrink/grow the vector from the loaded size to the output size.
- // We assume this operation has no cost in codegen if there was no offset.
- // Note that we could use freeze to avoid poison problems, but then we might
- // still need a shuffle to change the vector size.
auto *Ty = cast<FixedVectorType>(I.getType());
SmallVector<int> Mask;
assert(OffsetEltIndex + NumScalarChunks <= MinVecNumElts &&
"Address offset too big");
if (NeedCast) {
+ // Poison unused lanes, then gather the scalar chunks into the low lanes.
Mask.assign(MinVecNumElts, PoisonMaskElem);
std::iota(Mask.begin(), Mask.begin() + NumScalarChunks, OffsetEltIndex);
} else {
+ // Poison unused lanes, select the scalar into lane zero, and resize the
+ // vector if needed.
unsigned OutputNumElts = Ty->getNumElements();
Mask.assign(OutputNumElts, PoisonMaskElem);
Mask[0] = OffsetEltIndex;
}
+ // Assume an offset-zero shuffle is free because codegen can use the loaded
+ // vector directly.
if (OffsetEltIndex) {
if (NeedCast) {
+ // Account for the chunk shuffle in the unaligned example:
+ // %chunks = shufflevector <8 x i16> %wide.i16, <8 x i16> poison,
+ // <8 x i32> <i32 1, i32 2, i32 poison, i32 poison,
+ // i32 poison, i32 poison, i32 poison, i32 poison>
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
MinVecTy, Mask, CostKind);
} else {
+ // Account for the shuffle in the scalar-aligned example:
+ // %aligned = shufflevector <4 x i32> %wide.i32, <4 x i32> poison,
+ // <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>
NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, Ty, MinVecTy,
Mask, CostKind);
}
}
- if (NeedCast)
+ if (NeedCast) {
+ // Account for the bitcast after the unaligned chunk shuffle:
+ // %result = bitcast <8 x i16> %chunks to <4 x i32>
NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
TargetTransformInfo::CastContextHint::None,
CostKind);
+ }
// We can aggressively convert to the vector form because the backend can
// invert this transform if it does not result in a performance win.
>From 13b3292e26331bdbcf362299469fd73e2b2a1c3f Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Thu, 23 Jul 2026 15:15:40 +0900
Subject: [PATCH 22/24] Clarify chunked load state
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 14 ++++++++------
1 file changed, 8 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 5001af2646daf..262cd6225afe2 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -305,7 +305,9 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// An aligned scalar occupies one vector element. Unaligned accesses below
// split it into smaller chunks and update this count.
unsigned NumScalarChunks = 1;
- bool NeedCast = false;
+ // For an unaligned scalar, load and shuffle GCD-sized integer chunks, then
+ // bitcast the chunk vector to the requested result type.
+ bool UseChunkedLoad = false;
// Check the widened access with minimal alignment. The actual load alignment
// is derived after choosing a safe pointer.
@@ -351,7 +353,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
return false;
MinVecTy = ChunkVecTy;
- NeedCast = true;
+ UseChunkedLoad = true;
}
APInt OffsetEltIndexAP = Offset.udiv(ChunkSizeInBytes);
@@ -392,7 +394,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
SmallVector<int> Mask;
assert(OffsetEltIndex + NumScalarChunks <= MinVecNumElts &&
"Address offset too big");
- if (NeedCast) {
+ if (UseChunkedLoad) {
// Poison unused lanes, then gather the scalar chunks into the low lanes.
Mask.assign(MinVecNumElts, PoisonMaskElem);
std::iota(Mask.begin(), Mask.begin() + NumScalarChunks, OffsetEltIndex);
@@ -407,7 +409,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// Assume an offset-zero shuffle is free because codegen can use the loaded
// vector directly.
if (OffsetEltIndex) {
- if (NeedCast) {
+ if (UseChunkedLoad) {
// Account for the chunk shuffle in the unaligned example:
// %chunks = shufflevector <8 x i16> %wide.i16, <8 x i16> poison,
// <8 x i32> <i32 1, i32 2, i32 poison, i32 poison,
@@ -423,7 +425,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
}
}
- if (NeedCast) {
+ if (UseChunkedLoad) {
// Account for the bitcast after the unaligned chunk shuffle:
// %result = bitcast <8 x i16> %chunks to <4 x i32>
NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
@@ -445,7 +447,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
Result = Builder.CreateAlignedLoad(MinVecTy, CastedPtr, Alignment);
Worklist.pushValue(Result);
Result = Builder.CreateShuffleVector(Result, Mask);
- if (NeedCast) {
+ if (UseChunkedLoad) {
Worklist.pushValue(Result);
Result = Builder.CreateBitOrPointerCast(Result, I.getType());
}
>From 537c48de56189aa7c44647fa90364457eed08ea1 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Thu, 23 Jul 2026 15:23:28 +0900
Subject: [PATCH 23/24] Consolidate chunked load handling
---
.../Transforms/Vectorize/VectorCombine.cpp | 36 +++++++++----------
1 file changed, 16 insertions(+), 20 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 262cd6225afe2..827e8db7a8fe7 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -398,25 +398,29 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// Poison unused lanes, then gather the scalar chunks into the low lanes.
Mask.assign(MinVecNumElts, PoisonMaskElem);
std::iota(Mask.begin(), Mask.begin() + NumScalarChunks, OffsetEltIndex);
+
+ // The chunked path always shuffles from a non-zero source lane:
+ // %chunks = shufflevector <8 x i16> %wide.i16, <8 x i16> poison,
+ // <8 x i32> <i32 1, i32 2, i32 poison, i32 poison,
+ // i32 poison, i32 poison, i32 poison, i32 poison>
+ NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
+ MinVecTy, Mask, CostKind);
+
+ // Account for the bitcast after the unaligned chunk shuffle:
+ // %result = bitcast <8 x i16> %chunks to <4 x i32>
+ NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
+ TargetTransformInfo::CastContextHint::None,
+ CostKind);
} else {
// Poison unused lanes, select the scalar into lane zero, and resize the
// vector if needed.
unsigned OutputNumElts = Ty->getNumElements();
Mask.assign(OutputNumElts, PoisonMaskElem);
Mask[0] = OffsetEltIndex;
- }
- // Assume an offset-zero shuffle is free because codegen can use the loaded
- // vector directly.
- if (OffsetEltIndex) {
- if (UseChunkedLoad) {
- // Account for the chunk shuffle in the unaligned example:
- // %chunks = shufflevector <8 x i16> %wide.i16, <8 x i16> poison,
- // <8 x i32> <i32 1, i32 2, i32 poison, i32 poison,
- // i32 poison, i32 poison, i32 poison, i32 poison>
- NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
- MinVecTy, Mask, CostKind);
- } else {
+ // Assume an offset-zero shuffle is free because codegen can use the loaded
+ // vector directly.
+ if (OffsetEltIndex) {
// Account for the shuffle in the scalar-aligned example:
// %aligned = shufflevector <4 x i32> %wide.i32, <4 x i32> poison,
// <4 x i32> <i32 1, i32 poison, i32 poison, i32 poison>
@@ -425,14 +429,6 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
}
}
- if (UseChunkedLoad) {
- // Account for the bitcast after the unaligned chunk shuffle:
- // %result = bitcast <8 x i16> %chunks to <4 x i32>
- NewCost += TTI.getCastInstrCost(Instruction::BitCast, Ty, MinVecTy,
- TargetTransformInfo::CastContextHint::None,
- CostKind);
- }
-
// We can aggressively convert to the vector form because the backend can
// invert this transform if it does not result in a performance win.
if (OldCost < NewCost || !NewCost.isValid())
>From 8a9b9d8340a0b012973b5aa41c3b11d97d661517 Mon Sep 17 00:00:00 2001
From: hanbeom <kese111 at gmail.com>
Date: Thu, 23 Jul 2026 15:45:03 +0900
Subject: [PATCH 24/24] formatting
---
llvm/lib/Transforms/Vectorize/VectorCombine.cpp | 16 ++++++----------
1 file changed, 6 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
index 827e8db7a8fe7..11fd2097a0600 100644
--- a/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
+++ b/llvm/lib/Transforms/Vectorize/VectorCombine.cpp
@@ -298,8 +298,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
assert(isa<PointerType>(SrcPtr->getType()) && "Expected a pointer type");
unsigned MinVecNumElts = MinVectorSize / OriginalScalarSizeInBits;
- auto *MinVecTy =
- VectorType::get(OriginalScalarTy, MinVecNumElts, false);
+ auto *MinVecTy = VectorType::get(OriginalScalarTy, MinVecNumElts, false);
unsigned OffsetEltIndex = 0;
// An aligned scalar occupies one vector element. Unaligned accesses below
@@ -325,11 +324,9 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
if (Offset.isNegative())
return false;
- const uint64_t OriginalScalarSizeInBytes =
- OriginalScalarSizeInBits / 8;
+ const uint64_t OriginalScalarSizeInBytes = OriginalScalarSizeInBits / 8;
uint64_t ChunkSizeInBytes = OriginalScalarSizeInBytes;
- if (uint64_t UnalignedBytes =
- Offset.urem(OriginalScalarSizeInBytes)) {
+ if (uint64_t UnalignedBytes = Offset.urem(OriginalScalarSizeInBytes)) {
// Reconstruct an unaligned scalar from smaller integer chunks.
// Consecutive low-address chunks appear in increasing vector lanes on a
// little-endian target. Big-endian reconstruction would require a
@@ -339,8 +336,7 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// The GCD gives the largest chunk that exactly divides both the scalar
// width and byte offset. This minimizes the number of shuffle lanes.
- ChunkSizeInBytes =
- std::gcd(OriginalScalarSizeInBytes, UnalignedBytes);
+ ChunkSizeInBytes = std::gcd(OriginalScalarSizeInBytes, UnalignedBytes);
const uint64_t ChunkSizeInBits = ChunkSizeInBytes * 8;
NumScalarChunks = OriginalScalarSizeInBytes / ChunkSizeInBytes;
MinVecNumElts = MinVectorSize / ChunkSizeInBits;
@@ -403,8 +399,8 @@ bool VectorCombine::vectorizeLoadInsert(Instruction &I) {
// %chunks = shufflevector <8 x i16> %wide.i16, <8 x i16> poison,
// <8 x i32> <i32 1, i32 2, i32 poison, i32 poison,
// i32 poison, i32 poison, i32 poison, i32 poison>
- NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy,
- MinVecTy, Mask, CostKind);
+ NewCost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, MinVecTy, MinVecTy,
+ Mask, CostKind);
// Account for the bitcast after the unaligned chunk shuffle:
// %result = bitcast <8 x i16> %chunks to <4 x i32>
More information about the llvm-commits
mailing list