[llvm] [WebAssembly] Fix the i32x4 case of simd splat shuffle cost (PR #191964)

Jasmine Tang via llvm-commits llvm-commits at lists.llvm.org
Mon Apr 13 23:22:32 PDT 2026


https://github.com/badumbatish created https://github.com/llvm/llvm-project/pull/191964

Inspiration taken from llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp's AArch64TTIImpl::getVectorInstrCost

>From a77a72e45a991bf00b46d5d22836d09817cbca28 Mon Sep 17 00:00:00 2001
From: Jasmine Tang <jjasmine at igalia.com>
Date: Mon, 13 Apr 2026 23:19:53 -0700
Subject: [PATCH] [WebAssembly] Fix the i32x4 case of simd splat shuffle cost

---
 .../WebAssemblyTargetTransformInfo.cpp        |  7 +++++
 .../WebAssembly/simd-splat-shuffle-cost.ll    | 28 ++++++++-----------
 2 files changed, 18 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Target/WebAssembly/WebAssemblyTargetTransformInfo.cpp b/llvm/lib/Target/WebAssembly/WebAssemblyTargetTransformInfo.cpp
index 1bd3a6b950944..0b2f84d42cd2e 100644
--- a/llvm/lib/Target/WebAssembly/WebAssemblyTargetTransformInfo.cpp
+++ b/llvm/lib/Target/WebAssembly/WebAssemblyTargetTransformInfo.cpp
@@ -406,6 +406,13 @@ InstructionCost WebAssemblyTTIImpl::getInterleavedMemoryOpCost(
 InstructionCost WebAssemblyTTIImpl::getVectorInstrCost(
     unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
     const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
+  // Treat insert at lane 0 into a poison vector as having zero cost. The
+  // insert + broadcast shuffle pair will be lowered to a single splat
+  // instruction, so the insert is free.
+  if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
+      isa<PoisonValue>(Op0))
+    return 0;
+
   InstructionCost Cost = BasicTTIImplBase::getVectorInstrCost(
       Opcode, Val, CostKind, Index, Op0, Op1, VIC);
 
diff --git a/llvm/test/Transforms/SLPVectorizer/WebAssembly/simd-splat-shuffle-cost.ll b/llvm/test/Transforms/SLPVectorizer/WebAssembly/simd-splat-shuffle-cost.ll
index c1ea504a0c140..fc3fc012fda88 100644
--- a/llvm/test/Transforms/SLPVectorizer/WebAssembly/simd-splat-shuffle-cost.ll
+++ b/llvm/test/Transforms/SLPVectorizer/WebAssembly/simd-splat-shuffle-cost.ll
@@ -132,13 +132,9 @@ entry:
 define void @splat_i32x4(i32 %v, ptr noalias %p) {
 ; SIMD128-LABEL: @splat_i32x4(
 ; SIMD128-NEXT:  entry:
-; SIMD128-NEXT:    store i32 [[V:%.*]], ptr [[P:%.*]], align 1
-; SIMD128-NEXT:    [[IDX1:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 4
-; SIMD128-NEXT:    store i32 [[V]], ptr [[IDX1]], align 1
-; SIMD128-NEXT:    [[IDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 8
-; SIMD128-NEXT:    store i32 [[V]], ptr [[IDX2]], align 1
-; SIMD128-NEXT:    [[IDX3:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 12
-; SIMD128-NEXT:    store i32 [[V]], ptr [[IDX3]], align 1
+; SIMD128-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[V:%.*]], i32 0
+; SIMD128-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SIMD128-NEXT:    store <4 x i32> [[TMP1]], ptr [[P:%.*]], align 1
 ; SIMD128-NEXT:    ret void
 ;
 ; NO-SIMD128-LABEL: @splat_i32x4(
@@ -189,13 +185,9 @@ entry:
 define void @splat_f32x4(float %v, ptr noalias %p) {
 ; SIMD128-LABEL: @splat_f32x4(
 ; SIMD128-NEXT:  entry:
-; SIMD128-NEXT:    store float [[V:%.*]], ptr [[P:%.*]], align 1
-; SIMD128-NEXT:    [[IDX1:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 4
-; SIMD128-NEXT:    store float [[V]], ptr [[IDX1]], align 1
-; SIMD128-NEXT:    [[IDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 8
-; SIMD128-NEXT:    store float [[V]], ptr [[IDX2]], align 1
-; SIMD128-NEXT:    [[IDX3:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 12
-; SIMD128-NEXT:    store float [[V]], ptr [[IDX3]], align 1
+; SIMD128-NEXT:    [[TMP0:%.*]] = insertelement <4 x float> poison, float [[V:%.*]], i32 0
+; SIMD128-NEXT:    [[TMP1:%.*]] = shufflevector <4 x float> [[TMP0]], <4 x float> poison, <4 x i32> zeroinitializer
+; SIMD128-NEXT:    store <4 x float> [[TMP1]], ptr [[P:%.*]], align 1
 ; SIMD128-NEXT:    ret void
 ;
 ; NO-SIMD128-LABEL: @splat_f32x4(
@@ -245,9 +237,11 @@ entry:
 define void @splat_i32x8(i32 %v, ptr noalias %p) {
 ; SIMD128-LABEL: @splat_i32x8(
 ; SIMD128-NEXT:  entry:
-; SIMD128-NEXT:    [[TMP0:%.*]] = insertelement <8 x i32> poison, i32 [[V:%.*]], i32 0
-; SIMD128-NEXT:    [[TMP1:%.*]] = shufflevector <8 x i32> [[TMP0]], <8 x i32> poison, <8 x i32> zeroinitializer
-; SIMD128-NEXT:    store <8 x i32> [[TMP1]], ptr [[P:%.*]], align 1
+; SIMD128-NEXT:    [[TMP0:%.*]] = insertelement <4 x i32> poison, i32 [[V:%.*]], i32 0
+; SIMD128-NEXT:    [[TMP1:%.*]] = shufflevector <4 x i32> [[TMP0]], <4 x i32> poison, <4 x i32> zeroinitializer
+; SIMD128-NEXT:    store <4 x i32> [[TMP1]], ptr [[P:%.*]], align 1
+; SIMD128-NEXT:    [[IDX4:%.*]] = getelementptr inbounds nuw i8, ptr [[P]], i32 16
+; SIMD128-NEXT:    store <4 x i32> [[TMP1]], ptr [[IDX4]], align 1
 ; SIMD128-NEXT:    ret void
 ;
 ; NO-SIMD128-LABEL: @splat_i32x8(



More information about the llvm-commits mailing list