[llvm] [RISCV] Optimize repeated-half i64 vector splats on RV32 (PR #228119)

via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 22:07:26 PDT 2026


https://github.com/tinfengyu updated https://github.com/llvm/llvm-project/pull/228119

>From 6c7e2e900de68f310049ffbda0195b4056c0e02a Mon Sep 17 00:00:00 2001
From: tinfeng <1768309929 at qq.com>
Date: Thu, 1 Oct 2026 19:36:37 +0800
Subject: [PATCH 1/2] [RISCV] Optimize repeated-half i64 vector splats on RV32

RV32 lowering currently materializes identical dynamic i32 halves through
an internal stack temporary and a stride-zero e64 vector load.

When both halves are the same nonconstant SDValue and the passthrough is
undef-compatible, splat the i32 value into an e32 vector and bitcast it to
the required i64 vector type. Use e32 VLMAX to define the complete register
group, so every active e64 element has both halves initialized regardless
of the incoming AVL. Defining additional lanes refines the undef passthrough.

Keep the pre-existing constant handling and defined-passthrough fallback
unchanged. This removes the temporary stack traffic for repeated dynamic
halves.
---
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   |  11 ++
 .../RISCV/rvv/splat-repeated-i32-rv32.ll      | 153 ++++++++++++++++++
 2 files changed, 164 insertions(+)
 create mode 100644 llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll

diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 7730a52a2e26d..4d59bb1dd4c5c 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -5212,6 +5212,17 @@ static SDValue splatPartsI64WithVL(const SDLoc &DL, MVT VT, SDValue Passthru,
     }
   }
 
+  // With identical nonconstant halves and undefined passthru, fill the whole
+  // register group at EEW=32. This defines every active i64 element regardless
+  // of VL, and may define additional elements that were undefined.
+  if (!isa<ConstantSDNode>(Lo) && Lo == Hi && Passthru.isUndef()) {
+    MVT InterVT = MVT::getVectorVT(MVT::i32, VT.getVectorElementCount() * 2);
+    SDValue MaxVL = DAG.getRegister(RISCV::X0, MVT::i32);
+    SDValue InterVec = DAG.getNode(RISCVISD::VMV_V_X_VL, DL, InterVT,
+                                   DAG.getUNDEF(InterVT), Lo, MaxVL);
+    return DAG.getNode(ISD::BITCAST, DL, VT, InterVec);
+  }
+
   // Detect cases where Hi is (SRA Lo, 31) which means Hi is Lo sign extended.
   if (Hi.getOpcode() == ISD::SRA && Hi.getOperand(0) == Lo &&
       isa<ConstantSDNode>(Hi.getOperand(1)) &&
diff --git a/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
new file mode 100644
index 0000000000000..7fc20ad136f55
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
@@ -0,0 +1,153 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv32 -mattr=+v -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,VLEN128
+; RUN: llc -mtriple=riscv32 -mattr=+v -riscv-v-vector-bits-min=256 -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,VLEN256
+
+; A dynamic i64 splat with identical i32 halves can use an e32 vmv.v.x.
+define <vscale x 1 x i64> @splat_repeated_i32(i32 %x) {
+; CHECK-LABEL: splat_repeated_i32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a1, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <vscale x 1 x i64> poison, i64 %repeated, i64 0
+  %splat = shufflevector <vscale x 1 x i64> %head, <vscale x 1 x i64> poison, <vscale x 1 x i32> zeroinitializer
+  ret <vscale x 1 x i64> %splat
+}
+
+; Different dynamic halves still require the i64 splat path.
+define <vscale x 1 x i64> @splat_different_i32(i32 %low, i32 %high) {
+; CHECK-LABEL: splat_different_i32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    addi sp, sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    sw a0, 8(sp)
+; CHECK-NEXT:    sw a1, 12(sp)
+; CHECK-NEXT:    addi a0, sp, 8
+; CHECK-NEXT:    vsetvli a1, zero, e64, m1, ta, ma
+; CHECK-NEXT:    vlse64.v v8, (a0), zero
+; CHECK-NEXT:    addi sp, sp, 16
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    ret
+  %lo = zext i32 %low to i64
+  %hi = zext i32 %high to i64
+  %shifted = shl i64 %hi, 32
+  %both = or i64 %lo, %shifted
+  %head = insertelement <vscale x 1 x i64> poison, i64 %both, i64 0
+  %splat = shufflevector <vscale x 1 x i64> %head, <vscale x 1 x i64> poison, <vscale x 1 x i32> zeroinitializer
+  ret <vscale x 1 x i64> %splat
+}
+
+define <2 x i64> @splat_repeated_i32_fixed(i32 %x) {
+; CHECK-LABEL: splat_repeated_i32_fixed:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a1, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <2 x i64> poison, i64 %repeated, i32 0
+  %splat = shufflevector <2 x i64> %head, <2 x i64> poison, <2 x i32> zeroinitializer
+  ret <2 x i64> %splat
+}
+
+define <vscale x 1 x i64> @splat_repeated_i32_tu(<vscale x 1 x i64> %passthru, i32 %x, i32 %vl) {
+; CHECK-LABEL: splat_repeated_i32_tu:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    addi sp, sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    sw a0, 8(sp)
+; CHECK-NEXT:    sw a0, 12(sp)
+; CHECK-NEXT:    addi a0, sp, 8
+; CHECK-NEXT:    vsetvli zero, a1, e64, m1, tu, ma
+; CHECK-NEXT:    vlse64.v v8, (a0), zero
+; CHECK-NEXT:    addi sp, sp, 16
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> %passthru, i64 %repeated, i32 %vl)
+  ret <vscale x 1 x i64> %splat
+}
+
+declare <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64>, i64, i32)
+
+; A single-element fixed splat has numeric AVL 1. Fill the e32 container, then
+; restore the original e64 AVL for its user.
+define void @splat_repeated_i32_fixed1_store(ptr %p, i32 %x) {
+; CHECK-LABEL: splat_repeated_i32_fixed1_store:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a2, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a1
+; CHECK-NEXT:    vsetivli zero, 1, e64, m1, ta, ma
+; CHECK-NEXT:    vse64.v v8, (a0)
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <1 x i64> poison, i64 %repeated, i32 0
+  store <1 x i64> %head, ptr %p
+  ret void
+}
+
+; Legalization widens the fixed splat to four elements, giving it numeric AVL 4.
+; Fill the e32 container, then restore the e64 store's AVL 3.
+define void @splat_repeated_i32_fixed_store(ptr %p, i32 %x) {
+; VLEN128-LABEL: splat_repeated_i32_fixed_store:
+; VLEN128:       # %bb.0:
+; VLEN128-NEXT:    vsetvli a2, zero, e32, m2, ta, ma
+; VLEN128-NEXT:    vmv.v.x v8, a1
+; VLEN128-NEXT:    vsetivli zero, 3, e64, m2, ta, ma
+; VLEN128-NEXT:    vse64.v v8, (a0)
+; VLEN128-NEXT:    ret
+;
+; VLEN256-LABEL: splat_repeated_i32_fixed_store:
+; VLEN256:       # %bb.0:
+; VLEN256-NEXT:    vsetvli a2, zero, e32, m1, ta, ma
+; VLEN256-NEXT:    vmv.v.x v8, a1
+; VLEN256-NEXT:    vsetivli zero, 3, e64, m1, ta, ma
+; VLEN256-NEXT:    vse64.v v8, (a0)
+; VLEN256-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <3 x i64> poison, i64 %repeated, i32 0
+  %splat = shufflevector <3 x i64> %head, <3 x i64> poison, <3 x i32> zeroinitializer
+  store <3 x i64> %splat, ptr %p
+  ret void
+}
+
+; Preserve the existing equal-constant path, including its doubled numeric AVL.
+define <vscale x 1 x i64> @splat_equal_constant_avl1() {
+; CHECK-LABEL: splat_equal_constant_avl1:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetivli zero, 2, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 1
+; CHECK-NEXT:    ret
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> poison, i64 4294967297, i32 1)
+  ret <vscale x 1 x i64> %splat
+}
+
+define <vscale x 1 x i64> @splat_equal_constant_runtime(i32 %vl) {
+; CHECK-LABEL: splat_equal_constant_runtime:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a0, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 1
+; CHECK-NEXT:    ret
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> poison, i64 4294967297, i32 %vl)
+  ret <vscale x 1 x i64> %splat
+}
+
+; Equal sign-extension constants must keep using the e64 fast path.
+define <vscale x 1 x i64> @splat_zero_constant() {
+; CHECK-LABEL: splat_zero_constant:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a0, zero, e64, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 0
+; CHECK-NEXT:    ret
+  ret <vscale x 1 x i64> zeroinitializer
+}

>From 4e85d91db126283aef8e06b16e81d19af3625962 Mon Sep 17 00:00:00 2001
From: tinfeng <1768309929 at qq.com>
Date: Fri, 2 Oct 2026 13:06:00 +0800
Subject: [PATCH 2/2] [RISCV] Preserve passthru for equal-constant i64 splats

Only use the e32 fast path when the passthru is undef-compatible. Defined-passthru cases now use the existing fallback.
---
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   |  2 +-
 .../RISCV/rvv/splat-repeated-i32-rv32.ll      | 33 +++++++++++++++++++
 2 files changed, 34 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 4d59bb1dd4c5c..aee355d600299 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -5198,7 +5198,7 @@ static SDValue splatPartsI64WithVL(const SDLoc &DL, MVT VT, SDValue Passthru,
 
     // Use vmv.v.x with EEW=32.  Use either a vsetivli or vsetvli to change
     // VL.  This can temporarily increase VL if VL less than VLMAX.
-    if (LoC == HiC) {
+    if (LoC == HiC && Passthru.isUndef()) {
       SDValue NewVL;
       if (isa<ConstantSDNode>(VL) && isUInt<4>(VL->getAsZExtVal()))
         NewVL = DAG.getNode(ISD::ADD, DL, VL.getValueType(), VL, VL);
diff --git a/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
index 7fc20ad136f55..65a1688490bf4 100644
--- a/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
@@ -151,3 +151,36 @@ define <vscale x 1 x i64> @splat_zero_constant() {
 ; CHECK-NEXT:    ret
   ret <vscale x 1 x i64> zeroinitializer
 }
+
+declare <vscale x 2 x i64> @llvm.riscv.vle.nxv2i64(<vscale x 2 x i64>, ptr, i32)
+declare <vscale x 2 x i64> @llvm.riscv.vmv.v.x.nxv2i64(<vscale x 2 x i64>, i64, i32)
+declare void @llvm.riscv.vse.nxv2i64(<vscale x 2 x i64>, ptr, i32)
+
+; Preserve defined passthru for repeated constants.
+define void @splat_equal_constant_defined_passthru(ptr %in, ptr %out, i32 %vl) {
+; CHECK-LABEL: splat_equal_constant_defined_passthru:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    addi sp, sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    vsetivli zero, 2, e64, m2, ta, ma
+; CHECK-NEXT:    vle64.v v8, (a0)
+; CHECK-NEXT:    lui a0, 74565
+; CHECK-NEXT:    addi a0, a0, 1656
+; CHECK-NEXT:    sw a0, 8(sp)
+; CHECK-NEXT:    sw a0, 12(sp)
+; CHECK-NEXT:    addi a0, sp, 8
+; CHECK-NEXT:    vsetvli zero, a2, e64, m2, tu, ma
+; CHECK-NEXT:    vlse64.v v8, (a0), zero
+; CHECK-NEXT:    vsetivli zero, 2, e64, m2, ta, ma
+; CHECK-NEXT:    vse64.v v8, (a1)
+; CHECK-NEXT:    addi sp, sp, 16
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    ret
+  %passthru = call <vscale x 2 x i64> @llvm.riscv.vle.nxv2i64(
+      <vscale x 2 x i64> poison, ptr %in, i32 2)
+  %splat = call <vscale x 2 x i64> @llvm.riscv.vmv.v.x.nxv2i64(
+      <vscale x 2 x i64> %passthru, i64 1311768465173141112, i32 %vl)
+  call void @llvm.riscv.vse.nxv2i64(
+      <vscale x 2 x i64> %splat, ptr %out, i32 2)
+  ret void
+}



More information about the llvm-commits mailing list