[llvm] [RISCV] Optimize repeated-half i64 vector splats on RV32 (PR #228119)

via llvm-commits llvm-commits at lists.llvm.org
Thu Oct 1 09:02:19 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-risc-v

Author: tinfengyu

<details>
<summary>Changes</summary>

On RV32, split-i64 vector splats with identical dynamic i32 halves can be
materialized through a temporary stack object, two scalar stores, and a
stride-zero e64 vector load. The existing equal-constant optimization does
not cover identical nonconstant SDValues.

For example:

  %lo = zext i32 %x to i64
  %hi = shl i64 %lo, 32
  %value = or i64 %lo, %hi

Integer legalization can expose the two halves as the same SDValue.

When the two nonconstant halves are identical and the passthrough is
undef-compatible, splat the i32 value at e32 and bitcast the result to the
required e64 vector type. Fill the intermediate e32 register group at VLMAX.

The existing constant handling and the fallback for defined passthrough are
unchanged.

Lo == Hi compares both the node and result number, so the two i32 halves
are the exact same SDValue. Repeating that value across the e32 vector
therefore constructs the required repeated-half i64 pattern.

Doubling the element count while halving the element width preserves the
total register-group width and required LMUL. Filling the complete e32
register group defines both halves of every e64 lane that can become active,
independently of the incoming AVL. This also covers an incoming AVL greater
than the e64 VLMAX, since the eventual e64 operation still clamps its active
VL to the e64 VLMAX.

Downstream users retain their own VL and element-width configuration.

Defining additional lanes is permitted for the undef-compatible passthrough.
Defined passthrough retains the existing lowering. The optimization only
removes compiler-generated temporary memory used to construct the splat;
user memory operations are unchanged.

The focused RV32 tests cover:

- dynamic repeated halves;
- unequal dynamic halves;
- defined passthrough;
- fixed-length and scalable vectors;
- numeric AVL=1 with restoration of the e64 store configuration;
- VLEN128 and VLEN256;
- preservation of the existing constant cases.

All 14 focused and nearby RUN pipelines passed on current main with
MachineVerifier enabled. Supplemental same-revision comparisons also
confirmed unchanged RV64 output and unchanged existing constant/raw-AVL
behavior.

For a two-element i64 store on current main, the baseline is:

  addi      sp, sp, -16
  addi      a2, sp, 8
  sw        a1, 8(sp)
  sw        a1, 12(sp)
  vsetivli  zero, 2, e64, m1, ta, ma
  vlse64.v  v8, (a2), zero
  vse64.v   v8, (a0)
  addi      sp, sp, 16
  ret

With this change:

  vsetvli   a2, zero, e32, m1, ta, ma
  vmv.v.x   v8, a1
  vsetivli  zero, 2, e64, m1, ta, ma
  vse64.v   v8, (a0)
  ret

With identical RV32GCV/ilp32d options, the focused kernel changes from nine
to five static instructions and from 24 to 18 symbol/.text bytes at both
minimum VLEN128 and VLEN256. The optimized sequence removes the stack
adjustment, scalar stores, and stride-zero load while restoring the e64
store's VL.

AI tool usage: OpenAI Codex was used to assist with investigation and
implementation. I reviewed the resulting implementation, tests, and
validation results and take responsibility for the contribution.

Assisted-by: OpenAI Codex

---
Full diff: https://github.com/llvm/llvm-project/pull/228119.diff


2 Files Affected:

- (modified) llvm/lib/Target/RISCV/RISCVISelLowering.cpp (+11) 
- (added) llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll (+153) 


``````````diff
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 7730a52a2e26d9..4d59bb1dd4c5c1 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -5212,6 +5212,17 @@ static SDValue splatPartsI64WithVL(const SDLoc &DL, MVT VT, SDValue Passthru,
     }
   }
 
+  // With identical nonconstant halves and undefined passthru, fill the whole
+  // register group at EEW=32. This defines every active i64 element regardless
+  // of VL, and may define additional elements that were undefined.
+  if (!isa<ConstantSDNode>(Lo) && Lo == Hi && Passthru.isUndef()) {
+    MVT InterVT = MVT::getVectorVT(MVT::i32, VT.getVectorElementCount() * 2);
+    SDValue MaxVL = DAG.getRegister(RISCV::X0, MVT::i32);
+    SDValue InterVec = DAG.getNode(RISCVISD::VMV_V_X_VL, DL, InterVT,
+                                   DAG.getUNDEF(InterVT), Lo, MaxVL);
+    return DAG.getNode(ISD::BITCAST, DL, VT, InterVec);
+  }
+
   // Detect cases where Hi is (SRA Lo, 31) which means Hi is Lo sign extended.
   if (Hi.getOpcode() == ISD::SRA && Hi.getOperand(0) == Lo &&
       isa<ConstantSDNode>(Hi.getOperand(1)) &&
diff --git a/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
new file mode 100644
index 00000000000000..7fc20ad136f551
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvv/splat-repeated-i32-rv32.ll
@@ -0,0 +1,153 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv32 -mattr=+v -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,VLEN128
+; RUN: llc -mtriple=riscv32 -mattr=+v -riscv-v-vector-bits-min=256 -verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,VLEN256
+
+; A dynamic i64 splat with identical i32 halves can use an e32 vmv.v.x.
+define <vscale x 1 x i64> @splat_repeated_i32(i32 %x) {
+; CHECK-LABEL: splat_repeated_i32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a1, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <vscale x 1 x i64> poison, i64 %repeated, i64 0
+  %splat = shufflevector <vscale x 1 x i64> %head, <vscale x 1 x i64> poison, <vscale x 1 x i32> zeroinitializer
+  ret <vscale x 1 x i64> %splat
+}
+
+; Different dynamic halves still require the i64 splat path.
+define <vscale x 1 x i64> @splat_different_i32(i32 %low, i32 %high) {
+; CHECK-LABEL: splat_different_i32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    addi sp, sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    sw a0, 8(sp)
+; CHECK-NEXT:    sw a1, 12(sp)
+; CHECK-NEXT:    addi a0, sp, 8
+; CHECK-NEXT:    vsetvli a1, zero, e64, m1, ta, ma
+; CHECK-NEXT:    vlse64.v v8, (a0), zero
+; CHECK-NEXT:    addi sp, sp, 16
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    ret
+  %lo = zext i32 %low to i64
+  %hi = zext i32 %high to i64
+  %shifted = shl i64 %hi, 32
+  %both = or i64 %lo, %shifted
+  %head = insertelement <vscale x 1 x i64> poison, i64 %both, i64 0
+  %splat = shufflevector <vscale x 1 x i64> %head, <vscale x 1 x i64> poison, <vscale x 1 x i32> zeroinitializer
+  ret <vscale x 1 x i64> %splat
+}
+
+define <2 x i64> @splat_repeated_i32_fixed(i32 %x) {
+; CHECK-LABEL: splat_repeated_i32_fixed:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a1, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <2 x i64> poison, i64 %repeated, i32 0
+  %splat = shufflevector <2 x i64> %head, <2 x i64> poison, <2 x i32> zeroinitializer
+  ret <2 x i64> %splat
+}
+
+define <vscale x 1 x i64> @splat_repeated_i32_tu(<vscale x 1 x i64> %passthru, i32 %x, i32 %vl) {
+; CHECK-LABEL: splat_repeated_i32_tu:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    addi sp, sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    sw a0, 8(sp)
+; CHECK-NEXT:    sw a0, 12(sp)
+; CHECK-NEXT:    addi a0, sp, 8
+; CHECK-NEXT:    vsetvli zero, a1, e64, m1, tu, ma
+; CHECK-NEXT:    vlse64.v v8, (a0), zero
+; CHECK-NEXT:    addi sp, sp, 16
+; CHECK-NEXT:    .cfi_def_cfa_offset 0
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> %passthru, i64 %repeated, i32 %vl)
+  ret <vscale x 1 x i64> %splat
+}
+
+declare <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64>, i64, i32)
+
+; A single-element fixed splat has numeric AVL 1. Fill the e32 container, then
+; restore the original e64 AVL for its user.
+define void @splat_repeated_i32_fixed1_store(ptr %p, i32 %x) {
+; CHECK-LABEL: splat_repeated_i32_fixed1_store:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a2, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.x v8, a1
+; CHECK-NEXT:    vsetivli zero, 1, e64, m1, ta, ma
+; CHECK-NEXT:    vse64.v v8, (a0)
+; CHECK-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <1 x i64> poison, i64 %repeated, i32 0
+  store <1 x i64> %head, ptr %p
+  ret void
+}
+
+; Legalization widens the fixed splat to four elements, giving it numeric AVL 4.
+; Fill the e32 container, then restore the e64 store's AVL 3.
+define void @splat_repeated_i32_fixed_store(ptr %p, i32 %x) {
+; VLEN128-LABEL: splat_repeated_i32_fixed_store:
+; VLEN128:       # %bb.0:
+; VLEN128-NEXT:    vsetvli a2, zero, e32, m2, ta, ma
+; VLEN128-NEXT:    vmv.v.x v8, a1
+; VLEN128-NEXT:    vsetivli zero, 3, e64, m2, ta, ma
+; VLEN128-NEXT:    vse64.v v8, (a0)
+; VLEN128-NEXT:    ret
+;
+; VLEN256-LABEL: splat_repeated_i32_fixed_store:
+; VLEN256:       # %bb.0:
+; VLEN256-NEXT:    vsetvli a2, zero, e32, m1, ta, ma
+; VLEN256-NEXT:    vmv.v.x v8, a1
+; VLEN256-NEXT:    vsetivli zero, 3, e64, m1, ta, ma
+; VLEN256-NEXT:    vse64.v v8, (a0)
+; VLEN256-NEXT:    ret
+  %low = zext i32 %x to i64
+  %high = shl i64 %low, 32
+  %repeated = or i64 %low, %high
+  %head = insertelement <3 x i64> poison, i64 %repeated, i32 0
+  %splat = shufflevector <3 x i64> %head, <3 x i64> poison, <3 x i32> zeroinitializer
+  store <3 x i64> %splat, ptr %p
+  ret void
+}
+
+; Preserve the existing equal-constant path, including its doubled numeric AVL.
+define <vscale x 1 x i64> @splat_equal_constant_avl1() {
+; CHECK-LABEL: splat_equal_constant_avl1:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetivli zero, 2, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 1
+; CHECK-NEXT:    ret
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> poison, i64 4294967297, i32 1)
+  ret <vscale x 1 x i64> %splat
+}
+
+define <vscale x 1 x i64> @splat_equal_constant_runtime(i32 %vl) {
+; CHECK-LABEL: splat_equal_constant_runtime:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a0, zero, e32, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 1
+; CHECK-NEXT:    ret
+  %splat = call <vscale x 1 x i64> @llvm.riscv.vmv.v.x.nxv1i64(<vscale x 1 x i64> poison, i64 4294967297, i32 %vl)
+  ret <vscale x 1 x i64> %splat
+}
+
+; Equal sign-extension constants must keep using the e64 fast path.
+define <vscale x 1 x i64> @splat_zero_constant() {
+; CHECK-LABEL: splat_zero_constant:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vsetvli a0, zero, e64, m1, ta, ma
+; CHECK-NEXT:    vmv.v.i v8, 0
+; CHECK-NEXT:    ret
+  ret <vscale x 1 x i64> zeroinitializer
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/228119


More information about the llvm-commits mailing list