[llvm] [RISCV] Consider truncate semantics in performINSERT_VECTOR_ELTCombine (PR #228243)

Alex Bradbury via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 04:22:40 PDT 2026


https://github.com/asb updated https://github.com/llvm/llvm-project/pull/228243

>From 91f3a0ea3e33b0aebeceeed1f9c0f08dc2983d6c Mon Sep 17 00:00:00 2001
From: Alex Bradbury <asb at igalia.com>
Date: Thu, 1 Oct 2026 21:45:12 +0100
Subject: [PATCH 1/2] [RISCV][test] Add test for insert_vector_elt of truncated
 binop combine

performINSERT_VECTOR_ELTCombine currently miscompiles this test (the
lshrs happen after truncation).
---
 .../rvv/fixed-vectors-buildvec-of-binop.ll    | 39 +++++++++++++++++++
 1 file changed, 39 insertions(+)

diff --git a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
index a1ade4139fe516..2df65d27fc0711 100644
--- a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
@@ -624,3 +624,42 @@ entry:
   %3 = insertelement <2 x i32> %2, i32 %1, i64 1
   ret <2 x i32> %3
 }
+
+; The scalar lshrs are performed on i32 and then implicitly truncated by the
+; insert_vector_elt, so they must not be combined into a vector lshr on the i8
+; elements, which would truncate before the shift instead of after it.
+; FIXME: This is currently miscompiled (the lshrs happen after truncation).
+define <8 x i8> @insert_elt_of_trunc_op(i32 %a) {
+; RV32-LABEL: insert_elt_of_trunc_op:
+; RV32:       # %bb.0: # %entry
+; RV32-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
+; RV32-NEXT:    vid.v v8
+; RV32-NEXT:    vmv.v.x v9, a0
+; RV32-NEXT:    vadd.vi v8, v8, 2
+; RV32-NEXT:    vsrl.vv v8, v9, v8
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: insert_elt_of_trunc_op:
+; RV64:       # %bb.0: # %entry
+; RV64-NEXT:    li a1, 192
+; RV64-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
+; RV64-NEXT:    vmv.s.x v0, a1
+; RV64-NEXT:    vmv.v.x v8, a0
+; RV64-NEXT:    vid.v v9
+; RV64-NEXT:    vmerge.vxm v8, v8, a0, v0
+; RV64-NEXT:    vadd.vi v9, v9, 2
+; RV64-NEXT:    vsrl.vv v8, v8, v9
+; RV64-NEXT:    ret
+entry:
+  %b = trunc i32 %a to i8
+  %s8 = lshr i32 %a, 8
+  %s9 = lshr i32 %a, 9
+  %t8 = trunc i32 %s8 to i8
+  %t9 = trunc i32 %s9 to i8
+  %v0 = insertelement <8 x i8> poison, i8 %b, i64 0
+  %v1 = shufflevector <8 x i8> %v0, <8 x i8> poison, <8 x i32> zeroinitializer
+  %v2 = lshr <8 x i8> %v1, <i8 2, i8 3, i8 4, i8 5, i8 6, i8 7, i8 poison, i8 poison>
+  %v3 = insertelement <8 x i8> %v2, i8 %t8, i64 6
+  %v4 = insertelement <8 x i8> %v3, i8 %t9, i64 7
+  ret <8 x i8> %v4
+}

>From 6e0eba0fedbdbe3db2e8a4826596feaafadb85de Mon Sep 17 00:00:00 2001
From: Alex Bradbury <asb at igalia.com>
Date: Thu, 1 Oct 2026 21:54:47 +0100
Subject: [PATCH 2/2] [RISCV] Consider truncate semantics in
 performINSERT_VECTOR_ELTCombine

This fixes a miscompile in performINSERT_VECTOR_ELTCombine dating back
to when it was added (#72675), but only just showing up through testing
when a recent unrelated change triggered vectorisation for a function
within Clang, leading to broken builds on some of the two-stage RVV buildbots.

After type legalization the scalar binop can be wider than the vector
element type, with the insert implicitly truncating it. That isn't
equivalent for shifts and a similar bug was found and fixed in a
neighbouring combine in #81168. This patch just applies the same fix
(with the same comment even), avoiding the transform if the scalar type
doesn't match the element type.

I used an LLM to root cause the issue and produce the test case.
---
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   |  4 ++
 .../rvv/fixed-vectors-buildvec-of-binop.ll    | 37 ++++++++-----------
 2 files changed, 20 insertions(+), 21 deletions(-)

diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 13669fa56766de..f3119ad66f0c78 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -23698,6 +23698,10 @@ static SDValue performINSERT_VECTOR_ELTCombine(SDNode *N, SelectionDAG &DAG,
       return SDValue();
     if (!isa<ConstantSDNode>(InValRHS) && !isa<ConstantFPSDNode>(InValRHS))
       return SDValue();
+    // This INSERT_VECTOR_ELT involves an implicit truncation, and sinking
+    // truncates through binops is non-trivial.
+    if (InVal.getValueType() != VT.getVectorElementType())
+      return SDValue();
     // FIXME: Return failure if the RHS type doesn't match the LHS. Shifts may
     // have different LHS and RHS types.
     if (InVec.getOperand(0).getValueType() != InVec.getOperand(1).getValueType())
diff --git a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
index 2df65d27fc0711..0fb5ea38098a94 100644
--- a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
@@ -628,28 +628,23 @@ entry:
 ; The scalar lshrs are performed on i32 and then implicitly truncated by the
 ; insert_vector_elt, so they must not be combined into a vector lshr on the i8
 ; elements, which would truncate before the shift instead of after it.
-; FIXME: This is currently miscompiled (the lshrs happen after truncation).
 define <8 x i8> @insert_elt_of_trunc_op(i32 %a) {
-; RV32-LABEL: insert_elt_of_trunc_op:
-; RV32:       # %bb.0: # %entry
-; RV32-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
-; RV32-NEXT:    vid.v v8
-; RV32-NEXT:    vmv.v.x v9, a0
-; RV32-NEXT:    vadd.vi v8, v8, 2
-; RV32-NEXT:    vsrl.vv v8, v9, v8
-; RV32-NEXT:    ret
-;
-; RV64-LABEL: insert_elt_of_trunc_op:
-; RV64:       # %bb.0: # %entry
-; RV64-NEXT:    li a1, 192
-; RV64-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
-; RV64-NEXT:    vmv.s.x v0, a1
-; RV64-NEXT:    vmv.v.x v8, a0
-; RV64-NEXT:    vid.v v9
-; RV64-NEXT:    vmerge.vxm v8, v8, a0, v0
-; RV64-NEXT:    vadd.vi v9, v9, 2
-; RV64-NEXT:    vsrl.vv v8, v8, v9
-; RV64-NEXT:    ret
+; CHECK-LABEL: insert_elt_of_trunc_op:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
+; CHECK-NEXT:    vid.v v8
+; CHECK-NEXT:    vmv.v.x v9, a0
+; CHECK-NEXT:    vadd.vi v8, v8, 2
+; CHECK-NEXT:    srli a1, a0, 8
+; CHECK-NEXT:    vsrl.vv v8, v9, v8
+; CHECK-NEXT:    vmv.s.x v9, a1
+; CHECK-NEXT:    srli a0, a0, 9
+; CHECK-NEXT:    vsetivli zero, 7, e8, mf2, tu, ma
+; CHECK-NEXT:    vslideup.vi v8, v9, 6
+; CHECK-NEXT:    vmv.s.x v9, a0
+; CHECK-NEXT:    vsetivli zero, 8, e8, mf2, ta, ma
+; CHECK-NEXT:    vslideup.vi v8, v9, 7
+; CHECK-NEXT:    ret
 entry:
   %b = trunc i32 %a to i8
   %s8 = lshr i32 %a, 8



More information about the llvm-commits mailing list