[llvm] [RISCV] Consider truncate semantics in performINSERT_VECTOR_ELTCombine (PR #228243)
Alex Bradbury via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 04:22:40 PDT 2026
https://github.com/asb updated https://github.com/llvm/llvm-project/pull/228243
>From 91f3a0ea3e33b0aebeceeed1f9c0f08dc2983d6c Mon Sep 17 00:00:00 2001
From: Alex Bradbury <asb at igalia.com>
Date: Thu, 1 Oct 2026 21:45:12 +0100
Subject: [PATCH 1/2] [RISCV][test] Add test for insert_vector_elt of truncated
binop combine
performINSERT_VECTOR_ELTCombine currently miscompiles this test (the
lshrs happen after truncation).
---
.../rvv/fixed-vectors-buildvec-of-binop.ll | 39 +++++++++++++++++++
1 file changed, 39 insertions(+)
diff --git a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
index a1ade4139fe516..2df65d27fc0711 100644
--- a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
@@ -624,3 +624,42 @@ entry:
%3 = insertelement <2 x i32> %2, i32 %1, i64 1
ret <2 x i32> %3
}
+
+; The scalar lshrs are performed on i32 and then implicitly truncated by the
+; insert_vector_elt, so they must not be combined into a vector lshr on the i8
+; elements, which would truncate before the shift instead of after it.
+; FIXME: This is currently miscompiled (the lshrs happen after truncation).
+define <8 x i8> @insert_elt_of_trunc_op(i32 %a) {
+; RV32-LABEL: insert_elt_of_trunc_op:
+; RV32: # %bb.0: # %entry
+; RV32-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
+; RV32-NEXT: vid.v v8
+; RV32-NEXT: vmv.v.x v9, a0
+; RV32-NEXT: vadd.vi v8, v8, 2
+; RV32-NEXT: vsrl.vv v8, v9, v8
+; RV32-NEXT: ret
+;
+; RV64-LABEL: insert_elt_of_trunc_op:
+; RV64: # %bb.0: # %entry
+; RV64-NEXT: li a1, 192
+; RV64-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
+; RV64-NEXT: vmv.s.x v0, a1
+; RV64-NEXT: vmv.v.x v8, a0
+; RV64-NEXT: vid.v v9
+; RV64-NEXT: vmerge.vxm v8, v8, a0, v0
+; RV64-NEXT: vadd.vi v9, v9, 2
+; RV64-NEXT: vsrl.vv v8, v8, v9
+; RV64-NEXT: ret
+entry:
+ %b = trunc i32 %a to i8
+ %s8 = lshr i32 %a, 8
+ %s9 = lshr i32 %a, 9
+ %t8 = trunc i32 %s8 to i8
+ %t9 = trunc i32 %s9 to i8
+ %v0 = insertelement <8 x i8> poison, i8 %b, i64 0
+ %v1 = shufflevector <8 x i8> %v0, <8 x i8> poison, <8 x i32> zeroinitializer
+ %v2 = lshr <8 x i8> %v1, <i8 2, i8 3, i8 4, i8 5, i8 6, i8 7, i8 poison, i8 poison>
+ %v3 = insertelement <8 x i8> %v2, i8 %t8, i64 6
+ %v4 = insertelement <8 x i8> %v3, i8 %t9, i64 7
+ ret <8 x i8> %v4
+}
>From 6e0eba0fedbdbe3db2e8a4826596feaafadb85de Mon Sep 17 00:00:00 2001
From: Alex Bradbury <asb at igalia.com>
Date: Thu, 1 Oct 2026 21:54:47 +0100
Subject: [PATCH 2/2] [RISCV] Consider truncate semantics in
performINSERT_VECTOR_ELTCombine
This fixes a miscompile in performINSERT_VECTOR_ELTCombine dating back
to when it was added (#72675), but only just showing up through testing
when a recent unrelated change triggered vectorisation for a function
within Clang, leading to broken builds on some of the two-stage RVV buildbots.
After type legalization the scalar binop can be wider than the vector
element type, with the insert implicitly truncating it. That isn't
equivalent for shifts and a similar bug was found and fixed in a
neighbouring combine in #81168. This patch just applies the same fix
(with the same comment even), avoiding the transform if the scalar type
doesn't match the element type.
I used an LLM to root cause the issue and produce the test case.
---
llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 4 ++
.../rvv/fixed-vectors-buildvec-of-binop.ll | 37 ++++++++-----------
2 files changed, 20 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 13669fa56766de..f3119ad66f0c78 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -23698,6 +23698,10 @@ static SDValue performINSERT_VECTOR_ELTCombine(SDNode *N, SelectionDAG &DAG,
return SDValue();
if (!isa<ConstantSDNode>(InValRHS) && !isa<ConstantFPSDNode>(InValRHS))
return SDValue();
+ // This INSERT_VECTOR_ELT involves an implicit truncation, and sinking
+ // truncates through binops is non-trivial.
+ if (InVal.getValueType() != VT.getVectorElementType())
+ return SDValue();
// FIXME: Return failure if the RHS type doesn't match the LHS. Shifts may
// have different LHS and RHS types.
if (InVec.getOperand(0).getValueType() != InVec.getOperand(1).getValueType())
diff --git a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
index 2df65d27fc0711..0fb5ea38098a94 100644
--- a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-buildvec-of-binop.ll
@@ -628,28 +628,23 @@ entry:
; The scalar lshrs are performed on i32 and then implicitly truncated by the
; insert_vector_elt, so they must not be combined into a vector lshr on the i8
; elements, which would truncate before the shift instead of after it.
-; FIXME: This is currently miscompiled (the lshrs happen after truncation).
define <8 x i8> @insert_elt_of_trunc_op(i32 %a) {
-; RV32-LABEL: insert_elt_of_trunc_op:
-; RV32: # %bb.0: # %entry
-; RV32-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
-; RV32-NEXT: vid.v v8
-; RV32-NEXT: vmv.v.x v9, a0
-; RV32-NEXT: vadd.vi v8, v8, 2
-; RV32-NEXT: vsrl.vv v8, v9, v8
-; RV32-NEXT: ret
-;
-; RV64-LABEL: insert_elt_of_trunc_op:
-; RV64: # %bb.0: # %entry
-; RV64-NEXT: li a1, 192
-; RV64-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
-; RV64-NEXT: vmv.s.x v0, a1
-; RV64-NEXT: vmv.v.x v8, a0
-; RV64-NEXT: vid.v v9
-; RV64-NEXT: vmerge.vxm v8, v8, a0, v0
-; RV64-NEXT: vadd.vi v9, v9, 2
-; RV64-NEXT: vsrl.vv v8, v8, v9
-; RV64-NEXT: ret
+; CHECK-LABEL: insert_elt_of_trunc_op:
+; CHECK: # %bb.0: # %entry
+; CHECK-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
+; CHECK-NEXT: vid.v v8
+; CHECK-NEXT: vmv.v.x v9, a0
+; CHECK-NEXT: vadd.vi v8, v8, 2
+; CHECK-NEXT: srli a1, a0, 8
+; CHECK-NEXT: vsrl.vv v8, v9, v8
+; CHECK-NEXT: vmv.s.x v9, a1
+; CHECK-NEXT: srli a0, a0, 9
+; CHECK-NEXT: vsetivli zero, 7, e8, mf2, tu, ma
+; CHECK-NEXT: vslideup.vi v8, v9, 6
+; CHECK-NEXT: vmv.s.x v9, a0
+; CHECK-NEXT: vsetivli zero, 8, e8, mf2, ta, ma
+; CHECK-NEXT: vslideup.vi v8, v9, 7
+; CHECK-NEXT: ret
entry:
%b = trunc i32 %a to i8
%s8 = lshr i32 %a, 8
More information about the llvm-commits
mailing list