[llvm] add f16 vector bitcast patterns (PR #226287)
Folkert de Vries via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 12:56:46 PDT 2026
https://github.com/folkertdev created https://github.com/llvm/llvm-project/pull/226287
came across this failure
https://godbolt.org/z/YPK78ea76
```llvm
; -O0 -mtriple=mipsel-linux-gnu -mcpu=mips32r5 -mattr=+msa,+fp64
define <4 x float> @f(<8 x half> %a) {
%b = bitcast <8 x half> %a to <4 x float>
ret <4 x float> %b
}
```
Gives
```
LLVM ERROR: Cannot select: 0x3b4ac900: v8f16 = bitcast 0x3b4acf90
0x3b4acf90: v4i32 = insert_vector_elt 0x3b4aceb0, 0x3b4ac5f0, Constant:i32<3>
0x3b4aceb0: v4i32 = insert_vector_elt 0x3b4acdd0, 0x3b4ac510, Constant:i32<2>
0x3b4acdd0: v4i32 = insert_vector_elt 0x3b4accf0, 0x3b4ac3c0, Constant:i32<1>
0x3b4accf0: v4i32 = insert_vector_elt undef:v4i32, 0x3b4ac2e0, Constant:i32<0>
0x3b4ac2e0: i32,ch = CopyFromReg 0x3b3c7b88, Register:i32 %1
0x3b4ac3c0: i32,ch = CopyFromReg 0x3b3c7b88, Register:i32 %2
0x3b4ac510: i32,ch = load<(load (s32) from %fixed-stack.1)> 0x3b3c7b88, FrameIndex:i32<-1>, poison:i32
0x3b4ac5f0: i32,ch = load<(load (s32) from %fixed-stack.0)> 0x3b3c7b88, FrameIndex:i32<-2>, poison:i32
In function: f
```
Just adding the patterns fixes that, and seems straightforward enough. I'm using `fexupl` in the tests to force the bitcast, otherwise the bitcast just gets optimized out (and just a store of the input vector type happens).
This test should maybe start using the automatic update script?
>From 80e33766a61abe3f5feeb00200f313cb95778023 Mon Sep 17 00:00:00 2001
From: Folkert de Vries <folkert at folkertdev.nl>
Date: Thu, 24 Sep 2026 21:47:53 +0200
Subject: [PATCH] add f16 vector bitcast patterns
---
llvm/lib/Target/Mips/MipsMSAInstrInfo.td | 6 +++
llvm/test/CodeGen/Mips/msa/bitcast.ll | 59 ++++++++++++++++--------
2 files changed, 47 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Target/Mips/MipsMSAInstrInfo.td b/llvm/lib/Target/Mips/MipsMSAInstrInfo.td
index a30fef338a6a3..3ecb079fbcded 100644
--- a/llvm/lib/Target/Mips/MipsMSAInstrInfo.td
+++ b/llvm/lib/Target/Mips/MipsMSAInstrInfo.td
@@ -3510,6 +3510,12 @@ def : MSABitconvertPat<v2i64, v4i32, MSA128D, [HasMSA, IsLE]>;
def : MSABitconvertPat<v2i64, v8f16, MSA128D, [HasMSA, IsLE]>;
def : MSABitconvertPat<v2i64, v4f32, MSA128D, [HasMSA, IsLE]>;
+def : MSABitconvertPat<v8f16, v16i8, MSA128H, [HasMSA, IsLE]>;
+def : MSABitconvertPat<v8f16, v4i32, MSA128H, [HasMSA, IsLE]>;
+def : MSABitconvertPat<v8f16, v2i64, MSA128H, [HasMSA, IsLE]>;
+def : MSABitconvertPat<v8f16, v4f32, MSA128H, [HasMSA, IsLE]>;
+def : MSABitconvertPat<v8f16, v2f64, MSA128H, [HasMSA, IsLE]>;
+
def : MSABitconvertPat<v4f32, v16i8, MSA128W, [HasMSA, IsLE]>;
def : MSABitconvertPat<v4f32, v8i16, MSA128W, [HasMSA, IsLE]>;
def : MSABitconvertPat<v4f32, v2i64, MSA128W, [HasMSA, IsLE]>;
diff --git a/llvm/test/CodeGen/Mips/msa/bitcast.ll b/llvm/test/CodeGen/Mips/msa/bitcast.ll
index c34e89b196e8f..7f553f010ad2a 100644
--- a/llvm/test/CodeGen/Mips/msa/bitcast.ll
+++ b/llvm/test/CodeGen/Mips/msa/bitcast.ll
@@ -59,20 +59,24 @@ entry:
%0 = load volatile <16 x i8>, ptr %src
%1 = tail call <16 x i8> @llvm.mips.addv.b(<16 x i8> %0, <16 x i8> %0)
%2 = bitcast <16 x i8> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v16i8_to_v8f16:
; LITENDIAN: ld.b [[R1:\$w[0-9]+]],
; LITENDIAN: addv.b [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.b [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v16i8_to_v8f16
; BIGENDIAN: v16i8_to_v8f16:
; BIGENDIAN: ld.b [[R1:\$w[0-9]+]],
; BIGENDIAN: addv.b [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.b [[R2]],
+; BIGENDIAN: shf.b [[R3:\$w[0-9]+]], [[R2]], 177
+; BIGENDIAN: fexupl.w [[R4:\$w[0-9]+]], [[R3]]
+; BIGENDIAN: st.w [[R4]],
; BIGENDIAN: .size v16i8_to_v8f16
define void @v16i8_to_v4i32(ptr %src, ptr %dst) nounwind {
@@ -233,20 +237,23 @@ entry:
%0 = load volatile <8 x i16>, ptr %src
%1 = tail call <8 x i16> @llvm.mips.addv.h(<8 x i16> %0, <8 x i16> %0)
%2 = bitcast <8 x i16> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v8i16_to_v8f16:
; LITENDIAN: ld.h [[R1:\$w[0-9]+]],
; LITENDIAN: addv.h [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.h [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v8i16_to_v8f16
; BIGENDIAN: v8i16_to_v8f16:
; BIGENDIAN: ld.h [[R1:\$w[0-9]+]],
; BIGENDIAN: addv.h [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.h [[R2]],
+; BIGENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; BIGENDIAN: st.w [[R3]],
; BIGENDIAN: .size v8i16_to_v8f16
define void @v8i16_to_v4i32(ptr %src, ptr %dst) nounwind {
@@ -568,20 +575,24 @@ entry:
%0 = load volatile <4 x i32>, ptr %src
%1 = tail call <4 x i32> @llvm.mips.addv.w(<4 x i32> %0, <4 x i32> %0)
%2 = bitcast <4 x i32> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v4i32_to_v8f16:
; LITENDIAN: ld.w [[R1:\$w[0-9]+]],
; LITENDIAN: addv.w [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.w [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v4i32_to_v8f16
; BIGENDIAN: v4i32_to_v8f16:
; BIGENDIAN: ld.w [[R1:\$w[0-9]+]],
; BIGENDIAN: addv.w [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.w [[R2]],
+; BIGENDIAN: shf.h [[R3:\$w[0-9]+]], [[R2]], 177
+; BIGENDIAN: fexupl.w [[R4:\$w[0-9]+]], [[R3]]
+; BIGENDIAN: st.w [[R4]],
; BIGENDIAN: .size v4i32_to_v8f16
define void @v4i32_to_v4i32(ptr %src, ptr %dst) nounwind {
@@ -739,20 +750,24 @@ entry:
%0 = load volatile <4 x float>, ptr %src
%1 = tail call <4 x float> @llvm.mips.fadd.w(<4 x float> %0, <4 x float> %0)
%2 = bitcast <4 x float> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v4f32_to_v8f16:
; LITENDIAN: ld.w [[R1:\$w[0-9]+]],
; LITENDIAN: fadd.w [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.w [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v4f32_to_v8f16
; BIGENDIAN: v4f32_to_v8f16:
; BIGENDIAN: ld.w [[R1:\$w[0-9]+]],
; BIGENDIAN: fadd.w [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.w [[R2]],
+; BIGENDIAN: shf.h [[R3:\$w[0-9]+]], [[R2]], 177
+; BIGENDIAN: fexupl.w [[R4:\$w[0-9]+]], [[R3]]
+; BIGENDIAN: st.w [[R4]],
; BIGENDIAN: .size v4f32_to_v8f16
define void @v4f32_to_v4i32(ptr %src, ptr %dst) nounwind {
@@ -911,20 +926,24 @@ entry:
%0 = load volatile <2 x i64>, ptr %src
%1 = tail call <2 x i64> @llvm.mips.addv.d(<2 x i64> %0, <2 x i64> %0)
%2 = bitcast <2 x i64> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v2i64_to_v8f16:
; LITENDIAN: ld.d [[R1:\$w[0-9]+]],
; LITENDIAN: addv.d [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.d [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v2i64_to_v8f16
; BIGENDIAN: v2i64_to_v8f16:
; BIGENDIAN: ld.d [[R1:\$w[0-9]+]],
; BIGENDIAN: addv.d [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.d [[R2]],
+; BIGENDIAN: shf.h [[R3:\$w[0-9]+]], [[R2]], 27
+; BIGENDIAN: fexupl.w [[R4:\$w[0-9]+]], [[R3]]
+; BIGENDIAN: st.w [[R4]],
; BIGENDIAN: .size v2i64_to_v8f16
define void @v2i64_to_v4i32(ptr %src, ptr %dst) nounwind {
@@ -1083,20 +1102,24 @@ entry:
%0 = load volatile <2 x double>, ptr %src
%1 = tail call <2 x double> @llvm.mips.fadd.d(<2 x double> %0, <2 x double> %0)
%2 = bitcast <2 x double> %1 to <8 x half>
- store <8 x half> %2, ptr %dst
+ %3 = tail call <4 x float> @llvm.mips.fexupl.w(<8 x half> %2)
+ store <4 x float> %3, ptr %dst
ret void
}
; LITENDIAN: v2f64_to_v8f16:
; LITENDIAN: ld.d [[R1:\$w[0-9]+]],
; LITENDIAN: fadd.d [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; LITENDIAN: st.d [[R2]],
+; LITENDIAN: fexupl.w [[R3:\$w[0-9]+]], [[R2]]
+; LITENDIAN: st.w [[R3]],
; LITENDIAN: .size v2f64_to_v8f16
; BIGENDIAN: v2f64_to_v8f16:
; BIGENDIAN: ld.d [[R1:\$w[0-9]+]],
; BIGENDIAN: fadd.d [[R2:\$w[0-9]+]], [[R1]], [[R1]]
-; BIGENDIAN: st.d [[R2]],
+; BIGENDIAN: shf.h [[R3:\$w[0-9]+]], [[R2]], 27
+; BIGENDIAN: fexupl.w [[R4:\$w[0-9]+]], [[R3]]
+; BIGENDIAN: st.w [[R4]],
; BIGENDIAN: .size v2f64_to_v8f16
define void @v2f64_to_v4i32(ptr %src, ptr %dst) nounwind {
More information about the llvm-commits
mailing list