[llvm] [AArch64] Improve scalar fixed-point int-to-fp codegen (PR #220299)
Vimal Patel via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 07:03:44 PDT 2026
https://github.com/pvimal816a updated https://github.com/llvm/llvm-project/pull/220299
>From 261aa3020cb53339b0e349a3f9fee173f4a9ca65 Mon Sep 17 00:00:00 2001
From: Vimal Patel <vimal.patel at arm.com>
Date: Tue, 1 Sep 2026 14:08:38 +0000
Subject: [PATCH 1/2] [AArch64] Improve scalar fixed-point int-to-fp codegen
Select the scalar FPR shifted conversion forms when the integer input
is already in an FPR, and keep the GPR forms for integer inputs in
GPRs.
Add coverage for intrinsic and sitofp/uitofp fixed-point conversions.
---
llvm/lib/Target/AArch64/AArch64InstrInfo.td | 80 +++--
...arm64-fixed-point-scalar-cvt-dagcombine.ll | 87 ++++-
llvm/test/CodeGen/AArch64/fcvt-fixed.ll | 330 +++++++++++++++++-
3 files changed, 460 insertions(+), 37 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index 3b226e9a78105..b7b1c75332f80 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -5807,34 +5807,6 @@ let Predicates = [HasFPRCVT] in {
(UCVTFDSr (EXTRACT_SUBREG V64:$Rn, ssub))>;
}
-def : Pat<(f16 (fdiv (f16 (any_sint_to_fp (i32 GPR32:$Rn))), fixedpoint_f16_i32:$scale)),
- (SCVTFSWHri GPR32:$Rn, fixedpoint_f16_i32:$scale)>;
-def : Pat<(f32 (fdiv (f32 (any_sint_to_fp (i32 GPR32:$Rn))), fixedpoint_f32_i32:$scale)),
- (SCVTFSWSri GPR32:$Rn, fixedpoint_f32_i32:$scale)>;
-def : Pat<(f64 (fdiv (f64 (any_sint_to_fp (i32 GPR32:$Rn))), fixedpoint_f64_i32:$scale)),
- (SCVTFSWDri GPR32:$Rn, fixedpoint_f64_i32:$scale)>;
-
-def : Pat<(f16 (fdiv (f16 (any_sint_to_fp (i64 GPR64:$Rn))), fixedpoint_f16_i64:$scale)),
- (SCVTFSXHri GPR64:$Rn, fixedpoint_f16_i64:$scale)>;
-def : Pat<(f32 (fdiv (f32 (any_sint_to_fp (i64 GPR64:$Rn))), fixedpoint_f32_i64:$scale)),
- (SCVTFSXSri GPR64:$Rn, fixedpoint_f32_i64:$scale)>;
-def : Pat<(f64 (fdiv (f64 (any_sint_to_fp (i64 GPR64:$Rn))), fixedpoint_f64_i64:$scale)),
- (SCVTFSXDri GPR64:$Rn, fixedpoint_f64_i64:$scale)>;
-
-def : Pat<(f16 (fdiv (f16 (any_uint_to_fp (i64 GPR64:$Rn))), fixedpoint_f16_i64:$scale)),
- (UCVTFSXHri GPR64:$Rn, fixedpoint_f16_i64:$scale)>;
-def : Pat<(f32 (fdiv (f32 (any_uint_to_fp (i64 GPR64:$Rn))), fixedpoint_f32_i64:$scale)),
- (UCVTFSXSri GPR64:$Rn, fixedpoint_f32_i64:$scale)>;
-def : Pat<(f64 (fdiv (f64 (any_uint_to_fp (i64 GPR64:$Rn))), fixedpoint_f64_i64:$scale)),
- (UCVTFSXDri GPR64:$Rn, fixedpoint_f64_i64:$scale)>;
-
-def : Pat<(f16 (fdiv (f16 (any_uint_to_fp (i32 GPR32:$Rn))), fixedpoint_f16_i32:$scale)),
- (UCVTFSWHri GPR32:$Rn, fixedpoint_f16_i32:$scale)>;
-def : Pat<(f32 (fdiv (f32 (any_uint_to_fp (i32 GPR32:$Rn))), fixedpoint_f32_i32:$scale)),
- (UCVTFSWSri GPR32:$Rn, fixedpoint_f32_i32:$scale)>;
-def : Pat<(f64 (fdiv (f64 (any_uint_to_fp (i32 GPR32:$Rn))), fixedpoint_f64_i32:$scale)),
- (UCVTFSWDri GPR32:$Rn, fixedpoint_f64_i32:$scale)>;
-
//===----------------------------------------------------------------------===//
// Unscaled integer to floating point conversion instruction.
//===----------------------------------------------------------------------===//
@@ -9666,6 +9638,39 @@ def gi_fixedpoint_scalar_xform
: GICustomOperandRenderer<"renderFixedPointScalarXForm">,
GISDNodeXFormEquiv<fixedpoint_scalar_xform>;
+multiclass IntegerToFPScalarPats<SDPatternOperator OpNode, string INST> {
+ def : Pat<(f16 (fdiv (f16 (OpNode (i32 GPR32:$Rn))), fixedpoint_f16_i32:$scale)),
+ (!cast<Instruction>(INST # "SWHri") GPR32:$Rn, fixedpoint_f16_i32:$scale)>;
+ def : Pat<(f32 (fdiv (f32 (OpNode (i32 GPR32:$Rn))), fixedpoint_f32_i32:$scale)),
+ (!cast<Instruction>(INST # "SWSri") GPR32:$Rn, fixedpoint_f32_i32:$scale)>;
+ def : Pat<(f64 (fdiv (f64 (OpNode (i32 GPR32:$Rn))), fixedpoint_f64_i32:$scale)),
+ (!cast<Instruction>(INST # "SWDri") GPR32:$Rn, fixedpoint_f64_i32:$scale)>;
+ def : Pat<(f16 (fdiv (f16 (OpNode (i64 GPR64:$Rn))), fixedpoint_f16_i64:$scale)),
+ (!cast<Instruction>(INST # "SXHri") GPR64:$Rn, fixedpoint_f16_i64:$scale)>;
+ def : Pat<(f32 (fdiv (f32 (OpNode (i64 GPR64:$Rn))), fixedpoint_f32_i64:$scale)),
+ (!cast<Instruction>(INST # "SXSri") GPR64:$Rn, fixedpoint_f32_i64:$scale)>;
+ def : Pat<(f64 (fdiv (f64 (OpNode (i64 GPR64:$Rn))), fixedpoint_f64_i64:$scale)),
+ (!cast<Instruction>(INST # "SXDri") GPR64:$Rn, fixedpoint_f64_i64:$scale)>;
+
+ // Expect the integer operand to be in FPR register. The AdvSIMD scalar
+ // shifted conversion forms are same-width, so only i32->f32 and i64->f64
+ // can use them for generic integer-to-fp fixed-point conversions.
+ let GISelShouldIgnore = 1 in {
+ def : Pat<(f32 (fmul (f32 (OpNode (i32 (bitconvert (f32 FPR32:$Rn))))), fixedpoint_recip_f32_i32:$scale)),
+ (!cast<Instruction>(INST # "s") FPR32:$Rn, (fixedpoint_xform fixedpoint_recip_f32_i32:$scale))>;
+ def : Pat<(f64 (fmul (f64 (OpNode (i64 (bitconvert (f64 FPR64:$Rn))))), fixedpoint_recip_f64_i64:$scale)),
+ (!cast<Instruction>(INST # "d") FPR64:$Rn, (fixedpoint_xform fixedpoint_recip_f64_i64:$scale))>;
+ }
+
+ def : Pat<(f32 (fdiv (f32 (OpNode (i32 (bitconvert (f32 FPR32:$Rn))))), fixedpoint_f32_i32:$scale)),
+ (!cast<Instruction>(INST # "s") FPR32:$Rn, (fixedpoint_xform fixedpoint_f32_i32:$scale))>;
+ def : Pat<(f64 (fdiv (f64 (OpNode (i64 (bitconvert (f64 FPR64:$Rn))))), fixedpoint_f64_i64:$scale)),
+ (!cast<Instruction>(INST # "d") FPR64:$Rn, (fixedpoint_xform fixedpoint_f64_i64:$scale))>;
+}
+
+defm : IntegerToFPScalarPats<any_sint_to_fp, "SCVTF">;
+defm : IntegerToFPScalarPats<any_uint_to_fp, "UCVTF">;
+
multiclass FPToFixedScalarPats<SDPatternOperator OpN, string INST > {
// Allow integer result to remain in GPR register.
def : Pat<(i32 (OpN FPR32:$Rn, vecshiftR32:$imm)),
@@ -9698,21 +9703,32 @@ defm : FPToFixedScalarPats<int_aarch64_neon_vcvtfp2fxu, "FCVTZU">;
// Codegen patterns for SCVTF and UCVTF. We don't put these directly on the
// instructions because TableGen's type inference can't handle the truth.
// Having the same base pattern for fp <--> int totally freaks it out.
-def : Pat<(int_aarch64_neon_vcvtfxu2fp FPR32:$Rn, vecshiftR32:$imm),
+def : Pat<(int_aarch64_neon_vcvtfxu2fp (i32 (bitconvert (f32 FPR32:$Rn))), vecshiftR32:$imm),
(UCVTFs FPR32:$Rn, vecshiftR32:$imm)>;
-def : Pat<(f64 (int_aarch64_neon_vcvtfxu2fp (i64 FPR64:$Rn), vecshiftR64:$imm)),
+def : Pat<(f64 (int_aarch64_neon_vcvtfxu2fp (i64 (bitconvert (f64 FPR64:$Rn))),
+ vecshiftR64:$imm)),
(UCVTFd FPR64:$Rn, vecshiftR64:$imm)>;
def : Pat<(v1f64 (int_aarch64_neon_vcvtfxs2fp (v1i64 FPR64:$Rn),
vecshiftR64:$imm)),
(SCVTFd FPR64:$Rn, vecshiftR64:$imm)>;
-def : Pat<(f64 (int_aarch64_neon_vcvtfxs2fp (i64 FPR64:$Rn), vecshiftR64:$imm)),
+def : Pat<(f64 (int_aarch64_neon_vcvtfxs2fp (i64 (bitconvert (f64 FPR64:$Rn))),
+ vecshiftR64:$imm)),
(SCVTFd FPR64:$Rn, vecshiftR64:$imm)>;
def : Pat<(v1f64 (int_aarch64_neon_vcvtfxu2fp (v1i64 FPR64:$Rn),
vecshiftR64:$imm)),
(UCVTFd FPR64:$Rn, vecshiftR64:$imm)>;
-def : Pat<(int_aarch64_neon_vcvtfxs2fp FPR32:$Rn, vecshiftR32:$imm),
+def : Pat<(int_aarch64_neon_vcvtfxs2fp (i32 (bitconvert (f32 FPR32:$Rn))), vecshiftR32:$imm),
(SCVTFs FPR32:$Rn, vecshiftR32:$imm)>;
+def : Pat<(f32 (int_aarch64_neon_vcvtfxs2fp GPR32:$Rn, vecshiftR32:$imm)),
+ (SCVTFSWSri GPR32:$Rn, (fixedpoint_scalar_xform vecshiftR32:$imm))>;
+def : Pat<(f32 (int_aarch64_neon_vcvtfxu2fp GPR32:$Rn, vecshiftR32:$imm)),
+ (UCVTFSWSri GPR32:$Rn, (fixedpoint_scalar_xform vecshiftR32:$imm))>;
+def : Pat<(f64 (int_aarch64_neon_vcvtfxs2fp GPR64:$Rn, vecshiftR64:$imm)),
+ (SCVTFSXDri GPR64:$Rn, (fixedpoint_scalar_xform vecshiftR64:$imm))>;
+def : Pat<(f64 (int_aarch64_neon_vcvtfxu2fp GPR64:$Rn, vecshiftR64:$imm)),
+ (UCVTFSXDri GPR64:$Rn, (fixedpoint_scalar_xform vecshiftR64:$imm))>;
+
// Patterns for FP16 Intrinsics - requires reg copy to/from as i16s not supported.
def : Pat<(f16 (int_aarch64_neon_vcvtfxs2fp (i32 (sext_inreg FPR32:$Rn, i16)), vecshiftR16:$imm)),
diff --git a/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll b/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
index 35f62e52ffd76..040b800e9dc38 100644
--- a/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
@@ -43,7 +43,8 @@ define float @do_stuff(<8 x i16> noundef %var_135) {
; CHECK-LABEL: do_stuff:
; CHECK: // %bb.0: // %entry
; CHECK-NEXT: umaxv.8h h0, v0
-; CHECK-NEXT: ucvtf s0, s0, #1
+; CHECK-NEXT: fmov w8, s0
+; CHECK-NEXT: ucvtf s0, w8, #1
; CHECK-NEXT: ret
entry:
%vmaxv.i = call i32 @llvm.aarch64.neon.umaxv.i32.v8i16(<8 x i16> %var_135) #2
@@ -51,6 +52,90 @@ entry:
ret float %vcvts_n_f32_u32
}
+define float @neon_vcvtfxu2fp_i32_f32_gpr(i32 %a) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ucvtf s0, w0, #16
+; CHECK-NEXT: ret
+ %cvt = tail call float @llvm.aarch64.neon.vcvtfxu2fp.i32.f32(i32 %a, i32 16)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_fpr(i32 %a) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_fpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov s0, w0
+; CHECK-NEXT: usqadd s0, s0
+; CHECK-NEXT: ucvtf s0, s0, #16
+; CHECK-NEXT: ret
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %a, i32 %a)
+ %cvt = tail call float @llvm.aarch64.neon.vcvtfxu2fp.i32.f32(i32 %sum, i32 16)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxs2fp_i32_f32_gpr(i32 %a) {
+; CHECK-LABEL: neon_vcvtfxs2fp_i32_f32_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: scvtf s0, w0, #16
+; CHECK-NEXT: ret
+ %cvt = call float @llvm.aarch64.neon.vcvtfxs2fp.i32.f32(i32 %a, i32 16)
+ ret float %cvt
+}
+
+define double @neon_vcvtfxs2fp_i64_f64_gpr(i64 %a) {
+; CHECK-LABEL: neon_vcvtfxs2fp_i64_f64_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: scvtf d0, x0, #32
+; CHECK-NEXT: ret
+ %cvt = call double @llvm.aarch64.neon.vcvtfxs2fp.i64.f64(i64 %a, i32 32)
+ ret double %cvt
+}
+
+define double @neon_vcvtfxu2fp_i64_f64_gpr(i64 %a) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i64_f64_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: ucvtf d0, x0, #32
+; CHECK-NEXT: ret
+ %cvt = call double @llvm.aarch64.neon.vcvtfxu2fp.i64.f64(i64 %a, i32 32)
+ ret double %cvt
+}
+
+define float @neon_vcvtfxs2fp_i32_f32_fpr(i32 %a) {
+; CHECK-LABEL: neon_vcvtfxs2fp_i32_f32_fpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov s0, w0
+; CHECK-NEXT: usqadd s0, s0
+; CHECK-NEXT: scvtf s0, s0, #16
+; CHECK-NEXT: ret
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %a, i32 %a)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxs2fp.i32.f32(i32 %sum, i32 16)
+ ret float %cvt
+}
+
+define double @neon_vcvtfxs2fp_i64_f64_fpr(i64 %a) {
+; CHECK-LABEL: neon_vcvtfxs2fp_i64_f64_fpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov d0, x0
+; CHECK-NEXT: usqadd d0, d0
+; CHECK-NEXT: scvtf d0, d0, #32
+; CHECK-NEXT: ret
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %a, i64 %a)
+ %cvt = call double @llvm.aarch64.neon.vcvtfxs2fp.i64.f64(i64 %sum, i32 32)
+ ret double %cvt
+}
+
+define double @neon_vcvtfxu2fp_i64_f64_fpr(i64 %a) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i64_f64_fpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: fmov d0, x0
+; CHECK-NEXT: usqadd d0, d0
+; CHECK-NEXT: ucvtf d0, d0, #32
+; CHECK-NEXT: ret
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %a, i64 %a)
+ %cvt = call double @llvm.aarch64.neon.vcvtfxu2fp.i64.f64(i64 %sum, i32 32)
+ ret double %cvt
+}
+
declare <1 x i64> @llvm.aarch64.neon.vsri.v1i64(<1 x i64>, <1 x i64>, i32)
declare double @llvm.aarch64.neon.vcvtfxs2fp.f64.i64(i64, i32)
declare <1 x double> @llvm.nearbyint.v1f64(<1 x double>)
diff --git a/llvm/test/CodeGen/AArch64/fcvt-fixed.ll b/llvm/test/CodeGen/AArch64/fcvt-fixed.ll
index f42071126df1b..48bd1c4ddfa54 100644
--- a/llvm/test/CodeGen/AArch64/fcvt-fixed.ll
+++ b/llvm/test/CodeGen/AArch64/fcvt-fixed.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,CHECK-NO16,CHECK-SD-NO16
-; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -mattr=+fullfp16 | FileCheck %s --check-prefixes=CHECK,CHECK-FP16
-; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,CHECK-NO16,CHECK-GI-NO16
-; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -mattr=+fullfp16 -global-isel | FileCheck %s --check-prefixes=CHECK,CHECK-FP16
+; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu | FileCheck %s --check-prefixes=CHECK,CHECK-NO16,CHECK-SD-NO16,CHECK-SD
+; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -mattr=+fullfp16 | FileCheck %s --check-prefixes=CHECK,CHECK-FP16,CHECK-SD
+; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -global-isel | FileCheck %s --check-prefixes=CHECK,CHECK-NO16,CHECK-GI-NO16,CHECK-GI
+; RUN: llc -verify-machineinstrs < %s -mtriple=aarch64-none-linux-gnu -mattr=+fullfp16 -global-isel | FileCheck %s --check-prefixes=CHECK,CHECK-FP16,CHECK-GI
; fptoui
@@ -1301,3 +1301,325 @@ define float @neon_fcvtzu_f32_i32_32_bitcast(float %a) {
%bc = bitcast i32 %r to float
ret float %bc
}
+
+
+define float @scvtf_f32_i32_3_input_in_fpr(i32 %int) {
+; CHECK-SD-LABEL: scvtf_f32_i32_3_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov s0, w0
+; CHECK-SD-NEXT: usqadd s0, s0
+; CHECK-SD-NEXT: scvtf s0, s0, #3
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: scvtf_f32_i32_3_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov s0, w0
+; CHECK-GI-NEXT: fmov s1, #8.00000000
+; CHECK-GI-NEXT: usqadd s0, s0
+; CHECK-GI-NEXT: scvtf s0, s0
+; CHECK-GI-NEXT: fdiv s0, s0, s1
+; CHECK-GI-NEXT: ret
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %int, i32 %int)
+ %cvt = sitofp i32 %sum to float
+ %fix = fdiv float %cvt, 8.0
+ ret float %fix
+}
+
+define float @scvtf_f32_i32_5_fmul_input_in_fpr(i32 %int) {
+; CHECK-SD-LABEL: scvtf_f32_i32_5_fmul_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov s0, w0
+; CHECK-SD-NEXT: usqadd s0, s0
+; CHECK-SD-NEXT: scvtf s0, s0, #5
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: scvtf_f32_i32_5_fmul_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov s1, w0
+; CHECK-GI-NEXT: movi v0.2s, #61, lsl #24
+; CHECK-GI-NEXT: usqadd s1, s1
+; CHECK-GI-NEXT: scvtf s1, s1
+; CHECK-GI-NEXT: fmul s0, s1, s0
+; CHECK-GI-NEXT: ret
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %int, i32 %int)
+ %cvt = sitofp i32 %sum to float
+ %fix = fmul float %cvt, 0x1.0p-5
+ ret float %fix
+}
+
+define double @scvtf_f64_i64_6_input_in_fpr(i64 %long) {
+; CHECK-SD-LABEL: scvtf_f64_i64_6_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov d0, x0
+; CHECK-SD-NEXT: usqadd d0, d0
+; CHECK-SD-NEXT: scvtf d0, d0, #6
+; CHECK-SD-NEXT: ret
+
+; CHECK-GI-LABEL: scvtf_f64_i64_6_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov d0, x0
+; CHECK-GI-NEXT: mov x8, #4634204016564240384 // =0x4050000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: usqadd d0, d0
+; CHECK-GI-NEXT: scvtf d0, d0
+; CHECK-GI-NEXT: fdiv d0, d0, d1
+; CHECK-GI-NEXT: ret
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %long, i64 %long)
+ %cvt = sitofp i64 %sum to double
+ %fix = fdiv double %cvt, 64.0
+ ret double %fix
+}
+
+define double @scvtf_f64_i64_9_fmul_input_in_fpr(i64 %long) {
+; CHECK-SD-LABEL: scvtf_f64_i64_9_fmul_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov d0, x0
+; CHECK-SD-NEXT: usqadd d0, d0
+; CHECK-SD-NEXT: scvtf d0, d0, #9
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: scvtf_f64_i64_9_fmul_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov d0, x0
+; CHECK-GI-NEXT: mov x8, #4566650022153682944 // =0x3f60000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: usqadd d0, d0
+; CHECK-GI-NEXT: scvtf d0, d0
+; CHECK-GI-NEXT: fmul d0, d0, d1
+; CHECK-GI-NEXT: ret
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %long, i64 %long)
+ %cvt = sitofp i64 %sum to double
+ %fix = fmul double %cvt, 0x1.0p-9
+ ret double %fix
+}
+
+define float @ucvtf_f32_i32_3_input_in_fpr(i32 %int) {
+; CHECK-SD-LABEL: ucvtf_f32_i32_3_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov s0, w0
+; CHECK-SD-NEXT: usqadd s0, s0
+; CHECK-SD-NEXT: ucvtf s0, s0, #3
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f32_i32_3_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov s0, w0
+; CHECK-GI-NEXT: fmov s1, #8.00000000
+; CHECK-GI-NEXT: usqadd s0, s0
+; CHECK-GI-NEXT: ucvtf s0, s0
+; CHECK-GI-NEXT: fdiv s0, s0, s1
+; CHECK-GI-NEXT: ret
+;
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %int, i32 %int)
+ %cvt = uitofp i32 %sum to float
+ %fix = fdiv float %cvt, 8.0
+ ret float %fix
+}
+
+define float @ucvtf_f32_i32_5_fmul_input_in_fpr(i32 %int) {
+; CHECK-SD-LABEL: ucvtf_f32_i32_5_fmul_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov s0, w0
+; CHECK-SD-NEXT: usqadd s0, s0
+; CHECK-SD-NEXT: ucvtf s0, s0, #5
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f32_i32_5_fmul_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov s1, w0
+; CHECK-GI-NEXT: movi v0.2s, #61, lsl #24
+; CHECK-GI-NEXT: usqadd s1, s1
+; CHECK-GI-NEXT: ucvtf s1, s1
+; CHECK-GI-NEXT: fmul s0, s1, s0
+; CHECK-GI-NEXT: ret
+;
+ %sum = call i32 @llvm.aarch64.neon.usqadd.i32(i32 %int, i32 %int)
+ %cvt = uitofp i32 %sum to float
+ %fix = fmul float %cvt, 0x1.0p-5
+ ret float %fix
+}
+
+define double @ucvtf_f64_i64_6_input_in_fpr(i64 %long) {
+; CHECK-SD-LABEL: ucvtf_f64_i64_6_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov d0, x0
+; CHECK-SD-NEXT: usqadd d0, d0
+; CHECK-SD-NEXT: ucvtf d0, d0, #6
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f64_i64_6_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov d0, x0
+; CHECK-GI-NEXT: mov x8, #4634204016564240384 // =0x4050000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: usqadd d0, d0
+; CHECK-GI-NEXT: ucvtf d0, d0
+; CHECK-GI-NEXT: fdiv d0, d0, d1
+; CHECK-GI-NEXT: ret
+;
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %long, i64 %long)
+ %cvt = uitofp i64 %sum to double
+ %fix = fdiv double %cvt, 64.0
+ ret double %fix
+}
+
+define double @ucvtf_f64_i64_9_fmul_input_in_fpr(i64 %long) {
+; CHECK-SD-LABEL: ucvtf_f64_i64_9_fmul_input_in_fpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: fmov d0, x0
+; CHECK-SD-NEXT: usqadd d0, d0
+; CHECK-SD-NEXT: ucvtf d0, d0, #9
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f64_i64_9_fmul_input_in_fpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: fmov d0, x0
+; CHECK-GI-NEXT: mov x8, #4566650022153682944 // =0x3f60000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: usqadd d0, d0
+; CHECK-GI-NEXT: ucvtf d0, d0
+; CHECK-GI-NEXT: fmul d0, d0, d1
+; CHECK-GI-NEXT: ret
+;
+ %sum = call i64 @llvm.aarch64.neon.usqadd.i64(i64 %long, i64 %long)
+ %cvt = uitofp i64 %sum to double
+ %fix = fmul double %cvt, 0x1.0p-9
+ ret double %fix
+}
+
+define float @scvtf_f32_i32_3_input_in_gpr(i32 %int) {
+; CHECK-LABEL: scvtf_f32_i32_3_input_in_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: add w8, w0, w0
+; CHECK-NEXT: scvtf s0, w8, #3
+; CHECK-NEXT: ret
+ %sum = add i32 %int, %int
+ %cvt = sitofp i32 %sum to float
+ %fix = fdiv float %cvt, 8.0
+ ret float %fix
+}
+
+define float @scvtf_f32_i32_5_fmul_input_in_gpr(i32 %int) {
+; CHECK-SD-LABEL: scvtf_f32_i32_5_fmul_input_in_gpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: add w8, w0, w0
+; CHECK-SD-NEXT: scvtf s0, w8, #5
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: scvtf_f32_i32_5_fmul_input_in_gpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: add w8, w0, w0
+; CHECK-GI-NEXT: movi v0.2s, #61, lsl #24
+; CHECK-GI-NEXT: scvtf s1, w8
+; CHECK-GI-NEXT: fmul s0, s1, s0
+; CHECK-GI-NEXT: ret
+;
+ %sum = add i32 %int, %int
+ %cvt = sitofp i32 %sum to float
+ %fix = fmul float %cvt, 0x1.0p-5
+ ret float %fix
+}
+
+define double @scvtf_f64_i64_6_input_in_gpr(i64 %long) {
+; CHECK-LABEL: scvtf_f64_i64_6_input_in_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: add x8, x0, x0
+; CHECK-NEXT: scvtf d0, x8, #6
+; CHECK-NEXT: ret
+ %sum = add i64 %long, %long
+ %cvt = sitofp i64 %sum to double
+ %fix = fdiv double %cvt, 64.0
+ ret double %fix
+}
+
+define double @scvtf_f64_i64_9_fmul_input_in_gpr(i64 %long) {
+; CHECK-SD-LABEL: scvtf_f64_i64_9_fmul_input_in_gpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: add x8, x0, x0
+; CHECK-SD-NEXT: scvtf d0, x8, #9
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: scvtf_f64_i64_9_fmul_input_in_gpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: add x8, x0, x0
+; CHECK-GI-NEXT: scvtf d0, x8
+; CHECK-GI-NEXT: mov x8, #4566650022153682944 // =0x3f60000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: fmul d0, d0, d1
+; CHECK-GI-NEXT: ret
+;
+ %sum = add i64 %long, %long
+ %cvt = sitofp i64 %sum to double
+ %fix = fmul double %cvt, 0x1.0p-9
+ ret double %fix
+}
+
+define float @ucvtf_f32_i32_3_input_in_gpr(i32 %int) {
+; CHECK-LABEL: ucvtf_f32_i32_3_input_in_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: add w8, w0, w0
+; CHECK-NEXT: ucvtf s0, w8, #3
+; CHECK-NEXT: ret
+ %sum = add i32 %int, %int
+ %cvt = uitofp i32 %sum to float
+ %fix = fdiv float %cvt, 8.0
+ ret float %fix
+}
+
+define float @ucvtf_f32_i32_5_fmul_input_in_gpr(i32 %int) {
+; CHECK-SD-LABEL: ucvtf_f32_i32_5_fmul_input_in_gpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: add w8, w0, w0
+; CHECK-SD-NEXT: ucvtf s0, w8, #5
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f32_i32_5_fmul_input_in_gpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: add w8, w0, w0
+; CHECK-GI-NEXT: movi v0.2s, #61, lsl #24
+; CHECK-GI-NEXT: ucvtf s1, w8
+; CHECK-GI-NEXT: fmul s0, s1, s0
+; CHECK-GI-NEXT: ret
+;
+ %sum = add i32 %int, %int
+ %cvt = uitofp i32 %sum to float
+ %fix = fmul float %cvt, 0x1.0p-5
+ ret float %fix
+}
+
+define double @ucvtf_f64_i64_6_input_in_gpr(i64 %long) {
+; CHECK-LABEL: ucvtf_f64_i64_6_input_in_gpr:
+; CHECK: // %bb.0:
+; CHECK-NEXT: add x8, x0, x0
+; CHECK-NEXT: ucvtf d0, x8, #6
+; CHECK-NEXT: ret
+ %sum = add i64 %long, %long
+ %cvt = uitofp i64 %sum to double
+ %fix = fdiv double %cvt, 64.0
+ ret double %fix
+}
+
+define double @ucvtf_f64_i64_9_fmul_input_in_gpr(i64 %long) {
+; CHECK-SD-LABEL: ucvtf_f64_i64_9_fmul_input_in_gpr:
+; CHECK-SD: // %bb.0:
+; CHECK-SD-NEXT: add x8, x0, x0
+; CHECK-SD-NEXT: ucvtf d0, x8, #9
+; CHECK-SD-NEXT: ret
+;
+; CHECK-GI-LABEL: ucvtf_f64_i64_9_fmul_input_in_gpr:
+; CHECK-GI: // %bb.0:
+; CHECK-GI-NEXT: add x8, x0, x0
+; CHECK-GI-NEXT: ucvtf d0, x8
+; CHECK-GI-NEXT: mov x8, #4566650022153682944 // =0x3f60000000000000
+; CHECK-GI-NEXT: fmov d1, x8
+; CHECK-GI-NEXT: fmul d0, d0, d1
+; CHECK-GI-NEXT: ret
+;
+ %sum = add i64 %long, %long
+ %cvt = uitofp i64 %sum to double
+ %fix = fmul double %cvt, 0x1.0p-9
+ ret double %fix
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK-GI: {{.*}}
+; CHECK-SD: {{.*}}
>From 2c24ddb23fea55b0c4aae25c7066d7f337d857d7 Mon Sep 17 00:00:00 2001
From: Vimal Patel <vimal.patel at arm.com>
Date: Thu, 3 Sep 2026 13:38:18 +0000
Subject: [PATCH 2/2] [AArch64] Avoid GPR roundtrip for unsigned min/max
conversions
Lower UMINV and UMAXV intrinsics that return i32 from subword
vectors through widened target DAG nodes. This lets instruction
selection use the implicit zeroing of scalar SIMD&FP writes.
This avoids moving unsigned across-lane min/max results through a GPR
before converting to floating point.
---
.../Target/AArch64/AArch64ISelLowering.cpp | 23 ++-
llvm/lib/Target/AArch64/AArch64InstrInfo.td | 62 +++++---
...arm64-fixed-point-scalar-cvt-dagcombine.ll | 136 +++++++++++++++++-
3 files changed, 195 insertions(+), 26 deletions(-)
diff --git a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
index fa887b7543be9..2331452a81d14 100644
--- a/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
+++ b/llvm/lib/Target/AArch64/AArch64ISelLowering.cpp
@@ -7342,6 +7342,25 @@ SDValue AArch64TargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
ADDLV, DAG.getConstant(0, DL, MVT::i64));
return EXTRACT_VEC_ELT;
}
+ case Intrinsic::aarch64_neon_uminv:
+ case Intrinsic::aarch64_neon_umaxv: {
+ EVT OpVT = Op.getOperand(1).getValueType();
+ EVT ResVT = Op.getValueType();
+ assert((ResVT == MVT::i32 &&
+ (OpVT == MVT::v8i8 || OpVT == MVT::v16i8 || OpVT == MVT::v8i16 ||
+ OpVT == MVT::v4i16 || OpVT == MVT::v2i32 || OpVT == MVT::v4i32)) &&
+ "Unexpected aarch64_neon_u/saddlv type");
+ (void)OpVT;
+ // In order to avoid insert_subvector, use v4i32 rather than v2i32.
+ unsigned OpCode = IntNo == Intrinsic::aarch64_neon_uminv
+ ? AArch64ISD::UMINV
+ : AArch64ISD::UMAXV;
+ SDValue UMinMaxV = DAG.getNode(OpCode, DL, MVT::v4i32, Op.getOperand(1));
+ SDValue EXTRACT_VEC_ELT =
+ DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, UMinMaxV,
+ DAG.getConstant(0, DL, MVT::i64));
+ return EXTRACT_VEC_ELT;
+ }
case Intrinsic::aarch64_cls:
case Intrinsic::aarch64_cls64: {
SDValue Res = DAG.getNode(ISD::CTLS, DL, Op.getOperand(1).getValueType(),
@@ -25388,12 +25407,8 @@ static SDValue performIntrinsicCombine(SDNode *N,
return combineAcrossLanesIntrinsic(AArch64ISD::UADDV, N, DAG);
case Intrinsic::aarch64_neon_sminv:
return combineAcrossLanesIntrinsic(AArch64ISD::SMINV, N, DAG);
- case Intrinsic::aarch64_neon_uminv:
- return combineAcrossLanesIntrinsic(AArch64ISD::UMINV, N, DAG);
case Intrinsic::aarch64_neon_smaxv:
return combineAcrossLanesIntrinsic(AArch64ISD::SMAXV, N, DAG);
- case Intrinsic::aarch64_neon_umaxv:
- return combineAcrossLanesIntrinsic(AArch64ISD::UMAXV, N, DAG);
case Intrinsic::aarch64_neon_faddv:
return tryCombineFADDReductionWithZero(N, DAG, Subtarget, N->getOperand(1));
case Intrinsic::aarch64_neon_fmax:
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index b7b1c75332f80..9fdfd6b4fba31 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -541,7 +541,7 @@ def SDT_AArch64RANGE_PREFETCH: SDTypeProfile<0, 3, [SDTCisVT<0, i32>, SDTCisPtrT
def SDT_AArch64ITOF : SDTypeProfile<1, 1, [SDTCisFP<0>, SDTCisSameAs<0,1>]>;
-def SDT_AArch64uaddlp : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisVec<1>]>;
+def SDT_AArch64UnaryVecDiffOpResTy : SDTypeProfile<1, 1, [SDTCisVec<0>, SDTCisVec<1>]>;
def SDT_AArch64ldp : SDTypeProfile<2, 1, [SDTCisVT<0, i64>, SDTCisSameAs<0, 1>, SDTCisPtrTy<2>]>;
def SDT_AArch64ldiapp : SDTypeProfile<2, 1, [SDTCisVT<0, i64>, SDTCisSameAs<0, 1>, SDTCisPtrTy<2>]>;
@@ -1149,19 +1149,19 @@ def AArch64addhn : SDNode<"AArch64ISD::ADDHN", SDT_AArch64Addhn>;
// Vector across-lanes min/max
// Only the lower result lane is defined.
def AArch64sminv : SDNode<"AArch64ISD::SMINV", SDT_AArch64UnaryVec>;
-def AArch64uminv : SDNode<"AArch64ISD::UMINV", SDT_AArch64UnaryVec>;
+def AArch64uminv : SDNode<"AArch64ISD::UMINV", SDT_AArch64UnaryVecDiffOpResTy>;
def AArch64smaxv : SDNode<"AArch64ISD::SMAXV", SDT_AArch64UnaryVec>;
-def AArch64umaxv : SDNode<"AArch64ISD::UMAXV", SDT_AArch64UnaryVec>;
+def AArch64umaxv : SDNode<"AArch64ISD::UMAXV", SDT_AArch64UnaryVecDiffOpResTy>;
// Unsigned sum Long across Vector
-def AArch64uaddlv : SDNode<"AArch64ISD::UADDLV", SDT_AArch64uaddlp>;
-def AArch64saddlv : SDNode<"AArch64ISD::SADDLV", SDT_AArch64uaddlp>;
+def AArch64uaddlv : SDNode<"AArch64ISD::UADDLV", SDT_AArch64UnaryVecDiffOpResTy>;
+def AArch64saddlv : SDNode<"AArch64ISD::SADDLV", SDT_AArch64UnaryVecDiffOpResTy>;
// Add Pairwise of two vectors
def AArch64addp_n : SDNode<"AArch64ISD::ADDP", SDT_AArch64Zip>;
// Add Long Pairwise
-def AArch64uaddlp_n : SDNode<"AArch64ISD::UADDLP", SDT_AArch64uaddlp>;
-def AArch64saddlp_n : SDNode<"AArch64ISD::SADDLP", SDT_AArch64uaddlp>;
+def AArch64uaddlp_n : SDNode<"AArch64ISD::UADDLP", SDT_AArch64UnaryVecDiffOpResTy>;
+def AArch64saddlp_n : SDNode<"AArch64ISD::SADDLP", SDT_AArch64UnaryVecDiffOpResTy>;
def AArch64addp : PatFrags<(ops node:$Rn, node:$Rm),
[(AArch64addp_n node:$Rn, node:$Rm),
(int_aarch64_neon_addp node:$Rn, node:$Rm)]>;
@@ -8883,43 +8883,42 @@ multiclass SIMDAcrossLanesIntrinsic<string baseOpc,
SDPatternOperator opNode> {
// If a lane instruction caught the vector_extract around opNode, we can
// directly match the latter to the instruction.
-def : Pat<(v8i8 (opNode V64:$Rn)),
+def : Pat<(v8i8 (opNode (v8i8 V64:$Rn))),
(INSERT_SUBREG (v8i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v8i8v")) V64:$Rn), bsub)>;
-def : Pat<(v16i8 (opNode V128:$Rn)),
+def : Pat<(v16i8 (opNode (v16i8 V128:$Rn))),
(INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v16i8v")) V128:$Rn), bsub)>;
-def : Pat<(v4i16 (opNode V64:$Rn)),
+def : Pat<(v4i16 (opNode (v4i16 V64:$Rn))),
(INSERT_SUBREG (v4i16 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v4i16v")) V64:$Rn), hsub)>;
-def : Pat<(v8i16 (opNode V128:$Rn)),
+def : Pat<(v8i16 (opNode (v8i16 V128:$Rn))),
(INSERT_SUBREG (v8i16 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v8i16v")) V128:$Rn), hsub)>;
-def : Pat<(v4i32 (opNode V128:$Rn)),
+def : Pat<(v4i32 (opNode (v4i32 V128:$Rn))),
(INSERT_SUBREG (v4i32 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v4i32v")) V128:$Rn), ssub)>;
-
// If none did, fallback to the explicit patterns, consuming the vector_extract.
-def : Pat<(i32 (vector_extract (insert_subvector (v16i8 undef), (v8i8 (opNode V64:$Rn)),
+def : Pat<(i32 (vector_extract (insert_subvector (v16i8 undef), (v8i8 (opNode (v8i8 V64:$Rn))),
(i64 0)), (i64 0))),
(EXTRACT_SUBREG (INSERT_SUBREG (v8i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v8i8v")) V64:$Rn),
bsub), ssub)>;
-def : Pat<(i32 (vector_extract (v16i8 (opNode V128:$Rn)), (i64 0))),
+def : Pat<(i32 (vector_extract (v16i8 (opNode (v16i8 V128:$Rn))), (i64 0))),
(EXTRACT_SUBREG (INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v16i8v")) V128:$Rn),
bsub), ssub)>;
def : Pat<(i32 (vector_extract (insert_subvector (v8i16 undef),
- (v4i16 (opNode V64:$Rn)), (i64 0)), (i64 0))),
+ (v4i16 (opNode (v4i16 V64:$Rn))), (i64 0)), (i64 0))),
(EXTRACT_SUBREG (INSERT_SUBREG (v4i16 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v4i16v")) V64:$Rn),
hsub), ssub)>;
-def : Pat<(i32 (vector_extract (v8i16 (opNode V128:$Rn)), (i64 0))),
+def : Pat<(i32 (vector_extract (v8i16 (opNode (v8i16 V128:$Rn))), (i64 0))),
(EXTRACT_SUBREG (INSERT_SUBREG (v8i16 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v8i16v")) V128:$Rn),
hsub), ssub)>;
-def : Pat<(i32 (vector_extract (v4i32 (opNode V128:$Rn)), (i64 0))),
+def : Pat<(i32 (vector_extract (v4i32 (opNode (v4i32 V128:$Rn))), (i64 0))),
(EXTRACT_SUBREG (INSERT_SUBREG (v4i32 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v4i32v")) V128:$Rn),
ssub), ssub)>;
@@ -8968,7 +8967,7 @@ def : Pat<(i32 (and (i32 (vector_extract (insert_subvector (v16i8 undef),
(INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v8i8v")) V64:$Rn), bsub),
ssub))>;
-def : Pat<(i32 (and (i32 (vector_extract (opNode (v16i8 V128:$Rn)), (i64 0))),
+def : Pat<(i32 (and (i32 (vector_extract (v16i8 (opNode (v16i8 V128:$Rn))), (i64 0))),
maski8_or_more)),
(i32 (EXTRACT_SUBREG
(INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
@@ -8980,7 +8979,7 @@ def : Pat<(i32 (and (i32 (vector_extract (insert_subvector (v8i16 undef),
(INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
(!cast<Instruction>(!strconcat(baseOpc, "v4i16v")) V64:$Rn), hsub),
ssub))>;
-def : Pat<(i32 (and (i32 (vector_extract (opNode (v8i16 V128:$Rn)), (i64 0))),
+def : Pat<(i32 (and (i32 (vector_extract (v8i16 (opNode (v8i16 V128:$Rn))), (i64 0))),
maski16_or_more)),
(i32 (EXTRACT_SUBREG
(INSERT_SUBREG (v16i8 (IMPLICIT_DEF)),
@@ -9022,13 +9021,36 @@ defm : SIMDAcrossLanesSignedIntrinsic<"SMINV", AArch64sminv>;
def : Pat<(v2i32 (AArch64sminv (v2i32 V64:$Rn))),
(SMINPv2i32 V64:$Rn, V64:$Rn)>;
+multiclass SIMDAcrossLanesUnsignedMinMaxWideningIntrinsic<
+ string Opc, SDPatternOperator opNode, Instruction PairwiseInst> {
+def : Pat<(v4i32 (opNode (v8i8 V64:$Rn))),
+ (v4i32 (SUBREG_TO_REG
+ (!cast<Instruction>(Opc#"v8i8v") V64:$Rn), bsub))>;
+def : Pat<(v4i32 (opNode (v16i8 V128:$Rn))),
+ (v4i32 (SUBREG_TO_REG
+ (!cast<Instruction>(Opc#"v16i8v") V128:$Rn), bsub))>;
+def : Pat<(v4i32 (opNode (v4i16 V64:$Rn))),
+ (v4i32 (SUBREG_TO_REG
+ (!cast<Instruction>(Opc#"v4i16v") V64:$Rn), hsub))>;
+def : Pat<(v4i32 (opNode (v8i16 V128:$Rn))),
+ (v4i32 (SUBREG_TO_REG
+ (!cast<Instruction>(Opc#"v8i16v") V128:$Rn), hsub))>;
+def : Pat<(v4i32 (opNode (v2i32 V64:$Rn))),
+ (v4i32 (SUBREG_TO_REG
+ (PairwiseInst V64:$Rn, V64:$Rn), dsub))>;
+}
+
defm : SIMDAcrossLanesUnsignedIntrinsic<"UMAXV", AArch64umaxv>;
def : Pat<(v2i32 (AArch64umaxv (v2i32 V64:$Rn))),
(UMAXPv2i32 V64:$Rn, V64:$Rn)>;
+defm : SIMDAcrossLanesUnsignedMinMaxWideningIntrinsic<
+ "UMAXV", AArch64umaxv, UMAXPv2i32>;
defm : SIMDAcrossLanesUnsignedIntrinsic<"UMINV", AArch64uminv>;
def : Pat<(v2i32 (AArch64uminv (v2i32 V64:$Rn))),
(UMINPv2i32 V64:$Rn, V64:$Rn)>;
+defm : SIMDAcrossLanesUnsignedMinMaxWideningIntrinsic<
+ "UMINV", AArch64uminv, UMINPv2i32>;
// For vecreduce_{opc} used by GlobalISel, not SDAG at the moment
// because GlobalISel allows us to specify the return register to be a FPR
diff --git a/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll b/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
index 040b800e9dc38..e7c45d6ff6fc5 100644
--- a/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-fixed-point-scalar-cvt-dagcombine.ll
@@ -43,8 +43,8 @@ define float @do_stuff(<8 x i16> noundef %var_135) {
; CHECK-LABEL: do_stuff:
; CHECK: // %bb.0: // %entry
; CHECK-NEXT: umaxv.8h h0, v0
-; CHECK-NEXT: fmov w8, s0
-; CHECK-NEXT: ucvtf s0, w8, #1
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
; CHECK-NEXT: ret
entry:
%vmaxv.i = call i32 @llvm.aarch64.neon.umaxv.i32.v8i16(<8 x i16> %var_135) #2
@@ -52,6 +52,138 @@ entry:
ret float %vcvts_n_f32_u32
}
+define float @neon_vcvtfxu2fp_i32_f32_umaxv_v8i8(<8 x i8> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_umaxv_v8i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: umaxv.8b b0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.umaxv.i32.v8i8(<8 x i8> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_umaxv_v16i8(<16 x i8> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_umaxv_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: umaxv.16b b0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.umaxv.i32.v16i8(<16 x i8> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_umaxv_v4i16(<4 x i16> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_umaxv_v4i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: umaxv.4h h0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.umaxv.i32.v4i16(<4 x i16> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_umaxv_v8i16(<8 x i16> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_umaxv_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: umaxv.8h h0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.umaxv.i32.v8i16(<8 x i16> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_uminv_v8i8(<8 x i8> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_uminv_v8i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uminv.8b b0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.uminv.i32.v8i8(<8 x i8> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_uminv_v16i8(<16 x i8> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_uminv_v16i8:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uminv.16b b0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.uminv.i32.v16i8(<16 x i8> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_uminv_v4i16(<4 x i16> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_uminv_v4i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uminv.4h h0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.uminv.i32.v4i16(<4 x i16> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_uminv_v8i16(<8 x i16> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_uminv_v8i16:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uminv.8h h0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.uminv.i32.v8i16(<8 x i16> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_umaxv_v2i32(<2 x i32> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_umaxv_v2i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: umaxp.2s v0, v0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.umaxv.i32.v2i32(<2 x i32> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxu2fp_i32_f32_uminv_v2i32(<2 x i32> %v) {
+; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_uminv_v2i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: uminp.2s v0, v0, v0
+; CHECK-NEXT: ucvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.uminv.i32.v2i32(<2 x i32> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxu2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
+define float @neon_vcvtfxs2fp_i32_f32_smaxv_v4i32(<4 x i32> %v) {
+; CHECK-LABEL: neon_vcvtfxs2fp_i32_f32_smaxv_v4i32:
+; CHECK: // %bb.0:
+; CHECK-NEXT: smaxv.4s s0, v0
+; CHECK-NEXT: scvtf.4s v0, v0, #1
+; CHECK-NEXT: // kill: def $s0 killed $s0 killed $q0
+; CHECK-NEXT: ret
+ %elt = call i32 @llvm.aarch64.neon.smaxv.i32.v4i32(<4 x i32> %v)
+ %cvt = call float @llvm.aarch64.neon.vcvtfxs2fp.f32.i32(i32 %elt, i32 1)
+ ret float %cvt
+}
+
define float @neon_vcvtfxu2fp_i32_f32_gpr(i32 %a) {
; CHECK-LABEL: neon_vcvtfxu2fp_i32_f32_gpr:
; CHECK: // %bb.0:
More information about the llvm-commits
mailing list