[llvm] cc958e4 - [AArch64][GlobalISel] Avoid cross bank copies for NEON vcvtfp2fx results (#213277)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 10 04:53:42 PDT 2026
Author: Kieran B
Date: 2026-08-10T12:53:37+01:00
New Revision: cc958e4be42923572cd73d3e64c34ea8a6c1b974
URL: https://github.com/llvm/llvm-project/commit/cc958e4be42923572cd73d3e64c34ea8a6c1b974
DIFF: https://github.com/llvm/llvm-project/commit/cc958e4be42923572cd73d3e64c34ea8a6c1b974.diff
LOG: [AArch64][GlobalISel] Avoid cross bank copies for NEON vcvtfp2fx results (#213277)
Currently, patterns to avoid cross bank copies for the intrinsic
vcvtfp2fx only work with SelectionDAG. This patch allows the DAG
patterns to work with GlobalISel.
SelectionDAG PR: #210275
Added:
Modified:
llvm/lib/Target/AArch64/AArch64InstrInfo.td
llvm/lib/Target/AArch64/GISel/AArch64InstructionSelector.cpp
llvm/lib/Target/AArch64/GISel/AArch64RegisterBankInfo.cpp
llvm/test/CodeGen/AArch64/neon-scalar-vcvtfp2fx.ll
Removed:
################################################################################
diff --git a/llvm/lib/Target/AArch64/AArch64InstrInfo.td b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
index 53c7af890b6e5..da4ae5bc29675 100644
--- a/llvm/lib/Target/AArch64/AArch64InstrInfo.td
+++ b/llvm/lib/Target/AArch64/AArch64InstrInfo.td
@@ -9149,6 +9149,9 @@ def fixedpoint_scalar_xform : SDNodeXForm<timm, [{
(void)N;
return V;
}]>;
+def gi_fixedpoint_scalar_xform
+ : GICustomOperandRenderer<"renderFixedPointScalarXForm">,
+ GISDNodeXFormEquiv<fixedpoint_scalar_xform>;
multiclass FPToFixedScalarPats<SDPatternOperator OpN, string INST > {
// Allow integer result to remain in GPR register.
@@ -9162,12 +9165,7 @@ multiclass FPToFixedScalarPats<SDPatternOperator OpN, string INST > {
(!cast<Instruction>(INST # "s") FPR32:$Rn, vecshiftR32:$imm)>;
def : Pat<(f64 (bitconvert(i64 (OpN (f64 FPR64:$Rn), vecshiftR64:$imm)))),
(!cast<Instruction>(INST # "d") FPR64:$Rn, vecshiftR64:$imm)>;
-
- // FPR fallback patterns.
- def : Pat<(i32 (OpN FPR32:$Rn, vecshiftR32:$imm)),
- (!cast<Instruction>(INST # "s") FPR32:$Rn, vecshiftR32:$imm)>;
- def : Pat<(i64 (OpN (f64 FPR64:$Rn), vecshiftR64:$imm)),
- (!cast<Instruction>(INST # "d") FPR64:$Rn, vecshiftR64:$imm)>;
+
def : Pat<(v1i64 (OpN (v1f64 FPR64:$Rn), vecshiftR64:$imm)),
(!cast<Instruction>(INST # "d") FPR64:$Rn, vecshiftR64:$imm)>;
def : Pat<(i32 (OpN (f16 FPR16:$Rn), vecshiftR32:$imm)),
diff --git a/llvm/lib/Target/AArch64/GISel/AArch64InstructionSelector.cpp b/llvm/lib/Target/AArch64/GISel/AArch64InstructionSelector.cpp
index b3ea51a4de15c..ad8c1eb728e7b 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64InstructionSelector.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64InstructionSelector.cpp
@@ -488,11 +488,12 @@ class AArch64InstructionSelector : public InstructionSelector {
ComplexRendererFns
selectCVTFixedPointVecBase(const MachineOperand &Root,
bool isReciprocal = false) const;
+ void renderFixedPointScalarXForm(MachineInstrBuilder &MIB,
+ const MachineInstr &MI, int OpIdx) const;
void renderFixedPointXForm(MachineInstrBuilder &MIB, const MachineInstr &MI,
int OpIdx = -1) const;
void renderFixedPointRecipXForm(MachineInstrBuilder &MIB,
const MachineInstr &MI, int OpIdx = -1) const;
-
void renderTruncImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
int OpIdx = -1) const;
void renderLogicalImm32(MachineInstrBuilder &MIB, const MachineInstr &I,
@@ -7965,6 +7966,13 @@ AArch64InstructionSelector::selectCVTFixedPosRecipOperandVec(
return selectCVTFixedPointVecBase(Root, /*isReciprocal*/ true);
}
+void AArch64InstructionSelector::renderFixedPointScalarXForm(
+ MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
+ assert(OpIdx == 3 && MI.getOperand(OpIdx).isImm() &&
+ "Expected vecshift immediate operand");
+ MIB.addImm(MI.getOperand(OpIdx).getImm());
+}
+
void AArch64InstructionSelector::renderFixedPointXForm(MachineInstrBuilder &MIB,
const MachineInstr &MI,
int OpIdx) const {
diff --git a/llvm/lib/Target/AArch64/GISel/AArch64RegisterBankInfo.cpp b/llvm/lib/Target/AArch64/GISel/AArch64RegisterBankInfo.cpp
index 58af438482fa1..78c0a178ac5a3 100644
--- a/llvm/lib/Target/AArch64/GISel/AArch64RegisterBankInfo.cpp
+++ b/llvm/lib/Target/AArch64/GISel/AArch64RegisterBankInfo.cpp
@@ -1431,8 +1431,6 @@ AArch64RegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
}
case Intrinsic::aarch64_neon_vcvtfxs2fp:
case Intrinsic::aarch64_neon_vcvtfxu2fp:
- case Intrinsic::aarch64_neon_vcvtfp2fxs:
- case Intrinsic::aarch64_neon_vcvtfp2fxu:
// Override these intrinsics, because they would have a partial
// mapping. This is needed for 'half' types, which otherwise don't
// get legalised correctly.
@@ -1441,6 +1439,32 @@ AArch64RegisterBankInfo::getInstrMapping(const MachineInstr &MI) const {
// OpRegBankIdx[1] is the intrinsic ID.
// OpRegBankIdx[3] is an integer immediate.
break;
+ case Intrinsic::aarch64_neon_vcvtfp2fxs:
+ case Intrinsic::aarch64_neon_vcvtfp2fxu: {
+ OpRegBankIdx[2] = PMI_FirstFPR;
+ if (MRI.getType(MI.getOperand(0).getReg()).isVector()) {
+ OpRegBankIdx[0] = PMI_FirstFPR;
+ break;
+ }
+
+ TypeSize DstSize = getSizeInBits(MI.getOperand(0).getReg(), MRI, TRI);
+ TypeSize SrcSize = getSizeInBits(MI.getOperand(2).getReg(), MRI, TRI);
+
+ // Half-precision fixed-point FP-to-int scalar intrinsics are specified as
+ // producing an H-register result. The LLVM intrinsic may still return an
+ // i32/i64 type, so check the source size for 16 bits.
+ if (SrcSize == 16 ||
+ ((DstSize == SrcSize) &&
+ all_of(MRI.use_nodbg_instructions(MI.getOperand(0).getReg()),
+ [&](const MachineInstr &UseMI) {
+ return onlyUsesFP(UseMI, MRI, TRI) ||
+ prefersFPUse(UseMI, MRI, TRI);
+ })))
+ OpRegBankIdx[0] = PMI_FirstFPR;
+ else
+ OpRegBankIdx[0] = PMI_FirstGPR;
+ break;
+ }
default: {
// Check if we know that the intrinsic has any constraints on its register
// banks. If it does, then update the mapping accordingly.
diff --git a/llvm/test/CodeGen/AArch64/neon-scalar-vcvtfp2fx.ll b/llvm/test/CodeGen/AArch64/neon-scalar-vcvtfp2fx.ll
index 219a2950aa35d..784b9d42f1559 100644
--- a/llvm/test/CodeGen/AArch64/neon-scalar-vcvtfp2fx.ll
+++ b/llvm/test/CodeGen/AArch64/neon-scalar-vcvtfp2fx.ll
@@ -1,5 +1,6 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
-; RUN: llc -mtriple=aarch64 -global-isel=0 < %s | FileCheck %s
+; RUN: llc -mtriple=aarch64 -global-isel=0 < %s | FileCheck %s --check-prefixes=CHECK
+; RUN: llc -mtriple=aarch64 -global-isel=1 < %s | FileCheck %s --check-prefixes=CHECK
define i32 @vcvtfp2fxs_i32_f32(float %a) {
; CHECK-LABEL: vcvtfp2fxs_i32_f32:
@@ -76,3 +77,6 @@ define double @vcvtfp2fxu_i64_f64_bitcast(double %a) {
%b = bitcast i64 %r to double
ret double %b
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; CHECK-GI: {{.*}}
+; CHECK-SD: {{.*}}
More information about the llvm-commits
mailing list