[llvm] [SelectionDAG] Widen vector math libcalls when no routine is available (PR #218948)
Mattéo Rizza Murgier via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 02:25:20 PDT 2026
https://github.com/matteo-rm updated https://github.com/llvm/llvm-project/pull/218948
>From ad41386e586b7fa908e88636be05af0748672cbe Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Wed, 26 Aug 2026 13:23:36 +0200
Subject: [PATCH 1/6] [SelectionDAG] Widen vector math libcalls when no routine
is available
---
.../SelectionDAG/LegalizeVectorOps.cpp | 82 ++++---
.../CodeGen/AArch64/fp-veclib-expansion.ll | 203 ++++++++++++++++++
2 files changed, 253 insertions(+), 32 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 787f2683080e2..2a0af18338304 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -27,6 +27,7 @@
//===----------------------------------------------------------------------===//
#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/STLFunctionalExtras.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/Analysis/TargetLibraryInfo.h"
#include "llvm/Analysis/VectorUtils.h"
@@ -152,7 +153,8 @@ class VectorLegalizer {
void ExpandStrictFPOp(SDNode *Node, SmallVectorImpl<SDValue> &Results);
void ExpandREM(SDNode *Node, SmallVectorImpl<SDValue> &Results);
- bool tryExpandVecMathCall(SDNode *Node, RTLIB::Libcall LC,
+ bool tryExpandVecMathCall(SDNode *Node,
+ function_ref<RTLIB::Libcall(EVT)> GetLibcall,
SmallVectorImpl<SDValue> &Results);
void UnrollStrictFPOp(SDNode *Node, SmallVectorImpl<SDValue> &Results);
@@ -1244,13 +1246,10 @@ void VectorLegalizer::Expand(SDNode *Node, SmallVectorImpl<SDValue> &Results) {
return;
}
break;
- case ISD::FREM: {
- RTLIB::Libcall LC = RTLIB::getREM(Node->getValueType(0));
- if (tryExpandVecMathCall(Node, LC, Results))
+ case ISD::FREM:
+ if (tryExpandVecMathCall(Node, RTLIB::getREM, Results))
return;
-
break;
- }
case ISD::FSINCOS:
case ISD::FSINCOSPI: {
EVT VT = Node->getValueType(0);
@@ -1265,24 +1264,20 @@ void VectorLegalizer::Expand(SDNode *Node, SmallVectorImpl<SDValue> &Results) {
// scalarizing.
break;
}
- case ISD::FPOW: {
- RTLIB::Libcall LC = RTLIB::getPOW(Node->getValueType(0));
- if (tryExpandVecMathCall(Node, LC, Results))
+ case ISD::FPOW:
+ if (tryExpandVecMathCall(Node, RTLIB::getPOW, Results))
return;
// TODO: Try to see if there's a narrower call available to use before
// scalarizing.
break;
- }
- case ISD::FCBRT: {
- RTLIB::Libcall LC = RTLIB::getCBRT(Node->getValueType(0));
- if (tryExpandVecMathCall(Node, LC, Results))
+ case ISD::FCBRT:
+ if (tryExpandVecMathCall(Node, RTLIB::getCBRT, Results))
return;
// TODO: Try to see if there's a narrower call available to use before
// scalarizing.
break;
- }
case ISD::FMODF: {
EVT VT = Node->getValueType(0);
RTLIB::Libcall LC = RTLIB::getMODF(VT);
@@ -2147,24 +2142,33 @@ void VectorLegalizer::ExpandREM(SDNode *Node,
}
// Try to expand libm nodes into vector math routine calls. Callers provide the
-// LibFunc equivalent of the passed in Node, which is used to lookup mappings
-// within TargetLibraryInfo. The only mappings considered are those where the
-// result and all operands are the same vector type. While predicated nodes are
-// not supported, we will emit calls to masked routines by passing in an all
-// true mask.
-bool VectorLegalizer::tryExpandVecMathCall(SDNode *Node, RTLIB::Libcall LC,
- SmallVectorImpl<SDValue> &Results) {
+// RTLIB::get<OP>(EVT) selector of the node's libcall family, which is used to
+// look up mappings within RuntimeLibcallsInfo. The only mappings considered are
+// those where the result and all operands are the same vector type. While
+// predicated nodes are not supported, we will emit calls to masked routines by
+// passing in a mask that is true for the lanes computed by the node.
+bool VectorLegalizer::tryExpandVecMathCall(
+ SDNode *Node, function_ref<RTLIB::Libcall(EVT)> GetLibcall,
+ SmallVectorImpl<SDValue> &Results) {
// Chain must be propagated but currently strict fp operations are down
// converted to their none strict counterpart.
assert(!Node->isStrictFPOpcode() && "Unexpected strict fp operation!");
- RTLIB::LibcallImpl LCImpl = DAG.getLibcalls().getLibcallImpl(LC);
- if (LCImpl == RTLIB::Unsupported)
- return false;
-
EVT VT = Node->getValueType(0);
- const RTLIB::RuntimeLibcallsInfo &RTLCI = TLI.getRuntimeLibcallsInfo();
LLVMContext &Ctx = *DAG.getContext();
+ const LibcallLoweringInfo &Libcalls = DAG.getLibcalls();
+
+ // Try to widen the vector type when no libcall is available at that width.
+ EVT CallVT = VT;
+ RTLIB::LibcallImpl LCImpl = Libcalls.getLibcallImpl(GetLibcall(CallVT));
+ while (LCImpl == RTLIB::Unsupported) {
+ CallVT = CallVT.getDoubleNumVectorElementsVT(Ctx);
+ if (!TLI.isTypeLegal(CallVT))
+ return false;
+ LCImpl = Libcalls.getLibcallImpl(GetLibcall(CallVT));
+ }
+
+ const RTLIB::RuntimeLibcallsInfo &RTLCI = TLI.getRuntimeLibcallsInfo();
auto [FuncTy, FuncAttrs] = RTLCI.getFunctionTy(
Ctx, DAG.getSubtarget().getTargetTriple(), DAG.getDataLayout(), LCImpl);
@@ -2176,7 +2180,7 @@ bool VectorLegalizer::tryExpandVecMathCall(SDNode *Node, RTLIB::Libcall LC,
// Sanity check just in case function has unexpected parameters.
assert(FuncTy->getNumParams() == Node->getNumOperands() + HasMaskArg &&
- EVT::getEVT(FuncTy->getReturnType(), true) == VT &&
+ EVT::getEVT(FuncTy->getReturnType(), true) == CallVT &&
"mismatch in value type and call signature type");
for (unsigned I = 0, E = FuncTy->getNumParams(); I != E; ++I) {
@@ -2185,12 +2189,23 @@ bool VectorLegalizer::tryExpandVecMathCall(SDNode *Node, RTLIB::Libcall LC,
if (HasMaskArg && I == E - 1) {
assert(cast<VectorType>(ParamTy)->getElementType()->isIntegerTy(1) &&
"unexpected vector mask type");
- EVT MaskVT = TLI.getSetCCResultType(DAG.getDataLayout(), Ctx, VT);
- Args.emplace_back(DAG.getBoolConstant(true, DL, MaskVT, VT),
- MaskVT.getTypeForEVT(Ctx));
-
+ EVT MaskVT = TLI.getSetCCResultType(DAG.getDataLayout(), Ctx, CallVT);
+ SDValue Mask;
+ if (CallVT == VT) {
+ Mask = DAG.getBoolConstant(true, DL, MaskVT, CallVT);
+ } else {
+ // Only the lanes holding the node's elements need to be active.
+ EVT SubMaskVT = TLI.getSetCCResultType(DAG.getDataLayout(), Ctx, VT);
+ Mask = DAG.getInsertSubvector(
+ DL, DAG.getBoolConstant(false, DL, MaskVT, CallVT),
+ DAG.getBoolConstant(true, DL, SubMaskVT, VT), 0);
+ }
+ Args.emplace_back(Mask, MaskVT.getTypeForEVT(Ctx));
} else {
SDValue Op = Node->getOperand(I);
+ assert(Op.getValueType() == VT && "mismatch in vector types");
+ if (CallVT != VT)
+ Op = DAG.getInsertSubvector(DL, DAG.getPOISON(CallVT), Op, 0);
assert(Op.getValueType() == EVT::getEVT(ParamTy, true) &&
"mismatch in value type and call argument type");
Args.emplace_back(Op, ParamTy);
@@ -2208,7 +2223,10 @@ bool VectorLegalizer::tryExpandVecMathCall(SDNode *Node, RTLIB::Libcall LC,
.setLibCallee(CC, FuncTy->getReturnType(), Callee, std::move(Args));
std::pair<SDValue, SDValue> CallResult = TLI.LowerCallTo(CLI);
- Results.push_back(CallResult.first);
+ SDValue Result = CallResult.first;
+ if (CallVT != VT)
+ Result = DAG.getExtractSubvector(DL, VT, Result, 0);
+ Results.push_back(Result);
return true;
}
diff --git a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
index 2c8e2190f8209..166c5f0813fd2 100644
--- a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
+++ b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
@@ -112,5 +112,208 @@ define <vscale x 2 x double> @frem_strict_nxv2f64(<vscale x 2 x double> %unused,
ret <vscale x 2 x double> %res
}
+; Expected to be widened.
+define <2 x float> @frem_v2f32(<2 x float> %unused, <2 x float> %a, <2 x float> %b) #0 {
+; ARMPL-LABEL: frem_v2f32:
+; ARMPL: // %bb.0:
+; ARMPL-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; ARMPL-NEXT: .cfi_def_cfa_offset 16
+; ARMPL-NEXT: .cfi_offset w30, -16
+; ARMPL-NEXT: fmov d0, d1
+; ARMPL-NEXT: // kill: def $d2 killed $d2 def $q2
+; ARMPL-NEXT: mov v1.16b, v2.16b
+; ARMPL-NEXT: bl armpl_vfmodq_f32
+; ARMPL-NEXT: // kill: def $d0 killed $d0 killed $q0
+; ARMPL-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; ARMPL-NEXT: ret
+;
+; SLEEF-LABEL: frem_v2f32:
+; SLEEF: // %bb.0:
+; SLEEF-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; SLEEF-NEXT: .cfi_def_cfa_offset 16
+; SLEEF-NEXT: .cfi_offset w30, -16
+; SLEEF-NEXT: fmov d0, d1
+; SLEEF-NEXT: // kill: def $d2 killed $d2 def $q2
+; SLEEF-NEXT: mov v1.16b, v2.16b
+; SLEEF-NEXT: bl _ZGVnN4vv_fmodf
+; SLEEF-NEXT: // kill: def $d0 killed $d0 killed $q0
+; SLEEF-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; SLEEF-NEXT: ret
+ %res = frem <2 x float> %a, %b
+ ret <2 x float> %res
+}
+
+define <1 x double> @frem_v1f64(<1 x double> %unused, <1 x double> %a, <1 x double> %b) #0 {
+; ARMPL-LABEL: frem_v1f64:
+; ARMPL: // %bb.0:
+; ARMPL-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; ARMPL-NEXT: .cfi_def_cfa_offset 16
+; ARMPL-NEXT: .cfi_offset w30, -16
+; ARMPL-NEXT: fmov d0, d1
+; ARMPL-NEXT: // kill: def $d2 killed $d2 def $q2
+; ARMPL-NEXT: mov v1.16b, v2.16b
+; ARMPL-NEXT: bl armpl_vfmodq_f64
+; ARMPL-NEXT: // kill: def $d0 killed $d0 killed $q0
+; ARMPL-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; ARMPL-NEXT: ret
+;
+; SLEEF-LABEL: frem_v1f64:
+; SLEEF: // %bb.0:
+; SLEEF-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; SLEEF-NEXT: .cfi_def_cfa_offset 16
+; SLEEF-NEXT: .cfi_offset w30, -16
+; SLEEF-NEXT: fmov d0, d1
+; SLEEF-NEXT: // kill: def $d2 killed $d2 def $q2
+; SLEEF-NEXT: mov v1.16b, v2.16b
+; SLEEF-NEXT: bl _ZGVnN2vv_fmod
+; SLEEF-NEXT: // kill: def $d0 killed $d0 killed $q0
+; SLEEF-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; SLEEF-NEXT: ret
+ %res = frem <1 x double> %a, %b
+ ret <1 x double> %res
+}
+
+; Expected to widen with a mask.
+define <vscale x 2 x float> @frem_nxv2f32(<vscale x 2 x float> %unused, <vscale x 2 x float> %a, <vscale x 2 x float> %b) #0 {
+; ARMPL-LABEL: frem_nxv2f32:
+; ARMPL: // %bb.0:
+; ARMPL-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; ARMPL-NEXT: .cfi_def_cfa_offset 16
+; ARMPL-NEXT: .cfi_offset w30, -16
+; ARMPL-NEXT: pfalse p0.b
+; ARMPL-NEXT: uzp1 z0.s, z1.s, z1.s
+; ARMPL-NEXT: uzp1 z1.s, z2.s, z2.s
+; ARMPL-NEXT: ptrue p1.d
+; ARMPL-NEXT: uzp1 p0.s, p1.s, p0.s
+; ARMPL-NEXT: bl armpl_svfmod_f32_x
+; ARMPL-NEXT: uunpklo z0.d, z0.s
+; ARMPL-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; ARMPL-NEXT: ret
+;
+; SLEEF-LABEL: frem_nxv2f32:
+; SLEEF: // %bb.0:
+; SLEEF-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
+; SLEEF-NEXT: .cfi_def_cfa_offset 16
+; SLEEF-NEXT: .cfi_offset w30, -16
+; SLEEF-NEXT: pfalse p0.b
+; SLEEF-NEXT: uzp1 z0.s, z1.s, z1.s
+; SLEEF-NEXT: uzp1 z1.s, z2.s, z2.s
+; SLEEF-NEXT: ptrue p1.d
+; SLEEF-NEXT: uzp1 p0.s, p1.s, p0.s
+; SLEEF-NEXT: bl _ZGVsMxvv_fmodf
+; SLEEF-NEXT: uunpklo z0.d, z0.s
+; SLEEF-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
+; SLEEF-NEXT: ret
+ %res = frem <vscale x 2 x float> %a, %b
+ ret <vscale x 2 x float> %res
+}
+
+; Expected to get scalarized: no libcalls exist for this type.
+define <4 x half> @frem_v4f16(<4 x half> %unused, <4 x half> %a, <4 x half> %b) #0 {
+; ARMPL-LABEL: frem_v4f16:
+; ARMPL: // %bb.0:
+; ARMPL-NEXT: sub sp, sp, #64
+; ARMPL-NEXT: str x30, [sp, #48] // 8-byte Spill
+; ARMPL-NEXT: .cfi_def_cfa_offset 64
+; ARMPL-NEXT: .cfi_offset w30, -16
+; ARMPL-NEXT: // kill: def $d1 killed $d1 def $q1
+; ARMPL-NEXT: // kill: def $d2 killed $d2 def $q2
+; ARMPL-NEXT: stp q1, q2, [sp, #16] // 32-byte Folded Spill
+; ARMPL-NEXT: mov h0, v1.h[1]
+; ARMPL-NEXT: mov h1, v2.h[1]
+; ARMPL-NEXT: fcvt s0, h0
+; ARMPL-NEXT: fcvt s1, h1
+; ARMPL-NEXT: bl fmodf
+; ARMPL-NEXT: ldr q1, [sp, #16] // 16-byte Reload
+; ARMPL-NEXT: fcvt h0, s0
+; ARMPL-NEXT: fcvt s2, h1
+; ARMPL-NEXT: str q0, [sp] // 16-byte Spill
+; ARMPL-NEXT: ldr q0, [sp, #32] // 16-byte Reload
+; ARMPL-NEXT: fcvt s1, h0
+; ARMPL-NEXT: fmov s0, s2
+; ARMPL-NEXT: bl fmodf
+; ARMPL-NEXT: ldr q1, [sp, #16] // 16-byte Reload
+; ARMPL-NEXT: fcvt h3, s0
+; ARMPL-NEXT: ldr q0, [sp, #32] // 16-byte Reload
+; ARMPL-NEXT: mov h1, v1.h[2]
+; ARMPL-NEXT: mov h2, v0.h[2]
+; ARMPL-NEXT: ldr q0, [sp] // 16-byte Reload
+; ARMPL-NEXT: mov v3.h[1], v0.h[0]
+; ARMPL-NEXT: fcvt s0, h1
+; ARMPL-NEXT: fcvt s1, h2
+; ARMPL-NEXT: str q3, [sp] // 16-byte Spill
+; ARMPL-NEXT: bl fmodf
+; ARMPL-NEXT: ldp q1, q2, [sp, #16] // 32-byte Folded Reload
+; ARMPL-NEXT: fcvt h0, s0
+; ARMPL-NEXT: ldr q3, [sp] // 16-byte Reload
+; ARMPL-NEXT: mov h1, v1.h[3]
+; ARMPL-NEXT: mov h2, v2.h[3]
+; ARMPL-NEXT: mov v3.h[2], v0.h[0]
+; ARMPL-NEXT: fcvt s0, h1
+; ARMPL-NEXT: fcvt s1, h2
+; ARMPL-NEXT: str q3, [sp] // 16-byte Spill
+; ARMPL-NEXT: bl fmodf
+; ARMPL-NEXT: fcvt h1, s0
+; ARMPL-NEXT: ldr q0, [sp] // 16-byte Reload
+; ARMPL-NEXT: ldr x30, [sp, #48] // 8-byte Reload
+; ARMPL-NEXT: mov v0.h[3], v1.h[0]
+; ARMPL-NEXT: // kill: def $d0 killed $d0 killed $q0
+; ARMPL-NEXT: add sp, sp, #64
+; ARMPL-NEXT: ret
+;
+; SLEEF-LABEL: frem_v4f16:
+; SLEEF: // %bb.0:
+; SLEEF-NEXT: sub sp, sp, #64
+; SLEEF-NEXT: str x30, [sp, #48] // 8-byte Spill
+; SLEEF-NEXT: .cfi_def_cfa_offset 64
+; SLEEF-NEXT: .cfi_offset w30, -16
+; SLEEF-NEXT: // kill: def $d1 killed $d1 def $q1
+; SLEEF-NEXT: // kill: def $d2 killed $d2 def $q2
+; SLEEF-NEXT: stp q1, q2, [sp, #16] // 32-byte Folded Spill
+; SLEEF-NEXT: mov h0, v1.h[1]
+; SLEEF-NEXT: mov h1, v2.h[1]
+; SLEEF-NEXT: fcvt s0, h0
+; SLEEF-NEXT: fcvt s1, h1
+; SLEEF-NEXT: bl fmodf
+; SLEEF-NEXT: ldr q1, [sp, #16] // 16-byte Reload
+; SLEEF-NEXT: fcvt h0, s0
+; SLEEF-NEXT: fcvt s2, h1
+; SLEEF-NEXT: str q0, [sp] // 16-byte Spill
+; SLEEF-NEXT: ldr q0, [sp, #32] // 16-byte Reload
+; SLEEF-NEXT: fcvt s1, h0
+; SLEEF-NEXT: fmov s0, s2
+; SLEEF-NEXT: bl fmodf
+; SLEEF-NEXT: ldr q1, [sp, #16] // 16-byte Reload
+; SLEEF-NEXT: fcvt h3, s0
+; SLEEF-NEXT: ldr q0, [sp, #32] // 16-byte Reload
+; SLEEF-NEXT: mov h1, v1.h[2]
+; SLEEF-NEXT: mov h2, v0.h[2]
+; SLEEF-NEXT: ldr q0, [sp] // 16-byte Reload
+; SLEEF-NEXT: mov v3.h[1], v0.h[0]
+; SLEEF-NEXT: fcvt s0, h1
+; SLEEF-NEXT: fcvt s1, h2
+; SLEEF-NEXT: str q3, [sp] // 16-byte Spill
+; SLEEF-NEXT: bl fmodf
+; SLEEF-NEXT: ldp q1, q2, [sp, #16] // 32-byte Folded Reload
+; SLEEF-NEXT: fcvt h0, s0
+; SLEEF-NEXT: ldr q3, [sp] // 16-byte Reload
+; SLEEF-NEXT: mov h1, v1.h[3]
+; SLEEF-NEXT: mov h2, v2.h[3]
+; SLEEF-NEXT: mov v3.h[2], v0.h[0]
+; SLEEF-NEXT: fcvt s0, h1
+; SLEEF-NEXT: fcvt s1, h2
+; SLEEF-NEXT: str q3, [sp] // 16-byte Spill
+; SLEEF-NEXT: bl fmodf
+; SLEEF-NEXT: fcvt h1, s0
+; SLEEF-NEXT: ldr q0, [sp] // 16-byte Reload
+; SLEEF-NEXT: ldr x30, [sp, #48] // 8-byte Reload
+; SLEEF-NEXT: mov v0.h[3], v1.h[0]
+; SLEEF-NEXT: // kill: def $d0 killed $d0 killed $q0
+; SLEEF-NEXT: add sp, sp, #64
+; SLEEF-NEXT: ret
+ %res = frem <4 x half> %a, %b
+ ret <4 x half> %res
+}
+
attributes #0 = { "target-features"="+sve" }
attributes #1 = { "target-features"="+sve" strictfp }
>From 73b5ef2c3f02a726783652ea5ff1c66897ad7783 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Thu, 27 Aug 2026 10:40:47 +0200
Subject: [PATCH 2/6] Don't widen scalar calls
---
llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp | 2 ++
1 file changed, 2 insertions(+)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 2a0af18338304..0962adce768e4 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2161,6 +2161,8 @@ bool VectorLegalizer::tryExpandVecMathCall(
// Try to widen the vector type when no libcall is available at that width.
EVT CallVT = VT;
RTLIB::LibcallImpl LCImpl = Libcalls.getLibcallImpl(GetLibcall(CallVT));
+ if (LCImpl == RTLIB::Unsupported && VT.getVectorElementCount().isScalar())
+ return false;
while (LCImpl == RTLIB::Unsupported) {
CallVT = CallVT.getDoubleNumVectorElementsVT(Ctx);
if (!TLI.isTypeLegal(CallVT))
>From 9eb3313955c80d28779c6a6031367274acf9c291 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Thu, 27 Aug 2026 10:44:40 +0200
Subject: [PATCH 3/6] Don't stop at first illegal vec type
---
llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp | 5 +++--
1 file changed, 3 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 0962adce768e4..13c5f214f0c82 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2165,9 +2165,10 @@ bool VectorLegalizer::tryExpandVecMathCall(
return false;
while (LCImpl == RTLIB::Unsupported) {
CallVT = CallVT.getDoubleNumVectorElementsVT(Ctx);
- if (!TLI.isTypeLegal(CallVT))
+ if (!CallVT.isSimple())
return false;
- LCImpl = Libcalls.getLibcallImpl(GetLibcall(CallVT));
+ if (TLI.isTypeLegal(CallVT))
+ LCImpl = Libcalls.getLibcallImpl(GetLibcall(CallVT));
}
const RTLIB::RuntimeLibcallsInfo &RTLCI = TLI.getRuntimeLibcallsInfo();
>From 50bc9738f298fad3f5dd0286c8bb388966afce6e Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Thu, 27 Aug 2026 11:17:28 +0200
Subject: [PATCH 4/6] Fix mask/submask types discrepancy in some cases
---
.../CodeGen/SelectionDAG/LegalizeVectorOps.cpp | 15 ++++++---------
1 file changed, 6 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 13c5f214f0c82..8c04d069d59b7 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2193,16 +2193,13 @@ bool VectorLegalizer::tryExpandVecMathCall(
assert(cast<VectorType>(ParamTy)->getElementType()->isIntegerTy(1) &&
"unexpected vector mask type");
EVT MaskVT = TLI.getSetCCResultType(DAG.getDataLayout(), Ctx, CallVT);
- SDValue Mask;
- if (CallVT == VT) {
- Mask = DAG.getBoolConstant(true, DL, MaskVT, CallVT);
- } else {
- // Only the lanes holding the node's elements need to be active.
- EVT SubMaskVT = TLI.getSetCCResultType(DAG.getDataLayout(), Ctx, VT);
+ EVT SubMaskVT =
+ MaskVT.changeVectorElementCount(Ctx, VT.getVectorElementCount());
+ SDValue Mask = DAG.getBoolConstant(true, DL, SubMaskVT, VT);
+ // Only the lanes holding the node's elements need to be active.
+ if (CallVT != VT)
Mask = DAG.getInsertSubvector(
- DL, DAG.getBoolConstant(false, DL, MaskVT, CallVT),
- DAG.getBoolConstant(true, DL, SubMaskVT, VT), 0);
- }
+ DL, DAG.getBoolConstant(false, DL, MaskVT, CallVT), Mask, 0);
Args.emplace_back(Mask, MaskVT.getTypeForEVT(Ctx));
} else {
SDValue Op = Node->getOperand(I);
>From c53ef1337975dbab5baa6df9c2d05bd421a104fd Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Thu, 27 Aug 2026 11:18:54 +0200
Subject: [PATCH 5/6] Pad vectors with operand copy rather than poison
---
.../SelectionDAG/LegalizeVectorOps.cpp | 8 ++++++--
.../CodeGen/AArch64/fp-veclib-expansion.ll | 20 +++++++++----------
2 files changed, 16 insertions(+), 12 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
index 8c04d069d59b7..574b5c20aa139 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorOps.cpp
@@ -2204,8 +2204,12 @@ bool VectorLegalizer::tryExpandVecMathCall(
} else {
SDValue Op = Node->getOperand(I);
assert(Op.getValueType() == VT && "mismatch in vector types");
- if (CallVT != VT)
- Op = DAG.getInsertSubvector(DL, DAG.getPOISON(CallVT), Op, 0);
+ if (CallVT != VT) {
+ unsigned NumConcat =
+ CallVT.getVectorMinNumElements() / VT.getVectorMinNumElements();
+ SmallVector<SDValue, 4> Ops(NumConcat, Op);
+ Op = DAG.getNode(ISD::CONCAT_VECTORS, DL, CallVT, Ops);
+ }
assert(Op.getValueType() == EVT::getEVT(ParamTy, true) &&
"mismatch in value type and call argument type");
Args.emplace_back(Op, ParamTy);
diff --git a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
index 166c5f0813fd2..6b6ffb16404b3 100644
--- a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
+++ b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
@@ -119,8 +119,10 @@ define <2 x float> @frem_v2f32(<2 x float> %unused, <2 x float> %a, <2 x float>
; ARMPL-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; ARMPL-NEXT: .cfi_def_cfa_offset 16
; ARMPL-NEXT: .cfi_offset w30, -16
-; ARMPL-NEXT: fmov d0, d1
; ARMPL-NEXT: // kill: def $d2 killed $d2 def $q2
+; ARMPL-NEXT: fmov d0, d1
+; ARMPL-NEXT: mov v2.d[1], v2.d[0]
+; ARMPL-NEXT: mov v0.d[1], v0.d[0]
; ARMPL-NEXT: mov v1.16b, v2.16b
; ARMPL-NEXT: bl armpl_vfmodq_f32
; ARMPL-NEXT: // kill: def $d0 killed $d0 killed $q0
@@ -132,8 +134,10 @@ define <2 x float> @frem_v2f32(<2 x float> %unused, <2 x float> %a, <2 x float>
; SLEEF-NEXT: str x30, [sp, #-16]! // 8-byte Folded Spill
; SLEEF-NEXT: .cfi_def_cfa_offset 16
; SLEEF-NEXT: .cfi_offset w30, -16
-; SLEEF-NEXT: fmov d0, d1
; SLEEF-NEXT: // kill: def $d2 killed $d2 def $q2
+; SLEEF-NEXT: fmov d0, d1
+; SLEEF-NEXT: mov v2.d[1], v2.d[0]
+; SLEEF-NEXT: mov v0.d[1], v0.d[0]
; SLEEF-NEXT: mov v1.16b, v2.16b
; SLEEF-NEXT: bl _ZGVnN4vv_fmodf
; SLEEF-NEXT: // kill: def $d0 killed $d0 killed $q0
@@ -150,10 +154,8 @@ define <1 x double> @frem_v1f64(<1 x double> %unused, <1 x double> %a, <1 x doub
; ARMPL-NEXT: .cfi_def_cfa_offset 16
; ARMPL-NEXT: .cfi_offset w30, -16
; ARMPL-NEXT: fmov d0, d1
-; ARMPL-NEXT: // kill: def $d2 killed $d2 def $q2
-; ARMPL-NEXT: mov v1.16b, v2.16b
-; ARMPL-NEXT: bl armpl_vfmodq_f64
-; ARMPL-NEXT: // kill: def $d0 killed $d0 killed $q0
+; ARMPL-NEXT: fmov d1, d2
+; ARMPL-NEXT: bl fmod
; ARMPL-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
; ARMPL-NEXT: ret
;
@@ -163,10 +165,8 @@ define <1 x double> @frem_v1f64(<1 x double> %unused, <1 x double> %a, <1 x doub
; SLEEF-NEXT: .cfi_def_cfa_offset 16
; SLEEF-NEXT: .cfi_offset w30, -16
; SLEEF-NEXT: fmov d0, d1
-; SLEEF-NEXT: // kill: def $d2 killed $d2 def $q2
-; SLEEF-NEXT: mov v1.16b, v2.16b
-; SLEEF-NEXT: bl _ZGVnN2vv_fmod
-; SLEEF-NEXT: // kill: def $d0 killed $d0 killed $q0
+; SLEEF-NEXT: fmov d1, d2
+; SLEEF-NEXT: bl fmod
; SLEEF-NEXT: ldr x30, [sp], #16 // 8-byte Folded Reload
; SLEEF-NEXT: ret
%res = frem <1 x double> %a, %b
>From 64317a6a7981827ad0ccac95afe47a03cb00f7c6 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Matt=C3=A9o=20Rizza=20Murgier?=
<matteo.rizza-murgier at sipearl.com>
Date: Thu, 27 Aug 2026 11:25:01 +0200
Subject: [PATCH 6/6] Add comment to test
---
llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll | 1 +
1 file changed, 1 insertion(+)
diff --git a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
index 6b6ffb16404b3..88aa019414583 100644
--- a/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
+++ b/llvm/test/CodeGen/AArch64/fp-veclib-expansion.ll
@@ -147,6 +147,7 @@ define <2 x float> @frem_v2f32(<2 x float> %unused, <2 x float> %a, <2 x float>
ret <2 x float> %res
}
+; Expected not to be widened.
define <1 x double> @frem_v1f64(<1 x double> %unused, <1 x double> %a, <1 x double> %b) #0 {
; ARMPL-LABEL: frem_v1f64:
; ARMPL: // %bb.0:
More information about the llvm-commits
mailing list