[llvm] [AArch64] Support lowering of wider interleaved load (PR #222998)
Paul Walker via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 08:58:16 PDT 2026
================
@@ -31345,8 +31279,78 @@ static bool isDeinterleave4ForWideningUToFP(SDNode *N) {
return llvm::all_of(Users, [](SDNode *User) { return User != nullptr; });
}
+static SDValue performLegalizedVectorDeinterleaveCombine(
+ SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG) {
+ // Type legalization splits a wide load into consecutive legal loads. Combine
+ // each group used by a legal VECTOR_DEINTERLEAVE into a structured load.
+ if (DCI.getDAGCombineLevel() < AfterLegalizeTypes ||
+ !DAG.getSubtarget<AArch64Subtarget>().isNeonAvailable())
+ return SDValue();
+
+ if (N->getOpcode() != ISD::VECTOR_DEINTERLEAVE)
+ return SDValue();
+
+ unsigned NumParts = N->getNumOperands();
+ if (NumParts < 2 || NumParts > 4)
+ return SDValue();
+
+ EVT SubVecTy = N->getValueType(0);
+ if (!SubVecTy.is64BitVector() && !SubVecTy.is128BitVector())
+ return SDValue();
+
+ SmallVector<LoadSDNode *, 4> Loads;
+ for (const SDValue &Operand : N->op_values()) {
+ auto *Load = dyn_cast<LoadSDNode>(Operand);
+ if (!Load || !Operand.hasOneUse())
+ return SDValue();
+ Loads.push_back(Load);
+ }
+
+ LoadSDNode *BaseLoad = Loads[0];
+ unsigned Bytes = SubVecTy.getStoreSize().getFixedValue();
+ for (auto [Idx, Load] : enumerate(Loads)) {
+ if (!DAG.areNonVolatileConsecutiveLoads(Load, BaseLoad, Bytes, Idx))
+ return SDValue();
+ }
+
+ static constexpr Intrinsic::ID NEONLoads[] = {Intrinsic::aarch64_neon_ld2,
+ Intrinsic::aarch64_neon_ld3,
+ Intrinsic::aarch64_neon_ld4};
+ SDLoc DL(N);
+ EVT MemVT =
+ EVT::getVectorVT(*DAG.getContext(), SubVecTy.getVectorElementType(),
+ SubVecTy.getVectorElementCount() * NumParts);
+ MachineFunction &MF = DAG.getMachineFunction();
+ MachineMemOperand *MMO =
+ MF.getMachineMemOperand(BaseLoad->getMemOperand(), 0, NumParts * Bytes);
+
+ SmallVector<EVT, 5> ResVTs(NumParts, SubVecTy);
+ ResVTs.push_back(MVT::Other);
+
+ SDValue NewLoad = DAG.getMemIntrinsicNode(
+ ISD::INTRINSIC_W_CHAIN, DL, DAG.getVTList(ResVTs),
+ {BaseLoad->getChain(),
+ DAG.getTargetConstant(NEONLoads[NumParts - 2], DL, MVT::i64),
+ BaseLoad->getBasePtr()},
+ MemVT, MMO);
+
+ // We can now generate a structured load!
----------------
paulwalker-arm wrote:
The comment seems more applicable to the previous line where NewLoad is created?
https://github.com/llvm/llvm-project/pull/222998
More information about the llvm-commits
mailing list