[llvm] [ARM] Add combine for vector_deinterleave load (PR #225747)
David Green via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 06:26:35 PDT 2026
================
@@ -15446,6 +15446,169 @@ static SDValue PerformBUILD_VECTORCombine(SDNode *N,
return DAG.getNode(ISD::BITCAST, dl, VT, BV);
}
+static SDValue performLegalizedVECTOR_DEINTERLEAVECombine(
+ SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG,
+ const ARMSubtarget *Subtarget) {
+ // Type legalization splits a wide load into consecutive legal loads. Combine
+ // each group used by a legal VECTOR_DEINTERLEAVE into a structured load.
+ if (DCI.getDAGCombineLevel() < AfterLegalizeTypes)
+ return SDValue();
+
+ unsigned NumParts = N->getNumOperands();
+ if (NumParts < 2 || NumParts > 4)
+ return SDValue();
+
+ if (NumParts == 3 && !Subtarget->hasNEON())
+ return SDValue();
+
+ EVT SubVecTy = N->getValueType(0);
+ if (!SubVecTy.is64BitVector() && !SubVecTy.is128BitVector())
+ return SDValue();
+ unsigned EltBits = SubVecTy.getScalarSizeInBits();
+ if (EltBits != 8 && EltBits != 16 && EltBits != 32)
+ return SDValue();
+
+ SmallVector<LoadSDNode *, 4> Loads;
+ for (const SDValue &Operand : N->op_values()) {
+ auto *Load = dyn_cast<LoadSDNode>(Operand);
+ if (!Load || !Operand.hasOneUse())
+ return SDValue();
+ Loads.push_back(Load);
+ }
+
+ LoadSDNode *BaseLoad = Loads[0];
+ unsigned Bytes = SubVecTy.getStoreSize().getFixedValue();
+ for (auto [Idx, Load] : enumerate(Loads)) {
+ if (!DAG.areNonVolatileConsecutiveLoads(Load, BaseLoad, Bytes, Idx))
+ return SDValue();
+ }
+
+ static constexpr Intrinsic::ID NEONLoads[] = {Intrinsic::arm_neon_vld2,
+ Intrinsic::arm_neon_vld3,
+ Intrinsic::arm_neon_vld4};
+ SDLoc DL(N);
+ EVT MemVT =
+ EVT::getVectorVT(*DAG.getContext(), SubVecTy.getVectorElementType(),
+ SubVecTy.getVectorElementCount() * NumParts);
+ MachineFunction &MF = DAG.getMachineFunction();
+ MachineMemOperand *MMO =
+ MF.getMachineMemOperand(BaseLoad->getMemOperand(), 0, NumParts * Bytes);
+
+ SmallVector<EVT, 5> ResVTs(NumParts, SubVecTy);
+ ResVTs.push_back(MVT::Other);
+
+ SmallVector<SDValue> NewLdOps;
+ NewLdOps.push_back(BaseLoad->getChain());
+ Intrinsic::ID IID = 0;
+ if (Subtarget->hasNEON())
+ IID = NEONLoads[NumParts - 2];
+ else
+ IID = NumParts == 2 ? Intrinsic::arm_mve_vld2q : Intrinsic::arm_mve_vld4q;
+
+ NewLdOps.push_back(DAG.getTargetConstant(IID, DL, MVT::i64));
+ NewLdOps.push_back(BaseLoad->getBasePtr());
+ // We can now generate a structured load!
+ SDValue NewLoad = DAG.getMemIntrinsicNode(
+ ISD::INTRINSIC_W_CHAIN, DL, DAG.getVTList(ResVTs), NewLdOps, MemVT, MMO);
+
+ SmallVector<SDValue, 4> ResOps;
+ for (unsigned I = 0; I != NumParts; ++I)
+ ResOps.push_back(NewLoad.getValue(I));
+
+ // Replace uses of the original chain result with the new chain result.
+ for (LoadSDNode *Load : Loads)
+ DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), NewLoad.getValue(NumParts));
+
+ return DCI.CombineTo(N, ResOps, false);
+}
+
+// Combine a deinterleave of adjacent subvectors from one load into a
+// structured NEON load.
+static SDValue
+PerformVECTOR_DEINTERLEAVECombine(SDNode *N,
+ TargetLowering::DAGCombinerInfo &DCI,
+ const ARMSubtarget *Subtarget) {
+ if (SDValue Res = performLegalizedVECTOR_DEINTERLEAVECombine(N, DCI, DCI.DAG,
+ Subtarget))
+ return Res;
+
+ if (!DCI.isBeforeLegalize())
+ return SDValue();
+
+ SelectionDAG &DAG = DCI.DAG;
+ unsigned NumParts = N->getNumOperands();
+ if (NumParts != 2 && NumParts != 3 && NumParts != 4)
+ return SDValue();
+
+ if (NumParts == 3 && !Subtarget->hasNEON())
----------------
davemgreen wrote:
!Subtarget->hasNEON() -> Subtarget->hasMVE()
https://github.com/llvm/llvm-project/pull/225747
More information about the llvm-commits
mailing list