[llvm] [AArch64] [CostModel] Improve costs for scalar inserts into fixed-length SVE constant vector (PR #223638)
Sander de Smalen via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 02:58:27 PDT 2026
================
@@ -4866,9 +4866,57 @@ InstructionCost AArch64TTIImpl::getScalarizationOverhead(
TTI::VectorInstrContext VIC) const {
if (isa<ScalableVectorType>(Ty))
return InstructionCost::getInvalid();
- if (Ty->getElementType()->isFloatingPointTy())
- return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
- CostKind);
+ if (Ty->getElementType()->isFloatingPointTy()) {
+ InstructionCost Cost = BaseT::getScalarizationOverhead(
+ Ty, DemandedElts, Insert, Extract, CostKind);
+
+ if (!Insert || VL.empty())
+ return Cost;
+
+ auto LT = getTypeLegalizationCost(Ty);
+ if (!ST->isNeonAvailable() || !LT.second.isFixedLengthVector() ||
+ LT.second.getFixedSizeInBits() <= 128)
+ return Cost;
+
+ auto HasNonUniformConstants = [&VL]() -> bool {
+ Value *FirstConst = nullptr;
+ for (Value *V : VL) {
+ if (isa<UndefValue>(V) || !isa<Constant>(V) ||
+ isa<ConstantExpr, GlobalValue>(V))
+ continue;
+ if (!FirstConst)
+ FirstConst = V;
+ else if (V != FirstConst)
+ return true;
+ }
+ return false;
+ };
+
+ if (!HasNonUniformConstants())
+ return Cost;
+
+ // Conservatively assume each legalized vector part is assembled from
+ // 128-bit subvectors, requiring NumSubParts - 1 splices and one shared
+ // predicate.
+ unsigned Num128BitSubParts =
+ LT.second.getFixedSizeInBits() / AArch64::SVEBitsPerBlock;
+ InstructionCost InsertExtractCost = LT.first * Num128BitSubParts *
+ DemandedElts.popcount() *
+ (Insert + Extract);
+
+ auto *ContainerTy = ScalableVectorType::get(Ty->getElementType(),
+ AArch64::SVEBitsPerBlock /
+ Ty->getScalarSizeInBits());
+ InstructionCost SpliceCost =
+ LT.first * (Num128BitSubParts - 1) *
+ getSpliceCost(ContainerTy, /*Index=*/0, CostKind);
+ Cost = InsertExtractCost + SpliceCost;
+ Cost += 1; // shared predicate cost
+ Cost += 2; // constant-pool address materialization.
----------------
sdesmalen-arm wrote:
These two costs can probably be removed; the predicate will be the same for all splice operations and is likely to be hoisted from a loop. The constant pool materialization cost should be included in the `getScalarizationOverhead` of the smaller vector (see my suggestion above).
https://github.com/llvm/llvm-project/pull/223638
More information about the llvm-commits
mailing list