[llvm] [AArch64] [CostModel] Improve costs for scalar inserts into fixed-length SVE constant vector (PR #223638)

Sander de Smalen via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 02:58:27 PDT 2026


================
@@ -4866,9 +4866,57 @@ InstructionCost AArch64TTIImpl::getScalarizationOverhead(
     TTI::VectorInstrContext VIC) const {
   if (isa<ScalableVectorType>(Ty))
     return InstructionCost::getInvalid();
-  if (Ty->getElementType()->isFloatingPointTy())
-    return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
-                                           CostKind);
+  if (Ty->getElementType()->isFloatingPointTy()) {
+    InstructionCost Cost = BaseT::getScalarizationOverhead(
+        Ty, DemandedElts, Insert, Extract, CostKind);
+
+    if (!Insert || VL.empty())
+      return Cost;
+
+    auto LT = getTypeLegalizationCost(Ty);
+    if (!ST->isNeonAvailable() || !LT.second.isFixedLengthVector() ||
+        LT.second.getFixedSizeInBits() <= 128)
+      return Cost;
+
+    auto HasNonUniformConstants = [&VL]() -> bool {
+      Value *FirstConst = nullptr;
+      for (Value *V : VL) {
+        if (isa<UndefValue>(V) || !isa<Constant>(V) ||
+            isa<ConstantExpr, GlobalValue>(V))
+          continue;
+        if (!FirstConst)
+          FirstConst = V;
+        else if (V != FirstConst)
+          return true;
+      }
+      return false;
+    };
+
+    if (!HasNonUniformConstants())
+      return Cost;
+
+    // Conservatively assume each legalized vector part is assembled from
+    // 128-bit subvectors, requiring NumSubParts - 1 splices and one shared
+    // predicate.
+    unsigned Num128BitSubParts =
+        LT.second.getFixedSizeInBits() / AArch64::SVEBitsPerBlock;
+    InstructionCost InsertExtractCost = LT.first * Num128BitSubParts *
+                                        DemandedElts.popcount() *
+                                        (Insert + Extract);
+
+    auto *ContainerTy = ScalableVectorType::get(Ty->getElementType(),
+                                                AArch64::SVEBitsPerBlock /
+                                                    Ty->getScalarSizeInBits());
+    InstructionCost SpliceCost =
+        LT.first * (Num128BitSubParts - 1) *
+        getSpliceCost(ContainerTy, /*Index=*/0, CostKind);
+    Cost = InsertExtractCost + SpliceCost;
+    Cost += 1; // shared predicate cost
+    Cost += 2; // constant-pool address materialization.
----------------
sdesmalen-arm wrote:

These two costs can probably be removed; the predicate will be the same for all splice operations and is likely to be hoisted from a loop. The constant pool materialization cost should be included in the `getScalarizationOverhead` of the smaller vector (see my suggestion above).

https://github.com/llvm/llvm-project/pull/223638


More information about the llvm-commits mailing list