[clang] [llvm] [AArch64][SME] Allow more inlining when SME attributes are incompatible. (PR #223393)

Sander de Smalen via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 00:29:24 PDT 2026


================
@@ -233,23 +233,96 @@ static bool isSMEABIRoutineCall(const CallInst &CI,
          SMEAttrs(F->getName(), TLI.getRuntimeLibcallsInfo()).isSMEABIRoutine();
 }
 
+/// Returns true if \p I is an intrinsic that may not be compatible with a
+/// different streaming mode (because it depends on vscale).
+static bool isPossiblyIncompatibleIntrinsic(const Instruction *I) {
+  if (I->isDebugOrPseudoInst())
+    return false;
+
+  if (auto *II = dyn_cast<IntrinsicInst>(I)) {
+    unsigned IID = II->getIntrinsicID();
+    switch (IID) {
+    default:
+      return Intrinsic::isTargetIntrinsic(IID);
+    case Intrinsic::vscale:
+    case Intrinsic::masked_gather:
+    case Intrinsic::masked_scatter:
+      return true;
+    }
+  }
+
+  return false;
+}
+
+/// Returns true if \p IA has "za" in its clobber list.
+static bool hasZAClobber(const InlineAsm *IA) {
+  for (const InlineAsm::ConstraintInfo &CI : IA->ParseConstraints()) {
+    if (CI.Type != llvm::InlineAsm::ConstraintPrefix::isClobber)
+      continue;
+    if (any_of(CI.Codes, [](StringRef S) { return S == "{za}"; }))
+      return true;
+  }
+  return false;
+}
+
 /// Returns true if the function has explicit operations that can only be
 /// lowered using incompatible instructions for the selected mode. This also
 /// returns true if the function F may use or modify ZA state.
 static bool hasPossibleIncompatibleOps(const Function *F,
-                                       const AArch64TargetLowering &TLI) {
+                                       const AArch64TargetLowering &TLI,
+                                       bool ConsiderZA, bool ConsiderSM) {
+  assert((ConsiderZA || ConsiderSM) && "No SME state to consider");
+
+  bool IsAlwaysInline = F->hasFnAttribute(Attribute::AlwaysInline);
+  bool HasVLDependentArgsOrRet =
+      F->getReturnType()->isScalableTy() ||
+      any_of(F->getFunctionType()->params(),
+             [](const Type *T) { return T->isScalableTy(); });
+
   for (const BasicBlock &BB : *F) {
     for (const Instruction &I : BB) {
-      // Be conservative for now and assume that any call to inline asm or to
-      // intrinsics could could result in non-streaming ops (e.g. calls to
-      // @llvm.aarch64.* or @llvm.gather/scatter intrinsics). We can assume that
-      // all native LLVM instructions can be lowered to compatible instructions.
-      if (isa<CallInst>(I) && !I.isDebugOrPseudoInst() &&
-          (cast<CallInst>(I).isInlineAsm() || isa<IntrinsicInst>(I) ||
-           isSMEABIRoutineCall(cast<CallInst>(I), TLI)))
+      // Inlining operations on fixed-length vectors when the streaming
+      // mode does not match, is rejected because performance may be impacted.
+      // This decision should eventually be moved the cost-model.
+      if (!IsAlwaysInline && ConsiderSM &&
+          (isa<FixedVectorType>(I.getType()) ||
+           any_of(I.operand_values(), [](const Value *V) {
+             return isa<FixedVectorType>(V->getType());
+           })))
         return true;
+
+      // Inlining operations on scalable vectors is rejected because it is
+      // a vscale-dependent operation. The only exception is when the interface
+      // already has vscale-dependent arguments/return value, as the ACLE
+      // describes that in order for the program to have defined behaviour is
+      // for vscale to match in both modes.
+      if (ConsiderSM && !HasVLDependentArgsOrRet) {
+        if (I.getType()->isScalableTy() ||
+            any_of(
+                I.operand_values(),
+                [](const Value *V) { return V->getType()->isScalableTy(); }) ||
+            (isa<GetElementPtrInst>(I) &&
+             cast<GetElementPtrInst>(I).getSourceElementType()->isScalableTy()))
+          return true;
+      }
+
+      if (auto *CI = dyn_cast<CallInst>(&I)) {
+        // Inline asm must be rejected, unless we know that it contains no
+        // vscale dependent operations and does not use ZA.
+        if (CI->isInlineAsm() &&
+            (ConsiderSM || (ConsiderZA && hasZAClobber(cast<InlineAsm>(
+                                              CI->getCalledOperand())))))
+          return true;
+
+        if (isSMEABIRoutineCall(*CI, TLI))
+          return true;
+
+        if (ConsiderSM && isPossiblyIncompatibleIntrinsic(&I))
----------------
sdesmalen-arm wrote:

The reason I didn't do that is because in order to use ZA in the callee, the callee needs to be either a shared-ZA or new-ZA function. In case of a shared-ZA function, the callee can be inlined as there's no lazy-save required. In case of New-ZA, the case is already rejected for inlining.

https://github.com/llvm/llvm-project/pull/223393


More information about the llvm-commits mailing list