[clang] [llvm] [AArch64][SME] Allow more inlining when SME attributes are incompatible. (PR #223393)

Benjamin Maxwell via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 07:53:13 PDT 2026


================
@@ -237,23 +237,100 @@ static bool isSMEABIRoutineCall(const CallInst &CI,
          SMEAttrs(F->getName(), TLI.getRuntimeLibcallsInfo()).isSMEABIRoutine();
 }
 
+/// Returns true if \p I is an intrinsic that may not be compatible with a
+/// different streaming mode (because it depends on vscale).
+static bool isPossiblyIncompatibleIntrinsic(const Instruction *I) {
+  if (I->isDebugOrPseudoInst())
+    return false;
+
+  if (auto *II = dyn_cast<IntrinsicInst>(I)) {
+    unsigned IID = II->getIntrinsicID();
+    switch (IID) {
+    default:
+      return Intrinsic::isTargetIntrinsic(IID);
+    case Intrinsic::vscale:
+    case Intrinsic::masked_gather:
+    case Intrinsic::masked_scatter:
+      return true;
+    }
+  }
+
+  return false;
+}
+
 /// Returns true if the function has explicit operations that can only be
 /// lowered using incompatible instructions for the selected mode. This also
 /// returns true if the function F may use or modify ZA state.
 static bool hasPossibleIncompatibleOps(const Function *F,
-                                       const AArch64TargetLowering &TLI) {
+                                       const AArch64TargetLowering &TLI,
+                                       bool ConsiderZA, bool ConsiderZT,
+                                       bool ConsiderSM) {
+  assert((ConsiderZA || ConsiderZT || ConsiderSM) &&
+         "No SME state to consider");
+
+  bool IsAlwaysInline = F->hasFnAttribute(Attribute::AlwaysInline);
+  bool HasVLDependentArgsOrRet =
+      F->getReturnType()->isScalableTy() ||
+      any_of(F->getFunctionType()->params(),
+             [](const Type *T) { return T->isScalableTy(); });
+
   for (const BasicBlock &BB : *F) {
     for (const Instruction &I : BB) {
-      // Be conservative for now and assume that any call to inline asm or to
-      // intrinsics could could result in non-streaming ops (e.g. calls to
-      // @llvm.aarch64.* or @llvm.gather/scatter intrinsics). We can assume that
-      // all native LLVM instructions can be lowered to compatible instructions.
-      if (isa<CallInst>(I) && !I.isDebugOrPseudoInst() &&
-          (cast<CallInst>(I).isInlineAsm() || isa<IntrinsicInst>(I) ||
-           isSMEABIRoutineCall(cast<CallInst>(I), TLI)))
+      // Inlining operations on fixed-length vectors when the streaming
+      // mode does not match, is rejected because performance may be impacted.
+      // This decision should eventually be moved the cost-model.
+      if (!IsAlwaysInline && ConsiderSM &&
+          (isa<FixedVectorType>(I.getType()) ||
+           any_of(I.operand_values(), [](const Value *V) {
+             return isa<FixedVectorType>(V->getType());
+           })))
+        return true;
+
+      // Inlining operations on scalable vectors is rejected because it is
+      // a vscale-dependent operation. The only exception is when the interface
+      // already has vscale-dependent arguments/return value, as the ACLE
+      // describes that in order for the program to have defined behaviour is
+      // for vscale to match in both modes.
+      if (ConsiderSM && !HasVLDependentArgsOrRet) {
+        if (I.getType()->isScalableTy() ||
+            any_of(
+                I.operand_values(),
+                [](const Value *V) { return V->getType()->isScalableTy(); }) ||
+            (isa<GetElementPtrInst>(I) && cast<GetElementPtrInst>(I)
+                                              .getSourceElementType()
+                                              ->isScalableTy()) ||
+            (isa<AllocaInst>(I) && cast<AllocaInst>(I).isScalable()))
+          return true;
+      }
+
+      auto *CB = dyn_cast<CallBase>(&I);
+      if (!CB)
+        continue;
+
+      // Inline asm must be rejected as it could use SME state.
+      if (CB->isInlineAsm())
         return true;
+
+      if (auto *CI = dyn_cast<CallInst>(&I)) {
+        // If the callee has calls to streaming compatible functions, then those
+        // may have vl-dependent statements. Be cautious about inlining such
+        // calls, as the streaming-compatible calls would otherwise be executed
+        // in a different streaming mode.
+        if (ConsiderSM && CI->getCalledFunction()) {
+          SMEAttrs CalleeAttrs(*CI->getCalledFunction());
+          if (CalleeAttrs.hasStreamingCompatibleInterface())
+            return true;
+        }
+
+        if (isSMEABIRoutineCall(*CI, TLI))
+          return true;
+
+        if (ConsiderSM && isPossiblyIncompatibleIntrinsic(&I))
+          return true;
+      }
----------------
MacDue wrote:

I think this should be rewritten to use the `SMECallAttrs` as that handles a couple of cases this currently misses: 
1. Indirect calls to streaming-compatible functions (where `getCalledFunction()` is null)
2. Invokes of streaming compatible functions (which are a `CallBase` but not a `CallInst`).
```suggestion
      SMECallAttrs CallAttrs(*CB, &TLI.getRuntimeLibcallsInfo());

      // If the callee has calls to streaming compatible functions, then those
      // may have vl-dependent statements. Be cautious about inlining such
      // calls, as the streaming-compatible calls would otherwise be executed
      // in a different streaming mode.
      if (ConsiderSM && CallAttrs.callee().hasStreamingCompatibleInterface())
        return true;

      if (CallAttrs.callee().isSMEABIRoutine())
        return true;

      if (auto *CI = dyn_cast<CallInst>(&I)) {
        if (ConsiderSM && isPossiblyIncompatibleIntrinsic(&I))
          return true;
      }
```

https://github.com/llvm/llvm-project/pull/223393


More information about the llvm-commits mailing list