[clang] [llvm] [AArch64][SME] Allow more inlining when SME attributes are incompatible. (PR #223393)
Sander de Smalen via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 00:29:24 PDT 2026
================
@@ -233,23 +233,96 @@ static bool isSMEABIRoutineCall(const CallInst &CI,
SMEAttrs(F->getName(), TLI.getRuntimeLibcallsInfo()).isSMEABIRoutine();
}
+/// Returns true if \p I is an intrinsic that may not be compatible with a
+/// different streaming mode (because it depends on vscale).
+static bool isPossiblyIncompatibleIntrinsic(const Instruction *I) {
+ if (I->isDebugOrPseudoInst())
+ return false;
+
+ if (auto *II = dyn_cast<IntrinsicInst>(I)) {
+ unsigned IID = II->getIntrinsicID();
+ switch (IID) {
+ default:
+ return Intrinsic::isTargetIntrinsic(IID);
+ case Intrinsic::vscale:
+ case Intrinsic::masked_gather:
+ case Intrinsic::masked_scatter:
+ return true;
+ }
+ }
+
+ return false;
+}
+
+/// Returns true if \p IA has "za" in its clobber list.
+static bool hasZAClobber(const InlineAsm *IA) {
+ for (const InlineAsm::ConstraintInfo &CI : IA->ParseConstraints()) {
+ if (CI.Type != llvm::InlineAsm::ConstraintPrefix::isClobber)
+ continue;
+ if (any_of(CI.Codes, [](StringRef S) { return S == "{za}"; }))
+ return true;
+ }
+ return false;
+}
+
/// Returns true if the function has explicit operations that can only be
/// lowered using incompatible instructions for the selected mode. This also
/// returns true if the function F may use or modify ZA state.
static bool hasPossibleIncompatibleOps(const Function *F,
- const AArch64TargetLowering &TLI) {
+ const AArch64TargetLowering &TLI,
+ bool ConsiderZA, bool ConsiderSM) {
+ assert((ConsiderZA || ConsiderSM) && "No SME state to consider");
+
+ bool IsAlwaysInline = F->hasFnAttribute(Attribute::AlwaysInline);
+ bool HasVLDependentArgsOrRet =
+ F->getReturnType()->isScalableTy() ||
+ any_of(F->getFunctionType()->params(),
+ [](const Type *T) { return T->isScalableTy(); });
+
for (const BasicBlock &BB : *F) {
for (const Instruction &I : BB) {
- // Be conservative for now and assume that any call to inline asm or to
- // intrinsics could could result in non-streaming ops (e.g. calls to
- // @llvm.aarch64.* or @llvm.gather/scatter intrinsics). We can assume that
- // all native LLVM instructions can be lowered to compatible instructions.
- if (isa<CallInst>(I) && !I.isDebugOrPseudoInst() &&
- (cast<CallInst>(I).isInlineAsm() || isa<IntrinsicInst>(I) ||
- isSMEABIRoutineCall(cast<CallInst>(I), TLI)))
+ // Inlining operations on fixed-length vectors when the streaming
+ // mode does not match, is rejected because performance may be impacted.
+ // This decision should eventually be moved the cost-model.
+ if (!IsAlwaysInline && ConsiderSM &&
+ (isa<FixedVectorType>(I.getType()) ||
+ any_of(I.operand_values(), [](const Value *V) {
+ return isa<FixedVectorType>(V->getType());
+ })))
return true;
+
+ // Inlining operations on scalable vectors is rejected because it is
+ // a vscale-dependent operation. The only exception is when the interface
+ // already has vscale-dependent arguments/return value, as the ACLE
+ // describes that in order for the program to have defined behaviour is
+ // for vscale to match in both modes.
+ if (ConsiderSM && !HasVLDependentArgsOrRet) {
+ if (I.getType()->isScalableTy() ||
+ any_of(
+ I.operand_values(),
+ [](const Value *V) { return V->getType()->isScalableTy(); }) ||
+ (isa<GetElementPtrInst>(I) &&
+ cast<GetElementPtrInst>(I).getSourceElementType()->isScalableTy()))
+ return true;
+ }
+
+ if (auto *CI = dyn_cast<CallInst>(&I)) {
+ // Inline asm must be rejected, unless we know that it contains no
+ // vscale dependent operations and does not use ZA.
+ if (CI->isInlineAsm() &&
+ (ConsiderSM || (ConsiderZA && hasZAClobber(cast<InlineAsm>(
+ CI->getCalledOperand())))))
+ return true;
+
+ if (isSMEABIRoutineCall(*CI, TLI))
+ return true;
+
+ if (ConsiderSM && isPossiblyIncompatibleIntrinsic(&I))
----------------
sdesmalen-arm wrote:
The reason I didn't do that is because in order to use ZA in the callee, the callee needs to be either a shared-ZA or new-ZA function. In case of a shared-ZA function, the callee can be inlined as there's no lazy-save required. In case of New-ZA, the case is already rejected for inlining.
https://github.com/llvm/llvm-project/pull/223393
More information about the llvm-commits
mailing list