[llvm] [X86] Reject incompatible vector library calls (PR #226364)

Tõnu Samuel via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 00:02:27 PDT 2026


https://github.com/tonuonu updated https://github.com/llvm/llvm-project/pull/226364

>From c48f7d5e820f563f20683692bf2fe19cbae20815 Mon Sep 17 00:00:00 2001
From: Tonu Samuel <tonu at spam.ee>
Date: Fri, 25 Sep 2026 08:44:12 +0300
Subject: [PATCH 1/2] [X86] Reject incompatible vector library calls

Check vector-library call compatibility through TTI before selecting a
variant in LV, VPlan, SLP, or ReplaceWithVeclib. The x86 implementation
checks the GNU vector ABI ISA class and the actual argument/return vector
widths against the caller's subtarget and preferred vector width.

This prevents a float loop containing double-precision calls from selecting
an incompatible wider library ABI. Unsupported intrinsic replacements are
left for codegen legalization; this does not introduce vector-call splitting.

Add regression coverage for mixed precision, AVX versus AVX2, per-function
features, and AVX-512 with a 256-bit preference. Make ISA requirements
explicit in existing positive mapping tests and update sincos expectations.

Assisted-by: OpenAI Codex
---
 .../llvm/Analysis/TargetTransformInfo.h       |   7 +
 .../llvm/Analysis/TargetTransformInfoImpl.h   |   5 +
 llvm/include/llvm/Analysis/VectorUtils.h      |  22 +-
 llvm/lib/Analysis/TargetTransformInfo.cpp     |   5 +
 llvm/lib/Analysis/VectorUtils.cpp             |  13 +
 llvm/lib/CodeGen/ReplaceWithVeclib.cpp        |  26 +-
 .../lib/Target/X86/X86TargetTransformInfo.cpp |  36 ++
 llvm/lib/Target/X86/X86TargetTransformInfo.h  |   2 +
 .../Vectorize/LoopVectorizationLegality.cpp   |   7 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    |  14 +-
 .../Transforms/Vectorize/SLPVectorizer.cpp    |   8 +-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |   9 +-
 llvm/test/CodeGen/X86/veclib-llvm.sincos.ll   |  56 ++-
 .../LoopVectorize/X86/amdlibm-calls-finite.ll |   2 +-
 .../X86/amdlibm-target-features.ll            |  75 ++++
 .../X86/libm-vector-calls-finite.ll           | 114 +++---
 .../LoopVectorize/X86/libm-vector-calls.ll    | 376 +++++++++---------
 .../X86/libmvec-target-features.ll            |  82 ++++
 .../LoopVectorize/X86/svml-calls-finite.ll    |   2 +-
 .../X86/amdlibm-target-features.ll            |  44 ++
 .../X86/libmvec-target-features.ll            |  61 +++
 .../ReplaceWithVeclib/X86/lit.local.cfg       |   2 +
 .../X86/libmvec-target-features.ll            |  66 +++
 23 files changed, 738 insertions(+), 296 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
 create mode 100644 llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
 create mode 100644 llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
 create mode 100644 llvm/test/Transforms/ReplaceWithVeclib/X86/lit.local.cfg
 create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index e30cbc61a5420b..1cbc17aa074616 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -1038,6 +1038,13 @@ class TargetTransformInfo {
   /// Return true if this type is legal.
   LLVM_ABI bool isTypeLegal(Type *Ty) const;
 
+  /// Whether a vector function can be called directly from this function.
+  /// Unlike an intrinsic, an external vector call cannot be legalized by
+  /// splitting its operands without changing the callee's ABI. Targets may
+  /// also impose ISA requirements encoded in the vector function's name.
+  LLVM_ABI bool isLegalToCallVectorFunction(FunctionType *FTy,
+                                            StringRef Name) const;
+
   /// Returns the estimated number of registers required to represent \p Ty.
   LLVM_ABI unsigned getRegUsageForType(Type *Ty) const;
 
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 625433e2e0a0b9..82879eedc013e8 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -469,6 +469,11 @@ class LLVM_ABI TargetTransformInfoImplBase {
 
   virtual bool isTypeLegal(Type *Ty) const { return false; }
 
+  virtual bool isLegalToCallVectorFunction(FunctionType *FTy,
+                                           StringRef Name) const {
+    return true;
+  }
+
   virtual unsigned getRegUsageForType(Type *Ty) const { return 1; }
 
   virtual bool shouldBuildLookupTables() const { return true; }
diff --git a/llvm/include/llvm/Analysis/VectorUtils.h b/llvm/include/llvm/Analysis/VectorUtils.h
index b177d9eec21896..8faf285764a098 100644
--- a/llvm/include/llvm/Analysis/VectorUtils.h
+++ b/llvm/include/llvm/Analysis/VectorUtils.h
@@ -26,6 +26,7 @@
 
 namespace llvm {
 class TargetLibraryInfo;
+class TargetTransformInfo;
 class IntrinsicInst;
 
 /// The Vector Function Database.
@@ -73,23 +74,16 @@ class VFDatabase {
 
 public:
   /// Retrieve all the VFInfo instances associated to the CallInst CI.
-  static SmallVector<VFInfo, 8> getMappings(const CallInst &CI) {
-    SmallVector<VFInfo, 8> Ret;
-
-    // Get mappings from the Vector Function ABI variants.
-    getVFABIMappings(CI, Ret);
-
-    // Other non-VFABI variants should be retrieved here.
-
-    return Ret;
-  }
+  LLVM_ABI static SmallVector<VFInfo, 8>
+  getMappings(const CallInst &CI, const TargetTransformInfo *TTI = nullptr);
 
   static bool hasMaskedVariant(const CallInst &CI,
-                               std::optional<ElementCount> VF = std::nullopt) {
+                               std::optional<ElementCount> VF = std::nullopt,
+                               const TargetTransformInfo *TTI = nullptr) {
     // Check whether we have at least one masked vector version of a scalar
     // function. If no VF is specified then we check for any masked variant,
     // otherwise we look for one that matches the supplied VF.
-    auto Mappings = VFDatabase::getMappings(CI);
+    auto Mappings = VFDatabase::getMappings(CI, TTI);
     for (VFInfo Info : Mappings)
       if (!VF || Info.Shape.VF == *VF)
         if (Info.isMasked())
@@ -99,9 +93,9 @@ class VFDatabase {
   }
 
   /// Constructor, requires a CallInst instance.
-  VFDatabase(CallInst &CI)
+  VFDatabase(CallInst &CI, const TargetTransformInfo *TTI = nullptr)
       : M(CI.getModule()), CI(CI),
-        ScalarToVectorMappings(VFDatabase::getMappings(CI)) {}
+        ScalarToVectorMappings(VFDatabase::getMappings(CI, TTI)) {}
 
   /// \defgroup VFDatabase query interface.
   ///
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 4c2cac9c440a08..3dde0e3f3df3d3 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -601,6 +601,11 @@ bool TargetTransformInfo::isProfitableToHoist(Instruction *I) const {
 
 bool TargetTransformInfo::useAA() const { return TTIImpl->useAA(); }
 
+bool TargetTransformInfo::isLegalToCallVectorFunction(FunctionType *FTy,
+                                                      StringRef Name) const {
+  return TTIImpl->isLegalToCallVectorFunction(FTy, Name);
+}
+
 bool TargetTransformInfo::isTypeLegal(Type *Ty) const {
   return TTIImpl->isTypeLegal(Ty);
 }
diff --git a/llvm/lib/Analysis/VectorUtils.cpp b/llvm/lib/Analysis/VectorUtils.cpp
index 1c105ebb772b35..7649b729de5050 100644
--- a/llvm/lib/Analysis/VectorUtils.cpp
+++ b/llvm/lib/Analysis/VectorUtils.cpp
@@ -33,6 +33,19 @@
 using namespace llvm;
 using namespace llvm::PatternMatch;
 
+SmallVector<VFInfo, 8> VFDatabase::getMappings(const CallInst &CI,
+                                               const TargetTransformInfo *TTI) {
+  SmallVector<VFInfo, 8> Mappings;
+  getVFABIMappings(CI, Mappings);
+  if (TTI)
+    llvm::erase_if(Mappings, [&](const VFInfo &Info) {
+      const Function *VF = CI.getModule()->getFunction(Info.VectorName);
+      return !TTI->isLegalToCallVectorFunction(VF->getFunctionType(),
+                                               VF->getName());
+    });
+  return Mappings;
+}
+
 /// Maximum factor for an interleaved memory access.
 static cl::opt<unsigned> MaxInterleaveGroupFactor(
     "max-interleave-group-factor", cl::Hidden,
diff --git a/llvm/lib/CodeGen/ReplaceWithVeclib.cpp b/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
index c2a7835504e674..f7242b5c1e9306 100644
--- a/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
+++ b/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
@@ -19,6 +19,7 @@
 #include "llvm/Analysis/GlobalsModRef.h"
 #include "llvm/Analysis/OptimizationRemarkEmitter.h"
 #include "llvm/Analysis/TargetLibraryInfo.h"
+#include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/CodeGen/Passes.h"
 #include "llvm/IR/DerivedTypes.h"
@@ -105,6 +106,7 @@ static void replaceWithTLIFunction(IntrinsicInst *II, VFInfo &Info,
 /// vectorized intrinsic, with a suitable function taking vector arguments,
 /// based on available mappings in the \p TLI.
 static bool replaceWithCallToVeclib(const TargetLibraryInfo &TLI,
+                                    const TargetTransformInfo &TTI,
                                     IntrinsicInst *II) {
   assert(II != nullptr && "Intrinsic cannot be null");
   Intrinsic::ID IID = II->getIntrinsicID();
@@ -197,7 +199,8 @@ static bool replaceWithCallToVeclib(const TargetLibraryInfo &TLI,
   }
 
   FunctionType *VectorFTy = VFABI::createFunctionType(*OptInfo, ScalarFTy);
-  if (!VectorFTy)
+  if (!VectorFTy ||
+      !TTI.isLegalToCallVectorFunction(VectorFTy, VD->getVectorFnName()))
     return false;
 
   Function *TLIFunc =
@@ -232,6 +235,7 @@ static bool hasIntrinsicVectorMapping(const TargetLibraryInfo &TLI,
 /// same element count, replace it with separate llvm.sin and llvm.cos calls
 /// and run the standard veclib replacement on each.
 static bool trySplitVectorSinCos(const TargetLibraryInfo &TLI,
+                                 const TargetTransformInfo &TTI,
                                  IntrinsicInst *II,
                                  SmallVectorImpl<Instruction *> &Replaced) {
   if (II->getIntrinsicID() != Intrinsic::sincos)
@@ -288,15 +292,16 @@ static bool trySplitVectorSinCos(const TargetLibraryInfo &TLI,
   }
 
   // Replace each new call with the vector library function.
-  if (replaceWithCallToVeclib(TLI, cast<IntrinsicInst>(SinCall)))
+  if (replaceWithCallToVeclib(TLI, TTI, cast<IntrinsicInst>(SinCall)))
     Replaced.push_back(SinCall);
-  if (replaceWithCallToVeclib(TLI, cast<IntrinsicInst>(CosCall)))
+  if (replaceWithCallToVeclib(TLI, TTI, cast<IntrinsicInst>(CosCall)))
     Replaced.push_back(CosCall);
 
   return true;
 }
 
-static bool runImpl(const TargetLibraryInfo &TLI, Function &F) {
+static bool runImpl(const TargetLibraryInfo &TLI,
+                    const TargetTransformInfo &TTI, Function &F) {
   SmallVector<Instruction *> ReplacedCalls;
   for (auto &I : instructions(F)) {
     auto *II = dyn_cast<IntrinsicInst>(&I);
@@ -306,7 +311,7 @@ static bool runImpl(const TargetLibraryInfo &TLI, Function &F) {
     // Vector llvm.sincos returns a struct so it does not fit the generic
     // path below; try to split it into separate sin and cos calls when the
     // target has vector mappings for them.
-    if (trySplitVectorSinCos(TLI, II, ReplacedCalls)) {
+    if (trySplitVectorSinCos(TLI, TTI, II, ReplacedCalls)) {
       ReplacedCalls.push_back(&I);
       continue;
     }
@@ -315,7 +320,7 @@ static bool runImpl(const TargetLibraryInfo &TLI, Function &F) {
     if (!II->getType()->isVectorTy() && !II->getType()->isVoidTy())
       continue;
 
-    if (replaceWithCallToVeclib(TLI, II))
+    if (replaceWithCallToVeclib(TLI, TTI, II))
       ReplacedCalls.push_back(&I);
   }
   // Erase any intrinsic calls that were replaced with vector library calls.
@@ -330,7 +335,8 @@ static bool runImpl(const TargetLibraryInfo &TLI, Function &F) {
 PreservedAnalyses ReplaceWithVeclib::run(Function &F,
                                          FunctionAnalysisManager &AM) {
   const TargetLibraryInfo &TLI = AM.getResult<TargetLibraryAnalysis>(F);
-  auto Changed = runImpl(TLI, F);
+  const TargetTransformInfo &TTI = AM.getResult<TargetIRAnalysis>(F);
+  auto Changed = runImpl(TLI, TTI, F);
   if (Changed) {
     LLVM_DEBUG(dbgs() << "Intrinsic calls replaced with vector libraries: "
                       << NumCallsReplaced << "\n");
@@ -355,12 +361,15 @@ PreservedAnalyses ReplaceWithVeclib::run(Function &F,
 bool ReplaceWithVeclibLegacy::runOnFunction(Function &F) {
   const TargetLibraryInfo &TLI =
       getAnalysis<TargetLibraryInfoWrapperPass>().getTLI(F);
-  return runImpl(TLI, F);
+  const TargetTransformInfo &TTI =
+      getAnalysis<TargetTransformInfoWrapperPass>().getTTI(F);
+  return runImpl(TLI, TTI, F);
 }
 
 void ReplaceWithVeclibLegacy::getAnalysisUsage(AnalysisUsage &AU) const {
   AU.setPreservesCFG();
   AU.addRequired<TargetLibraryInfoWrapperPass>();
+  AU.addRequired<TargetTransformInfoWrapperPass>();
   AU.addPreserved<TargetLibraryInfoWrapperPass>();
   AU.addPreserved<ScalarEvolutionWrapperPass>();
   AU.addPreserved<AAResultsWrapperPass>();
@@ -377,6 +386,7 @@ INITIALIZE_PASS_BEGIN(ReplaceWithVeclibLegacy, DEBUG_TYPE,
                       "Replace intrinsics with calls to vector library", false,
                       false)
 INITIALIZE_PASS_DEPENDENCY(TargetLibraryInfoWrapperPass)
+INITIALIZE_PASS_DEPENDENCY(TargetTransformInfoWrapperPass)
 INITIALIZE_PASS_END(ReplaceWithVeclibLegacy, DEBUG_TYPE,
                     "Replace intrinsics with calls to vector library", false,
                     false)
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index 846e8c8968ccc7..d2e122f9bc4d9f 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -207,6 +207,42 @@ bool X86TTIImpl::hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const {
   }
 }
 
+static bool hasSupportedVectorTypes(Type *Ty, const DataLayout &DL,
+                                    TypeSize MaxWidth) {
+  if (auto *VT = dyn_cast<VectorType>(Ty))
+    return !VT->getElementCount().isScalable() &&
+           TypeSize::isKnownLE(DL.getTypeSizeInBits(VT), MaxWidth);
+  if (auto *ST = dyn_cast<StructType>(Ty))
+    return all_of(ST->elements(), [&](Type *Elt) {
+      return hasSupportedVectorTypes(Elt, DL, MaxWidth);
+    });
+  if (auto *AT = dyn_cast<ArrayType>(Ty))
+    return hasSupportedVectorTypes(AT->getElementType(), DL, MaxWidth);
+  return true;
+}
+
+bool X86TTIImpl::isLegalToCallVectorFunction(FunctionType *FTy,
+                                             StringRef Name) const {
+  // The x86 Vector Function ABI encodes the required ISA in the symbol.
+  // In particular, a 256-bit AVX caller cannot use the AVX2 ('d') variant,
+  // even though its vector arguments fit in a YMM register.
+  if ((Name.starts_with("_ZGVb") && !ST->hasSSE2()) ||
+      (Name.starts_with("_ZGVc") && !ST->hasAVX()) ||
+      (Name.starts_with("_ZGVd") && !ST->hasAVX2()) ||
+      (Name.starts_with("_ZGVe") && !ST->hasAVX512()))
+    return false;
+
+  // Check the actual call signature, not the vectorized loop's element type.
+  // A float loop can contain a double-precision call with twice its width.
+  // Conservatively avoid calls wider than the target's preferred vector width:
+  // codegen must not split the operands of an already selected external call.
+  TypeSize MaxWidth = getRegisterBitWidth(TTI::RGK_FixedWidthVector);
+  return hasSupportedVectorTypes(FTy->getReturnType(), DL, MaxWidth) &&
+         all_of(FTy->params(), [&](Type *Ty) {
+           return hasSupportedVectorTypes(Ty, DL, MaxWidth);
+         });
+}
+
 TypeSize
 X86TTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
   unsigned PreferVectorWidth = ST->getPreferVectorWidth();
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index f4197c260ba81a..32033700fc1106 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -58,6 +58,8 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
   /// \name Vector TTI Implementations
   /// @{
 
+  bool isLegalToCallVectorFunction(FunctionType *FTy,
+                                   StringRef Name) const override;
   unsigned getNumberOfRegisters(unsigned ClassID) const override;
   unsigned getRegisterClassForType(bool Vector, Type *Ty) const override;
   bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const override;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 862470ce0de6d2..fe384fd74b2722 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -925,7 +925,8 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
 
   if (CI && !getVectorIntrinsicIDForCall(CI, TLI) &&
       !(CI->getCalledFunction() && TLI &&
-        (!VFDatabase::getMappings(*CI).empty() || isTLIScalarize(*TLI, *CI)))) {
+        (!VFDatabase::getMappings(*CI, TTI).empty() ||
+         isTLIScalarize(*TLI, *CI)))) {
     // If the call is a recognized math libary call, it is likely that
     // we can vectorize it given loosened floating-point constraints.
     bool IsMathLibCall =
@@ -971,7 +972,7 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
 
   // If we found a vectorized variant of a function, note that so LV can
   // make better decisions about maximum VF.
-  if (CI && !VFDatabase::getMappings(*CI).empty())
+  if (CI && !VFDatabase::getMappings(*CI, TTI).empty())
     VecCallVariantsFound = true;
 
   auto CanWidenInstructionTy = [](Instruction const &Inst) {
@@ -1399,7 +1400,7 @@ bool LoopVectorizationLegality::blockCanBePredicated(
     // TODO: Allow other calls if they have appropriate attributes... readonly
     // and argmemonly?
     if (CallInst *CI = dyn_cast<CallInst>(&I))
-      if (VFDatabase::hasMaskedVariant(*CI)) {
+      if (VFDatabase::hasMaskedVariant(*CI, std::nullopt, TTI)) {
         MaskedOp.insert(CI);
         continue;
       }
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 2f1fc4398654ae..a470a93994314a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2088,10 +2088,11 @@ static unsigned estimateElementCount(ElementCount VF,
 /// module.
 static Function *getVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
                                             bool MaskRequired,
-                                            const TargetLibraryInfo *TLI) {
+                                            const TargetLibraryInfo *TLI,
+                                            const TargetTransformInfo &TTI) {
   if (!TLI || CI.isNoBuiltin())
     return nullptr;
-  for (const VFInfo &Info : VFDatabase::getMappings(CI))
+  for (const VFInfo &Info : VFDatabase::getMappings(CI, &TTI))
     if (Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()))
       if (Function *F = CI.getModule()->getFunction(Info.VectorName))
         return F;
@@ -2101,8 +2102,9 @@ static Function *getVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
 /// Returns true iff \p CI has a library vector variant usable at \p VF.
 static bool hasVectorLibraryVariantFor(const CallInst &CI, ElementCount VF,
                                        bool MaskRequired,
-                                       const TargetLibraryInfo *TLI) {
-  return getVectorLibraryVariantFor(CI, VF, MaskRequired, TLI) != nullptr;
+                                       const TargetLibraryInfo *TLI,
+                                       const TargetTransformInfo &TTI) {
+  return getVectorLibraryVariantFor(CI, VF, MaskRequired, TLI, TTI) != nullptr;
 }
 
 InstructionCost
@@ -2129,7 +2131,7 @@ LoopVectorizationCostModel::getVectorCallCost(CallInst *CI,
     Cost = std::min(Cost, getVectorIntrinsicCost(CI, VF));
 
   if (Function *Variant =
-          getVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI))
+          getVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI, TTI))
     Cost = std::min(Cost,
                     TTI.getCallInstrCost(
                         /*F=*/nullptr, Variant->getReturnType(),
@@ -2402,7 +2404,7 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
     auto *CI = cast<CallInst>(I);
     // A vector intrinsic or library variant lowering avoids scalarization.
     return !getVectorIntrinsicIDForCall(CI, TLI) &&
-           !hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI);
+           !hasVectorLibraryVariantFor(*CI, VF, isMaskRequired(CI), TLI, TTI);
   }
   case Instruction::Load:
   case Instruction::Store: {
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b76cf0677c2f6e..b9eb50fcd895a6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -9125,7 +9125,7 @@ getVectorCallCosts(CallInst *CI, Type *VecTy, const TargetTransformInfo *TTI,
   auto Shape = VFShape::get(CI->getFunctionType(),
                             ElementCount::getFixed(getNumElements(VecTy)),
                             false /*HasGlobalPred*/);
-  Function *VecFunc = VFDatabase(*CI).getVectorizedFunction(Shape);
+  Function *VecFunc = VFDatabase(*CI, TTI).getVectorizedFunction(Shape);
   auto LibCost = InstructionCost::getInvalid();
   if (!CI->isNoBuiltin() && VecFunc) {
     // Calculate the cost of the vector library call.
@@ -9790,7 +9790,7 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
         CI->getFunctionType(),
         ElementCount::getFixed(static_cast<unsigned int>(VL.size())),
         false /*HasGlobalPred*/);
-    Function *VecFunc = VFDatabase(*CI).getVectorizedFunction(Shape);
+    Function *VecFunc = VFDatabase(*CI, TTI).getVectorizedFunction(Shape);
 
     if (!VecFunc && !isTriviallyVectorizable(ID)) {
       LLVM_DEBUG(dbgs() << "SLP: Non-vectorizable call.\n");
@@ -9822,7 +9822,7 @@ BoUpSLP::TreeEntry::EntryState BoUpSLP::getScalarsVectorizationState(
                Intrinsic::not_intrinsic) ||
           (ID != ID2 && Equivalent == Intrinsic::not_intrinsic) ||
           (VecFunc &&
-           VecFunc != VFDatabase(*CI2).getVectorizedFunction(Shape)) ||
+           VecFunc != VFDatabase(*CI2, TTI).getVectorizedFunction(Shape)) ||
           !CI->hasIdenticalOperandBundleSchema(*CI2)) {
         LLVM_DEBUG(dbgs() << "SLP: mismatched calls:" << *CI << "!=" << *V
                           << "\n");
@@ -24850,7 +24850,7 @@ Value *BoUpSLP::vectorizeTree(TreeEntry *E) {
             VFShape::get(CI->getFunctionType(),
                          ElementCount::getFixed(getNumElements(VecTy)),
                          false /*HasGlobalPred*/);
-        CF = VFDatabase(*CI).getVectorizedFunction(Shape);
+        CF = VFDatabase(*CI, TTI).getVectorizedFunction(Shape);
       } else {
         CF = Intrinsic::getOrInsertDeclaration(F->getParent(), ID, TysForDecl);
       }
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 7f135c97e63928..6c9ba6d64a7cbd 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -5791,10 +5791,11 @@ static bool areVFParamsOk(const VFInfo &Info, ArrayRef<VPValue *> Args,
 static Function *findVectorVariant(CallInst *CI, ArrayRef<VPValue *> Args,
                                    ElementCount VF, bool MaskRequired,
                                    PredicatedScalarEvolution &PSE,
-                                   const Loop *L) {
+                                   const Loop *L,
+                                   const TargetTransformInfo &TTI) {
   if (CI->isNoBuiltin())
     return nullptr;
-  auto Mappings = VFDatabase::getMappings(*CI);
+  auto Mappings = VFDatabase::getMappings(*CI, &TTI);
   const auto *It = find_if(Mappings, [&](const VFInfo &Info) {
     return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
            areVFParamsOk(Info, Args, PSE, L);
@@ -5847,8 +5848,8 @@ static CallWideningDecision decideCallWidening(VPInstruction &VPI,
       VPReplicateRecipe::computeCallCost(CalledFn, ResultTy, Ops,
                                          /*IsSingleScalar=*/false, VF, CostCtx);
 
-  Function *VecFunc =
-      findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE, CostCtx.L);
+  Function *VecFunc = findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE,
+                                        CostCtx.L, CostCtx.TTI);
   InstructionCost VecCallCost = InstructionCost::getInvalid();
   if (VecFunc)
     VecCallCost = VPWidenCallRecipe::computeCallCost(VecFunc, CostCtx);
diff --git a/llvm/test/CodeGen/X86/veclib-llvm.sincos.ll b/llvm/test/CodeGen/X86/veclib-llvm.sincos.ll
index b7ce01cfe56228..f167c3bfb204b3 100644
--- a/llvm/test/CodeGen/X86/veclib-llvm.sincos.ll
+++ b/llvm/test/CodeGen/X86/veclib-llvm.sincos.ll
@@ -38,9 +38,33 @@ define void @test_sincos_v8f32(<8 x float> %x, ptr noalias %out_sin, ptr noalias
 ; AMD-AVX512-LABEL: test_sincos_v8f32:
 ; AMD-AVX512:    callq amd_vrs8_sincosf at PLT
 ;
-; GLIBC-LABEL: test_sincos_v8f32:
-; GLIBC:    callq _ZGVdN8v_sinf at PLT
-; GLIBC:    callq _ZGVdN8v_cosf at PLT
+; GLIBC-SSE-LABEL: test_sincos_v8f32:
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+; GLIBC-SSE:    callq sincosf at PLT
+;
+; GLIBC-AVX-LABEL: test_sincos_v8f32:
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+; GLIBC-AVX:    callq sincosf at PLT
+;
+; GLIBC-AVX2-LABEL: test_sincos_v8f32:
+; GLIBC-AVX2:    callq _ZGVdN8v_sinf at PLT
+; GLIBC-AVX2:    callq _ZGVdN8v_cosf at PLT
+;
+; GLIBC-AVX512-LABEL: test_sincos_v8f32:
+; GLIBC-AVX512:    callq _ZGVdN8v_sinf at PLT
+; GLIBC-AVX512:    callq _ZGVdN8v_cosf at PLT
   %result = call { <8 x float>, <8 x float> } @llvm.sincos.v8f32(<8 x float> %x)
   %result.0 = extractvalue { <8 x float>, <8 x float> } %result, 0
   %result.1 = extractvalue { <8 x float>, <8 x float> } %result, 1
@@ -124,9 +148,25 @@ define void @test_sincos_v4f64(<4 x double> %x, ptr noalias %out_sin, ptr noalia
 ; AMD-AVX512-LABEL: test_sincos_v4f64:
 ; AMD-AVX512:    callq amd_vrd4_sincos at PLT
 ;
-; GLIBC-LABEL: test_sincos_v4f64:
-; GLIBC:    callq _ZGVdN4v_sin at PLT
-; GLIBC:    callq _ZGVdN4v_cos at PLT
+; GLIBC-SSE-LABEL: test_sincos_v4f64:
+; GLIBC-SSE:    callq sincos at PLT
+; GLIBC-SSE:    callq sincos at PLT
+; GLIBC-SSE:    callq sincos at PLT
+; GLIBC-SSE:    callq sincos at PLT
+;
+; GLIBC-AVX-LABEL: test_sincos_v4f64:
+; GLIBC-AVX:    callq sincos at PLT
+; GLIBC-AVX:    callq sincos at PLT
+; GLIBC-AVX:    callq sincos at PLT
+; GLIBC-AVX:    callq sincos at PLT
+;
+; GLIBC-AVX2-LABEL: test_sincos_v4f64:
+; GLIBC-AVX2:    callq _ZGVdN4v_sin at PLT
+; GLIBC-AVX2:    callq _ZGVdN4v_cos at PLT
+;
+; GLIBC-AVX512-LABEL: test_sincos_v4f64:
+; GLIBC-AVX512:    callq _ZGVdN4v_sin at PLT
+; GLIBC-AVX512:    callq _ZGVdN4v_cos at PLT
   %result = call { <4 x double>, <4 x double> } @llvm.sincos.v4f64(<4 x double> %x)
   %result.0 = extractvalue { <4 x double>, <4 x double> } %result, 0
   %result.1 = extractvalue { <4 x double>, <4 x double> } %result, 1
@@ -173,7 +213,3 @@ define void @test_sincos_v8f64(<8 x double> %x, ptr noalias %out_sin, ptr noalia
 }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; CHECK: {{.*}}
-; GLIBC-AVX: {{.*}}
-; GLIBC-AVX2: {{.*}}
-; GLIBC-AVX512: {{.*}}
-; GLIBC-SSE: {{.*}}
diff --git a/llvm/test/Transforms/LoopVectorize/X86/amdlibm-calls-finite.ll b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-calls-finite.ll
index 985a2c693adeda..f570a48b84c213 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/amdlibm-calls-finite.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-calls-finite.ll
@@ -1,4 +1,4 @@
-; RUN: opt -vector-library=AMDLIBM -passes=inject-tli-mappings,loop-vectorize -S < %s | FileCheck %s
+; RUN: opt -mattr=+avx512f -vector-library=AMDLIBM -passes=inject-tli-mappings,loop-vectorize -S < %s | FileCheck %s
 
 ; Test to verify that when math headers are built with
 ; __FINITE_MATH_ONLY__ enabled, causing use of __<func>_finite
diff --git a/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
new file mode 100644
index 00000000000000..06559a540b705c
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
@@ -0,0 +1,75 @@
+; RUN: opt -passes=inject-tli-mappings,loop-vectorize,replace-with-veclib -vector-library=AMDLIBM -force-vector-width=8 -force-vector-interleave=1 -S %s | FileCheck %s
+; Eight floats occupy 256 bits, but a double-precision call at VF8 needs 512.
+; Check both the loop vectorizer and subsequent intrinsic replacement.
+target triple = "x86_64-unknown-linux-gnu"
+
+define void @avx2(ptr noalias %out, ptr noalias %in) #0 {
+; CHECK-LABEL: define void @avx2(
+; CHECK-NOT: call {{.*}}@amd_vrd8_log
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @llvm.log.f64(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @prefer256(ptr noalias %out, ptr noalias %in) #1 {
+; CHECK-LABEL: define void @prefer256(
+; CHECK-NOT: call {{.*}}@amd_vrd8_log
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @llvm.log.f64(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @avx512(ptr noalias %out, ptr noalias %in) #2 {
+; CHECK-LABEL: define void @avx512(
+; CHECK: call <8 x double> @amd_vrd8_log
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @llvm.log.f64(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+declare double @llvm.log.f64(double)
+attributes #0 = { "target-cpu"="haswell" }
+attributes #1 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #2 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
diff --git a/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls-finite.ll b/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls-finite.ll
index e495fa6676f6cf..ddf7748b69a675 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls-finite.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls-finite.ll
@@ -1,22 +1,22 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter "call.*@" --version 6
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=2 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF2
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF4
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=8 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF8
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=2 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF2
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF4
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=8 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF8
 target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
 target triple = "x86_64-unknown-linux-gnu"
 
 define void @exp_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @exp_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @__expf_finite(float [[TMP1:%.*]]) #[[ATTR0:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @__expf_finite(float [[TMP3:%.*]]) #[[ATTR0]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @__expf_finite(float [[TMP1:%.*]]) #[[ATTR1:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @__expf_finite(float [[TMP3:%.*]]) #[[ATTR1]]
 ;
 ; CHECK-VF4-LABEL: define void @exp_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v___expf_finite(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @exp_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v___expf_finite(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -39,23 +39,23 @@ for.end:
 
 define void @exp_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @exp_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v___exp_finite(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @exp_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v___exp_finite(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @exp_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @__exp_finite(double [[TMP1:%.*]]) #[[ATTR0:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @__exp_finite(double [[TMP3:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__exp_finite(double [[TMP5:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @__exp_finite(double [[TMP7:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @__exp_finite(double [[TMP9:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__exp_finite(double [[TMP11:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @__exp_finite(double [[TMP13:%.*]]) #[[ATTR0]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @__exp_finite(double [[TMP15:%.*]]) #[[ATTR0]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @__exp_finite(double [[TMP1:%.*]]) #[[ATTR1:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @__exp_finite(double [[TMP3:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__exp_finite(double [[TMP5:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @__exp_finite(double [[TMP7:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @__exp_finite(double [[TMP9:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__exp_finite(double [[TMP11:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @__exp_finite(double [[TMP13:%.*]]) #[[ATTR1]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @__exp_finite(double [[TMP15:%.*]]) #[[ATTR1]]
 ;
 entry:
   br label %for.body
@@ -77,16 +77,16 @@ for.end:
 
 define void @log_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @log_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @__logf_finite(float [[TMP1:%.*]]) #[[ATTR1:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @__logf_finite(float [[TMP3:%.*]]) #[[ATTR1]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @__logf_finite(float [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @__logf_finite(float [[TMP3:%.*]]) #[[ATTR2]]
 ;
 ; CHECK-VF4-LABEL: define void @log_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v___logf_finite(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @log_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v___logf_finite(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -109,23 +109,23 @@ for.end:
 
 define void @log_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @log_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v___log_finite(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @log_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v___log_finite(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @log_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @__log_finite(double [[TMP1:%.*]]) #[[ATTR1:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @__log_finite(double [[TMP3:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__log_finite(double [[TMP5:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @__log_finite(double [[TMP7:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @__log_finite(double [[TMP9:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__log_finite(double [[TMP11:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @__log_finite(double [[TMP13:%.*]]) #[[ATTR1]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @__log_finite(double [[TMP15:%.*]]) #[[ATTR1]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR0]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @__log_finite(double [[TMP1:%.*]]) #[[ATTR2:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @__log_finite(double [[TMP3:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__log_finite(double [[TMP5:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @__log_finite(double [[TMP7:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @__log_finite(double [[TMP9:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__log_finite(double [[TMP11:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @__log_finite(double [[TMP13:%.*]]) #[[ATTR2]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @__log_finite(double [[TMP15:%.*]]) #[[ATTR2]]
 ;
 entry:
   br label %for.body
@@ -148,20 +148,20 @@ for.end:
 
 define void @pow_f32(ptr nocapture %varray, ptr nocapture readonly %exp) {
 ; CHECK-VF2-LABEL: define void @pow_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
-; CHECK-VF2:    [[TMP6:%.*]] = tail call fast float @__powf_finite(float [[TMP4:%.*]], float [[TMP5:%.*]]) #[[ATTR2:[0-9]+]]
-; CHECK-VF2:    [[TMP9:%.*]] = tail call fast float @__powf_finite(float [[TMP7:%.*]], float [[TMP8:%.*]]) #[[ATTR2]]
-; CHECK-VF2:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR2]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
+; CHECK-VF2:    [[TMP6:%.*]] = tail call fast float @__powf_finite(float [[TMP4:%.*]], float [[TMP5:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-VF2:    [[TMP9:%.*]] = tail call fast float @__powf_finite(float [[TMP7:%.*]], float [[TMP8:%.*]]) #[[ATTR3]]
+; CHECK-VF2:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR3]]
 ;
 ; CHECK-VF4-LABEL: define void @pow_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
 ; CHECK-VF4:    [[TMP3:%.*]] = call fast <4 x float> @_ZGVbN4vv___powf_finite(<4 x float> [[TMP1:%.*]], <4 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF4:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR0:[0-9]+]]
+; CHECK-VF4:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR1:[0-9]+]]
 ;
 ; CHECK-VF8-LABEL: define void @pow_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
 ; CHECK-VF8:    [[TMP3:%.*]] = call fast <8 x float> @_ZGVdN8vv___powf_finite(<8 x float> [[TMP1:%.*]], <8 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF8:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR2:[0-9]+]]
+; CHECK-VF8:    [[I2:%.*]] = tail call fast float @__powf_finite(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR3:[0-9]+]]
 ;
 entry:
   br label %for.body
@@ -185,26 +185,26 @@ for.end:
 
 define void @pow_f64(ptr nocapture %varray, ptr nocapture readonly %exp) {
 ; CHECK-VF2-LABEL: define void @pow_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
 ; CHECK-VF2:    [[TMP3:%.*]] = call fast <2 x double> @_ZGVbN2vv___pow_finite(<2 x double> [[TMP1:%.*]], <2 x double> [[WIDE_LOAD:%.*]])
-; CHECK-VF2:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-VF2:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR4:[0-9]+]]
 ;
 ; CHECK-VF4-LABEL: define void @pow_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
 ; CHECK-VF4:    [[TMP3:%.*]] = call fast <4 x double> @_ZGVdN4vv___pow_finite(<4 x double> [[TMP1:%.*]], <4 x double> [[WIDE_LOAD:%.*]])
-; CHECK-VF4:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR1:[0-9]+]]
+; CHECK-VF4:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR2:[0-9]+]]
 ;
 ; CHECK-VF8-LABEL: define void @pow_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__pow_finite(double [[TMP4:%.*]], double [[TMP5:%.*]]) #[[ATTR3:[0-9]+]]
-; CHECK-VF8:    [[TMP9:%.*]] = tail call fast double @__pow_finite(double [[TMP7:%.*]], double [[TMP8:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__pow_finite(double [[TMP10:%.*]], double [[TMP11:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP15:%.*]] = tail call fast double @__pow_finite(double [[TMP13:%.*]], double [[TMP14:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP18:%.*]] = tail call fast double @__pow_finite(double [[TMP16:%.*]], double [[TMP17:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP21:%.*]] = tail call fast double @__pow_finite(double [[TMP19:%.*]], double [[TMP20:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP24:%.*]] = tail call fast double @__pow_finite(double [[TMP22:%.*]], double [[TMP23:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[TMP27:%.*]] = tail call fast double @__pow_finite(double [[TMP25:%.*]], double [[TMP26:%.*]]) #[[ATTR3]]
-; CHECK-VF8:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR3]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR0]] {
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @__pow_finite(double [[TMP4:%.*]], double [[TMP5:%.*]]) #[[ATTR4:[0-9]+]]
+; CHECK-VF8:    [[TMP9:%.*]] = tail call fast double @__pow_finite(double [[TMP7:%.*]], double [[TMP8:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @__pow_finite(double [[TMP10:%.*]], double [[TMP11:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP15:%.*]] = tail call fast double @__pow_finite(double [[TMP13:%.*]], double [[TMP14:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP18:%.*]] = tail call fast double @__pow_finite(double [[TMP16:%.*]], double [[TMP17:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP21:%.*]] = tail call fast double @__pow_finite(double [[TMP19:%.*]], double [[TMP20:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP24:%.*]] = tail call fast double @__pow_finite(double [[TMP22:%.*]], double [[TMP23:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[TMP27:%.*]] = tail call fast double @__pow_finite(double [[TMP25:%.*]], double [[TMP26:%.*]]) #[[ATTR4]]
+; CHECK-VF8:    [[I2:%.*]] = tail call fast double @__pow_finite(double [[CONV:%.*]], double [[I1:%.*]]) #[[ATTR4]]
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls.ll b/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls.ll
index 84cab4bc959017..9afa1d64d70480 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/libm-vector-calls.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --filter "call.*@" --version 6
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=2 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF2
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF4
-; RUN: opt -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=8 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF8
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=2 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF2
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF4
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -passes=inject-tli-mappings,loop-vectorize -force-vector-width=8 -force-vector-interleave=1 -S < %s | FileCheck %s --check-prefixes=CHECK-VF8
 
 target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
 target triple = "x86_64-unknown-linux-gnu"
@@ -38,15 +38,15 @@ declare double @atanh(double) #0
 
 define void @sin_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @sin_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1:[0-9]+]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_sin(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @sin_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1:[0-9]+]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_sin(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @sin_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1:[0-9]+]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.sin.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -69,15 +69,15 @@ for.end:
 
 define void @sin_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @sin_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.sin.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @sin_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_sinf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @sin_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_sinf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -100,15 +100,15 @@ for.end:
 
 define void @sin_f64_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @sin_f64_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_sin(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @sin_f64_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_sin(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @sin_f64_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.sin.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -131,15 +131,15 @@ for.end:
 
 define void @sin_f32_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @sin_f32_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.sin.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @sin_f32_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_sinf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @sin_f32_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_sinf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -162,15 +162,15 @@ for.end:
 
 define void @cos_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cos_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_cos(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @cos_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_cos(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cos_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.cos.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -193,15 +193,15 @@ for.end:
 
 define void @cos_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cos_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.cos.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @cos_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_cosf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cos_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_cosf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -224,15 +224,15 @@ for.end:
 
 define void @cos_f64_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cos_f64_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_cos(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @cos_f64_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_cos(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cos_f64_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.cos.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -255,15 +255,15 @@ for.end:
 
 define void @cos_f32_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cos_f32_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.cos.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @cos_f32_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_cosf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cos_f32_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_cosf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -286,15 +286,15 @@ for.end:
 
 define void @tan_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @tan_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_tan(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @tan_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_tan(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @tan_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.tan.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -317,15 +317,15 @@ for.end:
 
 define void @tan_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @tan_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.tan.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @tan_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_tanf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @tan_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_tanf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -348,15 +348,15 @@ for.end:
 
 define void @tan_f64_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @tan_f64_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x double> @_ZGVbN2v_tan(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @tan_f64_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_tan(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @tan_f64_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x double> @llvm.tan.v8f64(<8 x double> [[TMP0:%.*]])
 ;
 entry:
@@ -379,15 +379,15 @@ for.end:
 
 define void @tan_f32_intrinsic(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @tan_f32_intrinsic(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call <2 x float> @llvm.tan.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @tan_f32_intrinsic(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_tanf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @tan_f32_intrinsic(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call <8 x float> @_ZGVdN8v_tanf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -410,15 +410,15 @@ for.end:
 
 define void @exp_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @exp_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x float> @llvm.exp.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @exp_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_expf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @exp_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_expf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -441,15 +441,15 @@ for.end:
 
 define void @exp_f32_intrin(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @exp_f32_intrin(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x float> @llvm.exp.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @exp_f32_intrin(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_expf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @exp_f32_intrin(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_expf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -472,15 +472,15 @@ for.end:
 
 define void @log_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @log_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x float> @llvm.log.v2f32(<2 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @log_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_logf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @log_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_logf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -503,19 +503,19 @@ for.end:
 
 define void @pow_f32(ptr nocapture %varray, ptr nocapture readonly %exp) {
 ; CHECK-VF2-LABEL: define void @pow_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP4:%.*]] = call fast <2 x float> @llvm.pow.v2f32(<2 x float> [[TMP2:%.*]], <2 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF2:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-VF2:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR6:[0-9]+]]
 ;
 ; CHECK-VF4-LABEL: define void @pow_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP4:%.*]] = call fast <4 x float> @_ZGVbN4vv_powf(<4 x float> [[TMP2:%.*]], <4 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF4:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-VF4:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR4:[0-9]+]]
 ;
 ; CHECK-VF8-LABEL: define void @pow_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP4:%.*]] = call fast <8 x float> @_ZGVdN8vv_powf(<8 x float> [[TMP2:%.*]], <8 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF8:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-VF8:    [[I2:%.*]] = tail call fast float @powf(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR6:[0-9]+]]
 ;
 entry:
   br label %for.body
@@ -539,19 +539,19 @@ for.end:
 
 define void @pow_f32_intrin(ptr nocapture %varray, ptr nocapture readonly %exp) {
 ; CHECK-VF2-LABEL: define void @pow_f32_intrin(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP4:%.*]] = call fast <2 x float> @llvm.pow.v2f32(<2 x float> [[TMP2:%.*]], <2 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF2:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR4:[0-9]+]]
+; CHECK-VF2:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR7:[0-9]+]]
 ;
 ; CHECK-VF4-LABEL: define void @pow_f32_intrin(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP4:%.*]] = call fast <4 x float> @_ZGVbN4vv_powf(<4 x float> [[TMP2:%.*]], <4 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF4:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR4:[0-9]+]]
+; CHECK-VF4:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR5:[0-9]+]]
 ;
 ; CHECK-VF8-LABEL: define void @pow_f32_intrin(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]], ptr readonly captures(none) [[EXP:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP4:%.*]] = call fast <8 x float> @_ZGVdN8vv_powf(<8 x float> [[TMP2:%.*]], <8 x float> [[WIDE_LOAD:%.*]])
-; CHECK-VF8:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR4:[0-9]+]]
+; CHECK-VF8:    [[I2:%.*]] = tail call fast float @llvm.pow.f32(float [[CONV:%.*]], float [[I1:%.*]]) #[[ATTR7:[0-9]+]]
 ;
 entry:
   br label %for.body
@@ -575,16 +575,16 @@ for.end:
 
 define void @erf_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @erf_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @erff(float [[TMP1:%.*]]) #[[ATTR5:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @erff(float [[TMP3:%.*]]) #[[ATTR5]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @erff(float [[TMP1:%.*]]) #[[ATTR8:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @erff(float [[TMP3:%.*]]) #[[ATTR8]]
 ;
 ; CHECK-VF4-LABEL: define void @erf_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_erff(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @erf_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_erff(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -607,16 +607,16 @@ for.end:
 
 define void @erfc_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @erfc_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @erfcf(float [[TMP1:%.*]]) #[[ATTR6:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @erfcf(float [[TMP3:%.*]]) #[[ATTR6]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @erfcf(float [[TMP1:%.*]]) #[[ATTR9:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @erfcf(float [[TMP3:%.*]]) #[[ATTR9]]
 ;
 ; CHECK-VF4-LABEL: define void @erfc_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_erfcf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @erfc_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_erfcf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -639,16 +639,16 @@ for.end:
 
 define void @cbrt_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cbrt_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @cbrtf(float [[TMP1:%.*]]) #[[ATTR7:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @cbrtf(float [[TMP3:%.*]]) #[[ATTR7]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @cbrtf(float [[TMP1:%.*]]) #[[ATTR10:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @cbrtf(float [[TMP3:%.*]]) #[[ATTR10]]
 ;
 ; CHECK-VF4-LABEL: define void @cbrt_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_cbrtf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cbrt_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_cbrtf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -671,16 +671,16 @@ for.end:
 
 define void @expm1_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @expm1_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @expm1f(float [[TMP1:%.*]]) #[[ATTR8:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @expm1f(float [[TMP3:%.*]]) #[[ATTR8]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @expm1f(float [[TMP1:%.*]]) #[[ATTR11:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @expm1f(float [[TMP3:%.*]]) #[[ATTR11]]
 ;
 ; CHECK-VF4-LABEL: define void @expm1_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_expm1f(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @expm1_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_expm1f(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -703,16 +703,16 @@ for.end:
 
 define void @log1p_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @log1p_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @log1pf(float [[TMP1:%.*]]) #[[ATTR9:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @log1pf(float [[TMP3:%.*]]) #[[ATTR9]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @log1pf(float [[TMP1:%.*]]) #[[ATTR12:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @log1pf(float [[TMP3:%.*]]) #[[ATTR12]]
 ;
 ; CHECK-VF4-LABEL: define void @log1p_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_log1pf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @log1p_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_log1pf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -735,16 +735,16 @@ for.end:
 
 define void @asinh_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @asinh_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @asinhf(float [[TMP1:%.*]]) #[[ATTR10:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @asinhf(float [[TMP3:%.*]]) #[[ATTR10]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @asinhf(float [[TMP1:%.*]]) #[[ATTR13:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @asinhf(float [[TMP3:%.*]]) #[[ATTR13]]
 ;
 ; CHECK-VF4-LABEL: define void @asinh_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_asinhf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @asinh_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_asinhf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -767,16 +767,16 @@ for.end:
 
 define void @acosh_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @acosh_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @acoshf(float [[TMP1:%.*]]) #[[ATTR11:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @acoshf(float [[TMP3:%.*]]) #[[ATTR11]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @acoshf(float [[TMP1:%.*]]) #[[ATTR14:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @acoshf(float [[TMP3:%.*]]) #[[ATTR14]]
 ;
 ; CHECK-VF4-LABEL: define void @acosh_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_acoshf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @acosh_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_acoshf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -799,16 +799,16 @@ for.end:
 
 define void @atanh_f32(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @atanh_f32(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @atanhf(float [[TMP1:%.*]]) #[[ATTR12:[0-9]+]]
-; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @atanhf(float [[TMP3:%.*]]) #[[ATTR12]]
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF2:    [[TMP2:%.*]] = tail call fast float @atanhf(float [[TMP1:%.*]]) #[[ATTR15:[0-9]+]]
+; CHECK-VF2:    [[TMP4:%.*]] = tail call fast float @atanhf(float [[TMP3:%.*]]) #[[ATTR15]]
 ;
 ; CHECK-VF4-LABEL: define void @atanh_f32(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x float> @_ZGVbN4v_atanhf(<4 x float> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @atanh_f32(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF8:    [[TMP1:%.*]] = call fast <8 x float> @_ZGVdN8v_atanhf(<8 x float> [[TMP0:%.*]])
 ;
 entry:
@@ -831,23 +831,23 @@ for.end:
 
 define void @erf_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @erf_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_erf(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @erf_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_erf(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @erf_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @erf(double [[TMP1:%.*]]) #[[ATTR5:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @erf(double [[TMP3:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @erf(double [[TMP5:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @erf(double [[TMP7:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @erf(double [[TMP9:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @erf(double [[TMP11:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @erf(double [[TMP13:%.*]]) #[[ATTR5]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @erf(double [[TMP15:%.*]]) #[[ATTR5]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @erf(double [[TMP1:%.*]]) #[[ATTR8:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @erf(double [[TMP3:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @erf(double [[TMP5:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @erf(double [[TMP7:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @erf(double [[TMP9:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @erf(double [[TMP11:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @erf(double [[TMP13:%.*]]) #[[ATTR8]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @erf(double [[TMP15:%.*]]) #[[ATTR8]]
 ;
 entry:
   br label %for.body
@@ -869,23 +869,23 @@ for.end:
 
 define void @erfc_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @erfc_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_erfc(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @erfc_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_erfc(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @erfc_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @erfc(double [[TMP1:%.*]]) #[[ATTR6:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @erfc(double [[TMP3:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @erfc(double [[TMP5:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @erfc(double [[TMP7:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @erfc(double [[TMP9:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @erfc(double [[TMP11:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @erfc(double [[TMP13:%.*]]) #[[ATTR6]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @erfc(double [[TMP15:%.*]]) #[[ATTR6]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @erfc(double [[TMP1:%.*]]) #[[ATTR9:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @erfc(double [[TMP3:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @erfc(double [[TMP5:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @erfc(double [[TMP7:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @erfc(double [[TMP9:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @erfc(double [[TMP11:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @erfc(double [[TMP13:%.*]]) #[[ATTR9]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @erfc(double [[TMP15:%.*]]) #[[ATTR9]]
 ;
 entry:
   br label %for.body
@@ -907,23 +907,23 @@ for.end:
 
 define void @cbrt_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @cbrt_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_cbrt(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @cbrt_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_cbrt(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @cbrt_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @cbrt(double [[TMP1:%.*]]) #[[ATTR7:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @cbrt(double [[TMP3:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @cbrt(double [[TMP5:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @cbrt(double [[TMP7:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @cbrt(double [[TMP9:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @cbrt(double [[TMP11:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @cbrt(double [[TMP13:%.*]]) #[[ATTR7]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @cbrt(double [[TMP15:%.*]]) #[[ATTR7]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @cbrt(double [[TMP1:%.*]]) #[[ATTR10:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @cbrt(double [[TMP3:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @cbrt(double [[TMP5:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @cbrt(double [[TMP7:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @cbrt(double [[TMP9:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @cbrt(double [[TMP11:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @cbrt(double [[TMP13:%.*]]) #[[ATTR10]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @cbrt(double [[TMP15:%.*]]) #[[ATTR10]]
 ;
 entry:
   br label %for.body
@@ -945,23 +945,23 @@ for.end:
 
 define void @expm1_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @expm1_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_expm1(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @expm1_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_expm1(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @expm1_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @expm1(double [[TMP1:%.*]]) #[[ATTR8:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @expm1(double [[TMP3:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @expm1(double [[TMP5:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @expm1(double [[TMP7:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @expm1(double [[TMP9:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @expm1(double [[TMP11:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @expm1(double [[TMP13:%.*]]) #[[ATTR8]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @expm1(double [[TMP15:%.*]]) #[[ATTR8]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @expm1(double [[TMP1:%.*]]) #[[ATTR11:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @expm1(double [[TMP3:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @expm1(double [[TMP5:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @expm1(double [[TMP7:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @expm1(double [[TMP9:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @expm1(double [[TMP11:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @expm1(double [[TMP13:%.*]]) #[[ATTR11]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @expm1(double [[TMP15:%.*]]) #[[ATTR11]]
 ;
 entry:
   br label %for.body
@@ -983,23 +983,23 @@ for.end:
 
 define void @log1p_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @log1p_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_log1p(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @log1p_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_log1p(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @log1p_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @log1p(double [[TMP1:%.*]]) #[[ATTR9:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @log1p(double [[TMP3:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @log1p(double [[TMP5:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @log1p(double [[TMP7:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @log1p(double [[TMP9:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @log1p(double [[TMP11:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @log1p(double [[TMP13:%.*]]) #[[ATTR9]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @log1p(double [[TMP15:%.*]]) #[[ATTR9]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @log1p(double [[TMP1:%.*]]) #[[ATTR12:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @log1p(double [[TMP3:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @log1p(double [[TMP5:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @log1p(double [[TMP7:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @log1p(double [[TMP9:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @log1p(double [[TMP11:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @log1p(double [[TMP13:%.*]]) #[[ATTR12]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @log1p(double [[TMP15:%.*]]) #[[ATTR12]]
 ;
 entry:
   br label %for.body
@@ -1021,23 +1021,23 @@ for.end:
 
 define void @asinh_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @asinh_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_asinh(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @asinh_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_asinh(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @asinh_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @asinh(double [[TMP1:%.*]]) #[[ATTR10:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @asinh(double [[TMP3:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @asinh(double [[TMP5:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @asinh(double [[TMP7:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @asinh(double [[TMP9:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @asinh(double [[TMP11:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @asinh(double [[TMP13:%.*]]) #[[ATTR10]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @asinh(double [[TMP15:%.*]]) #[[ATTR10]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @asinh(double [[TMP1:%.*]]) #[[ATTR13:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @asinh(double [[TMP3:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @asinh(double [[TMP5:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @asinh(double [[TMP7:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @asinh(double [[TMP9:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @asinh(double [[TMP11:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @asinh(double [[TMP13:%.*]]) #[[ATTR13]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @asinh(double [[TMP15:%.*]]) #[[ATTR13]]
 ;
 entry:
   br label %for.body
@@ -1059,23 +1059,23 @@ for.end:
 
 define void @acosh_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @acosh_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_acosh(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @acosh_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_acosh(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @acosh_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @acosh(double [[TMP1:%.*]]) #[[ATTR11:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @acosh(double [[TMP3:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @acosh(double [[TMP5:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @acosh(double [[TMP7:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @acosh(double [[TMP9:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @acosh(double [[TMP11:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @acosh(double [[TMP13:%.*]]) #[[ATTR11]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @acosh(double [[TMP15:%.*]]) #[[ATTR11]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @acosh(double [[TMP1:%.*]]) #[[ATTR14:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @acosh(double [[TMP3:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @acosh(double [[TMP5:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @acosh(double [[TMP7:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @acosh(double [[TMP9:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @acosh(double [[TMP11:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @acosh(double [[TMP13:%.*]]) #[[ATTR14]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @acosh(double [[TMP15:%.*]]) #[[ATTR14]]
 ;
 entry:
   br label %for.body
@@ -1097,23 +1097,23 @@ for.end:
 
 define void @atanh_f64(ptr nocapture %varray) {
 ; CHECK-VF2-LABEL: define void @atanh_f64(
-; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF2-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF2:    [[TMP1:%.*]] = call fast <2 x double> @_ZGVbN2v_atanh(<2 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF4-LABEL: define void @atanh_f64(
-; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) {
+; CHECK-VF4-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
 ; CHECK-VF4:    [[TMP1:%.*]] = call fast <4 x double> @_ZGVdN4v_atanh(<4 x double> [[TMP0:%.*]])
 ;
 ; CHECK-VF8-LABEL: define void @atanh_f64(
-; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) {
-; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @atanh(double [[TMP1:%.*]]) #[[ATTR12:[0-9]+]]
-; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @atanh(double [[TMP3:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @atanh(double [[TMP5:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @atanh(double [[TMP7:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @atanh(double [[TMP9:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @atanh(double [[TMP11:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @atanh(double [[TMP13:%.*]]) #[[ATTR12]]
-; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @atanh(double [[TMP15:%.*]]) #[[ATTR12]]
+; CHECK-VF8-SAME: ptr captures(none) [[VARRAY:%.*]]) #[[ATTR1]] {
+; CHECK-VF8:    [[TMP2:%.*]] = tail call fast double @atanh(double [[TMP1:%.*]]) #[[ATTR15:[0-9]+]]
+; CHECK-VF8:    [[TMP4:%.*]] = tail call fast double @atanh(double [[TMP3:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP6:%.*]] = tail call fast double @atanh(double [[TMP5:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP8:%.*]] = tail call fast double @atanh(double [[TMP7:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP10:%.*]] = tail call fast double @atanh(double [[TMP9:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP12:%.*]] = tail call fast double @atanh(double [[TMP11:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP14:%.*]] = tail call fast double @atanh(double [[TMP13:%.*]]) #[[ATTR15]]
+; CHECK-VF8:    [[TMP16:%.*]] = tail call fast double @atanh(double [[TMP15:%.*]]) #[[ATTR15]]
 ;
 entry:
   br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll b/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
new file mode 100644
index 00000000000000..16d7d6a238c3d0
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
@@ -0,0 +1,82 @@
+; RUN: opt -passes=inject-tli-mappings,loop-vectorize -vector-library=LIBMVEC -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+; The loop's elements are floats, but each library call operates on doubles.
+target triple = "x86_64-unknown-linux-gnu"
+
+define void @generic(ptr noalias %out, ptr noalias %in) #0 {
+; CHECK-LABEL: define void @generic(
+; CHECK-NOT: call {{.*}}@_ZGVd
+; CHECK: call double @erf
+; CHECK-NOT: call {{.*}}@_ZGVd
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @erf(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @avx(ptr noalias %out, ptr noalias %in) #1 {
+; CHECK-LABEL: define void @avx(
+; CHECK-NOT: call {{.*}}@_ZGVd
+; CHECK: call double @erf
+; CHECK-NOT: call {{.*}}@_ZGVd
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @erf(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @avx2(ptr noalias %out, ptr noalias %in) #2 {
+; CHECK-LABEL: define void @avx2(
+; CHECK: call <4 x double> @_ZGVdN4v_erf
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @erf(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+declare double @erf(double) nounwind memory(none)
+attributes #0 = { "target-cpu"="x86-64" }
+attributes #1 = { "target-cpu"="sandybridge" }
+attributes #2 = { "target-cpu"="haswell" }
+attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
+attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
+attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
diff --git a/llvm/test/Transforms/LoopVectorize/X86/svml-calls-finite.ll b/llvm/test/Transforms/LoopVectorize/X86/svml-calls-finite.ll
index a768a38da55930..cf3931a7b95159 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/svml-calls-finite.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/svml-calls-finite.ll
@@ -1,4 +1,4 @@
-; RUN: opt -vector-library=SVML -passes=inject-tli-mappings,loop-vectorize -S < %s | FileCheck %s
+; RUN: opt -mattr=+avx2 -vector-library=SVML -passes=inject-tli-mappings,loop-vectorize -S < %s | FileCheck %s
 
 ; Test to verify that when math headers are built with
 ; __FINITE_MATH_ONLY__ enabled, causing use of __<func>_finite
diff --git a/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll b/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
new file mode 100644
index 00000000000000..7968e2c7c8647c
--- /dev/null
+++ b/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
@@ -0,0 +1,44 @@
+; RUN: opt -passes=replace-with-veclib -vector-library=AMDLIBM -S %s | FileCheck %s
+; A 512-bit external call must not have its operands split by legalization.
+target triple = "x86_64-unknown-linux-gnu"
+
+define <8 x double> @generic(<8 x double> %x) #0 {
+; CHECK-LABEL: define <8 x double> @generic(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
+define <8 x double> @avx2(<8 x double> %x) #2 {
+; CHECK-LABEL: define <8 x double> @avx2(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
+define <8 x double> @prefer256(<8 x double> %x) #3 {
+; CHECK-LABEL: define <8 x double> @prefer256(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
+define <8 x double> @avx512(<8 x double> %x) #4 {
+; CHECK-LABEL: define <8 x double> @avx512(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @amd_vrd8_log(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
+declare <8 x double> @llvm.log.v8f64(<8 x double>)
+attributes #0 = { "target-cpu"="x86-64" }
+attributes #1 = { "target-cpu"="sandybridge" }
+attributes #2 = { "target-cpu"="haswell" }
+attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
+attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
+attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
diff --git a/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll b/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
new file mode 100644
index 00000000000000..8689d627ebf8ad
--- /dev/null
+++ b/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
@@ -0,0 +1,61 @@
+; RUN: opt -passes=replace-with-veclib -vector-library=LIBMVEC -S %s | FileCheck %s
+; The same module contains callers with different target features. Matching a
+; vector width alone is insufficient: class d requires AVX2, not just AVX.
+target triple = "x86_64-unknown-linux-gnu"
+
+define <4 x double> @generic(<4 x double> %x, <4 x double> %y) #0 {
+; CHECK-LABEL: define <4 x double> @generic(
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: ret <4 x double> [[R]]
+  %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+  ret <4 x double> %r
+}
+
+define <4 x double> @avx(<4 x double> %x, <4 x double> %y) #1 {
+; CHECK-LABEL: define <4 x double> @avx(
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: ret <4 x double> [[R]]
+  %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+  ret <4 x double> %r
+}
+
+define <4 x double> @avx2(<4 x double> %x, <4 x double> %y) #2 {
+; CHECK-LABEL: define <4 x double> @avx2(
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @_ZGVdN4vv_pow(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: ret <4 x double> [[R]]
+  %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+  ret <4 x double> %r
+}
+
+define <4 x double> @disabled_avx2(<4 x double> %x, <4 x double> %y) #5 {
+; CHECK-LABEL: define <4 x double> @disabled_avx2(
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: ret <4 x double> [[R]]
+  %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+  ret <4 x double> %r
+}
+
+define <4 x double> @prefer128(<4 x double> %x, <4 x double> %y) #6 {
+; CHECK-LABEL: define <4 x double> @prefer128(
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: ret <4 x double> [[R]]
+  %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+  ret <4 x double> %r
+}
+
+define <2 x double> @sse2(<2 x double> %x, <2 x double> %y) #0 {
+; CHECK-LABEL: define <2 x double> @sse2(
+; CHECK-NEXT: [[R:%.*]] = call fast <2 x double> @_ZGVbN2vv_pow(<2 x double> %x, <2 x double> %y)
+; CHECK-NEXT: ret <2 x double> [[R]]
+  %r = call fast <2 x double> @llvm.pow.v2f64(<2 x double> %x, <2 x double> %y)
+  ret <2 x double> %r
+}
+declare <2 x double> @llvm.pow.v2f64(<2 x double>, <2 x double>)
+declare <4 x double> @llvm.pow.v4f64(<4 x double>, <4 x double>)
+attributes #0 = { "target-cpu"="x86-64" }
+attributes #1 = { "target-cpu"="sandybridge" }
+attributes #2 = { "target-cpu"="haswell" }
+attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
+attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
+attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
diff --git a/llvm/test/Transforms/ReplaceWithVeclib/X86/lit.local.cfg b/llvm/test/Transforms/ReplaceWithVeclib/X86/lit.local.cfg
new file mode 100644
index 00000000000000..42bf50dcc13c35
--- /dev/null
+++ b/llvm/test/Transforms/ReplaceWithVeclib/X86/lit.local.cfg
@@ -0,0 +1,2 @@
+if not "X86" in config.root.targets:
+    config.unsupported = True
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll b/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll
new file mode 100644
index 00000000000000..ebaf7b0d15c8b7
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll
@@ -0,0 +1,66 @@
+; RUN: opt -passes=inject-tli-mappings,slp-vectorizer -vector-library=LIBMVEC -slp-threshold=-1000 -S %s | FileCheck %s
+; An AVX register can hold the arguments, but the d ABI variant requires AVX2.
+target triple = "x86_64-unknown-linux-gnu"
+
+define void @avx(ptr noalias %out, ptr noalias %in) #1 {
+; CHECK-LABEL: define void @avx(
+; CHECK-NOT: call {{.*}}@_ZGVd
+; CHECK: ret void
+  %p0 = getelementptr double, ptr %in, i64 0
+  %x0 = load double, ptr %p0, align 8
+  %r0 = call fast double @erf(double %x0)
+  %p1 = getelementptr double, ptr %in, i64 1
+  %x1 = load double, ptr %p1, align 8
+  %r1 = call fast double @erf(double %x1)
+  %p2 = getelementptr double, ptr %in, i64 2
+  %x2 = load double, ptr %p2, align 8
+  %r2 = call fast double @erf(double %x2)
+  %p3 = getelementptr double, ptr %in, i64 3
+  %x3 = load double, ptr %p3, align 8
+  %r3 = call fast double @erf(double %x3)
+  %q0 = getelementptr double, ptr %out, i64 0
+  store double %r0, ptr %q0, align 8
+  %q1 = getelementptr double, ptr %out, i64 1
+  store double %r1, ptr %q1, align 8
+  %q2 = getelementptr double, ptr %out, i64 2
+  store double %r2, ptr %q2, align 8
+  %q3 = getelementptr double, ptr %out, i64 3
+  store double %r3, ptr %q3, align 8
+  ret void
+}
+
+define void @avx2(ptr noalias %out, ptr noalias %in) #2 {
+; CHECK-LABEL: define void @avx2(
+; CHECK: call fast <4 x double> @_ZGVdN4v_erf
+; CHECK: ret void
+  %p0 = getelementptr double, ptr %in, i64 0
+  %x0 = load double, ptr %p0, align 8
+  %r0 = call fast double @erf(double %x0)
+  %p1 = getelementptr double, ptr %in, i64 1
+  %x1 = load double, ptr %p1, align 8
+  %r1 = call fast double @erf(double %x1)
+  %p2 = getelementptr double, ptr %in, i64 2
+  %x2 = load double, ptr %p2, align 8
+  %r2 = call fast double @erf(double %x2)
+  %p3 = getelementptr double, ptr %in, i64 3
+  %x3 = load double, ptr %p3, align 8
+  %r3 = call fast double @erf(double %x3)
+  %q0 = getelementptr double, ptr %out, i64 0
+  store double %r0, ptr %q0, align 8
+  %q1 = getelementptr double, ptr %out, i64 1
+  store double %r1, ptr %q1, align 8
+  %q2 = getelementptr double, ptr %out, i64 2
+  store double %r2, ptr %q2, align 8
+  %q3 = getelementptr double, ptr %out, i64 3
+  store double %r3, ptr %q3, align 8
+  ret void
+}
+
+declare double @erf(double) nounwind willreturn memory(none)
+attributes #0 = { "target-cpu"="x86-64" }
+attributes #1 = { "target-cpu"="sandybridge" }
+attributes #2 = { "target-cpu"="haswell" }
+attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
+attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
+attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }

>From 88ecc45d1f5c21aee3b05c5478d40ad5b8bfff00 Mon Sep 17 00:00:00 2001
From: Tonu Samuel <tonu at spam.ee>
Date: Fri, 25 Sep 2026 09:59:25 +0300
Subject: [PATCH 2/2] [X86] Preserve vector ABI requirements and enabled
 register widths

Keep VFABI ISA information through redirected symbols and reject GNU
variants whose multiple-register ABI is not represented by the IR signature.
Use the registers enabled for codegen rather than the preferred vector width,
so ABI-safe ZMM calls remain available when min-legal-vector-width allows them.

Add positive and negative coverage for redirected and overwide variants and
AVX-512 minimum legal widths. Enable AVX2 and regenerate the generic mapping
test that failed CI. Remove unused test attributes.

Assisted-by: OpenAI Codex
Assisted-by: Claude Opus
Assisted-by: Kimi K3
---
 .../llvm/Analysis/TargetTransformInfo.h       |   6 +-
 .../llvm/Analysis/TargetTransformInfoImpl.h   |   2 +-
 llvm/lib/Analysis/TargetTransformInfo.cpp     |   6 +-
 llvm/lib/Analysis/VectorUtils.cpp             |   3 +-
 llvm/lib/CodeGen/ReplaceWithVeclib.cpp        |   3 +-
 .../lib/Target/X86/X86TargetTransformInfo.cpp |  40 +++++--
 llvm/lib/Target/X86/X86TargetTransformInfo.h  |   2 +-
 .../Generic/replace-intrinsics-with-veclib.ll |  36 +++---
 .../X86/amdlibm-target-features.ll            |  49 +++++++-
 .../X86/libmvec-redirect-target-features.ll   | 108 ++++++++++++++++++
 .../X86/libmvec-target-features.ll            |   4 -
 .../X86/amdlibm-target-features.ll            |  24 +++-
 .../X86/libmvec-target-features.ll            |   4 +-
 .../X86/libmvec-target-features.ll            |   5 -
 14 files changed, 235 insertions(+), 57 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/libmvec-redirect-target-features.ll

diff --git a/llvm/include/llvm/Analysis/TargetTransformInfo.h b/llvm/include/llvm/Analysis/TargetTransformInfo.h
index 1cbc17aa074616..40cf84f15f2083 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfo.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfo.h
@@ -72,6 +72,7 @@ class TargetLibraryInfo;
 class Type;
 class VPIntrinsic;
 struct KnownBits;
+struct VFInfo;
 
 /// Information about a load/store intrinsic defined by the target.
 struct MemIntrinsicInfo {
@@ -1041,9 +1042,10 @@ class TargetTransformInfo {
   /// Whether a vector function can be called directly from this function.
   /// Unlike an intrinsic, an external vector call cannot be legalized by
   /// splitting its operands without changing the callee's ABI. Targets may
-  /// also impose ISA requirements encoded in the vector function's name.
+  /// also impose ISA requirements from the VFABI mapping or symbol name.
+  /// Targets may conservatively reject calls wider than their preferred width.
   LLVM_ABI bool isLegalToCallVectorFunction(FunctionType *FTy,
-                                            StringRef Name) const;
+                                            const VFInfo &Info) const;
 
   /// Returns the estimated number of registers required to represent \p Ty.
   LLVM_ABI unsigned getRegUsageForType(Type *Ty) const;
diff --git a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
index 82879eedc013e8..3d15dadd8e9555 100644
--- a/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
+++ b/llvm/include/llvm/Analysis/TargetTransformInfoImpl.h
@@ -470,7 +470,7 @@ class LLVM_ABI TargetTransformInfoImplBase {
   virtual bool isTypeLegal(Type *Ty) const { return false; }
 
   virtual bool isLegalToCallVectorFunction(FunctionType *FTy,
-                                           StringRef Name) const {
+                                           const VFInfo &Info) const {
     return true;
   }
 
diff --git a/llvm/lib/Analysis/TargetTransformInfo.cpp b/llvm/lib/Analysis/TargetTransformInfo.cpp
index 3dde0e3f3df3d3..5b1cd420adcb5c 100644
--- a/llvm/lib/Analysis/TargetTransformInfo.cpp
+++ b/llvm/lib/Analysis/TargetTransformInfo.cpp
@@ -601,9 +601,9 @@ bool TargetTransformInfo::isProfitableToHoist(Instruction *I) const {
 
 bool TargetTransformInfo::useAA() const { return TTIImpl->useAA(); }
 
-bool TargetTransformInfo::isLegalToCallVectorFunction(FunctionType *FTy,
-                                                      StringRef Name) const {
-  return TTIImpl->isLegalToCallVectorFunction(FTy, Name);
+bool TargetTransformInfo::isLegalToCallVectorFunction(
+    FunctionType *FTy, const VFInfo &Info) const {
+  return TTIImpl->isLegalToCallVectorFunction(FTy, Info);
 }
 
 bool TargetTransformInfo::isTypeLegal(Type *Ty) const {
diff --git a/llvm/lib/Analysis/VectorUtils.cpp b/llvm/lib/Analysis/VectorUtils.cpp
index 7649b729de5050..019e144be64cbf 100644
--- a/llvm/lib/Analysis/VectorUtils.cpp
+++ b/llvm/lib/Analysis/VectorUtils.cpp
@@ -40,8 +40,7 @@ SmallVector<VFInfo, 8> VFDatabase::getMappings(const CallInst &CI,
   if (TTI)
     llvm::erase_if(Mappings, [&](const VFInfo &Info) {
       const Function *VF = CI.getModule()->getFunction(Info.VectorName);
-      return !TTI->isLegalToCallVectorFunction(VF->getFunctionType(),
-                                               VF->getName());
+      return !TTI->isLegalToCallVectorFunction(VF->getFunctionType(), Info);
     });
   return Mappings;
 }
diff --git a/llvm/lib/CodeGen/ReplaceWithVeclib.cpp b/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
index f7242b5c1e9306..92168983042e2b 100644
--- a/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
+++ b/llvm/lib/CodeGen/ReplaceWithVeclib.cpp
@@ -199,8 +199,7 @@ static bool replaceWithCallToVeclib(const TargetLibraryInfo &TLI,
   }
 
   FunctionType *VectorFTy = VFABI::createFunctionType(*OptInfo, ScalarFTy);
-  if (!VectorFTy ||
-      !TTI.isLegalToCallVectorFunction(VectorFTy, VD->getVectorFnName()))
+  if (!VectorFTy || !TTI.isLegalToCallVectorFunction(VectorFTy, *OptInfo))
     return false;
 
   Function *TLIFunc =
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
index d2e122f9bc4d9f..f62f980f9b9a29 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.cpp
@@ -57,6 +57,7 @@
 #include "llvm/CodeGen/TargetLowering.h"
 #include "llvm/IR/InstIterator.h"
 #include "llvm/IR/IntrinsicInst.h"
+#include "llvm/IR/VFABIDemangler.h"
 #include <optional>
 
 using namespace llvm;
@@ -222,21 +223,38 @@ static bool hasSupportedVectorTypes(Type *Ty, const DataLayout &DL,
 }
 
 bool X86TTIImpl::isLegalToCallVectorFunction(FunctionType *FTy,
-                                             StringRef Name) const {
-  // The x86 Vector Function ABI encodes the required ISA in the symbol.
-  // In particular, a 256-bit AVX caller cannot use the AVX2 ('d') variant,
-  // even though its vector arguments fit in a YMM register.
-  if ((Name.starts_with("_ZGVb") && !ST->hasSSE2()) ||
-      (Name.starts_with("_ZGVc") && !ST->hasAVX()) ||
-      (Name.starts_with("_ZGVd") && !ST->hasAVX2()) ||
-      (Name.starts_with("_ZGVe") && !ST->hasAVX512()))
+                                             const VFInfo &Info) const {
+  // Preserve the ISA from the mapping even when the symbol is redirected.
+  // TLI also uses LLVM-internal mappings with GNU ABI symbol names, so check
+  // both sources. AVX and AVX2 variants have the same vector width but require
+  // different instruction sets.
+  StringRef Name = Info.VectorName;
+  if (((Info.ISA == VFISAKind::SSE || Name.starts_with("_ZGVb")) &&
+       !ST->hasSSE2()) ||
+      ((Info.ISA == VFISAKind::AVX || Name.starts_with("_ZGVc")) &&
+       !ST->hasAVX()) ||
+      ((Info.ISA == VFISAKind::AVX2 || Name.starts_with("_ZGVd")) &&
+       !ST->hasAVX2()) ||
+      ((Info.ISA == VFISAKind::AVX512 || Name.starts_with("_ZGVe")) &&
+       !ST->hasAVX512()))
     return false;
 
   // Check the actual call signature, not the vectorized loop's element type.
   // A float loop can contain a double-precision call with twice its width.
-  // Conservatively avoid calls wider than the target's preferred vector width:
-  // codegen must not split the operands of an already selected external call.
-  TypeSize MaxWidth = getRegisterBitWidth(TTI::RGK_FixedWidthVector);
+  // Use the registers enabled for legalization, not just the preferred width.
+  // In particular, min-legal-vector-width can enable ZMM arguments even when
+  // the caller prefers 256-bit vectors.
+  TypeSize MaxWidth = TypeSize::getFixed(ST->useAVX512Regs() ? 512
+                                         : ST->hasAVX()      ? 256
+                                         : ST->hasSSE1()     ? 128
+                                                             : 0);
+  // Overwide GNU variants use multiple registers of their ABI class (and
+  // may return indirectly). A single wider IR vector does not model that ABI.
+  if (Info.ISA == VFISAKind::SSE || Name.starts_with("_ZGVb"))
+    MaxWidth = std::min(MaxWidth, TypeSize::getFixed(128));
+  else if (Info.ISA == VFISAKind::AVX || Info.ISA == VFISAKind::AVX2 ||
+           Name.starts_with("_ZGVc") || Name.starts_with("_ZGVd"))
+    MaxWidth = std::min(MaxWidth, TypeSize::getFixed(256));
   return hasSupportedVectorTypes(FTy->getReturnType(), DL, MaxWidth) &&
          all_of(FTy->params(), [&](Type *Ty) {
            return hasSupportedVectorTypes(Ty, DL, MaxWidth);
diff --git a/llvm/lib/Target/X86/X86TargetTransformInfo.h b/llvm/lib/Target/X86/X86TargetTransformInfo.h
index 32033700fc1106..bc528513dcd7e1 100644
--- a/llvm/lib/Target/X86/X86TargetTransformInfo.h
+++ b/llvm/lib/Target/X86/X86TargetTransformInfo.h
@@ -59,7 +59,7 @@ class X86TTIImpl final : public BasicTTIImplBase<X86TTIImpl> {
   /// @{
 
   bool isLegalToCallVectorFunction(FunctionType *FTy,
-                                   StringRef Name) const override;
+                                   const VFInfo &Info) const override;
   unsigned getNumberOfRegisters(unsigned ClassID) const override;
   unsigned getRegisterClassForType(bool Vector, Type *Ty) const override;
   bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const override;
diff --git a/llvm/test/CodeGen/Generic/replace-intrinsics-with-veclib.ll b/llvm/test/CodeGen/Generic/replace-intrinsics-with-veclib.ll
index ff9c7486c099e3..add1549aff67d4 100644
--- a/llvm/test/CodeGen/Generic/replace-intrinsics-with-veclib.ll
+++ b/llvm/test/CodeGen/Generic/replace-intrinsics-with-veclib.ll
@@ -1,36 +1,36 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --function-signature --check-attributes
-; RUN: opt -vector-library=SVML -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,SVML
-; RUN: opt -vector-library=AMDLIBM -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,AMDLIBM
-; RUN: opt -vector-library=LIBMVEC -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,LIBMVEC-X86
-; RUN: opt -vector-library=MASSV -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,MASSV
-; RUN: opt -vector-library=Accelerate -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,ACCELERATE
+; RUN: opt -mattr=+avx2 -vector-library=SVML -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,SVML
+; RUN: opt -mattr=+avx2 -vector-library=AMDLIBM -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,AMDLIBM
+; RUN: opt -mattr=+avx2 -vector-library=LIBMVEC -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,LIBMVEC-X86
+; RUN: opt -mattr=+avx2 -vector-library=MASSV -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,MASSV
+; RUN: opt -mattr=+avx2 -vector-library=Accelerate -replace-with-veclib -S < %s | FileCheck %s  --check-prefixes=COMMON,ACCELERATE
 
 target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
 target triple = "x86_64-unknown-linux-gnu"
 
 define <4 x double> @exp_v4(<4 x double> %in) {
 ; SVML-LABEL: define {{[^@]+}}@exp_v4
-; SVML-SAME: (<4 x double> [[IN:%.*]]) {
+; SVML-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; SVML-NEXT:    [[TMP1:%.*]] = call <4 x double> @__svml_exp4(<4 x double> [[IN]])
 ; SVML-NEXT:    ret <4 x double> [[TMP1]]
 ;
 ; AMDLIBM-LABEL: define {{[^@]+}}@exp_v4
-; AMDLIBM-SAME: (<4 x double> [[IN:%.*]]) {
+; AMDLIBM-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; AMDLIBM-NEXT:    [[TMP1:%.*]] = call <4 x double> @amd_vrd4_exp(<4 x double> [[IN]])
 ; AMDLIBM-NEXT:    ret <4 x double> [[TMP1]]
 ;
 ; LIBMVEC-X86-LABEL: define {{[^@]+}}@exp_v4
-; LIBMVEC-X86-SAME: (<4 x double> [[IN:%.*]]) {
+; LIBMVEC-X86-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; LIBMVEC-X86-NEXT:    [[TMP1:%.*]] = call <4 x double> @_ZGVdN4v_exp(<4 x double> [[IN]])
 ; LIBMVEC-X86-NEXT:    ret <4 x double> [[TMP1]]
 ;
 ; MASSV-LABEL: define {{[^@]+}}@exp_v4
-; MASSV-SAME: (<4 x double> [[IN:%.*]]) {
+; MASSV-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; MASSV-NEXT:    [[CALL:%.*]] = call <4 x double> @llvm.exp.v4f64(<4 x double> [[IN]])
 ; MASSV-NEXT:    ret <4 x double> [[CALL]]
 ;
 ; ACCELERATE-LABEL: define {{[^@]+}}@exp_v4
-; ACCELERATE-SAME: (<4 x double> [[IN:%.*]]) {
+; ACCELERATE-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; ACCELERATE-NEXT:    [[CALL:%.*]] = call <4 x double> @llvm.exp.v4f64(<4 x double> [[IN]])
 ; ACCELERATE-NEXT:    ret <4 x double> [[CALL]]
 ;
@@ -42,27 +42,27 @@ declare <4 x double> @llvm.exp.v4f64(<4 x double>) #0
 
 define <4 x float> @exp_f32(<4 x float> %in) {
 ; SVML-LABEL: define {{[^@]+}}@exp_f32
-; SVML-SAME: (<4 x float> [[IN:%.*]]) {
+; SVML-SAME: (<4 x float> [[IN:%.*]]) #[[ATTR0]] {
 ; SVML-NEXT:    [[TMP1:%.*]] = call <4 x float> @__svml_expf4(<4 x float> [[IN]])
 ; SVML-NEXT:    ret <4 x float> [[TMP1]]
 ;
 ; AMDLIBM-LABEL: define {{[^@]+}}@exp_f32
-; AMDLIBM-SAME: (<4 x float> [[IN:%.*]]) {
+; AMDLIBM-SAME: (<4 x float> [[IN:%.*]]) #[[ATTR0]] {
 ; AMDLIBM-NEXT:    [[TMP1:%.*]] = call <4 x float> @amd_vrs4_expf(<4 x float> [[IN]])
 ; AMDLIBM-NEXT:    ret <4 x float> [[TMP1]]
 ;
 ; LIBMVEC-X86-LABEL: define {{[^@]+}}@exp_f32
-; LIBMVEC-X86-SAME: (<4 x float> [[IN:%.*]]) {
+; LIBMVEC-X86-SAME: (<4 x float> [[IN:%.*]]) #[[ATTR0]] {
 ; LIBMVEC-X86-NEXT:    [[TMP1:%.*]] = call <4 x float> @_ZGVbN4v_expf(<4 x float> [[IN]])
 ; LIBMVEC-X86-NEXT:    ret <4 x float> [[TMP1]]
 ;
 ; MASSV-LABEL: define {{[^@]+}}@exp_f32
-; MASSV-SAME: (<4 x float> [[IN:%.*]]) {
+; MASSV-SAME: (<4 x float> [[IN:%.*]]) #[[ATTR0]] {
 ; MASSV-NEXT:    [[TMP1:%.*]] = call <4 x float> @__expf4(<4 x float> [[IN]])
 ; MASSV-NEXT:    ret <4 x float> [[TMP1]]
 ;
 ; ACCELERATE-LABEL: define {{[^@]+}}@exp_f32
-; ACCELERATE-SAME: (<4 x float> [[IN:%.*]]) {
+; ACCELERATE-SAME: (<4 x float> [[IN:%.*]]) #[[ATTR0]] {
 ; ACCELERATE-NEXT:    [[TMP1:%.*]] = call <4 x float> @vexpf(<4 x float> [[IN]])
 ; ACCELERATE-NEXT:    ret <4 x float> [[TMP1]]
 ;
@@ -75,7 +75,7 @@ declare <4 x float> @llvm.exp.v4f32(<4 x float>) #0
 ; No replacement should take place for non-vector intrinsic.
 define double @exp_f64(double %in) {
 ; COMMON-LABEL: define {{[^@]+}}@exp_f64
-; COMMON-SAME: (double [[IN:%.*]]) {
+; COMMON-SAME: (double [[IN:%.*]]) #[[ATTR0:[0-9]+]] {
 ; COMMON-NEXT:    [[CALL:%.*]] = call double @llvm.exp.f64(double [[IN]])
 ; COMMON-NEXT:    ret double [[CALL]]
 ;
@@ -89,7 +89,7 @@ declare double @llvm.exp.f64(double) #0
 ; vector intrinsics. No vector library has a substitute for powi.
 define <4 x double> @powi_v4(<4 x double> %in){
 ; COMMON-LABEL: define {{[^@]+}}@powi_v4
-; COMMON-SAME: (<4 x double> [[IN:%.*]]) {
+; COMMON-SAME: (<4 x double> [[IN:%.*]]) #[[ATTR0]] {
 ; COMMON-NEXT:    [[CALL:%.*]] = call <4 x double> @llvm.powi.v4f64.i32(<4 x double> [[IN]], i32 3)
 ; COMMON-NEXT:    ret <4 x double> [[CALL]]
 ;
@@ -103,7 +103,7 @@ declare <4 x double> @llvm.powi.v4f64.i32(<4 x double>, i32) #0
 ; does not match exactly.
 define <3 x double> @exp_v3(<3 x double> %in) {
 ; COMMON-LABEL: define {{[^@]+}}@exp_v3
-; COMMON-SAME: (<3 x double> [[IN:%.*]]) {
+; COMMON-SAME: (<3 x double> [[IN:%.*]]) #[[ATTR0]] {
 ; COMMON-NEXT:    [[CALL:%.*]] = call <3 x double> @llvm.exp.v3f64(<3 x double> [[IN]])
 ; COMMON-NEXT:    ret <3 x double> [[CALL]]
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
index 06559a540b705c..8c1a0cbd581a53 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/amdlibm-target-features.ll
@@ -69,7 +69,54 @@ exit:
   ret void
 }
 
+define void @prefer256_no_min(ptr noalias %out, ptr noalias %in) #7 {
+; CHECK-LABEL: define void @prefer256_no_min(
+; CHECK: call <8 x double> @amd_vrd8_log
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @llvm.log.f64(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @prefer256_min512(ptr noalias %out, ptr noalias %in) #8 {
+; CHECK-LABEL: define void @prefer256_min512(
+; CHECK: call <8 x double> @amd_vrd8_log
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @llvm.log.f64(double %d)
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
 declare double @llvm.log.f64(double)
 attributes #0 = { "target-cpu"="haswell" }
-attributes #1 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #1 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" "min-legal-vector-width"="256" }
 attributes #2 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
+
+attributes #7 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #8 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" "min-legal-vector-width"="512" }
diff --git a/llvm/test/Transforms/LoopVectorize/X86/libmvec-redirect-target-features.ll b/llvm/test/Transforms/LoopVectorize/X86/libmvec-redirect-target-features.ll
new file mode 100644
index 00000000000000..21e6b01dd55025
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/libmvec-redirect-target-features.ll
@@ -0,0 +1,108 @@
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+target triple = "x86_64-unknown-linux-gnu"
+
+define void @generic(ptr noalias %out, ptr noalias %in) #0 {
+; CHECK-LABEL: define void @generic(
+; CHECK-NOT: call {{.*}}@custom_avx2
+; CHECK: call double @foo
+; CHECK-NOT: call {{.*}}@custom_avx2
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @foo(double %d) #9
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @avx(ptr noalias %out, ptr noalias %in) #1 {
+; CHECK-LABEL: define void @avx(
+; CHECK-NOT: call {{.*}}@custom_avx2
+; CHECK: call double @foo
+; CHECK-NOT: call {{.*}}@custom_avx2
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @foo(double %d) #9
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @avx2(ptr noalias %out, ptr noalias %in) #2 {
+; CHECK-LABEL: define void @avx2(
+; CHECK: call <4 x double> @custom_avx2
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @foo(double %d) #9
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+define void @overwide_sse(ptr noalias %out, ptr noalias %in) #2 {
+; CHECK-LABEL: define void @overwide_sse(
+; CHECK-NOT: call {{.*}}@custom_sse
+; CHECK: call double @foo
+; CHECK-NOT: call {{.*}}@custom_sse
+; CHECK: ret void
+entry:
+  br label %loop
+loop:
+  %i = phi i64 [ 0, %entry ], [ %next, %loop ]
+  %p = getelementptr float, ptr %in, i64 %i
+  %x = load float, ptr %p, align 4
+  %d = fpext float %x to double
+  %r = call double @foo(double %d) #10
+  %f = fptrunc double %r to float
+  %q = getelementptr float, ptr %out, i64 %i
+  store float %f, ptr %q, align 4
+  %next = add nuw nsw i64 %i, 1
+  %done = icmp eq i64 %next, 64
+  br i1 %done, label %exit, label %loop
+exit:
+  ret void
+}
+
+declare double @foo(double) nounwind memory(none)
+declare <4 x double> @custom_avx2(<4 x double>)
+attributes #0 = { "target-cpu"="x86-64" }
+attributes #1 = { "target-cpu"="sandybridge" }
+attributes #9 = { "vector-function-abi-variant"="_ZGVdN4v_foo(custom_avx2)" }
+
+attributes #2 = { "target-cpu"="haswell" }
+
+declare <4 x double> @custom_sse(<4 x double>)
+attributes #10 = { "vector-function-abi-variant"="_ZGVbN4v_foo(custom_sse)" }
diff --git a/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll b/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
index 16d7d6a238c3d0..b43a8cabda2902 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/libmvec-target-features.ll
@@ -76,7 +76,3 @@ declare double @erf(double) nounwind memory(none)
 attributes #0 = { "target-cpu"="x86-64" }
 attributes #1 = { "target-cpu"="sandybridge" }
 attributes #2 = { "target-cpu"="haswell" }
-attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
-attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
-attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
-attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
diff --git a/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll b/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
index 7968e2c7c8647c..e430d159ff25c9 100644
--- a/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
+++ b/llvm/test/Transforms/ReplaceWithVeclib/X86/amdlibm-target-features.ll
@@ -34,11 +34,27 @@ define <8 x double> @avx512(<8 x double> %x) #4 {
   ret <8 x double> %r
 }
 
+define <8 x double> @prefer256_no_min(<8 x double> %x) #7 {
+; CHECK-LABEL: define <8 x double> @prefer256_no_min(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @amd_vrd8_log(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
+define <8 x double> @prefer256_min512(<8 x double> %x) #8 {
+; CHECK-LABEL: define <8 x double> @prefer256_min512(
+; CHECK-NEXT: [[R:%.*]] = call fast <8 x double> @amd_vrd8_log(<8 x double> %x)
+; CHECK-NEXT: ret <8 x double> [[R]]
+  %r = call fast <8 x double> @llvm.log.v8f64(<8 x double> %x)
+  ret <8 x double> %r
+}
+
 declare <8 x double> @llvm.log.v8f64(<8 x double>)
 attributes #0 = { "target-cpu"="x86-64" }
-attributes #1 = { "target-cpu"="sandybridge" }
 attributes #2 = { "target-cpu"="haswell" }
-attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" "min-legal-vector-width"="256" }
 attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
-attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
-attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
+
+attributes #7 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
+attributes #8 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" "min-legal-vector-width"="512" }
diff --git a/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll b/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
index 8689d627ebf8ad..f268f1d0ed0b99 100644
--- a/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
+++ b/llvm/test/Transforms/ReplaceWithVeclib/X86/libmvec-target-features.ll
@@ -37,7 +37,7 @@ define <4 x double> @disabled_avx2(<4 x double> %x, <4 x double> %y) #5 {
 
 define <4 x double> @prefer128(<4 x double> %x, <4 x double> %y) #6 {
 ; CHECK-LABEL: define <4 x double> @prefer128(
-; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
+; CHECK-NEXT: [[R:%.*]] = call fast <4 x double> @_ZGVdN4vv_pow(<4 x double> %x, <4 x double> %y)
 ; CHECK-NEXT: ret <4 x double> [[R]]
   %r = call fast <4 x double> @llvm.pow.v4f64(<4 x double> %x, <4 x double> %y)
   ret <4 x double> %r
@@ -55,7 +55,5 @@ declare <4 x double> @llvm.pow.v4f64(<4 x double>, <4 x double>)
 attributes #0 = { "target-cpu"="x86-64" }
 attributes #1 = { "target-cpu"="sandybridge" }
 attributes #2 = { "target-cpu"="haswell" }
-attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
-attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
 attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
 attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll b/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll
index ebaf7b0d15c8b7..d28a3b15003813 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/libmvec-target-features.ll
@@ -57,10 +57,5 @@ define void @avx2(ptr noalias %out, ptr noalias %in) #2 {
 }
 
 declare double @erf(double) nounwind willreturn memory(none)
-attributes #0 = { "target-cpu"="x86-64" }
 attributes #1 = { "target-cpu"="sandybridge" }
 attributes #2 = { "target-cpu"="haswell" }
-attributes #3 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="256" }
-attributes #4 = { "target-cpu"="skylake-avx512" "prefer-vector-width"="512" }
-attributes #5 = { "target-cpu"="haswell" "target-features"="-avx2" }
-attributes #6 = { "target-cpu"="haswell" "prefer-vector-width"="128" }



More information about the llvm-commits mailing list