[llvm] [hexagon] Add `hexagon_hmx` function attribute. (PR #222340)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 07:25:02 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-hexagon

Author: Weiwei C (weiweichen)

<details>
<summary>Changes</summary>

- [x] Add `hexagon_hmx` function attribute to be used to annotate hmx functions.
- [x] Add logic to prevent hmx functions being inlined into hvx functions to avoid unintended accidental vectorization. 

---
Full diff: https://github.com/llvm/llvm-project/pull/222340.diff


4 Files Affected:

- (modified) llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.cpp (+13-1) 
- (modified) llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.h (+11-2) 
- (added) llvm/test/CodeGen/Hexagon/hmx-attr-inline-compat.ll (+41) 
- (added) llvm/test/CodeGen/Hexagon/hmx-attr-no-autohvx.ll (+59) 


``````````diff
diff --git a/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.cpp b/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.cpp
index 60449635aecbd..f5295d5fd6ff3 100644
--- a/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.cpp
+++ b/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.cpp
@@ -52,7 +52,7 @@ static cl::opt<bool> HexagonMaskedVMem("hexagon-masked-vmem", cl::init(true),
 static const unsigned FloatFactor = 4;
 
 bool HexagonTTIImpl::useHVX() const {
-  return ST.useHVXOps() && HexagonAutoHVX;
+  return ST.useHVXOps() && HexagonAutoHVX && !IsHMX;
 }
 
 bool HexagonTTIImpl::isHVXVectorType(Type *Ty) const {
@@ -455,3 +455,15 @@ HexagonTTIImpl::getInstructionCost(const User *U,
 bool HexagonTTIImpl::shouldBuildLookupTables() const {
   return EmitLookupTables;
 }
+
+bool HexagonTTIImpl::areInlineCompatible(const Function *Caller,
+                                         const Function *Callee) const {
+  // Inlining across the boundary would defeat the attribute in both
+  // directions: HVX code moved into an HMX function can take a vector unit
+  // the HMX thread needs, and an HMX body moved into a non-HMX function
+  // becomes a candidate for auto-vectorization again.
+  if (Caller->hasFnAttribute("hexagon_hmx") !=
+      Callee->hasFnAttribute("hexagon_hmx"))
+    return false;
+  return BaseT::areInlineCompatible(Caller, Callee);
+}
diff --git a/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.h b/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.h
index 7835c4b7e22ee..c912e706ef806 100644
--- a/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.h
+++ b/llvm/lib/Target/Hexagon/HexagonTargetTransformInfo.h
@@ -38,6 +38,11 @@ class HexagonTTIImpl final : public BasicTTIImplBase<HexagonTTIImpl> {
 
   const HexagonSubtarget &ST;
   const HexagonTargetLowering &TLI;
+  // Functions running on the HMX (matrix) unit must not acquire an HVX
+  // context: the two share a limited pool of vector units, so an HVX
+  // instruction reaching the HMX thread can block behind a barrier the
+  // HVX threads are themselves waiting on.
+  const bool IsHMX;
 
   const HexagonSubtarget *getST() const { return &ST; }
   const HexagonTargetLowering *getTLI() const { return &TLI; }
@@ -52,8 +57,9 @@ class HexagonTTIImpl final : public BasicTTIImplBase<HexagonTTIImpl> {
 
 public:
   explicit HexagonTTIImpl(const HexagonTargetMachine *TM, const Function &F)
-      : BaseT(TM, F.getDataLayout()),
-        ST(*TM->getSubtargetImpl(F)), TLI(*ST.getTargetLowering()) {}
+      : BaseT(TM, F.getDataLayout()), ST(*TM->getSubtargetImpl(F)),
+        TLI(*ST.getTargetLowering()),
+        IsHMX(F.hasFnAttribute("hexagon_hmx")) {}
 
   /// \name Scalar TTI Implementations
   /// @{
@@ -187,6 +193,9 @@ class HexagonTTIImpl final : public BasicTTIImplBase<HexagonTTIImpl> {
 
   // Hexagon specific decision to generate a lookup table.
   bool shouldBuildLookupTables() const override;
+
+  bool areInlineCompatible(const Function *Caller,
+                           const Function *Callee) const override;
 };
 
 } // end namespace llvm
diff --git a/llvm/test/CodeGen/Hexagon/hmx-attr-inline-compat.ll b/llvm/test/CodeGen/Hexagon/hmx-attr-inline-compat.ll
new file mode 100644
index 0000000000000..4723758f24dc2
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/hmx-attr-inline-compat.ll
@@ -0,0 +1,41 @@
+; Inlining must not mix HMX and non-HMX bodies in either direction, and the
+; check has to hold for alwaysinline callees, which is why this is enforced in
+; TTI rather than through an Attributes.td CompatRule.
+
+; RUN: opt -mtriple=hexagon -passes=always-inline -S < %s | FileCheck %s
+
+define internal void @hvx_helper(ptr %p) alwaysinline #1 {
+  store i32 1, ptr %p, align 4
+  ret void
+}
+
+define internal void @hmx_helper(ptr %p) alwaysinline #0 {
+  store i32 2, ptr %p, align 4
+  ret void
+}
+
+; CHECK-LABEL: define void @hmx_caller(
+; CHECK:         call void @hvx_helper(
+define void @hmx_caller(ptr %p) #0 {
+  call void @hvx_helper(ptr %p)
+  ret void
+}
+
+; CHECK-LABEL: define void @hvx_caller(
+; CHECK:         call void @hmx_helper(
+define void @hvx_caller(ptr %p) #1 {
+  call void @hmx_helper(ptr %p)
+  ret void
+}
+
+; Matching attributes still inline, so the checks above are not passing because
+; alwaysinline is broken.
+; CHECK-LABEL: define void @hmx_caller_same(
+; CHECK-NOT:     call void @hmx_helper(
+define void @hmx_caller_same(ptr %p) #0 {
+  call void @hmx_helper(ptr %p)
+  ret void
+}
+
+attributes #0 = { nounwind "hexagon_hmx" "target-features"="+hvxv68,+hvx-length128b" }
+attributes #1 = { nounwind "target-features"="+hvxv68,+hvx-length128b" }
diff --git a/llvm/test/CodeGen/Hexagon/hmx-attr-no-autohvx.ll b/llvm/test/CodeGen/Hexagon/hmx-attr-no-autohvx.ll
new file mode 100644
index 0000000000000..f5c5389806729
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/hmx-attr-no-autohvx.ll
@@ -0,0 +1,59 @@
+; A "hexagon_hmx" function must stay scalar even with the HVX loop vectorizer
+; enabled, so no HVX unit is acquired on the HMX thread.
+
+; RUN: opt -mtriple=hexagon -mattr=+hvxv68,+hvx-length128b -hexagon-autohvx \
+; RUN:     -passes=loop-vectorize -S < %s | FileCheck %s
+
+define void @hmx_stays_scalar(ptr %dst, ptr %a, ptr %b, i32 %n) #0 {
+; CHECK-LABEL: @hmx_stays_scalar(
+; CHECK-NOT:     <32 x i32>
+; CHECK-NOT:     <64 x i32>
+entry:
+  %cmp = icmp sgt i32 %n, 0
+  br i1 %cmp, label %loop, label %exit
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %pa = getelementptr inbounds i32, ptr %a, i32 %iv
+  %pb = getelementptr inbounds i32, ptr %b, i32 %iv
+  %pd = getelementptr inbounds i32, ptr %dst, i32 %iv
+  %va = load i32, ptr %pa, align 4
+  %vb = load i32, ptr %pb, align 4
+  %sum = add nsw i32 %va, %vb
+  store i32 %sum, ptr %pd, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %done = icmp eq i32 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; Same loop without the attribute is vectorized, so the check above is not
+; passing for an unrelated reason.
+define void @no_attr_vectorizes(ptr %dst, ptr %a, ptr %b, i32 %n) #1 {
+; CHECK-LABEL: @no_attr_vectorizes(
+; CHECK:         <32 x i32>
+entry:
+  %cmp = icmp sgt i32 %n, 0
+  br i1 %cmp, label %loop, label %exit
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %pa = getelementptr inbounds i32, ptr %a, i32 %iv
+  %pb = getelementptr inbounds i32, ptr %b, i32 %iv
+  %pd = getelementptr inbounds i32, ptr %dst, i32 %iv
+  %va = load i32, ptr %pa, align 4
+  %vb = load i32, ptr %pb, align 4
+  %sum = add nsw i32 %va, %vb
+  store i32 %sum, ptr %pd, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %done = icmp eq i32 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+attributes #0 = { nounwind "hexagon_hmx" "target-features"="+hvxv68,+hvx-length128b" }
+attributes #1 = { nounwind "target-features"="+hvxv68,+hvx-length128b" }

``````````

</details>


https://github.com/llvm/llvm-project/pull/222340


More information about the llvm-commits mailing list