[llvm] 2866d23 - [AArch64][Bitcode] Use target memory for SME state (#205829)

via llvm-commits llvm-commits at lists.llvm.org
Mon Jul 6 01:24:27 PDT 2026


Author: CarolineConcatto
Date: 2026-07-06T09:24:22+01:00
New Revision: 2866d2333c477dead9a37248316e8d172e3b612a

URL: https://github.com/llvm/llvm-project/commit/2866d2333c477dead9a37248316e8d172e3b612a
DIFF: https://github.com/llvm/llvm-project/commit/2866d2333c477dead9a37248316e8d172e3b612a.diff

LOG: [AArch64][Bitcode] Use target memory for SME state (#205829)

Model AArch64 ZA and ZT0 intrinsic state using target_mem instead of
inaccessiblemem.

Bump the bitcode memory-attribute encoding and upgrade old AArch64
bitcode so prior inaccessiblemem effects are preserved on the new target
memory locations. Non-AArch64 bitcode keeps the old interpretation.

Added: 
    llvm/test/Bitcode/Inputs/aarch64-memory-attribute-upgrade.bc
    llvm/test/Bitcode/Inputs/x86-memory-attribute-upgrade.bc
    llvm/test/Bitcode/target-memory-attribute.ll
    llvm/test/Transforms/LICM/AArch64/sme-fp8-hoist.ll

Modified: 
    llvm/docs/LangRef.rst
    llvm/include/llvm/IR/IntrinsicsAArch64.td
    llvm/lib/Bitcode/Reader/BitcodeReader.cpp
    llvm/lib/Bitcode/Writer/BitcodeWriter.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/docs/LangRef.rst b/llvm/docs/LangRef.rst
index 5ce6019660e0b..bac98bcbb2fad 100644
--- a/llvm/docs/LangRef.rst
+++ b/llvm/docs/LangRef.rst
@@ -2302,8 +2302,9 @@ For example:
     - ``errnomem``: This refers to accesses to the ``errno`` variable.
     - ``target_mem#`` : These refer to target specific state that cannot be
       accessed by any other means. # is a number between 0 and 1 inclusive.
-      Note: The target_mem locations are experimental and intended for internal
-      testing only. They must not be used in production code.
+      Note: The following target_mem locations are implemented in AArch64.
+      target_mem0 represents SME ZT0 state, target_mem1 represents SME ZA
+      state.
 
     - The default access kind (specified without a location prefix) applies to
       all locations that haven't been specified explicitly, including those that

diff  --git a/llvm/include/llvm/IR/IntrinsicsAArch64.td b/llvm/include/llvm/IR/IntrinsicsAArch64.td
index 5ba1f4ba861d2..6078e1438ceee 100644
--- a/llvm/include/llvm/IR/IntrinsicsAArch64.td
+++ b/llvm/include/llvm/IR/IntrinsicsAArch64.td
@@ -734,8 +734,8 @@ def int_aarch64_neon_tbx4 : AdvSIMD_Tbx4_Intrinsic;
 
 // Maps Memory locations to registers.
 defvar FPMR = InaccessibleMem;
-defvar ZT0 = InaccessibleMem;
-defvar ZA = InaccessibleMem;
+defvar ZT0 = TargetMem0;
+defvar ZA = TargetMem1;
 
 let TargetPrefix = "aarch64" in {
   class FPENV_Get_Intrinsic

diff  --git a/llvm/lib/Bitcode/Reader/BitcodeReader.cpp b/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
index 2bd251efd05ce..7bd569db09a76 100644
--- a/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
+++ b/llvm/lib/Bitcode/Reader/BitcodeReader.cpp
@@ -581,6 +581,7 @@ class BitcodeConstant final : public Value,
 class BitcodeReader : public BitcodeReaderBase, public GVMaterializer {
   LLVMContext &Context;
   Module *TheModule = nullptr;
+  Triple BitcodeTargetTriple;
   // Next offset to start scanning for lazy parsing of function bodies.
   uint64_t NextUnreadBit = 0;
   // Last function offset found in the VST.
@@ -704,7 +705,8 @@ class BitcodeReader : public BitcodeReaderBase, public GVMaterializer {
 
 public:
   BitcodeReader(BitstreamCursor Stream, StringRef Strtab,
-                StringRef ProducerIdentification, LLVMContext &Context);
+                StringRef ProducerIdentification, LLVMContext &Context,
+                Triple BitcodeTargetTriple);
 
   Error materializeForwardReferencedFunctions();
 
@@ -1062,8 +1064,9 @@ std::error_code llvm::errorToErrorCodeAndEmitErrors(LLVMContext &Ctx,
 
 BitcodeReader::BitcodeReader(BitstreamCursor Stream, StringRef Strtab,
                              StringRef ProducerIdentification,
-                             LLVMContext &Context)
+                             LLVMContext &Context, Triple TTriple)
     : BitcodeReaderBase(std::move(Stream), Strtab), Context(Context),
+      BitcodeTargetTriple(TTriple),
       ValueList(this->Stream.SizeInBytes(),
                 [this](unsigned ValID, BasicBlock *InsertBB) {
                   return materializeValue(ValID, InsertBB);
@@ -2470,12 +2473,29 @@ Error BitcodeReader::parseAttributeGroupBlock() {
                         MemoryEffects::argMemOnly(ArgMem) |
                         MemoryEffects::errnoMemOnly(OtherMem) |
                         MemoryEffects::otherMemOnly(OtherMem);
+              // Old versions dont have target memory location.
+              // It was represented as Inaccessible memory for AArch64.
+              if (BitcodeTargetTriple.isAArch64())
+                ME = ME.getWithModRef(IRMemLocation::TargetMem0,
+                                      InaccessibleMem) |
+                     ME.getWithModRef(IRMemLocation::TargetMem1,
+                                      InaccessibleMem);
               B.addMemoryAttr(ME);
             } else {
               // Construct the memory attribute directly from the encoded base
               // on newer versions.
-              B.addMemoryAttr(MemoryEffects::createFromIntValue(
-                  EncodedME & 0x00FFFFFFFFFFFFFFULL));
+              auto ME = MemoryEffects::createFromIntValue(
+                  EncodedME & 0x00FFFFFFFFFFFFFFULL);
+              // Only from Version=2 onwards target memory location exist.
+              // It was represented as Inaccessible memory for AArch64.
+              if (Version == 1 && BitcodeTargetTriple.isAArch64())
+                ME = ME.getWithModRef(
+                         IRMemLocation::TargetMem0,
+                         ME.getModRef(IRMemLocation::InaccessibleMem)) |
+                     ME.getWithModRef(
+                         IRMemLocation::TargetMem1,
+                         ME.getModRef(IRMemLocation::InaccessibleMem));
+              B.addMemoryAttr(ME);
             }
           } else if (Kind == Attribute::Captures)
             B.addCapturesAttr(CaptureInfo::createFromIntValue(Record[++i]));
@@ -8701,10 +8721,20 @@ BitcodeModule::getModuleImpl(LLVMContext &Context, bool MaterializeAll,
       return std::move(E);
   }
 
+  // Cache target triple early for target-memory attribute upgrading.
+  // Suppress target parser diagnostics during this early parse,
+  // because attribute parsing runs before target parsing.
+  Triple BitcodeTargetTriple;
+  BitstreamCursor TripleStream(Buffer);
+  if (Expected<std::string> TripleStr = readTriple(TripleStream))
+    BitcodeTargetTriple = Triple(*TripleStr);
+  else
+    consumeError(TripleStr.takeError());
+
   if (Error JumpFailed = Stream.JumpToBit(ModuleBit))
     return std::move(JumpFailed);
   auto *R = new BitcodeReader(std::move(Stream), Strtab, ProducerIdentification,
-                              Context);
+                              Context, BitcodeTargetTriple);
 
   std::unique_ptr<Module> M =
       std::make_unique<Module>(ModuleIdentifier, Context);

diff  --git a/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp b/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
index 4aa203b0d12d1..12b508c51cbfc 100644
--- a/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
+++ b/llvm/lib/Bitcode/Writer/BitcodeWriter.cpp
@@ -1083,7 +1083,7 @@ void ModuleBitcodeWriter::writeAttributeGroupTable() {
         Record.push_back(getAttrKindEncoding(Kind));
         if (Kind == Attribute::Memory) {
           // Version field for upgrading old memory effects.
-          const uint64_t Version = 1;
+          const uint64_t Version = 2;
           Record.push_back((Version << 56) | Attr.getValueAsInt());
         } else {
           Record.push_back(Attr.getValueAsInt());

diff  --git a/llvm/test/Bitcode/Inputs/aarch64-memory-attribute-upgrade.bc b/llvm/test/Bitcode/Inputs/aarch64-memory-attribute-upgrade.bc
new file mode 100644
index 0000000000000..0df2f355e8f91
Binary files /dev/null and b/llvm/test/Bitcode/Inputs/aarch64-memory-attribute-upgrade.bc 
diff er

diff  --git a/llvm/test/Bitcode/Inputs/x86-memory-attribute-upgrade.bc b/llvm/test/Bitcode/Inputs/x86-memory-attribute-upgrade.bc
new file mode 100644
index 0000000000000..a0b2e2ac64af5
Binary files /dev/null and b/llvm/test/Bitcode/Inputs/x86-memory-attribute-upgrade.bc 
diff er

diff  --git a/llvm/test/Bitcode/target-memory-attribute.ll b/llvm/test/Bitcode/target-memory-attribute.ll
new file mode 100644
index 0000000000000..0349f56aca534
--- /dev/null
+++ b/llvm/test/Bitcode/target-memory-attribute.ll
@@ -0,0 +1,35 @@
+; RUN: llvm-dis < %S/Inputs/aarch64-memory-attribute-upgrade.bc | FileCheck %s --check-prefix=AARCH64
+; RUN: llvm-dis < %S/Inputs/x86-memory-attribute-upgrade.bc | FileCheck %s --check-prefix=NON-AARCH64
+
+; The .bc inputs were generated by an older LLVM memory-attribute encoding,
+; before AArch64 target memory was split out from inaccessible memory.
+; The code was compiled using x86 and aarch64 hosts.
+
+
+define void @test_inaccessible_read() #0 {
+; AARCH64: ; Function Attrs: memory(inaccessiblemem: read, target_mem: read)
+; AARCH64-NEXT: define void @test_inaccessible_read()
+; NON-AARCH64: ; Function Attrs: memory(inaccessiblemem: read)
+; NON-AARCH64-NEXT: define void @test_inaccessible_read()
+  ret void
+}
+
+define void @test_inaccessible_readwrite() #1 {
+; AARCH64: ; Function Attrs: memory(inaccessiblemem: readwrite, target_mem: readwrite)
+; AARCH64-NEXT: define void @test_inaccessible_readwrite()
+; NON-AARCH64: ; Function Attrs: memory(inaccessiblemem: readwrite)
+; NON-AARCH64-NEXT: define void @test_inaccessible_readwrite()
+  ret void
+}
+
+define void @test_inaccessible_none() #2 {
+; AARCH64: ; Function Attrs: memory(none)
+; AARCH64-NEXT: define void @test_inaccessible_none()
+; NON-AARCH64: ; Function Attrs: memory(none)
+; NON-AARCH64-NEXT: define void @test_inaccessible_none()
+  ret void
+}
+
+attributes #0 = { memory(inaccessiblemem: read) }
+attributes #1 = { memory(inaccessiblemem: readwrite) }
+attributes #2 = { memory(inaccessiblemem: none) }

diff  --git a/llvm/test/Transforms/LICM/AArch64/sme-fp8-hoist.ll b/llvm/test/Transforms/LICM/AArch64/sme-fp8-hoist.ll
new file mode 100644
index 0000000000000..1efb6688e0208
--- /dev/null
+++ b/llvm/test/Transforms/LICM/AArch64/sme-fp8-hoist.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=licm < %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+;; The set.fprm should be hoisted once ZA does not use the same
+;; memory location as FPMR.
+define void @fp8_fmopa_loop(ptr %lhs, ptr %rhs, i64 %fpmr, i64 %n) {
+; CHECK-LABEL: define void @fp8_fmopa_loop(
+; CHECK-SAME: ptr [[LHS:%.*]], ptr [[RHS:%.*]], i64 [[FPMR:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[PTRUE:%.*]] = call <vscale x 16 x i1> @llvm.aarch64.sve.ptrue.nxv16i1(i32 31)
+; CHECK-NEXT:    [[LHS_LOAD:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.ld1.nxv16i8.p0(<vscale x 16 x i1> [[PTRUE]], ptr [[LHS]])
+; CHECK-NEXT:    [[RHS_LOAD:%.*]] = call <vscale x 16 x i8> @llvm.aarch64.sve.ld1.nxv16i8.p0(<vscale x 16 x i1> [[PTRUE]], ptr [[RHS]])
+; CHECK-NEXT:    call void @llvm.aarch64.set.fpmr(i64 [[FPMR]])
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    call void @llvm.aarch64.sme.fp8.fmopa.za32(i32 0, <vscale x 16 x i1> [[PTRUE]], <vscale x 16 x i1> [[PTRUE]], <vscale x 16 x i8> [[LHS_LOAD]], <vscale x 16 x i8> [[RHS_LOAD]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[COND:%.*]] = icmp ult i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[COND]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %ptrue = call <vscale x 16 x i1> @llvm.aarch64.sve.ptrue.nxv16i1(i32 31)
+  %lhs.load = call <vscale x 16 x i8> @llvm.aarch64.sve.ld1.nxv16i8(<vscale x 16 x i1> %ptrue, ptr %lhs)
+  %rhs.load = call <vscale x 16 x i8> @llvm.aarch64.sve.ld1.nxv16i8(<vscale x 16 x i1> %ptrue, ptr %rhs)
+  call void @llvm.aarch64.set.fpmr(i64 %fpmr)
+  call void @llvm.aarch64.sme.fp8.fmopa.za32(i32 0, <vscale x 16 x i1> %ptrue, <vscale x 16 x i1> %ptrue, <vscale x 16 x i8> %lhs.load, <vscale x 16 x i8> %rhs.load)
+  %iv.next = add nuw i64 %iv, 1
+  %cond = icmp ult i64 %iv.next, %n
+  br i1 %cond, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+declare <vscale x 16 x i1> @llvm.aarch64.sve.ptrue.nxv16i1(i32 immarg) #0
+declare <vscale x 16 x i8> @llvm.aarch64.sve.ld1.nxv16i8(<vscale x 16 x i1>, ptr) #1
+declare void @llvm.aarch64.set.fpmr(i64) #2
+declare void @llvm.aarch64.sme.fp8.fmopa.za32(i32 immarg, <vscale x 16 x i1>, <vscale x 16 x i1>, <vscale x 16 x i8>, <vscale x 16 x i8>) #3
+
+attributes #0 = { nocallback nofree nosync nounwind willreturn memory(none) }
+attributes #1 = { nocallback nofree nosync nounwind willreturn memory(argmem: read) }
+attributes #2 = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: write) }
+attributes #3 = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: readwrite) }


        


More information about the llvm-commits mailing list