[llvm-branch-commits] [llvm] [LTO] Add support for NewPM CodeGen (PR #210249)

Vikram Hegde via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Tue Aug 18 00:58:07 PDT 2026


https://github.com/vikramRH updated https://github.com/llvm/llvm-project/pull/210249

>From d520b90f7da6fb6fed00176140f2a81491102422 Mon Sep 17 00:00:00 2001
From: vikhegde <vikram.hegde at amd.com>
Date: Thu, 16 Jul 2026 16:23:49 +0530
Subject: [PATCH] [LTO] Add support for NewPM CodeGen

---
 llvm/lib/LTO/LTOBackend.cpp                   | 85 +++++++++++++++-
 .../CodeGen/AMDGPU/lto-lower-module-lds.ll    | 96 +++++++++++++++++++
 2 files changed, 177 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/LTO/LTOBackend.cpp b/llvm/lib/LTO/LTOBackend.cpp
index 73697a9d0d446..7cdd55fd43bbb 100644
--- a/llvm/lib/LTO/LTOBackend.cpp
+++ b/llvm/lib/LTO/LTOBackend.cpp
@@ -22,8 +22,10 @@
 #include "llvm/Bitcode/BitcodeReader.h"
 #include "llvm/Bitcode/BitcodeWriter.h"
 #include "llvm/CGData/CodeGenData.h"
+#include "llvm/CodeGen/MachineModuleInfo.h"
 #include "llvm/IR/LLVMRemarkStreamer.h"
 #include "llvm/IR/LegacyPassManager.h"
+#include "llvm/IR/ModuleSummaryIndex.h"
 #include "llvm/IR/PassManager.h"
 #include "llvm/IR/Verifier.h"
 #include "llvm/LTO/LTO.h"
@@ -58,6 +60,12 @@ enum class LTOBitcodeEmbedding {
   EmbedPostMergePreOptimized = 2
 };
 
+enum class LTONewPMEnablementLevel {
+  Auto,         // Use the target dependent default.
+  ForceEnable,  // Always enable regardless of the target default.
+  ForceDisable, // Always disable regardless of the target default.
+};
+
 static cl::opt<LTOBitcodeEmbedding> EmbedBitcode(
     "lto-embed-bitcode", cl::init(LTOBitcodeEmbedding::DoNotEmbed),
     cl::values(clEnumValN(LTOBitcodeEmbedding::DoNotEmbed, "none",
@@ -74,6 +82,18 @@ static cl::opt<bool> ThinLTOAssumeMerged(
     cl::desc("Assume the input has already undergone ThinLTO function "
              "importing and the other pre-optimization pipeline changes."));
 
+static cl::opt<LTONewPMEnablementLevel> EnableNPMForBackend(
+    "enable-npm-for-backend", cl::init(LTONewPMEnablementLevel::Auto),
+    cl::values(
+        clEnumValN(LTONewPMEnablementLevel::Auto, "auto",
+                   "Use the target dependent default"),
+        clEnumValN(LTONewPMEnablementLevel::ForceEnable, "force-on",
+                   "Always enable NPM regardless of the target default"),
+        clEnumValN(LTONewPMEnablementLevel::ForceDisable, "force-disable",
+                   "Always disable NewPM regardless of the target default")),
+    cl::desc(
+        "option to enable or disable NewPM to drive the CodeGen pipeline."));
+
 static cl::list<std::string>
     SaveModulesList("filter-save-modules", cl::value_desc("module names"),
                     cl::desc("Only save bitcode for module whose name without "
@@ -480,7 +500,64 @@ static void codegen(const Config &Conf, TargetMachine *TM,
   // Stream->commit() is called. The commit function of CacheStream deletes
   // the raw stream, which is too early as streamers (e.g. MCAsmStreamer)
   // keep the pointer and may use it until their destruction. See #138194.
-  {
+  if (EnableNPMForBackend == LTONewPMEnablementLevel::ForceEnable ||
+      (EnableNPMForBackend == LTONewPMEnablementLevel::Auto &&
+       TM->shouldDefaultToNewPM())) {
+    MachineModuleInfo MMI(TM);
+    PassInstrumentationCallbacks PIC;
+    MachineFunctionAnalysisManager MFAM;
+    LoopAnalysisManager LAM;
+    FunctionAnalysisManager FAM;
+    CGSCCAnalysisManager CGAM;
+    ModuleAnalysisManager MAM;
+    PassBuilder PB(TM, PipelineTuningOptions(), std::nullopt, &PIC);
+
+    StandardInstrumentations SI(Mod.getContext(), Conf.DebugPassManager,
+                                Conf.VerifyEach);
+    SI.registerCallbacks(PIC, &MAM);
+
+    TargetLibraryInfoImpl TLII(Mod.getTargetTriple(), TM->Options.VecLib);
+    FAM.registerPass([&] { return TargetLibraryAnalysis(TLII); });
+    MAM.registerPass([&] { return MachineModuleAnalysis(MMI); });
+    MAM.registerPass([&] {
+      return RuntimeLibraryAnalysis(
+          Mod.getTargetTriple(), TM->Options.ExceptionModel,
+          TM->Options.FloatABIType, TM->Options.EABIVersion,
+          TM->Options.MCOptions.ABIName, TM->Options.VecLib);
+    });
+
+    if (!isEmptyModule(Mod))
+      MAM.registerPass(
+          [&] { return ImmutableModuleSummaryIndexAnalysis(&CombinedIndex); });
+
+    PB.registerModuleAnalyses(MAM);
+    PB.registerCGSCCAnalyses(CGAM);
+    PB.registerFunctionAnalyses(FAM);
+    PB.registerLoopAnalyses(LAM);
+    PB.registerMachineFunctionAnalyses(MFAM);
+    PB.crossRegisterProxies(LAM, FAM, CGAM, MAM, &MFAM);
+
+    ModulePassManager MPM;
+    FunctionPassManager FPM;
+
+    if (Error Err = TM->buildCodeGenPipeline(
+            MPM, MAM, *Stream->OS, DwoOut ? &DwoOut->os() : nullptr,
+            Conf.CGFileType, CGPassBuilderOption(), MMI.getContext(), &PIC))
+      return;
+
+    if (PrintPipelinePasses) {
+      std::string PipelineStr;
+      raw_string_ostream OutS(PipelineStr);
+      MPM.printPipeline(OutS, [&PIC](StringRef ClassName) {
+        auto PassName = PIC.getPassNameForClassName(ClassName);
+        return PassName.empty() ? ClassName : PassName;
+      });
+      outs() << PipelineStr << '\n';
+    } else {
+      MPM.run(Mod, MAM);
+    }
+
+  } else {
     legacy::PassManager CodeGenPasses;
     TargetLibraryInfoImpl TLII(Mod.getTargetTriple(), TM->Options.VecLib);
     CodeGenPasses.add(new TargetLibraryInfoWrapperPass(TLII));
@@ -504,11 +581,11 @@ static void codegen(const Config &Conf, TargetMachine *TM,
                                 Conf.CGFileType))
       report_fatal_error("Failed to setup codegen");
     CodeGenPasses.run(Mod);
-
-    if (DwoOut)
-      DwoOut->keep();
   }
 
+  if (DwoOut)
+    DwoOut->keep();
+
   if (Error Err = Stream->commit())
     report_fatal_error(std::move(Err));
 }
diff --git a/llvm/test/CodeGen/AMDGPU/lto-lower-module-lds.ll b/llvm/test/CodeGen/AMDGPU/lto-lower-module-lds.ll
index a2e3a84f7c064..ef5f789c06d9d 100644
--- a/llvm/test/CodeGen/AMDGPU/lto-lower-module-lds.ll
+++ b/llvm/test/CodeGen/AMDGPU/lto-lower-module-lds.ll
@@ -38,6 +38,102 @@
 ; CHECK: ModulePass Manager
 ; CHECK:   Lower uses of LDS variables from non-kernel functions
 
+; Test -enable-npm-for-backend.
+
+; NPM Default O0
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -debug-pass-manager 2>&1 | FileCheck --check-prefix=NPM %s
+
+; NPM Unified O0
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -debug-pass-manager 2>&1 | FileCheck --check-prefix=NPM %s
+
+; NPM Default O2
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -debug-pass-manager 2>&1 | FileCheck --check-prefix=NPM %s
+
+; NPM Unified O2
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -debug-pass-manager 2>&1 | FileCheck --check-prefix=NPM %s
+
+; The New PM full-LTO pipeline still runs the module-LDS lowering, and the CG
+; pipeline is now driven by the New PM (no legacy "ModulePass Manager" structure).
+; NPM-NOT: ModulePass Manager
+; NPM: Running pass: AMDGPULowerModuleLDSPass on [module]
+; NPM: Running pass: SelectionDAGISelPass on test
+; NPM: Running pass: PrologEpilogInserterPass on test
+; NPM: Running pass: AMDGPUAsmPrinterPass on test
+
+; Test -print-pipeline-passes prints the New PM codegen pipeline.
+
+; PP Default O2
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -print-pipeline-passes 2>&1 | FileCheck --check-prefix=PP %s
+
+; PP Unified O2
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-on -print-pipeline-passes 2>&1 | FileCheck --check-prefix=PP %s
+
+; First line is the full-LTO (IR) pipeline, second line is the codegen (machine)
+; pipeline. Check a few codegen-specific passes (in order) on the second line.
+; PP: amdgpu-lower-module-lds
+; PP: require<MachineModuleAnalysis>
+; PP-SAME: amdgpu-isel
+; PP-SAME: prolog-epilog
+; PP-SAME: amdgpu-asm-printer
+
+; Test -enable-npm-for-backend=force-disable always drives the CodeGen pipeline
+; with the legacy PM, regardless of the target default.
+
+; DISABLE Default O0
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-disable -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=DISABLE %s
+
+; DISABLE Unified O0
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-disable -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=DISABLE %s
+
+; DISABLE Default O2
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-disable -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=DISABLE %s
+
+; DISABLE Unified O2
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=force-disable -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=DISABLE %s
+
+; force-disable keeps the legacy CodeGen PM even after the New PM CodeGen
+; pipeline becomes the AMDGPU default, so no New PM CodeGen passes should run.
+; DISABLE: Running pass: AMDGPULowerModuleLDSPass on [module]
+; DISABLE: ModulePass Manager
+; DISABLE:   Lower uses of LDS variables from non-kernel functions
+; DISABLE-NOT: Running pass: SelectionDAGISelPass
+
+; Test -enable-npm-for-backend=auto follows the target default, which is
+; currently the legacy CodeGen PM for AMDGPU. Update the checks below to the New
+; PM CodeGen pipeline when it becomes the default for AMDGPU.
+
+; AUTO Default O0
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=auto -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=AUTO %s
+
+; AUTO Unified O0
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O0 -cg-opt-level 0 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=auto -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=AUTO %s
+
+; AUTO Default O2
+; RUN: opt -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=auto -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=AUTO %s
+
+; AUTO Unified O2
+; RUN: opt -unified-lto -thinlto-split-lto-unit -thinlto-bc -mtriple=amdgpu10.30-- %s -o %t.bc
+; RUN: llvm-lto2 run -unified-lto=full -O2 -cg-opt-level 2 %t.bc -o %t.s -r %t.bc,test,px -enable-npm-for-backend=auto -debug-pass-manager -debug-pass=Structure 2>&1 | FileCheck --check-prefix=AUTO %s
+
+; auto currently maps to the legacy CodeGen PM for AMDGPU (target default false).
+; AUTO: Running pass: AMDGPULowerModuleLDSPass on [module]
+; AUTO: ModulePass Manager
+; AUTO:   Lower uses of LDS variables from non-kernel functions
+; AUTO-NOT: Running pass: SelectionDAGISelPass
+
 @lds = internal unnamed_addr addrspace(3) global i32 poison, align 4
 
 define amdgpu_kernel void @test() {



More information about the llvm-branch-commits mailing list