[llvm] [AMDGPU] Register HIP stdpar math fixuppass for LTO (PR #223828)
Joseph Huber via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 15 13:59:00 PDT 2026
https://github.com/jhuber6 created https://github.com/llvm/llvm-project/pull/223828
Summary:
This pass is necesasry to fixup math operations but seems to have been
mistakenly guarded such that it would not run during the LTO pipeline. I
am assuming this was unintentional and problems were triggered when the
HIP build moved to LTO by default.
The reported issue was a sincos merging that did not get lowered
properly.
>From c85681affa02e6c0edbfbd2f2cbabc460543910d Mon Sep 17 00:00:00 2001
From: Joseph Huber <huberjn at outlook.com>
Date: Tue, 15 Sep 2026 15:53:34 -0500
Subject: [PATCH] [AMDGPU] Register HIP stdpar math fixuppass for LTO
Summary:
This pass is necesasry to fixup math operations but seems to have been
mistakenly guarded such that it would not run during the LTO pipeline. I
am assuming this was unintentional and problems were triggered when the
HIP build moved to LTO by default.
The reported issue was a sincos merging that did not get lowered
properly.
---
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 7 ++--
.../Transforms/HipStdPar/math-fixup-lto.ll | 36 +++++++++++++++++++
2 files changed, 40 insertions(+), 3 deletions(-)
create mode 100644 llvm/test/Transforms/HipStdPar/math-fixup-lto.ll
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index b8a1e4a656fd8..a311cf17da20a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -1045,14 +1045,15 @@ void AMDGPUTargetMachine::registerPassBuilderCallbacks(PassBuilder &PB) {
PB.registerPipelineEarlySimplificationEPCallback(
[this](ModulePassManager &PM, OptimizationLevel Level,
ThinOrFullLTOPhase Phase) {
+ if (EnableHipStdPar && getTargetTriple().isAMDGCN())
+ PM.addPass(HipStdParMathFixupPass());
+
if (!isLTOPreLink(Phase) && getTargetTriple().isAMDGCN()) {
// When we are not using -fgpu-rdc, we can run accelerator code
// selection relatively early, but still after linking to prevent
// eager removal of potentially reachable symbols.
- if (EnableHipStdPar) {
- PM.addPass(HipStdParMathFixupPass());
+ if (EnableHipStdPar)
PM.addPass(HipStdParAcceleratorCodeSelectionPass());
- }
PM.addPass(AMDGPUPrintfRuntimeBindingPass());
}
diff --git a/llvm/test/Transforms/HipStdPar/math-fixup-lto.ll b/llvm/test/Transforms/HipStdPar/math-fixup-lto.ll
new file mode 100644
index 0000000000000..bea234c723dc0
--- /dev/null
+++ b/llvm/test/Transforms/HipStdPar/math-fixup-lto.ll
@@ -0,0 +1,36 @@
+; REQUIRES: amdgpu-registered-target
+; RUN: opt -mtriple=amdgcn-amd-amdhsa -passes='lto-pre-link<O1>' \
+; RUN: -amdgpu-enable-hipstdpar %s \
+; RUN: | opt -S -mtriple=amdgcn-amd-amdhsa -passes='lto<O1>' \
+; RUN: -amdgpu-enable-hipstdpar \
+; RUN: | FileCheck %s --implicit-check-not=llvm.sincos
+
+define linkonce_odr double @mSin(double %x) #0 {
+ %sin = call double @llvm.sin.f64(double %x)
+ ret double %sin
+}
+
+define linkonce_odr double @mCos(double %x) #0 {
+ %cos = call double @llvm.cos.f64(double %x)
+ ret double %cos
+}
+
+define amdgpu_kernel void @test(double %x, ptr addrspace(1) %out) {
+; CHECK-LABEL: define amdgpu_kernel void @test(
+; CHECK: [[SIN:%.*]] = {{.*}}call double @__hipstdpar_sin_f64(double %x)
+; CHECK: [[COS:%.*]] = {{.*}}call double @__hipstdpar_cos_f64(double %x)
+; CHECK: [[SUM:%.*]] = fadd double [[SIN]], [[COS]]
+; CHECK: store double [[SUM]], ptr addrspace(1) %out
+;
+entry:
+ %sin = call double @mSin(double %x)
+ %cos = call double @mCos(double %x)
+ %sum = fadd double %sin, %cos
+ store double %sum, ptr addrspace(1) %out
+ ret void
+}
+
+declare double @llvm.sin.f64(double)
+declare double @llvm.cos.f64(double)
+
+attributes #0 = { alwaysinline }
More information about the llvm-commits
mailing list