[llvm] [DirectX][SPIR-V] Fix `copysign` backend lowering (PR #217421)

Kaitlin Peng via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 19 11:20:54 PDT 2026


https://github.com/kmpeng created https://github.com/llvm/llvm-project/pull/217421

Fixes #216826.

Adds `copysign` DirectX backend lowering and fixes the SPIR-V backend lowering. Both lower with the bit manipulation `copysign(magnitude, sign) = bitcast((bitcast(magnitude) & ~signBit) | (bitcast(sign) & signBit))` except the OpenCL path, which still emits `OpExtInst ... copysign`.

>From 0a4891f48bf0bf2b593747994c6775ed4ec37ed9 Mon Sep 17 00:00:00 2001
From: kmpeng <kaitlinpeng at microsoft.com>
Date: Wed, 19 Aug 2026 10:47:26 -0700
Subject: [PATCH 1/2] add copysign directx backend lowering

---
 .../Target/DirectX/DXILIntrinsicExpansion.cpp | 29 +++++++++
 llvm/test/CodeGen/DirectX/copysign.ll         | 60 +++++++++++++++++++
 2 files changed, 89 insertions(+)
 create mode 100644 llvm/test/CodeGen/DirectX/copysign.ll

diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index 8ecc7ddc4b64f..87f7b1ab92564 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -203,6 +203,7 @@ static bool isIntrinsicExpansion(Function &F) {
   case Intrinsic::assume:
   case Intrinsic::abs:
   case Intrinsic::atan2:
+  case Intrinsic::copysign:
   case Intrinsic::fshl:
   case Intrinsic::fshr:
   case Intrinsic::exp:
@@ -1055,6 +1056,31 @@ static Value *expandSignIntrinsic(CallInst *Orig) {
   return Builder.CreateSub(ZextGT, ZextLT);
 }
 
+// Expand llvm.copysign by combining the sign bit with the magnitude bits using
+// bitwise operations.
+static Value *expandCopySignIntrinsic(CallInst *Orig) {
+  Value *Magnitude = Orig->getOperand(0);
+  Value *Sign = Orig->getOperand(1);
+  Type *Ty = Orig->getType();
+
+  IRBuilder<> Builder(Orig);
+
+  unsigned BitWidth = Ty->getScalarSizeInBits();
+  Type *IntTy = Ty->getWithNewType(Builder.getIntNTy(BitWidth));
+
+  // `ConstantInt::get` broadcasts to a splat when `IntTy` is a vector.
+  APInt SignMaskVal = APInt::getSignMask(BitWidth);
+  Constant *SignMask = ConstantInt::get(IntTy, SignMaskVal);
+  Constant *NotSignMask = ConstantInt::get(IntTy, ~SignMaskVal);
+
+  Value *MagnitudeInt = Builder.CreateBitCast(Magnitude, IntTy);
+  Value *SignInt = Builder.CreateBitCast(Sign, IntTy);
+  Value *MagnitudeBits = Builder.CreateAnd(MagnitudeInt, NotSignMask);
+  Value *SignBits = Builder.CreateAnd(SignInt, SignMask);
+  Value *Result = Builder.CreateOr(MagnitudeBits, SignBits);
+  return Builder.CreateBitCast(Result, Ty);
+}
+
 // Expand llvm.matrix.multiply by extracting row/column vectors and computing
 // dot products.
 // Result[r,c] = dot(row_r(LHS), col_c(RHS))
@@ -1249,6 +1275,9 @@ static bool expandIntrinsic(Function &F, CallInst *Orig) {
   case Intrinsic::atan2:
     Result = expandAtan2Intrinsic(Orig);
     break;
+  case Intrinsic::copysign:
+    Result = expandCopySignIntrinsic(Orig);
+    break;
   case Intrinsic::fshl:
     Result = expandFunnelShiftIntrinsic<true>(Orig);
     break;
diff --git a/llvm/test/CodeGen/DirectX/copysign.ll b/llvm/test/CodeGen/DirectX/copysign.ll
new file mode 100644
index 0000000000000..b683fc85520c2
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/copysign.ll
@@ -0,0 +1,60 @@
+; RUN: opt -S -dxil-intrinsic-expansion -dxil-op-lower -mtriple=dxil-pc-shadermodel6.3-library %s | FileCheck %s
+
+; Make sure the copysign intrinsic is expanded to bitwise operations.
+
+; CHECK-LABEL: copysign_half
+define noundef half @copysign_half(half noundef %a, half noundef %b) {
+entry:
+  ; CHECK: [[MAGNITUDE_INT:%.*]] = bitcast half %{{.*}} to i16
+  ; CHECK: [[SIGN_INT:%.*]] = bitcast half %{{.*}} to i16
+  ; CHECK: [[MAGNITUDE_BITS:%.*]] = and i16 [[MAGNITUDE_INT]], 32767
+  ; CHECK: [[SIGN_BITS:%.*]] = and i16 [[SIGN_INT]], -32768
+  ; CHECK: [[RES_INT:%.*]] = or i16 [[MAGNITUDE_BITS]], [[SIGN_BITS]]
+  ; CHECK: bitcast i16 [[RES_INT]] to half
+  %r = call half @llvm.copysign.f16(half %a, half %b)
+  ret half %r
+}
+
+; CHECK-LABEL: copysign_float
+define noundef float @copysign_float(float noundef %a, float noundef %b) {
+entry:
+  ; CHECK: [[MAGNITUDE_INT:%.*]] = bitcast float %{{.*}} to i32
+  ; CHECK: [[SIGN_INT:%.*]] = bitcast float %{{.*}} to i32
+  ; CHECK: [[MAGNITUDE_BITS:%.*]] = and i32 [[MAGNITUDE_INT]], 2147483647
+  ; CHECK: [[SIGN_BITS:%.*]] = and i32 [[SIGN_INT]], -2147483648
+  ; CHECK: [[RES_INT:%.*]] = or i32 [[MAGNITUDE_BITS]], [[SIGN_BITS]]
+  ; CHECK: bitcast i32 [[RES_INT]] to float
+  %r = call float @llvm.copysign.f32(float %a, float %b)
+  ret float %r
+}
+
+; CHECK-LABEL: copysign_double
+define noundef double @copysign_double(double noundef %a, double noundef %b) {
+entry:
+  ; CHECK: [[MAGNITUDE_INT:%.*]] = bitcast double %{{.*}} to i64
+  ; CHECK: [[SIGN_INT:%.*]] = bitcast double %{{.*}} to i64
+  ; CHECK: [[MAGNITUDE_BITS:%.*]] = and i64 [[MAGNITUDE_INT]], 9223372036854775807
+  ; CHECK: [[SIGN_BITS:%.*]] = and i64 [[SIGN_INT]], -9223372036854775808
+  ; CHECK: [[RES_INT:%.*]] = or i64 [[MAGNITUDE_BITS]], [[SIGN_BITS]]
+  ; CHECK: bitcast i64 [[RES_INT]] to double
+  %r = call double @llvm.copysign.f64(double %a, double %b)
+  ret double %r
+}
+
+; CHECK-LABEL: copysign_float4
+define noundef <4 x float> @copysign_float4(<4 x float> noundef %a, <4 x float> noundef %b) {
+entry:
+  ; CHECK: [[MAGNITUDE_INT:%.*]] = bitcast <4 x float> %{{.*}} to <4 x i32>
+  ; CHECK: [[SIGN_INT:%.*]] = bitcast <4 x float> %{{.*}} to <4 x i32>
+  ; CHECK: [[MAGNITUDE_BITS:%.*]] = and <4 x i32> [[MAGNITUDE_INT]], splat (i32 2147483647)
+  ; CHECK: [[SIGN_BITS:%.*]] = and <4 x i32> [[SIGN_INT]], splat (i32 -2147483648)
+  ; CHECK: [[RES_INT:%.*]] = or <4 x i32> [[MAGNITUDE_BITS]], [[SIGN_BITS]]
+  ; CHECK: bitcast <4 x i32> [[RES_INT]] to <4 x float>
+  %r = call <4 x float> @llvm.copysign.v4f32(<4 x float> %a, <4 x float> %b)
+  ret <4 x float> %r
+}
+
+declare half @llvm.copysign.f16(half, half)
+declare float @llvm.copysign.f32(float, float)
+declare double @llvm.copysign.f64(double, double)
+declare <4 x float> @llvm.copysign.v4f32(<4 x float>, <4 x float>)

>From 92e5af31933f872bc780e5f23f5fe577a99831d8 Mon Sep 17 00:00:00 2001
From: kmpeng <kaitlinpeng at microsoft.com>
Date: Wed, 19 Aug 2026 10:47:52 -0700
Subject: [PATCH 2/2] fix copysign spirv backend lowering

---
 .../Target/SPIRV/SPIRVInstructionSelector.cpp | 56 ++++++++++++-
 .../CodeGen/SPIRV/hlsl-intrinsics/copysign.ll | 82 +++++++++++++++++++
 2 files changed, 137 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/CodeGen/SPIRV/hlsl-intrinsics/copysign.ll

diff --git a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
index 9c6268ff0c618..6611c95ca9948 100644
--- a/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
+++ b/llvm/lib/Target/SPIRV/SPIRVInstructionSelector.cpp
@@ -480,6 +480,9 @@ class SPIRVInstructionSelector : public InstructionSelector {
   bool selectFrexp(Register ResVReg, SPIRVTypeInst ResType,
                    MachineInstr &I) const;
 
+  bool selectCopySign(Register ResVReg, SPIRVTypeInst ResType,
+                      MachineInstr &I) const;
+
   bool selectLdexp(Register ResVReg, SPIRVTypeInst ResType,
                    MachineInstr &I) const;
   bool selectSincos(Register ResVReg, SPIRVTypeInst ResType,
@@ -1188,7 +1191,7 @@ bool SPIRVInstructionSelector::spvSelect(Register ResVReg,
     return selectExtInst(ResVReg, ResType, I, CL::fmax, GL::NMax);
 
   case TargetOpcode::G_FCOPYSIGN:
-    return selectExtInst(ResVReg, ResType, I, CL::copysign);
+    return selectCopySign(ResVReg, ResType, I);
 
   case TargetOpcode::G_FCEIL:
     return selectExtInst(ResVReg, ResType, I, CL::ceil, GL::Ceil);
@@ -1575,6 +1578,57 @@ bool SPIRVInstructionSelector::selectExtInst(Register ResVReg,
   return false;
 }
 
+bool SPIRVInstructionSelector::selectCopySign(Register ResVReg,
+                                              SPIRVTypeInst ResType,
+                                              MachineInstr &I) const {
+  if (STI.canUseExtInstSet(SPIRV::InstructionSet::OpenCL_std))
+    return selectExtInst(ResVReg, ResType, I, CL::copysign);
+
+  // There is no copysign instruction in the GLSL Extended Instruction set, so
+  // it is implemented with bit manipulation:
+  //   bitcast((bitcast(magnitude) & ~signBit) | (bitcast(sign) & signBit))
+  Register MagnitudeReg = I.getOperand(1).getReg();
+  Register SignReg = I.getOperand(2).getReg();
+
+  const unsigned BitWidth = GR.getScalarOrVectorBitWidth(ResType);
+  const unsigned ComponentCount = GR.getScalarOrVectorComponentCount(ResType);
+  const bool IsVector = ComponentCount > 1;
+
+  SPIRVTypeInst IntType = GR.getOrCreateSPIRVIntegerType(BitWidth, I, TII);
+  if (IsVector)
+    IntType = GR.getOrCreateSPIRVVectorType(IntType, ComponentCount, I, TII);
+
+  const APInt SignMaskVal = APInt::getSignMask(BitWidth);
+  Register SignMask =
+      IsVector ? GR.getOrCreateConstVector(SignMaskVal, I, IntType, TII)
+               : GR.getOrCreateConstInt(SignMaskVal, I, IntType, TII);
+  Register NotSignMask =
+      IsVector ? GR.getOrCreateConstVector(~SignMaskVal, I, IntType, TII)
+               : GR.getOrCreateConstInt(~SignMaskVal, I, IntType, TII);
+
+  const unsigned AndOpcode =
+      IsVector ? SPIRV::OpBitwiseAndV : SPIRV::OpBitwiseAndS;
+  const unsigned OrOpcode =
+      IsVector ? SPIRV::OpBitwiseOrV : SPIRV::OpBitwiseOrS;
+
+  Register MagnitudeInt = MRI->createVirtualRegister(GR.getRegClass(IntType));
+  Register SignInt = MRI->createVirtualRegister(GR.getRegClass(IntType));
+  Register MagnitudeBits = MRI->createVirtualRegister(GR.getRegClass(IntType));
+  Register SignBits = MRI->createVirtualRegister(GR.getRegClass(IntType));
+  Register ResInt = MRI->createVirtualRegister(GR.getRegClass(IntType));
+
+  if (!selectOpWithSrcs(MagnitudeInt, IntType, I, {MagnitudeReg},
+                        SPIRV::OpBitcast) ||
+      !selectOpWithSrcs(SignInt, IntType, I, {SignReg}, SPIRV::OpBitcast) ||
+      !selectOpWithSrcs(MagnitudeBits, IntType, I, {MagnitudeInt, NotSignMask},
+                        AndOpcode) ||
+      !selectOpWithSrcs(SignBits, IntType, I, {SignInt, SignMask}, AndOpcode) ||
+      !selectOpWithSrcs(ResInt, IntType, I, {MagnitudeBits, SignBits},
+                        OrOpcode))
+    return false;
+  return selectOpWithSrcs(ResVReg, ResType, I, {ResInt}, SPIRV::OpBitcast);
+}
+
 bool SPIRVInstructionSelector::selectFrexp(Register ResVReg,
                                            SPIRVTypeInst ResType,
                                            MachineInstr &I) const {
diff --git a/llvm/test/CodeGen/SPIRV/hlsl-intrinsics/copysign.ll b/llvm/test/CodeGen/SPIRV/hlsl-intrinsics/copysign.ll
new file mode 100644
index 0000000000000..dce9438bc49e1
--- /dev/null
+++ b/llvm/test/CodeGen/SPIRV/hlsl-intrinsics/copysign.ll
@@ -0,0 +1,82 @@
+; RUN: llc -O0 -verify-machineinstrs -mtriple=spirv-unknown-vulkan %s -o - | FileCheck %s
+; RUN: %if spirv-tools %{ llc -O0 -mtriple=spirv-unknown-vulkan %s -o - -filetype=obj | spirv-val %}
+
+; Vulkan SPIR-V has no copysign in GLSL.std.450, so llvm.copysign is lowered with bit manipulation.
+
+; CHECK-DAG: %[[#float_16:]] = OpTypeFloat 16
+; CHECK-DAG: %[[#int_16:]] = OpTypeInt 16 0
+; CHECK-DAG: %[[#sign_16:]] = OpConstant %[[#int_16]] 32768
+; CHECK-DAG: %[[#magn_16:]] = OpConstant %[[#int_16]] 32767
+; CHECK-DAG: %[[#float_32:]] = OpTypeFloat 32
+; CHECK-DAG: %[[#int_32:]] = OpTypeInt 32 0
+; CHECK-DAG: %[[#sign_32:]] = OpConstant %[[#int_32]] 2147483648
+; CHECK-DAG: %[[#magn_32:]] = OpConstant %[[#int_32]] 2147483647
+; CHECK-DAG: %[[#vec4_float_32:]] = OpTypeVector %[[#float_32]] 4
+; CHECK-DAG: %[[#vec4_int_32:]] = OpTypeVector %[[#int_32]] 4
+; CHECK-DAG: %[[#vec4_sign_32:]] = OpConstantComposite %[[#vec4_int_32]] %[[#sign_32]] %[[#sign_32]] %[[#sign_32]] %[[#sign_32]]
+; CHECK-DAG: %[[#vec4_magn_32:]] = OpConstantComposite %[[#vec4_int_32]] %[[#magn_32]] %[[#magn_32]] %[[#magn_32]] %[[#magn_32]]
+; CHECK-DAG: %[[#float_64:]] = OpTypeFloat 64
+; CHECK-DAG: %[[#int_64:]] = OpTypeInt 64 0
+; CHECK-DAG: %[[#sign_64:]] = OpConstant %[[#int_64]] 9223372036854775808
+; CHECK-DAG: %[[#magn_64:]] = OpConstant %[[#int_64]] 9223372036854775807
+
+define noundef half @copysign_half(half noundef %a, half noundef %b) {
+entry:
+  ; CHECK: %[[#a:]] = OpFunctionParameter %[[#float_16]]
+  ; CHECK: %[[#b:]] = OpFunctionParameter %[[#float_16]]
+  ; CHECK: %[[#ai:]] = OpBitcast %[[#int_16]] %[[#a]]
+  ; CHECK: %[[#bi:]] = OpBitcast %[[#int_16]] %[[#b]]
+  ; CHECK: %[[#am:]] = OpBitwiseAnd %[[#int_16]] %[[#ai]] %[[#magn_16]]
+  ; CHECK: %[[#bs:]] = OpBitwiseAnd %[[#int_16]] %[[#bi]] %[[#sign_16]]
+  ; CHECK: %[[#or:]] = OpBitwiseOr %[[#int_16]] %[[#am]] %[[#bs]]
+  ; CHECK: %[[#]] = OpBitcast %[[#float_16]] %[[#or]]
+  %r = call half @llvm.copysign.f16(half %a, half %b)
+  ret half %r
+}
+
+define noundef float @copysign_float(float noundef %a, float noundef %b) {
+entry:
+  ; CHECK: %[[#a:]] = OpFunctionParameter %[[#float_32]]
+  ; CHECK: %[[#b:]] = OpFunctionParameter %[[#float_32]]
+  ; CHECK: %[[#ai:]] = OpBitcast %[[#int_32]] %[[#a]]
+  ; CHECK: %[[#bi:]] = OpBitcast %[[#int_32]] %[[#b]]
+  ; CHECK: %[[#am:]] = OpBitwiseAnd %[[#int_32]] %[[#ai]] %[[#magn_32]]
+  ; CHECK: %[[#bs:]] = OpBitwiseAnd %[[#int_32]] %[[#bi]] %[[#sign_32]]
+  ; CHECK: %[[#or:]] = OpBitwiseOr %[[#int_32]] %[[#am]] %[[#bs]]
+  ; CHECK: %[[#]] = OpBitcast %[[#float_32]] %[[#or]]
+  %r = call float @llvm.copysign.f32(float %a, float %b)
+  ret float %r
+}
+
+define noundef double @copysign_double(double noundef %a, double noundef %b) {
+entry:
+  ; CHECK: %[[#a:]] = OpFunctionParameter %[[#float_64]]
+  ; CHECK: %[[#b:]] = OpFunctionParameter %[[#float_64]]
+  ; CHECK: %[[#ai:]] = OpBitcast %[[#int_64]] %[[#a]]
+  ; CHECK: %[[#bi:]] = OpBitcast %[[#int_64]] %[[#b]]
+  ; CHECK: %[[#am:]] = OpBitwiseAnd %[[#int_64]] %[[#ai]] %[[#magn_64]]
+  ; CHECK: %[[#bs:]] = OpBitwiseAnd %[[#int_64]] %[[#bi]] %[[#sign_64]]
+  ; CHECK: %[[#or:]] = OpBitwiseOr %[[#int_64]] %[[#am]] %[[#bs]]
+  ; CHECK: %[[#]] = OpBitcast %[[#float_64]] %[[#or]]
+  %r = call double @llvm.copysign.f64(double %a, double %b)
+  ret double %r
+}
+
+define noundef <4 x float> @copysign_float4(<4 x float> noundef %a, <4 x float> noundef %b) {
+entry:
+  ; CHECK: %[[#a:]] = OpFunctionParameter %[[#vec4_float_32]]
+  ; CHECK: %[[#b:]] = OpFunctionParameter %[[#vec4_float_32]]
+  ; CHECK: %[[#ai:]] = OpBitcast %[[#vec4_int_32]] %[[#a]]
+  ; CHECK: %[[#bi:]] = OpBitcast %[[#vec4_int_32]] %[[#b]]
+  ; CHECK: %[[#am:]] = OpBitwiseAnd %[[#vec4_int_32]] %[[#ai]] %[[#vec4_magn_32]]
+  ; CHECK: %[[#bs:]] = OpBitwiseAnd %[[#vec4_int_32]] %[[#bi]] %[[#vec4_sign_32]]
+  ; CHECK: %[[#or:]] = OpBitwiseOr %[[#vec4_int_32]] %[[#am]] %[[#bs]]
+  ; CHECK: %[[#]] = OpBitcast %[[#vec4_float_32]] %[[#or]]
+  %r = call <4 x float> @llvm.copysign.v4f32(<4 x float> %a, <4 x float> %b)
+  ret <4 x float> %r
+}
+
+declare half @llvm.copysign.f16(half, half)
+declare float @llvm.copysign.f32(float, float)
+declare double @llvm.copysign.f64(double, double)
+declare <4 x float> @llvm.copysign.v4f32(<4 x float>, <4 x float>)



More information about the llvm-commits mailing list