[llvm] [HLSL][DirectX] Lower llvm.dx.load.input and llvm.dx.store.output to DXIL (PR #208871)
Dan Brown via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 5 17:04:29 PDT 2026
https://github.com/danbrown-amd updated https://github.com/llvm/llvm-project/pull/208871
>From 9ec1e174f976c081ca8a7b887fb61f53f0fee071 Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 8 Jul 2026 18:56:41 -0600
Subject: [PATCH 1/5] [HLSL][DirectX] Lower llvm.dx.load.input and
llvm.dx.store.output to DXIL
Addresses #189766.
Assisted-by: Claude Sonnet 4
---
llvm/include/llvm/IR/IntrinsicsDirectX.td | 20 +++--
llvm/lib/Target/DirectX/DXIL.td | 24 +++++
.../Target/DirectX/DXILIntrinsicExpansion.cpp | 90 +++++++++++++++++++
llvm/test/CodeGen/DirectX/LoadInput.ll | 51 +++++++++++
llvm/test/CodeGen/DirectX/StoreOutput.ll | 49 ++++++++++
5 files changed, 226 insertions(+), 8 deletions(-)
create mode 100644 llvm/test/CodeGen/DirectX/LoadInput.ll
create mode 100644 llvm/test/CodeGen/DirectX/StoreOutput.ll
diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 090656ffb36c8..876447b8c866a 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -323,14 +323,18 @@ def int_dx_group_memory_barrier_with_group_sync
: DefaultAttrsIntrinsic<[], [], [IntrConvergent]>;
def int_dx_load_input
- : DefaultAttrsIntrinsic<[llvm_any_ty],
- [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i8_ty,
- llvm_i32_ty],
- [IntrConvergent]>;
+ : DefaultAttrsIntrinsic<
+ [llvm_any_ty],
+ [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
+ llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
+ llvm_i32_ty /*gsVertexOrPrimIndex*/],
+ [IntrConvergent]>;
def int_dx_store_output
- : DefaultAttrsIntrinsic<[],
- [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i8_ty,
- llvm_i32_ty, llvm_any_ty],
- [IntrConvergent]>;
+ : DefaultAttrsIntrinsic<
+ [],
+ [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
+ llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
+ llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
+ [IntrConvergent]>;
}
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 3d978c207f104..fb2648b6dea0b 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -412,6 +412,30 @@ class DXILOp<int opcode, DXILOpClass opclass> {
//
// This are sorted by ascending value of the DXIL Opcodes
+def LoadInput : DXILOp<4, loadInput> {
+ let Doc = "Loads a scalar value from a shader input register component.";
+ let intrinsics = [IntrinSelect<int_dx_load_input,
+ [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
+ // inputSigId, rowIndex, colIndex
+ let arguments = [Int32Ty, Int32Ty, Int8Ty];
+ let result = OverloadTy;
+ let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
+ let stages = [Stages<DXIL1_0, [all_stages]>];
+ let attributes = [Attributes<DXIL1_0, [ReadOnly]>];
+}
+
+def StoreOutput : DXILOp<5, storeOutput> {
+ let Doc = "Stores a scalar value to a shader output register component.";
+ let intrinsics = [IntrinSelect<int_dx_store_output,
+ [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
+ IntrinArgIndex<5>]>];
+ // outputSigId, rowIndex, colIndex, value
+ let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
+ let result = VoidTy;
+ let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
+ let stages = [Stages<DXIL1_0, [all_stages]>];
+}
+
def Abs : DXILOp<6, unary> {
let Doc = "Returns the absolute value of the input.";
let intrinsics = [IntrinSelect<int_fabs>];
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index a251288a6ae42..efd1212c6f154 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -234,6 +234,8 @@ static bool isIntrinsicExpansion(Function &F) {
case Intrinsic::matrix_transpose:
case Intrinsic::umul_with_overflow:
case Intrinsic::smul_with_overflow:
+ case Intrinsic::dx_load_input:
+ case Intrinsic::dx_store_output:
return true;
case Intrinsic::dx_resource_load_rawbuffer:
return resourceAccessNeeds64BitExpansion(
@@ -1212,6 +1214,87 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
return Builder.CreateShuffleVector(Mat, Mask);
}
+// Scalarize a vector int_dx_store_output call into per-component scalar calls.
+// The DXIL StoreOutput op is per-component; vector intrinsics are split here
+// so that DXILOpLowering sees only scalar variants.
+static bool expandStoreOutput(CallInst *Orig) {
+ auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
+ if (!VT)
+ return false; // already scalar, nothing to expand
+
+ IRBuilder<> Builder(Orig);
+ Module *M = Orig->getModule();
+ Type *Int8Ty = Builder.getInt8Ty();
+ Type *Int32Ty = Builder.getInt32Ty();
+ Type *ScalarTy = VT->getElementType();
+ unsigned NumElems = VT->getNumElements();
+
+ Value *SigpointId = Orig->getArgOperand(0);
+ Value *SigElementId = Orig->getArgOperand(1);
+ Value *RowIndex = Orig->getArgOperand(2);
+ Value *StartCol = Orig->getArgOperand(3); // i8
+ Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+ Value *Data = Orig->getArgOperand(5);
+ Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
+
+ Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
+ M, Intrinsic::dx_store_output, {ScalarTy});
+
+ for (unsigned I = 0; I < NumElems; ++I) {
+ Value *Scalar =
+ Builder.CreateExtractElement(Data, ConstantInt::get(Int32Ty, I));
+ Value *ColIdx =
+ Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
+ Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
+ Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+ GsVertexOrPrimIndex, Scalar});
+ }
+
+ Orig->eraseFromParent();
+ return true;
+}
+
+// Scalarize a vector int_dx_load_input call into per-component scalar calls
+// and reassemble the vector. The DXIL LoadInput op is per-component.
+static Value *expandLoadInput(CallInst *Orig) {
+ auto *VT = dyn_cast<FixedVectorType>(Orig->getType());
+ if (!VT)
+ return nullptr; // already scalar, nothing to expand
+
+ IRBuilder<> Builder(Orig);
+ Module *M = Orig->getModule();
+ Type *Int8Ty = Builder.getInt8Ty();
+ Type *Int32Ty = Builder.getInt32Ty();
+ Type *ScalarTy = VT->getElementType();
+ unsigned NumElems = VT->getNumElements();
+
+ // Intrinsic args: (sigpointId, sigElementId, rowIndex, colIndex:i8,
+ // gsVertexOrPrimIndex)
+ Value *SigpointId = Orig->getArgOperand(0);
+ Value *SigElementId = Orig->getArgOperand(1);
+ Value *RowIndex = Orig->getArgOperand(2);
+ Value *StartCol = Orig->getArgOperand(3); // i8
+ Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+ Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
+
+ Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
+ M, Intrinsic::dx_load_input, {ScalarTy});
+
+ Value *Vec = PoisonValue::get(VT);
+ for (unsigned I = 0; I < NumElems; ++I) {
+ Value *ColIdx =
+ Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
+ Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
+ Value *Scalar =
+ Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+ GsVertexOrPrimIndex});
+ Vec =
+ Builder.CreateInsertElement(Vec, Scalar, ConstantInt::get(Int32Ty, I));
+ }
+
+ return Vec;
+}
+
static bool expandIntrinsic(Function &F, CallInst *Orig) {
Value *Result = nullptr;
Intrinsic::ID IntrinsicId = F.getIntrinsicID();
@@ -1287,6 +1370,13 @@ static bool expandIntrinsic(Function &F, CallInst *Orig) {
case Intrinsic::dx_radians:
Result = expandRadiansIntrinsic(Orig);
break;
+ case Intrinsic::dx_load_input:
+ Result = expandLoadInput(Orig);
+ break;
+ case Intrinsic::dx_store_output:
+ if (expandStoreOutput(Orig))
+ return true;
+ break;
case Intrinsic::dx_resource_load_rawbuffer:
if (expandBufferLoadIntrinsic(Orig, /*IsRaw*/ true))
return true;
diff --git a/llvm/test/CodeGen/DirectX/LoadInput.ll b/llvm/test/CodeGen/DirectX/LoadInput.ll
new file mode 100644
index 0000000000000..d9d6aea6a457e
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/LoadInput.ll
@@ -0,0 +1,51 @@
+; RUN: opt -S -dxil-intrinsic-expansion -dxil-op-lower %s | FileCheck %s
+
+target triple = "dxil-pc-shadermodel6.0-pixel"
+
+; Scalar float load: one LoadInput call, result forwarded directly.
+; CHECK-LABEL: define float @load_scalar_f32
+define float @load_scalar_f32() {
+ ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2)
+ ; CHECK-NEXT: ret float [[V]]
+ ; CHECK-NOT: llvm.dx.load.input
+ %v = call float @llvm.dx.load.input.f32(i32 99, i32 0, i32 1, i8 2, i32 0)
+ ret float %v
+}
+
+; Vector float4 load: four per-component LoadInput calls reassembled into a vector.
+; CHECK-LABEL: define <4 x float> @load_v4f32
+define <4 x float> @load_v4f32() {
+ ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0)
+ ; CHECK-NEXT: insertelement <4 x float> {{.*}}, float [[S0]], i32 0
+ ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1)
+ ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
+ ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2)
+ ; CHECK-NEXT: insertelement {{.*}}, float [[S2]], i32 2
+ ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3)
+ ; CHECK-NEXT: insertelement {{.*}}, float [[S3]], i32 3
+ ; CHECK-NOT: llvm.dx.load.input
+ %v = call <4 x float> @llvm.dx.load.input.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0)
+ ret <4 x float> %v
+}
+
+; Vector float2 load with non-zero start column: col indices must be 2 and 3.
+; CHECK-LABEL: define <2 x float> @load_v2f32_col2
+define <2 x float> @load_v2f32_col2() {
+ ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2)
+ ; CHECK-NEXT: insertelement <2 x float> {{.*}}, float [[S0]], i32 0
+ ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3)
+ ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
+ ; CHECK-NOT: llvm.dx.load.input
+ %v = call <2 x float> @llvm.dx.load.input.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0)
+ ret <2 x float> %v
+}
+
+; Scalar int load: one LoadInput call, result forwarded directly.
+; CHECK-LABEL: define i32 @load_scalar_i32
+define i32 @load_scalar_i32() {
+ ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0)
+ ; CHECK-NEXT: ret i32 [[V]]
+ ; CHECK-NOT: llvm.dx.load.input
+ %v = call i32 @llvm.dx.load.input.i32(i32 99, i32 2, i32 0, i8 0, i32 0)
+ ret i32 %v
+}
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
new file mode 100644
index 0000000000000..7e6d356f2dee4
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -0,0 +1,49 @@
+; RUN: opt -S -dxil-intrinsic-expansion -dxil-op-lower %s | FileCheck %s
+
+target triple = "dxil-pc-shadermodel6.0-pixel"
+
+; Scalar float store: one StoreOutput call, no residual intrinsic.
+; CHECK-LABEL: define void @store_scalar_f32
+define void @store_scalar_f32(float %val) {
+ ; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
+ ; CHECK-NOT: llvm.dx.store.output
+ call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
+ ret void
+}
+
+; Vector float4 store: four per-component StoreOutput calls, col indices 0..3.
+; CHECK-LABEL: define void @store_v4f32
+define void @store_v4f32(<4 x float> %val) {
+ ; CHECK: [[E0:%.*]] = extractelement <4 x float> %val, i32 0
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 0, float [[E0]])
+ ; CHECK-NEXT: [[E1:%.*]] = extractelement <4 x float> %val, i32 1
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 1, float [[E1]])
+ ; CHECK-NEXT: [[E2:%.*]] = extractelement <4 x float> %val, i32 2
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 2, float [[E2]])
+ ; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
+ ; CHECK-NOT: llvm.dx.store.output
+ call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
+ ret void
+}
+
+; Vector float2 store with non-zero start column: col indices must be 2 and 3.
+; CHECK-LABEL: define void @store_v2f32_col2
+define void @store_v2f32_col2(<2 x float> %val) {
+ ; CHECK: [[E0:%.*]] = extractelement <2 x float> %val, i32 0
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 2, float [[E0]])
+ ; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
+ ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
+ ; CHECK-NOT: llvm.dx.store.output
+ call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
+ ret void
+}
+
+; Scalar int store: one StoreOutput call, no residual intrinsic.
+; CHECK-LABEL: define void @store_scalar_i32
+define void @store_scalar_i32(i32 %val) {
+ ; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
+ ; CHECK-NOT: llvm.dx.store.output
+ call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
+ ret void
+}
>From fba1577b3b6fa1e0d6eb474eeb8924d699014180 Mon Sep 17 00:00:00 2001
From: Dan Brown <61992655+danbrown-amd at users.noreply.github.com>
Date: Wed, 5 Aug 2026 12:39:55 -0600
Subject: [PATCH 2/5] Update llvm/include/llvm/IR/IntrinsicsDirectX.td
Co-authored-by: Finn Plummer <mail at inbelic.dev>
---
llvm/include/llvm/IR/IntrinsicsDirectX.td | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 876447b8c866a..62489b58a0300 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -335,6 +335,6 @@ def int_dx_store_output
[],
[llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
- llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
+ llvm_any_ty /*value*/],
[IntrConvergent]>;
}
>From a6137afaf99ab2626429af9a953f99b91a6ec48a Mon Sep 17 00:00:00 2001
From: Dan Brown <61992655+danbrown-amd at users.noreply.github.com>
Date: Wed, 5 Aug 2026 12:40:45 -0600
Subject: [PATCH 3/5] Update llvm/lib/Target/DirectX/DXIL.td
Co-authored-by: Finn Plummer <mail at inbelic.dev>
---
llvm/lib/Target/DirectX/DXIL.td | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index fb2648b6dea0b..73d19ae9867d0 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -416,8 +416,8 @@ def LoadInput : DXILOp<4, loadInput> {
let Doc = "Loads a scalar value from a shader input register component.";
let intrinsics = [IntrinSelect<int_dx_load_input,
[IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
- // inputSigId, rowIndex, colIndex
- let arguments = [Int32Ty, Int32Ty, Int8Ty];
+ // inputSigId, rowIndex, colIndex, gsVertexOrPrimIndex
+ let arguments = [Int32Ty, Int32Ty, Int8Ty, Int32Ty];
let result = OverloadTy;
let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
let stages = [Stages<DXIL1_0, [all_stages]>];
>From 9259c87ce3b62041e308fb4737d8efa0740753f1 Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 5 Aug 2026 16:49:15 -0600
Subject: [PATCH 4/5] Make tests and other source code consistent with
suggested changes
---
llvm/lib/Target/DirectX/DXIL.td | 5 +++--
.../Target/DirectX/DXILIntrinsicExpansion.cpp | 11 ++++-------
llvm/test/CodeGen/DirectX/LoadInput.ll | 16 ++++++++--------
llvm/test/CodeGen/DirectX/StoreOutput.ll | 8 ++++----
4 files changed, 19 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 73d19ae9867d0..842c1b48efecf 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -415,7 +415,8 @@ class DXILOp<int opcode, DXILOpClass opclass> {
def LoadInput : DXILOp<4, loadInput> {
let Doc = "Loads a scalar value from a shader input register component.";
let intrinsics = [IntrinSelect<int_dx_load_input,
- [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
+ [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
+ IntrinArgIndex<4>]>];
// inputSigId, rowIndex, colIndex, gsVertexOrPrimIndex
let arguments = [Int32Ty, Int32Ty, Int8Ty, Int32Ty];
let result = OverloadTy;
@@ -428,7 +429,7 @@ def StoreOutput : DXILOp<5, storeOutput> {
let Doc = "Stores a scalar value to a shader output register component.";
let intrinsics = [IntrinSelect<int_dx_store_output,
[IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
- IntrinArgIndex<5>]>];
+ IntrinArgIndex<4>]>];
// outputSigId, rowIndex, colIndex, value
let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
let result = VoidTy;
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index efd1212c6f154..c5b8b4efee956 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -1218,7 +1218,7 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
// The DXIL StoreOutput op is per-component; vector intrinsics are split here
// so that DXILOpLowering sees only scalar variants.
static bool expandStoreOutput(CallInst *Orig) {
- auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
+ auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(4)->getType());
if (!VT)
return false; // already scalar, nothing to expand
@@ -1233,8 +1233,7 @@ static bool expandStoreOutput(CallInst *Orig) {
Value *SigElementId = Orig->getArgOperand(1);
Value *RowIndex = Orig->getArgOperand(2);
Value *StartCol = Orig->getArgOperand(3); // i8
- Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
- Value *Data = Orig->getArgOperand(5);
+ Value *Data = Orig->getArgOperand(4);
Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
@@ -1246,8 +1245,8 @@ static bool expandStoreOutput(CallInst *Orig) {
Value *ColIdx =
Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
- Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
- GsVertexOrPrimIndex, Scalar});
+ Builder.CreateCall(ScalarFn,
+ {SigpointId, SigElementId, RowIndex, ColI8, Scalar});
}
Orig->eraseFromParent();
@@ -1268,8 +1267,6 @@ static Value *expandLoadInput(CallInst *Orig) {
Type *ScalarTy = VT->getElementType();
unsigned NumElems = VT->getNumElements();
- // Intrinsic args: (sigpointId, sigElementId, rowIndex, colIndex:i8,
- // gsVertexOrPrimIndex)
Value *SigpointId = Orig->getArgOperand(0);
Value *SigElementId = Orig->getArgOperand(1);
Value *RowIndex = Orig->getArgOperand(2);
diff --git a/llvm/test/CodeGen/DirectX/LoadInput.ll b/llvm/test/CodeGen/DirectX/LoadInput.ll
index d9d6aea6a457e..b29167d2dfb27 100644
--- a/llvm/test/CodeGen/DirectX/LoadInput.ll
+++ b/llvm/test/CodeGen/DirectX/LoadInput.ll
@@ -5,7 +5,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
; Scalar float load: one LoadInput call, result forwarded directly.
; CHECK-LABEL: define float @load_scalar_f32
define float @load_scalar_f32() {
- ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2)
+ ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2, i32 0)
; CHECK-NEXT: ret float [[V]]
; CHECK-NOT: llvm.dx.load.input
%v = call float @llvm.dx.load.input.f32(i32 99, i32 0, i32 1, i8 2, i32 0)
@@ -15,13 +15,13 @@ define float @load_scalar_f32() {
; Vector float4 load: four per-component LoadInput calls reassembled into a vector.
; CHECK-LABEL: define <4 x float> @load_v4f32
define <4 x float> @load_v4f32() {
- ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0)
+ ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0, i32 0)
; CHECK-NEXT: insertelement <4 x float> {{.*}}, float [[S0]], i32 0
- ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1)
+ ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1, i32 0)
; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
- ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2)
+ ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2, i32 0)
; CHECK-NEXT: insertelement {{.*}}, float [[S2]], i32 2
- ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3)
+ ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3, i32 0)
; CHECK-NEXT: insertelement {{.*}}, float [[S3]], i32 3
; CHECK-NOT: llvm.dx.load.input
%v = call <4 x float> @llvm.dx.load.input.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0)
@@ -31,9 +31,9 @@ define <4 x float> @load_v4f32() {
; Vector float2 load with non-zero start column: col indices must be 2 and 3.
; CHECK-LABEL: define <2 x float> @load_v2f32_col2
define <2 x float> @load_v2f32_col2() {
- ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2)
+ ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2, i32 0)
; CHECK-NEXT: insertelement <2 x float> {{.*}}, float [[S0]], i32 0
- ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3)
+ ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3, i32 0)
; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
; CHECK-NOT: llvm.dx.load.input
%v = call <2 x float> @llvm.dx.load.input.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0)
@@ -43,7 +43,7 @@ define <2 x float> @load_v2f32_col2() {
; Scalar int load: one LoadInput call, result forwarded directly.
; CHECK-LABEL: define i32 @load_scalar_i32
define i32 @load_scalar_i32() {
- ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0)
+ ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0, i32 0)
; CHECK-NEXT: ret i32 [[V]]
; CHECK-NOT: llvm.dx.load.input
%v = call i32 @llvm.dx.load.input.i32(i32 99, i32 2, i32 0, i8 0, i32 0)
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
index 7e6d356f2dee4..04970a1488710 100644
--- a/llvm/test/CodeGen/DirectX/StoreOutput.ll
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -7,7 +7,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
define void @store_scalar_f32(float %val) {
; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
+ call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, float %val)
ret void
}
@@ -23,7 +23,7 @@ define void @store_v4f32(<4 x float> %val) {
; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
+ call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, <4 x float> %val)
ret void
}
@@ -35,7 +35,7 @@ define void @store_v2f32_col2(<2 x float> %val) {
; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
+ call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, <2 x float> %val)
ret void
}
@@ -44,6 +44,6 @@ define void @store_v2f32_col2(<2 x float> %val) {
define void @store_scalar_i32(i32 %val) {
; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
+ call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 %val)
ret void
}
>From c7259fa2349368af9d2ce2303386672793c5b36e Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 5 Aug 2026 17:59:42 -0600
Subject: [PATCH 5/5] Undo llvm.dx.store.output argument change
---
llvm/include/llvm/IR/IntrinsicsDirectX.td | 2 +-
llvm/lib/Target/DirectX/DXIL.td | 2 +-
llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp | 9 +++++----
llvm/test/CodeGen/DirectX/StoreOutput.ll | 8 ++++----
4 files changed, 11 insertions(+), 10 deletions(-)
diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 62489b58a0300..876447b8c866a 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -335,6 +335,6 @@ def int_dx_store_output
[],
[llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
- llvm_any_ty /*value*/],
+ llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
[IntrConvergent]>;
}
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 842c1b48efecf..b4e53e2dcbb6a 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -429,7 +429,7 @@ def StoreOutput : DXILOp<5, storeOutput> {
let Doc = "Stores a scalar value to a shader output register component.";
let intrinsics = [IntrinSelect<int_dx_store_output,
[IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
- IntrinArgIndex<4>]>];
+ IntrinArgIndex<5>]>];
// outputSigId, rowIndex, colIndex, value
let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
let result = VoidTy;
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index c5b8b4efee956..1447c3db473f9 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -1218,7 +1218,7 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
// The DXIL StoreOutput op is per-component; vector intrinsics are split here
// so that DXILOpLowering sees only scalar variants.
static bool expandStoreOutput(CallInst *Orig) {
- auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(4)->getType());
+ auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
if (!VT)
return false; // already scalar, nothing to expand
@@ -1233,7 +1233,8 @@ static bool expandStoreOutput(CallInst *Orig) {
Value *SigElementId = Orig->getArgOperand(1);
Value *RowIndex = Orig->getArgOperand(2);
Value *StartCol = Orig->getArgOperand(3); // i8
- Value *Data = Orig->getArgOperand(4);
+ Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+ Value *Data = Orig->getArgOperand(5);
Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
@@ -1245,8 +1246,8 @@ static bool expandStoreOutput(CallInst *Orig) {
Value *ColIdx =
Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
- Builder.CreateCall(ScalarFn,
- {SigpointId, SigElementId, RowIndex, ColI8, Scalar});
+ Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+ GsVertexOrPrimIndex, Scalar});
}
Orig->eraseFromParent();
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
index 04970a1488710..7e6d356f2dee4 100644
--- a/llvm/test/CodeGen/DirectX/StoreOutput.ll
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -7,7 +7,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
define void @store_scalar_f32(float %val) {
; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, float %val)
+ call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
ret void
}
@@ -23,7 +23,7 @@ define void @store_v4f32(<4 x float> %val) {
; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, <4 x float> %val)
+ call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
ret void
}
@@ -35,7 +35,7 @@ define void @store_v2f32_col2(<2 x float> %val) {
; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, <2 x float> %val)
+ call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
ret void
}
@@ -44,6 +44,6 @@ define void @store_v2f32_col2(<2 x float> %val) {
define void @store_scalar_i32(i32 %val) {
; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
; CHECK-NOT: llvm.dx.store.output
- call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 %val)
+ call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
ret void
}
More information about the llvm-commits
mailing list