[llvm] [HLSL][DirectX] Lower llvm.dx.load.input and llvm.dx.store.output to DXIL (PR #208871)

Dan Brown via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 5 17:04:29 PDT 2026


https://github.com/danbrown-amd updated https://github.com/llvm/llvm-project/pull/208871

>From 9ec1e174f976c081ca8a7b887fb61f53f0fee071 Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 8 Jul 2026 18:56:41 -0600
Subject: [PATCH 1/5] [HLSL][DirectX] Lower llvm.dx.load.input and
 llvm.dx.store.output to DXIL

Addresses #189766.

Assisted-by: Claude Sonnet 4
---
 llvm/include/llvm/IR/IntrinsicsDirectX.td     | 20 +++--
 llvm/lib/Target/DirectX/DXIL.td               | 24 +++++
 .../Target/DirectX/DXILIntrinsicExpansion.cpp | 90 +++++++++++++++++++
 llvm/test/CodeGen/DirectX/LoadInput.ll        | 51 +++++++++++
 llvm/test/CodeGen/DirectX/StoreOutput.ll      | 49 ++++++++++
 5 files changed, 226 insertions(+), 8 deletions(-)
 create mode 100644 llvm/test/CodeGen/DirectX/LoadInput.ll
 create mode 100644 llvm/test/CodeGen/DirectX/StoreOutput.ll

diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 090656ffb36c8..876447b8c866a 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -323,14 +323,18 @@ def int_dx_group_memory_barrier_with_group_sync
     : DefaultAttrsIntrinsic<[], [], [IntrConvergent]>;
 
 def int_dx_load_input
-    : DefaultAttrsIntrinsic<[llvm_any_ty],
-                            [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i8_ty,
-                             llvm_i32_ty],
-                            [IntrConvergent]>;
+    : DefaultAttrsIntrinsic<
+          [llvm_any_ty],
+          [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
+           llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
+           llvm_i32_ty /*gsVertexOrPrimIndex*/],
+          [IntrConvergent]>;
 
 def int_dx_store_output
-    : DefaultAttrsIntrinsic<[],
-                            [llvm_i32_ty, llvm_i32_ty, llvm_i32_ty, llvm_i8_ty,
-                             llvm_i32_ty, llvm_any_ty],
-                            [IntrConvergent]>;
+    : DefaultAttrsIntrinsic<
+          [],
+          [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
+           llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
+           llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
+          [IntrConvergent]>;
 }
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 3d978c207f104..fb2648b6dea0b 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -412,6 +412,30 @@ class DXILOp<int opcode, DXILOpClass opclass> {
 //
 // This are sorted by ascending value of the DXIL Opcodes
 
+def LoadInput : DXILOp<4, loadInput> {
+  let Doc = "Loads a scalar value from a shader input register component.";
+  let intrinsics = [IntrinSelect<int_dx_load_input,
+      [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
+  // inputSigId, rowIndex, colIndex
+  let arguments = [Int32Ty, Int32Ty, Int8Ty];
+  let result = OverloadTy;
+  let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
+  let stages = [Stages<DXIL1_0, [all_stages]>];
+  let attributes = [Attributes<DXIL1_0, [ReadOnly]>];
+}
+
+def StoreOutput : DXILOp<5, storeOutput> {
+  let Doc = "Stores a scalar value to a shader output register component.";
+  let intrinsics = [IntrinSelect<int_dx_store_output,
+      [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
+       IntrinArgIndex<5>]>];
+  // outputSigId, rowIndex, colIndex, value
+  let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
+  let result = VoidTy;
+  let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
+  let stages = [Stages<DXIL1_0, [all_stages]>];
+}
+
 def Abs : DXILOp<6, unary> {
   let Doc = "Returns the absolute value of the input.";
   let intrinsics = [IntrinSelect<int_fabs>];
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index a251288a6ae42..efd1212c6f154 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -234,6 +234,8 @@ static bool isIntrinsicExpansion(Function &F) {
   case Intrinsic::matrix_transpose:
   case Intrinsic::umul_with_overflow:
   case Intrinsic::smul_with_overflow:
+  case Intrinsic::dx_load_input:
+  case Intrinsic::dx_store_output:
     return true;
   case Intrinsic::dx_resource_load_rawbuffer:
     return resourceAccessNeeds64BitExpansion(
@@ -1212,6 +1214,87 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
   return Builder.CreateShuffleVector(Mat, Mask);
 }
 
+// Scalarize a vector int_dx_store_output call into per-component scalar calls.
+// The DXIL StoreOutput op is per-component; vector intrinsics are split here
+// so that DXILOpLowering sees only scalar variants.
+static bool expandStoreOutput(CallInst *Orig) {
+  auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
+  if (!VT)
+    return false; // already scalar, nothing to expand
+
+  IRBuilder<> Builder(Orig);
+  Module *M = Orig->getModule();
+  Type *Int8Ty = Builder.getInt8Ty();
+  Type *Int32Ty = Builder.getInt32Ty();
+  Type *ScalarTy = VT->getElementType();
+  unsigned NumElems = VT->getNumElements();
+
+  Value *SigpointId = Orig->getArgOperand(0);
+  Value *SigElementId = Orig->getArgOperand(1);
+  Value *RowIndex = Orig->getArgOperand(2);
+  Value *StartCol = Orig->getArgOperand(3); // i8
+  Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+  Value *Data = Orig->getArgOperand(5);
+  Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
+
+  Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
+      M, Intrinsic::dx_store_output, {ScalarTy});
+
+  for (unsigned I = 0; I < NumElems; ++I) {
+    Value *Scalar =
+        Builder.CreateExtractElement(Data, ConstantInt::get(Int32Ty, I));
+    Value *ColIdx =
+        Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
+    Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
+    Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+                                  GsVertexOrPrimIndex, Scalar});
+  }
+
+  Orig->eraseFromParent();
+  return true;
+}
+
+// Scalarize a vector int_dx_load_input call into per-component scalar calls
+// and reassemble the vector. The DXIL LoadInput op is per-component.
+static Value *expandLoadInput(CallInst *Orig) {
+  auto *VT = dyn_cast<FixedVectorType>(Orig->getType());
+  if (!VT)
+    return nullptr; // already scalar, nothing to expand
+
+  IRBuilder<> Builder(Orig);
+  Module *M = Orig->getModule();
+  Type *Int8Ty = Builder.getInt8Ty();
+  Type *Int32Ty = Builder.getInt32Ty();
+  Type *ScalarTy = VT->getElementType();
+  unsigned NumElems = VT->getNumElements();
+
+  // Intrinsic args: (sigpointId, sigElementId, rowIndex, colIndex:i8,
+  //                  gsVertexOrPrimIndex)
+  Value *SigpointId = Orig->getArgOperand(0);
+  Value *SigElementId = Orig->getArgOperand(1);
+  Value *RowIndex = Orig->getArgOperand(2);
+  Value *StartCol = Orig->getArgOperand(3); // i8
+  Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+  Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
+
+  Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
+      M, Intrinsic::dx_load_input, {ScalarTy});
+
+  Value *Vec = PoisonValue::get(VT);
+  for (unsigned I = 0; I < NumElems; ++I) {
+    Value *ColIdx =
+        Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
+    Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
+    Value *Scalar =
+        Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+                                      GsVertexOrPrimIndex});
+    Vec =
+        Builder.CreateInsertElement(Vec, Scalar, ConstantInt::get(Int32Ty, I));
+  }
+
+  return Vec;
+}
+
 static bool expandIntrinsic(Function &F, CallInst *Orig) {
   Value *Result = nullptr;
   Intrinsic::ID IntrinsicId = F.getIntrinsicID();
@@ -1287,6 +1370,13 @@ static bool expandIntrinsic(Function &F, CallInst *Orig) {
   case Intrinsic::dx_radians:
     Result = expandRadiansIntrinsic(Orig);
     break;
+  case Intrinsic::dx_load_input:
+    Result = expandLoadInput(Orig);
+    break;
+  case Intrinsic::dx_store_output:
+    if (expandStoreOutput(Orig))
+      return true;
+    break;
   case Intrinsic::dx_resource_load_rawbuffer:
     if (expandBufferLoadIntrinsic(Orig, /*IsRaw*/ true))
       return true;
diff --git a/llvm/test/CodeGen/DirectX/LoadInput.ll b/llvm/test/CodeGen/DirectX/LoadInput.ll
new file mode 100644
index 0000000000000..d9d6aea6a457e
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/LoadInput.ll
@@ -0,0 +1,51 @@
+; RUN: opt -S -dxil-intrinsic-expansion -dxil-op-lower %s | FileCheck %s
+
+target triple = "dxil-pc-shadermodel6.0-pixel"
+
+; Scalar float load: one LoadInput call, result forwarded directly.
+; CHECK-LABEL: define float @load_scalar_f32
+define float @load_scalar_f32() {
+  ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2)
+  ; CHECK-NEXT: ret float [[V]]
+  ; CHECK-NOT: llvm.dx.load.input
+  %v = call float @llvm.dx.load.input.f32(i32 99, i32 0, i32 1, i8 2, i32 0)
+  ret float %v
+}
+
+; Vector float4 load: four per-component LoadInput calls reassembled into a vector.
+; CHECK-LABEL: define <4 x float> @load_v4f32
+define <4 x float> @load_v4f32() {
+  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0)
+  ; CHECK-NEXT: insertelement <4 x float> {{.*}}, float [[S0]], i32 0
+  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1)
+  ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
+  ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2)
+  ; CHECK-NEXT: insertelement {{.*}}, float [[S2]], i32 2
+  ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3)
+  ; CHECK-NEXT: insertelement {{.*}}, float [[S3]], i32 3
+  ; CHECK-NOT: llvm.dx.load.input
+  %v = call <4 x float> @llvm.dx.load.input.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0)
+  ret <4 x float> %v
+}
+
+; Vector float2 load with non-zero start column: col indices must be 2 and 3.
+; CHECK-LABEL: define <2 x float> @load_v2f32_col2
+define <2 x float> @load_v2f32_col2() {
+  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2)
+  ; CHECK-NEXT: insertelement <2 x float> {{.*}}, float [[S0]], i32 0
+  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3)
+  ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
+  ; CHECK-NOT: llvm.dx.load.input
+  %v = call <2 x float> @llvm.dx.load.input.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0)
+  ret <2 x float> %v
+}
+
+; Scalar int load: one LoadInput call, result forwarded directly.
+; CHECK-LABEL: define i32 @load_scalar_i32
+define i32 @load_scalar_i32() {
+  ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0)
+  ; CHECK-NEXT: ret i32 [[V]]
+  ; CHECK-NOT: llvm.dx.load.input
+  %v = call i32 @llvm.dx.load.input.i32(i32 99, i32 2, i32 0, i8 0, i32 0)
+  ret i32 %v
+}
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
new file mode 100644
index 0000000000000..7e6d356f2dee4
--- /dev/null
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -0,0 +1,49 @@
+; RUN: opt -S -dxil-intrinsic-expansion -dxil-op-lower %s | FileCheck %s
+
+target triple = "dxil-pc-shadermodel6.0-pixel"
+
+; Scalar float store: one StoreOutput call, no residual intrinsic.
+; CHECK-LABEL: define void @store_scalar_f32
+define void @store_scalar_f32(float %val) {
+  ; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
+  ; CHECK-NOT: llvm.dx.store.output
+  call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
+  ret void
+}
+
+; Vector float4 store: four per-component StoreOutput calls, col indices 0..3.
+; CHECK-LABEL: define void @store_v4f32
+define void @store_v4f32(<4 x float> %val) {
+  ; CHECK: [[E0:%.*]] = extractelement <4 x float> %val, i32 0
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 0, float [[E0]])
+  ; CHECK-NEXT: [[E1:%.*]] = extractelement <4 x float> %val, i32 1
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 1, float [[E1]])
+  ; CHECK-NEXT: [[E2:%.*]] = extractelement <4 x float> %val, i32 2
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 2, float [[E2]])
+  ; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
+  ; CHECK-NOT: llvm.dx.store.output
+  call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
+  ret void
+}
+
+; Vector float2 store with non-zero start column: col indices must be 2 and 3.
+; CHECK-LABEL: define void @store_v2f32_col2
+define void @store_v2f32_col2(<2 x float> %val) {
+  ; CHECK: [[E0:%.*]] = extractelement <2 x float> %val, i32 0
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 2, float [[E0]])
+  ; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
+  ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
+  ; CHECK-NOT: llvm.dx.store.output
+  call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
+  ret void
+}
+
+; Scalar int store: one StoreOutput call, no residual intrinsic.
+; CHECK-LABEL: define void @store_scalar_i32
+define void @store_scalar_i32(i32 %val) {
+  ; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
+  ; CHECK-NOT: llvm.dx.store.output
+  call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
+  ret void
+}

>From fba1577b3b6fa1e0d6eb474eeb8924d699014180 Mon Sep 17 00:00:00 2001
From: Dan Brown <61992655+danbrown-amd at users.noreply.github.com>
Date: Wed, 5 Aug 2026 12:39:55 -0600
Subject: [PATCH 2/5] Update llvm/include/llvm/IR/IntrinsicsDirectX.td

Co-authored-by: Finn Plummer <mail at inbelic.dev>
---
 llvm/include/llvm/IR/IntrinsicsDirectX.td | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 876447b8c866a..62489b58a0300 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -335,6 +335,6 @@ def int_dx_store_output
           [],
           [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
            llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
-           llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
+           llvm_any_ty /*value*/],
           [IntrConvergent]>;
 }

>From a6137afaf99ab2626429af9a953f99b91a6ec48a Mon Sep 17 00:00:00 2001
From: Dan Brown <61992655+danbrown-amd at users.noreply.github.com>
Date: Wed, 5 Aug 2026 12:40:45 -0600
Subject: [PATCH 3/5] Update llvm/lib/Target/DirectX/DXIL.td

Co-authored-by: Finn Plummer <mail at inbelic.dev>
---
 llvm/lib/Target/DirectX/DXIL.td | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index fb2648b6dea0b..73d19ae9867d0 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -416,8 +416,8 @@ def LoadInput : DXILOp<4, loadInput> {
   let Doc = "Loads a scalar value from a shader input register component.";
   let intrinsics = [IntrinSelect<int_dx_load_input,
       [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
-  // inputSigId, rowIndex, colIndex
-  let arguments = [Int32Ty, Int32Ty, Int8Ty];
+  // inputSigId, rowIndex, colIndex, gsVertexOrPrimIndex
+  let arguments = [Int32Ty, Int32Ty, Int8Ty, Int32Ty];
   let result = OverloadTy;
   let overloads = [Overloads<DXIL1_0, [HalfTy, FloatTy, Int16Ty, Int32Ty]>];
   let stages = [Stages<DXIL1_0, [all_stages]>];

>From 9259c87ce3b62041e308fb4737d8efa0740753f1 Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 5 Aug 2026 16:49:15 -0600
Subject: [PATCH 4/5] Make tests and other source code consistent with
 suggested changes

---
 llvm/lib/Target/DirectX/DXIL.td                  |  5 +++--
 .../Target/DirectX/DXILIntrinsicExpansion.cpp    | 11 ++++-------
 llvm/test/CodeGen/DirectX/LoadInput.ll           | 16 ++++++++--------
 llvm/test/CodeGen/DirectX/StoreOutput.ll         |  8 ++++----
 4 files changed, 19 insertions(+), 21 deletions(-)

diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 73d19ae9867d0..842c1b48efecf 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -415,7 +415,8 @@ class DXILOp<int opcode, DXILOpClass opclass> {
 def LoadInput : DXILOp<4, loadInput> {
   let Doc = "Loads a scalar value from a shader input register component.";
   let intrinsics = [IntrinSelect<int_dx_load_input,
-      [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>]>];
+      [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
+       IntrinArgIndex<4>]>];
   // inputSigId, rowIndex, colIndex, gsVertexOrPrimIndex
   let arguments = [Int32Ty, Int32Ty, Int8Ty, Int32Ty];
   let result = OverloadTy;
@@ -428,7 +429,7 @@ def StoreOutput : DXILOp<5, storeOutput> {
   let Doc = "Stores a scalar value to a shader output register component.";
   let intrinsics = [IntrinSelect<int_dx_store_output,
       [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
-       IntrinArgIndex<5>]>];
+       IntrinArgIndex<4>]>];
   // outputSigId, rowIndex, colIndex, value
   let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
   let result = VoidTy;
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index efd1212c6f154..c5b8b4efee956 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -1218,7 +1218,7 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
 // The DXIL StoreOutput op is per-component; vector intrinsics are split here
 // so that DXILOpLowering sees only scalar variants.
 static bool expandStoreOutput(CallInst *Orig) {
-  auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
+  auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(4)->getType());
   if (!VT)
     return false; // already scalar, nothing to expand
 
@@ -1233,8 +1233,7 @@ static bool expandStoreOutput(CallInst *Orig) {
   Value *SigElementId = Orig->getArgOperand(1);
   Value *RowIndex = Orig->getArgOperand(2);
   Value *StartCol = Orig->getArgOperand(3); // i8
-  Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
-  Value *Data = Orig->getArgOperand(5);
+  Value *Data = Orig->getArgOperand(4);
   Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
 
   Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
@@ -1246,8 +1245,8 @@ static bool expandStoreOutput(CallInst *Orig) {
     Value *ColIdx =
         Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
     Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
-    Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
-                                  GsVertexOrPrimIndex, Scalar});
+    Builder.CreateCall(ScalarFn,
+                       {SigpointId, SigElementId, RowIndex, ColI8, Scalar});
   }
 
   Orig->eraseFromParent();
@@ -1268,8 +1267,6 @@ static Value *expandLoadInput(CallInst *Orig) {
   Type *ScalarTy = VT->getElementType();
   unsigned NumElems = VT->getNumElements();
 
-  // Intrinsic args: (sigpointId, sigElementId, rowIndex, colIndex:i8,
-  //                  gsVertexOrPrimIndex)
   Value *SigpointId = Orig->getArgOperand(0);
   Value *SigElementId = Orig->getArgOperand(1);
   Value *RowIndex = Orig->getArgOperand(2);
diff --git a/llvm/test/CodeGen/DirectX/LoadInput.ll b/llvm/test/CodeGen/DirectX/LoadInput.ll
index d9d6aea6a457e..b29167d2dfb27 100644
--- a/llvm/test/CodeGen/DirectX/LoadInput.ll
+++ b/llvm/test/CodeGen/DirectX/LoadInput.ll
@@ -5,7 +5,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
 ; Scalar float load: one LoadInput call, result forwarded directly.
 ; CHECK-LABEL: define float @load_scalar_f32
 define float @load_scalar_f32() {
-  ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2)
+  ; CHECK: [[V:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 0, i32 1, i8 2, i32 0)
   ; CHECK-NEXT: ret float [[V]]
   ; CHECK-NOT: llvm.dx.load.input
   %v = call float @llvm.dx.load.input.f32(i32 99, i32 0, i32 1, i8 2, i32 0)
@@ -15,13 +15,13 @@ define float @load_scalar_f32() {
 ; Vector float4 load: four per-component LoadInput calls reassembled into a vector.
 ; CHECK-LABEL: define <4 x float> @load_v4f32
 define <4 x float> @load_v4f32() {
-  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0)
+  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 0, i32 0)
   ; CHECK-NEXT: insertelement <4 x float> {{.*}}, float [[S0]], i32 0
-  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1)
+  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 1, i32 0)
   ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
-  ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2)
+  ; CHECK: [[S2:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 2, i32 0)
   ; CHECK-NEXT: insertelement {{.*}}, float [[S2]], i32 2
-  ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3)
+  ; CHECK: [[S3:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 1, i32 0, i8 3, i32 0)
   ; CHECK-NEXT: insertelement {{.*}}, float [[S3]], i32 3
   ; CHECK-NOT: llvm.dx.load.input
   %v = call <4 x float> @llvm.dx.load.input.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0)
@@ -31,9 +31,9 @@ define <4 x float> @load_v4f32() {
 ; Vector float2 load with non-zero start column: col indices must be 2 and 3.
 ; CHECK-LABEL: define <2 x float> @load_v2f32_col2
 define <2 x float> @load_v2f32_col2() {
-  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2)
+  ; CHECK: [[S0:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 2, i32 0)
   ; CHECK-NEXT: insertelement <2 x float> {{.*}}, float [[S0]], i32 0
-  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3)
+  ; CHECK: [[S1:%.*]] = call float @dx.op.loadInput.f32(i32 4, i32 2, i32 0, i8 3, i32 0)
   ; CHECK-NEXT: insertelement {{.*}}, float [[S1]], i32 1
   ; CHECK-NOT: llvm.dx.load.input
   %v = call <2 x float> @llvm.dx.load.input.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0)
@@ -43,7 +43,7 @@ define <2 x float> @load_v2f32_col2() {
 ; Scalar int load: one LoadInput call, result forwarded directly.
 ; CHECK-LABEL: define i32 @load_scalar_i32
 define i32 @load_scalar_i32() {
-  ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0)
+  ; CHECK: [[V:%.*]] = call i32 @dx.op.loadInput.i32(i32 4, i32 2, i32 0, i8 0, i32 0)
   ; CHECK-NEXT: ret i32 [[V]]
   ; CHECK-NOT: llvm.dx.load.input
   %v = call i32 @llvm.dx.load.input.i32(i32 99, i32 2, i32 0, i8 0, i32 0)
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
index 7e6d356f2dee4..04970a1488710 100644
--- a/llvm/test/CodeGen/DirectX/StoreOutput.ll
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -7,7 +7,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
 define void @store_scalar_f32(float %val) {
   ; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
+  call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, float %val)
   ret void
 }
 
@@ -23,7 +23,7 @@ define void @store_v4f32(<4 x float> %val) {
   ; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
   ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
+  call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, <4 x float> %val)
   ret void
 }
 
@@ -35,7 +35,7 @@ define void @store_v2f32_col2(<2 x float> %val) {
   ; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
   ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
+  call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, <2 x float> %val)
   ret void
 }
 
@@ -44,6 +44,6 @@ define void @store_v2f32_col2(<2 x float> %val) {
 define void @store_scalar_i32(i32 %val) {
   ; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
+  call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 %val)
   ret void
 }

>From c7259fa2349368af9d2ce2303386672793c5b36e Mon Sep 17 00:00:00 2001
From: danbrown-amd <danbrown at amd.com>
Date: Wed, 5 Aug 2026 17:59:42 -0600
Subject: [PATCH 5/5] Undo llvm.dx.store.output argument change

---
 llvm/include/llvm/IR/IntrinsicsDirectX.td          | 2 +-
 llvm/lib/Target/DirectX/DXIL.td                    | 2 +-
 llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp | 9 +++++----
 llvm/test/CodeGen/DirectX/StoreOutput.ll           | 8 ++++----
 4 files changed, 11 insertions(+), 10 deletions(-)

diff --git a/llvm/include/llvm/IR/IntrinsicsDirectX.td b/llvm/include/llvm/IR/IntrinsicsDirectX.td
index 62489b58a0300..876447b8c866a 100644
--- a/llvm/include/llvm/IR/IntrinsicsDirectX.td
+++ b/llvm/include/llvm/IR/IntrinsicsDirectX.td
@@ -335,6 +335,6 @@ def int_dx_store_output
           [],
           [llvm_i32_ty /*sigpointId*/, llvm_i32_ty /*sigElementId*/,
            llvm_i32_ty /*rowIndex*/, llvm_i8_ty /*colIndex*/,
-           llvm_any_ty /*value*/],
+           llvm_i32_ty /*gsVertexOrPrimIndex*/, llvm_any_ty /*value*/],
           [IntrConvergent]>;
 }
diff --git a/llvm/lib/Target/DirectX/DXIL.td b/llvm/lib/Target/DirectX/DXIL.td
index 842c1b48efecf..b4e53e2dcbb6a 100644
--- a/llvm/lib/Target/DirectX/DXIL.td
+++ b/llvm/lib/Target/DirectX/DXIL.td
@@ -429,7 +429,7 @@ def StoreOutput : DXILOp<5, storeOutput> {
   let Doc = "Stores a scalar value to a shader output register component.";
   let intrinsics = [IntrinSelect<int_dx_store_output,
       [IntrinArgIndex<1>, IntrinArgIndex<2>, IntrinArgIndex<3>,
-       IntrinArgIndex<4>]>];
+       IntrinArgIndex<5>]>];
   // outputSigId, rowIndex, colIndex, value
   let arguments = [Int32Ty, Int32Ty, Int8Ty, OverloadTy];
   let result = VoidTy;
diff --git a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
index c5b8b4efee956..1447c3db473f9 100644
--- a/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
+++ b/llvm/lib/Target/DirectX/DXILIntrinsicExpansion.cpp
@@ -1218,7 +1218,7 @@ static Value *expandMatrixTranspose(CallInst *Orig) {
 // The DXIL StoreOutput op is per-component; vector intrinsics are split here
 // so that DXILOpLowering sees only scalar variants.
 static bool expandStoreOutput(CallInst *Orig) {
-  auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(4)->getType());
+  auto *VT = dyn_cast<FixedVectorType>(Orig->getArgOperand(5)->getType());
   if (!VT)
     return false; // already scalar, nothing to expand
 
@@ -1233,7 +1233,8 @@ static bool expandStoreOutput(CallInst *Orig) {
   Value *SigElementId = Orig->getArgOperand(1);
   Value *RowIndex = Orig->getArgOperand(2);
   Value *StartCol = Orig->getArgOperand(3); // i8
-  Value *Data = Orig->getArgOperand(4);
+  Value *GsVertexOrPrimIndex = Orig->getArgOperand(4);
+  Value *Data = Orig->getArgOperand(5);
   Value *StartColI32 = Builder.CreateZExt(StartCol, Int32Ty);
 
   Function *ScalarFn = Intrinsic::getOrInsertDeclaration(
@@ -1245,8 +1246,8 @@ static bool expandStoreOutput(CallInst *Orig) {
     Value *ColIdx =
         Builder.CreateAdd(StartColI32, ConstantInt::get(Int32Ty, I));
     Value *ColI8 = Builder.CreateTrunc(ColIdx, Int8Ty);
-    Builder.CreateCall(ScalarFn,
-                       {SigpointId, SigElementId, RowIndex, ColI8, Scalar});
+    Builder.CreateCall(ScalarFn, {SigpointId, SigElementId, RowIndex, ColI8,
+                                  GsVertexOrPrimIndex, Scalar});
   }
 
   Orig->eraseFromParent();
diff --git a/llvm/test/CodeGen/DirectX/StoreOutput.ll b/llvm/test/CodeGen/DirectX/StoreOutput.ll
index 04970a1488710..7e6d356f2dee4 100644
--- a/llvm/test/CodeGen/DirectX/StoreOutput.ll
+++ b/llvm/test/CodeGen/DirectX/StoreOutput.ll
@@ -7,7 +7,7 @@ target triple = "dxil-pc-shadermodel6.0-pixel"
 define void @store_scalar_f32(float %val) {
   ; CHECK: call void @dx.op.storeOutput.f32(i32 5, i32 0, i32 1, i8 2, float %val)
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, float %val)
+  call void @llvm.dx.store.output.f32(i32 99, i32 0, i32 1, i8 2, i32 0, float %val)
   ret void
 }
 
@@ -23,7 +23,7 @@ define void @store_v4f32(<4 x float> %val) {
   ; CHECK-NEXT: [[E3:%.*]] = extractelement <4 x float> %val, i32 3
   ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 1, i32 0, i8 3, float [[E3]])
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, <4 x float> %val)
+  call void @llvm.dx.store.output.v4f32(i32 99, i32 1, i32 0, i8 0, i32 0, <4 x float> %val)
   ret void
 }
 
@@ -35,7 +35,7 @@ define void @store_v2f32_col2(<2 x float> %val) {
   ; CHECK-NEXT: [[E1:%.*]] = extractelement <2 x float> %val, i32 1
   ; CHECK-NEXT: call void @dx.op.storeOutput.f32(i32 5, i32 2, i32 0, i8 3, float [[E1]])
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, <2 x float> %val)
+  call void @llvm.dx.store.output.v2f32(i32 99, i32 2, i32 0, i8 2, i32 0, <2 x float> %val)
   ret void
 }
 
@@ -44,6 +44,6 @@ define void @store_v2f32_col2(<2 x float> %val) {
 define void @store_scalar_i32(i32 %val) {
   ; CHECK: call void @dx.op.storeOutput.i32(i32 5, i32 2, i32 0, i8 0, i32 %val)
   ; CHECK-NOT: llvm.dx.store.output
-  call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 %val)
+  call void @llvm.dx.store.output.i32(i32 99, i32 2, i32 0, i8 0, i32 0, i32 %val)
   ret void
 }



More information about the llvm-commits mailing list