[llvm] [AMDGPU] Canonicalize constant operands of sudot intrinsics (PR #226478)

Harrison Hao via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 06:03:31 PDT 2026


https://github.com/harrisonGPU created https://github.com/llvm/llvm-project/pull/226478

Move a constant multiplicand and its sign flag to the second operand.

For example:
```
  sudot4(true, 1, false, x, acc, clamp)
  ->
  sudot4(false, x, true, 1, acc, clamp)
```

---

<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>

>From b6ad737fa948eb6647e9208e16bf5dffa3466112 Mon Sep 17 00:00:00 2001
From: Harrison Hao <tsworld1314 at gmail.com>
Date: Fri, 25 Sep 2026 20:57:38 +0800
Subject: [PATCH] [AMDGPU] Canonicalize constant operands of sudot intrinsics

Move a constant multiplicand and its sign flag to the second operand.

For example:
```
  sudot4(true, 1, false, x, acc, clamp)
  ->
  sudot4(false, x, true, 1, acc, clamp)
```
---
 .../Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp  | 15 +++++++++++++++
 .../InstCombine/AMDGPU/llvm.amdgcn.sudot.ll       |  8 ++++----
 2 files changed, 19 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 06d2b84411604..0e81d503c5fd2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -2021,6 +2021,21 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
   }
   case Intrinsic::amdgcn_sudot4:
   case Intrinsic::amdgcn_sudot8: {
+    Value *Src0 = II.getArgOperand(1);
+    Value *Src1 = II.getArgOperand(3);
+
+    // Canonicalize the constant multiplicand to Src1, moving its sign flag with
+    // it.
+    if (isa<Constant>(Src0) && !isa<Constant>(Src1)) {
+      Value *Sign0 = II.getArgOperand(0);
+      Value *Sign1 = II.getArgOperand(2);
+      II.setArgOperand(0, Sign1);
+      II.setArgOperand(1, Src1);
+      II.setArgOperand(2, Sign0);
+      II.setArgOperand(3, Src0);
+      return &II;
+    }
+
     if (Instruction *I = foldConstantIntoDotAccumulator(II, 4, 5, IC))
       return I;
 
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/llvm.amdgcn.sudot.ll b/llvm/test/Transforms/InstCombine/AMDGPU/llvm.amdgcn.sudot.ll
index 97a32749dac39..e99e6bbf00821 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/llvm.amdgcn.sudot.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/llvm.amdgcn.sudot.ll
@@ -26,7 +26,7 @@ define i32 @sudot4_sub(i32 %a, i32 %b) {
 define i32 @sudot4_const_lhs(i32 %b, i32 %acc) {
 ; CHECK-LABEL: define i32 @sudot4_const_lhs(
 ; CHECK-SAME: i32 [[B:%.*]], i32 [[ACC:%.*]]) {
-; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot4(i1 true, i32 16843009, i1 false, i32 [[B]], i32 [[ACC]], i1 false)
+; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot4(i1 false, i32 [[B]], i1 true, i32 16843009, i32 [[ACC]], i1 false)
 ; CHECK-NEXT:    ret i32 [[DOT]]
 ;
   %dot = call i32 @llvm.amdgcn.sudot4(i1 true, i32 16843009, i1 false, i32 %b, i32 %acc, i1 false)
@@ -36,7 +36,7 @@ define i32 @sudot4_const_lhs(i32 %b, i32 %acc) {
 define i32 @sudot4_a_zero(i32 %b, i32 %acc) {
 ; CHECK-LABEL: define i32 @sudot4_a_zero(
 ; CHECK-SAME: i32 [[B:%.*]], i32 [[ACC:%.*]]) {
-; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot4(i1 true, i32 0, i1 false, i32 [[B]], i32 [[ACC]], i1 false)
+; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot4(i1 false, i32 [[B]], i1 true, i32 0, i32 [[ACC]], i1 false)
 ; CHECK-NEXT:    ret i32 [[DOT]]
 ;
   %dot = call i32 @llvm.amdgcn.sudot4(i1 true, i32 0, i1 false, i32 %b, i32 %acc, i1 false)
@@ -316,7 +316,7 @@ define i32 @sudot8_sub(i32 %a, i32 %b) {
 define i32 @sudot8_const_lhs(i32 %b, i32 %acc) {
 ; CHECK-LABEL: define i32 @sudot8_const_lhs(
 ; CHECK-SAME: i32 [[B:%.*]], i32 [[ACC:%.*]]) {
-; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot8(i1 false, i32 286331153, i1 true, i32 [[B]], i32 [[ACC]], i1 false)
+; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot8(i1 true, i32 [[B]], i1 false, i32 286331153, i32 [[ACC]], i1 false)
 ; CHECK-NEXT:    ret i32 [[DOT]]
 ;
   %dot = call i32 @llvm.amdgcn.sudot8(i1 false, i32 286331153, i1 true, i32 %b, i32 %acc, i1 false)
@@ -326,7 +326,7 @@ define i32 @sudot8_const_lhs(i32 %b, i32 %acc) {
 define i32 @sudot8_a_zero(i32 %b, i32 %acc) {
 ; CHECK-LABEL: define i32 @sudot8_a_zero(
 ; CHECK-SAME: i32 [[B:%.*]], i32 [[ACC:%.*]]) {
-; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot8(i1 false, i32 0, i1 true, i32 [[B]], i32 [[ACC]], i1 false)
+; CHECK-NEXT:    [[DOT:%.*]] = call i32 @llvm.amdgcn.sudot8(i1 true, i32 [[B]], i1 false, i32 0, i32 [[ACC]], i1 false)
 ; CHECK-NEXT:    ret i32 [[DOT]]
 ;
   %dot = call i32 @llvm.amdgcn.sudot8(i1 false, i32 0, i1 true, i32 %b, i32 %acc, i1 false)



More information about the llvm-commits mailing list