[clang] [CIR] Add Matrix transpose operation (PR #227372)

Amr Hesham via cfe-commits cfe-commits at lists.llvm.org
Tue Sep 29 12:47:42 PDT 2026


https://github.com/AmrDeveloper updated https://github.com/llvm/llvm-project/pull/227372

>From 38e8ae9f0d5136ea1e8d819b73603d8c86a0c349 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 17:59:11 +0200
Subject: [PATCH 1/4] [CIR] Add Matrix transpose operation

---
 clang/include/clang/CIR/Dialect/IR/CIROps.td  | 27 ++++++++++++++
 clang/lib/CIR/CodeGen/CIRGenBuilder.h         |  8 +++++
 clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp       |  7 +++-
 clang/lib/CIR/Dialect/IR/CIRDialect.cpp       | 22 ++++++++++++
 .../CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp | 12 +++++++
 clang/test/CIR/CodeGen/matrix.cpp             | 36 +++++++++++++++++++
 clang/test/CIR/IR/invalid-matrix.cir          | 26 ++++++++++++++
 7 files changed, 137 insertions(+), 1 deletion(-)

diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index de805d4c642a2..f4f418e031ea1 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6287,6 +6287,33 @@ def CIR_VecSplatOp : CIR_Op<"vec.splat", [
   }];
 }
 
+//===----------------------------------------------------------------------===//
+// MatrixTransposeOp
+//===----------------------------------------------------------------------===//
+
+def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
+  Pure,
+]> {
+  let summary = "Matrix transpose";
+  let description = [{
+    The `cir.matrix.transpose` operation transposing a 2-D matrix.
+
+    ```
+    %result = cir.matrix.transpose %value : <3 x 2 x !cir.float>, 
+        !cir.matrix<3 x 3 x !cir.float>
+    ```
+  }];
+
+  let arguments = (ins CIR_MatrixType:$value);
+  let results = (outs CIR_MatrixType:$result);
+
+  let assemblyFormat = [{
+    $value `:` type($value) `,` qualified(type($result)) attr-dict
+  }];
+
+  let hasVerifier = 1;
+}
+
 //===----------------------------------------------------------------------===//
 // BaseClassAddrOp
 //===----------------------------------------------------------------------===//
diff --git a/clang/lib/CIR/CodeGen/CIRGenBuilder.h b/clang/lib/CIR/CodeGen/CIRGenBuilder.h
index b581212b0db56..ccd4ed403e3b6 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuilder.h
+++ b/clang/lib/CIR/CodeGen/CIRGenBuilder.h
@@ -825,6 +825,14 @@ class CIRGenBuilderTy : public cir::CIRBaseBuilderTy {
     return createVecShuffle(loc, vec1, poison, mask);
   }
 
+  cir::MatrixTransposeOp createMatrixTranspose(mlir::Location loc,
+                                               mlir::Value matrix) {
+    auto inputTy = mlir::cast<cir::MatrixType>(matrix.getType());
+    auto resultTy = cir::MatrixType::get(
+        inputTy.getElementType(), inputTy.getColumnNum(), inputTy.getRowNum());
+    return cir::MatrixTransposeOp::create(*this, loc, resultTy, matrix);
+  }
+
   template <typename... Operands>
   mlir::Value emitIntrinsicCallOp(mlir::Location loc, const llvm::StringRef str,
                                   const mlir::Type &resTy, Operands &&...op) {
diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp b/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
index 245708691b7d9..fe7a63d7bb85d 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
@@ -2260,7 +2260,12 @@ RValue CIRGenFunction::emitBuiltinExpr(const GlobalDecl &gd, unsigned builtinID,
   }
   case Builtin::BI__builtin_reduce_maximum:
   case Builtin::BI__builtin_reduce_minimum:
-  case Builtin::BI__builtin_matrix_transpose:
+    return errorBuiltinNYI(*this, e, builtinID);
+  case Builtin::BI__builtin_matrix_transpose: {
+    mlir::Value matrix = emitScalarExpr(e->getArg(0));
+    mlir::Value result = builder.createMatrixTranspose(loc, matrix);
+    return RValue::get(result);
+  }
   case Builtin::BI__builtin_matrix_column_major_load:
   case Builtin::BI__builtin_matrix_column_major_store:
   case Builtin::BI__builtin_masked_load:
diff --git a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
index d5a587ff6d81e..1a50ffcb037c7 100644
--- a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
+++ b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
@@ -4183,6 +4183,28 @@ OpFoldResult cir::VecTernaryOp::fold(FoldAdaptor adaptor) {
       vecTy, mlir::ArrayAttr::get(getContext(), elements));
 }
 
+//===----------------------------------------------------------------------===//
+// MatrixTransposeOp
+//===----------------------------------------------------------------------===//
+
+LogicalResult cir::MatrixTransposeOp::verify() {
+  cir::MatrixType valueTy = getValue().getType();
+  cir::MatrixType resultTy = getResult().getType();
+  if (valueTy.getElementType() != resultTy.getElementType()) {
+    emitOpError() << "operand type doesn't match the result type";
+    return failure();
+  }
+
+  if ((valueTy.getRowNum() != resultTy.getColumnNum()) ||
+      (valueTy.getColumnNum() != resultTy.getRowNum())) {
+    emitOpError()
+        << "result type doesn't match the transpose type of the operand type";
+    return failure();
+  }
+
+  return success();
+}
+
 //===----------------------------------------------------------------------===//
 // ComplexCreateOp
 //===----------------------------------------------------------------------===//
diff --git a/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp b/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
index 4d2a99d5015fe..b4561c1ef684b 100644
--- a/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
+++ b/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
@@ -5224,6 +5224,18 @@ mlir::LogicalResult CIRToLLVMVecTernaryOpLowering::matchAndRewrite(
   return mlir::success();
 }
 
+mlir::LogicalResult CIRToLLVMMatrixTransposeOpLowering::matchAndRewrite(
+    cir::MatrixTransposeOp op, OpAdaptor adaptor,
+    mlir::ConversionPatternRewriter &rewriter) const {
+  cir::MatrixType matrixTy = op.getValue().getType();
+  mlir::Type resultTy =
+      typeConverter->convertType(op->getResultTypes().front());
+  rewriter.replaceOpWithNewOp<mlir::LLVM::MatrixTransposeOp>(
+      +op, resultTy, adaptor.getValue(), matrixTy.getRowNum(),
+      matrixTy.getColumnNum());
+  return mlir::success();
+}
+
 mlir::LogicalResult CIRToLLVMComplexAddOpLowering::matchAndRewrite(
     cir::ComplexAddOp op, OpAdaptor adaptor,
     mlir::ConversionPatternRewriter &rewriter) const {
diff --git a/clang/test/CIR/CodeGen/matrix.cpp b/clang/test/CIR/CodeGen/matrix.cpp
index f71bf38c2a1a9..5fc26b134eb55 100644
--- a/clang/test/CIR/CodeGen/matrix.cpp
+++ b/clang/test/CIR/CodeGen/matrix.cpp
@@ -6,6 +6,8 @@
 // RUN: FileCheck --input-file=%t.ll %s -check-prefix=LLVM
 
 typedef float matrix3x3 __attribute__((matrix_type(3, 3)));
+typedef float matrix3x2 __attribute__((matrix_type(3, 2)));
+typedef float matrix2x3 __attribute__((matrix_type(2, 3)));
 
 matrix3x3 a;
 
@@ -48,3 +50,37 @@ void load_global_store_in_local() {
 // LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
 // LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr @a, align 4
 // LLVM: store <9 x float> %[[TMP_A]], ptr %[[B_ADDR]], align 4
+
+void builtin_matrix_transpose() {
+  matrix3x3 a;
+  matrix3x3 b = __builtin_matrix_transpose(a);
+}
+
+// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+
+// LLVM: %[[A_ADDR:.*]] = alloca [9 x float], align 4
+// LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
+// LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr %[[A_ADDR]], align 4
+// LLVM: %[[TRANSPOE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
+// LLVM: store <9 x float> %[[TRANSPOE]], ptr %[[B_ADDR]], align 4
+
+void builtin_matrix_transpose_different_sizes() {
+  matrix3x2 a;
+  matrix2x3 b = __builtin_matrix_transpose(a);
+}
+
+// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+
+// LLVM: %[[A_ADDR:.*]] = alloca [6 x float], align 4
+// LLVM: %[[B_ADDR:.*]] = alloca [6 x float], align 4
+// LLVM: %[[TMP_A:.*]] = load <6 x float>, ptr %[[A_ADDR]], align 4
+// LLVM: %[[TRANSPOE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
+// LLVM: store <6 x float> %[[TRANSPOE:.*]], ptr %[[B_ADDR]], align 4
diff --git a/clang/test/CIR/IR/invalid-matrix.cir b/clang/test/CIR/IR/invalid-matrix.cir
index 7c234dcdf9a96..e8a8e8aab66f8 100644
--- a/clang/test/CIR/IR/invalid-matrix.cir
+++ b/clang/test/CIR/IR/invalid-matrix.cir
@@ -41,3 +41,29 @@ cir.func @negative_column_number() {
   cir.return
 
 }
+
+// -----
+
+!s32i = !cir.int<s, 32>
+
+cir.func @builtin_tranpose_different_element_type() {
+  %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+  %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+  %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+  // expected-error at +1 {{operand type doesn't match the result type}}
+  %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !s32i>
+  cir.return
+} 
+
+// -----
+
+!s32i = !cir.int<s, 32>
+
+cir.func @builtin_tranpose_different_sizes() {
+  %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+  %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+  %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+  // expected-error at +1 {{result type doesn't match the transpose type of the operand type}}
+  %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+  cir.return
+} 

>From c556fea823ba046a600e9cbd84cd2b61f55b50fd Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 19:12:29 +0200
Subject: [PATCH 2/4] Address code review comments and improve diagnostic

---
 clang/include/clang/CIR/Dialect/IR/CIROps.td |  2 +-
 clang/lib/CIR/Dialect/IR/CIRDialect.cpp      | 13 ++++++-------
 clang/test/CIR/IR/invalid-matrix.cir         |  4 ++--
 3 files changed, 9 insertions(+), 10 deletions(-)

diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index f4f418e031ea1..e093a928340b5 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6300,7 +6300,7 @@ def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
 
     ```
     %result = cir.matrix.transpose %value : <3 x 2 x !cir.float>, 
-        !cir.matrix<3 x 3 x !cir.float>
+        !cir.matrix<2 x 3 x !cir.float>
     ```
   }];
 
diff --git a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
index 1a50ffcb037c7..b3409befbce58 100644
--- a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
+++ b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
@@ -4190,15 +4190,14 @@ OpFoldResult cir::VecTernaryOp::fold(FoldAdaptor adaptor) {
 LogicalResult cir::MatrixTransposeOp::verify() {
   cir::MatrixType valueTy = getValue().getType();
   cir::MatrixType resultTy = getResult().getType();
-  if (valueTy.getElementType() != resultTy.getElementType()) {
-    emitOpError() << "operand type doesn't match the result type";
-    return failure();
-  }
 
-  if ((valueTy.getRowNum() != resultTy.getColumnNum()) ||
+  if ((valueTy.getElementType() != resultTy.getElementType()) ||
+      (valueTy.getRowNum() != resultTy.getColumnNum()) ||
       (valueTy.getColumnNum() != resultTy.getRowNum())) {
-    emitOpError()
-        << "result type doesn't match the transpose type of the operand type";
+    auto expectedTy = cir::MatrixType::get(
+        valueTy.getElementType(), valueTy.getColumnNum(), valueTy.getRowNum());
+    emitOpError() << "operand type " << valueTy << " expects result type of "
+                  << expectedTy << " but got " << resultTy;
     return failure();
   }
 
diff --git a/clang/test/CIR/IR/invalid-matrix.cir b/clang/test/CIR/IR/invalid-matrix.cir
index e8a8e8aab66f8..a68bdecd26012 100644
--- a/clang/test/CIR/IR/invalid-matrix.cir
+++ b/clang/test/CIR/IR/invalid-matrix.cir
@@ -50,7 +50,7 @@ cir.func @builtin_tranpose_different_element_type() {
   %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
   %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
   %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
-  // expected-error at +1 {{operand type doesn't match the result type}}
+  // expected-error at +1 {{op operand type '!cir.matrix<3 x 2 x !cir.float>' expects result type of '!cir.matrix<2 x 3 x !cir.float>' but got '!cir.matrix<2 x 3 x !cir.int<s, 32>>}}
   %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !s32i>
   cir.return
 } 
@@ -63,7 +63,7 @@ cir.func @builtin_tranpose_different_sizes() {
   %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
   %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
   %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
-  // expected-error at +1 {{result type doesn't match the transpose type of the operand type}}
+  // expected-error at +1 {{op operand type '!cir.matrix<3 x 2 x !cir.float>' expects result type of '!cir.matrix<2 x 3 x !cir.float>' but got '!cir.matrix<3 x 3 x !cir.float>}}
   %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
   cir.return
 } 

>From 9171588378d1c666bca0e0863e1a1a2accffb8d1 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 19:54:18 +0200
Subject: [PATCH 3/4] Update MatrixTranspose op description

---
 clang/include/clang/CIR/Dialect/IR/CIROps.td | 7 ++++++-
 1 file changed, 6 insertions(+), 1 deletion(-)

diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index e093a928340b5..239bf5cfa706e 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6296,7 +6296,12 @@ def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
 ]> {
   let summary = "Matrix transpose";
   let description = [{
-    The `cir.matrix.transpose` operation transposing a 2-D matrix.
+    The `cir.matrix.transpose` operation provides a representation for the
+    `__builtin_matrix_transpose` builtin and corresponds to the
+    `llvm.matrix.transpose` intrinsic in LLVM IR.
+
+    This operation performs transposition to switch the row and column
+    indices of the matrix and returns the transposed matrix,
 
     ```
     %result = cir.matrix.transpose %value : <3 x 2 x !cir.float>, 

>From 19e42165084fd8e7b80c8b5aba599e172414fc06 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 21:42:34 +0200
Subject: [PATCH 4/4] Fix variables names in LIT tests

---
 clang/test/CIR/CodeGen/matrix.cpp | 16 ++++++++--------
 1 file changed, 8 insertions(+), 8 deletions(-)

diff --git a/clang/test/CIR/CodeGen/matrix.cpp b/clang/test/CIR/CodeGen/matrix.cpp
index 5fc26b134eb55..9b8634f2e8247 100644
--- a/clang/test/CIR/CodeGen/matrix.cpp
+++ b/clang/test/CIR/CodeGen/matrix.cpp
@@ -59,14 +59,14 @@ void builtin_matrix_transpose() {
 // CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
 // CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
 // CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>, !cir.matrix<3 x 3 x !cir.float>
-// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
-// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[TRANSPOSE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOSE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
 
 // LLVM: %[[A_ADDR:.*]] = alloca [9 x float], align 4
 // LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
 // LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr %[[A_ADDR]], align 4
-// LLVM: %[[TRANSPOE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
-// LLVM: store <9 x float> %[[TRANSPOE]], ptr %[[B_ADDR]], align 4
+// LLVM: %[[TRANSPOSE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
+// LLVM: store <9 x float> %[[TRANSPOSE]], ptr %[[B_ADDR]], align 4
 
 void builtin_matrix_transpose_different_sizes() {
   matrix3x2 a;
@@ -76,11 +76,11 @@ void builtin_matrix_transpose_different_sizes() {
 // CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
 // CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
 // CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
-// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
-// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+// CIR: %[[TRANSPOSE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOSE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
 
 // LLVM: %[[A_ADDR:.*]] = alloca [6 x float], align 4
 // LLVM: %[[B_ADDR:.*]] = alloca [6 x float], align 4
 // LLVM: %[[TMP_A:.*]] = load <6 x float>, ptr %[[A_ADDR]], align 4
-// LLVM: %[[TRANSPOE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
-// LLVM: store <6 x float> %[[TRANSPOE:.*]], ptr %[[B_ADDR]], align 4
+// LLVM: %[[TRANSPOSE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
+// LLVM: store <6 x float> %[[TRANSPOSE:.*]], ptr %[[B_ADDR]], align 4



More information about the cfe-commits mailing list