[clang] [CIR] Add Matrix transpose operation (PR #227372)
Amr Hesham via cfe-commits
cfe-commits at lists.llvm.org
Tue Sep 29 12:47:42 PDT 2026
https://github.com/AmrDeveloper updated https://github.com/llvm/llvm-project/pull/227372
>From 38e8ae9f0d5136ea1e8d819b73603d8c86a0c349 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 17:59:11 +0200
Subject: [PATCH 1/4] [CIR] Add Matrix transpose operation
---
clang/include/clang/CIR/Dialect/IR/CIROps.td | 27 ++++++++++++++
clang/lib/CIR/CodeGen/CIRGenBuilder.h | 8 +++++
clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp | 7 +++-
clang/lib/CIR/Dialect/IR/CIRDialect.cpp | 22 ++++++++++++
.../CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp | 12 +++++++
clang/test/CIR/CodeGen/matrix.cpp | 36 +++++++++++++++++++
clang/test/CIR/IR/invalid-matrix.cir | 26 ++++++++++++++
7 files changed, 137 insertions(+), 1 deletion(-)
diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index de805d4c642a2..f4f418e031ea1 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6287,6 +6287,33 @@ def CIR_VecSplatOp : CIR_Op<"vec.splat", [
}];
}
+//===----------------------------------------------------------------------===//
+// MatrixTransposeOp
+//===----------------------------------------------------------------------===//
+
+def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
+ Pure,
+]> {
+ let summary = "Matrix transpose";
+ let description = [{
+ The `cir.matrix.transpose` operation transposing a 2-D matrix.
+
+ ```
+ %result = cir.matrix.transpose %value : <3 x 2 x !cir.float>,
+ !cir.matrix<3 x 3 x !cir.float>
+ ```
+ }];
+
+ let arguments = (ins CIR_MatrixType:$value);
+ let results = (outs CIR_MatrixType:$result);
+
+ let assemblyFormat = [{
+ $value `:` type($value) `,` qualified(type($result)) attr-dict
+ }];
+
+ let hasVerifier = 1;
+}
+
//===----------------------------------------------------------------------===//
// BaseClassAddrOp
//===----------------------------------------------------------------------===//
diff --git a/clang/lib/CIR/CodeGen/CIRGenBuilder.h b/clang/lib/CIR/CodeGen/CIRGenBuilder.h
index b581212b0db56..ccd4ed403e3b6 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuilder.h
+++ b/clang/lib/CIR/CodeGen/CIRGenBuilder.h
@@ -825,6 +825,14 @@ class CIRGenBuilderTy : public cir::CIRBaseBuilderTy {
return createVecShuffle(loc, vec1, poison, mask);
}
+ cir::MatrixTransposeOp createMatrixTranspose(mlir::Location loc,
+ mlir::Value matrix) {
+ auto inputTy = mlir::cast<cir::MatrixType>(matrix.getType());
+ auto resultTy = cir::MatrixType::get(
+ inputTy.getElementType(), inputTy.getColumnNum(), inputTy.getRowNum());
+ return cir::MatrixTransposeOp::create(*this, loc, resultTy, matrix);
+ }
+
template <typename... Operands>
mlir::Value emitIntrinsicCallOp(mlir::Location loc, const llvm::StringRef str,
const mlir::Type &resTy, Operands &&...op) {
diff --git a/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp b/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
index 245708691b7d9..fe7a63d7bb85d 100644
--- a/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
+++ b/clang/lib/CIR/CodeGen/CIRGenBuiltin.cpp
@@ -2260,7 +2260,12 @@ RValue CIRGenFunction::emitBuiltinExpr(const GlobalDecl &gd, unsigned builtinID,
}
case Builtin::BI__builtin_reduce_maximum:
case Builtin::BI__builtin_reduce_minimum:
- case Builtin::BI__builtin_matrix_transpose:
+ return errorBuiltinNYI(*this, e, builtinID);
+ case Builtin::BI__builtin_matrix_transpose: {
+ mlir::Value matrix = emitScalarExpr(e->getArg(0));
+ mlir::Value result = builder.createMatrixTranspose(loc, matrix);
+ return RValue::get(result);
+ }
case Builtin::BI__builtin_matrix_column_major_load:
case Builtin::BI__builtin_matrix_column_major_store:
case Builtin::BI__builtin_masked_load:
diff --git a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
index d5a587ff6d81e..1a50ffcb037c7 100644
--- a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
+++ b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
@@ -4183,6 +4183,28 @@ OpFoldResult cir::VecTernaryOp::fold(FoldAdaptor adaptor) {
vecTy, mlir::ArrayAttr::get(getContext(), elements));
}
+//===----------------------------------------------------------------------===//
+// MatrixTransposeOp
+//===----------------------------------------------------------------------===//
+
+LogicalResult cir::MatrixTransposeOp::verify() {
+ cir::MatrixType valueTy = getValue().getType();
+ cir::MatrixType resultTy = getResult().getType();
+ if (valueTy.getElementType() != resultTy.getElementType()) {
+ emitOpError() << "operand type doesn't match the result type";
+ return failure();
+ }
+
+ if ((valueTy.getRowNum() != resultTy.getColumnNum()) ||
+ (valueTy.getColumnNum() != resultTy.getRowNum())) {
+ emitOpError()
+ << "result type doesn't match the transpose type of the operand type";
+ return failure();
+ }
+
+ return success();
+}
+
//===----------------------------------------------------------------------===//
// ComplexCreateOp
//===----------------------------------------------------------------------===//
diff --git a/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp b/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
index 4d2a99d5015fe..b4561c1ef684b 100644
--- a/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
+++ b/clang/lib/CIR/Lowering/DirectToLLVM/LowerToLLVM.cpp
@@ -5224,6 +5224,18 @@ mlir::LogicalResult CIRToLLVMVecTernaryOpLowering::matchAndRewrite(
return mlir::success();
}
+mlir::LogicalResult CIRToLLVMMatrixTransposeOpLowering::matchAndRewrite(
+ cir::MatrixTransposeOp op, OpAdaptor adaptor,
+ mlir::ConversionPatternRewriter &rewriter) const {
+ cir::MatrixType matrixTy = op.getValue().getType();
+ mlir::Type resultTy =
+ typeConverter->convertType(op->getResultTypes().front());
+ rewriter.replaceOpWithNewOp<mlir::LLVM::MatrixTransposeOp>(
+ +op, resultTy, adaptor.getValue(), matrixTy.getRowNum(),
+ matrixTy.getColumnNum());
+ return mlir::success();
+}
+
mlir::LogicalResult CIRToLLVMComplexAddOpLowering::matchAndRewrite(
cir::ComplexAddOp op, OpAdaptor adaptor,
mlir::ConversionPatternRewriter &rewriter) const {
diff --git a/clang/test/CIR/CodeGen/matrix.cpp b/clang/test/CIR/CodeGen/matrix.cpp
index f71bf38c2a1a9..5fc26b134eb55 100644
--- a/clang/test/CIR/CodeGen/matrix.cpp
+++ b/clang/test/CIR/CodeGen/matrix.cpp
@@ -6,6 +6,8 @@
// RUN: FileCheck --input-file=%t.ll %s -check-prefix=LLVM
typedef float matrix3x3 __attribute__((matrix_type(3, 3)));
+typedef float matrix3x2 __attribute__((matrix_type(3, 2)));
+typedef float matrix2x3 __attribute__((matrix_type(2, 3)));
matrix3x3 a;
@@ -48,3 +50,37 @@ void load_global_store_in_local() {
// LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
// LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr @a, align 4
// LLVM: store <9 x float> %[[TMP_A]], ptr %[[B_ADDR]], align 4
+
+void builtin_matrix_transpose() {
+ matrix3x3 a;
+ matrix3x3 b = __builtin_matrix_transpose(a);
+}
+
+// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+
+// LLVM: %[[A_ADDR:.*]] = alloca [9 x float], align 4
+// LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
+// LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr %[[A_ADDR]], align 4
+// LLVM: %[[TRANSPOE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
+// LLVM: store <9 x float> %[[TRANSPOE]], ptr %[[B_ADDR]], align 4
+
+void builtin_matrix_transpose_different_sizes() {
+ matrix3x2 a;
+ matrix2x3 b = __builtin_matrix_transpose(a);
+}
+
+// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+
+// LLVM: %[[A_ADDR:.*]] = alloca [6 x float], align 4
+// LLVM: %[[B_ADDR:.*]] = alloca [6 x float], align 4
+// LLVM: %[[TMP_A:.*]] = load <6 x float>, ptr %[[A_ADDR]], align 4
+// LLVM: %[[TRANSPOE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
+// LLVM: store <6 x float> %[[TRANSPOE:.*]], ptr %[[B_ADDR]], align 4
diff --git a/clang/test/CIR/IR/invalid-matrix.cir b/clang/test/CIR/IR/invalid-matrix.cir
index 7c234dcdf9a96..e8a8e8aab66f8 100644
--- a/clang/test/CIR/IR/invalid-matrix.cir
+++ b/clang/test/CIR/IR/invalid-matrix.cir
@@ -41,3 +41,29 @@ cir.func @negative_column_number() {
cir.return
}
+
+// -----
+
+!s32i = !cir.int<s, 32>
+
+cir.func @builtin_tranpose_different_element_type() {
+ %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+ %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+ %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+ // expected-error at +1 {{operand type doesn't match the result type}}
+ %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !s32i>
+ cir.return
+}
+
+// -----
+
+!s32i = !cir.int<s, 32>
+
+cir.func @builtin_tranpose_different_sizes() {
+ %0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
+ %1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+ %2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
+ // expected-error at +1 {{result type doesn't match the transpose type of the operand type}}
+ %3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+ cir.return
+}
>From c556fea823ba046a600e9cbd84cd2b61f55b50fd Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 19:12:29 +0200
Subject: [PATCH 2/4] Address code review comments and improve diagnostic
---
clang/include/clang/CIR/Dialect/IR/CIROps.td | 2 +-
clang/lib/CIR/Dialect/IR/CIRDialect.cpp | 13 ++++++-------
clang/test/CIR/IR/invalid-matrix.cir | 4 ++--
3 files changed, 9 insertions(+), 10 deletions(-)
diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index f4f418e031ea1..e093a928340b5 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6300,7 +6300,7 @@ def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
```
%result = cir.matrix.transpose %value : <3 x 2 x !cir.float>,
- !cir.matrix<3 x 3 x !cir.float>
+ !cir.matrix<2 x 3 x !cir.float>
```
}];
diff --git a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
index 1a50ffcb037c7..b3409befbce58 100644
--- a/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
+++ b/clang/lib/CIR/Dialect/IR/CIRDialect.cpp
@@ -4190,15 +4190,14 @@ OpFoldResult cir::VecTernaryOp::fold(FoldAdaptor adaptor) {
LogicalResult cir::MatrixTransposeOp::verify() {
cir::MatrixType valueTy = getValue().getType();
cir::MatrixType resultTy = getResult().getType();
- if (valueTy.getElementType() != resultTy.getElementType()) {
- emitOpError() << "operand type doesn't match the result type";
- return failure();
- }
- if ((valueTy.getRowNum() != resultTy.getColumnNum()) ||
+ if ((valueTy.getElementType() != resultTy.getElementType()) ||
+ (valueTy.getRowNum() != resultTy.getColumnNum()) ||
(valueTy.getColumnNum() != resultTy.getRowNum())) {
- emitOpError()
- << "result type doesn't match the transpose type of the operand type";
+ auto expectedTy = cir::MatrixType::get(
+ valueTy.getElementType(), valueTy.getColumnNum(), valueTy.getRowNum());
+ emitOpError() << "operand type " << valueTy << " expects result type of "
+ << expectedTy << " but got " << resultTy;
return failure();
}
diff --git a/clang/test/CIR/IR/invalid-matrix.cir b/clang/test/CIR/IR/invalid-matrix.cir
index e8a8e8aab66f8..a68bdecd26012 100644
--- a/clang/test/CIR/IR/invalid-matrix.cir
+++ b/clang/test/CIR/IR/invalid-matrix.cir
@@ -50,7 +50,7 @@ cir.func @builtin_tranpose_different_element_type() {
%0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
%1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
%2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
- // expected-error at +1 {{operand type doesn't match the result type}}
+ // expected-error at +1 {{op operand type '!cir.matrix<3 x 2 x !cir.float>' expects result type of '!cir.matrix<2 x 3 x !cir.float>' but got '!cir.matrix<2 x 3 x !cir.int<s, 32>>}}
%3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !s32i>
cir.return
}
@@ -63,7 +63,7 @@ cir.func @builtin_tranpose_different_sizes() {
%0 = cir.alloca "a" align(4) : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
%1 = cir.alloca "b" align(4) init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
%2 = cir.load align(4) %0 : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
- // expected-error at +1 {{result type doesn't match the transpose type of the operand type}}
+ // expected-error at +1 {{op operand type '!cir.matrix<3 x 2 x !cir.float>' expects result type of '!cir.matrix<2 x 3 x !cir.float>' but got '!cir.matrix<3 x 3 x !cir.float>}}
%3 = cir.matrix.transpose %2 : <3 x 2 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
cir.return
}
>From 9171588378d1c666bca0e0863e1a1a2accffb8d1 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 19:54:18 +0200
Subject: [PATCH 3/4] Update MatrixTranspose op description
---
clang/include/clang/CIR/Dialect/IR/CIROps.td | 7 ++++++-
1 file changed, 6 insertions(+), 1 deletion(-)
diff --git a/clang/include/clang/CIR/Dialect/IR/CIROps.td b/clang/include/clang/CIR/Dialect/IR/CIROps.td
index e093a928340b5..239bf5cfa706e 100644
--- a/clang/include/clang/CIR/Dialect/IR/CIROps.td
+++ b/clang/include/clang/CIR/Dialect/IR/CIROps.td
@@ -6296,7 +6296,12 @@ def CIR_MatrixTransposeOp : CIR_Op<"matrix.transpose", [
]> {
let summary = "Matrix transpose";
let description = [{
- The `cir.matrix.transpose` operation transposing a 2-D matrix.
+ The `cir.matrix.transpose` operation provides a representation for the
+ `__builtin_matrix_transpose` builtin and corresponds to the
+ `llvm.matrix.transpose` intrinsic in LLVM IR.
+
+ This operation performs transposition to switch the row and column
+ indices of the matrix and returns the transposed matrix,
```
%result = cir.matrix.transpose %value : <3 x 2 x !cir.float>,
>From 19e42165084fd8e7b80c8b5aba599e172414fc06 Mon Sep 17 00:00:00 2001
From: Amr Hesham <amr96 at programmer.net>
Date: Tue, 29 Sep 2026 21:42:34 +0200
Subject: [PATCH 4/4] Fix variables names in LIT tests
---
clang/test/CIR/CodeGen/matrix.cpp | 16 ++++++++--------
1 file changed, 8 insertions(+), 8 deletions(-)
diff --git a/clang/test/CIR/CodeGen/matrix.cpp b/clang/test/CIR/CodeGen/matrix.cpp
index 5fc26b134eb55..9b8634f2e8247 100644
--- a/clang/test/CIR/CodeGen/matrix.cpp
+++ b/clang/test/CIR/CodeGen/matrix.cpp
@@ -59,14 +59,14 @@ void builtin_matrix_transpose() {
// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>, !cir.matrix<3 x 3 x !cir.float>
-// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
-// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
+// CIR: %[[TRANSPOSE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 3 x !cir.float>, !cir.matrix<3 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOSE]], %[[B_ADDR]] : !cir.matrix<3 x 3 x !cir.float>, !cir.ptr<!cir.matrix<3 x 3 x !cir.float>>
// LLVM: %[[A_ADDR:.*]] = alloca [9 x float], align 4
// LLVM: %[[B_ADDR:.*]] = alloca [9 x float], align 4
// LLVM: %[[TMP_A:.*]] = load <9 x float>, ptr %[[A_ADDR]], align 4
-// LLVM: %[[TRANSPOE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
-// LLVM: store <9 x float> %[[TRANSPOE]], ptr %[[B_ADDR]], align 4
+// LLVM: %[[TRANSPOSE:.*]] = call <9 x float> @llvm.matrix.transpose.v9f32(<9 x float> %[[TMP_A]], i32 3, i32 3)
+// LLVM: store <9 x float> %[[TRANSPOSE]], ptr %[[B_ADDR]], align 4
void builtin_matrix_transpose_different_sizes() {
matrix3x2 a;
@@ -76,11 +76,11 @@ void builtin_matrix_transpose_different_sizes() {
// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>
// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
// CIR: %[[TMP_A:.*]] = cir.load {{.*}} %[[A_ADDR]] : !cir.ptr<!cir.matrix<3 x 2 x !cir.float>>, !cir.matrix<3 x 2 x !cir.float>
-// CIR: %[[TRANSPOE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
-// CIR: cir.store {{.*}} %[[TRANSPOE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
+// CIR: %[[TRANSPOSE:.*]] = cir.matrix.transpose %[[TMP_A]] : <3 x 2 x !cir.float>, !cir.matrix<2 x 3 x !cir.float>
+// CIR: cir.store {{.*}} %[[TRANSPOSE]], %[[B_ADDR]] : !cir.matrix<2 x 3 x !cir.float>, !cir.ptr<!cir.matrix<2 x 3 x !cir.float>>
// LLVM: %[[A_ADDR:.*]] = alloca [6 x float], align 4
// LLVM: %[[B_ADDR:.*]] = alloca [6 x float], align 4
// LLVM: %[[TMP_A:.*]] = load <6 x float>, ptr %[[A_ADDR]], align 4
-// LLVM: %[[TRANSPOE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
-// LLVM: store <6 x float> %[[TRANSPOE:.*]], ptr %[[B_ADDR]], align 4
+// LLVM: %[[TRANSPOSE:.*]] = call <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[TMP_A]], i32 3, i32 2)
+// LLVM: store <6 x float> %[[TRANSPOSE:.*]], ptr %[[B_ADDR]], align 4
More information about the cfe-commits
mailing list