[clang] [HLSL] Preserve matrix layout as AST storage metadata (PR #225519)

Farzon Lotfi via cfe-commits cfe-commits at lists.llvm.org
Tue Sep 22 13:46:27 PDT 2026


https://github.com/farzonl created https://github.com/llvm/llvm-project/pull/225519

fixes #213996
fixes #211977
fixes https://godbolt.org/z/rhTYx1KGf

Keep matrix layout metadata on noncanonical matrix types without affecting type identity, overload resolution, deduction, or mangling.

Normalize matrix values to column-major register representation when loading from memory, and convert them back to the destination layout when storing. Preserve layout metadata through typedefs, arrays, records, resources, serialization, AST import, and template substitution.

>From fdf93b9ec50f1a57545a9f0968c2715257bbedfc Mon Sep 17 00:00:00 2001
From: Farzon Lotfi <farzonlotfi at microsoft.com>
Date: Tue, 22 Sep 2026 15:38:17 -0400
Subject: [PATCH] [HLSL] Preserve matrix layout as AST storage metadata

fixes #213996
fixes #211977
fixes https://godbolt.org/z/rhTYx1KGf

Keep matrix layout metadata on noncanonical matrix types without affecting
type identity, overload resolution, deduction, or mangling.

Normalize matrix values to column-major register representation when loading
from memory, and convert them back to the destination layout when storing.
Preserve layout metadata through typedefs, arrays, records, resources,
serialization, AST import, and template substitution.
---
 clang/include/clang/AST/ASTContext.h          |   8 +-
 clang/include/clang/AST/MatrixUtils.h         |  25 +-
 clang/include/clang/AST/PropertiesBase.td     |   7 +
 clang/include/clang/AST/TypeBase.h            |  19 +-
 clang/include/clang/AST/TypeProperties.td     |   5 +-
 clang/include/clang/Sema/SemaHLSL.h           |   1 -
 clang/lib/AST/ASTContext.cpp                  |  41 +-
 clang/lib/AST/ASTImporter.cpp                 |   2 +-
 clang/lib/AST/Type.cpp                        |  14 +-
 clang/lib/CodeGen/CGExpr.cpp                  |  17 +-
 clang/lib/CodeGen/CGExprScalar.cpp            |  25 +-
 clang/lib/CodeGen/CGHLSLBuiltins.cpp          |  36 +-
 clang/lib/CodeGen/CodeGenTypes.cpp            |   5 +
 clang/lib/Sema/SemaExpr.cpp                   |   4 +
 clang/lib/Sema/SemaHLSL.cpp                   |  31 --
 clang/lib/Sema/SemaStmt.cpp                   |   5 -
 clang/lib/Sema/SemaType.cpp                   |  11 +-
 clang/lib/Sema/TreeTransform.h                |   9 +
 clang/test/AST/HLSL/matrix_layout_attr.hlsl   |  21 +
 .../BasicFeatures/MatrixElementTypeCast.hlsl  | 479 ++++++++++++------
 .../MatrixExplicitTruncation.hlsl             | 361 +++++++++----
 .../MatrixImplicitTruncation.hlsl             | 318 ++++++++----
 .../MatrixInitializerListOrder.hlsl           |  34 +-
 .../MatrixToAndFromVectorConstructors.hlsl    |  48 +-
 .../BasicFeatures/VectorElementwiseCast.hlsl  |  27 +-
 .../BasicFeatures/matrix-type-indexing.hlsl   |  18 +-
 clang/test/CodeGenHLSL/builtins/mul.hlsl      |  50 +-
 .../test/CodeGenHLSL/builtins/transpose.hlsl  |  13 +-
 .../matrix-layout-attr-overrides-default.hlsl |  82 ++-
 ...matrix-layout-register-representation.hlsl |  37 ++
 .../resources/MatrixElement_cbuffer.hlsl      |   3 +-
 clang/test/SemaHLSL/matrix_layout_attr.hlsl   |  12 +
 32 files changed, 1150 insertions(+), 618 deletions(-)
 create mode 100644 clang/test/CodeGenHLSL/matrix-layout-register-representation.hlsl

diff --git a/clang/include/clang/AST/ASTContext.h b/clang/include/clang/AST/ASTContext.h
index b2d407e412b3d9..d5ec38c5f2bd24 100644
--- a/clang/include/clang/AST/ASTContext.h
+++ b/clang/include/clang/AST/ASTContext.h
@@ -1906,8 +1906,9 @@ class ASTContext : public RefCountedBase<ASTContext> {
   ///
   /// \pre \p ElementType must be a valid matrix element type (see
   /// MatrixType::isValidElementType).
-  QualType getConstantMatrixType(QualType ElementType, unsigned NumRows,
-                                 unsigned NumColumns) const;
+  QualType getConstantMatrixType(
+      QualType ElementType, unsigned NumRows, unsigned NumColumns,
+      std::optional<MatrixType::LayoutKind> Layout = std::nullopt) const;
 
   /// Return the unique reference to the matrix type of the specified element
   /// type and size
@@ -1915,6 +1916,9 @@ class ASTContext : public RefCountedBase<ASTContext> {
                                        Expr *ColumnExpr,
                                        SourceLocation AttrLoc) const;
 
+  QualType getMatrixTypeWithLayout(QualType T,
+                                   MatrixType::LayoutKind Layout) const;
+
   QualType getDependentAddressSpaceType(QualType PointeeType,
                                         Expr *AddrSpaceExpr,
                                         SourceLocation AttrLoc) const;
diff --git a/clang/include/clang/AST/MatrixUtils.h b/clang/include/clang/AST/MatrixUtils.h
index ef6cbba6ba7c07..81aed613074a21 100644
--- a/clang/include/clang/AST/MatrixUtils.h
+++ b/clang/include/clang/AST/MatrixUtils.h
@@ -15,32 +15,17 @@
 #define LLVM_CLANG_AST_MATRIXUTILS_H
 
 #include "clang/AST/Type.h"
-#include "clang/Basic/AttrKinds.h"
 #include "clang/Basic/LangOptions.h"
 
 namespace clang {
 /// Returns true if matrices of \p T should be laid out in row-major order.
 ///
-/// In HLSL mode, an `HLSLRowMajor` / `HLSLColumnMajor` AttributedType anywhere
-/// in the sugar chain of \p T (imprinted by Sema when a source decl carries
-/// `[[hlsl::row_major]]` / `[[hlsl::column_major]]`) takes precedence over the
-/// `-fmatrix-memory-layout=` default carried in \p LangOpts. Otherwise the
-/// LangOptions default is used.
+/// An explicit layout stored on the matrix type takes precedence over the
+/// `-fmatrix-memory-layout=` default carried in \p LangOpts.
 inline bool isMatrixRowMajor(const LangOptions &LangOpts, QualType T) {
-  if (LangOpts.HLSL && !T.isNull()) {
-    QualType Cur = T;
-    while (const auto *AT = Cur->getAs<AttributedType>()) {
-      switch (AT->getAttrKind()) {
-      case attr::HLSLRowMajor:
-        return true;
-      case attr::HLSLColumnMajor:
-        return false;
-      default:
-        break;
-      }
-      Cur = AT->getModifiedType();
-    }
-  }
+  if (const auto *MT = T.isNull() ? nullptr : T->getAs<ConstantMatrixType>();
+      LangOpts.HLSL && MT && MT->getLayout())
+    return MT->getLayout() == MatrixType::LayoutKind::RowMajor;
   return LangOpts.getDefaultMatrixMemoryLayout() ==
          LangOptions::MatrixMemoryLayout::MatrixRowMajor;
 }
diff --git a/clang/include/clang/AST/PropertiesBase.td b/clang/include/clang/AST/PropertiesBase.td
index 347f58d45ffa86..0e76302dcc13fd 100644
--- a/clang/include/clang/AST/PropertiesBase.td
+++ b/clang/include/clang/AST/PropertiesBase.td
@@ -130,6 +130,13 @@ def LValuePathSerializationHelper :
     PropertyType<"APValue::LValuePathSerializationHelper"> {
   let BufferElementTypes = [ LValuePathEntry ];
 }
+def MatrixLayoutKind : EnumPropertyType<"MatrixType::LayoutKind"> {
+  let PackOptional =
+    "value.value_or(static_cast<MatrixType::LayoutKind>(2))";
+  let UnpackOptional =
+    "value == static_cast<MatrixType::LayoutKind>(2) ? std::nullopt : "
+    "std::optional<MatrixType::LayoutKind>(value)";
+}
 def NestedNameSpecifier : PropertyType<"NestedNameSpecifier">;
 def NestedNameSpecifierKind : EnumPropertyType<"NestedNameSpecifier::Kind">;
 def OverloadedOperatorKind : EnumPropertyType;
diff --git a/clang/include/clang/AST/TypeBase.h b/clang/include/clang/AST/TypeBase.h
index 28f102fdaf534c..cc8cb69493825d 100644
--- a/clang/include/clang/AST/TypeBase.h
+++ b/clang/include/clang/AST/TypeBase.h
@@ -4429,9 +4429,14 @@ class MatrixType : public Type, public llvm::FoldingSetNode {
 protected:
   friend class ASTContext;
 
+public:
+  enum class LayoutKind : uint8_t { RowMajor, ColumnMajor };
+
+private:
   /// The element type of the matrix.
   QualType ElementType;
 
+protected:
   MatrixType(QualType ElementTy, QualType CanonElementTy);
 
   MatrixType(TypeClass TypeClass, QualType ElementTy, QualType CanonElementTy,
@@ -4482,12 +4487,15 @@ class ConstantMatrixType final : public MatrixType {
   /// Number of rows and columns.
   unsigned NumRows;
   unsigned NumColumns;
+  std::optional<LayoutKind> Layout;
 
   ConstantMatrixType(QualType MatrixElementType, unsigned NRows,
-                     unsigned NColumns, QualType CanonElementType);
+                     unsigned NColumns, QualType CanonElementType,
+                     std::optional<LayoutKind> Layout);
 
   ConstantMatrixType(TypeClass typeClass, QualType MatrixType, unsigned NRows,
-                     unsigned NColumns, QualType CanonElementType);
+                     unsigned NColumns, QualType CanonElementType,
+                     std::optional<LayoutKind> Layout);
 
 public:
   /// Returns the number of rows in the matrix.
@@ -4496,6 +4504,8 @@ class ConstantMatrixType final : public MatrixType {
   /// Returns the number of columns in the matrix.
   unsigned getNumColumns() const { return NumColumns; }
 
+  std::optional<LayoutKind> getLayout() const { return Layout; }
+
   /// Returns the number of elements required to embed the matrix into a vector.
   unsigned getNumElementsFlattened() const {
     return getNumRows() * getNumColumns();
@@ -4541,16 +4551,17 @@ class ConstantMatrixType final : public MatrixType {
   }
 
   void Profile(llvm::FoldingSetNodeID &ID) {
-    Profile(ID, getElementType(), getNumRows(), getNumColumns(),
+    Profile(ID, getElementType(), getNumRows(), getNumColumns(), getLayout(),
             getTypeClass());
   }
 
   static void Profile(llvm::FoldingSetNodeID &ID, QualType ElementType,
                       unsigned NumRows, unsigned NumColumns,
-                      TypeClass TypeClass) {
+                      std::optional<LayoutKind> Layout, TypeClass TypeClass) {
     ID.AddPointer(ElementType.getAsOpaquePtr());
     ID.AddInteger(NumRows);
     ID.AddInteger(NumColumns);
+    ID.AddInteger(Layout ? llvm::to_underlying(*Layout) + 1 : 0);
     ID.AddInteger(TypeClass);
   }
 
diff --git a/clang/include/clang/AST/TypeProperties.td b/clang/include/clang/AST/TypeProperties.td
index dc2a45ec857296..9174382485faff 100644
--- a/clang/include/clang/AST/TypeProperties.td
+++ b/clang/include/clang/AST/TypeProperties.td
@@ -254,9 +254,12 @@ let Class = ConstantMatrixType in {
   def : Property<"numColumns", UInt32> {
     let Read = [{ node->getNumColumns() }];
   }
+  def : Property<"layout", Optional<MatrixLayoutKind>> {
+    let Read = [{ node->getLayout() }];
+  }
 
   def : Creator<[{
-    return ctx.getConstantMatrixType(elementType, numRows, numColumns);
+    return ctx.getConstantMatrixType(elementType, numRows, numColumns, layout);
   }]>;
 }
 
diff --git a/clang/include/clang/Sema/SemaHLSL.h b/clang/include/clang/Sema/SemaHLSL.h
index 6c0e5b52f7cb3b..b2c73b67eaa26c 100644
--- a/clang/include/clang/Sema/SemaHLSL.h
+++ b/clang/include/clang/Sema/SemaHLSL.h
@@ -196,7 +196,6 @@ class SemaHLSL : public SemaBase {
                                          SourceLocation Loc);
   // Re-type a layout-adapting matrix builtin call \p E with \p DestType's
   // row_major/column_major sugar so CodeGen lowers it into that layout.
-  void propagateContextualMatrixLayout(Expr *E, QualType DestType);
   bool handleResourceTypeAttr(QualType T, const ParsedAttr &AL);
 
   template <typename T>
diff --git a/clang/lib/AST/ASTContext.cpp b/clang/lib/AST/ASTContext.cpp
index e74423ca8c8a1b..15252a9eac5165 100644
--- a/clang/lib/AST/ASTContext.cpp
+++ b/clang/lib/AST/ASTContext.cpp
@@ -4870,10 +4870,11 @@ ASTContext::getDependentSizedExtVectorType(QualType vecType,
   return QualType(New, 0);
 }
 
-QualType ASTContext::getConstantMatrixType(QualType ElementTy, unsigned NumRows,
-                                           unsigned NumColumns) const {
+QualType ASTContext::getConstantMatrixType(
+    QualType ElementTy, unsigned NumRows, unsigned NumColumns,
+    std::optional<MatrixType::LayoutKind> Layout) const {
   llvm::FoldingSetNodeID ID;
-  ConstantMatrixType::Profile(ID, ElementTy, NumRows, NumColumns,
+  ConstantMatrixType::Profile(ID, ElementTy, NumRows, NumColumns, Layout,
                               Type::ConstantMatrix);
 
   assert(MatrixType::isValidElementType(ElementTy, getLangOpts()) &&
@@ -4886,9 +4887,9 @@ QualType ASTContext::getConstantMatrixType(QualType ElementTy, unsigned NumRows,
     return QualType(MTP, 0);
 
   QualType Canonical;
-  if (!ElementTy.isCanonical()) {
-    Canonical =
-        getConstantMatrixType(getCanonicalType(ElementTy), NumRows, NumColumns);
+  if (Layout || !ElementTy.isCanonical()) {
+    Canonical = getConstantMatrixType(getCanonicalType(ElementTy), NumRows,
+                                      NumColumns, std::nullopt);
 
     ConstantMatrixType *NewIP = MatrixTypes.lookup(ID, Token);
     assert(!NewIP && "Matrix type shouldn't already exist in the map");
@@ -4896,7 +4897,7 @@ QualType ASTContext::getConstantMatrixType(QualType ElementTy, unsigned NumRows,
   }
 
   auto *New = new (*this, alignof(ConstantMatrixType))
-      ConstantMatrixType(ElementTy, NumRows, NumColumns, Canonical);
+      ConstantMatrixType(ElementTy, NumRows, NumColumns, Canonical, Layout);
   MatrixTypes.insert(New, Token);
   Types.push_back(New);
   return QualType(New, 0);
@@ -4942,6 +4943,32 @@ QualType ASTContext::getDependentSizedMatrixType(QualType ElementTy,
   return QualType(New, 0);
 }
 
+QualType
+ASTContext::getMatrixTypeWithLayout(QualType T,
+                                    MatrixType::LayoutKind Layout) const {
+  Qualifiers Quals = T.getQualifiers();
+  const Type *Ty = T->getUnqualifiedDesugaredType();
+
+  if (const auto *MT = dyn_cast<ConstantMatrixType>(Ty))
+    return getQualifiedType(getConstantMatrixType(MT->getElementType(),
+                                                  MT->getNumRows(),
+                                                  MT->getNumColumns(), Layout),
+                            Quals);
+
+  const auto *CAT = dyn_cast<ConstantArrayType>(Ty);
+  if (!CAT)
+    return T;
+
+  QualType Result = getConstantArrayType(
+      getMatrixTypeWithLayout(CAT->getElementType(), Layout), CAT->getSize(),
+      CAT->getSizeExpr(), CAT->getSizeModifier(),
+      CAT->getIndexTypeCVRQualifiers());
+  if (isa<ArrayParameterType>(CAT))
+    Result = getArrayParameterType(Result);
+
+  return getQualifiedType(Result, Quals);
+}
+
 QualType ASTContext::getDependentAddressSpaceType(QualType PointeeType,
                                                   Expr *AddrSpaceExpr,
                                                   SourceLocation AttrLoc) const {
diff --git a/clang/lib/AST/ASTImporter.cpp b/clang/lib/AST/ASTImporter.cpp
index bec73d820d0099..89b34aba3ec4ac 100644
--- a/clang/lib/AST/ASTImporter.cpp
+++ b/clang/lib/AST/ASTImporter.cpp
@@ -2108,7 +2108,7 @@ ExpectedType clang::ASTNodeImporter::VisitConstantMatrixType(
     return ToElementTypeOrErr.takeError();
 
   return Importer.getToContext().getConstantMatrixType(
-      *ToElementTypeOrErr, T->getNumRows(), T->getNumColumns());
+      *ToElementTypeOrErr, T->getNumRows(), T->getNumColumns(), T->getLayout());
 }
 
 ExpectedType clang::ASTNodeImporter::VisitDependentAddressSpaceType(
diff --git a/clang/lib/AST/Type.cpp b/clang/lib/AST/Type.cpp
index 5fdac154725c3c..f855f005f03221 100644
--- a/clang/lib/AST/Type.cpp
+++ b/clang/lib/AST/Type.cpp
@@ -496,15 +496,17 @@ MatrixType::MatrixType(TypeClass tc, QualType matrixType, QualType canonType,
       ElementType(matrixType) {}
 
 ConstantMatrixType::ConstantMatrixType(QualType matrixType, unsigned nRows,
-                                       unsigned nColumns, QualType canonType)
-    : ConstantMatrixType(ConstantMatrix, matrixType, nRows, nColumns,
-                         canonType) {}
+                                       unsigned nColumns, QualType canonType,
+                                       std::optional<LayoutKind> Layout)
+    : ConstantMatrixType(ConstantMatrix, matrixType, nRows, nColumns, canonType,
+                         Layout) {}
 
 ConstantMatrixType::ConstantMatrixType(TypeClass tc, QualType matrixType,
                                        unsigned nRows, unsigned nColumns,
-                                       QualType canonType)
+                                       QualType canonType,
+                                       std::optional<LayoutKind> Layout)
     : MatrixType(tc, matrixType, canonType), NumRows(nRows),
-      NumColumns(nColumns) {}
+      NumColumns(nColumns), Layout(Layout) {}
 
 DependentSizedMatrixType::DependentSizedMatrixType(QualType ElementType,
                                                    QualType CanonicalType,
@@ -1279,7 +1281,7 @@ struct SimpleTransformVisitor : public TypeVisitor<Derived, QualType> {
       return QualType(T, 0);
 
     return Ctx.getConstantMatrixType(elementType, T->getNumRows(),
-                                     T->getNumColumns());
+                                     T->getNumColumns(), T->getLayout());
   }
 
   QualType VisitOverflowBehaviorType(const OverflowBehaviorType *T) {
diff --git a/clang/lib/CodeGen/CGExpr.cpp b/clang/lib/CodeGen/CGExpr.cpp
index aca0415d7f5449..91c49639e40823 100644
--- a/clang/lib/CodeGen/CGExpr.cpp
+++ b/clang/lib/CodeGen/CGExpr.cpp
@@ -2425,6 +2425,13 @@ LValue CodeGenFunction::EmitMatrixElementExpr(const MatrixElementExpr *E) {
 // (VectorType).
 static void EmitStoreOfMatrixScalar(llvm::Value *value, LValue lvalue,
                                     bool isInit, CodeGenFunction &CGF) {
+  if (CGF.getLangOpts().HLSL &&
+      isMatrixRowMajor(CGF.getLangOpts(), lvalue.getType())) {
+    const auto *MatrixTy = lvalue.getType()->castAs<ConstantMatrixType>();
+    llvm::MatrixBuilder MB(CGF.Builder);
+    value = MB.CreateColumnMajorToRowMajorTransform(
+        value, MatrixTy->getNumRows(), MatrixTy->getNumColumns());
+  }
   Address Addr = MaybeConvertMatrixAddress(lvalue.getAddress(), CGF,
                                            value->getType()->isVectorTy());
   CGF.EmitStoreOfScalar(value, Addr, lvalue.isVolatile(), lvalue.getType(),
@@ -2515,7 +2522,15 @@ static RValue EmitLoadOfMatrixLValue(LValue LV, SourceLocation Loc,
 
   Address Addr = MaybeConvertMatrixAddress(DestAddr, CGF);
   LV.setAddress(Addr);
-  return RValue::get(CGF.EmitLoadOfScalar(LV, Loc));
+  llvm::Value *Value = CGF.EmitLoadOfScalar(LV, Loc);
+  if (CGF.getLangOpts().HLSL &&
+      isMatrixRowMajor(CGF.getLangOpts(), LV.getType())) {
+    const auto *MatrixTy = LV.getType()->castAs<ConstantMatrixType>();
+    llvm::MatrixBuilder MB(CGF.Builder);
+    Value = MB.CreateRowMajorToColumnMajorTransform(
+        Value, MatrixTy->getNumRows(), MatrixTy->getNumColumns());
+  }
+  return RValue::get(Value);
 }
 
 RValue CodeGenFunction::EmitLoadOfAnyValue(LValue LV, AggValueSlot Slot,
diff --git a/clang/lib/CodeGen/CGExprScalar.cpp b/clang/lib/CodeGen/CGExprScalar.cpp
index 1410433237c49e..0c310816bc268e 100644
--- a/clang/lib/CodeGen/CGExprScalar.cpp
+++ b/clang/lib/CodeGen/CGExprScalar.cpp
@@ -2229,13 +2229,10 @@ Value *ScalarExprEmitter::VisitMatrixSingleSubscriptExpr(
   auto *ResultTy = llvm::FixedVectorType::get(ElemTy, NumColumns);
   Value *RowVec = llvm::PoisonValue::get(ResultTy);
 
-  bool IsMatrixRowMajor =
-      isMatrixRowMajor(CGF.getLangOpts(), E->getBase()->getType());
-
   for (unsigned Col = 0; Col != NumColumns; ++Col) {
     Value *ColVal = llvm::ConstantInt::get(RowIdx->getType(), Col);
     Value *EltIdx = MB.CreateIndex(RowIdx, ColVal, NumRows, NumColumns,
-                                   IsMatrixRowMajor, "matrix_row_idx");
+                                   /*IsRowMajor=*/false, "matrix_row_idx");
     Value *Elt =
         Builder.CreateExtractElement(FlatMatrix, EltIdx, "matrix_elem");
     Value *Lane = llvm::ConstantInt::get(Builder.getInt32Ty(), Col);
@@ -2259,9 +2256,8 @@ Value *ScalarExprEmitter::VisitMatrixSubscriptExpr(MatrixSubscriptExpr *E) {
   Value *Idx;
   unsigned NumCols = MatrixTy->getNumColumns();
   unsigned NumRows = MatrixTy->getNumRows();
-  bool IsMatrixRowMajor =
-      isMatrixRowMajor(CGF.getLangOpts(), E->getBase()->getType());
-  Idx = MB.CreateIndex(RowIdx, ColumnIdx, NumRows, NumCols, IsMatrixRowMajor);
+  Idx = MB.CreateIndex(RowIdx, ColumnIdx, NumRows, NumCols,
+                       /*IsRowMajor=*/false);
 
   if (CGF.CGM.getCodeGenOpts().OptimizationLevel > 0)
     MB.CreateIndexAssumption(Idx, MatrixTy->getNumElementsFlattened());
@@ -2343,10 +2339,8 @@ Value *ScalarExprEmitter::VisitInitListExpr(InitListExpr *E) {
 
   // For column-major matrix types, we insert elements directly at their
   // column-major positions rather than inserting sequentially and shuffling.
-  const ConstantMatrixType *ColMajorMT = nullptr;
-  if (const auto *MT = E->getType()->getAs<ConstantMatrixType>();
-      MT && !isMatrixRowMajor(CGF.getLangOpts(), E->getType()))
-    ColMajorMT = MT;
+  const ConstantMatrixType *ColMajorMT =
+      E->getType()->getAs<ConstantMatrixType>();
 
   // Loop over initializers collecting the Value for each, and remembering
   // whether the source was swizzle (ExtVectorElementExpr).  This will allow
@@ -3182,15 +3176,10 @@ Value *ScalarExprEmitter::VisitCastExpr(CastExpr *CE) {
       assert(NumRows <= SrcMatTy->getNumRows());
       assert(NumCols <= SrcMatTy->getNumColumns());
 
-      // isMatrix[Src|Dst]RowMajor needs the full sugared QualType to find
-      // matrix layout attrs. So use E->getType() &  DestTy rather than SrcMatTy
-      // & MatTy b/c getAs<ConstantMatrixType>() strips the sugar.
-      bool IsSrcRowMajor = isMatrixRowMajor(CGF.getLangOpts(), E->getType());
-      bool IsDstRowMajor = isMatrixRowMajor(CGF.getLangOpts(), DestTy);
       for (unsigned R = 0; R < NumRows; R++)
         for (unsigned C = 0; C < NumCols; C++)
-          Mask[MatTy->getFlattenedIndex(R, C, IsDstRowMajor)] =
-              SrcMatTy->getFlattenedIndex(R, C, IsSrcRowMajor);
+          Mask[MatTy->getColumnMajorFlattenedIndex(R, C)] =
+              SrcMatTy->getColumnMajorFlattenedIndex(R, C);
 
       return Builder.CreateShuffleVector(Mat, Mask, "trunc");
     }
diff --git a/clang/lib/CodeGen/CGHLSLBuiltins.cpp b/clang/lib/CodeGen/CGHLSLBuiltins.cpp
index bb0fe135ff8b08..ea9ce8ad0d19c7 100644
--- a/clang/lib/CodeGen/CGHLSLBuiltins.cpp
+++ b/clang/lib/CodeGen/CGHLSLBuiltins.cpp
@@ -1259,13 +1259,6 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID,
     bool IsMat0 = QTy0->isConstantMatrixType();
     bool IsMat1 = QTy1->isConstantMatrixType();
 
-    // The matrix multiply intrinsic only operates on column-major order
-    // matrices. Therefore matrix memory layout transforms must be inserted
-    // before and after matrix multiply intrinsics.
-    // Use whichever operand is a matrix to discover its declared layout.
-    bool IsRowMajorMat0 = IsMat0 && isMatrixRowMajor(getLangOpts(), QTy0);
-    bool IsRowMajorMat1 = IsMat1 && isMatrixRowMajor(getLangOpts(), QTy1);
-
     llvm::MatrixBuilder MB(Builder);
     if (IsVec0 && IsMat1) {
       unsigned N = QTy0->castAs<VectorType>()->getNumElements();
@@ -1273,8 +1266,6 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID,
       unsigned Rows = MatTy->getNumRows();
       unsigned Cols = MatTy->getNumColumns();
       assert(N == Rows && "vector length must match matrix row count");
-      if (IsRowMajorMat1)
-        Op1 = MB.CreateRowMajorToColumnMajorTransform(Op1, Rows, Cols);
       return MB.CreateMatrixMultiply(Op0, Op1, 1, N, Cols, "hlsl.mul");
     }
     if (IsMat0 && IsVec1) {
@@ -1283,8 +1274,6 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID,
       unsigned Cols = MatTy->getNumColumns();
       assert(QTy1->castAs<VectorType>()->getNumElements() == Cols &&
              "vector length must match matrix column count");
-      if (IsRowMajorMat0)
-        Op0 = MB.CreateRowMajorToColumnMajorTransform(Op0, Rows, Cols);
       return MB.CreateMatrixMultiply(Op0, Op1, Rows, Cols, 1, "hlsl.mul");
     }
     assert(IsMat0 && IsMat1);
@@ -1296,18 +1285,7 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID,
     unsigned Cols1 = MatTy1->getNumColumns();
     assert(Cols0 == Rows1 &&
            "inner matrix dimensions must match for multiplication");
-    if (IsRowMajorMat0)
-      Op0 = MB.CreateRowMajorToColumnMajorTransform(Op0, Rows0, Cols0);
-    if (IsRowMajorMat1)
-      Op1 = MB.CreateRowMajorToColumnMajorTransform(Op1, Rows1, Cols1);
-
-    Value *Result =
-        MB.CreateMatrixMultiply(Op0, Op1, Rows0, Cols0, Cols1, "hlsl.mul");
-
-    bool IsResultRowMajor = isMatrixRowMajor(getLangOpts(), E->getType());
-    if (IsResultRowMajor)
-      Result = MB.CreateColumnMajorToRowMajorTransform(Result, Rows0, Cols1);
-    return Result;
+    return MB.CreateMatrixMultiply(Op0, Op1, Rows0, Cols0, Cols1, "hlsl.mul");
   }
   case Builtin::BI__builtin_hlsl_transpose: {
     Value *Op0 = EmitScalarExpr(E->getArg(0));
@@ -1315,18 +1293,6 @@ Value *CodeGenFunction::EmitHLSLBuiltinExpr(unsigned BuiltinID,
     unsigned Rows = MatTy->getNumRows();
     unsigned Cols = MatTy->getNumColumns();
     llvm::MatrixBuilder MB(Builder);
-    // The correct lowering of a transpose depends on both the source layout
-    // and the result layout.
-    bool SrcRowMajor = isMatrixRowMajor(getLangOpts(), E->getArg(0)->getType());
-    bool DstRowMajor = isMatrixRowMajor(getLangOpts(), E->getType());
-    //  When the source & result layouts differ, the operand already holds the
-    //  transposed result, ie transpose is a no-op on the underlying vector.
-    if (SrcRowMajor != DstRowMajor)
-      return Op0;
-    // When the source and result share a layout, emit a transpose.
-    if (SrcRowMajor)
-      // For row-major operands the dimensions are swapped
-      return MB.CreateMatrixTranspose(Op0, Cols, Rows);
     return MB.CreateMatrixTranspose(Op0, Rows, Cols);
   }
   case Builtin::BI__builtin_hlsl_elementwise_rcp: {
diff --git a/clang/lib/CodeGen/CodeGenTypes.cpp b/clang/lib/CodeGen/CodeGenTypes.cpp
index 99ead1295bc584..999867e5e40cec 100644
--- a/clang/lib/CodeGen/CodeGenTypes.cpp
+++ b/clang/lib/CodeGen/CodeGenTypes.cpp
@@ -102,6 +102,11 @@ void CodeGenTypes::addRecordTypeName(const RecordDecl *RD,
 /// But the size does need to be exactly right or else things like struct
 /// layout will break.
 llvm::Type *CodeGenTypes::ConvertTypeForMem(QualType T) {
+  if (const auto *ArrayTy = dyn_cast<ConstantArrayType>(T.getTypePtr())) {
+    llvm::Type *ElementTy = ConvertTypeForMem(ArrayTy->getElementType());
+    return llvm::ArrayType::get(ElementTy, ArrayTy->getZExtSize());
+  }
+
   if (T->isConstantMatrixType()) {
     const Type *Ty = Context.getCanonicalType(T).getTypePtr();
     const ConstantMatrixType *MT = cast<ConstantMatrixType>(Ty);
diff --git a/clang/lib/Sema/SemaExpr.cpp b/clang/lib/Sema/SemaExpr.cpp
index be1dc9f85d4f7d..729186875128d5 100644
--- a/clang/lib/Sema/SemaExpr.cpp
+++ b/clang/lib/Sema/SemaExpr.cpp
@@ -722,6 +722,10 @@ ExprResult Sema::DefaultLvalueConversion(Expr *E) {
   if (T.hasQualifiers())
     T = T.getUnqualifiedType();
 
+  if (getLangOpts().HLSL)
+    if (const auto *MT = T->getAs<ConstantMatrixType>(); MT && MT->getLayout())
+      T = Context.getCanonicalType(T);
+
   // Under the MS ABI, lock down the inheritance model now.
   if (T->isMemberPointerType() &&
       Context.getTargetInfo().getCXXABI().isMicrosoft())
diff --git a/clang/lib/Sema/SemaHLSL.cpp b/clang/lib/Sema/SemaHLSL.cpp
index a5429a90f962ba..88ef91a1680440 100644
--- a/clang/lib/Sema/SemaHLSL.cpp
+++ b/clang/lib/Sema/SemaHLSL.cpp
@@ -2752,37 +2752,6 @@ bool SemaHLSL::diagnoseMatrixLayoutInstantiation(attr::Kind K, QualType T,
 
 // Transpose and matrix mul need to read the destination layout.
 // Elementwise builtins reuse the operand layout instead.
-static bool isLayoutAdaptingMatrixBuiltin(unsigned BuiltinID) {
-  switch (BuiltinID) {
-  case Builtin::BI__builtin_hlsl_mul:
-  case Builtin::BI__builtin_hlsl_transpose:
-    return true;
-  default:
-    return false;
-  }
-}
-
-void SemaHLSL::propagateContextualMatrixLayout(Expr *E, QualType DestType) {
-  if (!E || DestType.isNull())
-    return;
-  const auto *DestMat = DestType->getAs<ConstantMatrixType>();
-  if (!DestMat)
-    return;
-  auto *Call = dyn_cast<CallExpr>(E->IgnoreParenImpCasts());
-  if (!Call)
-    return;
-  const FunctionDecl *Callee = Call->getDirectCallee();
-  if (!Callee || !isLayoutAdaptingMatrixBuiltin(Callee->getBuiltinID()))
-    return;
-  const auto *CallMat = Call->getType()->getAs<ConstantMatrixType>();
-  if (!CallMat || CallMat->getNumRows() != DestMat->getNumRows() ||
-      CallMat->getNumColumns() != DestMat->getNumColumns())
-    return;
-  // Re-type the call with the destination sugar so CodeGen lowers into that
-  // layout, not the TU default.
-  Call->setType(DestType.getUnqualifiedType());
-}
-
 namespace {
 
 /// This class implements HLSL availability diagnostics for default
diff --git a/clang/lib/Sema/SemaStmt.cpp b/clang/lib/Sema/SemaStmt.cpp
index 74fe253efa1374..25db6087d8d25f 100644
--- a/clang/lib/Sema/SemaStmt.cpp
+++ b/clang/lib/Sema/SemaStmt.cpp
@@ -4301,11 +4301,6 @@ StmtResult Sema::BuildReturnStmt(SourceLocation ReturnLoc, Expr *RetValExp,
       }
       RetValExp = Res.getAs<Expr>();
 
-      // A returned HLSL matrix may need its layout reconciled with the
-      // function's row_major/column_major return type.
-      if (getLangOpts().HLSL && RetValExp && RetType->isMatrixType())
-        HLSL().propagateContextualMatrixLayout(RetValExp, RetType);
-
       // If we have a related result type, we need to implicitly
       // convert back to the formal result type.  We can't pretend to
       // initialize the result again --- we might end double-retaining
diff --git a/clang/lib/Sema/SemaType.cpp b/clang/lib/Sema/SemaType.cpp
index 2796ac2929f460..adeb26bcd54abe 100644
--- a/clang/lib/Sema/SemaType.cpp
+++ b/clang/lib/Sema/SemaType.cpp
@@ -9168,8 +9168,15 @@ static void processTypeAttrs(TypeProcessingState &state, QualType &type,
     case ParsedAttr::AT_HLSLRowMajor:
     case ParsedAttr::AT_HLSLColumnMajor:
       if (Attr *A =
-              state.getSema().HLSL().buildMatrixLayoutTypeAttr(type, attr))
-        type = state.getAttributedType(A, type, type);
+              state.getSema().HLSL().buildMatrixLayoutTypeAttr(type, attr)) {
+        MatrixType::LayoutKind Layout =
+            attr.getKind() == ParsedAttr::AT_HLSLRowMajor
+                ? MatrixType::LayoutKind::RowMajor
+                : MatrixType::LayoutKind::ColumnMajor;
+        QualType Equivalent =
+            state.getSema().Context.getMatrixTypeWithLayout(type, Layout);
+        type = state.getAttributedType(A, type, Equivalent);
+      }
       attr.setUsedAsTypeAttr();
       break;
     OBJC_POINTER_TYPE_ATTRS_CASELIST:
diff --git a/clang/lib/Sema/TreeTransform.h b/clang/lib/Sema/TreeTransform.h
index 942fb586b1f239..d3ceddd4622572 100644
--- a/clang/lib/Sema/TreeTransform.h
+++ b/clang/lib/Sema/TreeTransform.h
@@ -7857,6 +7857,15 @@ QualType TreeTransform<Derived>::TransformAttributedType(TypeLocBuilder &TLB,
         return QualType();
     }
 
+    if (SemaRef.getLangOpts().HLSL) {
+      if (oldType->getAttrKind() == attr::HLSLRowMajor)
+        equivalentType = SemaRef.Context.getMatrixTypeWithLayout(
+            equivalentType, MatrixType::LayoutKind::RowMajor);
+      else if (oldType->getAttrKind() == attr::HLSLColumnMajor)
+        equivalentType = SemaRef.Context.getMatrixTypeWithLayout(
+            equivalentType, MatrixType::LayoutKind::ColumnMajor);
+    }
+
     // Check whether we can add nullability; it is only represented as
     // type sugar, and therefore cannot be diagnosed in any other way.
     if (auto nullability = oldType->getImmediateNullability()) {
diff --git a/clang/test/AST/HLSL/matrix_layout_attr.hlsl b/clang/test/AST/HLSL/matrix_layout_attr.hlsl
index 5e5084feb143f5..f10aea99e35df7 100644
--- a/clang/test/AST/HLSL/matrix_layout_attr.hlsl
+++ b/clang/test/AST/HLSL/matrix_layout_attr.hlsl
@@ -1,5 +1,9 @@
 // RUN: %clang_cc1 -triple dxil-pc-shadermodel6.6-library -finclude-default-header \
 // RUN:   -std=hlsl202x -ast-dump -x hlsl %s | FileCheck %s
+// RUN: %clang_cc1 -triple dxil-pc-shadermodel6.6-library -finclude-default-header \
+// RUN:   -std=hlsl202x -emit-pch -o %t %s
+// RUN: %clang_cc1 -triple dxil-pc-shadermodel6.6-library -finclude-default-header \
+// RUN:   -std=hlsl202x -include-pch %t -ast-dump-all -x hlsl %S/Inputs/empty.hlsl | FileCheck %s
 
 // CHECK: VarDecl {{.*}} rm_mat 'float3x3 hlsl_constant __attribute__((row_major))':'matrix<float, 3, 3> hlsl_constant'
 row_major float3x3 rm_mat;
@@ -21,3 +25,20 @@ typedef row_major float4x4 RM44;
 // CHECK-LABEL: TypedefDecl {{.*}} CM44 'float4x4 __attribute__((column_major))':'matrix<float, 4, 4>'
 // CHECK-NEXT:  AttributedType {{.*}} 'float4x4 __attribute__((column_major))' sugar
 typedef column_major float4x4 CM44;
+
+// CHECK: VarDecl {{.*}} rm_array 'float2x3 hlsl_constant[2] __attribute__((row_major))'
+row_major float2x3 rm_array[2];
+
+template <typename T>
+struct Holder {
+  T value;
+};
+
+// CHECK: VarDecl {{.*}} rm_holder 'hlsl_constant Holder<float2x3 __attribute__((row_major))>':'hlsl_constant Holder<matrix<float, 2, 3>>'
+Holder<row_major float2x3> rm_holder;
+
+// CHECK: VarDecl {{.*}} rm_resource 'StructuredBuffer<float2x3 __attribute__((row_major))>':'hlsl::StructuredBuffer<matrix<float, 2, 3>>'
+StructuredBuffer<row_major float2x3> rm_resource;
+
+// CHECK: VarDecl {{.*}} cm_resource 'StructuredBuffer<float2x3 __attribute__((column_major))>':'hlsl::StructuredBuffer<matrix<float, 2, 3>>'
+StructuredBuffer<column_major float2x3> cm_resource;
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/MatrixElementTypeCast.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/MatrixElementTypeCast.hlsl
index b4436cc39e443f..78988265c3f98a 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/MatrixElementTypeCast.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/MatrixElementTypeCast.hlsl
@@ -1,127 +1,223 @@
+// NOTE: Assertions have been autogenerated by utils/update_cc_test_checks.py UTC_ARGS: --version 6
 // RUN: %clang_cc1 -finclude-default-header -triple dxil-pc-shadermodel6.3-library -x hlsl -emit-llvm -disable-llvm-passes -fnative-half-type -fnative-int16-type -fmatrix-memory-layout=row-major -o - %s | FileCheck %s --check-prefixes=CHECK,ROW-CHECK
 // RUN: %clang_cc1 -finclude-default-header -triple dxil-pc-shadermodel6.3-library -x hlsl -emit-llvm -disable-llvm-passes -fnative-half-type -fnative-int16-type -fmatrix-memory-layout=column-major -o - %s | FileCheck %s --check-prefixes=CHECK,COL-CHECK
 
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast0u11matrix_typeILm3ELm2EfE(
-// CHECK-SAME: <6 x float> noundef nofpclass(nan inf) [[F32:%.*]]) #[[ATTR0:[0-9]+]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:  %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast0u11matrix_typeILm3ELm2EfE(
+// ROW-CHECK-SAME: <6 x float> noundef nofpclass(nan inf) [[F32:%.*]]) #[[ATTR0:[0-9]+]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // ROW-CHECK-NEXT:    [[F32_ADDR:%.*]] = alloca [3 x <2 x float>], align 4
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[F32]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x float> [[TMP1]], ptr [[F32_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x float>, ptr [[F32_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[TMP2]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x float> [[TMP3]] to <6 x i32>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[CONV]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast0u11matrix_typeILm3ELm2EfE(
+// COL-CHECK-SAME: <6 x float> noundef nofpclass(nan inf) [[F32:%.*]]) #[[ATTR0:[0-9]+]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[F32_ADDR:%.*]] = alloca [2 x <3 x float>], align 4
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <6 x float> [[F32]], ptr [[F32_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <6 x float>, ptr [[F32_ADDR]], align 4
-// CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x float> [[TMP0]] to <6 x i32>
-// CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <6 x float> [[F32]], ptr [[F32_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <6 x float>, ptr [[F32_ADDR]], align 4
+// COL-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x float> [[TMP1]] to <6 x i32>
+// COL-CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
 int3x2 elementwise_type_cast0(float3x2 f32) {
     int3x2 i32 = (int3x2)f32;
     return i32;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast1u11matrix_typeILm3ELm2EsE(
-// CHECK-SAME: <6 x i16> noundef [[I16_32:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:  %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast1u11matrix_typeILm3ELm2EsE(
+// ROW-CHECK-SAME: <6 x i16> noundef [[I16_32:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // ROW-CHECK-NEXT:    [[I16_32_ADDR:%.*]] = alloca [3 x <2 x i16>], align 2
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <6 x i16> @llvm.matrix.transpose.v6i16(<6 x i16> [[I16_32]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i16> [[TMP1]], ptr [[I16_32_ADDR]], align 2
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i16>, ptr [[I16_32_ADDR]], align 2
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <6 x i16> @llvm.matrix.transpose.v6i16(<6 x i16> [[TMP2]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = sext <6 x i16> [[TMP3]] to <6 x i32>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[CONV]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast1u11matrix_typeILm3ELm2EsE(
+// COL-CHECK-SAME: <6 x i16> noundef [[I16_32:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[I16_32_ADDR:%.*]] = alloca [2 x <3 x i16>], align 2
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <6 x i16> [[I16_32]], ptr [[I16_32_ADDR]], align 2
-// CHECK-NEXT:    [[TMP0:%.*]] = load <6 x i16>, ptr [[I16_32_ADDR]], align 2
-// CHECK-NEXT:    [[CONV:%.*]] = sext <6 x i16> [[TMP0]] to <6 x i32>
-// CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <6 x i16> [[I16_32]], ptr [[I16_32_ADDR]], align 2
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i16>, ptr [[I16_32_ADDR]], align 2
+// COL-CHECK-NEXT:    [[CONV:%.*]] = sext <6 x i16> [[TMP1]] to <6 x i32>
+// COL-CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
 int3x2 elementwise_type_cast1(int16_t3x2 i16_32) {
     int3x2 i32 = (int3x2)i16_32;
     return i32;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast2u11matrix_typeILm3ELm2ElE(
-// CHECK-SAME: <6 x i64> noundef [[I64_32:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:  %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast2u11matrix_typeILm3ELm2ElE(
+// ROW-CHECK-SAME: <6 x i64> noundef [[I64_32:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // ROW-CHECK-NEXT:    [[I64_32_ADDR:%.*]] = alloca [3 x <2 x i64>], align 8
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <6 x i64> @llvm.matrix.transpose.v6i64(<6 x i64> [[I64_32]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i64> [[TMP1]], ptr [[I64_32_ADDR]], align 8
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i64>, ptr [[I64_32_ADDR]], align 8
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <6 x i64> @llvm.matrix.transpose.v6i64(<6 x i64> [[TMP2]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = trunc <6 x i64> [[TMP3]] to <6 x i32>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[CONV]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast2u11matrix_typeILm3ELm2ElE(
+// COL-CHECK-SAME: <6 x i64> noundef [[I64_32:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[I64_32_ADDR:%.*]] = alloca [2 x <3 x i64>], align 8
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <6 x i64> [[I64_32]], ptr [[I64_32_ADDR]], align 8
-// CHECK-NEXT:    [[TMP0:%.*]] = load <6 x i64>, ptr [[I64_32_ADDR]], align 8
-// CHECK-NEXT:    [[CONV:%.*]] = trunc <6 x i64> [[TMP0]] to <6 x i32>
-// CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <6 x i64> [[I64_32]], ptr [[I64_32_ADDR]], align 8
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i64>, ptr [[I64_32_ADDR]], align 8
+// COL-CHECK-NEXT:    [[CONV:%.*]] = trunc <6 x i64> [[TMP1]] to <6 x i32>
+// COL-CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
 int3x2 elementwise_type_cast2(int64_t3x2 i64_32) {
     int3x2 i32 = (int3x2)i64_32;
     return i32;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i16> @_Z22elementwise_type_cast3u11matrix_typeILm2ELm3EDhE(
-// CHECK-SAME: <6 x half> noundef nofpclass(nan inf) [[H23:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:  %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-LABEL: define hidden noundef <6 x i16> @_Z22elementwise_type_cast3u11matrix_typeILm2ELm3EDhE(
+// ROW-CHECK-SAME: <6 x half> noundef nofpclass(nan inf) [[H23:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // ROW-CHECK-NEXT:    [[H23_ADDR:%.*]] = alloca [2 x <3 x half>], align 2
 // ROW-CHECK-NEXT:    [[I23:%.*]] = alloca [2 x <3 x i16>], align 2
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> [[H23]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    store <6 x half> [[TMP1]], ptr [[H23_ADDR]], align 2
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x half>, ptr [[H23_ADDR]], align 2
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> [[TMP2]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x half> [[TMP3]] to <6 x i16>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i16> @llvm.matrix.transpose.v6i16(<6 x i16> [[CONV]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    store <6 x i16> [[TMP4]], ptr [[I23]], align 2
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i16>, ptr [[I23]], align 2
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i16> @llvm.matrix.transpose.v6i16(<6 x i16> [[TMP5]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    ret <6 x i16> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i16> @_Z22elementwise_type_cast3u11matrix_typeILm2ELm3EDhE(
+// COL-CHECK-SAME: <6 x half> noundef nofpclass(nan inf) [[H23:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[H23_ADDR:%.*]] = alloca [3 x <2 x half>], align 2
 // COL-CHECK-NEXT:    [[I23:%.*]] = alloca [3 x <2 x i16>], align 2
-// CHECK-NEXT:    store <6 x half> [[H23]], ptr [[H23_ADDR]], align 2
-// CHECK-NEXT:    [[TMP0:%.*]] = load <6 x half>, ptr [[H23_ADDR]], align 2
-// CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x half> [[TMP0]] to <6 x i16>
-// CHECK-NEXT:    store <6 x i16> [[CONV]], ptr [[I23]], align 2
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i16>, ptr [[I23]], align 2
-// CHECK-NEXT:    ret <6 x i16> [[TMP1]]
+// COL-CHECK-NEXT:    store <6 x half> [[H23]], ptr [[H23_ADDR]], align 2
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <6 x half>, ptr [[H23_ADDR]], align 2
+// COL-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x half> [[TMP1]] to <6 x i16>
+// COL-CHECK-NEXT:    store <6 x i16> [[CONV]], ptr [[I23]], align 2
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i16>, ptr [[I23]], align 2
+// COL-CHECK-NEXT:    ret <6 x i16> [[TMP2]]
 //
 int16_t2x3 elementwise_type_cast3(half2x3 h23) {
     int16_t2x3 i23 = (int16_t2x3)h23;
     return i23;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast4u11matrix_typeILm3ELm2EdE(
-// CHECK-SAME: <6 x double> noundef nofpclass(nan inf) [[D32:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:  %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast4u11matrix_typeILm3ELm2EdE(
+// ROW-CHECK-SAME: <6 x double> noundef nofpclass(nan inf) [[D32:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // ROW-CHECK-NEXT:    [[D32_ADDR:%.*]] = alloca [3 x <2 x double>], align 8
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x double> @llvm.matrix.transpose.v6f64(<6 x double> [[D32]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x double> [[TMP1]], ptr [[D32_ADDR]], align 8
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x double>, ptr [[D32_ADDR]], align 8
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call reassoc nnan ninf nsz arcp afn <6 x double> @llvm.matrix.transpose.v6f64(<6 x double> [[TMP2]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x double> [[TMP3]] to <6 x i32>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[CONV]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z22elementwise_type_cast4u11matrix_typeILm3ELm2EdE(
+// COL-CHECK-SAME: <6 x double> noundef nofpclass(nan inf) [[D32:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[D32_ADDR:%.*]] = alloca [2 x <3 x double>], align 8
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <6 x double> [[D32]], ptr [[D32_ADDR]], align 8
-// CHECK-NEXT:    [[TMP0:%.*]] = load <6 x double>, ptr [[D32_ADDR]], align 8
-// CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x double> [[TMP0]] to <6 x i32>
-// CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <6 x double> [[D32]], ptr [[D32_ADDR]], align 8
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <6 x double>, ptr [[D32_ADDR]], align 8
+// COL-CHECK-NEXT:    [[CONV:%.*]] = fptosi <6 x double> [[TMP1]] to <6 x i32>
+// COL-CHECK-NEXT:    store <6 x i32> [[CONV]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
 int3x2 elementwise_type_cast4(double3x2 d32) {
     int3x2 i32 = (int3x2)d32;
     return i32;
 }
 
-// CHECK-LABEL: define hidden void @_Z5call2v(
-// CHECK-SAME: ) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[A:%.*]] = alloca [2 x [1 x i32]], align 4
+// ROW-CHECK-LABEL: define hidden void @_Z5call2v(
+// ROW-CHECK-SAME: ) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[A:%.*]] = alloca [2 x [1 x i32]], align 4
 // ROW-CHECK-NEXT:    [[B:%.*]] = alloca [2 x <1 x i32>], align 4
+// ROW-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [2 x [1 x i32]], align 4
+// ROW-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
+// ROW-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[A]], ptr align 4 @__const._Z5call2v.A, i32 8, i1 false)
+// ROW-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[AGG_TEMP]], ptr align 4 [[A]], i32 8, i1 false)
+// ROW-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
+// ROW-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 1, i32 0
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[GEP]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 0
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr [[GEP1]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP3]], i32 [[TMP4]], i64 1
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP5]], i32 2, i32 1)
+// ROW-CHECK-NEXT:    store <2 x i32> [[TMP6]], ptr [[B]], align 4
+// ROW-CHECK-NEXT:    ret void
+//
+// COL-CHECK-LABEL: define hidden void @_Z5call2v(
+// COL-CHECK-SAME: ) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[A:%.*]] = alloca [2 x [1 x i32]], align 4
 // COL-CHECK-NEXT:    [[B:%.*]] = alloca [1 x <2 x i32>], align 4
-// CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [2 x [1 x i32]], align 4
-// CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
-// CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[A]], ptr align 4 @__const._Z5call2v.A, i32 8, i1 false)
-// CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[AGG_TEMP]], ptr align 4 [[A]], i32 8, i1 false)
-// CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
-// CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 1, i32 0
-// CHECK-NEXT:    [[TMP0:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr [[GEP]], align 4
-// CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> [[TMP0]], i32 [[TMP1]], i64 0
-// CHECK-NEXT:    [[TMP3:%.*]] = load i32, ptr [[GEP1]], align 4
-// CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[TMP3]], i64 1
-// CHECK-NEXT:    store <2 x i32> [[TMP4]], ptr [[B]], align 4
-// CHECK-NEXT:    ret void
+// COL-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [2 x [1 x i32]], align 4
+// COL-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
+// COL-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[A]], ptr align 4 @__const._Z5call2v.A, i32 8, i1 false)
+// COL-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 4 [[AGG_TEMP]], ptr align 4 [[A]], i32 8, i1 false)
+// COL-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
+// COL-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [2 x [1 x i32]], ptr [[AGG_TEMP]], i32 0, i32 1, i32 0
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[GEP]], align 4
+// COL-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 0
+// COL-CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr [[GEP1]], align 4
+// COL-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP3]], i32 [[TMP4]], i64 1
+// COL-CHECK-NEXT:    store <2 x i32> [[TMP5]], ptr [[B]], align 4
+// COL-CHECK-NEXT:    ret void
 //
 void call2() {
   int A[2][1] = {{1},{2}};
@@ -133,27 +229,48 @@ struct S {
   float Y;
 };
 
-// CHECK-LABEL: define hidden void @_Z5call3v(
-// CHECK-SAME: ) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[S:%.*]] = alloca [[STRUCT_S:%.*]], align 1
+// ROW-CHECK-LABEL: define hidden void @_Z5call3v(
+// ROW-CHECK-SAME: ) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[S:%.*]] = alloca [[STRUCT_S:%.*]], align 1
 // ROW-CHECK-NEXT:    [[A:%.*]] = alloca [2 x <1 x i32>], align 4
+// ROW-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_S]], align 1
+// ROW-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
+// ROW-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[S]], ptr align 1 @__const._Z5call3v.s, i32 8, i1 false)
+// ROW-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[S]], i32 8, i1 false)
+// ROW-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 0
+// ROW-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 1
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[GEP]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 0
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = load float, ptr [[GEP1]], align 4
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = fptosi float [[TMP4]] to i32
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP3]], i32 [[CONV]], i64 1
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP5]], i32 2, i32 1)
+// ROW-CHECK-NEXT:    store <2 x i32> [[TMP6]], ptr [[A]], align 4
+// ROW-CHECK-NEXT:    ret void
+//
+// COL-CHECK-LABEL: define hidden void @_Z5call3v(
+// COL-CHECK-SAME: ) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[S:%.*]] = alloca [[STRUCT_S:%.*]], align 1
 // COL-CHECK-NEXT:    [[A:%.*]] = alloca [1 x <2 x i32>], align 4
-// CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_S]], align 1
-// CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
-// CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[S]], ptr align 1 @__const._Z5call3v.s, i32 8, i1 false)
-// CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[S]], i32 8, i1 false)
-// CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 0
-// CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 1
-// CHECK-NEXT:    [[TMP0:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr [[GEP]], align 4
-// CHECK-NEXT:    [[TMP2:%.*]] = insertelement <2 x i32> [[TMP0]], i32 [[TMP1]], i64 0
-// CHECK-NEXT:    [[TMP3:%.*]] = load float, ptr [[GEP1]], align 4
-// CHECK-NEXT:    [[CONV:%.*]] = fptosi float [[TMP3]] to i32
-// CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> [[TMP2]], i32 [[CONV]], i64 1
-// CHECK-NEXT:    store <2 x i32> [[TMP4]], ptr [[A]], align 4
-// CHECK-NEXT:    ret void
+// COL-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_S]], align 1
+// COL-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i32>, align 4
+// COL-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[S]], ptr align 1 @__const._Z5call3v.s, i32 8, i1 false)
+// COL-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[S]], i32 8, i1 false)
+// COL-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 0
+// COL-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_S]], ptr [[AGG_TEMP]], i32 0, i32 1
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[FLATCAST_TMP]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[GEP]], align 4
+// COL-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP1]], i32 [[TMP2]], i64 0
+// COL-CHECK-NEXT:    [[TMP4:%.*]] = load float, ptr [[GEP1]], align 4
+// COL-CHECK-NEXT:    [[CONV:%.*]] = fptosi float [[TMP4]] to i32
+// COL-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <2 x i32> [[TMP3]], i32 [[CONV]], i64 1
+// COL-CHECK-NEXT:    store <2 x i32> [[TMP5]], ptr [[A]], align 4
+// COL-CHECK-NEXT:    ret void
 //
 void call3() {
   S s = {1, 2.0};
@@ -171,76 +288,138 @@ struct Derived : BFields {
   int G;
 };
 
-// CHECK-LABEL: define hidden void @_Z5call47Derived(
-// CHECK-SAME: ptr nofreeobj noundef align 1 dead_on_return dereferenceable(19) [[D:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[D_INDIRECT_ADDR:%.*]] = alloca ptr, align 4
-// CHECK-NEXT:    [[A:%.*]] = alloca [2 x <2 x i32>], align 4
-// CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_DERIVED:.*]], align 1
-// CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x i32>, align 4
-// CHECK-NEXT:    store ptr %D, ptr [[D_INDIRECT_ADDR]], align 4
-// CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[D]], i32 19, i1 false)
-// CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0
-// CHECK-NEXT:    [[E:%.*]] = getelementptr inbounds nuw [[STRUCT_BFIELDS:%.*]], ptr [[GEP]], i32 0, i32 1
-// CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
-// CHECK-NEXT:    [[GEP2:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 2
-// CHECK-NEXT:    [[GEP3:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 1
-// CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr [[FLATCAST_TMP]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load double, ptr [[GEP1]], align 8
-// CHECK-NEXT:    [[CONV:%.*]] = fptosi double [[TMP1]] to i32
-// CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x i32> [[TMP0]], i32 [[CONV]], i64 0
-// CHECK-NEXT:    [[BF_LOAD:%.*]] = load i24, ptr [[E]], align 1
-// CHECK-NEXT:    [[BF_SHL:%.*]] = shl i24 [[BF_LOAD]], 9
-// CHECK-NEXT:    [[BF_ASHR:%.*]] = ashr i24 [[BF_SHL]], 9
-// CHECK-NEXT:    [[BF_CAST:%.*]] = sext i24 [[BF_ASHR]] to i32
-// ROW-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 [[BF_CAST]], i64 1
-// COL-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP2]], i32 [[BF_CAST]], i64 2
-// CHECK-NEXT:    [[TMP4:%.*]] = load float, ptr [[GEP2]], align 4
-// CHECK-NEXT:    [[CONV4:%.*]] = fptosi float [[TMP4]] to i32
-// ROW-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x i32> [[TMP3]], i32 [[CONV4]], i64 2
-// COL-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x i32> [[TMP3]], i32 [[CONV4]], i64 1
-// CHECK-NEXT:    [[TMP6:%.*]] = load i32, ptr [[GEP3]], align 4
-// CHECK-NEXT:    [[TMP7:%.*]] = insertelement <4 x i32> [[TMP5]], i32 [[TMP6]], i64 3
-// CHECK-NEXT:    store <4 x i32> [[TMP7]], ptr [[A]], align 4
-// CHECK-NEXT:    ret void
+// ROW-CHECK-LABEL: define hidden void @_Z5call47Derived(
+// ROW-CHECK-SAME: ptr nofreeobj noundef align 1 dead_on_return dereferenceable(19) [[D:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[D_INDIRECT_ADDR:%.*]] = alloca ptr, align 4
+// ROW-CHECK-NEXT:    [[A:%.*]] = alloca [2 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_DERIVED:%.*]], align 1
+// ROW-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x i32>, align 4
+// ROW-CHECK-NEXT:    store ptr [[D]], ptr [[D_INDIRECT_ADDR]], align 4
+// ROW-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[D]], i32 19, i1 false)
+// ROW-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0
+// ROW-CHECK-NEXT:    [[E:%.*]] = getelementptr inbounds nuw [[STRUCT_BFIELDS:%.*]], ptr [[GEP]], i32 0, i32 1
+// ROW-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
+// ROW-CHECK-NEXT:    [[GEP2:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 2
+// ROW-CHECK-NEXT:    [[GEP3:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 1
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr [[FLATCAST_TMP]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load double, ptr [[GEP1]], align 8
+// ROW-CHECK-NEXT:    [[CONV:%.*]] = fptosi double [[TMP2]] to i32
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP1]], i32 [[CONV]], i64 0
+// ROW-CHECK-NEXT:    [[BF_LOAD:%.*]] = load i24, ptr [[E]], align 1
+// ROW-CHECK-NEXT:    [[BF_SHL:%.*]] = shl i24 [[BF_LOAD]], 9
+// ROW-CHECK-NEXT:    [[BF_ASHR:%.*]] = ashr i24 [[BF_SHL]], 9
+// ROW-CHECK-NEXT:    [[BF_CAST:%.*]] = sext i24 [[BF_ASHR]] to i32
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 [[BF_CAST]], i64 1
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load float, ptr [[GEP2]], align 4
+// ROW-CHECK-NEXT:    [[CONV4:%.*]] = fptosi float [[TMP5]] to i32
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x i32> [[TMP4]], i32 [[CONV4]], i64 2
+// ROW-CHECK-NEXT:    [[TMP7:%.*]] = load i32, ptr [[GEP3]], align 4
+// ROW-CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x i32> [[TMP6]], i32 [[TMP7]], i64 3
+// ROW-CHECK-NEXT:    [[TMP9:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TMP8]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x i32> [[TMP9]], ptr [[A]], align 4
+// ROW-CHECK-NEXT:    ret void
+//
+// COL-CHECK-LABEL: define hidden void @_Z5call47Derived(
+// COL-CHECK-SAME: ptr nofreeobj noundef align 1 dead_on_return dereferenceable(19) [[D:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[D_INDIRECT_ADDR:%.*]] = alloca ptr, align 4
+// COL-CHECK-NEXT:    [[A:%.*]] = alloca [2 x <2 x i32>], align 4
+// COL-CHECK-NEXT:    [[AGG_TEMP:%.*]] = alloca [[STRUCT_DERIVED:%.*]], align 1
+// COL-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x i32>, align 4
+// COL-CHECK-NEXT:    store ptr [[D]], ptr [[D_INDIRECT_ADDR]], align 4
+// COL-CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i32(ptr align 1 [[AGG_TEMP]], ptr align 1 [[D]], i32 19, i1 false)
+// COL-CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0
+// COL-CHECK-NEXT:    [[E:%.*]] = getelementptr inbounds nuw [[STRUCT_BFIELDS:%.*]], ptr [[GEP]], i32 0, i32 1
+// COL-CHECK-NEXT:    [[GEP1:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 0
+// COL-CHECK-NEXT:    [[GEP2:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 0, i32 2
+// COL-CHECK-NEXT:    [[GEP3:%.*]] = getelementptr inbounds [[STRUCT_DERIVED]], ptr [[AGG_TEMP]], i32 0, i32 1
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr [[FLATCAST_TMP]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load double, ptr [[GEP1]], align 8
+// COL-CHECK-NEXT:    [[CONV:%.*]] = fptosi double [[TMP2]] to i32
+// COL-CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x i32> [[TMP1]], i32 [[CONV]], i64 0
+// COL-CHECK-NEXT:    [[BF_LOAD:%.*]] = load i24, ptr [[E]], align 1
+// COL-CHECK-NEXT:    [[BF_SHL:%.*]] = shl i24 [[BF_LOAD]], 9
+// COL-CHECK-NEXT:    [[BF_ASHR:%.*]] = ashr i24 [[BF_SHL]], 9
+// COL-CHECK-NEXT:    [[BF_CAST:%.*]] = sext i24 [[BF_ASHR]] to i32
+// COL-CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x i32> [[TMP3]], i32 [[BF_CAST]], i64 2
+// COL-CHECK-NEXT:    [[TMP5:%.*]] = load float, ptr [[GEP2]], align 4
+// COL-CHECK-NEXT:    [[CONV4:%.*]] = fptosi float [[TMP5]] to i32
+// COL-CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x i32> [[TMP4]], i32 [[CONV4]], i64 1
+// COL-CHECK-NEXT:    [[TMP7:%.*]] = load i32, ptr [[GEP3]], align 4
+// COL-CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x i32> [[TMP6]], i32 [[TMP7]], i64 3
+// COL-CHECK-NEXT:    store <4 x i32> [[TMP8]], ptr [[A]], align 4
+// COL-CHECK-NEXT:    ret void
 //
 void call4(Derived D) {
   int2x2 A = (int2x2)D;
 }
 
-// CHECK-LABEL: define hidden noundef nofpclass(nan inf) <4 x float> @_Z5call5Dv4_f(
-// CHECK-SAME: <4 x float> noundef nofpclass(nan inf) [[V:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[V_ADDR:%.*]] = alloca <4 x float>, align 4
-// CHECK-NEXT:    [[M:%.*]] = alloca [2 x <2 x float>], align 4
-// CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca <4 x float>, align 4
-// CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x float>, align 4
-// CHECK-NEXT:    store <4 x float> [[V]], ptr [[V_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[V_ADDR]], align 4
-// CHECK-NEXT:    store <4 x float> [[TMP0]], ptr [[HLSL_EWCAST_SRC]], align 4
-// CHECK-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr inbounds <4 x float>, ptr [[HLSL_EWCAST_SRC]], i32 0
-// CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[FLATCAST_TMP]], align 4
-// CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
-// CHECK-NEXT:    [[VECEXT:%.*]] = extractelement <4 x float> [[TMP2]], i32 0
-// CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x float> [[TMP1]], float [[VECEXT]], i64 0
-// CHECK-NEXT:    [[TMP4:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
-// CHECK-NEXT:    [[VECEXT1:%.*]] = extractelement <4 x float> [[TMP4]], i32 1
-// ROW-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x float> [[TMP3]], float [[VECEXT1]], i64 1
-// COL-CHECK-NEXT:    [[TMP5:%.*]] = insertelement <4 x float> [[TMP3]], float [[VECEXT1]], i64 2
-// CHECK-NEXT:    [[TMP6:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
-// CHECK-NEXT:    [[VECEXT2:%.*]] = extractelement <4 x float> [[TMP6]], i32 2
-// ROW-CHECK-NEXT:    [[TMP7:%.*]] = insertelement <4 x float> [[TMP5]], float [[VECEXT2]], i64 2
-// COL-CHECK-NEXT:    [[TMP7:%.*]] = insertelement <4 x float> [[TMP5]], float [[VECEXT2]], i64 1
-// CHECK-NEXT:    [[TMP8:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
-// CHECK-NEXT:    [[VECEXT3:%.*]] = extractelement <4 x float> [[TMP8]], i32 3
-// CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x float> [[TMP7]], float [[VECEXT3]], i64 3
-// CHECK-NEXT:    store <4 x float> [[TMP9]], ptr [[M]], align 4
-// CHECK-NEXT:    [[TMP10:%.*]] = load <4 x float>, ptr [[M]], align 4
-// CHECK-NEXT:    ret <4 x float> [[TMP10]]
+// ROW-CHECK-LABEL: define hidden noundef nofpclass(nan inf) <4 x float> @_Z5call5Dv4_f(
+// ROW-CHECK-SAME: <4 x float> noundef nofpclass(nan inf) [[V:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[V_ADDR:%.*]] = alloca <4 x float>, align 4
+// ROW-CHECK-NEXT:    [[M:%.*]] = alloca [2 x <2 x float>], align 4
+// ROW-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca <4 x float>, align 4
+// ROW-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x float>, align 4
+// ROW-CHECK-NEXT:    store <4 x float> [[V]], ptr [[V_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[V_ADDR]], align 4
+// ROW-CHECK-NEXT:    store <4 x float> [[TMP1]], ptr [[HLSL_EWCAST_SRC]], align 4
+// ROW-CHECK-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr inbounds <4 x float>, ptr [[HLSL_EWCAST_SRC]], i32 0
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[FLATCAST_TMP]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// ROW-CHECK-NEXT:    [[VECEXT:%.*]] = extractelement <4 x float> [[TMP3]], i32 0
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x float> [[TMP2]], float [[VECEXT]], i64 0
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// ROW-CHECK-NEXT:    [[VECEXT1:%.*]] = extractelement <4 x float> [[TMP5]], i32 1
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x float> [[TMP4]], float [[VECEXT1]], i64 1
+// ROW-CHECK-NEXT:    [[TMP7:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// ROW-CHECK-NEXT:    [[VECEXT2:%.*]] = extractelement <4 x float> [[TMP7]], i32 2
+// ROW-CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x float> [[TMP6]], float [[VECEXT2]], i64 2
+// ROW-CHECK-NEXT:    [[TMP9:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// ROW-CHECK-NEXT:    [[VECEXT3:%.*]] = extractelement <4 x float> [[TMP9]], i32 3
+// ROW-CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> [[TMP8]], float [[VECEXT3]], i64 3
+// ROW-CHECK-NEXT:    [[TMP11:%.*]] = call reassoc nnan ninf nsz arcp afn <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP10]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x float> [[TMP11]], ptr [[M]], align 4
+// ROW-CHECK-NEXT:    [[TMP12:%.*]] = load <4 x float>, ptr [[M]], align 4
+// ROW-CHECK-NEXT:    [[TMP13:%.*]] = call reassoc nnan ninf nsz arcp afn <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP12]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    ret <4 x float> [[TMP13]]
+//
+// COL-CHECK-LABEL: define hidden noundef nofpclass(nan inf) <4 x float> @_Z5call5Dv4_f(
+// COL-CHECK-SAME: <4 x float> noundef nofpclass(nan inf) [[V:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[V_ADDR:%.*]] = alloca <4 x float>, align 4
+// COL-CHECK-NEXT:    [[M:%.*]] = alloca [2 x <2 x float>], align 4
+// COL-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca <4 x float>, align 4
+// COL-CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x float>, align 4
+// COL-CHECK-NEXT:    store <4 x float> [[V]], ptr [[V_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[V_ADDR]], align 4
+// COL-CHECK-NEXT:    store <4 x float> [[TMP1]], ptr [[HLSL_EWCAST_SRC]], align 4
+// COL-CHECK-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr inbounds <4 x float>, ptr [[HLSL_EWCAST_SRC]], i32 0
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[FLATCAST_TMP]], align 4
+// COL-CHECK-NEXT:    [[TMP3:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// COL-CHECK-NEXT:    [[VECEXT:%.*]] = extractelement <4 x float> [[TMP3]], i32 0
+// COL-CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x float> [[TMP2]], float [[VECEXT]], i64 0
+// COL-CHECK-NEXT:    [[TMP5:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// COL-CHECK-NEXT:    [[VECEXT1:%.*]] = extractelement <4 x float> [[TMP5]], i32 1
+// COL-CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x float> [[TMP4]], float [[VECEXT1]], i64 2
+// COL-CHECK-NEXT:    [[TMP7:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// COL-CHECK-NEXT:    [[VECEXT2:%.*]] = extractelement <4 x float> [[TMP7]], i32 2
+// COL-CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x float> [[TMP6]], float [[VECEXT2]], i64 1
+// COL-CHECK-NEXT:    [[TMP9:%.*]] = load <4 x float>, ptr [[VECTOR_GEP]], align 4
+// COL-CHECK-NEXT:    [[VECEXT3:%.*]] = extractelement <4 x float> [[TMP9]], i32 3
+// COL-CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> [[TMP8]], float [[VECEXT3]], i64 3
+// COL-CHECK-NEXT:    store <4 x float> [[TMP10]], ptr [[M]], align 4
+// COL-CHECK-NEXT:    [[TMP11:%.*]] = load <4 x float>, ptr [[M]], align 4
+// COL-CHECK-NEXT:    ret <4 x float> [[TMP11]]
 //
 float2x2 call5(float4 v) {
     float2x2 m = (float2x2)v;
     return m;
 }
+//// NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+// CHECK: {{.*}}
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/MatrixExplicitTruncation.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/MatrixExplicitTruncation.hlsl
index 4ab82faac41784..85dd38f8b26477 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/MatrixExplicitTruncation.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/MatrixExplicitTruncation.hlsl
@@ -2,176 +2,317 @@
 // RUN: %clang_cc1 -triple dxil-pc-shadermodel6.7-library -disable-llvm-passes -emit-llvm -finclude-default-header -fmatrix-memory-layout=row-major -o - %s | FileCheck %s --check-prefixes=CHECK,ROW-CHECK
 // RUN: %clang_cc1 -triple dxil-pc-shadermodel6.7-library -disable-llvm-passes -emit-llvm -finclude-default-header -fmatrix-memory-layout=column-major -o - %s | FileCheck %s --check-prefixes=CHECK,COL-CHECK
 
-// CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I34:%.*]] = alloca [3 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TRUNC]], i32 3, i32 4)
+// ROW-CHECK-NEXT:    store <12 x i32> [[TMP4]], ptr [[I34]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <12 x i32>, ptr [[I34]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TMP5]], i32 4, i32 3)
+// ROW-CHECK-NEXT:    ret <12 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I34:%.*]] = alloca [4 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
-// CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I34]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <12 x i32>, ptr [[I34]], align 4
-// CHECK-NEXT:    ret <12 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// COL-CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I34]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <12 x i32>, ptr [[I34]], align 4
+// COL-CHECK-NEXT:    ret <12 x i32> [[TMP2]]
 //
  int3x4 trunc_cast(int4x4 i44) {
     int3x4 i34 = (int3x4)i44;
     return i34;
 }
 
-// CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I43:%.*]] = alloca [4 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TRUNC]], i32 4, i32 3)
+// ROW-CHECK-NEXT:    store <12 x i32> [[TMP4]], ptr [[I43]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <12 x i32>, ptr [[I43]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TMP5]], i32 3, i32 4)
+// ROW-CHECK-NEXT:    ret <12 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I43:%.*]] = alloca [3 x <4 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
-// CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I43]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <12 x i32>, ptr [[I43]], align 4
-// CHECK-NEXT:    ret <12 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
+// COL-CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I43]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <12 x i32>, ptr [[I43]], align 4
+// COL-CHECK-NEXT:    ret <12 x i32> [[TMP2]]
 //
  int4x3 trunc_cast0(int4x4 i44) {
     int4x3 i43 = (int4x3)i44;
     return i43;
 }
 
-// CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
-// CHECK-NEXT:    store <9 x i32> [[TRUNC]], ptr [[I33]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <9 x i32>, ptr [[I33]], align 4
-// CHECK-NEXT:    ret <9 x i32> [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <9 x i32> @llvm.matrix.transpose.v9i32(<9 x i32> [[TRUNC]], i32 3, i32 3)
+// ROW-CHECK-NEXT:    store <9 x i32> [[TMP4]], ptr [[I33]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <9 x i32>, ptr [[I33]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <9 x i32> @llvm.matrix.transpose.v9i32(<9 x i32> [[TMP5]], i32 3, i32 3)
+// ROW-CHECK-NEXT:    ret <9 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
+// COL-CHECK-NEXT:    store <9 x i32> [[TRUNC]], ptr [[I33]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <9 x i32>, ptr [[I33]], align 4
+// COL-CHECK-NEXT:    ret <9 x i32> [[TMP2]]
 //
  int3x3 trunc_cast1(int4x4 i44) {
     int3x3 i33 = (int3x3)i44;
     return i33;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TRUNC]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
-// CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
+// COL-CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
  int3x2 trunc_cast2(int4x4 i44) {
     int3x2 i32 = (int3x2)i44;
     return i32;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I23:%.*]] = alloca [2 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TRUNC]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I23]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I23]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I23:%.*]] = alloca [3 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
-// CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I23]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I23]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
+// COL-CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I23]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I23]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
  int2x3 trunc_cast3(int4x4 i44) {
     int2x3 i23 = (int2x3)i44;
     return i23;
 }
 
-// CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-// CHECK-NEXT:    store <4 x i32> [[TRUNC]], ptr [[I22]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr [[I22]], align 4
-// CHECK-NEXT:    ret <4 x i32> [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TRUNC]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x i32> [[TMP4]], ptr [[I22]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <4 x i32>, ptr [[I22]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TMP5]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    ret <4 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+// COL-CHECK-NEXT:    store <4 x i32> [[TRUNC]], ptr [[I22]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr [[I22]], align 4
+// COL-CHECK-NEXT:    ret <4 x i32> [[TMP2]]
 //
  int2x2 trunc_cast4(int4x4 i44) {
     int2x2 i22 = (int2x2)i44;
     return i22;
 }
 
-// CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I21:%.*]] = alloca [2 x <1 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TRUNC]], i32 2, i32 1)
+// ROW-CHECK-NEXT:    store <2 x i32> [[TMP4]], ptr [[I21]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <2 x i32>, ptr [[I21]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP5]], i32 1, i32 2)
+// ROW-CHECK-NEXT:    ret <2 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I21:%.*]] = alloca [1 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <2 x i32> <i32 0, i32 4>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
-// CHECK-NEXT:    store <2 x i32> [[TRUNC]], ptr [[I21]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[I21]], align 4
-// CHECK-NEXT:    ret <2 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
+// COL-CHECK-NEXT:    store <2 x i32> [[TRUNC]], ptr [[I21]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <2 x i32>, ptr [[I21]], align 4
+// COL-CHECK-NEXT:    ret <2 x i32> [[TMP2]]
 //
  int2x1 trunc_cast5(int4x4 i44) {
     int2x1 i21 = (int2x1)i44;
     return i21;
 }
 
-// CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP0]], i32 0
-// CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr [[I1]], align 4
-// CHECK-NEXT:    ret i32 [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP3]], i32 0
+// ROW-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    ret i32 [[TMP4]]
+//
+// COL-CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP1]], i32 0
+// COL-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[I1]], align 4
+// COL-CHECK-NEXT:    ret i32 [[TMP2]]
 //
  int trunc_cast6(int4x4 i44) {
     int i1 = (int)i44;
     return i1;
 }
 
-// CHECK-LABEL: define hidden noundef i32 @_Z16trunc_multi_castu11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
-// CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <12 x i32> [[TRUNC]], i32 0
-// CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr [[I1]], align 4
-// CHECK-NEXT:    ret i32 [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef i32 @_Z16trunc_multi_castu11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// ROW-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <12 x i32> [[TRUNC]], i32 0
+// ROW-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    ret i32 [[TMP4]]
+//
+// COL-CHECK-LABEL: define hidden noundef i32 @_Z16trunc_multi_castu11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// COL-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <12 x i32> [[TRUNC]], i32 0
+// COL-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[I1]], align 4
+// COL-CHECK-NEXT:    ret i32 [[TMP2]]
 //
  int trunc_multi_cast(int4x4 i44) {
     int i1 = (int)(int3x4)i44;
     return i1;
 }
+//// NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+// CHECK: {{.*}}
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/MatrixImplicitTruncation.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/MatrixImplicitTruncation.hlsl
index 52105b8c615d0e..05de54a6f01451 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/MatrixImplicitTruncation.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/MatrixImplicitTruncation.hlsl
@@ -2,156 +2,282 @@
 // RUN: %clang_cc1 -triple dxil-pc-shadermodel6.7-library -disable-llvm-passes -emit-llvm -finclude-default-header -fmatrix-memory-layout=row-major -o - %s | FileCheck %s --check-prefixes=CHECK,ROW-CHECK
 // RUN: %clang_cc1 -triple dxil-pc-shadermodel6.7-library -disable-llvm-passes -emit-llvm -finclude-default-header -fmatrix-memory-layout=column-major -o - %s | FileCheck %s --check-prefixes=CHECK,COL-CHECK
 
-// CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I34:%.*]] = alloca [3 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TRUNC]], i32 3, i32 4)
+// ROW-CHECK-NEXT:    store <12 x i32> [[TMP4]], ptr [[I34]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <12 x i32>, ptr [[I34]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TMP5]], i32 4, i32 3)
+// ROW-CHECK-NEXT:    ret <12 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <12 x i32> @_Z10trunc_castu11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0:[0-9]+]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I34:%.*]] = alloca [4 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
-// CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I34]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <12 x i32>, ptr [[I34]], align 4
-// CHECK-NEXT:    ret <12 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
+// COL-CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I34]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <12 x i32>, ptr [[I34]], align 4
+// COL-CHECK-NEXT:    ret <12 x i32> [[TMP2]]
 //
  int3x4 trunc_cast(int4x4 i44) {
     int3x4 i34 = i44;
     return i34;
 }
 
-// CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I43:%.*]] = alloca [4 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TRUNC]], i32 4, i32 3)
+// ROW-CHECK-NEXT:    store <12 x i32> [[TMP4]], ptr [[I43]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <12 x i32>, ptr [[I43]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[TMP5]], i32 3, i32 4)
+// ROW-CHECK-NEXT:    ret <12 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <12 x i32> @_Z11trunc_cast0u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I43:%.*]] = alloca [3 x <4 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10, i32 12, i32 13, i32 14>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
-// CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I43]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <12 x i32>, ptr [[I43]], align 4
-// CHECK-NEXT:    ret <12 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <12 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8, i32 9, i32 10, i32 11>
+// COL-CHECK-NEXT:    store <12 x i32> [[TRUNC]], ptr [[I43]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <12 x i32>, ptr [[I43]], align 4
+// COL-CHECK-NEXT:    ret <12 x i32> [[TMP2]]
 //
  int4x3 trunc_cast0(int4x4 i44) {
     int4x3 i43 = i44;
     return i43;
 }
 
-// CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
-// CHECK-NEXT:    store <9 x i32> [[TRUNC]], ptr [[I33]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <9 x i32>, ptr [[I33]], align 4
-// CHECK-NEXT:    ret <9 x i32> [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <9 x i32> @llvm.matrix.transpose.v9i32(<9 x i32> [[TRUNC]], i32 3, i32 3)
+// ROW-CHECK-NEXT:    store <9 x i32> [[TMP4]], ptr [[I33]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <9 x i32>, ptr [[I33]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <9 x i32> @llvm.matrix.transpose.v9i32(<9 x i32> [[TMP5]], i32 3, i32 3)
+// ROW-CHECK-NEXT:    ret <9 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <9 x i32> @_Z11trunc_cast1u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I33:%.*]] = alloca [3 x <3 x i32>], align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6, i32 8, i32 9, i32 10>
+// COL-CHECK-NEXT:    store <9 x i32> [[TRUNC]], ptr [[I33]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <9 x i32>, ptr [[I33]], align 4
+// COL-CHECK-NEXT:    ret <9 x i32> [[TMP2]]
 //
  int3x3 trunc_cast1(int4x4 i44) {
     int3x3 i33 = i44;
     return i33;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I32:%.*]] = alloca [3 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TRUNC]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast2u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I32:%.*]] = alloca [2 x <3 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
-// CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I32]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I32]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
+// COL-CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I32]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I32]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
  int3x2 trunc_cast2(int4x4 i44) {
     int3x2 i32 = i44;
     return i32;
 }
 
-// CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I23:%.*]] = alloca [2 x <3 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TRUNC]], i32 2, i32 3)
+// ROW-CHECK-NEXT:    store <6 x i32> [[TMP4]], ptr [[I23]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <6 x i32>, ptr [[I23]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[TMP5]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    ret <6 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <6 x i32> @_Z11trunc_cast3u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I23:%.*]] = alloca [3 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 2, i32 4, i32 5, i32 6>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
-// CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I23]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <6 x i32>, ptr [[I23]], align 4
-// CHECK-NEXT:    ret <6 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <6 x i32> <i32 0, i32 1, i32 4, i32 5, i32 8, i32 9>
+// COL-CHECK-NEXT:    store <6 x i32> [[TRUNC]], ptr [[I23]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <6 x i32>, ptr [[I23]], align 4
+// COL-CHECK-NEXT:    ret <6 x i32> [[TMP2]]
 //
  int2x3 trunc_cast3(int4x4 i44) {
     int2x3 i23 = i44;
     return i23;
 }
 
-// CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
-// CHECK-NEXT:    store <4 x i32> [[TRUNC]], ptr [[I22]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr [[I22]], align 4
-// CHECK-NEXT:    ret <4 x i32> [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TRUNC]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x i32> [[TMP4]], ptr [[I22]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <4 x i32>, ptr [[I22]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TMP5]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    ret <4 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <4 x i32> @_Z11trunc_cast4u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I22:%.*]] = alloca [2 x <2 x i32>], align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <4 x i32> <i32 0, i32 1, i32 4, i32 5>
+// COL-CHECK-NEXT:    store <4 x i32> [[TRUNC]], ptr [[I22]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr [[I22]], align 4
+// COL-CHECK-NEXT:    ret <4 x i32> [[TMP2]]
 //
  int2x2 trunc_cast4(int4x4 i44) {
     int2x2 i22 = i44;
     return i22;
 }
 
-// CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // ROW-CHECK-NEXT:    [[I21:%.*]] = alloca [2 x <1 x i32>], align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP3]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TRUNC]], i32 2, i32 1)
+// ROW-CHECK-NEXT:    store <2 x i32> [[TMP4]], ptr [[I21]], align 4
+// ROW-CHECK-NEXT:    [[TMP5:%.*]] = load <2 x i32>, ptr [[I21]], align 4
+// ROW-CHECK-NEXT:    [[TMP6:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP5]], i32 1, i32 2)
+// ROW-CHECK-NEXT:    ret <2 x i32> [[TMP6]]
+//
+// COL-CHECK-LABEL: define hidden noundef <2 x i32> @_Z11trunc_cast5u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
 // COL-CHECK-NEXT:    [[I21:%.*]] = alloca [1 x <2 x i32>], align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// ROW-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <2 x i32> <i32 0, i32 4>
-// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP0]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
-// CHECK-NEXT:    store <2 x i32> [[TRUNC]], ptr [[I21]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[I21]], align 4
-// CHECK-NEXT:    ret <2 x i32> [[TMP1]]
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TRUNC:%.*]] = shufflevector <16 x i32> [[TMP1]], <16 x i32> poison, <2 x i32> <i32 0, i32 1>
+// COL-CHECK-NEXT:    store <2 x i32> [[TRUNC]], ptr [[I21]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load <2 x i32>, ptr [[I21]], align 4
+// COL-CHECK-NEXT:    ret <2 x i32> [[TMP2]]
 //
  int2x1 trunc_cast5(int4x4 i44) {
     int2x1 i21 = i44;
     return i21;
 }
 
-// CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
-// CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
-// CHECK-NEXT:  [[ENTRY:.*:]]
-// CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
-// CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
-// CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
-// CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
-// CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP0]], i32 0
-// CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
-// CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr [[I1]], align 4
-// CHECK-NEXT:    ret i32 [[TMP1]]
+// ROW-CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
+// ROW-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// ROW-CHECK-NEXT:  [[ENTRY:.*:]]
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// ROW-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// ROW-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// ROW-CHECK-NEXT:    [[TMP1:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[I44]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    store <16 x i32> [[TMP1]], ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP2:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[TMP3:%.*]] = call <16 x i32> @llvm.matrix.transpose.v16i32(<16 x i32> [[TMP2]], i32 4, i32 4)
+// ROW-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP3]], i32 0
+// ROW-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    [[TMP4:%.*]] = load i32, ptr [[I1]], align 4
+// ROW-CHECK-NEXT:    ret i32 [[TMP4]]
+//
+// COL-CHECK-LABEL: define hidden noundef i32 @_Z11trunc_cast6u11matrix_typeILm4ELm4EiE(
+// COL-CHECK-SAME: <16 x i32> noundef [[I44:%.*]]) #[[ATTR0]] {
+// COL-CHECK-NEXT:  [[ENTRY:.*:]]
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = call token @llvm.experimental.convergence.entry()
+// COL-CHECK-NEXT:    [[I44_ADDR:%.*]] = alloca [4 x <4 x i32>], align 4
+// COL-CHECK-NEXT:    [[I1:%.*]] = alloca i32, align 4
+// COL-CHECK-NEXT:    store <16 x i32> [[I44]], ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[TMP1:%.*]] = load <16 x i32>, ptr [[I44_ADDR]], align 4
+// COL-CHECK-NEXT:    [[CAST_MTRUNC:%.*]] = extractelement <16 x i32> [[TMP1]], i32 0
+// COL-CHECK-NEXT:    store i32 [[CAST_MTRUNC]], ptr [[I1]], align 4
+// COL-CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr [[I1]], align 4
+// COL-CHECK-NEXT:    ret i32 [[TMP2]]
 //
  int trunc_cast6(int4x4 i44) {
     int i1 = i44;
     return i1;
 }
+//// NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+// CHECK: {{.*}}
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/MatrixInitializerListOrder.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/MatrixInitializerListOrder.hlsl
index da4a522ea28ea8..97ebe1712e9c76 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/MatrixInitializerListOrder.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/MatrixInitializerListOrder.hlsl
@@ -4,21 +4,18 @@
 // RUN:   -emit-llvm -finclude-default-header -fmatrix-memory-layout=row-major -o - %s \
 // RUN:   | FileCheck %s --check-prefix=CHECK,ROW-CHECK
 
-// Verify that matrix initializer lists store elements in the correct memory
-// layout. The initializer list {1,2,3,4,5,6} for a float2x3 (2 rows, 3 cols)
-// is in row-major order: row0=[1,2,3], row1=[4,5,6].
+// Verify that matrix initializer lists produce values in canonical column-major
+// register layout. The initializer list {1,2,3,4,5,6} for a float2x3 (2 rows,
+// 3 cols) is in row-major order: row0=[1,2,3], row1=[4,5,6].
 //
-// With column-major (default) memory layout, the stored vector should be
-// reordered to: col0=[1,4], col1=[2,5], col2=[3,6] = <1,4,2,5,3,6>.
-//
-// With row-major memory layout, the stored vector stays as-is: <1,2,3,4,5,6>.
+// The register value is reordered to col0=[1,4], col1=[2,5], col2=[3,6] =
+// <1,4,2,5,3,6>. Row-major memory layout transposes that value at the store.
 
 export float test_row0_col2() {
 // CHECK-LABEL: define {{.*}} float @_Z14test_row0_col2v
 // COL-CHECK: store <6 x float> <float 1.000000e+00, float 4.000000e+00, float 2.000000e+00, float 5.000000e+00, float 3.000000e+00, float 6.000000e+00>
-// COL-CHECK: extractelement <6 x float> %{{.*}}, i32 4
-// ROW-CHECK: store <6 x float> <float 1.000000e+00, float 2.000000e+00, float 3.000000e+00, float 4.000000e+00, float 5.000000e+00, float 6.000000e+00>
-// ROW-CHECK: extractelement <6 x float> %{{.*}}, i32 2
+// ROW-CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> <float 1.000000e+00, float 4.000000e+00, float 2.000000e+00, float 5.000000e+00, float 3.000000e+00, float 6.000000e+00>, i32 2, i32 3)
+// CHECK: extractelement <6 x float> %{{.*}}, i32 4
   float2x3 M = {1.0, 2.0, 3.0, 4.0, 5.0, 6.0};
   // Row 0, Col 2 in row-major is the 3rd element = 3.0
   return M[0][2];
@@ -27,9 +24,8 @@ export float test_row0_col2() {
 export float test_row1_col0() {
 // CHECK-LABEL: define {{.*}} float @_Z14test_row1_col0v
 // COL-CHECK: store <6 x float> <float 1.000000e+00, float 4.000000e+00, float 2.000000e+00, float 5.000000e+00, float 3.000000e+00, float 6.000000e+00>
-// COL-CHECK: extractelement <6 x float> %{{.*}}, i32 1
-// ROW-CHECK: store <6 x float> <float 1.000000e+00, float 2.000000e+00, float 3.000000e+00, float 4.000000e+00, float 5.000000e+00, float 6.000000e+00>
-// ROW-CHECK: extractelement <6 x float> %{{.*}}, i32 3
+// ROW-CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> <float 1.000000e+00, float 4.000000e+00, float 2.000000e+00, float 5.000000e+00, float 3.000000e+00, float 6.000000e+00>, i32 2, i32 3)
+// CHECK: extractelement <6 x float> %{{.*}}, i32 1
   float2x3 M = {1.0, 2.0, 3.0, 4.0, 5.0, 6.0};
   // Row 1, Col 0 in row-major is the 4th element = 4.0
   return M[1][0];
@@ -43,17 +39,13 @@ export float2x3 test_dynamic(float a, float b, float c,
 // CHECK: [[A:%.*]] = load float, ptr %a.addr
 // CHECK: [[VECINIT0:%.*]] = insertelement <6 x float> poison, float [[A]], i32 0
 // CHECK: [[B:%.*]] = load float, ptr %b.addr
-// COL-CHECK: [[VECINIT1:%.*]] = insertelement <6 x float> [[VECINIT0]], float [[B]], i32 2
-// ROW-CHECK: [[VECINIT1:%.*]] = insertelement <6 x float> [[VECINIT0]], float [[B]], i32 1
+// CHECK: [[VECINIT1:%.*]] = insertelement <6 x float> [[VECINIT0]], float [[B]], i32 2
 // CHECK: [[C:%.*]] = load float, ptr %c.addr
-// COL-CHECK: [[VECINIT2:%.*]] = insertelement <6 x float> [[VECINIT1]], float [[C]], i32 4
-// ROW-CHECK: [[VECINIT2:%.*]] = insertelement <6 x float> [[VECINIT1]], float [[C]], i32 2
+// CHECK: [[VECINIT2:%.*]] = insertelement <6 x float> [[VECINIT1]], float [[C]], i32 4
 // CHECK: [[D:%.*]] = load float, ptr %d.addr
-// COL-CHECK: [[VECINIT3:%.*]] = insertelement <6 x float> [[VECINIT2]], float [[D]], i32 1
-// ROW-CHECK: [[VECINIT3:%.*]] = insertelement <6 x float> [[VECINIT2]], float [[D]], i32 3
+// CHECK: [[VECINIT3:%.*]] = insertelement <6 x float> [[VECINIT2]], float [[D]], i32 1
 // CHECK: [[E:%.*]] = load float, ptr %e.addr
-// COL-CHECK: [[VECINIT4:%.*]] = insertelement <6 x float> [[VECINIT3]], float [[E]], i32 3
-// ROW-CHECK: [[VECINIT4:%.*]] = insertelement <6 x float> [[VECINIT3]], float [[E]], i32 4
+// CHECK: [[VECINIT4:%.*]] = insertelement <6 x float> [[VECINIT3]], float [[E]], i32 3
 // CHECK: [[F:%.*]] = load float, ptr %f.addr
 // CHECK: [[VECINIT5:%.*]] = insertelement <6 x float> [[VECINIT4]], float [[F]], i32 5
   return (float2x3){a, b, c, d, e, f};
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/MatrixToAndFromVectorConstructors.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/MatrixToAndFromVectorConstructors.hlsl
index 358f976126f55c..b8d47f8f30b41b 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/MatrixToAndFromVectorConstructors.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/MatrixToAndFromVectorConstructors.hlsl
@@ -7,20 +7,28 @@
 // CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
 // CHECK-NEXT:    [[M_ADDR:%.*]] = alloca [2 x <2 x float>], align 4
 // CHECK-NEXT:    [[V:%.*]] = alloca <4 x float>, align 4
-// CHECK-NEXT:    store <4 x float> [[M]], ptr [[M_ADDR]], align 4
+// COL-CHECK-NEXT:    store <4 x float> [[M]], ptr [[M_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[M]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x float> [[M_ROW]], ptr [[M_ADDR]], align 4
 // CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[M_ADDR]], align 4
-// CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <4 x float> [[TMP0]], i32 0
+// COL-CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <4 x float> [[TMP0]], i32 0
+// ROW-CHECK-NEXT:    [[TMP0_COL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP0]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <4 x float> [[TMP0_COL]], i32 0
 // CHECK-NEXT:    [[VECINIT:%.*]] = insertelement <4 x float> poison, float [[MATRIXEXT]], i32 0
 // CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[M_ADDR]], align 4
 // COL-CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <4 x float> [[TMP1]], i32 2
-// ROW-CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <4 x float> [[TMP1]], i32 1
+// ROW-CHECK-NEXT:    [[TMP1_COL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP1]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <4 x float> [[TMP1_COL]], i32 2
 // CHECK-NEXT:    [[VECINIT2:%.*]] = insertelement <4 x float> [[VECINIT]], float [[MATRIXEXT1]], i32 1
 // CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[M_ADDR]], align 4
 // COL-CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <4 x float> [[TMP2]], i32 1
-// ROW-CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <4 x float> [[TMP2]], i32 2
+// ROW-CHECK-NEXT:    [[TMP2_COL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP2]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <4 x float> [[TMP2_COL]], i32 1
 // CHECK-NEXT:    [[VECINIT4:%.*]] = insertelement <4 x float> [[VECINIT2]], float [[MATRIXEXT3]], i32 2
 // CHECK-NEXT:    [[TMP3:%.*]] = load <4 x float>, ptr [[M_ADDR]], align 4
-// CHECK-NEXT:    [[MATRIXEXT5:%.*]] = extractelement <4 x float> [[TMP3]], i32 3
+// COL-CHECK-NEXT:    [[MATRIXEXT5:%.*]] = extractelement <4 x float> [[TMP3]], i32 3
+// ROW-CHECK-NEXT:    [[TMP3_COL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP3]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    [[MATRIXEXT5:%.*]] = extractelement <4 x float> [[TMP3_COL]], i32 3
 // CHECK-NEXT:    [[VECINIT6:%.*]] = insertelement <4 x float> [[VECINIT4]], float [[MATRIXEXT5]], i32 3
 // CHECK-NEXT:    store <4 x float> [[VECINIT6]], ptr [[V]], align 4
 // CHECK-NEXT:    [[TMP4:%.*]] = load <4 x float>, ptr [[V]], align 4
@@ -43,18 +51,20 @@ float4 fn(float2x2 m) {
 // CHECK-NEXT:    [[VECINIT:%.*]] = insertelement <4 x i32> poison, i32 [[VECEXT]], i32 0
 // CHECK-NEXT:    [[TMP1:%.*]] = load <4 x i32>, ptr [[V_ADDR]], align 4
 // CHECK-NEXT:    [[VECEXT1:%.*]] = extractelement <4 x i32> [[TMP1]], i64 1
-// COL-CHECK-NEXT:    [[VECINIT2:%.*]] = insertelement <4 x i32> [[VECINIT]], i32 [[VECEXT1]], i32 2
-// ROW-CHECK-NEXT:    [[VECINIT2:%.*]] = insertelement <4 x i32> [[VECINIT]], i32 [[VECEXT1]], i32 1
+// CHECK-NEXT:    [[VECINIT2:%.*]] = insertelement <4 x i32> [[VECINIT]], i32 [[VECEXT1]], i32 2
 // CHECK-NEXT:    [[TMP2:%.*]] = load <4 x i32>, ptr [[V_ADDR]], align 4
 // CHECK-NEXT:    [[VECEXT3:%.*]] = extractelement <4 x i32> [[TMP2]], i64 2
-// COL-CHECK-NEXT:    [[VECINIT4:%.*]] = insertelement <4 x i32> [[VECINIT2]], i32 [[VECEXT3]], i32 1
-// ROW-CHECK-NEXT:    [[VECINIT4:%.*]] = insertelement <4 x i32> [[VECINIT2]], i32 [[VECEXT3]], i32 2
+// CHECK-NEXT:    [[VECINIT4:%.*]] = insertelement <4 x i32> [[VECINIT2]], i32 [[VECEXT3]], i32 1
 // CHECK-NEXT:    [[TMP3:%.*]] = load <4 x i32>, ptr [[V_ADDR]], align 4
 // CHECK-NEXT:    [[VECEXT5:%.*]] = extractelement <4 x i32> [[TMP3]], i64 3
 // CHECK-NEXT:    [[VECINIT6:%.*]] = insertelement <4 x i32> [[VECINIT4]], i32 [[VECEXT5]], i32 3
-// CHECK-NEXT:    store <4 x i32> [[VECINIT6]], ptr [[M]], align 4
+// COL-CHECK-NEXT:    store <4 x i32> [[VECINIT6]], ptr [[M]], align 4
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[VECINIT6]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x i32> [[M_ROW]], ptr [[M]], align 4
 // CHECK-NEXT:    [[TMP4:%.*]] = load <4 x i32>, ptr [[M]], align 4
-// CHECK-NEXT:    ret <4 x i32> [[TMP4]]
+// COL-CHECK-NEXT:    ret <4 x i32> [[TMP4]]
+// ROW-CHECK-NEXT:    [[TMP4_COL:%.*]] = call <4 x i32> @llvm.matrix.transpose.v4i32(<4 x i32> [[TMP4]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    ret <4 x i32> [[TMP4_COL]]
 //
 int2x2 fn(int4 v) {
     int2x2 m = v;
@@ -110,16 +120,24 @@ bool3x1 fn2(bool3 b) {
 // CHECK-NEXT:    %[[#C_ENTRY:]] = call token @llvm.experimental.convergence.entry()
 // COL-CHECK-NEXT:    [[B_ADDR:%.*]] = alloca [3 x <1 x i32>], align 4
 // ROW-CHECK-NEXT:    [[B_ADDR:%.*]] = alloca [1 x <3 x i32>], align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = zext <3 x i1> [[B]] to <3 x i32>
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = zext <3 x i1> [[B]] to <3 x i32>
+// ROW-CHECK-NEXT:    [[B_ROW:%.*]] = call <3 x i1> @llvm.matrix.transpose.v3i1(<3 x i1> [[B]], i32 1, i32 3)
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = zext <3 x i1> [[B_ROW]] to <3 x i32>
 // CHECK-NEXT:    store <3 x i32> [[TMP0]], ptr [[B_ADDR]], align 4
 // CHECK-NEXT:    [[TMP1:%.*]] = load <3 x i32>, ptr [[B_ADDR]], align 4
-// CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <3 x i32> [[TMP1]], i32 0
+// COL-CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <3 x i32> [[TMP1]], i32 0
+// ROW-CHECK-NEXT:    [[TMP1_COL:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> [[TMP1]], i32 3, i32 1)
+// ROW-CHECK-NEXT:    [[MATRIXEXT:%.*]] = extractelement <3 x i32> [[TMP1_COL]], i32 0
 // CHECK-NEXT:    [[VECINIT:%.*]] = insertelement <3 x i32> poison, i32 [[MATRIXEXT]], i32 0
 // CHECK-NEXT:    [[TMP2:%.*]] = load <3 x i32>, ptr [[B_ADDR]], align 4
-// CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <3 x i32> [[TMP2]], i32 1
+// COL-CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <3 x i32> [[TMP2]], i32 1
+// ROW-CHECK-NEXT:    [[TMP2_COL:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> [[TMP2]], i32 3, i32 1)
+// ROW-CHECK-NEXT:    [[MATRIXEXT1:%.*]] = extractelement <3 x i32> [[TMP2_COL]], i32 1
 // CHECK-NEXT:    [[VECINIT2:%.*]] = insertelement <3 x i32> [[VECINIT]], i32 [[MATRIXEXT1]], i32 1
 // CHECK-NEXT:    [[TMP3:%.*]] = load <3 x i32>, ptr [[B_ADDR]], align 4
-// CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <3 x i32> [[TMP3]], i32 2
+// COL-CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <3 x i32> [[TMP3]], i32 2
+// ROW-CHECK-NEXT:    [[TMP3_COL:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> [[TMP3]], i32 3, i32 1)
+// ROW-CHECK-NEXT:    [[MATRIXEXT3:%.*]] = extractelement <3 x i32> [[TMP3_COL]], i32 2
 // CHECK-NEXT:    [[VECINIT4:%.*]] = insertelement <3 x i32> [[VECINIT2]], i32 [[MATRIXEXT3]], i32 2
 // CHECK-NEXT:    ret <3 x i32> [[VECINIT4]]
 //
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/VectorElementwiseCast.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/VectorElementwiseCast.hlsl
index 89f5ff8f168259..5156dccb26eca2 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/VectorElementwiseCast.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/VectorElementwiseCast.hlsl
@@ -133,9 +133,14 @@ export void call6(Derived D) {
 // CHECK-NEXT:    [[V:%.*]] = alloca <4 x float>, align 4
 // CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca [2 x <2 x float>], align 4
 // CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <4 x float>, align 4
-// CHECK-NEXT:    store <4 x float> %M, ptr [[M_ADDR]], align 4
+// COL-CHECK-NEXT:    store <4 x float> %M, ptr [[M_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> %M, i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x float> [[M_ROW]], ptr [[M_ADDR]], align 4
 // CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[M_ADDR]], align 4
-// CHECK-NEXT:    store <4 x float> [[TMP0]], ptr [[HLSL_EWCAST_SRC]], align 4
+// COL-CHECK-NEXT:    store <4 x float> [[TMP0]], ptr [[HLSL_EWCAST_SRC]], align 4
+// ROW-CHECK-NEXT:    [[TMP0_COL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP0]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    [[TMP0_ROW:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[TMP0_COL]], i32 2, i32 2)
+// ROW-CHECK-NEXT:    store <4 x float> [[TMP0_ROW]], ptr [[HLSL_EWCAST_SRC]], align 4
 // CHECK-NEXT:    [[MATRIX_GEP:%.*]] = getelementptr inbounds <4 x float>, ptr [[HLSL_EWCAST_SRC]], i32 0
 // CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[FLATCAST_TMP]], align 4
 // CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[MATRIX_GEP]], align 4
@@ -166,9 +171,14 @@ export void call7(float2x2 M) {
 // COL-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca [1 x <3 x i32>], align 4
 // ROW-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca [3 x <1 x i32>], align 4
 // CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <3 x i32>, align 4
-// CHECK-NEXT:    store <3 x i32> %M, ptr [[M_ADDR]], align 4
+// COL-CHECK-NEXT:    store <3 x i32> %M, ptr [[M_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> %M, i32 3, i32 1)
+// ROW-CHECK-NEXT:    store <3 x i32> [[M_ROW]], ptr [[M_ADDR]], align 4
 // CHECK-NEXT:    [[TMP0:%.*]] = load <3 x i32>, ptr [[M_ADDR]], align 4
-// CHECK-NEXT:    store <3 x i32> [[TMP0]], ptr [[HLSL_EWCAST_SRC]], align 4
+// COL-CHECK-NEXT:    store <3 x i32> [[TMP0]], ptr [[HLSL_EWCAST_SRC]], align 4
+// ROW-CHECK-NEXT:    [[TMP0_COL:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> [[TMP0]], i32 1, i32 3)
+// ROW-CHECK-NEXT:    [[TMP0_ROW:%.*]] = call <3 x i32> @llvm.matrix.transpose.v3i32(<3 x i32> [[TMP0_COL]], i32 3, i32 1)
+// ROW-CHECK-NEXT:    store <3 x i32> [[TMP0_ROW]], ptr [[HLSL_EWCAST_SRC]], align 4
 // CHECK-NEXT:    [[MATRIX_GEP:%.*]] = getelementptr inbounds <3 x i32>, ptr [[HLSL_EWCAST_SRC]], i32 0
 // CHECK-NEXT:    [[TMP1:%.*]] = load <3 x i32>, ptr [[FLATCAST_TMP]], align 4
 // CHECK-NEXT:    [[TMP2:%.*]] = load <3 x i32>, ptr [[MATRIX_GEP]], align 4
@@ -194,10 +204,15 @@ export void call8(int3x1 M) {
 // COL-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca [2 x <1 x i32>], align 4
 // ROW-CHECK-NEXT:    [[HLSL_EWCAST_SRC:%.*]] = alloca [1 x <2 x i32>], align 4
 // CHECK-NEXT:    [[FLATCAST_TMP:%.*]] = alloca <2 x i1>, align 4
-// CHECK-NEXT:    [[TMP0:%.*]] = zext <2 x i1> %M to <2 x i32>
+// COL-CHECK-NEXT:    [[TMP0:%.*]] = zext <2 x i1> %M to <2 x i32>
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call <2 x i1> @llvm.matrix.transpose.v2i1(<2 x i1> %M, i32 1, i32 2)
+// ROW-CHECK-NEXT:    [[TMP0:%.*]] = zext <2 x i1> [[M_ROW]] to <2 x i32>
 // CHECK-NEXT:    store <2 x i32> [[TMP0]], ptr [[M_ADDR]], align 4
 // CHECK-NEXT:    [[TMP1:%.*]] = load <2 x i32>, ptr [[M_ADDR]], align 4
-// CHECK-NEXT:    store <2 x i32> [[TMP1]], ptr [[HLSL_EWCAST_SRC]], align 4
+// COL-CHECK-NEXT:    store <2 x i32> [[TMP1]], ptr [[HLSL_EWCAST_SRC]], align 4
+// ROW-CHECK-NEXT:    [[TMP1_COL:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP1]], i32 2, i32 1)
+// ROW-CHECK-NEXT:    [[TMP1_ROW:%.*]] = call <2 x i32> @llvm.matrix.transpose.v2i32(<2 x i32> [[TMP1_COL]], i32 1, i32 2)
+// ROW-CHECK-NEXT:    store <2 x i32> [[TMP1_ROW]], ptr [[HLSL_EWCAST_SRC]], align 4
 // CHECK-NEXT:    [[MATRIX_GEP:%.*]] = getelementptr inbounds <2 x i32>, ptr [[HLSL_EWCAST_SRC]], i32 0
 // CHECK-NEXT:    [[TMP2:%.*]] = load <2 x i1>, ptr [[FLATCAST_TMP]], align 4
 // CHECK-NEXT:    [[TMP3:%.*]] = load <2 x i32>, ptr [[MATRIX_GEP]], align 4
diff --git a/clang/test/CodeGenHLSL/BasicFeatures/matrix-type-indexing.hlsl b/clang/test/CodeGenHLSL/BasicFeatures/matrix-type-indexing.hlsl
index 3fff4976a93877..85c65d14616a5c 100644
--- a/clang/test/CodeGenHLSL/BasicFeatures/matrix-type-indexing.hlsl
+++ b/clang/test/CodeGenHLSL/BasicFeatures/matrix-type-indexing.hlsl
@@ -12,12 +12,11 @@ void binaryOpMatrixSubscriptExpr(int index, half2x3 M) {
     // CHECK: %col = alloca i32, align 4
     // CHECK: [[row_load:%.*]] = load i32, ptr %row, align 4
     // CHECK-NEXT: [[col_load:%.*]] = load i32, ptr %col, align 4
-    // ROW-CHECK-NEXT: [[row_offset:%.*]] = mul i32 [[row_load]], 3
-    // ROW-CHECK-NEXT: [[row_major_index:%.*]] = add i32 [[row_offset]], [[col_load]]
-    // COL-CHECK-NEXT: [[col_offset:%.*]] = mul i32 [[col_load]], 2
-    // COL-CHECK-NEXT: [[col_major_index:%.*]] = add i32 [[col_offset]], [[row_load]]
+    // CHECK-NEXT: [[col_offset:%.*]] = mul i32 [[col_load]], 2
+    // CHECK-NEXT: [[col_major_index:%.*]] = add i32 [[col_offset]], [[row_load]]
     // CHECK-NEXT: [[matrix_as_vec:%.*]] = load <6 x half>, ptr %M.addr, align 2
-    // ROW-CHECK-NEXT: %matrixext = extractelement <6 x half> [[matrix_as_vec]], i32 [[row_major_index]]
+    // ROW-CHECK-NEXT: [[normalized:%.*]] = call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> [[matrix_as_vec]], i32 3, i32 2)
+    // ROW-CHECK-NEXT: %matrixext = extractelement <6 x half> [[normalized]], i32 [[col_major_index]]
     // COL-CHECK-NEXT: %matrixext = extractelement <6 x half> [[matrix_as_vec]], i32 [[col_major_index]]
     const uint COLS = 3;
     uint row = index / COLS;
@@ -27,12 +26,11 @@ void binaryOpMatrixSubscriptExpr(int index, half2x3 M) {
 
 half returnMatrixSubscriptExpr(int row, int col, half2x3 M) {
     // CHECK-LABEL: returnMatrixSubscriptExpr
-    // ROW-CHECK: [[row_offset:%.*]] = mul i32 [[row_load:%.*]], 3
-    // ROW-CHECK-NEXT: [[row_major_index:%.*]] = add i32 [[row_offset]], [[col_load:%.*]]
-    // COL-CHECK: [[col_offset:%.*]] = mul i32 [[col_load:%.*]], 2
-    // COL-CHECK-NEXT: [[col_major_index:%.*]] = add i32 [[col_offset]], [[row_load:%.*]]
+    // CHECK: [[col_offset:%.*]] = mul i32 [[col_load:%.*]], 2
+    // CHECK-NEXT: [[col_major_index:%.*]] = add i32 [[col_offset]], [[row_load:%.*]]
     // CHECK-NEXT: [[matrix_as_vec:%.*]] = load <6 x half>, ptr %M.addr, align 2
-    // ROW-CHECK-NEXT: %matrixext = extractelement <6 x half> [[matrix_as_vec]], i32 [[row_major_index]]
+    // ROW-CHECK-NEXT: [[normalized:%.*]] = call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> [[matrix_as_vec]], i32 3, i32 2)
+    // ROW-CHECK-NEXT: %matrixext = extractelement <6 x half> [[normalized]], i32 [[col_major_index]]
     // COL-CHECK-NEXT: %matrixext = extractelement <6 x half> [[matrix_as_vec]], i32 [[col_major_index]]
     return M[row][col];
 }
diff --git a/clang/test/CodeGenHLSL/builtins/mul.hlsl b/clang/test/CodeGenHLSL/builtins/mul.hlsl
index 5e7468763654b9..3af97f7ca31fc6 100644
--- a/clang/test/CodeGenHLSL/builtins/mul.hlsl
+++ b/clang/test/CodeGenHLSL/builtins/mul.hlsl
@@ -29,7 +29,7 @@ export float3 test_scalar_vec_mul(float a, float3 b) { return mul(a, b); }
 // CHECK-LABEL: test_scalar_mat_mul
 // CHECK: %scalar.splat.splatinsert.i = insertelement <6 x float> poison, float %a, i64 0
 // CHECK: %scalar.splat.splat.i = shufflevector <6 x float> %scalar.splat.splatinsert.i, <6 x float> poison, <6 x i32> zeroinitializer
-// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x float> %scalar.splat.splat.i, %b
+// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x float> {{.*}}%scalar.splat.splat.i{{.*}}
 // CHECK: ret <6 x float> [[MUL]]
 export float2x3 test_scalar_mat_mul(float a, float2x3 b) { return mul(a, b); }
 
@@ -76,7 +76,8 @@ export double test_vec_vec_muld(double3 a, double3 b) { return mul(a, b); }
 // -- Case 6: vector * matrix -> vector --
 
 // CHECK-LABEL: test_vec_mat_mul
-// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %m, i32 3, i32 2)
+// ROWMAJOR: %[[STORED:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %m, i32 2, i32 3)
+// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[STORED]], i32 3, i32 2)
 // ROWMAJOR: %hlsl.mul = {{.*}} call {{.*}} <3 x float> @llvm.matrix.multiply.v3f32.v2f32.v6f32(<2 x float> %v, <6 x float> %[[TRANSPOSE]], i32 1, i32 2, i32 3)
 // COLMAJOR: %hlsl.mul = {{.*}} call {{.*}} <3 x float> @llvm.matrix.multiply.v3f32.v2f32.v6f32(<2 x float> %v, <6 x float> %m, i32 1, i32 2, i32 3)
 // CHECK: ret <3 x float> %hlsl.mul
@@ -87,14 +88,15 @@ export float3 test_vec_mat_mul(float2 v, float2x3 m) { return mul(v, m); }
 // CHECK-LABEL: test_mat_scalar_mul
 // CHECK: %scalar.splat.splatinsert.i = insertelement <6 x float> poison, float %b, i64 0
 // CHECK: %scalar.splat.splat.i = shufflevector <6 x float> %scalar.splat.splatinsert.i, <6 x float> poison, <6 x i32> zeroinitializer
-// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x float> %scalar.splat.splat.i, %a
+// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x float> {{.*}}%scalar.splat.splat.i{{.*}}
 // CHECK: ret <6 x float> [[MUL]]
 export float2x3 test_mat_scalar_mul(float2x3 a, float b) { return mul(a, b); }
 
 // -- Case 8: matrix * vector -> vector --
 
 // CHECK-LABEL: test_mat_vec_mul
-// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %m, i32 3, i32 2)
+// ROWMAJOR: %[[STORED:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %m, i32 2, i32 3)
+// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[STORED]], i32 3, i32 2)
 // ROWMAJOR: %hlsl.mul = {{.*}} call {{.*}} <2 x float> @llvm.matrix.multiply.v2f32.v6f32.v3f32(<6 x float> %[[TRANSPOSE]], <3 x float> %v, i32 2, i32 3, i32 1)
 // COLMAJOR: %hlsl.mul = {{.*}} call {{.*}} <2 x float> @llvm.matrix.multiply.v2f32.v6f32.v3f32(<6 x float> %m, <3 x float> %v, i32 2, i32 3, i32 1)
 // CHECK: ret <2 x float> %hlsl.mul
@@ -103,25 +105,25 @@ export float2 test_mat_vec_mul(float2x3 m, float3 v) { return mul(m, v); }
 // -- Case 9: matrix * matrix -> matrix --
 
 // CHECK-LABEL: test_mat_mat_mul
-// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %a, i32 3, i32 2)
-// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call {{.*}} <12 x float> @llvm.matrix.transpose.v12f32(<12 x float> %b, i32 4, i32 3)
+// ROWMAJOR: %[[STORED_A:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %a, i32 2, i32 3)
+// ROWMAJOR: %[[STORED_B:.*]] = {{.*}} call {{.*}} <12 x float> @llvm.matrix.transpose.v12f32(<12 x float> %b, i32 3, i32 4)
+// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %[[STORED_A]], i32 3, i32 2)
+// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call {{.*}} <12 x float> @llvm.matrix.transpose.v12f32(<12 x float> %[[STORED_B]], i32 4, i32 3)
 // ROWMAJOR: %hlsl.mul = {{.*}} call {{.*}} <8 x float> @llvm.matrix.multiply.v8f32.v6f32.v12f32(<6 x float> %[[TRANSPOSE_A]], <12 x float> %[[TRANSPOSE_B]], i32 2, i32 3, i32 4)
 // COLMAJOR: %hlsl.mul = {{.*}} call {{.*}} <8 x float> @llvm.matrix.multiply.v8f32.v6f32.v12f32(<6 x float> %a, <12 x float> %b, i32 2, i32 3, i32 4)
-// COLMAJOR: ret <8 x float> %hlsl.mul
-// ROWMAJOR: %[[TRANSPOSE_RES:.*]] = {{.*}} call {{.*}} <8 x float> @llvm.matrix.transpose.v8f32(<8 x float> %hlsl.mul, i32 2, i32 4)
-// ROWMAJOR: ret <8 x float> %[[TRANSPOSE_RES]]
+// CHECK: ret <8 x float> %hlsl.mul
 export float2x4 test_mat_mat_mul(float2x3 a, float3x4 b) { return mul(a, b); }
 
 // -- Integer matrix multiply --
 
 // CHECK-LABEL: test_mat_mat_muli
-// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> %a, i32 3, i32 2)
-// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> %b, i32 4, i32 3)
+// ROWMAJOR: %[[STORED_A:.*]] = {{.*}} call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> %a, i32 2, i32 3)
+// ROWMAJOR: %[[STORED_B:.*]] = {{.*}} call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> %b, i32 3, i32 4)
+// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> %[[STORED_A]], i32 3, i32 2)
+// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> %[[STORED_B]], i32 4, i32 3)
 // ROWMAJOR: %hlsl.mul = {{.*}} call <8 x i32> @llvm.matrix.multiply.v8i32.v6i32.v12i32(<6 x i32> %[[TRANSPOSE_A]], <12 x i32> %[[TRANSPOSE_B]], i32 2, i32 3, i32 4)
 // COLMAJOR: %hlsl.mul = {{.*}} call <8 x i32> @llvm.matrix.multiply.v8i32.v6i32.v12i32(<6 x i32> %a, <12 x i32> %b, i32 2, i32 3, i32 4)
-// COLMAJOR: ret <8 x i32> %hlsl.mul
-// ROWMAJOR: %[[TRANSPOSE_RES:.*]] = {{.*}} call <8 x i32> @llvm.matrix.transpose.v8i32(<8 x i32> %hlsl.mul, i32 2, i32 4)
-// ROWMAJOR: ret <8 x i32> %[[TRANSPOSE_RES]]
+// CHECK: ret <8 x i32> %hlsl.mul
 export int2x4 test_mat_mat_muli(int2x3 a, int3x4 b) { return mul(a, b); }
 
 // -- Half-type overloads (native half) --
@@ -141,7 +143,7 @@ export half3 test_scalar_vec_mulh(half a, half3 b) { return mul(a, b); }
 // CHECK-LABEL: test_scalar_mat_mulh
 // CHECK: %scalar.splat.splatinsert.i = insertelement <6 x half> poison, half %a, i64 0
 // CHECK: %scalar.splat.splat.i = shufflevector <6 x half> %scalar.splat.splatinsert.i, <6 x half> poison, <6 x i32> zeroinitializer
-// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x half> %scalar.splat.splat.i, %b
+// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x half> {{.*}}%scalar.splat.splat.i{{.*}}
 // CHECK: ret <6 x half> [[MUL]]
 export half2x3 test_scalar_mat_mulh(half a, half2x3 b) { return mul(a, b); }
 
@@ -161,30 +163,32 @@ export half test_vec_vec_mulh(half3 a, half3 b) { return mul(a, b); }
 // CHECK-LABEL: test_mat_scalar_mulh
 // CHECK: %scalar.splat.splatinsert.i = insertelement <6 x half> poison, half %b, i64 0
 // CHECK: %scalar.splat.splat.i = shufflevector <6 x half> %scalar.splat.splatinsert.i, <6 x half> poison, <6 x i32> zeroinitializer
-// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x half> %scalar.splat.splat.i, %a
+// CHECK: [[MUL:%.*]] = fmul {{.*}} <6 x half> {{.*}}%scalar.splat.splat.i{{.*}}
 // CHECK: ret <6 x half> [[MUL]]
 export half2x3 test_mat_scalar_mulh(half2x3 a, half b) { return mul(a, b); }
 
 // CHECK-LABEL: test_vec_mat_mulh
-// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %m, i32 3, i32 2)
+// ROWMAJOR: %[[STORED:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %m, i32 2, i32 3)
+// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %[[STORED]], i32 3, i32 2)
 // ROWMAJOR: %hlsl.mul = {{.*}}call {{.*}} <3 x half> @llvm.matrix.multiply.v3f16.v2f16.v6f16(<2 x half> %v, <6 x half> %[[TRANSPOSE]], i32 1, i32 2, i32 3)
 // COLMAJOR: %hlsl.mul = {{.*}}call {{.*}} <3 x half> @llvm.matrix.multiply.v3f16.v2f16.v6f16(<2 x half> %v, <6 x half> %m, i32 1, i32 2, i32 3)
 // CHECK: ret <3 x half> %hlsl.mul
 export half3 test_vec_mat_mulh(half2 v, half2x3 m) { return mul(v, m); }
 
 // CHECK-LABEL: test_mat_vec_mulh
-// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %m, i32 3, i32 2)
+// ROWMAJOR: %[[STORED:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %m, i32 2, i32 3)
+// ROWMAJOR: %[[TRANSPOSE:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %[[STORED]], i32 3, i32 2)
 // ROWMAJOR: %hlsl.mul = {{.*}}call {{.*}} <2 x half> @llvm.matrix.multiply.v2f16.v6f16.v3f16(<6 x half> %[[TRANSPOSE]], <3 x half> %v, i32 2, i32 3, i32 1)
 // COLMAJOR: %hlsl.mul = {{.*}}call {{.*}} <2 x half> @llvm.matrix.multiply.v2f16.v6f16.v3f16(<6 x half> %m, <3 x half> %v, i32 2, i32 3, i32 1)
 // CHECK: ret <2 x half> %hlsl.mul
 export half2 test_mat_vec_mulh(half2x3 m, half3 v) { return mul(m, v); }
 
 // CHECK-LABEL: test_mat_mat_mulh
-// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %a, i32 3, i32 2)
-// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call {{.*}} <12 x half> @llvm.matrix.transpose.v12f16(<12 x half> %b, i32 4, i32 3)
+// ROWMAJOR: %[[STORED_A:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %a, i32 2, i32 3)
+// ROWMAJOR: %[[STORED_B:.*]] = {{.*}} call {{.*}} <12 x half> @llvm.matrix.transpose.v12f16(<12 x half> %b, i32 3, i32 4)
+// ROWMAJOR: %[[TRANSPOSE_A:.*]] = {{.*}} call {{.*}} <6 x half> @llvm.matrix.transpose.v6f16(<6 x half> %[[STORED_A]], i32 3, i32 2)
+// ROWMAJOR: %[[TRANSPOSE_B:.*]] = {{.*}} call {{.*}} <12 x half> @llvm.matrix.transpose.v12f16(<12 x half> %[[STORED_B]], i32 4, i32 3)
 // ROWMAJOR: %hlsl.mul = {{.*}}call {{.*}} <8 x half> @llvm.matrix.multiply.v8f16.v6f16.v12f16(<6 x half> %[[TRANSPOSE_A]], <12 x half> %[[TRANSPOSE_B]], i32 2, i32 3, i32 4)
 // COLMAJOR: %hlsl.mul = {{.*}}call {{.*}} <8 x half> @llvm.matrix.multiply.v8f16.v6f16.v12f16(<6 x half> %a, <12 x half> %b, i32 2, i32 3, i32 4)
-// COLMAJOR: ret <8 x half> %hlsl.mul
-// ROWMAJOR: %[[TRANSPOSE_RES:.*]] = {{.*}} call {{.*}} <8 x half> @llvm.matrix.transpose.v8f16(<8 x half> %hlsl.mul, i32 2, i32 4)
-// ROWMAJOR: ret <8 x half> %[[TRANSPOSE_RES]]
+// CHECK: ret <8 x half> %hlsl.mul
 export half2x4 test_mat_mat_mulh(half2x3 a, half3x4 b) { return mul(a, b); }
diff --git a/clang/test/CodeGenHLSL/builtins/transpose.hlsl b/clang/test/CodeGenHLSL/builtins/transpose.hlsl
index d8430fcf5bf9d8..96198d8b08e9db 100644
--- a/clang/test/CodeGenHLSL/builtins/transpose.hlsl
+++ b/clang/test/CodeGenHLSL/builtins/transpose.hlsl
@@ -10,7 +10,8 @@
 // CHECK:       store <6 x i32> [[A_EXT]], ptr [[A_ADDR]], align 4
 // CHECK:       [[A:%.*]] = load <6 x i32>, ptr [[A_ADDR]], align 4
 // COLMAJOR:    [[TRANS:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[A]], i32 2, i32 3)
-// ROWMAJOR:    [[TRANS:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[A]], i32 3, i32 2)
+// ROWMAJOR:    [[NORMALIZED:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[A]], i32 3, i32 2)
+// ROWMAJOR:    [[TRANS:%.*]] = call <6 x i32> @llvm.matrix.transpose.v6i32(<6 x i32> [[NORMALIZED]], i32 2, i32 3)
 bool3x2 test_transpose_bool2x3(bool2x3 a) {
   return transpose(a);
 }
@@ -21,7 +22,8 @@ bool3x2 test_transpose_bool2x3(bool2x3 a) {
 // CHECK:       store <12 x i32> %{{.*}}, ptr [[A_ADDR]], align 4
 // CHECK:       [[A:%.*]] = load <12 x i32>, ptr [[A_ADDR]], align 4
 // COLMAJOR:    [[TRANS:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[A]], i32 4, i32 3)
-// ROWMAJOR:    [[TRANS:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[A]], i32 3, i32 4)
+// ROWMAJOR:    [[NORMALIZED:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[A]], i32 3, i32 4)
+// ROWMAJOR:    [[TRANS:%.*]] = call <12 x i32> @llvm.matrix.transpose.v12i32(<12 x i32> [[NORMALIZED]], i32 4, i32 3)
 // CHECK:       ret <12 x i32> [[TRANS]]
 int3x4 test_transpose_int4x3(int4x3 a) {
   return transpose(a);
@@ -31,7 +33,9 @@ int3x4 test_transpose_int4x3(int4x3 a) {
 // CHECK:       [[A_ADDR:%.*]] = alloca [4 x <4 x float>], align 4
 // CHECK:       store <16 x float> %{{.*}}, ptr [[A_ADDR]], align 4
 // CHECK:       [[A:%.*]] = load <16 x float>, ptr [[A_ADDR]], align 4
-// CHECK:       [[TRANS:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[A]], i32 4, i32 4)
+// COLMAJOR:    [[TRANS:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[A]], i32 4, i32 4)
+// ROWMAJOR:    [[NORMALIZED:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[A]], i32 4, i32 4)
+// ROWMAJOR:    [[TRANS:%.*]] = call {{.*}}<16 x float> @llvm.matrix.transpose.v16f32(<16 x float> [[NORMALIZED]], i32 4, i32 4)
 // CHECK:       ret <16 x float> [[TRANS]]
 float4x4 test_transpose_float4x4(float4x4 a) {
   return transpose(a);
@@ -43,7 +47,8 @@ float4x4 test_transpose_float4x4(float4x4 a) {
 // CHECK:       store <4 x double> %{{.*}}, ptr [[A_ADDR]], align 8
 // CHECK:       [[A:%.*]] = load <4 x double>, ptr [[A_ADDR]], align 8
 // COLMAJOR:    [[TRANS:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[A]], i32 1, i32 4)
-// ROWMAJOR:    [[TRANS:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[A]], i32 4, i32 1)
+// ROWMAJOR:    [[NORMALIZED:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[A]], i32 4, i32 1)
+// ROWMAJOR:    [[TRANS:%.*]] = call {{.*}}<4 x double> @llvm.matrix.transpose.v4f64(<4 x double> [[NORMALIZED]], i32 1, i32 4)
 // CHECK:       ret <4 x double> [[TRANS]]
 double4x1 test_transpose_double1x4(double1x4 a) {
   return transpose(a);
diff --git a/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl b/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl
index dfafa2b0b7e613..a04cf58b45b7c2 100644
--- a/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl
+++ b/clang/test/CodeGenHLSL/matrix-layout-attr-overrides-default.hlsl
@@ -1,5 +1,5 @@
-// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -x hlsl -triple dxil-pc-shadermodel6.3-library %s -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=column-major -o - | FileCheck %s --check-prefixes=CHECK,COLMAJOR
-// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -x hlsl -triple dxil-pc-shadermodel6.3-library %s -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=row-major -o - | FileCheck %s --check-prefixes=CHECK,ROWMAJOR
+// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -x hlsl -triple dxil-pc-shadermodel6.3-library %s -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=column-major -o - | FileCheck %s
+// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -x hlsl -triple dxil-pc-shadermodel6.3-library %s -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=row-major -o - | FileCheck %s
 
 // Verifies that a per-decl `[[hlsl::row_major]]` / `[[hlsl::column_major]]`
 // (spelled `row_major` / `column_major` in HLSL) overrides the
@@ -14,7 +14,7 @@
 // The decl-level attribute should win regardless of the TU default.
 
 // -----------------------------------------------------------------------------
-// MatrixSubscriptExpr indexing: row-major attr -> Row*NumCols + Col
+// MatrixSubscriptExpr indexing uses canonical column-major prvalues.
 // -----------------------------------------------------------------------------
 export float subscript_rm(int row, int col, row_major float2x3 m) {
   return m[row][col];
@@ -22,8 +22,8 @@ export float subscript_rm(int row, int col, row_major float2x3 m) {
 // CHECK-LABEL: define {{.*}} float @_Z12subscript_rmiiu11matrix_typeILm2ELm3EfE
 // CHECK: [[ROW:%.*]] = load i32, ptr %row.addr
 // CHECK: [[COL:%.*]] = load i32, ptr %col.addr
-// CHECK: [[OFFSET:%.*]] = mul i32 [[ROW]], 3
-// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[COL]]
+// CHECK: [[OFFSET:%.*]] = mul i32 [[COL]], 2
+// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[ROW]]
 // CHECK: extractelement <6 x float> %{{.*}}, i32 [[IDX]]
 
 // -----------------------------------------------------------------------------
@@ -44,19 +44,15 @@ export float subscript_cm(int row, int col, column_major float2x3 m) {
 // index formula even when the TU default disagrees.
 // -----------------------------------------------------------------------------
 
-// Row-major: per-column element index is Row*NumCols + Col, materialized as a
-// constant-zero / constant-one / constant-two add to (Row*3).
+// Row extraction also indexes the canonical column-major prvalue.
 export float3 row_extract_rm(int row, row_major float2x3 m) {
   return m[row];
 }
 // CHECK-LABEL: define {{.*}} <3 x float> @_Z14row_extract_rmiu11matrix_typeILm2ELm3EfE
 // CHECK: [[ROW:%.*]] = load i32, ptr %row.addr
-// CHECK: [[ROW_OFFSET0:%.*]] = mul i32 [[ROW]], 3
-// CHECK: add i32 [[ROW_OFFSET0]], 0
-// CHECK: [[ROW_OFFSET1:%.*]] = mul i32 [[ROW]], 3
-// CHECK: add i32 [[ROW_OFFSET1]], 1
-// CHECK: [[ROW_OFFSET2:%.*]] = mul i32 [[ROW]], 3
-// CHECK: add i32 [[ROW_OFFSET2]], 2
+// CHECK: add i32 0, [[ROW]]
+// CHECK: add i32 2, [[ROW]]
+// CHECK: add i32 4, [[ROW]]
 
 // Column-major: per-column element index is Col*NumRows + Row, so we *don't*
 // see the Row*NumCols multiply; instead each column folds the constant
@@ -94,9 +90,9 @@ export float3 vec_mat_cm(float2 v, column_major float2x3 m) { return mul(v, m);
 export float2x2 mat_mat_rm_cm(row_major float2x3 a, column_major float3x2 b) { return mul(a, b); }
 // CHECK-LABEL: define {{.*}} <4 x float> @_Z13mat_mat_rm_cm
 // CHECK: [[AMat:%.*]] = load <6 x float>, ptr %a.addr, align 4
+// CHECK: [[A:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[AMat]], i32 3, i32 2)
 // CHECK: [[BMat:%.*]] = load <6 x float>, ptr %b.addr, align 4
-// CHECK: [[T:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[AMat]], i32 3, i32 2)
-// CHECK: call {{.*}} <4 x float> @llvm.matrix.multiply.v4f32.v6f32.v6f32(<6 x float> [[T]], <6 x float> [[BMat]], i32 2, i32 3, i32 2)
+// CHECK: call {{.*}} <4 x float> @llvm.matrix.multiply.v4f32.v6f32.v6f32(<6 x float> [[A]], <6 x float> [[BMat]], i32 2, i32 3, i32 2)
 
 // LHS column-major, RHS row-major: only RHS is transposed.
 export float2x2 mat_mat_cm_rm(column_major float2x3 a, row_major float3x2 b) { return mul(a, b); }
@@ -116,47 +112,42 @@ export column_major float2x2 mat_mat_dst_cm(column_major float2x3 a, column_majo
 export row_major float2x2 mat_mat_dst_rm(column_major float2x3 a, column_major float3x2 b) { return mul(a, b); }
 // CHECK-LABEL: define {{.*}} <4 x float> @_Z14mat_mat_dst_rm
 // CHECK: [[MUL:%.*]] = call {{.*}} <4 x float> @llvm.matrix.multiply.v4f32.v6f32.v6f32(<6 x float> %{{.*}}, <6 x float> %{{.*}}, i32 2, i32 3, i32 2)
-// CHECK: call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[MUL]], i32 2, i32 2)
+// CHECK-NOT: @llvm.matrix.transpose
+// CHECK: ret <4 x float> [[MUL]]
 
 
-// Row-major source -> column-major destination: bits already transposed, no-op.
+// Transpose operates on the canonical column-major value after the load.
 export column_major float3x2 transpose_rm_to_cm(row_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_rm_to_cmu11matrix_typeILm2ELm3EfE
-// CHECK-NOT: @llvm.matrix.transpose
-// CHECK: ret <6 x float>
+// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2)
+// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3)
 
-// Column-major source -> row-major destination: bits already transposed, no-op.
+// Return layout metadata does not change the canonical value representation.
 export row_major float3x2 transpose_cm_to_rm(column_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_cm_to_rmu11matrix_typeILm2ELm3EfE
-// CHECK-NOT: @llvm.matrix.transpose
-// CHECK: ret <6 x float>
+// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3)
 
 // Row-major source -> row-major destination: real transpose, dims swapped.
 export row_major float3x2 transpose_rm_to_rm(row_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_rm_to_rmu11matrix_typeILm2ELm3EfE
-// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2)
+// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2)
+// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3)
 
 // Column-major source -> column-major destination: real transpose, natural dims.
 export column_major float3x2 transpose_cm_to_cm(column_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z18transpose_cm_to_cmu11matrix_typeILm2ELm3EfE
 // CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3)
 
-// Default-layout return type: the TU `-fmatrix-memory-layout=` default 
-// flips between a real transpose and a no-op depending on the default.
+// The TU memory-layout default does not affect matrix prvalues.
 export float3x2 transpose_rm(row_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z12transpose_rmu11matrix_typeILm2ELm3EfE
-// COLMAJOR-NOT: @llvm.matrix.transpose
-// COLMAJOR: ret <6 x float>
-// ROWMAJOR: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2)
+// CHECK: [[RM:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 3, i32 2)
+// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[RM]], i32 2, i32 3)
 
 
-// column-major default: src/dst match -> real transpose, natural dims.
-// row-major default: src/dst differ -> bits already transposed, no-op.
 export float3x2 transpose_cm(column_major float2x3 m) { return transpose(m); }
 // CHECK-LABEL: define {{.*}} <6 x float> @_Z12transpose_cmu11matrix_typeILm2ELm3EfE
-// COLMAJOR: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3)
-// ROWMAJOR-NOT: @llvm.matrix.transpose
-// ROWMAJOR: ret <6 x float>
+// CHECK: call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> %{{.*}}, i32 2, i32 3)
 
 // -----------------------------------------------------------------------------
 // CK_HLSLMatrixTruncation: the shuffle mask that picks elements from the
@@ -168,10 +159,10 @@ typedef column_major float2x2 CM22;
 typedef row_major    float3x3 RM33;
 typedef column_major float3x3 CM33;
 
-// Row-major source 3x2 -> row-major dest 2x2: flat row-major mask is {0,1,2,3}.
+// Matrix truncation uses canonical column-major indices.
 export row_major float2x2 truncate_rm(row_major float3x2 m) { return (RM22)m; }
 // CHECK-LABEL: define {{.*}} <4 x float> @_Z11truncate_rmu11matrix_typeILm3ELm2EfE
-// CHECK: shufflevector <6 x float> %{{.*}}, <6 x float> poison, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+// CHECK: shufflevector <6 x float> %{{.*}}, <6 x float> poison, <4 x i32> <i32 0, i32 1, i32 3, i32 4>
 
 // Column-major source 3x2 -> column-major dest 2x2: flat column-major mask is {0,1,3,4}.
 export column_major float2x2 truncate_cm(column_major float3x2 m) { return (CM22)m; }
@@ -186,14 +177,9 @@ export column_major float2x2 truncate_cm(column_major float3x2 m) { return (CM22
 // `-fmatrix-memory-layout=` default.
 // -----------------------------------------------------------------------------
 
-// Row-major src 3x4 -> column-major dst 3x3.
-// src idx (R,C) = R*4+C; dst slot (R,C) = C*3+R.
-//   (0,0)->mask[0]=0  (0,1)->mask[3]=1  (0,2)->mask[6]=2
-//   (1,0)->mask[1]=4  (1,1)->mask[4]=5  (1,2)->mask[7]=6
-//   (2,0)->mask[2]=8  (2,1)->mask[5]=9  (2,2)->mask[8]=10
 export column_major float3x3 truncate_rm_to_cm(row_major float3x4 m) { return (CM33)m; }
 // CHECK-LABEL: define {{.*}} <9 x float> @_Z17truncate_rm_to_cmu11matrix_typeILm3ELm4EfE
-// CHECK: shufflevector <12 x float> %{{.*}}, <12 x float> poison, <9 x i32> <i32 0, i32 4, i32 8, i32 1, i32 5, i32 9, i32 2, i32 6, i32 10>
+// CHECK: shufflevector <12 x float> %{{.*}}, <12 x float> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
 
 // Column-major src 3x4 -> row-major dst 3x3.
 // src idx (R,C) = C*3+R; dst slot (R,C) = R*3+C.
@@ -202,7 +188,7 @@ export column_major float3x3 truncate_rm_to_cm(row_major float3x4 m) { return (C
 //   (2,0)->mask[6]=2  (2,1)->mask[7]=5  (2,2)->mask[8]=8
 export row_major float3x3 truncate_cm_to_rm(column_major float3x4 m) { return (RM33)m; }
 // CHECK-LABEL: define {{.*}} <9 x float> @_Z17truncate_cm_to_rmu11matrix_typeILm3ELm4EfE
-// CHECK: shufflevector <12 x float> %{{.*}}, <12 x float> poison, <9 x i32> <i32 0, i32 3, i32 6, i32 1, i32 4, i32 7, i32 2, i32 5, i32 8>
+// CHECK: shufflevector <12 x float> %{{.*}}, <12 x float> poison, <9 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8>
 
 // -----------------------------------------------------------------------------
 // Array of matrix: the per-decl layout attribute propagates through
@@ -210,15 +196,15 @@ export row_major float3x3 truncate_cm_to_rm(column_major float3x4 m) { return (R
 // array element still uses the correct layout.
 // -----------------------------------------------------------------------------
 
-// Row-major array element subscript: Row*NumCols + Col
+// Row-major array storage is normalized before column-major scalar extraction.
 export float arr_subscript_rm(int row, int col, row_major float2x3 arr[2]) {
   return arr[1][row][col];
 }
 // CHECK-LABEL: define {{.*}} float @_Z16arr_subscript_rm
 // CHECK: [[ROW:%.*]] = load i32, ptr %row.addr
 // CHECK: [[COL:%.*]] = load i32, ptr %col.addr
-// CHECK: [[OFFSET:%.*]] = mul i32 [[ROW]], 3
-// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[COL]]
+// CHECK: [[OFFSET:%.*]] = mul i32 [[COL]], 2
+// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[ROW]]
 // CHECK: extractelement <6 x float> %{{.*}}, i32 [[IDX]]
 
 // Column-major array element subscript: Col*NumRows + Row
@@ -237,15 +223,15 @@ export float arr_subscript_cm(int row, int col, column_major float2x3 arr[2]) {
 // through nested ConstantArrayType layers.
 // -----------------------------------------------------------------------------
 
-// Row-major 2D array element subscript: Row*NumCols + Col
+// Nested row-major array storage is normalized before scalar extraction.
 export float arr2d_subscript_rm(int row, int col, row_major float2x3 arr[2][3]) {
   return arr[0][1][row][col];
 }
 // CHECK-LABEL: define {{.*}} float @_Z18arr2d_subscript_rm
 // CHECK: [[ROW:%.*]] = load i32, ptr %row.addr
 // CHECK: [[COL:%.*]] = load i32, ptr %col.addr
-// CHECK: [[OFFSET:%.*]] = mul i32 [[ROW]], 3
-// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[COL]]
+// CHECK: [[OFFSET:%.*]] = mul i32 [[COL]], 2
+// CHECK: [[IDX:%.*]] = add i32 [[OFFSET]], [[ROW]]
 // CHECK: extractelement <6 x float> %{{.*}}, i32 [[IDX]]
 
 // Column-major 2D array element subscript: Col*NumRows + Row
diff --git a/clang/test/CodeGenHLSL/matrix-layout-register-representation.hlsl b/clang/test/CodeGenHLSL/matrix-layout-register-representation.hlsl
new file mode 100644
index 00000000000000..eff4dff366f526
--- /dev/null
+++ b/clang/test/CodeGenHLSL/matrix-layout-register-representation.hlsl
@@ -0,0 +1,37 @@
+// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -triple dxil-pc-shadermodel6.3-library \
+// RUN:   -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=column-major -o - %s | FileCheck %s
+// RUN: %clang_cc1 -std=hlsl202x -finclude-default-header -triple dxil-pc-shadermodel6.3-library \
+// RUN:   -emit-llvm -disable-llvm-passes -fmatrix-memory-layout=row-major -o - %s | FileCheck %s
+
+export float2x2 load_row_major(row_major float2x2 matrix) {
+  return matrix;
+}
+
+// CHECK-LABEL: define {{.*}} <4 x float> @_Z14load_row_major
+// CHECK: [[TO_MEMORY:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> %matrix, i32 2, i32 2)
+// CHECK: store <4 x float> [[TO_MEMORY]], ptr %matrix.addr
+// CHECK: [[FROM_MEMORY:%.*]] = load <4 x float>, ptr %matrix.addr
+// CHECK: [[TO_REGISTER:%.*]] = call {{.*}} <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> [[FROM_MEMORY]], i32 2, i32 2)
+// CHECK: ret <4 x float> [[TO_REGISTER]]
+
+export void store_row_major(out row_major float2x3 destination,
+                            column_major float2x3 source) {
+  destination = source;
+}
+
+// CHECK-LABEL: define {{.*}} @_Z15store_row_major
+// CHECK: [[SOURCE:%.*]] = load <6 x float>, ptr %source.addr
+// CHECK: [[TO_MEMORY:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[SOURCE]], i32 2, i32 3)
+// CHECK: store <6 x float> [[TO_MEMORY]], ptr %{{.*}}
+
+export float2x2 multiply(row_major float2x3 lhs,
+                         column_major float3x2 rhs) {
+  return mul(lhs, rhs);
+}
+
+// CHECK-LABEL: define {{.*}} <4 x float> @_Z8multiply
+// CHECK: [[LHS_MEMORY:%.*]] = load <6 x float>, ptr %lhs.addr
+// CHECK: [[LHS_REGISTER:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[LHS_MEMORY]], i32 3, i32 2)
+// CHECK: [[RHS_REGISTER:%.*]] = load <6 x float>, ptr %rhs.addr
+// CHECK-NOT: @llvm.matrix.transpose
+// CHECK: call {{.*}} <4 x float> @llvm.matrix.multiply.v4f32.v6f32.v6f32(<6 x float> [[LHS_REGISTER]], <6 x float> [[RHS_REGISTER]], i32 2, i32 3, i32 2)
diff --git a/clang/test/CodeGenHLSL/resources/MatrixElement_cbuffer.hlsl b/clang/test/CodeGenHLSL/resources/MatrixElement_cbuffer.hlsl
index bcb8b4bb9e82dd..8d2493bee2c0ba 100644
--- a/clang/test/CodeGenHLSL/resources/MatrixElement_cbuffer.hlsl
+++ b/clang/test/CodeGenHLSL/resources/MatrixElement_cbuffer.hlsl
@@ -89,7 +89,8 @@ float4 getCBufferSwizzleAccess() {
 // COL-CHECK-NEXT:    ret <2 x float> [[TMP1]]
 //
 // ROW-CHECK-NEXT:    [[M_ADDR:%.*]] = alloca [3 x <2 x float>], align 4
-// ROW-CHECK-NEXT:    store <6 x float> [[M]], ptr [[M_ADDR]], align 4
+// ROW-CHECK-NEXT:    [[M_ROW:%.*]] = call {{.*}} <6 x float> @llvm.matrix.transpose.v6f32(<6 x float> [[M]], i32 3, i32 2)
+// ROW-CHECK-NEXT:    store <6 x float> [[M_ROW]], ptr [[M_ADDR]], align 4
 // ROW-CHECK-NEXT:    [[TMP0:%.*]] = load <6 x float>, ptr [[M_ADDR]], align 4
 // ROW-CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <6 x float> [[TMP0]], <6 x float> poison, <2 x i32> <i32 2, i32 1>
 // ROW-CHECK-NEXT:    ret <2 x float> [[TMP1]]
diff --git a/clang/test/SemaHLSL/matrix_layout_attr.hlsl b/clang/test/SemaHLSL/matrix_layout_attr.hlsl
index 3e953f07557e6c..9c243ea8ce5342 100644
--- a/clang/test/SemaHLSL/matrix_layout_attr.hlsl
+++ b/clang/test/SemaHLSL/matrix_layout_attr.hlsl
@@ -44,6 +44,18 @@ column_major float4x4 Col2Row(row_major float4x4 M) {
 
 void bar(row_major float4x4 M, column_major float4x4 M2) {}
 
+// Layout metadata does not create distinct overloads.
+void same_overload(row_major float2x2 M);
+void same_overload(column_major float2x2 M);
+
+template <typename T>
+T preserve_layout(T M) {
+  return M;
+}
+
+row_major float2x3 substitution_source;
+row_major float2x3 substituted = preserve_layout(substitution_source);
+
 //Invalid: 
 // expected-error at +1 {{'row_major' attribute can only be applied to a matrix type}}
 void foo(column_major float4x4 mat, row_major int i) {}



More information about the cfe-commits mailing list