[llvm] [Matrix] De-duplicate reshaped matrixes used as incoming values for phi. (PR #211210)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 22 02:17:15 PDT 2026
https://github.com/fhahn created https://github.com/llvm/llvm-project/pull/211210
Phis can have multiple incoming entries for the same block. In that case, all incoming values for the block must be the same.
Update visitPHI to avoid expanding the incoming matrix multiple times for the some incoming block.
Fixes a verifier error for the newly added test case.
>From 6e5ac8fc0703735f716b296b627128a760d2af50 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 21 Jul 2026 20:41:50 +0100
Subject: [PATCH] [Matrix] De-duplicate reshaped matrixes used as incoming
values for phi.
Phis can have multiple incoming entries for the same block. In that
case, all incoming values for the block must be the same.
Update visitPHI to avoid expanding the incoming matrix multiple times
for the some incoming block.
Fixes a verifier error for the newly added test case.
---
.../Scalar/LowerMatrixIntrinsics.cpp | 9 +++--
.../Transforms/LowerMatrixIntrinsics/phi.ll | 33 +++++++++++++++++++
2 files changed, 40 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/LowerMatrixIntrinsics.cpp b/llvm/lib/Transforms/Scalar/LowerMatrixIntrinsics.cpp
index 2871431b12b31..059436ab82b5c 100644
--- a/llvm/lib/Transforms/Scalar/LowerMatrixIntrinsics.cpp
+++ b/llvm/lib/Transforms/Scalar/LowerMatrixIntrinsics.cpp
@@ -2397,9 +2397,11 @@ class LowerMatrixIntrinsics {
Builder.SetInsertPoint(BlockIP);
MatrixTy PhiM = getMatrix(Inst, SI, Builder);
+ // Cache the reshaped columns per incoming block, so that a block listed
+ // more than once contributes identical incoming values to the new PHIs.
+ SmallDenseMap<BasicBlock *, MatrixTy> ReshapedIncoming;
for (auto [IncomingV, IncomingB] :
llvm::zip_equal(Inst->incoming_values(), Inst->blocks())) {
- // getMatrix() may insert some instructions to help with reshaping. The
// safest place for those is at the top of the block after the rest of the
// PHI's. Even better, if we can put it in the incoming block.
Builder.SetInsertPoint(BlockIP);
@@ -2407,7 +2409,10 @@ class LowerMatrixIntrinsics {
if (auto MaybeIP = IncomingInst->getInsertionPointAfterDef())
Builder.SetInsertPoint(*MaybeIP);
- MatrixTy OpM = getMatrix(IncomingV, SI, Builder);
+ auto [It, Inserted] = ReshapedIncoming.try_emplace(IncomingB);
+ if (Inserted)
+ It->second = getMatrix(IncomingV, SI, Builder);
+ const MatrixTy &OpM = getMatrix(IncomingV, SI, Builder);
for (unsigned VI = 0, VE = PhiM.getNumVectors(); VI != VE; ++VI) {
PHINode *NewPHI = cast<PHINode>(PhiM.getVector(VI));
diff --git a/llvm/test/Transforms/LowerMatrixIntrinsics/phi.ll b/llvm/test/Transforms/LowerMatrixIntrinsics/phi.ll
index c35924d85aa29..c99c0cbf175b7 100644
--- a/llvm/test/Transforms/LowerMatrixIntrinsics/phi.ll
+++ b/llvm/test/Transforms/LowerMatrixIntrinsics/phi.ll
@@ -787,3 +787,36 @@ if.end: ; preds = %if.then, %if.else
%res = tail call <9 x double> @llvm.matrix.multiply.v9f64.v9f64.v9f64(<9 x double> %C, <9 x double> %merge, i32 3, i32 3, i32 3)
ret <9 x double> %res
}
+
+define <4 x float> @matrix_phi_duplicate_predecessor(ptr %arg, i32 %sw) {
+; CHECK-LABEL: @matrix_phi_duplicate_predecessor(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[SPLIT:%.*]] = shufflevector <4 x float> [[ARG:%.*]], <4 x float> poison, <2 x i32> <i32 0, i32 1>
+; CHECK-NEXT: [[SPLIT3:%.*]] = shufflevector <4 x float> [[ARG]], <4 x float> poison, <2 x i32> <i32 2, i32 3>
+; CHECK-NEXT: switch i32 [[SW:%.*]], label [[BB:%.*]] [
+; CHECK-NEXT: i32 0, label [[EXIT:%.*]]
+; CHECK-NEXT: i32 1, label [[EXIT]]
+; CHECK-NEXT: ]
+; CHECK: bb:
+; CHECK-NEXT: br label [[EXIT]]
+; CHECK: exit:
+; CHECK-NEXT: [[PHI1:%.*]] = phi <2 x float> [ [[SPLIT]], [[ENTRY:%.*]] ], [ [[SPLIT]], [[ENTRY]] ], [ zeroinitializer, [[BB]] ]
+; CHECK-NEXT: [[PHI2:%.*]] = phi <2 x float> [ [[SPLIT3]], [[ENTRY]] ], [ [[SPLIT3]], [[ENTRY]] ], [ zeroinitializer, [[BB]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = shufflevector <2 x float> [[PHI1]], <2 x float> [[PHI2]], <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+; CHECK-NEXT: ret <4 x float> [[TMP0]]
+;
+entry:
+ %m = load <4 x float>, ptr %arg
+ switch i32 %sw, label %bb [
+ i32 0, label %exit
+ i32 1, label %exit
+ ]
+
+bb:
+ %t = call <4 x float> @llvm.matrix.transpose.v4f32(<4 x float> zeroinitializer, i32 2, i32 2)
+ br label %exit
+
+exit:
+ %phi = phi <4 x float> [ %m, %entry ], [ %m, %entry ], [ %t, %bb ]
+ ret <4 x float> %phi
+}
More information about the llvm-commits
mailing list