[polly] [Polly] Allocate packed arrays of matrix multiplication on heap (PR #226163)
Timur Baidusenov via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 06:30:50 PDT 2026
https://github.com/bai-tim created https://github.com/llvm/llvm-project/pull/226163
The matrix multiplication optimization copies blocks of its operands into the arrays Packed_A and Packed_B. Their sizes are derived from the cache parameters rather than from the operands, so with the default parameters Packed_B alone takes 3 to 4 MB. They were allocated with alloca in the entry block of the function, and two optimized multiplications in one function exceeded the default 8 MB stack: two products of 100x100 int32 matrices declared as VLAs and inlined into main crash with a segmentation fault. Allocate them with malloc at the start of the SCoP and free them at its exit instead, as is already done for the arrays created by JSCoP import and by maximal static expansion.
Assisted-by: Claude (Anthropic)
>From d2bbc55d51a34001a7b33f2d3449207cc2f9764d Mon Sep 17 00:00:00 2001
From: Timur Baidusenov <timurbaidusenov at gmail.com>
Date: Thu, 24 Sep 2026 04:05:12 +0300
Subject: [PATCH] [Polly] Allocate the packed arrays of the matrix
multiplication on the heap
The matrix multiplication optimization copies blocks of its operands into the arrays Packed_A and Packed_B. Their sizes are derived from the cache parameters rather than from the operands, so with the default parameters Packed_B alone takes 3 to 4 MB. They were allocated with alloca in the entry block of the function, and two optimized multiplications in one function exceeded the default 8 MB stack: two products of 100x100 int32 matrices declared as VLAs and inlined into main crash with a segmentation fault. Allocate them with malloc at the start of the SCoP and free them at its exit instead, as is already done for the arrays created by JSCoP import and by maximal static expansion.
Assisted-by: Claude (Anthropic)
---
polly/lib/Transform/MatmulOptimizer.cpp | 5 ++
...ttern-matching-based-opts-packed-arrays.ll | 70 +++++++++++++++++++
2 files changed, 75 insertions(+)
create mode 100644 polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
diff --git a/polly/lib/Transform/MatmulOptimizer.cpp b/polly/lib/Transform/MatmulOptimizer.cpp
index 7a6b3d25871c39..e85a23646d6b51 100644
--- a/polly/lib/Transform/MatmulOptimizer.cpp
+++ b/polly/lib/Transform/MatmulOptimizer.cpp
@@ -810,6 +810,10 @@ static isl::schedule_node optimizePackedB(isl::schedule_node Node,
ScopArrayInfo *PackedB =
S->createScopArrayInfo(MMI.B->getElementType(), "Packed_B",
{FirstDimSize, SecondDimSize, ThirdDimSize});
+ // The packed arrays are sized by the cache parameters rather than by the
+ // operands and take megabytes. On the stack, a few of them in one function
+ // would overflow it.
+ PackedB->setIsOnHeap(true);
// Compute the access relation for copying from B to PackedB.
isl::map AccRelB = MMI.B->getLatestAccessRelation();
@@ -849,6 +853,7 @@ static isl::schedule_node optimizePackedA(isl::schedule_node Node, ScopStmt *,
ScopArrayInfo *PackedA = Stmt->getParent()->createScopArrayInfo(
MMI.A->getElementType(), "Packed_A",
{FirstDimSize, SecondDimSize, ThirdDimSize});
+ PackedA->setIsOnHeap(true);
// Compute the access relation for copying from A to PackedA.
isl::map AccRelA = MMI.A->getLatestAccessRelation();
diff --git a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
new file mode 100644
index 00000000000000..23cbecd2b4d924
--- /dev/null
+++ b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
@@ -0,0 +1,70 @@
+; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s
+;
+; The packed arrays of the matrix multiplication optimization are sized by the
+; cache parameters, which makes them megabytes large. Check that they are
+; allocated on the heap, since a few of them in one function would overflow
+; the stack.
+;
+; /* C := alpha*A*B + beta*C */
+; for (i = 0; i < _PB_NI; i++)
+; for (j = 0; j < _PB_NJ; j++)
+; {
+; C[i][j] *= beta;
+; for (k = 0; k < _PB_NK; ++k)
+; C[i][j] += alpha * A[i][k] * B[k][j];
+; }
+;
+; CHECK-LABEL: define internal void @kernel_gemm(
+; CHECK-NOT: %Packed_{{[AB]}} = alloca
+; CHECK: %Packed_B = tail call ptr @malloc(i64 4194304)
+; CHECK-NEXT: %Packed_A = tail call ptr @malloc(i64 196608)
+; CHECK: tail call void @free(ptr %Packed_B)
+; CHECK-NEXT: tail call void @free(ptr %Packed_A)
+
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-unknown"
+
+define internal void @kernel_gemm(i32 %arg, i32 %arg1, i32 %arg2, double %arg3, double %arg4, ptr %arg5, ptr %arg6, ptr %arg7) #0 {
+bb:
+ br label %bb8
+
+bb8: ; preds = %bb29, %bb
+ %tmp = phi i64 [ 0, %bb ], [ %tmp30, %bb29 ]
+ br label %bb9
+
+bb9: ; preds = %bb26, %bb8
+ %tmp10 = phi i64 [ 0, %bb8 ], [ %tmp27, %bb26 ]
+ %tmp11 = getelementptr inbounds [1056 x double], ptr %arg5, i64 %tmp, i64 %tmp10
+ %tmp12 = load double, ptr %tmp11, align 8
+ %tmp13 = fmul double %tmp12, %arg4
+ store double %tmp13, ptr %tmp11, align 8
+ br label %Copy_0
+
+Copy_0: ; preds = %Copy_0, %bb9
+ %tmp15 = phi i64 [ 0, %bb9 ], [ %tmp24, %Copy_0 ]
+ %tmp16 = getelementptr inbounds [1024 x double], ptr %arg6, i64 %tmp, i64 %tmp15
+ %tmp17 = load double, ptr %tmp16, align 8
+ %tmp18 = fmul double %tmp17, %arg3
+ %tmp19 = getelementptr inbounds [1056 x double], ptr %arg7, i64 %tmp15, i64 %tmp10
+ %tmp20 = load double, ptr %tmp19, align 8
+ %tmp21 = fmul double %tmp18, %tmp20
+ %tmp22 = load double, ptr %tmp11, align 8
+ %tmp23 = fadd double %tmp22, %tmp21
+ store double %tmp23, ptr %tmp11, align 8
+ %tmp24 = add nuw nsw i64 %tmp15, 1
+ %tmp25 = icmp ne i64 %tmp24, 1024
+ br i1 %tmp25, label %Copy_0, label %bb26
+
+bb26: ; preds = %Copy_0
+ %tmp27 = add nuw nsw i64 %tmp10, 1
+ %tmp28 = icmp ne i64 %tmp27, 1056
+ br i1 %tmp28, label %bb9, label %bb29
+
+bb29: ; preds = %bb26
+ %tmp30 = add nuw nsw i64 %tmp, 1
+ %tmp31 = icmp ne i64 %tmp30, 1056
+ br i1 %tmp31, label %bb8, label %bb32
+
+bb32: ; preds = %bb29
+ ret void
+}
More information about the llvm-commits
mailing list