[polly] [Polly] Allocate packed arrays of matrix multiplication on heap (PR #226163)
Timur Baidusenov via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 03:28:32 PDT 2026
https://github.com/bai-tim updated https://github.com/llvm/llvm-project/pull/226163
>From d2bbc55d51a34001a7b33f2d3449207cc2f9764d Mon Sep 17 00:00:00 2001
From: Timur Baidusenov <timurbaidusenov at gmail.com>
Date: Thu, 24 Sep 2026 04:05:12 +0300
Subject: [PATCH 1/4] [Polly] Allocate the packed arrays of the matrix
multiplication on the heap
The matrix multiplication optimization copies blocks of its operands into the arrays Packed_A and Packed_B. Their sizes are derived from the cache parameters rather than from the operands, so with the default parameters Packed_B alone takes 3 to 4 MB. They were allocated with alloca in the entry block of the function, and two optimized multiplications in one function exceeded the default 8 MB stack: two products of 100x100 int32 matrices declared as VLAs and inlined into main crash with a segmentation fault. Allocate them with malloc at the start of the SCoP and free them at its exit instead, as is already done for the arrays created by JSCoP import and by maximal static expansion.
Assisted-by: Claude (Anthropic)
---
polly/lib/Transform/MatmulOptimizer.cpp | 5 ++
...ttern-matching-based-opts-packed-arrays.ll | 70 +++++++++++++++++++
2 files changed, 75 insertions(+)
create mode 100644 polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
diff --git a/polly/lib/Transform/MatmulOptimizer.cpp b/polly/lib/Transform/MatmulOptimizer.cpp
index 7a6b3d25871c3..e85a23646d6b5 100644
--- a/polly/lib/Transform/MatmulOptimizer.cpp
+++ b/polly/lib/Transform/MatmulOptimizer.cpp
@@ -810,6 +810,10 @@ static isl::schedule_node optimizePackedB(isl::schedule_node Node,
ScopArrayInfo *PackedB =
S->createScopArrayInfo(MMI.B->getElementType(), "Packed_B",
{FirstDimSize, SecondDimSize, ThirdDimSize});
+ // The packed arrays are sized by the cache parameters rather than by the
+ // operands and take megabytes. On the stack, a few of them in one function
+ // would overflow it.
+ PackedB->setIsOnHeap(true);
// Compute the access relation for copying from B to PackedB.
isl::map AccRelB = MMI.B->getLatestAccessRelation();
@@ -849,6 +853,7 @@ static isl::schedule_node optimizePackedA(isl::schedule_node Node, ScopStmt *,
ScopArrayInfo *PackedA = Stmt->getParent()->createScopArrayInfo(
MMI.A->getElementType(), "Packed_A",
{FirstDimSize, SecondDimSize, ThirdDimSize});
+ PackedA->setIsOnHeap(true);
// Compute the access relation for copying from A to PackedA.
isl::map AccRelA = MMI.A->getLatestAccessRelation();
diff --git a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
new file mode 100644
index 0000000000000..23cbecd2b4d92
--- /dev/null
+++ b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
@@ -0,0 +1,70 @@
+; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s
+;
+; The packed arrays of the matrix multiplication optimization are sized by the
+; cache parameters, which makes them megabytes large. Check that they are
+; allocated on the heap, since a few of them in one function would overflow
+; the stack.
+;
+; /* C := alpha*A*B + beta*C */
+; for (i = 0; i < _PB_NI; i++)
+; for (j = 0; j < _PB_NJ; j++)
+; {
+; C[i][j] *= beta;
+; for (k = 0; k < _PB_NK; ++k)
+; C[i][j] += alpha * A[i][k] * B[k][j];
+; }
+;
+; CHECK-LABEL: define internal void @kernel_gemm(
+; CHECK-NOT: %Packed_{{[AB]}} = alloca
+; CHECK: %Packed_B = tail call ptr @malloc(i64 4194304)
+; CHECK-NEXT: %Packed_A = tail call ptr @malloc(i64 196608)
+; CHECK: tail call void @free(ptr %Packed_B)
+; CHECK-NEXT: tail call void @free(ptr %Packed_A)
+
+target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-unknown"
+
+define internal void @kernel_gemm(i32 %arg, i32 %arg1, i32 %arg2, double %arg3, double %arg4, ptr %arg5, ptr %arg6, ptr %arg7) #0 {
+bb:
+ br label %bb8
+
+bb8: ; preds = %bb29, %bb
+ %tmp = phi i64 [ 0, %bb ], [ %tmp30, %bb29 ]
+ br label %bb9
+
+bb9: ; preds = %bb26, %bb8
+ %tmp10 = phi i64 [ 0, %bb8 ], [ %tmp27, %bb26 ]
+ %tmp11 = getelementptr inbounds [1056 x double], ptr %arg5, i64 %tmp, i64 %tmp10
+ %tmp12 = load double, ptr %tmp11, align 8
+ %tmp13 = fmul double %tmp12, %arg4
+ store double %tmp13, ptr %tmp11, align 8
+ br label %Copy_0
+
+Copy_0: ; preds = %Copy_0, %bb9
+ %tmp15 = phi i64 [ 0, %bb9 ], [ %tmp24, %Copy_0 ]
+ %tmp16 = getelementptr inbounds [1024 x double], ptr %arg6, i64 %tmp, i64 %tmp15
+ %tmp17 = load double, ptr %tmp16, align 8
+ %tmp18 = fmul double %tmp17, %arg3
+ %tmp19 = getelementptr inbounds [1056 x double], ptr %arg7, i64 %tmp15, i64 %tmp10
+ %tmp20 = load double, ptr %tmp19, align 8
+ %tmp21 = fmul double %tmp18, %tmp20
+ %tmp22 = load double, ptr %tmp11, align 8
+ %tmp23 = fadd double %tmp22, %tmp21
+ store double %tmp23, ptr %tmp11, align 8
+ %tmp24 = add nuw nsw i64 %tmp15, 1
+ %tmp25 = icmp ne i64 %tmp24, 1024
+ br i1 %tmp25, label %Copy_0, label %bb26
+
+bb26: ; preds = %Copy_0
+ %tmp27 = add nuw nsw i64 %tmp10, 1
+ %tmp28 = icmp ne i64 %tmp27, 1056
+ br i1 %tmp28, label %bb9, label %bb29
+
+bb29: ; preds = %bb26
+ %tmp30 = add nuw nsw i64 %tmp, 1
+ %tmp31 = icmp ne i64 %tmp30, 1056
+ br i1 %tmp31, label %bb8, label %bb32
+
+bb32: ; preds = %bb29
+ ret void
+}
>From a129a451b7f248d94b962888e05a151b4d3c9c92 Mon Sep 17 00:00:00 2001
From: Timur Baidusenov <timurbaidusenov at gmail.com>
Date: Mon, 28 Sep 2026 22:57:30 +0300
Subject: [PATCH 2/4] Address review: keep packed arrays up to a size limit on
the stack
Allocate a packed array of the matrix multiplication optimization on the heap only if it is larger than -polly-pattern-matching-max-stack-array-size, 1 MiB by default, and on the stack otherwise; -1 keeps all of them on the stack, 0 puts all of them on the heap. With the default cache parameters, Packed_B (4 MiB) goes to the heap, which avoids the stack overflow of two optimized multiplications in one function, while Packed_A (192 KiB) stays on the stack.
Assisted-by: Claude (Anthropic)
---
polly/lib/Transform/MatmulOptimizer.cpp | 30 +++++++++++++++----
...ttern-matching-based-opts-packed-arrays.ll | 27 +++++++++++++----
2 files changed, 46 insertions(+), 11 deletions(-)
diff --git a/polly/lib/Transform/MatmulOptimizer.cpp b/polly/lib/Transform/MatmulOptimizer.cpp
index e85a23646d6b5..fcf96ecf47e28 100644
--- a/polly/lib/Transform/MatmulOptimizer.cpp
+++ b/polly/lib/Transform/MatmulOptimizer.cpp
@@ -128,6 +128,14 @@ static cl::opt<int> PollyPatternMatchingNcQuotient(
"macro-kernel, by Nr, the parameter of the micro-kernel"),
cl::Hidden, cl::init(256), cl::cat(PollyCategory));
+static cl::opt<int> MaxStackArraySize(
+ "polly-pattern-matching-max-stack-array-size",
+ cl::desc("The maximal size in bytes of a packed array of the matrix "
+ "multiplication optimization that is allocated on the stack; "
+ "larger ones are allocated on the heap (-1: all on the stack, "
+ "0: all on the heap)"),
+ cl::Hidden, cl::init(1024 * 1024), cl::cat(PollyCategory));
+
static cl::opt<bool>
PMBasedTCOpts("polly-tc-opt",
cl::desc("Perform optimizations of tensor contractions based "
@@ -795,6 +803,19 @@ static isl::schedule_node createExtensionNode(isl::schedule_node Node,
return Node.graft_before(NewNode);
}
+/// Allocate the packed array @p SAI, whose dimensions have the sizes
+/// @p DimSizes, on the heap if it is larger than
+/// -polly-pattern-matching-max-stack-array-size and that is not negative, and
+/// on the stack otherwise.
+static void setPackedArrayAllocation(ScopArrayInfo *SAI,
+ ArrayRef<unsigned> DimSizes) {
+ uint64_t Size = SAI->getElemSizeInBytes();
+ for (unsigned DimSize : DimSizes)
+ Size *= DimSize;
+ SAI->setIsOnHeap(MaxStackArraySize >= 0 &&
+ Size > uint64_t(MaxStackArraySize));
+}
+
static isl::schedule_node optimizePackedB(isl::schedule_node Node,
ScopStmt *Stmt, isl::map MapOldIndVar,
MicroKernelParamsTy MicroParams,
@@ -810,10 +831,8 @@ static isl::schedule_node optimizePackedB(isl::schedule_node Node,
ScopArrayInfo *PackedB =
S->createScopArrayInfo(MMI.B->getElementType(), "Packed_B",
{FirstDimSize, SecondDimSize, ThirdDimSize});
- // The packed arrays are sized by the cache parameters rather than by the
- // operands and take megabytes. On the stack, a few of them in one function
- // would overflow it.
- PackedB->setIsOnHeap(true);
+ setPackedArrayAllocation(PackedB,
+ {FirstDimSize, SecondDimSize, ThirdDimSize});
// Compute the access relation for copying from B to PackedB.
isl::map AccRelB = MMI.B->getLatestAccessRelation();
@@ -853,7 +872,8 @@ static isl::schedule_node optimizePackedA(isl::schedule_node Node, ScopStmt *,
ScopArrayInfo *PackedA = Stmt->getParent()->createScopArrayInfo(
MMI.A->getElementType(), "Packed_A",
{FirstDimSize, SecondDimSize, ThirdDimSize});
- PackedA->setIsOnHeap(true);
+ setPackedArrayAllocation(PackedA,
+ {FirstDimSize, SecondDimSize, ThirdDimSize});
// Compute the access relation for copying from A to PackedA.
isl::map AccRelA = MMI.A->getLatestAccessRelation();
diff --git a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
index 23cbecd2b4d92..26ae32930acbf 100644
--- a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
+++ b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
@@ -1,9 +1,12 @@
; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s
+; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 -polly-pattern-matching-max-stack-array-size=-1 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=STACK
+; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 -polly-pattern-matching-max-stack-array-size=0 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=HEAP
;
; The packed arrays of the matrix multiplication optimization are sized by the
-; cache parameters, which makes them megabytes large. Check that they are
-; allocated on the heap, since a few of them in one function would overflow
-; the stack.
+; cache parameters: here Packed_A takes 192 KiB and Packed_B 4 MiB. Arrays
+; larger than -polly-pattern-matching-max-stack-array-size (1 MiB by default)
+; are allocated on the heap, the others on the stack. With -1 all of them are
+; allocated on the stack, with 0 all of them on the heap.
;
; /* C := alpha*A*B + beta*C */
; for (i = 0; i < _PB_NI; i++)
@@ -15,11 +18,23 @@
; }
;
; CHECK-LABEL: define internal void @kernel_gemm(
-; CHECK-NOT: %Packed_{{[AB]}} = alloca
+; CHECK: %Packed_A = alloca [24 x [256 x [4 x double]]]
; CHECK: %Packed_B = tail call ptr @malloc(i64 4194304)
-; CHECK-NEXT: %Packed_A = tail call ptr @malloc(i64 196608)
+; CHECK-NOT: call ptr @malloc
; CHECK: tail call void @free(ptr %Packed_B)
-; CHECK-NEXT: tail call void @free(ptr %Packed_A)
+; CHECK-NOT: call void @free
+;
+; STACK-LABEL: define internal void @kernel_gemm(
+; STACK: %Packed_B = alloca [256 x [256 x [8 x double]]]
+; STACK-NEXT: %Packed_A = alloca [24 x [256 x [4 x double]]]
+; STACK-NOT: call ptr @malloc
+;
+; HEAP-LABEL: define internal void @kernel_gemm(
+; HEAP-NOT: %Packed_{{[AB]}} = alloca
+; HEAP: %Packed_B = tail call ptr @malloc(i64 4194304)
+; HEAP-NEXT: %Packed_A = tail call ptr @malloc(i64 196608)
+; HEAP: tail call void @free(ptr %Packed_B)
+; HEAP-NEXT: tail call void @free(ptr %Packed_A)
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-unknown-unknown"
>From ae054dfdf528506e7d1f2c176073872f719afe14 Mon Sep 17 00:00:00 2001
From: Timur Baidusenov <timurbaidusenov at gmail.com>
Date: Tue, 29 Sep 2026 22:52:17 +0300
Subject: [PATCH 3/4] Address review
---
polly/docs/ReleaseNotes.rst | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/polly/docs/ReleaseNotes.rst b/polly/docs/ReleaseNotes.rst
index 618a4265f09cf..1cb5262e58937 100644
--- a/polly/docs/ReleaseNotes.rst
+++ b/polly/docs/ReleaseNotes.rst
@@ -19,3 +19,7 @@ In Polly |version| the following important changes have been incorporated.
* The infrastructure around ScopPasses has been removed.
+ * The matrix multiplication optimization allocates packed arrays larger than
+ ``-polly-pattern-matching-max-stack-array-size`` (1 MiB by default) on the
+ heap instead of the stack. ``-1`` keeps all of them on the stack.
+
>From 2d596adb5dd27c61fca38f28d39de3465de394f9 Mon Sep 17 00:00:00 2001
From: Timur Baidusenov <timurbaidusenov at gmail.com>
Date: Wed, 30 Sep 2026 13:22:44 +0300
Subject: [PATCH 4/4] Pass the Polly options of the new test with -plugin-arg
Assisted-by: Claude (Anthropic)
---
.../pattern-matching-based-opts-packed-arrays.ll | 6 +++---
1 file changed, 3 insertions(+), 3 deletions(-)
diff --git a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
index 26ae32930acbf..9a453302d78ad 100644
--- a/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
+++ b/polly/test/ScheduleOptimizer/pattern-matching-based-opts-packed-arrays.ll
@@ -1,6 +1,6 @@
-; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s
-; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 -polly-pattern-matching-max-stack-array-size=-1 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=STACK
-; RUN: opt %loadNPMPolly -polly-pattern-matching-based-opts=true -polly-target-throughput-vector-fma=1 -polly-target-latency-vector-fma=8 -polly-target-1st-cache-level-associativity=8 -polly-target-2nd-cache-level-associativity=8 -polly-target-1st-cache-level-size=32768 -polly-target-vector-register-bitwidth=256 -polly-target-2nd-cache-level-size=262144 -polly-pattern-matching-max-stack-array-size=0 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=HEAP
+; RUN: opt %loadNPMPolly -plugin-arg=Polly,-polly-pattern-matching-based-opts=true -plugin-arg=Polly,-polly-target-throughput-vector-fma=1 -plugin-arg=Polly,-polly-target-latency-vector-fma=8 -plugin-arg=Polly,-polly-target-1st-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-2nd-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-1st-cache-level-size=32768 -plugin-arg=Polly,-polly-target-vector-register-bitwidth=256 -plugin-arg=Polly,-polly-target-2nd-cache-level-size=262144 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s
+; RUN: opt %loadNPMPolly -plugin-arg=Polly,-polly-pattern-matching-based-opts=true -plugin-arg=Polly,-polly-target-throughput-vector-fma=1 -plugin-arg=Polly,-polly-target-latency-vector-fma=8 -plugin-arg=Polly,-polly-target-1st-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-2nd-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-1st-cache-level-size=32768 -plugin-arg=Polly,-polly-target-vector-register-bitwidth=256 -plugin-arg=Polly,-polly-target-2nd-cache-level-size=262144 -plugin-arg=Polly,-polly-pattern-matching-max-stack-array-size=-1 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=STACK
+; RUN: opt %loadNPMPolly -plugin-arg=Polly,-polly-pattern-matching-based-opts=true -plugin-arg=Polly,-polly-target-throughput-vector-fma=1 -plugin-arg=Polly,-polly-target-latency-vector-fma=8 -plugin-arg=Polly,-polly-target-1st-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-2nd-cache-level-associativity=8 -plugin-arg=Polly,-polly-target-1st-cache-level-size=32768 -plugin-arg=Polly,-polly-target-vector-register-bitwidth=256 -plugin-arg=Polly,-polly-target-2nd-cache-level-size=262144 -plugin-arg=Polly,-polly-pattern-matching-max-stack-array-size=0 '-passes=polly<no-default-opts;opt-isl>' -S < %s | FileCheck %s --check-prefix=HEAP
;
; The packed arrays of the matrix multiplication optimization are sized by the
; cache parameters: here Packed_A takes 192 KiB and Packed_B 4 MiB. Arrays
More information about the llvm-commits
mailing list