[llvm] [MemCpyOpt] Extend call slot optimization for non-dereferenceable destinations. (PR #217436)
Alina Sbirlea via llvm-commits
llvm-commits at lists.llvm.org
Sun Aug 23 15:58:20 PDT 2026
https://github.com/alinas updated https://github.com/llvm/llvm-project/pull/217436
>From 894a8584b4d01a8cd2e7070664cd96d7a9c9a501 Mon Sep 17 00:00:00 2001
From: Alina Sbirlea <asbirlea at google.com>
Date: Tue, 18 Aug 2026 19:35:11 +0000
Subject: [PATCH 1/2] [MemCpyOpt] Extend call slot optimization.
Allow call slot optimization for non-dereferenceable destinations when
execution is guaranteed to reach the store. For this, check if the call has
both willreturn and nounwind attributes, and there are no instructions
between the call and the store that might trap or throw.
Since the store would trap anyway if the destination pointer was not
dereferenceable, we can forward the pointer to the call.
This optimization is restricted to non-memcpy/memset calls because these
are handled separately after the callSlotOptimization.
---
.../lib/Transforms/Scalar/MemCpyOptimizer.cpp | 14 +++-
.../Transforms/MemCpyOpt/callslot_deref.ll | 76 ++++++++++++++++++-
2 files changed, 87 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Scalar/MemCpyOptimizer.cpp b/llvm/lib/Transforms/Scalar/MemCpyOptimizer.cpp
index 92e1d9cf21240..09760694c9730 100644
--- a/llvm/lib/Transforms/Scalar/MemCpyOptimizer.cpp
+++ b/llvm/lib/Transforms/Scalar/MemCpyOptimizer.cpp
@@ -924,8 +924,18 @@ bool MemCpyOptPass::performCallSlotOptzn(Instruction *cpyLoad,
ExplicitlyDereferenceableOnly) ||
!isDereferenceablePointer(cpyDest, APInt(64, cpySize),
SimplifyQuery(DL, DT, AC, C))) {
- LLVM_DEBUG(dbgs() << "Call Slot: Dest pointer not dereferenceable\n");
- return false;
+ // If the call is guaranteed to return normally (willreturn + nounwind),
+ // and there are no instructions between the call and the store that might
+ // trap or throw, execution will reach the store. Since the store would
+ // trap anyway if the pointer was not dereferenceable, we can forward the
+ // pointer to the call. Perform optimization only for non-memcpy/memset
+ // calls, as those are special cased later.
+ if (isa<AnyMemCpyInst>(C) || isa<AnyMemSetInst>(C) ||
+ !isGuaranteedToTransferExecutionToSuccessor(C->getIterator(),
+ cpyStore->getIterator())) {
+ LLVM_DEBUG(dbgs() << "Call Slot: Dest pointer not dereferenceable\n");
+ return false;
+ }
}
// Make sure that nothing can observe cpyDest being written early. There are
diff --git a/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll b/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
index dcbede105a3b1..f806277d6b7bf 100644
--- a/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
+++ b/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
@@ -2,7 +2,7 @@
; RUN: opt < %s -S -passes=memcpyopt -verify-memoryssa | FileCheck %s
target datalayout = "e-i64:64-f80:128-n8:16:32:64-S128"
-declare void @llvm.memcpy.p0.p0.i64(ptr nocapture, ptr nocapture readonly, i64, i1) unnamed_addr nounwind
+declare void @llvm.memcpy.p0.p0.i64(ptr nocapture writeonly, ptr nocapture readonly, i64, i1) unnamed_addr nounwind
declare void @llvm.memset.p0.i64(ptr nocapture, i8, i64, i1) nounwind
; all bytes of %dst that are touch by the memset are dereferenceable
@@ -32,3 +32,77 @@ define void @must_not_remove_memcpy(ptr noalias nocapture writable dereferenceab
call void @llvm.memcpy.p0.p0.i64(ptr %dst, ptr %src, i64 4096, i1 false) #2
ret void
}
+
+; enable copy elision when call has both willreturn and nounwind.
+define void @test_positive_willreturn_nounwind(ptr %val) {
+; CHECK-LABEL: @test_positive_willreturn_nounwind(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
+; CHECK-NEXT: call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 [[VAL:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-NEXT: ret void
+;
+entry:
+ %tmp = alloca [1024 x i32], align 4
+ call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #0
+ call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
+ ret void
+}
+
+; missing willreturn attribute; optimization should not occur.
+define void @test_negative_missing_willreturn(ptr %val) {
+; CHECK-LABEL: @test_negative_missing_willreturn(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
+; CHECK-NEXT: call void @bar_missing_willreturn(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR4:[0-9]+]]
+; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
+; CHECK-NEXT: ret void
+;
+entry:
+ %tmp = alloca [1024 x i32], align 4
+ call void @bar_missing_willreturn(ptr sret([1024 x i32]) align 4 %tmp) #1
+ call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
+ ret void
+}
+
+; missing nounwind attribute; optimization should not occur.
+define void @test_negative_missing_nounwind(ptr %val) {
+; CHECK-LABEL: @test_negative_missing_nounwind(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
+; CHECK-NEXT: call void @bar_missing_nounwind(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR5:[0-9]+]]
+; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
+; CHECK-NEXT: ret void
+;
+entry:
+ %tmp = alloca [1024 x i32], align 4
+ call void @bar_missing_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #2
+ call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
+ ret void
+}
+
+; instruction between call and memcpy that may trap.
+define void @test_negative_may_trap_between(ptr %val, ptr %unknown) {
+; CHECK-LABEL: @test_negative_may_trap_between(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
+; CHECK-NEXT: call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR3]]
+; CHECK-NEXT: [[VAL_LOADED:%.*]] = load volatile i32, ptr [[UNKNOWN:%.*]], align 4
+; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
+; CHECK-NEXT: ret void
+;
+entry:
+ %tmp = alloca [1024 x i32], align 4
+ call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #0
+ ; A volatile load can trap and has side effects.
+ %val_loaded = load volatile i32, ptr %unknown
+ call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
+ ret void
+}
+
+declare void @bar_willreturn_nounwind(ptr nocapture sret([1024 x i32]) align 4) #0
+declare void @bar_missing_willreturn(ptr nocapture sret([1024 x i32]) align 4) #1
+declare void @bar_missing_nounwind(ptr nocapture sret([1024 x i32]) align 4) #2
+
+attributes #0 = { willreturn nounwind memory(argmem: write) }
+attributes #1 = { nounwind memory(argmem: write) }
+attributes #2 = { willreturn memory(argmem: write) }
>From 26a59ad6ede934b19179217e0bcc73e83311451c Mon Sep 17 00:00:00 2001
From: Alina Sbirlea <asbirlea at google.com>
Date: Sun, 23 Aug 2026 22:57:43 +0000
Subject: [PATCH 2/2] update test
---
.../Transforms/MemCpyOpt/callslot_deref.ll | 26 +++++++------------
1 file changed, 10 insertions(+), 16 deletions(-)
diff --git a/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll b/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
index f806277d6b7bf..38d47fb04e879 100644
--- a/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
+++ b/llvm/test/Transforms/MemCpyOpt/callslot_deref.ll
@@ -38,12 +38,12 @@ define void @test_positive_willreturn_nounwind(ptr %val) {
; CHECK-LABEL: @test_positive_willreturn_nounwind(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
-; CHECK-NEXT: call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 [[VAL:%.*]]) #[[ATTR3:[0-9]+]]
+; CHECK-NEXT: call void @bar(ptr sret([1024 x i32]) align 4 [[VAL:%.*]]) #[[ATTR3:[0-9]+]]
; CHECK-NEXT: ret void
;
entry:
%tmp = alloca [1024 x i32], align 4
- call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #0
+ call void @bar(ptr sret([1024 x i32]) align 4 %tmp) willreturn nounwind memory(argmem: write)
call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
ret void
}
@@ -53,13 +53,13 @@ define void @test_negative_missing_willreturn(ptr %val) {
; CHECK-LABEL: @test_negative_missing_willreturn(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
-; CHECK-NEXT: call void @bar_missing_willreturn(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR4:[0-9]+]]
+; CHECK-NEXT: call void @bar(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR4:[0-9]+]]
; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
; CHECK-NEXT: ret void
;
entry:
%tmp = alloca [1024 x i32], align 4
- call void @bar_missing_willreturn(ptr sret([1024 x i32]) align 4 %tmp) #1
+ call void @bar(ptr sret([1024 x i32]) align 4 %tmp) nounwind memory(argmem: write)
call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
ret void
}
@@ -69,40 +69,34 @@ define void @test_negative_missing_nounwind(ptr %val) {
; CHECK-LABEL: @test_negative_missing_nounwind(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
-; CHECK-NEXT: call void @bar_missing_nounwind(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR5:[0-9]+]]
+; CHECK-NEXT: call void @bar(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR5:[0-9]+]]
; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
; CHECK-NEXT: ret void
;
entry:
%tmp = alloca [1024 x i32], align 4
- call void @bar_missing_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #2
+ call void @bar(ptr sret([1024 x i32]) align 4 %tmp) willreturn memory(argmem: write)
call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
ret void
}
-; instruction between call and memcpy that may trap.
+; willreturn nounwind attributes, but instruction between call and memcpy that may trap.
define void @test_negative_may_trap_between(ptr %val, ptr %unknown) {
; CHECK-LABEL: @test_negative_may_trap_between(
; CHECK-NEXT: entry:
; CHECK-NEXT: [[TMP:%.*]] = alloca [1024 x i32], align 4
-; CHECK-NEXT: call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR3]]
+; CHECK-NEXT: call void @bar(ptr sret([1024 x i32]) align 4 [[TMP]]) #[[ATTR3]]
; CHECK-NEXT: [[VAL_LOADED:%.*]] = load volatile i32, ptr [[UNKNOWN:%.*]], align 4
; CHECK-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 4 [[VAL:%.*]], ptr align 4 [[TMP]], i64 4096, i1 false)
; CHECK-NEXT: ret void
;
entry:
%tmp = alloca [1024 x i32], align 4
- call void @bar_willreturn_nounwind(ptr sret([1024 x i32]) align 4 %tmp) #0
+ call void @bar(ptr sret([1024 x i32]) align 4 %tmp) willreturn nounwind memory(argmem: write)
; A volatile load can trap and has side effects.
%val_loaded = load volatile i32, ptr %unknown
call void @llvm.memcpy.p0.p0.i64(ptr align 4 %val, ptr align 4 %tmp, i64 4096, i1 false)
ret void
}
-declare void @bar_willreturn_nounwind(ptr nocapture sret([1024 x i32]) align 4) #0
-declare void @bar_missing_willreturn(ptr nocapture sret([1024 x i32]) align 4) #1
-declare void @bar_missing_nounwind(ptr nocapture sret([1024 x i32]) align 4) #2
-
-attributes #0 = { willreturn nounwind memory(argmem: write) }
-attributes #1 = { nounwind memory(argmem: write) }
-attributes #2 = { willreturn memory(argmem: write) }
+declare void @bar(ptr nocapture sret([1024 x i32]) align 4)
More information about the llvm-commits
mailing list