[clang] [CIR] Copy byval records as bytes in CallConvLowering (PR #227170)
Bruno Cardoso Lopes via cfe-commits
cfe-commits at lists.llvm.org
Mon Sep 28 18:50:38 PDT 2026
https://github.com/bcardosolopes updated https://github.com/llvm/llvm-project/pull/227170
>From 4975c573e1a5af46ab472135e7d1fbed89fd779c Mon Sep 17 00:00:00 2001
From: Bruno Cardoso Lopes <bruno.cardoso at gmail.com>
Date: Mon, 28 Sep 2026 15:52:50 -0700
Subject: [PATCH 1/2] [CIR] Copy byval records as bytes in CallConvLowering
A byval record argument was copied with a whole-record cir.load/cir.store,
both in the callee (into the parameter's slot) and at call sites (into the
byval temporary). A record's LLVM type is built from one member for a union,
so bytes that are padding in that member but data in another were dropped.
A clang built with -fclangir miscompiled itself this way: TemplateArgument,
passed by value, lost its integer bit width.
Use cir.copy instead when the value comes straight from memory.
---
.../TargetLowering/CIRABIRewriteContext.cpp | 43 ++++++++++++++++++-
.../call-conv-lowering-x86_64-byval-union.c | 41 ++++++++++++++++++
.../call-conv-lowering-x86_64-non-byval.cpp | 6 +--
.../call-conv-lowering-x86_64-variadic.c | 32 +++++++-------
clang/test/CIR/CodeGen/call.c | 8 ++--
.../abi-lowering/indirect-byval.cir | 7 ++-
.../Transforms/abi-lowering/indirect-call.cir | 5 ++-
.../indirect-non-byval-forward-param.cir | 29 +++++--------
.../abi-lowering/x86_64-struct-indirect.cir | 6 +--
.../Transforms/abi-lowering/x86_64-union.cir | 5 +--
10 files changed, 125 insertions(+), 57 deletions(-)
create mode 100644 clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
index a7639abb62466..0ed6ad22797a4 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
@@ -800,8 +800,28 @@ void insertArgCoercion(
if (destAlloca)
pendingParamSlots.emplace_back(destAlloca, blockArg);
} else {
- // byval: load the incoming pointer so the body sees a T value (and
- // any CIRGen param-slot store becomes a local copy of that value).
+ // byval: the body gets a local copy of the incoming value. When a
+ // record's only use is CIRGen's param spill, copy the bytes into the
+ // slot: a loaded record value carries only the fields of the LLVM
+ // type, which for a union is one member's, so bytes that are padding
+ // in that member but data in another would be lost.
+ auto spill =
+ isa<cir::RecordType>(blockArg.getType()) && blockArg.hasOneUse()
+ ? dyn_cast<cir::StoreOp>(*blockArg.user_begin())
+ : cir::StoreOp();
+ if (spill && spill.getValue() == blockArg) {
+ builder.setInsertionPoint(spill);
+ cir::CopyOp::create(
+ builder, spill.getLoc(), spill.getAddr(), blockArg,
+ /*dst_alignment=*/{},
+ builder.getI64IntegerAttr(ac.indirectAlign.value()));
+ spill->erase();
+ blockArg.setType(ptrTy);
+ ++blockArgIdx;
+ continue;
+ }
+
+ // Otherwise load the incoming pointer so the body sees a T value.
blockArg.setType(ptrTy);
builder.setInsertionPointToStart(&entry);
@@ -1549,6 +1569,25 @@ CIRABIRewriteContext::rewriteCallSite(mlir::Operation *callOp,
continue;
}
auto ptrTy = cir::PointerType::get(arg.getType());
+ // A record loaded from memory is copied as bytes, at the load so it
+ // sees the same value: a record value carries only the fields of its
+ // LLVM type, which for a union is one member's.
+ cir::LoadOp srcLoad = isa<cir::RecordType>(arg.getType())
+ ? maybeGetSimpleLoad(arg)
+ : cir::LoadOp();
+ if (srcLoad && srcLoad.getAddr().getType() == ptrTy) {
+ mlir::OpBuilder::InsertionGuard guard(builder);
+ builder.setInsertionPointAfter(srcLoad);
+ auto slot = cir::AllocaOp::create(
+ builder, call.getLoc(), ptrTy, builder.getStringAttr("byval"),
+ builder.getI64IntegerAttr(ac.indirectAlign.value()));
+ cir::CopyOp::create(builder, call.getLoc(), slot, srcLoad.getAddr(),
+ builder.getI64IntegerAttr(ac.indirectAlign.value()),
+ srcLoad.getAlignmentAttr());
+ newArgs.push_back(slot);
+ deadRecordLoads.push_back(srcLoad);
+ continue;
+ }
auto slot = cir::AllocaOp::create(
builder, call.getLoc(), ptrTy, builder.getStringAttr("byval"),
builder.getI64IntegerAttr(ac.indirectAlign.value()));
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
new file mode 100644
index 0000000000000..5000f72080b52
--- /dev/null
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
@@ -0,0 +1,41 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+
+// A union passed byval is copied as bytes. Its LLVM type is built from one
+// member, so a record load/store would drop bytes that are padding in that
+// member but data in another: bytes 4-7, y.b, here.
+union U {
+ struct { int a; void *p, *q; } x;
+ struct { int a, b; void *p, *q; } y;
+};
+
+int get_b(union U u) { return u.y.b; }
+
+void pass(void) {
+ union U u;
+ u.y.b = 42;
+ get_b(u);
+}
+
+// CIR-LABEL: cir.func {{.*}}@get_b(%arg0: !cir.ptr<!rec_U> {llvm.align = 8 : i64, llvm.byval = !rec_U, llvm.noundef}
+// CIR: %[[U:.*]] = cir.alloca "u" align(8) init : !cir.ptr<!rec_U>
+// CIR: cir.copy %arg0 align(8) to %[[U]] : !cir.ptr<!rec_U>
+// CIR-NOT: cir.load {{.*}} !rec_U
+
+// CIR-LABEL: cir.func {{.*}}@pass()
+// CIR: %[[U:.*]] = cir.alloca "u" align(8) : !cir.ptr<!rec_U>
+// CIR: %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_U>
+// CIR-NEXT: cir.copy %[[U]] align(8) to %[[SLOT]] align(8) : !cir.ptr<!rec_U>
+// CIR-NEXT: cir.call @get_b(%[[SLOT]])
+
+// LLVM-LABEL: define {{.*}}i32 @get_b(ptr noundef byval(%union.U) align 8 %0)
+// LLVM: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.+}}, ptr align 8 %0, i64 24, i1 false)
+// LLVM-NOT: load %union.U
+
+// LLVM-LABEL: define {{.*}}void @pass()
+// LLVM: %[[U:.+]] = alloca %union.U, align 8
+// LLVM: %[[SLOT:.+]] = alloca %union.U, align 8
+// LLVM-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 8 %[[U]], i64 24, i1 false)
+// LLVM-NEXT: call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[SLOT]])
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
index 374ad6683d66d..58c683798f50a 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
@@ -100,15 +100,13 @@ void callByval() {
// CIR-LABEL: cir.func {{.*}}@_Z9callByvalv
// CIR: %[[TMP:.*]] = cir.alloca "agg.tmp0" align(8) : !cir.ptr<!rec_Big>
-// CIR: %[[V:.*]] = cir.load align(8) %[[TMP]] : !cir.ptr<!rec_Big>, !rec_Big
// CIR: %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR: cir.store %[[V]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+// CIR: cir.copy %[[TMP]] align(8) to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
// CIR: cir.call @_Z9takeByval3Big(%[[SLOT]]) : (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
// LLVM-LABEL: define dso_local void @_Z9callByvalv()
// LLVM: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[TMP:[^,]+]], ptr align 8 %{{[^,]+}}, i64 32, i1 false)
-// LLVM-CIR: %[[V:.*]] = load %struct.Big, ptr %[[TMP]], align 8
-// LLVM-CIR: store %struct.Big %[[V]], ptr %[[SLOT:.*]], align 8
+// LLVM-CIR: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT:[^,]+]], ptr align 8 %[[TMP]], i64 32, i1 false)
// LLVM-CIR: call void @_Z9takeByval3Big(ptr noundef byval(%struct.Big) align 8 %[[SLOT]])
// OGCG: call void @_Z9takeByval3Big(ptr noundef byval(%struct.Big) align 8 %[[TMP]])
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
index c7d3d4aa8c335..9505f01c2f3be 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
@@ -67,19 +67,19 @@ int call_small(Pair2 p, Pair16 q) { return vf(p, q); }
int call_big(Pair2 p, Big b) { return vf(p, b); }
// CIR-LABEL: cir.func {{.*}}@call_big(%arg0: !u64i loc({{.+}}), %arg1: !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef} loc({{.+}})) -> !s32i
-// CIR: %{{[0-9]+}} = cir.load %arg1 : !cir.ptr<!rec_Big>, !rec_Big
+// CIR: cir.copy %arg1 align(8) to %[[LOCAL:[0-9]+]] : !cir.ptr<!rec_Big>
+// CIR: %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+// CIR-NEXT: cir.copy %[[LOCAL]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
// CIR: %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT: %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR-NEXT: cir.store %{{[0-9]+}}, %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
// CIR-NEXT: %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[COPY]]) : (!u64i, !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s32i
// CIR copies the incoming byval slot before forwarding it. OGCG does not.
// LLVM-CIR-LABEL: define dso_local i32 @call_big(
// LLVM-CIR-SAME: i64 %[[P:[0-9a-zA-Z._]+]], ptr noundef byval(%struct.Big) align 8 %[[B:[0-9a-zA-Z._]+]])
-// LLVM-CIR: %{{[0-9a-zA-Z._]+}} = load %struct.Big, ptr %[[B]], align 8
+// LLVM-CIR: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[LOCAL:[0-9a-zA-Z._]+]], ptr align 8 %[[B]], i64 32, i1 false)
+// LLVM-CIR: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Big, align 8
+// LLVM-CIR-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[COPY]], ptr align 8 %[[LOCAL]], i64 32, i1 false)
// LLVM-CIR: %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Big, align 8
-// LLVM-CIR-NEXT: store %struct.Big %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 8
// LLVM-CIR-NEXT: %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], ptr noundef byval(%struct.Big) align 8 %[[COPY]])
// LLVM-OGCG-LABEL: define dso_local i32 @call_big(
@@ -103,9 +103,9 @@ int call_exhausted(Pair2 p, long a, long b, long c, long d, Pair16 q) {
// CIR: %[[BV:[0-9]+]] = cir.load align(8) %[[BS]] : !cir.ptr<!s64i>, !s64i
// CIR: %[[CV:[0-9]+]] = cir.load align(8) %[[CS]] : !cir.ptr<!s64i>, !s64i
// CIR: %[[DV:[0-9]+]] = cir.load align(8) %[[DS]] : !cir.ptr<!s64i>, !s64i
+// CIR: %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Pair16>
+// CIR-NEXT: cir.copy %{{[0-9]+}} align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Pair16>
// CIR: %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT: %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Pair16>
-// CIR-NEXT: cir.store %{{[0-9]+}}, %[[COPY]] : !rec_Pair16, !cir.ptr<!rec_Pair16>
// CIR-NEXT: %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[AV]], %[[BV]], %[[CV]], %[[DV]], %[[COPY]]) : (!u64i, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !cir.ptr<!rec_Pair16> {llvm.align = 8 : i64, llvm.byval = !rec_Pair16, llvm.noundef}) -> !s32i
// LLVM-CIR-LABEL: define dso_local i32 @call_exhausted(
@@ -120,9 +120,9 @@ int call_exhausted(Pair2 p, long a, long b, long c, long d, Pair16 q) {
// LLVM: %[[BV:[0-9a-zA-Z._]+]] = load i64, ptr %[[BS]], align 8
// LLVM: %[[CV:[0-9a-zA-Z._]+]] = load i64, ptr %[[CS]], align 8
// LLVM: %[[DV:[0-9a-zA-Z._]+]] = load i64, ptr %[[DS]], align 8
+// LLVM-CIR: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Pair16, align 8
+// LLVM-CIR-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[COPY]], ptr align 8 %{{[0-9a-zA-Z._]+}}, i64 16, i1 false)
// LLVM-CIR: %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Pair16, align 8
-// LLVM-CIR-NEXT: store %struct.Pair16 %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 8
// LLVM-CIR-NEXT: %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], i64 noundef %[[AV]], i64 noundef %[[BV]], i64 noundef %[[CV]], i64 noundef %[[DV]], ptr noundef byval(%struct.Pair16) align 8 %[[COPY]])
// LLVM-OGCG: %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 4
// LLVM-OGCG-NEXT: %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], i64 noundef %[[AV]], i64 noundef %[[BV]], i64 noundef %[[CV]], i64 noundef %[[DV]], ptr noundef byval(%struct.Pair16) align 8 %[[Q]])
@@ -178,18 +178,18 @@ int call_wide(Pair2 p, Wide w) { return vf(p, w); }
int call_wide_char(Pair2 p, WideChar w) { return vf(p, w); }
// CIR-LABEL: cir.func {{.*}}@call_wide_char(%arg0: !u64i loc({{.+}}), %arg1: !cir.ptr<!rec_WideChar> {llvm.align = 16 : i64, llvm.byval = !rec_WideChar, llvm.noundef} loc({{.+}})) -> !s32i
-// CIR: %{{[0-9]+}} = cir.load %arg1 : !cir.ptr<!rec_WideChar>, !rec_WideChar
+// CIR: cir.copy %arg1 align(16) to %[[LOCAL:[0-9]+]] : !cir.ptr<!rec_WideChar>
+// CIR: %[[COPY:[0-9]+]] = cir.alloca "byval" align(16) : !cir.ptr<!rec_WideChar>
+// CIR-NEXT: cir.copy %[[LOCAL]] align(16) to %[[COPY]] align(16) : !cir.ptr<!rec_WideChar>
// CIR: %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT: %[[COPY:[0-9]+]] = cir.alloca "byval" align(16) : !cir.ptr<!rec_WideChar>
-// CIR-NEXT: cir.store %{{[0-9]+}}, %[[COPY]] : !rec_WideChar, !cir.ptr<!rec_WideChar>
// CIR-NEXT: %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[COPY]]) : (!u64i, !cir.ptr<!rec_WideChar> {llvm.align = 16 : i64, llvm.byval = !rec_WideChar, llvm.noundef}) -> !s32i
// LLVM-CIR-LABEL: define dso_local i32 @call_wide_char(
// LLVM-CIR-SAME: i64 %[[P:[0-9a-zA-Z._]+]], ptr noundef byval(%struct.WideChar) align 16 %[[W:[0-9a-zA-Z._]+]])
-// LLVM-CIR: %{{[0-9a-zA-Z._]+}} = load %struct.WideChar, ptr %[[W]], align 16
+// LLVM-CIR: call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[LOCAL:[0-9a-zA-Z._]+]], ptr align 16 %[[W]], i64 32, i1 false)
+// LLVM-CIR: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.WideChar, align 16
+// LLVM-CIR-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[COPY]], ptr align 16 %[[LOCAL]], i64 32, i1 false)
// LLVM-CIR: %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT: %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.WideChar, align 16
-// LLVM-CIR-NEXT: store %struct.WideChar %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 16
// LLVM-CIR-NEXT: %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], ptr noundef byval(%struct.WideChar) align 16 %[[COPY]])
// LLVM-OGCG-LABEL: define dso_local i32 @call_wide_char(
diff --git a/clang/test/CIR/CodeGen/call.c b/clang/test/CIR/CodeGen/call.c
index ef67f8704b256..9a9c0c75bd4b3 100644
--- a/clang/test/CIR/CodeGen/call.c
+++ b/clang/test/CIR/CodeGen/call.c
@@ -72,15 +72,15 @@ void f7(void) {
}
// CIR-LABEL: cir.func{{.*}} @f7(){{.*}} {
-// CIR: %[[B:.+]] = cir.load align(4) %{{.+}} : !cir.ptr<!rec_Big>, !rec_Big
+// CIR: %[[B:.+]] = cir.alloca "b" align(4) : !cir.ptr<!rec_Big>
// CIR-NEXT: %[[SLOT:.+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR-NEXT: cir.store %[[B]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+// CIR-NEXT: cir.copy %[[B]] align(4) to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
// CIR-NEXT: cir.call @f5(%[[SLOT]]) : (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
// LLVM-LABEL: define{{.*}} void @f7(){{.*}} {
-// LLVM: %[[B:.+]] = load %struct.Big, ptr %{{.+}}, align 4
+// LLVM: %[[B:.+]] = alloca %struct.Big, align 4
// LLVM-NEXT: %[[SLOT:.+]] = alloca %struct.Big, align 8
-// LLVM-NEXT: store %struct.Big %[[B]], ptr %[[SLOT]], align 4
+// LLVM-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 4 %[[B]], i64 40, i1 false)
// LLVM-NEXT: call void @f5(ptr noundef byval(%struct.Big) align 8 %[[SLOT]])
// OGCG-LABEL: define{{.*}} void @f7() #0 {
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
index 90322dc342568..ddf53f8666c5b 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
@@ -231,8 +231,8 @@ module attributes {
// CHECK-NEXT: %[[V:.*]] = cir.load %[[M]] : !cir.ptr<!s64i>, !s64i
// CHECK-NEXT: cir.return %[[V]] : !s64i
- // byval with the same CIRGen spill keeps a local copy: load the byval
- // pointer at entry and store into the param-slot alloca.
+ // byval with the same CIRGen spill keeps a local copy: the byval memory is
+ // copied into the param-slot alloca.
cir.func @takes_big_byval_field(%arg0: !rec_Big) -> !s64i
attributes { test_classify = #byval_arg } {
%0 = cir.alloca "arg0" align(8) init : !cir.ptr<!rec_Big>
@@ -244,9 +244,8 @@ module attributes {
// CHECK: cir.func{{.*}} @takes_big_byval_field(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: llvm.byval = !rec_Big, llvm.noundef
- // CHECK: %[[LOADED:.*]] = cir.load %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
// CHECK: %[[SLOT:.*]] = cir.alloca "arg0" align(8) init : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[LOADED]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %[[SLOT]] : !cir.ptr<!rec_Big>
// CHECK: %[[M:.*]] = cir.get_member %[[SLOT]][0] {name = "a"} : !cir.ptr<!rec_Big> -> !cir.ptr<!s64i>
// CHECK: %[[V:.*]] = cir.load %[[M]] : !cir.ptr<!s64i>, !s64i
// CHECK: cir.return %[[V]] : !s64i
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
index eca64c5975ea1..b0805e8e60cee 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
@@ -75,7 +75,7 @@ module attributes {
// CHECK: cir.call %[[ICAST]](%arg1) : (!cir.ptr<!cir.func<(!s32i) -> !s32i>>, !s32i) -> !s32i
// Indirect call with a by-value (byval) struct argument: the argument is
- // spilled to a stack slot and the callee pointer is bitcast to the coerced
+ // copied to a stack slot and the callee pointer is bitcast to the coerced
// signature before the call is rebuilt.
cir.func @call_byval(%fp: !cir.ptr<!cir.func<(!rec_Big) -> !s64i>>) -> !s64i {
%0 = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
@@ -86,8 +86,9 @@ module attributes {
}
// CHECK: cir.func{{.*}} @call_byval(%arg0: !cir.ptr<!cir.func<(!rec_Big) -> !s64i>>)
+ // CHECK: %[[B:.*]] = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
// CHECK: %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %{{.+}}, %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[B]] to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
// CHECK: %[[CAST:.*]] = cir.cast bitcast %arg0 : !cir.ptr<!cir.func<(!rec_Big) -> !s64i>> -> !cir.ptr<!cir.func<(!cir.ptr<!rec_Big>) -> !s64i>>
// CHECK: cir.call %[[CAST]](%[[SLOT]]) : (!cir.ptr<!cir.func<(!cir.ptr<!rec_Big>) -> !s64i>>, !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s64i
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
index 4019d1394bf79..46fae7daa6476 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
@@ -546,10 +546,9 @@ module attributes {
// CHECK: cir.func{{.*}} @unloaded_param_to_byval(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
// CHECK-NOT: cir.load
// CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
// CHECK: cir.call @takes_big_byval(%[[COPY]]) :
// CHECK-SAME: (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
@@ -632,11 +631,10 @@ module attributes {
// CHECK: cir.func{{.*}} @unloaded_param_to_two_calls(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
// CHECK-NOT: cir.load
- // CHECK: cir.call @takes_big_non_byval(%[[PTR]]) :
// CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
+ // CHECK: cir.call @takes_big_non_byval(%[[PTR]]) :
// CHECK: cir.call @takes_big_byval(%[[COPY]]) :
cir.func private @takes_big_non_byval(%arg0: !rec_Big)
@@ -720,11 +718,10 @@ module attributes {
// CHECK: cir.func{{.*}} @unloaded_param_read_before_overwrite(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
+ // CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
// CHECK: %[[ZERO:.*]] = cir.const #cir.zero : !rec_Big
// CHECK: cir.store %[[ZERO]], %[[PTR]] : !rec_Big, !cir.ptr<!rec_Big>
- // CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
// CHECK: cir.call @takes_big_byval(%[[COPY]]) :
cir.func private @takes_big_byval(%arg0: !rec_Big)
@@ -856,8 +853,7 @@ module attributes {
// CHECK: cir.func{{.*}} @unloaded_param_over_overaligned_slot(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
- // CHECK: cir.store %[[VAL]], %{{.*}} : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %{{.*}} align(8) : !cir.ptr<!rec_Big>
cir.func private @takes_big_byval(%arg0: !rec_Big)
attributes { test_classify = #byval_arg }
@@ -996,11 +992,9 @@ module attributes {
// CHECK: cir.func{{.*}} @unspilled_param_to_byval(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK-NOT: cir.alloca
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
- // CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
- // CHECK: cir.call @takes_big_byval(%[[COPY]]) :
+ // CHECK-NEXT: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+ // CHECK-NEXT: cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
+ // CHECK-NEXT: cir.call @takes_big_byval(%[[COPY]]) :
}
@@ -1038,10 +1032,9 @@ module attributes {
// CHECK: cir.func{{.*}} @unloaded_param_read_in_scope(%[[PTR:.*]]: !cir.ptr<!rec_Big>
// CHECK-SAME: {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
+ // CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
// CHECK: cir.scope {
- // CHECK: %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
// CHECK: cir.call @takes_big_byval(%[[COPY]]) :
cir.func private @takes_big_byval(%arg0: !rec_Big)
diff --git a/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir b/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
index c843ff0026604..0f6c7d9baa6ae 100644
--- a/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
@@ -20,8 +20,7 @@ module attributes {
} {
// A 24-byte struct does not fit in registers: passed byval. The body reads
- // a field, so the load the rewriter inserts at entry (turning the byval
- // pointer back into the record value) is visible.
+ // a field, so the copy of the byval memory into the local is visible.
cir.func @take_big(%arg0: !rec_Big) -> !s64i {
%0 = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
cir.store %arg0, %0 : !rec_Big, !cir.ptr<!rec_Big>
@@ -31,9 +30,8 @@ module attributes {
}
// CHECK: cir.func{{.*}} @take_big(%arg0: !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s64i
- // CHECK: %[[VAL:.*]] = cir.load %arg0 : !cir.ptr<!rec_Big>, !rec_Big
// CHECK: %[[LOCAL:.*]] = cir.alloca "b" {{.*}} : !cir.ptr<!rec_Big>
- // CHECK: cir.store %[[VAL]], %[[LOCAL]] : !rec_Big, !cir.ptr<!rec_Big>
+ // CHECK: cir.copy %arg0 align(8) to %[[LOCAL]] : !cir.ptr<!rec_Big>
// CHECK: %[[FLD:.*]] = cir.get_member %[[LOCAL]][2] {name = "c"} : !cir.ptr<!rec_Big> -> !cir.ptr<!s64i>
// CHECK: %{{.*}} = cir.load %[[FLD]] : !cir.ptr<!s64i>, !s64i
diff --git a/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir b/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
index 74a10e34f43c4..ccfdf813fe4ca 100644
--- a/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
@@ -681,7 +681,7 @@ module attributes {
// CHECK: cir.func{{.*}} @call_nua_big_empty_int(%arg0: !cir.ptr<!rec_UNuaBigEmptyInt> {llvm.align = 32 : i64, llvm.byval = !rec_UNuaBigEmptyInt, llvm.noundef})
// CHECK: %[[BYVAL:.*]] = cir.alloca "byval" align(32) : !cir.ptr<!rec_UNuaBigEmptyInt>
- // CHECK: cir.store %{{.*}}, %[[BYVAL]] : !rec_UNuaBigEmptyInt, !cir.ptr<!rec_UNuaBigEmptyInt>
+ // CHECK: cir.copy %arg0 align(32) to %[[BYVAL]] : !cir.ptr<!rec_UNuaBigEmptyInt>
// CHECK: cir.call @take_nua_big_empty_int(%[[BYVAL]]) : (!cir.ptr<!rec_UNuaBigEmptyInt> {llvm.align = 32 : i64, llvm.byval = !rec_UNuaBigEmptyInt, llvm.noundef}) -> ()
cir.func @ret_nua_empty_floats(%arg0: !rec_UNuaEmptyFloats) -> !rec_UNuaEmptyFloats {
@@ -792,8 +792,7 @@ module attributes {
}
// CHECK: cir.func{{.*}} @ret_big(%arg0: !cir.ptr<!rec_UBig> {llvm.align = 1 : i64, llvm.dead_on_unwind, llvm.noalias, llvm.sret = !rec_UBig, llvm.writable}, %arg1: !cir.ptr<!rec_UBig> {llvm.align = 8 : i64, llvm.byval = !rec_UBig, llvm.noundef})
- // CHECK: %[[VAL:.*]] = cir.load %arg1 : !cir.ptr<!rec_UBig>, !rec_UBig
- // CHECK: cir.store %[[VAL]], %arg0 : !rec_UBig, !cir.ptr<!rec_UBig>
+ // CHECK: cir.copy %arg1 align(8) to %arg0 : !cir.ptr<!rec_UBig>
// The call site coerces the union argument the same way the callee expects
// it.
>From 097d47d50b38ee4d5812cb8609de56710f719f2b Mon Sep 17 00:00:00 2001
From: Bruno Cardoso Lopes <bruno.cardoso at gmail.com>
Date: Mon, 28 Sep 2026 18:40:33 -0700
Subject: [PATCH 2/2] [CIR] Check classic codegen in the byval-union test
---
.../call-conv-lowering-x86_64-byval-union.c | 23 +++++++++++++++++++
1 file changed, 23 insertions(+)
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
index 5000f72080b52..b6047902a0c0c 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
@@ -2,6 +2,8 @@
// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-llvm %s -o %t-cir.ll
// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=OGCG --input-file=%t.ll %s
// A union passed byval is copied as bytes. Its LLVM type is built from one
// member, so a record load/store would drop bytes that are padding in that
@@ -39,3 +41,24 @@ void pass(void) {
// LLVM: %[[SLOT:.+]] = alloca %union.U, align 8
// LLVM-NEXT: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 8 %[[U]], i64 24, i1 false)
// LLVM-NEXT: call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[SLOT]])
+
+// FIXME: Classic codegen copies neither the parameter nor the argument, so
+// LLVM and OGCG differ at -O0 (not at -O2, where the copies go away). To match:
+// - callee: use the byval pointer as the parameter's storage instead of
+// copying it into the spill slot, as CallConvLowering already does for
+// non-byval indirect parameters; copy only when the slot needs more
+// alignment than the byval pointer has.
+// - caller: pass the address the argument was loaded from straight to the
+// byval call (byval already gives the callee its own copy), when it is
+// aligned enough and nothing writes that memory between the load and call.
+
+// OGCG-LABEL: define {{.*}}i32 @get_b(
+// OGCG-SAME: ptr noundef byval(%union.U) align 8 %[[U:.+]])
+// OGCG-NOT: call void @llvm.memcpy
+// OGCG: %[[B:.+]] = getelementptr inbounds nuw %struct.anon.0, ptr %[[U]], i32 0, i32 1
+// OGCG-NEXT: load i32, ptr %[[B]], align 4
+
+// OGCG-LABEL: define {{.*}}void @pass()
+// OGCG: %[[U:.+]] = alloca %union.U, align 8
+// OGCG-NOT: call void @llvm.memcpy
+// OGCG: call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[U]])
More information about the cfe-commits
mailing list