[clang] [CIR] Copy byval records as bytes in CallConvLowering (PR #227170)

Bruno Cardoso Lopes via cfe-commits cfe-commits at lists.llvm.org
Mon Sep 28 18:50:38 PDT 2026


https://github.com/bcardosolopes updated https://github.com/llvm/llvm-project/pull/227170

>From 4975c573e1a5af46ab472135e7d1fbed89fd779c Mon Sep 17 00:00:00 2001
From: Bruno Cardoso Lopes <bruno.cardoso at gmail.com>
Date: Mon, 28 Sep 2026 15:52:50 -0700
Subject: [PATCH 1/2] [CIR] Copy byval records as bytes in CallConvLowering

A byval record argument was copied with a whole-record cir.load/cir.store,
both in the callee (into the parameter's slot) and at call sites (into the
byval temporary). A record's LLVM type is built from one member for a union,
so bytes that are padding in that member but data in another were dropped.
A clang built with -fclangir miscompiled itself this way: TemplateArgument,
passed by value, lost its integer bit width.

Use cir.copy instead when the value comes straight from memory.
---
 .../TargetLowering/CIRABIRewriteContext.cpp   | 43 ++++++++++++++++++-
 .../call-conv-lowering-x86_64-byval-union.c   | 41 ++++++++++++++++++
 .../call-conv-lowering-x86_64-non-byval.cpp   |  6 +--
 .../call-conv-lowering-x86_64-variadic.c      | 32 +++++++-------
 clang/test/CIR/CodeGen/call.c                 |  8 ++--
 .../abi-lowering/indirect-byval.cir           |  7 ++-
 .../Transforms/abi-lowering/indirect-call.cir |  5 ++-
 .../indirect-non-byval-forward-param.cir      | 29 +++++--------
 .../abi-lowering/x86_64-struct-indirect.cir   |  6 +--
 .../Transforms/abi-lowering/x86_64-union.cir  |  5 +--
 10 files changed, 125 insertions(+), 57 deletions(-)
 create mode 100644 clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c

diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
index a7639abb62466..0ed6ad22797a4 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
@@ -800,8 +800,28 @@ void insertArgCoercion(
         if (destAlloca)
           pendingParamSlots.emplace_back(destAlloca, blockArg);
       } else {
-        // byval: load the incoming pointer so the body sees a T value (and
-        // any CIRGen param-slot store becomes a local copy of that value).
+        // byval: the body gets a local copy of the incoming value.  When a
+        // record's only use is CIRGen's param spill, copy the bytes into the
+        // slot: a loaded record value carries only the fields of the LLVM
+        // type, which for a union is one member's, so bytes that are padding
+        // in that member but data in another would be lost.
+        auto spill =
+            isa<cir::RecordType>(blockArg.getType()) && blockArg.hasOneUse()
+                ? dyn_cast<cir::StoreOp>(*blockArg.user_begin())
+                : cir::StoreOp();
+        if (spill && spill.getValue() == blockArg) {
+          builder.setInsertionPoint(spill);
+          cir::CopyOp::create(
+              builder, spill.getLoc(), spill.getAddr(), blockArg,
+              /*dst_alignment=*/{},
+              builder.getI64IntegerAttr(ac.indirectAlign.value()));
+          spill->erase();
+          blockArg.setType(ptrTy);
+          ++blockArgIdx;
+          continue;
+        }
+
+        // Otherwise load the incoming pointer so the body sees a T value.
         blockArg.setType(ptrTy);
 
         builder.setInsertionPointToStart(&entry);
@@ -1549,6 +1569,25 @@ CIRABIRewriteContext::rewriteCallSite(mlir::Operation *callOp,
         continue;
       }
       auto ptrTy = cir::PointerType::get(arg.getType());
+      // A record loaded from memory is copied as bytes, at the load so it
+      // sees the same value: a record value carries only the fields of its
+      // LLVM type, which for a union is one member's.
+      cir::LoadOp srcLoad = isa<cir::RecordType>(arg.getType())
+                                ? maybeGetSimpleLoad(arg)
+                                : cir::LoadOp();
+      if (srcLoad && srcLoad.getAddr().getType() == ptrTy) {
+        mlir::OpBuilder::InsertionGuard guard(builder);
+        builder.setInsertionPointAfter(srcLoad);
+        auto slot = cir::AllocaOp::create(
+            builder, call.getLoc(), ptrTy, builder.getStringAttr("byval"),
+            builder.getI64IntegerAttr(ac.indirectAlign.value()));
+        cir::CopyOp::create(builder, call.getLoc(), slot, srcLoad.getAddr(),
+                            builder.getI64IntegerAttr(ac.indirectAlign.value()),
+                            srcLoad.getAlignmentAttr());
+        newArgs.push_back(slot);
+        deadRecordLoads.push_back(srcLoad);
+        continue;
+      }
       auto slot = cir::AllocaOp::create(
           builder, call.getLoc(), ptrTy, builder.getStringAttr("byval"),
           builder.getI64IntegerAttr(ac.indirectAlign.value()));
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
new file mode 100644
index 0000000000000..5000f72080b52
--- /dev/null
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
@@ -0,0 +1,41 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+
+// A union passed byval is copied as bytes. Its LLVM type is built from one
+// member, so a record load/store would drop bytes that are padding in that
+// member but data in another: bytes 4-7, y.b, here.
+union U {
+  struct { int a; void *p, *q; } x;
+  struct { int a, b; void *p, *q; } y;
+};
+
+int get_b(union U u) { return u.y.b; }
+
+void pass(void) {
+  union U u;
+  u.y.b = 42;
+  get_b(u);
+}
+
+// CIR-LABEL: cir.func {{.*}}@get_b(%arg0: !cir.ptr<!rec_U> {llvm.align = 8 : i64, llvm.byval = !rec_U, llvm.noundef}
+// CIR:         %[[U:.*]] = cir.alloca "u" align(8) init : !cir.ptr<!rec_U>
+// CIR:         cir.copy %arg0 align(8) to %[[U]] : !cir.ptr<!rec_U>
+// CIR-NOT:     cir.load {{.*}} !rec_U
+
+// CIR-LABEL: cir.func {{.*}}@pass()
+// CIR:         %[[U:.*]] = cir.alloca "u" align(8) : !cir.ptr<!rec_U>
+// CIR:         %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_U>
+// CIR-NEXT:    cir.copy %[[U]] align(8) to %[[SLOT]] align(8) : !cir.ptr<!rec_U>
+// CIR-NEXT:    cir.call @get_b(%[[SLOT]])
+
+// LLVM-LABEL: define {{.*}}i32 @get_b(ptr noundef byval(%union.U) align 8 %0)
+// LLVM:         call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.+}}, ptr align 8 %0, i64 24, i1 false)
+// LLVM-NOT:     load %union.U
+
+// LLVM-LABEL: define {{.*}}void @pass()
+// LLVM:         %[[U:.+]] = alloca %union.U, align 8
+// LLVM:         %[[SLOT:.+]] = alloca %union.U, align 8
+// LLVM-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 8 %[[U]], i64 24, i1 false)
+// LLVM-NEXT:    call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[SLOT]])
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
index 374ad6683d66d..58c683798f50a 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-non-byval.cpp
@@ -100,15 +100,13 @@ void callByval() {
 
 // CIR-LABEL: cir.func {{.*}}@_Z9callByvalv
 // CIR:         %[[TMP:.*]] = cir.alloca "agg.tmp0" align(8) : !cir.ptr<!rec_Big>
-// CIR:         %[[V:.*]] = cir.load align(8) %[[TMP]] : !cir.ptr<!rec_Big>, !rec_Big
 // CIR:         %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR:         cir.store %[[V]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+// CIR:         cir.copy %[[TMP]] align(8) to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
 // CIR:         cir.call @_Z9takeByval3Big(%[[SLOT]]) : (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
 
 // LLVM-LABEL: define dso_local void @_Z9callByvalv()
 // LLVM:         call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[TMP:[^,]+]], ptr align 8 %{{[^,]+}}, i64 32, i1 false)
-// LLVM-CIR:     %[[V:.*]] = load %struct.Big, ptr %[[TMP]], align 8
-// LLVM-CIR:     store %struct.Big %[[V]], ptr %[[SLOT:.*]], align 8
+// LLVM-CIR:     call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT:[^,]+]], ptr align 8 %[[TMP]], i64 32, i1 false)
 // LLVM-CIR:     call void @_Z9takeByval3Big(ptr noundef byval(%struct.Big) align 8 %[[SLOT]])
 // OGCG:         call void @_Z9takeByval3Big(ptr noundef byval(%struct.Big) align 8 %[[TMP]])
 
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
index c7d3d4aa8c335..9505f01c2f3be 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-variadic.c
@@ -67,19 +67,19 @@ int call_small(Pair2 p, Pair16 q) { return vf(p, q); }
 int call_big(Pair2 p, Big b) { return vf(p, b); }
 
 // CIR-LABEL: cir.func {{.*}}@call_big(%arg0: !u64i loc({{.+}}), %arg1: !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef} loc({{.+}})) -> !s32i
-// CIR:         %{{[0-9]+}} = cir.load %arg1 : !cir.ptr<!rec_Big>, !rec_Big
+// CIR:         cir.copy %arg1 align(8) to %[[LOCAL:[0-9]+]] : !cir.ptr<!rec_Big>
+// CIR:         %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+// CIR-NEXT:    cir.copy %[[LOCAL]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
 // CIR:         %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT:    %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR-NEXT:    cir.store %{{[0-9]+}}, %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
 // CIR-NEXT:    %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[COPY]]) : (!u64i, !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s32i
 
 // CIR copies the incoming byval slot before forwarding it.  OGCG does not.
 // LLVM-CIR-LABEL: define dso_local i32 @call_big(
 // LLVM-CIR-SAME:    i64 %[[P:[0-9a-zA-Z._]+]], ptr noundef byval(%struct.Big) align 8 %[[B:[0-9a-zA-Z._]+]])
-// LLVM-CIR:       %{{[0-9a-zA-Z._]+}} = load %struct.Big, ptr %[[B]], align 8
+// LLVM-CIR:       call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[LOCAL:[0-9a-zA-Z._]+]], ptr align 8 %[[B]], i64 32, i1 false)
+// LLVM-CIR:       %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Big, align 8
+// LLVM-CIR-NEXT:  call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[COPY]], ptr align 8 %[[LOCAL]], i64 32, i1 false)
 // LLVM-CIR:       %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT:  %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Big, align 8
-// LLVM-CIR-NEXT:  store %struct.Big %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 8
 // LLVM-CIR-NEXT:  %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], ptr noundef byval(%struct.Big) align 8 %[[COPY]])
 
 // LLVM-OGCG-LABEL: define dso_local i32 @call_big(
@@ -103,9 +103,9 @@ int call_exhausted(Pair2 p, long a, long b, long c, long d, Pair16 q) {
 // CIR:         %[[BV:[0-9]+]] = cir.load align(8) %[[BS]] : !cir.ptr<!s64i>, !s64i
 // CIR:         %[[CV:[0-9]+]] = cir.load align(8) %[[CS]] : !cir.ptr<!s64i>, !s64i
 // CIR:         %[[DV:[0-9]+]] = cir.load align(8) %[[DS]] : !cir.ptr<!s64i>, !s64i
+// CIR:         %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Pair16>
+// CIR-NEXT:    cir.copy %{{[0-9]+}} align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Pair16>
 // CIR:         %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT:    %[[COPY:[0-9]+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Pair16>
-// CIR-NEXT:    cir.store %{{[0-9]+}}, %[[COPY]] : !rec_Pair16, !cir.ptr<!rec_Pair16>
 // CIR-NEXT:    %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[AV]], %[[BV]], %[[CV]], %[[DV]], %[[COPY]]) : (!u64i, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !s64i {llvm.noundef}, !cir.ptr<!rec_Pair16> {llvm.align = 8 : i64, llvm.byval = !rec_Pair16, llvm.noundef}) -> !s32i
 
 // LLVM-CIR-LABEL: define dso_local i32 @call_exhausted(
@@ -120,9 +120,9 @@ int call_exhausted(Pair2 p, long a, long b, long c, long d, Pair16 q) {
 // LLVM:         %[[BV:[0-9a-zA-Z._]+]] = load i64, ptr %[[BS]], align 8
 // LLVM:         %[[CV:[0-9a-zA-Z._]+]] = load i64, ptr %[[CS]], align 8
 // LLVM:         %[[DV:[0-9a-zA-Z._]+]] = load i64, ptr %[[DS]], align 8
+// LLVM-CIR:       %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Pair16, align 8
+// LLVM-CIR-NEXT:  call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[COPY]], ptr align 8 %{{[0-9a-zA-Z._]+}}, i64 16, i1 false)
 // LLVM-CIR:       %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT:  %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.Pair16, align 8
-// LLVM-CIR-NEXT:  store %struct.Pair16 %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 8
 // LLVM-CIR-NEXT:  %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], i64 noundef %[[AV]], i64 noundef %[[BV]], i64 noundef %[[CV]], i64 noundef %[[DV]], ptr noundef byval(%struct.Pair16) align 8 %[[COPY]])
 // LLVM-OGCG:      %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 4
 // LLVM-OGCG-NEXT: %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], i64 noundef %[[AV]], i64 noundef %[[BV]], i64 noundef %[[CV]], i64 noundef %[[DV]], ptr noundef byval(%struct.Pair16) align 8 %[[Q]])
@@ -178,18 +178,18 @@ int call_wide(Pair2 p, Wide w) { return vf(p, w); }
 int call_wide_char(Pair2 p, WideChar w) { return vf(p, w); }
 
 // CIR-LABEL: cir.func {{.*}}@call_wide_char(%arg0: !u64i loc({{.+}}), %arg1: !cir.ptr<!rec_WideChar> {llvm.align = 16 : i64, llvm.byval = !rec_WideChar, llvm.noundef} loc({{.+}})) -> !s32i
-// CIR:         %{{[0-9]+}} = cir.load %arg1 : !cir.ptr<!rec_WideChar>, !rec_WideChar
+// CIR:         cir.copy %arg1 align(16) to %[[LOCAL:[0-9]+]] : !cir.ptr<!rec_WideChar>
+// CIR:         %[[COPY:[0-9]+]] = cir.alloca "byval" align(16) : !cir.ptr<!rec_WideChar>
+// CIR-NEXT:    cir.copy %[[LOCAL]] align(16) to %[[COPY]] align(16) : !cir.ptr<!rec_WideChar>
 // CIR:         %[[PV:[0-9]+]] = cir.load %{{[0-9]+}} : !cir.ptr<!u64i>, !u64i
-// CIR-NEXT:    %[[COPY:[0-9]+]] = cir.alloca "byval" align(16) : !cir.ptr<!rec_WideChar>
-// CIR-NEXT:    cir.store %{{[0-9]+}}, %[[COPY]] : !rec_WideChar, !cir.ptr<!rec_WideChar>
 // CIR-NEXT:    %{{[0-9]+}} = cir.call @vf(%[[PV]], %[[COPY]]) : (!u64i, !cir.ptr<!rec_WideChar> {llvm.align = 16 : i64, llvm.byval = !rec_WideChar, llvm.noundef}) -> !s32i
 
 // LLVM-CIR-LABEL: define dso_local i32 @call_wide_char(
 // LLVM-CIR-SAME:    i64 %[[P:[0-9a-zA-Z._]+]], ptr noundef byval(%struct.WideChar) align 16 %[[W:[0-9a-zA-Z._]+]])
-// LLVM-CIR:       %{{[0-9a-zA-Z._]+}} = load %struct.WideChar, ptr %[[W]], align 16
+// LLVM-CIR:       call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[LOCAL:[0-9a-zA-Z._]+]], ptr align 16 %[[W]], i64 32, i1 false)
+// LLVM-CIR:       %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.WideChar, align 16
+// LLVM-CIR-NEXT:  call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[COPY]], ptr align 16 %[[LOCAL]], i64 32, i1 false)
 // LLVM-CIR:       %[[PV:[0-9a-zA-Z._]+]] = load i64, ptr %{{[0-9a-zA-Z._]+}}, align 8
-// LLVM-CIR-NEXT:  %[[COPY:[0-9a-zA-Z._]+]] = alloca %struct.WideChar, align 16
-// LLVM-CIR-NEXT:  store %struct.WideChar %{{[0-9a-zA-Z._]+}}, ptr %[[COPY]], align 16
 // LLVM-CIR-NEXT:  %{{[0-9a-zA-Z._]+}} = call i32 (i64, ...) @vf(i64 %[[PV]], ptr noundef byval(%struct.WideChar) align 16 %[[COPY]])
 
 // LLVM-OGCG-LABEL: define dso_local i32 @call_wide_char(
diff --git a/clang/test/CIR/CodeGen/call.c b/clang/test/CIR/CodeGen/call.c
index ef67f8704b256..9a9c0c75bd4b3 100644
--- a/clang/test/CIR/CodeGen/call.c
+++ b/clang/test/CIR/CodeGen/call.c
@@ -72,15 +72,15 @@ void f7(void) {
 }
 
 // CIR-LABEL: cir.func{{.*}} @f7(){{.*}} {
-// CIR:         %[[B:.+]] = cir.load align(4) %{{.+}} : !cir.ptr<!rec_Big>, !rec_Big
+// CIR:         %[[B:.+]] = cir.alloca "b" align(4) : !cir.ptr<!rec_Big>
 // CIR-NEXT:    %[[SLOT:.+]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-// CIR-NEXT:    cir.store %[[B]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+// CIR-NEXT:    cir.copy %[[B]] align(4) to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
 // CIR-NEXT:    cir.call @f5(%[[SLOT]]) : (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
 
 // LLVM-LABEL: define{{.*}} void @f7(){{.*}} {
-// LLVM:         %[[B:.+]] = load %struct.Big, ptr %{{.+}}, align 4
+// LLVM:         %[[B:.+]] = alloca %struct.Big, align 4
 // LLVM-NEXT:    %[[SLOT:.+]] = alloca %struct.Big, align 8
-// LLVM-NEXT:    store %struct.Big %[[B]], ptr %[[SLOT]], align 4
+// LLVM-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 4 %[[B]], i64 40, i1 false)
 // LLVM-NEXT:    call void @f5(ptr noundef byval(%struct.Big) align 8 %[[SLOT]])
 
 // OGCG-LABEL: define{{.*}} void @f7() #0 {
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
index 90322dc342568..ddf53f8666c5b 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-byval.cir
@@ -231,8 +231,8 @@ module attributes {
   // CHECK-NEXT:   %[[V:.*]] = cir.load %[[M]] : !cir.ptr<!s64i>, !s64i
   // CHECK-NEXT:   cir.return %[[V]] : !s64i
 
-  // byval with the same CIRGen spill keeps a local copy: load the byval
-  // pointer at entry and store into the param-slot alloca.
+  // byval with the same CIRGen spill keeps a local copy: the byval memory is
+  // copied into the param-slot alloca.
   cir.func @takes_big_byval_field(%arg0: !rec_Big) -> !s64i
       attributes { test_classify = #byval_arg } {
     %0 = cir.alloca "arg0" align(8) init : !cir.ptr<!rec_Big>
@@ -244,9 +244,8 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @takes_big_byval_field(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     llvm.byval = !rec_Big, llvm.noundef
-  // CHECK:        %[[LOADED:.*]] = cir.load %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
   // CHECK:        %[[SLOT:.*]] = cir.alloca "arg0" align(8) init : !cir.ptr<!rec_Big>
-  // CHECK:        cir.store %[[LOADED]], %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %[[SLOT]] : !cir.ptr<!rec_Big>
   // CHECK:        %[[M:.*]] = cir.get_member %[[SLOT]][0] {name = "a"} : !cir.ptr<!rec_Big> -> !cir.ptr<!s64i>
   // CHECK:        %[[V:.*]] = cir.load %[[M]] : !cir.ptr<!s64i>, !s64i
   // CHECK:        cir.return %[[V]] : !s64i
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
index eca64c5975ea1..b0805e8e60cee 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-call.cir
@@ -75,7 +75,7 @@ module attributes {
   // CHECK:   cir.call %[[ICAST]](%arg1) : (!cir.ptr<!cir.func<(!s32i) -> !s32i>>, !s32i) -> !s32i
 
   // Indirect call with a by-value (byval) struct argument: the argument is
-  // spilled to a stack slot and the callee pointer is bitcast to the coerced
+  // copied to a stack slot and the callee pointer is bitcast to the coerced
   // signature before the call is rebuilt.
   cir.func @call_byval(%fp: !cir.ptr<!cir.func<(!rec_Big) -> !s64i>>) -> !s64i {
     %0 = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
@@ -86,8 +86,9 @@ module attributes {
   }
 
   // CHECK: cir.func{{.*}} @call_byval(%arg0: !cir.ptr<!cir.func<(!rec_Big) -> !s64i>>)
+  // CHECK:   %[[B:.*]] = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
   // CHECK:   %[[SLOT:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:   cir.store %{{.+}}, %[[SLOT]] : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:   cir.copy %[[B]] to %[[SLOT]] align(8) : !cir.ptr<!rec_Big>
   // CHECK:   %[[CAST:.*]] = cir.cast bitcast %arg0 : !cir.ptr<!cir.func<(!rec_Big) -> !s64i>> -> !cir.ptr<!cir.func<(!cir.ptr<!rec_Big>) -> !s64i>>
   // CHECK:   cir.call %[[CAST]](%[[SLOT]]) : (!cir.ptr<!cir.func<(!cir.ptr<!rec_Big>) -> !s64i>>, !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s64i
 
diff --git a/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir b/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
index 4019d1394bf79..46fae7daa6476 100644
--- a/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/indirect-non-byval-forward-param.cir
@@ -546,10 +546,9 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unloaded_param_to_byval(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
   // CHECK-NOT:    cir.load
   // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:        cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
   // CHECK:        cir.call @takes_big_byval(%[[COPY]]) :
   // CHECK-SAME:     (!cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> ()
 
@@ -632,11 +631,10 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unloaded_param_to_two_calls(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
   // CHECK-NOT:    cir.load
-  // CHECK:        cir.call @takes_big_non_byval(%[[PTR]]) :
   // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:        cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
+  // CHECK:        cir.call @takes_big_non_byval(%[[PTR]]) :
   // CHECK:        cir.call @takes_big_byval(%[[COPY]]) :
 
   cir.func private @takes_big_non_byval(%arg0: !rec_Big)
@@ -720,11 +718,10 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unloaded_param_read_before_overwrite(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
+  // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
   // CHECK:        %[[ZERO:.*]] = cir.const #cir.zero : !rec_Big
   // CHECK:        cir.store %[[ZERO]], %[[PTR]] : !rec_Big, !cir.ptr<!rec_Big>
-  // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:        cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
   // CHECK:        cir.call @takes_big_byval(%[[COPY]]) :
 
   cir.func private @takes_big_byval(%arg0: !rec_Big)
@@ -856,8 +853,7 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unloaded_param_over_overaligned_slot(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
-  // CHECK:        cir.store %[[VAL]], %{{.*}} : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %{{.*}} align(8) : !cir.ptr<!rec_Big>
 
   cir.func private @takes_big_byval(%arg0: !rec_Big)
       attributes { test_classify = #byval_arg }
@@ -996,11 +992,9 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unspilled_param_to_byval(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK-NOT:    cir.alloca
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
-  // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:        cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
-  // CHECK:        cir.call @takes_big_byval(%[[COPY]]) :
+  // CHECK-NEXT:   %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+  // CHECK-NEXT:   cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
+  // CHECK-NEXT:   cir.call @takes_big_byval(%[[COPY]]) :
 
 }
 
@@ -1038,10 +1032,9 @@ module attributes {
 
   // CHECK:      cir.func{{.*}} @unloaded_param_read_in_scope(%[[PTR:.*]]: !cir.ptr<!rec_Big>
   // CHECK-SAME:     {llvm.align = 8 : i64, llvm.dereferenceable = 32 : i64, llvm.nofreeobj, llvm.noundef})
-  // CHECK:        %[[VAL:.*]] = cir.load align(8) %[[PTR]] : !cir.ptr<!rec_Big>, !rec_Big
+  // CHECK:        %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
+  // CHECK:        cir.copy %[[PTR]] align(8) to %[[COPY]] align(8) : !cir.ptr<!rec_Big>
   // CHECK:        cir.scope {
-  // CHECK:          %[[COPY:.*]] = cir.alloca "byval" align(8) : !cir.ptr<!rec_Big>
-  // CHECK:          cir.store %[[VAL]], %[[COPY]] : !rec_Big, !cir.ptr<!rec_Big>
   // CHECK:          cir.call @takes_big_byval(%[[COPY]]) :
 
   cir.func private @takes_big_byval(%arg0: !rec_Big)
diff --git a/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir b/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
index c843ff0026604..0f6c7d9baa6ae 100644
--- a/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/x86_64-struct-indirect.cir
@@ -20,8 +20,7 @@ module attributes {
 } {
 
   // A 24-byte struct does not fit in registers: passed byval.  The body reads
-  // a field, so the load the rewriter inserts at entry (turning the byval
-  // pointer back into the record value) is visible.
+  // a field, so the copy of the byval memory into the local is visible.
   cir.func @take_big(%arg0: !rec_Big) -> !s64i {
     %0 = cir.alloca "b" align(8) : !cir.ptr<!rec_Big>
     cir.store %arg0, %0 : !rec_Big, !cir.ptr<!rec_Big>
@@ -31,9 +30,8 @@ module attributes {
   }
 
   // CHECK: cir.func{{.*}} @take_big(%arg0: !cir.ptr<!rec_Big> {llvm.align = 8 : i64, llvm.byval = !rec_Big, llvm.noundef}) -> !s64i
-  // CHECK:   %[[VAL:.*]] = cir.load %arg0 : !cir.ptr<!rec_Big>, !rec_Big
   // CHECK:   %[[LOCAL:.*]] = cir.alloca "b" {{.*}} : !cir.ptr<!rec_Big>
-  // CHECK:   cir.store %[[VAL]], %[[LOCAL]] : !rec_Big, !cir.ptr<!rec_Big>
+  // CHECK:   cir.copy %arg0 align(8) to %[[LOCAL]] : !cir.ptr<!rec_Big>
   // CHECK:   %[[FLD:.*]] = cir.get_member %[[LOCAL]][2] {name = "c"} : !cir.ptr<!rec_Big> -> !cir.ptr<!s64i>
   // CHECK:   %{{.*}} = cir.load %[[FLD]] : !cir.ptr<!s64i>, !s64i
 
diff --git a/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir b/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
index 74a10e34f43c4..ccfdf813fe4ca 100644
--- a/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
+++ b/clang/test/CIR/Transforms/abi-lowering/x86_64-union.cir
@@ -681,7 +681,7 @@ module attributes {
 
   // CHECK: cir.func{{.*}} @call_nua_big_empty_int(%arg0: !cir.ptr<!rec_UNuaBigEmptyInt> {llvm.align = 32 : i64, llvm.byval = !rec_UNuaBigEmptyInt, llvm.noundef})
   // CHECK:   %[[BYVAL:.*]] = cir.alloca "byval" align(32) : !cir.ptr<!rec_UNuaBigEmptyInt>
-  // CHECK:   cir.store %{{.*}}, %[[BYVAL]] : !rec_UNuaBigEmptyInt, !cir.ptr<!rec_UNuaBigEmptyInt>
+  // CHECK:   cir.copy %arg0 align(32) to %[[BYVAL]] : !cir.ptr<!rec_UNuaBigEmptyInt>
   // CHECK:   cir.call @take_nua_big_empty_int(%[[BYVAL]]) : (!cir.ptr<!rec_UNuaBigEmptyInt> {llvm.align = 32 : i64, llvm.byval = !rec_UNuaBigEmptyInt, llvm.noundef}) -> ()
 
   cir.func @ret_nua_empty_floats(%arg0: !rec_UNuaEmptyFloats) -> !rec_UNuaEmptyFloats {
@@ -792,8 +792,7 @@ module attributes {
   }
 
   // CHECK: cir.func{{.*}} @ret_big(%arg0: !cir.ptr<!rec_UBig> {llvm.align = 1 : i64, llvm.dead_on_unwind, llvm.noalias, llvm.sret = !rec_UBig, llvm.writable}, %arg1: !cir.ptr<!rec_UBig> {llvm.align = 8 : i64, llvm.byval = !rec_UBig, llvm.noundef})
-  // CHECK:   %[[VAL:.*]] = cir.load %arg1 : !cir.ptr<!rec_UBig>, !rec_UBig
-  // CHECK:   cir.store %[[VAL]], %arg0 : !rec_UBig, !cir.ptr<!rec_UBig>
+  // CHECK:   cir.copy %arg1 align(8) to %arg0 : !cir.ptr<!rec_UBig>
 
   // The call site coerces the union argument the same way the callee expects
   // it.

>From 097d47d50b38ee4d5812cb8609de56710f719f2b Mon Sep 17 00:00:00 2001
From: Bruno Cardoso Lopes <bruno.cardoso at gmail.com>
Date: Mon, 28 Sep 2026 18:40:33 -0700
Subject: [PATCH 2/2] [CIR] Check classic codegen in the byval-union test

---
 .../call-conv-lowering-x86_64-byval-union.c   | 23 +++++++++++++++++++
 1 file changed, 23 insertions(+)

diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
index 5000f72080b52..b6047902a0c0c 100644
--- a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-byval-union.c
@@ -2,6 +2,8 @@
 // RUN: FileCheck --check-prefix=CIR --input-file=%t.cir %s
 // RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -fclangir -emit-llvm %s -o %t-cir.ll
 // RUN: FileCheck --check-prefix=LLVM --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=OGCG --input-file=%t.ll %s
 
 // A union passed byval is copied as bytes. Its LLVM type is built from one
 // member, so a record load/store would drop bytes that are padding in that
@@ -39,3 +41,24 @@ void pass(void) {
 // LLVM:         %[[SLOT:.+]] = alloca %union.U, align 8
 // LLVM-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr align 8 %[[SLOT]], ptr align 8 %[[U]], i64 24, i1 false)
 // LLVM-NEXT:    call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[SLOT]])
+
+// FIXME: Classic codegen copies neither the parameter nor the argument, so
+// LLVM and OGCG differ at -O0 (not at -O2, where the copies go away). To match:
+// - callee: use the byval pointer as the parameter's storage instead of
+//   copying it into the spill slot, as CallConvLowering already does for
+//   non-byval indirect parameters; copy only when the slot needs more
+//   alignment than the byval pointer has.
+// - caller: pass the address the argument was loaded from straight to the
+//   byval call (byval already gives the callee its own copy), when it is
+//   aligned enough and nothing writes that memory between the load and call.
+
+// OGCG-LABEL: define {{.*}}i32 @get_b(
+// OGCG-SAME:    ptr noundef byval(%union.U) align 8 %[[U:.+]])
+// OGCG-NOT:     call void @llvm.memcpy
+// OGCG:         %[[B:.+]] = getelementptr inbounds nuw %struct.anon.0, ptr %[[U]], i32 0, i32 1
+// OGCG-NEXT:    load i32, ptr %[[B]], align 4
+
+// OGCG-LABEL: define {{.*}}void @pass()
+// OGCG:         %[[U:.+]] = alloca %union.U, align 8
+// OGCG-NOT:     call void @llvm.memcpy
+// OGCG:         call i32 @get_b(ptr noundef byval(%union.U) align 8 %[[U]])



More information about the cfe-commits mailing list