[clang] [llvm] [mlir] [CIR] Expand callee-side va_arg on x86_64 (PR #222420)
Adam Smith via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 16 12:10:07 PDT 2026
https://github.com/adams381 updated https://github.com/llvm/llvm-project/pull/222420
>From 3f152804b6d6a27a33f8deecf38a1a62a5830876 Mon Sep 17 00:00:00 2001
From: Adam Smith <adams at nvidia.com>
Date: Wed, 9 Sep 2026 11:21:09 -0700
Subject: [PATCH 1/3] [CIR] Expand callee-side va_arg on x86_64
CallConvLowering now classifies every `cir.va_arg` on its own, as an
unnamed argument with a full register budget, and expands it in place.
A fetch that fits the budget reads from the saved register area, and
everything else reads the overflow area and bumps the cursor. An
aggregate, an x87 long double, and an `__int128` all lower correctly
now.
A non-trivially-copyable type still reports NYI.
Assisted-by: Cursor / claude-opus-5
---
.../Transforms/CallConvLoweringPass.cpp | 39 +++
.../TargetLowering/CIRABIRewriteContext.cpp | 297 +++++++++++++++++
.../TargetLowering/CIRABIRewriteContext.h | 6 +
...conv-lowering-x86_64-vaarg-non-pod-nyi.cpp | 20 ++
clang/test/CIR/CodeGen/complex.cpp | 38 ++-
clang/test/CIR/CodeGen/var-arg-aggregate.c | 312 ++++++++++++++++--
clang/test/CIR/CodeGen/var-arg-empty.c | 55 +++
clang/test/CIR/CodeGen/var-arg-int128.c | 65 ++++
clang/test/CIR/CodeGen/var-arg-long-double.c | 51 +++
clang/test/CIR/CodeGen/var-arg-vector.c | 56 ++++
clang/test/CIR/CodeGen/var_arg.c | 78 ++++-
mlir/include/mlir/ABI/ABIRewriteContext.h | 27 ++
12 files changed, 1013 insertions(+), 31 deletions(-)
create mode 100644 clang/test/CIR/CodeGen/call-conv-lowering-x86_64-vaarg-non-pod-nyi.cpp
create mode 100644 clang/test/CIR/CodeGen/var-arg-empty.c
create mode 100644 clang/test/CIR/CodeGen/var-arg-int128.c
create mode 100644 clang/test/CIR/CodeGen/var-arg-long-double.c
create mode 100644 clang/test/CIR/CodeGen/var-arg-vector.c
diff --git a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
index 7af53c0698885..8635666523a9a 100644
--- a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
@@ -712,6 +712,22 @@ classifyX86_64Function(cir::FuncOp func, const DataLayout &dl,
[&]() { return func.emitOpError(); });
}
+/// Classify the single type fetched by a `cir.va_arg` as an unnamed argument
+/// (RequiredArgs(0)) with a full register budget. Returns std::nullopt and
+/// emits an NYI error if the type is not one this bridge handles.
+static std::optional<ArgClassification> classifyX86_64VarArgType(
+ mlir::Type ty, MLIRContext *ctx, const DataLayout &dl,
+ mlir::abi::ABITypeMapper &typeMapper,
+ const llvm::abi::TargetInfo &targetInfo, ModuleOp modOp,
+ llvm::function_ref<mlir::InFlightDiagnostic()> emitError) {
+ std::optional<FunctionClassification> fc = classifyX86_64Signature(
+ cir::VoidType::get(ctx), mlir::TypeRange(ty), llvm::abi::RequiredArgs(0),
+ ctx, dl, typeMapper, targetInfo, modOp, emitError);
+ if (!fc)
+ return std::nullopt;
+ return fc->argInfos[0];
+}
+
/// Classify a call that passes arguments through an ellipsis. The callee's
/// own classification covers only its declared parameters, but an ellipsis
/// argument competes for the same argument registers as a declared one, so
@@ -1109,6 +1125,29 @@ void CallConvLoweringPass::runOnOperation() {
return;
}
}
+
+ // Fetches on targets other than x86-64 are left in place here and still
+ // lower to a generic vararg instruction that cannot select an aggregate or
+ // an x87 long double result.
+ if (isX86) {
+ SmallVector<cir::VAArgOp> vaArgs;
+ moduleOp.walk([&](cir::VAArgOp v) { vaArgs.push_back(v); });
+ for (cir::VAArgOp v : vaArgs) {
+ cir::FuncOp enclosing = v->getParentOfType<cir::FuncOp>();
+ std::optional<ArgClassification> ac = classifyX86_64VarArgType(
+ v.getType(), ctx, dl, *x86TypeMapper,
+ x86TargetFor(avxLevelFor(enclosing)), moduleOp,
+ [&]() { return v->emitOpError(); });
+ if (!ac) {
+ signalPassFailure();
+ return;
+ }
+ if (failed(rewriteCtx.rewriteVAArg(v.getOperation(), *ac, builder))) {
+ signalPassFailure();
+ return;
+ }
+ }
+ }
}
} // namespace
diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
index 918f3f17b9933..fb858718d0dd2 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
@@ -10,8 +10,11 @@
#include "mlir/Dialect/LLVMIR/LLVMDialect.h"
#include "mlir/IR/Builders.h"
#include "mlir/IR/Dominance.h"
+#include "clang/CIR/Dialect/Builder/CIRBaseBuilder.h"
#include "clang/CIR/Dialect/IR/CIRAttrs.h"
#include "clang/CIR/Dialect/IR/CIRTypes.h"
+#include "llvm/ADT/APFloat.h"
+#include <array>
#include <utility>
using namespace cir;
@@ -977,6 +980,48 @@ void rewriteIndirectReturnCall(cir::CallOp call,
call->erase();
}
+/// Whether \p ty is an eightbyte that travels in a vector register. An x87
+/// long double is floating-point but travels in neither register class, so it
+/// is excluded. A caller reads a false result as the integer class.
+bool isSSERegisterClass(mlir::Type ty) {
+ if (mlir::isa<cir::VectorType>(ty))
+ return true;
+ if (auto fp = mlir::dyn_cast<cir::FPTypeInterface>(ty))
+ return &fp.getFloatSemantics() != &llvm::APFloat::x87DoubleExtended();
+ return false;
+}
+
+/// The boundary the caller aligned \p ty to in the argument area. An
+/// alignment attribute can raise a record above what its members imply, and
+/// only the record-layout metadata carries that, so the member-derived value
+/// alone can be too small.
+uint64_t argumentAreaAlign(mlir::Type ty, mlir::ModuleOp modOp,
+ const mlir::DataLayout &dl) {
+ uint64_t align = dl.getTypeABIAlignment(ty);
+ if (auto recTy = mlir::dyn_cast<cir::RecordType>(ty))
+ if (auto layout = cir::tryGetRecordLayout(modOp, recTy.getName()))
+ align = std::max<uint64_t>(align, layout.getRecordAlign());
+ return align;
+}
+
+/// Whether an SSE-class type fits the single 16-byte vector-register slot a
+/// fetch can read from. A wider vector travels in memory no matter how many
+/// vector registers the target has, because a fetch has no declared parameter
+/// to pin it to one. The width has to be measured here rather than read off
+/// the classification, which spells a memory-class scalar the same way it
+/// spells one that really does travel in a register.
+bool fitsOneVectorSlot(mlir::Type ty, const mlir::DataLayout &dl) {
+ return dl.getTypeSize(ty).getFixedValue() <= 16;
+}
+
+/// Whether a scalar INTEGER-class type occupies two eightbytes rather than
+/// one, which is the case for any 128-bit container: `__int128`, or a
+/// narrower `_BitInt` widened to 128 bits.
+bool isWide128BitInt(mlir::Type ty) {
+ auto intTy = mlir::dyn_cast<cir::IntType>(ty);
+ return intTy && intTy.getWidth() == 128;
+}
+
} // namespace
mlir::LogicalResult CIRABIRewriteContext::rewriteFunctionDefinition(
@@ -1380,3 +1425,255 @@ void CIRABIRewriteContext::rewriteFunctionAddress(cir::GetGlobalOp addrOp,
cir::CastKind::bitcast, addrOp.getAddr());
addrOp.getAddr().replaceAllUsesExcept(bitcast.getResult(), bitcast);
}
+
+mlir::LogicalResult
+CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
+ const ArgClassification &ac,
+ mlir::OpBuilder &opBuilder) {
+ auto op = mlir::cast<cir::VAArgOp>(vaArgOp);
+ CIRBaseBuilderTy builder(opBuilder);
+ mlir::Location loc = op.getLoc();
+ mlir::Type resultTy = op.getType();
+ mlir::Value valist = op.getArgList();
+
+ auto reportNYI = [&](llvm::StringRef what) {
+ op->emitOpError() << "va_arg of " << what
+ << " not yet implemented in CallConvLowering";
+ return mlir::failure();
+ };
+
+ // An ignored type is passed in no register and no stack slot, so the fetch
+ // has nothing to read and must leave the va_list cursor where it found it,
+ // for the fetches that come after this one. The type holds no bytes, so
+ // the value the fetch produces carries no information either.
+ if (ac.kind == ArgKind::Ignore) {
+ builder.setInsertionPoint(op);
+ op.getResult().replaceAllUsesWith(builder.createDummyValue(
+ loc, resultTy,
+ clang::CharUnits::fromQuantity(dl.getTypeABIAlignment(resultTy))));
+ op->erase();
+ return mlir::success();
+ }
+
+ if (ac.kind == ArgKind::Indirect && !ac.byVal)
+ return reportNYI("a non-trivially-copyable type");
+
+ // How many eightbytes of each register class the fetched type occupies.
+ // Zero of both means the type travels in memory and is read straight from
+ // the overflow area.
+ unsigned neededInt = 0, neededSse = 0;
+ // Which coerced-pair element (0 = low eightbyte, 1 = high) is SSE rather
+ // than INTEGER class. Only meaningful when isRegPair is set.
+ std::array<bool, 2> pairIsSse = {false, false};
+ bool isRegPair = false;
+
+ if (ac.kind == ArgKind::Extend) {
+ neededInt = 1;
+ } else if (ac.kind == ArgKind::Direct) {
+ if (mlir::Type coerced = ac.coercedType) {
+ if (auto pairTy = mlir::dyn_cast<cir::RecordType>(coerced)) {
+ assert(pairTy.getNumElements() == 2 &&
+ "a register-class coercion spans at most two eightbytes");
+ for (auto [i, memberTy] : llvm::enumerate(pairTy.getMembers())) {
+ if (isSSERegisterClass(memberTy)) {
+ pairIsSse[i] = true;
+ ++neededSse;
+ } else {
+ ++neededInt;
+ }
+ }
+ isRegPair = true;
+ } else if (isSSERegisterClass(coerced)) {
+ if (fitsOneVectorSlot(coerced, dl))
+ neededSse = 1;
+ } else {
+ neededInt = isWide128BitInt(coerced) ? 2 : 1;
+ }
+ } else if (auto fp = mlir::dyn_cast<cir::FPTypeInterface>(resultTy)) {
+ // An uncoerced scalar float is SSE, unless it is x87 long double,
+ // which the SysV ABI always classifies MEMORY. That is recognized
+ // from the semantics here, because the classification spells a
+ // memory-class scalar the same way it spells a register one.
+ if (&fp.getFloatSemantics() != &llvm::APFloat::x87DoubleExtended())
+ neededSse = 1;
+ } else if (mlir::isa<cir::VectorType>(resultTy)) {
+ if (fitsOneVectorSlot(resultTy, dl))
+ neededSse = 1;
+ } else if (mlir::isa<cir::IntType, cir::PointerType, cir::BoolType>(
+ resultTy)) {
+ neededInt = isWide128BitInt(resultTy) ? 2 : 1;
+ } else {
+ return reportNYI("this type");
+ }
+ } else if (ac.kind != ArgKind::Indirect) {
+ return reportNYI("a type with an argument classification this fetch does "
+ "not model");
+ }
+ // Indirect is the one remaining kind, and it takes no register.
+
+ auto vaListRecTy = mlir::cast<cir::RecordType>(
+ mlir::cast<cir::PointerType>(valist.getType()).getPointee());
+ llvm::ArrayRef<mlir::Type> vaFields = vaListRecTy.getMembers();
+ assert(vaFields.size() == 4 &&
+ "expected the four-field gp_offset / fp_offset / overflow_arg_area / "
+ "reg_save_area argument cursor");
+ cir::IntType byteTy = builder.getUIntNTy(8);
+ cir::PointerType bytePtrTy = builder.getPointerTo(byteTy);
+
+ builder.setInsertionPoint(op);
+
+ // Reading the overflow area also advances the cursor past the argument.
+ auto buildMemAddr = [&](CIRBaseBuilderTy &b) -> mlir::Value {
+ mlir::Value overflowP = b.createGetMember(loc, b.getPointerTo(vaFields[2]),
+ valist, "overflow_arg_area", 2);
+ mlir::Value overflow = b.createLoad(loc, overflowP);
+ mlir::Value bytePtr = b.createPtrBitcast(overflow, byteTy);
+ uint64_t tyAlign = argumentAreaAlign(resultTy, module, dl);
+ if (tyAlign > 8) {
+ mlir::Type wordTy = b.getUIntNTy(64);
+ mlir::Value asInt = cir::CastOp::create(
+ b, loc, wordTy, cir::CastKind::ptr_to_int, bytePtr);
+ mlir::Value bumped = b.createNUWAdd(
+ loc, asInt, b.getConstantInt(loc, wordTy, tyAlign - 1));
+ mlir::Value rounded = b.createAnd(
+ loc, bumped, b.getConstantInt(loc, wordTy, ~(tyAlign - 1)));
+ bytePtr = cir::CastOp::create(b, loc, bytePtrTy,
+ cir::CastKind::int_to_ptr, rounded);
+ }
+ uint64_t tySize = dl.getTypeSize(resultTy).getFixedValue();
+ uint64_t stride = (tySize + 7) & ~UINT64_C(7);
+ mlir::Value strideVal = b.getSignedInt(loc, stride, 32);
+ mlir::Value next = b.createPtrStride(loc, bytePtr, strideVal);
+ b.createStore(loc, next, overflowP);
+ return bytePtr;
+ };
+
+ mlir::Value addr;
+ if (neededInt == 0 && neededSse == 0) {
+ addr = buildMemAddr(builder);
+ } else {
+ mlir::Value gpOffsetP, fpOffsetP, gpOffset, fpOffset, inRegs;
+ if (neededInt) {
+ gpOffsetP = builder.createGetMember(
+ loc, builder.getPointerTo(vaFields[0]), valist, "gp_offset", 0);
+ gpOffset = builder.createLoad(loc, gpOffsetP);
+ mlir::Value limit =
+ builder.getConstantInt(loc, gpOffset.getType(), 48 - neededInt * 8);
+ inRegs = builder.createCompare(loc, cir::CmpOpKind::le, gpOffset, limit);
+ }
+ if (neededSse) {
+ fpOffsetP = builder.createGetMember(
+ loc, builder.getPointerTo(vaFields[1]), valist, "fp_offset", 1);
+ fpOffset = builder.createLoad(loc, fpOffsetP);
+ mlir::Value limit =
+ builder.getConstantInt(loc, fpOffset.getType(), 176 - neededSse * 16);
+ mlir::Value fitsInFp =
+ builder.createCompare(loc, cir::CmpOpKind::le, fpOffset, limit);
+ inRegs =
+ inRegs ? builder.createLogicalAnd(loc, inRegs, fitsInFp) : fitsInFp;
+ }
+
+ // A two-eightbyte pair that is purely INTEGER class is contiguous in the
+ // register-save area (GP slots are 8-byte packed), so one load of the
+ // coerced pair type at gp_offset reads both eightbytes. A pure SSE pair
+ // or a mixed pair is not contiguous (SSE slots are 16-byte spaced, and a
+ // mixed pair's GP and SSE halves live in disjoint areas of the
+ // register-save area), so each half is copied into a temp laid out as the
+ // coerced pair before the whole value is reinterpreted as resultTy.
+ bool pairNeedsReassembly = isRegPair && neededSse != 0;
+ mlir::Value regPairTemp;
+ if (pairNeedsReassembly)
+ regPairTemp = builder.createAlloca(
+ loc, builder.getPointerTo(ac.coercedType), "vaarg.reg",
+ clang::CharUnits::fromQuantity(
+ dl.getTypeABIAlignment(ac.coercedType)));
+
+ addr =
+ cir::TernaryOp::create(
+ builder, loc, inRegs,
+ /*trueBuilder=*/
+ [&](mlir::OpBuilder &ob, mlir::Location l) {
+ CIRBaseBuilderTy b(ob);
+ mlir::Value regSaveArea = b.createLoad(
+ l, b.createGetMember(l, b.getPointerTo(vaFields[3]), valist,
+ "reg_save_area", 3));
+ regSaveArea = b.createPtrBitcast(regSaveArea, byteTy);
+
+ mlir::Value regAddr;
+ if (pairNeedsReassembly) {
+ auto pairTy = mlir::cast<cir::RecordType>(ac.coercedType);
+ // Same-class eightbytes sit regSize bytes apart within
+ // their own area (8 for GP, 16 for SSE), so track how
+ // many of each class came before.
+ unsigned seenOfClass[2] = {0, 0};
+ for (unsigned i = 0; i < 2; ++i) {
+ bool isSse = pairIsSse[i];
+ mlir::Value base = isSse ? fpOffset : gpOffset;
+ unsigned regSize = isSse ? 16 : 8;
+ unsigned prior = seenOfClass[isSse]++;
+ mlir::Value off = base;
+ if (prior)
+ off = b.createAdd(
+ l, base,
+ b.getConstantInt(l, base.getType(), prior * regSize));
+ mlir::Value src = b.createPtrStride(l, regSaveArea, off);
+ mlir::Type elemTy = pairTy.getElementType(i);
+ mlir::Value val =
+ b.createLoad(l, b.createPtrBitcast(src, elemTy));
+ b.createStore(l, val,
+ b.createGetMember(l, b.getPointerTo(elemTy),
+ regPairTemp, "", i));
+ }
+ regAddr = b.createPtrBitcast(regPairTemp, byteTy);
+ } else {
+ mlir::Value off = neededSse ? fpOffset : gpOffset;
+ regAddr = b.createPtrStride(l, regSaveArea, off);
+ // A GP register slot is only 8-byte aligned, so a
+ // wider-aligned GP-class fetch (e.g. __int128, align 16)
+ // is copied through a temp. SSE slots are already
+ // 16-byte aligned and need no copy.
+ uint64_t resultAlign = argumentAreaAlign(resultTy, module, dl);
+ if (!neededSse && resultAlign > 8) {
+ mlir::Value temp = b.createAlloca(
+ l, b.getPointerTo(resultTy), "vaarg.reg",
+ clang::CharUnits::fromQuantity(resultAlign));
+ // The slot this reads from is only 8-byte aligned, so the
+ // load cannot claim the wider alignment the temp has.
+ mlir::Value val = b.createAlignedLoad(
+ l, b.createPtrBitcast(regAddr, resultTy), 8);
+ b.createStore(l, val, temp);
+ regAddr = b.createPtrBitcast(temp, byteTy);
+ }
+ }
+
+ if (neededInt)
+ b.createStore(
+ l,
+ b.createAdd(
+ l, gpOffset,
+ b.getConstantInt(l, gpOffset.getType(), neededInt * 8)),
+ gpOffsetP);
+ if (neededSse)
+ b.createStore(
+ l,
+ b.createAdd(l, fpOffset,
+ b.getConstantInt(l, fpOffset.getType(),
+ neededSse * 16)),
+ fpOffsetP);
+
+ cir::YieldOp::create(b, l, regAddr);
+ },
+ /*falseBuilder=*/
+ [&](mlir::OpBuilder &ob, mlir::Location l) {
+ CIRBaseBuilderTy b(ob);
+ cir::YieldOp::create(b, l, buildMemAddr(b));
+ })
+ .getResult();
+ }
+
+ mlir::Value result =
+ builder.createLoad(loc, builder.createPtrBitcast(addr, resultTy));
+ op.getResult().replaceAllUsesWith(result);
+ op->erase();
+ return mlir::success();
+}
diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.h b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.h
index 705036fc2f620..e88face5dbfa9 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.h
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.h
@@ -50,6 +50,12 @@ class CIRABIRewriteContext : public mlir::abi::ABIRewriteContext {
const mlir::abi::FunctionClassification &fc,
mlir::OpBuilder &builder) override;
+ /// Expand a `cir.va_arg` into the x86-64 SysV register-save-area /
+ /// overflow-area sequence.
+ mlir::LogicalResult rewriteVAArg(mlir::Operation *vaArgOp,
+ const mlir::abi::ArgClassification &ac,
+ mlir::OpBuilder &builder) override;
+
/// Retype \p addrOp, which holds the address of \p funcOp, to the signature
/// funcOp was rewritten to, and cast it back so existing uses keep the type
/// they were built for. A no-op when the ABI left funcOp's type alone.
diff --git a/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-vaarg-non-pod-nyi.cpp b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-vaarg-non-pod-nyi.cpp
new file mode 100644
index 0000000000000..2d151a8e63c1f
--- /dev/null
+++ b/clang/test/CIR/CodeGen/call-conv-lowering-x86_64-vaarg-non-pod-nyi.cpp
@@ -0,0 +1,20 @@
+// RUN: not %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -Wno-non-pod-varargs -fclangir -emit-llvm %s -o - 2>&1 | FileCheck %s
+
+// A non-trivially-copyable class is passed as a reference to caller-owned
+// storage, which the fetch cannot yet read. Reaching it needs
+// -Wno-non-pod-varargs.
+struct NonTrivial {
+ int a;
+ NonTrivial(const NonTrivial &);
+ ~NonTrivial();
+};
+
+NonTrivial take_non_pod(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ NonTrivial res = __builtin_va_arg(args, NonTrivial);
+ __builtin_va_end(args);
+ return res;
+}
+
+// CHECK: error: 'cir.va_arg' op va_arg of a non-trivially-copyable type not yet implemented in CallConvLowering
diff --git a/clang/test/CIR/CodeGen/complex.cpp b/clang/test/CIR/CodeGen/complex.cpp
index abf70f89fa7e7..b287eab052cf9 100644
--- a/clang/test/CIR/CodeGen/complex.cpp
+++ b/clang/test/CIR/CodeGen/complex.cpp
@@ -936,22 +936,48 @@ void foo33(__builtin_va_list a) {
float _Complex b = __builtin_va_arg(a, float _Complex);
}
+// `_Complex float` coerces to `<2 x float>`, one SSE eightbyte.
// CIR: %[[A_ADDR:.*]] = cir.alloca "a" {{.*}} init : !cir.ptr<!cir.ptr<!rec___va_list_tag>>
// CIR: %[[B_ADDR:.*]] = cir.alloca "b" {{.*}} init : !cir.ptr<!cir.complex<!cir.float>>
// CIR: cir.store %[[ARG_0:.*]], %[[A_ADDR]] : !cir.ptr<!rec___va_list_tag>, !cir.ptr<!cir.ptr<!rec___va_list_tag>>
// CIR: %[[VA_TAG:.*]] = cir.load{{.*}} %[[A_ADDR]] : !cir.ptr<!cir.ptr<!rec___va_list_tag>>, !cir.ptr<!rec___va_list_tag>
-// CIR: %[[COMPLEX:.*]] = cir.va_arg %[[VA_TAG]] : (!cir.ptr<!rec___va_list_tag>) -> !cir.complex<!cir.float>
+// CIR: %[[FP_OFFSET_P:.*]] = cir.get_member %[[VA_TAG]][1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.*]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[FP_LIMIT:.*]] = cir.const #cir.int<160> : !u32i
+// CIR: %[[FITS_FP:.*]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[COMPLEX_ADDR:.*]] = cir.ternary(%[[FITS_FP]], true {
+// CIR: %[[REG_ADDR:.*]] = cir.ptr_stride %{{.*}}, %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[FP_BUMP:.*]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[FP_NEXT:.*]] = cir.add %[[FP_OFFSET]], %[[FP_BUMP]] : !u32i
+// CIR: cir.store %[[FP_NEXT]], %[[FP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: cir.yield %{{.*}} : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[COMPLEX_ADDR_B:.*]] = cir.cast bitcast %[[COMPLEX_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.complex<!cir.float>>
+// CIR: %[[COMPLEX:.*]] = cir.load %[[COMPLEX_ADDR_B]] : !cir.ptr<!cir.complex<!cir.float>>, !cir.complex<!cir.float>
// CIR: cir.store{{.*}} %[[COMPLEX]], %[[B_ADDR]] : !cir.complex<!cir.float>, !cir.ptr<!cir.complex<!cir.float>>
// LLVM: %[[A_ADDR:.*]] = alloca ptr, align 8
// LLVM: %[[B_ADDR:.*]] = alloca { float, float }, align 4
// LLVM: store ptr %[[ARG_0:.*]], ptr %[[A_ADDR]], align 8
// LLVM: %[[TMP_A:.*]] = load ptr, ptr %[[A_ADDR]], align 8
-// LLVM: %[[COMPLEX:.*]] = va_arg ptr %[[TMP_A]], { float, float }
-// LLVM: store { float, float } %[[COMPLEX]], ptr %[[B_ADDR]], align 4
-
-// TODO(CIR): the difference between the CIR LLVM and OGCG is because the lack of calling convention lowering,
-// Test will be updated when that is implemented
+// LLVM: %[[FP_OFFSET_P:.*]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %[[TMP_A]], i32 0, i32 1
+// LLVM: %[[FP_OFFSET:.*]] = load i32, ptr %[[FP_OFFSET_P]], align 4
+// LLVM: %[[FITS_FP:.*]] = icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: br i1 %[[FITS_FP]], label %[[REG_BB:.*]], label %[[MEM_BB:.*]]
+// LLVM: [[REG_BB]]:
+// LLVM: %[[REG_SAVE:.*]] = load ptr, ptr %{{.*}}, align 8
+// LLVM: %[[COMPLEX:.*]] = getelementptr i8, ptr %[[REG_SAVE]], i64 %{{.*}}
+// LLVM: %[[FP_NEXT:.*]] = add i32 %[[FP_OFFSET]], 16
+// LLVM: store i32 %[[FP_NEXT]], ptr %[[FP_OFFSET_P]], align 4
+// LLVM: br label %[[END_BB:.*]]
+// LLVM: [[MEM_BB]]:
+// LLVM: br label %[[END_BB]]
+// LLVM: [[END_BB]]:
+// LLVM: %[[COMPLEX_ADDR:.*]] = phi ptr [ %{{.*}}, %[[MEM_BB]] ], [ %[[COMPLEX]], %[[REG_BB]] ]
+// LLVM: %[[COMPLEX_V:.*]] = load { float, float }, ptr %[[COMPLEX_ADDR]], align 4
+// LLVM: store { float, float } %[[COMPLEX_V]], ptr %[[B_ADDR]], align 4
// OGCG: %[[A_ADDR:.*]] = alloca ptr, align 8
// OGCG: %[[B_ADDR:.*]] = alloca { float, float }, align 4
diff --git a/clang/test/CIR/CodeGen/var-arg-aggregate.c b/clang/test/CIR/CodeGen/var-arg-aggregate.c
index 94da8e2b80ad9..e3ca73d10102b 100644
--- a/clang/test/CIR/CodeGen/var-arg-aggregate.c
+++ b/clang/test/CIR/CodeGen/var-arg-aggregate.c
@@ -1,9 +1,9 @@
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
-// RUN: FileCheck --input-file=%t-cir.ll %s -check-prefix=LLVM
+// RUN: FileCheck --input-file=%t-cir.ll %s --check-prefixes=LLVM,LLVMCIR
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
-// RUN: FileCheck --input-file=%t.ll %s -check-prefix=OGCG
+// RUN: FileCheck --input-file=%t.ll %s --check-prefixes=LLVM,OGCG
struct Bar {
float f1;
@@ -11,7 +11,7 @@ struct Bar {
unsigned u;
};
-struct Bar varargs_aggregate(int count, ...) {
+struct Bar varargs_aggregate_mixed_pair(int count, ...) {
__builtin_va_list args;
__builtin_va_start(args, count);
struct Bar res = __builtin_va_arg(args, struct Bar);
@@ -19,36 +19,310 @@ struct Bar varargs_aggregate(int count, ...) {
return res;
}
+struct LongPair {
+ long a;
+ long b;
+};
+
+struct LongPair varargs_aggregate_gp_pair(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct LongPair res = __builtin_va_arg(args, struct LongPair);
+ __builtin_va_end(args);
+ return res;
+}
-// CIR-LABEL: cir.func {{.*}} @varargs_aggregate(
-// CIR-SAME: -> !rec_anon_struct
-// CIR: %[[COERCE:.+]] = cir.alloca "coerce" {{.*}} : !cir.ptr<!rec_anon_struct>
+struct DoublePair {
+ double a;
+ double b;
+};
+
+struct DoublePair varargs_aggregate_sse_pair(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct DoublePair res = __builtin_va_arg(args, struct DoublePair);
+ __builtin_va_end(args);
+ return res;
+}
+
+struct Big {
+ long a;
+ long b;
+ long c;
+};
+
+struct Big varargs_aggregate_memory(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct Big res = __builtin_va_arg(args, struct Big);
+ __builtin_va_end(args);
+ return res;
+}
+
+// Mixed pair: one SSE eightbyte, one INTEGER eightbyte, reassembled via a temp.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_mixed_pair(
+// CIR-SAME: -> !rec_anon_struct{{[0-9]*}}
+// CIR: %[[COERCE:.+]] = cir.alloca "coerce" {{.*}} : !cir.ptr<!rec_anon_struct{{[0-9]*}}>
// CIR: %[[RET_ADDR:.+]] = cir.alloca "__retval" {{.*}} init : !cir.ptr<!rec_Bar>
// CIR: %[[VAAREA:.+]] = cir.alloca "args" {{.*}} : !cir.ptr<!cir.array<!rec___va_list_tag x 1>>
// CIR: %[[TMP_ADDR:.+]] = cir.alloca "vaarg.tmp" {{.*}} : !cir.ptr<!rec_Bar>
// CIR: %[[VA_PTR0:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_start %[[VA_PTR0]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[VA_PTR1:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
-// CIR: %[[VA_ARG:.+]] = cir.va_arg %[[VA_PTR1]] : (!cir.ptr<!rec___va_list_tag>) -> !rec_Bar
-// CIR: cir.store{{.*}} %[[VA_ARG]], %[[TMP_ADDR]] : !rec_Bar, !cir.ptr<!rec_Bar>
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %[[VA_PTR1]][0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %[[VA_PTR1]][1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
+// CIR: %[[FITS_FP:.+]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[IN_REGS:.+]] = cir.select if %[[FITS_GP]] then %[[FITS_FP]] else %{{.+}} : (!cir.bool, !cir.bool, !cir.bool) -> !cir.bool
+// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!rec_anon_struct{{[0-9]*}}>
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[IN_REGS]], true {
+// CIR: %[[REG_SAVE_P:.+]] = cir.get_member %[[VA_PTR1]][3] {name = "reg_save_area"}
+// CIR: %[[REG_SAVE:.+]] = cir.load %[[REG_SAVE_P]]
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[FP_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[FP_ADDR_V:.+]] = cir.cast bitcast %[[FP_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.vector<2 x !cir.float>>
+// CIR: %[[SSE_VAL:.+]] = cir.load %[[FP_ADDR_V]] : !cir.ptr<!cir.vector<2 x !cir.float>>, !cir.vector<2 x !cir.float>
+// CIR: %[[TMP_SSE:.+]] = cir.get_member %[[REG_TMP]][0] {{.*}} -> !cir.ptr<!cir.vector<2 x !cir.float>>
+// CIR: cir.store %[[SSE_VAL]], %[[TMP_SSE]] : !cir.vector<2 x !cir.float>, !cir.ptr<!cir.vector<2 x !cir.float>>
+// CIR: %[[GP_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[GP_ADDR_V:.+]] = cir.cast bitcast %[[GP_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!u32i>
+// CIR: %[[INT_VAL:.+]] = cir.load %[[GP_ADDR_V]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[TMP_INT:.+]] = cir.get_member %[[REG_TMP]][1] {{.*}} -> !cir.ptr<!u32i>
+// CIR: cir.store %[[INT_VAL]], %[[TMP_INT]] : !u32i, !cir.ptr<!u32i>
+// CIR: %[[REG_TMP_B:.+]] = cir.cast bitcast %[[REG_TMP]] : !cir.ptr<!rec_anon_struct{{[0-9]*}}> -> !cir.ptr<!u8i>
+// CIR: %[[GP_BUMP:.+]] = cir.const #cir.int<8> : !u32i
+// CIR: %[[GP_NEXT:.+]] = cir.add %[[GP_OFFSET]], %[[GP_BUMP]] : !u32i
+// CIR: cir.store %[[GP_NEXT]], %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: %[[FP_BUMP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[FP_NEXT:.+]] = cir.add %[[FP_OFFSET]], %[[FP_BUMP]] : !u32i
+// CIR: cir.store %[[FP_NEXT]], %[[FP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_TMP_B]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %[[VA_PTR1]][2] {name = "overflow_arg_area"}
+// CIR: %[[OVERFLOW:.+]] = cir.load %[[OVERFLOW_P]]
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<16> : !s32i
+// CIR: %[[OVERFLOW_NEXT:.+]] = cir.ptr_stride %[[OVERFLOW_B]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[OVERFLOW_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: cir.yield %[[OVERFLOW_B]] : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_Bar>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!rec_Bar>, !rec_Bar
+// CIR: cir.store{{.*}} %[[VA_ARG_V]], %[[TMP_ADDR]] : !rec_Bar, !cir.ptr<!rec_Bar>
// CIR: cir.copy %[[TMP_ADDR]] align(4) to %[[RET_ADDR]] align(4) : !cir.ptr<!rec_Bar>
// CIR: %[[VA_PTR2:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_end %[[VA_PTR2]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[RETVAL:.+]] = cir.load{{.*}} %[[RET_ADDR]] : !cir.ptr<!rec_Bar>, !rec_Bar
-// CIR: %[[SLOT:.+]] = cir.cast bitcast %[[COERCE]] : !cir.ptr<!rec_anon_struct> -> !cir.ptr<!rec_Bar>
+// CIR: %[[SLOT:.+]] = cir.cast bitcast %[[COERCE]] : !cir.ptr<!rec_anon_struct{{[0-9]*}}> -> !cir.ptr<!rec_Bar>
// CIR: cir.store %[[RETVAL]], %[[SLOT]] : !rec_Bar, !cir.ptr<!rec_Bar>
-// CIR: %[[COERCED:.+]] = cir.load %[[COERCE]] : !cir.ptr<!rec_anon_struct>, !rec_anon_struct
-// CIR: cir.return %[[COERCED]] : !rec_anon_struct
+// CIR: %[[COERCED:.+]] = cir.load %[[COERCE]] : !cir.ptr<!rec_anon_struct{{[0-9]*}}>, !rec_anon_struct{{[0-9]*}}
+// CIR: cir.return %[[COERCED]] : !rec_anon_struct{{[0-9]*}}
+
+// GP pair: both eightbytes INTEGER, contiguous, so a single load suffices.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_gp_pair(
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<32> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
+// CIR: %[[REG_SAVE:.+]] = cir.load %{{.+}}
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[GP_BUMP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[GP_NEXT:.+]] = cir.add %[[GP_OFFSET]], %[[GP_BUMP]] : !u32i
+// CIR: cir.store %[[GP_NEXT]], %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: %[[OVERFLOW:.+]] = cir.load %{{.+}}
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<16> : !s32i
+// CIR: %[[OVERFLOW_NEXT:.+]] = cir.ptr_stride %[[OVERFLOW_B]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[OVERFLOW_B]] : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_LongPair>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!rec_LongPair>, !rec_LongPair
-// LLVM-LABEL: define dso_local { <2 x float>, i32 } @varargs_aggregate(i32 noundef %{{.*}}, ...)
+// SSE pair: both eightbytes SSE, 16 bytes apart, reassembled via a temp.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_sse_pair(
+// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<144> : !u32i
+// CIR: %[[FITS_FP:.+]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!rec_anon_struct{{[0-9]*}}>
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_FP]], true {
+// CIR: %[[REG_SAVE:.+]] = cir.load %{{.+}}
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[LO_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[LO_ADDR_V:.+]] = cir.cast bitcast %[[LO_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: %[[LO_VAL:.+]] = cir.load %[[LO_ADDR_V]] : !cir.ptr<!cir.double>, !cir.double
+// CIR: %[[TMP_LO:.+]] = cir.get_member %[[REG_TMP]][0] {{.*}} -> !cir.ptr<!cir.double>
+// CIR: cir.store %[[LO_VAL]], %[[TMP_LO]] : !cir.double, !cir.ptr<!cir.double>
+// CIR: %[[HI_BUMP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[HI_OFF:.+]] = cir.add %[[FP_OFFSET]], %[[HI_BUMP]] : !u32i
+// CIR: %[[HI_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[HI_OFF]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[HI_ADDR_V:.+]] = cir.cast bitcast %[[HI_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: %[[HI_VAL:.+]] = cir.load %[[HI_ADDR_V]] : !cir.ptr<!cir.double>, !cir.double
+// CIR: %[[TMP_HI:.+]] = cir.get_member %[[REG_TMP]][1] {{.*}} -> !cir.ptr<!cir.double>
+// CIR: cir.store %[[HI_VAL]], %[[TMP_HI]] : !cir.double, !cir.ptr<!cir.double>
+// CIR: %[[REG_TMP_B:.+]] = cir.cast bitcast %[[REG_TMP]] : !cir.ptr<!rec_anon_struct{{[0-9]*}}> -> !cir.ptr<!u8i>
+// CIR: %[[FP_BUMP:.+]] = cir.const #cir.int<32> : !u32i
+// CIR: %[[FP_NEXT:.+]] = cir.add %[[FP_OFFSET]], %[[FP_BUMP]] : !u32i
+// CIR: cir.store %[[FP_NEXT]], %[[FP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_TMP_B]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: %[[OVERFLOW:.+]] = cir.load %{{.+}}
+// CIR: cir.yield %{{.+}} : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_DoublePair>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!rec_DoublePair>, !rec_DoublePair
+
+// MEMORY class: three eightbytes never fit in registers.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_memory(
+// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %{{.+}}[2] {name = "overflow_arg_area"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!cir.ptr<!void>>
+// CIR: %[[OVERFLOW:.+]] = cir.load %[[OVERFLOW_P]] : !cir.ptr<!cir.ptr<!void>>, !cir.ptr<!void>
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<24> : !s32i
+// CIR: %[[OVERFLOW_NEXT:.+]] = cir.ptr_stride %[[OVERFLOW_B]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[OVERFLOW_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[OVERFLOW_B]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_Big>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!rec_Big>, !rec_Big
+
+// LLVM-LABEL: define dso_local { <2 x float>, i32 } @varargs_aggregate_mixed_pair(i32 noundef %{{.*}}, ...)
// LLVM: call void @llvm.va_start.p0(ptr %{{.*}})
-// LLVM: %[[VA_PTR1:.+]] = getelementptr %struct.__va_list_tag, ptr %{{.*}}, i32 0
-// LLVM: %[[VA_ARG:.+]] = va_arg ptr %[[VA_PTR1]], %struct.Bar
-// LLVM: store %struct.Bar %[[VA_ARG]], ptr %{{.*}}
+// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FITS_GP:.+]] = icmp ule i32 %[[GP_OFFSET]], 40
+// LLVM: %[[FP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 1
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FITS_FP:.+]] = icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: %[[IN_REGS:.+]] = and i1 %[[FITS_GP]], %[[FITS_FP]]
+// LLVMCIR: %[[REG_TMP:.+]] = alloca { <2 x float>, i32 }, align 8
+// LLVM: br i1 %[[IN_REGS]], label %[[REG_BB:.+]], label %[[MEM_BB:.+]]
+// LLVM: [[REG_BB]]:
+// LLVM: %[[REG_SAVE:.+]] = load ptr, ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[SSE_VAL:.+]] = load <2 x float>, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: store <2 x float> %[[SSE_VAL]], ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[INT_VAL:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: store i32 %[[INT_VAL]], ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[GP_NEXT:.+]] = add i32 %[[GP_OFFSET]], 8
+// LLVM: store i32 %[[GP_NEXT]], ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FP_NEXT:.+]] = add i32 %[[FP_OFFSET]], 16
+// LLVM: store i32 %[[FP_NEXT]], ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: br label %[[END_BB:.+]]
+// LLVM: [[MEM_BB]]:
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[OVERFLOW_NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 16
+// LLVM: store ptr %[[OVERFLOW_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVM: br label %[[END_BB]]
+// LLVM: [[END_BB]]:
+// The fetched value is materialized by loading the record on one side and by
+// copying it on the other, so only the address feeding it is shared.
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[OVERFLOW]], %[[MEM_BB]] ], [ %[[REG_TMP]], %[[REG_BB]] ]
+// LLVMCIR: %[[VA_ARG:.+]] = load %struct.Bar, ptr %[[ADDR]], align 4
+// LLVMCIR: store %struct.Bar %[[VA_ARG]], ptr %{{.*}}, align 4
+// OGCG: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[REG_BB]] ], [ %{{.*}}, %[[MEM_BB]] ]
// LLVM: call void @llvm.memcpy.p0.p0.i64(ptr align 4 %{{.*}}, ptr align 4 %{{.*}}, i64 12, i1 false)
-// OGCG-LABEL: define dso_local { <2 x float>, i32 } @varargs_aggregate
-// OGCG: call void @llvm.va_start.p0(ptr %{{.*}})
-// OGCG: %[[VAARG_ADDR:.+]] = phi ptr [ %{{.*}}, %vaarg.in_reg ], [ %{{.*}}, %vaarg.in_mem ]
-// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 4 %{{.*}}, ptr align 4 %[[VAARG_ADDR]], i64 12, i1 false)
+// LLVM-LABEL: define dso_local { i64, i64 } @varargs_aggregate_gp_pair(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FITS_GP:.+]] = icmp ule i32 %[[GP_OFFSET]], 32
+// LLVM: br i1 %[[FITS_GP]], label %[[REG_BB:.+]], label %[[MEM_BB:.+]]
+// LLVM: [[REG_BB]]:
+// LLVM: %[[REG_SAVE:.+]] = load ptr, ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[REG_SAVE]], i{{32|64}} %{{.+}}
+// LLVM: %[[GP_NEXT:.+]] = add i32 %[[GP_OFFSET]], 16
+// LLVM: store i32 %[[GP_NEXT]], ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: br label %[[END_BB:.+]]
+// LLVM: [[MEM_BB]]:
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[OVERFLOW_NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 16
+// LLVM: store ptr %[[OVERFLOW_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVM: br label %[[END_BB]]
+// LLVM: [[END_BB]]:
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[OVERFLOW]], %[[MEM_BB]] ], [ %[[REG_ADDR]], %[[REG_BB]] ]
+// LLVMCIR: %[[VA_ARG:.+]] = load %struct.LongPair, ptr %[[ADDR]], align 8
+// LLVMCIR: store %struct.LongPair %[[VA_ARG]], ptr %{{.*}}, align 8
+// OGCG: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[REG_BB]] ], [ %{{.*}}, %[[MEM_BB]] ]
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.*}}, ptr align 8 %[[ADDR]], i64 16, i1 false)
+
+// LLVM-LABEL: define dso_local { double, double } @varargs_aggregate_sse_pair(i32 noundef %{{.*}}, ...)
+// LLVM: %[[FP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 1
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FITS_FP:.+]] = icmp ule i32 %[[FP_OFFSET]], 144
+// LLVMCIR: %[[REG_TMP:.+]] = alloca { double, double }, align 8
+// LLVM: br i1 %[[FITS_FP]], label %[[REG_BB:.+]], label %[[MEM_BB:.+]]
+// LLVM: [[REG_BB]]:
+// LLVM: %[[REG_SAVE:.+]] = load ptr, ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[LO_VAL:.+]] = load double, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: store double %[[LO_VAL]], ptr %{{.*}}, align {{[0-9]+}}
+// The high eightbyte sits one 16-byte vector slot past the low one, reached by
+// bumping the offset on one side and by folding it into the address on the
+// other.
+// LLVMCIR: %[[HI_OFF:.+]] = add i32 %[[FP_OFFSET]], 16
+// LLVM: %[[HI_VAL:.+]] = load double, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: store double %[[HI_VAL]], ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[FP_NEXT:.+]] = add i32 %[[FP_OFFSET]], 32
+// LLVM: store i32 %[[FP_NEXT]], ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: br label %[[END_BB:.+]]
+// LLVM: [[MEM_BB]]:
+// LLVM: br label %[[END_BB]]
+// LLVM: [[END_BB]]:
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[MEM_BB]] ], [ %[[REG_TMP]], %[[REG_BB]] ]
+// LLVMCIR: %[[VA_ARG:.+]] = load %struct.DoublePair, ptr %[[ADDR]], align 8
+// LLVMCIR: store %struct.DoublePair %[[VA_ARG]], ptr %{{.*}}, align 8
+// OGCG: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[REG_BB]] ], [ %{{.*}}, %[[MEM_BB]] ]
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.*}}, ptr align 8 %[[ADDR]], i64 16, i1 false)
+
+// LLVM-LABEL: define dso_local void @varargs_aggregate_memory(ptr dead_on_unwind noalias writable sret(%struct.Big) align 8 %{{.*}}, i32 noundef %{{.*}}, ...)
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[OVERFLOW_NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 24
+// LLVM: store ptr %[[OVERFLOW_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVMCIR: %[[VA_ARG:.+]] = load %struct.Big, ptr %[[OVERFLOW]], align 8
+// LLVMCIR: store %struct.Big %[[VA_ARG]], ptr %{{.*}}, align 8
+// LLVM: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.*}}, ptr align 8 %{{.*}}, i64 24, i1 false)
+
+struct RevMixed {
+ long a;
+ double b;
+};
+
+struct RevMixed varargs_aggregate_mixed_pair_rev(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct RevMixed res = __builtin_va_arg(args, struct RevMixed);
+ __builtin_va_end(args);
+ return res;
+}
+
+// The mirror of the mixed pair, with INTEGER in the low eightbyte and SSE in
+// the high one, so each half has to land in the member that matches its own
+// class rather than in the other one.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_mixed_pair_rev(
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
+// CIR: %[[LO_ADDR_V:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
+// CIR: %[[LO_VAL:.+]] = cir.load %[[LO_ADDR_V]] : !cir.ptr<!s64i>, !s64i
+// CIR: %[[TMP_LO:.+]] = cir.get_member %{{.+}}[0] {{.*}} -> !cir.ptr<!s64i>
+// CIR: cir.store %[[LO_VAL]], %[[TMP_LO]] : !s64i, !cir.ptr<!s64i>
+// CIR: %[[HI_ADDR_V:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: %[[HI_VAL:.+]] = cir.load %[[HI_ADDR_V]] : !cir.ptr<!cir.double>, !cir.double
+// CIR: %[[TMP_HI:.+]] = cir.get_member %{{.+}}[1] {{.*}} -> !cir.ptr<!cir.double>
+// CIR: cir.store %[[HI_VAL]], %[[TMP_HI]] : !cir.double, !cir.ptr<!cir.double>
+// LLVM-LABEL: define dso_local { i64, double } @varargs_aggregate_mixed_pair_rev(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[GP_OFFSET]], 40
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: load i64, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: load double, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: add i32 %[[GP_OFFSET]], 8
+// LLVM: add i32 %[[FP_OFFSET]], 16
diff --git a/clang/test/CIR/CodeGen/var-arg-empty.c b/clang/test/CIR/CodeGen/var-arg-empty.c
new file mode 100644
index 0000000000000..d2f2de8a2bb21
--- /dev/null
+++ b/clang/test/CIR/CodeGen/var-arg-empty.c
@@ -0,0 +1,55 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefixes=LLVM --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefixes=LLVM --input-file=%t.ll %s
+
+struct Empty {};
+
+struct Empty only_empty(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct Empty res = __builtin_va_arg(args, struct Empty);
+ __builtin_va_end(args);
+ return res;
+}
+
+// An empty record travels in no register and no stack slot, so fetching one
+// reads no argument and advances no field of the cursor.
+// CIR-LABEL: cir.func {{.*}} @only_empty(
+// CIR-NOT: cir.va_arg
+// CIR-NOT: gp_offset
+// CIR-NOT: fp_offset
+// CIR-NOT: overflow_arg_area
+// CIR: cir.va_end
+
+// LLVM-LABEL: define dso_local void @only_empty(i32 noundef %{{.*}}, ...)
+// LLVM-NOT: va_arg
+// LLVM-NOT: getelementptr inbounds nuw %struct.__va_list_tag
+// LLVM: call void @llvm.va_end.p0(
+
+int empty_then_int(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct Empty e = __builtin_va_arg(args, struct Empty);
+ int i = __builtin_va_arg(args, int);
+ __builtin_va_end(args);
+ return i;
+}
+
+// The empty fetch leaves the cursor alone, so the int fetch that follows is
+// the only one that touches gp_offset.
+// CIR-LABEL: cir.func {{.*}} @empty_then_int(
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR-NOT: {name = "gp_offset"}
+// CIR: cir.va_end
+
+// LLVM-LABEL: define dso_local i32 @empty_then_int(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]]
+// LLVM: icmp ule i32 %[[GP_OFFSET]], 40
+// LLVM: call void @llvm.va_end.p0(
diff --git a/clang/test/CIR/CodeGen/var-arg-int128.c b/clang/test/CIR/CodeGen/var-arg-int128.c
new file mode 100644
index 0000000000000..6df946fa79482
--- /dev/null
+++ b/clang/test/CIR/CodeGen/var-arg-int128.c
@@ -0,0 +1,65 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM,LLVMCIR --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=LLVM,OGCG --input-file=%t.ll %s
+
+__int128 varargs_int128(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ __int128 res = __builtin_va_arg(args, __int128);
+ __builtin_va_end(args);
+ return res;
+}
+
+// __int128 spans two INTEGER eightbytes, so it needs two integer registers
+// rather than one, which lowers the gate and doubles the bump.
+// CIR-LABEL: cir.func {{.*}} @varargs_int128(
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<32> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
+// CIR: %[[REG_SAVE:.+]] = cir.load %{{.+}}
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!s128i>
+// CIR: %[[REG_ADDR_V:.+]] = cir.cast bitcast %[[REG_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!s128i>
+// CIR: %[[REG_VAL:.+]] = cir.load align(8) %[[REG_ADDR_V]] : !cir.ptr<!s128i>, !s128i
+// CIR: cir.store %[[REG_VAL]], %[[REG_TMP]] : !s128i, !cir.ptr<!s128i>
+// CIR: %[[REG_TMP_B:.+]] = cir.cast bitcast %[[REG_TMP]] : !cir.ptr<!s128i> -> !cir.ptr<!u8i>
+// CIR: %[[GP_BUMP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[GP_NEXT:.+]] = cir.add %[[GP_OFFSET]], %[[GP_BUMP]] : !u32i
+// CIR: cir.store %[[GP_NEXT]], %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_TMP_B]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: %[[OVERFLOW:.+]] = cir.load %{{.+}}
+// CIR: cir.yield %{{.+}} : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!s128i>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!s128i>, !s128i
+
+// LLVM-LABEL: define dso_local i128 @varargs_int128(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[FITS_GP:.+]] = icmp ule i32 %[[GP_OFFSET]], 32
+// LLVM: br i1 %[[FITS_GP]], label %[[REG_BB:.+]], label %[[MEM_BB:.+]]
+// LLVM: [[REG_BB]]:
+// LLVM: %[[REG_SAVE:.+]] = load ptr, ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[REG_SAVE]], i{{32|64}} %{{.*}}
+// A GP register slot is only 8-byte aligned, so the value is copied to a
+// 16-byte-aligned temp before it is read as a whole.
+// LLVMCIR: %[[VAL:.+]] = load i128, ptr %[[REG_ADDR]], align 8
+// LLVMCIR: store i128 %[[VAL]], ptr %[[REG_TMP:.+]], align 16
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[REG_TMP:.+]], ptr align 8 %[[REG_ADDR]], i64 16, i1 false)
+// LLVM: %[[GP_NEXT:.+]] = add i32 %[[GP_OFFSET]], 16
+// LLVM: store i32 %[[GP_NEXT]], ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: br label %[[END_BB:.+]]
+// LLVM: [[MEM_BB]]:
+// LLVM: br label %[[END_BB]]
+// LLVM: [[END_BB]]:
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[MEM_BB]] ], [ %[[REG_TMP]], %[[REG_BB]] ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_TMP]], %[[REG_BB]] ], [ %{{.*}}, %[[MEM_BB]] ]
+// LLVM: %[[VA_ARG:.+]] = load i128, ptr %[[ADDR]], align 16
+// LLVM: store i128 %[[VA_ARG]], ptr %{{.*}}, align 16
diff --git a/clang/test/CIR/CodeGen/var-arg-long-double.c b/clang/test/CIR/CodeGen/var-arg-long-double.c
new file mode 100644
index 0000000000000..ee2216b0a7213
--- /dev/null
+++ b/clang/test/CIR/CodeGen/var-arg-long-double.c
@@ -0,0 +1,51 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefix=LLVM,LLVMCIR --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefix=LLVM,OGCG --input-file=%t.ll %s
+
+long double varargs_long_double(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ long double res = __builtin_va_arg(args, long double);
+ __builtin_va_end(args);
+ return res;
+}
+
+// x87 long double is always MEMORY class: no register-save-area branch.
+// CIR-LABEL: cir.func {{.*}} @varargs_long_double(
+// CIR-SAME: -> !cir.long_double<!cir.f80>
+// CIR: cir.va_start %{{.+}} : !cir.ptr<!rec___va_list_tag>
+// CIR: %[[VA_PTR:.+]] = cir.cast array_to_ptrdecay %{{.+}} : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
+// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %[[VA_PTR]][2] {name = "overflow_arg_area"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!cir.ptr<!void>>
+// CIR: %[[OVERFLOW:.+]] = cir.load %[[OVERFLOW_P]] : !cir.ptr<!cir.ptr<!void>>, !cir.ptr<!void>
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[AS_INT:.+]] = cir.cast ptr_to_int %[[OVERFLOW_B]] : !cir.ptr<!u8i> -> !u64i
+// CIR: %[[BUMP:.+]] = cir.const #cir.int<15> : !u64i
+// CIR: %[[BUMPED:.+]] = cir.add nuw %[[AS_INT]], %[[BUMP]] : !u64i
+// CIR: %[[MASK:.+]] = cir.const #cir.int<18446744073709551600> : !u64i
+// CIR: %[[ROUNDED:.+]] = cir.and %[[BUMPED]], %[[MASK]] : !u64i
+// CIR: %[[ALIGNED:.+]] = cir.cast int_to_ptr %[[ROUNDED]] : !u64i -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<16> : !s32i
+// CIR: %[[OVERFLOW_NEXT:.+]] = cir.ptr_stride %[[ALIGNED]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[OVERFLOW_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[ALIGNED]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.long_double<!cir.f80>>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!cir.long_double<!cir.f80>>, !cir.long_double<!cir.f80>
+
+// LLVM-LABEL: define dso_local x86_fp80 @varargs_long_double(i32 noundef %{{.*}}, ...)
+// LLVM: call void @llvm.va_start.p0(ptr %{{.*}})
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// The overflow cursor is rounded up to 16 before the read, spelled as plain
+// arithmetic on one side and as a pointer mask on the other.
+// LLVMCIR: %[[AS_INT:.+]] = ptrtoint ptr %[[OVERFLOW]] to i64
+// LLVMCIR: %[[BUMPED:.+]] = add nuw i64 %[[AS_INT]], 15
+// LLVMCIR: %[[ROUNDED:.+]] = and i64 %[[BUMPED]], -16
+// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %[[ROUNDED]] to ptr
+// OGCG: %[[UNALIGNED:.+]] = getelementptr inbounds i8, ptr %[[OVERFLOW]], i32 15
+// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %[[UNALIGNED]], i64 -16)
+// LLVM: %[[OVERFLOW_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 16
+// LLVM: store ptr %[[OVERFLOW_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[VA_ARG:.+]] = load x86_fp80, ptr %[[ALIGNED]], align 16
+// LLVM: store x86_fp80 %[[VA_ARG]], ptr %{{.*}}, align 16
diff --git a/clang/test/CIR/CodeGen/var-arg-vector.c b/clang/test/CIR/CodeGen/var-arg-vector.c
new file mode 100644
index 0000000000000..660523be5e0e7
--- /dev/null
+++ b/clang/test/CIR/CodeGen/var-arg-vector.c
@@ -0,0 +1,56 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -target-feature +avx -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s --check-prefixes=CIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -target-feature +avx -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --input-file=%t-cir.ll %s --check-prefixes=LLVM,LLVMCIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -target-feature +avx -Wno-unused-value -emit-llvm %s -o %t.ll
+// RUN: FileCheck --input-file=%t.ll %s --check-prefixes=LLVM,OGCG
+
+typedef float v4f __attribute__((ext_vector_type(4)));
+typedef float v8f __attribute__((ext_vector_type(8)));
+
+v4f take_16(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ v4f res = __builtin_va_arg(args, v4f);
+ __builtin_va_end(args);
+ return res;
+}
+
+// A vector filling one 16-byte slot is fetched from the vector register area.
+// CIR-LABEL: cir.func {{.*}} @take_16(
+// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
+// CIR: cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+
+// LLVM-LABEL: define dso_local <4 x float> @take_16(i32 noundef %{{.*}}, ...)
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.*}}, align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: add i32 %[[FP_OFFSET]], 16
+
+v8f take_32(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ v8f res = __builtin_va_arg(args, v8f);
+ __builtin_va_end(args);
+ return res;
+}
+
+// A vector too wide for one slot travels in memory, whatever vector registers
+// the target has, so there is no register arm to bound and the cursor is
+// rounded up to the vector's own 32-byte alignment.
+// CIR-LABEL: cir.func {{.*}} @take_32(
+// CIR-NOT: fp_offset
+// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %{{.+}}[2] {name = "overflow_arg_area"}
+// CIR: %[[MASK:.+]] = cir.const #cir.int<18446744073709551584> : !u64i
+// CIR: cir.and %{{.+}}, %[[MASK]] : !u64i
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<32> : !s32i
+// CIR-NOT: fp_offset
+// CIR: cir.va_end
+
+// LLVM-LABEL: define dso_local <8 x float> @take_32(i32 noundef %{{.*}}, ...)
+// LLVM-NOT: icmp ule i32 %{{.*}}, 160
+// LLVMCIR: and i64 %{{.+}}, -32
+// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
+// LLVM: getelementptr i8, ptr %{{.+}}, i{{32|64}} 32
+// LLVM: call void @llvm.va_end.p0(
diff --git a/clang/test/CIR/CodeGen/var_arg.c b/clang/test/CIR/CodeGen/var_arg.c
index bc744fef4bb51..0f19ca293c69a 100644
--- a/clang/test/CIR/CodeGen/var_arg.c
+++ b/clang/test/CIR/CodeGen/var_arg.c
@@ -21,6 +21,7 @@ int varargs(int count, ...) {
return res;
}
+// `int` classifies Extend, one INTEGER eightbyte.
// CIR-LABEL: cir.func {{.*}} @varargs(
// CIR: %[[RET_ADDR:.+]] = cir.alloca "__retval" {{.*}} : !cir.ptr<!s32i>
// CIR: %[[VAAREA:.+]] = cir.alloca "args" {{.*}} : !cir.ptr<!cir.array<!rec___va_list_tag x 1>>
@@ -28,8 +29,28 @@ int varargs(int count, ...) {
// CIR: %[[VA_PTR0:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_start %[[VA_PTR0]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[VA_PTR1:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
-// CIR: %[[VA_ARG:.+]] = cir.va_arg %[[VA_PTR1]] : (!cir.ptr<!rec___va_list_tag>) -> !s32i
-// CIR: cir.store{{.*}} %[[VA_ARG]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %[[VA_PTR1]][0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
+// CIR: %[[REG_SAVE:.+]] = cir.load %{{.+}}
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[GP_BUMP:.+]] = cir.const #cir.int<8> : !u32i
+// CIR: %[[GP_NEXT:.+]] = cir.add %[[GP_OFFSET]], %[[GP_BUMP]] : !u32i
+// CIR: cir.store %[[GP_NEXT]], %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: %[[OVERFLOW:.+]] = cir.load %{{.+}}
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<8> : !s32i
+// CIR: %[[OVERFLOW_NEXT:.+]] = cir.ptr_stride %[[OVERFLOW_B]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[OVERFLOW_B]] : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!s32i>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!s32i>, !s32i
+// CIR: cir.store{{.*}} %[[VA_ARG_V]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
// CIR: %[[VA_PTR2:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_end %[[VA_PTR2]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[RESULT:.+]] = cir.load{{.*}} %[[RES_ADDR]] : !cir.ptr<!s32i>, !s32i
@@ -65,8 +86,20 @@ int stdarg_start(int count, ...) {
// CIR: %[[VA_PTR0:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_start %[[VA_PTR0]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[VA_PTR1:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
-// CIR: %[[VA_ARG:.+]] = cir.va_arg %[[VA_PTR1]] : (!cir.ptr<!rec___va_list_tag>) -> !s32i
-// CIR: cir.store{{.*}} %[[VA_ARG]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %[[VA_PTR1]][0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %{{.+}}, %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %{{.+}}, %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: cir.yield %{{.+}} : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!s32i>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!s32i>, !s32i
+// CIR: cir.store{{.*}} %[[VA_ARG_V]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
// CIR: %[[VA_PTR2:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_end %[[VA_PTR2]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[RESULT:.+]] = cir.load{{.*}} %[[RES_ADDR]] : !cir.ptr<!s32i>, !s32i
@@ -121,8 +154,20 @@ int varargs_new(char *fmt, ...) {
// CIR: %[[VA_PTR0:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_start %[[VA_PTR0]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[VA_PTR1:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
-// CIR: %[[VA_ARG:.+]] = cir.va_arg %[[VA_PTR1]] : (!cir.ptr<!rec___va_list_tag>) -> !s32i
-// CIR: cir.store{{.*}} %[[VA_ARG]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %[[VA_PTR1]][0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %{{.+}}, %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %{{.+}}, %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: }, false {
+// CIR: cir.yield %{{.+}} : !cir.ptr<!u8i>
+// CIR: }) : (!cir.bool) -> !cir.ptr<!u8i>
+// CIR: %[[VA_ARG_B:.+]] = cir.cast bitcast %[[VA_ARG]] : !cir.ptr<!u8i> -> !cir.ptr<!s32i>
+// CIR: %[[VA_ARG_V:.+]] = cir.load %[[VA_ARG_B]] : !cir.ptr<!s32i>, !s32i
+// CIR: cir.store{{.*}} %[[VA_ARG_V]], %[[RES_ADDR]] : !s32i, !cir.ptr<!s32i>
// CIR: %[[VA_PTR2:.+]] = cir.cast array_to_ptrdecay %[[VAAREA]] : !cir.ptr<!cir.array<!rec___va_list_tag x 1>> -> !cir.ptr<!rec___va_list_tag>
// CIR: cir.va_end %[[VA_PTR2]] : !cir.ptr<!rec___va_list_tag>
// CIR: %[[RESULT:.+]] = cir.load{{.*}} %[[RES_ADDR]] : !cir.ptr<!s32i>, !s32i
@@ -190,3 +235,24 @@ void with_param(int count, ...) {
// LLVM: %[[VA_PTR1:.+]] = getelementptr {{.*}}%struct.__va_list_tag{{.?}}, ptr %[[VAAREA]]
// LLVM: call void @llvm.va_end.p0(ptr %[[VA_PTR1]])
// LLVM: ret void
+
+double varargs_double(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ double res = __builtin_va_arg(args, double);
+ __builtin_va_end(args);
+ return res;
+}
+
+// `double` is Direct with no coercion, filling one vector-register eightbyte,
+// so it reads the vector cursor rather than the integer one.
+// CIR-LABEL: cir.func {{.*}} @varargs_double(
+// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
+// CIR: cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+
+// LLVM-LABEL: define dso_local double @varargs_double(
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: add i32 %[[FP_OFFSET]], 16
diff --git a/mlir/include/mlir/ABI/ABIRewriteContext.h b/mlir/include/mlir/ABI/ABIRewriteContext.h
index 8394aeccc75ce..c462665f9f0bf 100644
--- a/mlir/include/mlir/ABI/ABIRewriteContext.h
+++ b/mlir/include/mlir/ABI/ABIRewriteContext.h
@@ -190,6 +190,33 @@ class ABIRewriteContext {
const FunctionClassification &fc,
OpBuilder &builder) = 0;
+ /// Rewrite a single "fetch the next vararg" operation (e.g. C `va_arg`) in
+ /// place to match how \p ac says the fetched type is passed at the ABI
+ /// level.
+ ///
+ /// This is deliberately separate from rewriteFunctionDefinition and
+ /// rewriteCallSite, which classify a whole signature ahead of time. A
+ /// vararg fetch instead advances a runtime cursor (the platform va_list)
+ /// through registers and then memory, so whether a given fetch lands in a
+ /// register depends on how much of the register budget earlier variadic
+ /// arguments already consumed at run time, not on the fetch's static
+ /// position. \p ac therefore classifies only the one type being fetched,
+ /// in isolation.
+ ///
+ /// The default implementation reports failure, so a dialect that has not
+ /// implemented vararg fetches does not need to override this. An overrider
+ /// that fails is responsible for emitting its own diagnostic.
+ ///
+ /// \param vaArgOp The operation to rewrite in place.
+ /// \param ac The ABI classification of the fetched type.
+ /// \param builder The OpBuilder to use for modifications.
+ /// \returns success() if the operation was rewritten.
+ virtual LogicalResult rewriteVAArg(Operation *vaArgOp,
+ const ArgClassification &ac,
+ OpBuilder &builder) {
+ return failure();
+ }
+
/// Return the dialect namespace this context handles (e.g. "cir").
virtual StringRef getDialectNamespace() const = 0;
};
>From f02c864ca0f89639add19794549ba394a032b133 Mon Sep 17 00:00:00 2001
From: Adam Smith <adams at nvidia.com>
Date: Tue, 15 Sep 2026 09:23:09 -0700
Subject: [PATCH 2/3] [CIR] Take the va_arg register demand from the classifier
rewriteVAArg rebuilt how many registers of each class the fetched type
needs by inspecting that type. ArgInfo and ArgClassification now carry
the demand the x86-64 classifier already computes.
The register arm also has to copy into a full-size temp whenever the
registers carry less than the whole result, not only when the value
starts at a byte offset. A record whose tail holds no field took the
neighbouring slot as part of its value.
Assisted-by: Cursor / claude-opus-5
---
.../Transforms/CallConvLoweringPass.cpp | 10 +-
.../TargetLowering/CIRABIRewriteContext.cpp | 175 ++++++++----------
clang/test/CIR/CodeGen/var-arg-aggregate.c | 100 ++++++++--
.../CIR/CodeGen/var-arg-direct-offset.cpp | 168 +++++++++++++++++
clang/test/CIR/CodeGen/var-arg-empty.c | 9 +-
clang/test/CIR/CodeGen/var-arg-int128.c | 10 +-
clang/test/CIR/CodeGen/var-arg-vector.c | 57 +++++-
clang/test/CIR/CodeGen/var_arg.c | 86 ++++++++-
.../x86_64-vaarg-valist-shape-nyi.cir | 21 +++
llvm/include/llvm/ABI/FunctionInfo.h | 16 +-
llvm/lib/ABI/Targets/X86.cpp | 4 +-
mlir/include/mlir/ABI/ABIRewriteContext.h | 35 ++--
12 files changed, 545 insertions(+), 146 deletions(-)
create mode 100644 clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
create mode 100644 clang/test/CIR/Transforms/abi-lowering/x86_64-vaarg-valist-shape-nyi.cir
diff --git a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
index a98d09fc01792..37320bf8c659a 100644
--- a/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/CallConvLoweringPass.cpp
@@ -685,12 +685,15 @@ static std::optional<FunctionClassification> classifyX86_64Signature(
fc.returnInfo = *retAc;
for (unsigned i = 0, e = fi->arg_size(); i < e; ++i) {
mlir::Type origArg = i < inputs.size() ? inputs[i] : mlir::Type();
+ const llvm::abi::ArgInfo &argInfo = fi->getArgInfo(i).Info;
std::optional<ArgClassification> ac =
- convertABIArgInfo(fi->getArgInfo(i).Info, ctx, origArg);
+ convertABIArgInfo(argInfo, ctx, origArg);
if (!ac) {
nyiCoercion(origArg);
return std::nullopt;
}
+ ac->neededIntRegs = argInfo.getNeededIntRegs();
+ ac->neededSseRegs = argInfo.getNeededSseRegs();
fc.argInfos.push_back(*ac);
}
return fc;
@@ -1165,10 +1168,9 @@ void CallConvLoweringPass::runOnOperation() {
}
}
- // Fetches on targets other than x86-64 are left in place here and still
- // lower to a generic vararg instruction that cannot select an aggregate or
- // an x87 long double result.
if (isX86) {
+ // Collect the fetches before rewriting any, because rewriting one erases
+ // it, and a live walk holds the op it is about to visit next.
SmallVector<cir::VAArgOp> vaArgs;
moduleOp.walk([&](cir::VAArgOp v) { vaArgs.push_back(v); });
for (cir::VAArgOp v : vaArgs) {
diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
index 45162691f3feb..b328bb3fdbca1 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
@@ -15,7 +15,6 @@
#include "clang/CIR/Dialect/IR/CIRDialect.h"
#include "clang/CIR/Dialect/IR/CIRTypes.h"
#include "clang/CIR/MissingFeatures.h"
-#include "llvm/ADT/APFloat.h"
#include "llvm/ADT/STLExtras.h"
#include <algorithm>
#include <array>
@@ -1064,21 +1063,16 @@ void rewriteIndirectReturnCall(cir::CallOp call,
call->erase();
}
-/// Whether \p ty is an eightbyte that travels in a vector register. An x87
-/// long double is floating-point but travels in neither register class, so it
-/// is excluded. A caller reads a false result as the integer class.
+/// Whether \p ty, a type the classifier named for one register of a
+/// coercion, is carried in a vector register.
bool isSSERegisterClass(mlir::Type ty) {
- if (mlir::isa<cir::VectorType>(ty))
- return true;
- if (auto fp = mlir::dyn_cast<cir::FPTypeInterface>(ty))
- return &fp.getFloatSemantics() != &llvm::APFloat::x87DoubleExtended();
- return false;
+ return mlir::isa<cir::VectorType, cir::FPTypeInterface>(ty);
}
-/// The boundary the caller aligned \p ty to in the argument area. An
-/// alignment attribute can raise a record above what its members imply, and
-/// only the record-layout metadata carries that, so the member-derived value
-/// alone can be too small.
+/// The alignment in bytes the ABI gives \p ty as an argument. An alignment
+/// attribute can raise a record above what its members imply, and only the
+/// record-layout metadata carries that, so the member-derived value alone can
+/// be too small.
uint64_t argumentAreaAlign(mlir::Type ty, mlir::ModuleOp modOp,
const mlir::DataLayout &dl) {
uint64_t align = dl.getTypeABIAlignment(ty);
@@ -1088,24 +1082,6 @@ uint64_t argumentAreaAlign(mlir::Type ty, mlir::ModuleOp modOp,
return align;
}
-/// Whether an SSE-class type fits the single 16-byte vector-register slot a
-/// fetch can read from. A wider vector travels in memory no matter how many
-/// vector registers the target has, because a fetch has no declared parameter
-/// to pin it to one. The width has to be measured here rather than read off
-/// the classification, which spells a memory-class scalar the same way it
-/// spells one that really does travel in a register.
-bool fitsOneVectorSlot(mlir::Type ty, const mlir::DataLayout &dl) {
- return dl.getTypeSize(ty).getFixedValue() <= 16;
-}
-
-/// Whether a scalar INTEGER-class type occupies two eightbytes rather than
-/// one, which is the case for any 128-bit container: `__int128`, or a
-/// narrower `_BitInt` widened to 128 bits.
-bool isWide128BitInt(mlir::Type ty) {
- auto intTy = mlir::dyn_cast<cir::IntType>(ty);
- return intTy && intTy.getWidth() == 128;
-}
-
} // namespace
void CIRABIRewriteContext::normalizeParameterSlotAlignments(
@@ -1589,65 +1565,40 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
if (ac.kind == ArgKind::Indirect && !ac.byVal)
return reportNYI("a non-trivially-copyable type");
- // How many eightbytes of each register class the fetched type occupies.
- // Zero of both means the type travels in memory and is read straight from
- // the overflow area.
- unsigned neededInt = 0, neededSse = 0;
+ // neededInt counts 8-byte integer slots and neededSse counts 16-byte vector
+ // slots. Zero of both means the type travels in memory and is read
+ // straight from the overflow area.
+ unsigned neededInt = ac.neededIntRegs;
+ unsigned neededSse = ac.neededSseRegs;
+
// Which coerced-pair element (0 = low eightbyte, 1 = high) is SSE rather
// than INTEGER class. Only meaningful when isRegPair is set.
std::array<bool, 2> pairIsSse = {false, false};
bool isRegPair = false;
- if (ac.kind == ArgKind::Extend) {
- neededInt = 1;
- } else if (ac.kind == ArgKind::Direct) {
- if (mlir::Type coerced = ac.coercedType) {
- if (auto pairTy = mlir::dyn_cast<cir::RecordType>(coerced)) {
- assert(pairTy.getNumElements() == 2 &&
- "a register-class coercion spans at most two eightbytes");
- for (auto [i, memberTy] : llvm::enumerate(pairTy.getMembers())) {
- if (isSSERegisterClass(memberTy)) {
- pairIsSse[i] = true;
- ++neededSse;
- } else {
- ++neededInt;
- }
- }
- isRegPair = true;
- } else if (isSSERegisterClass(coerced)) {
- if (fitsOneVectorSlot(coerced, dl))
- neededSse = 1;
- } else {
- neededInt = isWide128BitInt(coerced) ? 2 : 1;
- }
- } else if (auto fp = mlir::dyn_cast<cir::FPTypeInterface>(resultTy)) {
- // An uncoerced scalar float is SSE, unless it is x87 long double,
- // which the SysV ABI always classifies MEMORY. That is recognized
- // from the semantics here, because the classification spells a
- // memory-class scalar the same way it spells a register one.
- if (&fp.getFloatSemantics() != &llvm::APFloat::x87DoubleExtended())
- neededSse = 1;
- } else if (mlir::isa<cir::VectorType>(resultTy)) {
- if (fitsOneVectorSlot(resultTy, dl))
- neededSse = 1;
- } else if (mlir::isa<cir::IntType, cir::PointerType, cir::BoolType>(
- resultTy)) {
- neededInt = isWide128BitInt(resultTy) ? 2 : 1;
- } else {
- return reportNYI("this type");
+ // A value needing two registers is either a coerced pair, one register per
+ // member, or a single wide scalar read from one address.
+ if (ac.kind == ArgKind::Direct && neededInt + neededSse == 2 &&
+ ac.coercedType) {
+ if (auto pairTy = mlir::dyn_cast<cir::RecordType>(ac.coercedType)) {
+ if (pairTy.getNumElements() != 2)
+ return reportNYI("a register coercion that is not two eightbytes");
+ assert(!ac.directOffset &&
+ "a pair already spans both eightbytes, so it cannot also start "
+ "partway into the value");
+ for (auto [i, memberTy] : llvm::enumerate(pairTy.getMembers()))
+ pairIsSse[i] = isSSERegisterClass(memberTy);
+ isRegPair = true;
}
- } else if (ac.kind != ArgKind::Indirect) {
- return reportNYI("a type with an argument classification this fetch does "
- "not model");
}
- // Indirect is the one remaining kind, and it takes no register.
- auto vaListRecTy = mlir::cast<cir::RecordType>(
+ auto vaListRecTy = mlir::dyn_cast<cir::RecordType>(
mlir::cast<cir::PointerType>(valist.getType()).getPointee());
+ if (!vaListRecTy || vaListRecTy.getNumElements() != 4) {
+ return reportNYI("a va_list that is not the four-field gp_offset / "
+ "fp_offset / overflow_arg_area / reg_save_area cursor");
+ }
llvm::ArrayRef<mlir::Type> vaFields = vaListRecTy.getMembers();
- assert(vaFields.size() == 4 &&
- "expected the four-field gp_offset / fp_offset / overflow_arg_area / "
- "reg_save_area argument cursor");
cir::IntType byteTy = builder.getUIntNTy(8);
cir::PointerType bytePtrTy = builder.getPointerTo(byteTy);
@@ -1705,19 +1656,23 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
}
// A two-eightbyte pair that is purely INTEGER class is contiguous in the
- // register-save area (GP slots are 8-byte packed), so one load of the
- // coerced pair type at gp_offset reads both eightbytes. A pure SSE pair
- // or a mixed pair is not contiguous (SSE slots are 16-byte spaced, and a
- // mixed pair's GP and SSE halves live in disjoint areas of the
- // register-save area), so each half is copied into a temp laid out as the
- // coerced pair before the whole value is reinterpreted as resultTy.
+ // register-save area, since GP slots are 8-byte packed, so the address of
+ // its low eightbyte is already the address of the whole value. A pure
+ // SSE pair or a mixed pair is not contiguous, since SSE slots are 16-byte
+ // spaced and a mixed pair's halves live in disjoint areas, so each half is
+ // copied into a temp laid out as the coerced pair.
bool pairNeedsReassembly = isRegPair && neededSse != 0;
mlir::Value regPairTemp;
- if (pairNeedsReassembly)
+ if (pairNeedsReassembly) {
+ // The temp is written one member at a time through the coerced pair, so
+ // it has to meet that type's alignment, and read back as the result, so
+ // it has to meet the result's declared one too.
regPairTemp = builder.createAlloca(
loc, builder.getPointerTo(ac.coercedType), "vaarg.reg",
clang::CharUnits::fromQuantity(
- dl.getTypeABIAlignment(ac.coercedType)));
+ std::max(dl.getTypeABIAlignment(ac.coercedType),
+ argumentAreaAlign(resultTy, module, dl))));
+ }
addr =
cir::TernaryOp::create(
@@ -1743,10 +1698,11 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
unsigned regSize = isSse ? 16 : 8;
unsigned prior = seenOfClass[isSse]++;
mlir::Value off = base;
- if (prior)
+ if (prior) {
off = b.createAdd(
l, base,
b.getConstantInt(l, base.getType(), prior * regSize));
+ }
mlir::Value src = b.createPtrStride(l, regSaveArea, off);
mlir::Type elemTy = pairTy.getElementType(i);
mlir::Value val =
@@ -1759,17 +1715,38 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
} else {
mlir::Value off = neededSse ? fpOffset : gpOffset;
regAddr = b.createPtrStride(l, regSaveArea, off);
- // A GP register slot is only 8-byte aligned, so a
- // wider-aligned GP-class fetch (e.g. __int128, align 16)
- // is copied through a temp. SSE slots are already
- // 16-byte aligned and need no copy.
uint64_t resultAlign = argumentAreaAlign(resultTy, module, dl);
- if (!neededSse && resultAlign > 8) {
+ uint64_t regSize = neededInt ? neededInt * 8 : 16;
+ uint64_t tySize = dl.getTypeSize(resultTy).getFixedValue();
+ if (ac.coercedType && (ac.directOffset || regSize < tySize)) {
+ // The registers carry less than the whole result, either
+ // because the eightbytes below directOffset hold no field or
+ // because the result is wider than the registers carrying
+ // it. Copy what they do carry into a full-size temp, so the
+ // load below reads a whole result rather than reading on
+ // into the neighbouring slot.
+ mlir::Value temp = b.createAlloca(
+ l, b.getPointerTo(resultTy), "vaarg.reg",
+ clang::CharUnits::fromQuantity(resultAlign));
+ mlir::Value val = b.createAlignedLoad(
+ l, b.createPtrBitcast(regAddr, ac.coercedType), 8);
+ mlir::Value dst = temp;
+ if (ac.directOffset) {
+ dst = b.createPtrStride(
+ l, b.createPtrBitcast(temp, byteTy),
+ b.getSignedInt(l, ac.directOffset, 32));
+ }
+ b.createStore(l, val,
+ b.createPtrBitcast(dst, ac.coercedType));
+ regAddr = b.createPtrBitcast(temp, byteTy);
+ } else if (!neededSse && resultAlign > 8) {
+ // A GP register slot is only 8-byte aligned, so a
+ // wider-aligned GP-class fetch (e.g. __int128, align 16)
+ // is copied through a temp. SSE slots are already
+ // 16-byte aligned and need no copy.
mlir::Value temp = b.createAlloca(
l, b.getPointerTo(resultTy), "vaarg.reg",
clang::CharUnits::fromQuantity(resultAlign));
- // The slot this reads from is only 8-byte aligned, so the
- // load cannot claim the wider alignment the temp has.
mlir::Value val = b.createAlignedLoad(
l, b.createPtrBitcast(regAddr, resultTy), 8);
b.createStore(l, val, temp);
@@ -1777,20 +1754,22 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
}
}
- if (neededInt)
+ if (neededInt) {
b.createStore(
l,
b.createAdd(
l, gpOffset,
b.getConstantInt(l, gpOffset.getType(), neededInt * 8)),
gpOffsetP);
- if (neededSse)
+ }
+ if (neededSse) {
b.createStore(
l,
b.createAdd(l, fpOffset,
b.getConstantInt(l, fpOffset.getType(),
neededSse * 16)),
fpOffsetP);
+ }
cir::YieldOp::create(b, l, regAddr);
},
diff --git a/clang/test/CIR/CodeGen/var-arg-aggregate.c b/clang/test/CIR/CodeGen/var-arg-aggregate.c
index e3ca73d10102b..7fe8b336d1237 100644
--- a/clang/test/CIR/CodeGen/var-arg-aggregate.c
+++ b/clang/test/CIR/CodeGen/var-arg-aggregate.c
@@ -303,26 +303,102 @@ struct RevMixed varargs_aggregate_mixed_pair_rev(int count, ...) {
}
// The mirror of the mixed pair, with INTEGER in the low eightbyte and SSE in
-// the high one, so each half has to land in the member that matches its own
-// class rather than in the other one.
+// the high one, so each half has to be read from the cursor for its own class
+// and stored to the member for that class. Reading the SSE half from
+// gp_offset would still produce a well-formed sequence, so each load is bound
+// to the cursor it came from.
// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_mixed_pair_rev(
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
-// CIR: %[[LO_ADDR_V:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
-// CIR: %[[LO_VAL:.+]] = cir.load %[[LO_ADDR_V]] : !cir.ptr<!s64i>, !s64i
-// CIR: %[[TMP_LO:.+]] = cir.get_member %{{.+}}[0] {{.*}} -> !cir.ptr<!s64i>
-// CIR: cir.store %[[LO_VAL]], %[[TMP_LO]] : !s64i, !cir.ptr<!s64i>
-// CIR: %[[HI_ADDR_V:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
-// CIR: %[[HI_VAL:.+]] = cir.load %[[HI_ADDR_V]] : !cir.ptr<!cir.double>, !cir.double
-// CIR: %[[TMP_HI:.+]] = cir.get_member %{{.+}}[1] {{.*}} -> !cir.ptr<!cir.double>
-// CIR: cir.store %[[HI_VAL]], %[[TMP_HI]] : !cir.double, !cir.ptr<!cir.double>
+// CIR: %[[FITS_FP:.+]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!rec_anon_struct{{[0-9]*}}>
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[LO_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[LO_ADDR_V:.+]] = cir.cast bitcast %[[LO_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
+// CIR: %[[LO_VAL:.+]] = cir.load %[[LO_ADDR_V]] : !cir.ptr<!s64i>, !s64i
+// CIR: %[[TMP_LO:.+]] = cir.get_member %[[REG_TMP]][0] {{.*}} -> !cir.ptr<!s64i>
+// CIR: cir.store %[[LO_VAL]], %[[TMP_LO]] : !s64i, !cir.ptr<!s64i>
+// CIR: %[[HI_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[HI_ADDR_V:.+]] = cir.cast bitcast %[[HI_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: %[[HI_VAL:.+]] = cir.load %[[HI_ADDR_V]] : !cir.ptr<!cir.double>, !cir.double
+// CIR: %[[TMP_HI:.+]] = cir.get_member %[[REG_TMP]][1] {{.*}} -> !cir.ptr<!cir.double>
+// CIR: cir.store %[[HI_VAL]], %[[TMP_HI]] : !cir.double, !cir.ptr<!cir.double>
+// CIR: %[[REG_TMP_B:.+]] = cir.cast bitcast %[[REG_TMP]] : !cir.ptr<!rec_anon_struct{{[0-9]*}}> -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[REG_TMP_B]] : !cir.ptr<!u8i>
// LLVM-LABEL: define dso_local { i64, double } @varargs_aggregate_mixed_pair_rev(i32 noundef %{{.*}}, ...)
// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
// LLVM: icmp ule i32 %[[GP_OFFSET]], 40
// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
-// LLVM: load i64, ptr %{{.+}}, align {{[0-9]+}}
-// LLVM: load double, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: %[[RSA:.+]] = load ptr, ptr %{{.+}}, align {{[0-9]+}}
+// Both halves index the same save area, each by its own cursor.
+// LLVMCIR: %[[GP64:.+]] = zext i32 %[[GP_OFFSET]] to i64
+// LLVMCIR: %[[LO_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[GP64]]
+// LLVMCIR: %[[LO_VAL:.+]] = load i64, ptr %[[LO_ADDR]], align {{[0-9]+}}
+// LLVMCIR: store i64 %[[LO_VAL]], ptr %{{.+}}, align {{[0-9]+}}
+// LLVMCIR: %[[FP64:.+]] = zext i32 %[[FP_OFFSET]] to i64
+// LLVMCIR: %[[HI_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[FP64]]
+// LLVMCIR: %[[HI_VAL:.+]] = load double, ptr %[[HI_ADDR]], align {{[0-9]+}}
+// LLVMCIR: store double %[[HI_VAL]], ptr %{{.+}}, align {{[0-9]+}}
+// OGCG: %[[LO_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[GP_OFFSET]]
+// OGCG: %[[HI_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[FP_OFFSET]]
+// OGCG: %[[LO_VAL:.+]] = load i64, ptr %[[LO_ADDR]], align {{[0-9]+}}
+// OGCG: store i64 %[[LO_VAL]], ptr %{{.+}}, align {{[0-9]+}}
+// OGCG: %[[HI_VAL:.+]] = load double, ptr %[[HI_ADDR]], align {{[0-9]+}}
+// OGCG: store double %[[HI_VAL]], ptr %{{.+}}, align {{[0-9]+}}
// LLVM: add i32 %[[GP_OFFSET]], 8
// LLVM: add i32 %[[FP_OFFSET]], 16
+
+struct __attribute__((aligned(16))) OverAligned {
+ long a;
+ long b;
+};
+
+struct OverAligned varargs_aggregate_overaligned(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct OverAligned res = __builtin_va_arg(args, struct OverAligned);
+ __builtin_va_end(args);
+ return res;
+}
+
+// An alignment attribute raises the record above what its two long members
+// imply, and only the record layout records that. Both arms have to honor
+// it: the register arm copies through a 16-byte-aligned temp, and the
+// overflow arm rounds the cursor up to 16 before reading.
+// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_overaligned(
+// CIR: %[[GP_OFFSET:.+]] = cir.load %{{.+}} : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<32> : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %{{.+}}, %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(16) : !cir.ptr<!rec_OverAligned>
+// CIR: %[[REG_ADDR_V:.+]] = cir.cast bitcast %[[REG_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_OverAligned>
+// CIR: %[[VAL:.+]] = cir.load align(8) %[[REG_ADDR_V]] : !cir.ptr<!rec_OverAligned>, !rec_OverAligned
+// CIR: cir.store %[[VAL]], %[[TEMP]] : !rec_OverAligned, !cir.ptr<!rec_OverAligned>
+// CIR: %[[TEMP_B:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_OverAligned> -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[TEMP_B]] : !cir.ptr<!u8i>
+// CIR: %[[MASK:.+]] = cir.const #cir.int<18446744073709551600> : !u64i
+// CIR: %[[ROUNDED:.+]] = cir.and %{{.+}}, %[[MASK]] : !u64i
+// CIR: %[[ALIGNED:.+]] = cir.cast int_to_ptr %[[ROUNDED]] : !u64i -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[ALIGNED]] : !cir.ptr<!u8i>
+
+// LLVM-LABEL: define dso_local { i64, i64 } @varargs_aggregate_overaligned(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[GP_OFFSET]], 32
+// The register slot is only 8-aligned, so the copy reads at 8 and stores
+// into a temp carrying the record's declared 16.
+// LLVMCIR: %[[VAL:.+]] = load %struct.OverAligned, ptr %{{.+}}, align 8
+// LLVMCIR: store %struct.OverAligned %[[VAL]], ptr %[[TEMP:.+]], align 8
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[TEMP:.+]], ptr align 8 %{{.+}}, i64 16, i1 false)
+// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %{{.+}} to ptr
+// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -16)
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 16
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[ALIGNED]], %{{.+}} ], [ %[[TEMP]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[TEMP]], %{{.+}} ], [ %[[ALIGNED]], %{{.+}} ]
diff --git a/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp b/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
new file mode 100644
index 0000000000000..bc545dd7bc507
--- /dev/null
+++ b/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
@@ -0,0 +1,168 @@
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
+// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
+// RUN: FileCheck --check-prefixes=LLVM,LLVMCIR --input-file=%t-cir.ll %s
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
+// RUN: FileCheck --check-prefixes=LLVM,OGCG --input-file=%t.ll %s
+
+struct Empty {};
+
+// No field with data lies in the low eightbyte, so the register carries bytes
+// 8 through 15 and the classification reports that byte offset.
+struct EmptyLow {
+ Empty e;
+ long x;
+};
+
+struct EmptyLowSse {
+ Empty e;
+ double d;
+};
+
+// One eightbyte holds the only field with data and the rest of the record
+// carries none, so the record is wider than the register that carries it.
+// Reading the record straight from the slot would also read the next one.
+struct TailPad {
+ long x;
+ Empty e;
+};
+
+EmptyLow varargs_empty_low(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ EmptyLow res = __builtin_va_arg(args, EmptyLow);
+ __builtin_va_end(args);
+ return res;
+}
+
+// CIR-LABEL: cir.func {{.*}} @_Z17varargs_empty_lowiz(
+// CIR: %[[GP_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP:.+]] = cir.load %[[GP_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[GP]], %[[LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[RSA_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[SLOT:.+]] = cir.ptr_stride %[[RSA_B]], %[[GP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLow>
+// CIR: %[[SLOT_I:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
+// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_I]] : !cir.ptr<!s64i>, !s64i
+// CIR: %[[TEMP_B:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLow> -> !cir.ptr<!u8i>
+// CIR: %[[OFF:.+]] = cir.const #cir.int<8> : !s32i
+// CIR: %[[DST:.+]] = cir.ptr_stride %[[TEMP_B]], %[[OFF]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: %[[DST_I:.+]] = cir.cast bitcast %[[DST]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
+// CIR: cir.store %[[VAL]], %[[DST_I]] : !s64i, !cir.ptr<!s64i>
+// CIR: %[[YIELDED:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLow> -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[YIELDED]] : !cir.ptr<!u8i>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_EmptyLow>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!rec_EmptyLow>, !rec_EmptyLow
+
+// LLVM-LABEL: define dso_local i64 @_Z17varargs_empty_lowiz(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 0
+// LLVM: %[[GP:.+]] = load i32, ptr %[[GP_P]], align {{4|16}}
+// LLVM: %[[FITS:.+]] = icmp ule i32 %[[GP]], 40
+// LLVM: %[[RSA:.+]] = load ptr, ptr %{{.+}}, align {{8|16}}
+// LLVMCIR: %[[GP64:.+]] = zext i32 %[[GP]] to i64
+// LLVMCIR: %[[SLOT:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[GP64]]
+// OGCG: %[[SLOT:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[GP]]
+// LLVM: %[[VAL:.+]] = load i64, ptr %[[SLOT]], align 8
+// LLVM: %[[DST:.+]] = getelementptr i8, ptr %[[TEMP:.+]], i{{32|64}} 8
+// LLVM: store i64 %[[VAL]], ptr %[[DST]], align 8
+// LLVM: %[[BUMPED:.+]] = add i32 %[[GP]], 8
+// LLVM: store i32 %[[BUMPED]], ptr %[[GP_P]], align {{4|16}}
+
+// The overflow area holds the whole record laid out normally, so the memory
+// path reads it from its base and the byte offset does not apply.
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 16
+// LLVM: store ptr %[[NEXT]], ptr %[[OVERFLOW_P]], align 8
+
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[OVERFLOW]], %{{.+}} ], [ %[[TEMP]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[TEMP]], %{{.+}} ], [ %[[OVERFLOW]], %{{.+}} ]
+// LLVMCIR: load %struct.EmptyLow, ptr %[[ADDR]], align 8
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.+}}, ptr align 8 %[[ADDR]], i64 16, i1 false)
+
+double varargs_empty_low_sse(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ EmptyLowSse res = __builtin_va_arg(args, EmptyLowSse);
+ __builtin_va_end(args);
+ return res.d;
+}
+
+// The same offset applies when the carrying register is SSE, so the fetch
+// gates on fp_offset and still lands the value at byte 8.
+// CIR-LABEL: cir.func {{.*}} @_Z21varargs_empty_low_sseiz(
+// CIR: %[[FP_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[FP:.+]] = cir.load %[[FP_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[LIMIT:.+]] = cir.const #cir.int<160> : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[FP]], %[[LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[SLOT:.+]] = cir.ptr_stride %{{.+}}, %[[FP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLowSse>
+// CIR: %[[SLOT_D:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_D]] : !cir.ptr<!cir.double>, !cir.double
+// CIR: %[[TEMP_B:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLowSse> -> !cir.ptr<!u8i>
+// CIR: %[[OFF:.+]] = cir.const #cir.int<8> : !s32i
+// CIR: %[[DST:.+]] = cir.ptr_stride %[[TEMP_B]], %[[OFF]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: %[[DST_D:.+]] = cir.cast bitcast %[[DST]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: cir.store %[[VAL]], %[[DST_D]] : !cir.double, !cir.ptr<!cir.double>
+// CIR: %[[YIELDED:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLowSse> -> !cir.ptr<!u8i>
+// An SSE slot is 16 bytes, so the cursor advances by 16 rather than by 8.
+// CIR: %[[STEP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: cir.store %{{.+}}, %[[FP_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[YIELDED]] : !cir.ptr<!u8i>
+
+// LLVM-LABEL: define dso_local noundef double @_Z21varargs_empty_low_sseiz(i32 noundef %{{.*}}, ...)
+// LLVM: %[[FP_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 1
+// LLVM: %[[FP:.+]] = load i32, ptr %[[FP_P]], align {{4|8|16}}
+// LLVM: %[[FITS:.+]] = icmp ule i32 %[[FP]], 160
+// LLVM: %[[VAL:.+]] = load double, ptr %{{.+}}, align 8
+// LLVM: %[[DST:.+]] = getelementptr i8, ptr %[[TEMP:.+]], i{{32|64}} 8
+// LLVM: store double %[[VAL]], ptr %[[DST]], align 8
+// LLVM: %[[BUMPED:.+]] = add i32 %[[FP]], 16
+// LLVM: store i32 %[[BUMPED]], ptr %[[FP_P]], align {{4|8|16}}
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %{{.+}}, %{{.+}} ], [ %[[TEMP]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[TEMP]], %{{.+}} ], [ %{{.+}}, %{{.+}} ]
+// LLVMCIR: load %struct.EmptyLowSse, ptr %[[ADDR]], align 8
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.+}}, ptr align 8 %[[ADDR]], i64 16, i1 false)
+
+long varargs_tail_pad(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ TailPad res = __builtin_va_arg(args, TailPad);
+ __builtin_va_end(args);
+ return res.x;
+}
+
+// With no byte offset the value lands at the temp's base.
+// CIR-LABEL: cir.func {{.*}} @_Z16varargs_tail_padiz(
+// CIR: %[[GP:.+]] = cir.load %{{.+}} : !cir.ptr<!u32i>, !u32i
+// CIR: %[[LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[GP]], %[[LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[SLOT:.+]] = cir.ptr_stride %{{.+}}, %[[GP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_TailPad>
+// CIR: %[[SLOT_I:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
+// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_I]] : !cir.ptr<!s64i>, !s64i
+// CIR: %[[TEMP_I:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_TailPad> -> !cir.ptr<!s64i>
+// CIR: cir.store %[[VAL]], %[[TEMP_I]] : !s64i, !cir.ptr<!s64i>
+// CIR: %[[YIELDED:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_TailPad> -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[YIELDED]] : !cir.ptr<!u8i>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_TailPad>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!rec_TailPad>, !rec_TailPad
+
+// LLVM-LABEL: define dso_local noundef i64 @_Z16varargs_tail_padiz(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP:.+]] = load i32, ptr %{{.+}}, align {{4|16}}
+// LLVM: %[[FITS:.+]] = icmp ule i32 %[[GP]], 40
+// LLVM: %[[VAL:.+]] = load i64, ptr %{{.+}}, align 8
+// The load is of the eightbyte the register holds, never of the whole record,
+// so no byte of the neighbouring slot reaches the result.
+// LLVM-NOT: load %struct.TailPad, ptr %[[RSA:.+]]
+// LLVMCIR: store i64 %[[VAL]], ptr %[[TEMP:.+]], align 8
+// OGCG: %[[DST:.+]] = getelementptr i8, ptr %[[TEMP:.+]], i32 0
+// OGCG: store i64 %[[VAL]], ptr %[[DST]], align 8
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %{{.+}}, %{{.+}} ], [ %[[TEMP]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[TEMP]], %{{.+}} ], [ %{{.+}}, %{{.+}} ]
+// LLVMCIR: load %struct.TailPad, ptr %[[ADDR]], align 8
+// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 8 %{{.+}}, ptr align 8 %[[ADDR]], i64 16, i1 false)
diff --git a/clang/test/CIR/CodeGen/var-arg-empty.c b/clang/test/CIR/CodeGen/var-arg-empty.c
index d2f2de8a2bb21..fef5ea743bbe7 100644
--- a/clang/test/CIR/CodeGen/var-arg-empty.c
+++ b/clang/test/CIR/CodeGen/var-arg-empty.c
@@ -1,9 +1,9 @@
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
-// RUN: FileCheck --check-prefixes=LLVM --input-file=%t-cir.ll %s
+// RUN: FileCheck --check-prefixes=LLVM,LLVMCIR --input-file=%t-cir.ll %s
// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
-// RUN: FileCheck --check-prefixes=LLVM --input-file=%t.ll %s
+// RUN: FileCheck --check-prefixes=LLVM,OGCG --input-file=%t.ll %s
struct Empty {};
@@ -27,6 +27,10 @@ struct Empty only_empty(int count, ...) {
// LLVM-LABEL: define dso_local void @only_empty(i32 noundef %{{.*}}, ...)
// LLVM-NOT: va_arg
// LLVM-NOT: getelementptr inbounds nuw %struct.__va_list_tag
+// The fetch produces a value with no bytes, which LLVMCIR initializes the
+// variable from with a zero-length copy. OGCG emits nothing at all.
+// LLVMCIR: call void @llvm.memcpy.p0.p0.i64(ptr align 1 %{{.+}}, ptr align 1 %{{.+}}, i64 0, i1 false)
+// OGCG-NOT: memcpy
// LLVM: call void @llvm.va_end.p0(
int empty_then_int(int count, ...) {
@@ -52,4 +56,5 @@ int empty_then_int(int count, ...) {
// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]]
// LLVM: icmp ule i32 %[[GP_OFFSET]], 40
+// LLVM-NOT: %struct.__va_list_tag, ptr %{{.*}}, i32 0, i32 0
// LLVM: call void @llvm.va_end.p0(
diff --git a/clang/test/CIR/CodeGen/var-arg-int128.c b/clang/test/CIR/CodeGen/var-arg-int128.c
index 6df946fa79482..70c2e693d9dea 100644
--- a/clang/test/CIR/CodeGen/var-arg-int128.c
+++ b/clang/test/CIR/CodeGen/var-arg-int128.c
@@ -57,9 +57,15 @@ __int128 varargs_int128(int count, ...) {
// LLVM: store i32 %[[GP_NEXT]], ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
// LLVM: br label %[[END_BB:.+]]
// LLVM: [[MEM_BB]]:
+// The overflow arm rounds the cursor up to 16 as well, and both the advance
+// and the fetched value start from the rounded pointer.
+// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %{{.+}} to ptr
+// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -16)
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 16
+// LLVM: store ptr %[[MEM_NEXT]], ptr %{{.+}}, align 8
// LLVM: br label %[[END_BB]]
// LLVM: [[END_BB]]:
-// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %{{.*}}, %[[MEM_BB]] ], [ %[[REG_TMP]], %[[REG_BB]] ]
-// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_TMP]], %[[REG_BB]] ], [ %{{.*}}, %[[MEM_BB]] ]
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[ALIGNED]], %[[MEM_BB]] ], [ %[[REG_TMP]], %[[REG_BB]] ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_TMP]], %[[REG_BB]] ], [ %[[ALIGNED]], %[[MEM_BB]] ]
// LLVM: %[[VA_ARG:.+]] = load i128, ptr %[[ADDR]], align 16
// LLVM: store i128 %[[VA_ARG]], ptr %{{.*}}, align 16
diff --git a/clang/test/CIR/CodeGen/var-arg-vector.c b/clang/test/CIR/CodeGen/var-arg-vector.c
index 660523be5e0e7..d9d82b73f95be 100644
--- a/clang/test/CIR/CodeGen/var-arg-vector.c
+++ b/clang/test/CIR/CodeGen/var-arg-vector.c
@@ -16,17 +16,49 @@ v4f take_16(int count, ...) {
return res;
}
-// A vector filling one 16-byte slot is fetched from the vector register area.
+// A vector filling one 16-byte slot is fetched from the vector register area,
+// at the vector cursor, and the overflow arm rounds up to the vector's own
+// 16-byte alignment.
// CIR-LABEL: cir.func {{.*}} @take_16(
// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
-// CIR: cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// The overflow arm rounds the cursor up before reading, and both the read and
+// the advance start from the rounded pointer.
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[AS_INT:.+]] = cir.cast ptr_to_int %[[OVERFLOW_B]] : !cir.ptr<!u8i> -> !u64i
+// CIR: %[[BUMP:.+]] = cir.const #cir.int<15> : !u64i
+// CIR: %[[BUMPED:.+]] = cir.add nuw %[[AS_INT]], %[[BUMP]] : !u64i
+// CIR: %[[MASK:.+]] = cir.const #cir.int<18446744073709551600> : !u64i
+// CIR: %[[ROUNDED:.+]] = cir.and %[[BUMPED]], %[[MASK]] : !u64i
+// CIR: %[[ALIGNED:.+]] = cir.cast int_to_ptr %[[ROUNDED]] : !u64i -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<16> : !s32i
+// CIR: %[[MEM_NEXT:.+]] = cir.ptr_stride %[[ALIGNED]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[MEM_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: cir.yield %[[ALIGNED]] : !cir.ptr<!u8i>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.vector<4 x !cir.float>>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!cir.vector<4 x !cir.float>>, !cir.vector<4 x !cir.float>
// LLVM-LABEL: define dso_local <4 x float> @take_16(i32 noundef %{{.*}}, ...)
// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.*}}, align {{[0-9]+}}
// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
+// LLVM: %[[RSA:.+]] = load ptr, ptr %{{.+}}, align {{[0-9]+}}
+// LLVMCIR: %[[FP64:.+]] = zext i32 %[[FP_OFFSET]] to i64
+// LLVMCIR: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[FP64]]
+// OGCG: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[FP_OFFSET]]
// LLVM: add i32 %[[FP_OFFSET]], 16
+// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %{{.+}} to ptr
+// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -16)
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 16
+// LLVM: store ptr %[[MEM_NEXT]], ptr %{{.+}}, align 8
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[ALIGNED]], %{{.+}} ], [ %[[REG_ADDR]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_ADDR]], %{{.+}} ], [ %[[ALIGNED]], %{{.+}} ]
+// LLVM: load <4 x float>, ptr %[[ADDR]], align 16
v8f take_32(int count, ...) {
__builtin_va_list args;
@@ -37,20 +69,31 @@ v8f take_32(int count, ...) {
}
// A vector too wide for one slot travels in memory, whatever vector registers
-// the target has, so there is no register arm to bound and the cursor is
+// the target has, so the fetch has no register arm at all and the cursor is
// rounded up to the vector's own 32-byte alignment.
// CIR-LABEL: cir.func {{.*}} @take_32(
// CIR-NOT: fp_offset
// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %{{.+}}[2] {name = "overflow_arg_area"}
+// CIR: %[[OVERFLOW:.+]] = cir.load %[[OVERFLOW_P]] : !cir.ptr<!cir.ptr<!void>>, !cir.ptr<!void>
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[AS_INT:.+]] = cir.cast ptr_to_int %[[OVERFLOW_B]] : !cir.ptr<!u8i> -> !u64i
+// CIR: %[[BUMPED:.+]] = cir.add nuw %[[AS_INT]], %{{.+}} : !u64i
// CIR: %[[MASK:.+]] = cir.const #cir.int<18446744073709551584> : !u64i
-// CIR: cir.and %{{.+}}, %[[MASK]] : !u64i
+// CIR: %[[ROUNDED:.+]] = cir.and %[[BUMPED]], %[[MASK]] : !u64i
+// CIR: %[[ALIGNED:.+]] = cir.cast int_to_ptr %[[ROUNDED]] : !u64i -> !cir.ptr<!u8i>
// CIR: %[[STRIDE:.+]] = cir.const #cir.int<32> : !s32i
+// CIR: %[[MEM_NEXT:.+]] = cir.ptr_stride %[[ALIGNED]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[MEM_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ALIGNED]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.vector<8 x !cir.float>>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!cir.vector<8 x !cir.float>>, !cir.vector<8 x !cir.float>
// CIR-NOT: fp_offset
// CIR: cir.va_end
// LLVM-LABEL: define dso_local <8 x float> @take_32(i32 noundef %{{.*}}, ...)
// LLVM-NOT: icmp ule i32 %{{.*}}, 160
-// LLVMCIR: and i64 %{{.+}}, -32
-// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
-// LLVM: getelementptr i8, ptr %{{.+}}, i{{32|64}} 32
+// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %{{.+}} to ptr
+// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 32
+// LLVM: store ptr %[[MEM_NEXT]], ptr %{{.+}}, align 8
+// LLVM: load <8 x float>, ptr %[[ALIGNED]], align 32
// LLVM: call void @llvm.va_end.p0(
diff --git a/clang/test/CIR/CodeGen/var_arg.c b/clang/test/CIR/CodeGen/var_arg.c
index 0f19ca293c69a..8d11fe8fa9962 100644
--- a/clang/test/CIR/CodeGen/var_arg.c
+++ b/clang/test/CIR/CodeGen/var_arg.c
@@ -1,9 +1,9 @@
// RUN: %clang_cc1 -std=c23 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-cir %s -o %t.cir
// RUN: FileCheck --input-file=%t.cir %s -check-prefix=CIR
// RUN: %clang_cc1 -std=c23 -triple x86_64-unknown-linux-gnu -Wno-unused-value -fclangir -emit-llvm %s -o %t-cir.ll
-// RUN: FileCheck --input-file=%t-cir.ll %s -check-prefix=LLVM
+// RUN: FileCheck --input-file=%t-cir.ll %s -check-prefixes=LLVM,LLVMCIR
// RUN: %clang_cc1 -std=c23 -triple x86_64-unknown-linux-gnu -Wno-unused-value -emit-llvm %s -o %t.ll
-// RUN: FileCheck --input-file=%t.ll %s -check-prefix=LLVM
+// RUN: FileCheck --input-file=%t.ll %s -check-prefixes=LLVM,OGCG
//
// C23 is required for __builtin_c23_va_start and for variadic functions with
// no named parameters. The LLVM checks are shared between the ClangIR
@@ -21,7 +21,7 @@ int varargs(int count, ...) {
return res;
}
-// `int` classifies Extend, one INTEGER eightbyte.
+// `int` is INTEGER class, one eightbyte, so the fetch gates on gp_offset.
// CIR-LABEL: cir.func {{.*}} @varargs(
// CIR: %[[RET_ADDR:.+]] = cir.alloca "__retval" {{.*}} : !cir.ptr<!s32i>
// CIR: %[[VAAREA:.+]] = cir.alloca "args" {{.*}} : !cir.ptr<!cir.array<!rec___va_list_tag x 1>>
@@ -250,9 +250,83 @@ double varargs_double(int count, ...) {
// CIR: %[[FP_OFFSET_P:.+]] = cir.get_member %{{.+}}[1] {name = "fp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
// CIR: %[[FP_OFFSET:.+]] = cir.load %[[FP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[FP_LIMIT:.+]] = cir.const #cir.int<160> : !u32i
-// CIR: cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[FP_OFFSET]], %[[FP_LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[FP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// Taking the register also spends it, so the cursor moves on by one slot.
+// CIR: %[[STEP:.+]] = cir.const #cir.int<16> : !u32i
+// CIR: %[[FP_NEXT:.+]] = cir.add %[[FP_OFFSET]], %[[STEP]] : !u32i
+// CIR: cir.store %[[FP_NEXT]], %[[FP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// The overflow arm reads at the cursor and advances it by the argument size.
+// CIR: %[[OVERFLOW_P:.+]] = cir.get_member %{{.+}}[2] {name = "overflow_arg_area"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!cir.ptr<!void>>
+// CIR: %[[OVERFLOW:.+]] = cir.load %[[OVERFLOW_P]] : !cir.ptr<!cir.ptr<!void>>, !cir.ptr<!void>
+// CIR: %[[OVERFLOW_B:.+]] = cir.cast bitcast %[[OVERFLOW]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[STRIDE:.+]] = cir.const #cir.int<8> : !s32i
+// CIR: %[[MEM_NEXT:.+]] = cir.ptr_stride %[[OVERFLOW_B]], %[[STRIDE]] : (!cir.ptr<!u8i>, !s32i) -> !cir.ptr<!u8i>
+// CIR: cir.store %[[MEM_NEXT]], %{{.+}} : !cir.ptr<!u8i>, !cir.ptr<!cir.ptr<!u8i>>
+// CIR: cir.yield %[[OVERFLOW_B]] : !cir.ptr<!u8i>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!cir.double>, !cir.double
// LLVM-LABEL: define dso_local double @varargs_double(
-// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
+// LLVM: %[[FP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 1
+// LLVM: %[[FP_OFFSET:.+]] = load i32, ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
// LLVM: icmp ule i32 %[[FP_OFFSET]], 160
-// LLVM: add i32 %[[FP_OFFSET]], 16
+// LLVM: %[[RSA:.+]] = load ptr, ptr %{{.+}}, align {{[0-9]+}}
+// LLVMCIR: %[[FP64:.+]] = zext i32 %[[FP_OFFSET]] to i64
+// LLVMCIR: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[FP64]]
+// OGCG: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[FP_OFFSET]]
+// LLVM: %[[FP_NEXT:.+]] = add i32 %[[FP_OFFSET]], 16
+// LLVM: store i32 %[[FP_NEXT]], ptr %[[FP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[OVERFLOW_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 2
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P]], align 8
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 8
+// LLVM: store ptr %[[MEM_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[OVERFLOW]], %{{.+}} ], [ %[[REG_ADDR]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_ADDR]], %{{.+}} ], [ %[[OVERFLOW]], %{{.+}} ]
+// LLVM: load double, ptr %[[ADDR]], align 8
+
+short varargs_short(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ short res = __builtin_va_arg(args, short);
+ __builtin_va_end(args);
+ return res;
+}
+
+// A type narrower than a register is sign-extended into one. That is a
+// separate classification from the `int` above, though it produces the same
+// one-register sequence.
+// CIR-LABEL: cir.func {{.*}} @varargs_short(
+// CIR: %[[GP_OFFSET_P:.+]] = cir.get_member %{{.+}}[0] {name = "gp_offset"} : !cir.ptr<!rec___va_list_tag> -> !cir.ptr<!u32i>
+// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
+// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<40> : !u32i
+// CIR: %[[FITS:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
+// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
+// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
+// CIR: %[[STEP:.+]] = cir.const #cir.int<8> : !u32i
+// CIR: %[[GP_NEXT:.+]] = cir.add %[[GP_OFFSET]], %[[STEP]] : !u32i
+// CIR: cir.store %[[GP_NEXT]], %[[GP_OFFSET_P]] : !u32i, !cir.ptr<!u32i>
+// CIR: cir.yield %[[REG_ADDR]] : !cir.ptr<!u8i>
+// CIR: %[[RESULT_P:.+]] = cir.cast bitcast %[[ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!s16i>
+// CIR: cir.load %[[RESULT_P]] : !cir.ptr<!s16i>, !s16i
+
+// LLVM-LABEL: define dso_local signext i16 @varargs_short(i32 noundef %{{.*}}, ...)
+// LLVM: %[[GP_OFFSET_P:.+]] = getelementptr inbounds nuw %struct.__va_list_tag, ptr %{{.+}}, i32 0, i32 0
+// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: icmp ule i32 %[[GP_OFFSET]], 40
+// LLVM: %[[RSA:.+]] = load ptr, ptr %{{.+}}, align {{[0-9]+}}
+// LLVMCIR: %[[GP64:.+]] = zext i32 %[[GP_OFFSET]] to i64
+// LLVMCIR: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i64 %[[GP64]]
+// OGCG: %[[REG_ADDR:.+]] = getelementptr i8, ptr %[[RSA]], i32 %[[GP_OFFSET]]
+// LLVM: %[[GP_NEXT:.+]] = add i32 %[[GP_OFFSET]], 8
+// LLVM: store i32 %[[GP_NEXT]], ptr %[[GP_OFFSET_P]], align {{[0-9]+}}
+// LLVM: %[[OVERFLOW:.+]] = load ptr, ptr %[[OVERFLOW_P:.+]], align 8
+// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[OVERFLOW]], i{{32|64}} 8
+// LLVM: store ptr %[[MEM_NEXT]], ptr %[[OVERFLOW_P]], align 8
+// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[OVERFLOW]], %{{.+}} ], [ %[[REG_ADDR]], %{{.+}} ]
+// OGCG: %[[ADDR:.+]] = phi ptr [ %[[REG_ADDR]], %{{.+}} ], [ %[[OVERFLOW]], %{{.+}} ]
+// LLVM: load i16, ptr %[[ADDR]], align 2
diff --git a/clang/test/CIR/Transforms/abi-lowering/x86_64-vaarg-valist-shape-nyi.cir b/clang/test/CIR/Transforms/abi-lowering/x86_64-vaarg-valist-shape-nyi.cir
new file mode 100644
index 0000000000000..9fe24dd68b58b
--- /dev/null
+++ b/clang/test/CIR/Transforms/abi-lowering/x86_64-vaarg-valist-shape-nyi.cir
@@ -0,0 +1,21 @@
+// RUN: not cir-opt %s -cir-call-conv-lowering=target=x86_64 2>&1 | FileCheck %s
+
+!s8i = !cir.int<s, 8>
+!s32i = !cir.int<s, 32>
+
+// The expansion reads the SysV cursor's four fields by position, and the
+// operation's type constraint admits any pointer, so the shape is checked.
+module attributes {
+ dlti.dl_spec = #dlti.dl_spec<
+ #dlti.dl_entry<i8, dense<8>: vector<2xi64>>,
+ #dlti.dl_entry<i32, dense<32>: vector<2xi64>>,
+ #dlti.dl_entry<i64, dense<64>: vector<2xi64>>,
+ #dlti.dl_entry<i128, dense<128>: vector<2xi64>>>
+} {
+ cir.func @scalar_valist(%arg0: !cir.ptr<!cir.ptr<!s8i>>) -> !s32i {
+ %0 = cir.va_arg %arg0 : (!cir.ptr<!cir.ptr<!s8i>>) -> !s32i
+ cir.return %0 : !s32i
+ }
+}
+
+// CHECK: error: 'cir.va_arg' op va_arg of a va_list that is not the four-field gp_offset / fp_offset / overflow_arg_area / reg_save_area cursor not yet implemented in CallConvLowering
diff --git a/llvm/include/llvm/ABI/FunctionInfo.h b/llvm/include/llvm/ABI/FunctionInfo.h
index caedafcbe9d22..700419cc04235 100644
--- a/llvm/include/llvm/ABI/FunctionInfo.h
+++ b/llvm/include/llvm/ABI/FunctionInfo.h
@@ -71,10 +71,12 @@ class ArgInfo {
bool ZeroExt : 1;
bool IndirectByVal : 1;
bool IndirectRealign : 1;
+ unsigned NeededIntRegs : 3;
+ unsigned NeededSseRegs : 3;
ArgInfo(Kind K = Direct)
: TheKind(K), SignExt(false), ZeroExt(false), IndirectByVal(false),
- IndirectRealign(false) {}
+ IndirectRealign(false), NeededIntRegs(0), NeededSseRegs(0) {}
public:
/// \param T The type to coerce to. If null, the argument's original type is
@@ -151,6 +153,18 @@ class ArgInfo {
return DirectAttr.Offset;
}
+ /// How many integer and vector argument registers this argument occupies.
+ /// Both zero means it occupies none and travels in memory, which is also
+ /// what a target whose classifier does not record the demand reports.
+ unsigned getNeededIntRegs() const { return NeededIntRegs; }
+ unsigned getNeededSseRegs() const { return NeededSseRegs; }
+
+ void setNeededRegs(unsigned IntRegs, unsigned SseRegs) {
+ assert(IntRegs <= 7 && SseRegs <= 7 && "Register demand does not fit");
+ NeededIntRegs = IntRegs;
+ NeededSseRegs = SseRegs;
+ }
+
MaybeAlign getDirectAlign() const {
assert((isDirect() || isExtend()) && "Not a direct or extend kind");
return Alignment;
diff --git a/llvm/lib/ABI/Targets/X86.cpp b/llvm/lib/ABI/Targets/X86.cpp
index fd7a5e15b3548..d14faab15fc27 100644
--- a/llvm/lib/ABI/Targets/X86.cpp
+++ b/llvm/lib/ABI/Targets/X86.cpp
@@ -1442,9 +1442,11 @@ void X86_64TargetInfo::computeInfo(FunctionInfo &FI) const {
if (FreeIntRegs >= NeededInt && FreeSSERegs >= NeededSSE) {
FreeIntRegs -= NeededInt;
FreeSSERegs -= NeededSSE;
+ AI.setNeededRegs(NeededInt, NeededSSE);
IT->Info = AI;
} else {
- // Not enough registers, pass on stack
+ // Not enough registers, pass on stack. The demand the classification
+ // reports is what the argument ends up occupying, which is nothing.
IT->Info = getIndirectResult(ArgTy, FreeIntRegs);
}
}
diff --git a/mlir/include/mlir/ABI/ABIRewriteContext.h b/mlir/include/mlir/ABI/ABIRewriteContext.h
index c770431913213..7760d015d8919 100644
--- a/mlir/include/mlir/ABI/ABIRewriteContext.h
+++ b/mlir/include/mlir/ABI/ABIRewriteContext.h
@@ -82,6 +82,12 @@ struct ArgClassification {
/// NO_CLASS and the value is carried in a later eightbyte (x86-64 SysV).
unsigned directOffset = 0;
+ /// How many integer and vector argument registers the value occupies. Both
+ /// zero means it travels in memory, which is also what a target whose
+ /// classifier does not record the demand reports.
+ unsigned neededIntRegs = 0;
+ unsigned neededSseRegs = 0;
+
/// Whether the value is passed as-is, so a rewriter can leave it alone.
/// Only an uncoerced Direct qualifies. Extend counts as needing a rewrite
/// even though it only adds an attribute, because the attribute changes
@@ -94,7 +100,9 @@ struct ArgClassification {
return kind == other.kind && coercedType == other.coercedType &&
indirectAlign == other.indirectAlign &&
signExtend == other.signExtend && canFlatten == other.canFlatten &&
- byVal == other.byVal && directOffset == other.directOffset;
+ byVal == other.byVal && directOffset == other.directOffset &&
+ neededIntRegs == other.neededIntRegs &&
+ neededSseRegs == other.neededSseRegs;
}
static ArgClassification getDirect() {
@@ -206,24 +214,25 @@ class ABIRewriteContext {
const FunctionClassification &fc,
OpBuilder &builder) = 0;
- /// Rewrite a single "fetch the next vararg" operation (e.g. C `va_arg`) in
- /// place to match how \p ac says the fetched type is passed at the ABI
- /// level.
+ /// Rewrite a single "fetch the next vararg" operation (e.g. C `va_arg`) to
+ /// match how \p ac says the fetched type is passed at the ABI level.
+ ///
+ /// \p ac classifies only the one type being fetched, in isolation, with the
+ /// whole register budget available. A vararg fetch advances a runtime
+ /// cursor (the platform va_list) through registers and then memory, so
+ /// whether a given fetch lands in a register depends on how much of the
+ /// budget earlier variadic arguments already consumed at run time, not on
+ /// the fetch's static position.
///
- /// This is deliberately separate from rewriteFunctionDefinition and
- /// rewriteCallSite, which classify a whole signature ahead of time. A
- /// vararg fetch instead advances a runtime cursor (the platform va_list)
- /// through registers and then memory, so whether a given fetch lands in a
- /// register depends on how much of the register budget earlier variadic
- /// arguments already consumed at run time, not on the fetch's static
- /// position. \p ac therefore classifies only the one type being fetched,
- /// in isolation.
+ /// An implementation may erase \p vaArgOp and replace its result, so a
+ /// caller walking the IR must collect the fetches before rewriting any of
+ /// them.
///
/// The default implementation reports failure, so a dialect that has not
/// implemented vararg fetches does not need to override this. An overrider
/// that fails is responsible for emitting its own diagnostic.
///
- /// \param vaArgOp The operation to rewrite in place.
+ /// \param vaArgOp The fetch to rewrite. May be erased.
/// \param ac The ABI classification of the fetched type.
/// \param builder The OpBuilder to use for modifications.
/// \returns success() if the operation was rewritten.
>From 423030d98ad4159c0ee3c8bb6a500fc1e4987431 Mon Sep 17 00:00:00 2001
From: Adam Smith <adams at nvidia.com>
Date: Tue, 15 Sep 2026 19:42:05 -0700
Subject: [PATCH 3/3] [CIR] Address review on the x86-64 va_arg expansion
rewriteVAArg had grown to 254 lines, with the register arm alone taking
105 of them inside a nested builder callback. The round-up, the overflow
arm, the register gate, the pair reassembly, the temp copy and the cursor
write-back are now named functions taking a small context struct, and the
48 and 176 register-file bounds sit in one place.
Four fixes come with it.
A va_arg of an empty record yielded a dummy value, which is an alloca and
a load of uninitialized memory. It now yields the poison the file already
hands to an ignored block argument.
The register slot size came from the integer register count while the
address it described came from whichever area the vector count selected,
so the two disagreed for a fetch needing one register of each class. Only
one class is ever in play, since a fetch needing both is a coerced pair
that takes the reassembly path, and an assert now says so. Reachable
fetches are unaffected.
The temp the register arm copies through was allocated inside the arm, and
its address left the arm through the yield. That worked only because
HoistAllocas moved it somewhere legal afterwards. Whether a fetch copies
through a temp depends on the classification and the result type rather
than on anything the arm computes, so the decision and the allocation both
happen before the ternary.
A record can need more alignment than any of its types imply: an `aligned`
attribute on a field raises the record without changing the field's type,
and a member record's raised alignment does not survive into the storage
it lowers to. Only one over-aligned shape was tested, a struct carrying
the attribute on its own declaration. Four more cover a union, a field
attribute, a member union and a packed record, and all four fail if the
record-layout lookup is removed.
Assisted-by: Cursor / claude-opus-5
---
.../TargetLowering/CIRABIRewriteContext.cpp | 503 +++++++++++-------
clang/test/CIR/CodeGen/var-arg-aggregate.c | 91 +++-
.../CIR/CodeGen/var-arg-direct-offset.cpp | 6 +-
clang/test/CIR/CodeGen/var-arg-empty.c | 14 +-
clang/test/CIR/CodeGen/var-arg-int128.c | 2 +-
5 files changed, 406 insertions(+), 210 deletions(-)
diff --git a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
index b328bb3fdbca1..eacee276bd130 100644
--- a/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
+++ b/clang/lib/CIR/Dialect/Transforms/TargetLowering/CIRABIRewriteContext.cpp
@@ -1069,19 +1069,6 @@ bool isSSERegisterClass(mlir::Type ty) {
return mlir::isa<cir::VectorType, cir::FPTypeInterface>(ty);
}
-/// The alignment in bytes the ABI gives \p ty as an argument. An alignment
-/// attribute can raise a record above what its members imply, and only the
-/// record-layout metadata carries that, so the member-derived value alone can
-/// be too small.
-uint64_t argumentAreaAlign(mlir::Type ty, mlir::ModuleOp modOp,
- const mlir::DataLayout &dl) {
- uint64_t align = dl.getTypeABIAlignment(ty);
- if (auto recTy = mlir::dyn_cast<cir::RecordType>(ty))
- if (auto layout = cir::tryGetRecordLayout(modOp, recTy.getName()))
- align = std::max<uint64_t>(align, layout.getRecordAlign());
- return align;
-}
-
} // namespace
void CIRABIRewriteContext::normalizeParameterSlotAlignments(
@@ -1533,6 +1520,247 @@ void CIRABIRewriteContext::rewriteFunctionAddress(cir::GetGlobalOp addrOp,
addrOp.getAddr().replaceAllUsesExcept(bitcast.getResult(), bitcast);
}
+namespace {
+
+/// What one `va_arg` expansion needs from the op it rewrites. The x86-64
+/// cursor fields are reached through `vaFields` by index: 0 gp_offset,
+/// 1 fp_offset, 2 overflow_arg_area, 3 reg_save_area.
+struct VAArgFetch {
+ mlir::Location loc;
+ mlir::Value valist;
+ llvm::ArrayRef<mlir::Type> vaFields;
+ mlir::Type resultTy;
+ const ArgClassification ∾
+ const mlir::DataLayout &dl;
+ mlir::ModuleOp module;
+};
+
+/// The cursor fields a register fetch reads, and the predicate saying every
+/// class it needs still has room. An offset is null when the fetch needs no
+/// register of that class.
+struct RegisterCursor {
+ mlir::Value gpOffsetP;
+ mlir::Value fpOffsetP;
+ mlir::Value gpOffset;
+ mlir::Value fpOffset;
+ mlir::Value inRegs;
+};
+
+mlir::LogicalResult reportVAArgNYI(cir::VAArgOp op, llvm::StringRef what) {
+ op->emitOpError() << "va_arg of " << what
+ << " not yet implemented in CallConvLowering";
+ return mlir::failure();
+}
+
+/// Sets \p isRegPair when a two-register argument is a coerced pair, one
+/// register per member, rather than a single wide scalar, and records in
+/// \p pairIsSse which element is SSE class. Fails on a coercion that is not
+/// two eightbytes.
+mlir::LogicalResult classifyRegisterPair(cir::VAArgOp op,
+ const ArgClassification &ac,
+ std::array<bool, 2> &pairIsSse,
+ bool &isRegPair) {
+ auto pairTy = mlir::dyn_cast<cir::RecordType>(ac.coercedType);
+ if (!pairTy)
+ return mlir::success();
+
+ if (pairTy.getNumElements() != 2)
+ return reportVAArgNYI(op, "a register coercion that is not two eightbytes");
+
+ assert(!ac.directOffset &&
+ "a pair already spans both eightbytes, so it cannot also start "
+ "partway into the value");
+ for (auto [i, memberTy] : llvm::enumerate(pairTy.getMembers()))
+ pairIsSse[i] = isSSERegisterClass(memberTy);
+ isRegPair = true;
+ return mlir::success();
+}
+
+/// The alignment an argument is placed at. A record can require more than
+/// the types of its members imply, from an `aligned` attribute on the record
+/// or on one of its fields, and neither raises the alignment of any type.
+/// Only the record layout knows, so the member-derived value alone can be too
+/// small.
+uint64_t argumentAreaAlign(mlir::Type ty, mlir::ModuleOp modOp,
+ const mlir::DataLayout &dl) {
+ uint64_t align = dl.getTypeABIAlignment(ty);
+ if (auto recTy = mlir::dyn_cast<cir::RecordType>(ty))
+ if (auto layout = cir::tryGetRecordLayout(modOp, recTy.getName()))
+ align = std::max<uint64_t>(align, layout.getRecordAlign());
+ return align;
+}
+
+/// Rounds \p bytePtr up to \p align. CIR has no pointer-mask op, so this
+/// leaves the pointer domain and comes back.
+mlir::Value roundPointerUpToAlignment(CIRBaseBuilderTy &b, mlir::Location loc,
+ mlir::Value bytePtr, uint64_t align) {
+ mlir::Type wordTy = b.getUIntNTy(64);
+ mlir::Value asInt =
+ cir::CastOp::create(b, loc, wordTy, cir::CastKind::ptr_to_int, bytePtr);
+ mlir::Value bumped =
+ b.createNUWAdd(loc, asInt, b.getConstantInt(loc, wordTy, align - 1));
+ mlir::Value rounded =
+ b.createAnd(loc, bumped, b.getConstantInt(loc, wordTy, ~(align - 1)));
+ return cir::CastOp::create(b, loc, bytePtr.getType(),
+ cir::CastKind::int_to_ptr, rounded);
+}
+
+/// Reads the argument's address out of the overflow area, which also advances
+/// the cursor past the argument.
+mlir::Value buildOverflowAddr(CIRBaseBuilderTy &b, const VAArgFetch &f) {
+ cir::IntType byteTy = b.getUIntNTy(8);
+ mlir::Value overflowP = b.createGetMember(
+ f.loc, b.getPointerTo(f.vaFields[2]), f.valist, "overflow_arg_area", 2);
+ mlir::Value overflow = b.createLoad(f.loc, overflowP);
+ mlir::Value bytePtr = b.createPtrBitcast(overflow, byteTy);
+
+ uint64_t tyAlign = argumentAreaAlign(f.resultTy, f.module, f.dl);
+ if (tyAlign > 8)
+ bytePtr = roundPointerUpToAlignment(b, f.loc, bytePtr, tyAlign);
+
+ uint64_t tySize = f.dl.getTypeSize(f.resultTy).getFixedValue();
+ uint64_t stride = (tySize + 7) & ~UINT64_C(7);
+ mlir::Value strideVal = b.getSignedInt(f.loc, stride, 32);
+ mlir::Value next = b.createPtrStride(f.loc, bytePtr, strideVal);
+ b.createStore(f.loc, next, overflowP);
+ return bytePtr;
+}
+
+/// Loads the cursor offsets and builds the predicate that sends the fetch to
+/// the register-save area. Both offsets count from the start of that area, so
+/// the integer limit is the six GP registers at 48 and the vector limit is
+/// those plus the eight SSE registers at 176.
+RegisterCursor buildRegisterGate(CIRBaseBuilderTy &b, const VAArgFetch &f,
+ unsigned neededInt, unsigned neededSse) {
+ RegisterCursor cursor;
+ if (neededInt) {
+ cursor.gpOffsetP = b.createGetMember(f.loc, b.getPointerTo(f.vaFields[0]),
+ f.valist, "gp_offset", 0);
+ cursor.gpOffset = b.createLoad(f.loc, cursor.gpOffsetP);
+ mlir::Value limit =
+ b.getConstantInt(f.loc, cursor.gpOffset.getType(), 48 - neededInt * 8);
+ cursor.inRegs =
+ b.createCompare(f.loc, cir::CmpOpKind::le, cursor.gpOffset, limit);
+ }
+ if (neededSse) {
+ cursor.fpOffsetP = b.createGetMember(f.loc, b.getPointerTo(f.vaFields[1]),
+ f.valist, "fp_offset", 1);
+ cursor.fpOffset = b.createLoad(f.loc, cursor.fpOffsetP);
+ mlir::Value limit = b.getConstantInt(f.loc, cursor.fpOffset.getType(),
+ 176 - neededSse * 16);
+ mlir::Value fitsInFp =
+ b.createCompare(f.loc, cir::CmpOpKind::le, cursor.fpOffset, limit);
+ cursor.inRegs = cursor.inRegs
+ ? b.createLogicalAnd(f.loc, cursor.inRegs, fitsInFp)
+ : fitsInFp;
+ }
+ return cursor;
+}
+
+/// Copies each half of a non-contiguous pair out of the register-save area
+/// into \p regPairTemp, laid out as the coerced pair.
+void reassembleRegisterPair(CIRBaseBuilderTy &b, mlir::Location loc,
+ const ArgClassification &ac,
+ const RegisterCursor &cursor,
+ llvm::ArrayRef<bool> pairIsSse,
+ mlir::Value regSaveArea, mlir::Value regPairTemp) {
+ auto pairTy = mlir::cast<cir::RecordType>(ac.coercedType);
+ // SSE slots sit 16 bytes apart within their own area, so track how many of
+ // each class came before.
+ unsigned seenOfClass[2] = {0, 0};
+ for (unsigned i = 0; i < 2; ++i) {
+ bool isSse = pairIsSse[i];
+ mlir::Value base = isSse ? cursor.fpOffset : cursor.gpOffset;
+ unsigned regSize = isSse ? 16 : 8;
+ unsigned prior = seenOfClass[isSse]++;
+ mlir::Value off = base;
+ if (prior) {
+ off = b.createAdd(loc, base,
+ b.getConstantInt(loc, base.getType(), prior * regSize));
+ }
+ mlir::Value src = b.createPtrStride(loc, regSaveArea, off);
+ mlir::Type elemTy = pairTy.getElementType(i);
+ mlir::Value val = b.createLoad(loc, b.createPtrBitcast(src, elemTy));
+ b.createStore(
+ loc, val,
+ b.createGetMember(loc, b.getPointerTo(elemTy), regPairTemp, "", i));
+ }
+}
+
+/// The bytes carried by the registers of this fetch's one class.
+uint64_t registerSlotSize(unsigned neededInt, unsigned neededSse) {
+ assert(!(neededInt && neededSse) &&
+ "a fetch needing both classes is a pair, reassembled elsewhere");
+ return neededSse ? neededSse * 16 : neededInt * 8;
+}
+
+/// Whether the register cannot be read in place, either because it carries
+/// less than the whole result or because its slot is under-aligned for it.
+bool needsTempCopy(const VAArgFetch &f, unsigned neededInt,
+ unsigned neededSse) {
+ uint64_t tySize = f.dl.getTypeSize(f.resultTy).getFixedValue();
+ if (f.ac.coercedType &&
+ (f.ac.directOffset || registerSlotSize(neededInt, neededSse) < tySize))
+ return true;
+ // A GP register slot is only 8-byte aligned, so a wider-aligned GP-class
+ // fetch (e.g. __int128, align 16) is copied through a temp. SSE slots are
+ // already 16-byte aligned and need no copy.
+ return !neededSse && argumentAreaAlign(f.resultTy, f.module, f.dl) > 8;
+}
+
+/// Copies what the registers carry into \p temp, which is the size of the
+/// whole result, and returns the address to read the result from.
+mlir::Value copyRegisterToTemp(CIRBaseBuilderTy &b, mlir::Location loc,
+ const VAArgFetch &f, mlir::Value regAddr,
+ mlir::Value temp, unsigned neededInt,
+ unsigned neededSse) {
+ cir::IntType byteTy = b.getUIntNTy(8);
+ uint64_t tySize = f.dl.getTypeSize(f.resultTy).getFixedValue();
+
+ if (f.ac.coercedType &&
+ (f.ac.directOffset || registerSlotSize(neededInt, neededSse) < tySize)) {
+ // The registers carry less than the whole result, either because the
+ // eightbytes below directOffset hold no field or because the result is
+ // wider than the registers carrying it. Copy only what they carry, so
+ // that reading the temp cannot run on into the neighboring slot.
+ mlir::Value val = b.createAlignedLoad(
+ loc, b.createPtrBitcast(regAddr, f.ac.coercedType), 8);
+ mlir::Value dst = temp;
+ if (f.ac.directOffset) {
+ dst = b.createPtrStride(loc, b.createPtrBitcast(temp, byteTy),
+ b.getSignedInt(loc, f.ac.directOffset, 32));
+ }
+ b.createStore(loc, val, b.createPtrBitcast(dst, f.ac.coercedType));
+ return b.createPtrBitcast(temp, byteTy);
+ }
+
+ mlir::Value val =
+ b.createAlignedLoad(loc, b.createPtrBitcast(regAddr, f.resultTy), 8);
+ b.createStore(loc, val, temp);
+ return b.createPtrBitcast(temp, byteTy);
+}
+
+void advanceRegisterCursors(CIRBaseBuilderTy &b, mlir::Location loc,
+ const RegisterCursor &cursor, unsigned neededInt,
+ unsigned neededSse) {
+ if (neededInt) {
+ b.createStore(loc,
+ b.createAdd(loc, cursor.gpOffset,
+ b.getConstantInt(loc, cursor.gpOffset.getType(),
+ neededInt * 8)),
+ cursor.gpOffsetP);
+ }
+ if (neededSse) {
+ b.createStore(loc,
+ b.createAdd(loc, cursor.fpOffset,
+ b.getConstantInt(loc, cursor.fpOffset.getType(),
+ neededSse * 16)),
+ cursor.fpOffsetP);
+ }
+}
+
+} // namespace
+
mlir::LogicalResult
CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
const ArgClassification &ac,
@@ -1543,27 +1771,19 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
mlir::Type resultTy = op.getType();
mlir::Value valist = op.getArgList();
- auto reportNYI = [&](llvm::StringRef what) {
- op->emitOpError() << "va_arg of " << what
- << " not yet implemented in CallConvLowering";
- return mlir::failure();
- };
-
- // An ignored type is passed in no register and no stack slot, so the fetch
- // has nothing to read and must leave the va_list cursor where it found it,
- // for the fetches that come after this one. The type holds no bytes, so
- // the value the fetch produces carries no information either.
+ // An ignored type travels in no register and no stack slot, so the fetch
+ // reads nothing and must leave the cursor unchanged. The value holds no
+ // bytes, so poison stands in for it.
if (ac.kind == ArgKind::Ignore) {
builder.setInsertionPoint(op);
- op.getResult().replaceAllUsesWith(builder.createDummyValue(
- loc, resultTy,
- clang::CharUnits::fromQuantity(dl.getTypeABIAlignment(resultTy))));
+ op.getResult().replaceAllUsesWith(
+ createIgnoredValue(builder, loc, resultTy));
op->erase();
return mlir::success();
}
if (ac.kind == ArgKind::Indirect && !ac.byVal)
- return reportNYI("a non-trivially-copyable type");
+ return reportVAArgNYI(op, "a non-trivially-copyable type");
// neededInt counts 8-byte integer slots and neededSse counts 16-byte vector
// slots. Zero of both means the type travels in memory and is read
@@ -1575,85 +1795,33 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
// than INTEGER class. Only meaningful when isRegPair is set.
std::array<bool, 2> pairIsSse = {false, false};
bool isRegPair = false;
-
- // A value needing two registers is either a coerced pair, one register per
- // member, or a single wide scalar read from one address.
if (ac.kind == ArgKind::Direct && neededInt + neededSse == 2 &&
ac.coercedType) {
- if (auto pairTy = mlir::dyn_cast<cir::RecordType>(ac.coercedType)) {
- if (pairTy.getNumElements() != 2)
- return reportNYI("a register coercion that is not two eightbytes");
- assert(!ac.directOffset &&
- "a pair already spans both eightbytes, so it cannot also start "
- "partway into the value");
- for (auto [i, memberTy] : llvm::enumerate(pairTy.getMembers()))
- pairIsSse[i] = isSSERegisterClass(memberTy);
- isRegPair = true;
- }
+ if (classifyRegisterPair(op, ac, pairIsSse, isRegPair).failed())
+ return mlir::failure();
}
auto vaListRecTy = mlir::dyn_cast<cir::RecordType>(
mlir::cast<cir::PointerType>(valist.getType()).getPointee());
if (!vaListRecTy || vaListRecTy.getNumElements() != 4) {
- return reportNYI("a va_list that is not the four-field gp_offset / "
- "fp_offset / overflow_arg_area / reg_save_area cursor");
+ return reportVAArgNYI(op,
+ "a va_list that is not the four-field gp_offset / "
+ "fp_offset / overflow_arg_area / reg_save_area "
+ "cursor");
}
- llvm::ArrayRef<mlir::Type> vaFields = vaListRecTy.getMembers();
+
+ const VAArgFetch fetch{loc, valist, vaListRecTy.getMembers(), resultTy, ac,
+ dl, module};
cir::IntType byteTy = builder.getUIntNTy(8);
- cir::PointerType bytePtrTy = builder.getPointerTo(byteTy);
builder.setInsertionPoint(op);
- // Reading the overflow area also advances the cursor past the argument.
- auto buildMemAddr = [&](CIRBaseBuilderTy &b) -> mlir::Value {
- mlir::Value overflowP = b.createGetMember(loc, b.getPointerTo(vaFields[2]),
- valist, "overflow_arg_area", 2);
- mlir::Value overflow = b.createLoad(loc, overflowP);
- mlir::Value bytePtr = b.createPtrBitcast(overflow, byteTy);
- uint64_t tyAlign = argumentAreaAlign(resultTy, module, dl);
- if (tyAlign > 8) {
- mlir::Type wordTy = b.getUIntNTy(64);
- mlir::Value asInt = cir::CastOp::create(
- b, loc, wordTy, cir::CastKind::ptr_to_int, bytePtr);
- mlir::Value bumped = b.createNUWAdd(
- loc, asInt, b.getConstantInt(loc, wordTy, tyAlign - 1));
- mlir::Value rounded = b.createAnd(
- loc, bumped, b.getConstantInt(loc, wordTy, ~(tyAlign - 1)));
- bytePtr = cir::CastOp::create(b, loc, bytePtrTy,
- cir::CastKind::int_to_ptr, rounded);
- }
- uint64_t tySize = dl.getTypeSize(resultTy).getFixedValue();
- uint64_t stride = (tySize + 7) & ~UINT64_C(7);
- mlir::Value strideVal = b.getSignedInt(loc, stride, 32);
- mlir::Value next = b.createPtrStride(loc, bytePtr, strideVal);
- b.createStore(loc, next, overflowP);
- return bytePtr;
- };
-
mlir::Value addr;
if (neededInt == 0 && neededSse == 0) {
- addr = buildMemAddr(builder);
+ addr = buildOverflowAddr(builder, fetch);
} else {
- mlir::Value gpOffsetP, fpOffsetP, gpOffset, fpOffset, inRegs;
- if (neededInt) {
- gpOffsetP = builder.createGetMember(
- loc, builder.getPointerTo(vaFields[0]), valist, "gp_offset", 0);
- gpOffset = builder.createLoad(loc, gpOffsetP);
- mlir::Value limit =
- builder.getConstantInt(loc, gpOffset.getType(), 48 - neededInt * 8);
- inRegs = builder.createCompare(loc, cir::CmpOpKind::le, gpOffset, limit);
- }
- if (neededSse) {
- fpOffsetP = builder.createGetMember(
- loc, builder.getPointerTo(vaFields[1]), valist, "fp_offset", 1);
- fpOffset = builder.createLoad(loc, fpOffsetP);
- mlir::Value limit =
- builder.getConstantInt(loc, fpOffset.getType(), 176 - neededSse * 16);
- mlir::Value fitsInFp =
- builder.createCompare(loc, cir::CmpOpKind::le, fpOffset, limit);
- inRegs =
- inRegs ? builder.createLogicalAnd(loc, inRegs, fitsInFp) : fitsInFp;
- }
+ RegisterCursor cursor =
+ buildRegisterGate(builder, fetch, neededInt, neededSse);
// A two-eightbyte pair that is purely INTEGER class is contiguous in the
// register-save area, since GP slots are 8-byte packed, so the address of
@@ -1662,11 +1830,13 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
// spaced and a mixed pair's halves live in disjoint areas, so each half is
// copied into a temp laid out as the coerced pair.
bool pairNeedsReassembly = isRegPair && neededSse != 0;
+
+ // Both temps are allocated here, since a ternary arm yields their address.
mlir::Value regPairTemp;
if (pairNeedsReassembly) {
// The temp is written one member at a time through the coerced pair, so
// it has to meet that type's alignment, and read back as the result, so
- // it has to meet the result's declared one too.
+ // it has to meet the result's alignment too.
regPairTemp = builder.createAlloca(
loc, builder.getPointerTo(ac.coercedType), "vaarg.reg",
clang::CharUnits::fromQuantity(
@@ -1674,111 +1844,50 @@ CIRABIRewriteContext::rewriteVAArg(mlir::Operation *vaArgOp,
argumentAreaAlign(resultTy, module, dl))));
}
- addr =
- cir::TernaryOp::create(
- builder, loc, inRegs,
- /*trueBuilder=*/
- [&](mlir::OpBuilder &ob, mlir::Location l) {
- CIRBaseBuilderTy b(ob);
- mlir::Value regSaveArea = b.createLoad(
- l, b.createGetMember(l, b.getPointerTo(vaFields[3]), valist,
- "reg_save_area", 3));
- regSaveArea = b.createPtrBitcast(regSaveArea, byteTy);
-
- mlir::Value regAddr;
- if (pairNeedsReassembly) {
- auto pairTy = mlir::cast<cir::RecordType>(ac.coercedType);
- // Same-class eightbytes sit regSize bytes apart within
- // their own area (8 for GP, 16 for SSE), so track how
- // many of each class came before.
- unsigned seenOfClass[2] = {0, 0};
- for (unsigned i = 0; i < 2; ++i) {
- bool isSse = pairIsSse[i];
- mlir::Value base = isSse ? fpOffset : gpOffset;
- unsigned regSize = isSse ? 16 : 8;
- unsigned prior = seenOfClass[isSse]++;
- mlir::Value off = base;
- if (prior) {
- off = b.createAdd(
- l, base,
- b.getConstantInt(l, base.getType(), prior * regSize));
- }
- mlir::Value src = b.createPtrStride(l, regSaveArea, off);
- mlir::Type elemTy = pairTy.getElementType(i);
- mlir::Value val =
- b.createLoad(l, b.createPtrBitcast(src, elemTy));
- b.createStore(l, val,
- b.createGetMember(l, b.getPointerTo(elemTy),
- regPairTemp, "", i));
- }
- regAddr = b.createPtrBitcast(regPairTemp, byteTy);
- } else {
- mlir::Value off = neededSse ? fpOffset : gpOffset;
- regAddr = b.createPtrStride(l, regSaveArea, off);
- uint64_t resultAlign = argumentAreaAlign(resultTy, module, dl);
- uint64_t regSize = neededInt ? neededInt * 8 : 16;
- uint64_t tySize = dl.getTypeSize(resultTy).getFixedValue();
- if (ac.coercedType && (ac.directOffset || regSize < tySize)) {
- // The registers carry less than the whole result, either
- // because the eightbytes below directOffset hold no field or
- // because the result is wider than the registers carrying
- // it. Copy what they do carry into a full-size temp, so the
- // load below reads a whole result rather than reading on
- // into the neighbouring slot.
- mlir::Value temp = b.createAlloca(
- l, b.getPointerTo(resultTy), "vaarg.reg",
- clang::CharUnits::fromQuantity(resultAlign));
- mlir::Value val = b.createAlignedLoad(
- l, b.createPtrBitcast(regAddr, ac.coercedType), 8);
- mlir::Value dst = temp;
- if (ac.directOffset) {
- dst = b.createPtrStride(
- l, b.createPtrBitcast(temp, byteTy),
- b.getSignedInt(l, ac.directOffset, 32));
- }
- b.createStore(l, val,
- b.createPtrBitcast(dst, ac.coercedType));
- regAddr = b.createPtrBitcast(temp, byteTy);
- } else if (!neededSse && resultAlign > 8) {
- // A GP register slot is only 8-byte aligned, so a
- // wider-aligned GP-class fetch (e.g. __int128, align 16)
- // is copied through a temp. SSE slots are already
- // 16-byte aligned and need no copy.
- mlir::Value temp = b.createAlloca(
- l, b.getPointerTo(resultTy), "vaarg.reg",
- clang::CharUnits::fromQuantity(resultAlign));
- mlir::Value val = b.createAlignedLoad(
- l, b.createPtrBitcast(regAddr, resultTy), 8);
- b.createStore(l, val, temp);
- regAddr = b.createPtrBitcast(temp, byteTy);
- }
- }
-
- if (neededInt) {
- b.createStore(
- l,
- b.createAdd(
- l, gpOffset,
- b.getConstantInt(l, gpOffset.getType(), neededInt * 8)),
- gpOffsetP);
- }
- if (neededSse) {
- b.createStore(
- l,
- b.createAdd(l, fpOffset,
- b.getConstantInt(l, fpOffset.getType(),
- neededSse * 16)),
- fpOffsetP);
- }
-
- cir::YieldOp::create(b, l, regAddr);
- },
- /*falseBuilder=*/
- [&](mlir::OpBuilder &ob, mlir::Location l) {
- CIRBaseBuilderTy b(ob);
- cir::YieldOp::create(b, l, buildMemAddr(b));
- })
- .getResult();
+ mlir::Value regTemp;
+ bool copyThroughTemp =
+ !pairNeedsReassembly && needsTempCopy(fetch, neededInt, neededSse);
+ if (copyThroughTemp) {
+ regTemp =
+ builder.createAlloca(loc, builder.getPointerTo(resultTy), "vaarg.reg",
+ clang::CharUnits::fromQuantity(
+ argumentAreaAlign(resultTy, module, dl)));
+ }
+
+ addr = cir::TernaryOp::create(
+ builder, loc, cursor.inRegs,
+ /*trueBuilder=*/
+ [&](mlir::OpBuilder &ob, mlir::Location l) {
+ CIRBaseBuilderTy b(ob);
+ mlir::Value regSaveArea = b.createLoad(
+ l, b.createGetMember(l, b.getPointerTo(fetch.vaFields[3]),
+ valist, "reg_save_area", 3));
+ regSaveArea = b.createPtrBitcast(regSaveArea, byteTy);
+
+ mlir::Value regAddr;
+ if (pairNeedsReassembly) {
+ reassembleRegisterPair(b, l, ac, cursor, pairIsSse,
+ regSaveArea, regPairTemp);
+ regAddr = b.createPtrBitcast(regPairTemp, byteTy);
+ } else {
+ mlir::Value off =
+ neededSse ? cursor.fpOffset : cursor.gpOffset;
+ regAddr = b.createPtrStride(l, regSaveArea, off);
+ if (copyThroughTemp) {
+ regAddr = copyRegisterToTemp(b, l, fetch, regAddr, regTemp,
+ neededInt, neededSse);
+ }
+ }
+
+ advanceRegisterCursors(b, l, cursor, neededInt, neededSse);
+ cir::YieldOp::create(b, l, regAddr);
+ },
+ /*falseBuilder=*/
+ [&](mlir::OpBuilder &ob, mlir::Location l) {
+ CIRBaseBuilderTy b(ob);
+ cir::YieldOp::create(b, l, buildOverflowAddr(b, fetch));
+ })
+ .getResult();
}
mlir::Value result =
diff --git a/clang/test/CIR/CodeGen/var-arg-aggregate.c b/clang/test/CIR/CodeGen/var-arg-aggregate.c
index 7fe8b336d1237..6fcff9188bbd6 100644
--- a/clang/test/CIR/CodeGen/var-arg-aggregate.c
+++ b/clang/test/CIR/CodeGen/var-arg-aggregate.c
@@ -369,16 +369,16 @@ struct OverAligned varargs_aggregate_overaligned(int count, ...) {
}
// An alignment attribute raises the record above what its two long members
-// imply, and only the record layout records that. Both arms have to honor
-// it: the register arm copies through a 16-byte-aligned temp, and the
-// overflow arm rounds the cursor up to 16 before reading.
+// imply, and only the record layout carries that. Both arms have to honor it:
+// the register arm copies through a 16-byte-aligned temp, and the overflow arm
+// rounds the cursor up to 16 before reading.
// CIR-LABEL: cir.func {{.*}} @varargs_aggregate_overaligned(
// CIR: %[[GP_OFFSET:.+]] = cir.load %{{.+}} : !cir.ptr<!u32i>, !u32i
// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<32> : !u32i
// CIR: %[[FITS:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(16) : !cir.ptr<!rec_OverAligned>
// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %{{.+}}, %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
-// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(16) : !cir.ptr<!rec_OverAligned>
// CIR: %[[REG_ADDR_V:.+]] = cir.cast bitcast %[[REG_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!rec_OverAligned>
// CIR: %[[VAL:.+]] = cir.load align(8) %[[REG_ADDR_V]] : !cir.ptr<!rec_OverAligned>, !rec_OverAligned
// CIR: cir.store %[[VAL]], %[[TEMP]] : !rec_OverAligned, !cir.ptr<!rec_OverAligned>
@@ -393,12 +393,91 @@ struct OverAligned varargs_aggregate_overaligned(int count, ...) {
// LLVM: %[[GP_OFFSET:.+]] = load i32, ptr %{{.+}}, align {{[0-9]+}}
// LLVM: icmp ule i32 %[[GP_OFFSET]], 32
// The register slot is only 8-aligned, so the copy reads at 8 and stores
-// into a temp carrying the record's declared 16.
+// into a temp carrying the record's declared 16. The temp is bound at its
+// alloca, so that an under-aligned one cannot satisfy the store below.
+// LLVMCIR: %[[TEMP:.+]] = alloca %struct.OverAligned, align 16
// LLVMCIR: %[[VAL:.+]] = load %struct.OverAligned, ptr %{{.+}}, align 8
-// LLVMCIR: store %struct.OverAligned %[[VAL]], ptr %[[TEMP:.+]], align 8
+// LLVMCIR: store %struct.OverAligned %[[VAL]], ptr %[[TEMP]], align 8
// OGCG: call void @llvm.memcpy.p0.p0.i64(ptr align 16 %[[TEMP:.+]], ptr align 8 %{{.+}}, i64 16, i1 false)
// LLVMCIR: %[[ALIGNED:.+]] = inttoptr i64 %{{.+}} to ptr
// OGCG: %[[ALIGNED:.+]] = call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -16)
// LLVM: %[[MEM_NEXT:.+]] = getelementptr i8, ptr %[[ALIGNED]], i{{32|64}} 16
// LLVMCIR: %[[ADDR:.+]] = phi ptr [ %[[ALIGNED]], %{{.+}} ], [ %[[TEMP]], %{{.+}} ]
// OGCG: %[[ADDR:.+]] = phi ptr [ %[[TEMP]], %{{.+}} ], [ %[[ALIGNED]], %{{.+}} ]
+
+// A record can require more alignment than the types of its members imply,
+// and the attribute that raises it need not sit on the record: it can sit on a
+// field, or on a member record whose own storage is only byte-aligned. None
+// of those raise the alignment of any type, so the fetch takes the alignment
+// from the record layout. Each of these reads from the overflow area, since
+// 32 bytes is too large for registers.
+
+union OverAlignedUnion {
+ char buf[32];
+} __attribute__((aligned(32)));
+
+long varargs_overaligned_union(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ union OverAlignedUnion res = __builtin_va_arg(args, union OverAlignedUnion);
+ __builtin_va_end(args);
+ return res.buf[0];
+}
+
+// LLVM-LABEL: define dso_local {{.*}} @varargs_overaligned_union(
+// LLVMCIR: and i64 %{{.+}}, -32
+// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
+
+struct FieldOverAligned {
+ __attribute__((aligned(32))) char c;
+ char rest[31];
+};
+
+long varargs_field_overaligned(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct FieldOverAligned res = __builtin_va_arg(args, struct FieldOverAligned);
+ __builtin_va_end(args);
+ return res.c;
+}
+
+// LLVM-LABEL: define dso_local {{.*}} @varargs_field_overaligned(
+// LLVMCIR: and i64 %{{.+}}, -32
+// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
+
+struct HoldsOverAlignedUnion {
+ union OverAlignedUnion u;
+};
+
+long varargs_holds_overaligned_union(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct HoldsOverAlignedUnion res =
+ __builtin_va_arg(args, struct HoldsOverAlignedUnion);
+ __builtin_va_end(args);
+ return res.u.buf[0];
+}
+
+// LLVM-LABEL: define dso_local {{.*}} @varargs_holds_overaligned_union(
+// LLVMCIR: and i64 %{{.+}}, -32
+// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
+
+// Packing lowers the alignment the members imply, and an attribute can still
+// raise the record above it.
+struct __attribute__((packed, aligned(32))) PackedOverAligned {
+ char c;
+ int i;
+};
+
+long varargs_packed_overaligned(int count, ...) {
+ __builtin_va_list args;
+ __builtin_va_start(args, count);
+ struct PackedOverAligned res =
+ __builtin_va_arg(args, struct PackedOverAligned);
+ __builtin_va_end(args);
+ return res.i;
+}
+
+// LLVM-LABEL: define dso_local {{.*}} @varargs_packed_overaligned(
+// LLVMCIR: and i64 %{{.+}}, -32
+// OGCG: call ptr @llvm.ptrmask.p0.i64(ptr %{{.+}}, i64 -32)
diff --git a/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp b/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
index bc545dd7bc507..59d95b5ab96f3 100644
--- a/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
+++ b/clang/test/CIR/CodeGen/var-arg-direct-offset.cpp
@@ -40,10 +40,10 @@ EmptyLow varargs_empty_low(int count, ...) {
// CIR: %[[GP:.+]] = cir.load %[[GP_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[LIMIT:.+]] = cir.const #cir.int<40> : !u32i
// CIR: %[[FITS:.+]] = cir.cmp le %[[GP]], %[[LIMIT]] : !u32i
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLow>
// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
// CIR: %[[RSA_B:.+]] = cir.cast bitcast %{{.+}} : !cir.ptr<!void> -> !cir.ptr<!u8i>
// CIR: %[[SLOT:.+]] = cir.ptr_stride %[[RSA_B]], %[[GP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
-// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLow>
// CIR: %[[SLOT_I:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_I]] : !cir.ptr<!s64i>, !s64i
// CIR: %[[TEMP_B:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLow> -> !cir.ptr<!u8i>
@@ -97,9 +97,9 @@ double varargs_empty_low_sse(int count, ...) {
// CIR: %[[FP:.+]] = cir.load %[[FP_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[LIMIT:.+]] = cir.const #cir.int<160> : !u32i
// CIR: %[[FITS:.+]] = cir.cmp le %[[FP]], %[[LIMIT]] : !u32i
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLowSse>
// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
// CIR: %[[SLOT:.+]] = cir.ptr_stride %{{.+}}, %[[FP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
-// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_EmptyLowSse>
// CIR: %[[SLOT_D:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!cir.double>
// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_D]] : !cir.ptr<!cir.double>, !cir.double
// CIR: %[[TEMP_B:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_EmptyLowSse> -> !cir.ptr<!u8i>
@@ -140,9 +140,9 @@ long varargs_tail_pad(int count, ...) {
// CIR: %[[GP:.+]] = cir.load %{{.+}} : !cir.ptr<!u32i>, !u32i
// CIR: %[[LIMIT:.+]] = cir.const #cir.int<40> : !u32i
// CIR: %[[FITS:.+]] = cir.cmp le %[[GP]], %[[LIMIT]] : !u32i
+// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_TailPad>
// CIR: %[[ADDR:.+]] = cir.ternary(%[[FITS]], true {
// CIR: %[[SLOT:.+]] = cir.ptr_stride %{{.+}}, %[[GP]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
-// CIR: %[[TEMP:.+]] = cir.alloca "vaarg.reg" align(8) : !cir.ptr<!rec_TailPad>
// CIR: %[[SLOT_I:.+]] = cir.cast bitcast %[[SLOT]] : !cir.ptr<!u8i> -> !cir.ptr<!s64i>
// CIR: %[[VAL:.+]] = cir.load align(8) %[[SLOT_I]] : !cir.ptr<!s64i>, !s64i
// CIR: %[[TEMP_I:.+]] = cir.cast bitcast %[[TEMP]] : !cir.ptr<!rec_TailPad> -> !cir.ptr<!s64i>
diff --git a/clang/test/CIR/CodeGen/var-arg-empty.c b/clang/test/CIR/CodeGen/var-arg-empty.c
index fef5ea743bbe7..fa4e0c092d01f 100644
--- a/clang/test/CIR/CodeGen/var-arg-empty.c
+++ b/clang/test/CIR/CodeGen/var-arg-empty.c
@@ -16,19 +16,27 @@ struct Empty only_empty(int count, ...) {
}
// An empty record travels in no register and no stack slot, so fetching one
-// reads no argument and advances no field of the cursor.
+// reads no argument, advances no field of the cursor, and yields poison.
// CIR-LABEL: cir.func {{.*}} @only_empty(
// CIR-NOT: cir.va_arg
// CIR-NOT: gp_offset
// CIR-NOT: fp_offset
// CIR-NOT: overflow_arg_area
+// CIR: %{{.+}} = cir.const #cir.poison : !rec_Empty
+// The negatives are restated because the match above ends their range.
+// CIR-NOT: cir.va_arg
+// CIR-NOT: gp_offset
+// CIR-NOT: fp_offset
+// CIR-NOT: overflow_arg_area
// CIR: cir.va_end
// LLVM-LABEL: define dso_local void @only_empty(i32 noundef %{{.*}}, ...)
// LLVM-NOT: va_arg
// LLVM-NOT: getelementptr inbounds nuw %struct.__va_list_tag
-// The fetch produces a value with no bytes, which LLVMCIR initializes the
-// variable from with a zero-length copy. OGCG emits nothing at all.
+// LLVMCIR: store %struct.Empty poison, ptr %{{.+}}, align 1
+// LLVMCIR-NOT: getelementptr inbounds nuw %struct.__va_list_tag
+// The variable is initialized from that value with a copy of no bytes. OGCG
+// emits nothing at all.
// LLVMCIR: call void @llvm.memcpy.p0.p0.i64(ptr align 1 %{{.+}}, ptr align 1 %{{.+}}, i64 0, i1 false)
// OGCG-NOT: memcpy
// LLVM: call void @llvm.va_end.p0(
diff --git a/clang/test/CIR/CodeGen/var-arg-int128.c b/clang/test/CIR/CodeGen/var-arg-int128.c
index 70c2e693d9dea..3f685b015c217 100644
--- a/clang/test/CIR/CodeGen/var-arg-int128.c
+++ b/clang/test/CIR/CodeGen/var-arg-int128.c
@@ -20,11 +20,11 @@ __int128 varargs_int128(int count, ...) {
// CIR: %[[GP_OFFSET:.+]] = cir.load %[[GP_OFFSET_P]] : !cir.ptr<!u32i>, !u32i
// CIR: %[[GP_LIMIT:.+]] = cir.const #cir.int<32> : !u32i
// CIR: %[[FITS_GP:.+]] = cir.cmp le %[[GP_OFFSET]], %[[GP_LIMIT]] : !u32i
+// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!s128i>
// CIR: %[[VA_ARG:.+]] = cir.ternary(%[[FITS_GP]], true {
// CIR: %[[REG_SAVE:.+]] = cir.load %{{.+}}
// CIR: %[[REG_SAVE_B:.+]] = cir.cast bitcast %[[REG_SAVE]] : !cir.ptr<!void> -> !cir.ptr<!u8i>
// CIR: %[[REG_ADDR:.+]] = cir.ptr_stride %[[REG_SAVE_B]], %[[GP_OFFSET]] : (!cir.ptr<!u8i>, !u32i) -> !cir.ptr<!u8i>
-// CIR: %[[REG_TMP:.+]] = cir.alloca "vaarg.reg" {{.*}} : !cir.ptr<!s128i>
// CIR: %[[REG_ADDR_V:.+]] = cir.cast bitcast %[[REG_ADDR]] : !cir.ptr<!u8i> -> !cir.ptr<!s128i>
// CIR: %[[REG_VAL:.+]] = cir.load align(8) %[[REG_ADDR_V]] : !cir.ptr<!s128i>, !s128i
// CIR: cir.store %[[REG_VAL]], %[[REG_TMP]] : !s128i, !cir.ptr<!s128i>
More information about the llvm-commits
mailing list