[clang] [Clang][CodeGen][X86] Fix dropped union bytes in SSE register coercion (PR #227229)

via cfe-commits cfe-commits at lists.llvm.org
Tue Sep 29 02:10:44 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-clang

Author: MaoJian (MaooJian)

<details>
<summary>Changes</summary>

A union is lowered to the IR type of its largest member, so bytes that are padding in that member are invisible to the IR-type walk in GetSSETypeAtOffset. A smaller member can have a field in those bytes, and the narrow coercion it picks then drops them when the union is passed or returned by value in xmm registers. Check BitsContainNoUserData on the source type before returning a narrow type, as GetINTEGERTypeAtOffset already does, and fall back to double so the full eightbyte is transferred.

Fixes #<!-- -->76017

---
Full diff: https://github.com/llvm/llvm-project/pull/227229.diff


2 Files Affected:

- (modified) clang/lib/CodeGen/Targets/X86.cpp (+22-2) 
- (added) clang/test/CodeGen/X86/x86_64-union-sse-abi.c (+123) 


``````````diff
diff --git a/clang/lib/CodeGen/Targets/X86.cpp b/clang/lib/CodeGen/Targets/X86.cpp
index 21f9929cfa882..934c702d76209 100644
--- a/clang/lib/CodeGen/Targets/X86.cpp
+++ b/clang/lib/CodeGen/Targets/X86.cpp
@@ -2562,8 +2562,21 @@ GetSSETypeAtOffset(llvm::Type *IRType, unsigned IROffset,
     // If we can't get a second FP type, return a simple half or float.
     // avx512fp16-abi.c:pr51813_2 shows it works to return float for
     // {float, i8} too.
-    if (T1 == nullptr)
+    if (T1 == nullptr) {
+      // T0 does not cover the whole eightbyte.  If the rest of the
+      // eightbyte holds user data in the source type, the whole eightbyte
+      // has to be transferred.  A union is lowered to the IR type of its
+      // largest member, so a smaller member may have a field in bytes that
+      // are padding in the largest member and therefore invisible here;
+      // returning T0 would drop those bytes when the union is passed or
+      // returned in SSE registers.  As in GetINTEGERTypeAtOffset, this
+      // analysis has to run on the source type, because we can't depend
+      // on unions being lowered a specific way.
+      if (!BitsContainNoUserData(SourceTy, (SourceOffset + T0Size) * 8,
+                                 (SourceOffset + 8) * 8, getContext()))
+        return llvm::Type::getDoubleTy(getVMContext());
       return T0;
+    }
   }
 
   if (T0->isFloatTy() && T1->isFloatTy())
@@ -2573,8 +2586,15 @@ GetSSETypeAtOffset(llvm::Type *IRType, unsigned IROffset,
     llvm::Type *T2 = nullptr;
     if (SourceSize > 4)
       T2 = getFPTypeAtOffset(IRType, IROffset + 4, TD);
-    if (T2 == nullptr)
+    if (T2 == nullptr) {
+      // <2 x half> covers only the first half of the eightbyte; fall back
+      // to double if another union member has data in the rest, for the
+      // same reason as above.
+      if (!BitsContainNoUserData(SourceTy, (SourceOffset + 4) * 8,
+                                 (SourceOffset + 8) * 8, getContext()))
+        return llvm::Type::getDoubleTy(getVMContext());
       return llvm::FixedVectorType::get(T0, 2);
+    }
     return llvm::FixedVectorType::get(T0, 4);
   }
 
diff --git a/clang/test/CodeGen/X86/x86_64-union-sse-abi.c b/clang/test/CodeGen/X86/x86_64-union-sse-abi.c
new file mode 100644
index 0000000000000..f107b27678593
--- /dev/null
+++ b/clang/test/CodeGen/X86/x86_64-union-sse-abi.c
@@ -0,0 +1,123 @@
+// NOTE: Assertions have been autogenerated by utils/update_cc_test_checks.py UTC_ARGS: --version 6
+// RUN: %clang_cc1 -triple x86_64-unknown-linux-gnu -emit-llvm %s -o - | FileCheck %s
+
+// SSE counterpart of x86_64-union-abi.c.  A union is lowered to the IR type
+// of its largest member, so bytes that are padding in that member are
+// invisible to the IR-type walk that picks the SSE coercion type.  When a
+// smaller member has a field in those bytes, the coercion has to cover the
+// whole eightbyte (double), otherwise those bytes are dropped when the union
+// is passed or returned in SSE registers.
+
+union FloatFloatDouble {
+  struct { float a; float b; } s;
+  struct { float x; double y; } t;
+};
+
+void take_ffd(union FloatFloatDouble u);
+// CHECK-LABEL: define dso_local void @call_take_ffd(
+// CHECK-SAME: double [[U_COERCE0:%.*]], double [[U_COERCE1:%.*]]) #[[ATTR0:[0-9]+]] {
+// CHECK-NEXT:  [[ENTRY:.*:]]
+// CHECK-NEXT:    [[U:%.*]] = alloca [[UNION_FLOATFLOATDOUBLE:%.*]], align 8
+// CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    store double [[U_COERCE0]], ptr [[TMP0]], align 8
+// CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 1
+// CHECK-NEXT:    store double [[U_COERCE1]], ptr [[TMP1]], align 8
+// CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP3:%.*]] = load double, ptr [[TMP2]], align 8
+// CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 1
+// CHECK-NEXT:    [[TMP5:%.*]] = load double, ptr [[TMP4]], align 8
+// CHECK-NEXT:    call void @take_ffd(double [[TMP3]], double [[TMP5]])
+// CHECK-NEXT:    ret void
+//
+void call_take_ffd(union FloatFloatDouble u) { take_ffd(u); }
+
+// The second float lives in the padding of the larger member, so both
+// eightbytes have to travel whole.
+
+union FloatFloatDouble ret_ffd(void);
+// CHECK-LABEL: define dso_local void @call_ret_ffd(
+// CHECK-SAME: ) #[[ATTR0]] {
+// CHECK-NEXT:  [[ENTRY:.*:]]
+// CHECK-NEXT:    [[COERCE:%.*]] = alloca [[UNION_FLOATFLOATDOUBLE:%.*]], align 8
+// CHECK-NEXT:    [[CALL:%.*]] = call { double, double } @ret_ffd()
+// CHECK-NEXT:    [[COERCE_DIVE:%.*]] = getelementptr inbounds nuw [[UNION_FLOATFLOATDOUBLE]], ptr [[COERCE]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[COERCE_DIVE]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP1:%.*]] = extractvalue { double, double } [[CALL]], 0
+// CHECK-NEXT:    store double [[TMP1]], ptr [[TMP0]], align 8
+// CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[COERCE_DIVE]], i32 0, i32 1
+// CHECK-NEXT:    [[TMP3:%.*]] = extractvalue { double, double } [[CALL]], 1
+// CHECK-NEXT:    store double [[TMP3]], ptr [[TMP2]], align 8
+// CHECK-NEXT:    ret void
+//
+void call_ret_ffd(void) { ret_ffd(); }
+
+
+// The <2 x half> coercion would only cover the first half of the low
+// eightbyte; the float member t.c forces the full eightbyte again.
+union HalfHalfDouble {
+  struct { _Float16 a; _Float16 b; double c; } s;
+  struct { _Float16 a; _Float16 b; float c; } t;
+};
+
+void take_hhd(union HalfHalfDouble u);
+// CHECK-LABEL: define dso_local void @call_take_hhd(
+// CHECK-SAME: double [[U_COERCE0:%.*]], double [[U_COERCE1:%.*]]) #[[ATTR0]] {
+// CHECK-NEXT:  [[ENTRY:.*:]]
+// CHECK-NEXT:    [[U:%.*]] = alloca [[UNION_HALFHALFDOUBLE:%.*]], align 8
+// CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    store double [[U_COERCE0]], ptr [[TMP0]], align 8
+// CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 1
+// CHECK-NEXT:    store double [[U_COERCE1]], ptr [[TMP1]], align 8
+// CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP3:%.*]] = load double, ptr [[TMP2]], align 8
+// CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw { double, double }, ptr [[U]], i32 0, i32 1
+// CHECK-NEXT:    [[TMP5:%.*]] = load double, ptr [[TMP4]], align 8
+// CHECK-NEXT:    call void @take_hhd(double [[TMP3]], double [[TMP5]])
+// CHECK-NEXT:    ret void
+//
+void call_take_hhd(union HalfHalfDouble u) { take_hhd(u); }
+
+
+// Nothing covers the tail of an over-aligned union, so those bytes are
+// padding and the coercion is the float alone rather than the full
+// eightbyte.
+union FloatOverAligned8 {
+  float f;
+} __attribute__((aligned(8)));
+
+void take_foa8(union FloatOverAligned8 u);
+// CHECK-LABEL: define dso_local void @call_take_foa8(
+// CHECK-SAME: float [[U_COERCE:%.*]]) #[[ATTR0]] {
+// CHECK-NEXT:  [[ENTRY:.*:]]
+// CHECK-NEXT:    [[U:%.*]] = alloca [[UNION_FLOATOVERALIGNED8:%.*]], align 8
+// CHECK-NEXT:    [[COERCE_DIVE:%.*]] = getelementptr inbounds nuw [[UNION_FLOATOVERALIGNED8]], ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    store float [[U_COERCE]], ptr [[COERCE_DIVE]], align 8
+// CHECK-NEXT:    [[COERCE_DIVE1:%.*]] = getelementptr inbounds nuw [[UNION_FLOATOVERALIGNED8]], ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP0:%.*]] = load float, ptr [[COERCE_DIVE1]], align 8
+// CHECK-NEXT:    call void @take_foa8(float [[TMP0]])
+// CHECK-NEXT:    ret void
+//
+void call_take_foa8(union FloatOverAligned8 u) { take_foa8(u); }
+
+
+// When every byte of the eightbyte is covered by fields of the largest
+// member, the natural <2 x float> coercion keeps working.
+union TwoFloats {
+  struct { float a; float b; } s;
+  float f;
+};
+
+void take_twof(union TwoFloats u);
+// CHECK-LABEL: define dso_local void @call_take_twof(
+// CHECK-SAME: <2 x float> [[U_COERCE:%.*]]) #[[ATTR2:[0-9]+]] {
+// CHECK-NEXT:  [[ENTRY:.*:]]
+// CHECK-NEXT:    [[U:%.*]] = alloca [[UNION_TWOFLOATS:%.*]], align 4
+// CHECK-NEXT:    [[COERCE_DIVE:%.*]] = getelementptr inbounds nuw [[UNION_TWOFLOATS]], ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    store <2 x float> [[U_COERCE]], ptr [[COERCE_DIVE]], align 4
+// CHECK-NEXT:    [[COERCE_DIVE1:%.*]] = getelementptr inbounds nuw [[UNION_TWOFLOATS]], ptr [[U]], i32 0, i32 0
+// CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr [[COERCE_DIVE1]], align 4
+// CHECK-NEXT:    call void @take_twof(<2 x float> [[TMP0]])
+// CHECK-NEXT:    ret void
+//
+void call_take_twof(union TwoFloats u) { take_twof(u); }
+

``````````

</details>


https://github.com/llvm/llvm-project/pull/227229


More information about the cfe-commits mailing list