[llvm] [X86] combineVTRUNCSAT - don't split 128-bit concatenated vectors when folding to PACKSS/US (PR #194347)
Simon Pilgrim via llvm-commits
llvm-commits at lists.llvm.org
Tue Apr 28 00:00:48 PDT 2026
https://github.com/RKSimon updated https://github.com/llvm/llvm-project/pull/194347
>From b48d5a6e78e435862396022dcd9ee2b126fa19a1 Mon Sep 17 00:00:00 2001
From: Simon Pilgrim <llvm-dev at redking.me.uk>
Date: Mon, 27 Apr 2026 12:31:41 +0100
Subject: [PATCH] [X86] combineVTRUNCSAT - don't split 128-bit concatenated
vectors when folding to PACKSS/US
If the VTRUNCS/US node has 128-bit src/dst types, then ensure we don't split into sub-128-bit vectors - just treat it as padded with zeros (matches VTRUNC behaviour)
Fixes #194344
---
llvm/lib/Target/X86/X86ISelLowering.cpp | 17 ++++++++++++++---
llvm/test/CodeGen/X86/packus.ll | 19 +++++++++++++++++++
2 files changed, 33 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index cbae4a97527a8..22685a28d31b9 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -55624,7 +55624,8 @@ static SDValue combineVTRUNC(SDNode *N, SelectionDAG &DAG,
}
static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
- TargetLowering::DAGCombinerInfo &DCI) {
+ TargetLowering::DAGCombinerInfo &DCI,
+ const X86Subtarget &Subtarget) {
using namespace SDPatternMatch;
unsigned Opc = N->getOpcode();
EVT VT = N->getValueType(0);
@@ -55643,7 +55644,17 @@ static SDValue combineVTRUNCSAT(SDNode *N, SelectionDAG &DAG,
(EltSizeInBits * 2) == Src.getScalarValueSizeInBits() &&
isFreeToSplitVector(Src, DAG)) {
SDLoc DL(N);
- auto [LHS, RHS] = splitVector(Src, DAG, DL);
+ SDValue LHS, RHS;
+ if (Src.getValueSizeInBits() == VT.getSizeInBits()) {
+ assert(VT.is128BitVector() && "128-bit VTRUNC source expected");
+ LHS = Src;
+ RHS = getZeroVector(Src.getSimpleValueType(), Subtarget, DAG, DL);
+ } else {
+ std::tie(LHS, RHS) = splitVector(Src, DAG, DL);
+ }
+ assert(LHS.getValueSizeInBits() == VT.getSizeInBits() &&
+ RHS.getValueSizeInBits() == VT.getSizeInBits() &&
+ "PACK src/dst size mismatch");
unsigned PackOpc = Opc == X86ISD::VTRUNCS ? X86ISD::PACKSS : X86ISD::PACKUS;
SDValue Pack = DAG.getNode(PackOpc, DL, VT, LHS, RHS);
if (VT.is128BitVector())
@@ -62367,7 +62378,7 @@ SDValue X86TargetLowering::PerformDAGCombine(SDNode *N,
case ISD::TRUNCATE: return combineTruncate(N, DAG, Subtarget);
case X86ISD::VTRUNC: return combineVTRUNC(N, DAG, DCI);
case X86ISD::VTRUNCS:
- case X86ISD::VTRUNCUS: return combineVTRUNCSAT(N, DAG, DCI);
+ case X86ISD::VTRUNCUS: return combineVTRUNCSAT(N, DAG, DCI, Subtarget);
case X86ISD::ANDNP: return combineAndnp(N, DAG, DCI, Subtarget);
case X86ISD::FAND: return combineFAnd(N, DAG, Subtarget);
case X86ISD::FANDN: return combineFAndn(N, DAG, Subtarget);
diff --git a/llvm/test/CodeGen/X86/packus.ll b/llvm/test/CodeGen/X86/packus.ll
index e678991fcec78..7de7caff18d6b 100644
--- a/llvm/test/CodeGen/X86/packus.ll
+++ b/llvm/test/CodeGen/X86/packus.ll
@@ -511,6 +511,25 @@ define <32 x i8> @packuswb_icmp_zero_trunc_256(<16 x i16> %a0) {
ret <32 x i8> %4
}
+define <8 x i8> @_mm_packs_pu16_manual(<4 x i16> %a, <4 x i16> %b) nounwind {
+; SSE-LABEL: _mm_packs_pu16_manual:
+; SSE: # %bb.0:
+; SSE-NEXT: punpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm1[0]
+; SSE-NEXT: packuswb %xmm0, %xmm0
+; SSE-NEXT: ret{{[l|q]}}
+;
+; AVX-LABEL: _mm_packs_pu16_manual:
+; AVX: # %bb.0:
+; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm1[0]
+; AVX-NEXT: vpackuswb %xmm0, %xmm0, %xmm0
+; AVX-NEXT: ret{{[l|q]}}
+ %sh = shufflevector <4 x i16> %a, <4 x i16> %b, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7>
+ %minv = tail call <8 x i16> @llvm.smax.v8i16(<8 x i16> %sh, <8 x i16> splat (i16 0))
+ %sat = tail call <8 x i16> @llvm.umin.v8i16(<8 x i16> %minv, <8 x i16> splat (i16 255))
+ %tr = trunc nuw <8 x i16> %sat to <8 x i8>
+ ret <8 x i8> %tr
+}
+
define <16 x i8> @_mm_packus_epi16_manual(<8 x i16> %a, <8 x i16> %b) nounwind {
; SSE-LABEL: _mm_packus_epi16_manual:
; SSE: # %bb.0:
More information about the llvm-commits
mailing list