[llvm] [X86] Fix VECREDUCE_XOR to PARITY lowering for 16-bit elements (PR #219216)

Nikita Popov via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 07:06:04 PDT 2026


https://github.com/nikic created https://github.com/llvm/llvm-project/pull/219216

We can't lower vecreduce.xor to parity(movmsk) if the movmsk doesn't match the element size. For 16-bit elements we'd use movmskb, which would result in two bits per element, which are always the same. Thus the parity is always zero.

Disclosure: Test case identified by AI, patch is my own.

>From 2dc6aa3617e2f1931cfd858913782322455f9088 Mon Sep 17 00:00:00 2001
From: Nikita Popov <npopov at redhat.com>
Date: Thu, 27 Aug 2026 15:56:42 +0200
Subject: [PATCH 1/2] Add test

---
 llvm/test/CodeGen/X86/vector-reduce-xor.ll | 28 ++++++++++++++++++++++
 1 file changed, 28 insertions(+)

diff --git a/llvm/test/CodeGen/X86/vector-reduce-xor.ll b/llvm/test/CodeGen/X86/vector-reduce-xor.ll
index b46ee32b3ee78..9cbb45ad0d274 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-xor.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-xor.ll
@@ -1572,6 +1572,34 @@ define i32 @PR215069() {
   ret i32 %i2
 }
 
+define i16 @test_v8i16_signbits(<8 x i16> %a0, <8 x i16> %a1) {
+; SSE-LABEL: test_v8i16_signbits:
+; SSE:       # %bb.0:
+; SSE-NEXT:    pcmpeqw %xmm1, %xmm0
+; SSE-NEXT:    pmovmskb %xmm0, %ecx
+; SSE-NEXT:    xorl %eax, %eax
+; SSE-NEXT:    xorb %ch, %cl
+; SSE-NEXT:    setnp %al
+; SSE-NEXT:    negl %eax
+; SSE-NEXT:    # kill: def $ax killed $ax killed $eax
+; SSE-NEXT:    ret{{[l|q]}}
+;
+; AVX-LABEL: test_v8i16_signbits:
+; AVX:       # %bb.0:
+; AVX-NEXT:    vpcmpeqw %xmm1, %xmm0, %xmm0
+; AVX-NEXT:    vpmovmskb %xmm0, %ecx
+; AVX-NEXT:    xorl %eax, %eax
+; AVX-NEXT:    xorb %ch, %cl
+; AVX-NEXT:    setnp %al
+; AVX-NEXT:    negl %eax
+; AVX-NEXT:    # kill: def $ax killed $ax killed $eax
+; AVX-NEXT:    ret{{[l|q]}}
+  %c = icmp eq <8 x i16> %a0, %a1
+  %s = sext <8 x i1> %c to <8 x i16>
+  %r = call i16 @llvm.vector.reduce.xor.v8i16(<8 x i16> %s)
+  ret i16 %r
+}
+
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; AVX1OR2: {{.*}}
 ; AVX512BW: {{.*}}

>From fc09b392fcd68f5b161c377c7906fb77b3ac47e3 Mon Sep 17 00:00:00 2001
From: Nikita Popov <npopov at redhat.com>
Date: Thu, 27 Aug 2026 16:00:44 +0200
Subject: [PATCH 2/2] Fix

---
 llvm/lib/Target/X86/X86ISelLowering.cpp    |  8 ++++++-
 llvm/test/CodeGen/X86/vector-reduce-xor.ll | 25 +++++++++++++---------
 2 files changed, 22 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 3a66310ff1b3c..72fa89fb5d1ae 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -47325,8 +47325,14 @@ static SDValue combineVECREDUCE_LOGIC(SDNode *Reduce, SelectionDAG &DAG,
     if (64 == BitWidth || 32 == BitWidth)
       MaskSrcVT = MVT::getVectorVT(MVT::getFloatingPointVT(BitWidth),
                                    MatchSizeInBits / BitWidth);
-    else
+    else {
+      // Lowering via parity is not valid when using pmovmskb for vectors
+      // with 16 bit elements. In that case we get two bits for every element,
+      // such that the parity is always zero.
+      if (BinOp == ISD::XOR && BitWidth != 8)
+        return SDValue();
       MaskSrcVT = MVT::getVectorVT(MVT::i8, MatchSizeInBits / 8);
+    }
 
     SDValue BitcastLogicOp = DAG.getBitcast(MaskSrcVT, Match);
     Movmsk = getMOVMSK(DL, BitcastLogicOp, DAG, Subtarget);
diff --git a/llvm/test/CodeGen/X86/vector-reduce-xor.ll b/llvm/test/CodeGen/X86/vector-reduce-xor.ll
index 9cbb45ad0d274..8fc68928c3a15 100644
--- a/llvm/test/CodeGen/X86/vector-reduce-xor.ll
+++ b/llvm/test/CodeGen/X86/vector-reduce-xor.ll
@@ -1576,22 +1576,27 @@ define i16 @test_v8i16_signbits(<8 x i16> %a0, <8 x i16> %a1) {
 ; SSE-LABEL: test_v8i16_signbits:
 ; SSE:       # %bb.0:
 ; SSE-NEXT:    pcmpeqw %xmm1, %xmm0
-; SSE-NEXT:    pmovmskb %xmm0, %ecx
-; SSE-NEXT:    xorl %eax, %eax
-; SSE-NEXT:    xorb %ch, %cl
-; SSE-NEXT:    setnp %al
-; SSE-NEXT:    negl %eax
+; SSE-NEXT:    pshufd {{.*#+}} xmm1 = xmm0[2,3,2,3]
+; SSE-NEXT:    pxor %xmm0, %xmm1
+; SSE-NEXT:    pshufd {{.*#+}} xmm0 = xmm1[1,1,1,1]
+; SSE-NEXT:    pxor %xmm1, %xmm0
+; SSE-NEXT:    movdqa %xmm0, %xmm1
+; SSE-NEXT:    psrld $16, %xmm1
+; SSE-NEXT:    pxor %xmm0, %xmm1
+; SSE-NEXT:    movd %xmm1, %eax
 ; SSE-NEXT:    # kill: def $ax killed $ax killed $eax
 ; SSE-NEXT:    ret{{[l|q]}}
 ;
 ; AVX-LABEL: test_v8i16_signbits:
 ; AVX:       # %bb.0:
 ; AVX-NEXT:    vpcmpeqw %xmm1, %xmm0, %xmm0
-; AVX-NEXT:    vpmovmskb %xmm0, %ecx
-; AVX-NEXT:    xorl %eax, %eax
-; AVX-NEXT:    xorb %ch, %cl
-; AVX-NEXT:    setnp %al
-; AVX-NEXT:    negl %eax
+; AVX-NEXT:    vpshufd {{.*#+}} xmm1 = xmm0[2,3,2,3]
+; AVX-NEXT:    vpxor %xmm1, %xmm0, %xmm0
+; AVX-NEXT:    vpshufd {{.*#+}} xmm1 = xmm0[1,1,1,1]
+; AVX-NEXT:    vpxor %xmm1, %xmm0, %xmm0
+; AVX-NEXT:    vpsrld $16, %xmm0, %xmm1
+; AVX-NEXT:    vpxor %xmm0, %xmm1, %xmm0
+; AVX-NEXT:    vmovd %xmm0, %eax
 ; AVX-NEXT:    # kill: def $ax killed $ax killed $eax
 ; AVX-NEXT:    ret{{[l|q]}}
   %c = icmp eq <8 x i16> %a0, %a1



More information about the llvm-commits mailing list