[llvm] [X86] Limit the result of XOR8rr_NOREX unused (PR #218640)
Phoebe Wang via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 25 07:06:29 PDT 2026
https://github.com/phoebewang updated https://github.com/llvm/llvm-project/pull/218640
>From 0ce9eeaf7bc2ff61c1e885fef637e6bd3f665707 Mon Sep 17 00:00:00 2001
From: Phoebe Wang <phoebe.wang at intel.com>
Date: Tue, 25 Aug 2026 17:10:39 +0800
Subject: [PATCH 1/2] [X86] Limit the result of XOR8rr_NOREX unused
In case it may be zero/sign-extended into an REX/REX2 register.
Fixes: #218583
---
llvm/lib/Target/X86/X86InstrCompiler.td | 7 ++-
llvm/lib/Target/X86/X86InstrFragments.td | 12 ++++
llvm/test/CodeGen/X86/pr218583.ll | 77 ++++++++++++++++++++++++
3 files changed, 93 insertions(+), 3 deletions(-)
create mode 100644 llvm/test/CodeGen/X86/pr218583.ll
diff --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index d3e8fdbdcd53c..3752c8783970a 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1927,10 +1927,11 @@ def : Pat<(store (i8 (trunc_su (srl_su GR16:$src, (i8 8)))), addr:$dst),
// we need to be more careful. We're using a NOREX instruction here in case
// register allocation fails to keep the two registers together. So we need to
// make sure we can't accidentally mix R8-R15 with an h-register.
-def : Pat<(X86xor_flag (i8 (trunc GR32:$src)),
- (i8 (trunc (srl_su GR32:$src, (i8 8))))),
+def : Pat<(xor_flag_dead_result (i8 (trunc GR32:$src)),
+ (i8 (trunc (srl_su GR32:$src, (i8 8))))),
(XOR8rr_NOREX (EXTRACT_SUBREG GR32:$src, sub_8bit),
- (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>;
+ (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>,
+ Requires<[In64BitMode]>;
// (shl x, 1) ==> (add x, x)
// Note that if x is undef (immediate or otherwise), we could theoretically
diff --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index c183849d4f575..259586f36889b 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -884,6 +884,18 @@ def xor_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
return hasNoCarryFlagUses(SDValue(N, 1));
}]>;
+// Only matches when the data (non-flags) result of the xor is unused, i.e. the
+// __builtin_parity idiom where only EFLAGS are consumed. This guards the 64-bit
+// h-register XOR8rr_NOREX pattern: its GR8_NOREX result may be allocated to a
+// high-byte register (AH/BH/CH/DH), which cannot be encoded in a REX-prefixed
+// instruction. If the result value escaped (e.g. got zero/sign-extended into an
+// R8-R15 register) the encoder would fail, so restrict the trick to the case
+// where the value result has no uses at all.
+def xor_flag_dead_result : PatFrag<(ops node:$lhs, node:$rhs),
+ (X86xor_flag node:$lhs, node:$rhs), [{
+ return SDValue(N, 0).use_empty();
+}]>;
+
def and_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
(X86and_flag node:$lhs, node:$rhs), [{
return hasNoCarryFlagUses(SDValue(N, 1));
diff --git a/llvm/test/CodeGen/X86/pr218583.ll b/llvm/test/CodeGen/X86/pr218583.ll
new file mode 100644
index 0000000000000..89f6fe888f268
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr218583.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=i686 | FileCheck %s --check-prefix=X86
+; RUN: llc < %s -mtriple=x86_64 | FileCheck %s --check-prefix=X64
+
+; The xor of two bytes extracted from a vector matches the h-register
+; __builtin_parity idiom (xor(trunc(x), trunc(srl(x, 8)))). Previously this was
+; lowered to the GR8_NOREX XOR8rr_NOREX instruction whose result could be
+; allocated to a high-byte register (e.g. %bh). Because the xor result is used
+; here (zero-extended into a GR32/GR64 register that may require a REX prefix),
+; the encoder failed with "Cannot encode high byte register in REX-prefixed
+; instruction". The h-register trick must only be used when the xor value
+; result is dead and only EFLAGS are consumed.
+
+define i32 @main(<4 x i8> %j.0, i32 %conv81) nounwind {
+; X86-LABEL: main:
+; X86: # %bb.0: # %entry
+; X86-NEXT: pushl %ebx
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: xorb %cl, %al
+; X86-NEXT: js .LBB0_2
+; X86-NEXT: # %bb.1: # %if.then.i
+; X86-NEXT: movzbl %al, %ebx
+; X86-NEXT: xorl $1, %ebx
+; X86-NEXT: pushl %ebx
+; X86-NEXT: calll 0
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: addl $4, %esp
+; X86-NEXT: movb %bl, 0
+; X86-NEXT: .LBB0_2: # %h.exit
+; X86-NEXT: movsbl %cl, %eax
+; X86-NEXT: popl %ebx
+; X86-NEXT: retl
+;
+; X64-LABEL: main:
+; X64: # %bb.0: # %entry
+; X64-NEXT: pushq %rbp
+; X64-NEXT: pushq %rbx
+; X64-NEXT: pushq %rax
+; X64-NEXT: movd %xmm0, %eax
+; X64-NEXT: movl %eax, %ebp
+; X64-NEXT: shrl $8, %eax
+; X64-NEXT: xorb %bpl, %al
+; X64-NEXT: js .LBB0_2
+; X64-NEXT: # %bb.1: # %if.then.i
+; X64-NEXT: movzbl %al, %ebx
+; X64-NEXT: xorq $1, %rbx
+; X64-NEXT: xorl %eax, %eax
+; X64-NEXT: movl %ebx, %edi
+; X64-NEXT: callq *%rax
+; X64-NEXT: movb %bl, 0
+; X64-NEXT: .LBB0_2: # %h.exit
+; X64-NEXT: movsbl %bpl, %eax
+; X64-NEXT: addq $8, %rsp
+; X64-NEXT: popq %rbx
+; X64-NEXT: popq %rbp
+; X64-NEXT: retq
+entry:
+ %vecext = extractelement <4 x i8> %j.0, i64 0
+ %vecext4 = extractelement <4 x i8> %j.0, i64 1
+ %xor13 = xor i8 %vecext, %vecext4
+ %cmp.i = icmp sgt i8 %xor13, -1
+ br i1 %cmp.i, label %if.then.i, label %h.exit
+
+if.then.i:
+ %xor = zext i8 %xor13 to i64
+ %xor6 = xor i64 %xor, 1
+ %conv2.i = trunc i64 %xor6 to i32
+ %call.i = tail call i32 null(i32 %conv2.i)
+ %conv3.i = trunc i64 %xor6 to i8
+ store i8 %conv3.i, ptr null, align 1
+ br label %h.exit
+
+h.exit:
+ %conv811 = sext i8 %vecext to i32
+ ret i32 %conv811
+}
>From 8250a8d6895dfd13ccb0faf4525eace5de927b96 Mon Sep 17 00:00:00 2001
From: Phoebe Wang <phoebe.wang at intel.com>
Date: Tue, 25 Aug 2026 22:06:10 +0800
Subject: [PATCH 2/2] Move comments to X86InstrCompiler.td
---
llvm/lib/Target/X86/X86InstrCompiler.td | 6 ++++++
llvm/lib/Target/X86/X86InstrFragments.td | 7 -------
2 files changed, 6 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index 3752c8783970a..40ad1eb9b72cb 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1927,6 +1927,12 @@ def : Pat<(store (i8 (trunc_su (srl_su GR16:$src, (i8 8)))), addr:$dst),
// we need to be more careful. We're using a NOREX instruction here in case
// register allocation fails to keep the two registers together. So we need to
// make sure we can't accidentally mix R8-R15 with an h-register.
+// We additionally require the (non-flags) result of the xor to be unused: the
+// GR8_NOREX result may end up in a high-byte register (AH/BH/CH/DH), and any
+// user that widens or otherwise consumes it (e.g. by zero/sign-extending it
+// into an R8-R15 register) could require a REX prefix, which cannot encode a
+// high-byte register. The encoder would then fail. So restrict the trick to the
+// case where the value result has no uses at all (see xor_flag_dead_result).
def : Pat<(xor_flag_dead_result (i8 (trunc GR32:$src)),
(i8 (trunc (srl_su GR32:$src, (i8 8))))),
(XOR8rr_NOREX (EXTRACT_SUBREG GR32:$src, sub_8bit),
diff --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index 259586f36889b..a9910dfedbfcc 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -884,13 +884,6 @@ def xor_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
return hasNoCarryFlagUses(SDValue(N, 1));
}]>;
-// Only matches when the data (non-flags) result of the xor is unused, i.e. the
-// __builtin_parity idiom where only EFLAGS are consumed. This guards the 64-bit
-// h-register XOR8rr_NOREX pattern: its GR8_NOREX result may be allocated to a
-// high-byte register (AH/BH/CH/DH), which cannot be encoded in a REX-prefixed
-// instruction. If the result value escaped (e.g. got zero/sign-extended into an
-// R8-R15 register) the encoder would fail, so restrict the trick to the case
-// where the value result has no uses at all.
def xor_flag_dead_result : PatFrag<(ops node:$lhs, node:$rhs),
(X86xor_flag node:$lhs, node:$rhs), [{
return SDValue(N, 0).use_empty();
More information about the llvm-commits
mailing list