[llvm] f872635 - [X86] Limit the result of XOR8rr_NOREX unused (#218640)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 25 20:19:58 PDT 2026
Author: Phoebe Wang
Date: 2026-08-26T11:19:54+08:00
New Revision: f8726351889e41e0c5b7e04670cd46e47fa8b878
URL: https://github.com/llvm/llvm-project/commit/f8726351889e41e0c5b7e04670cd46e47fa8b878
DIFF: https://github.com/llvm/llvm-project/commit/f8726351889e41e0c5b7e04670cd46e47fa8b878.diff
LOG: [X86] Limit the result of XOR8rr_NOREX unused (#218640)
In case it may be zero/sign-extended into an REX/REX2 register.
Fixes: #218583
Assisted-by: Claude Opus 4.8
Added:
llvm/test/CodeGen/X86/pr218583.ll
Modified:
llvm/lib/Target/X86/X86InstrCompiler.td
llvm/lib/Target/X86/X86InstrFragments.td
Removed:
################################################################################
diff --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index d3e8fdbdcd53c..40ad1eb9b72cb 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1927,10 +1927,17 @@ def : Pat<(store (i8 (trunc_su (srl_su GR16:$src, (i8 8)))), addr:$dst),
// we need to be more careful. We're using a NOREX instruction here in case
// register allocation fails to keep the two registers together. So we need to
// make sure we can't accidentally mix R8-R15 with an h-register.
-def : Pat<(X86xor_flag (i8 (trunc GR32:$src)),
- (i8 (trunc (srl_su GR32:$src, (i8 8))))),
+// We additionally require the (non-flags) result of the xor to be unused: the
+// GR8_NOREX result may end up in a high-byte register (AH/BH/CH/DH), and any
+// user that widens or otherwise consumes it (e.g. by zero/sign-extending it
+// into an R8-R15 register) could require a REX prefix, which cannot encode a
+// high-byte register. The encoder would then fail. So restrict the trick to the
+// case where the value result has no uses at all (see xor_flag_dead_result).
+def : Pat<(xor_flag_dead_result (i8 (trunc GR32:$src)),
+ (i8 (trunc (srl_su GR32:$src, (i8 8))))),
(XOR8rr_NOREX (EXTRACT_SUBREG GR32:$src, sub_8bit),
- (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>;
+ (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>,
+ Requires<[In64BitMode]>;
// (shl x, 1) ==> (add x, x)
// Note that if x is undef (immediate or otherwise), we could theoretically
diff --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index c183849d4f575..a9910dfedbfcc 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -884,6 +884,11 @@ def xor_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
return hasNoCarryFlagUses(SDValue(N, 1));
}]>;
+def xor_flag_dead_result : PatFrag<(ops node:$lhs, node:$rhs),
+ (X86xor_flag node:$lhs, node:$rhs), [{
+ return SDValue(N, 0).use_empty();
+}]>;
+
def and_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
(X86and_flag node:$lhs, node:$rhs), [{
return hasNoCarryFlagUses(SDValue(N, 1));
diff --git a/llvm/test/CodeGen/X86/pr218583.ll b/llvm/test/CodeGen/X86/pr218583.ll
new file mode 100644
index 0000000000000..89f6fe888f268
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr218583.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=i686 | FileCheck %s --check-prefix=X86
+; RUN: llc < %s -mtriple=x86_64 | FileCheck %s --check-prefix=X64
+
+; The xor of two bytes extracted from a vector matches the h-register
+; __builtin_parity idiom (xor(trunc(x), trunc(srl(x, 8)))). Previously this was
+; lowered to the GR8_NOREX XOR8rr_NOREX instruction whose result could be
+; allocated to a high-byte register (e.g. %bh). Because the xor result is used
+; here (zero-extended into a GR32/GR64 register that may require a REX prefix),
+; the encoder failed with "Cannot encode high byte register in REX-prefixed
+; instruction". The h-register trick must only be used when the xor value
+; result is dead and only EFLAGS are consumed.
+
+define i32 @main(<4 x i8> %j.0, i32 %conv81) nounwind {
+; X86-LABEL: main:
+; X86: # %bb.0: # %entry
+; X86-NEXT: pushl %ebx
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT: xorb %cl, %al
+; X86-NEXT: js .LBB0_2
+; X86-NEXT: # %bb.1: # %if.then.i
+; X86-NEXT: movzbl %al, %ebx
+; X86-NEXT: xorl $1, %ebx
+; X86-NEXT: pushl %ebx
+; X86-NEXT: calll 0
+; X86-NEXT: movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT: addl $4, %esp
+; X86-NEXT: movb %bl, 0
+; X86-NEXT: .LBB0_2: # %h.exit
+; X86-NEXT: movsbl %cl, %eax
+; X86-NEXT: popl %ebx
+; X86-NEXT: retl
+;
+; X64-LABEL: main:
+; X64: # %bb.0: # %entry
+; X64-NEXT: pushq %rbp
+; X64-NEXT: pushq %rbx
+; X64-NEXT: pushq %rax
+; X64-NEXT: movd %xmm0, %eax
+; X64-NEXT: movl %eax, %ebp
+; X64-NEXT: shrl $8, %eax
+; X64-NEXT: xorb %bpl, %al
+; X64-NEXT: js .LBB0_2
+; X64-NEXT: # %bb.1: # %if.then.i
+; X64-NEXT: movzbl %al, %ebx
+; X64-NEXT: xorq $1, %rbx
+; X64-NEXT: xorl %eax, %eax
+; X64-NEXT: movl %ebx, %edi
+; X64-NEXT: callq *%rax
+; X64-NEXT: movb %bl, 0
+; X64-NEXT: .LBB0_2: # %h.exit
+; X64-NEXT: movsbl %bpl, %eax
+; X64-NEXT: addq $8, %rsp
+; X64-NEXT: popq %rbx
+; X64-NEXT: popq %rbp
+; X64-NEXT: retq
+entry:
+ %vecext = extractelement <4 x i8> %j.0, i64 0
+ %vecext4 = extractelement <4 x i8> %j.0, i64 1
+ %xor13 = xor i8 %vecext, %vecext4
+ %cmp.i = icmp sgt i8 %xor13, -1
+ br i1 %cmp.i, label %if.then.i, label %h.exit
+
+if.then.i:
+ %xor = zext i8 %xor13 to i64
+ %xor6 = xor i64 %xor, 1
+ %conv2.i = trunc i64 %xor6 to i32
+ %call.i = tail call i32 null(i32 %conv2.i)
+ %conv3.i = trunc i64 %xor6 to i8
+ store i8 %conv3.i, ptr null, align 1
+ br label %h.exit
+
+h.exit:
+ %conv811 = sext i8 %vecext to i32
+ ret i32 %conv811
+}
More information about the llvm-commits
mailing list