[llvm] f872635 - [X86] Limit the result of XOR8rr_NOREX unused (#218640)

via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 25 20:19:58 PDT 2026


Author: Phoebe Wang
Date: 2026-08-26T11:19:54+08:00
New Revision: f8726351889e41e0c5b7e04670cd46e47fa8b878

URL: https://github.com/llvm/llvm-project/commit/f8726351889e41e0c5b7e04670cd46e47fa8b878
DIFF: https://github.com/llvm/llvm-project/commit/f8726351889e41e0c5b7e04670cd46e47fa8b878.diff

LOG: [X86] Limit the result of XOR8rr_NOREX unused (#218640)

In case it may be zero/sign-extended into an REX/REX2 register.

Fixes: #218583

Assisted-by: Claude Opus 4.8

Added: 
    llvm/test/CodeGen/X86/pr218583.ll

Modified: 
    llvm/lib/Target/X86/X86InstrCompiler.td
    llvm/lib/Target/X86/X86InstrFragments.td

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/X86/X86InstrCompiler.td b/llvm/lib/Target/X86/X86InstrCompiler.td
index d3e8fdbdcd53c..40ad1eb9b72cb 100644
--- a/llvm/lib/Target/X86/X86InstrCompiler.td
+++ b/llvm/lib/Target/X86/X86InstrCompiler.td
@@ -1927,10 +1927,17 @@ def : Pat<(store (i8 (trunc_su (srl_su GR16:$src, (i8 8)))), addr:$dst),
 // we need to be more careful. We're using a NOREX instruction here in case
 // register allocation fails to keep the two registers together. So we need to
 // make sure we can't accidentally mix R8-R15 with an h-register.
-def : Pat<(X86xor_flag (i8 (trunc GR32:$src)),
-                       (i8 (trunc (srl_su GR32:$src, (i8 8))))),
+// We additionally require the (non-flags) result of the xor to be unused: the
+// GR8_NOREX result may end up in a high-byte register (AH/BH/CH/DH), and any
+// user that widens or otherwise consumes it (e.g. by zero/sign-extending it
+// into an R8-R15 register) could require a REX prefix, which cannot encode a
+// high-byte register. The encoder would then fail. So restrict the trick to the
+// case where the value result has no uses at all (see xor_flag_dead_result).
+def : Pat<(xor_flag_dead_result (i8 (trunc GR32:$src)),
+                                (i8 (trunc (srl_su GR32:$src, (i8 8))))),
           (XOR8rr_NOREX (EXTRACT_SUBREG GR32:$src, sub_8bit),
-                        (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>;
+                        (EXTRACT_SUBREG GR32:$src, sub_8bit_hi))>,
+      Requires<[In64BitMode]>;
 
 // (shl x, 1) ==> (add x, x)
 // Note that if x is undef (immediate or otherwise), we could theoretically

diff  --git a/llvm/lib/Target/X86/X86InstrFragments.td b/llvm/lib/Target/X86/X86InstrFragments.td
index c183849d4f575..a9910dfedbfcc 100644
--- a/llvm/lib/Target/X86/X86InstrFragments.td
+++ b/llvm/lib/Target/X86/X86InstrFragments.td
@@ -884,6 +884,11 @@ def xor_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
   return hasNoCarryFlagUses(SDValue(N, 1));
 }]>;
 
+def xor_flag_dead_result : PatFrag<(ops node:$lhs, node:$rhs),
+                                   (X86xor_flag node:$lhs, node:$rhs), [{
+  return SDValue(N, 0).use_empty();
+}]>;
+
 def and_flag_nocf : PatFrag<(ops node:$lhs, node:$rhs),
                             (X86and_flag node:$lhs, node:$rhs), [{
   return hasNoCarryFlagUses(SDValue(N, 1));

diff  --git a/llvm/test/CodeGen/X86/pr218583.ll b/llvm/test/CodeGen/X86/pr218583.ll
new file mode 100644
index 0000000000000..89f6fe888f268
--- /dev/null
+++ b/llvm/test/CodeGen/X86/pr218583.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=i686 | FileCheck %s --check-prefix=X86
+; RUN: llc < %s -mtriple=x86_64 | FileCheck %s --check-prefix=X64
+
+; The xor of two bytes extracted from a vector matches the h-register
+; __builtin_parity idiom (xor(trunc(x), trunc(srl(x, 8)))). Previously this was
+; lowered to the GR8_NOREX XOR8rr_NOREX instruction whose result could be
+; allocated to a high-byte register (e.g. %bh). Because the xor result is used
+; here (zero-extended into a GR32/GR64 register that may require a REX prefix),
+; the encoder failed with "Cannot encode high byte register in REX-prefixed
+; instruction". The h-register trick must only be used when the xor value
+; result is dead and only EFLAGS are consumed.
+
+define i32 @main(<4 x i8> %j.0, i32 %conv81) nounwind {
+; X86-LABEL: main:
+; X86:       # %bb.0: # %entry
+; X86-NEXT:    pushl %ebx
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %eax
+; X86-NEXT:    xorb %cl, %al
+; X86-NEXT:    js .LBB0_2
+; X86-NEXT:  # %bb.1: # %if.then.i
+; X86-NEXT:    movzbl %al, %ebx
+; X86-NEXT:    xorl $1, %ebx
+; X86-NEXT:    pushl %ebx
+; X86-NEXT:    calll 0
+; X86-NEXT:    movzbl {{[0-9]+}}(%esp), %ecx
+; X86-NEXT:    addl $4, %esp
+; X86-NEXT:    movb %bl, 0
+; X86-NEXT:  .LBB0_2: # %h.exit
+; X86-NEXT:    movsbl %cl, %eax
+; X86-NEXT:    popl %ebx
+; X86-NEXT:    retl
+;
+; X64-LABEL: main:
+; X64:       # %bb.0: # %entry
+; X64-NEXT:    pushq %rbp
+; X64-NEXT:    pushq %rbx
+; X64-NEXT:    pushq %rax
+; X64-NEXT:    movd %xmm0, %eax
+; X64-NEXT:    movl %eax, %ebp
+; X64-NEXT:    shrl $8, %eax
+; X64-NEXT:    xorb %bpl, %al
+; X64-NEXT:    js .LBB0_2
+; X64-NEXT:  # %bb.1: # %if.then.i
+; X64-NEXT:    movzbl %al, %ebx
+; X64-NEXT:    xorq $1, %rbx
+; X64-NEXT:    xorl %eax, %eax
+; X64-NEXT:    movl %ebx, %edi
+; X64-NEXT:    callq *%rax
+; X64-NEXT:    movb %bl, 0
+; X64-NEXT:  .LBB0_2: # %h.exit
+; X64-NEXT:    movsbl %bpl, %eax
+; X64-NEXT:    addq $8, %rsp
+; X64-NEXT:    popq %rbx
+; X64-NEXT:    popq %rbp
+; X64-NEXT:    retq
+entry:
+  %vecext = extractelement <4 x i8> %j.0, i64 0
+  %vecext4 = extractelement <4 x i8> %j.0, i64 1
+  %xor13 = xor i8 %vecext, %vecext4
+  %cmp.i = icmp sgt i8 %xor13, -1
+  br i1 %cmp.i, label %if.then.i, label %h.exit
+
+if.then.i:
+  %xor = zext i8 %xor13 to i64
+  %xor6 = xor i64 %xor, 1
+  %conv2.i = trunc i64 %xor6 to i32
+  %call.i = tail call i32 null(i32 %conv2.i)
+  %conv3.i = trunc i64 %xor6 to i8
+  store i8 %conv3.i, ptr null, align 1
+  br label %h.exit
+
+h.exit:
+  %conv811 = sext i8 %vecext to i32
+  ret i32 %conv811
+}


        


More information about the llvm-commits mailing list