[llvm] [X86][APX] Use EVEX BLSI/BLSMSK for i8 patterns with EGPR (PR #226796)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 27 07:56:42 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-x86

Author: Evgenii Kudriashov (e-kud)

<details>
<summary>Changes</summary>

This is a follow-up to #<!-- -->204746 and #<!-- -->205093, which added the i8 BLSI and
BLSMSK patterns.

The i8 BLSI/BLSMSK patterns were defined outside of Bls_Pats and always
selected the VEX forms, so with EGPR the register allocator could not
assign r16-r31 to their operands. Move them into Bls_Pats so that the
_EVEX variants are selected when EGPR is available.

Assisted-by: Claude Code

---
Full diff: https://github.com/llvm/llvm-project/pull/226796.diff


3 Files Affected:

- (modified) llvm/lib/Target/X86/X86InstrMisc.td (+22-24) 
- (added) llvm/test/CodeGen/X86/apx/bls-egpr.ll (+125) 
- (modified) llvm/test/CodeGen/X86/bmi.ll (+4-4) 


``````````diff
diff --git a/llvm/lib/Target/X86/X86InstrMisc.td b/llvm/lib/Target/X86/X86InstrMisc.td
index 14a2f477450a3..d5adc8686bd14 100644
--- a/llvm/lib/Target/X86/X86InstrMisc.td
+++ b/llvm/lib/Target/X86/X86InstrMisc.td
@@ -1276,38 +1276,36 @@ multiclass Bls_Pats<string suffix> {
             (!cast<Instruction>(BLSI32rr#suffix) GR32:$src)>;
   def : Pat<(and_flag_nocf GR64:$src, (ineg_su GR64:$src)),
             (!cast<Instruction>(BLSI64rr#suffix) GR64:$src)>;
-}
-
-let Predicates = [HasBMI, NoEGPR] in
-  defm : Bls_Pats<"">;
-
-let Predicates = [HasBMI, HasEGPR] in
-  defm : Bls_Pats<"_EVEX">;
 
-// BLSI only has GR32/GR64 forms. For an i8 operand we any-extend into a
-// 32-bit register, isolate the lowest set bit with BLSI32, and extract
-// the low byte. Any-extend is safe because BLSI keeps only the lowest
-// set bit, so upper-bit garbage cannot affect the extracted byte.
-let Predicates = [HasBMI] in
+  // BLSI only has GR32/GR64 forms. For an i8 operand we any-extend into a
+  // 32-bit register, isolate the lowest set bit with BLSI32, and extract
+  // the low byte. Any-extend is safe because BLSI keeps only the lowest
+  // set bit, so upper-bit garbage cannot affect the extracted byte.
   def : Pat<(and GR8:$src, (ineg_su GR8:$src)),
             (EXTRACT_SUBREG
-              (BLSI32rr (INSERT_SUBREG (i32 (IMPLICIT_DEF)),
-                                       GR8:$src, sub_8bit)),
+              (!cast<Instruction>(BLSI32rr#suffix)
+                (INSERT_SUBREG (i32 (IMPLICIT_DEF)), GR8:$src, sub_8bit)),
               sub_8bit)>;
 
-// BLSMSK only has GR32/GR64 forms. For an i8 operand we use INSERT_SUBREG
-// with IMPLICIT_DEF (any-extend) rather than SUBREG_TO_REG (zero-extend).
-// Although BLSMSK computes (src XOR (src-1)), dirty upper bits [31:8] cannot
-// corrupt bits [7:0] of the result: borrow from (src-1) only propagates
-// downward through contiguous zeros, and even when [7:0] is zero the
-// EXTRACT_SUBREG still yields the correct all-ones result. This matches the
-// same approach used for BLSI.
-let Predicates = [HasBMI] in
+  // BLSMSK only has GR32/GR64 forms. For an i8 operand we use INSERT_SUBREG
+  // with IMPLICIT_DEF (any-extend) rather than SUBREG_TO_REG (zero-extend).
+  // Although BLSMSK computes (src XOR (src-1)), dirty upper bits [31:8] cannot
+  // corrupt bits [7:0] of the result: borrow from (src-1) only propagates
+  // downward through contiguous zeros, and even when [7:0] is zero the
+  // EXTRACT_SUBREG still yields the correct all-ones result. This matches the
+  // same approach used for BLSI.
   def : Pat<(xor GR8:$src, (add GR8:$src, -1)),
             (EXTRACT_SUBREG
-              (BLSMSK32rr (INSERT_SUBREG (i32 (IMPLICIT_DEF)),
-                                         GR8:$src, sub_8bit)),
+              (!cast<Instruction>(BLSMSK32rr#suffix)
+                (INSERT_SUBREG (i32 (IMPLICIT_DEF)), GR8:$src, sub_8bit)),
               sub_8bit)>;
+}
+
+let Predicates = [HasBMI, NoEGPR] in
+  defm : Bls_Pats<"">;
+
+let Predicates = [HasBMI, HasEGPR] in
+  defm : Bls_Pats<"_EVEX">;
 
 multiclass Bmi4VOp3<bits<8> o, string m, X86TypeInfo t, SDPatternOperator node,
                     X86FoldableSchedWrite sched, string Suffix = ""> {
diff --git a/llvm/test/CodeGen/X86/apx/bls-egpr.ll b/llvm/test/CodeGen/X86/apx/bls-egpr.ll
new file mode 100644
index 0000000000000..7885e8ffada8f
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/bls-egpr.ll
@@ -0,0 +1,125 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+bmi,+egpr --show-mc-encoding | FileCheck %s
+
+; Clobber all legacy GPRs so that the BLS* operands have to live in r16-r31,
+; which requires the EVEX forms of the instructions.
+
+define i8 @blsi8(i8 %x) nounwind {
+; CHECK-LABEL: blsi8:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp # encoding: [0x55]
+; CHECK-NEXT:    pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT:    pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT:    pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT:    pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT:    pushq %rbx # encoding: [0x53]
+; CHECK-NEXT:    movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    blsil %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT:    popq %rbx # encoding: [0x5b]
+; CHECK-NEXT:    popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT:    popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT:    popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT:    popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT:    popq %rbp # encoding: [0x5d]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  %neg = sub i8 0, %x
+  %r = and i8 %x, %neg
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  ret i8 %r
+}
+
+define i32 @blsi32(i32 %x) nounwind {
+; CHECK-LABEL: blsi32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp # encoding: [0x55]
+; CHECK-NEXT:    pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT:    pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT:    pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT:    pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT:    pushq %rbx # encoding: [0x53]
+; CHECK-NEXT:    movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    blsil %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT:    popq %rbx # encoding: [0x5b]
+; CHECK-NEXT:    popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT:    popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT:    popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT:    popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT:    popq %rbp # encoding: [0x5d]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  %neg = sub i32 0, %x
+  %r = and i32 %x, %neg
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  ret i32 %r
+}
+
+define i8 @blsmsk8(i8 %x) nounwind {
+; CHECK-LABEL: blsmsk8:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp # encoding: [0x55]
+; CHECK-NEXT:    pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT:    pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT:    pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT:    pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT:    pushq %rbx # encoding: [0x53]
+; CHECK-NEXT:    movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    blsmskl %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd0]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT:    popq %rbx # encoding: [0x5b]
+; CHECK-NEXT:    popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT:    popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT:    popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT:    popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT:    popq %rbp # encoding: [0x5d]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  %y = add i8 %x, -1
+  %r = xor i8 %x, %y
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  ret i8 %r
+}
+
+define i32 @blsmsk32(i32 %x) nounwind {
+; CHECK-LABEL: blsmsk32:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    pushq %rbp # encoding: [0x55]
+; CHECK-NEXT:    pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT:    pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT:    pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT:    pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT:    pushq %rbx # encoding: [0x53]
+; CHECK-NEXT:    movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    blsmskl %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd0]
+; CHECK-NEXT:    #APP
+; CHECK-NEXT:    #NO_APP
+; CHECK-NEXT:    movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT:    popq %rbx # encoding: [0x5b]
+; CHECK-NEXT:    popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT:    popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT:    popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT:    popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT:    popq %rbp # encoding: [0x5d]
+; CHECK-NEXT:    retq # encoding: [0xc3]
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  %y = add i32 %x, -1
+  %r = xor i32 %x, %y
+  call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+  ret i32 %r
+}
diff --git a/llvm/test/CodeGen/X86/bmi.ll b/llvm/test/CodeGen/X86/bmi.ll
index b0a56bd9b5fbd..e0bfef40ed358 100644
--- a/llvm/test/CodeGen/X86/bmi.ll
+++ b/llvm/test/CodeGen/X86/bmi.ll
@@ -2068,7 +2068,7 @@ define i8 @blsi8(i8 %x) {
 ;
 ; EGPR-LABEL: blsi8:
 ; EGPR:       # %bb.0:
-; EGPR-NEXT:    blsil %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
+; EGPR-NEXT:    blsil %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
 ; EGPR-NEXT:    # kill: def $al killed $al killed $eax
 ; EGPR-NEXT:    retq # encoding: [0xc3]
   %neg = sub i8 0, %x
@@ -2115,7 +2115,7 @@ define i8 @blsi8_trunc(i32 %x) {
 ;
 ; EGPR-LABEL: blsi8_trunc:
 ; EGPR:       # %bb.0:
-; EGPR-NEXT:    blsil %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
+; EGPR-NEXT:    blsil %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
 ; EGPR-NEXT:    # kill: def $al killed $al killed $eax
 ; EGPR-NEXT:    retq # encoding: [0xc3]
   %t = trunc i32 %x to i8
@@ -2298,7 +2298,7 @@ define i8 @blsmsk8(i8 %x) nounwind {
 ;
 ; EGPR-LABEL: blsmsk8:
 ; EGPR:       # %bb.0:
-; EGPR-NEXT:    blsmskl %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
+; EGPR-NEXT:    blsmskl %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
 ; EGPR-NEXT:    # kill: def $al killed $al killed $eax
 ; EGPR-NEXT:    retq # encoding: [0xc3]
   %y = add i8 %x, -1
@@ -2322,7 +2322,7 @@ define i8 @blsmsk8_trunc(i32 %x) nounwind {
 ;
 ; EGPR-LABEL: blsmsk8_trunc:
 ; EGPR:       # %bb.0:
-; EGPR-NEXT:    blsmskl %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
+; EGPR-NEXT:    blsmskl %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
 ; EGPR-NEXT:    # kill: def $al killed $al killed $eax
 ; EGPR-NEXT:    retq # encoding: [0xc3]
   %t = trunc i32 %x to i8

``````````

</details>


https://github.com/llvm/llvm-project/pull/226796


More information about the llvm-commits mailing list