[llvm] [X86][APX] Use EVEX BLSI/BLSMSK for i8 patterns with EGPR (PR #226796)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 07:56:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-x86
Author: Evgenii Kudriashov (e-kud)
<details>
<summary>Changes</summary>
This is a follow-up to #<!-- -->204746 and #<!-- -->205093, which added the i8 BLSI and
BLSMSK patterns.
The i8 BLSI/BLSMSK patterns were defined outside of Bls_Pats and always
selected the VEX forms, so with EGPR the register allocator could not
assign r16-r31 to their operands. Move them into Bls_Pats so that the
_EVEX variants are selected when EGPR is available.
Assisted-by: Claude Code
---
Full diff: https://github.com/llvm/llvm-project/pull/226796.diff
3 Files Affected:
- (modified) llvm/lib/Target/X86/X86InstrMisc.td (+22-24)
- (added) llvm/test/CodeGen/X86/apx/bls-egpr.ll (+125)
- (modified) llvm/test/CodeGen/X86/bmi.ll (+4-4)
``````````diff
diff --git a/llvm/lib/Target/X86/X86InstrMisc.td b/llvm/lib/Target/X86/X86InstrMisc.td
index 14a2f477450a3..d5adc8686bd14 100644
--- a/llvm/lib/Target/X86/X86InstrMisc.td
+++ b/llvm/lib/Target/X86/X86InstrMisc.td
@@ -1276,38 +1276,36 @@ multiclass Bls_Pats<string suffix> {
(!cast<Instruction>(BLSI32rr#suffix) GR32:$src)>;
def : Pat<(and_flag_nocf GR64:$src, (ineg_su GR64:$src)),
(!cast<Instruction>(BLSI64rr#suffix) GR64:$src)>;
-}
-
-let Predicates = [HasBMI, NoEGPR] in
- defm : Bls_Pats<"">;
-
-let Predicates = [HasBMI, HasEGPR] in
- defm : Bls_Pats<"_EVEX">;
-// BLSI only has GR32/GR64 forms. For an i8 operand we any-extend into a
-// 32-bit register, isolate the lowest set bit with BLSI32, and extract
-// the low byte. Any-extend is safe because BLSI keeps only the lowest
-// set bit, so upper-bit garbage cannot affect the extracted byte.
-let Predicates = [HasBMI] in
+ // BLSI only has GR32/GR64 forms. For an i8 operand we any-extend into a
+ // 32-bit register, isolate the lowest set bit with BLSI32, and extract
+ // the low byte. Any-extend is safe because BLSI keeps only the lowest
+ // set bit, so upper-bit garbage cannot affect the extracted byte.
def : Pat<(and GR8:$src, (ineg_su GR8:$src)),
(EXTRACT_SUBREG
- (BLSI32rr (INSERT_SUBREG (i32 (IMPLICIT_DEF)),
- GR8:$src, sub_8bit)),
+ (!cast<Instruction>(BLSI32rr#suffix)
+ (INSERT_SUBREG (i32 (IMPLICIT_DEF)), GR8:$src, sub_8bit)),
sub_8bit)>;
-// BLSMSK only has GR32/GR64 forms. For an i8 operand we use INSERT_SUBREG
-// with IMPLICIT_DEF (any-extend) rather than SUBREG_TO_REG (zero-extend).
-// Although BLSMSK computes (src XOR (src-1)), dirty upper bits [31:8] cannot
-// corrupt bits [7:0] of the result: borrow from (src-1) only propagates
-// downward through contiguous zeros, and even when [7:0] is zero the
-// EXTRACT_SUBREG still yields the correct all-ones result. This matches the
-// same approach used for BLSI.
-let Predicates = [HasBMI] in
+ // BLSMSK only has GR32/GR64 forms. For an i8 operand we use INSERT_SUBREG
+ // with IMPLICIT_DEF (any-extend) rather than SUBREG_TO_REG (zero-extend).
+ // Although BLSMSK computes (src XOR (src-1)), dirty upper bits [31:8] cannot
+ // corrupt bits [7:0] of the result: borrow from (src-1) only propagates
+ // downward through contiguous zeros, and even when [7:0] is zero the
+ // EXTRACT_SUBREG still yields the correct all-ones result. This matches the
+ // same approach used for BLSI.
def : Pat<(xor GR8:$src, (add GR8:$src, -1)),
(EXTRACT_SUBREG
- (BLSMSK32rr (INSERT_SUBREG (i32 (IMPLICIT_DEF)),
- GR8:$src, sub_8bit)),
+ (!cast<Instruction>(BLSMSK32rr#suffix)
+ (INSERT_SUBREG (i32 (IMPLICIT_DEF)), GR8:$src, sub_8bit)),
sub_8bit)>;
+}
+
+let Predicates = [HasBMI, NoEGPR] in
+ defm : Bls_Pats<"">;
+
+let Predicates = [HasBMI, HasEGPR] in
+ defm : Bls_Pats<"_EVEX">;
multiclass Bmi4VOp3<bits<8> o, string m, X86TypeInfo t, SDPatternOperator node,
X86FoldableSchedWrite sched, string Suffix = ""> {
diff --git a/llvm/test/CodeGen/X86/apx/bls-egpr.ll b/llvm/test/CodeGen/X86/apx/bls-egpr.ll
new file mode 100644
index 0000000000000..7885e8ffada8f
--- /dev/null
+++ b/llvm/test/CodeGen/X86/apx/bls-egpr.ll
@@ -0,0 +1,125 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+bmi,+egpr --show-mc-encoding | FileCheck %s
+
+; Clobber all legacy GPRs so that the BLS* operands have to live in r16-r31,
+; which requires the EVEX forms of the instructions.
+
+define i8 @blsi8(i8 %x) nounwind {
+; CHECK-LABEL: blsi8:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp # encoding: [0x55]
+; CHECK-NEXT: pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT: pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT: pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT: pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT: pushq %rbx # encoding: [0x53]
+; CHECK-NEXT: movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: blsil %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT: popq %rbx # encoding: [0x5b]
+; CHECK-NEXT: popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT: popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT: popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT: popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT: popq %rbp # encoding: [0x5d]
+; CHECK-NEXT: retq # encoding: [0xc3]
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ %neg = sub i8 0, %x
+ %r = and i8 %x, %neg
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ ret i8 %r
+}
+
+define i32 @blsi32(i32 %x) nounwind {
+; CHECK-LABEL: blsi32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp # encoding: [0x55]
+; CHECK-NEXT: pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT: pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT: pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT: pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT: pushq %rbx # encoding: [0x53]
+; CHECK-NEXT: movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: blsil %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT: popq %rbx # encoding: [0x5b]
+; CHECK-NEXT: popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT: popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT: popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT: popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT: popq %rbp # encoding: [0x5d]
+; CHECK-NEXT: retq # encoding: [0xc3]
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ %neg = sub i32 0, %x
+ %r = and i32 %x, %neg
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ ret i32 %r
+}
+
+define i8 @blsmsk8(i8 %x) nounwind {
+; CHECK-LABEL: blsmsk8:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp # encoding: [0x55]
+; CHECK-NEXT: pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT: pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT: pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT: pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT: pushq %rbx # encoding: [0x53]
+; CHECK-NEXT: movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: blsmskl %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd0]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT: popq %rbx # encoding: [0x5b]
+; CHECK-NEXT: popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT: popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT: popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT: popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT: popq %rbp # encoding: [0x5d]
+; CHECK-NEXT: retq # encoding: [0xc3]
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ %y = add i8 %x, -1
+ %r = xor i8 %x, %y
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ ret i8 %r
+}
+
+define i32 @blsmsk32(i32 %x) nounwind {
+; CHECK-LABEL: blsmsk32:
+; CHECK: # %bb.0:
+; CHECK-NEXT: pushq %rbp # encoding: [0x55]
+; CHECK-NEXT: pushq %r15 # encoding: [0x41,0x57]
+; CHECK-NEXT: pushq %r14 # encoding: [0x41,0x56]
+; CHECK-NEXT: pushq %r13 # encoding: [0x41,0x55]
+; CHECK-NEXT: pushq %r12 # encoding: [0x41,0x54]
+; CHECK-NEXT: pushq %rbx # encoding: [0x53]
+; CHECK-NEXT: movl %edi, %r16d # encoding: [0xd5,0x10,0x89,0xf8]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: blsmskl %r16d, %r16d # encoding: [0x62,0xfa,0x7c,0x00,0xf3,0xd0]
+; CHECK-NEXT: #APP
+; CHECK-NEXT: #NO_APP
+; CHECK-NEXT: movl %r16d, %eax # encoding: [0xd5,0x40,0x89,0xc0]
+; CHECK-NEXT: popq %rbx # encoding: [0x5b]
+; CHECK-NEXT: popq %r12 # encoding: [0x41,0x5c]
+; CHECK-NEXT: popq %r13 # encoding: [0x41,0x5d]
+; CHECK-NEXT: popq %r14 # encoding: [0x41,0x5e]
+; CHECK-NEXT: popq %r15 # encoding: [0x41,0x5f]
+; CHECK-NEXT: popq %rbp # encoding: [0x5d]
+; CHECK-NEXT: retq # encoding: [0xc3]
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ %y = add i32 %x, -1
+ %r = xor i32 %x, %y
+ call void asm sideeffect "", "~{rax},~{rbx},~{rcx},~{rdx},~{rsi},~{rdi},~{rbp},~{r8},~{r9},~{r10},~{r11},~{r12},~{r13},~{r14},~{r15}"()
+ ret i32 %r
+}
diff --git a/llvm/test/CodeGen/X86/bmi.ll b/llvm/test/CodeGen/X86/bmi.ll
index b0a56bd9b5fbd..e0bfef40ed358 100644
--- a/llvm/test/CodeGen/X86/bmi.ll
+++ b/llvm/test/CodeGen/X86/bmi.ll
@@ -2068,7 +2068,7 @@ define i8 @blsi8(i8 %x) {
;
; EGPR-LABEL: blsi8:
; EGPR: # %bb.0:
-; EGPR-NEXT: blsil %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
+; EGPR-NEXT: blsil %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
; EGPR-NEXT: # kill: def $al killed $al killed $eax
; EGPR-NEXT: retq # encoding: [0xc3]
%neg = sub i8 0, %x
@@ -2115,7 +2115,7 @@ define i8 @blsi8_trunc(i32 %x) {
;
; EGPR-LABEL: blsi8_trunc:
; EGPR: # %bb.0:
-; EGPR-NEXT: blsil %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
+; EGPR-NEXT: blsil %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xdf]
; EGPR-NEXT: # kill: def $al killed $al killed $eax
; EGPR-NEXT: retq # encoding: [0xc3]
%t = trunc i32 %x to i8
@@ -2298,7 +2298,7 @@ define i8 @blsmsk8(i8 %x) nounwind {
;
; EGPR-LABEL: blsmsk8:
; EGPR: # %bb.0:
-; EGPR-NEXT: blsmskl %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
+; EGPR-NEXT: blsmskl %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
; EGPR-NEXT: # kill: def $al killed $al killed $eax
; EGPR-NEXT: retq # encoding: [0xc3]
%y = add i8 %x, -1
@@ -2322,7 +2322,7 @@ define i8 @blsmsk8_trunc(i32 %x) nounwind {
;
; EGPR-LABEL: blsmsk8_trunc:
; EGPR: # %bb.0:
-; EGPR-NEXT: blsmskl %edi, %eax # encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
+; EGPR-NEXT: blsmskl %edi, %eax # EVEX TO VEX Compression encoding: [0xc4,0xe2,0x78,0xf3,0xd7]
; EGPR-NEXT: # kill: def $al killed $al killed $eax
; EGPR-NEXT: retq # encoding: [0xc3]
%t = trunc i32 %x to i8
``````````
</details>
https://github.com/llvm/llvm-project/pull/226796
More information about the llvm-commits
mailing list