[llvm] 57b06e6 - [X86] Add test showing missed write-mask fusion for multi-use setcc (#213669)

via llvm-commits llvm-commits at lists.llvm.org
Tue Aug 4 04:09:53 PDT 2026


Author: Timur Golubovich
Date: 2026-08-04T11:09:48Z
New Revision: 57b06e655bc3f46adb2171d89c4f304350c3ee1e

URL: https://github.com/llvm/llvm-project/commit/57b06e655bc3f46adb2171d89c4f304350c3ee1e
DIFF: https://github.com/llvm/llvm-project/commit/57b06e655bc3f46adb2171d89c4f304350c3ee1e.diff

LOG: [X86] Add test showing missed write-mask fusion for multi-use setcc (#213669)

Precommit the test for #213645

Added: 
    llvm/test/CodeGen/X86/avx512-masked-op-fusion.ll

Modified: 
    

Removed: 
    


################################################################################
diff  --git a/llvm/test/CodeGen/X86/avx512-masked-op-fusion.ll b/llvm/test/CodeGen/X86/avx512-masked-op-fusion.ll
new file mode 100644
index 0000000000000..9280366c78b9c
--- /dev/null
+++ b/llvm/test/CodeGen/X86/avx512-masked-op-fusion.ll
@@ -0,0 +1,58 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc < %s -mtriple=x86_64-unknown-linux-gnu -mcpu=x86-64-v4 | FileCheck %s
+
+; Verify that commuteSelect handles multi-use setcc conditions shared between
+; min and max vselects. The setcc should be inverted once and both selects
+; commuted, enabling ISel to emit fused write-masked vminps/vmaxps {%k}.
+
+define void @masked_min_max(ptr %pSrc, ptr %pMsk, i64 %n, ptr %pMin, ptr %pMax) {
+; CHECK-LABEL: masked_min_max:
+; CHECK:       # %bb.0: # %entry
+; CHECK-NEXT:    vbroadcastss {{.*#+}} zmm1 = [-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf,-Inf]
+; CHECK-NEXT:    vbroadcastss {{.*#+}} zmm0 = [+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf,+Inf]
+; CHECK-NEXT:    xorl %eax, %eax
+; CHECK-NEXT:    .p2align 4
+; CHECK-NEXT:  .LBB0_1: # %loop
+; CHECK-NEXT:    # =>This Inner Loop Header: Depth=1
+; CHECK-NEXT:    vmovaps %zmm1, %zmm2
+; CHECK-NEXT:    vmovaps %zmm0, %zmm1
+; CHECK-NEXT:    vmovdqu (%rsi,%rax), %xmm0
+; CHECK-NEXT:    vptestnmb %xmm0, %xmm0, %k1
+; CHECK-NEXT:    vmovups (%rdi,%rax,4), %zmm3
+; CHECK-NEXT:    vminps %zmm3, %zmm1, %zmm0
+; CHECK-NEXT:    vmovaps %zmm1, %zmm0 {%k1}
+; CHECK-NEXT:    vmaxps %zmm3, %zmm2, %zmm1
+; CHECK-NEXT:    vmovaps %zmm2, %zmm1 {%k1}
+; CHECK-NEXT:    addq $16, %rax
+; CHECK-NEXT:    cmpq %rdx, %rax
+; CHECK-NEXT:    jb .LBB0_1
+; CHECK-NEXT:  # %bb.2: # %exit
+; CHECK-NEXT:    vmovaps %zmm0, (%rcx)
+; CHECK-NEXT:    vmovaps %zmm1, (%r8)
+; CHECK-NEXT:    vzeroupper
+; CHECK-NEXT:    retq
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc_min = phi <16 x float> [ splat (float 0x7FF0000000000000), %entry ], [ %res_min, %loop ]
+  %acc_max = phi <16 x float> [ splat (float 0xFFF0000000000000), %entry ], [ %res_max, %loop ]
+  %msk_ptr = getelementptr inbounds i8, ptr %pMsk, i64 %iv
+  %msk_bytes = load <16 x i8>, ptr %msk_ptr, align 1
+  %cmp = icmp eq <16 x i8> %msk_bytes, zeroinitializer
+  %src_ptr = getelementptr inbounds float, ptr %pSrc, i64 %iv
+  %src = load <16 x float>, ptr %src_ptr, align 1
+  %min = tail call <16 x float> @llvm.x86.avx512.min.ps.512(<16 x float> %acc_min, <16 x float> %src, i32 4)
+  %res_min = select <16 x i1> %cmp, <16 x float> %acc_min, <16 x float> %min
+  %max = tail call <16 x float> @llvm.x86.avx512.max.ps.512(<16 x float> %acc_max, <16 x float> %src, i32 4)
+  %res_max = select <16 x i1> %cmp, <16 x float> %acc_max, <16 x float> %max
+  %iv.next = add nuw nsw i64 %iv, 16
+  %done = icmp uge i64 %iv.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  store <16 x float> %res_min, ptr %pMin, align 64
+  store <16 x float> %res_max, ptr %pMax, align 64
+  ret void
+}


        


More information about the llvm-commits mailing list