[llvm] [LoongArch] Use [X]VSETNEZ.V/[X]VSETEQZ.V for whole-vector zero check (PR #226814)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 27 10:34:55 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-loongarch

Author: lrzlin

<details>
<summary>Changes</summary>

Introduce `loongarch_vanynonzero` and `loongarch_vallzero` for checking wether a vector has any non zero or all of them are zero, which could simply some vector comparison such as the one below.

Before
```
xvseq.d $xr0, $xr0, $xr1
xvmskltz.d $xr0, $xr0
xvpickve2gr.wu $a0, $xr0, 0
xvpickve2gr.wu $a1, $xr0, 4
bstrins.d $a0, $a1, 3, 2
beqz $a0, .LBB2_2
```
After
```
xvseq.d $xr0, $xr0, $xr1
xvseteqz.v $fcc0, $xr0
bcnez $fcc0, .LBB2_2
```

This was largely used in early-exit loops such as `std::find`, the code snippet after the patch could match the one witch gcc will generate.

---
Full diff: https://github.com/llvm/llvm-project/pull/226814.diff


5 Files Affected:

- (modified) llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp (+49) 
- (modified) llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td (+15) 
- (modified) llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td (+21) 
- (added) llvm/test/CodeGen/LoongArch/lasx/vset-zero-test.ll (+129) 
- (added) llvm/test/CodeGen/LoongArch/lsx/vset-zero-test.ll (+100) 


``````````diff
diff --git a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
index 07f6a438bbb1b..66a0cd4ccff53 100644
--- a/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
+++ b/llvm/lib/Target/LoongArch/LoongArchISelLowering.cpp
@@ -7096,6 +7096,42 @@ static bool checkValueWidth(SDValue V, ISD::LoadExtType &ExtType) {
   return false;
 }
 
+// Fold a lane-mask extraction that is only compared against zero into one of
+// all-lane check instructions, which write the results into fcc register, so
+// BCEQZ could use it directly.
+//
+//   (VMSKLTZ X) != 0  ->  VSETNEZ.V X
+//   (VMSKLTZ X) == 0  ->  VSETEQZ.V X
+//
+// This is valid when X is the result of a vector compare instruction,
+// so each lane of X has to be all-ones or all-zeros.
+static SDValue foldVMskZeroTest(SDValue LHS, SDValue RHS, ISD::CondCode CC,
+                                const SDLoc &DL, SelectionDAG &DAG,
+                                const LoongArchSubtarget &Subtarget) {
+  if (CC != ISD::SETEQ && CC != ISD::SETNE)
+    return SDValue();
+  if (!isNullConstant(RHS))
+    return SDValue();
+
+  unsigned MskOpc = LHS.getOpcode();
+  if (MskOpc != LoongArchISD::VMSKLTZ && MskOpc != LoongArchISD::XVMSKLTZ)
+    return SDValue();
+  // Keeping the mask alive for another user would defeat the purpose.
+  if (!LHS.hasOneUse())
+    return SDValue();
+
+  SDValue Src = LHS.getOperand(0);
+  EVT SrcVT = Src.getValueType();
+  // Make sure every lane is all-ones or all-zeros.
+  if (!SrcVT.isVector() ||
+      DAG.ComputeNumSignBits(Src) != SrcVT.getScalarSizeInBits())
+    return SDValue();
+
+  return DAG.getNode(CC == ISD::SETNE ? LoongArchISD::VANYNONZERO
+                                      : LoongArchISD::VALLZERO,
+                     DL, Subtarget.getGRLenVT(), Src);
+}
+
 // Eliminate redundant truncation and zero-extension nodes.
 // * Case 1:
 //  +------------+ +------------+ +------------+
@@ -7160,6 +7196,11 @@ static SDValue performSETCCCombine(SDNode *N, SelectionDAG &DAG,
                                    const LoongArchSubtarget &Subtarget) {
   ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
 
+  if (N->getValueType(0) == Subtarget.getGRLenVT())
+    if (SDValue V = foldVMskZeroTest(N->getOperand(0), N->getOperand(1), CC,
+                                     SDLoc(N), DAG, Subtarget))
+      return V;
+
   SDNode *AndNode = N->getOperand(0).getNode();
   if (AndNode->getOpcode() != ISD::AND)
     return SDValue();
@@ -7474,6 +7515,14 @@ static SDValue performBR_CCCombine(SDNode *N, SelectionDAG &DAG,
   SDValue CC = N->getOperand(3);
   SDLoc DL(N);
 
+  // CC was folded into V, so always return ISD::SETNE is fine.
+  if (SDValue V = foldVMskZeroTest(LHS, RHS, cast<CondCodeSDNode>(CC)->get(),
+                                   DL, DAG, Subtarget))
+    return DAG.getNode(LoongArchISD::BR_CC, DL, N->getValueType(0),
+                       N->getOperand(0), V,
+                       DAG.getConstant(0, DL, Subtarget.getGRLenVT()),
+                       DAG.getCondCode(ISD::SETNE), N->getOperand(4));
+
   if (combine_CC(LHS, RHS, CC, DL, DAG, Subtarget))
     return DAG.getNode(LoongArchISD::BR_CC, DL, N->getValueType(0),
                        N->getOperand(0), LHS, RHS, CC, N->getOperand(4));
diff --git a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
index f3f19264ce3d3..fcd134770332d 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLASXInstrInfo.td
@@ -2789,4 +2789,19 @@ def : Pat<(int_loongarch_lasx_xvstelm_d v4i64:$xd, GPR:$rj, timm:$imm, timm:$idx
           (XVSTELM_D v4i64:$xd, GPR:$rj, (to_valid_timm timm:$imm),
                     (to_valid_timm timm:$idx))>;
 
+
+// Whole-vector zero tests, see the LSX counterparts.
+foreach vt = [v32i8, v16i16, v8i32, v4i64] in {
+def : Pat<(GRLenVT (loongarch_vanynonzero (vt LASX256:$xj))),
+          (MOVCF2GR (XVSETNEZ_V LASX256:$xj))>;
+def : Pat<(GRLenVT (loongarch_vallzero (vt LASX256:$xj))),
+          (MOVCF2GR (XVSETEQZ_V LASX256:$xj))>;
+def : Pat<(loongarch_brcc (GRLenVT (loongarch_vanynonzero (vt LASX256:$xj))), 0,
+                          SETNE, bb:$imm21),
+          (BCNEZ (XVSETNEZ_V LASX256:$xj), bb:$imm21)>;
+def : Pat<(loongarch_brcc (GRLenVT (loongarch_vallzero (vt LASX256:$xj))), 0,
+                          SETNE, bb:$imm21),
+          (BCNEZ (XVSETEQZ_V LASX256:$xj), bb:$imm21)>;
+}
+
 } // Predicates = [HasExtLASX]
diff --git a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
index 351f0a4cf86d2..e1489f9e36d69 100644
--- a/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
+++ b/llvm/lib/Target/LoongArch/LoongArchLSXInstrInfo.td
@@ -133,6 +133,12 @@ def loongarch_vftintrz_w_d: SDNode<"LoongArchISD::VFTINTRZ", SDT_LoongArchVFTINT
 // Vector 64-bit integer convert to single-precision
 def loongarch_vffint_s_l: SDNode<"LoongArchISD::VFFINT", SDT_LoongArchVFFINT_S_L>;
 
+// Whole-vector zero tests
+def loongarch_vanynonzero
+    : SDNode<"LoongArchISD::VANYNONZERO", SDT_LoongArchVMSKCOND>;
+def loongarch_vallzero
+    : SDNode<"LoongArchISD::VALLZERO", SDT_LoongArchVMSKCOND>;
+
 def immZExt1 : ImmLeaf<GRLenVT, [{return isUInt<1>(Imm);}]>;
 def immZExt2 : ImmLeaf<GRLenVT, [{return isUInt<2>(Imm);}]>;
 def immZExt3 : ImmLeaf<GRLenVT, [{return isUInt<3>(Imm);}]>;
@@ -2462,6 +2468,21 @@ def : Pat<(loongarch_vmskgez (v16i8 LSX128:$vj)), (PseudoVMSKGEZ_B LSX128:$vj)>;
 def : Pat<(loongarch_vmskeqz (v16i8 LSX128:$vj)), (PseudoVMSKEQZ_B LSX128:$vj)>;
 def : Pat<(loongarch_vmsknez (v16i8 LSX128:$vj)), (PseudoVMSKNEZ_B LSX128:$vj)>;
 
+// Whole-vector zero tests. The result lives in a condition flag register, so a
+// branch consumes it directly and only a materialization needs MOVCF2GR.
+foreach vt = [v16i8, v8i16, v4i32, v2i64] in {
+def : Pat<(GRLenVT (loongarch_vanynonzero (vt LSX128:$vj))),
+          (MOVCF2GR (VSETNEZ_V LSX128:$vj))>;
+def : Pat<(GRLenVT (loongarch_vallzero (vt LSX128:$vj))),
+          (MOVCF2GR (VSETEQZ_V LSX128:$vj))>;
+def : Pat<(loongarch_brcc (GRLenVT (loongarch_vanynonzero (vt LSX128:$vj))), 0,
+                          SETNE, bb:$imm21),
+          (BCNEZ (VSETNEZ_V LSX128:$vj), bb:$imm21)>;
+def : Pat<(loongarch_brcc (GRLenVT (loongarch_vallzero (vt LSX128:$vj))), 0,
+                          SETNE, bb:$imm21),
+          (BCNEZ (VSETEQZ_V LSX128:$vj), bb:$imm21)>;
+}
+
 } // Predicates = [HasExtLSX]
 
 /// Intrinsic pattern
diff --git a/llvm/test/CodeGen/LoongArch/lasx/vset-zero-test.ll b/llvm/test/CodeGen/LoongArch/lasx/vset-zero-test.ll
new file mode 100644
index 0000000000000..9226119da2183
--- /dev/null
+++ b/llvm/test/CodeGen/LoongArch/lasx/vset-zero-test.ll
@@ -0,0 +1,129 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lasx --verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,LA32
+; RUN: llc --mtriple=loongarch64 --mattr=+lasx --verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,LA64
+
+declare void @sink()
+
+define void @any_lane_set(ptr %src, ptr %dst) {
+; CHECK-LABEL: any_lane_set:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    xvld $xr0, $a0, 0
+; CHECK-NEXT:    xvld $xr1, $a1, 0
+; CHECK-NEXT:    xvseq.d $xr0, $xr0, $xr1
+; CHECK-NEXT:    xvsetnez.v $fcc0, $xr0
+; CHECK-NEXT:    movcf2gr $a0, $fcc0
+; CHECK-NEXT:    andi $a0, $a0, 1
+; CHECK-NEXT:    st.b $a0, $a1, 0
+; CHECK-NEXT:    ret
+  %a = load <4 x i64>, ptr %src
+  %b = load <4 x i64>, ptr %dst
+  %cmp = icmp eq <4 x i64> %a, %b
+  %bc = bitcast <4 x i1> %cmp to i4
+  %any = icmp ne i4 %bc, 0
+  store i1 %any, ptr %dst
+  ret void
+}
+
+define void @no_lane_set(ptr %src, ptr %dst) {
+; CHECK-LABEL: no_lane_set:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    xvld $xr0, $a0, 0
+; CHECK-NEXT:    xvld $xr1, $a1, 0
+; CHECK-NEXT:    xvseq.w $xr0, $xr0, $xr1
+; CHECK-NEXT:    xvseteqz.v $fcc0, $xr0
+; CHECK-NEXT:    movcf2gr $a0, $fcc0
+; CHECK-NEXT:    andi $a0, $a0, 1
+; CHECK-NEXT:    st.b $a0, $a1, 0
+; CHECK-NEXT:    ret
+  %a = load <8 x i32>, ptr %src
+  %b = load <8 x i32>, ptr %dst
+  %cmp = icmp eq <8 x i32> %a, %b
+  %bc = bitcast <8 x i1> %cmp to i8
+  %none = icmp eq i8 %bc, 0
+  store i1 %none, ptr %dst
+  ret void
+}
+
+define void @branch_any_lane_set(ptr %src, ptr %dst) {
+; LA32-LABEL: branch_any_lane_set:
+; LA32:       # %bb.0:
+; LA32-NEXT:    xvld $xr0, $a0, 0
+; LA32-NEXT:    xvld $xr1, $a1, 0
+; LA32-NEXT:    xvseq.d $xr0, $xr0, $xr1
+; LA32-NEXT:    xvseteqz.v $fcc0, $xr0
+; LA32-NEXT:    bcnez $fcc0, .LBB2_2
+; LA32-NEXT:  # %bb.1: # %hit
+; LA32-NEXT:    addi.w $sp, $sp, -16
+; LA32-NEXT:    .cfi_def_cfa_offset 16
+; LA32-NEXT:    st.w $ra, $sp, 12 # 4-byte Folded Spill
+; LA32-NEXT:    .cfi_offset 1, -4
+; LA32-NEXT:    bl sink
+; LA32-NEXT:    ld.w $ra, $sp, 12 # 4-byte Folded Reload
+; LA32-NEXT:    addi.w $sp, $sp, 16
+; LA32-NEXT:  .LBB2_2: # %exit
+; LA32-NEXT:    ret
+;
+; LA64-LABEL: branch_any_lane_set:
+; LA64:       # %bb.0:
+; LA64-NEXT:    xvld $xr0, $a0, 0
+; LA64-NEXT:    xvld $xr1, $a1, 0
+; LA64-NEXT:    xvseq.d $xr0, $xr0, $xr1
+; LA64-NEXT:    xvseteqz.v $fcc0, $xr0
+; LA64-NEXT:    bcnez $fcc0, .LBB2_2
+; LA64-NEXT:  # %bb.1: # %hit
+; LA64-NEXT:    addi.d $sp, $sp, -16
+; LA64-NEXT:    .cfi_def_cfa_offset 16
+; LA64-NEXT:    st.d $ra, $sp, 8 # 8-byte Folded Spill
+; LA64-NEXT:    .cfi_offset 1, -8
+; LA64-NEXT:    pcaddu18i $ra, %call36(sink)
+; LA64-NEXT:    jirl $ra, $ra, 0
+; LA64-NEXT:    ld.d $ra, $sp, 8 # 8-byte Folded Reload
+; LA64-NEXT:    addi.d $sp, $sp, 16
+; LA64-NEXT:  .LBB2_2: # %exit
+; LA64-NEXT:    ret
+  %a = load <4 x i64>, ptr %src
+  %b = load <4 x i64>, ptr %dst
+  %cmp = icmp eq <4 x i64> %a, %b
+  %bc = bitcast <4 x i1> %cmp to i4
+  %any = icmp ne i4 %bc, 0
+  br i1 %any, label %hit, label %exit
+hit:
+  call void @sink()
+  br label %exit
+exit:
+  ret void
+}
+
+;; x < 0 lowers to vmskltz.d applied to the raw vector rather
+;; than to a mask, so we must not trigger the test.
+;; For <1, 0> the mask is zero while the vector is nonzero.
+
+define void @any_lane_negative(ptr %src, ptr %dst) {
+; LA32-LABEL: any_lane_negative:
+; LA32:       # %bb.0:
+; LA32-NEXT:    xvld $xr0, $a0, 0
+; LA32-NEXT:    xvmskltz.d $xr0, $xr0
+; LA32-NEXT:    xvpickve2gr.wu $a0, $xr0, 0
+; LA32-NEXT:    xvpickve2gr.wu $a2, $xr0, 4
+; LA32-NEXT:    bstrins.w $a0, $a2, 3, 2
+; LA32-NEXT:    sltu $a0, $zero, $a0
+; LA32-NEXT:    st.b $a0, $a1, 0
+; LA32-NEXT:    ret
+;
+; LA64-LABEL: any_lane_negative:
+; LA64:       # %bb.0:
+; LA64-NEXT:    xvld $xr0, $a0, 0
+; LA64-NEXT:    xvmskltz.d $xr0, $xr0
+; LA64-NEXT:    xvpickve2gr.wu $a0, $xr0, 0
+; LA64-NEXT:    xvpickve2gr.wu $a2, $xr0, 4
+; LA64-NEXT:    bstrins.d $a0, $a2, 3, 2
+; LA64-NEXT:    sltu $a0, $zero, $a0
+; LA64-NEXT:    st.b $a0, $a1, 0
+; LA64-NEXT:    ret
+  %v = load <4 x i64>, ptr %src
+  %cmp = icmp slt <4 x i64> %v, zeroinitializer
+  %bc = bitcast <4 x i1> %cmp to i4
+  %any = icmp ne i4 %bc, 0
+  store i1 %any, ptr %dst
+  ret void
+}
diff --git a/llvm/test/CodeGen/LoongArch/lsx/vset-zero-test.ll b/llvm/test/CodeGen/LoongArch/lsx/vset-zero-test.ll
new file mode 100644
index 0000000000000..eb82b925ce0f7
--- /dev/null
+++ b/llvm/test/CodeGen/LoongArch/lsx/vset-zero-test.ll
@@ -0,0 +1,100 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
+;; RUN: llc --mtriple=loongarch32 --mattr=+32s,+lsx --verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,LA32
+; RUN: llc --mtriple=loongarch64 --mattr=+lsx --verify-machineinstrs < %s | FileCheck %s --check-prefixes=CHECK,LA64
+
+declare void @sink()
+
+define void @any_lane_set(ptr %src, ptr %dst) {
+; CHECK-LABEL: any_lane_set:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vld $vr0, $a0, 0
+; CHECK-NEXT:    vld $vr1, $a1, 0
+; CHECK-NEXT:    vseq.d $vr0, $vr0, $vr1
+; CHECK-NEXT:    vsetnez.v $fcc0, $vr0
+; CHECK-NEXT:    movcf2gr $a0, $fcc0
+; CHECK-NEXT:    andi $a0, $a0, 1
+; CHECK-NEXT:    st.b $a0, $a1, 0
+; CHECK-NEXT:    ret
+  %a = load <2 x i64>, ptr %src
+  %b = load <2 x i64>, ptr %dst
+  %cmp = icmp eq <2 x i64> %a, %b
+  %bc = bitcast <2 x i1> %cmp to i2
+  %any = icmp ne i2 %bc, 0
+  store i1 %any, ptr %dst
+  ret void
+}
+
+define void @no_lane_set(ptr %src, ptr %dst) {
+; CHECK-LABEL: no_lane_set:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vld $vr0, $a0, 0
+; CHECK-NEXT:    vld $vr1, $a1, 0
+; CHECK-NEXT:    vseq.w $vr0, $vr0, $vr1
+; CHECK-NEXT:    vseteqz.v $fcc0, $vr0
+; CHECK-NEXT:    movcf2gr $a0, $fcc0
+; CHECK-NEXT:    andi $a0, $a0, 1
+; CHECK-NEXT:    st.b $a0, $a1, 0
+; CHECK-NEXT:    ret
+  %a = load <4 x i32>, ptr %src
+  %b = load <4 x i32>, ptr %dst
+  %cmp = icmp eq <4 x i32> %a, %b
+  %bc = bitcast <4 x i1> %cmp to i4
+  %none = icmp eq i4 %bc, 0
+  store i1 %none, ptr %dst
+  ret void
+}
+
+define void @branch_any_lane_set(ptr %src, ptr %dst) {
+; CHECK-LABEL: branch_any_lane_set:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vld $vr0, $a0, 0
+; CHECK-NEXT:    vld $vr1, $a1, 0
+; CHECK-NEXT:    vseq.d $vr0, $vr0, $vr1
+; CHECK-NEXT:    vseteqz.v $fcc0, $vr0
+; CHECK-NEXT:    bcnez $fcc0, .LBB2_2
+; CHECK-NEXT:  # %bb.1: # %hit
+; CHECK-NEXT:    addi.d $sp, $sp, -16
+; CHECK-NEXT:    .cfi_def_cfa_offset 16
+; CHECK-NEXT:    st.d $ra, $sp, 8 # 8-byte Folded Spill
+; CHECK-NEXT:    .cfi_offset 1, -8
+; CHECK-NEXT:    pcaddu18i $ra, %call36(sink)
+; CHECK-NEXT:    jirl $ra, $ra, 0
+; CHECK-NEXT:    ld.d $ra, $sp, 8 # 8-byte Folded Reload
+; CHECK-NEXT:    addi.d $sp, $sp, 16
+; CHECK-NEXT:  .LBB2_2: # %exit
+; CHECK-NEXT:    ret
+  %a = load <2 x i64>, ptr %src
+  %b = load <2 x i64>, ptr %dst
+  %cmp = icmp eq <2 x i64> %a, %b
+  %bc = bitcast <2 x i1> %cmp to i2
+  %any = icmp ne i2 %bc, 0
+  br i1 %any, label %hit, label %exit
+hit:
+  call void @sink()
+  br label %exit
+exit:
+  ret void
+}
+
+;; x < 0 lowers to vmskltz.d applied to the raw vector rather
+;; than to a mask, so we must not trigger the test.
+;; For <1, 0> the mask is zero while the vector is nonzero.
+
+define void @any_lane_negative(ptr %src, ptr %dst) {
+; CHECK-LABEL: any_lane_negative:
+; CHECK:       # %bb.0:
+; CHECK-NEXT:    vld $vr0, $a0, 0
+; CHECK-NEXT:    vmskltz.d $vr0, $vr0
+; CHECK-NEXT:    vpickve2gr.hu $a0, $vr0, 0
+; CHECK-NEXT:    sltu $a0, $zero, $a0
+; CHECK-NEXT:    st.b $a0, $a1, 0
+; CHECK-NEXT:    ret
+  %v = load <2 x i64>, ptr %src
+  %cmp = icmp slt <2 x i64> %v, zeroinitializer
+  %bc = bitcast <2 x i1> %cmp to i2
+  %any = icmp ne i2 %bc, 0
+  store i1 %any, ptr %dst
+  ret void
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; LA64: {{.*}}

``````````

</details>


https://github.com/llvm/llvm-project/pull/226814


More information about the llvm-commits mailing list