[llvm] [IR][RFC] Add @llvm.mask.beforefirst intrinsic (PR #203874)

Luke Lau via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 07:18:10 PDT 2026


================
@@ -1741,6 +1747,40 @@ SDValue VectorLegalizer::ExpandLOOP_DEPENDENCE_MASK(SDNode *N) {
   return TLI.expandLoopDependenceMask(N, DAG);
 }
 
+SDValue VectorLegalizer::ExpandMASK_BEFOREFIRST(SDNode *N) {
+  // Expand to (setcc ult (step-vector), (cttz_elts x))
+  SDLoc DL(N);
+  EVT VT = N->getValueType(0);
+
+  // Try to expand via get_active_lane_mask if supported.
+  EVT BoolVT = VT.changeVectorElementType(*DAG.getContext(), MVT::i1);
+  EVT VecIdxVT = TLI.getVectorIdxTy(DAG.getDataLayout());
+  if (!TLI.shouldExpandGetActiveLaneMask(BoolVT, VecIdxVT)) {
+    // Perform cttz_elts in bool VT to get the custom target lowering.
+    SDValue CttzElts =
+        DAG.getNode(ISD::CTTZ_ELTS, DL, VecIdxVT,
+                    DAG.getNode(ISD::TRUNCATE, DL, BoolVT, N->getOperand(0)));
----------------
lukel97 wrote:

I think it's related to the AArch64 `optimizeBrk` combine. Without this we get:

```
--- a/llvm/test/CodeGen/AArch64/mask-beforefirst-sve.ll
+++ b/llvm/test/CodeGen/AArch64/mask-beforefirst-sve.ll
@@ -45,10 +45,16 @@ define <16 x i1> @v16i1(<16 x i1> %m) {
 ; CHECK-LABEL: v16i1:
 ; CHECK:       // %bb.0:
 ; CHECK-NEXT:    shl v0.16b, v0.16b, #7
-; CHECK-NEXT:    ptrue p0.b, vl16
-; CHECK-NEXT:    cmpne p1.b, p0/z, z0.b, #0
-; CHECK-NEXT:    brkb p1.b, p0/z, p1.b
-; CHECK-NEXT:    mov z0.b, p1/z, #-1 // =0xffffffffffffffff
+; CHECK-NEXT:    adrp x8, .LCPI4_0
+; CHECK-NEXT:    mov w9, #16 // =0x10
+; CHECK-NEXT:    ldr q1, [x8, :lo12:.LCPI4_0]
+; CHECK-NEXT:    cmlt v0.16b, v0.16b, #0
+; CHECK-NEXT:    and v0.16b, v0.16b, v1.16b
+; CHECK-NEXT:    umaxv b0, v0.16b
+; CHECK-NEXT:    fmov w8, s0
+; CHECK-NEXT:    sub w8, w9, w8, uxtb
+; CHECK-NEXT:    whilelo p0.b, xzr, x8
+; CHECK-NEXT:    mov z0.b, p0/z, #-1 // =0xffffffffffffffff
 ; CHECK-NEXT:    // kill: def $q0 killed $q0 killed $z0
 ; CHECK-NEXT:    ret

```

But yeah we can probably teach AArch64ISelLowering about this instead. Removed the truncate in 14bc2205c7764af9d05fc3e0766c215506afd30e

https://github.com/llvm/llvm-project/pull/203874


More information about the llvm-commits mailing list