[llvm] [CodeGen] Add widening support for ISD::CTTZ_ELTS (PR #205841)

Chandana Mudda via llvm-commits llvm-commits at lists.llvm.org
Sun Jun 28 23:23:24 PDT 2026


https://github.com/chandmudda updated https://github.com/llvm/llvm-project/pull/205841

>From 5bf5abfb7ba3f5ae634158275beb450b8348cab2 Mon Sep 17 00:00:00 2001
From: Chandana Mudda <quic_csinderi at quicinc.com>
Date: Thu, 25 Jun 2026 07:55:47 -0700
Subject: [PATCH 1/2] [CodeGen] Add widening support for ISD::CTTZ_ELTS

WidenVectorOperand had no handler forCTTZ_ELTS/
CTTZ_ELTS_ZERO_POISON, causing a fatal error when the
input vector type needed widening.

Add WidenVecOp_CttzElements which widens the input vector
and pads the extra lanes with all-ones, ensuring they do
not contribute spurious trailing zeros to the count.
This follows the same pattern as the existing
WidenVecOp_VP_CttzElements.
---
 llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h |  1 +
 .../SelectionDAG/LegalizeVectorTypes.cpp      | 19 +++++++
 llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll  | 56 +++++++++++++++++++
 3 files changed, 76 insertions(+)
 create mode 100644 llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll

diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
index 71d3e1c66be86..d4d56a9563f71 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeTypes.h
@@ -1132,6 +1132,7 @@ class LLVM_LIBRARY_VISIBILITY DAGTypeLegalizer {
   SDValue WidenVecOp_VECREDUCE_SEQ(SDNode *N);
   SDValue WidenVecOp_VP_REDUCE(SDNode *N);
   SDValue WidenVecOp_ExpOp(SDNode *N);
+  SDValue WidenVecOp_CttzElements(SDNode *N);
   SDValue WidenVecOp_VP_CttzElements(SDNode *N);
   SDValue WidenVecOp_VECTOR_FIND_LAST_ACTIVE(SDNode *N);
 
diff --git a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
index 97aa765642ea7..78b99c49e07d0 100644
--- a/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/LegalizeVectorTypes.cpp
@@ -7741,6 +7741,10 @@ bool DAGTypeLegalizer::WidenVectorOperand(SDNode *N, unsigned OpNo) {
   case ISD::VP_REDUCE_FMINIMUM:
     Res = WidenVecOp_VP_REDUCE(N);
     break;
+  case ISD::CTTZ_ELTS:
+  case ISD::CTTZ_ELTS_ZERO_POISON:
+    Res = WidenVecOp_CttzElements(N);
+    break;
   case ISD::VP_CTTZ_ELTS:
   case ISD::VP_CTTZ_ELTS_ZERO_POISON:
     Res = WidenVecOp_VP_CttzElements(N);
@@ -8723,6 +8727,21 @@ SDValue DAGTypeLegalizer::WidenVecOp_VSELECT(SDNode *N) {
   return DAG.getExtractSubvector(DL, VT, Select, 0);
 }
 
+SDValue DAGTypeLegalizer::WidenVecOp_CttzElements(SDNode *N) {
+  SDLoc DL(N);
+  SDValue Source = N->getOperand(0);
+  EVT WideVT = GetWidenedVector(Source).getValueType();
+
+  // Pad the widened portion with all-ones so the extra lanes appear as
+  // active (non-zero) elements and do not contribute trailing zeros.
+  SDValue AllOnes = DAG.getAllOnesConstant(DL, WideVT);
+  SDValue WideSource = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, WideVT, AllOnes,
+                                   Source, DAG.getVectorIdxConstant(0, DL));
+
+  return DAG.getNode(N->getOpcode(), DL, N->getValueType(0), WideSource,
+                     N->getFlags());
+}
+
 SDValue DAGTypeLegalizer::WidenVecOp_VP_CttzElements(SDNode *N) {
   SDLoc DL(N);
   SDValue Source = GetWidenedVector(N->getOperand(0));
diff --git a/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll b/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll
new file mode 100644
index 0000000000000..ac262ad431ae5
--- /dev/null
+++ b/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll
@@ -0,0 +1,56 @@
+; RUN: llc -march=hexagon < %s | FileCheck %s
+
+; Widening of ISD::CTTZ_ELTS / CTTZ_ELTS_ZERO_POISON when the operand
+; vector type is illegal (<3 x i32> widens to <4 x i32>). Padded lanes
+; must be all-ones so the count never falls into the synthetic lane.
+
+; All-zero input must return OrigElts (3), never WideElts (4).
+; CHECK-LABEL: cttz_elts_zero_v3i32:
+; CHECK:        r0 = #3
+; CHECK-NEXT:   jumpr r31
+; CHECK-NOT:    r0 = #4
+define i32 @cttz_elts_zero_v3i32() {
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> zeroinitializer, i1 false)
+  ret i32 %res
+}
+
+; First non-zero element at the highest *original* lane (lane 2 of
+; <3 x i32>). The padded lane 3 must not steal the result: answer is 2.
+; CHECK-LABEL: cttz_elts_high_lane_v3i32:
+; CHECK:        r0 = #2
+; CHECK-NEXT:   jumpr r31
+; CHECK-NOT:    r0 = #3
+define i32 @cttz_elts_high_lane_v3i32() {
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> <i32 0, i32 0, i32 9>, i1 false)
+  ret i32 %res
+}
+
+; Symbolic input: confirm the operand is actually widened to <4 x i32>.
+; Two vcmpw.eq compares against the zero pair cover both halves of the
+; widened vector, and the final sub(#4, ...) reflects the widened lane
+; count. A regression that fails to widen, or widens without all-ones
+; padding, would change these.
+; CHECK-LABEL: cttz_elts_v3i32:
+; CHECK:        r{{[0-9]+}}:{{[0-9]+}} = combine(#0,#0)
+; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
+; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
+; CHECK:        r0 = sub(#4,r{{[0-9]+}})
+; CHECK-NEXT:   jumpr r31
+define i32 @cttz_elts_v3i32(<3 x i32> %v) {
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> %v, i1 false)
+  ret i32 %res
+}
+
+; Same shape for the zero-poison variant; padding must still be emitted.
+; CHECK-LABEL: cttz_elts_zero_poison_v3i32:
+; CHECK:        r{{[0-9]+}}:{{[0-9]+}} = combine(#0,#0)
+; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
+; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
+; CHECK:        r0 = sub(#4,r{{[0-9]+}})
+; CHECK-NEXT:   jumpr r31
+define i32 @cttz_elts_zero_poison_v3i32(<3 x i32> %v) {
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> %v, i1 true)
+  ret i32 %res
+}
+
+declare i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32>, i1)

>From 7f9b3c29bbf1cbd3331ecd03dd7f1b7032b663bc Mon Sep 17 00:00:00 2001
From: Chandana Mudda <quic_csinderi at quicinc.com>
Date: Sun, 28 Jun 2026 23:19:21 -0700
Subject: [PATCH 2/2] [CodeGen] Address review comments for CTTZ_ELTS widening

Add RISC-V <3 x i32> and AArch64 SVE <vscale x 3 x i1> predicate tests,
plus a <7 x i1> predicate case on Hexagon, for the new CTTZ_ELTS operand
widening. Regenerate the Hexagon checks with update_llc_test_checks.py
(--filter) so the assertions are auto-generated but scoped to the result
values and widening signature instead of full scheduling output.
---
 .../AArch64/intrinsic-cttz-elts-sve.ll        | 36 +++++++++++
 llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll  | 46 +++++++-------
 llvm/test/CodeGen/RISCV/rvv/cttz-elts.ll      | 60 +++++++++++++++++++
 3 files changed, 122 insertions(+), 20 deletions(-)

diff --git a/llvm/test/CodeGen/AArch64/intrinsic-cttz-elts-sve.ll b/llvm/test/CodeGen/AArch64/intrinsic-cttz-elts-sve.ll
index d7beaa5addddb..fa3673feaee46 100644
--- a/llvm/test/CodeGen/AArch64/intrinsic-cttz-elts-sve.ll
+++ b/llvm/test/CodeGen/AArch64/intrinsic-cttz-elts-sve.ll
@@ -121,6 +121,42 @@ define i32 @add_i32_ctz_nxv2i1_poison(<vscale x 2 x i1> %a, i32 %b) {
   ret i32 %add
 }
 
+define i32 @ctz_nxv3i1(<vscale x 3 x i1> %a) {
+; CHECK-LABEL: ctz_nxv3i1:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    rdvl x8, #1
+; CHECK-NEXT:    mov w9, #3 // =0x3
+; CHECK-NEXT:    ptrue p2.s
+; CHECK-NEXT:    lsr x8, x8, #4
+; CHECK-NEXT:    mul x8, x8, x9
+; CHECK-NEXT:    whilelo p1.s, xzr, x8
+; CHECK-NEXT:    not p1.b, p2/z, p1.b
+; CHECK-NEXT:    mov p0.b, p1/m, p1.b
+; CHECK-NEXT:    brkb p0.b, p2/z, p0.b
+; CHECK-NEXT:    cntp x0, p0, p0.s
+; CHECK-NEXT:    ret
+  %res = call i32 @llvm.experimental.cttz.elts.i32.nxv3i1(<vscale x 3 x i1> %a, i1 0)
+  ret i32 %res
+}
+
+define i32 @ctz_nxv3i1_poison(<vscale x 3 x i1> %a) {
+; CHECK-LABEL: ctz_nxv3i1_poison:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    rdvl x8, #1
+; CHECK-NEXT:    mov w9, #3 // =0x3
+; CHECK-NEXT:    ptrue p2.s
+; CHECK-NEXT:    lsr x8, x8, #4
+; CHECK-NEXT:    mul x8, x8, x9
+; CHECK-NEXT:    whilelo p1.s, xzr, x8
+; CHECK-NEXT:    not p1.b, p2/z, p1.b
+; CHECK-NEXT:    mov p0.b, p1/m, p1.b
+; CHECK-NEXT:    brkb p0.b, p2/z, p0.b
+; CHECK-NEXT:    cntp x0, p0, p0.s
+; CHECK-NEXT:    ret
+  %res = call i32 @llvm.experimental.cttz.elts.i32.nxv3i1(<vscale x 3 x i1> %a, i1 1)
+  ret i32 %res
+}
+
 define i32 @ctz_nxv4i1(<vscale x 4 x i1> %a) {
 ; CHECK-LABEL: ctz_nxv4i1:
 ; CHECK:       // %bb.0:
diff --git a/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll b/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll
index ac262ad431ae5..2f32a640dc5b0 100644
--- a/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll
+++ b/llvm/test/CodeGen/Hexagon/cttz-elts-widen.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --filter "r0 = #|vcmpw\.eq|sub\(#4|sub\(#8" --version 6
 ; RUN: llc -march=hexagon < %s | FileCheck %s
 
 ; Widening of ISD::CTTZ_ELTS / CTTZ_ELTS_ZERO_POISON when the operand
@@ -5,22 +6,18 @@
 ; must be all-ones so the count never falls into the synthetic lane.
 
 ; All-zero input must return OrigElts (3), never WideElts (4).
-; CHECK-LABEL: cttz_elts_zero_v3i32:
-; CHECK:        r0 = #3
-; CHECK-NEXT:   jumpr r31
-; CHECK-NOT:    r0 = #4
 define i32 @cttz_elts_zero_v3i32() {
+; CHECK-LABEL: cttz_elts_zero_v3i32:
+; CHECK:     r0 = #3
   %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> zeroinitializer, i1 false)
   ret i32 %res
 }
 
 ; First non-zero element at the highest *original* lane (lane 2 of
 ; <3 x i32>). The padded lane 3 must not steal the result: answer is 2.
-; CHECK-LABEL: cttz_elts_high_lane_v3i32:
-; CHECK:        r0 = #2
-; CHECK-NEXT:   jumpr r31
-; CHECK-NOT:    r0 = #3
 define i32 @cttz_elts_high_lane_v3i32() {
+; CHECK-LABEL: cttz_elts_high_lane_v3i32:
+; CHECK:     r0 = #2
   %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> <i32 0, i32 0, i32 9>, i1 false)
   ret i32 %res
 }
@@ -30,27 +27,36 @@ define i32 @cttz_elts_high_lane_v3i32() {
 ; widened vector, and the final sub(#4, ...) reflects the widened lane
 ; count. A regression that fails to widen, or widens without all-ones
 ; padding, would change these.
-; CHECK-LABEL: cttz_elts_v3i32:
-; CHECK:        r{{[0-9]+}}:{{[0-9]+}} = combine(#0,#0)
-; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
-; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
-; CHECK:        r0 = sub(#4,r{{[0-9]+}})
-; CHECK-NEXT:   jumpr r31
 define i32 @cttz_elts_v3i32(<3 x i32> %v) {
+; CHECK-LABEL: cttz_elts_v3i32:
+; CHECK:     p2 = vcmpw.eq(r1:0,r5:4)
+; CHECK:     p0 = vcmpw.eq(r7:6,r5:4)
+; CHECK:     r0 = sub(#4,r0)
   %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> %v, i1 false)
   ret i32 %res
 }
 
 ; Same shape for the zero-poison variant; padding must still be emitted.
-; CHECK-LABEL: cttz_elts_zero_poison_v3i32:
-; CHECK:        r{{[0-9]+}}:{{[0-9]+}} = combine(#0,#0)
-; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
-; CHECK:        vcmpw.eq(r{{[0-9]+}}:{{[0-9]+}},r{{[0-9]+}}:{{[0-9]+}})
-; CHECK:        r0 = sub(#4,r{{[0-9]+}})
-; CHECK-NEXT:   jumpr r31
 define i32 @cttz_elts_zero_poison_v3i32(<3 x i32> %v) {
+; CHECK-LABEL: cttz_elts_zero_poison_v3i32:
+; CHECK:     p2 = vcmpw.eq(r1:0,r5:4)
+; CHECK:     p0 = vcmpw.eq(r7:6,r5:4)
+; CHECK:     r0 = sub(#4,r0)
   %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> %v, i1 true)
   ret i32 %res
 }
 
 declare i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32>, i1)
+
+; Predicate-style operand: <7 x i1> widens to <8 x i1>. This is the most
+; common use case. The padded lane must be all-ones so the widened lane
+; count (8) only appears when no original lane is active; the final
+; sub(#8, ...) reflects the widened vector.
+define i32 @cttz_elts_v7i1(<7 x i1> %v) {
+; CHECK-LABEL: cttz_elts_v7i1:
+; CHECK:     r0 = sub(#8,r0)
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v7i1(<7 x i1> %v, i1 true)
+  ret i32 %res
+}
+
+declare i32 @llvm.experimental.cttz.elts.i32.v7i1(<7 x i1>, i1)
diff --git a/llvm/test/CodeGen/RISCV/rvv/cttz-elts.ll b/llvm/test/CodeGen/RISCV/rvv/cttz-elts.ll
index fc892c1a5cae0..e761f14b0d4e7 100644
--- a/llvm/test/CodeGen/RISCV/rvv/cttz-elts.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/cttz-elts.ll
@@ -201,4 +201,64 @@ define i16 @ctz_v8i1_i16_ret(<8 x i1> %a) {
   ret i16 %res
 }
 
+define i32 @ctz_v3i32(<3 x i32> %a) {
+; RV32-LABEL: ctz_v3i32:
+; RV32:       # %bb.0:
+; RV32-NEXT:    vsetivli zero, 4, e8, mf4, ta, ma
+; RV32-NEXT:    vmv.v.i v9, 0
+; RV32-NEXT:    vsetvli zero, zero, e32, m1, ta, ma
+; RV32-NEXT:    vmsne.vi v0, v8, 0
+; RV32-NEXT:    vfirst.m a0, v0
+; RV32-NEXT:    vsetvli zero, zero, e8, mf4, ta, ma
+; RV32-NEXT:    vmerge.vim v8, v9, 1, v0
+; RV32-NEXT:    vslidedown.vi v9, v8, 1
+; RV32-NEXT:    vmv.x.s a1, v9
+; RV32-NEXT:    vslidedown.vi v8, v8, 2
+; RV32-NEXT:    vmv.x.s a2, v8
+; RV32-NEXT:    seqz a0, a0
+; RV32-NEXT:    vmv.v.x v8, a0
+; RV32-NEXT:    vslide1down.vx v8, v8, a1
+; RV32-NEXT:    vslide1down.vx v8, v8, a2
+; RV32-NEXT:    li a0, 1
+; RV32-NEXT:    vslide1down.vx v8, v8, a0
+; RV32-NEXT:    vand.vi v8, v8, 1
+; RV32-NEXT:    vmsne.vi v8, v8, 0
+; RV32-NEXT:    vfirst.m a0, v8
+; RV32-NEXT:    bgez a0, .LBB9_2
+; RV32-NEXT:  # %bb.1:
+; RV32-NEXT:    li a0, 4
+; RV32-NEXT:  .LBB9_2:
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: ctz_v3i32:
+; RV64:       # %bb.0:
+; RV64-NEXT:    vsetivli zero, 4, e8, mf4, ta, ma
+; RV64-NEXT:    vmv.v.i v9, 0
+; RV64-NEXT:    vsetvli zero, zero, e32, m1, ta, ma
+; RV64-NEXT:    vmsne.vi v0, v8, 0
+; RV64-NEXT:    vfirst.m a0, v0
+; RV64-NEXT:    vsetvli zero, zero, e8, mf4, ta, ma
+; RV64-NEXT:    vmerge.vim v8, v9, 1, v0
+; RV64-NEXT:    vslidedown.vi v9, v8, 1
+; RV64-NEXT:    vmv.x.s a1, v9
+; RV64-NEXT:    vslidedown.vi v8, v8, 2
+; RV64-NEXT:    vmv.x.s a2, v8
+; RV64-NEXT:    seqz a0, a0
+; RV64-NEXT:    vmv.v.x v8, a0
+; RV64-NEXT:    vslide1down.vx v8, v8, a1
+; RV64-NEXT:    vslide1down.vx v8, v8, a2
+; RV64-NEXT:    li a0, 1
+; RV64-NEXT:    vslide1down.vx v8, v8, a0
+; RV64-NEXT:    vand.vi v8, v8, 1
+; RV64-NEXT:    vmsne.vi v8, v8, 0
+; RV64-NEXT:    vfirst.m a0, v8
+; RV64-NEXT:    bgez a0, .LBB9_2
+; RV64-NEXT:  # %bb.1:
+; RV64-NEXT:    li a0, 4
+; RV64-NEXT:  .LBB9_2:
+; RV64-NEXT:    ret
+  %res = call i32 @llvm.experimental.cttz.elts.i32.v3i32(<3 x i32> %a, i1 0)
+  ret i32 %res
+}
+
 attributes #0 = { vscale_range(2,1024) }



More information about the llvm-commits mailing list