[llvm] a6e5d88 - [ValueTracking] Compute known bits for llvm.stepvector (#219779)

via llvm-commits llvm-commits at lists.llvm.org
Sun Aug 30 12:52:36 PDT 2026


Author: Oscar Priego
Date: 2026-08-30T19:52:30Z
New Revision: a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c

URL: https://github.com/llvm/llvm-project/commit/a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c
DIFF: https://github.com/llvm/llvm-project/commit/a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c.diff

LOG: [ValueTracking] Compute known bits for llvm.stepvector (#219779)

Teach ValueTracking to infer high zero bits for `llvm.stepvector` from
the
vector element count and a finite `vscale_range`.

Conservatively give up when the lane-count calculation overflows the
element
width, since `llvm.stepvector` truncates out-of-range lane indices.

This allows existing sign-bit reasoning to eliminate redundant
extensions for
bounded step vectors.

The regression tests cover scalable and fixed vectors, bounded and
unbounded
`vscale_range`, lane-index truncation, unconstrained inputs, and
signed-i32
boundary cases.

Fixes #219776.

Added: 
    llvm/test/Transforms/InstCombine/stepvector-known-bits.ll

Modified: 
    llvm/lib/Analysis/ValueTracking.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index e028768f3a487..45748c4f35189 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -2285,6 +2285,30 @@ static void computeKnownBitsFromOperator(const Operator *I,
         Known = getVScaleRange(II->getFunction(), BitWidth).toKnownBits();
         break;
       }
+      case Intrinsic::stepvector: {
+        auto *VecTy = cast<VectorType>(II->getType());
+        unsigned MinNumElts = VecTy->getElementCount().getKnownMinValue();
+        if (!isUIntN(BitWidth, MinNumElts))
+          break;
+
+        bool Overflow = false;
+        APInt MaxNumElts(BitWidth, MinNumElts);
+        if (VecTy->isScalableTy()) {
+          if (!II->getParent() || !II->getFunction())
+            break;
+          MaxNumElts = getVScaleRange(II->getFunction(), BitWidth)
+                           .getUnsignedMax()
+                           .umul_ov(MaxNumElts, Overflow);
+        }
+
+        // Give up if the lane count could wrap. Stepvector truncates lane
+        // indices that do not fit in the element type.
+        if (Overflow)
+          break;
+
+        Known.Zero.setHighBits((MaxNumElts - 1).countl_zero());
+        break;
+      }
       }
     }
     break;

diff  --git a/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll b/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll
new file mode 100644
index 0000000000000..40d4986804ecd
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll
@@ -0,0 +1,117 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+define <vscale x 2 x i64> @bounded(i32 %base) vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @bounded(
+; CHECK-SAME: i32 [[BASE:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[IN_RANGE:%.*]] = icmp ult i32 [[BASE]], 3841
+; CHECK-NEXT:    call void @llvm.assume(i1 [[IN_RANGE]])
+; CHECK-NEXT:    [[BASE64:%.*]] = zext nneg i32 [[BASE]] to i64
+; CHECK-NEXT:    [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[BASE64]], i64 0
+; CHECK-NEXT:    [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[SUM:%.*]] = add nuw nsw <vscale x 2 x i64> [[SPLAT]], [[LANE]]
+; CHECK-NEXT:    ret <vscale x 2 x i64> [[SUM]]
+;
+  %in.range = icmp ult i32 %base, 3841
+  call void @llvm.assume(i1 %in.range)
+  %base64 = zext i32 %base to i64
+  %insert = insertelement <vscale x 2 x i64> poison, i64 %base64, i64 0
+  %splat = shufflevector <vscale x 2 x i64> %insert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %sum = add <vscale x 2 x i64> %splat, %lane
+  %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+  %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+  ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @unconstrained_base(i32 %base) vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @unconstrained_base(
+; CHECK-SAME: i32 [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[BASE64:%.*]] = sext i32 [[BASE]] to i64
+; CHECK-NEXT:    [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[BASE64]], i64 0
+; CHECK-NEXT:    [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[SUM:%.*]] = add <vscale x 2 x i64> [[SPLAT]], [[LANE]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl <vscale x 2 x i64> [[SUM]], splat (i64 32)
+; CHECK-NEXT:    [[SEXT:%.*]] = ashr exact <vscale x 2 x i64> [[TMP1]], splat (i64 32)
+; CHECK-NEXT:    ret <vscale x 2 x i64> [[SEXT]]
+;
+  %base64 = sext i32 %base to i64
+  %insert = insertelement <vscale x 2 x i64> poison, i64 %base64, i64 0
+  %splat = shufflevector <vscale x 2 x i64> %insert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %sum = add <vscale x 2 x i64> %splat, %lane
+  %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+  %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+  ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @unbounded_vscale() vscale_range(2, 0) {
+; CHECK-LABEL: define <vscale x 2 x i64> @unbounded_vscale(
+; CHECK-SAME: ) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP1:%.*]] = shl <vscale x 2 x i64> [[LANE]], splat (i64 32)
+; CHECK-NEXT:    [[SEXT:%.*]] = ashr exact <vscale x 2 x i64> [[TMP1]], splat (i64 32)
+; CHECK-NEXT:    ret <vscale x 2 x i64> [[SEXT]]
+;
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %trunc = trunc <vscale x 2 x i64> %lane to <vscale x 2 x i32>
+  %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+  ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 192 x i16> @lane_index_truncates() vscale_range(1, 2) {
+; CHECK-LABEL: define <vscale x 192 x i16> @lane_index_truncates(
+; CHECK-SAME: ) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 192 x i8> @llvm.stepvector.nxv192i8()
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <vscale x 192 x i8> [[LANE]] to <vscale x 192 x i16>
+; CHECK-NEXT:    ret <vscale x 192 x i16> [[SEXT]]
+;
+  %lane = call <vscale x 192 x i8> @llvm.stepvector.nxv192i8()
+  %wide = zext <vscale x 192 x i8> %lane to <vscale x 192 x i16>
+  %trunc = trunc <vscale x 192 x i16> %wide to <vscale x 192 x i8>
+  %sext = sext <vscale x 192 x i8> %trunc to <vscale x 192 x i16>
+  ret <vscale x 192 x i16> %sext
+}
+
+define <vscale x 2 x i64> @signed_max() vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @signed_max(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[SUM:%.*]] = or disjoint <vscale x 2 x i64> [[LANE]], splat (i64 2147481600)
+; CHECK-NEXT:    ret <vscale x 2 x i64> [[SUM]]
+;
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %sum = add <vscale x 2 x i64> %lane, splat (i64 2147481600)
+  %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+  %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+  ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @past_signed_max() vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @past_signed_max(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT:    [[TMP1:%.*]] = trunc nuw nsw <vscale x 2 x i64> [[LANE]] to <vscale x 2 x i32>
+; CHECK-NEXT:    [[TRUNC:%.*]] = add nuw <vscale x 2 x i32> [[TMP1]], splat (i32 2147481601)
+; CHECK-NEXT:    [[SEXT:%.*]] = sext <vscale x 2 x i32> [[TRUNC]] to <vscale x 2 x i64>
+; CHECK-NEXT:    ret <vscale x 2 x i64> [[SEXT]]
+;
+  %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+  %sum = add <vscale x 2 x i64> %lane, splat (i64 2147481601)
+  %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+  %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+  ret <vscale x 2 x i64> %sext
+}
+
+define <4 x i64> @fixed() {
+; CHECK-LABEL: define <4 x i64> @fixed() {
+; CHECK-NEXT:    [[LANE:%.*]] = call <4 x i64> @llvm.stepvector.v4i64()
+; CHECK-NEXT:    ret <4 x i64> [[LANE]]
+;
+  %lane = call <4 x i64> @llvm.stepvector.v4i64()
+  %trunc = trunc <4 x i64> %lane to <4 x i32>
+  %sext = sext <4 x i32> %trunc to <4 x i64>
+  ret <4 x i64> %sext
+}


        


More information about the llvm-commits mailing list