[llvm] a6e5d88 - [ValueTracking] Compute known bits for llvm.stepvector (#219779)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Aug 30 12:52:36 PDT 2026
Author: Oscar Priego
Date: 2026-08-30T19:52:30Z
New Revision: a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c
URL: https://github.com/llvm/llvm-project/commit/a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c
DIFF: https://github.com/llvm/llvm-project/commit/a6e5d88e8607e4e6ad283ea06c5c6f51110dff2c.diff
LOG: [ValueTracking] Compute known bits for llvm.stepvector (#219779)
Teach ValueTracking to infer high zero bits for `llvm.stepvector` from
the
vector element count and a finite `vscale_range`.
Conservatively give up when the lane-count calculation overflows the
element
width, since `llvm.stepvector` truncates out-of-range lane indices.
This allows existing sign-bit reasoning to eliminate redundant
extensions for
bounded step vectors.
The regression tests cover scalable and fixed vectors, bounded and
unbounded
`vscale_range`, lane-index truncation, unconstrained inputs, and
signed-i32
boundary cases.
Fixes #219776.
Added:
llvm/test/Transforms/InstCombine/stepvector-known-bits.ll
Modified:
llvm/lib/Analysis/ValueTracking.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index e028768f3a487..45748c4f35189 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -2285,6 +2285,30 @@ static void computeKnownBitsFromOperator(const Operator *I,
Known = getVScaleRange(II->getFunction(), BitWidth).toKnownBits();
break;
}
+ case Intrinsic::stepvector: {
+ auto *VecTy = cast<VectorType>(II->getType());
+ unsigned MinNumElts = VecTy->getElementCount().getKnownMinValue();
+ if (!isUIntN(BitWidth, MinNumElts))
+ break;
+
+ bool Overflow = false;
+ APInt MaxNumElts(BitWidth, MinNumElts);
+ if (VecTy->isScalableTy()) {
+ if (!II->getParent() || !II->getFunction())
+ break;
+ MaxNumElts = getVScaleRange(II->getFunction(), BitWidth)
+ .getUnsignedMax()
+ .umul_ov(MaxNumElts, Overflow);
+ }
+
+ // Give up if the lane count could wrap. Stepvector truncates lane
+ // indices that do not fit in the element type.
+ if (Overflow)
+ break;
+
+ Known.Zero.setHighBits((MaxNumElts - 1).countl_zero());
+ break;
+ }
}
}
break;
diff --git a/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll b/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll
new file mode 100644
index 0000000000000..40d4986804ecd
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/stepvector-known-bits.ll
@@ -0,0 +1,117 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=instcombine -S | FileCheck %s
+
+define <vscale x 2 x i64> @bounded(i32 %base) vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @bounded(
+; CHECK-SAME: i32 [[BASE:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[IN_RANGE:%.*]] = icmp ult i32 [[BASE]], 3841
+; CHECK-NEXT: call void @llvm.assume(i1 [[IN_RANGE]])
+; CHECK-NEXT: [[BASE64:%.*]] = zext nneg i32 [[BASE]] to i64
+; CHECK-NEXT: [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[BASE64]], i64 0
+; CHECK-NEXT: [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[SUM:%.*]] = add nuw nsw <vscale x 2 x i64> [[SPLAT]], [[LANE]]
+; CHECK-NEXT: ret <vscale x 2 x i64> [[SUM]]
+;
+ %in.range = icmp ult i32 %base, 3841
+ call void @llvm.assume(i1 %in.range)
+ %base64 = zext i32 %base to i64
+ %insert = insertelement <vscale x 2 x i64> poison, i64 %base64, i64 0
+ %splat = shufflevector <vscale x 2 x i64> %insert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %sum = add <vscale x 2 x i64> %splat, %lane
+ %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+ %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @unconstrained_base(i32 %base) vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @unconstrained_base(
+; CHECK-SAME: i32 [[BASE:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[BASE64:%.*]] = sext i32 [[BASE]] to i64
+; CHECK-NEXT: [[INSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[BASE64]], i64 0
+; CHECK-NEXT: [[SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[INSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[SUM:%.*]] = add <vscale x 2 x i64> [[SPLAT]], [[LANE]]
+; CHECK-NEXT: [[TMP1:%.*]] = shl <vscale x 2 x i64> [[SUM]], splat (i64 32)
+; CHECK-NEXT: [[SEXT:%.*]] = ashr exact <vscale x 2 x i64> [[TMP1]], splat (i64 32)
+; CHECK-NEXT: ret <vscale x 2 x i64> [[SEXT]]
+;
+ %base64 = sext i32 %base to i64
+ %insert = insertelement <vscale x 2 x i64> poison, i64 %base64, i64 0
+ %splat = shufflevector <vscale x 2 x i64> %insert, <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %sum = add <vscale x 2 x i64> %splat, %lane
+ %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+ %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @unbounded_vscale() vscale_range(2, 0) {
+; CHECK-LABEL: define <vscale x 2 x i64> @unbounded_vscale(
+; CHECK-SAME: ) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl <vscale x 2 x i64> [[LANE]], splat (i64 32)
+; CHECK-NEXT: [[SEXT:%.*]] = ashr exact <vscale x 2 x i64> [[TMP1]], splat (i64 32)
+; CHECK-NEXT: ret <vscale x 2 x i64> [[SEXT]]
+;
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %trunc = trunc <vscale x 2 x i64> %lane to <vscale x 2 x i32>
+ %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 192 x i16> @lane_index_truncates() vscale_range(1, 2) {
+; CHECK-LABEL: define <vscale x 192 x i16> @lane_index_truncates(
+; CHECK-SAME: ) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 192 x i8> @llvm.stepvector.nxv192i8()
+; CHECK-NEXT: [[SEXT:%.*]] = sext <vscale x 192 x i8> [[LANE]] to <vscale x 192 x i16>
+; CHECK-NEXT: ret <vscale x 192 x i16> [[SEXT]]
+;
+ %lane = call <vscale x 192 x i8> @llvm.stepvector.nxv192i8()
+ %wide = zext <vscale x 192 x i8> %lane to <vscale x 192 x i16>
+ %trunc = trunc <vscale x 192 x i16> %wide to <vscale x 192 x i8>
+ %sext = sext <vscale x 192 x i8> %trunc to <vscale x 192 x i16>
+ ret <vscale x 192 x i16> %sext
+}
+
+define <vscale x 2 x i64> @signed_max() vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @signed_max(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[SUM:%.*]] = or disjoint <vscale x 2 x i64> [[LANE]], splat (i64 2147481600)
+; CHECK-NEXT: ret <vscale x 2 x i64> [[SUM]]
+;
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %sum = add <vscale x 2 x i64> %lane, splat (i64 2147481600)
+ %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+ %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %sext
+}
+
+define <vscale x 2 x i64> @past_signed_max() vscale_range(2, 1024) {
+; CHECK-LABEL: define <vscale x 2 x i64> @past_signed_max(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT: [[LANE:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-NEXT: [[TMP1:%.*]] = trunc nuw nsw <vscale x 2 x i64> [[LANE]] to <vscale x 2 x i32>
+; CHECK-NEXT: [[TRUNC:%.*]] = add nuw <vscale x 2 x i32> [[TMP1]], splat (i32 2147481601)
+; CHECK-NEXT: [[SEXT:%.*]] = sext <vscale x 2 x i32> [[TRUNC]] to <vscale x 2 x i64>
+; CHECK-NEXT: ret <vscale x 2 x i64> [[SEXT]]
+;
+ %lane = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+ %sum = add <vscale x 2 x i64> %lane, splat (i64 2147481601)
+ %trunc = trunc <vscale x 2 x i64> %sum to <vscale x 2 x i32>
+ %sext = sext <vscale x 2 x i32> %trunc to <vscale x 2 x i64>
+ ret <vscale x 2 x i64> %sext
+}
+
+define <4 x i64> @fixed() {
+; CHECK-LABEL: define <4 x i64> @fixed() {
+; CHECK-NEXT: [[LANE:%.*]] = call <4 x i64> @llvm.stepvector.v4i64()
+; CHECK-NEXT: ret <4 x i64> [[LANE]]
+;
+ %lane = call <4 x i64> @llvm.stepvector.v4i64()
+ %trunc = trunc <4 x i64> %lane to <4 x i32>
+ %sext = sext <4 x i32> %trunc to <4 x i64>
+ ret <4 x i64> %sext
+}
More information about the llvm-commits
mailing list