[llvm] c3a8b92 - [RISCV] Attach VLMAX range attribute for vsetvli/vsetvlimax in InstCombine

via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 23:10:22 PDT 2026


Author: Pengcheng Wang
Date: 2026-08-28T14:10:17+08:00
New Revision: c3a8b929c432b8d3adb69c9a2e397838a7441044

URL: https://github.com/llvm/llvm-project/commit/c3a8b929c432b8d3adb69c9a2e397838a7441044
DIFF: https://github.com/llvm/llvm-project/commit/c3a8b929c432b8d3adb69c9a2e397838a7441044.diff

LOG: [RISCV] Attach VLMAX range attribute for vsetvli/vsetvlimax in InstCombine

Attach a range return attribute to riscv_vsetvli/vsetvlimax so the generic
value analyses can reason about the result via CallBase::getRange(), using the
subtarget's real VLEN instead of the architectural maximum.

VLMAX = VLEN * LMUL / SEW. vsetvlimax returns exactly VLMAX; vsetvli returns
0 <= vl <= min(AVL, VLMAX), which equals AVL only when AVL cannot exceed the
smallest possible VLMAX. Otherwise vl may shrink below VLMAX (to 0 at runtime),
so we only claim the VLMAX-derived upper bound.

Fixes #217784.

Assisted-by: TRAE CLI (Opus 4.8)

Reviewers: efriedma-quic, preames, lenary, lukel97

Reviewed By: lukel97

Pull Request: https://github.com/llvm/llvm-project/pull/218652

Added: 
    llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
    llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvlimax-range.ll

Modified: 
    llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
index 59e60aef0305f..b6e4ea982f889 100644
--- a/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
+++ b/llvm/lib/Target/RISCV/RISCVTargetTransformInfo.cpp
@@ -3752,6 +3752,65 @@ bool RISCVTTIImpl::shouldCopyAttributeWhenOutliningFrom(
 
 std::optional<Instruction *>
 RISCVTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
+  // Attach a range return attribute describing the result of vsetvli/vsetvlimax
+  // so generic value analyses can reason about it. The verifier guarantees an
+  // XLen result and constant VSEW/VLMUL encoding a valid vtype, so no defensive
+  // validation is needed here.
+  if (is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
+                   II.getIntrinsicID())) {
+    // These intrinsics require the V extension; without it the VLEN queries
+    // below would assert. Such IR would fail isel anyway, so just bail out.
+    if (!ST->hasVInstructions())
+      return {};
+
+    bool HasAVL = II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
+    unsigned Offset = HasAVL ? 1 : 0;
+    unsigned BitWidth = II.getType()->getIntegerBitWidth();
+    ConstantRange VLenRange(APInt(BitWidth, ST->getRealMinVLen()),
+                            APInt(BitWidth, ST->getRealMaxVLen()) + 1);
+
+    uint64_t VSEW = cast<ConstantInt>(II.getArgOperand(Offset))->getZExtValue();
+    auto VLMUL = static_cast<RISCVVType::VLMUL>(
+        cast<ConstantInt>(II.getArgOperand(Offset + 1))->getZExtValue());
+    unsigned SEW = RISCVVType::decodeVSEW(VSEW);
+    unsigned Ratio = RISCVVType::getSEWLMULRatio(SEW, VLMUL);
+
+    // VLMAX = VLEN / (SEW / LMUL), clamped to >= 1 for any usable vtype.
+    ConstantRange VLMAXRange =
+        VLenRange.udiv(ConstantRange(APInt(BitWidth, Ratio)))
+            .umax(ConstantRange(APInt(BitWidth, 1)));
+
+    // vsetvlimax returns exactly VLMAX; vsetvli returns vl with
+    // 0 <= vl <= min(AVL, VLMAX). vl == AVL only when AVL <= the smallest
+    // possible VLMAX; otherwise vl can shrink below VLMAX (to 0 at runtime), so
+    // only the VLMAX upper bound is sound.
+    ConstantRange VLRange = VLMAXRange;
+    if (HasAVL) {
+      APInt MaxVL = VLMAXRange.getUnsignedMax();
+      if (auto *AVL = dyn_cast<ConstantInt>(II.getArgOperand(0))) {
+        const APInt &C = AVL->getValue();
+        // A constant AVL not exceeding the smallest possible VLMAX means vl is
+        // exactly AVL, so replace the intrinsic with that constant.
+        if (C.ule(VLMAXRange.getUnsignedMin()))
+          return IC.replaceInstUsesWith(II, ConstantInt::get(II.getType(), C));
+        VLRange = ConstantRange::getNonEmpty(APInt::getZero(BitWidth),
+                                             APIntOps::umin(C, MaxVL) + 1);
+      } else {
+        VLRange =
+            ConstantRange::getNonEmpty(APInt::getZero(BitWidth), MaxVL + 1);
+      }
+    }
+
+    ConstantRange OldRange =
+        II.getRange().value_or(ConstantRange::getFull(BitWidth));
+    ConstantRange NewRange = VLRange.intersectWith(OldRange);
+    if (NewRange != OldRange) {
+      II.addRangeRetAttr(NewRange);
+      return &II;
+    }
+    return {};
+  }
+
   // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
   // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
   // creating redundant masks.

diff  --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
new file mode 100644
index 0000000000000..ef1fe05c79d60
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvli-range.ll
@@ -0,0 +1,125 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -p instcombine -mtriple=riscv64 -mattr=+v -S %s | FileCheck %s --check-prefixes=CHECK,VLEN128
+; RUN: opt -p instcombine -mtriple=riscv64 -mattr=+v,+zvl512b -S %s | FileCheck %s --check-prefixes=CHECK,VLEN512
+
+; RISCVTTIImpl::instCombineIntrinsic attaches a range return attribute to
+; vsetvli. The result is vl = f(AVL, VLMAX) with 0 <= vl <= min(AVL, VLMAX).
+; A constant AVL not exceeding the smallest possible VLMAX folds to that
+; constant. For any larger (or runtime) AVL, vl may shrink below VLMAX all the
+; way to 0, so only the VLMAX-derived upper bound is sound.
+
+;------------------------------------------------------------------------------
+; Constant AVL.
+;------------------------------------------------------------------------------
+
+; AVL below the smallest VLMAX (16 at VLEN128, 64 at VLEN512): vl == AVL, so the
+; intrinsic folds to the constant.
+define i64 @vsetvli_const_avl_below_min() {
+; CHECK-LABEL: define i64 @vsetvli_const_avl_below_min(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    ret i64 5
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 5, i64 0, i64 0)
+  ret i64 %vl
+}
+
+; AVL == 16: still <= the smallest VLMAX on both subtargets, so vl == AVL.
+define i64 @vsetvli_const_avl_at_min128() {
+; CHECK-LABEL: define i64 @vsetvli_const_avl_at_min128(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    ret i64 16
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 16, i64 0, i64 0)
+  ret i64 %vl
+}
+
+; AVL == 20 sits in (MinVLMAX, ...) at VLEN128, so vl is only known to be
+; [0, 20]; at VLEN512 it is still below VLMAX (64), so vl == 20 folds.
+define i64 @vsetvli_const_avl_mid() {
+; VLEN128-LABEL: define i64 @vsetvli_const_avl_mid(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 0, 21) i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvli_const_avl_mid(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT:    ret i64 20
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 20, i64 0, i64 0)
+  ret i64 %vl
+}
+
+; AVL far above the largest VLMAX (8192): vl is capped by VLMAX, giving the
+; VLEN-independent range [0, 8192].
+define i64 @vsetvli_const_avl_above_max() {
+; CHECK-LABEL: define i64 @vsetvli_const_avl_above_max(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    [[VL:%.*]] = call range(i64 0, 8193) i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+; CHECK-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 100000, i64 0, i64 0)
+  ret i64 %vl
+}
+
+; A small constant AVL (vl == 5) lets a masking AND disappear.
+define i64 @vsetvli_const_avl_and() {
+; CHECK-LABEL: define i64 @vsetvli_const_avl_and(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    ret i64 5
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 5, i64 0, i64 0)
+  %m = and i64 %vl, 7
+  ret i64 %m
+}
+
+; i32 result: small constant AVL is exact regardless of VLEN.
+define i32 @vsetvli_i32_const_avl_below_min() {
+; CHECK-LABEL: define i32 @vsetvli_i32_const_avl_below_min(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    ret i32 5
+;
+  %vl = call i32 @llvm.riscv.vsetvli.i32(i32 5, i32 0, i32 0)
+  ret i32 %vl
+}
+
+;------------------------------------------------------------------------------
+; Runtime AVL: vl in [0, VLMAX]. These guard against claiming an unsound VLMAX
+; lower bound, which would miscompile the compares below.
+;------------------------------------------------------------------------------
+
+; vl can be as small as 0 (AVL == 0), so this compare must NOT fold.
+define i1 @vsetvli_runtime_avl_eq0_not_folded(i64 %avl) {
+; CHECK-LABEL: define i1 @vsetvli_runtime_avl_eq0_not_folded(
+; CHECK-SAME: i64 [[AVL:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[VL:%.*]] = call range(i64 0, 8193) i64 @llvm.riscv.vsetvli.i64(i64 [[AVL]], i64 0, i64 0)
+; CHECK-NEXT:    [[C:%.*]] = icmp eq i64 [[VL]], 0
+; CHECK-NEXT:    ret i1 [[C]]
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
+  %c = icmp eq i64 %vl, 0
+  ret i1 %c
+}
+
+; vl can be below the smallest VLMAX (e.g. AVL == 3), so this must NOT fold.
+define i1 @vsetvli_runtime_avl_lt16_not_folded(i64 %avl) {
+; CHECK-LABEL: define i1 @vsetvli_runtime_avl_lt16_not_folded(
+; CHECK-SAME: i64 [[AVL:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    [[VL:%.*]] = call range(i64 0, 8193) i64 @llvm.riscv.vsetvli.i64(i64 [[AVL]], i64 0, i64 0)
+; CHECK-NEXT:    [[C:%.*]] = icmp samesign ult i64 [[VL]], 16
+; CHECK-NEXT:    ret i1 [[C]]
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
+  %c = icmp ult i64 %vl, 16
+  ret i1 %c
+}
+
+; vl <= VLMAX <= 8192 is sound, so an out-of-range compare DOES fold to false.
+define i1 @vsetvli_runtime_avl_gt_max_folds(i64 %avl) {
+; CHECK-LABEL: define i1 @vsetvli_runtime_avl_gt_max_folds(
+; CHECK-SAME: i64 [[AVL:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:    ret i1 false
+;
+  %vl = call i64 @llvm.riscv.vsetvli.i64(i64 %avl, i64 0, i64 0)
+  %c = icmp ugt i64 %vl, 8192
+  ret i1 %c
+}

diff  --git a/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvlimax-range.ll b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvlimax-range.ll
new file mode 100644
index 0000000000000..95430da922f00
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/RISCV/riscv-vsetvlimax-range.ll
@@ -0,0 +1,107 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -p instcombine -mtriple=riscv64 -mattr=+v -S %s | FileCheck %s --check-prefixes=CHECK,VLEN128
+; RUN: opt -p instcombine -mtriple=riscv64 -mattr=+v,+zvl512b -S %s | FileCheck %s --check-prefixes=CHECK,VLEN512
+
+; RISCVTTIImpl::instCombineIntrinsic attaches a range return attribute to
+; vsetvlimax. The result is exactly VLMAX = VLEN * LMUL / SEW, so the range uses
+; the subtarget's real VLEN bounds: a wider minimum VLEN (zvl512b) yields a
+; tighter lower bound.
+
+; e8m1: ratio 8, VLMAX = VLEN / 8.
+define i64 @vsetvlimax_e8m1() {
+; VLEN128-LABEL: define i64 @vsetvlimax_e8m1(
+; VLEN128-SAME: ) #[[ATTR0:[0-9]+]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 16, 8193) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvlimax_e8m1(
+; VLEN512-SAME: ) #[[ATTR0:[0-9]+]] {
+; VLEN512-NEXT:    [[VL:%.*]] = call range(i64 64, 8193) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+; VLEN512-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+  ret i64 %vl
+}
+
+; e8m8: ratio 1, VLMAX = VLEN.
+define i64 @vsetvlimax_e8m8() {
+; VLEN128-LABEL: define i64 @vsetvlimax_e8m8(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 128, 65537) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 3)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvlimax_e8m8(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT:    [[VL:%.*]] = call range(i64 512, 65537) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 3)
+; VLEN512-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 3)
+  ret i64 %vl
+}
+
+; e64m1: ratio 64, VLMAX = VLEN / 64.
+define i64 @vsetvlimax_e64m1() {
+; VLEN128-LABEL: define i64 @vsetvlimax_e64m1(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 2, 1025) i64 @llvm.riscv.vsetvlimax.i64(i64 3, i64 0)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvlimax_e64m1(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT:    [[VL:%.*]] = call range(i64 8, 1025) i64 @llvm.riscv.vsetvlimax.i64(i64 3, i64 0)
+; VLEN512-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 3, i64 0)
+  ret i64 %vl
+}
+
+; e64mf8: ratio 512. At small VLEN the lower bound underflows to 0, so VLMAX is
+; clamped to be non-zero; the result is VLEN-independent here.
+define i64 @vsetvlimax_e64mf8() {
+; CHECK-LABEL: define i64 @vsetvlimax_e64mf8(
+; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[VL:%.*]] = call range(i64 1, 129) i64 @llvm.riscv.vsetvlimax.i64(i64 3, i64 5)
+; CHECK-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 3, i64 5)
+  ret i64 %vl
+}
+
+; The result is always >= the minimum VLMAX (>= 16), so uge 8 folds to true.
+define i1 @vsetvlimax_cmp_folds() {
+; CHECK-LABEL: define i1 @vsetvlimax_cmp_folds(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    ret i1 true
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+  %c = icmp uge i64 %vl, 8
+  ret i1 %c
+}
+
+; The result fits in 14 bits (<= 8192), so masking the high bits is a nop.
+define i64 @vsetvlimax_and_high_bits() {
+; VLEN128-LABEL: define i64 @vsetvlimax_and_high_bits(
+; VLEN128-SAME: ) #[[ATTR0]] {
+; VLEN128-NEXT:    [[VL:%.*]] = call range(i64 16, 8193) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+; VLEN128-NEXT:    ret i64 [[VL]]
+;
+; VLEN512-LABEL: define i64 @vsetvlimax_and_high_bits(
+; VLEN512-SAME: ) #[[ATTR0]] {
+; VLEN512-NEXT:    [[VL:%.*]] = call range(i64 64, 8193) i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+; VLEN512-NEXT:    ret i64 [[VL]]
+;
+  %vl = call i64 @llvm.riscv.vsetvlimax.i64(i64 0, i64 0)
+  %m = and i64 %vl, 16383
+  ret i64 %m
+}
+
+; i32 result: e64mf8 clamps to a VLEN-independent [1, 128].
+define i32 @vsetvlimax_i32_e64mf8() {
+; CHECK-LABEL: define i32 @vsetvlimax_i32_e64mf8(
+; CHECK-SAME: ) #[[ATTR0]] {
+; CHECK-NEXT:    [[VL:%.*]] = call range(i32 1, 129) i32 @llvm.riscv.vsetvlimax.i32(i32 3, i32 5)
+; CHECK-NEXT:    ret i32 [[VL]]
+;
+  %vl = call i32 @llvm.riscv.vsetvlimax.i32(i32 3, i32 5)
+  ret i32 %vl
+}


        


More information about the llvm-commits mailing list