[llvm] afab119 - [SelectionDAG] Add computeKnownBits support for VP_LOAD_FF's second result (#227389)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 16:52:57 PDT 2026
Author: Min-Yih Hsu
Date: 2026-09-29T16:52:51-07:00
New Revision: afab119f6f50b23b54b538c98dc82c6ddb2391f6
URL: https://github.com/llvm/llvm-project/commit/afab119f6f50b23b54b538c98dc82c6ddb2391f6
DIFF: https://github.com/llvm/llvm-project/commit/afab119f6f50b23b54b538c98dc82c6ddb2391f6.diff
LOG: [SelectionDAG] Add computeKnownBits support for VP_LOAD_FF's second result (#227389)
The second result of `ISD::VP_LOAD_FF` is the new EVL, which is always
less than or equal to(1) the EVL operand of this intrinsic, and (2) the
largest length of the loaded vector. This patch teaches computeKnownBits
about these ranges so that DAGCombiner can eliminate things like
redundant z/sext + truncate on this new EVL.
In RISC-V, this means that if there are two consecutive fault only first
loads where the second one is using the EVL from the first load, there
won't be any redundant zext + truncate generated for the intermediate
EVL in between.
Added:
Modified:
llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
llvm/test/CodeGen/RISCV/rvv/fixed-vectors-vploadff.ll
llvm/test/CodeGen/RISCV/rvv/vploadff.ll
Removed:
################################################################################
diff --git a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
index f0b1613e73f7b..1bbc8b4e782df 100644
--- a/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/SelectionDAG.cpp
@@ -4541,6 +4541,24 @@ KnownBits SelectionDAG::computeKnownBits(SDValue Op, const APInt &DemandedElts,
Known, MF, MF.getFrameInfo().getObjectAlign(FrameIdx));
break;
}
+ case ISD::VP_LOAD_FF: {
+ if (Op.getResNo() != 1)
+ break;
+ // The second result of vp.load.ff is an unsigned value that is less than or
+ // equal to the EVL operand.
+ KnownBits VLKB =
+ computeKnownBits(Op.getOperand(3), DemandedElts, Depth + 1);
+ // The new VL is also bounded by the largest vector length.
+ EVT ResVT = Op->getValueType(0);
+ auto ResKB = KnownBits::makeConstant(
+ APInt(BitWidth, ResVT.getVectorMinNumElements()));
+ if (ResVT.isScalableVector()) {
+ const Function &F = getMachineFunction().getFunction();
+ ResKB = KnownBits::mul(getVScaleRange(&F, BitWidth).toKnownBits(), ResKB);
+ }
+ Known.Zero.setHighBits(KnownBits::umin(VLKB, ResKB).countMinLeadingZeros());
+ break;
+ }
default:
if (Opcode < ISD::BUILTIN_OP_END)
diff --git a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-vploadff.ll b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-vploadff.ll
index 5b01976dbbebd..5390fa11f9a4a 100644
--- a/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-vploadff.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/fixed-vectors-vploadff.ll
@@ -584,3 +584,16 @@ define { <7 x i8>, i32 } @vploadff_v7i8(ptr %ptr, <7 x i1> %m, i32 zeroext %evl)
%load = call { <7 x i8>, i32 } @llvm.vp.load.ff.v7i8.p0(ptr %ptr, <7 x i1> %m, i32 %evl)
ret { <7 x i8>, i32 } %load
}
+
+define i32 @vploadff_new_evl_knownbits(ptr %p, i32 zeroext %evl) {
+; CHECK-LABEL: vploadff_new_evl_knownbits:
+; CHECK: # %bb.0:
+; CHECK-NEXT: vsetvli zero, a1, e8, mf2, ta, ma
+; CHECK-NEXT: vle8ff.v v8, (a0)
+; CHECK-NEXT: csrr a0, vl
+; CHECK-NEXT: ret
+ %load = call {<8 x i8>, i32} @llvm.vp.load.ff(ptr %p, <8 x i1> splat (i1 true), i32 %evl)
+ %new.evl = extractvalue {<8 x i8>, i32} %load, 1
+ %ret = and i32 %new.evl, 31
+ ret i32 %ret
+}
diff --git a/llvm/test/CodeGen/RISCV/rvv/vploadff.ll b/llvm/test/CodeGen/RISCV/rvv/vploadff.ll
index 9e08938a9fe6c..782e0b796cb69 100644
--- a/llvm/test/CodeGen/RISCV/rvv/vploadff.ll
+++ b/llvm/test/CodeGen/RISCV/rvv/vploadff.ll
@@ -1006,3 +1006,41 @@ define { <vscale x 3 x i8>, i32 } @vploadff_nxv3i8(ptr %ptr, <vscale x 3 x i1> %
%load = call { <vscale x 3 x i8>, i32 } @llvm.vp.load.ff.nxv3i8.p0(ptr %ptr, <vscale x 3 x i1> %m, i32 %evl)
ret { <vscale x 3 x i8>, i32 } %load
}
+
+define <vscale x 8 x i8> @consecutive_vploadff(ptr %p0, ptr %p1) {
+; CHECK-LABEL: consecutive_vploadff:
+; CHECK: # %bb.0:
+; CHECK-NEXT: vsetivli zero, 16, e8, m1, ta, ma
+; CHECK-NEXT: vle8ff.v v8, (a0)
+; CHECK-NEXT: vle8ff.v v9, (a1)
+; CHECK-NEXT: vsetvli a0, zero, e8, m1, ta, ma
+; CHECK-NEXT: vadd.vv v8, v8, v9
+; CHECK-NEXT: ret
+ %ff1 = call {<vscale x 8 x i8>, i32} @llvm.vp.load.ff(ptr %p0, <vscale x 8 x i1> splat (i1 true), i32 16)
+ %vl1.new = extractvalue {<vscale x 8 x i8>, i32} %ff1, 1
+ %ff2 = call {<vscale x 8 x i8>, i32} @llvm.vp.load.ff(ptr %p1, <vscale x 8 x i1> splat (i1 true), i32 %vl1.new)
+
+ %val1 = extractvalue {<vscale x 8 x i8>, i32} %ff1, 0
+ %val2 = extractvalue {<vscale x 8 x i8>, i32} %ff2, 0
+ %ret = add nuw <vscale x 8 x i8> %val1, %val2
+ ret <vscale x 8 x i8> %ret
+}
+
+define <vscale x 8 x i8> @consecutive_vploadff_non_const_evl(ptr %p0, ptr %p1, i32 zeroext %vl) {
+; CHECK-LABEL: consecutive_vploadff_non_const_evl:
+; CHECK: # %bb.0:
+; CHECK-NEXT: vsetvli zero, a2, e8, m1, ta, ma
+; CHECK-NEXT: vle8ff.v v8, (a0)
+; CHECK-NEXT: vle8ff.v v9, (a1)
+; CHECK-NEXT: vsetvli a0, zero, e8, m1, ta, ma
+; CHECK-NEXT: vadd.vv v8, v8, v9
+; CHECK-NEXT: ret
+ %ff1 = call {<vscale x 8 x i8>, i32} @llvm.vp.load.ff(ptr %p0, <vscale x 8 x i1> splat (i1 true), i32 %vl)
+ %vl1.new = extractvalue {<vscale x 8 x i8>, i32} %ff1, 1
+ %ff2 = call {<vscale x 8 x i8>, i32} @llvm.vp.load.ff(ptr %p1, <vscale x 8 x i1> splat (i1 true), i32 %vl1.new)
+
+ %val1 = extractvalue {<vscale x 8 x i8>, i32} %ff1, 0
+ %val2 = extractvalue {<vscale x 8 x i8>, i32} %ff2, 0
+ %ret = add nuw <vscale x 8 x i8> %val1, %val2
+ ret <vscale x 8 x i8> %ret
+}
More information about the llvm-commits
mailing list