[llvm] [GlobalISel] Fold instructions with fully-known bits to a constant. (PR #224254)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 03:10:32 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-globalisel
Author: Vikash Gupta (vg0204)
<details>
<summary>Changes</summary>
Port SelectionDAG's `SimplifyDemandedBits` "all demanded bits known -> constant" shortcut to GlobalISel: when value tracking proves every bit of a def is known, replacing the instruction with the materialized constant.
It adds a `known_bits_to_constant` combine (rooted on `G_AND`, `G_OR`, `G_ZEXT`, `G_SEXT`, `G_TRUNC`) to the `known_bits_simplifications` group. Unlike the `constant_fold` rules, this fires when the result is fully known even though the operands are only partially known.
---
Patch is 41.62 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/224254.diff
18 Files Affected:
- (modified) llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h (+5)
- (modified) llvm/include/llvm/Target/GlobalISel/Combine.td (+9)
- (modified) llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp (+29)
- (added) llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll (+35)
- (added) llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir (+202)
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir (+2-3)
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir (+2-3)
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir (+14-8)
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll (+12-9)
- (modified) llvm/test/CodeGen/AArch64/arm64-vshift.ll (+1-2)
- (modified) llvm/test/CodeGen/AArch64/hadd-combine.ll (+13-43)
- (modified) llvm/test/CodeGen/AArch64/neon-compare-instructions.ll (+9-26)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir (+2-3)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shl-from-extend-narrow.postlegal.mir (+4-4)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shl-from-extend-narrow.prelegal.mir (+6-6)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-short-clamp-inverted-bounds.mir (+6-2)
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll (+30-48)
- (modified) llvm/test/CodeGen/RISCV/GlobalISel/rotl-rotr.ll (+17-6)
``````````diff
diff --git a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
index 9f020ba7be528..2059a92f59594 100644
--- a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
+++ b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
@@ -574,6 +574,11 @@ class CombinerHelper {
LLVM_ABI bool matchRedundantAnd(MachineInstr &MI,
Register &Replacement) const;
+ /// \return true if all bits of \p MI's result are known, storing that
+ /// constant in \p MatchInfo so \p MI can be replaced by that constant.
+ LLVM_ABI bool matchKnownBitsToConstant(MachineInstr &MI,
+ APInt &MatchInfo) const;
+
/// \return true if \p MI is a G_OR instruction whose operands are x and y
/// where x | y == x or x | y == y. (E.g., one of operands is all-zeros
/// value.)
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 3735e95ee8f68..0c7d8681f17f7 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -1005,6 +1005,14 @@ def overlapping_and: GICombineRule <
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
>;
+// Fold a fully-known result to its constant (SelectionDAG's SimplifyDemandedBits
+// "all bits known -> constant" shortcut).
+def known_bits_to_constant : GICombineRule<
+ (defs root:$root, apint_matchinfo:$matchinfo),
+ (match (wip_match_opcode G_AND, G_OR, G_ZEXT, G_SEXT, G_TRUNC):$root,
+ [{ return Helper.matchKnownBitsToConstant(*${root}, ${matchinfo}); }]),
+ (apply [{ Helper.replaceInstWithConstant(*${root}, ${matchinfo}); }])>;
+
// Fold (x & y) -> x or (x & y) -> y when (x & y) is known to equal x or equal y.
def redundant_and: GICombineRule <
(defs root:$root, register_matchinfo:$matchinfo),
@@ -2775,6 +2783,7 @@ def const_combines : GICombineGroup<[constant_fold_fp_ops, const_ptradd_to_i2p,
combine_minmax_nan, expand_const_fpowi]>;
def known_bits_simplifications : GICombineGroup<[
+ known_bits_to_constant,
redundant_sext_inreg, redundant_zext_sext_inreg, redundant_aext_sext_inreg,
redundant_zext_unmerge_sext_inreg, redundant_aext_unmerge_sext_inreg,
redundant_and, redundant_or, urem_pow2_to_mask, zext_trunc_fold,
diff --git a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
index 4dce66a74cacc..413fabbff0c33 100644
--- a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
@@ -3363,6 +3363,35 @@ bool CombinerHelper::matchRedundantAnd(MachineInstr &MI,
return false;
}
+bool CombinerHelper::matchKnownBitsToConstant(MachineInstr &MI,
+ APInt &MatchInfo) const {
+ if (!VT)
+ return false;
+
+ Register Dst = MI.getOperand(0).getReg();
+ LLT Ty = MRI.getType(Dst);
+
+ // Scalars and fixed vectors. getKnownBits intersects all lanes, so a constant
+ // vector result is necessarily a splat-constant.
+ if (!Ty.isScalar() && !Ty.isFixedVector())
+ return false;
+
+ // Don't materialize a def that already has a class/bank constraint into a
+ // constant.
+ if (!MRI.getRegClassOrRegBank(Dst).isNull())
+ return false;
+
+ if (!isConstantLegalOrBeforeLegalizer(Ty))
+ return false;
+
+ KnownBits Known = VT->getKnownBits(Dst);
+ if (!Known.isConstant())
+ return false;
+
+ MatchInfo = Known.getConstant();
+ return true;
+}
+
bool CombinerHelper::matchRedundantOr(MachineInstr &MI,
Register &Replacement) const {
// Given
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll
new file mode 100644
index 0000000000000..cd0c6f1e99d83
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll
@@ -0,0 +1,35 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=aarch64 -global-isel -global-isel-abort=1 -O1 %s -o - | FileCheck %s
+
+; lshr -> top byte only; the AND masks it away, so the result is known 0.
+define i32 @and_masks_to_zero(i32 %x) {
+; CHECK-LABEL: and_masks_to_zero:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov w0, wzr
+; CHECK-NEXT: ret
+ %s = lshr i32 %x, 24
+ %a = and i32 %s, 4294967040
+ ret i32 %a
+}
+
+; Each OR pins a disjoint half of the bits to 1, so every bit is known 1 -> -1.
+define i32 @or_pins_all_bits(i32 %x, i32 %y) {
+; CHECK-LABEL: or_pins_all_bits:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mov w0, #-1 // =0xffffffff
+; CHECK-NEXT: ret
+ %a = or i32 %x, 15
+ %b = or i32 %y, 4294967280
+ %o = or i32 %a, %b
+ ret i32 %o
+}
+
+; Negative: low byte unknown, so nothing is folded.
+define i32 @not_fully_known(i32 %x) {
+; CHECK-LABEL: not_fully_known:
+; CHECK: // %bb.0:
+; CHECK-NEXT: and w0, w0, #0xff
+; CHECK-NEXT: ret
+ %a = and i32 %x, 255
+ ret i32 %a
+}
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir
new file mode 100644
index 0000000000000..eb2e18392135b
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir
@@ -0,0 +1,202 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
+# RUN: llc -run-pass=aarch64-prelegalizer-combiner -mtriple aarch64-unknown-unknown %s -o - | FileCheck %s
+
+# known_bits_to_constant folds a fully-known result to a constant even when it is
+# built from partially-known operands, so the result equals neither operand and
+# redundant_and/redundant_or cannot fire.
+
+---
+# lshr by 24 -> bits 8..31 known 0; anding with 0xFFFFFF00 clears the low 8 -> 0.
+name: and_of_lshr_to_zero
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: and_of_lshr_to_zero
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 0
+ ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 24
+ %2:_(s32) = G_LSHR %0, %1
+ %3:_(s32) = G_CONSTANT i32 -256
+ %4:_(s32) = G_AND %2, %3
+ $w0 = COPY %4(s32)
+...
+---
+# Each or pins a disjoint half of the bits to 1, so every bit is known 1 -> -1.
+name: or_partial_to_allones
+body: |
+ bb.0:
+ liveins: $w0, $w1
+ ; CHECK-LABEL: name: or_partial_to_allones
+ ; CHECK: liveins: $w0, $w1
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 -1
+ ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = COPY $w1
+ %2:_(s32) = G_CONSTANT i32 15
+ %3:_(s32) = G_OR %0, %2
+ %4:_(s32) = G_CONSTANT i32 -16
+ %5:_(s32) = G_OR %1, %4
+ %6:_(s32) = G_OR %3, %5
+ $w0 = COPY %6(s32)
+...
+---
+# Vector: every lane is fully known, so getKnownBits proves a splat -> splat 0.
+name: vec_and_of_lshr_to_zero
+body: |
+ bb.0:
+ liveins: $q0
+ ; CHECK-LABEL: name: vec_and_of_lshr_to_zero
+ ; CHECK: liveins: $q0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 0
+ ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<4 x s32>) = G_BUILD_VECTOR [[C]](s32), [[C]](s32), [[C]](s32), [[C]](s32)
+ ; CHECK-NEXT: $q0 = COPY [[BUILD_VECTOR]](<4 x s32>)
+ %0:_(<4 x s32>) = COPY $q0
+ %1:_(s32) = G_CONSTANT i32 24
+ %2:_(<4 x s32>) = G_BUILD_VECTOR %1, %1, %1, %1
+ %3:_(<4 x s32>) = G_LSHR %0, %2
+ %4:_(s32) = G_CONSTANT i32 -256
+ %5:_(<4 x s32>) = G_BUILD_VECTOR %4, %4, %4, %4
+ %6:_(<4 x s32>) = G_AND %3, %5
+ $q0 = COPY %6(<4 x s32>)
+...
+---
+# trunc reads only the low 16 bits, which the AND forces to 0, so the trunc is
+# known 0 even though the wider AND result is not fully known.
+name: trunc_of_and_low_zero
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: trunc_of_and_low_zero
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s16) = G_CONSTANT i16 0
+ ; CHECK-NEXT: $h0 = COPY [[C]](s16)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 -65536
+ %2:_(s32) = G_AND %0, %1
+ %3:_(s16) = G_TRUNC %2
+ $h0 = COPY %3(s16)
+...
+---
+# zext of a known-0 narrow value (lshr result) is known 0 in the wider type.
+name: zext_of_known_zero
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: zext_of_known_zero
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 0
+ ; CHECK-NEXT: $x0 = COPY [[C]](s64)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 255
+ %2:_(s32) = G_AND %0, %1
+ %3:_(s32) = G_CONSTANT i32 8
+ %4:_(s32) = G_LSHR %2, %3
+ %5:_(s64) = G_ZEXT %4
+ $x0 = COPY %5(s64)
+...
+---
+# sext of a known-0 narrow value is known 0 (0 sign-extends to 0).
+name: sext_of_known_zero
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: sext_of_known_zero
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 0
+ ; CHECK-NEXT: $x0 = COPY [[C]](s64)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 65535
+ %2:_(s32) = G_AND %0, %1
+ %3:_(s32) = G_CONSTANT i32 16
+ %4:_(s32) = G_LSHR %2, %3
+ %5:_(s64) = G_SEXT %4
+ $x0 = COPY %5(s64)
+...
+---
+# trunc to a non-zero constant: (or x,0x0F)<<4 forces the low byte to 0xF0
+# (bits 0-3 shifted-in 0, bits 4-7 the forced ones); trunc s8 -> 0xF0.
+name: trunc_to_nonzero_const
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: trunc_to_nonzero_const
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s8) = G_CONSTANT i8 -16
+ ; CHECK-NEXT: $b0 = COPY [[C]](s8)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 15
+ %2:_(s32) = G_OR %0, %1
+ %3:_(s32) = G_CONSTANT i32 4
+ %4:_(s32) = G_SHL %2, %3
+ %5:_(s8) = G_TRUNC %4
+ $b0 = COPY %5(s8)
+...
+---
+# zext of a fully-known non-zero narrow value: the s8 (or t,0x0F)<<4 is 0xF0,
+# so zext to s32 is 0x000000F0 = 240.
+name: zext_to_nonzero_const
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: zext_to_nonzero_const
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 240
+ ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+ %0:_(s32) = COPY $w0
+ %t:_(s8) = G_TRUNC %0
+ %1:_(s8) = G_CONSTANT i8 15
+ %2:_(s8) = G_OR %t, %1
+ %3:_(s8) = G_CONSTANT i8 4
+ %4:_(s8) = G_SHL %2, %3
+ %5:_(s32) = G_ZEXT %4
+ $w0 = COPY %5(s32)
+...
+---
+# sext of the same 0xF0 s8: the sign bit is set, so sext to s32 is
+# 0xFFFFFFF0 = -16 (contrast with the zext case above).
+name: sext_to_nonzero_const
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: sext_to_nonzero_const
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 -16
+ ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+ %0:_(s32) = COPY $w0
+ %t:_(s8) = G_TRUNC %0
+ %1:_(s8) = G_CONSTANT i8 15
+ %2:_(s8) = G_OR %t, %1
+ %3:_(s8) = G_CONSTANT i8 4
+ %4:_(s8) = G_SHL %2, %3
+ %5:_(s32) = G_SEXT %4
+ $w0 = COPY %5(s32)
+...
+---
+# Negative: low 8 bits unknown, so the AND is preserved.
+name: partial_known_no_fold
+body: |
+ bb.0:
+ liveins: $w0
+ ; CHECK-LABEL: name: partial_known_no_fold
+ ; CHECK: liveins: $w0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(s32) = COPY $w0
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 255
+ ; CHECK-NEXT: [[AND:%[0-9]+]]:_(s32) = G_AND [[COPY]], [[C]]
+ ; CHECK-NEXT: $w0 = COPY [[AND]](s32)
+ %0:_(s32) = COPY $w0
+ %1:_(s32) = G_CONSTANT i32 255
+ %2:_(s32) = G_AND %0, %1
+ $w0 = COPY %2(s32)
+...
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
index 7bb166e0312c2..92d1a81098e1d 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
@@ -386,9 +386,8 @@ body: |
; CHECK-LABEL: name: test_fcmp_and_fcmp_with_vectors
; CHECK: liveins: $x0, $x1
; CHECK-NEXT: {{ $}}
- ; CHECK-NEXT: [[C:%[0-9]+]]:_(i1) = G_CONSTANT i1 false
- ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<2 x i1>) = G_BUILD_VECTOR [[C]](i1), [[C]](i1)
- ; CHECK-NEXT: %zext:_(<2 x i64>) = G_ZEXT [[BUILD_VECTOR]](<2 x i1>)
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
+ ; CHECK-NEXT: %zext:_(<2 x i64>) = G_BUILD_VECTOR [[C]](i64), [[C]](i64)
; CHECK-NEXT: $q0 = COPY %zext(<2 x i64>)
%0:_(f64) = COPY $x0
%1:_(f64) = COPY $x1
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
index df2cdb15ef899..9b98cd45da0de 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
@@ -157,9 +157,8 @@ body: |
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
; CHECK-NEXT: [[COPY1:%[0-9]+]]:_(i32) = COPY $w1
; CHECK-NEXT: %bv0:_(<4 x i32>) = G_BUILD_VECTOR [[COPY]](i32), [[COPY1]](i32), [[COPY]](i32), [[COPY1]](i32)
- ; CHECK-NEXT: [[C:%[0-9]+]]:_(i1) = G_CONSTANT i1 false
- ; CHECK-NEXT: %o:_(<4 x i1>) = G_BUILD_VECTOR [[C]](i1), [[C]](i1), [[C]](i1), [[C]](i1)
- ; CHECK-NEXT: %o_wide:_(<4 x i32>) = G_ZEXT %o(<4 x i1>)
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
+ ; CHECK-NEXT: %o_wide:_(<4 x i32>) = G_BUILD_VECTOR [[C]](i32), [[C]](i32), [[C]](i32), [[C]](i32)
; CHECK-NEXT: $q0 = COPY %bv0(<4 x i32>)
; CHECK-NEXT: $q1 = COPY %o_wide(<4 x i32>)
; CHECK-NEXT: RET_ReallyLR implicit $w0
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
index f478aa80e7fa2..09723b7b9959e 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
@@ -171,14 +171,20 @@ legalized: true
body: |
bb.1:
liveins: $w0
- ; CHECK-LABEL: name: test_combine_trunc_shl_s32_by_17
- ; CHECK: liveins: $w0
- ; CHECK-NEXT: {{ $}}
- ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
- ; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 17
- ; CHECK-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[COPY]], [[C]](i32)
- ; CHECK-NEXT: [[TRUNC:%[0-9]+]]:_(i16) = G_TRUNC [[SHL]](i32)
- ; CHECK-NEXT: $h0 = COPY [[TRUNC]](i16)
+ ; CHECK-PRE-LABEL: name: test_combine_trunc_shl_s32_by_17
+ ; CHECK-PRE: liveins: $w0
+ ; CHECK-PRE-NEXT: {{ $}}
+ ; CHECK-PRE-NEXT: [[C:%[0-9]+]]:_(i16) = G_CONSTANT i16 0
+ ; CHECK-PRE-NEXT: $h0 = COPY [[C]](i16)
+ ;
+ ; CHECK-POST-LABEL: name: test_combine_trunc_shl_s32_by_17
+ ; CHECK-POST: liveins: $w0
+ ; CHECK-POST-NEXT: {{ $}}
+ ; CHECK-POST-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
+ ; CHECK-POST-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 17
+ ; CHECK-POST-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[COPY]], [[C]](i32)
+ ; CHECK-POST-NEXT: [[TRUNC:%[0-9]+]]:_(i16) = G_TRUNC [[SHL]](i32)
+ ; CHECK-POST-NEXT: $h0 = COPY [[TRUNC]](i16)
%0:_(i32) = COPY $w0
%1:_(i32) = G_CONSTANT i32 17
%2:_(i32) = G_SHL %0(i32), %1(i32)
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll b/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
index ba53cb57c2ef2..76197f959559e 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
@@ -5,25 +5,28 @@ target triple = "arm64-apple-macosx11.0.0"
declare i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32>) #0
-define i32 @bar() {
+; The vectors are function arguments (not constants) so the shuffle result is not
+; known, ensuring the <4 x i1> shuffle still reaches the widen legalization path
+; this test guards rather than being folded to a constant beforehand.
+define i32 @bar(<8 x i1> %a, <8 x i1> %b) {
; CHECK-LABEL: bar:
; CHECK: ; %bb.0: ; %bb
-; CHECK-NEXT: movi.2d v0, #0000000000000000
+; CHECK-NEXT: ; kill: def $d0 killed $d0 def $q0
; CHECK-NEXT: umov.b w8, v0[0]
; CHECK-NEXT: umov.b w9, v0[1]
-; CHECK-NEXT: fmov s1, w8
+; CHECK-NEXT: movi.4s v1, #1
+; CHECK-NEXT: fmov s2, w8
; CHECK-NEXT: umov.b w8, v0[2]
-; CHECK-NEXT: mov.s v1[1], w9
+; CHECK-NEXT: mov.s v2[1], w9
; CHECK-NEXT: umov.b w9, v0[3]
-; CHECK-NEXT: movi.4s v0, #1
-; CHECK-NEXT: mov.s v1[2], w8
-; CHECK-NEXT: mov.s v1[3], w9
-; CHECK-NEXT: and.16b v0, v1, v0
+; CHECK-NEXT: mov.s v2[2], w8
+; CHECK-NEXT: mov.s v2[3], w9
+; CHECK-NEXT: and.16b v0, v2, v1
; CHECK-NEXT: addv.4s s0, v0
; CHECK-NEXT: fmov w0, s0
; CHECK-NEXT: ret
bb:
- %shufflevector = shufflevector <8 x i1> zeroinitializer, <8 x i1> zeroinitializer, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+ %shufflevector = shufflevector <8 x i1> %a, <8 x i1> %b, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
%zext = zext <4 x i1> %shufflevector to <4 x i32>
%call = call i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32> %zext)
%icmp = icmp eq i32 %call, 0
diff --git a/llvm/test/CodeGen/AArch64/arm64-vshift.ll b/llvm/test/CodeGen/AArch64/arm64-vshift.ll
index a7411457430c4..05db56ec848f1 100644
--- a/llvm/test/CodeGen/AArch64/arm64-vshift.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-vshift.ll
@@ -4658,8 +4658,7 @@ define <2 x i8> @shl_trunc_v2i64_v2i8(<2 x i64> %a) {
;
; CHECK-GI-LABEL: shl_trunc_v2i64_v2i8:
; CHECK-GI: // %bb.0:
-; CHECK-GI-NEXT: shl v0.2d, v0.2d, #16
-; CHECK-GI-NEXT: xtn v0.2s, v0.2d
+; CHECK-GI-NEXT: movi v0.2d, #0000000000000000
; CHECK-GI-NEXT: ret
%b = shl <2 x i64> %a, <i64 16, i64 16>
%c = trunc <2 x i64> %b to <2 x i8>
diff --git a/llvm/test/CodeGen/AArch64/hadd-combine.ll b/llvm/test/CodeGen/AArch64/hadd-combine.ll
index 450069cd27428..6dbe5b2c3ecf7 100644
--- a/llvm/test/CodeGen/AArch64/hadd-combine.ll
+++ b/llvm/test/CodeGen/AArch64/hadd-combine.ll
@@ -95,19 +95,10 @@ define <8 x i16> @haddu_const_both() {
}
define <8 x i16> @haddu_const_bothhigh() {
-; CHECK-SD-LABEL: haddu_const_bothhigh:
-; CHECK-SD: // %bb.0:
-; CHECK-SD-NEXT: mvni v0.8h, #1
-; CHECK-SD-NEXT: ret
-;
-; CHECK-GI-LABEL: haddu_const_bothhigh:
-; CHECK-GI: // %bb.0:
-; CHECK-GI-NEXT: movi d0, #0xffffffffffffffff
-; CHECK-GI-NEXT: mvni v1.4h, #1
-; CHECK-GI-NEXT: uaddl v1.4s, v1.4h, v0.4h
-; CHECK-GI-NEXT: shrn v0.4h, v1.4s, #1
-; CHECK-GI-NEXT: shrn2 v0.8h, v1.4s, #1
-; CHECK-GI-NEXT: ret
+; CHECK-LABEL: haddu_const_bothhigh:
+; CHECK: // %bb.0:
+; CHECK-NEXT: mvni v0.8h, #1
+; CHECK-NEXT: ret
%ext1 = zext <8 x i16> <i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534> to <8 x i32>
%ext2 = zext <8 x i16> <i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535> to <8 x i32>
%add = add <8 x i32> %ext1, %ext2
@@ -341,9 +332,7 @@ define <8 x i16> @hadds_const_bothhigh() {
; CHECK-GI-LABEL: hadds_const_bothhigh:
; CHECK-GI: // %bb.0:
; CHECK-GI-NEXT: adrp x8, .LCPI19_0
-; CHECK-GI-NEXT: mvni v0.8h, #128, lsl #8
-; CHECK-GI-NEXT: ldr q1, [x8, :lo12:.LCPI19_0]
-; CHECK-GI-NEXT: shadd v0.8h, v1.8h, v0.8h
+; CHECK-GI-NEXT: ldr q0, [x8, :lo12:.LCPI19_0]
; CHECK-GI-NEXT: ret
%ext1 = sext <8 x i16> <i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766> to <8 x i32>
%ext2 = sext <8 x i16> <i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767> to <8 x i32>
@@ -615,21 +604,10 @@ define <8 x i16> @rhaddu_const_both() {
}
define <8 x i16> @rhaddu_const_bothhigh() {
-; CHECK-SD-LABEL: rhaddu_const_bothhigh:
-; CHECK-SD: // %bb.0:
-; CHECK-SD-NEXT: movi v0.2d, #0xffffffffffffffff
-; CHECK-SD-NEXT: ret
-;
-; CHECK-GI-...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/224254
More information about the llvm-commits
mailing list