[llvm] [GlobalISel] Fold instructions with fully-known bits to a constant. (PR #224254)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 17 03:10:32 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-globalisel

Author: Vikash Gupta (vg0204)

<details>
<summary>Changes</summary>

Port SelectionDAG's `SimplifyDemandedBits` "all demanded bits known -> constant" shortcut to GlobalISel: when value tracking proves every bit of a def is known, replacing the instruction with the materialized constant.

It adds a `known_bits_to_constant` combine (rooted on `G_AND`, `G_OR`, `G_ZEXT`, `G_SEXT`, `G_TRUNC`) to the `known_bits_simplifications` group. Unlike the `constant_fold` rules, this fires when the result is fully known even though the operands are only partially known.

---

Patch is 41.62 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/224254.diff


18 Files Affected:

- (modified) llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h (+5) 
- (modified) llvm/include/llvm/Target/GlobalISel/Combine.td (+9) 
- (modified) llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp (+29) 
- (added) llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll (+35) 
- (added) llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir (+202) 
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir (+2-3) 
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir (+2-3) 
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir (+14-8) 
- (modified) llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll (+12-9) 
- (modified) llvm/test/CodeGen/AArch64/arm64-vshift.ll (+1-2) 
- (modified) llvm/test/CodeGen/AArch64/hadd-combine.ll (+13-43) 
- (modified) llvm/test/CodeGen/AArch64/neon-compare-instructions.ll (+9-26) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir (+2-3) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shl-from-extend-narrow.postlegal.mir (+4-4) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shl-from-extend-narrow.prelegal.mir (+6-6) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/combine-short-clamp-inverted-bounds.mir (+6-2) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll (+30-48) 
- (modified) llvm/test/CodeGen/RISCV/GlobalISel/rotl-rotr.ll (+17-6) 


``````````diff
diff --git a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
index 9f020ba7be528..2059a92f59594 100644
--- a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
+++ b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
@@ -574,6 +574,11 @@ class CombinerHelper {
   LLVM_ABI bool matchRedundantAnd(MachineInstr &MI,
                                   Register &Replacement) const;
 
+  /// \return true if all bits of \p MI's result are known, storing that
+  /// constant in \p MatchInfo so \p MI can be replaced by that constant.
+  LLVM_ABI bool matchKnownBitsToConstant(MachineInstr &MI,
+                                         APInt &MatchInfo) const;
+
   /// \return true if \p MI is a G_OR instruction whose operands are x and y
   /// where x | y == x or x | y == y. (E.g., one of operands is all-zeros
   /// value.)
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 3735e95ee8f68..0c7d8681f17f7 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -1005,6 +1005,14 @@ def overlapping_and: GICombineRule <
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
 >;
 
+// Fold a fully-known result to its constant (SelectionDAG's SimplifyDemandedBits
+// "all bits known -> constant" shortcut).
+def known_bits_to_constant : GICombineRule<
+  (defs root:$root, apint_matchinfo:$matchinfo),
+  (match (wip_match_opcode G_AND, G_OR, G_ZEXT, G_SEXT, G_TRUNC):$root,
+   [{ return Helper.matchKnownBitsToConstant(*${root}, ${matchinfo}); }]),
+  (apply [{ Helper.replaceInstWithConstant(*${root}, ${matchinfo}); }])>;
+
 // Fold (x & y) -> x or (x & y) -> y when (x & y) is known to equal x or equal y.
 def redundant_and: GICombineRule <
   (defs root:$root, register_matchinfo:$matchinfo),
@@ -2775,6 +2783,7 @@ def const_combines : GICombineGroup<[constant_fold_fp_ops, const_ptradd_to_i2p,
                                      combine_minmax_nan, expand_const_fpowi]>;
 
 def known_bits_simplifications : GICombineGroup<[
+  known_bits_to_constant,
   redundant_sext_inreg, redundant_zext_sext_inreg, redundant_aext_sext_inreg,
   redundant_zext_unmerge_sext_inreg, redundant_aext_unmerge_sext_inreg,
   redundant_and, redundant_or, urem_pow2_to_mask, zext_trunc_fold,
diff --git a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
index 4dce66a74cacc..413fabbff0c33 100644
--- a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
@@ -3363,6 +3363,35 @@ bool CombinerHelper::matchRedundantAnd(MachineInstr &MI,
   return false;
 }
 
+bool CombinerHelper::matchKnownBitsToConstant(MachineInstr &MI,
+                                              APInt &MatchInfo) const {
+  if (!VT)
+    return false;
+
+  Register Dst = MI.getOperand(0).getReg();
+  LLT Ty = MRI.getType(Dst);
+
+  // Scalars and fixed vectors. getKnownBits intersects all lanes, so a constant
+  // vector result is necessarily a splat-constant.
+  if (!Ty.isScalar() && !Ty.isFixedVector())
+    return false;
+
+  // Don't materialize a def that already has a class/bank constraint into a
+  // constant.
+  if (!MRI.getRegClassOrRegBank(Dst).isNull())
+    return false;
+
+  if (!isConstantLegalOrBeforeLegalizer(Ty))
+    return false;
+
+  KnownBits Known = VT->getKnownBits(Dst);
+  if (!Known.isConstant())
+    return false;
+
+  MatchInfo = Known.getConstant();
+  return true;
+}
+
 bool CombinerHelper::matchRedundantOr(MachineInstr &MI,
                                       Register &Replacement) const {
   // Given
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll
new file mode 100644
index 0000000000000..cd0c6f1e99d83
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.ll
@@ -0,0 +1,35 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
+; RUN: llc -mtriple=aarch64 -global-isel -global-isel-abort=1 -O1 %s -o - | FileCheck %s
+
+; lshr -> top byte only; the AND masks it away, so the result is known 0.
+define i32 @and_masks_to_zero(i32 %x) {
+; CHECK-LABEL: and_masks_to_zero:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mov w0, wzr
+; CHECK-NEXT:    ret
+  %s = lshr i32 %x, 24
+  %a = and i32 %s, 4294967040
+  ret i32 %a
+}
+
+; Each OR pins a disjoint half of the bits to 1, so every bit is known 1 -> -1.
+define i32 @or_pins_all_bits(i32 %x, i32 %y) {
+; CHECK-LABEL: or_pins_all_bits:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mov w0, #-1 // =0xffffffff
+; CHECK-NEXT:    ret
+  %a = or i32 %x, 15
+  %b = or i32 %y, 4294967280
+  %o = or i32 %a, %b
+  ret i32 %o
+}
+
+; Negative: low byte unknown, so nothing is folded.
+define i32 @not_fully_known(i32 %x) {
+; CHECK-LABEL: not_fully_known:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    and w0, w0, #0xff
+; CHECK-NEXT:    ret
+  %a = and i32 %x, 255
+  ret i32 %a
+}
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir
new file mode 100644
index 0000000000000..eb2e18392135b
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-known-bits-to-constant.mir
@@ -0,0 +1,202 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
+# RUN: llc -run-pass=aarch64-prelegalizer-combiner -mtriple aarch64-unknown-unknown %s -o - | FileCheck %s
+
+# known_bits_to_constant folds a fully-known result to a constant even when it is
+# built from partially-known operands, so the result equals neither operand and
+# redundant_and/redundant_or cannot fire.
+
+---
+# lshr by 24 -> bits 8..31 known 0; anding with 0xFFFFFF00 clears the low 8 -> 0.
+name:            and_of_lshr_to_zero
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: and_of_lshr_to_zero
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 0
+    ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 24
+    %2:_(s32) = G_LSHR %0, %1
+    %3:_(s32) = G_CONSTANT i32 -256
+    %4:_(s32) = G_AND %2, %3
+    $w0 = COPY %4(s32)
+...
+---
+# Each or pins a disjoint half of the bits to 1, so every bit is known 1 -> -1.
+name:            or_partial_to_allones
+body:             |
+  bb.0:
+    liveins: $w0, $w1
+    ; CHECK-LABEL: name: or_partial_to_allones
+    ; CHECK: liveins: $w0, $w1
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 -1
+    ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = COPY $w1
+    %2:_(s32) = G_CONSTANT i32 15
+    %3:_(s32) = G_OR %0, %2
+    %4:_(s32) = G_CONSTANT i32 -16
+    %5:_(s32) = G_OR %1, %4
+    %6:_(s32) = G_OR %3, %5
+    $w0 = COPY %6(s32)
+...
+---
+# Vector: every lane is fully known, so getKnownBits proves a splat -> splat 0.
+name:            vec_and_of_lshr_to_zero
+body:             |
+  bb.0:
+    liveins: $q0
+    ; CHECK-LABEL: name: vec_and_of_lshr_to_zero
+    ; CHECK: liveins: $q0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 0
+    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<4 x s32>) = G_BUILD_VECTOR [[C]](s32), [[C]](s32), [[C]](s32), [[C]](s32)
+    ; CHECK-NEXT: $q0 = COPY [[BUILD_VECTOR]](<4 x s32>)
+    %0:_(<4 x s32>) = COPY $q0
+    %1:_(s32) = G_CONSTANT i32 24
+    %2:_(<4 x s32>) = G_BUILD_VECTOR %1, %1, %1, %1
+    %3:_(<4 x s32>) = G_LSHR %0, %2
+    %4:_(s32) = G_CONSTANT i32 -256
+    %5:_(<4 x s32>) = G_BUILD_VECTOR %4, %4, %4, %4
+    %6:_(<4 x s32>) = G_AND %3, %5
+    $q0 = COPY %6(<4 x s32>)
+...
+---
+# trunc reads only the low 16 bits, which the AND forces to 0, so the trunc is
+# known 0 even though the wider AND result is not fully known.
+name:            trunc_of_and_low_zero
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: trunc_of_and_low_zero
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s16) = G_CONSTANT i16 0
+    ; CHECK-NEXT: $h0 = COPY [[C]](s16)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 -65536
+    %2:_(s32) = G_AND %0, %1
+    %3:_(s16) = G_TRUNC %2
+    $h0 = COPY %3(s16)
+...
+---
+# zext of a known-0 narrow value (lshr result) is known 0 in the wider type.
+name:            zext_of_known_zero
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: zext_of_known_zero
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 0
+    ; CHECK-NEXT: $x0 = COPY [[C]](s64)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 255
+    %2:_(s32) = G_AND %0, %1
+    %3:_(s32) = G_CONSTANT i32 8
+    %4:_(s32) = G_LSHR %2, %3
+    %5:_(s64) = G_ZEXT %4
+    $x0 = COPY %5(s64)
+...
+---
+# sext of a known-0 narrow value is known 0 (0 sign-extends to 0).
+name:            sext_of_known_zero
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: sext_of_known_zero
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 0
+    ; CHECK-NEXT: $x0 = COPY [[C]](s64)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 65535
+    %2:_(s32) = G_AND %0, %1
+    %3:_(s32) = G_CONSTANT i32 16
+    %4:_(s32) = G_LSHR %2, %3
+    %5:_(s64) = G_SEXT %4
+    $x0 = COPY %5(s64)
+...
+---
+# trunc to a non-zero constant: (or x,0x0F)<<4 forces the low byte to 0xF0
+# (bits 0-3 shifted-in 0, bits 4-7 the forced ones); trunc s8 -> 0xF0.
+name:            trunc_to_nonzero_const
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: trunc_to_nonzero_const
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s8) = G_CONSTANT i8 -16
+    ; CHECK-NEXT: $b0 = COPY [[C]](s8)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 15
+    %2:_(s32) = G_OR %0, %1
+    %3:_(s32) = G_CONSTANT i32 4
+    %4:_(s32) = G_SHL %2, %3
+    %5:_(s8) = G_TRUNC %4
+    $b0 = COPY %5(s8)
+...
+---
+# zext of a fully-known non-zero narrow value: the s8 (or t,0x0F)<<4 is 0xF0,
+# so zext to s32 is 0x000000F0 = 240.
+name:            zext_to_nonzero_const
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: zext_to_nonzero_const
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 240
+    ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+    %0:_(s32) = COPY $w0
+    %t:_(s8) = G_TRUNC %0
+    %1:_(s8) = G_CONSTANT i8 15
+    %2:_(s8) = G_OR %t, %1
+    %3:_(s8) = G_CONSTANT i8 4
+    %4:_(s8) = G_SHL %2, %3
+    %5:_(s32) = G_ZEXT %4
+    $w0 = COPY %5(s32)
+...
+---
+# sext of the same 0xF0 s8: the sign bit is set, so sext to s32 is
+# 0xFFFFFFF0 = -16 (contrast with the zext case above).
+name:            sext_to_nonzero_const
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: sext_to_nonzero_const
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 -16
+    ; CHECK-NEXT: $w0 = COPY [[C]](s32)
+    %0:_(s32) = COPY $w0
+    %t:_(s8) = G_TRUNC %0
+    %1:_(s8) = G_CONSTANT i8 15
+    %2:_(s8) = G_OR %t, %1
+    %3:_(s8) = G_CONSTANT i8 4
+    %4:_(s8) = G_SHL %2, %3
+    %5:_(s32) = G_SEXT %4
+    $w0 = COPY %5(s32)
+...
+---
+# Negative: low 8 bits unknown, so the AND is preserved.
+name:            partial_known_no_fold
+body:             |
+  bb.0:
+    liveins: $w0
+    ; CHECK-LABEL: name: partial_known_no_fold
+    ; CHECK: liveins: $w0
+    ; CHECK-NEXT: {{  $}}
+    ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(s32) = COPY $w0
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 255
+    ; CHECK-NEXT: [[AND:%[0-9]+]]:_(s32) = G_AND [[COPY]], [[C]]
+    ; CHECK-NEXT: $w0 = COPY [[AND]](s32)
+    %0:_(s32) = COPY $w0
+    %1:_(s32) = G_CONSTANT i32 255
+    %2:_(s32) = G_AND %0, %1
+    $w0 = COPY %2(s32)
+...
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
index 7bb166e0312c2..92d1a81098e1d 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-logic-of-compare.mir
@@ -386,9 +386,8 @@ body:             |
     ; CHECK-LABEL: name: test_fcmp_and_fcmp_with_vectors
     ; CHECK: liveins: $x0, $x1
     ; CHECK-NEXT: {{  $}}
-    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i1) = G_CONSTANT i1 false
-    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<2 x i1>) = G_BUILD_VECTOR [[C]](i1), [[C]](i1)
-    ; CHECK-NEXT: %zext:_(<2 x i64>) = G_ZEXT [[BUILD_VECTOR]](<2 x i1>)
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
+    ; CHECK-NEXT: %zext:_(<2 x i64>) = G_BUILD_VECTOR [[C]](i64), [[C]](i64)
     ; CHECK-NEXT: $q0 = COPY %zext(<2 x i64>)
     %0:_(f64) = COPY $x0
     %1:_(f64) = COPY $x1
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
index df2cdb15ef899..9b98cd45da0de 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-overflow.mir
@@ -157,9 +157,8 @@ body:             |
     ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
     ; CHECK-NEXT: [[COPY1:%[0-9]+]]:_(i32) = COPY $w1
     ; CHECK-NEXT: %bv0:_(<4 x i32>) = G_BUILD_VECTOR [[COPY]](i32), [[COPY1]](i32), [[COPY]](i32), [[COPY1]](i32)
-    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i1) = G_CONSTANT i1 false
-    ; CHECK-NEXT: %o:_(<4 x i1>) = G_BUILD_VECTOR [[C]](i1), [[C]](i1), [[C]](i1), [[C]](i1)
-    ; CHECK-NEXT: %o_wide:_(<4 x i32>) = G_ZEXT %o(<4 x i1>)
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
+    ; CHECK-NEXT: %o_wide:_(<4 x i32>) = G_BUILD_VECTOR [[C]](i32), [[C]](i32), [[C]](i32), [[C]](i32)
     ; CHECK-NEXT: $q0 = COPY %bv0(<4 x i32>)
     ; CHECK-NEXT: $q1 = COPY %o_wide(<4 x i32>)
     ; CHECK-NEXT: RET_ReallyLR implicit $w0
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
index f478aa80e7fa2..09723b7b9959e 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-trunc.mir
@@ -171,14 +171,20 @@ legalized: true
 body:             |
   bb.1:
   liveins: $w0
-    ; CHECK-LABEL: name: test_combine_trunc_shl_s32_by_17
-    ; CHECK: liveins: $w0
-    ; CHECK-NEXT: {{  $}}
-    ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
-    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 17
-    ; CHECK-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[COPY]], [[C]](i32)
-    ; CHECK-NEXT: [[TRUNC:%[0-9]+]]:_(i16) = G_TRUNC [[SHL]](i32)
-    ; CHECK-NEXT: $h0 = COPY [[TRUNC]](i16)
+    ; CHECK-PRE-LABEL: name: test_combine_trunc_shl_s32_by_17
+    ; CHECK-PRE: liveins: $w0
+    ; CHECK-PRE-NEXT: {{  $}}
+    ; CHECK-PRE-NEXT: [[C:%[0-9]+]]:_(i16) = G_CONSTANT i16 0
+    ; CHECK-PRE-NEXT: $h0 = COPY [[C]](i16)
+    ;
+    ; CHECK-POST-LABEL: name: test_combine_trunc_shl_s32_by_17
+    ; CHECK-POST: liveins: $w0
+    ; CHECK-POST-NEXT: {{  $}}
+    ; CHECK-POST-NEXT: [[COPY:%[0-9]+]]:_(i32) = COPY $w0
+    ; CHECK-POST-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 17
+    ; CHECK-POST-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[COPY]], [[C]](i32)
+    ; CHECK-POST-NEXT: [[TRUNC:%[0-9]+]]:_(i16) = G_TRUNC [[SHL]](i32)
+    ; CHECK-POST-NEXT: $h0 = COPY [[TRUNC]](i16)
     %0:_(i32) = COPY $w0
     %1:_(i32) = G_CONSTANT i32 17
     %2:_(i32) = G_SHL %0(i32), %1(i32)
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll b/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
index ba53cb57c2ef2..76197f959559e 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/legalize-shuffle-vector-widen-crash.ll
@@ -5,25 +5,28 @@ target triple = "arm64-apple-macosx11.0.0"
 
 declare i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32>) #0
 
-define i32 @bar() {
+; The vectors are function arguments (not constants) so the shuffle result is not
+; known, ensuring the <4 x i1> shuffle still reaches the widen legalization path
+; this test guards rather than being folded to a constant beforehand.
+define i32 @bar(<8 x i1> %a, <8 x i1> %b) {
 ; CHECK-LABEL: bar:
 ; CHECK:       ; %bb.0: ; %bb
-; CHECK-NEXT:    movi.2d v0, #0000000000000000
+; CHECK-NEXT:    ; kill: def $d0 killed $d0 def $q0
 ; CHECK-NEXT:    umov.b w8, v0[0]
 ; CHECK-NEXT:    umov.b w9, v0[1]
-; CHECK-NEXT:    fmov s1, w8
+; CHECK-NEXT:    movi.4s v1, #1
+; CHECK-NEXT:    fmov s2, w8
 ; CHECK-NEXT:    umov.b w8, v0[2]
-; CHECK-NEXT:    mov.s v1[1], w9
+; CHECK-NEXT:    mov.s v2[1], w9
 ; CHECK-NEXT:    umov.b w9, v0[3]
-; CHECK-NEXT:    movi.4s v0, #1
-; CHECK-NEXT:    mov.s v1[2], w8
-; CHECK-NEXT:    mov.s v1[3], w9
-; CHECK-NEXT:    and.16b v0, v1, v0
+; CHECK-NEXT:    mov.s v2[2], w8
+; CHECK-NEXT:    mov.s v2[3], w9
+; CHECK-NEXT:    and.16b v0, v2, v1
 ; CHECK-NEXT:    addv.4s s0, v0
 ; CHECK-NEXT:    fmov w0, s0
 ; CHECK-NEXT:    ret
 bb:
-  %shufflevector = shufflevector <8 x i1> zeroinitializer, <8 x i1> zeroinitializer, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
+  %shufflevector = shufflevector <8 x i1> %a, <8 x i1> %b, <4 x i32> <i32 0, i32 1, i32 2, i32 3>
   %zext = zext <4 x i1> %shufflevector to <4 x i32>
   %call = call i32 @llvm.aarch64.neon.uaddv.i32.v4i32(<4 x i32> %zext)
   %icmp = icmp eq i32 %call, 0
diff --git a/llvm/test/CodeGen/AArch64/arm64-vshift.ll b/llvm/test/CodeGen/AArch64/arm64-vshift.ll
index a7411457430c4..05db56ec848f1 100644
--- a/llvm/test/CodeGen/AArch64/arm64-vshift.ll
+++ b/llvm/test/CodeGen/AArch64/arm64-vshift.ll
@@ -4658,8 +4658,7 @@ define <2 x i8> @shl_trunc_v2i64_v2i8(<2 x i64> %a) {
 ;
 ; CHECK-GI-LABEL: shl_trunc_v2i64_v2i8:
 ; CHECK-GI:       // %bb.0:
-; CHECK-GI-NEXT:    shl v0.2d, v0.2d, #16
-; CHECK-GI-NEXT:    xtn v0.2s, v0.2d
+; CHECK-GI-NEXT:    movi v0.2d, #0000000000000000
 ; CHECK-GI-NEXT:    ret
   %b = shl <2 x i64> %a, <i64 16, i64 16>
   %c = trunc <2 x i64> %b to <2 x i8>
diff --git a/llvm/test/CodeGen/AArch64/hadd-combine.ll b/llvm/test/CodeGen/AArch64/hadd-combine.ll
index 450069cd27428..6dbe5b2c3ecf7 100644
--- a/llvm/test/CodeGen/AArch64/hadd-combine.ll
+++ b/llvm/test/CodeGen/AArch64/hadd-combine.ll
@@ -95,19 +95,10 @@ define <8 x i16> @haddu_const_both() {
 }
 
 define <8 x i16> @haddu_const_bothhigh() {
-; CHECK-SD-LABEL: haddu_const_bothhigh:
-; CHECK-SD:       // %bb.0:
-; CHECK-SD-NEXT:    mvni v0.8h, #1
-; CHECK-SD-NEXT:    ret
-;
-; CHECK-GI-LABEL: haddu_const_bothhigh:
-; CHECK-GI:       // %bb.0:
-; CHECK-GI-NEXT:    movi d0, #0xffffffffffffffff
-; CHECK-GI-NEXT:    mvni v1.4h, #1
-; CHECK-GI-NEXT:    uaddl v1.4s, v1.4h, v0.4h
-; CHECK-GI-NEXT:    shrn v0.4h, v1.4s, #1
-; CHECK-GI-NEXT:    shrn2 v0.8h, v1.4s, #1
-; CHECK-GI-NEXT:    ret
+; CHECK-LABEL: haddu_const_bothhigh:
+; CHECK:       // %bb.0:
+; CHECK-NEXT:    mvni v0.8h, #1
+; CHECK-NEXT:    ret
   %ext1 = zext <8 x i16> <i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534, i16 65534> to <8 x i32>
   %ext2 = zext <8 x i16> <i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535, i16 65535> to <8 x i32>
   %add = add <8 x i32> %ext1, %ext2
@@ -341,9 +332,7 @@ define <8 x i16> @hadds_const_bothhigh() {
 ; CHECK-GI-LABEL: hadds_const_bothhigh:
 ; CHECK-GI:       // %bb.0:
 ; CHECK-GI-NEXT:    adrp x8, .LCPI19_0
-; CHECK-GI-NEXT:    mvni v0.8h, #128, lsl #8
-; CHECK-GI-NEXT:    ldr q1, [x8, :lo12:.LCPI19_0]
-; CHECK-GI-NEXT:    shadd v0.8h, v1.8h, v0.8h
+; CHECK-GI-NEXT:    ldr q0, [x8, :lo12:.LCPI19_0]
 ; CHECK-GI-NEXT:    ret
   %ext1 = sext <8 x i16> <i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766, i16 32766> to <8 x i32>
   %ext2 = sext <8 x i16> <i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767, i16 32767> to <8 x i32>
@@ -615,21 +604,10 @@ define <8 x i16> @rhaddu_const_both() {
 }
 
 define <8 x i16> @rhaddu_const_bothhigh() {
-; CHECK-SD-LABEL: rhaddu_const_bothhigh:
-; CHECK-SD:       // %bb.0:
-; CHECK-SD-NEXT:    movi v0.2d, #0xffffffffffffffff
-; CHECK-SD-NEXT:    ret
-;
-; CHECK-GI-...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/224254


More information about the llvm-commits mailing list