[llvm] [GlobalISel] Migrate various generic wip_match_opcode combines to MIR-pattern. (PR #220213)

Vikash Gupta via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 8 04:03:03 PDT 2026


https://github.com/vg0204 updated https://github.com/llvm/llvm-project/pull/220213

>From 606e83e89044e7997cf39dc634d90007efa5d0bb Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Tue, 1 Sep 2026 15:31:29 +0530
Subject: [PATCH 1/3] [GlobalISel] Migrate various wip_match_opcode combines to
 MIR-pattern.
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

This patch converts a batch of GlobalISel combine rules from
hand-written C++ matchers to declarative MIR patterns preserving the
behavior. It also along with adds two functional changes (a rule-ordering
fix and an out-of-bounds bug fix) are described below.

- `select_same_val` → `select_same_val_trivial` +
  `select_same_val_equiv` group
- `select_constant_cmp` → `_false` / `_true` / `_general` group
- `simplify_add_to_sub`, `add_p2i_to_ptradd` → `GICombinePatFrag`s
- `commute_shift` → `commute_shift_frags` (C++ residue reduced to
  `isDesirableToCommuteWithShift`)
- `combine_i2p_to_p2i`, `ptr_add_zero`, `sext_trunc_sext_load` →
  patterns.
- funnel-shift / rotate / `ashr_lshr` / `constant_fold_fma` /
  `constant_fold_cast_op` / `combine_minmax_nan` → pattern fragments

- **`select_zero_true` / `select_zero_false`:** skip a constant
  condition as they fold it to an arm instead strictly better and
  avoids a `G_FREEZE` that post-legalization can't clean up.
- **`match_bitfield_extract_from_and`:** mask the AND immediate to the
  operand width and reject fields where `lsb + width > size`,
  preventing an out-of-bounds `G_UBFX` (fixes an AMDGPU
  `knownBitsForSBFE` crash).

28 CodeGen tests updated. All diffs are semantically equivalent.
---
 .../llvm/CodeGen/GlobalISel/CombinerHelper.h  |  10 +-
 .../include/llvm/Target/GlobalISel/Combine.td | 307 +++++++++++-------
 .../lib/CodeGen/GlobalISel/CombinerHelper.cpp |  84 +----
 .../GlobalISel/combine-insert-vec-elt.mir     |  14 +-
 .../form-bitfield-extract-from-and.mir        |   5 +-
 ...galizer-combiner-divrem-insertpt-crash.mir |   5 +-
 .../AArch64/neon-shuffle-vector-tbl.ll        |   2 +-
 llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll   | 152 ++++-----
 .../combine-extract-vector-load.mir           |  10 +-
 .../GlobalISel/combine-redundant-and.mir      |  10 +-
 .../combine-shift-of-shifted-logic.ll         |   4 +-
 llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll   |  18 +-
 llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll   | 148 ++++-----
 .../GlobalISel/llvm.amdgcn.intersect_ray.ll   |  46 +--
 .../llvm.amdgcn.make.buffer.rsrc.ll           |   4 +-
 llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll   | 156 ++++-----
 .../GlobalISel/merge-values-s16-true16.ll     |  11 +-
 .../AMDGPU/GlobalISel/mul-known-bits.i64.ll   |  21 +-
 llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll    |  80 ++---
 .../CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll  |   4 +-
 .../AMDGPU/llvm.amdgcn.intersect_ray.ll       | 219 +++++--------
 .../lower-work-group-id-intrinsics-hsa.ll     |  71 ++--
 .../lower-work-group-id-intrinsics-opt.ll     |   8 +-
 .../lower-work-group-id-intrinsics-pal.ll     |  72 ++--
 .../AMDGPU/lower-work-group-id-intrinsics.ll  |  18 +-
 llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll |  56 ++--
 llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll |  36 +-
 llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll |  76 ++---
 .../test/CodeGen/AMDGPU/vector-reduce-umax.ll | 102 +++---
 .../test/CodeGen/AMDGPU/vector-reduce-umin.ll | 114 +++----
 llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll |  36 +-
 .../AMDGPU/workgroup-id-in-arch-sgprs.ll      |  65 ++--
 .../CodeGen/AMDGPU/workitem-intrinsic-opts.ll | 125 ++++---
 33 files changed, 1073 insertions(+), 1016 deletions(-)

diff --git a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
index a8bf13b5760c1..6b80367d786a5 100644
--- a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
+++ b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
@@ -245,7 +245,6 @@ class CombinerHelper {
                                IndexedLoadStoreMatchInfo &MatchInfo) const;
 
   LLVM_ABI bool matchSextTruncSextLoad(MachineInstr &MI) const;
-  LLVM_ABI void applySextTruncSextLoad(MachineInstr &MI) const;
 
   /// Match sext_inreg(load p), imm -> sextload p
   LLVM_ABI bool
@@ -379,7 +378,9 @@ class CombinerHelper {
   LLVM_ABI void applyShiftOfShiftedLogic(MachineInstr &MI,
                                          ShiftOfShiftedLogic &MatchInfo) const;
 
-  LLVM_ABI bool matchCommuteShift(MachineInstr &MI, BuildFnTy &MatchInfo) const;
+  /// \return true if the target's TargetLowering::isDesirableToCommuteWithShift
+  /// hook approves of commuting \p MI (a G_SHL) with the binop feeding it.
+  LLVM_ABI bool isDesirableToCommuteWithShift(const MachineInstr &MI) const;
 
   /// Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (shift x, (C1 + C2))
   LLVM_ABI bool matchLshrOfTruncOfLshr(MachineInstr &MI,
@@ -454,10 +455,6 @@ class CombinerHelper {
   LLVM_ABI bool matchConstantFoldUnaryIntOp(MachineInstr &MI,
                                             BuildFnTy &MatchInfo) const;
 
-  /// Transform IntToPtr(PtrToInt(x)) to x if cast is in the same address space.
-  LLVM_ABI bool matchCombineI2PToP2I(MachineInstr &MI, Register &Reg) const;
-  LLVM_ABI void applyCombineI2PToP2I(MachineInstr &MI, Register &Reg) const;
-
   /// Transform PtrToInt(IntToPtr(x)) to x.
   LLVM_ABI void applyCombineP2IToI2P(MachineInstr &MI, Register &Reg) const;
 
@@ -649,7 +646,6 @@ class CombinerHelper {
 
   /// Combine G_PTR_ADD with nullptr to G_INTTOPTR
   LLVM_ABI bool matchPtrAddZero(MachineInstr &MI) const;
-  LLVM_ABI void applyPtrAddZero(MachineInstr &MI) const;
 
   /// Combine G_UREM x, (known power of 2) to an add and bitmasking.
   LLVM_ABI void applySimplifyURemByPow2(MachineInstr &MI) const;
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 4b0a438cd6455..8942fd3a2e155 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -251,21 +251,21 @@ def extending_loads : GICombineRule<
 
 def load_and_mask : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_AND):$root,
+  (match (G_AND $dst, $src1, $src2):$root,
         [{ return Helper.matchCombineLoadWithAndMask(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 def combines_for_extload: GICombineGroup<[extending_loads, load_and_mask]>;
 
 def sext_trunc_sextload : GICombineRule<
   (defs root:$d),
-  (match (wip_match_opcode G_SEXT_INREG):$d,
+  (match (G_SEXT_INREG $dst, $src, $sz):$d,
          [{ return Helper.matchSextTruncSextLoad(*${d}); }]),
-  (apply [{ Helper.applySextTruncSextLoad(*${d}); }])>;
+  (apply (GIReplaceReg $dst, $src))>;
 
 def sext_inreg_of_load_matchdata : GIDefMatchData<"std::tuple<Register, unsigned>">;
 def sext_inreg_of_load : GICombineRule<
   (defs root:$root, sext_inreg_of_load_matchdata:$matchinfo),
-  (match (wip_match_opcode G_SEXT_INREG):$root,
+  (match (G_SEXT_INREG $dst, $src, $sz):$root,
          [{ return Helper.matchSextInRegOfLoad(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applySextInRegOfLoad(*${root}, ${matchinfo}); }])>;
 
@@ -286,7 +286,7 @@ def sext_inreg_to_zext_inreg : GICombineRule<
 
 def combine_extracted_vector_load : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+  (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
         [{ return Helper.matchCombineExtractedVectorLoad(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
@@ -339,14 +339,14 @@ def memcpy_family_combines : GICombineGroup<[combine_memcpy_inline,
 def opt_brcond_by_inverting_cond_matchdata : GIDefMatchData<"MachineInstr *">;
 def opt_brcond_by_inverting_cond : GICombineRule<
   (defs root:$root, opt_brcond_by_inverting_cond_matchdata:$matchinfo),
-  (match (wip_match_opcode G_BR):$root,
+  (match (G_BR $tgt):$root,
          [{ return Helper.matchOptBrCondByInvertingCond(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyOptBrCondByInvertingCond(*${root}, ${matchinfo}); }])>;
 
 def ptr_add_immed_matchdata : GIDefMatchData<"PtrAddChain">;
 def ptr_add_immed_chain : GICombineRule<
   (defs root:$d, ptr_add_immed_matchdata:$matchinfo),
-  (match (wip_match_opcode G_PTR_ADD):$d,
+  (match (G_PTR_ADD $dst, $base, $offset):$d,
          [{ return Helper.matchPtrAddImmedChain(*${d}, ${matchinfo}); }]),
   (apply [{ Helper.applyPtrAddImmedChain(*${d}, ${matchinfo}); }])>;
 
@@ -447,12 +447,26 @@ def bitreverse_lshr : GICombineRule<
   (apply (G_SHL $d, $val, $amt))>;
 
 // Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
-// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
+// Combine (shl (or  x, c1), c2) -> (or  (shl x, c2), c1 << c2)
+// Apply rebuilds the matched opcode, so it stays inline C++.
+def commute_shift_frags : GICombinePatFrag<
+  (outs root:$dst, $binop), (ins $x, $c1, $c2),
+  !foreach(op, [G_ADD, G_OR],
+    (pattern (op $binop, $x, $c1), (G_SHL $dst, $binop, $c2)))>;
 def commute_shift : GICombineRule<
-  (defs root:$d, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SHL):$d,
-         [{ return Helper.matchCommuteShift(*${d}, ${matchinfo}); }]),
-  (apply [{ Helper.applyBuildFn(*${d}, ${matchinfo}); }])>;
+  (defs root:$dst),
+  (match (commute_shift_frags $dst, $binop, $x, $c1, $c2):$shl,
+         [{ return MRI.hasOneNonDBGUse(${binop}.getReg()) &&
+                   isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
+                   isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
+                   Helper.isDesirableToCommuteWithShift(*${shl}); }]),
+  (apply [{ auto &B = Helper.getBuilder();
+            LLT Ty = MRI.getType(${dst}.getReg());
+            unsigned Opc = MRI.getVRegDef(${binop}.getReg())->getOpcode();
+            auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
+            auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
+            auto New = B.buildInstr(Opc, {Ty}, {S1, S2});
+            Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
 
 // Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (lshr x, (C1 + C2))
 def lshr_of_trunc_of_lshr_matchdata : GIDefMatchData<"LshrOfTruncOfLshr">;
@@ -466,7 +480,7 @@ def lshr_of_trunc_of_lshr : GICombineRule<
 
 def narrow_binop_feeding_and : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_AND):$root,
+  (match (G_AND $dst, $src1, $src2):$root,
          [{ return Helper.matchNarrowBinopFeedingAnd(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFnNoErase(*${root}, ${matchinfo}); }])>;
 
@@ -562,9 +576,9 @@ def propagate_undef_all_ops: GICombineRule<
 // Replace a G_SHUFFLE_VECTOR with an undef mask with a G_IMPLICIT_DEF.
 def propagate_undef_shuffle_mask: GICombineRule<
   (defs root:$root),
-  (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+  (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
          [{ return Helper.matchUndefShuffleVectorMask(*${root}); }]),
-  (apply [{ Helper.replaceInstWithUndef(*${root}); }])>;
+  (apply (G_IMPLICIT_DEF $dst))>;
 
   // Replace an insert/extract element of an out of bounds index with undef.
   def insert_extract_vec_elt_out_of_bounds : GICombineRule<
@@ -574,12 +588,20 @@ def propagate_undef_shuffle_mask: GICombineRule<
   (apply [{ Helper.replaceInstWithUndef(*${root}); }])>;
 
 // Fold (cond ? x : x) -> x
-def select_same_val: GICombineRule<
+// _trivial: arms are the same register; _equiv: arms are provably equivalent.
+def select_same_val_trivial : GICombineRule<
+  (defs root:$dst),
+  (match (G_SELECT $dst, $cond, $x, $x)),
+  (apply (GIReplaceReg $dst, $x))
+>;
+def select_same_val_equiv: GICombineRule<
   (defs root:$root),
-  (match (wip_match_opcode G_SELECT):$root,
+  (match (G_SELECT $dst, $cond, $tval, $fval):$root,
     [{ return Helper.matchSelectSameVal(*${root}); }]),
-  (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, 2); }])
+  (apply (GIReplaceReg $dst, $tval))
 >;
+def select_same_val : GICombineGroup<[select_same_val_trivial,
+                                      select_same_val_equiv]>;
 
 // Fold (undef ? x : y) -> y
 def select_undef_cmp: GICombineRule<
@@ -589,20 +611,36 @@ def select_undef_cmp: GICombineRule<
   (apply (GIReplaceReg $dst, $y))
 >;
 
-// Fold (true ? x : y) -> x
-// Fold (false ? x : y) -> y
-def select_constant_cmp: GICombineRule<
+def select_constant_cmp_false : GICombineRule<
+  (defs root:$dst),
+  (match (G_CONSTANT $cond, 0),
+         (G_SELECT $dst, $cond, $tval, $fval)),
+  (apply (GIReplaceReg $dst, $fval))
+>;
+def select_constant_cmp_true : GICombineRule<
+  (defs root:$dst),
+  (match (G_CONSTANT $cond, 1),
+         (G_SELECT $dst, $cond, $tval, $fval)),
+  (apply (GIReplaceReg $dst, $tval))
+>;
+
+def select_constant_cmp_general: GICombineRule<
   (defs root:$root, unsigned_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SELECT):$root,
+  (match (G_SELECT $dst, $cond, $tval, $fval):$root,
     [{ return Helper.matchConstantSelectCmp(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, ${matchinfo}); }])
 >;
+def select_constant_cmp : GICombineGroup<[select_constant_cmp_false,
+                                          select_constant_cmp_true,
+                                          select_constant_cmp_general]>;
 
 // select c, 0, x -> and (not c), x
+// Skip constant c: select_constant_cmp folds it directly, avoiding the freeze.
 def select_zero_true: GICombineRule<
   (defs root:$root),
   (match (G_SELECT $dst, $c, 0, $x):$root,
-    [{ return MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
+    [{ return !isConstantOrConstantSplatVector(${c}.getReg(), MRI) &&
+              MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
               VT->computeNumSignBits(${c}.getReg()) == MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
   (apply (G_XOR $xor, $c, -1),
          (G_FREEZE $f, $x),
@@ -610,10 +648,12 @@ def select_zero_true: GICombineRule<
 >;
 
 // select c, x, 0 -> and c, x
+// Skip constant c: select_constant_cmp folds it directly, avoiding the freeze.
 def select_zero_false: GICombineRule<
   (defs root:$root),
   (match (G_SELECT $dst, $c, $x, 0):$root,
-    [{ return MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
+    [{ return !isConstantOrConstantSplatVector(${c}.getReg(), MRI) &&
+              MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
               VT->computeNumSignBits(${c}.getReg()) == MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
   (apply (G_FREEZE $f, $x),
          (G_AND $dst, $c, $f))
@@ -813,12 +853,18 @@ def erase_undef_store : GICombineRule<
   (apply [{ Helper.eraseInst(*${root}); }])
 >;
 
-def simplify_add_to_sub_matchinfo: GIDefMatchData<"std::tuple<Register, Register>">;
-def simplify_add_to_sub: GICombineRule <
-  (defs root:$root, simplify_add_to_sub_matchinfo:$info),
-  (match (wip_match_opcode G_ADD):$root,
-    [{ return Helper.matchSimplifyAddToSub(*${root}, ${info}); }]),
-  (apply [{ Helper.applySimplifyAddToSub(*${root}, ${info});}])
+// Fold ((0-A) + B) -> B - A, (A + (0-B)) -> A - B (negation is G_SUB 0, x).
+// !foreach covers both G_ADD operand orders.
+def simplify_add_to_sub_frags : GICombinePatFrag<
+  (outs root:$dst), (ins $newlhs, $newrhs),
+  !foreach(inst, [(G_ADD $dst, $neg, $newlhs), (G_ADD $dst, $newlhs, $neg)],
+    (pattern (G_CONSTANT $zero, 0),
+             (G_SUB $neg, $zero, $newrhs),
+             inst))>;
+def simplify_add_to_sub : GICombineRule <
+  (defs root:$dst),
+  (match (simplify_add_to_sub_frags $dst, $newlhs, $newrhs)),
+  (apply (G_SUB $dst, $newlhs, $newrhs))
 >;
 
 // Fold fp_op(cst) to the constant result of the floating point operation.
@@ -873,10 +919,11 @@ def constant_fold_fp_ops : GICombineGroup<[
 
 // Fold int2ptr(ptr2int(x)) -> x
 def p2i_to_i2p: GICombineRule<
-  (defs root:$root, register_matchinfo:$info),
-  (match (wip_match_opcode G_INTTOPTR):$root,
-    [{ return Helper.matchCombineI2PToP2I(*${root}, ${info}); }]),
-  (apply [{ Helper.applyCombineI2PToP2I(*${root}, ${info}); }])
+  (defs root:$root),
+  (match (G_PTRTOINT $mid, $x),
+         (G_INTTOPTR $dst, $mid):$root,
+    [{ return MRI.getType(${x}.getReg()) == MRI.getType(${dst}.getReg()); }]),
+  (apply (GIReplaceReg $dst, $x))
 >;
 
 // Fold ptr2int(int2ptr(x)) -> x
@@ -889,20 +936,37 @@ def i2p_to_p2i: GICombineRule<
 >;
 
 // Fold add ptrtoint(x), y -> ptrtoint (ptr_add x), y
-def add_p2i_to_ptradd_matchinfo : GIDefMatchData<"std::pair<Register, bool>">;
+// !foreach covers both G_ADD operand orders; residue checks equal bitwidths.
+def add_p2i_to_ptradd_frags : GICombinePatFrag<
+  (outs root:$dst), (ins $ptr, $other),
+  !foreach(inst, [(G_ADD $dst, $x, $other), (G_ADD $dst, $other, $x)],
+    (pattern (G_PTRTOINT $x, $ptr), inst))>;
 def add_p2i_to_ptradd : GICombineRule<
-  (defs root:$root, add_p2i_to_ptradd_matchinfo:$info),
-  (match (wip_match_opcode G_ADD):$root,
-    [{ return Helper.matchCombineAddP2IToPtrAdd(*${root}, ${info}); }]),
-  (apply [{ Helper.applyCombineAddP2IToPtrAdd(*${root}, ${info}); }])
+  (defs root:$dst),
+  (match (add_p2i_to_ptradd_frags $dst, $ptr, $other),
+    [{ return MRI.getType(${ptr}.getReg()).getScalarSizeInBits() ==
+              MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
+  (apply (G_PTR_ADD $padd, $ptr, $other),
+         (G_PTRTOINT $dst, $padd))
 >;
 
-// Fold (ptr_add (int2ptr C1), C2) -> C1 + C2
+// Fold (ptr_add (int2ptr C1), C2) -> C1 + C2 (zext(C1)+sext(C2) folded in C++).
 def const_ptradd_to_i2p: GICombineRule<
   (defs root:$root, apint_matchinfo:$info),
-  (match (wip_match_opcode G_PTR_ADD):$root,
-    [{ return Helper.matchCombineConstPtrAddToI2P(*${root}, ${info}); }]),
-  (apply [{ Helper.applyCombineConstPtrAddToI2P(*${root}, ${info}); }])
+  (match (G_CONSTANT $c1reg, $c1imm),
+         (G_INTTOPTR $base, $c1reg),
+         (G_CONSTANT $c2reg, $c2imm),
+         (G_PTR_ADD $dst, $base, $c2reg):$root,
+    [{ LLT DstTy = MRI.getType(${dst}.getReg());
+       APInt NewCst = ${c1imm}.getCImm()->getValue().zextOrTrunc(
+           DstTy.getSizeInBits());
+       NewCst += ${c2imm}.getCImm()->getValue().sextOrTrunc(
+           DstTy.getSizeInBits());
+       ${info} = NewCst;
+       return true; }]),
+  (apply [{ Helper.getBuilder().setInstrAndDebugLoc(*${root});
+            Helper.getBuilder().buildConstant(${dst}, ${info});
+            ${root}->eraseFromParent(); }])
 >;
 
 // Simplify: (logic_op (op x...), (op y...)) -> (op (logic_op x, y))
@@ -917,7 +981,7 @@ def hoist_logic_op_with_same_opcode_hands: GICombineRule <
 def shl_ashr_to_sext_inreg_matchinfo : GIDefMatchData<"std::tuple<Register, int64_t>">;
 def shl_ashr_to_sext_inreg : GICombineRule<
   (defs root:$root, shl_ashr_to_sext_inreg_matchinfo:$info),
-  (match (wip_match_opcode G_ASHR): $root,
+  (match (G_ASHR $dst, $src1, $src2): $root,
     [{ return Helper.matchAshrShlToSextInreg(*${root}, ${info}); }]),
   (apply [{ Helper.applyAshShlToSextInreg(*${root}, ${info});}])
 >;
@@ -939,7 +1003,7 @@ def neg_and_one_to_sext_inreg : GICombineRule<
 >;
 
 // Fold and(and(x, C1), C2) -> C1&C2 ? and(x, C1&C2) : 0
-def overlapping_and: GICombineRule <
+def overlapping_and : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
   (match (wip_match_opcode G_AND):$root,
          [{ return Helper.matchOverlappingAnd(*${root}, ${info}); }]),
@@ -949,7 +1013,7 @@ def overlapping_and: GICombineRule <
 // Fold (x & y) -> x or (x & y) -> y when (x & y) is known to equal x or equal y.
 def redundant_and: GICombineRule <
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_AND):$root,
+  (match (G_AND $dst, $src1, $src2):$root,
          [{ return Helper.matchRedundantAnd(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
 >;
@@ -957,7 +1021,7 @@ def redundant_and: GICombineRule <
 // Fold (x | y) -> x or (x | y) -> y when (x | y) is known to equal x or equal y.
 def redundant_or: GICombineRule <
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_OR):$root,
+  (match (G_OR $dst, $src1, $src2):$root,
          [{ return Helper.matchRedundantOr(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
 >;
@@ -967,9 +1031,9 @@ def redundant_or: GICombineRule <
 //   if computeNumSignBits(x) >= (x.getScalarSizeInBits() - K + 1)
 def redundant_sext_inreg: GICombineRule <
   (defs root:$root),
-  (match (wip_match_opcode G_SEXT_INREG):$root,
+  (match (G_SEXT_INREG $dst, $src, $sz):$root,
          [{ return Helper.matchRedundantSExtInReg(*${root}); }]),
-     (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, 1); }])
+     (apply (GIReplaceReg $dst, $src))
 >;
 
 // Convert G_SEXT_INREG(G_ZEXT) -> G_SEXT
@@ -1012,7 +1076,7 @@ def redundant_aext_unmerge_sext_inreg : redundant_ext_unmerge_sext_inreg<G_ANYEX
 // the destination type.
 def anyext_trunc_fold: GICombineRule <
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_ANYEXT):$root,
+  (match (G_ANYEXT $dst, $src):$root,
          [{ return Helper.matchCombineAnyExtTrunc(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
 >;
@@ -1021,14 +1085,14 @@ def anyext_trunc_fold: GICombineRule <
 // and truncated bits are known to be zero.
 def zext_trunc_fold: GICombineRule <
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_ZEXT):$root,
+  (match (G_ZEXT $dst, $src):$root,
          [{ return Helper.matchCombineZextTrunc(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
 >;
 
 def not_cmp_fold : GICombineRule<
   (defs root:$d, register_vector_matchinfo:$info),
-  (match (wip_match_opcode G_XOR): $d,
+  (match (G_XOR $dst, $src1, $src2): $d,
   [{ return Helper.matchNotCmp(*${d}, ${info}); }]),
   (apply [{ Helper.applyNotCmp(*${d}, ${info}); }])
 >;
@@ -1203,7 +1267,7 @@ def merge_combines: GICombineGroup<[
 def trunc_shift_matchinfo : GIDefMatchData<"std::pair<MachineInstr*, LLT>">;
 def trunc_shift: GICombineRule <
   (defs root:$root, trunc_shift_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_TRUNC):$root,
+  (match (G_TRUNC $dst, $src):$root,
          [{ return Helper.matchCombineTruncOfShift(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyCombineTruncOfShift(*${root}, ${matchinfo}); }])
 >;
@@ -1220,7 +1284,7 @@ def xor_of_and_with_same_reg_matchinfo :
     GIDefMatchData<"std::pair<Register, Register>">;
 def xor_of_and_with_same_reg: GICombineRule <
   (defs root:$root, xor_of_and_with_same_reg_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_XOR):$root,
+  (match (G_XOR $dst, $src1, $src2):$root,
          [{ return Helper.matchXorOfAndWithSameReg(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyXorOfAndWithSameReg(*${root}, ${matchinfo}); }])
 >;
@@ -1228,26 +1292,26 @@ def xor_of_and_with_same_reg: GICombineRule <
 // Transform (ptr_add 0, x) -> (int_to_ptr x)
 def ptr_add_with_zero: GICombineRule<
   (defs root:$root),
-  (match (wip_match_opcode G_PTR_ADD):$root,
+  (match (G_PTR_ADD $dst, $base, $offset):$root,
          [{ return Helper.matchPtrAddZero(*${root}); }]),
-  (apply [{ Helper.applyPtrAddZero(*${root}); }])>;
+  (apply (G_INTTOPTR $dst, $offset))>;
 
 def combine_insert_vec_elts_build_vector : GICombineRule<
   (defs root:$root, register_vector_matchinfo:$info),
-  (match (wip_match_opcode G_INSERT_VECTOR_ELT):$root,
+  (match (G_INSERT_VECTOR_ELT $dst, $src, $elt, $idx):$root,
     [{ return Helper.matchCombineInsertVecElts(*${root}, ${info}); }]),
   (apply [{ Helper.applyCombineInsertVecElts(*${root}, ${info}); }])>;
 
 def load_or_combine : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_OR):$root,
+  (match (G_OR $dst, $src1, $src2):$root,
     [{ return Helper.matchLoadOrCombine(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
 def extend_through_phis_matchdata: GIDefMatchData<"MachineInstr*">;
 def extend_through_phis : GICombineRule<
   (defs root:$root, extend_through_phis_matchdata:$matchinfo),
-  (match (wip_match_opcode G_PHI):$root,
+  (match (G_PHI $dst, GIVariadic<>:$srcs):$root,
     [{ return Helper.matchExtendThroughPhis(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyExtendThroughPhis(*${root}, ${matchinfo}); }])>;
 
@@ -1257,7 +1321,7 @@ def insert_vec_elt_combines : GICombineGroup<
 
 def extract_vec_elt_build_vec : GICombineRule<
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+  (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
     [{ return Helper.matchExtractVecEltBuildVec(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyExtractVecEltBuildVec(*${root}, ${matchinfo}); }])>;
 
@@ -1266,7 +1330,7 @@ def extract_all_elts_from_build_vector_matchinfo :
   GIDefMatchData<"SmallVector<std::pair<Register, MachineInstr*>>">;
 def extract_all_elts_from_build_vector : GICombineRule<
   (defs root:$root, extract_all_elts_from_build_vector_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_BUILD_VECTOR):$root,
+  (match (G_BUILD_VECTOR $dst, GIVariadic<>:$srcs):$root,
     [{ return Helper.matchExtractAllEltsFromBuildVector(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyExtractAllEltsFromBuildVector(*${root}, ${matchinfo}); }])>;
 
@@ -1276,22 +1340,26 @@ def extract_vec_elt_combines : GICombineGroup<[
 
 def funnel_shift_from_or_shift : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_OR):$root,
+  (match (G_OR $dst, $src1, $src2):$root,
     [{ return Helper.matchOrShiftToFunnelShift(*${root}, false, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
 >;
 
 def funnel_shift_from_or_shift_constants_are_legal : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_OR):$root,
+  (match (G_OR $dst, $src1, $src2):$root,
     [{ return Helper.matchOrShiftToFunnelShift(*${root}, true, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
 >;
 
 
+def funnel_shift_op_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_FSHL, G_FSHR], (pattern (op $dst, $x, $y, $amt)))>;
+
 def funnel_shift_to_rotate : GICombineRule<
-  (defs root:$root),
-  (match (wip_match_opcode G_FSHL, G_FSHR):$root,
+  (defs root:$dst),
+  (match (funnel_shift_op_frags $dst):$root,
     [{ return Helper.matchFunnelShiftToRotate(*${root}); }]),
   (apply [{ Helper.applyFunnelShiftToRotate(*${root}); }])
 >;
@@ -1312,8 +1380,8 @@ def funnel_shift_left_zero: GICombineRule<
 
 // Fold fsh(l/r) x, y, C -> fsh(l/r) x, y, C % bw
 def funnel_shift_overshift: GICombineRule<
-  (defs root:$root),
-  (match (wip_match_opcode G_FSHL, G_FSHR):$root,
+  (defs root:$dst),
+  (match (funnel_shift_op_frags $dst):$root,
     [{ return Helper.matchConstantLargerBitWidth(*${root}, 3); }]),
   (apply [{ Helper.applyFunnelShiftConstantModulo(*${root}); }])
 >;
@@ -1346,28 +1414,32 @@ def funnel_shift_or_shift_to_funnel_shift_right: GICombineRule<
   (apply (GIReplaceReg $root, $out1))
 >;
 
+def rotate_op_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_ROTR, G_ROTL], (pattern (op $dst, $x, $amt)))>;
+
 def rotate_out_of_range : GICombineRule<
-  (defs root:$root),
-  (match (wip_match_opcode G_ROTR, G_ROTL):$root,
+  (defs root:$dst),
+  (match (rotate_op_frags $dst):$root,
     [{ return Helper.matchRotateOutOfRange(*${root}); }]),
   (apply [{ Helper.applyRotateOutOfRange(*${root}); }])
 >;
 
 def icmp_to_true_false_known_bits : GICombineRule<
   (defs root:$d, int64_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_ICMP):$d,
+  (match (G_ICMP $dst, $pred, $src1, $src2):$d,
          [{ return Helper.matchICmpToTrueFalseKnownBits(*${d}, ${matchinfo}); }]),
   (apply [{ Helper.replaceInstWithConstant(*${d}, ${matchinfo}); }])>;
 
 def icmp_to_lhs_known_bits : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_ICMP):$root,
+  (match (G_ICMP $dst, $pred, $src1, $src2):$root,
          [{ return Helper.matchICmpToLHSKnownBits(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
 def redundant_binop_in_equality : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_ICMP):$root,
+  (match (G_ICMP $dst, $pred, $src1, $src2):$root,
          [{ return Helper.matchRedundantBinOpInEquality(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
@@ -1401,7 +1473,7 @@ def double_icmp_zero_or_combine: GICombineRule<
 
 def and_or_disjoint_mask : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_AND):$root,
+  (match (G_AND $dst, $src1, $src2):$root,
          [{ return Helper.matchAndOrDisjointMask(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFnNoErase(*${root}, ${info}); }])>;
 
@@ -1481,19 +1553,23 @@ def funnel_shift_combines : GICombineGroup<[funnel_shift_from_or_shift,
 
 def bitfield_extract_from_sext_inreg : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_SEXT_INREG):$root,
+  (match (G_SEXT_INREG $dst, $src, $sz):$root,
     [{ return Helper.matchBitfieldExtractFromSExtInReg(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
+def ashr_lshr_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_ASHR, G_LSHR], (pattern (op $dst, $src, $amt)))>;
+
 def bitfield_extract_from_shr : GICombineRule<
-  (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_ASHR, G_LSHR):$root,
+  (defs root:$dst, build_fn_matchinfo:$info),
+  (match (ashr_lshr_frags $dst):$root,
     [{ return Helper.matchBitfieldExtractFromShr(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
 def bitfield_extract_from_shr_and : GICombineRule<
-  (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_ASHR, G_LSHR):$root,
+  (defs root:$dst, build_fn_matchinfo:$info),
+  (match (ashr_lshr_frags $dst):$root,
     [{ return Helper.matchBitfieldExtractFromShrAnd(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
@@ -1553,7 +1629,7 @@ def intrem_combines : GICombineGroup<[srem_pow2_to_mask, urem_by_const,
 
 def reassoc_ptradd : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_PTR_ADD):$root,
+  (match (G_PTR_ADD $dst, $base, $offset):$root,
     [{ return Helper.matchReassocPtrAdd(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFnNoErase(*${root}, ${matchinfo}); }])>;
 
@@ -1589,15 +1665,23 @@ def constant_fold_fp_binop : GICombineRule<
   (apply [{ Helper.replaceInstWithFConstant(*${mi}, ${matchinfo}); }])>;
 
 
+def constant_fold_fma_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_FMAD, G_FMA], (pattern (op $dst, $src0, $src1, $src2)))>;
+
 def constant_fold_fma : GICombineRule<
-  (defs root:$d, constantfp_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_FMAD, G_FMA):$d,
+  (defs root:$dst, constantfp_matchinfo:$matchinfo),
+  (match (constant_fold_fma_frags $dst):$d,
    [{ return Helper.matchConstantFoldFMA(*${d}, ${matchinfo}); }]),
   (apply [{ Helper.replaceInstWithFConstant(*${d}, ${matchinfo}); }])>;
 
+def constant_fold_cast_op_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_ZEXT, G_SEXT, G_ANYEXT], (pattern (op $dst, $src)))>;
+
 def constant_fold_cast_op : GICombineRule<
-  (defs root:$d, apint_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_ZEXT, G_SEXT, G_ANYEXT):$d,
+  (defs root:$dst, apint_matchinfo:$matchinfo),
+  (match (constant_fold_cast_op_frags $dst):$d,
    [{ return Helper.matchConstantFoldCastOp(*${d}, ${matchinfo}); }]),
   (apply [{ Helper.replaceInstWithConstant(*${d}, ${matchinfo}); }])>;
 
@@ -1638,7 +1722,7 @@ def adde_to_addo: GICombineRule<
 
 def mulh_to_lshr : GICombineRule<
   (defs root:$root),
-  (match (wip_match_opcode G_UMULH):$root,
+  (match (G_UMULH $dst, $src1, $src2):$root,
          [{ return Helper.matchUMulHToLShr(*${root}); }]),
   (apply [{ Helper.applyUMulHToLShr(*${root}); }])>;
 
@@ -1709,7 +1793,7 @@ def redundant_neg_operands : GICombineGroup<
 // Transform (fsub +-0.0, X) -> (fneg X)
 def fsub_to_fneg: GICombineRule<
   (defs root:$root, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_FSUB):$root,
+  (match (G_FSUB $dst, $src1, $src2):$root,
     [{ return Helper.matchFsubToFneg(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyFsubToFneg(*${root}, ${matchinfo}); }])>;
 
@@ -1719,7 +1803,7 @@ def fsub_to_fneg: GICombineRule<
 //           (fadd (fmul x, y), z) -> (fmad x, y, z)
 def combine_fadd_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FADD):$root,
+  (match (G_FADD $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFAddFMulToFMadOrFMA(*${root},
                                                           ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1730,7 +1814,7 @@ def combine_fadd_fmul_to_fmad_or_fma: GICombineRule<
 //                                         -> (fmad (fpext y), (fpext z), x)
 def combine_fadd_fpext_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FADD):$root,
+  (match (G_FADD $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFAddFpExtFMulToFMadOrFMA(*${root},
                                                                ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1741,7 +1825,7 @@ def combine_fadd_fpext_fmul_to_fmad_or_fma: GICombineRule<
 //           (fadd v, (fmad x, y, (fmul z, u))) -> (fmad x, y, (fmad z, u, v))
 def combine_fadd_fma_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FADD):$root,
+  (match (G_FADD $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFAddFMAFMulToFMadOrFMA(*${root},
                                                              ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1750,7 +1834,7 @@ def combine_fadd_fma_fmul_to_fmad_or_fma: GICombineRule<
 //           (fma x, y, (fma (fpext u), (fpext v), z))
 def combine_fadd_fpext_fma_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FADD):$root,
+  (match (G_FADD $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFAddFpExtFMulToFMadOrFMAAggressive(
                                                   *${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1759,7 +1843,7 @@ def combine_fadd_fpext_fma_fmul_to_fmad_or_fma: GICombineRule<
 //                                 -> (fmad x, y, -z)
 def combine_fsub_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FSUB):$root,
+  (match (G_FSUB $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFSubFMulToFMadOrFMA(*${root},
                                                           ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1768,7 +1852,7 @@ def combine_fsub_fmul_to_fmad_or_fma: GICombineRule<
 //           (fsub x, (fneg (fmul, y, z))) -> (fma y, z, x)
 def combine_fsub_fneg_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FSUB):$root,
+  (match (G_FSUB $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFSubFNegFMulToFMadOrFMA(*${root},
                                                               ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1777,7 +1861,7 @@ def combine_fsub_fneg_fmul_to_fmad_or_fma: GICombineRule<
 //           (fma (fpext x), (fpext y), (fneg z))
 def combine_fsub_fpext_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FSUB):$root,
+  (match (G_FSUB $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFSubFpExtFMulToFMadOrFMA(*${root},
                                                                ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1786,14 +1870,19 @@ def combine_fsub_fpext_fmul_to_fmad_or_fma: GICombineRule<
 //           (fneg (fma (fpext x), (fpext y), z))
 def combine_fsub_fpext_fneg_fmul_to_fmad_or_fma: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_FSUB):$root,
+  (match (G_FSUB $dst, $src1, $src2):$root,
          [{ return Helper.matchCombineFSubFpExtFNegFMulToFMadOrFMA(
                                             *${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
+def combine_minmax_nan_frags : GICombinePatFrag<
+  (outs root:$dst), (ins),
+  !foreach(op, [G_FMINNUM, G_FMAXNUM, G_FMINIMUM, G_FMAXIMUM],
+           (pattern (op $dst, $src0, $src1)))>;
+
 def combine_minmax_nan: GICombineRule<
-  (defs root:$root, unsigned_matchinfo:$info),
-  (match (wip_match_opcode G_FMINNUM, G_FMAXNUM, G_FMINIMUM, G_FMAXIMUM):$root,
+  (defs root:$dst, unsigned_matchinfo:$info),
+  (match (combine_minmax_nan_frags $dst):$root,
          [{ return Helper.matchCombineFMinMaxNaN(*${root}, ${info}); }]),
   (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, ${info}); }])>;
 
@@ -1826,13 +1915,13 @@ def buildvector_identity_fold : GICombineRule<
 
 def trunc_buildvector_fold : GICombineRule<
   (defs root:$op, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_TRUNC):$op,
+  (match (G_TRUNC $dst, $src):$op,
       [{ return Helper.matchTruncBuildVectorFold(*${op}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${op}, ${matchinfo}); }])>;
 
 def trunc_lshr_buildvector_fold : GICombineRule<
   (defs root:$op, register_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_TRUNC):$op,
+  (match (G_TRUNC $dst, $src):$op,
       [{ return Helper.matchTruncLshrBuildVectorFold(*${op}, ${matchinfo}); }]),
   (apply [{ Helper.replaceSingleDefInstWithReg(*${op}, ${matchinfo}); }])>;
 
@@ -1843,7 +1932,7 @@ def trunc_lshr_buildvector_fold : GICombineRule<
 //   x - (x + z) -> 0 - z
 def sub_add_reg: GICombineRule <
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SUB):$root,
+  (match (G_SUB $dst, $src1, $src2):$root,
          [{ return Helper.matchSubAddSameReg(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
@@ -1868,7 +1957,7 @@ def fptrunc_fpext_fold : GICombineRule<
 
 def select_to_minmax: GICombineRule<
   (defs root:$root, build_fn_matchinfo:$info),
-  (match (wip_match_opcode G_SELECT):$root,
+  (match (G_SELECT $dst, $cond, $tval, $fval):$root,
          [{ return Helper.matchSimplifySelectToMinMax(*${root}, ${info}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
 
@@ -1881,25 +1970,25 @@ def select_to_iminmax: GICombineRule<
 
 def simplify_neg_minmax : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SUB):$root,
+  (match (G_SUB $dst, $src1, $src2):$root,
          [{ return Helper.matchSimplifyNegMinMax(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
 def match_selects : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SELECT):$root,
+  (match (G_SELECT $dst, $cond, $tval, $fval):$root,
         [{ return Helper.matchSelect(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
 def match_ands : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_AND):$root,
+  (match (G_AND $dst, $src1, $src2):$root,
         [{ return Helper.matchAnd(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
 def match_ors : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_OR):$root,
+  (match (G_OR $dst, $src1, $src2):$root,
         [{ return Helper.matchOr(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
@@ -1926,7 +2015,7 @@ def extract_vector_element_undef : GICombineRule <
 
 def match_extract_of_element : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+  (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
         [{ return Helper.matchExtractVectorElement(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
@@ -2033,14 +2122,14 @@ def nneg_zext : GICombineRule<
 // Combines concat operations
 def combine_concat_vector : GICombineRule<
   (defs root:$root, register_vector_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_CONCAT_VECTORS):$root,
+  (match (G_CONCAT_VECTORS $dst, GIVariadic<>:$srcs):$root,
         [{ return Helper.matchCombineConcatVectors(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyCombineConcatVectors(*${root}, ${matchinfo}); }])>;
 
 // Combines shuffle operations
 def combine_shuffle_vector : GICombineRule<
   (defs root:$root, register_vector_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+  (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
         [{ return Helper.matchCombineShuffleVector(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyCombineShuffleVector(*${root}, ${matchinfo}); }])>;
 
@@ -2052,7 +2141,7 @@ def combine_shuffle_vector : GICombineRule<
 // c = G_CONCAT_VECTORS x, y, z, undef
 def combine_shuffle_concat : GICombineRule<
   (defs root:$root, register_vector_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+  (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
         [{ return Helper.matchCombineShuffleConcat(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyCombineShuffleConcat(*${root}, ${matchinfo}); }])>;
 
@@ -2090,7 +2179,7 @@ def insert_vector_element_extract_vector_element : GICombineRule<
 
 def insert_vector_elt_oob : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_INSERT_VECTOR_ELT):$root,
+  (match (G_INSERT_VECTOR_ELT $dst, $src, $elt, $idx):$root,
          [{ return Helper.matchInsertVectorElementOOB(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
 
@@ -2149,7 +2238,7 @@ def combine_shuffle_undef_rhs : GICombineRule<
 
 def combine_shuffle_disjoint_mask : GICombineRule<
   (defs root:$root, build_fn_matchinfo:$matchinfo),
-  (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+  (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
         [{ return Helper.matchShuffleDisjointMask(*${root}, ${matchinfo}); }]),
   (apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])
 >;
diff --git a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
index 0ee8a9a204c09..cd672ed0b6f53 100644
--- a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
@@ -1111,12 +1111,6 @@ bool CombinerHelper::matchSextTruncSextLoad(MachineInstr &MI) const {
   return false;
 }
 
-void CombinerHelper::applySextTruncSextLoad(MachineInstr &MI) const {
-  assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
-  Builder.buildCopy(MI.getOperand(0).getReg(), MI.getOperand(1).getReg());
-  MI.eraseFromParent();
-}
-
 bool CombinerHelper::matchSextInRegOfLoad(
     MachineInstr &MI, std::tuple<Register, unsigned> &MatchInfo) const {
   assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
@@ -2129,40 +2123,10 @@ void CombinerHelper::applyShiftOfShiftedLogic(
   MI.eraseFromParent();
 }
 
-bool CombinerHelper::matchCommuteShift(MachineInstr &MI,
-                                       BuildFnTy &MatchInfo) const {
-  assert(MI.getOpcode() == TargetOpcode::G_SHL && "Expected G_SHL");
-  // Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
-  // Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
-  auto &Shl = cast<GenericMachineInstr>(MI);
-  Register DstReg = Shl.getReg(0);
-  Register SrcReg = Shl.getReg(1);
-  Register ShiftReg = Shl.getReg(2);
-  Register X, C1;
-
-  if (!getTargetLowering().isDesirableToCommuteWithShift(MI, !isPreLegalize()))
-    return false;
-
-  MachineInstr *SrcDef;
-  if (!mi_match(SrcReg, MRI,
-                m_OneNonDBGUse(m_any_of(m_GAdd(m_Reg(X), m_Reg(C1)),
-                                        m_GOr(m_Reg(X), m_Reg(C1))))) ||
-      !mi_match(SrcReg, MRI, m_MInstr(SrcDef)))
-    return false;
-
-  APInt C1Val, C2Val;
-  if (!mi_match(C1, MRI, m_ICstOrSplat(C1Val)) ||
-      !mi_match(ShiftReg, MRI, m_ICstOrSplat(C2Val)))
-    return false;
-
-  unsigned SrcOpc = SrcDef->getOpcode();
-  LLT SrcTy = MRI.getType(SrcReg);
-  MatchInfo = [=](MachineIRBuilder &B) {
-    auto S1 = B.buildShl(SrcTy, X, ShiftReg);
-    auto S2 = B.buildShl(SrcTy, C1, ShiftReg);
-    B.buildInstr(SrcOpc, {DstReg}, {S1, S2});
-  };
-  return true;
+bool CombinerHelper::isDesirableToCommuteWithShift(
+    const MachineInstr &MI) const {
+  return getTargetLowering().isDesirableToCommuteWithShift(MI,
+                                                           !isPreLegalize());
 }
 
 bool CombinerHelper::matchLshrOfTruncOfLshr(MachineInstr &MI,
@@ -2651,24 +2615,6 @@ bool CombinerHelper::tryCombineShiftToUnmerge(
   return false;
 }
 
-bool CombinerHelper::matchCombineI2PToP2I(MachineInstr &MI,
-                                          Register &Reg) const {
-  assert(MI.getOpcode() == TargetOpcode::G_INTTOPTR && "Expected a G_INTTOPTR");
-  Register DstReg = MI.getOperand(0).getReg();
-  LLT DstTy = MRI.getType(DstReg);
-  Register SrcReg = MI.getOperand(1).getReg();
-  return mi_match(SrcReg, MRI,
-                  m_GPtrToInt(m_all_of(m_SpecificType(DstTy), m_Reg(Reg))));
-}
-
-void CombinerHelper::applyCombineI2PToP2I(MachineInstr &MI,
-                                          Register &Reg) const {
-  assert(MI.getOpcode() == TargetOpcode::G_INTTOPTR && "Expected a G_INTTOPTR");
-  Register DstReg = MI.getOperand(0).getReg();
-  Builder.buildCopy(DstReg, Reg);
-  MI.eraseFromParent();
-}
-
 void CombinerHelper::applyCombineP2IToI2P(MachineInstr &MI,
                                           Register &Reg) const {
   assert(MI.getOpcode() == TargetOpcode::G_PTRTOINT && "Expected a G_PTRTOINT");
@@ -3992,12 +3938,6 @@ bool CombinerHelper::matchPtrAddZero(MachineInstr &MI) const {
   return isBuildVectorAllZeros(*VecMI, MRI);
 }
 
-void CombinerHelper::applyPtrAddZero(MachineInstr &MI) const {
-  auto &PtrAdd = cast<GPtrAdd>(MI);
-  Builder.buildIntToPtr(PtrAdd.getReg(0), PtrAdd.getOffsetReg());
-  PtrAdd.eraseFromParent();
-}
-
 /// The second source operand is known to be a power of 2.
 void CombinerHelper::applySimplifyURemByPow2(MachineInstr &MI) const {
   Register DstReg = MI.getOperand(0).getReg();
@@ -4959,8 +4899,13 @@ bool CombinerHelper::matchBitfieldExtractFromAnd(MachineInstr &MI,
                        m_ICst(AndImm))))
     return false;
 
+  // AndImm is sign-extended to 64 bits by m_ICst; restrict it to the operand
+  // width so an all-ones mask (a redundant AND) is not misread as a wider mask.
+  uint64_t MaybeMask = static_cast<uint64_t>(AndImm);
+  if (Size < 64)
+    MaybeMask &= maskTrailingOnes<uint64_t>(Size);
+
   // The mask is a mask of the low bits iff imm & (imm+1) == 0.
-  auto MaybeMask = static_cast<uint64_t>(AndImm);
   if (MaybeMask & (MaybeMask + 1))
     return false;
 
@@ -4968,7 +4913,14 @@ bool CombinerHelper::matchBitfieldExtractFromAnd(MachineInstr &MI,
   if (static_cast<uint64_t>(LSBImm) >= Size)
     return false;
 
-  uint64_t Width = APInt(Size, AndImm).countr_one();
+  uint64_t Width = APInt(Size, MaybeMask).countr_one();
+  // The extracted field [LSB, LSB+Width) must fit within the register.
+  // Otherwise this is a redundant AND (e.g. an all-ones mask combined with a
+  // non-zero shift) that is better handled by other combines, and would form
+  // an out-of-range bitfield extract.
+  if (static_cast<uint64_t>(LSBImm) + Width > Size)
+    return false;
+
   MatchInfo = [=](MachineIRBuilder &B) {
     auto WidthCst = B.buildConstant(ExtractTy, Width);
     auto LSBCst = B.buildConstant(ExtractTy, LSBImm);
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
index a9adf7a2e46ae..f4bcf38670196 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
@@ -229,9 +229,8 @@ body:             |
     ; CHECK: liveins: $x0
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: [[C:%[0-9]+]]:_(i8) = G_CONSTANT i8 127
-    ; CHECK-NEXT: [[DEF:%[0-9]+]]:_(i8) = G_IMPLICIT_DEF
+    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
     ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
-    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[DEF]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
     ; CHECK-NEXT: G_STORE [[BUILD_VECTOR]](<32 x i8>), [[COPY]](p0) :: (store (<32 x i8>))
     ; CHECK-NEXT: RET_ReallyLR
     %3:_(i8) = G_CONSTANT i8 127
@@ -253,9 +252,8 @@ body:             |
     ; CHECK: liveins: $x0
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: [[C:%[0-9]+]]:_(i8) = G_CONSTANT i8 127
-    ; CHECK-NEXT: [[DEF:%[0-9]+]]:_(i8) = G_IMPLICIT_DEF
+    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
     ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
-    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[DEF]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
     ; CHECK-NEXT: G_STORE [[BUILD_VECTOR]](<32 x i8>), [[COPY]](p0) :: (store (<32 x i8>))
     ; CHECK-NEXT: RET_ReallyLR
     %3:_(i8) = G_CONSTANT i8 127
@@ -319,13 +317,13 @@ body:             |
     ; CHECK-LABEL: name: test_inlineasm_base
     ; CHECK: liveins: $x0
     ; CHECK-NEXT: {{  $}}
-    ; CHECK-NEXT: [[CONST1:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
-    ; CHECK-NEXT: [[CONST2:%[0-9]+]]:_(i32) = G_CONSTANT i32 42
+    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
+    ; CHECK-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 42
     ; CHECK-NEXT: [[DEF:%[0-9]+]]:gpr64 = IMPLICIT_DEF
     ; CHECK-NEXT: INLINEASM &"ldr $0, [$1]", sideeffect attdialect, regdef:FPR128, def %3(<4 x i32>), reguse:GPR64, [[DEF]]
-    ; CHECK-NEXT: [[INSERT_VECTOR_ELT:%[0-9]+]]:_(<4 x i32>) = G_INSERT_VECTOR_ELT %3, [[CONST2]](i32), [[CONST1]](i64)
+    ; CHECK-NEXT: [[IVEC:%[0-9]+]]:_(<4 x i32>) = G_INSERT_VECTOR_ELT %3, [[C1]](i32), [[C]](i64)
     ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
-    ; CHECK-NEXT: G_STORE [[INSERT_VECTOR_ELT]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
+    ; CHECK-NEXT: G_STORE [[IVEC]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
     ; CHECK-NEXT: RET_ReallyLR
     %0:_(i64) = G_CONSTANT i64 0
     %1:_(i32) = G_CONSTANT i32 42
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir b/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
index 3abd11b3d5762..e5c014edc4d60 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
@@ -286,8 +286,9 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: %x:_(i64) = COPY $x0
     ; CHECK-NEXT: %lsb:_(i64) = G_CONSTANT i64 5
-    ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 64
-    ; CHECK-NEXT: %and:_(i64) = G_UBFX %x, %lsb(i64), [[C]]
+    ; CHECK-NEXT: %mask:_(i64) = G_CONSTANT i64 -1
+    ; CHECK-NEXT: %shift:_(i64) = G_LSHR %x, %lsb(i64)
+    ; CHECK-NEXT: %and:_(i64) = G_AND %shift, %mask
     ; CHECK-NEXT: $x0 = COPY %and(i64)
     ; CHECK-NEXT: RET_ReallyLR implicit $x0
     %x:_(i64) = COPY $x0
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
index b66ed3b50feb7..c9e8adfeaa94a 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
@@ -16,7 +16,6 @@ body:             |
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT:   [[COPY:%[0-9]+]]:_(p0) = COPY $x0
   ; CHECK-NEXT:   [[DEF:%[0-9]+]]:_(i1) = G_IMPLICIT_DEF
-  ; CHECK-NEXT:   [[DEF1:%[0-9]+]]:_(i64) = G_IMPLICIT_DEF
   ; CHECK-NEXT:   [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
   ; CHECK-NEXT:   G_BRCOND [[DEF]](i1), %bb.2
   ; CHECK-NEXT:   G_BR %bb.1
@@ -24,8 +23,8 @@ body:             |
   ; CHECK-NEXT: bb.1:
   ; CHECK-NEXT:   successors: %bb.2(0x80000000)
   ; CHECK-NEXT: {{  $}}
-  ; CHECK-NEXT:   [[FREEZE:%[0-9]+]]:_(i64) = G_FREEZE [[DEF1]]
-  ; CHECK-NEXT:   [[UDIV:%[0-9]+]]:_(i64) = G_UDIV [[FREEZE]], [[C]]
+  ; CHECK-NEXT:   [[C1:%[0-9]+]]:_(i64) = G_CONSTANT i64 -1
+  ; CHECK-NEXT:   [[UDIV:%[0-9]+]]:_(i64) = G_UDIV [[C1]], [[C]]
   ; CHECK-NEXT:   G_STORE [[UDIV]](i64), [[COPY]](p0) :: (store (i64))
   ; CHECK-NEXT: {{  $}}
   ; CHECK-NEXT: bb.2:
diff --git a/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll b/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
index 128b67663bf04..93c22573c0511 100644
--- a/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
+++ b/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
@@ -559,7 +559,7 @@ define <8 x i8> @no_shuffle_only_some_and_constants(<8 x i8> %src, <8 x i8> %mas
 ; CHECK-GI-NEXT:    mov x8, sp
 ; CHECK-GI-NEXT:    str d0, [sp]
 ; CHECK-GI-NEXT:    and w9, w9, #0x7
-; CHECK-GI-NEXT:    and x9, x9, #0xff
+; CHECK-GI-NEXT:    and x9, x9, #0x7
 ; CHECK-GI-NEXT:    lsl x11, x9, #1
 ; CHECK-GI-NEXT:    sub x9, x11, x9
 ; CHECK-GI-NEXT:    ldr b2, [x8, x9]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
index 5a8027bf75881..8a4d6c1d67b4d 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
@@ -1024,25 +1024,25 @@ define <2 x float> @v_ashr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
 ; GFX6-LABEL: v_ashr_v4i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v4, 16, v2
-; GFX6-NEXT:    v_bfe_i32 v6, v0, 0, 16
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT:    v_bfe_i32 v5, v0, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v0, v0, 16, 16
-; GFX6-NEXT:    v_lshrrev_b32_e32 v5, 16, v3
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_ashrrev_i32_e32 v0, v4, v0
-; GFX6-NEXT:    v_bfe_i32 v4, v1, 0, 16
+; GFX6-NEXT:    v_ashrrev_i32_e32 v4, v4, v5
+; GFX6-NEXT:    v_ashrrev_i32_e32 v0, v2, v0
+; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT:    v_bfe_i32 v5, v1, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v1, v1, 16, 16
-; GFX6-NEXT:    v_ashrrev_i32_e32 v2, v2, v6
-; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT:    v_ashrrev_i32_e32 v1, v5, v1
+; GFX6-NEXT:    v_ashrrev_i32_e32 v1, v3, v1
+; GFX6-NEXT:    v_ashrrev_i32_e32 v2, v2, v5
 ; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT:    v_ashrrev_i32_e32 v3, v3, v4
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
 ; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT:    v_or_b32_e32 v0, v2, v0
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v4
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT:    v_or_b32_e32 v0, v3, v0
 ; GFX6-NEXT:    v_or_b32_e32 v1, v2, v1
 ; GFX6-NEXT:    s_setpc_b64 s[30:31]
 ;
@@ -1078,23 +1078,23 @@ define <2 x float> @v_ashr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
 define amdgpu_ps <2 x i32> @s_ashr_v4i16(<4 x i16> inreg %value, <4 x i16> inreg %amount) {
 ; GFX6-LABEL: s_ashr_v4i16:
 ; GFX6:       ; %bb.0:
-; GFX6-NEXT:    s_lshr_b32 s4, s2, 16
-; GFX6-NEXT:    s_sext_i32_i16 s6, s0
+; GFX6-NEXT:    s_sext_i32_i16 s4, s0
+; GFX6-NEXT:    s_ashr_i32 s4, s4, s2
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s0, s0, 0x100010
-; GFX6-NEXT:    s_lshr_b32 s5, s3, 16
-; GFX6-NEXT:    s_ashr_i32 s0, s0, s4
-; GFX6-NEXT:    s_sext_i32_i16 s4, s1
+; GFX6-NEXT:    s_ashr_i32 s0, s0, s2
+; GFX6-NEXT:    s_sext_i32_i16 s2, s1
+; GFX6-NEXT:    s_ashr_i32 s2, s2, s3
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s1, s1, 0x100010
-; GFX6-NEXT:    s_ashr_i32 s2, s6, s2
-; GFX6-NEXT:    s_ashr_i32 s1, s1, s5
+; GFX6-NEXT:    s_ashr_i32 s1, s1, s3
 ; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT:    s_ashr_i32 s3, s4, s3
-; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
 ; GFX6-NEXT:    s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT:    s_or_b32 s0, s2, s0
-; GFX6-NEXT:    s_and_b32 s2, s3, 0xffff
+; GFX6-NEXT:    s_and_b32 s3, s4, 0xffff
+; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
+; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, 16
+; GFX6-NEXT:    s_or_b32 s0, s3, s0
 ; GFX6-NEXT:    s_or_b32 s1, s2, s1
 ; GFX6-NEXT:    ; return to shader part epilog
 ;
@@ -1189,45 +1189,45 @@ define <4 x float> @v_ashr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
 ; GFX6-LABEL: v_ashr_v8i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v4
-; GFX6-NEXT:    v_bfe_i32 v12, v0, 0, 16
+; GFX6-NEXT:    v_and_b32_e32 v8, 0xffff, v4
+; GFX6-NEXT:    v_bfe_i32 v9, v0, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v4, v4, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v0, v0, 16, 16
-; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v5
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT:    v_ashrrev_i32_e32 v0, v8, v0
-; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v5
-; GFX6-NEXT:    v_bfe_i32 v8, v1, 0, 16
+; GFX6-NEXT:    v_ashrrev_i32_e32 v8, v8, v9
+; GFX6-NEXT:    v_ashrrev_i32_e32 v0, v4, v0
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT:    v_bfe_i32 v9, v1, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v5, v5, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v1, v1, 16, 16
-; GFX6-NEXT:    v_lshrrev_b32_e32 v10, 16, v6
-; GFX6-NEXT:    v_ashrrev_i32_e32 v4, v4, v12
-; GFX6-NEXT:    v_ashrrev_i32_e32 v5, v5, v8
-; GFX6-NEXT:    v_ashrrev_i32_e32 v1, v9, v1
-; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v6
-; GFX6-NEXT:    v_bfe_i32 v8, v2, 0, 16
+; GFX6-NEXT:    v_ashrrev_i32_e32 v4, v4, v9
+; GFX6-NEXT:    v_ashrrev_i32_e32 v1, v5, v1
+; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v6
+; GFX6-NEXT:    v_bfe_i32 v9, v2, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v6, v6, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v2, v2, 16, 16
-; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v11, 16, v7
-; GFX6-NEXT:    v_ashrrev_i32_e32 v6, v6, v8
-; GFX6-NEXT:    v_ashrrev_i32_e32 v2, v10, v2
-; GFX6-NEXT:    v_bfe_i32 v8, v3, 0, 16
+; GFX6-NEXT:    v_ashrrev_i32_e32 v5, v5, v9
+; GFX6-NEXT:    v_ashrrev_i32_e32 v2, v6, v2
+; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v7
+; GFX6-NEXT:    v_bfe_i32 v9, v3, 0, 16
+; GFX6-NEXT:    v_bfe_u32 v7, v7, 16, 16
 ; GFX6-NEXT:    v_bfe_i32 v3, v3, 16, 16
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
 ; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT:    v_and_b32_e32 v7, 0xffff, v7
-; GFX6-NEXT:    v_ashrrev_i32_e32 v3, v11, v3
-; GFX6-NEXT:    v_or_b32_e32 v0, v4, v0
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT:    v_ashrrev_i32_e32 v3, v7, v3
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_ashrrev_i32_e32 v7, v7, v8
+; GFX6-NEXT:    v_ashrrev_i32_e32 v6, v6, v9
+; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX6-NEXT:    v_or_b32_e32 v1, v4, v1
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v6
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v5
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
 ; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT:    v_and_b32_e32 v7, 0xffff, v8
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
 ; GFX6-NEXT:    v_or_b32_e32 v2, v4, v2
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v7
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v6
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v3, 16, v3
+; GFX6-NEXT:    v_or_b32_e32 v0, v7, v0
 ; GFX6-NEXT:    v_or_b32_e32 v3, v4, v3
 ; GFX6-NEXT:    s_setpc_b64 s[30:31]
 ;
@@ -1273,41 +1273,41 @@ define <4 x float> @v_ashr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
 define amdgpu_ps <4 x i32> @s_ashr_v8i16(<8 x i16> inreg %value, <8 x i16> inreg %amount) {
 ; GFX6-LABEL: s_ashr_v8i16:
 ; GFX6:       ; %bb.0:
-; GFX6-NEXT:    s_lshr_b32 s8, s4, 16
-; GFX6-NEXT:    s_sext_i32_i16 s12, s0
+; GFX6-NEXT:    s_sext_i32_i16 s8, s0
+; GFX6-NEXT:    s_ashr_i32 s8, s8, s4
+; GFX6-NEXT:    s_bfe_u32 s4, s4, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s0, s0, 0x100010
-; GFX6-NEXT:    s_lshr_b32 s9, s5, 16
-; GFX6-NEXT:    s_ashr_i32 s0, s0, s8
-; GFX6-NEXT:    s_sext_i32_i16 s8, s1
+; GFX6-NEXT:    s_ashr_i32 s0, s0, s4
+; GFX6-NEXT:    s_sext_i32_i16 s4, s1
+; GFX6-NEXT:    s_ashr_i32 s4, s4, s5
+; GFX6-NEXT:    s_bfe_u32 s5, s5, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s1, s1, 0x100010
-; GFX6-NEXT:    s_lshr_b32 s10, s6, 16
-; GFX6-NEXT:    s_ashr_i32 s4, s12, s4
-; GFX6-NEXT:    s_ashr_i32 s5, s8, s5
-; GFX6-NEXT:    s_ashr_i32 s1, s1, s9
-; GFX6-NEXT:    s_sext_i32_i16 s8, s2
+; GFX6-NEXT:    s_ashr_i32 s1, s1, s5
+; GFX6-NEXT:    s_sext_i32_i16 s5, s2
+; GFX6-NEXT:    s_ashr_i32 s5, s5, s6
+; GFX6-NEXT:    s_bfe_u32 s6, s6, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s2, s2, 0x100010
-; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT:    s_lshr_b32 s11, s7, 16
-; GFX6-NEXT:    s_ashr_i32 s6, s8, s6
-; GFX6-NEXT:    s_ashr_i32 s2, s2, s10
-; GFX6-NEXT:    s_sext_i32_i16 s8, s3
+; GFX6-NEXT:    s_ashr_i32 s2, s2, s6
+; GFX6-NEXT:    s_sext_i32_i16 s6, s3
+; GFX6-NEXT:    s_ashr_i32 s6, s6, s7
+; GFX6-NEXT:    s_bfe_u32 s7, s7, 0x100010
 ; GFX6-NEXT:    s_bfe_i32 s3, s3, 0x100010
-; GFX6-NEXT:    s_and_b32 s4, s4, 0xffff
-; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
 ; GFX6-NEXT:    s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT:    s_ashr_i32 s3, s3, s11
-; GFX6-NEXT:    s_or_b32 s0, s4, s0
-; GFX6-NEXT:    s_and_b32 s4, s5, 0xffff
+; GFX6-NEXT:    s_ashr_i32 s3, s3, s7
+; GFX6-NEXT:    s_and_b32 s4, s4, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, 16
 ; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT:    s_ashr_i32 s7, s8, s7
+; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
 ; GFX6-NEXT:    s_or_b32 s1, s4, s1
-; GFX6-NEXT:    s_and_b32 s4, s6, 0xffff
+; GFX6-NEXT:    s_and_b32 s4, s5, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
 ; GFX6-NEXT:    s_and_b32 s3, s3, 0xffff
+; GFX6-NEXT:    s_and_b32 s7, s8, 0xffff
+; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
 ; GFX6-NEXT:    s_or_b32 s2, s4, s2
-; GFX6-NEXT:    s_and_b32 s4, s7, 0xffff
+; GFX6-NEXT:    s_and_b32 s4, s6, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s3, s3, 16
+; GFX6-NEXT:    s_or_b32 s0, s7, s0
 ; GFX6-NEXT:    s_or_b32 s3, s4, s3
 ; GFX6-NEXT:    ; return to shader part epilog
 ;
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
index c595f8b085739..d4b1b75fa82a9 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
@@ -8,9 +8,8 @@ tracksRegLiveness: true
 body:             |
   bb.0:
     ; CHECK-LABEL: name: test_ptradd_crash__offset_smaller
-    ; CHECK: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 12
-    ; CHECK-NEXT: [[INTTOPTR:%[0-9]+]]:_(p1) = G_INTTOPTR [[C]](s64)
-    ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[INTTOPTR]](p1) :: (load (s32), addrspace 1)
+    ; CHECK: [[C:%[0-9]+]]:_(p1) = G_CONSTANT i64 12
+    ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[C]](p1) :: (load (s32), addrspace 1)
     ; CHECK-NEXT: $sgpr0 = COPY [[LOAD]](s32)
     ; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
     %1:_(p1) = G_CONSTANT i64 0
@@ -28,9 +27,8 @@ tracksRegLiveness: true
 body:             |
   bb.0:
     ; CHECK-LABEL: name: test_ptradd_crash__offset_wider
-    ; CHECK: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 12
-    ; CHECK-NEXT: [[INTTOPTR:%[0-9]+]]:_(p1) = G_INTTOPTR [[C]](s64)
-    ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[INTTOPTR]](p1) :: (load (s32), addrspace 1)
+    ; CHECK: [[C:%[0-9]+]]:_(p1) = G_CONSTANT i64 12
+    ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[C]](p1) :: (load (s32), addrspace 1)
     ; CHECK-NEXT: $sgpr0 = COPY [[LOAD]](s32)
     ; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
     %1:_(p1) = G_CONSTANT i64 0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
index cb6de736d13e9..70bcd3d95c662 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
@@ -110,8 +110,9 @@ body:             |
     ; CHECK-NEXT: {{  $}}
     ; CHECK-NEXT: [[COPY:%[0-9]+]]:_(s32) = COPY $vgpr0
     ; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 5
-    ; CHECK-NEXT: [[LSHR:%[0-9]+]]:_(s32) = G_LSHR [[COPY]], [[C]](s32)
-    ; CHECK-NEXT: $vgpr0 = COPY [[LSHR]](s32)
+    ; CHECK-NEXT: [[C1:%[0-9]+]]:_(s32) = G_CONSTANT i32 27
+    ; CHECK-NEXT: [[UBFX:%[0-9]+]]:_(s32) = G_UBFX [[COPY]], [[C]](s32), [[C1]]
+    ; CHECK-NEXT: $vgpr0 = COPY [[UBFX]](s32)
     ; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $vgpr0
     %0:_(s32) = COPY $vgpr0
     %1:_(s32) = G_CONSTANT i32 5
@@ -153,8 +154,9 @@ tracksRegLiveness: true
 body:             |
   bb.0:
     ; CHECK-LABEL: name: test_sext_inreg
-    ; CHECK: %cst_1:_(s32) = G_CONSTANT i32 -5
-    ; CHECK-NEXT: $sgpr0 = COPY %cst_1(s32)
+    ; CHECK: %cst_11:_(s32) = G_CONSTANT i32 11
+    ; CHECK-NEXT: %sext_inreg_11:_(s32) = G_SEXT_INREG %cst_11, 4
+    ; CHECK-NEXT: $sgpr0 = COPY %sext_inreg_11(s32)
     ; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
     %cst_1:_(s32) = G_CONSTANT i32 -5
 
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
index 98de0a416e5b9..ce1112d8b496a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
@@ -41,7 +41,7 @@ define amdgpu_cs i32 @test_shl_and_3(i32 inreg %arg1) {
 define amdgpu_cs i32 @test_lshr_and_1(i32 inreg %arg1) {
 ; CHECK-LABEL: test_lshr_and_1:
 ; CHECK:       ; %bb.0: ; %.entry
-; CHECK-NEXT:    s_lshr_b32 s0, s0, 4
+; CHECK-NEXT:    s_bfe_u32 s0, s0, 0x1c0004
 ; CHECK-NEXT:    ; return to shader part epilog
 .entry:
   %z1 = lshr i32 %arg1, 2
@@ -66,7 +66,7 @@ define amdgpu_cs i32 @test_lshr_and_2(i32 inreg %arg1) {
 define amdgpu_cs i32 @test_lshr_and_3(i32 inreg %arg1) {
 ; CHECK-LABEL: test_lshr_and_3:
 ; CHECK:       ; %bb.0: ; %.entry
-; CHECK-NEXT:    s_lshr_b32 s0, s0, 5
+; CHECK-NEXT:    s_bfe_u32 s0, s0, 0x1b0005
 ; CHECK-NEXT:    ; return to shader part epilog
 .entry:
   %z1 = lshr i32 %arg1, 3
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
index dd6e3bd7ebb70..270626c35ba60 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
@@ -4822,10 +4822,11 @@ define amdgpu_ps i48 @s_fshl_v3i16(<3 x i16> inreg %lhs, <3 x i16> inreg %rhs, <
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s8
 ; GFX6-NEXT:    s_bfe_u32 s8, s2, 0xf0001
 ; GFX6-NEXT:    s_lshr_b32 s4, s8, s4
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
 ; GFX6-NEXT:    s_or_b32 s0, s0, s4
 ; GFX6-NEXT:    s_and_b32 s4, s7, 15
 ; GFX6-NEXT:    s_andn2_b32 s7, 15, s7
-; GFX6-NEXT:    s_lshr_b32 s2, s2, 17
+; GFX6-NEXT:    s_lshr_b32 s2, s2, 1
 ; GFX6-NEXT:    s_lshl_b32 s4, s6, s4
 ; GFX6-NEXT:    s_lshr_b32 s2, s2, s7
 ; GFX6-NEXT:    s_or_b32 s2, s4, s2
@@ -5075,8 +5076,9 @@ define <3 x half> @v_fshl_v3i16(<3 x i16> %lhs, <3 x i16> %rhs, <3 x i16> %amt)
 ; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v7
 ; GFX6-NEXT:    v_xor_b32_e32 v7, -1, v7
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
 ; GFX6-NEXT:    v_and_b32_e32 v7, 15, v7
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, 17, v2
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, 1, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v6
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v7, v2
 ; GFX6-NEXT:    v_or_b32_e32 v2, v4, v2
@@ -5231,10 +5233,11 @@ define amdgpu_ps <2 x i32> @s_fshl_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s10
 ; GFX6-NEXT:    s_bfe_u32 s10, s2, 0xf0001
 ; GFX6-NEXT:    s_lshr_b32 s4, s10, s4
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
 ; GFX6-NEXT:    s_or_b32 s0, s0, s4
 ; GFX6-NEXT:    s_and_b32 s4, s8, 15
 ; GFX6-NEXT:    s_andn2_b32 s8, 15, s8
-; GFX6-NEXT:    s_lshr_b32 s2, s2, 17
+; GFX6-NEXT:    s_lshr_b32 s2, s2, 1
 ; GFX6-NEXT:    s_lshl_b32 s4, s6, s4
 ; GFX6-NEXT:    s_lshr_b32 s2, s2, s8
 ; GFX6-NEXT:    s_or_b32 s2, s4, s2
@@ -5245,10 +5248,11 @@ define amdgpu_ps <2 x i32> @s_fshl_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, s4
 ; GFX6-NEXT:    s_bfe_u32 s4, s3, 0xf0001
 ; GFX6-NEXT:    s_lshr_b32 s4, s4, s5
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
 ; GFX6-NEXT:    s_or_b32 s1, s1, s4
 ; GFX6-NEXT:    s_and_b32 s4, s9, 15
 ; GFX6-NEXT:    s_andn2_b32 s5, 15, s9
-; GFX6-NEXT:    s_lshr_b32 s3, s3, 17
+; GFX6-NEXT:    s_lshr_b32 s3, s3, 1
 ; GFX6-NEXT:    s_lshl_b32 s4, s7, s4
 ; GFX6-NEXT:    s_lshr_b32 s3, s3, s5
 ; GFX6-NEXT:    s_and_b32 s2, 0xffff, s2
@@ -5449,8 +5453,9 @@ define <4 x half> @v_fshl_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
 ; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v8
 ; GFX6-NEXT:    v_xor_b32_e32 v8, -1, v8
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
 ; GFX6-NEXT:    v_and_b32_e32 v8, 15, v8
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, 17, v2
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, 1, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v6
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v8, v2
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v5
@@ -5463,10 +5468,11 @@ define <4 x half> @v_fshl_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
 ; GFX6-NEXT:    v_bfe_u32 v4, v3, 1, 15
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v5, v4
 ; GFX6-NEXT:    v_xor_b32_e32 v5, -1, v9
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
 ; GFX6-NEXT:    v_or_b32_e32 v1, v1, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v9
 ; GFX6-NEXT:    v_and_b32_e32 v5, 15, v5
-; GFX6-NEXT:    v_lshrrev_b32_e32 v3, 17, v3
+; GFX6-NEXT:    v_lshrrev_b32_e32 v3, 1, v3
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v7
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v3, v5, v3
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
index 6bd74d89deb57..1b9d573c60aa4 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
@@ -4522,21 +4522,21 @@ define amdgpu_ps i48 @s_fshr_v3i16(<3 x i16> inreg %lhs, <3 x i16> inreg %rhs, <
 ; GFX6-LABEL: s_fshr_v3i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_lshr_b32 s6, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s7, s2, 16
-; GFX6-NEXT:    s_lshr_b32 s8, s4, 16
-; GFX6-NEXT:    s_and_b32 s9, s4, 15
+; GFX6-NEXT:    s_lshr_b32 s7, s4, 16
+; GFX6-NEXT:    s_and_b32 s8, s4, 15
 ; GFX6-NEXT:    s_andn2_b32 s4, 15, s4
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, 1
-; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s4
-; GFX6-NEXT:    s_lshr_b32 s2, s2, s9
-; GFX6-NEXT:    s_or_b32 s0, s0, s2
-; GFX6-NEXT:    s_and_b32 s2, s8, 15
-; GFX6-NEXT:    s_andn2_b32 s4, 15, s8
+; GFX6-NEXT:    s_and_b32 s4, s2, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s4, s4, s8
+; GFX6-NEXT:    s_or_b32 s0, s0, s4
+; GFX6-NEXT:    s_and_b32 s4, s7, 15
+; GFX6-NEXT:    s_andn2_b32 s7, 15, s7
 ; GFX6-NEXT:    s_lshl_b32 s6, s6, 1
-; GFX6-NEXT:    s_lshl_b32 s4, s6, s4
-; GFX6-NEXT:    s_lshr_b32 s2, s7, s2
-; GFX6-NEXT:    s_or_b32 s2, s4, s2
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT:    s_lshl_b32 s6, s6, s7
+; GFX6-NEXT:    s_lshr_b32 s2, s2, s4
+; GFX6-NEXT:    s_or_b32 s2, s6, s2
 ; GFX6-NEXT:    s_and_b32 s4, s5, 15
 ; GFX6-NEXT:    s_andn2_b32 s5, 15, s5
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, 1
@@ -4767,26 +4767,26 @@ define <3 x half> @v_fshr_v3i16(<3 x i16> %lhs, <3 x i16> %rhs, <3 x i16> %amt)
 ; GFX6-LABEL: v_fshr_v3i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v4
-; GFX6-NEXT:    v_and_b32_e32 v9, 15, v4
+; GFX6-NEXT:    v_lshrrev_b32_e32 v7, 16, v4
+; GFX6-NEXT:    v_and_b32_e32 v8, 15, v4
 ; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v4
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v6, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v7, 16, v2
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 1, v0
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v9, v2
-; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v8
-; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 15, v8
-; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v8, v4
+; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
+; GFX6-NEXT:    v_and_b32_e32 v4, 15, v7
+; GFX6-NEXT:    v_xor_b32_e32 v7, -1, v7
+; GFX6-NEXT:    v_and_b32_e32 v7, 15, v7
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v6, 1, v6
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v6
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v7
-; GFX6-NEXT:    v_or_b32_e32 v2, v4, v2
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT:    v_lshlrev_b32_e32 v6, v7, v6
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v4, v2
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v5
 ; GFX6-NEXT:    v_xor_b32_e32 v5, -1, v5
+; GFX6-NEXT:    v_or_b32_e32 v2, v6, v2
 ; GFX6-NEXT:    v_and_b32_e32 v5, 15, v5
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 1, v1
 ; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
@@ -4932,41 +4932,41 @@ define amdgpu_ps <2 x i32> @s_fshr_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
 ; GFX6-LABEL: s_fshr_v4i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_lshr_b32 s6, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s8, s2, 16
-; GFX6-NEXT:    s_lshr_b32 s10, s4, 16
-; GFX6-NEXT:    s_and_b32 s12, s4, 15
+; GFX6-NEXT:    s_lshr_b32 s8, s4, 16
+; GFX6-NEXT:    s_and_b32 s10, s4, 15
 ; GFX6-NEXT:    s_andn2_b32 s4, 15, s4
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, 1
-; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s4
-; GFX6-NEXT:    s_lshr_b32 s2, s2, s12
-; GFX6-NEXT:    s_or_b32 s0, s0, s2
-; GFX6-NEXT:    s_and_b32 s2, s10, 15
-; GFX6-NEXT:    s_andn2_b32 s4, 15, s10
+; GFX6-NEXT:    s_and_b32 s4, s2, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s4, s4, s10
+; GFX6-NEXT:    s_or_b32 s0, s0, s4
+; GFX6-NEXT:    s_and_b32 s4, s8, 15
+; GFX6-NEXT:    s_andn2_b32 s8, 15, s8
 ; GFX6-NEXT:    s_lshl_b32 s6, s6, 1
-; GFX6-NEXT:    s_lshl_b32 s4, s6, s4
-; GFX6-NEXT:    s_lshr_b32 s2, s8, s2
-; GFX6-NEXT:    s_or_b32 s2, s4, s2
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT:    s_lshl_b32 s6, s6, s8
+; GFX6-NEXT:    s_lshr_b32 s2, s2, s4
+; GFX6-NEXT:    s_or_b32 s2, s6, s2
 ; GFX6-NEXT:    s_and_b32 s2, 0xffff, s2
+; GFX6-NEXT:    s_lshr_b32 s7, s1, 16
 ; GFX6-NEXT:    s_and_b32 s0, 0xffff, s0
 ; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
-; GFX6-NEXT:    s_lshr_b32 s7, s1, 16
-; GFX6-NEXT:    s_lshr_b32 s9, s3, 16
-; GFX6-NEXT:    s_or_b32 s0, s0, s2
-; GFX6-NEXT:    s_and_b32 s2, s5, 15
 ; GFX6-NEXT:    s_andn2_b32 s4, 15, s5
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, 1
-; GFX6-NEXT:    s_and_b32 s3, s3, 0xffff
-; GFX6-NEXT:    s_lshr_b32 s11, s5, 16
+; GFX6-NEXT:    s_or_b32 s0, s0, s2
+; GFX6-NEXT:    s_and_b32 s2, s5, 15
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, s4
-; GFX6-NEXT:    s_lshr_b32 s2, s3, s2
+; GFX6-NEXT:    s_and_b32 s4, s3, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s9, s5, 16
+; GFX6-NEXT:    s_lshr_b32 s2, s4, s2
 ; GFX6-NEXT:    s_or_b32 s1, s1, s2
-; GFX6-NEXT:    s_and_b32 s2, s11, 15
-; GFX6-NEXT:    s_andn2_b32 s3, 15, s11
-; GFX6-NEXT:    s_lshl_b32 s4, s7, 1
-; GFX6-NEXT:    s_lshl_b32 s3, s4, s3
-; GFX6-NEXT:    s_lshr_b32 s2, s9, s2
-; GFX6-NEXT:    s_or_b32 s2, s3, s2
+; GFX6-NEXT:    s_and_b32 s2, s9, 15
+; GFX6-NEXT:    s_andn2_b32 s4, 15, s9
+; GFX6-NEXT:    s_lshl_b32 s5, s7, 1
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT:    s_lshl_b32 s4, s5, s4
+; GFX6-NEXT:    s_lshr_b32 s2, s3, s2
+; GFX6-NEXT:    s_or_b32 s2, s4, s2
 ; GFX6-NEXT:    s_and_b32 s2, 0xffff, s2
 ; GFX6-NEXT:    s_and_b32 s1, 0xffff, s1
 ; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
@@ -5145,46 +5145,46 @@ define <4 x half> @v_fshr_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
 ; GFX6-LABEL: v_fshr_v4i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v10, 16, v4
-; GFX6-NEXT:    v_and_b32_e32 v12, 15, v4
+; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v4
+; GFX6-NEXT:    v_and_b32_e32 v10, 15, v4
 ; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v4
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v6, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v2
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 1, v0
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v12, v2
-; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v10
-; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 15, v10
-; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v10, v4
+; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
+; GFX6-NEXT:    v_and_b32_e32 v4, 15, v8
+; GFX6-NEXT:    v_xor_b32_e32 v8, -1, v8
+; GFX6-NEXT:    v_and_b32_e32 v8, 15, v8
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v6, 1, v6
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v6
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v8
-; GFX6-NEXT:    v_or_b32_e32 v2, v4, v2
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT:    v_lshlrev_b32_e32 v6, v8, v6
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v4, v2
+; GFX6-NEXT:    v_or_b32_e32 v2, v6, v2
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
 ; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v5
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v7, 16, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v3
-; GFX6-NEXT:    v_lshrrev_b32_e32 v11, 16, v5
-; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 15, v5
+; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
 ; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 1, v1
-; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v5
+; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
+; GFX6-NEXT:    v_and_b32_e32 v2, 15, v5
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v1, v4, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v3
-; GFX6-NEXT:    v_xor_b32_e32 v3, -1, v11
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v3
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v4
+; GFX6-NEXT:    v_xor_b32_e32 v4, -1, v9
 ; GFX6-NEXT:    v_or_b32_e32 v1, v1, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 15, v11
-; GFX6-NEXT:    v_and_b32_e32 v3, 15, v3
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 1, v7
-; GFX6-NEXT:    v_lshlrev_b32_e32 v3, v3, v4
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v9
-; GFX6-NEXT:    v_or_b32_e32 v2, v3, v2
+; GFX6-NEXT:    v_and_b32_e32 v2, 15, v9
+; GFX6-NEXT:    v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT:    v_lshlrev_b32_e32 v5, 1, v7
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v5
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v3
+; GFX6-NEXT:    v_or_b32_e32 v2, v4, v2
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
index ee93f413566e5..dbb753c0b3072 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
@@ -53,12 +53,13 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_flat(i32 %node_ptr, float
 define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16(i32 %node_ptr, float %ray_extent, <3 x float> %ray_origin, <3 x half> %ray_dir, <3 x half> %ray_inv_dir, <4 x i32> inreg %tdescr) {
 ; GFX10-LABEL: image_bvh_intersect_ray_a16:
 ; GFX10:       ; %bb.0:
-; GFX10-NEXT:    v_lshrrev_b32_e32 v9, 16, v5
+; GFX10-NEXT:    v_mov_b32_e32 v9, 16
 ; GFX10-NEXT:    v_and_b32_e32 v10, 0xffff, v7
+; GFX10-NEXT:    v_bfe_u32 v7, v7, 16, 16
 ; GFX10-NEXT:    v_and_b32_e32 v8, 0xffff, v8
-; GFX10-NEXT:    v_lshlrev_b32_e32 v9, 16, v9
+; GFX10-NEXT:    v_lshlrev_b32_sdwa v9, v9, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX10-NEXT:    v_lshlrev_b32_e32 v10, 16, v10
-; GFX10-NEXT:    v_alignbit_b32 v7, v8, v7, 16
+; GFX10-NEXT:    v_lshl_or_b32 v7, v8, 16, v7
 ; GFX10-NEXT:    v_and_or_b32 v5, 0xffff, v5, v9
 ; GFX10-NEXT:    v_and_or_b32 v6, 0xffff, v6, v10
 ; GFX10-NEXT:    image_bvh_intersect_ray v[0:3], v[0:7], s[0:3] a16
@@ -128,12 +129,13 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_flat(<2 x i32> %node_ptr
 define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16(i64 %node_ptr, float %ray_extent, <3 x float> %ray_origin, <3 x half> %ray_dir, <3 x half> %ray_inv_dir, <4 x i32> inreg %tdescr) {
 ; GFX10-LABEL: image_bvh64_intersect_ray_a16:
 ; GFX10:       ; %bb.0:
-; GFX10-NEXT:    v_lshrrev_b32_e32 v10, 16, v6
+; GFX10-NEXT:    v_mov_b32_e32 v10, 16
 ; GFX10-NEXT:    v_and_b32_e32 v11, 0xffff, v8
+; GFX10-NEXT:    v_bfe_u32 v8, v8, 16, 16
 ; GFX10-NEXT:    v_and_b32_e32 v9, 0xffff, v9
-; GFX10-NEXT:    v_lshlrev_b32_e32 v10, 16, v10
+; GFX10-NEXT:    v_lshlrev_b32_sdwa v10, v10, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX10-NEXT:    v_lshlrev_b32_e32 v11, 16, v11
-; GFX10-NEXT:    v_alignbit_b32 v8, v9, v8, 16
+; GFX10-NEXT:    v_lshl_or_b32 v8, v9, 16, v8
 ; GFX10-NEXT:    v_and_or_b32 v6, 0xffff, v6, v10
 ; GFX10-NEXT:    v_and_or_b32 v7, 0xffff, v7, v11
 ; GFX10-NEXT:    image_bvh64_intersect_ray v[0:3], v[0:8], s[0:3] a16
@@ -282,18 +284,19 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
 ; GFX1030:       ; %bb.0:
 ; GFX1030-NEXT:    v_mov_b32_e32 v13, v0
 ; GFX1030-NEXT:    v_mov_b32_e32 v14, v1
-; GFX1030-NEXT:    v_lshrrev_b32_e32 v0, 16, v5
+; GFX1030-NEXT:    v_mov_b32_e32 v0, 16
 ; GFX1030-NEXT:    v_and_b32_e32 v1, 0xffff, v7
 ; GFX1030-NEXT:    v_mov_b32_e32 v15, v2
-; GFX1030-NEXT:    v_and_b32_e32 v2, 0xffff, v8
 ; GFX1030-NEXT:    v_mov_b32_e32 v16, v3
-; GFX1030-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX1030-NEXT:    v_bfe_u32 v2, v7, 16, 16
+; GFX1030-NEXT:    v_lshlrev_b32_sdwa v0, v0, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX1030-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
+; GFX1030-NEXT:    v_and_b32_e32 v3, 0xffff, v8
 ; GFX1030-NEXT:    v_mov_b32_e32 v17, v4
-; GFX1030-NEXT:    v_alignbit_b32 v20, v2, v7, 16
 ; GFX1030-NEXT:    s_mov_b32 s1, exec_lo
 ; GFX1030-NEXT:    v_and_or_b32 v18, 0xffff, v5, v0
 ; GFX1030-NEXT:    v_and_or_b32 v19, 0xffff, v6, v1
+; GFX1030-NEXT:    v_lshl_or_b32 v20, v3, 16, v2
 ; GFX1030-NEXT:  .LBB7_1: ; =>This Inner Loop Header: Depth=1
 ; GFX1030-NEXT:    v_readfirstlane_b32 s4, v9
 ; GFX1030-NEXT:    v_readfirstlane_b32 s5, v10
@@ -323,13 +326,14 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
 ;
 ; GFX1013-LABEL: image_bvh_intersect_ray_a16_vgpr_descr:
 ; GFX1013:       ; %bb.0:
-; GFX1013-NEXT:    v_lshrrev_b32_e32 v13, 16, v5
+; GFX1013-NEXT:    v_mov_b32_e32 v13, 16
 ; GFX1013-NEXT:    v_and_b32_e32 v14, 0xffff, v7
+; GFX1013-NEXT:    v_bfe_u32 v7, v7, 16, 16
 ; GFX1013-NEXT:    v_and_b32_e32 v8, 0xffff, v8
 ; GFX1013-NEXT:    s_mov_b32 s1, exec_lo
-; GFX1013-NEXT:    v_lshlrev_b32_e32 v13, 16, v13
+; GFX1013-NEXT:    v_lshlrev_b32_sdwa v13, v13, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX1013-NEXT:    v_lshlrev_b32_e32 v14, 16, v14
-; GFX1013-NEXT:    v_alignbit_b32 v7, v8, v7, 16
+; GFX1013-NEXT:    v_lshl_or_b32 v7, v8, 16, v7
 ; GFX1013-NEXT:    v_and_or_b32 v5, 0xffff, v5, v13
 ; GFX1013-NEXT:    v_and_or_b32 v6, 0xffff, v6, v14
 ; GFX1013-NEXT:  .LBB7_1: ; =>This Inner Loop Header: Depth=1
@@ -549,18 +553,19 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
 ; GFX1030:       ; %bb.0:
 ; GFX1030-NEXT:    v_mov_b32_e32 v14, v0
 ; GFX1030-NEXT:    v_mov_b32_e32 v15, v1
-; GFX1030-NEXT:    v_lshrrev_b32_e32 v0, 16, v6
+; GFX1030-NEXT:    v_mov_b32_e32 v0, 16
 ; GFX1030-NEXT:    v_and_b32_e32 v1, 0xffff, v8
 ; GFX1030-NEXT:    v_mov_b32_e32 v16, v2
-; GFX1030-NEXT:    v_and_b32_e32 v2, 0xffff, v9
 ; GFX1030-NEXT:    v_mov_b32_e32 v17, v3
-; GFX1030-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX1030-NEXT:    v_bfe_u32 v2, v8, 16, 16
+; GFX1030-NEXT:    v_lshlrev_b32_sdwa v0, v0, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX1030-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
+; GFX1030-NEXT:    v_and_b32_e32 v3, 0xffff, v9
 ; GFX1030-NEXT:    v_mov_b32_e32 v18, v4
 ; GFX1030-NEXT:    v_mov_b32_e32 v19, v5
-; GFX1030-NEXT:    v_alignbit_b32 v22, v2, v8, 16
 ; GFX1030-NEXT:    v_and_or_b32 v20, 0xffff, v6, v0
 ; GFX1030-NEXT:    v_and_or_b32 v21, 0xffff, v7, v1
+; GFX1030-NEXT:    v_lshl_or_b32 v22, v3, 16, v2
 ; GFX1030-NEXT:    s_mov_b32 s1, exec_lo
 ; GFX1030-NEXT:  .LBB9_1: ; =>This Inner Loop Header: Depth=1
 ; GFX1030-NEXT:    v_readfirstlane_b32 s4, v10
@@ -592,13 +597,14 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
 ;
 ; GFX1013-LABEL: image_bvh64_intersect_ray_a16_vgpr_descr:
 ; GFX1013:       ; %bb.0:
-; GFX1013-NEXT:    v_lshrrev_b32_e32 v14, 16, v6
+; GFX1013-NEXT:    v_mov_b32_e32 v14, 16
 ; GFX1013-NEXT:    v_and_b32_e32 v15, 0xffff, v8
+; GFX1013-NEXT:    v_bfe_u32 v8, v8, 16, 16
 ; GFX1013-NEXT:    v_and_b32_e32 v9, 0xffff, v9
 ; GFX1013-NEXT:    s_mov_b32 s1, exec_lo
-; GFX1013-NEXT:    v_lshlrev_b32_e32 v14, 16, v14
+; GFX1013-NEXT:    v_lshlrev_b32_sdwa v14, v14, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
 ; GFX1013-NEXT:    v_lshlrev_b32_e32 v15, 16, v15
-; GFX1013-NEXT:    v_alignbit_b32 v8, v9, v8, 16
+; GFX1013-NEXT:    v_lshl_or_b32 v8, v9, 16, v8
 ; GFX1013-NEXT:    v_and_or_b32 v6, 0xffff, v6, v14
 ; GFX1013-NEXT:    v_and_or_b32 v7, 0xffff, v7, v15
 ; GFX1013-NEXT:  .LBB9_1: ; =>This Inner Loop Header: Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
index 5739d01e61d95..1f79f289ac719 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
@@ -37,9 +37,9 @@ define amdgpu_ps ptr addrspace(8) @basic_raw_buffer(ptr inreg %p) {
   ; CHECK45-NEXT:   [[S_MOV_B32_:%[0-9]+]]:sreg_32 = S_MOV_B32 0
   ; CHECK45-NEXT:   [[S_MOV_B:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO -6629298651489370112
   ; CHECK45-NEXT:   [[S_OR_B64_:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE]], [[S_MOV_B]], implicit-def dead $scc
-  ; CHECK45-NEXT:   [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 9
   ; CHECK45-NEXT:   [[S_MOV_B32_1:%[0-9]+]]:sreg_32 = S_MOV_B32 -536870912
   ; CHECK45-NEXT:   [[REG_SEQUENCE1:%[0-9]+]]:sreg_64 = REG_SEQUENCE [[S_MOV_B32_]], %subreg.sub0, [[S_MOV_B32_1]], %subreg.sub1
+  ; CHECK45-NEXT:   [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 9
   ; CHECK45-NEXT:   [[S_OR_B64_1:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE1]], [[S_MOV_B64_]], implicit-def dead $scc
   ; CHECK45-NEXT:   [[COPY2:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub0
   ; CHECK45-NEXT:   [[COPY3:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub1
@@ -97,9 +97,9 @@ define amdgpu_ps ptr addrspace(8) @large_num_records_raw_buffer(ptr inreg %p) {
   ; CHECK45-NEXT:   [[S_MOV_B32_:%[0-9]+]]:sreg_32 = S_MOV_B32 0
   ; CHECK45-NEXT:   [[S_MOV_B:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO -6629298651489370112
   ; CHECK45-NEXT:   [[S_OR_B64_:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE]], [[S_MOV_B]], implicit-def dead $scc
-  ; CHECK45-NEXT:   [[S_MOV_B1:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO 33554441
   ; CHECK45-NEXT:   [[S_MOV_B32_1:%[0-9]+]]:sreg_32 = S_MOV_B32 -536870912
   ; CHECK45-NEXT:   [[REG_SEQUENCE1:%[0-9]+]]:sreg_64 = REG_SEQUENCE [[S_MOV_B32_]], %subreg.sub0, [[S_MOV_B32_1]], %subreg.sub1
+  ; CHECK45-NEXT:   [[S_MOV_B1:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO 33554441
   ; CHECK45-NEXT:   [[S_OR_B64_1:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE1]], [[S_MOV_B1]], implicit-def dead $scc
   ; CHECK45-NEXT:   [[COPY2:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub0
   ; CHECK45-NEXT:   [[COPY3:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
index 83e6d007f299a..1f10aaf81646a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
@@ -1035,22 +1035,22 @@ define <2 x float> @v_lshr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
 ; GFX6-LABEL: v_lshr_v4i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v4, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v6, 16, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v5, 16, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v7, 16, v3
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v0
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v4, v5
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v0, v2, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v6, v4
-; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
+; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v1
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v1, v1, 16, 16
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v1, v3, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v3, v7, v5
-; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
-; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v3
-; GFX6-NEXT:    v_or_b32_e32 v1, v1, v2
+; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v2, v5
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT:    v_or_b32_e32 v0, v4, v0
+; GFX6-NEXT:    v_or_b32_e32 v1, v2, v1
 ; GFX6-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-LABEL: v_lshr_v4i16:
@@ -1085,20 +1085,20 @@ define <2 x float> @v_lshr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
 define amdgpu_ps <2 x i32> @s_lshr_v4i16(<4 x i16> inreg %value, <4 x i16> inreg %amount) {
 ; GFX6-LABEL: s_lshr_v4i16:
 ; GFX6:       ; %bb.0:
-; GFX6-NEXT:    s_lshr_b32 s4, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s6, s2, 16
-; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT:    s_lshr_b32 s5, s1, 16
-; GFX6-NEXT:    s_lshr_b32 s7, s3, 16
+; GFX6-NEXT:    s_and_b32 s4, s0, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s4, s4, s2
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s0, s0, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s0, s0, s2
-; GFX6-NEXT:    s_lshr_b32 s2, s4, s6
-; GFX6-NEXT:    s_and_b32 s1, s1, 0xffff
+; GFX6-NEXT:    s_and_b32 s2, s1, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s2, s2, s3
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s1, s1, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s1, s1, s3
-; GFX6-NEXT:    s_lshr_b32 s3, s5, s7
-; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
-; GFX6-NEXT:    s_or_b32 s0, s0, s2
-; GFX6-NEXT:    s_lshl_b32 s2, s3, 16
-; GFX6-NEXT:    s_or_b32 s1, s1, s2
+; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
+; GFX6-NEXT:    s_lshl_b32 s1, s1, 16
+; GFX6-NEXT:    s_or_b32 s0, s4, s0
+; GFX6-NEXT:    s_or_b32 s1, s2, s1
 ; GFX6-NEXT:    ; return to shader part epilog
 ;
 ; GFX8-LABEL: s_lshr_v4i16:
@@ -1182,38 +1182,38 @@ define <4 x float> @v_lshr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
 ; GFX6-LABEL: v_lshr_v8i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v12, 16, v4
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v13, 16, v5
+; GFX6-NEXT:    v_and_b32_e32 v8, 0xffff, v4
+; GFX6-NEXT:    v_and_b32_e32 v9, 0xffff, v0
+; GFX6-NEXT:    v_bfe_u32 v4, v4, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX6-NEXT:    v_lshrrev_b32_e32 v8, v8, v9
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v0, v4, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v12, v8
-; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v5
-; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v10, 16, v2
-; GFX6-NEXT:    v_lshrrev_b32_e32 v14, 16, v6
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT:    v_and_b32_e32 v9, 0xffff, v1
+; GFX6-NEXT:    v_bfe_u32 v5, v5, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX6-NEXT:    v_lshrrev_b32_e32 v4, v4, v9
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v1, v5, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v5, v13, v9
-; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v6
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
-; GFX6-NEXT:    v_lshrrev_b32_e32 v11, 16, v3
-; GFX6-NEXT:    v_lshrrev_b32_e32 v15, 16, v7
+; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v6
+; GFX6-NEXT:    v_and_b32_e32 v9, 0xffff, v2
+; GFX6-NEXT:    v_bfe_u32 v6, v6, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT:    v_lshrrev_b32_e32 v5, v5, v9
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v2, v6, v2
-; GFX6-NEXT:    v_lshrrev_b32_e32 v6, v14, v10
-; GFX6-NEXT:    v_and_b32_e32 v7, 0xffff, v7
-; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v5
+; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v7
+; GFX6-NEXT:    v_and_b32_e32 v9, 0xffff, v3
+; GFX6-NEXT:    v_bfe_u32 v7, v7, 16, 16
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v3, v7, v3
-; GFX6-NEXT:    v_lshrrev_b32_e32 v7, v15, v11
-; GFX6-NEXT:    v_or_b32_e32 v1, v1, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v6
-; GFX6-NEXT:    v_or_b32_e32 v2, v2, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v7
-; GFX6-NEXT:    v_or_b32_e32 v3, v3, v4
+; GFX6-NEXT:    v_lshrrev_b32_e32 v6, v6, v9
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
+; GFX6-NEXT:    v_lshlrev_b32_e32 v3, 16, v3
+; GFX6-NEXT:    v_or_b32_e32 v0, v8, v0
+; GFX6-NEXT:    v_or_b32_e32 v1, v4, v1
+; GFX6-NEXT:    v_or_b32_e32 v2, v5, v2
+; GFX6-NEXT:    v_or_b32_e32 v3, v6, v3
 ; GFX6-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-LABEL: v_lshr_v8i16:
@@ -1258,34 +1258,34 @@ define <4 x float> @v_lshr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
 define amdgpu_ps <4 x i32> @s_lshr_v8i16(<8 x i16> inreg %value, <8 x i16> inreg %amount) {
 ; GFX6-LABEL: s_lshr_v8i16:
 ; GFX6:       ; %bb.0:
-; GFX6-NEXT:    s_lshr_b32 s8, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s12, s4, 16
-; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT:    s_lshr_b32 s9, s1, 16
-; GFX6-NEXT:    s_lshr_b32 s13, s5, 16
+; GFX6-NEXT:    s_and_b32 s8, s0, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s8, s8, s4
+; GFX6-NEXT:    s_bfe_u32 s4, s4, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s0, s0, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s0, s0, s4
-; GFX6-NEXT:    s_lshr_b32 s4, s8, s12
-; GFX6-NEXT:    s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT:    s_lshr_b32 s10, s2, 16
-; GFX6-NEXT:    s_lshr_b32 s14, s6, 16
+; GFX6-NEXT:    s_and_b32 s4, s1, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s4, s4, s5
+; GFX6-NEXT:    s_bfe_u32 s5, s5, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s1, s1, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s1, s1, s5
-; GFX6-NEXT:    s_lshr_b32 s5, s9, s13
-; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT:    s_lshl_b32 s4, s4, 16
-; GFX6-NEXT:    s_lshr_b32 s11, s3, 16
-; GFX6-NEXT:    s_lshr_b32 s15, s7, 16
+; GFX6-NEXT:    s_and_b32 s5, s2, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s5, s5, s6
+; GFX6-NEXT:    s_bfe_u32 s6, s6, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s2, s2, s6
-; GFX6-NEXT:    s_lshr_b32 s6, s10, s14
-; GFX6-NEXT:    s_and_b32 s3, s3, 0xffff
-; GFX6-NEXT:    s_or_b32 s0, s0, s4
-; GFX6-NEXT:    s_lshl_b32 s4, s5, 16
+; GFX6-NEXT:    s_and_b32 s6, s3, 0xffff
+; GFX6-NEXT:    s_lshr_b32 s6, s6, s7
+; GFX6-NEXT:    s_bfe_u32 s7, s7, 0x100010
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
 ; GFX6-NEXT:    s_lshr_b32 s3, s3, s7
-; GFX6-NEXT:    s_lshr_b32 s7, s11, s15
-; GFX6-NEXT:    s_or_b32 s1, s1, s4
-; GFX6-NEXT:    s_lshl_b32 s4, s6, 16
-; GFX6-NEXT:    s_or_b32 s2, s2, s4
-; GFX6-NEXT:    s_lshl_b32 s4, s7, 16
-; GFX6-NEXT:    s_or_b32 s3, s3, s4
+; GFX6-NEXT:    s_lshl_b32 s0, s0, 16
+; GFX6-NEXT:    s_lshl_b32 s1, s1, 16
+; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
+; GFX6-NEXT:    s_lshl_b32 s3, s3, 16
+; GFX6-NEXT:    s_or_b32 s0, s8, s0
+; GFX6-NEXT:    s_or_b32 s1, s4, s1
+; GFX6-NEXT:    s_or_b32 s2, s5, s2
+; GFX6-NEXT:    s_or_b32 s3, s6, s3
 ; GFX6-NEXT:    ; return to shader part epilog
 ;
 ; GFX8-LABEL: s_lshr_v8i16:
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
index dfa76d8d0acfa..fc1be777b09a3 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
@@ -28,20 +28,17 @@ define amdgpu_kernel void @store_i136_divergent(ptr %p) {
 ; GFX1150-LABEL: store_i136_divergent:
 ; GFX1150:       ; %bb.0:
 ; GFX1150-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0
-; GFX1150-NEXT:    s_pack_ll_b32_b16 s2, 0, 0
+; GFX1150-NEXT:    v_dual_mov_b32 v3, 0 :: v_dual_and_b32 v0, 0x3ff, v0
+; GFX1150-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX1150-NEXT:    v_mov_b32_e32 v6, 0
-; GFX1150-NEXT:    s_mov_b32 s3, s2
-; GFX1150-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1150-NEXT:    v_dual_mov_b32 v3, s3 :: v_dual_and_b32 v0, 0x3ff, v0
-; GFX1150-NEXT:    v_mov_b32_e32 v2, s2
+; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
 ; GFX1150-NEXT:    v_lshrrev_b32_e32 v1, 8, v0
 ; GFX1150-NEXT:    v_mov_b16_e32 v0.h, 0
 ; GFX1150-NEXT:    v_and_b16 v0.l, 0xff, v0.l
-; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
 ; GFX1150-NEXT:    v_lshlrev_b16 v4.l, 8, v1.l
+; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
 ; GFX1150-NEXT:    v_mov_b16_e32 v1.l, v0.h
 ; GFX1150-NEXT:    v_mov_b16_e32 v1.h, v0.h
-; GFX1150-NEXT:    s_delay_alu instid0(VALU_DEP_3)
 ; GFX1150-NEXT:    v_or_b16 v0.l, v0.l, v4.l
 ; GFX1150-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX1150-NEXT:    v_dual_mov_b32 v5, s1 :: v_dual_mov_b32 v4, s0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
index d1621dbc267da..b41e1a1374480 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
@@ -201,13 +201,16 @@ define amdgpu_kernel void @v_mul_i64_masked_src0_hi(ptr addrspace(1) %out, ptr a
 ; GFX10-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x24
 ; GFX10-NEXT:    s_load_dwordx2 s[6:7], s[4:5], 0x34
 ; GFX10-NEXT:    v_lshlrev_b32_e32 v0, 3, v0
+; GFX10-NEXT:    ; kill: killed $vgpr0
+; GFX10-NEXT:    ; kill: killed $sgpr2_sgpr3
 ; GFX10-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX10-NEXT:    s_clause 0x1
-; GFX10-NEXT:    global_load_dword v4, v0, s[2:3]
-; GFX10-NEXT:    global_load_dwordx2 v[2:3], v0, s[6:7]
+; GFX10-NEXT:    global_load_dwordx2 v[2:3], v0, s[2:3]
+; GFX10-NEXT:    global_load_dwordx2 v[3:4], v0, s[6:7]
+; GFX10-NEXT:    ; kill: killed $sgpr6_sgpr7
 ; GFX10-NEXT:    s_waitcnt vmcnt(0)
-; GFX10-NEXT:    v_mad_u64_u32 v[0:1], s2, v4, v2, 0
-; GFX10-NEXT:    v_mad_u64_u32 v[1:2], s2, v4, v3, v[1:2]
+; GFX10-NEXT:    v_mad_u64_u32 v[0:1], s2, v2, v3, 0
+; GFX10-NEXT:    v_mad_u64_u32 v[1:2], s2, v2, v4, v[1:2]
 ; GFX10-NEXT:    v_mov_b32_e32 v2, 0
 ; GFX10-NEXT:    global_store_dwordx2 v2, v[0:1], s[0:1]
 ; GFX10-NEXT:    s_endpgm
@@ -222,13 +225,13 @@ define amdgpu_kernel void @v_mul_i64_masked_src0_hi(ptr addrspace(1) %out, ptr a
 ; GFX11-NEXT:    v_lshlrev_b32_e32 v0, 3, v0
 ; GFX11-NEXT:    s_waitcnt lgkmcnt(0)
 ; GFX11-NEXT:    s_clause 0x1
-; GFX11-NEXT:    global_load_b32 v6, v0, s[2:3]
-; GFX11-NEXT:    global_load_b64 v[2:3], v0, s[4:5]
+; GFX11-NEXT:    global_load_b64 v[2:3], v0, s[2:3]
+; GFX11-NEXT:    global_load_b64 v[3:4], v0, s[4:5]
 ; GFX11-NEXT:    s_waitcnt vmcnt(0)
-; GFX11-NEXT:    v_mad_u64_u32 v[0:1], null, v6, v2, 0
+; GFX11-NEXT:    v_mad_u64_u32 v[0:1], null, v2, v3, 0
 ; GFX11-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-NEXT:    v_mad_u64_u32 v[4:5], null, v6, v3, v[1:2]
-; GFX11-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, v4
+; GFX11-NEXT:    v_mad_u64_u32 v[5:6], null, v2, v4, v[1:2]
+; GFX11-NEXT:    v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, v5
 ; GFX11-NEXT:    global_store_b64 v2, v[0:1], s[0:1]
 ; GFX11-NEXT:    s_endpgm
  %tid = call i32 @llvm.amdgcn.workitem.id.x()
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
index e4035042e9ca0..efb8fc0ed4c4c 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
@@ -980,18 +980,18 @@ define <2 x float> @v_shl_v4i16(<4 x i16> %value, <4 x i16> %amount) {
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v4, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v6, 16, v2
-; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v2, v0
-; GFX6-NEXT:    v_lshlrev_b32_e32 v2, v6, v4
+; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v2
+; GFX6-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT:    v_lshlrev_b32_e32 v2, v2, v4
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v5, 16, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v7, 16, v3
-; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v6, v0
+; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v3
+; GFX6-NEXT:    v_bfe_u32 v3, v3, 16, 16
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v1, v3, v1
-; GFX6-NEXT:    v_lshlrev_b32_e32 v3, v7, v5
+; GFX6-NEXT:    v_lshlrev_b32_e32 v3, v3, v5
 ; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v2, 16, v2
+; GFX6-NEXT:    v_lshlrev_b32_e32 v1, v4, v1
 ; GFX6-NEXT:    v_or_b32_e32 v0, v0, v2
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v3
 ; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
@@ -1032,14 +1032,14 @@ define amdgpu_ps <2 x i32> @s_shl_v4i16(<4 x i16> inreg %value, <4 x i16> inreg
 ; GFX6-LABEL: s_shl_v4i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_lshr_b32 s4, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s6, s2, 16
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s2
-; GFX6-NEXT:    s_lshl_b32 s2, s4, s6
+; GFX6-NEXT:    s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT:    s_lshl_b32 s2, s4, s2
 ; GFX6-NEXT:    s_lshr_b32 s5, s1, 16
-; GFX6-NEXT:    s_lshr_b32 s7, s3, 16
-; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, s3
-; GFX6-NEXT:    s_lshl_b32 s3, s5, s7
+; GFX6-NEXT:    s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
+; GFX6-NEXT:    s_lshl_b32 s3, s5, s3
 ; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s2, s2, 16
 ; GFX6-NEXT:    s_or_b32 s0, s0, s2
@@ -1129,36 +1129,36 @@ define <4 x float> @v_shl_v8i16(<8 x i16> %value, <8 x i16> %amount) {
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v8, 16, v0
-; GFX6-NEXT:    v_lshrrev_b32_e32 v12, 16, v4
-; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v12, v8
+; GFX6-NEXT:    v_and_b32_e32 v12, 0xffff, v4
+; GFX6-NEXT:    v_bfe_u32 v4, v4, 16, 16
+; GFX6-NEXT:    v_lshlrev_b32_e32 v4, v4, v8
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v9, 16, v1
-; GFX6-NEXT:    v_lshrrev_b32_e32 v13, 16, v5
-; GFX6-NEXT:    v_and_b32_e32 v5, 0xffff, v5
+; GFX6-NEXT:    v_lshlrev_b32_e32 v0, v12, v0
+; GFX6-NEXT:    v_and_b32_e32 v8, 0xffff, v5
+; GFX6-NEXT:    v_bfe_u32 v5, v5, 16, 16
 ; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT:    v_lshlrev_b32_e32 v1, v5, v1
-; GFX6-NEXT:    v_lshlrev_b32_e32 v5, v13, v9
+; GFX6-NEXT:    v_lshlrev_b32_e32 v5, v5, v9
 ; GFX6-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v10, 16, v2
-; GFX6-NEXT:    v_lshrrev_b32_e32 v14, 16, v6
-; GFX6-NEXT:    v_and_b32_e32 v6, 0xffff, v6
+; GFX6-NEXT:    v_lshlrev_b32_e32 v1, v8, v1
+; GFX6-NEXT:    v_and_b32_e32 v8, 0xffff, v6
+; GFX6-NEXT:    v_bfe_u32 v6, v6, 16, 16
 ; GFX6-NEXT:    v_or_b32_e32 v0, v0, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v5
-; GFX6-NEXT:    v_lshlrev_b32_e32 v2, v6, v2
-; GFX6-NEXT:    v_lshlrev_b32_e32 v6, v14, v10
+; GFX6-NEXT:    v_lshlrev_b32_e32 v6, v6, v10
 ; GFX6-NEXT:    v_and_b32_e32 v1, 0xffff, v1
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
 ; GFX6-NEXT:    v_lshrrev_b32_e32 v11, 16, v3
-; GFX6-NEXT:    v_lshrrev_b32_e32 v15, 16, v7
-; GFX6-NEXT:    v_and_b32_e32 v7, 0xffff, v7
+; GFX6-NEXT:    v_lshlrev_b32_e32 v2, v8, v2
+; GFX6-NEXT:    v_and_b32_e32 v8, 0xffff, v7
+; GFX6-NEXT:    v_bfe_u32 v7, v7, 16, 16
 ; GFX6-NEXT:    v_or_b32_e32 v1, v1, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v6
-; GFX6-NEXT:    v_lshlrev_b32_e32 v3, v7, v3
-; GFX6-NEXT:    v_lshlrev_b32_e32 v7, v15, v11
+; GFX6-NEXT:    v_lshlrev_b32_e32 v7, v7, v11
 ; GFX6-NEXT:    v_and_b32_e32 v2, 0xffff, v2
 ; GFX6-NEXT:    v_lshlrev_b32_e32 v4, 16, v4
+; GFX6-NEXT:    v_lshlrev_b32_e32 v3, v8, v3
 ; GFX6-NEXT:    v_or_b32_e32 v2, v2, v4
 ; GFX6-NEXT:    v_and_b32_e32 v4, 0xffff, v7
 ; GFX6-NEXT:    v_and_b32_e32 v3, 0xffff, v3
@@ -1209,30 +1209,30 @@ define amdgpu_ps <4 x i32> @s_shl_v8i16(<8 x i16> inreg %value, <8 x i16> inreg
 ; GFX6-LABEL: s_shl_v8i16:
 ; GFX6:       ; %bb.0:
 ; GFX6-NEXT:    s_lshr_b32 s8, s0, 16
-; GFX6-NEXT:    s_lshr_b32 s12, s4, 16
 ; GFX6-NEXT:    s_lshl_b32 s0, s0, s4
-; GFX6-NEXT:    s_lshl_b32 s4, s8, s12
+; GFX6-NEXT:    s_bfe_u32 s4, s4, 0x100010
+; GFX6-NEXT:    s_lshl_b32 s4, s8, s4
 ; GFX6-NEXT:    s_lshr_b32 s9, s1, 16
-; GFX6-NEXT:    s_lshr_b32 s13, s5, 16
-; GFX6-NEXT:    s_and_b32 s4, s4, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s1, s1, s5
-; GFX6-NEXT:    s_lshl_b32 s5, s9, s13
+; GFX6-NEXT:    s_bfe_u32 s5, s5, 0x100010
+; GFX6-NEXT:    s_and_b32 s4, s4, 0xffff
+; GFX6-NEXT:    s_lshl_b32 s5, s9, s5
 ; GFX6-NEXT:    s_and_b32 s0, s0, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s4, s4, 16
 ; GFX6-NEXT:    s_lshr_b32 s10, s2, 16
-; GFX6-NEXT:    s_lshr_b32 s14, s6, 16
+; GFX6-NEXT:    s_lshl_b32 s2, s2, s6
+; GFX6-NEXT:    s_bfe_u32 s6, s6, 0x100010
 ; GFX6-NEXT:    s_or_b32 s0, s0, s4
 ; GFX6-NEXT:    s_and_b32 s4, s5, 0xffff
-; GFX6-NEXT:    s_lshl_b32 s2, s2, s6
-; GFX6-NEXT:    s_lshl_b32 s6, s10, s14
+; GFX6-NEXT:    s_lshl_b32 s6, s10, s6
 ; GFX6-NEXT:    s_and_b32 s1, s1, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s4, s4, 16
 ; GFX6-NEXT:    s_lshr_b32 s11, s3, 16
-; GFX6-NEXT:    s_lshr_b32 s15, s7, 16
+; GFX6-NEXT:    s_lshl_b32 s3, s3, s7
+; GFX6-NEXT:    s_bfe_u32 s7, s7, 0x100010
 ; GFX6-NEXT:    s_or_b32 s1, s1, s4
 ; GFX6-NEXT:    s_and_b32 s4, s6, 0xffff
-; GFX6-NEXT:    s_lshl_b32 s3, s3, s7
-; GFX6-NEXT:    s_lshl_b32 s7, s11, s15
+; GFX6-NEXT:    s_lshl_b32 s7, s11, s7
 ; GFX6-NEXT:    s_and_b32 s2, s2, 0xffff
 ; GFX6-NEXT:    s_lshl_b32 s4, s4, 16
 ; GFX6-NEXT:    s_or_b32 s2, s2, s4
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
index cbbd5e69bff12..467f02fe970ea 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
@@ -492,7 +492,7 @@ define amdgpu_kernel void @test_cluster_id_z(ptr addrspace(1) %out) #1 {
 ; CHECK-G-UNKNOWN-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; CHECK-G-UNKNOWN-NEXT:    s_load_b64 s[2:3], s[0:1], 0x24 nv
 ; CHECK-G-UNKNOWN-NEXT:    s_wait_xcnt 0x0
-; CHECK-G-UNKNOWN-NEXT:    s_lshr_b32 s0, ttmp7, 16
+; CHECK-G-UNKNOWN-NEXT:    s_bfe_u32 s0, ttmp7, 0x100010
 ; CHECK-G-UNKNOWN-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
 ; CHECK-G-UNKNOWN-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s0
 ; CHECK-G-UNKNOWN-NEXT:    s_wait_kmcnt 0x0
@@ -574,7 +574,7 @@ define amdgpu_kernel void @test_cluster_id_z(ptr addrspace(1) %out) #1 {
 ; CHECK-G-MESA3D-NEXT:    v_nop
 ; CHECK-G-MESA3D-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; CHECK-G-MESA3D-NEXT:    s_load_b64 s[0:1], s[0:1], 0x0 nv
-; CHECK-G-MESA3D-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; CHECK-G-MESA3D-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; CHECK-G-MESA3D-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
 ; CHECK-G-MESA3D-NEXT:    v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
 ; CHECK-G-MESA3D-NEXT:    s_wait_kmcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
index 173739fa8f35f..70209924cd077 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
@@ -228,77 +228,43 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16(i32 inreg %node_ptr, f
 ; GFX10-SDAG-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-SDAG-NEXT:    ; return to shader part epilog
 ;
-; GFX1013-GISEL-LABEL: image_bvh_intersect_ray_a16:
-; GFX1013-GISEL:       ; %bb.0: ; %main_body
-; GFX1013-GISEL-NEXT:    s_and_b32 s8, s8, 0xffff
-; GFX1013-GISEL-NEXT:    s_mov_b32 s16, s9
-; GFX1013-GISEL-NEXT:    v_alignbit_b32 v0, s8, s7, 16
-; GFX1013-GISEL-NEXT:    s_lshr_b32 s9, s5, 16
-; GFX1013-GISEL-NEXT:    s_and_b32 s5, s5, 0xffff
-; GFX1013-GISEL-NEXT:    s_lshl_b32 s8, s9, 16
-; GFX1013-GISEL-NEXT:    s_and_b32 s9, s7, 0xffff
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s7, v0
-; GFX1013-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
-; GFX1013-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
-; GFX1013-GISEL-NEXT:    s_or_b32 s5, s5, s8
-; GFX1013-GISEL-NEXT:    s_or_b32 s6, s6, s9
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v4, s4
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v5, s5
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v6, s6
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v7, s7
-; GFX1013-GISEL-NEXT:    s_mov_b32 s17, s10
-; GFX1013-GISEL-NEXT:    s_mov_b32 s18, s11
-; GFX1013-GISEL-NEXT:    s_mov_b32 s19, s12
-; GFX1013-GISEL-NEXT:    image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
-; GFX1013-GISEL-NEXT:    s_waitcnt vmcnt(0)
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT:    ; return to shader part epilog
-;
-; GFX1030-GISEL-LABEL: image_bvh_intersect_ray_a16:
-; GFX1030-GISEL:       ; %bb.0: ; %main_body
-; GFX1030-GISEL-NEXT:    s_mov_b32 s16, s9
-; GFX1030-GISEL-NEXT:    s_lshr_b32 s9, s5, 16
-; GFX1030-GISEL-NEXT:    s_and_b32 s5, s5, 0xffff
-; GFX1030-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
-; GFX1030-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
-; GFX1030-GISEL-NEXT:    s_or_b32 s5, s5, s9
-; GFX1030-GISEL-NEXT:    s_and_b32 s9, s7, 0xffff
-; GFX1030-GISEL-NEXT:    s_and_b32 s8, s8, 0xffff
-; GFX1030-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
-; GFX1030-GISEL-NEXT:    v_alignbit_b32 v7, s8, s7, 16
-; GFX1030-GISEL-NEXT:    s_or_b32 s6, s6, s9
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v4, s4
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v5, s5
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v6, s6
-; GFX1030-GISEL-NEXT:    s_mov_b32 s17, s10
-; GFX1030-GISEL-NEXT:    s_mov_b32 s18, s11
-; GFX1030-GISEL-NEXT:    s_mov_b32 s19, s12
-; GFX1030-GISEL-NEXT:    image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
-; GFX1030-GISEL-NEXT:    s_waitcnt vmcnt(0)
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT:    ; return to shader part epilog
+; GFX10-GISEL-LABEL: image_bvh_intersect_ray_a16:
+; GFX10-GISEL:       ; %bb.0: ; %main_body
+; GFX10-GISEL-NEXT:    s_mov_b32 s16, s9
+; GFX10-GISEL-NEXT:    s_bfe_u32 s9, s5, 0x100010
+; GFX10-GISEL-NEXT:    s_and_b32 s5, s5, 0xffff
+; GFX10-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT:    s_and_b32 s8, s8, 0xffff
+; GFX10-GISEL-NEXT:    s_or_b32 s5, s5, s9
+; GFX10-GISEL-NEXT:    s_and_b32 s9, s7, 0xffff
+; GFX10-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
+; GFX10-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT:    s_bfe_u32 s7, s7, 0x100010
+; GFX10-GISEL-NEXT:    s_lshl_b32 s8, s8, 16
+; GFX10-GISEL-NEXT:    s_or_b32 s6, s6, s9
+; GFX10-GISEL-NEXT:    s_or_b32 s7, s7, s8
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v4, s4
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v5, s5
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v6, s6
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v7, s7
+; GFX10-GISEL-NEXT:    s_mov_b32 s17, s10
+; GFX10-GISEL-NEXT:    s_mov_b32 s18, s11
+; GFX10-GISEL-NEXT:    s_mov_b32 s19, s12
+; GFX10-GISEL-NEXT:    image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
+; GFX10-GISEL-NEXT:    s_waitcnt vmcnt(0)
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT:    ; return to shader part epilog
 ;
 ; GFX11-SDAG-LABEL: image_bvh_intersect_ray_a16:
 ; GFX11-SDAG:       ; %bb.0: ; %main_body
@@ -530,79 +496,44 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16(i64 inreg %node_ptr,
 ; GFX10-SDAG-NEXT:    s_waitcnt vmcnt(0)
 ; GFX10-SDAG-NEXT:    ; return to shader part epilog
 ;
-; GFX1013-GISEL-LABEL: image_bvh64_intersect_ray_a16:
-; GFX1013-GISEL:       ; %bb.0: ; %main_body
-; GFX1013-GISEL-NEXT:    s_and_b32 s9, s9, 0xffff
-; GFX1013-GISEL-NEXT:    s_mov_b32 s16, s10
-; GFX1013-GISEL-NEXT:    v_alignbit_b32 v0, s9, s8, 16
-; GFX1013-GISEL-NEXT:    s_lshr_b32 s10, s6, 16
-; GFX1013-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
-; GFX1013-GISEL-NEXT:    s_lshl_b32 s9, s10, 16
-; GFX1013-GISEL-NEXT:    s_and_b32 s10, s8, 0xffff
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s8, v0
-; GFX1013-GISEL-NEXT:    s_and_b32 s7, s7, 0xffff
-; GFX1013-GISEL-NEXT:    s_lshl_b32 s10, s10, 16
-; GFX1013-GISEL-NEXT:    s_or_b32 s6, s6, s9
-; GFX1013-GISEL-NEXT:    s_or_b32 s7, s7, s10
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v4, s4
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v5, s5
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v6, s6
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v7, s7
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v8, s8
-; GFX1013-GISEL-NEXT:    s_mov_b32 s17, s11
-; GFX1013-GISEL-NEXT:    s_mov_b32 s18, s12
-; GFX1013-GISEL-NEXT:    s_mov_b32 s19, s13
-; GFX1013-GISEL-NEXT:    image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
-; GFX1013-GISEL-NEXT:    s_waitcnt vmcnt(0)
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
-; GFX1013-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT:    ; return to shader part epilog
-;
-; GFX1030-GISEL-LABEL: image_bvh64_intersect_ray_a16:
-; GFX1030-GISEL:       ; %bb.0: ; %main_body
-; GFX1030-GISEL-NEXT:    s_mov_b32 s16, s10
-; GFX1030-GISEL-NEXT:    s_lshr_b32 s10, s6, 16
-; GFX1030-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
-; GFX1030-GISEL-NEXT:    s_lshl_b32 s10, s10, 16
-; GFX1030-GISEL-NEXT:    s_and_b32 s7, s7, 0xffff
-; GFX1030-GISEL-NEXT:    s_or_b32 s6, s6, s10
-; GFX1030-GISEL-NEXT:    s_and_b32 s10, s8, 0xffff
-; GFX1030-GISEL-NEXT:    s_and_b32 s9, s9, 0xffff
-; GFX1030-GISEL-NEXT:    s_lshl_b32 s10, s10, 16
-; GFX1030-GISEL-NEXT:    v_alignbit_b32 v8, s9, s8, 16
-; GFX1030-GISEL-NEXT:    s_or_b32 s7, s7, s10
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v4, s4
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v5, s5
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v6, s6
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v7, s7
-; GFX1030-GISEL-NEXT:    s_mov_b32 s17, s11
-; GFX1030-GISEL-NEXT:    s_mov_b32 s18, s12
-; GFX1030-GISEL-NEXT:    s_mov_b32 s19, s13
-; GFX1030-GISEL-NEXT:    image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
-; GFX1030-GISEL-NEXT:    s_waitcnt vmcnt(0)
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
-; GFX1030-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT:    v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT:    ; return to shader part epilog
+; GFX10-GISEL-LABEL: image_bvh64_intersect_ray_a16:
+; GFX10-GISEL:       ; %bb.0: ; %main_body
+; GFX10-GISEL-NEXT:    s_mov_b32 s16, s10
+; GFX10-GISEL-NEXT:    s_bfe_u32 s10, s6, 0x100010
+; GFX10-GISEL-NEXT:    s_and_b32 s6, s6, 0xffff
+; GFX10-GISEL-NEXT:    s_lshl_b32 s10, s10, 16
+; GFX10-GISEL-NEXT:    s_and_b32 s9, s9, 0xffff
+; GFX10-GISEL-NEXT:    s_or_b32 s6, s6, s10
+; GFX10-GISEL-NEXT:    s_and_b32 s10, s8, 0xffff
+; GFX10-GISEL-NEXT:    s_and_b32 s7, s7, 0xffff
+; GFX10-GISEL-NEXT:    s_lshl_b32 s10, s10, 16
+; GFX10-GISEL-NEXT:    s_bfe_u32 s8, s8, 0x100010
+; GFX10-GISEL-NEXT:    s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT:    s_or_b32 s7, s7, s10
+; GFX10-GISEL-NEXT:    s_or_b32 s8, s8, s9
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v4, s4
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v5, s5
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v6, s6
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v7, s7
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v8, s8
+; GFX10-GISEL-NEXT:    s_mov_b32 s17, s11
+; GFX10-GISEL-NEXT:    s_mov_b32 s18, s12
+; GFX10-GISEL-NEXT:    s_mov_b32 s19, s13
+; GFX10-GISEL-NEXT:    image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
+; GFX10-GISEL-NEXT:    s_waitcnt vmcnt(0)
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s0, v0
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s1, v1
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s2, v2
+; GFX10-GISEL-NEXT:    v_readfirstlane_b32 s3, v3
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT:    v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT:    ; return to shader part epilog
 ;
 ; GFX11-SDAG-LABEL: image_bvh64_intersect_ray_a16:
 ; GFX11-SDAG:       ; %bb.0: ; %main_body
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
index 44ff13dd81303..2c25f250d8439 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
@@ -29,7 +29,7 @@ define amdgpu_kernel void @workgroup_ids_kernel() {
 ; GFX9ARCH-GISEL:       ; %bb.0: ; %.entry
 ; GFX9ARCH-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX9ARCH-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v1, s1
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v2, s2
@@ -49,7 +49,7 @@ define amdgpu_kernel void @workgroup_ids_kernel() {
 ; GFX12-GISEL:       ; %bb.0: ; %.entry
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX12-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX12-GISEL-NEXT:    v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
 ; GFX12-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX12-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -251,7 +251,7 @@ define void @workgroup_ids_device_func(ptr addrspace(1) %outx, ptr addrspace(1)
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v6, ttmp9
 ; GFX9ARCH-GISEL-NEXT:    s_and_b32 s4, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT:    s_lshr_b32 s5, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT:    s_bfe_u32 s5, ttmp7, 0x100010
 ; GFX9ARCH-GISEL-NEXT:    global_store_dword v[0:1], v6, off
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v0, s4
@@ -262,27 +262,49 @@ define void @workgroup_ids_device_func(ptr addrspace(1) %outx, ptr addrspace(1)
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
-; GFX12-LABEL: workgroup_ids_device_func:
-; GFX12:       ; %bb.0:
-; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT:    s_wait_expcnt 0x0
-; GFX12-NEXT:    s_wait_samplecnt 0x0
-; GFX12-NEXT:    s_wait_bvhcnt 0x0
-; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
-; GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_mov_b32_e32 v8, s1
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    s_setpc_b64 s[30:31]
+; GFX12-SDAG-LABEL: workgroup_ids_device_func:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-SDAG-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v8, s1
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: workgroup_ids_device_func:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-GISEL-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v8, s1
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %id.x = call i32 @llvm.amdgcn.workgroup.id.x()
   %id.y = call i32 @llvm.amdgcn.workgroup.id.y()
   %id.z = call i32 @llvm.amdgcn.workgroup.id.z()
@@ -299,4 +321,5 @@ declare void @llvm.amdgcn.raw.ptr.buffer.store.v3i32(<3 x i32>, ptr addrspace(8)
 
 attributes #0 = { nounwind "amdgpu-no-workgroup-id-y" "amdgpu-no-cluster-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-cluster-id-z" }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
 ; GFX9ARCH: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
index 63d02e09d611e..cb2539b75438d 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
@@ -275,7 +275,7 @@ define void @test_workgroup_id_z_non_kernel(ptr addrspace(1) %out) {
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s0, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_add_co_i32 s0, s0, 1
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s2, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_mul_i32 s0, s1, s0
@@ -314,7 +314,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_used(ptr addrspace(1) %out
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s0, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_add_co_i32 s0, s0, 1
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s2, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_mul_i32 s1, s1, s0
@@ -343,7 +343,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_not_used(ptr addrspace(1)
 ; GFX1250-GISEL:       ; %bb.0:
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s0, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s0, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
 ; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v2, s0
 ; GFX1250-GISEL-NEXT:    global_store_b32 v[0:1], v2, off
@@ -371,7 +371,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_fixed(ptr addrspace(1) %ou
 ; GFX1250-GISEL:       ; %bb.0:
 ; GFX1250-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
 ; GFX1250-GISEL-NEXT:    s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s0, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s0, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s1, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
 ; GFX1250-GISEL-NEXT:    s_lshl1_add_u32 s0, s0, s1
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
index b2ee9119fbcad..e3c4cab6fa2f0 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
@@ -36,7 +36,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
 ; GFX9ARCH-GISEL:       ; %bb.0: ; %.entry
 ; GFX9ARCH-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX9ARCH-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v1, s1
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v2, s2
@@ -56,7 +56,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
 ; GFX12-GISEL:       ; %bb.0: ; %.entry
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX12-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX12-GISEL-NEXT:    v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
 ; GFX12-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX12-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -203,7 +203,7 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v6, ttmp9
 ; GFX9ARCH-GISEL-NEXT:    s_and_b32 s34, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT:    s_lshr_b32 s35, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT:    s_bfe_u32 s35, ttmp7, 0x100010
 ; GFX9ARCH-GISEL-NEXT:    global_store_dword v[0:1], v6, off
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    v_mov_b32_e32 v0, s34
@@ -214,27 +214,49 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
 ; GFX9ARCH-GISEL-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9ARCH-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
-; GFX12-LABEL: workgroup_ids_gfx:
-; GFX12:       ; %bb.0:
-; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT:    s_wait_expcnt 0x0
-; GFX12-NEXT:    s_wait_samplecnt 0x0
-; GFX12-NEXT:    s_wait_bvhcnt 0x0
-; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
-; GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_mov_b32_e32 v8, s1
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
-; GFX12-NEXT:    s_wait_storecnt 0x0
-; GFX12-NEXT:    s_setpc_b64 s[30:31]
+; GFX12-SDAG-LABEL: workgroup_ids_gfx:
+; GFX12-SDAG:       ; %bb.0:
+; GFX12-SDAG-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT:    s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-SDAG-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GFX12-SDAG-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT:    v_mov_b32_e32 v8, s1
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT:    s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: workgroup_ids_gfx:
+; GFX12-GISEL:       ; %bb.0:
+; GFX12-GISEL-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT:    s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-GISEL-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
+; GFX12-GISEL-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT:    v_mov_b32_e32 v8, s1
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT:    s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
   %id.x = call i32 @llvm.amdgcn.workgroup.id.x()
   %id.y = call i32 @llvm.amdgcn.workgroup.id.y()
   %id.z = call i32 @llvm.amdgcn.workgroup.id.z()
@@ -243,3 +265,5 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
   store volatile i32 %id.z, ptr addrspace(1) %outz
   ret void
 }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
index 3ffe9c76175ae..4d15b36584716 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
@@ -21,7 +21,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
 ; GFX9-GISEL:       ; %bb.0: ; %.entry
 ; GFX9-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX9-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v1, s1
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v2, s2
@@ -41,7 +41,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
 ; GFX12-GISEL:       ; %bb.0: ; %.entry
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX12-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX12-GISEL-NEXT:    v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
 ; GFX12-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX12-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -106,7 +106,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
 ; GFX1250-GISEL-NEXT:    s_cmp_eq_u32 s2, 0
 ; GFX1250-GISEL-NEXT:    s_cselect_b32 s1, s3, s4
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s3, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s4, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s4, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_add_co_i32 s3, s3, 1
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s5, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_mul_i32 s3, s4, s3
@@ -144,7 +144,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
 ; GFX9-GISEL:       ; %bb.0: ; %.entry
 ; GFX9-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX9-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v1, s1
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v2, s2
@@ -164,7 +164,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
 ; GFX12-GISEL:       ; %bb.0: ; %.entry
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX12-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX12-GISEL-NEXT:    v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
 ; GFX12-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX12-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -191,7 +191,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
 ; GFX1250-GISEL-NEXT:    global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
 ; GFX1250-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX1250-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    v_mov_b64_e32 v[0:1], s[0:1]
 ; GFX1250-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX1250-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -222,7 +222,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
 ; GFX9-GISEL:       ; %bb.0: ; %.entry
 ; GFX9-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX9-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v1, s1
 ; GFX9-GISEL-NEXT:    v_mov_b32_e32 v2, s2
@@ -242,7 +242,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
 ; GFX12-GISEL:       ; %bb.0: ; %.entry
 ; GFX12-GISEL-NEXT:    s_mov_b32 s0, ttmp9
 ; GFX12-GISEL-NEXT:    s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT:    s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT:    s_bfe_u32 s2, ttmp7, 0x100010
 ; GFX12-GISEL-NEXT:    v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
 ; GFX12-GISEL-NEXT:    v_mov_b32_e32 v2, s2
 ; GFX12-GISEL-NEXT:    buffer_store_b96 v[0:2], off, s[0:3], null
@@ -280,7 +280,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
 ; GFX1250-GISEL-NEXT:    s_and_b32 s0, ttmp6, 15
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s2, ttmp6, 0x40004
 ; GFX1250-GISEL-NEXT:    s_mul_i32 s1, s1, 3
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s3, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s3, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s4, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_lshl1_add_u32 s0, ttmp9, s0
 ; GFX1250-GISEL-NEXT:    s_add_co_i32 s1, s2, s1
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
index 01b7293dcd7ab..d42b9fcaaccd5 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
@@ -270,7 +270,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v2
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v1, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -289,7 +289,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -308,7 +308,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -329,7 +329,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v4i8:
@@ -371,7 +371,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v1.l, v3.l
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v4i8:
@@ -381,7 +381,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v4i8:
@@ -435,7 +435,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v1.l, v3.l
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v4i8:
@@ -449,7 +449,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.add.v4i8(<4 x i8> %v)
@@ -484,7 +484,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v2
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v1, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -511,7 +511,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -538,7 +538,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -567,7 +567,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v8i8:
@@ -624,7 +624,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v0.h, v1.h
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v8i8:
@@ -639,7 +639,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v8i8:
@@ -708,7 +708,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v0.h, v1.h
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v8i8:
@@ -727,7 +727,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.add.v8i8(<8 x i8> %v)
@@ -778,7 +778,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v2
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v1, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -821,7 +821,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -864,7 +864,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -909,7 +909,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v16i8:
@@ -994,7 +994,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v0.h, v1.h
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1019,7 +1019,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1116,7 +1116,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.h, v0.h, v1.h
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1145,7 +1145,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.add.v16i8(<16 x i8> %v)
@@ -1165,7 +1165,7 @@ define i16 @test_vector_reduce_add_v2i16(<2 x i16> %v) {
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v2i16:
@@ -1463,7 +1463,7 @@ define i16 @test_vector_reduce_add_v4i16(<4 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v2, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v4i16:
@@ -1645,7 +1645,7 @@ define i16 @test_vector_reduce_add_v8i16(<8 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v2, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v8i16:
@@ -1893,7 +1893,7 @@ define i16 @test_vector_reduce_add_v16i16(<16 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v1, vcc, v2, v3
 ; GFX7-GISEL-NEXT:    v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_add_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
index 0ae4b05aa7c89..7bd805dd8c575 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
@@ -297,7 +297,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -315,7 +315,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -333,7 +333,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -351,7 +351,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v4i8:
@@ -381,7 +381,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v4i8:
@@ -423,7 +423,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.and.v4i8(<4 x i8> %v)
@@ -454,7 +454,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -480,7 +480,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -506,7 +506,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -532,7 +532,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v8i8:
@@ -577,7 +577,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v8i8:
@@ -634,7 +634,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.and.v8i8(<8 x i8> %v)
@@ -681,7 +681,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -723,7 +723,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -765,7 +765,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -807,7 +807,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v16i8:
@@ -880,7 +880,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v16i8:
@@ -965,7 +965,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v1, v1, v3
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.and.v16i8(<16 x i8> %v)
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
index b931b312ff251..3a5a705e479f6 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
@@ -328,7 +328,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -346,7 +346,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -364,7 +364,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -382,7 +382,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -412,7 +412,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v1.l, v3.l
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -422,7 +422,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -464,7 +464,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v1.l, v3.l
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -478,7 +478,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.mul.v4i8(<4 x i8> %v)
@@ -509,7 +509,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -535,7 +535,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -561,7 +561,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -587,7 +587,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -632,7 +632,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v0.h, v1.h
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -647,7 +647,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -704,7 +704,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v0.h, v1.h
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -723,7 +723,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.mul.v8i8(<8 x i8> %v)
@@ -770,7 +770,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -812,7 +812,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -854,7 +854,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -896,7 +896,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v2
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -969,7 +969,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v0.h, v1.h
 ; GFX11-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -994,7 +994,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX11-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -1079,7 +1079,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.h, v0.h, v1.h
 ; GFX12-GISEL-TRUE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-TRUE16-NEXT:    v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-TRUE16-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -1108,7 +1108,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v1, v1, v3
 ; GFX12-GISEL-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-FAKE16-NEXT:    v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-FAKE16-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.mul.v16i8(<16 x i8> %v)
@@ -1116,21 +1116,13 @@ entry:
 }
 
 define i16 @test_vector_reduce_mul_v2i16(<2 x i16> %v) {
-; GFX7-SDAG-LABEL: test_vector_reduce_mul_v2i16:
-; GFX7-SDAG:       ; %bb.0: ; %entry
-; GFX7-SDAG-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-SDAG-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
-; GFX7-SDAG-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-SDAG-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-SDAG-NEXT:    s_setpc_b64 s[30:31]
-;
-; GFX7-GISEL-LABEL: test_vector_reduce_mul_v2i16:
-; GFX7-GISEL:       ; %bb.0: ; %entry
-; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
-; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
-; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
+; GFX7-LABEL: test_vector_reduce_mul_v2i16:
+; GFX7:       ; %bb.0: ; %entry
+; GFX7-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX7-NEXT:    v_lshrrev_b32_e32 v1, 16, v0
+; GFX7-NEXT:    v_mul_lo_u32 v0, v0, v1
+; GFX7-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX7-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-LABEL: test_vector_reduce_mul_v2i16:
 ; GFX8:       ; %bb.0: ; %entry
@@ -1404,7 +1396,7 @@ define i16 @test_vector_reduce_mul_v4i16(<4 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v2, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v4i16:
@@ -1564,7 +1556,7 @@ define i16 @test_vector_reduce_mul_v8i16(<8 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v2, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v8i16:
@@ -1807,7 +1799,7 @@ define i16 @test_vector_reduce_mul_v16i16(<16 x i16> %v) {
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v1, v2, v3
 ; GFX7-GISEL-NEXT:    v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_mul_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
index 087832601598a..d1fa2b23282c1 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
@@ -1593,10 +1593,10 @@ define i16 @test_vector_reduce_umax_v3i16(<3 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umax_v3i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_max3_u32 v0, v0, v2, v1
+; GFX7-GISEL-NEXT:    v_max3_u32 v0, v2, v0, v1
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-LABEL: test_vector_reduce_umax_v3i16:
@@ -1730,15 +1730,15 @@ define i16 @test_vector_reduce_umax_v4i16(<4 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umax_v4i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v3, 16, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT:    v_max3_u32 v0, v0, v1, v2
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v1, 0, v2
-; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_max3_u32 v1, v2, v3, v0
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v1, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umax_v4i16:
@@ -1886,22 +1886,22 @@ define i16 @test_vector_reduce_umax_v8i16(<8 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umax_v8i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v5, 16, v1
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v7, 16, v3
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v4, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v6, 16, v2
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v7, 0xffff, v3
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v4, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v5, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v6, v6, v7
 ; GFX7-GISEL-NEXT:    v_max_u32_e32 v1, v1, v3
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v3, v5, v7
+; GFX7-GISEL-NEXT:    v_max3_u32 v3, v4, v5, v6
 ; GFX7-GISEL-NEXT:    v_max3_u32 v0, v0, v2, v1
-; GFX7-GISEL-NEXT:    v_max3_u32 v1, v4, v6, v3
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v1, 0, v1
-; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v1, v3, v0
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v1, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umax_v8i16:
@@ -2099,35 +2099,35 @@ define i16 @test_vector_reduce_umax_v16i16(<16 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umax_v16i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v10, 16, v2
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v11, 16, v3
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v14, 16, v6
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v15, 16, v7
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v6
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v7, 0xffff, v7
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v8, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v9, 16, v1
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v12, 16, v4
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v13, 16, v5
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v5, 0xffff, v5
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v12, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v13, 0xffff, v6
+; GFX7-GISEL-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v6, v6, 16, 16
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v12, v12, v13
 ; GFX7-GISEL-NEXT:    v_max_u32_e32 v2, v2, v6
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v6, v10, v14
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v3
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v13, 0xffff, v7
+; GFX7-GISEL-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v7, v7, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v8, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v9, 0xffff, v4
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v4, v4, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v10, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v11, 0xffff, v5
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v5, v5, 16, 16
 ; GFX7-GISEL-NEXT:    v_max_u32_e32 v3, v3, v7
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v7, v11, v15
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v6, v6, v13
 ; GFX7-GISEL-NEXT:    v_max3_u32 v0, v0, v4, v2
-; GFX7-GISEL-NEXT:    v_max3_u32 v2, v8, v12, v6
 ; GFX7-GISEL-NEXT:    v_max3_u32 v1, v1, v5, v3
-; GFX7-GISEL-NEXT:    v_max3_u32 v3, v9, v13, v7
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT:    v_max3_u32 v0, v0, v1, v2
-; GFX7-GISEL-NEXT:    v_max_u32_e32 v1, 0, v2
-; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_max3_u32 v7, v8, v9, v12
+; GFX7-GISEL-NEXT:    v_max3_u32 v2, v10, v11, v6
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_max3_u32 v1, v7, v2, v0
+; GFX7-GISEL-NEXT:    v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT:    v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT:    v_or_b32_e32 v0, v1, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umax_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
index 5443cce424a5c..f3aa7eba3e105 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
@@ -315,7 +315,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_min_u16_sdwa v0, v0, v2 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
 ; GFX8-GISEL-NEXT:    v_min_u16_sdwa v1, v1, v3 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_umin_v4i8:
@@ -335,7 +335,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_and_b32_e32 v2, 0xff, v2
 ; GFX9-GISEL-NEXT:    v_min_u16_sdwa v1, v1, v3 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
 ; GFX9-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_umin_v4i8:
@@ -361,7 +361,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_and_b32_e32 v2, 0xff, v2
 ; GFX10-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v4i8:
@@ -409,7 +409,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX11-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v4i8:
@@ -469,7 +469,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX12-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.umin.v4i8(<4 x i8> %v)
@@ -536,7 +536,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_umin_v8i8:
@@ -567,7 +567,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_min3_u16 v0, v0, v4, v2
 ; GFX9-GISEL-NEXT:    v_min3_u16 v1, v1, v5, v3
 ; GFX9-GISEL-NEXT:    v_min_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_umin_v8i8:
@@ -607,7 +607,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_min3_u16 v0, v0, v4, v2
 ; GFX10-GISEL-NEXT:    v_min3_u16 v1, v1, v5, v3
 ; GFX10-GISEL-NEXT:    v_min_u16 v0, v0, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v8i8:
@@ -678,7 +678,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_min3_u16 v1, v1, v5, v3
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-NEXT:    v_min_u16 v0, v0, v1
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v8i8:
@@ -761,7 +761,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_min3_u16 v1, v1, v5, v3
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-NEXT:    v_min_u16 v0, v0, v1
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.umin.v8i8(<8 x i8> %v)
@@ -873,7 +873,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_umin_v16i8:
@@ -921,7 +921,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_min3_u16 v2, v2, v10, v6
 ; GFX9-GISEL-NEXT:    v_min_u16_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_umin_v16i8:
@@ -990,7 +990,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_min3_u16 v2, v2, v10, v6
 ; GFX10-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v16i8:
@@ -1107,7 +1107,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX11-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v16i8:
@@ -1236,7 +1236,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_min_u16 v1, v1, v3
 ; GFX12-GISEL-NEXT:    v_min3_u16 v0, v0, v2, v1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.umin.v16i8(<16 x i8> %v)
@@ -1374,10 +1374,10 @@ define i16 @test_vector_reduce_umin_v3i16(<3 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umin_v3i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
 ; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_min3_u32 v0, v0, v2, v1
+; GFX7-GISEL-NEXT:    v_min3_u32 v0, v2, v0, v1
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-LABEL: test_vector_reduce_umin_v3i16:
@@ -1512,12 +1512,12 @@ define i16 @test_vector_reduce_umin_v4i16(<4 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umin_v4i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v3, 16, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT:    v_min3_u32 v0, v0, v1, v2
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_min3_u32 v0, v2, v3, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umin_v4i16:
@@ -1665,19 +1665,19 @@ define i16 @test_vector_reduce_umin_v8i16(<8 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umin_v8i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v5, 16, v1
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v7, 16, v3
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v4, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v6, 16, v2
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v7, 0xffff, v3
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v4, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v5, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v6, v6, v7
 ; GFX7-GISEL-NEXT:    v_min_u32_e32 v1, v1, v3
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v3, v5, v7
+; GFX7-GISEL-NEXT:    v_min3_u32 v3, v4, v5, v6
 ; GFX7-GISEL-NEXT:    v_min3_u32 v0, v0, v2, v1
-; GFX7-GISEL-NEXT:    v_min3_u32 v1, v4, v6, v3
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v0, v3, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umin_v8i16:
@@ -1875,32 +1875,32 @@ define i16 @test_vector_reduce_umin_v16i16(<16 x i16> %v) {
 ; GFX7-GISEL-LABEL: test_vector_reduce_umin_v16i16:
 ; GFX7-GISEL:       ; %bb.0: ; %entry
 ; GFX7-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v10, 16, v2
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v11, 16, v3
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v14, 16, v6
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v15, 16, v7
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v2, 0xffff, v2
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v6
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v7, 0xffff, v7
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v8, 16, v0
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v9, 16, v1
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v12, 16, v4
-; GFX7-GISEL-NEXT:    v_lshrrev_b32_e32 v13, 16, v5
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v4, 0xffff, v4
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT:    v_and_b32_e32 v5, 0xffff, v5
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v12, 0xffff, v2
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v13, 0xffff, v6
+; GFX7-GISEL-NEXT:    v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v6, v6, 16, 16
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v12, v12, v13
 ; GFX7-GISEL-NEXT:    v_min_u32_e32 v2, v2, v6
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v6, v10, v14
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v6, 0xffff, v3
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v13, 0xffff, v7
+; GFX7-GISEL-NEXT:    v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v7, v7, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v8, 0xffff, v0
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v9, 0xffff, v4
+; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v4, v4, 16, 16
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v10, 0xffff, v1
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v11, 0xffff, v5
+; GFX7-GISEL-NEXT:    v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT:    v_bfe_u32 v5, v5, 16, 16
 ; GFX7-GISEL-NEXT:    v_min_u32_e32 v3, v3, v7
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v7, v11, v15
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v6, v6, v13
 ; GFX7-GISEL-NEXT:    v_min3_u32 v0, v0, v4, v2
-; GFX7-GISEL-NEXT:    v_min3_u32 v2, v8, v12, v6
 ; GFX7-GISEL-NEXT:    v_min3_u32 v1, v1, v5, v3
-; GFX7-GISEL-NEXT:    v_min3_u32 v3, v9, v13, v7
-; GFX7-GISEL-NEXT:    v_min_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT:    v_min3_u32 v0, v0, v1, v2
+; GFX7-GISEL-NEXT:    v_min3_u32 v7, v8, v9, v12
+; GFX7-GISEL-NEXT:    v_min3_u32 v2, v10, v11, v6
+; GFX7-GISEL-NEXT:    v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT:    v_min3_u32 v0, v7, v2, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_umin_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
index b9a8ea279ad15..b68654a2f9ff4 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
@@ -292,7 +292,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -310,7 +310,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -328,7 +328,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -345,7 +345,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX10-GISEL-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
 ; GFX10-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v4i8:
@@ -374,7 +374,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX11-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v4i8:
@@ -415,7 +415,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
 ; GFX12-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.xor.v4i8(<4 x i8> %v)
@@ -446,7 +446,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -472,7 +472,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -498,7 +498,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -522,7 +522,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_xor_b32_e32 v2, v2, v6
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v8i8:
@@ -565,7 +565,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX11-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v8i8:
@@ -620,7 +620,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX12-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.xor.v8i8(<8 x i8> %v)
@@ -667,7 +667,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX7-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX7-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX8-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -709,7 +709,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX8-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX8-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX9-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -751,7 +751,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v2
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v1, v1, v3
 ; GFX9-GISEL-NEXT:    v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX9-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX10-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -788,7 +788,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v2, v2, v10, v6
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX10-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX10-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v16i8:
@@ -855,7 +855,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX11-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX11-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
 ; GFX11-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX11-GISEL-NEXT:    s_setpc_b64 s[30:31]
 ;
 ; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v16i8:
@@ -934,7 +934,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
 ; GFX12-GISEL-NEXT:    v_xor3_b32 v1, v1, v5, v3
 ; GFX12-GISEL-NEXT:    v_xor3_b32 v0, v0, v2, v1
 ; GFX12-GISEL-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT:    v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT:    v_and_b32_e32 v0, 0xff, v0
 ; GFX12-GISEL-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %res = call i8 @llvm.vector.reduce.xor.v16i8(<16 x i8> %v)
diff --git a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
index ac54854ff6bfe..ef02f967d645d 100644
--- a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
+++ b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
@@ -191,38 +191,37 @@ define amdgpu_kernel void @workgroup_id_xy(ptr addrspace(1) %ptrx, ptr addrspace
 }
 
 define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspace(1) %ptry, ptr addrspace(1) %ptrz) {
-; GFX9-LABEL: workgroup_id_xyz:
-; GFX9:       ; %bb.0:
-; GFX9-NEXT:    s_load_dwordx4 s[0:3], s[8:9], 0x0
-; GFX9-NEXT:    s_load_dwordx2 s[4:5], s[8:9], 0x10
-; GFX9-NEXT:    v_mov_b32_e32 v0, ttmp9
-; GFX9-NEXT:    v_mov_b32_e32 v1, 0
-; GFX9-NEXT:    s_and_b32 s6, ttmp7, 0xffff
-; GFX9-NEXT:    s_waitcnt lgkmcnt(0)
-; GFX9-NEXT:    global_store_dword v1, v0, s[0:1]
-; GFX9-NEXT:    v_mov_b32_e32 v0, s6
-; GFX9-NEXT:    s_lshr_b32 s0, ttmp7, 16
-; GFX9-NEXT:    global_store_dword v1, v0, s[2:3]
-; GFX9-NEXT:    v_mov_b32_e32 v0, s0
-; GFX9-NEXT:    global_store_dword v1, v0, s[4:5]
-; GFX9-NEXT:    s_endpgm
+; GFX9-SDAG-LABEL: workgroup_id_xyz:
+; GFX9-SDAG:       ; %bb.0:
+; GFX9-SDAG-NEXT:    s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX9-SDAG-NEXT:    s_load_dwordx2 s[4:5], s[8:9], 0x10
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, ttmp9
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v1, 0
+; GFX9-SDAG-NEXT:    s_and_b32 s6, ttmp7, 0xffff
+; GFX9-SDAG-NEXT:    s_waitcnt lgkmcnt(0)
+; GFX9-SDAG-NEXT:    global_store_dword v1, v0, s[0:1]
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, s6
+; GFX9-SDAG-NEXT:    s_lshr_b32 s0, ttmp7, 16
+; GFX9-SDAG-NEXT:    global_store_dword v1, v0, s[2:3]
+; GFX9-SDAG-NEXT:    v_mov_b32_e32 v0, s0
+; GFX9-SDAG-NEXT:    global_store_dword v1, v0, s[4:5]
+; GFX9-SDAG-NEXT:    s_endpgm
 ;
-; GFX1200-LABEL: workgroup_id_xyz:
-; GFX1200:       ; %bb.0:
-; GFX1200-NEXT:    s_clause 0x1
-; GFX1200-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0
-; GFX1200-NEXT:    s_load_b64 s[4:5], s[4:5], 0x10
-; GFX1200-NEXT:    s_and_b32 s6, ttmp7, 0xffff
-; GFX1200-NEXT:    v_dual_mov_b32 v0, ttmp9 :: v_dual_mov_b32 v1, 0
-; GFX1200-NEXT:    s_lshr_b32 s7, ttmp7, 16
-; GFX1200-NEXT:    s_delay_alu instid0(SALU_CYCLE_1)
-; GFX1200-NEXT:    v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX1200-NEXT:    s_wait_kmcnt 0x0
-; GFX1200-NEXT:    s_clause 0x2
-; GFX1200-NEXT:    global_store_b32 v1, v0, s[0:1]
-; GFX1200-NEXT:    global_store_b32 v1, v2, s[2:3]
-; GFX1200-NEXT:    global_store_b32 v1, v3, s[4:5]
-; GFX1200-NEXT:    s_endpgm
+; GFX9-GISEL-LABEL: workgroup_id_xyz:
+; GFX9-GISEL:       ; %bb.0:
+; GFX9-GISEL-NEXT:    s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX9-GISEL-NEXT:    s_load_dwordx2 s[4:5], s[8:9], 0x10
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, ttmp9
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v1, 0
+; GFX9-GISEL-NEXT:    s_and_b32 s6, ttmp7, 0xffff
+; GFX9-GISEL-NEXT:    s_waitcnt lgkmcnt(0)
+; GFX9-GISEL-NEXT:    global_store_dword v1, v0, s[0:1]
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, s6
+; GFX9-GISEL-NEXT:    s_bfe_u32 s0, ttmp7, 0x100010
+; GFX9-GISEL-NEXT:    global_store_dword v1, v0, s[2:3]
+; GFX9-GISEL-NEXT:    v_mov_b32_e32 v0, s0
+; GFX9-GISEL-NEXT:    global_store_dword v1, v0, s[4:5]
+; GFX9-GISEL-NEXT:    s_endpgm
 ;
 ; GFX1250-SDAG-LABEL: workgroup_id_xyz:
 ; GFX1250-SDAG:       ; %bb.0:
@@ -295,7 +294,7 @@ define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspac
 ; GFX1250-GISEL-NEXT:    s_wait_xcnt 0x0
 ; GFX1250-GISEL-NEXT:    s_cselect_b32 s4, s10, s11
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s5, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT:    s_lshr_b32 s10, ttmp7, 16
+; GFX1250-GISEL-NEXT:    s_bfe_u32 s10, ttmp7, 0x100010
 ; GFX1250-GISEL-NEXT:    s_add_co_i32 s5, s5, 1
 ; GFX1250-GISEL-NEXT:    s_bfe_u32 s11, ttmp6, 0x40008
 ; GFX1250-GISEL-NEXT:    s_mul_i32 s5, s10, s5
@@ -341,5 +340,3 @@ declare i32 @llvm.amdgcn.workgroup.id.y()
 declare i32 @llvm.amdgcn.workgroup.id.z()
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GFX1250: {{.*}}
-; GFX9-GISEL: {{.*}}
-; GFX9-SDAG: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
index 07937b347a622..3f145cd1c59e1 100644
--- a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
+++ b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
@@ -240,7 +240,7 @@ define i1 @workgroup_zero() {
 ; GISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
 ; GISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
 ; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GISEL-GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GISEL-GFX12-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
 ; GISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
 ; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
 ; GISEL-GFX12-NEXT:    s_or_b32 s0, s0, s1
@@ -280,23 +280,41 @@ define i1 @workgroup_nonzero() {
 ; GFX942-NEXT:    v_mov_b32_e32 v0, s0
 ; GFX942-NEXT:    s_setpc_b64 s[30:31]
 ;
-; GFX12-LABEL: workgroup_nonzero:
-; GFX12:       ; %bb.0: ; %entry
-; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT:    s_wait_expcnt 0x0
-; GFX12-NEXT:    s_wait_samplecnt 0x0
-; GFX12-NEXT:    s_wait_bvhcnt 0x0
-; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    s_or_b32 s0, s0, s1
-; GFX12-NEXT:    s_cselect_b32 s0, 1, 0
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_mov_b32_e32 v0, s0
-; GFX12-NEXT:    s_setpc_b64 s[30:31]
+; DAGISEL-GFX12-LABEL: workgroup_nonzero:
+; DAGISEL-GFX12:       ; %bb.0: ; %entry
+; DAGISEL-GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_expcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_samplecnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_bvhcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; DAGISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    s_or_b32 s0, s0, s1
+; DAGISEL-GFX12-NEXT:    s_cselect_b32 s0, 1, 0
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    v_mov_b32_e32 v0, s0
+; DAGISEL-GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX12-LABEL: workgroup_nonzero:
+; GISEL-GFX12:       ; %bb.0: ; %entry
+; GISEL-GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_expcnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_samplecnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
+; GISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
+; GISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    s_or_b32 s0, s0, s1
+; GISEL-GFX12-NEXT:    s_cselect_b32 s0, 1, 0
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    v_mov_b32_e32 v0, s0
+; GISEL-GFX12-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %0 = tail call i32 @llvm.amdgcn.workgroup.id.x()
   %1 = tail call i32 @llvm.amdgcn.workgroup.id.y()
@@ -335,28 +353,51 @@ define i1 @workitem_workgroup_zero() {
 ; GFX942-NEXT:    v_cndmask_b32_e64 v0, 0, 1, vcc
 ; GFX942-NEXT:    s_setpc_b64 s[30:31]
 ;
-; GFX12-LABEL: workitem_workgroup_zero:
-; GFX12:       ; %bb.0: ; %entry
-; GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT:    s_wait_expcnt 0x0
-; GFX12-NEXT:    s_wait_samplecnt 0x0
-; GFX12-NEXT:    s_wait_bvhcnt 0x0
-; GFX12-NEXT:    s_wait_kmcnt 0x0
-; GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v31
-; GFX12-NEXT:    v_bfe_u32 v1, v31, 10, 10
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    s_or_b32 s0, s0, s1
-; GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT:    v_or3_b32 v0, s0, v0, v1
-; GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0
-; GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
-; GFX12-NEXT:    v_cndmask_b32_e64 v0, 0, 1, vcc_lo
-; GFX12-NEXT:    s_setpc_b64 s[30:31]
+; DAGISEL-GFX12-LABEL: workitem_workgroup_zero:
+; DAGISEL-GFX12:       ; %bb.0: ; %entry
+; DAGISEL-GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_expcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_samplecnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_bvhcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
+; DAGISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; DAGISEL-GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v31
+; DAGISEL-GFX12-NEXT:    v_bfe_u32 v1, v31, 10, 10
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; DAGISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    s_or_b32 s0, s0, s1
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT:    v_or3_b32 v0, s0, v0, v1
+; DAGISEL-GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL-GFX12-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0
+; DAGISEL-GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; DAGISEL-GFX12-NEXT:    v_cndmask_b32_e64 v0, 0, 1, vcc_lo
+; DAGISEL-GFX12-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX12-LABEL: workitem_workgroup_zero:
+; GISEL-GFX12:       ; %bb.0: ; %entry
+; GISEL-GFX12-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_expcnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_samplecnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_bvhcnt 0x0
+; GISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
+; GISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
+; GISEL-GFX12-NEXT:    v_and_b32_e32 v0, 0x3ff, v31
+; GISEL-GFX12-NEXT:    v_bfe_u32 v1, v31, 10, 10
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
+; GISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    s_or_b32 s0, s0, s1
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT:    v_or3_b32 v0, s0, v0, v1
+; GISEL-GFX12-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GISEL-GFX12-NEXT:    v_cmp_eq_u32_e32 vcc_lo, 0, v0
+; GISEL-GFX12-NEXT:    s_wait_alu depctr_va_vcc(0)
+; GISEL-GFX12-NEXT:    v_cndmask_b32_e64 v0, 0, 1, vcc_lo
+; GISEL-GFX12-NEXT:    s_setpc_b64 s[30:31]
 entry:
   %0 = tail call i32 @llvm.amdgcn.workgroup.id.x()
   %1 = tail call i32 @llvm.amdgcn.workgroup.id.y()
@@ -452,7 +493,7 @@ define i1 @workitem_workgroup_nonzero() {
 ; GISEL-GFX12-NEXT:    s_wait_kmcnt 0x0
 ; GISEL-GFX12-NEXT:    s_and_b32 s0, ttmp7, 0xffff
 ; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
-; GISEL-GFX12-NEXT:    s_lshr_b32 s1, ttmp7, 16
+; GISEL-GFX12-NEXT:    s_bfe_u32 s1, ttmp7, 0x100010
 ; GISEL-GFX12-NEXT:    s_or_b32 s0, ttmp9, s0
 ; GISEL-GFX12-NEXT:    v_bfe_u32 v0, v31, 10, 10
 ; GISEL-GFX12-NEXT:    s_wait_alu depctr_sa_sdst(0)
@@ -481,3 +522,5 @@ entry:
   %cmp = icmp ne i32 %or4, 0
   ret i1 %cmp
 }
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}

>From 6ac2e41d1044ac406b0e4d18c3f6d866d7894be0 Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Wed, 2 Sep 2026 17:12:44 +0530
Subject: [PATCH 2/3] Utilized class rule in place of GICOmbinePatfrag for
 commute_shift.

---
 .../include/llvm/Target/GlobalISel/Combine.td | 24 +++++++++----------
 1 file changed, 12 insertions(+), 12 deletions(-)

diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 8942fd3a2e155..699ca9ca9ba82 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -447,26 +447,26 @@ def bitreverse_lshr : GICombineRule<
   (apply (G_SHL $d, $val, $amt))>;
 
 // Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
-// Combine (shl (or  x, c1), c2) -> (or  (shl x, c2), c1 << c2)
-// Apply rebuilds the matched opcode, so it stays inline C++.
-def commute_shift_frags : GICombinePatFrag<
-  (outs root:$dst, $binop), (ins $x, $c1, $c2),
-  !foreach(op, [G_ADD, G_OR],
-    (pattern (op $binop, $x, $c1), (G_SHL $dst, $binop, $c2)))>;
-def commute_shift : GICombineRule<
+// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
+// Apply rebuilds the matched opcode, so it stays inline C++. The builder is
+// used rather than an apply pattern so that (shl c1, c2) is constant folded.
+class commute_shift_binop<Instruction binop> : GICombineRule<
   (defs root:$dst),
-  (match (commute_shift_frags $dst, $binop, $x, $c1, $c2):$shl,
-         [{ return MRI.hasOneNonDBGUse(${binop}.getReg()) &&
+  (match (binop $binop_dst, $x, $c1):$binop,
+         (G_SHL $dst, $binop_dst, $c2):$shl,
+         [{ return MRI.hasOneNonDBGUse(${binop_dst}.getReg()) &&
                    isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
                    isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
                    Helper.isDesirableToCommuteWithShift(*${shl}); }]),
   (apply [{ auto &B = Helper.getBuilder();
             LLT Ty = MRI.getType(${dst}.getReg());
-            unsigned Opc = MRI.getVRegDef(${binop}.getReg())->getOpcode();
             auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
             auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
-            auto New = B.buildInstr(Opc, {Ty}, {S1, S2});
+            auto New = B.buildInstr(${binop}->getOpcode(), {Ty}, {S1, S2});
             Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
+def commute_shift_add : commute_shift_binop<G_ADD>;
+def commute_shift_or : commute_shift_binop<G_OR>;
+def commute_shift : GICombineGroup<[commute_shift_add, commute_shift_or]>;
 
 // Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (lshr x, (C1 + C2))
 def lshr_of_trunc_of_lshr_matchdata : GIDefMatchData<"LshrOfTruncOfLshr">;
@@ -1003,7 +1003,7 @@ def neg_and_one_to_sext_inreg : GICombineRule<
 >;
 
 // Fold and(and(x, C1), C2) -> C1&C2 ? and(x, C1&C2) : 0
-def overlapping_and : GICombineRule<
+def overlapping_and: GICombineRule <
   (defs root:$root, build_fn_matchinfo:$info),
   (match (wip_match_opcode G_AND):$root,
          [{ return Helper.matchOverlappingAnd(*${root}, ${info}); }]),

>From 36d346cb329412f35fae5becd90c4a32f21e4d9c Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Thu, 3 Sep 2026 13:07:22 +0530
Subject: [PATCH 3/3] Add MIR pattern for apply rule in commute_shift.

---
 llvm/include/llvm/Target/GlobalISel/Combine.td        | 11 +++--------
 .../GlobalISel/prelegalizercombiner-commute-shift.mir |  5 ++---
 2 files changed, 5 insertions(+), 11 deletions(-)

diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 699ca9ca9ba82..3735e95ee8f68 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -448,8 +448,6 @@ def bitreverse_lshr : GICombineRule<
 
 // Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
 // Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
-// Apply rebuilds the matched opcode, so it stays inline C++. The builder is
-// used rather than an apply pattern so that (shl c1, c2) is constant folded.
 class commute_shift_binop<Instruction binop> : GICombineRule<
   (defs root:$dst),
   (match (binop $binop_dst, $x, $c1):$binop,
@@ -458,12 +456,9 @@ class commute_shift_binop<Instruction binop> : GICombineRule<
                    isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
                    isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
                    Helper.isDesirableToCommuteWithShift(*${shl}); }]),
-  (apply [{ auto &B = Helper.getBuilder();
-            LLT Ty = MRI.getType(${dst}.getReg());
-            auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
-            auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
-            auto New = B.buildInstr(${binop}->getOpcode(), {Ty}, {S1, S2});
-            Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
+  (apply (G_SHL GITypeOf<"$dst">:$s1, $x, $c2),
+         (G_SHL GITypeOf<"$dst">:$s2, $c1, $c2),
+         (binop $dst, $s1, $s2))>;
 def commute_shift_add : commute_shift_binop<G_ADD>;
 def commute_shift_or : commute_shift_binop<G_OR>;
 def commute_shift : GICombineGroup<[commute_shift_add, commute_shift_or]>;
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
index 8f78c1dacaf2a..7bc1503e9da32 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
@@ -108,9 +108,8 @@ body:             |
     ; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 2
     ; CHECK-NEXT: %veccst2:_(<4 x i32>) = G_BUILD_VECTOR [[C]](i32), [[C]](i32), [[C]](i32), [[C]](i32)
     ; CHECK-NEXT: [[SHL:%[0-9]+]]:_(<4 x i32>) = G_SHL %xvec, %veccst2(<4 x i32>)
-    ; CHECK-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 8
-    ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<4 x i32>) = G_BUILD_VECTOR [[C1]](i32), [[C1]](i32), [[C1]](i32), [[C1]](i32)
-    ; CHECK-NEXT: [[ADD:%[0-9]+]]:_(<4 x i32>) = G_ADD [[SHL]], [[BUILD_VECTOR]]
+    ; CHECK-NEXT: [[SHL1:%[0-9]+]]:_(<4 x i32>) = G_SHL %veccst2, %veccst2(<4 x i32>)
+    ; CHECK-NEXT: [[ADD:%[0-9]+]]:_(<4 x i32>) = G_ADD [[SHL]], [[SHL1]]
     ; CHECK-NEXT: G_STORE [[ADD]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
     ; CHECK-NEXT: RET_ReallyLR
     %0:_(p0) = COPY $x0



More information about the llvm-commits mailing list