[llvm] [GlobalISel] Migrate various generic wip_match_opcode combines to MIR-pattern. (PR #220213)
Vikash Gupta via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 04:03:03 PDT 2026
https://github.com/vg0204 updated https://github.com/llvm/llvm-project/pull/220213
>From 606e83e89044e7997cf39dc634d90007efa5d0bb Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Tue, 1 Sep 2026 15:31:29 +0530
Subject: [PATCH 1/3] [GlobalISel] Migrate various wip_match_opcode combines to
MIR-pattern.
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
This patch converts a batch of GlobalISel combine rules from
hand-written C++ matchers to declarative MIR patterns preserving the
behavior. It also along with adds two functional changes (a rule-ordering
fix and an out-of-bounds bug fix) are described below.
- `select_same_val` → `select_same_val_trivial` +
`select_same_val_equiv` group
- `select_constant_cmp` → `_false` / `_true` / `_general` group
- `simplify_add_to_sub`, `add_p2i_to_ptradd` → `GICombinePatFrag`s
- `commute_shift` → `commute_shift_frags` (C++ residue reduced to
`isDesirableToCommuteWithShift`)
- `combine_i2p_to_p2i`, `ptr_add_zero`, `sext_trunc_sext_load` →
patterns.
- funnel-shift / rotate / `ashr_lshr` / `constant_fold_fma` /
`constant_fold_cast_op` / `combine_minmax_nan` → pattern fragments
- **`select_zero_true` / `select_zero_false`:** skip a constant
condition as they fold it to an arm instead strictly better and
avoids a `G_FREEZE` that post-legalization can't clean up.
- **`match_bitfield_extract_from_and`:** mask the AND immediate to the
operand width and reject fields where `lsb + width > size`,
preventing an out-of-bounds `G_UBFX` (fixes an AMDGPU
`knownBitsForSBFE` crash).
28 CodeGen tests updated. All diffs are semantically equivalent.
---
.../llvm/CodeGen/GlobalISel/CombinerHelper.h | 10 +-
.../include/llvm/Target/GlobalISel/Combine.td | 307 +++++++++++-------
.../lib/CodeGen/GlobalISel/CombinerHelper.cpp | 84 +----
.../GlobalISel/combine-insert-vec-elt.mir | 14 +-
.../form-bitfield-extract-from-and.mir | 5 +-
...galizer-combiner-divrem-insertpt-crash.mir | 5 +-
.../AArch64/neon-shuffle-vector-tbl.ll | 2 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll | 152 ++++-----
.../combine-extract-vector-load.mir | 10 +-
.../GlobalISel/combine-redundant-and.mir | 10 +-
.../combine-shift-of-shifted-logic.ll | 4 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll | 18 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll | 148 ++++-----
.../GlobalISel/llvm.amdgcn.intersect_ray.ll | 46 +--
.../llvm.amdgcn.make.buffer.rsrc.ll | 4 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll | 156 ++++-----
.../GlobalISel/merge-values-s16-true16.ll | 11 +-
.../AMDGPU/GlobalISel/mul-known-bits.i64.ll | 21 +-
llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll | 80 ++---
.../CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll | 4 +-
.../AMDGPU/llvm.amdgcn.intersect_ray.ll | 219 +++++--------
.../lower-work-group-id-intrinsics-hsa.ll | 71 ++--
.../lower-work-group-id-intrinsics-opt.ll | 8 +-
.../lower-work-group-id-intrinsics-pal.ll | 72 ++--
.../AMDGPU/lower-work-group-id-intrinsics.ll | 18 +-
llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll | 56 ++--
llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll | 36 +-
llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll | 76 ++---
.../test/CodeGen/AMDGPU/vector-reduce-umax.ll | 102 +++---
.../test/CodeGen/AMDGPU/vector-reduce-umin.ll | 114 +++----
llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll | 36 +-
.../AMDGPU/workgroup-id-in-arch-sgprs.ll | 65 ++--
.../CodeGen/AMDGPU/workitem-intrinsic-opts.ll | 125 ++++---
33 files changed, 1073 insertions(+), 1016 deletions(-)
diff --git a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
index a8bf13b5760c1..6b80367d786a5 100644
--- a/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
+++ b/llvm/include/llvm/CodeGen/GlobalISel/CombinerHelper.h
@@ -245,7 +245,6 @@ class CombinerHelper {
IndexedLoadStoreMatchInfo &MatchInfo) const;
LLVM_ABI bool matchSextTruncSextLoad(MachineInstr &MI) const;
- LLVM_ABI void applySextTruncSextLoad(MachineInstr &MI) const;
/// Match sext_inreg(load p), imm -> sextload p
LLVM_ABI bool
@@ -379,7 +378,9 @@ class CombinerHelper {
LLVM_ABI void applyShiftOfShiftedLogic(MachineInstr &MI,
ShiftOfShiftedLogic &MatchInfo) const;
- LLVM_ABI bool matchCommuteShift(MachineInstr &MI, BuildFnTy &MatchInfo) const;
+ /// \return true if the target's TargetLowering::isDesirableToCommuteWithShift
+ /// hook approves of commuting \p MI (a G_SHL) with the binop feeding it.
+ LLVM_ABI bool isDesirableToCommuteWithShift(const MachineInstr &MI) const;
/// Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (shift x, (C1 + C2))
LLVM_ABI bool matchLshrOfTruncOfLshr(MachineInstr &MI,
@@ -454,10 +455,6 @@ class CombinerHelper {
LLVM_ABI bool matchConstantFoldUnaryIntOp(MachineInstr &MI,
BuildFnTy &MatchInfo) const;
- /// Transform IntToPtr(PtrToInt(x)) to x if cast is in the same address space.
- LLVM_ABI bool matchCombineI2PToP2I(MachineInstr &MI, Register &Reg) const;
- LLVM_ABI void applyCombineI2PToP2I(MachineInstr &MI, Register &Reg) const;
-
/// Transform PtrToInt(IntToPtr(x)) to x.
LLVM_ABI void applyCombineP2IToI2P(MachineInstr &MI, Register &Reg) const;
@@ -649,7 +646,6 @@ class CombinerHelper {
/// Combine G_PTR_ADD with nullptr to G_INTTOPTR
LLVM_ABI bool matchPtrAddZero(MachineInstr &MI) const;
- LLVM_ABI void applyPtrAddZero(MachineInstr &MI) const;
/// Combine G_UREM x, (known power of 2) to an add and bitmasking.
LLVM_ABI void applySimplifyURemByPow2(MachineInstr &MI) const;
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 4b0a438cd6455..8942fd3a2e155 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -251,21 +251,21 @@ def extending_loads : GICombineRule<
def load_and_mask : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_AND):$root,
+ (match (G_AND $dst, $src1, $src2):$root,
[{ return Helper.matchCombineLoadWithAndMask(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
def combines_for_extload: GICombineGroup<[extending_loads, load_and_mask]>;
def sext_trunc_sextload : GICombineRule<
(defs root:$d),
- (match (wip_match_opcode G_SEXT_INREG):$d,
+ (match (G_SEXT_INREG $dst, $src, $sz):$d,
[{ return Helper.matchSextTruncSextLoad(*${d}); }]),
- (apply [{ Helper.applySextTruncSextLoad(*${d}); }])>;
+ (apply (GIReplaceReg $dst, $src))>;
def sext_inreg_of_load_matchdata : GIDefMatchData<"std::tuple<Register, unsigned>">;
def sext_inreg_of_load : GICombineRule<
(defs root:$root, sext_inreg_of_load_matchdata:$matchinfo),
- (match (wip_match_opcode G_SEXT_INREG):$root,
+ (match (G_SEXT_INREG $dst, $src, $sz):$root,
[{ return Helper.matchSextInRegOfLoad(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applySextInRegOfLoad(*${root}, ${matchinfo}); }])>;
@@ -286,7 +286,7 @@ def sext_inreg_to_zext_inreg : GICombineRule<
def combine_extracted_vector_load : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+ (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
[{ return Helper.matchCombineExtractedVectorLoad(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
@@ -339,14 +339,14 @@ def memcpy_family_combines : GICombineGroup<[combine_memcpy_inline,
def opt_brcond_by_inverting_cond_matchdata : GIDefMatchData<"MachineInstr *">;
def opt_brcond_by_inverting_cond : GICombineRule<
(defs root:$root, opt_brcond_by_inverting_cond_matchdata:$matchinfo),
- (match (wip_match_opcode G_BR):$root,
+ (match (G_BR $tgt):$root,
[{ return Helper.matchOptBrCondByInvertingCond(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyOptBrCondByInvertingCond(*${root}, ${matchinfo}); }])>;
def ptr_add_immed_matchdata : GIDefMatchData<"PtrAddChain">;
def ptr_add_immed_chain : GICombineRule<
(defs root:$d, ptr_add_immed_matchdata:$matchinfo),
- (match (wip_match_opcode G_PTR_ADD):$d,
+ (match (G_PTR_ADD $dst, $base, $offset):$d,
[{ return Helper.matchPtrAddImmedChain(*${d}, ${matchinfo}); }]),
(apply [{ Helper.applyPtrAddImmedChain(*${d}, ${matchinfo}); }])>;
@@ -447,12 +447,26 @@ def bitreverse_lshr : GICombineRule<
(apply (G_SHL $d, $val, $amt))>;
// Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
-// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
+// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
+// Apply rebuilds the matched opcode, so it stays inline C++.
+def commute_shift_frags : GICombinePatFrag<
+ (outs root:$dst, $binop), (ins $x, $c1, $c2),
+ !foreach(op, [G_ADD, G_OR],
+ (pattern (op $binop, $x, $c1), (G_SHL $dst, $binop, $c2)))>;
def commute_shift : GICombineRule<
- (defs root:$d, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SHL):$d,
- [{ return Helper.matchCommuteShift(*${d}, ${matchinfo}); }]),
- (apply [{ Helper.applyBuildFn(*${d}, ${matchinfo}); }])>;
+ (defs root:$dst),
+ (match (commute_shift_frags $dst, $binop, $x, $c1, $c2):$shl,
+ [{ return MRI.hasOneNonDBGUse(${binop}.getReg()) &&
+ isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
+ isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
+ Helper.isDesirableToCommuteWithShift(*${shl}); }]),
+ (apply [{ auto &B = Helper.getBuilder();
+ LLT Ty = MRI.getType(${dst}.getReg());
+ unsigned Opc = MRI.getVRegDef(${binop}.getReg())->getOpcode();
+ auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
+ auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
+ auto New = B.buildInstr(Opc, {Ty}, {S1, S2});
+ Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
// Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (lshr x, (C1 + C2))
def lshr_of_trunc_of_lshr_matchdata : GIDefMatchData<"LshrOfTruncOfLshr">;
@@ -466,7 +480,7 @@ def lshr_of_trunc_of_lshr : GICombineRule<
def narrow_binop_feeding_and : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_AND):$root,
+ (match (G_AND $dst, $src1, $src2):$root,
[{ return Helper.matchNarrowBinopFeedingAnd(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFnNoErase(*${root}, ${matchinfo}); }])>;
@@ -562,9 +576,9 @@ def propagate_undef_all_ops: GICombineRule<
// Replace a G_SHUFFLE_VECTOR with an undef mask with a G_IMPLICIT_DEF.
def propagate_undef_shuffle_mask: GICombineRule<
(defs root:$root),
- (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+ (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
[{ return Helper.matchUndefShuffleVectorMask(*${root}); }]),
- (apply [{ Helper.replaceInstWithUndef(*${root}); }])>;
+ (apply (G_IMPLICIT_DEF $dst))>;
// Replace an insert/extract element of an out of bounds index with undef.
def insert_extract_vec_elt_out_of_bounds : GICombineRule<
@@ -574,12 +588,20 @@ def propagate_undef_shuffle_mask: GICombineRule<
(apply [{ Helper.replaceInstWithUndef(*${root}); }])>;
// Fold (cond ? x : x) -> x
-def select_same_val: GICombineRule<
+// _trivial: arms are the same register; _equiv: arms are provably equivalent.
+def select_same_val_trivial : GICombineRule<
+ (defs root:$dst),
+ (match (G_SELECT $dst, $cond, $x, $x)),
+ (apply (GIReplaceReg $dst, $x))
+>;
+def select_same_val_equiv: GICombineRule<
(defs root:$root),
- (match (wip_match_opcode G_SELECT):$root,
+ (match (G_SELECT $dst, $cond, $tval, $fval):$root,
[{ return Helper.matchSelectSameVal(*${root}); }]),
- (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, 2); }])
+ (apply (GIReplaceReg $dst, $tval))
>;
+def select_same_val : GICombineGroup<[select_same_val_trivial,
+ select_same_val_equiv]>;
// Fold (undef ? x : y) -> y
def select_undef_cmp: GICombineRule<
@@ -589,20 +611,36 @@ def select_undef_cmp: GICombineRule<
(apply (GIReplaceReg $dst, $y))
>;
-// Fold (true ? x : y) -> x
-// Fold (false ? x : y) -> y
-def select_constant_cmp: GICombineRule<
+def select_constant_cmp_false : GICombineRule<
+ (defs root:$dst),
+ (match (G_CONSTANT $cond, 0),
+ (G_SELECT $dst, $cond, $tval, $fval)),
+ (apply (GIReplaceReg $dst, $fval))
+>;
+def select_constant_cmp_true : GICombineRule<
+ (defs root:$dst),
+ (match (G_CONSTANT $cond, 1),
+ (G_SELECT $dst, $cond, $tval, $fval)),
+ (apply (GIReplaceReg $dst, $tval))
+>;
+
+def select_constant_cmp_general: GICombineRule<
(defs root:$root, unsigned_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SELECT):$root,
+ (match (G_SELECT $dst, $cond, $tval, $fval):$root,
[{ return Helper.matchConstantSelectCmp(*${root}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, ${matchinfo}); }])
>;
+def select_constant_cmp : GICombineGroup<[select_constant_cmp_false,
+ select_constant_cmp_true,
+ select_constant_cmp_general]>;
// select c, 0, x -> and (not c), x
+// Skip constant c: select_constant_cmp folds it directly, avoiding the freeze.
def select_zero_true: GICombineRule<
(defs root:$root),
(match (G_SELECT $dst, $c, 0, $x):$root,
- [{ return MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
+ [{ return !isConstantOrConstantSplatVector(${c}.getReg(), MRI) &&
+ MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
VT->computeNumSignBits(${c}.getReg()) == MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
(apply (G_XOR $xor, $c, -1),
(G_FREEZE $f, $x),
@@ -610,10 +648,12 @@ def select_zero_true: GICombineRule<
>;
// select c, x, 0 -> and c, x
+// Skip constant c: select_constant_cmp folds it directly, avoiding the freeze.
def select_zero_false: GICombineRule<
(defs root:$root),
(match (G_SELECT $dst, $c, $x, 0):$root,
- [{ return MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
+ [{ return !isConstantOrConstantSplatVector(${c}.getReg(), MRI) &&
+ MRI.getType(${c}.getReg()) == MRI.getType(${dst}.getReg()) &&
VT->computeNumSignBits(${c}.getReg()) == MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
(apply (G_FREEZE $f, $x),
(G_AND $dst, $c, $f))
@@ -813,12 +853,18 @@ def erase_undef_store : GICombineRule<
(apply [{ Helper.eraseInst(*${root}); }])
>;
-def simplify_add_to_sub_matchinfo: GIDefMatchData<"std::tuple<Register, Register>">;
-def simplify_add_to_sub: GICombineRule <
- (defs root:$root, simplify_add_to_sub_matchinfo:$info),
- (match (wip_match_opcode G_ADD):$root,
- [{ return Helper.matchSimplifyAddToSub(*${root}, ${info}); }]),
- (apply [{ Helper.applySimplifyAddToSub(*${root}, ${info});}])
+// Fold ((0-A) + B) -> B - A, (A + (0-B)) -> A - B (negation is G_SUB 0, x).
+// !foreach covers both G_ADD operand orders.
+def simplify_add_to_sub_frags : GICombinePatFrag<
+ (outs root:$dst), (ins $newlhs, $newrhs),
+ !foreach(inst, [(G_ADD $dst, $neg, $newlhs), (G_ADD $dst, $newlhs, $neg)],
+ (pattern (G_CONSTANT $zero, 0),
+ (G_SUB $neg, $zero, $newrhs),
+ inst))>;
+def simplify_add_to_sub : GICombineRule <
+ (defs root:$dst),
+ (match (simplify_add_to_sub_frags $dst, $newlhs, $newrhs)),
+ (apply (G_SUB $dst, $newlhs, $newrhs))
>;
// Fold fp_op(cst) to the constant result of the floating point operation.
@@ -873,10 +919,11 @@ def constant_fold_fp_ops : GICombineGroup<[
// Fold int2ptr(ptr2int(x)) -> x
def p2i_to_i2p: GICombineRule<
- (defs root:$root, register_matchinfo:$info),
- (match (wip_match_opcode G_INTTOPTR):$root,
- [{ return Helper.matchCombineI2PToP2I(*${root}, ${info}); }]),
- (apply [{ Helper.applyCombineI2PToP2I(*${root}, ${info}); }])
+ (defs root:$root),
+ (match (G_PTRTOINT $mid, $x),
+ (G_INTTOPTR $dst, $mid):$root,
+ [{ return MRI.getType(${x}.getReg()) == MRI.getType(${dst}.getReg()); }]),
+ (apply (GIReplaceReg $dst, $x))
>;
// Fold ptr2int(int2ptr(x)) -> x
@@ -889,20 +936,37 @@ def i2p_to_p2i: GICombineRule<
>;
// Fold add ptrtoint(x), y -> ptrtoint (ptr_add x), y
-def add_p2i_to_ptradd_matchinfo : GIDefMatchData<"std::pair<Register, bool>">;
+// !foreach covers both G_ADD operand orders; residue checks equal bitwidths.
+def add_p2i_to_ptradd_frags : GICombinePatFrag<
+ (outs root:$dst), (ins $ptr, $other),
+ !foreach(inst, [(G_ADD $dst, $x, $other), (G_ADD $dst, $other, $x)],
+ (pattern (G_PTRTOINT $x, $ptr), inst))>;
def add_p2i_to_ptradd : GICombineRule<
- (defs root:$root, add_p2i_to_ptradd_matchinfo:$info),
- (match (wip_match_opcode G_ADD):$root,
- [{ return Helper.matchCombineAddP2IToPtrAdd(*${root}, ${info}); }]),
- (apply [{ Helper.applyCombineAddP2IToPtrAdd(*${root}, ${info}); }])
+ (defs root:$dst),
+ (match (add_p2i_to_ptradd_frags $dst, $ptr, $other),
+ [{ return MRI.getType(${ptr}.getReg()).getScalarSizeInBits() ==
+ MRI.getType(${dst}.getReg()).getScalarSizeInBits(); }]),
+ (apply (G_PTR_ADD $padd, $ptr, $other),
+ (G_PTRTOINT $dst, $padd))
>;
-// Fold (ptr_add (int2ptr C1), C2) -> C1 + C2
+// Fold (ptr_add (int2ptr C1), C2) -> C1 + C2 (zext(C1)+sext(C2) folded in C++).
def const_ptradd_to_i2p: GICombineRule<
(defs root:$root, apint_matchinfo:$info),
- (match (wip_match_opcode G_PTR_ADD):$root,
- [{ return Helper.matchCombineConstPtrAddToI2P(*${root}, ${info}); }]),
- (apply [{ Helper.applyCombineConstPtrAddToI2P(*${root}, ${info}); }])
+ (match (G_CONSTANT $c1reg, $c1imm),
+ (G_INTTOPTR $base, $c1reg),
+ (G_CONSTANT $c2reg, $c2imm),
+ (G_PTR_ADD $dst, $base, $c2reg):$root,
+ [{ LLT DstTy = MRI.getType(${dst}.getReg());
+ APInt NewCst = ${c1imm}.getCImm()->getValue().zextOrTrunc(
+ DstTy.getSizeInBits());
+ NewCst += ${c2imm}.getCImm()->getValue().sextOrTrunc(
+ DstTy.getSizeInBits());
+ ${info} = NewCst;
+ return true; }]),
+ (apply [{ Helper.getBuilder().setInstrAndDebugLoc(*${root});
+ Helper.getBuilder().buildConstant(${dst}, ${info});
+ ${root}->eraseFromParent(); }])
>;
// Simplify: (logic_op (op x...), (op y...)) -> (op (logic_op x, y))
@@ -917,7 +981,7 @@ def hoist_logic_op_with_same_opcode_hands: GICombineRule <
def shl_ashr_to_sext_inreg_matchinfo : GIDefMatchData<"std::tuple<Register, int64_t>">;
def shl_ashr_to_sext_inreg : GICombineRule<
(defs root:$root, shl_ashr_to_sext_inreg_matchinfo:$info),
- (match (wip_match_opcode G_ASHR): $root,
+ (match (G_ASHR $dst, $src1, $src2): $root,
[{ return Helper.matchAshrShlToSextInreg(*${root}, ${info}); }]),
(apply [{ Helper.applyAshShlToSextInreg(*${root}, ${info});}])
>;
@@ -939,7 +1003,7 @@ def neg_and_one_to_sext_inreg : GICombineRule<
>;
// Fold and(and(x, C1), C2) -> C1&C2 ? and(x, C1&C2) : 0
-def overlapping_and: GICombineRule <
+def overlapping_and : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
(match (wip_match_opcode G_AND):$root,
[{ return Helper.matchOverlappingAnd(*${root}, ${info}); }]),
@@ -949,7 +1013,7 @@ def overlapping_and: GICombineRule <
// Fold (x & y) -> x or (x & y) -> y when (x & y) is known to equal x or equal y.
def redundant_and: GICombineRule <
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_AND):$root,
+ (match (G_AND $dst, $src1, $src2):$root,
[{ return Helper.matchRedundantAnd(*${root}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
>;
@@ -957,7 +1021,7 @@ def redundant_and: GICombineRule <
// Fold (x | y) -> x or (x | y) -> y when (x | y) is known to equal x or equal y.
def redundant_or: GICombineRule <
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_OR):$root,
+ (match (G_OR $dst, $src1, $src2):$root,
[{ return Helper.matchRedundantOr(*${root}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
>;
@@ -967,9 +1031,9 @@ def redundant_or: GICombineRule <
// if computeNumSignBits(x) >= (x.getScalarSizeInBits() - K + 1)
def redundant_sext_inreg: GICombineRule <
(defs root:$root),
- (match (wip_match_opcode G_SEXT_INREG):$root,
+ (match (G_SEXT_INREG $dst, $src, $sz):$root,
[{ return Helper.matchRedundantSExtInReg(*${root}); }]),
- (apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, 1); }])
+ (apply (GIReplaceReg $dst, $src))
>;
// Convert G_SEXT_INREG(G_ZEXT) -> G_SEXT
@@ -1012,7 +1076,7 @@ def redundant_aext_unmerge_sext_inreg : redundant_ext_unmerge_sext_inreg<G_ANYEX
// the destination type.
def anyext_trunc_fold: GICombineRule <
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_ANYEXT):$root,
+ (match (G_ANYEXT $dst, $src):$root,
[{ return Helper.matchCombineAnyExtTrunc(*${root}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
>;
@@ -1021,14 +1085,14 @@ def anyext_trunc_fold: GICombineRule <
// and truncated bits are known to be zero.
def zext_trunc_fold: GICombineRule <
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_ZEXT):$root,
+ (match (G_ZEXT $dst, $src):$root,
[{ return Helper.matchCombineZextTrunc(*${root}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${root}, ${matchinfo}); }])
>;
def not_cmp_fold : GICombineRule<
(defs root:$d, register_vector_matchinfo:$info),
- (match (wip_match_opcode G_XOR): $d,
+ (match (G_XOR $dst, $src1, $src2): $d,
[{ return Helper.matchNotCmp(*${d}, ${info}); }]),
(apply [{ Helper.applyNotCmp(*${d}, ${info}); }])
>;
@@ -1203,7 +1267,7 @@ def merge_combines: GICombineGroup<[
def trunc_shift_matchinfo : GIDefMatchData<"std::pair<MachineInstr*, LLT>">;
def trunc_shift: GICombineRule <
(defs root:$root, trunc_shift_matchinfo:$matchinfo),
- (match (wip_match_opcode G_TRUNC):$root,
+ (match (G_TRUNC $dst, $src):$root,
[{ return Helper.matchCombineTruncOfShift(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyCombineTruncOfShift(*${root}, ${matchinfo}); }])
>;
@@ -1220,7 +1284,7 @@ def xor_of_and_with_same_reg_matchinfo :
GIDefMatchData<"std::pair<Register, Register>">;
def xor_of_and_with_same_reg: GICombineRule <
(defs root:$root, xor_of_and_with_same_reg_matchinfo:$matchinfo),
- (match (wip_match_opcode G_XOR):$root,
+ (match (G_XOR $dst, $src1, $src2):$root,
[{ return Helper.matchXorOfAndWithSameReg(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyXorOfAndWithSameReg(*${root}, ${matchinfo}); }])
>;
@@ -1228,26 +1292,26 @@ def xor_of_and_with_same_reg: GICombineRule <
// Transform (ptr_add 0, x) -> (int_to_ptr x)
def ptr_add_with_zero: GICombineRule<
(defs root:$root),
- (match (wip_match_opcode G_PTR_ADD):$root,
+ (match (G_PTR_ADD $dst, $base, $offset):$root,
[{ return Helper.matchPtrAddZero(*${root}); }]),
- (apply [{ Helper.applyPtrAddZero(*${root}); }])>;
+ (apply (G_INTTOPTR $dst, $offset))>;
def combine_insert_vec_elts_build_vector : GICombineRule<
(defs root:$root, register_vector_matchinfo:$info),
- (match (wip_match_opcode G_INSERT_VECTOR_ELT):$root,
+ (match (G_INSERT_VECTOR_ELT $dst, $src, $elt, $idx):$root,
[{ return Helper.matchCombineInsertVecElts(*${root}, ${info}); }]),
(apply [{ Helper.applyCombineInsertVecElts(*${root}, ${info}); }])>;
def load_or_combine : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_OR):$root,
+ (match (G_OR $dst, $src1, $src2):$root,
[{ return Helper.matchLoadOrCombine(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
def extend_through_phis_matchdata: GIDefMatchData<"MachineInstr*">;
def extend_through_phis : GICombineRule<
(defs root:$root, extend_through_phis_matchdata:$matchinfo),
- (match (wip_match_opcode G_PHI):$root,
+ (match (G_PHI $dst, GIVariadic<>:$srcs):$root,
[{ return Helper.matchExtendThroughPhis(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyExtendThroughPhis(*${root}, ${matchinfo}); }])>;
@@ -1257,7 +1321,7 @@ def insert_vec_elt_combines : GICombineGroup<
def extract_vec_elt_build_vec : GICombineRule<
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+ (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
[{ return Helper.matchExtractVecEltBuildVec(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyExtractVecEltBuildVec(*${root}, ${matchinfo}); }])>;
@@ -1266,7 +1330,7 @@ def extract_all_elts_from_build_vector_matchinfo :
GIDefMatchData<"SmallVector<std::pair<Register, MachineInstr*>>">;
def extract_all_elts_from_build_vector : GICombineRule<
(defs root:$root, extract_all_elts_from_build_vector_matchinfo:$matchinfo),
- (match (wip_match_opcode G_BUILD_VECTOR):$root,
+ (match (G_BUILD_VECTOR $dst, GIVariadic<>:$srcs):$root,
[{ return Helper.matchExtractAllEltsFromBuildVector(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyExtractAllEltsFromBuildVector(*${root}, ${matchinfo}); }])>;
@@ -1276,22 +1340,26 @@ def extract_vec_elt_combines : GICombineGroup<[
def funnel_shift_from_or_shift : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_OR):$root,
+ (match (G_OR $dst, $src1, $src2):$root,
[{ return Helper.matchOrShiftToFunnelShift(*${root}, false, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
>;
def funnel_shift_from_or_shift_constants_are_legal : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_OR):$root,
+ (match (G_OR $dst, $src1, $src2):$root,
[{ return Helper.matchOrShiftToFunnelShift(*${root}, true, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])
>;
+def funnel_shift_op_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_FSHL, G_FSHR], (pattern (op $dst, $x, $y, $amt)))>;
+
def funnel_shift_to_rotate : GICombineRule<
- (defs root:$root),
- (match (wip_match_opcode G_FSHL, G_FSHR):$root,
+ (defs root:$dst),
+ (match (funnel_shift_op_frags $dst):$root,
[{ return Helper.matchFunnelShiftToRotate(*${root}); }]),
(apply [{ Helper.applyFunnelShiftToRotate(*${root}); }])
>;
@@ -1312,8 +1380,8 @@ def funnel_shift_left_zero: GICombineRule<
// Fold fsh(l/r) x, y, C -> fsh(l/r) x, y, C % bw
def funnel_shift_overshift: GICombineRule<
- (defs root:$root),
- (match (wip_match_opcode G_FSHL, G_FSHR):$root,
+ (defs root:$dst),
+ (match (funnel_shift_op_frags $dst):$root,
[{ return Helper.matchConstantLargerBitWidth(*${root}, 3); }]),
(apply [{ Helper.applyFunnelShiftConstantModulo(*${root}); }])
>;
@@ -1346,28 +1414,32 @@ def funnel_shift_or_shift_to_funnel_shift_right: GICombineRule<
(apply (GIReplaceReg $root, $out1))
>;
+def rotate_op_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_ROTR, G_ROTL], (pattern (op $dst, $x, $amt)))>;
+
def rotate_out_of_range : GICombineRule<
- (defs root:$root),
- (match (wip_match_opcode G_ROTR, G_ROTL):$root,
+ (defs root:$dst),
+ (match (rotate_op_frags $dst):$root,
[{ return Helper.matchRotateOutOfRange(*${root}); }]),
(apply [{ Helper.applyRotateOutOfRange(*${root}); }])
>;
def icmp_to_true_false_known_bits : GICombineRule<
(defs root:$d, int64_matchinfo:$matchinfo),
- (match (wip_match_opcode G_ICMP):$d,
+ (match (G_ICMP $dst, $pred, $src1, $src2):$d,
[{ return Helper.matchICmpToTrueFalseKnownBits(*${d}, ${matchinfo}); }]),
(apply [{ Helper.replaceInstWithConstant(*${d}, ${matchinfo}); }])>;
def icmp_to_lhs_known_bits : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_ICMP):$root,
+ (match (G_ICMP $dst, $pred, $src1, $src2):$root,
[{ return Helper.matchICmpToLHSKnownBits(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
def redundant_binop_in_equality : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_ICMP):$root,
+ (match (G_ICMP $dst, $pred, $src1, $src2):$root,
[{ return Helper.matchRedundantBinOpInEquality(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1401,7 +1473,7 @@ def double_icmp_zero_or_combine: GICombineRule<
def and_or_disjoint_mask : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_AND):$root,
+ (match (G_AND $dst, $src1, $src2):$root,
[{ return Helper.matchAndOrDisjointMask(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFnNoErase(*${root}, ${info}); }])>;
@@ -1481,19 +1553,23 @@ def funnel_shift_combines : GICombineGroup<[funnel_shift_from_or_shift,
def bitfield_extract_from_sext_inreg : GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_SEXT_INREG):$root,
+ (match (G_SEXT_INREG $dst, $src, $sz):$root,
[{ return Helper.matchBitfieldExtractFromSExtInReg(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
+def ashr_lshr_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_ASHR, G_LSHR], (pattern (op $dst, $src, $amt)))>;
+
def bitfield_extract_from_shr : GICombineRule<
- (defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_ASHR, G_LSHR):$root,
+ (defs root:$dst, build_fn_matchinfo:$info),
+ (match (ashr_lshr_frags $dst):$root,
[{ return Helper.matchBitfieldExtractFromShr(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
def bitfield_extract_from_shr_and : GICombineRule<
- (defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_ASHR, G_LSHR):$root,
+ (defs root:$dst, build_fn_matchinfo:$info),
+ (match (ashr_lshr_frags $dst):$root,
[{ return Helper.matchBitfieldExtractFromShrAnd(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1553,7 +1629,7 @@ def intrem_combines : GICombineGroup<[srem_pow2_to_mask, urem_by_const,
def reassoc_ptradd : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_PTR_ADD):$root,
+ (match (G_PTR_ADD $dst, $base, $offset):$root,
[{ return Helper.matchReassocPtrAdd(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFnNoErase(*${root}, ${matchinfo}); }])>;
@@ -1589,15 +1665,23 @@ def constant_fold_fp_binop : GICombineRule<
(apply [{ Helper.replaceInstWithFConstant(*${mi}, ${matchinfo}); }])>;
+def constant_fold_fma_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_FMAD, G_FMA], (pattern (op $dst, $src0, $src1, $src2)))>;
+
def constant_fold_fma : GICombineRule<
- (defs root:$d, constantfp_matchinfo:$matchinfo),
- (match (wip_match_opcode G_FMAD, G_FMA):$d,
+ (defs root:$dst, constantfp_matchinfo:$matchinfo),
+ (match (constant_fold_fma_frags $dst):$d,
[{ return Helper.matchConstantFoldFMA(*${d}, ${matchinfo}); }]),
(apply [{ Helper.replaceInstWithFConstant(*${d}, ${matchinfo}); }])>;
+def constant_fold_cast_op_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_ZEXT, G_SEXT, G_ANYEXT], (pattern (op $dst, $src)))>;
+
def constant_fold_cast_op : GICombineRule<
- (defs root:$d, apint_matchinfo:$matchinfo),
- (match (wip_match_opcode G_ZEXT, G_SEXT, G_ANYEXT):$d,
+ (defs root:$dst, apint_matchinfo:$matchinfo),
+ (match (constant_fold_cast_op_frags $dst):$d,
[{ return Helper.matchConstantFoldCastOp(*${d}, ${matchinfo}); }]),
(apply [{ Helper.replaceInstWithConstant(*${d}, ${matchinfo}); }])>;
@@ -1638,7 +1722,7 @@ def adde_to_addo: GICombineRule<
def mulh_to_lshr : GICombineRule<
(defs root:$root),
- (match (wip_match_opcode G_UMULH):$root,
+ (match (G_UMULH $dst, $src1, $src2):$root,
[{ return Helper.matchUMulHToLShr(*${root}); }]),
(apply [{ Helper.applyUMulHToLShr(*${root}); }])>;
@@ -1709,7 +1793,7 @@ def redundant_neg_operands : GICombineGroup<
// Transform (fsub +-0.0, X) -> (fneg X)
def fsub_to_fneg: GICombineRule<
(defs root:$root, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_FSUB):$root,
+ (match (G_FSUB $dst, $src1, $src2):$root,
[{ return Helper.matchFsubToFneg(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyFsubToFneg(*${root}, ${matchinfo}); }])>;
@@ -1719,7 +1803,7 @@ def fsub_to_fneg: GICombineRule<
// (fadd (fmul x, y), z) -> (fmad x, y, z)
def combine_fadd_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FADD):$root,
+ (match (G_FADD $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFAddFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1730,7 +1814,7 @@ def combine_fadd_fmul_to_fmad_or_fma: GICombineRule<
// -> (fmad (fpext y), (fpext z), x)
def combine_fadd_fpext_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FADD):$root,
+ (match (G_FADD $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFAddFpExtFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1741,7 +1825,7 @@ def combine_fadd_fpext_fmul_to_fmad_or_fma: GICombineRule<
// (fadd v, (fmad x, y, (fmul z, u))) -> (fmad x, y, (fmad z, u, v))
def combine_fadd_fma_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FADD):$root,
+ (match (G_FADD $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFAddFMAFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1750,7 +1834,7 @@ def combine_fadd_fma_fmul_to_fmad_or_fma: GICombineRule<
// (fma x, y, (fma (fpext u), (fpext v), z))
def combine_fadd_fpext_fma_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FADD):$root,
+ (match (G_FADD $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFAddFpExtFMulToFMadOrFMAAggressive(
*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1759,7 +1843,7 @@ def combine_fadd_fpext_fma_fmul_to_fmad_or_fma: GICombineRule<
// -> (fmad x, y, -z)
def combine_fsub_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FSUB):$root,
+ (match (G_FSUB $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFSubFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1768,7 +1852,7 @@ def combine_fsub_fmul_to_fmad_or_fma: GICombineRule<
// (fsub x, (fneg (fmul, y, z))) -> (fma y, z, x)
def combine_fsub_fneg_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FSUB):$root,
+ (match (G_FSUB $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFSubFNegFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1777,7 +1861,7 @@ def combine_fsub_fneg_fmul_to_fmad_or_fma: GICombineRule<
// (fma (fpext x), (fpext y), (fneg z))
def combine_fsub_fpext_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FSUB):$root,
+ (match (G_FSUB $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFSubFpExtFMulToFMadOrFMA(*${root},
${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1786,14 +1870,19 @@ def combine_fsub_fpext_fmul_to_fmad_or_fma: GICombineRule<
// (fneg (fma (fpext x), (fpext y), z))
def combine_fsub_fpext_fneg_fmul_to_fmad_or_fma: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_FSUB):$root,
+ (match (G_FSUB $dst, $src1, $src2):$root,
[{ return Helper.matchCombineFSubFpExtFNegFMulToFMadOrFMA(
*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
+def combine_minmax_nan_frags : GICombinePatFrag<
+ (outs root:$dst), (ins),
+ !foreach(op, [G_FMINNUM, G_FMAXNUM, G_FMINIMUM, G_FMAXIMUM],
+ (pattern (op $dst, $src0, $src1)))>;
+
def combine_minmax_nan: GICombineRule<
- (defs root:$root, unsigned_matchinfo:$info),
- (match (wip_match_opcode G_FMINNUM, G_FMAXNUM, G_FMINIMUM, G_FMAXIMUM):$root,
+ (defs root:$dst, unsigned_matchinfo:$info),
+ (match (combine_minmax_nan_frags $dst):$root,
[{ return Helper.matchCombineFMinMaxNaN(*${root}, ${info}); }]),
(apply [{ Helper.replaceSingleDefInstWithOperand(*${root}, ${info}); }])>;
@@ -1826,13 +1915,13 @@ def buildvector_identity_fold : GICombineRule<
def trunc_buildvector_fold : GICombineRule<
(defs root:$op, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_TRUNC):$op,
+ (match (G_TRUNC $dst, $src):$op,
[{ return Helper.matchTruncBuildVectorFold(*${op}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${op}, ${matchinfo}); }])>;
def trunc_lshr_buildvector_fold : GICombineRule<
(defs root:$op, register_matchinfo:$matchinfo),
- (match (wip_match_opcode G_TRUNC):$op,
+ (match (G_TRUNC $dst, $src):$op,
[{ return Helper.matchTruncLshrBuildVectorFold(*${op}, ${matchinfo}); }]),
(apply [{ Helper.replaceSingleDefInstWithReg(*${op}, ${matchinfo}); }])>;
@@ -1843,7 +1932,7 @@ def trunc_lshr_buildvector_fold : GICombineRule<
// x - (x + z) -> 0 - z
def sub_add_reg: GICombineRule <
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SUB):$root,
+ (match (G_SUB $dst, $src1, $src2):$root,
[{ return Helper.matchSubAddSameReg(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
@@ -1868,7 +1957,7 @@ def fptrunc_fpext_fold : GICombineRule<
def select_to_minmax: GICombineRule<
(defs root:$root, build_fn_matchinfo:$info),
- (match (wip_match_opcode G_SELECT):$root,
+ (match (G_SELECT $dst, $cond, $tval, $fval):$root,
[{ return Helper.matchSimplifySelectToMinMax(*${root}, ${info}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${info}); }])>;
@@ -1881,25 +1970,25 @@ def select_to_iminmax: GICombineRule<
def simplify_neg_minmax : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SUB):$root,
+ (match (G_SUB $dst, $src1, $src2):$root,
[{ return Helper.matchSimplifyNegMinMax(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
def match_selects : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SELECT):$root,
+ (match (G_SELECT $dst, $cond, $tval, $fval):$root,
[{ return Helper.matchSelect(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
def match_ands : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_AND):$root,
+ (match (G_AND $dst, $src1, $src2):$root,
[{ return Helper.matchAnd(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
def match_ors : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_OR):$root,
+ (match (G_OR $dst, $src1, $src2):$root,
[{ return Helper.matchOr(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
@@ -1926,7 +2015,7 @@ def extract_vector_element_undef : GICombineRule <
def match_extract_of_element : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_EXTRACT_VECTOR_ELT):$root,
+ (match (G_EXTRACT_VECTOR_ELT $dst, $src, $idx):$root,
[{ return Helper.matchExtractVectorElement(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
@@ -2033,14 +2122,14 @@ def nneg_zext : GICombineRule<
// Combines concat operations
def combine_concat_vector : GICombineRule<
(defs root:$root, register_vector_matchinfo:$matchinfo),
- (match (wip_match_opcode G_CONCAT_VECTORS):$root,
+ (match (G_CONCAT_VECTORS $dst, GIVariadic<>:$srcs):$root,
[{ return Helper.matchCombineConcatVectors(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyCombineConcatVectors(*${root}, ${matchinfo}); }])>;
// Combines shuffle operations
def combine_shuffle_vector : GICombineRule<
(defs root:$root, register_vector_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+ (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
[{ return Helper.matchCombineShuffleVector(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyCombineShuffleVector(*${root}, ${matchinfo}); }])>;
@@ -2052,7 +2141,7 @@ def combine_shuffle_vector : GICombineRule<
// c = G_CONCAT_VECTORS x, y, z, undef
def combine_shuffle_concat : GICombineRule<
(defs root:$root, register_vector_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+ (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
[{ return Helper.matchCombineShuffleConcat(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyCombineShuffleConcat(*${root}, ${matchinfo}); }])>;
@@ -2090,7 +2179,7 @@ def insert_vector_element_extract_vector_element : GICombineRule<
def insert_vector_elt_oob : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_INSERT_VECTOR_ELT):$root,
+ (match (G_INSERT_VECTOR_ELT $dst, $src, $elt, $idx):$root,
[{ return Helper.matchInsertVectorElementOOB(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])>;
@@ -2149,7 +2238,7 @@ def combine_shuffle_undef_rhs : GICombineRule<
def combine_shuffle_disjoint_mask : GICombineRule<
(defs root:$root, build_fn_matchinfo:$matchinfo),
- (match (wip_match_opcode G_SHUFFLE_VECTOR):$root,
+ (match (G_SHUFFLE_VECTOR $dst, $src1, $src2, $mask):$root,
[{ return Helper.matchShuffleDisjointMask(*${root}, ${matchinfo}); }]),
(apply [{ Helper.applyBuildFn(*${root}, ${matchinfo}); }])
>;
diff --git a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
index 0ee8a9a204c09..cd672ed0b6f53 100644
--- a/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
+++ b/llvm/lib/CodeGen/GlobalISel/CombinerHelper.cpp
@@ -1111,12 +1111,6 @@ bool CombinerHelper::matchSextTruncSextLoad(MachineInstr &MI) const {
return false;
}
-void CombinerHelper::applySextTruncSextLoad(MachineInstr &MI) const {
- assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
- Builder.buildCopy(MI.getOperand(0).getReg(), MI.getOperand(1).getReg());
- MI.eraseFromParent();
-}
-
bool CombinerHelper::matchSextInRegOfLoad(
MachineInstr &MI, std::tuple<Register, unsigned> &MatchInfo) const {
assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
@@ -2129,40 +2123,10 @@ void CombinerHelper::applyShiftOfShiftedLogic(
MI.eraseFromParent();
}
-bool CombinerHelper::matchCommuteShift(MachineInstr &MI,
- BuildFnTy &MatchInfo) const {
- assert(MI.getOpcode() == TargetOpcode::G_SHL && "Expected G_SHL");
- // Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
- // Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
- auto &Shl = cast<GenericMachineInstr>(MI);
- Register DstReg = Shl.getReg(0);
- Register SrcReg = Shl.getReg(1);
- Register ShiftReg = Shl.getReg(2);
- Register X, C1;
-
- if (!getTargetLowering().isDesirableToCommuteWithShift(MI, !isPreLegalize()))
- return false;
-
- MachineInstr *SrcDef;
- if (!mi_match(SrcReg, MRI,
- m_OneNonDBGUse(m_any_of(m_GAdd(m_Reg(X), m_Reg(C1)),
- m_GOr(m_Reg(X), m_Reg(C1))))) ||
- !mi_match(SrcReg, MRI, m_MInstr(SrcDef)))
- return false;
-
- APInt C1Val, C2Val;
- if (!mi_match(C1, MRI, m_ICstOrSplat(C1Val)) ||
- !mi_match(ShiftReg, MRI, m_ICstOrSplat(C2Val)))
- return false;
-
- unsigned SrcOpc = SrcDef->getOpcode();
- LLT SrcTy = MRI.getType(SrcReg);
- MatchInfo = [=](MachineIRBuilder &B) {
- auto S1 = B.buildShl(SrcTy, X, ShiftReg);
- auto S2 = B.buildShl(SrcTy, C1, ShiftReg);
- B.buildInstr(SrcOpc, {DstReg}, {S1, S2});
- };
- return true;
+bool CombinerHelper::isDesirableToCommuteWithShift(
+ const MachineInstr &MI) const {
+ return getTargetLowering().isDesirableToCommuteWithShift(MI,
+ !isPreLegalize());
}
bool CombinerHelper::matchLshrOfTruncOfLshr(MachineInstr &MI,
@@ -2651,24 +2615,6 @@ bool CombinerHelper::tryCombineShiftToUnmerge(
return false;
}
-bool CombinerHelper::matchCombineI2PToP2I(MachineInstr &MI,
- Register &Reg) const {
- assert(MI.getOpcode() == TargetOpcode::G_INTTOPTR && "Expected a G_INTTOPTR");
- Register DstReg = MI.getOperand(0).getReg();
- LLT DstTy = MRI.getType(DstReg);
- Register SrcReg = MI.getOperand(1).getReg();
- return mi_match(SrcReg, MRI,
- m_GPtrToInt(m_all_of(m_SpecificType(DstTy), m_Reg(Reg))));
-}
-
-void CombinerHelper::applyCombineI2PToP2I(MachineInstr &MI,
- Register &Reg) const {
- assert(MI.getOpcode() == TargetOpcode::G_INTTOPTR && "Expected a G_INTTOPTR");
- Register DstReg = MI.getOperand(0).getReg();
- Builder.buildCopy(DstReg, Reg);
- MI.eraseFromParent();
-}
-
void CombinerHelper::applyCombineP2IToI2P(MachineInstr &MI,
Register &Reg) const {
assert(MI.getOpcode() == TargetOpcode::G_PTRTOINT && "Expected a G_PTRTOINT");
@@ -3992,12 +3938,6 @@ bool CombinerHelper::matchPtrAddZero(MachineInstr &MI) const {
return isBuildVectorAllZeros(*VecMI, MRI);
}
-void CombinerHelper::applyPtrAddZero(MachineInstr &MI) const {
- auto &PtrAdd = cast<GPtrAdd>(MI);
- Builder.buildIntToPtr(PtrAdd.getReg(0), PtrAdd.getOffsetReg());
- PtrAdd.eraseFromParent();
-}
-
/// The second source operand is known to be a power of 2.
void CombinerHelper::applySimplifyURemByPow2(MachineInstr &MI) const {
Register DstReg = MI.getOperand(0).getReg();
@@ -4959,8 +4899,13 @@ bool CombinerHelper::matchBitfieldExtractFromAnd(MachineInstr &MI,
m_ICst(AndImm))))
return false;
+ // AndImm is sign-extended to 64 bits by m_ICst; restrict it to the operand
+ // width so an all-ones mask (a redundant AND) is not misread as a wider mask.
+ uint64_t MaybeMask = static_cast<uint64_t>(AndImm);
+ if (Size < 64)
+ MaybeMask &= maskTrailingOnes<uint64_t>(Size);
+
// The mask is a mask of the low bits iff imm & (imm+1) == 0.
- auto MaybeMask = static_cast<uint64_t>(AndImm);
if (MaybeMask & (MaybeMask + 1))
return false;
@@ -4968,7 +4913,14 @@ bool CombinerHelper::matchBitfieldExtractFromAnd(MachineInstr &MI,
if (static_cast<uint64_t>(LSBImm) >= Size)
return false;
- uint64_t Width = APInt(Size, AndImm).countr_one();
+ uint64_t Width = APInt(Size, MaybeMask).countr_one();
+ // The extracted field [LSB, LSB+Width) must fit within the register.
+ // Otherwise this is a redundant AND (e.g. an all-ones mask combined with a
+ // non-zero shift) that is better handled by other combines, and would form
+ // an out-of-range bitfield extract.
+ if (static_cast<uint64_t>(LSBImm) + Width > Size)
+ return false;
+
MatchInfo = [=](MachineIRBuilder &B) {
auto WidthCst = B.buildConstant(ExtractTy, Width);
auto LSBCst = B.buildConstant(ExtractTy, LSBImm);
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir b/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
index a9adf7a2e46ae..f4bcf38670196 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/combine-insert-vec-elt.mir
@@ -229,9 +229,8 @@ body: |
; CHECK: liveins: $x0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[C:%[0-9]+]]:_(i8) = G_CONSTANT i8 127
- ; CHECK-NEXT: [[DEF:%[0-9]+]]:_(i8) = G_IMPLICIT_DEF
+ ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
- ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[DEF]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
; CHECK-NEXT: G_STORE [[BUILD_VECTOR]](<32 x i8>), [[COPY]](p0) :: (store (<32 x i8>))
; CHECK-NEXT: RET_ReallyLR
%3:_(i8) = G_CONSTANT i8 127
@@ -253,9 +252,8 @@ body: |
; CHECK: liveins: $x0
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[C:%[0-9]+]]:_(i8) = G_CONSTANT i8 127
- ; CHECK-NEXT: [[DEF:%[0-9]+]]:_(i8) = G_IMPLICIT_DEF
+ ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
- ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<32 x i8>) = G_BUILD_VECTOR [[C]](i8), [[C]](i8), [[C]](i8), [[DEF]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8), [[C]](i8)
; CHECK-NEXT: G_STORE [[BUILD_VECTOR]](<32 x i8>), [[COPY]](p0) :: (store (<32 x i8>))
; CHECK-NEXT: RET_ReallyLR
%3:_(i8) = G_CONSTANT i8 127
@@ -319,13 +317,13 @@ body: |
; CHECK-LABEL: name: test_inlineasm_base
; CHECK: liveins: $x0
; CHECK-NEXT: {{ $}}
- ; CHECK-NEXT: [[CONST1:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
- ; CHECK-NEXT: [[CONST2:%[0-9]+]]:_(i32) = G_CONSTANT i32 42
+ ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
+ ; CHECK-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 42
; CHECK-NEXT: [[DEF:%[0-9]+]]:gpr64 = IMPLICIT_DEF
; CHECK-NEXT: INLINEASM &"ldr $0, [$1]", sideeffect attdialect, regdef:FPR128, def %3(<4 x i32>), reguse:GPR64, [[DEF]]
- ; CHECK-NEXT: [[INSERT_VECTOR_ELT:%[0-9]+]]:_(<4 x i32>) = G_INSERT_VECTOR_ELT %3, [[CONST2]](i32), [[CONST1]](i64)
+ ; CHECK-NEXT: [[IVEC:%[0-9]+]]:_(<4 x i32>) = G_INSERT_VECTOR_ELT %3, [[C1]](i32), [[C]](i64)
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
- ; CHECK-NEXT: G_STORE [[INSERT_VECTOR_ELT]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
+ ; CHECK-NEXT: G_STORE [[IVEC]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
; CHECK-NEXT: RET_ReallyLR
%0:_(i64) = G_CONSTANT i64 0
%1:_(i32) = G_CONSTANT i32 42
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir b/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
index 3abd11b3d5762..e5c014edc4d60 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/form-bitfield-extract-from-and.mir
@@ -286,8 +286,9 @@ body: |
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: %x:_(i64) = COPY $x0
; CHECK-NEXT: %lsb:_(i64) = G_CONSTANT i64 5
- ; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 64
- ; CHECK-NEXT: %and:_(i64) = G_UBFX %x, %lsb(i64), [[C]]
+ ; CHECK-NEXT: %mask:_(i64) = G_CONSTANT i64 -1
+ ; CHECK-NEXT: %shift:_(i64) = G_LSHR %x, %lsb(i64)
+ ; CHECK-NEXT: %and:_(i64) = G_AND %shift, %mask
; CHECK-NEXT: $x0 = COPY %and(i64)
; CHECK-NEXT: RET_ReallyLR implicit $x0
%x:_(i64) = COPY $x0
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
index b66ed3b50feb7..c9e8adfeaa94a 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizer-combiner-divrem-insertpt-crash.mir
@@ -16,7 +16,6 @@ body: |
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $x0
; CHECK-NEXT: [[DEF:%[0-9]+]]:_(i1) = G_IMPLICIT_DEF
- ; CHECK-NEXT: [[DEF1:%[0-9]+]]:_(i64) = G_IMPLICIT_DEF
; CHECK-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 0
; CHECK-NEXT: G_BRCOND [[DEF]](i1), %bb.2
; CHECK-NEXT: G_BR %bb.1
@@ -24,8 +23,8 @@ body: |
; CHECK-NEXT: bb.1:
; CHECK-NEXT: successors: %bb.2(0x80000000)
; CHECK-NEXT: {{ $}}
- ; CHECK-NEXT: [[FREEZE:%[0-9]+]]:_(i64) = G_FREEZE [[DEF1]]
- ; CHECK-NEXT: [[UDIV:%[0-9]+]]:_(i64) = G_UDIV [[FREEZE]], [[C]]
+ ; CHECK-NEXT: [[C1:%[0-9]+]]:_(i64) = G_CONSTANT i64 -1
+ ; CHECK-NEXT: [[UDIV:%[0-9]+]]:_(i64) = G_UDIV [[C1]], [[C]]
; CHECK-NEXT: G_STORE [[UDIV]](i64), [[COPY]](p0) :: (store (i64))
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: bb.2:
diff --git a/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll b/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
index 128b67663bf04..93c22573c0511 100644
--- a/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
+++ b/llvm/test/CodeGen/AArch64/neon-shuffle-vector-tbl.ll
@@ -559,7 +559,7 @@ define <8 x i8> @no_shuffle_only_some_and_constants(<8 x i8> %src, <8 x i8> %mas
; CHECK-GI-NEXT: mov x8, sp
; CHECK-GI-NEXT: str d0, [sp]
; CHECK-GI-NEXT: and w9, w9, #0x7
-; CHECK-GI-NEXT: and x9, x9, #0xff
+; CHECK-GI-NEXT: and x9, x9, #0x7
; CHECK-GI-NEXT: lsl x11, x9, #1
; CHECK-GI-NEXT: sub x9, x11, x9
; CHECK-GI-NEXT: ldr b2, [x8, x9]
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
index 5a8027bf75881..8a4d6c1d67b4d 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/ashr.ll
@@ -1024,25 +1024,25 @@ define <2 x float> @v_ashr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
; GFX6-LABEL: v_ashr_v4i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v4, 16, v2
-; GFX6-NEXT: v_bfe_i32 v6, v0, 0, 16
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT: v_bfe_i32 v5, v0, 0, 16
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
; GFX6-NEXT: v_bfe_i32 v0, v0, 16, 16
-; GFX6-NEXT: v_lshrrev_b32_e32 v5, 16, v3
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_ashrrev_i32_e32 v0, v4, v0
-; GFX6-NEXT: v_bfe_i32 v4, v1, 0, 16
+; GFX6-NEXT: v_ashrrev_i32_e32 v4, v4, v5
+; GFX6-NEXT: v_ashrrev_i32_e32 v0, v2, v0
+; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT: v_bfe_i32 v5, v1, 0, 16
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
; GFX6-NEXT: v_bfe_i32 v1, v1, 16, 16
-; GFX6-NEXT: v_ashrrev_i32_e32 v2, v2, v6
-; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT: v_ashrrev_i32_e32 v1, v5, v1
+; GFX6-NEXT: v_ashrrev_i32_e32 v1, v3, v1
+; GFX6-NEXT: v_ashrrev_i32_e32 v2, v2, v5
; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT: v_ashrrev_i32_e32 v3, v3, v4
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT: v_or_b32_e32 v0, v2, v0
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v4
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT: v_or_b32_e32 v0, v3, v0
; GFX6-NEXT: v_or_b32_e32 v1, v2, v1
; GFX6-NEXT: s_setpc_b64 s[30:31]
;
@@ -1078,23 +1078,23 @@ define <2 x float> @v_ashr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
define amdgpu_ps <2 x i32> @s_ashr_v4i16(<4 x i16> inreg %value, <4 x i16> inreg %amount) {
; GFX6-LABEL: s_ashr_v4i16:
; GFX6: ; %bb.0:
-; GFX6-NEXT: s_lshr_b32 s4, s2, 16
-; GFX6-NEXT: s_sext_i32_i16 s6, s0
+; GFX6-NEXT: s_sext_i32_i16 s4, s0
+; GFX6-NEXT: s_ashr_i32 s4, s4, s2
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
; GFX6-NEXT: s_bfe_i32 s0, s0, 0x100010
-; GFX6-NEXT: s_lshr_b32 s5, s3, 16
-; GFX6-NEXT: s_ashr_i32 s0, s0, s4
-; GFX6-NEXT: s_sext_i32_i16 s4, s1
+; GFX6-NEXT: s_ashr_i32 s0, s0, s2
+; GFX6-NEXT: s_sext_i32_i16 s2, s1
+; GFX6-NEXT: s_ashr_i32 s2, s2, s3
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
; GFX6-NEXT: s_bfe_i32 s1, s1, 0x100010
-; GFX6-NEXT: s_ashr_i32 s2, s6, s2
-; GFX6-NEXT: s_ashr_i32 s1, s1, s5
+; GFX6-NEXT: s_ashr_i32 s1, s1, s3
; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT: s_ashr_i32 s3, s4, s3
-; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT: s_lshl_b32 s0, s0, 16
; GFX6-NEXT: s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT: s_or_b32 s0, s2, s0
-; GFX6-NEXT: s_and_b32 s2, s3, 0xffff
+; GFX6-NEXT: s_and_b32 s3, s4, 0xffff
+; GFX6-NEXT: s_lshl_b32 s0, s0, 16
+; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
; GFX6-NEXT: s_lshl_b32 s1, s1, 16
+; GFX6-NEXT: s_or_b32 s0, s3, s0
; GFX6-NEXT: s_or_b32 s1, s2, s1
; GFX6-NEXT: ; return to shader part epilog
;
@@ -1189,45 +1189,45 @@ define <4 x float> @v_ashr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
; GFX6-LABEL: v_ashr_v8i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v4
-; GFX6-NEXT: v_bfe_i32 v12, v0, 0, 16
+; GFX6-NEXT: v_and_b32_e32 v8, 0xffff, v4
+; GFX6-NEXT: v_bfe_i32 v9, v0, 0, 16
+; GFX6-NEXT: v_bfe_u32 v4, v4, 16, 16
; GFX6-NEXT: v_bfe_i32 v0, v0, 16, 16
-; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v5
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT: v_ashrrev_i32_e32 v0, v8, v0
-; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v5
-; GFX6-NEXT: v_bfe_i32 v8, v1, 0, 16
+; GFX6-NEXT: v_ashrrev_i32_e32 v8, v8, v9
+; GFX6-NEXT: v_ashrrev_i32_e32 v0, v4, v0
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT: v_bfe_i32 v9, v1, 0, 16
+; GFX6-NEXT: v_bfe_u32 v5, v5, 16, 16
; GFX6-NEXT: v_bfe_i32 v1, v1, 16, 16
-; GFX6-NEXT: v_lshrrev_b32_e32 v10, 16, v6
-; GFX6-NEXT: v_ashrrev_i32_e32 v4, v4, v12
-; GFX6-NEXT: v_ashrrev_i32_e32 v5, v5, v8
-; GFX6-NEXT: v_ashrrev_i32_e32 v1, v9, v1
-; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v6
-; GFX6-NEXT: v_bfe_i32 v8, v2, 0, 16
+; GFX6-NEXT: v_ashrrev_i32_e32 v4, v4, v9
+; GFX6-NEXT: v_ashrrev_i32_e32 v1, v5, v1
+; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v6
+; GFX6-NEXT: v_bfe_i32 v9, v2, 0, 16
+; GFX6-NEXT: v_bfe_u32 v6, v6, 16, 16
; GFX6-NEXT: v_bfe_i32 v2, v2, 16, 16
-; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v11, 16, v7
-; GFX6-NEXT: v_ashrrev_i32_e32 v6, v6, v8
-; GFX6-NEXT: v_ashrrev_i32_e32 v2, v10, v2
-; GFX6-NEXT: v_bfe_i32 v8, v3, 0, 16
+; GFX6-NEXT: v_ashrrev_i32_e32 v5, v5, v9
+; GFX6-NEXT: v_ashrrev_i32_e32 v2, v6, v2
+; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v7
+; GFX6-NEXT: v_bfe_i32 v9, v3, 0, 16
+; GFX6-NEXT: v_bfe_u32 v7, v7, 16, 16
; GFX6-NEXT: v_bfe_i32 v3, v3, 16, 16
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT: v_and_b32_e32 v7, 0xffff, v7
-; GFX6-NEXT: v_ashrrev_i32_e32 v3, v11, v3
-; GFX6-NEXT: v_or_b32_e32 v0, v4, v0
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT: v_ashrrev_i32_e32 v3, v7, v3
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
; GFX6-NEXT: v_lshlrev_b32_e32 v1, 16, v1
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_ashrrev_i32_e32 v7, v7, v8
+; GFX6-NEXT: v_ashrrev_i32_e32 v6, v6, v9
+; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX6-NEXT: v_or_b32_e32 v1, v4, v1
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v6
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v5
; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT: v_and_b32_e32 v7, 0xffff, v8
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
; GFX6-NEXT: v_or_b32_e32 v2, v4, v2
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v7
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v6
; GFX6-NEXT: v_lshlrev_b32_e32 v3, 16, v3
+; GFX6-NEXT: v_or_b32_e32 v0, v7, v0
; GFX6-NEXT: v_or_b32_e32 v3, v4, v3
; GFX6-NEXT: s_setpc_b64 s[30:31]
;
@@ -1273,41 +1273,41 @@ define <4 x float> @v_ashr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
define amdgpu_ps <4 x i32> @s_ashr_v8i16(<8 x i16> inreg %value, <8 x i16> inreg %amount) {
; GFX6-LABEL: s_ashr_v8i16:
; GFX6: ; %bb.0:
-; GFX6-NEXT: s_lshr_b32 s8, s4, 16
-; GFX6-NEXT: s_sext_i32_i16 s12, s0
+; GFX6-NEXT: s_sext_i32_i16 s8, s0
+; GFX6-NEXT: s_ashr_i32 s8, s8, s4
+; GFX6-NEXT: s_bfe_u32 s4, s4, 0x100010
; GFX6-NEXT: s_bfe_i32 s0, s0, 0x100010
-; GFX6-NEXT: s_lshr_b32 s9, s5, 16
-; GFX6-NEXT: s_ashr_i32 s0, s0, s8
-; GFX6-NEXT: s_sext_i32_i16 s8, s1
+; GFX6-NEXT: s_ashr_i32 s0, s0, s4
+; GFX6-NEXT: s_sext_i32_i16 s4, s1
+; GFX6-NEXT: s_ashr_i32 s4, s4, s5
+; GFX6-NEXT: s_bfe_u32 s5, s5, 0x100010
; GFX6-NEXT: s_bfe_i32 s1, s1, 0x100010
-; GFX6-NEXT: s_lshr_b32 s10, s6, 16
-; GFX6-NEXT: s_ashr_i32 s4, s12, s4
-; GFX6-NEXT: s_ashr_i32 s5, s8, s5
-; GFX6-NEXT: s_ashr_i32 s1, s1, s9
-; GFX6-NEXT: s_sext_i32_i16 s8, s2
+; GFX6-NEXT: s_ashr_i32 s1, s1, s5
+; GFX6-NEXT: s_sext_i32_i16 s5, s2
+; GFX6-NEXT: s_ashr_i32 s5, s5, s6
+; GFX6-NEXT: s_bfe_u32 s6, s6, 0x100010
; GFX6-NEXT: s_bfe_i32 s2, s2, 0x100010
-; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT: s_lshr_b32 s11, s7, 16
-; GFX6-NEXT: s_ashr_i32 s6, s8, s6
-; GFX6-NEXT: s_ashr_i32 s2, s2, s10
-; GFX6-NEXT: s_sext_i32_i16 s8, s3
+; GFX6-NEXT: s_ashr_i32 s2, s2, s6
+; GFX6-NEXT: s_sext_i32_i16 s6, s3
+; GFX6-NEXT: s_ashr_i32 s6, s6, s7
+; GFX6-NEXT: s_bfe_u32 s7, s7, 0x100010
; GFX6-NEXT: s_bfe_i32 s3, s3, 0x100010
-; GFX6-NEXT: s_and_b32 s4, s4, 0xffff
-; GFX6-NEXT: s_lshl_b32 s0, s0, 16
; GFX6-NEXT: s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT: s_ashr_i32 s3, s3, s11
-; GFX6-NEXT: s_or_b32 s0, s4, s0
-; GFX6-NEXT: s_and_b32 s4, s5, 0xffff
+; GFX6-NEXT: s_ashr_i32 s3, s3, s7
+; GFX6-NEXT: s_and_b32 s4, s4, 0xffff
; GFX6-NEXT: s_lshl_b32 s1, s1, 16
; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT: s_ashr_i32 s7, s8, s7
+; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
; GFX6-NEXT: s_or_b32 s1, s4, s1
-; GFX6-NEXT: s_and_b32 s4, s6, 0xffff
+; GFX6-NEXT: s_and_b32 s4, s5, 0xffff
; GFX6-NEXT: s_lshl_b32 s2, s2, 16
; GFX6-NEXT: s_and_b32 s3, s3, 0xffff
+; GFX6-NEXT: s_and_b32 s7, s8, 0xffff
+; GFX6-NEXT: s_lshl_b32 s0, s0, 16
; GFX6-NEXT: s_or_b32 s2, s4, s2
-; GFX6-NEXT: s_and_b32 s4, s7, 0xffff
+; GFX6-NEXT: s_and_b32 s4, s6, 0xffff
; GFX6-NEXT: s_lshl_b32 s3, s3, 16
+; GFX6-NEXT: s_or_b32 s0, s7, s0
; GFX6-NEXT: s_or_b32 s3, s4, s3
; GFX6-NEXT: ; return to shader part epilog
;
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
index c595f8b085739..d4b1b75fa82a9 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-extract-vector-load.mir
@@ -8,9 +8,8 @@ tracksRegLiveness: true
body: |
bb.0:
; CHECK-LABEL: name: test_ptradd_crash__offset_smaller
- ; CHECK: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 12
- ; CHECK-NEXT: [[INTTOPTR:%[0-9]+]]:_(p1) = G_INTTOPTR [[C]](s64)
- ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[INTTOPTR]](p1) :: (load (s32), addrspace 1)
+ ; CHECK: [[C:%[0-9]+]]:_(p1) = G_CONSTANT i64 12
+ ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[C]](p1) :: (load (s32), addrspace 1)
; CHECK-NEXT: $sgpr0 = COPY [[LOAD]](s32)
; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
%1:_(p1) = G_CONSTANT i64 0
@@ -28,9 +27,8 @@ tracksRegLiveness: true
body: |
bb.0:
; CHECK-LABEL: name: test_ptradd_crash__offset_wider
- ; CHECK: [[C:%[0-9]+]]:_(s64) = G_CONSTANT i64 12
- ; CHECK-NEXT: [[INTTOPTR:%[0-9]+]]:_(p1) = G_INTTOPTR [[C]](s64)
- ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[INTTOPTR]](p1) :: (load (s32), addrspace 1)
+ ; CHECK: [[C:%[0-9]+]]:_(p1) = G_CONSTANT i64 12
+ ; CHECK-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[C]](p1) :: (load (s32), addrspace 1)
; CHECK-NEXT: $sgpr0 = COPY [[LOAD]](s32)
; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
%1:_(p1) = G_CONSTANT i64 0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
index cb6de736d13e9..70bcd3d95c662 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-redundant-and.mir
@@ -110,8 +110,9 @@ body: |
; CHECK-NEXT: {{ $}}
; CHECK-NEXT: [[COPY:%[0-9]+]]:_(s32) = COPY $vgpr0
; CHECK-NEXT: [[C:%[0-9]+]]:_(s32) = G_CONSTANT i32 5
- ; CHECK-NEXT: [[LSHR:%[0-9]+]]:_(s32) = G_LSHR [[COPY]], [[C]](s32)
- ; CHECK-NEXT: $vgpr0 = COPY [[LSHR]](s32)
+ ; CHECK-NEXT: [[C1:%[0-9]+]]:_(s32) = G_CONSTANT i32 27
+ ; CHECK-NEXT: [[UBFX:%[0-9]+]]:_(s32) = G_UBFX [[COPY]], [[C]](s32), [[C1]]
+ ; CHECK-NEXT: $vgpr0 = COPY [[UBFX]](s32)
; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $vgpr0
%0:_(s32) = COPY $vgpr0
%1:_(s32) = G_CONSTANT i32 5
@@ -153,8 +154,9 @@ tracksRegLiveness: true
body: |
bb.0:
; CHECK-LABEL: name: test_sext_inreg
- ; CHECK: %cst_1:_(s32) = G_CONSTANT i32 -5
- ; CHECK-NEXT: $sgpr0 = COPY %cst_1(s32)
+ ; CHECK: %cst_11:_(s32) = G_CONSTANT i32 11
+ ; CHECK-NEXT: %sext_inreg_11:_(s32) = G_SEXT_INREG %cst_11, 4
+ ; CHECK-NEXT: $sgpr0 = COPY %sext_inreg_11(s32)
; CHECK-NEXT: SI_RETURN_TO_EPILOG implicit $sgpr0
%cst_1:_(s32) = G_CONSTANT i32 -5
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
index 98de0a416e5b9..ce1112d8b496a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/combine-shift-of-shifted-logic.ll
@@ -41,7 +41,7 @@ define amdgpu_cs i32 @test_shl_and_3(i32 inreg %arg1) {
define amdgpu_cs i32 @test_lshr_and_1(i32 inreg %arg1) {
; CHECK-LABEL: test_lshr_and_1:
; CHECK: ; %bb.0: ; %.entry
-; CHECK-NEXT: s_lshr_b32 s0, s0, 4
+; CHECK-NEXT: s_bfe_u32 s0, s0, 0x1c0004
; CHECK-NEXT: ; return to shader part epilog
.entry:
%z1 = lshr i32 %arg1, 2
@@ -66,7 +66,7 @@ define amdgpu_cs i32 @test_lshr_and_2(i32 inreg %arg1) {
define amdgpu_cs i32 @test_lshr_and_3(i32 inreg %arg1) {
; CHECK-LABEL: test_lshr_and_3:
; CHECK: ; %bb.0: ; %.entry
-; CHECK-NEXT: s_lshr_b32 s0, s0, 5
+; CHECK-NEXT: s_bfe_u32 s0, s0, 0x1b0005
; CHECK-NEXT: ; return to shader part epilog
.entry:
%z1 = lshr i32 %arg1, 3
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
index dd6e3bd7ebb70..270626c35ba60 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshl.ll
@@ -4822,10 +4822,11 @@ define amdgpu_ps i48 @s_fshl_v3i16(<3 x i16> inreg %lhs, <3 x i16> inreg %rhs, <
; GFX6-NEXT: s_lshl_b32 s0, s0, s8
; GFX6-NEXT: s_bfe_u32 s8, s2, 0xf0001
; GFX6-NEXT: s_lshr_b32 s4, s8, s4
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
; GFX6-NEXT: s_or_b32 s0, s0, s4
; GFX6-NEXT: s_and_b32 s4, s7, 15
; GFX6-NEXT: s_andn2_b32 s7, 15, s7
-; GFX6-NEXT: s_lshr_b32 s2, s2, 17
+; GFX6-NEXT: s_lshr_b32 s2, s2, 1
; GFX6-NEXT: s_lshl_b32 s4, s6, s4
; GFX6-NEXT: s_lshr_b32 s2, s2, s7
; GFX6-NEXT: s_or_b32 s2, s4, s2
@@ -5075,8 +5076,9 @@ define <3 x half> @v_fshl_v3i16(<3 x i16> %lhs, <3 x i16> %rhs, <3 x i16> %amt)
; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
; GFX6-NEXT: v_and_b32_e32 v4, 15, v7
; GFX6-NEXT: v_xor_b32_e32 v7, -1, v7
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
; GFX6-NEXT: v_and_b32_e32 v7, 15, v7
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, 17, v2
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, 1, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v6
; GFX6-NEXT: v_lshrrev_b32_e32 v2, v7, v2
; GFX6-NEXT: v_or_b32_e32 v2, v4, v2
@@ -5231,10 +5233,11 @@ define amdgpu_ps <2 x i32> @s_fshl_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
; GFX6-NEXT: s_lshl_b32 s0, s0, s10
; GFX6-NEXT: s_bfe_u32 s10, s2, 0xf0001
; GFX6-NEXT: s_lshr_b32 s4, s10, s4
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
; GFX6-NEXT: s_or_b32 s0, s0, s4
; GFX6-NEXT: s_and_b32 s4, s8, 15
; GFX6-NEXT: s_andn2_b32 s8, 15, s8
-; GFX6-NEXT: s_lshr_b32 s2, s2, 17
+; GFX6-NEXT: s_lshr_b32 s2, s2, 1
; GFX6-NEXT: s_lshl_b32 s4, s6, s4
; GFX6-NEXT: s_lshr_b32 s2, s2, s8
; GFX6-NEXT: s_or_b32 s2, s4, s2
@@ -5245,10 +5248,11 @@ define amdgpu_ps <2 x i32> @s_fshl_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
; GFX6-NEXT: s_lshl_b32 s1, s1, s4
; GFX6-NEXT: s_bfe_u32 s4, s3, 0xf0001
; GFX6-NEXT: s_lshr_b32 s4, s4, s5
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
; GFX6-NEXT: s_or_b32 s1, s1, s4
; GFX6-NEXT: s_and_b32 s4, s9, 15
; GFX6-NEXT: s_andn2_b32 s5, 15, s9
-; GFX6-NEXT: s_lshr_b32 s3, s3, 17
+; GFX6-NEXT: s_lshr_b32 s3, s3, 1
; GFX6-NEXT: s_lshl_b32 s4, s7, s4
; GFX6-NEXT: s_lshr_b32 s3, s3, s5
; GFX6-NEXT: s_and_b32 s2, 0xffff, s2
@@ -5449,8 +5453,9 @@ define <4 x half> @v_fshl_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
; GFX6-NEXT: v_and_b32_e32 v4, 15, v8
; GFX6-NEXT: v_xor_b32_e32 v8, -1, v8
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
; GFX6-NEXT: v_and_b32_e32 v8, 15, v8
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, 17, v2
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, 1, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v6
; GFX6-NEXT: v_lshrrev_b32_e32 v2, v8, v2
; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v5
@@ -5463,10 +5468,11 @@ define <4 x half> @v_fshl_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
; GFX6-NEXT: v_bfe_u32 v4, v3, 1, 15
; GFX6-NEXT: v_lshrrev_b32_e32 v4, v5, v4
; GFX6-NEXT: v_xor_b32_e32 v5, -1, v9
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
; GFX6-NEXT: v_or_b32_e32 v1, v1, v4
; GFX6-NEXT: v_and_b32_e32 v4, 15, v9
; GFX6-NEXT: v_and_b32_e32 v5, 15, v5
-; GFX6-NEXT: v_lshrrev_b32_e32 v3, 17, v3
+; GFX6-NEXT: v_lshrrev_b32_e32 v3, 1, v3
; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v7
; GFX6-NEXT: v_lshrrev_b32_e32 v3, v5, v3
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
index 6bd74d89deb57..1b9d573c60aa4 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/fshr.ll
@@ -4522,21 +4522,21 @@ define amdgpu_ps i48 @s_fshr_v3i16(<3 x i16> inreg %lhs, <3 x i16> inreg %rhs, <
; GFX6-LABEL: s_fshr_v3i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_lshr_b32 s6, s0, 16
-; GFX6-NEXT: s_lshr_b32 s7, s2, 16
-; GFX6-NEXT: s_lshr_b32 s8, s4, 16
-; GFX6-NEXT: s_and_b32 s9, s4, 15
+; GFX6-NEXT: s_lshr_b32 s7, s4, 16
+; GFX6-NEXT: s_and_b32 s8, s4, 15
; GFX6-NEXT: s_andn2_b32 s4, 15, s4
; GFX6-NEXT: s_lshl_b32 s0, s0, 1
-; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
; GFX6-NEXT: s_lshl_b32 s0, s0, s4
-; GFX6-NEXT: s_lshr_b32 s2, s2, s9
-; GFX6-NEXT: s_or_b32 s0, s0, s2
-; GFX6-NEXT: s_and_b32 s2, s8, 15
-; GFX6-NEXT: s_andn2_b32 s4, 15, s8
+; GFX6-NEXT: s_and_b32 s4, s2, 0xffff
+; GFX6-NEXT: s_lshr_b32 s4, s4, s8
+; GFX6-NEXT: s_or_b32 s0, s0, s4
+; GFX6-NEXT: s_and_b32 s4, s7, 15
+; GFX6-NEXT: s_andn2_b32 s7, 15, s7
; GFX6-NEXT: s_lshl_b32 s6, s6, 1
-; GFX6-NEXT: s_lshl_b32 s4, s6, s4
-; GFX6-NEXT: s_lshr_b32 s2, s7, s2
-; GFX6-NEXT: s_or_b32 s2, s4, s2
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT: s_lshl_b32 s6, s6, s7
+; GFX6-NEXT: s_lshr_b32 s2, s2, s4
+; GFX6-NEXT: s_or_b32 s2, s6, s2
; GFX6-NEXT: s_and_b32 s4, s5, 15
; GFX6-NEXT: s_andn2_b32 s5, 15, s5
; GFX6-NEXT: s_lshl_b32 s1, s1, 1
@@ -4767,26 +4767,26 @@ define <3 x half> @v_fshr_v3i16(<3 x i16> %lhs, <3 x i16> %rhs, <3 x i16> %amt)
; GFX6-LABEL: v_fshr_v3i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v4
-; GFX6-NEXT: v_and_b32_e32 v9, 15, v4
+; GFX6-NEXT: v_lshrrev_b32_e32 v7, 16, v4
+; GFX6-NEXT: v_and_b32_e32 v8, 15, v4
; GFX6-NEXT: v_xor_b32_e32 v4, -1, v4
; GFX6-NEXT: v_lshrrev_b32_e32 v6, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v7, 16, v2
; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
; GFX6-NEXT: v_lshlrev_b32_e32 v0, 1, v0
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v9, v2
-; GFX6-NEXT: v_xor_b32_e32 v4, -1, v8
-; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 15, v8
-; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT: v_lshrrev_b32_e32 v4, v8, v4
+; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
+; GFX6-NEXT: v_and_b32_e32 v4, 15, v7
+; GFX6-NEXT: v_xor_b32_e32 v7, -1, v7
+; GFX6-NEXT: v_and_b32_e32 v7, 15, v7
; GFX6-NEXT: v_lshlrev_b32_e32 v6, 1, v6
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v6
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v7
-; GFX6-NEXT: v_or_b32_e32 v2, v4, v2
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT: v_lshlrev_b32_e32 v6, v7, v6
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, v4, v2
; GFX6-NEXT: v_and_b32_e32 v4, 15, v5
; GFX6-NEXT: v_xor_b32_e32 v5, -1, v5
+; GFX6-NEXT: v_or_b32_e32 v2, v6, v2
; GFX6-NEXT: v_and_b32_e32 v5, 15, v5
; GFX6-NEXT: v_lshlrev_b32_e32 v1, 1, v1
; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
@@ -4932,41 +4932,41 @@ define amdgpu_ps <2 x i32> @s_fshr_v4i16(<4 x i16> inreg %lhs, <4 x i16> inreg %
; GFX6-LABEL: s_fshr_v4i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_lshr_b32 s6, s0, 16
-; GFX6-NEXT: s_lshr_b32 s8, s2, 16
-; GFX6-NEXT: s_lshr_b32 s10, s4, 16
-; GFX6-NEXT: s_and_b32 s12, s4, 15
+; GFX6-NEXT: s_lshr_b32 s8, s4, 16
+; GFX6-NEXT: s_and_b32 s10, s4, 15
; GFX6-NEXT: s_andn2_b32 s4, 15, s4
; GFX6-NEXT: s_lshl_b32 s0, s0, 1
-; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
; GFX6-NEXT: s_lshl_b32 s0, s0, s4
-; GFX6-NEXT: s_lshr_b32 s2, s2, s12
-; GFX6-NEXT: s_or_b32 s0, s0, s2
-; GFX6-NEXT: s_and_b32 s2, s10, 15
-; GFX6-NEXT: s_andn2_b32 s4, 15, s10
+; GFX6-NEXT: s_and_b32 s4, s2, 0xffff
+; GFX6-NEXT: s_lshr_b32 s4, s4, s10
+; GFX6-NEXT: s_or_b32 s0, s0, s4
+; GFX6-NEXT: s_and_b32 s4, s8, 15
+; GFX6-NEXT: s_andn2_b32 s8, 15, s8
; GFX6-NEXT: s_lshl_b32 s6, s6, 1
-; GFX6-NEXT: s_lshl_b32 s4, s6, s4
-; GFX6-NEXT: s_lshr_b32 s2, s8, s2
-; GFX6-NEXT: s_or_b32 s2, s4, s2
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT: s_lshl_b32 s6, s6, s8
+; GFX6-NEXT: s_lshr_b32 s2, s2, s4
+; GFX6-NEXT: s_or_b32 s2, s6, s2
; GFX6-NEXT: s_and_b32 s2, 0xffff, s2
+; GFX6-NEXT: s_lshr_b32 s7, s1, 16
; GFX6-NEXT: s_and_b32 s0, 0xffff, s0
; GFX6-NEXT: s_lshl_b32 s2, s2, 16
-; GFX6-NEXT: s_lshr_b32 s7, s1, 16
-; GFX6-NEXT: s_lshr_b32 s9, s3, 16
-; GFX6-NEXT: s_or_b32 s0, s0, s2
-; GFX6-NEXT: s_and_b32 s2, s5, 15
; GFX6-NEXT: s_andn2_b32 s4, 15, s5
; GFX6-NEXT: s_lshl_b32 s1, s1, 1
-; GFX6-NEXT: s_and_b32 s3, s3, 0xffff
-; GFX6-NEXT: s_lshr_b32 s11, s5, 16
+; GFX6-NEXT: s_or_b32 s0, s0, s2
+; GFX6-NEXT: s_and_b32 s2, s5, 15
; GFX6-NEXT: s_lshl_b32 s1, s1, s4
-; GFX6-NEXT: s_lshr_b32 s2, s3, s2
+; GFX6-NEXT: s_and_b32 s4, s3, 0xffff
+; GFX6-NEXT: s_lshr_b32 s9, s5, 16
+; GFX6-NEXT: s_lshr_b32 s2, s4, s2
; GFX6-NEXT: s_or_b32 s1, s1, s2
-; GFX6-NEXT: s_and_b32 s2, s11, 15
-; GFX6-NEXT: s_andn2_b32 s3, 15, s11
-; GFX6-NEXT: s_lshl_b32 s4, s7, 1
-; GFX6-NEXT: s_lshl_b32 s3, s4, s3
-; GFX6-NEXT: s_lshr_b32 s2, s9, s2
-; GFX6-NEXT: s_or_b32 s2, s3, s2
+; GFX6-NEXT: s_and_b32 s2, s9, 15
+; GFX6-NEXT: s_andn2_b32 s4, 15, s9
+; GFX6-NEXT: s_lshl_b32 s5, s7, 1
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT: s_lshl_b32 s4, s5, s4
+; GFX6-NEXT: s_lshr_b32 s2, s3, s2
+; GFX6-NEXT: s_or_b32 s2, s4, s2
; GFX6-NEXT: s_and_b32 s2, 0xffff, s2
; GFX6-NEXT: s_and_b32 s1, 0xffff, s1
; GFX6-NEXT: s_lshl_b32 s2, s2, 16
@@ -5145,46 +5145,46 @@ define <4 x half> @v_fshr_v4i16(<4 x i16> %lhs, <4 x i16> %rhs, <4 x i16> %amt)
; GFX6-LABEL: v_fshr_v4i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v10, 16, v4
-; GFX6-NEXT: v_and_b32_e32 v12, 15, v4
+; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v4
+; GFX6-NEXT: v_and_b32_e32 v10, 15, v4
; GFX6-NEXT: v_xor_b32_e32 v4, -1, v4
; GFX6-NEXT: v_lshrrev_b32_e32 v6, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v2
; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
; GFX6-NEXT: v_lshlrev_b32_e32 v0, 1, v0
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v12, v2
-; GFX6-NEXT: v_xor_b32_e32 v4, -1, v10
-; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 15, v10
-; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT: v_lshrrev_b32_e32 v4, v10, v4
+; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
+; GFX6-NEXT: v_and_b32_e32 v4, 15, v8
+; GFX6-NEXT: v_xor_b32_e32 v8, -1, v8
+; GFX6-NEXT: v_and_b32_e32 v8, 15, v8
; GFX6-NEXT: v_lshlrev_b32_e32 v6, 1, v6
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v6
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v8
-; GFX6-NEXT: v_or_b32_e32 v2, v4, v2
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT: v_lshlrev_b32_e32 v6, v8, v6
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, v4, v2
+; GFX6-NEXT: v_or_b32_e32 v2, v6, v2
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
; GFX6-NEXT: v_xor_b32_e32 v4, -1, v5
; GFX6-NEXT: v_lshrrev_b32_e32 v7, 16, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v3
-; GFX6-NEXT: v_lshrrev_b32_e32 v11, 16, v5
-; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 15, v5
+; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
; GFX6-NEXT: v_lshlrev_b32_e32 v1, 1, v1
-; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v5
+; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
+; GFX6-NEXT: v_and_b32_e32 v2, 15, v5
; GFX6-NEXT: v_lshlrev_b32_e32 v1, v4, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v3
-; GFX6-NEXT: v_xor_b32_e32 v3, -1, v11
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v3
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v4
+; GFX6-NEXT: v_xor_b32_e32 v4, -1, v9
; GFX6-NEXT: v_or_b32_e32 v1, v1, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 15, v11
-; GFX6-NEXT: v_and_b32_e32 v3, 15, v3
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, 1, v7
-; GFX6-NEXT: v_lshlrev_b32_e32 v3, v3, v4
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v9
-; GFX6-NEXT: v_or_b32_e32 v2, v3, v2
+; GFX6-NEXT: v_and_b32_e32 v2, 15, v9
+; GFX6-NEXT: v_and_b32_e32 v4, 15, v4
+; GFX6-NEXT: v_lshlrev_b32_e32 v5, 1, v7
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v5
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v3
+; GFX6-NEXT: v_or_b32_e32 v2, v4, v2
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
index ee93f413566e5..dbb753c0b3072 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.intersect_ray.ll
@@ -53,12 +53,13 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_flat(i32 %node_ptr, float
define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16(i32 %node_ptr, float %ray_extent, <3 x float> %ray_origin, <3 x half> %ray_dir, <3 x half> %ray_inv_dir, <4 x i32> inreg %tdescr) {
; GFX10-LABEL: image_bvh_intersect_ray_a16:
; GFX10: ; %bb.0:
-; GFX10-NEXT: v_lshrrev_b32_e32 v9, 16, v5
+; GFX10-NEXT: v_mov_b32_e32 v9, 16
; GFX10-NEXT: v_and_b32_e32 v10, 0xffff, v7
+; GFX10-NEXT: v_bfe_u32 v7, v7, 16, 16
; GFX10-NEXT: v_and_b32_e32 v8, 0xffff, v8
-; GFX10-NEXT: v_lshlrev_b32_e32 v9, 16, v9
+; GFX10-NEXT: v_lshlrev_b32_sdwa v9, v9, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX10-NEXT: v_lshlrev_b32_e32 v10, 16, v10
-; GFX10-NEXT: v_alignbit_b32 v7, v8, v7, 16
+; GFX10-NEXT: v_lshl_or_b32 v7, v8, 16, v7
; GFX10-NEXT: v_and_or_b32 v5, 0xffff, v5, v9
; GFX10-NEXT: v_and_or_b32 v6, 0xffff, v6, v10
; GFX10-NEXT: image_bvh_intersect_ray v[0:3], v[0:7], s[0:3] a16
@@ -128,12 +129,13 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_flat(<2 x i32> %node_ptr
define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16(i64 %node_ptr, float %ray_extent, <3 x float> %ray_origin, <3 x half> %ray_dir, <3 x half> %ray_inv_dir, <4 x i32> inreg %tdescr) {
; GFX10-LABEL: image_bvh64_intersect_ray_a16:
; GFX10: ; %bb.0:
-; GFX10-NEXT: v_lshrrev_b32_e32 v10, 16, v6
+; GFX10-NEXT: v_mov_b32_e32 v10, 16
; GFX10-NEXT: v_and_b32_e32 v11, 0xffff, v8
+; GFX10-NEXT: v_bfe_u32 v8, v8, 16, 16
; GFX10-NEXT: v_and_b32_e32 v9, 0xffff, v9
-; GFX10-NEXT: v_lshlrev_b32_e32 v10, 16, v10
+; GFX10-NEXT: v_lshlrev_b32_sdwa v10, v10, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX10-NEXT: v_lshlrev_b32_e32 v11, 16, v11
-; GFX10-NEXT: v_alignbit_b32 v8, v9, v8, 16
+; GFX10-NEXT: v_lshl_or_b32 v8, v9, 16, v8
; GFX10-NEXT: v_and_or_b32 v6, 0xffff, v6, v10
; GFX10-NEXT: v_and_or_b32 v7, 0xffff, v7, v11
; GFX10-NEXT: image_bvh64_intersect_ray v[0:3], v[0:8], s[0:3] a16
@@ -282,18 +284,19 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
; GFX1030: ; %bb.0:
; GFX1030-NEXT: v_mov_b32_e32 v13, v0
; GFX1030-NEXT: v_mov_b32_e32 v14, v1
-; GFX1030-NEXT: v_lshrrev_b32_e32 v0, 16, v5
+; GFX1030-NEXT: v_mov_b32_e32 v0, 16
; GFX1030-NEXT: v_and_b32_e32 v1, 0xffff, v7
; GFX1030-NEXT: v_mov_b32_e32 v15, v2
-; GFX1030-NEXT: v_and_b32_e32 v2, 0xffff, v8
; GFX1030-NEXT: v_mov_b32_e32 v16, v3
-; GFX1030-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX1030-NEXT: v_bfe_u32 v2, v7, 16, 16
+; GFX1030-NEXT: v_lshlrev_b32_sdwa v0, v0, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX1030-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX1030-NEXT: v_and_b32_e32 v3, 0xffff, v8
; GFX1030-NEXT: v_mov_b32_e32 v17, v4
-; GFX1030-NEXT: v_alignbit_b32 v20, v2, v7, 16
; GFX1030-NEXT: s_mov_b32 s1, exec_lo
; GFX1030-NEXT: v_and_or_b32 v18, 0xffff, v5, v0
; GFX1030-NEXT: v_and_or_b32 v19, 0xffff, v6, v1
+; GFX1030-NEXT: v_lshl_or_b32 v20, v3, 16, v2
; GFX1030-NEXT: .LBB7_1: ; =>This Inner Loop Header: Depth=1
; GFX1030-NEXT: v_readfirstlane_b32 s4, v9
; GFX1030-NEXT: v_readfirstlane_b32 s5, v10
@@ -323,13 +326,14 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16_vgpr_descr(i32 %node_p
;
; GFX1013-LABEL: image_bvh_intersect_ray_a16_vgpr_descr:
; GFX1013: ; %bb.0:
-; GFX1013-NEXT: v_lshrrev_b32_e32 v13, 16, v5
+; GFX1013-NEXT: v_mov_b32_e32 v13, 16
; GFX1013-NEXT: v_and_b32_e32 v14, 0xffff, v7
+; GFX1013-NEXT: v_bfe_u32 v7, v7, 16, 16
; GFX1013-NEXT: v_and_b32_e32 v8, 0xffff, v8
; GFX1013-NEXT: s_mov_b32 s1, exec_lo
-; GFX1013-NEXT: v_lshlrev_b32_e32 v13, 16, v13
+; GFX1013-NEXT: v_lshlrev_b32_sdwa v13, v13, v5 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX1013-NEXT: v_lshlrev_b32_e32 v14, 16, v14
-; GFX1013-NEXT: v_alignbit_b32 v7, v8, v7, 16
+; GFX1013-NEXT: v_lshl_or_b32 v7, v8, 16, v7
; GFX1013-NEXT: v_and_or_b32 v5, 0xffff, v5, v13
; GFX1013-NEXT: v_and_or_b32 v6, 0xffff, v6, v14
; GFX1013-NEXT: .LBB7_1: ; =>This Inner Loop Header: Depth=1
@@ -549,18 +553,19 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
; GFX1030: ; %bb.0:
; GFX1030-NEXT: v_mov_b32_e32 v14, v0
; GFX1030-NEXT: v_mov_b32_e32 v15, v1
-; GFX1030-NEXT: v_lshrrev_b32_e32 v0, 16, v6
+; GFX1030-NEXT: v_mov_b32_e32 v0, 16
; GFX1030-NEXT: v_and_b32_e32 v1, 0xffff, v8
; GFX1030-NEXT: v_mov_b32_e32 v16, v2
-; GFX1030-NEXT: v_and_b32_e32 v2, 0xffff, v9
; GFX1030-NEXT: v_mov_b32_e32 v17, v3
-; GFX1030-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX1030-NEXT: v_bfe_u32 v2, v8, 16, 16
+; GFX1030-NEXT: v_lshlrev_b32_sdwa v0, v0, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX1030-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX1030-NEXT: v_and_b32_e32 v3, 0xffff, v9
; GFX1030-NEXT: v_mov_b32_e32 v18, v4
; GFX1030-NEXT: v_mov_b32_e32 v19, v5
-; GFX1030-NEXT: v_alignbit_b32 v22, v2, v8, 16
; GFX1030-NEXT: v_and_or_b32 v20, 0xffff, v6, v0
; GFX1030-NEXT: v_and_or_b32 v21, 0xffff, v7, v1
+; GFX1030-NEXT: v_lshl_or_b32 v22, v3, 16, v2
; GFX1030-NEXT: s_mov_b32 s1, exec_lo
; GFX1030-NEXT: .LBB9_1: ; =>This Inner Loop Header: Depth=1
; GFX1030-NEXT: v_readfirstlane_b32 s4, v10
@@ -592,13 +597,14 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16_vgpr_descr(i64 %node
;
; GFX1013-LABEL: image_bvh64_intersect_ray_a16_vgpr_descr:
; GFX1013: ; %bb.0:
-; GFX1013-NEXT: v_lshrrev_b32_e32 v14, 16, v6
+; GFX1013-NEXT: v_mov_b32_e32 v14, 16
; GFX1013-NEXT: v_and_b32_e32 v15, 0xffff, v8
+; GFX1013-NEXT: v_bfe_u32 v8, v8, 16, 16
; GFX1013-NEXT: v_and_b32_e32 v9, 0xffff, v9
; GFX1013-NEXT: s_mov_b32 s1, exec_lo
-; GFX1013-NEXT: v_lshlrev_b32_e32 v14, 16, v14
+; GFX1013-NEXT: v_lshlrev_b32_sdwa v14, v14, v6 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:DWORD src1_sel:WORD_1
; GFX1013-NEXT: v_lshlrev_b32_e32 v15, 16, v15
-; GFX1013-NEXT: v_alignbit_b32 v8, v9, v8, 16
+; GFX1013-NEXT: v_lshl_or_b32 v8, v9, 16, v8
; GFX1013-NEXT: v_and_or_b32 v6, 0xffff, v6, v14
; GFX1013-NEXT: v_and_or_b32 v7, 0xffff, v7, v15
; GFX1013-NEXT: .LBB9_1: ; =>This Inner Loop Header: Depth=1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
index 5739d01e61d95..1f79f289ac719 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/llvm.amdgcn.make.buffer.rsrc.ll
@@ -37,9 +37,9 @@ define amdgpu_ps ptr addrspace(8) @basic_raw_buffer(ptr inreg %p) {
; CHECK45-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32 = S_MOV_B32 0
; CHECK45-NEXT: [[S_MOV_B:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO -6629298651489370112
; CHECK45-NEXT: [[S_OR_B64_:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE]], [[S_MOV_B]], implicit-def dead $scc
- ; CHECK45-NEXT: [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 9
; CHECK45-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32 = S_MOV_B32 -536870912
; CHECK45-NEXT: [[REG_SEQUENCE1:%[0-9]+]]:sreg_64 = REG_SEQUENCE [[S_MOV_B32_]], %subreg.sub0, [[S_MOV_B32_1]], %subreg.sub1
+ ; CHECK45-NEXT: [[S_MOV_B64_:%[0-9]+]]:sreg_64 = S_MOV_B64 9
; CHECK45-NEXT: [[S_OR_B64_1:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE1]], [[S_MOV_B64_]], implicit-def dead $scc
; CHECK45-NEXT: [[COPY2:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub0
; CHECK45-NEXT: [[COPY3:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub1
@@ -97,9 +97,9 @@ define amdgpu_ps ptr addrspace(8) @large_num_records_raw_buffer(ptr inreg %p) {
; CHECK45-NEXT: [[S_MOV_B32_:%[0-9]+]]:sreg_32 = S_MOV_B32 0
; CHECK45-NEXT: [[S_MOV_B:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO -6629298651489370112
; CHECK45-NEXT: [[S_OR_B64_:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE]], [[S_MOV_B]], implicit-def dead $scc
- ; CHECK45-NEXT: [[S_MOV_B1:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO 33554441
; CHECK45-NEXT: [[S_MOV_B32_1:%[0-9]+]]:sreg_32 = S_MOV_B32 -536870912
; CHECK45-NEXT: [[REG_SEQUENCE1:%[0-9]+]]:sreg_64 = REG_SEQUENCE [[S_MOV_B32_]], %subreg.sub0, [[S_MOV_B32_1]], %subreg.sub1
+ ; CHECK45-NEXT: [[S_MOV_B1:%[0-9]+]]:sreg_64 = S_MOV_B64_IMM_PSEUDO 33554441
; CHECK45-NEXT: [[S_OR_B64_1:%[0-9]+]]:sreg_64 = S_OR_B64 [[REG_SEQUENCE1]], [[S_MOV_B1]], implicit-def dead $scc
; CHECK45-NEXT: [[COPY2:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub0
; CHECK45-NEXT: [[COPY3:%[0-9]+]]:sreg_32 = COPY [[S_OR_B64_]].sub1
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
index 83e6d007f299a..1f10aaf81646a 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/lshr.ll
@@ -1035,22 +1035,22 @@ define <2 x float> @v_lshr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
; GFX6-LABEL: v_lshr_v4i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v7, 16, v3
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v2
+; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v0
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX6-NEXT: v_lshrrev_b32_e32 v4, v4, v5
; GFX6-NEXT: v_lshrrev_b32_e32 v0, v2, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v2, v6, v4
-; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
+; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v3
+; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v1
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX6-NEXT: v_bfe_u32 v1, v1, 16, 16
; GFX6-NEXT: v_lshrrev_b32_e32 v1, v3, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v3, v7, v5
-; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
-; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v3
-; GFX6-NEXT: v_or_b32_e32 v1, v1, v2
+; GFX6-NEXT: v_lshrrev_b32_e32 v2, v2, v5
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT: v_or_b32_e32 v0, v4, v0
+; GFX6-NEXT: v_or_b32_e32 v1, v2, v1
; GFX6-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-LABEL: v_lshr_v4i16:
@@ -1085,20 +1085,20 @@ define <2 x float> @v_lshr_v4i16(<4 x i16> %value, <4 x i16> %amount) {
define amdgpu_ps <2 x i32> @s_lshr_v4i16(<4 x i16> inreg %value, <4 x i16> inreg %amount) {
; GFX6-LABEL: s_lshr_v4i16:
; GFX6: ; %bb.0:
-; GFX6-NEXT: s_lshr_b32 s4, s0, 16
-; GFX6-NEXT: s_lshr_b32 s6, s2, 16
-; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT: s_lshr_b32 s5, s1, 16
-; GFX6-NEXT: s_lshr_b32 s7, s3, 16
+; GFX6-NEXT: s_and_b32 s4, s0, 0xffff
+; GFX6-NEXT: s_lshr_b32 s4, s4, s2
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT: s_bfe_u32 s0, s0, 0x100010
; GFX6-NEXT: s_lshr_b32 s0, s0, s2
-; GFX6-NEXT: s_lshr_b32 s2, s4, s6
-; GFX6-NEXT: s_and_b32 s1, s1, 0xffff
+; GFX6-NEXT: s_and_b32 s2, s1, 0xffff
+; GFX6-NEXT: s_lshr_b32 s2, s2, s3
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT: s_bfe_u32 s1, s1, 0x100010
; GFX6-NEXT: s_lshr_b32 s1, s1, s3
-; GFX6-NEXT: s_lshr_b32 s3, s5, s7
-; GFX6-NEXT: s_lshl_b32 s2, s2, 16
-; GFX6-NEXT: s_or_b32 s0, s0, s2
-; GFX6-NEXT: s_lshl_b32 s2, s3, 16
-; GFX6-NEXT: s_or_b32 s1, s1, s2
+; GFX6-NEXT: s_lshl_b32 s0, s0, 16
+; GFX6-NEXT: s_lshl_b32 s1, s1, 16
+; GFX6-NEXT: s_or_b32 s0, s4, s0
+; GFX6-NEXT: s_or_b32 s1, s2, s1
; GFX6-NEXT: ; return to shader part epilog
;
; GFX8-LABEL: s_lshr_v4i16:
@@ -1182,38 +1182,38 @@ define <4 x float> @v_lshr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
; GFX6-LABEL: v_lshr_v8i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v12, 16, v4
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v13, 16, v5
+; GFX6-NEXT: v_and_b32_e32 v8, 0xffff, v4
+; GFX6-NEXT: v_and_b32_e32 v9, 0xffff, v0
+; GFX6-NEXT: v_bfe_u32 v4, v4, 16, 16
+; GFX6-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX6-NEXT: v_lshrrev_b32_e32 v8, v8, v9
; GFX6-NEXT: v_lshrrev_b32_e32 v0, v4, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v4, v12, v8
-; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v5
-; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX6-NEXT: v_lshrrev_b32_e32 v14, 16, v6
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v5
+; GFX6-NEXT: v_and_b32_e32 v9, 0xffff, v1
+; GFX6-NEXT: v_bfe_u32 v5, v5, 16, 16
+; GFX6-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX6-NEXT: v_lshrrev_b32_e32 v4, v4, v9
; GFX6-NEXT: v_lshrrev_b32_e32 v1, v5, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v5, v13, v9
-; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v6
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v4
-; GFX6-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX6-NEXT: v_lshrrev_b32_e32 v15, 16, v7
+; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v6
+; GFX6-NEXT: v_and_b32_e32 v9, 0xffff, v2
+; GFX6-NEXT: v_bfe_u32 v6, v6, 16, 16
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT: v_lshrrev_b32_e32 v5, v5, v9
; GFX6-NEXT: v_lshrrev_b32_e32 v2, v6, v2
-; GFX6-NEXT: v_lshrrev_b32_e32 v6, v14, v10
-; GFX6-NEXT: v_and_b32_e32 v7, 0xffff, v7
-; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v5
+; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v7
+; GFX6-NEXT: v_and_b32_e32 v9, 0xffff, v3
+; GFX6-NEXT: v_bfe_u32 v7, v7, 16, 16
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
; GFX6-NEXT: v_lshrrev_b32_e32 v3, v7, v3
-; GFX6-NEXT: v_lshrrev_b32_e32 v7, v15, v11
-; GFX6-NEXT: v_or_b32_e32 v1, v1, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v6
-; GFX6-NEXT: v_or_b32_e32 v2, v2, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v7
-; GFX6-NEXT: v_or_b32_e32 v3, v3, v4
+; GFX6-NEXT: v_lshrrev_b32_e32 v6, v6, v9
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX6-NEXT: v_lshlrev_b32_e32 v1, 16, v1
+; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX6-NEXT: v_lshlrev_b32_e32 v3, 16, v3
+; GFX6-NEXT: v_or_b32_e32 v0, v8, v0
+; GFX6-NEXT: v_or_b32_e32 v1, v4, v1
+; GFX6-NEXT: v_or_b32_e32 v2, v5, v2
+; GFX6-NEXT: v_or_b32_e32 v3, v6, v3
; GFX6-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-LABEL: v_lshr_v8i16:
@@ -1258,34 +1258,34 @@ define <4 x float> @v_lshr_v8i16(<8 x i16> %value, <8 x i16> %amount) {
define amdgpu_ps <4 x i32> @s_lshr_v8i16(<8 x i16> inreg %value, <8 x i16> inreg %amount) {
; GFX6-LABEL: s_lshr_v8i16:
; GFX6: ; %bb.0:
-; GFX6-NEXT: s_lshr_b32 s8, s0, 16
-; GFX6-NEXT: s_lshr_b32 s12, s4, 16
-; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
-; GFX6-NEXT: s_lshr_b32 s9, s1, 16
-; GFX6-NEXT: s_lshr_b32 s13, s5, 16
+; GFX6-NEXT: s_and_b32 s8, s0, 0xffff
+; GFX6-NEXT: s_lshr_b32 s8, s8, s4
+; GFX6-NEXT: s_bfe_u32 s4, s4, 0x100010
+; GFX6-NEXT: s_bfe_u32 s0, s0, 0x100010
; GFX6-NEXT: s_lshr_b32 s0, s0, s4
-; GFX6-NEXT: s_lshr_b32 s4, s8, s12
-; GFX6-NEXT: s_and_b32 s1, s1, 0xffff
-; GFX6-NEXT: s_lshr_b32 s10, s2, 16
-; GFX6-NEXT: s_lshr_b32 s14, s6, 16
+; GFX6-NEXT: s_and_b32 s4, s1, 0xffff
+; GFX6-NEXT: s_lshr_b32 s4, s4, s5
+; GFX6-NEXT: s_bfe_u32 s5, s5, 0x100010
+; GFX6-NEXT: s_bfe_u32 s1, s1, 0x100010
; GFX6-NEXT: s_lshr_b32 s1, s1, s5
-; GFX6-NEXT: s_lshr_b32 s5, s9, s13
-; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
-; GFX6-NEXT: s_lshl_b32 s4, s4, 16
-; GFX6-NEXT: s_lshr_b32 s11, s3, 16
-; GFX6-NEXT: s_lshr_b32 s15, s7, 16
+; GFX6-NEXT: s_and_b32 s5, s2, 0xffff
+; GFX6-NEXT: s_lshr_b32 s5, s5, s6
+; GFX6-NEXT: s_bfe_u32 s6, s6, 0x100010
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
; GFX6-NEXT: s_lshr_b32 s2, s2, s6
-; GFX6-NEXT: s_lshr_b32 s6, s10, s14
-; GFX6-NEXT: s_and_b32 s3, s3, 0xffff
-; GFX6-NEXT: s_or_b32 s0, s0, s4
-; GFX6-NEXT: s_lshl_b32 s4, s5, 16
+; GFX6-NEXT: s_and_b32 s6, s3, 0xffff
+; GFX6-NEXT: s_lshr_b32 s6, s6, s7
+; GFX6-NEXT: s_bfe_u32 s7, s7, 0x100010
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
; GFX6-NEXT: s_lshr_b32 s3, s3, s7
-; GFX6-NEXT: s_lshr_b32 s7, s11, s15
-; GFX6-NEXT: s_or_b32 s1, s1, s4
-; GFX6-NEXT: s_lshl_b32 s4, s6, 16
-; GFX6-NEXT: s_or_b32 s2, s2, s4
-; GFX6-NEXT: s_lshl_b32 s4, s7, 16
-; GFX6-NEXT: s_or_b32 s3, s3, s4
+; GFX6-NEXT: s_lshl_b32 s0, s0, 16
+; GFX6-NEXT: s_lshl_b32 s1, s1, 16
+; GFX6-NEXT: s_lshl_b32 s2, s2, 16
+; GFX6-NEXT: s_lshl_b32 s3, s3, 16
+; GFX6-NEXT: s_or_b32 s0, s8, s0
+; GFX6-NEXT: s_or_b32 s1, s4, s1
+; GFX6-NEXT: s_or_b32 s2, s5, s2
+; GFX6-NEXT: s_or_b32 s3, s6, s3
; GFX6-NEXT: ; return to shader part epilog
;
; GFX8-LABEL: s_lshr_v8i16:
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
index dfa76d8d0acfa..fc1be777b09a3 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/merge-values-s16-true16.ll
@@ -28,20 +28,17 @@ define amdgpu_kernel void @store_i136_divergent(ptr %p) {
; GFX1150-LABEL: store_i136_divergent:
; GFX1150: ; %bb.0:
; GFX1150-NEXT: s_load_b64 s[0:1], s[4:5], 0x0
-; GFX1150-NEXT: s_pack_ll_b32_b16 s2, 0, 0
+; GFX1150-NEXT: v_dual_mov_b32 v3, 0 :: v_dual_and_b32 v0, 0x3ff, v0
+; GFX1150-NEXT: v_mov_b32_e32 v2, 0
; GFX1150-NEXT: v_mov_b32_e32 v6, 0
-; GFX1150-NEXT: s_mov_b32 s3, s2
-; GFX1150-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_1) | instid1(VALU_DEP_2)
-; GFX1150-NEXT: v_dual_mov_b32 v3, s3 :: v_dual_and_b32 v0, 0x3ff, v0
-; GFX1150-NEXT: v_mov_b32_e32 v2, s2
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_2) | instid1(VALU_DEP_3)
; GFX1150-NEXT: v_lshrrev_b32_e32 v1, 8, v0
; GFX1150-NEXT: v_mov_b16_e32 v0.h, 0
; GFX1150-NEXT: v_and_b16 v0.l, 0xff, v0.l
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(NEXT) | instid1(VALU_DEP_3)
; GFX1150-NEXT: v_lshlrev_b16 v4.l, 8, v1.l
+; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_3) | instskip(SKIP_1) | instid1(VALU_DEP_3)
; GFX1150-NEXT: v_mov_b16_e32 v1.l, v0.h
; GFX1150-NEXT: v_mov_b16_e32 v1.h, v0.h
-; GFX1150-NEXT: s_delay_alu instid0(VALU_DEP_3)
; GFX1150-NEXT: v_or_b16 v0.l, v0.l, v4.l
; GFX1150-NEXT: s_waitcnt lgkmcnt(0)
; GFX1150-NEXT: v_dual_mov_b32 v5, s1 :: v_dual_mov_b32 v4, s0
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
index d1621dbc267da..b41e1a1374480 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/mul-known-bits.i64.ll
@@ -201,13 +201,16 @@ define amdgpu_kernel void @v_mul_i64_masked_src0_hi(ptr addrspace(1) %out, ptr a
; GFX10-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX10-NEXT: s_load_dwordx2 s[6:7], s[4:5], 0x34
; GFX10-NEXT: v_lshlrev_b32_e32 v0, 3, v0
+; GFX10-NEXT: ; kill: killed $vgpr0
+; GFX10-NEXT: ; kill: killed $sgpr2_sgpr3
; GFX10-NEXT: s_waitcnt lgkmcnt(0)
; GFX10-NEXT: s_clause 0x1
-; GFX10-NEXT: global_load_dword v4, v0, s[2:3]
-; GFX10-NEXT: global_load_dwordx2 v[2:3], v0, s[6:7]
+; GFX10-NEXT: global_load_dwordx2 v[2:3], v0, s[2:3]
+; GFX10-NEXT: global_load_dwordx2 v[3:4], v0, s[6:7]
+; GFX10-NEXT: ; kill: killed $sgpr6_sgpr7
; GFX10-NEXT: s_waitcnt vmcnt(0)
-; GFX10-NEXT: v_mad_u64_u32 v[0:1], s2, v4, v2, 0
-; GFX10-NEXT: v_mad_u64_u32 v[1:2], s2, v4, v3, v[1:2]
+; GFX10-NEXT: v_mad_u64_u32 v[0:1], s2, v2, v3, 0
+; GFX10-NEXT: v_mad_u64_u32 v[1:2], s2, v2, v4, v[1:2]
; GFX10-NEXT: v_mov_b32_e32 v2, 0
; GFX10-NEXT: global_store_dwordx2 v2, v[0:1], s[0:1]
; GFX10-NEXT: s_endpgm
@@ -222,13 +225,13 @@ define amdgpu_kernel void @v_mul_i64_masked_src0_hi(ptr addrspace(1) %out, ptr a
; GFX11-NEXT: v_lshlrev_b32_e32 v0, 3, v0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-NEXT: s_clause 0x1
-; GFX11-NEXT: global_load_b32 v6, v0, s[2:3]
-; GFX11-NEXT: global_load_b64 v[2:3], v0, s[4:5]
+; GFX11-NEXT: global_load_b64 v[2:3], v0, s[2:3]
+; GFX11-NEXT: global_load_b64 v[3:4], v0, s[4:5]
; GFX11-NEXT: s_waitcnt vmcnt(0)
-; GFX11-NEXT: v_mad_u64_u32 v[0:1], null, v6, v2, 0
+; GFX11-NEXT: v_mad_u64_u32 v[0:1], null, v2, v3, 0
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-NEXT: v_mad_u64_u32 v[4:5], null, v6, v3, v[1:2]
-; GFX11-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, v4
+; GFX11-NEXT: v_mad_u64_u32 v[5:6], null, v2, v4, v[1:2]
+; GFX11-NEXT: v_dual_mov_b32 v2, 0 :: v_dual_mov_b32 v1, v5
; GFX11-NEXT: global_store_b64 v2, v[0:1], s[0:1]
; GFX11-NEXT: s_endpgm
%tid = call i32 @llvm.amdgcn.workitem.id.x()
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
index e4035042e9ca0..efb8fc0ed4c4c 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/shl.ll
@@ -980,18 +980,18 @@ define <2 x float> @v_shl_v4i16(<4 x i16> %value, <4 x i16> %amount) {
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX6-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v0, v2, v0
-; GFX6-NEXT: v_lshlrev_b32_e32 v2, v6, v4
+; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v2
+; GFX6-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX6-NEXT: v_lshlrev_b32_e32 v2, v2, v4
; GFX6-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, v6, v0
+; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v3
+; GFX6-NEXT: v_bfe_u32 v3, v3, 16, 16
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v1, v3, v1
-; GFX6-NEXT: v_lshlrev_b32_e32 v3, v7, v5
+; GFX6-NEXT: v_lshlrev_b32_e32 v3, v3, v5
; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX6-NEXT: v_lshlrev_b32_e32 v2, 16, v2
+; GFX6-NEXT: v_lshlrev_b32_e32 v1, v4, v1
; GFX6-NEXT: v_or_b32_e32 v0, v0, v2
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v3
; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
@@ -1032,14 +1032,14 @@ define amdgpu_ps <2 x i32> @s_shl_v4i16(<4 x i16> inreg %value, <4 x i16> inreg
; GFX6-LABEL: s_shl_v4i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_lshr_b32 s4, s0, 16
-; GFX6-NEXT: s_lshr_b32 s6, s2, 16
; GFX6-NEXT: s_lshl_b32 s0, s0, s2
-; GFX6-NEXT: s_lshl_b32 s2, s4, s6
+; GFX6-NEXT: s_bfe_u32 s2, s2, 0x100010
+; GFX6-NEXT: s_lshl_b32 s2, s4, s2
; GFX6-NEXT: s_lshr_b32 s5, s1, 16
-; GFX6-NEXT: s_lshr_b32 s7, s3, 16
-; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
; GFX6-NEXT: s_lshl_b32 s1, s1, s3
-; GFX6-NEXT: s_lshl_b32 s3, s5, s7
+; GFX6-NEXT: s_bfe_u32 s3, s3, 0x100010
+; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
+; GFX6-NEXT: s_lshl_b32 s3, s5, s3
; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
; GFX6-NEXT: s_lshl_b32 s2, s2, 16
; GFX6-NEXT: s_or_b32 s0, s0, s2
@@ -1129,36 +1129,36 @@ define <4 x float> @v_shl_v8i16(<8 x i16> %value, <8 x i16> %amount) {
; GFX6: ; %bb.0:
; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX6-NEXT: v_lshrrev_b32_e32 v8, 16, v0
-; GFX6-NEXT: v_lshrrev_b32_e32 v12, 16, v4
-; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v0, v4, v0
-; GFX6-NEXT: v_lshlrev_b32_e32 v4, v12, v8
+; GFX6-NEXT: v_and_b32_e32 v12, 0xffff, v4
+; GFX6-NEXT: v_bfe_u32 v4, v4, 16, 16
+; GFX6-NEXT: v_lshlrev_b32_e32 v4, v4, v8
; GFX6-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX6-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX6-NEXT: v_and_b32_e32 v5, 0xffff, v5
+; GFX6-NEXT: v_lshlrev_b32_e32 v0, v12, v0
+; GFX6-NEXT: v_and_b32_e32 v8, 0xffff, v5
+; GFX6-NEXT: v_bfe_u32 v5, v5, 16, 16
; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX6-NEXT: v_lshlrev_b32_e32 v1, v5, v1
-; GFX6-NEXT: v_lshlrev_b32_e32 v5, v13, v9
+; GFX6-NEXT: v_lshlrev_b32_e32 v5, v5, v9
; GFX6-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX6-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX6-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX6-NEXT: v_and_b32_e32 v6, 0xffff, v6
+; GFX6-NEXT: v_lshlrev_b32_e32 v1, v8, v1
+; GFX6-NEXT: v_and_b32_e32 v8, 0xffff, v6
+; GFX6-NEXT: v_bfe_u32 v6, v6, 16, 16
; GFX6-NEXT: v_or_b32_e32 v0, v0, v4
; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v5
-; GFX6-NEXT: v_lshlrev_b32_e32 v2, v6, v2
-; GFX6-NEXT: v_lshlrev_b32_e32 v6, v14, v10
+; GFX6-NEXT: v_lshlrev_b32_e32 v6, v6, v10
; GFX6-NEXT: v_and_b32_e32 v1, 0xffff, v1
; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v4
; GFX6-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX6-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX6-NEXT: v_and_b32_e32 v7, 0xffff, v7
+; GFX6-NEXT: v_lshlrev_b32_e32 v2, v8, v2
+; GFX6-NEXT: v_and_b32_e32 v8, 0xffff, v7
+; GFX6-NEXT: v_bfe_u32 v7, v7, 16, 16
; GFX6-NEXT: v_or_b32_e32 v1, v1, v4
; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v6
-; GFX6-NEXT: v_lshlrev_b32_e32 v3, v7, v3
-; GFX6-NEXT: v_lshlrev_b32_e32 v7, v15, v11
+; GFX6-NEXT: v_lshlrev_b32_e32 v7, v7, v11
; GFX6-NEXT: v_and_b32_e32 v2, 0xffff, v2
; GFX6-NEXT: v_lshlrev_b32_e32 v4, 16, v4
+; GFX6-NEXT: v_lshlrev_b32_e32 v3, v8, v3
; GFX6-NEXT: v_or_b32_e32 v2, v2, v4
; GFX6-NEXT: v_and_b32_e32 v4, 0xffff, v7
; GFX6-NEXT: v_and_b32_e32 v3, 0xffff, v3
@@ -1209,30 +1209,30 @@ define amdgpu_ps <4 x i32> @s_shl_v8i16(<8 x i16> inreg %value, <8 x i16> inreg
; GFX6-LABEL: s_shl_v8i16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_lshr_b32 s8, s0, 16
-; GFX6-NEXT: s_lshr_b32 s12, s4, 16
; GFX6-NEXT: s_lshl_b32 s0, s0, s4
-; GFX6-NEXT: s_lshl_b32 s4, s8, s12
+; GFX6-NEXT: s_bfe_u32 s4, s4, 0x100010
+; GFX6-NEXT: s_lshl_b32 s4, s8, s4
; GFX6-NEXT: s_lshr_b32 s9, s1, 16
-; GFX6-NEXT: s_lshr_b32 s13, s5, 16
-; GFX6-NEXT: s_and_b32 s4, s4, 0xffff
; GFX6-NEXT: s_lshl_b32 s1, s1, s5
-; GFX6-NEXT: s_lshl_b32 s5, s9, s13
+; GFX6-NEXT: s_bfe_u32 s5, s5, 0x100010
+; GFX6-NEXT: s_and_b32 s4, s4, 0xffff
+; GFX6-NEXT: s_lshl_b32 s5, s9, s5
; GFX6-NEXT: s_and_b32 s0, s0, 0xffff
; GFX6-NEXT: s_lshl_b32 s4, s4, 16
; GFX6-NEXT: s_lshr_b32 s10, s2, 16
-; GFX6-NEXT: s_lshr_b32 s14, s6, 16
+; GFX6-NEXT: s_lshl_b32 s2, s2, s6
+; GFX6-NEXT: s_bfe_u32 s6, s6, 0x100010
; GFX6-NEXT: s_or_b32 s0, s0, s4
; GFX6-NEXT: s_and_b32 s4, s5, 0xffff
-; GFX6-NEXT: s_lshl_b32 s2, s2, s6
-; GFX6-NEXT: s_lshl_b32 s6, s10, s14
+; GFX6-NEXT: s_lshl_b32 s6, s10, s6
; GFX6-NEXT: s_and_b32 s1, s1, 0xffff
; GFX6-NEXT: s_lshl_b32 s4, s4, 16
; GFX6-NEXT: s_lshr_b32 s11, s3, 16
-; GFX6-NEXT: s_lshr_b32 s15, s7, 16
+; GFX6-NEXT: s_lshl_b32 s3, s3, s7
+; GFX6-NEXT: s_bfe_u32 s7, s7, 0x100010
; GFX6-NEXT: s_or_b32 s1, s1, s4
; GFX6-NEXT: s_and_b32 s4, s6, 0xffff
-; GFX6-NEXT: s_lshl_b32 s3, s3, s7
-; GFX6-NEXT: s_lshl_b32 s7, s11, s15
+; GFX6-NEXT: s_lshl_b32 s7, s11, s7
; GFX6-NEXT: s_and_b32 s2, s2, 0xffff
; GFX6-NEXT: s_lshl_b32 s4, s4, 16
; GFX6-NEXT: s_or_b32 s2, s2, s4
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
index cbbd5e69bff12..467f02fe970ea 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cluster.id.ll
@@ -492,7 +492,7 @@ define amdgpu_kernel void @test_cluster_id_z(ptr addrspace(1) %out) #1 {
; CHECK-G-UNKNOWN-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; CHECK-G-UNKNOWN-NEXT: s_load_b64 s[2:3], s[0:1], 0x24 nv
; CHECK-G-UNKNOWN-NEXT: s_wait_xcnt 0x0
-; CHECK-G-UNKNOWN-NEXT: s_lshr_b32 s0, ttmp7, 16
+; CHECK-G-UNKNOWN-NEXT: s_bfe_u32 s0, ttmp7, 0x100010
; CHECK-G-UNKNOWN-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-G-UNKNOWN-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s0
; CHECK-G-UNKNOWN-NEXT: s_wait_kmcnt 0x0
@@ -574,7 +574,7 @@ define amdgpu_kernel void @test_cluster_id_z(ptr addrspace(1) %out) #1 {
; CHECK-G-MESA3D-NEXT: v_nop
; CHECK-G-MESA3D-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; CHECK-G-MESA3D-NEXT: s_load_b64 s[0:1], s[0:1], 0x0 nv
-; CHECK-G-MESA3D-NEXT: s_lshr_b32 s2, ttmp7, 16
+; CHECK-G-MESA3D-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; CHECK-G-MESA3D-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; CHECK-G-MESA3D-NEXT: v_dual_mov_b32 v1, 0 :: v_dual_mov_b32 v0, s2
; CHECK-G-MESA3D-NEXT: s_wait_kmcnt 0x0
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
index 173739fa8f35f..70209924cd077 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.intersect_ray.ll
@@ -228,77 +228,43 @@ define amdgpu_ps <4 x float> @image_bvh_intersect_ray_a16(i32 inreg %node_ptr, f
; GFX10-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX10-SDAG-NEXT: ; return to shader part epilog
;
-; GFX1013-GISEL-LABEL: image_bvh_intersect_ray_a16:
-; GFX1013-GISEL: ; %bb.0: ; %main_body
-; GFX1013-GISEL-NEXT: s_and_b32 s8, s8, 0xffff
-; GFX1013-GISEL-NEXT: s_mov_b32 s16, s9
-; GFX1013-GISEL-NEXT: v_alignbit_b32 v0, s8, s7, 16
-; GFX1013-GISEL-NEXT: s_lshr_b32 s9, s5, 16
-; GFX1013-GISEL-NEXT: s_and_b32 s5, s5, 0xffff
-; GFX1013-GISEL-NEXT: s_lshl_b32 s8, s9, 16
-; GFX1013-GISEL-NEXT: s_and_b32 s9, s7, 0xffff
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s7, v0
-; GFX1013-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
-; GFX1013-GISEL-NEXT: s_lshl_b32 s9, s9, 16
-; GFX1013-GISEL-NEXT: s_or_b32 s5, s5, s8
-; GFX1013-GISEL-NEXT: s_or_b32 s6, s6, s9
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v4, s4
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v5, s5
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v6, s6
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v7, s7
-; GFX1013-GISEL-NEXT: s_mov_b32 s17, s10
-; GFX1013-GISEL-NEXT: s_mov_b32 s18, s11
-; GFX1013-GISEL-NEXT: s_mov_b32 s19, s12
-; GFX1013-GISEL-NEXT: image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
-; GFX1013-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT: ; return to shader part epilog
-;
-; GFX1030-GISEL-LABEL: image_bvh_intersect_ray_a16:
-; GFX1030-GISEL: ; %bb.0: ; %main_body
-; GFX1030-GISEL-NEXT: s_mov_b32 s16, s9
-; GFX1030-GISEL-NEXT: s_lshr_b32 s9, s5, 16
-; GFX1030-GISEL-NEXT: s_and_b32 s5, s5, 0xffff
-; GFX1030-GISEL-NEXT: s_lshl_b32 s9, s9, 16
-; GFX1030-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
-; GFX1030-GISEL-NEXT: s_or_b32 s5, s5, s9
-; GFX1030-GISEL-NEXT: s_and_b32 s9, s7, 0xffff
-; GFX1030-GISEL-NEXT: s_and_b32 s8, s8, 0xffff
-; GFX1030-GISEL-NEXT: s_lshl_b32 s9, s9, 16
-; GFX1030-GISEL-NEXT: v_alignbit_b32 v7, s8, s7, 16
-; GFX1030-GISEL-NEXT: s_or_b32 s6, s6, s9
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v4, s4
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v5, s5
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v6, s6
-; GFX1030-GISEL-NEXT: s_mov_b32 s17, s10
-; GFX1030-GISEL-NEXT: s_mov_b32 s18, s11
-; GFX1030-GISEL-NEXT: s_mov_b32 s19, s12
-; GFX1030-GISEL-NEXT: image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
-; GFX1030-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT: ; return to shader part epilog
+; GFX10-GISEL-LABEL: image_bvh_intersect_ray_a16:
+; GFX10-GISEL: ; %bb.0: ; %main_body
+; GFX10-GISEL-NEXT: s_mov_b32 s16, s9
+; GFX10-GISEL-NEXT: s_bfe_u32 s9, s5, 0x100010
+; GFX10-GISEL-NEXT: s_and_b32 s5, s5, 0xffff
+; GFX10-GISEL-NEXT: s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT: s_and_b32 s8, s8, 0xffff
+; GFX10-GISEL-NEXT: s_or_b32 s5, s5, s9
+; GFX10-GISEL-NEXT: s_and_b32 s9, s7, 0xffff
+; GFX10-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
+; GFX10-GISEL-NEXT: s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT: s_bfe_u32 s7, s7, 0x100010
+; GFX10-GISEL-NEXT: s_lshl_b32 s8, s8, 16
+; GFX10-GISEL-NEXT: s_or_b32 s6, s6, s9
+; GFX10-GISEL-NEXT: s_or_b32 s7, s7, s8
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v4, s4
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v5, s5
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v6, s6
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v7, s7
+; GFX10-GISEL-NEXT: s_mov_b32 s17, s10
+; GFX10-GISEL-NEXT: s_mov_b32 s18, s11
+; GFX10-GISEL-NEXT: s_mov_b32 s19, s12
+; GFX10-GISEL-NEXT: image_bvh_intersect_ray v[0:3], v[0:7], s[16:19] a16
+; GFX10-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s0, v0
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s1, v1
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s2, v2
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s3, v3
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT: ; return to shader part epilog
;
; GFX11-SDAG-LABEL: image_bvh_intersect_ray_a16:
; GFX11-SDAG: ; %bb.0: ; %main_body
@@ -530,79 +496,44 @@ define amdgpu_ps <4 x float> @image_bvh64_intersect_ray_a16(i64 inreg %node_ptr,
; GFX10-SDAG-NEXT: s_waitcnt vmcnt(0)
; GFX10-SDAG-NEXT: ; return to shader part epilog
;
-; GFX1013-GISEL-LABEL: image_bvh64_intersect_ray_a16:
-; GFX1013-GISEL: ; %bb.0: ; %main_body
-; GFX1013-GISEL-NEXT: s_and_b32 s9, s9, 0xffff
-; GFX1013-GISEL-NEXT: s_mov_b32 s16, s10
-; GFX1013-GISEL-NEXT: v_alignbit_b32 v0, s9, s8, 16
-; GFX1013-GISEL-NEXT: s_lshr_b32 s10, s6, 16
-; GFX1013-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
-; GFX1013-GISEL-NEXT: s_lshl_b32 s9, s10, 16
-; GFX1013-GISEL-NEXT: s_and_b32 s10, s8, 0xffff
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s8, v0
-; GFX1013-GISEL-NEXT: s_and_b32 s7, s7, 0xffff
-; GFX1013-GISEL-NEXT: s_lshl_b32 s10, s10, 16
-; GFX1013-GISEL-NEXT: s_or_b32 s6, s6, s9
-; GFX1013-GISEL-NEXT: s_or_b32 s7, s7, s10
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v4, s4
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v5, s5
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v6, s6
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v7, s7
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v8, s8
-; GFX1013-GISEL-NEXT: s_mov_b32 s17, s11
-; GFX1013-GISEL-NEXT: s_mov_b32 s18, s12
-; GFX1013-GISEL-NEXT: s_mov_b32 s19, s13
-; GFX1013-GISEL-NEXT: image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
-; GFX1013-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX1013-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1013-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1013-GISEL-NEXT: ; return to shader part epilog
-;
-; GFX1030-GISEL-LABEL: image_bvh64_intersect_ray_a16:
-; GFX1030-GISEL: ; %bb.0: ; %main_body
-; GFX1030-GISEL-NEXT: s_mov_b32 s16, s10
-; GFX1030-GISEL-NEXT: s_lshr_b32 s10, s6, 16
-; GFX1030-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
-; GFX1030-GISEL-NEXT: s_lshl_b32 s10, s10, 16
-; GFX1030-GISEL-NEXT: s_and_b32 s7, s7, 0xffff
-; GFX1030-GISEL-NEXT: s_or_b32 s6, s6, s10
-; GFX1030-GISEL-NEXT: s_and_b32 s10, s8, 0xffff
-; GFX1030-GISEL-NEXT: s_and_b32 s9, s9, 0xffff
-; GFX1030-GISEL-NEXT: s_lshl_b32 s10, s10, 16
-; GFX1030-GISEL-NEXT: v_alignbit_b32 v8, s9, s8, 16
-; GFX1030-GISEL-NEXT: s_or_b32 s7, s7, s10
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v4, s4
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v5, s5
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v6, s6
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v7, s7
-; GFX1030-GISEL-NEXT: s_mov_b32 s17, s11
-; GFX1030-GISEL-NEXT: s_mov_b32 s18, s12
-; GFX1030-GISEL-NEXT: s_mov_b32 s19, s13
-; GFX1030-GISEL-NEXT: image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
-; GFX1030-GISEL-NEXT: s_waitcnt vmcnt(0)
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s0, v0
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s1, v1
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s2, v2
-; GFX1030-GISEL-NEXT: v_readfirstlane_b32 s3, v3
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v0, s0
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v1, s1
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v2, s2
-; GFX1030-GISEL-NEXT: v_mov_b32_e32 v3, s3
-; GFX1030-GISEL-NEXT: ; return to shader part epilog
+; GFX10-GISEL-LABEL: image_bvh64_intersect_ray_a16:
+; GFX10-GISEL: ; %bb.0: ; %main_body
+; GFX10-GISEL-NEXT: s_mov_b32 s16, s10
+; GFX10-GISEL-NEXT: s_bfe_u32 s10, s6, 0x100010
+; GFX10-GISEL-NEXT: s_and_b32 s6, s6, 0xffff
+; GFX10-GISEL-NEXT: s_lshl_b32 s10, s10, 16
+; GFX10-GISEL-NEXT: s_and_b32 s9, s9, 0xffff
+; GFX10-GISEL-NEXT: s_or_b32 s6, s6, s10
+; GFX10-GISEL-NEXT: s_and_b32 s10, s8, 0xffff
+; GFX10-GISEL-NEXT: s_and_b32 s7, s7, 0xffff
+; GFX10-GISEL-NEXT: s_lshl_b32 s10, s10, 16
+; GFX10-GISEL-NEXT: s_bfe_u32 s8, s8, 0x100010
+; GFX10-GISEL-NEXT: s_lshl_b32 s9, s9, 16
+; GFX10-GISEL-NEXT: s_or_b32 s7, s7, s10
+; GFX10-GISEL-NEXT: s_or_b32 s8, s8, s9
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v4, s4
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v5, s5
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v6, s6
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v7, s7
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v8, s8
+; GFX10-GISEL-NEXT: s_mov_b32 s17, s11
+; GFX10-GISEL-NEXT: s_mov_b32 s18, s12
+; GFX10-GISEL-NEXT: s_mov_b32 s19, s13
+; GFX10-GISEL-NEXT: image_bvh64_intersect_ray v[0:3], v[0:8], s[16:19] a16
+; GFX10-GISEL-NEXT: s_waitcnt vmcnt(0)
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s0, v0
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s1, v1
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s2, v2
+; GFX10-GISEL-NEXT: v_readfirstlane_b32 s3, v3
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v0, s0
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v1, s1
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v2, s2
+; GFX10-GISEL-NEXT: v_mov_b32_e32 v3, s3
+; GFX10-GISEL-NEXT: ; return to shader part epilog
;
; GFX11-SDAG-LABEL: image_bvh64_intersect_ray_a16:
; GFX11-SDAG: ; %bb.0: ; %main_body
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
index 44ff13dd81303..2c25f250d8439 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-hsa.ll
@@ -29,7 +29,7 @@ define amdgpu_kernel void @workgroup_ids_kernel() {
; GFX9ARCH-GISEL: ; %bb.0: ; %.entry
; GFX9ARCH-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX9ARCH-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -49,7 +49,7 @@ define amdgpu_kernel void @workgroup_ids_kernel() {
; GFX12-GISEL: ; %bb.0: ; %.entry
; GFX12-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX12-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -251,7 +251,7 @@ define void @workgroup_ids_device_func(ptr addrspace(1) %outx, ptr addrspace(1)
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v6, ttmp9
; GFX9ARCH-GISEL-NEXT: s_and_b32 s4, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT: s_lshr_b32 s5, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT: s_bfe_u32 s5, ttmp7, 0x100010
; GFX9ARCH-GISEL-NEXT: global_store_dword v[0:1], v6, off
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v0, s4
@@ -262,27 +262,49 @@ define void @workgroup_ids_device_func(ptr addrspace(1) %outx, ptr addrspace(1)
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX9ARCH-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-LABEL: workgroup_ids_device_func:
-; GFX12: ; %bb.0:
-; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: s_wait_expcnt 0x0
-; GFX12-NEXT: s_wait_samplecnt 0x0
-; GFX12-NEXT: s_wait_bvhcnt 0x0
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
-; GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_mov_b32_e32 v8, s1
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: s_setpc_b64 s[30:31]
+; GFX12-SDAG-LABEL: workgroup_ids_device_func:
+; GFX12-SDAG: ; %bb.0:
+; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-SDAG-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: v_mov_b32_e32 v8, s1
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: workgroup_ids_device_func:
+; GFX12-GISEL: ; %bb.0:
+; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-GISEL-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
+; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT: v_mov_b32_e32 v8, s1
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
%id.x = call i32 @llvm.amdgcn.workgroup.id.x()
%id.y = call i32 @llvm.amdgcn.workgroup.id.y()
%id.z = call i32 @llvm.amdgcn.workgroup.id.z()
@@ -299,4 +321,5 @@ declare void @llvm.amdgcn.raw.ptr.buffer.store.v3i32(<3 x i32>, ptr addrspace(8)
attributes #0 = { nounwind "amdgpu-no-workgroup-id-y" "amdgpu-no-cluster-id-y" "amdgpu-no-workgroup-id-z" "amdgpu-no-cluster-id-z" }
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
; GFX9ARCH: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
index 63d02e09d611e..cb2539b75438d 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-opt.ll
@@ -275,7 +275,7 @@ define void @test_workgroup_id_z_non_kernel(ptr addrspace(1) %out) {
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_bfe_u32 s0, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s2, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s0, s1, s0
@@ -314,7 +314,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_used(ptr addrspace(1) %out
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
; GFX1250-GISEL-NEXT: s_bfe_u32 s0, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_add_co_i32 s0, s0, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s2, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s1, s1, s0
@@ -343,7 +343,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_not_used(ptr addrspace(1)
; GFX1250-GISEL: ; %bb.0:
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: s_lshr_b32 s0, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s0, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s0
; GFX1250-GISEL-NEXT: global_store_b32 v[0:1], v2, off
@@ -371,7 +371,7 @@ define void @test_workgroup_id_z_non_kernel_optimized_fixed(ptr addrspace(1) %ou
; GFX1250-GISEL: ; %bb.0:
; GFX1250-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
; GFX1250-GISEL-NEXT: s_wait_kmcnt 0x0
-; GFX1250-GISEL-NEXT: s_lshr_b32 s0, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s0, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_bfe_u32 s1, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
; GFX1250-GISEL-NEXT: s_lshl1_add_u32 s0, s0, s1
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
index b2ee9119fbcad..e3c4cab6fa2f0 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics-pal.ll
@@ -36,7 +36,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX9ARCH-GISEL: ; %bb.0: ; %.entry
; GFX9ARCH-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX9ARCH-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -56,7 +56,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX12-GISEL: ; %bb.0: ; %.entry
; GFX12-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX12-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -203,7 +203,7 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v6, ttmp9
; GFX9ARCH-GISEL-NEXT: s_and_b32 s34, ttmp7, 0xffff
-; GFX9ARCH-GISEL-NEXT: s_lshr_b32 s35, ttmp7, 16
+; GFX9ARCH-GISEL-NEXT: s_bfe_u32 s35, ttmp7, 0x100010
; GFX9ARCH-GISEL-NEXT: global_store_dword v[0:1], v6, off
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX9ARCH-GISEL-NEXT: v_mov_b32_e32 v0, s34
@@ -214,27 +214,49 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
; GFX9ARCH-GISEL-NEXT: s_waitcnt vmcnt(0)
; GFX9ARCH-GISEL-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-LABEL: workgroup_ids_gfx:
-; GFX12: ; %bb.0:
-; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: s_wait_expcnt 0x0
-; GFX12-NEXT: s_wait_samplecnt 0x0
-; GFX12-NEXT: s_wait_bvhcnt 0x0
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
-; GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_mov_b32_e32 v8, s1
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
-; GFX12-NEXT: s_wait_storecnt 0x0
-; GFX12-NEXT: s_setpc_b64 s[30:31]
+; GFX12-SDAG-LABEL: workgroup_ids_gfx:
+; GFX12-SDAG: ; %bb.0:
+; GFX12-SDAG-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-SDAG-NEXT: s_wait_expcnt 0x0
+; GFX12-SDAG-NEXT: s_wait_samplecnt 0x0
+; GFX12-SDAG-NEXT: s_wait_bvhcnt 0x0
+; GFX12-SDAG-NEXT: s_wait_kmcnt 0x0
+; GFX12-SDAG-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-SDAG-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GFX12-SDAG-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-SDAG-NEXT: v_mov_b32_e32 v8, s1
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-SDAG-NEXT: s_wait_storecnt 0x0
+; GFX12-SDAG-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX12-GISEL-LABEL: workgroup_ids_gfx:
+; GFX12-GISEL: ; %bb.0:
+; GFX12-GISEL-NEXT: s_wait_loadcnt_dscnt 0x0
+; GFX12-GISEL-NEXT: s_wait_expcnt 0x0
+; GFX12-GISEL-NEXT: s_wait_samplecnt 0x0
+; GFX12-GISEL-NEXT: s_wait_bvhcnt 0x0
+; GFX12-GISEL-NEXT: s_wait_kmcnt 0x0
+; GFX12-GISEL-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT: v_dual_mov_b32 v6, ttmp9 :: v_dual_mov_b32 v7, s0
+; GFX12-GISEL-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
+; GFX12-GISEL-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GFX12-GISEL-NEXT: v_mov_b32_e32 v8, s1
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[0:1], v6, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[2:3], v7, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: global_store_b32 v[4:5], v8, off scope:SCOPE_SYS
+; GFX12-GISEL-NEXT: s_wait_storecnt 0x0
+; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
%id.x = call i32 @llvm.amdgcn.workgroup.id.x()
%id.y = call i32 @llvm.amdgcn.workgroup.id.y()
%id.z = call i32 @llvm.amdgcn.workgroup.id.z()
@@ -243,3 +265,5 @@ define amdgpu_gfx void @workgroup_ids_gfx(ptr addrspace(1) %outx, ptr addrspace(
store volatile i32 %id.z, ptr addrspace(1) %outz
ret void
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
index 3ffe9c76175ae..4d15b36584716 100644
--- a/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
+++ b/llvm/test/CodeGen/AMDGPU/lower-work-group-id-intrinsics.ll
@@ -21,7 +21,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX9-GISEL: ; %bb.0: ; %.entry
; GFX9-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX9-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX9-GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX9-GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -41,7 +41,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX12-GISEL: ; %bb.0: ; %.entry
; GFX12-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX12-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -106,7 +106,7 @@ define amdgpu_cs void @_amdgpu_cs_main() {
; GFX1250-GISEL-NEXT: s_cmp_eq_u32 s2, 0
; GFX1250-GISEL-NEXT: s_cselect_b32 s1, s3, s4
; GFX1250-GISEL-NEXT: s_bfe_u32 s3, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT: s_lshr_b32 s4, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s4, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_add_co_i32 s3, s3, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s5, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s3, s4, s3
@@ -144,7 +144,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
; GFX9-GISEL: ; %bb.0: ; %.entry
; GFX9-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX9-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX9-GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX9-GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -164,7 +164,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
; GFX12-GISEL: ; %bb.0: ; %.entry
; GFX12-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX12-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -191,7 +191,7 @@ define amdgpu_cs void @workgroup_id_no_clusters() "amdgpu-cluster-dims"="0,0,0"
; GFX1250-GISEL-NEXT: global_prefetch_b8 v0, s[64:65] scope:SCOPE_SE
; GFX1250-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX1250-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX1250-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: v_mov_b64_e32 v[0:1], s[0:1]
; GFX1250-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX1250-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -222,7 +222,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
; GFX9-GISEL: ; %bb.0: ; %.entry
; GFX9-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX9-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX9-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX9-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, s0
; GFX9-GISEL-NEXT: v_mov_b32_e32 v1, s1
; GFX9-GISEL-NEXT: v_mov_b32_e32 v2, s2
@@ -242,7 +242,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
; GFX12-GISEL: ; %bb.0: ; %.entry
; GFX12-GISEL-NEXT: s_mov_b32 s0, ttmp9
; GFX12-GISEL-NEXT: s_and_b32 s1, ttmp7, 0xffff
-; GFX12-GISEL-NEXT: s_lshr_b32 s2, ttmp7, 16
+; GFX12-GISEL-NEXT: s_bfe_u32 s2, ttmp7, 0x100010
; GFX12-GISEL-NEXT: v_dual_mov_b32 v0, s0 :: v_dual_mov_b32 v1, s1
; GFX12-GISEL-NEXT: v_mov_b32_e32 v2, s2
; GFX12-GISEL-NEXT: buffer_store_b96 v[0:2], off, s[0:3], null
@@ -280,7 +280,7 @@ define amdgpu_cs void @workgroup_id_optimized() "amdgpu-cluster-dims"="2,3,4" {
; GFX1250-GISEL-NEXT: s_and_b32 s0, ttmp6, 15
; GFX1250-GISEL-NEXT: s_bfe_u32 s2, ttmp6, 0x40004
; GFX1250-GISEL-NEXT: s_mul_i32 s1, s1, 3
-; GFX1250-GISEL-NEXT: s_lshr_b32 s3, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s3, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_bfe_u32 s4, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_lshl1_add_u32 s0, ttmp9, s0
; GFX1250-GISEL-NEXT: s_add_co_i32 s1, s2, s1
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
index 01b7293dcd7ab..d42b9fcaaccd5 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-add.ll
@@ -270,7 +270,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v2
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v1, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -289,7 +289,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -308,7 +308,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_add_v4i8:
@@ -329,7 +329,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_add_nc_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v4i8:
@@ -371,7 +371,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v1.l, v3.l
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v4i8:
@@ -381,7 +381,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v4i8:
@@ -435,7 +435,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v1.l, v3.l
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v4i8:
@@ -449,7 +449,7 @@ define i8 @test_vector_reduce_add_v4i8(<4 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.add.v4i8(<4 x i8> %v)
@@ -484,7 +484,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v2
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v1, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -511,7 +511,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -538,7 +538,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_add_v8i8:
@@ -567,7 +567,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_add_nc_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v8i8:
@@ -624,7 +624,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, v1.h
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v8i8:
@@ -639,7 +639,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v8i8:
@@ -708,7 +708,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v8i8:
@@ -727,7 +727,7 @@ define i8 @test_vector_reduce_add_v8i8(<8 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.add.v8i8(<8 x i8> %v)
@@ -778,7 +778,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v2
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v1, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -821,7 +821,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -864,7 +864,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_add_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_add_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_add_v16i8:
@@ -909,7 +909,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_add_nc_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_add_v16i8:
@@ -994,7 +994,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, v1.h
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1019,7 +1019,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1116,7 +1116,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.h, v0.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_add_nc_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_add_v16i8:
@@ -1145,7 +1145,7 @@ define i8 @test_vector_reduce_add_v16i8(<16 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_add_nc_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.add.v16i8(<16 x i8> %v)
@@ -1165,7 +1165,7 @@ define i16 @test_vector_reduce_add_v2i16(<2 x i16> %v) {
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v1, 16, v0
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v2i16:
@@ -1463,7 +1463,7 @@ define i16 @test_vector_reduce_add_v4i16(<4 x i16> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v2, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v4i16:
@@ -1645,7 +1645,7 @@ define i16 @test_vector_reduce_add_v8i16(<8 x i16> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v2, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v8i16:
@@ -1893,7 +1893,7 @@ define i16 @test_vector_reduce_add_v16i16(<16 x i16> %v) {
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
; GFX7-GISEL-NEXT: v_add_i32_e32 v1, vcc, v2, v3
; GFX7-GISEL-NEXT: v_add_i32_e32 v0, vcc, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_add_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
index 0ae4b05aa7c89..7bd805dd8c575 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-and.ll
@@ -297,7 +297,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -315,7 +315,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -333,7 +333,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_and_v4i8:
@@ -351,7 +351,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX10-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v4i8:
@@ -381,7 +381,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX11-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v4i8:
@@ -423,7 +423,7 @@ define i8 @test_vector_reduce_and_v4i8(<4 x i8> %v) {
; GFX12-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.and.v4i8(<4 x i8> %v)
@@ -454,7 +454,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -480,7 +480,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -506,7 +506,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_and_v8i8:
@@ -532,7 +532,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX10-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v8i8:
@@ -577,7 +577,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX11-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v8i8:
@@ -634,7 +634,7 @@ define i8 @test_vector_reduce_and_v8i8(<8 x i8> %v) {
; GFX12-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.and.v8i8(<8 x i8> %v)
@@ -681,7 +681,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -723,7 +723,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -765,7 +765,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_and_v16i8:
@@ -807,7 +807,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v2
; GFX10-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX10-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_and_v16i8:
@@ -880,7 +880,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX11-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_and_v16i8:
@@ -965,7 +965,7 @@ define i8 @test_vector_reduce_and_v16i8(<16 x i8> %v) {
; GFX12-GISEL-NEXT: v_and_b32_e32 v1, v1, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_and_b32_e32 v0, v0, v1
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.and.v16i8(<16 x i8> %v)
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
index b931b312ff251..3a5a705e479f6 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-mul.ll
@@ -328,7 +328,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v2
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v1, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -346,7 +346,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -364,7 +364,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_mul_v4i8:
@@ -382,7 +382,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -412,7 +412,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v1.l, v3.l
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -422,7 +422,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -464,7 +464,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v1.l, v3.l
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v4i8:
@@ -478,7 +478,7 @@ define i8 @test_vector_reduce_mul_v4i8(<4 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.mul.v4i8(<4 x i8> %v)
@@ -509,7 +509,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v2
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v1, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -535,7 +535,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -561,7 +561,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_mul_v8i8:
@@ -587,7 +587,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -632,7 +632,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v0.h, v1.h
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -647,7 +647,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -704,7 +704,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v0.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v8i8:
@@ -723,7 +723,7 @@ define i8 @test_vector_reduce_mul_v8i8(<8 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.mul.v8i8(<8 x i8> %v)
@@ -770,7 +770,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v2
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v1, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -812,7 +812,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -854,7 +854,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_mul_lo_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_mul_v16i8:
@@ -896,7 +896,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v2
; GFX10-GISEL-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -969,7 +969,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v0.h, v1.h
; GFX11-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX11-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -994,7 +994,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX11-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX11-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -1079,7 +1079,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.h, v0.h, v1.h
; GFX12-GISEL-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-TRUE16-NEXT: v_mul_lo_u16 v0.l, v0.l, v0.h
-; GFX12-GISEL-TRUE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-TRUE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-TRUE16-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-GISEL-FAKE16-LABEL: test_vector_reduce_mul_v16i8:
@@ -1108,7 +1108,7 @@ define i8 @test_vector_reduce_mul_v16i8(<16 x i8> %v) {
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v1, v1, v3
; GFX12-GISEL-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-FAKE16-NEXT: v_mul_lo_u16 v0, v0, v1
-; GFX12-GISEL-FAKE16-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-FAKE16-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-FAKE16-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.mul.v16i8(<16 x i8> %v)
@@ -1116,21 +1116,13 @@ entry:
}
define i16 @test_vector_reduce_mul_v2i16(<2 x i16> %v) {
-; GFX7-SDAG-LABEL: test_vector_reduce_mul_v2i16:
-; GFX7-SDAG: ; %bb.0: ; %entry
-; GFX7-SDAG-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-SDAG-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX7-SDAG-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-SDAG-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-SDAG-NEXT: s_setpc_b64 s[30:31]
-;
-; GFX7-GISEL-LABEL: test_vector_reduce_mul_v2i16:
-; GFX7-GISEL: ; %bb.0: ; %entry
-; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v1, 16, v0
-; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
-; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
+; GFX7-LABEL: test_vector_reduce_mul_v2i16:
+; GFX7: ; %bb.0: ; %entry
+; GFX7-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX7-NEXT: v_lshrrev_b32_e32 v1, 16, v0
+; GFX7-NEXT: v_mul_lo_u32 v0, v0, v1
+; GFX7-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX7-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-LABEL: test_vector_reduce_mul_v2i16:
; GFX8: ; %bb.0: ; %entry
@@ -1404,7 +1396,7 @@ define i16 @test_vector_reduce_mul_v4i16(<4 x i16> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v2, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v4i16:
@@ -1564,7 +1556,7 @@ define i16 @test_vector_reduce_mul_v8i16(<8 x i16> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v2, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v8i16:
@@ -1807,7 +1799,7 @@ define i16 @test_vector_reduce_mul_v16i16(<16 x i16> %v) {
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
; GFX7-GISEL-NEXT: v_mul_lo_u32 v1, v2, v3
; GFX7-GISEL-NEXT: v_mul_lo_u32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_mul_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
index 087832601598a..d1fa2b23282c1 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umax.ll
@@ -1593,10 +1593,10 @@ define i16 @test_vector_reduce_umax_v3i16(<3 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umax_v3i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_max3_u32 v0, v0, v2, v1
+; GFX7-GISEL-NEXT: v_max3_u32 v0, v2, v0, v1
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-LABEL: test_vector_reduce_umax_v3i16:
@@ -1730,15 +1730,15 @@ define i16 @test_vector_reduce_umax_v4i16(<4 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umax_v4i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v3, 16, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_max_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT: v_max3_u32 v0, v0, v1, v2
-; GFX7-GISEL-NEXT: v_max_u32_e32 v1, 0, v2
-; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v1
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_max3_u32 v1, v2, v3, v0
+; GFX7-GISEL-NEXT: v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v1, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umax_v4i16:
@@ -1886,22 +1886,22 @@ define i16 @test_vector_reduce_umax_v8i16(<8 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umax_v8i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v2
+; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v1
+; GFX7-GISEL-NEXT: v_and_b32_e32 v7, 0xffff, v3
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v4, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v5, 0xffff, v2
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT: v_max_u32_e32 v6, v6, v7
; GFX7-GISEL-NEXT: v_max_u32_e32 v1, v1, v3
-; GFX7-GISEL-NEXT: v_max_u32_e32 v3, v5, v7
+; GFX7-GISEL-NEXT: v_max3_u32 v3, v4, v5, v6
; GFX7-GISEL-NEXT: v_max3_u32 v0, v0, v2, v1
-; GFX7-GISEL-NEXT: v_max3_u32 v1, v4, v6, v3
-; GFX7-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_max_u32_e32 v1, 0, v1
-; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_max_u32_e32 v1, v3, v0
+; GFX7-GISEL-NEXT: v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v1, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umax_v8i16:
@@ -2099,35 +2099,35 @@ define i16 @test_vector_reduce_umax_v16i16(<16 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umax_v16i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v6
-; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT: v_and_b32_e32 v7, 0xffff, v7
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v8, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v12, 16, v4
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v5, 0xffff, v5
+; GFX7-GISEL-NEXT: v_and_b32_e32 v12, 0xffff, v2
+; GFX7-GISEL-NEXT: v_and_b32_e32 v13, 0xffff, v6
+; GFX7-GISEL-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v6, v6, 16, 16
+; GFX7-GISEL-NEXT: v_max_u32_e32 v12, v12, v13
; GFX7-GISEL-NEXT: v_max_u32_e32 v2, v2, v6
-; GFX7-GISEL-NEXT: v_max_u32_e32 v6, v10, v14
+; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v3
+; GFX7-GISEL-NEXT: v_and_b32_e32 v13, 0xffff, v7
+; GFX7-GISEL-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v7, v7, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v8, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v9, 0xffff, v4
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v4, v4, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v10, 0xffff, v1
+; GFX7-GISEL-NEXT: v_and_b32_e32 v11, 0xffff, v5
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v5, v5, 16, 16
; GFX7-GISEL-NEXT: v_max_u32_e32 v3, v3, v7
-; GFX7-GISEL-NEXT: v_max_u32_e32 v7, v11, v15
+; GFX7-GISEL-NEXT: v_max_u32_e32 v6, v6, v13
; GFX7-GISEL-NEXT: v_max3_u32 v0, v0, v4, v2
-; GFX7-GISEL-NEXT: v_max3_u32 v2, v8, v12, v6
; GFX7-GISEL-NEXT: v_max3_u32 v1, v1, v5, v3
-; GFX7-GISEL-NEXT: v_max3_u32 v3, v9, v13, v7
-; GFX7-GISEL-NEXT: v_max_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT: v_max3_u32 v0, v0, v1, v2
-; GFX7-GISEL-NEXT: v_max_u32_e32 v1, 0, v2
-; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v1, 16, v1
-; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_max3_u32 v7, v8, v9, v12
+; GFX7-GISEL-NEXT: v_max3_u32 v2, v10, v11, v6
+; GFX7-GISEL-NEXT: v_max_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_max3_u32 v1, v7, v2, v0
+; GFX7-GISEL-NEXT: v_max_u32_e32 v0, 0, v0
+; GFX7-GISEL-NEXT: v_lshlrev_b32_e32 v0, 16, v0
+; GFX7-GISEL-NEXT: v_or_b32_e32 v0, v1, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umax_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
index 5443cce424a5c..f3aa7eba3e105 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-umin.ll
@@ -315,7 +315,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
; GFX8-GISEL-NEXT: v_min_u16_sdwa v0, v0, v2 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
; GFX8-GISEL-NEXT: v_min_u16_sdwa v1, v1, v3 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
; GFX8-GISEL-NEXT: v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_umin_v4i8:
@@ -335,7 +335,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
; GFX9-GISEL-NEXT: v_and_b32_e32 v2, 0xff, v2
; GFX9-GISEL-NEXT: v_min_u16_sdwa v1, v1, v3 dst_sel:DWORD dst_unused:UNUSED_PAD src0_sel:BYTE_0 src1_sel:BYTE_0
; GFX9-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_umin_v4i8:
@@ -361,7 +361,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
; GFX10-GISEL-NEXT: v_and_b32_e32 v2, 0xff, v2
; GFX10-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v4i8:
@@ -409,7 +409,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
; GFX11-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX11-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v4i8:
@@ -469,7 +469,7 @@ define i8 @test_vector_reduce_umin_v4i8(<4 x i8> %v) {
; GFX12-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX12-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.umin.v4i8(<4 x i8> %v)
@@ -536,7 +536,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
; GFX8-GISEL-NEXT: v_min_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_min_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_umin_v8i8:
@@ -567,7 +567,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
; GFX9-GISEL-NEXT: v_min3_u16 v0, v0, v4, v2
; GFX9-GISEL-NEXT: v_min3_u16 v1, v1, v5, v3
; GFX9-GISEL-NEXT: v_min_u16_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_umin_v8i8:
@@ -607,7 +607,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
; GFX10-GISEL-NEXT: v_min3_u16 v0, v0, v4, v2
; GFX10-GISEL-NEXT: v_min3_u16 v1, v1, v5, v3
; GFX10-GISEL-NEXT: v_min_u16 v0, v0, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v8i8:
@@ -678,7 +678,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
; GFX11-GISEL-NEXT: v_min3_u16 v1, v1, v5, v3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_min_u16 v0, v0, v1
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v8i8:
@@ -761,7 +761,7 @@ define i8 @test_vector_reduce_umin_v8i8(<8 x i8> %v) {
; GFX12-GISEL-NEXT: v_min3_u16 v1, v1, v5, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_min_u16 v0, v0, v1
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.umin.v8i8(<8 x i8> %v)
@@ -873,7 +873,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
; GFX8-GISEL-NEXT: v_min_u16_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_min_u16_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_min_u16_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_umin_v16i8:
@@ -921,7 +921,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
; GFX9-GISEL-NEXT: v_min3_u16 v2, v2, v10, v6
; GFX9-GISEL-NEXT: v_min_u16_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_umin_v16i8:
@@ -990,7 +990,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
; GFX10-GISEL-NEXT: v_min3_u16 v2, v2, v10, v6
; GFX10-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX10-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v16i8:
@@ -1107,7 +1107,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
; GFX11-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX11-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_umin_v16i8:
@@ -1236,7 +1236,7 @@ define i8 @test_vector_reduce_umin_v16i8(<16 x i8> %v) {
; GFX12-GISEL-NEXT: v_min_u16 v1, v1, v3
; GFX12-GISEL-NEXT: v_min3_u16 v0, v0, v2, v1
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.umin.v16i8(<16 x i8> %v)
@@ -1374,10 +1374,10 @@ define i16 @test_vector_reduce_umin_v3i16(<3 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umin_v3i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_min3_u32 v0, v0, v2, v1
+; GFX7-GISEL-NEXT: v_min3_u32 v0, v2, v0, v1
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-LABEL: test_vector_reduce_umin_v3i16:
@@ -1512,12 +1512,12 @@ define i16 @test_vector_reduce_umin_v4i16(<4 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umin_v4i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v2, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v3, 16, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_min_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT: v_min3_u32 v0, v0, v1, v2
+; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v1
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_min3_u32 v0, v2, v3, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umin_v4i16:
@@ -1665,19 +1665,19 @@ define i16 @test_vector_reduce_umin_v8i16(<8 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umin_v8i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v5, 16, v1
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v7, 16, v3
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v4, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v6, 16, v2
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v2
+; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v1
+; GFX7-GISEL-NEXT: v_and_b32_e32 v7, 0xffff, v3
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v4, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v5, 0xffff, v2
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT: v_min_u32_e32 v6, v6, v7
; GFX7-GISEL-NEXT: v_min_u32_e32 v1, v1, v3
-; GFX7-GISEL-NEXT: v_min_u32_e32 v3, v5, v7
+; GFX7-GISEL-NEXT: v_min3_u32 v3, v4, v5, v6
; GFX7-GISEL-NEXT: v_min3_u32 v0, v0, v2, v1
-; GFX7-GISEL-NEXT: v_min3_u32 v1, v4, v6, v3
-; GFX7-GISEL-NEXT: v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_min_u32_e32 v0, v3, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umin_v8i16:
@@ -1875,32 +1875,32 @@ define i16 @test_vector_reduce_umin_v16i16(<16 x i16> %v) {
; GFX7-GISEL-LABEL: test_vector_reduce_umin_v16i16:
; GFX7-GISEL: ; %bb.0: ; %entry
; GFX7-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v10, 16, v2
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v11, 16, v3
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v14, 16, v6
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v15, 16, v7
-; GFX7-GISEL-NEXT: v_and_b32_e32 v2, 0xffff, v2
-; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v6
-; GFX7-GISEL-NEXT: v_and_b32_e32 v3, 0xffff, v3
-; GFX7-GISEL-NEXT: v_and_b32_e32 v7, 0xffff, v7
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v8, 16, v0
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v9, 16, v1
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v12, 16, v4
-; GFX7-GISEL-NEXT: v_lshrrev_b32_e32 v13, 16, v5
-; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xffff, v0
-; GFX7-GISEL-NEXT: v_and_b32_e32 v4, 0xffff, v4
-; GFX7-GISEL-NEXT: v_and_b32_e32 v1, 0xffff, v1
-; GFX7-GISEL-NEXT: v_and_b32_e32 v5, 0xffff, v5
+; GFX7-GISEL-NEXT: v_and_b32_e32 v12, 0xffff, v2
+; GFX7-GISEL-NEXT: v_and_b32_e32 v13, 0xffff, v6
+; GFX7-GISEL-NEXT: v_bfe_u32 v2, v2, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v6, v6, 16, 16
+; GFX7-GISEL-NEXT: v_min_u32_e32 v12, v12, v13
; GFX7-GISEL-NEXT: v_min_u32_e32 v2, v2, v6
-; GFX7-GISEL-NEXT: v_min_u32_e32 v6, v10, v14
+; GFX7-GISEL-NEXT: v_and_b32_e32 v6, 0xffff, v3
+; GFX7-GISEL-NEXT: v_and_b32_e32 v13, 0xffff, v7
+; GFX7-GISEL-NEXT: v_bfe_u32 v3, v3, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v7, v7, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v8, 0xffff, v0
+; GFX7-GISEL-NEXT: v_and_b32_e32 v9, 0xffff, v4
+; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v4, v4, 16, 16
+; GFX7-GISEL-NEXT: v_and_b32_e32 v10, 0xffff, v1
+; GFX7-GISEL-NEXT: v_and_b32_e32 v11, 0xffff, v5
+; GFX7-GISEL-NEXT: v_bfe_u32 v1, v1, 16, 16
+; GFX7-GISEL-NEXT: v_bfe_u32 v5, v5, 16, 16
; GFX7-GISEL-NEXT: v_min_u32_e32 v3, v3, v7
-; GFX7-GISEL-NEXT: v_min_u32_e32 v7, v11, v15
+; GFX7-GISEL-NEXT: v_min_u32_e32 v6, v6, v13
; GFX7-GISEL-NEXT: v_min3_u32 v0, v0, v4, v2
-; GFX7-GISEL-NEXT: v_min3_u32 v2, v8, v12, v6
; GFX7-GISEL-NEXT: v_min3_u32 v1, v1, v5, v3
-; GFX7-GISEL-NEXT: v_min3_u32 v3, v9, v13, v7
-; GFX7-GISEL-NEXT: v_min_u32_e32 v2, v2, v3
-; GFX7-GISEL-NEXT: v_min3_u32 v0, v0, v1, v2
+; GFX7-GISEL-NEXT: v_min3_u32 v7, v8, v9, v12
+; GFX7-GISEL-NEXT: v_min3_u32 v2, v10, v11, v6
+; GFX7-GISEL-NEXT: v_min_u32_e32 v0, v0, v1
+; GFX7-GISEL-NEXT: v_min3_u32 v0, v7, v2, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_umin_v16i16:
diff --git a/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll b/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
index b9a8ea279ad15..b68654a2f9ff4 100644
--- a/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
+++ b/llvm/test/CodeGen/AMDGPU/vector-reduce-xor.ll
@@ -292,7 +292,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -310,7 +310,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -328,7 +328,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_xor_v4i8:
@@ -345,7 +345,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX10-GISEL-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX10-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v4i8:
@@ -374,7 +374,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX11-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX11-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v4i8:
@@ -415,7 +415,7 @@ define i8 @test_vector_reduce_xor_v4i8(<4 x i8> %v) {
; GFX12-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
; GFX12-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.xor.v4i8(<4 x i8> %v)
@@ -446,7 +446,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -472,7 +472,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -498,7 +498,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_xor_v8i8:
@@ -522,7 +522,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX10-GISEL-NEXT: v_xor_b32_e32 v2, v2, v6
; GFX10-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX10-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v8i8:
@@ -565,7 +565,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX11-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX11-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v8i8:
@@ -620,7 +620,7 @@ define i8 @test_vector_reduce_xor_v8i8(<8 x i8> %v) {
; GFX12-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX12-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.xor.v8i8(<8 x i8> %v)
@@ -667,7 +667,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX7-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX7-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX7-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX7-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX7-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX8-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -709,7 +709,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX8-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX8-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX8-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX8-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX8-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX9-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -751,7 +751,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v2
; GFX9-GISEL-NEXT: v_xor_b32_e32 v1, v1, v3
; GFX9-GISEL-NEXT: v_xor_b32_e32 v0, v0, v1
-; GFX9-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX9-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX9-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX10-SDAG-LABEL: test_vector_reduce_xor_v16i8:
@@ -788,7 +788,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX10-GISEL-NEXT: v_xor3_b32 v2, v2, v10, v6
; GFX10-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX10-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
-; GFX10-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX10-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX10-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX11-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v16i8:
@@ -855,7 +855,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX11-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX11-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX11-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX11-GISEL-NEXT: s_setpc_b64 s[30:31]
;
; GFX12-SDAG-TRUE16-LABEL: test_vector_reduce_xor_v16i8:
@@ -934,7 +934,7 @@ define i8 @test_vector_reduce_xor_v16i8(<16 x i8> %v) {
; GFX12-GISEL-NEXT: v_xor3_b32 v1, v1, v5, v3
; GFX12-GISEL-NEXT: v_xor3_b32 v0, v0, v2, v1
; GFX12-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-GISEL-NEXT: v_bfe_u32 v0, v0, 0, 8
+; GFX12-GISEL-NEXT: v_and_b32_e32 v0, 0xff, v0
; GFX12-GISEL-NEXT: s_setpc_b64 s[30:31]
entry:
%res = call i8 @llvm.vector.reduce.xor.v16i8(<16 x i8> %v)
diff --git a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
index ac54854ff6bfe..ef02f967d645d 100644
--- a/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
+++ b/llvm/test/CodeGen/AMDGPU/workgroup-id-in-arch-sgprs.ll
@@ -191,38 +191,37 @@ define amdgpu_kernel void @workgroup_id_xy(ptr addrspace(1) %ptrx, ptr addrspace
}
define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspace(1) %ptry, ptr addrspace(1) %ptrz) {
-; GFX9-LABEL: workgroup_id_xyz:
-; GFX9: ; %bb.0:
-; GFX9-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
-; GFX9-NEXT: s_load_dwordx2 s[4:5], s[8:9], 0x10
-; GFX9-NEXT: v_mov_b32_e32 v0, ttmp9
-; GFX9-NEXT: v_mov_b32_e32 v1, 0
-; GFX9-NEXT: s_and_b32 s6, ttmp7, 0xffff
-; GFX9-NEXT: s_waitcnt lgkmcnt(0)
-; GFX9-NEXT: global_store_dword v1, v0, s[0:1]
-; GFX9-NEXT: v_mov_b32_e32 v0, s6
-; GFX9-NEXT: s_lshr_b32 s0, ttmp7, 16
-; GFX9-NEXT: global_store_dword v1, v0, s[2:3]
-; GFX9-NEXT: v_mov_b32_e32 v0, s0
-; GFX9-NEXT: global_store_dword v1, v0, s[4:5]
-; GFX9-NEXT: s_endpgm
+; GFX9-SDAG-LABEL: workgroup_id_xyz:
+; GFX9-SDAG: ; %bb.0:
+; GFX9-SDAG-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX9-SDAG-NEXT: s_load_dwordx2 s[4:5], s[8:9], 0x10
+; GFX9-SDAG-NEXT: v_mov_b32_e32 v0, ttmp9
+; GFX9-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX9-SDAG-NEXT: s_and_b32 s6, ttmp7, 0xffff
+; GFX9-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-SDAG-NEXT: global_store_dword v1, v0, s[0:1]
+; GFX9-SDAG-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-SDAG-NEXT: s_lshr_b32 s0, ttmp7, 16
+; GFX9-SDAG-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX9-SDAG-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-SDAG-NEXT: global_store_dword v1, v0, s[4:5]
+; GFX9-SDAG-NEXT: s_endpgm
;
-; GFX1200-LABEL: workgroup_id_xyz:
-; GFX1200: ; %bb.0:
-; GFX1200-NEXT: s_clause 0x1
-; GFX1200-NEXT: s_load_b128 s[0:3], s[4:5], 0x0
-; GFX1200-NEXT: s_load_b64 s[4:5], s[4:5], 0x10
-; GFX1200-NEXT: s_and_b32 s6, ttmp7, 0xffff
-; GFX1200-NEXT: v_dual_mov_b32 v0, ttmp9 :: v_dual_mov_b32 v1, 0
-; GFX1200-NEXT: s_lshr_b32 s7, ttmp7, 16
-; GFX1200-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX1200-NEXT: v_dual_mov_b32 v2, s6 :: v_dual_mov_b32 v3, s7
-; GFX1200-NEXT: s_wait_kmcnt 0x0
-; GFX1200-NEXT: s_clause 0x2
-; GFX1200-NEXT: global_store_b32 v1, v0, s[0:1]
-; GFX1200-NEXT: global_store_b32 v1, v2, s[2:3]
-; GFX1200-NEXT: global_store_b32 v1, v3, s[4:5]
-; GFX1200-NEXT: s_endpgm
+; GFX9-GISEL-LABEL: workgroup_id_xyz:
+; GFX9-GISEL: ; %bb.0:
+; GFX9-GISEL-NEXT: s_load_dwordx4 s[0:3], s[8:9], 0x0
+; GFX9-GISEL-NEXT: s_load_dwordx2 s[4:5], s[8:9], 0x10
+; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, ttmp9
+; GFX9-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX9-GISEL-NEXT: s_and_b32 s6, ttmp7, 0xffff
+; GFX9-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX9-GISEL-NEXT: global_store_dword v1, v0, s[0:1]
+; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, s6
+; GFX9-GISEL-NEXT: s_bfe_u32 s0, ttmp7, 0x100010
+; GFX9-GISEL-NEXT: global_store_dword v1, v0, s[2:3]
+; GFX9-GISEL-NEXT: v_mov_b32_e32 v0, s0
+; GFX9-GISEL-NEXT: global_store_dword v1, v0, s[4:5]
+; GFX9-GISEL-NEXT: s_endpgm
;
; GFX1250-SDAG-LABEL: workgroup_id_xyz:
; GFX1250-SDAG: ; %bb.0:
@@ -295,7 +294,7 @@ define amdgpu_kernel void @workgroup_id_xyz(ptr addrspace(1) %ptrx, ptr addrspac
; GFX1250-GISEL-NEXT: s_wait_xcnt 0x0
; GFX1250-GISEL-NEXT: s_cselect_b32 s4, s10, s11
; GFX1250-GISEL-NEXT: s_bfe_u32 s5, ttmp6, 0x40014
-; GFX1250-GISEL-NEXT: s_lshr_b32 s10, ttmp7, 16
+; GFX1250-GISEL-NEXT: s_bfe_u32 s10, ttmp7, 0x100010
; GFX1250-GISEL-NEXT: s_add_co_i32 s5, s5, 1
; GFX1250-GISEL-NEXT: s_bfe_u32 s11, ttmp6, 0x40008
; GFX1250-GISEL-NEXT: s_mul_i32 s5, s10, s5
@@ -341,5 +340,3 @@ declare i32 @llvm.amdgcn.workgroup.id.y()
declare i32 @llvm.amdgcn.workgroup.id.z()
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; GFX1250: {{.*}}
-; GFX9-GISEL: {{.*}}
-; GFX9-SDAG: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
index 07937b347a622..3f145cd1c59e1 100644
--- a/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
+++ b/llvm/test/CodeGen/AMDGPU/workitem-intrinsic-opts.ll
@@ -240,7 +240,7 @@ define i1 @workgroup_zero() {
; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
; GISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GISEL-GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GISEL-GFX12-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
; GISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
; GISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
@@ -280,23 +280,41 @@ define i1 @workgroup_nonzero() {
; GFX942-NEXT: v_mov_b32_e32 v0, s0
; GFX942-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-LABEL: workgroup_nonzero:
-; GFX12: ; %bb.0: ; %entry
-; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: s_wait_expcnt 0x0
-; GFX12-NEXT: s_wait_samplecnt 0x0
-; GFX12-NEXT: s_wait_bvhcnt 0x0
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT: s_or_b32 s0, ttmp9, s0
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_or_b32 s0, s0, s1
-; GFX12-NEXT: s_cselect_b32 s0, 1, 0
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_mov_b32_e32 v0, s0
-; GFX12-NEXT: s_setpc_b64 s[30:31]
+; DAGISEL-GFX12-LABEL: workgroup_nonzero:
+; DAGISEL-GFX12: ; %bb.0: ; %entry
+; DAGISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_expcnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_samplecnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_kmcnt 0x0
+; DAGISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
+; DAGISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
+; DAGISEL-GFX12-NEXT: s_cselect_b32 s0, 1, 0
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: v_mov_b32_e32 v0, s0
+; DAGISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
+;
+; GISEL-GFX12-LABEL: workgroup_nonzero:
+; GISEL-GFX12: ; %bb.0: ; %entry
+; GISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GISEL-GFX12-NEXT: s_wait_expcnt 0x0
+; GISEL-GFX12-NEXT: s_wait_samplecnt 0x0
+; GISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
+; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
+; GISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
+; GISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
+; GISEL-GFX12-NEXT: s_cselect_b32 s0, 1, 0
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: v_mov_b32_e32 v0, s0
+; GISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%0 = tail call i32 @llvm.amdgcn.workgroup.id.x()
%1 = tail call i32 @llvm.amdgcn.workgroup.id.y()
@@ -335,28 +353,51 @@ define i1 @workitem_workgroup_zero() {
; GFX942-NEXT: v_cndmask_b32_e64 v0, 0, 1, vcc
; GFX942-NEXT: s_setpc_b64 s[30:31]
;
-; GFX12-LABEL: workitem_workgroup_zero:
-; GFX12: ; %bb.0: ; %entry
-; GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
-; GFX12-NEXT: s_wait_expcnt 0x0
-; GFX12-NEXT: s_wait_samplecnt 0x0
-; GFX12-NEXT: s_wait_bvhcnt 0x0
-; GFX12-NEXT: s_wait_kmcnt 0x0
-; GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
-; GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v31
-; GFX12-NEXT: v_bfe_u32 v1, v31, 10, 10
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
-; GFX12-NEXT: s_or_b32 s0, ttmp9, s0
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: s_or_b32 s0, s0, s1
-; GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GFX12-NEXT: v_or3_b32 v0, s0, v0, v1
-; GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX12-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
-; GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
-; GFX12-NEXT: v_cndmask_b32_e64 v0, 0, 1, vcc_lo
-; GFX12-NEXT: s_setpc_b64 s[30:31]
+; DAGISEL-GFX12-LABEL: workitem_workgroup_zero:
+; DAGISEL-GFX12: ; %bb.0: ; %entry
+; DAGISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_expcnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_samplecnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
+; DAGISEL-GFX12-NEXT: s_wait_kmcnt 0x0
+; DAGISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; DAGISEL-GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v31
+; DAGISEL-GFX12-NEXT: v_bfe_u32 v1, v31, 10, 10
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
+; DAGISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; DAGISEL-GFX12-NEXT: v_or3_b32 v0, s0, v0, v1
+; DAGISEL-GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; DAGISEL-GFX12-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
+; DAGISEL-GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; DAGISEL-GFX12-NEXT: v_cndmask_b32_e64 v0, 0, 1, vcc_lo
+; DAGISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
+;
+; GISEL-GFX12-LABEL: workitem_workgroup_zero:
+; GISEL-GFX12: ; %bb.0: ; %entry
+; GISEL-GFX12-NEXT: s_wait_loadcnt_dscnt 0x0
+; GISEL-GFX12-NEXT: s_wait_expcnt 0x0
+; GISEL-GFX12-NEXT: s_wait_samplecnt 0x0
+; GISEL-GFX12-NEXT: s_wait_bvhcnt 0x0
+; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
+; GISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
+; GISEL-GFX12-NEXT: v_and_b32_e32 v0, 0x3ff, v31
+; GISEL-GFX12-NEXT: v_bfe_u32 v1, v31, 10, 10
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
+; GISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: s_or_b32 s0, s0, s1
+; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
+; GISEL-GFX12-NEXT: v_or3_b32 v0, s0, v0, v1
+; GISEL-GFX12-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GISEL-GFX12-NEXT: v_cmp_eq_u32_e32 vcc_lo, 0, v0
+; GISEL-GFX12-NEXT: s_wait_alu depctr_va_vcc(0)
+; GISEL-GFX12-NEXT: v_cndmask_b32_e64 v0, 0, 1, vcc_lo
+; GISEL-GFX12-NEXT: s_setpc_b64 s[30:31]
entry:
%0 = tail call i32 @llvm.amdgcn.workgroup.id.x()
%1 = tail call i32 @llvm.amdgcn.workgroup.id.y()
@@ -452,7 +493,7 @@ define i1 @workitem_workgroup_nonzero() {
; GISEL-GFX12-NEXT: s_wait_kmcnt 0x0
; GISEL-GFX12-NEXT: s_and_b32 s0, ttmp7, 0xffff
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
-; GISEL-GFX12-NEXT: s_lshr_b32 s1, ttmp7, 16
+; GISEL-GFX12-NEXT: s_bfe_u32 s1, ttmp7, 0x100010
; GISEL-GFX12-NEXT: s_or_b32 s0, ttmp9, s0
; GISEL-GFX12-NEXT: v_bfe_u32 v0, v31, 10, 10
; GISEL-GFX12-NEXT: s_wait_alu depctr_sa_sdst(0)
@@ -481,3 +522,5 @@ entry:
%cmp = icmp ne i32 %or4, 0
ret i1 %cmp
}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
+; GFX12: {{.*}}
>From 6ac2e41d1044ac406b0e4d18c3f6d866d7894be0 Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Wed, 2 Sep 2026 17:12:44 +0530
Subject: [PATCH 2/3] Utilized class rule in place of GICOmbinePatfrag for
commute_shift.
---
.../include/llvm/Target/GlobalISel/Combine.td | 24 +++++++++----------
1 file changed, 12 insertions(+), 12 deletions(-)
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 8942fd3a2e155..699ca9ca9ba82 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -447,26 +447,26 @@ def bitreverse_lshr : GICombineRule<
(apply (G_SHL $d, $val, $amt))>;
// Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
-// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
-// Apply rebuilds the matched opcode, so it stays inline C++.
-def commute_shift_frags : GICombinePatFrag<
- (outs root:$dst, $binop), (ins $x, $c1, $c2),
- !foreach(op, [G_ADD, G_OR],
- (pattern (op $binop, $x, $c1), (G_SHL $dst, $binop, $c2)))>;
-def commute_shift : GICombineRule<
+// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
+// Apply rebuilds the matched opcode, so it stays inline C++. The builder is
+// used rather than an apply pattern so that (shl c1, c2) is constant folded.
+class commute_shift_binop<Instruction binop> : GICombineRule<
(defs root:$dst),
- (match (commute_shift_frags $dst, $binop, $x, $c1, $c2):$shl,
- [{ return MRI.hasOneNonDBGUse(${binop}.getReg()) &&
+ (match (binop $binop_dst, $x, $c1):$binop,
+ (G_SHL $dst, $binop_dst, $c2):$shl,
+ [{ return MRI.hasOneNonDBGUse(${binop_dst}.getReg()) &&
isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
Helper.isDesirableToCommuteWithShift(*${shl}); }]),
(apply [{ auto &B = Helper.getBuilder();
LLT Ty = MRI.getType(${dst}.getReg());
- unsigned Opc = MRI.getVRegDef(${binop}.getReg())->getOpcode();
auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
- auto New = B.buildInstr(Opc, {Ty}, {S1, S2});
+ auto New = B.buildInstr(${binop}->getOpcode(), {Ty}, {S1, S2});
Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
+def commute_shift_add : commute_shift_binop<G_ADD>;
+def commute_shift_or : commute_shift_binop<G_OR>;
+def commute_shift : GICombineGroup<[commute_shift_add, commute_shift_or]>;
// Fold (lshr (trunc (lshr x, C1)), C2) -> trunc (lshr x, (C1 + C2))
def lshr_of_trunc_of_lshr_matchdata : GIDefMatchData<"LshrOfTruncOfLshr">;
@@ -1003,7 +1003,7 @@ def neg_and_one_to_sext_inreg : GICombineRule<
>;
// Fold and(and(x, C1), C2) -> C1&C2 ? and(x, C1&C2) : 0
-def overlapping_and : GICombineRule<
+def overlapping_and: GICombineRule <
(defs root:$root, build_fn_matchinfo:$info),
(match (wip_match_opcode G_AND):$root,
[{ return Helper.matchOverlappingAnd(*${root}, ${info}); }]),
>From 36d346cb329412f35fae5becd90c4a32f21e4d9c Mon Sep 17 00:00:00 2001
From: vg0204 <Vikash.Gupta at amd.com>
Date: Thu, 3 Sep 2026 13:07:22 +0530
Subject: [PATCH 3/3] Add MIR pattern for apply rule in commute_shift.
---
llvm/include/llvm/Target/GlobalISel/Combine.td | 11 +++--------
.../GlobalISel/prelegalizercombiner-commute-shift.mir | 5 ++---
2 files changed, 5 insertions(+), 11 deletions(-)
diff --git a/llvm/include/llvm/Target/GlobalISel/Combine.td b/llvm/include/llvm/Target/GlobalISel/Combine.td
index 699ca9ca9ba82..3735e95ee8f68 100644
--- a/llvm/include/llvm/Target/GlobalISel/Combine.td
+++ b/llvm/include/llvm/Target/GlobalISel/Combine.td
@@ -448,8 +448,6 @@ def bitreverse_lshr : GICombineRule<
// Combine (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
// Combine (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2)
-// Apply rebuilds the matched opcode, so it stays inline C++. The builder is
-// used rather than an apply pattern so that (shl c1, c2) is constant folded.
class commute_shift_binop<Instruction binop> : GICombineRule<
(defs root:$dst),
(match (binop $binop_dst, $x, $c1):$binop,
@@ -458,12 +456,9 @@ class commute_shift_binop<Instruction binop> : GICombineRule<
isConstantOrConstantSplatVector(${c1}.getReg(), MRI) &&
isConstantOrConstantSplatVector(${c2}.getReg(), MRI) &&
Helper.isDesirableToCommuteWithShift(*${shl}); }]),
- (apply [{ auto &B = Helper.getBuilder();
- LLT Ty = MRI.getType(${dst}.getReg());
- auto S1 = B.buildShl(Ty, ${x}.getReg(), ${c2}.getReg());
- auto S2 = B.buildShl(Ty, ${c1}.getReg(), ${c2}.getReg());
- auto New = B.buildInstr(${binop}->getOpcode(), {Ty}, {S1, S2});
- Helper.replaceSingleDefInstWithReg(*${shl}, New.getReg(0)); }])>;
+ (apply (G_SHL GITypeOf<"$dst">:$s1, $x, $c2),
+ (G_SHL GITypeOf<"$dst">:$s2, $c1, $c2),
+ (binop $dst, $s1, $s2))>;
def commute_shift_add : commute_shift_binop<G_ADD>;
def commute_shift_or : commute_shift_binop<G_OR>;
def commute_shift : GICombineGroup<[commute_shift_add, commute_shift_or]>;
diff --git a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
index 8f78c1dacaf2a..7bc1503e9da32 100644
--- a/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
+++ b/llvm/test/CodeGen/AArch64/GlobalISel/prelegalizercombiner-commute-shift.mir
@@ -108,9 +108,8 @@ body: |
; CHECK-NEXT: [[C:%[0-9]+]]:_(i32) = G_CONSTANT i32 2
; CHECK-NEXT: %veccst2:_(<4 x i32>) = G_BUILD_VECTOR [[C]](i32), [[C]](i32), [[C]](i32), [[C]](i32)
; CHECK-NEXT: [[SHL:%[0-9]+]]:_(<4 x i32>) = G_SHL %xvec, %veccst2(<4 x i32>)
- ; CHECK-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 8
- ; CHECK-NEXT: [[BUILD_VECTOR:%[0-9]+]]:_(<4 x i32>) = G_BUILD_VECTOR [[C1]](i32), [[C1]](i32), [[C1]](i32), [[C1]](i32)
- ; CHECK-NEXT: [[ADD:%[0-9]+]]:_(<4 x i32>) = G_ADD [[SHL]], [[BUILD_VECTOR]]
+ ; CHECK-NEXT: [[SHL1:%[0-9]+]]:_(<4 x i32>) = G_SHL %veccst2, %veccst2(<4 x i32>)
+ ; CHECK-NEXT: [[ADD:%[0-9]+]]:_(<4 x i32>) = G_ADD [[SHL]], [[SHL1]]
; CHECK-NEXT: G_STORE [[ADD]](<4 x i32>), [[COPY]](p0) :: (store (<4 x i32>))
; CHECK-NEXT: RET_ReallyLR
%0:_(p0) = COPY $x0
More information about the llvm-commits
mailing list