[llvm-branch-commits] [llvm] AMDGPU/GlobalISel: Use integer types when narrowing loads and stores (PR #229547)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Tue Oct 6 12:52:57 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-globalisel

Author: Matt Arsenault (arsenm)

<details>
<summary>Changes</summary>

The narrowScalar mutation for loads and stores produced untyped scalars.
For FP-typed values this resulted in an untyped G_OR in the lowerLoad
expansion for unaligned private accesses, which failed to select.

Co-authored-by: Claude Opus 5.5 <noreply@<!-- -->anthropic.com>

---

Patch is 293.13 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/229547.diff


6 Files Affected:

- (modified) llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp (+3-3) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw-fmin-fmax.ll (+16-16) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-flat.mir (+268-270) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-private.mir (+441-448) 
- (added) llvm/test/CodeGen/AMDGPU/GlobalISel/load-store-private-unaligned-f64.ll (+361) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/regbankselect-amdgcn.s.buffer.load.ll (+16-16) 


``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index 03987fb3f36b7..a874c5271e4be 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -1707,16 +1707,16 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
 
               // Split extloads.
               if (DstSize > MemSize)
-                return std::pair(0, LLT::scalar(MemSize));
+                return std::pair(0, LLT::integer(MemSize));
 
               unsigned MaxSize = maxSizeForAddrSpace(
                   ST, PtrTy.getAddressSpace(), Op == G_LOAD,
                   Query.MMODescrs[0].Ordering != AtomicOrdering::NotAtomic);
               if (MemSize > MaxSize)
-                return std::pair(0, LLT::scalar(MaxSize));
+                return std::pair(0, LLT::integer(MaxSize));
 
               uint64_t Align = Query.MMODescrs[0].AlignInBits;
-              return std::pair(0, LLT::scalar(Align));
+              return std::pair(0, LLT::integer(Align));
             })
         .fewerElementsIf(
             [=](const LegalityQuery &Query) -> bool {
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw-fmin-fmax.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw-fmin-fmax.ll
index 5831542bead6b..2f9a68804bcda 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw-fmin-fmax.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/atomicrmw-fmin-fmax.ll
@@ -139,16 +139,16 @@ define void @atomicrmw_fmax_flat_f64_vv_noret(ptr %ptr, double %val) {
   ; GFX10-NEXT:   [[V_CMP_NE_U64_e64_:%[0-9]+]]:sreg_32_xm0_xexec = V_CMP_NE_U64_e64 [[REG_SEQUENCE]], [[COPY9]], implicit $exec
   ; GFX10-NEXT:   [[COPY10:%[0-9]+]]:vgpr_32 = COPY [[S_MOV_B32_]]
   ; GFX10-NEXT:   [[V_CNDMASK_B32_e64_:%[0-9]+]]:vgpr_32 = V_CNDMASK_B32_e64 0, [[COPY10]], 0, [[COPY]], [[V_CMP_NE_U64_e64_]], implicit $exec
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (s32) from %ir.7, align 8, addrspace 5)
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (s32) from %ir.7 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (i32) from %ir.7, align 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (i32) from %ir.7 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   [[REG_SEQUENCE2:%[0-9]+]]:vreg_64 = REG_SEQUENCE [[BUFFER_LOAD_DWORD_OFFEN]], %subreg.sub0, [[BUFFER_LOAD_DWORD_OFFEN1]], %subreg.sub1
   ; GFX10-NEXT:   [[V_MAX_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE2]], 0, [[REG_SEQUENCE2]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_1:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE1]], 0, [[REG_SEQUENCE1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_2:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[V_MAX_F64_e64_]], 0, [[V_MAX_F64_e64_1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[COPY11:%[0-9]+]]:vgpr_32 = COPY [[V_MAX_F64_e64_2]].sub0
   ; GFX10-NEXT:   [[COPY12:%[0-9]+]]:vgpr_32 = COPY [[V_MAX_F64_e64_2]].sub1
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (s32) into %ir.7, align 8, addrspace 5)
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (s32) into %ir.7 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (i32) into %ir.7, align 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (i32) into %ir.7 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   S_BRANCH %bb.5
   ; GFX10-NEXT: {{  $}}
   ; GFX10-NEXT: bb.4.atomicrmw.global:
@@ -202,16 +202,16 @@ define double @atomicrmw_fmax_flat_f64_vv_ret(ptr %ptr, double %val) {
   ; GFX10-NEXT:   [[V_CMP_NE_U64_e64_:%[0-9]+]]:sreg_32_xm0_xexec = V_CMP_NE_U64_e64 [[REG_SEQUENCE]], [[COPY9]], implicit $exec
   ; GFX10-NEXT:   [[COPY10:%[0-9]+]]:vgpr_32 = COPY [[S_MOV_B32_]]
   ; GFX10-NEXT:   [[V_CNDMASK_B32_e64_:%[0-9]+]]:vgpr_32 = V_CNDMASK_B32_e64 0, [[COPY10]], 0, [[COPY]], [[V_CMP_NE_U64_e64_]], implicit $exec
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (s32) from %ir.8, align 8, addrspace 5)
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (s32) from %ir.8 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (i32) from %ir.8, align 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (i32) from %ir.8 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   [[REG_SEQUENCE2:%[0-9]+]]:vreg_64 = REG_SEQUENCE [[BUFFER_LOAD_DWORD_OFFEN]], %subreg.sub0, [[BUFFER_LOAD_DWORD_OFFEN1]], %subreg.sub1
   ; GFX10-NEXT:   [[V_MAX_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE2]], 0, [[REG_SEQUENCE2]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_1:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE1]], 0, [[REG_SEQUENCE1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_2:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[V_MAX_F64_e64_]], 0, [[V_MAX_F64_e64_1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[COPY11:%[0-9]+]]:vgpr_32 = COPY [[V_MAX_F64_e64_2]].sub0
   ; GFX10-NEXT:   [[COPY12:%[0-9]+]]:vgpr_32 = COPY [[V_MAX_F64_e64_2]].sub1
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (s32) into %ir.8, align 8, addrspace 5)
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (s32) into %ir.8 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (i32) into %ir.8, align 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (i32) into %ir.8 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   S_BRANCH %bb.5
   ; GFX10-NEXT: {{  $}}
   ; GFX10-NEXT: bb.4.atomicrmw.global:
@@ -435,16 +435,16 @@ define void @atomicrmw_fmin_flat_f64_vv_noret(ptr %ptr, double %val) {
   ; GFX10-NEXT:   [[V_CMP_NE_U64_e64_:%[0-9]+]]:sreg_32_xm0_xexec = V_CMP_NE_U64_e64 [[REG_SEQUENCE]], [[COPY9]], implicit $exec
   ; GFX10-NEXT:   [[COPY10:%[0-9]+]]:vgpr_32 = COPY [[S_MOV_B32_]]
   ; GFX10-NEXT:   [[V_CNDMASK_B32_e64_:%[0-9]+]]:vgpr_32 = V_CNDMASK_B32_e64 0, [[COPY10]], 0, [[COPY]], [[V_CMP_NE_U64_e64_]], implicit $exec
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (s32) from %ir.7, align 8, addrspace 5)
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (s32) from %ir.7 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (i32) from %ir.7, align 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (i32) from %ir.7 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   [[REG_SEQUENCE2:%[0-9]+]]:vreg_64 = REG_SEQUENCE [[BUFFER_LOAD_DWORD_OFFEN]], %subreg.sub0, [[BUFFER_LOAD_DWORD_OFFEN1]], %subreg.sub1
   ; GFX10-NEXT:   [[V_MAX_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE2]], 0, [[REG_SEQUENCE2]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_1:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE1]], 0, [[REG_SEQUENCE1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MIN_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MIN_F64_e64 0, [[V_MAX_F64_e64_]], 0, [[V_MAX_F64_e64_1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[COPY11:%[0-9]+]]:vgpr_32 = COPY [[V_MIN_F64_e64_]].sub0
   ; GFX10-NEXT:   [[COPY12:%[0-9]+]]:vgpr_32 = COPY [[V_MIN_F64_e64_]].sub1
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (s32) into %ir.7, align 8, addrspace 5)
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (s32) into %ir.7 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (i32) into %ir.7, align 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (i32) into %ir.7 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   S_BRANCH %bb.5
   ; GFX10-NEXT: {{  $}}
   ; GFX10-NEXT: bb.4.atomicrmw.global:
@@ -498,16 +498,16 @@ define double @atomicrmw_fmin_flat_f64_vv_ret(ptr %ptr, double %val) {
   ; GFX10-NEXT:   [[V_CMP_NE_U64_e64_:%[0-9]+]]:sreg_32_xm0_xexec = V_CMP_NE_U64_e64 [[REG_SEQUENCE]], [[COPY9]], implicit $exec
   ; GFX10-NEXT:   [[COPY10:%[0-9]+]]:vgpr_32 = COPY [[S_MOV_B32_]]
   ; GFX10-NEXT:   [[V_CNDMASK_B32_e64_:%[0-9]+]]:vgpr_32 = V_CNDMASK_B32_e64 0, [[COPY10]], 0, [[COPY]], [[V_CMP_NE_U64_e64_]], implicit $exec
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (s32) from %ir.8, align 8, addrspace 5)
-  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (s32) from %ir.8 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (load (i32) from %ir.8, align 8, addrspace 5)
+  ; GFX10-NEXT:   [[BUFFER_LOAD_DWORD_OFFEN1:%[0-9]+]]:vgpr_32 = BUFFER_LOAD_DWORD_OFFEN [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (load (i32) from %ir.8 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   [[REG_SEQUENCE2:%[0-9]+]]:vreg_64 = REG_SEQUENCE [[BUFFER_LOAD_DWORD_OFFEN]], %subreg.sub0, [[BUFFER_LOAD_DWORD_OFFEN1]], %subreg.sub1
   ; GFX10-NEXT:   [[V_MAX_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE2]], 0, [[REG_SEQUENCE2]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MAX_F64_e64_1:%[0-9]+]]:vreg_64 = nofpexcept V_MAX_F64_e64 0, [[REG_SEQUENCE1]], 0, [[REG_SEQUENCE1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[V_MIN_F64_e64_:%[0-9]+]]:vreg_64 = nofpexcept V_MIN_F64_e64 0, [[V_MAX_F64_e64_]], 0, [[V_MAX_F64_e64_1]], 0, 0, implicit $mode, implicit $exec
   ; GFX10-NEXT:   [[COPY11:%[0-9]+]]:vgpr_32 = COPY [[V_MIN_F64_e64_]].sub0
   ; GFX10-NEXT:   [[COPY12:%[0-9]+]]:vgpr_32 = COPY [[V_MIN_F64_e64_]].sub1
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (s32) into %ir.8, align 8, addrspace 5)
-  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (s32) into %ir.8 + 4, basealign 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY11]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 0, 0, 0, implicit $exec :: (store (i32) into %ir.8, align 8, addrspace 5)
+  ; GFX10-NEXT:   BUFFER_STORE_DWORD_OFFEN [[COPY12]], [[V_CNDMASK_B32_e64_]], $sgpr0_sgpr1_sgpr2_sgpr3, 0, 4, 0, 0, implicit $exec :: (store (i32) into %ir.8 + 4, basealign 8, addrspace 5)
   ; GFX10-NEXT:   S_BRANCH %bb.5
   ; GFX10-NEXT: {{  $}}
   ; GFX10-NEXT: bb.4.atomicrmw.global:
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-flat.mir b/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-flat.mir
index 5c187f74ceddf..c3cc357a31506 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-flat.mir
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/legalize-load-flat.mir
@@ -915,48 +915,46 @@ body: |
     ; CI: liveins: $vgpr0_vgpr1
     ; CI-NEXT: {{  $}}
     ; CI-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $vgpr0_vgpr1
-    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[COPY]](p0) :: (load (s32), align 8)
+    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(i32) = G_LOAD [[COPY]](p0) :: (load (i32), align 8)
     ; CI-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 4
     ; CI-NEXT: [[PTR_ADD:%[0-9]+]]:_(p0) = nuw inbounds G_PTR_ADD [[COPY]], [[C]](i64)
     ; CI-NEXT: [[LOAD1:%[0-9]+]]:_(i32) = G_LOAD [[PTR_ADD]](p0) :: (load (i16) from unknown-address + 4, align 4)
-    ; CI-NEXT: [[C1:%[0-9]+]]:_(s32) = G_CONSTANT i32 16
-    ; CI-NEXT: [[LSHR:%[0-9]+]]:_(s32) = G_LSHR [[LOAD]], [[C1]](s32)
+    ; CI-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 16
+    ; CI-NEXT: [[LSHR:%[0-9]+]]:_(i32) = G_LSHR [[LOAD]], [[C1]](i32)
     ; CI-NEXT: [[C2:%[0-9]+]]:_(i32) = G_CONSTANT i32 65535
     ; CI-NEXT: [[AND:%[0-9]+]]:_(i32) = G_AND [[LOAD]], [[C2]]
-    ; CI-NEXT: [[C3:%[0-9]+]]:_(i32) = G_CONSTANT i32 16
-    ; CI-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[LSHR]], [[C3]](i32)
+    ; CI-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[LSHR]], [[C1]](i32)
     ; CI-NEXT: [[OR:%[0-9]+]]:_(i32) = G_OR [[AND]], [[SHL]]
     ; CI-NEXT: [[AND1:%[0-9]+]]:_(i32) = G_AND [[LOAD1]], [[C2]]
-    ; CI-NEXT: [[C4:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
-    ; CI-NEXT: [[SHL1:%[0-9]+]]:_(i32) = G_SHL [[C4]], [[C3]](i32)
+    ; CI-NEXT: [[C3:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
+    ; CI-NEXT: [[SHL1:%[0-9]+]]:_(i32) = G_SHL [[C3]], [[C1]](i32)
     ; CI-NEXT: [[OR1:%[0-9]+]]:_(i32) = G_OR [[AND1]], [[SHL1]]
     ; CI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[OR]](i32), [[OR1]](i32)
-    ; CI-NEXT: [[C5:%[0-9]+]]:_(i64) = G_CONSTANT i64 281474976710655
-    ; CI-NEXT: [[AND2:%[0-9]+]]:_(i64) = G_AND [[MV]], [[C5]]
+    ; CI-NEXT: [[C4:%[0-9]+]]:_(i64) = G_CONSTANT i64 281474976710655
+    ; CI-NEXT: [[AND2:%[0-9]+]]:_(i64) = G_AND [[MV]], [[C4]]
     ; CI-NEXT: $vgpr0_vgpr1 = COPY [[AND2]](i64)
     ;
     ; VI-LABEL: name: test_load_flat_s48_align8
     ; VI: liveins: $vgpr0_vgpr1
     ; VI-NEXT: {{  $}}
     ; VI-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $vgpr0_vgpr1
-    ; VI-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[COPY]](p0) :: (load (s32), align 8)
+    ; VI-NEXT: [[LOAD:%[0-9]+]]:_(i32) = G_LOAD [[COPY]](p0) :: (load (i32), align 8)
     ; VI-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 4
     ; VI-NEXT: [[PTR_ADD:%[0-9]+]]:_(p0) = nuw inbounds G_PTR_ADD [[COPY]], [[C]](i64)
     ; VI-NEXT: [[LOAD1:%[0-9]+]]:_(i32) = G_LOAD [[PTR_ADD]](p0) :: (load (i16) from unknown-address + 4, align 4)
-    ; VI-NEXT: [[C1:%[0-9]+]]:_(s32) = G_CONSTANT i32 16
-    ; VI-NEXT: [[LSHR:%[0-9]+]]:_(s32) = G_LSHR [[LOAD]], [[C1]](s32)
+    ; VI-NEXT: [[C1:%[0-9]+]]:_(i32) = G_CONSTANT i32 16
+    ; VI-NEXT: [[LSHR:%[0-9]+]]:_(i32) = G_LSHR [[LOAD]], [[C1]](i32)
     ; VI-NEXT: [[C2:%[0-9]+]]:_(i32) = G_CONSTANT i32 65535
     ; VI-NEXT: [[AND:%[0-9]+]]:_(i32) = G_AND [[LOAD]], [[C2]]
-    ; VI-NEXT: [[C3:%[0-9]+]]:_(i32) = G_CONSTANT i32 16
-    ; VI-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[LSHR]], [[C3]](i32)
+    ; VI-NEXT: [[SHL:%[0-9]+]]:_(i32) = G_SHL [[LSHR]], [[C1]](i32)
     ; VI-NEXT: [[OR:%[0-9]+]]:_(i32) = G_OR [[AND]], [[SHL]]
     ; VI-NEXT: [[AND1:%[0-9]+]]:_(i32) = G_AND [[LOAD1]], [[C2]]
-    ; VI-NEXT: [[C4:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
-    ; VI-NEXT: [[SHL1:%[0-9]+]]:_(i32) = G_SHL [[C4]], [[C3]](i32)
+    ; VI-NEXT: [[C3:%[0-9]+]]:_(i32) = G_CONSTANT i32 0
+    ; VI-NEXT: [[SHL1:%[0-9]+]]:_(i32) = G_SHL [[C3]], [[C1]](i32)
     ; VI-NEXT: [[OR1:%[0-9]+]]:_(i32) = G_OR [[AND1]], [[SHL1]]
     ; VI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[OR]](i32), [[OR1]](i32)
-    ; VI-NEXT: [[C5:%[0-9]+]]:_(i64) = G_CONSTANT i64 281474976710655
-    ; VI-NEXT: [[AND2:%[0-9]+]]:_(i64) = G_AND [[MV]], [[C5]]
+    ; VI-NEXT: [[C4:%[0-9]+]]:_(i64) = G_CONSTANT i64 281474976710655
+    ; VI-NEXT: [[AND2:%[0-9]+]]:_(i64) = G_AND [[MV]], [[C4]]
     ; VI-NEXT: $vgpr0_vgpr1 = COPY [[AND2]](i64)
     ;
     ; GFX9PLUS-LABEL: name: test_load_flat_s48_align8
@@ -1028,22 +1026,22 @@ body: |
     ; CI: liveins: $vgpr0_vgpr1
     ; CI-NEXT: {{  $}}
     ; CI-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $vgpr0_vgpr1
-    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[COPY]](p0) :: (load (s32), align 8)
+    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(i32) = G_LOAD [[COPY]](p0) :: (load (i32), align 8)
     ; CI-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 4
     ; CI-NEXT: [[PTR_ADD:%[0-9]+]]:_(p0) = nuw inbounds G_PTR_ADD [[COPY]], [[C]](i64)
-    ; CI-NEXT: [[LOAD1:%[0-9]+]]:_(s32) = G_LOAD [[PTR_ADD]](p0) :: (load (s32) from unknown-address + 4)
-    ; CI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](s32), [[LOAD1]](s32)
+    ; CI-NEXT: [[LOAD1:%[0-9]+]]:_(i32) = G_LOAD [[PTR_ADD]](p0) :: (load (i32) from unknown-address + 4)
+    ; CI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](i32), [[LOAD1]](i32)
     ; CI-NEXT: $vgpr0_vgpr1 = COPY [[MV]](i64)
     ;
     ; VI-LABEL: name: test_load_flat_s64_align8
     ; VI: liveins: $vgpr0_vgpr1
     ; VI-NEXT: {{  $}}
     ; VI-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $vgpr0_vgpr1
-    ; VI-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[COPY]](p0) :: (load (s32), align 8)
+    ; VI-NEXT: [[LOAD:%[0-9]+]]:_(i32) = G_LOAD [[COPY]](p0) :: (load (i32), align 8)
     ; VI-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 4
     ; VI-NEXT: [[PTR_ADD:%[0-9]+]]:_(p0) = nuw inbounds G_PTR_ADD [[COPY]], [[C]](i64)
-    ; VI-NEXT: [[LOAD1:%[0-9]+]]:_(s32) = G_LOAD [[PTR_ADD]](p0) :: (load (s32) from unknown-address + 4)
-    ; VI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](s32), [[LOAD1]](s32)
+    ; VI-NEXT: [[LOAD1:%[0-9]+]]:_(i32) = G_LOAD [[PTR_ADD]](p0) :: (load (i32) from unknown-address + 4)
+    ; VI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](i32), [[LOAD1]](i32)
     ; VI-NEXT: $vgpr0_vgpr1 = COPY [[MV]](i64)
     ;
     ; GFX9PLUS-LABEL: name: test_load_flat_s64_align8
@@ -1102,22 +1100,22 @@ body: |
     ; CI: liveins: $vgpr0_vgpr1
     ; CI-NEXT: {{  $}}
     ; CI-NEXT: [[COPY:%[0-9]+]]:_(p0) = COPY $vgpr0_vgpr1
-    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(s32) = G_LOAD [[COPY]](p0) :: (load (s32))
+    ; CI-NEXT: [[LOAD:%[0-9]+]]:_(i32) = G_LOAD [[COPY]](p0) :: (load (i32))
     ; CI-NEXT: [[C:%[0-9]+]]:_(i64) = G_CONSTANT i64 4
     ; CI-NEXT: [[PTR_ADD:%[0-9]+]]:_(p0) = nuw inbounds G_PTR_ADD [[COPY]], [[C]](i64)
-    ; CI-NEXT: [[LOAD1:%[0-9]+]]:_(s32) = G_LOAD [[PTR_ADD]](p0) :: (load (s32) from unknown-address + 4)
-    ; CI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](s32), [[LOAD1]](s32)
+    ; CI-NEXT: [[LOAD1:%[0-9]+]]:_(i32) = G_LOAD [[PTR_ADD]](p0) :: (load (i32) from unknown-address + 4)
+    ; CI-NEXT: [[MV:%[0-9]+]]:_(i64) = G_MERGE_VALUES [[LOAD]](i32), [[LOAD1]](i32)
     ; CI-NEXT: $vgpr0_vgpr1 = COPY [[MV]](i64)
     ...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/229547


More information about the llvm-branch-commits mailing list