[llvm] [AMDGPU][NFC] Explicitly narrow conversions in the legalizer (PR #215206)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 06:49:38 PDT 2026


https://github.com/gretay-amd updated https://github.com/llvm/llvm-project/pull/215206

>From babc164ee55e813ecf1c4aecc900d982440ea92b Mon Sep 17 00:00:00 2001
From: Greta Y <Greta.Yorsh at amd.com>
Date: Thu, 6 Aug 2026 16:15:24 +0100
Subject: [PATCH] [AMDGPU][NFC] Explicitly narrow conversions in the legalizer

This patch handles the following cases:

LLT::getSizeInBits() returns TypeSize, which is assigned to 32-bit locals or
passed to 32-bit parameters holding a type width in bits. Add a static_cast to
make the existing narrowing conversion explicit. This is the dominant shape:
the legalization rules are expressed in terms of widths, and the
LegalityPredicate and LegalizeMutation builders take unsigned. The LLTs are
fixed-width, so the value is a type width of at most 1024.

Element counts computed with 64-bit arithmetic are passed to interfaces taking
an unsigned element count: PowerOf2Ceil returns uint64_t and its result is
handed to ElementCount::getFixed and LLT::changeElementSize, and the
LLT::fixed_vector builders likewise take unsigned. Add a static_cast to make the
existing narrowing conversion explicit. The inputs are LLT::getNumElements(),
which is uint16_t, and fixed LLT widths, so the rounded-up value stays far
inside unsigned.

MachineOperand::getImm() returns int64_t and is assigned to 32-bit locals
holding intrinsic operands and address-space numbers. Add a static_cast to make
the existing narrowing conversion explicit.

Container size() returns size_t and is assigned to unsigned or int locals
holding operand and part counts when splitting a legalization. Add a
static_cast to make the existing narrowing conversion explicit.

Values assigned to the uint16_t fields of the image-intrinsic dimension tables
are cast to uint16_t, the width those tables are generated with.

This fixes 89 instances of MSVC warning C4244 and 7 of C4267 ("possible loss of
data") in llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp.

Assisted-by: Claude <noreply at anthropic.com>
---
 .../lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp | 249 ++++++++++--------
 1 file changed, 141 insertions(+), 108 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
index c922d14576af8..9c7692d560d6b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULegalizerInfo.cpp
@@ -64,7 +64,7 @@ static LLT getPow2VectorType(LLT Ty) {
 
 // Round the number of bits to the next power of two bits
 static LLT getPow2ScalarType(LLT Ty) {
-  unsigned Bits = Ty.getSizeInBits();
+  unsigned Bits = static_cast<unsigned>(Ty.getSizeInBits());
   unsigned Pow2Bits = 1 <<  Log2_32_Ceil(Bits);
   return LLT::scalar(Pow2Bits);
 }
@@ -79,7 +79,7 @@ static LegalityPredicate isSmallOddVector(unsigned TypeIdx) {
       return false;
 
     const LLT EltTy = Ty.getElementType();
-    const unsigned EltSize = EltTy.getSizeInBits();
+    const unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
     return Ty.getNumElements() % 2 != 0 &&
            EltSize > 1 && EltSize < 32 &&
            Ty.getSizeInBits() % 32 != 0;
@@ -114,7 +114,7 @@ static LegalizeMutation fewerEltsToSize64Vector(unsigned TypeIdx) {
   return [=](const LegalityQuery &Query) {
     const LLT Ty = Query.Types[TypeIdx];
     const LLT EltTy = Ty.getElementType();
-    unsigned Size = Ty.getSizeInBits();
+    unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
     unsigned Pieces = (Size + 63) / 64;
     unsigned NewNumElts = (Ty.getNumElements() + 1) / Pieces;
     return std::pair(TypeIdx, LLT::scalarOrVector(
@@ -129,8 +129,8 @@ static LegalizeMutation moreEltsToNext32Bit(unsigned TypeIdx) {
     const LLT Ty = Query.Types[TypeIdx];
 
     const LLT EltTy = Ty.getElementType();
-    const int Size = Ty.getSizeInBits();
-    const int EltSize = EltTy.getSizeInBits();
+    const int Size = static_cast<int>(Ty.getSizeInBits());
+    const int EltSize = static_cast<int>(EltTy.getSizeInBits());
     const int NextMul32 = (Size + 31) / 32;
 
     assert(EltSize < 32);
@@ -143,7 +143,8 @@ static LegalizeMutation moreEltsToNext32Bit(unsigned TypeIdx) {
 // Retrieves the scalar type that's the same size as the mem desc
 static LegalizeMutation getScalarTypeFromMemDesc(unsigned TypeIdx) {
   return [=](const LegalityQuery &Query) {
-    unsigned MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
+    unsigned MemSize =
+        static_cast<unsigned>(Query.MMODescrs[0].MemoryTy.getSizeInBits());
     return std::make_pair(TypeIdx, LLT::integer(MemSize));
   };
 }
@@ -153,7 +154,8 @@ static LegalizeMutation moreElementsToNextExistingRegClass(unsigned TypeIdx) {
   return [=](const LegalityQuery &Query) {
     const LLT Ty = Query.Types[TypeIdx];
     const unsigned NumElts = Ty.getNumElements();
-    const unsigned EltSize = Ty.getElementType().getSizeInBits();
+    const unsigned EltSize =
+        static_cast<unsigned>(Ty.getElementType().getSizeInBits());
     const unsigned MaxNumElts = MaxRegisterSize / EltSize;
 
     assert(EltSize == 32 || EltSize == 64);
@@ -185,7 +187,7 @@ static LLT getBufferRsrcRegisterType(const LLT Ty) {
 }
 
 static LLT getBitcastRegisterType(const LLT Ty) {
-  const unsigned Size = Ty.getSizeInBits();
+  const unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
 
   if (Size <= 32) {
     // <2 x i8> -> i16
@@ -206,7 +208,7 @@ static LegalizeMutation bitcastToRegisterType(unsigned TypeIdx) {
 static LegalizeMutation bitcastToVectorElement32(unsigned TypeIdx) {
   return [=](const LegalityQuery &Query) {
     const LLT Ty = Query.Types[TypeIdx];
-    unsigned Size = Ty.getSizeInBits();
+    unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
     assert(Size % 32 == 0);
     return std::pair(TypeIdx,
                      LLT::scalarOrVector(ElementCount::getFixed(Size / 32),
@@ -241,12 +243,12 @@ static bool isRegisterSize(const GCNSubtarget &ST, unsigned Size) {
 }
 
 static bool isRegisterVectorElementType(LLT EltTy) {
-  const int EltSize = EltTy.getSizeInBits();
+  const int EltSize = static_cast<int>(EltTy.getSizeInBits());
   return EltSize == 16 || EltSize % 32 == 0;
 }
 
 static bool isRegisterVectorType(LLT Ty) {
-  const int EltSize = Ty.getElementType().getSizeInBits();
+  const int EltSize = static_cast<int>(Ty.getElementType().getSizeInBits());
   return EltSize == 32 || EltSize == 64 ||
          (EltSize == 16 && Ty.getNumElements() % 2 == 0) ||
          EltSize == 128 || EltSize == 256;
@@ -254,7 +256,7 @@ static bool isRegisterVectorType(LLT Ty) {
 
 // TODO: replace all uses of isRegisterType with isRegisterClassType
 static bool isRegisterType(const GCNSubtarget &ST, LLT Ty) {
-  if (!isRegisterSize(ST, Ty.getSizeInBits()))
+  if (!isRegisterSize(ST, static_cast<unsigned>(Ty.getSizeInBits())))
     return false;
 
   if (Ty.isVector())
@@ -280,7 +282,8 @@ static LegalityPredicate isIllegalRegisterType(const GCNSubtarget &ST,
   return [=, &ST](const LegalityQuery &Query) {
     LLT Ty = Query.Types[TypeIdx];
     return isRegisterType(ST, Ty) &&
-           !SIRegisterInfo::getSGPRClassForBitWidth(Ty.getSizeInBits());
+           !SIRegisterInfo::getSGPRClassForBitWidth(
+               static_cast<unsigned>(Ty.getSizeInBits()));
   };
 }
 
@@ -404,7 +407,8 @@ static LegalityPredicate isWideScalarExtLoadTruncStore(unsigned TypeIdx) {
 // than 32-bits and mem location is a power of 2
 static LegalityPredicate isTruncStoreToSizePowerOf2(unsigned TypeIdx) {
   return [=](const LegalityQuery &Query) {
-    unsigned MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
+    unsigned MemSize =
+        static_cast<unsigned>(Query.MMODescrs[0].MemoryTy.getSizeInBits());
     return isWideScalarExtLoadTruncStore(TypeIdx)(Query) &&
            isPowerOf2_64(MemSize);
   };
@@ -447,7 +451,7 @@ static bool isLoadStoreSizeLegal(const GCNSubtarget &ST,
   // Handle G_LOAD, G_ZEXTLOAD, G_SEXTLOAD
   const bool IsLoad = Query.Opcode != AMDGPU::G_STORE;
 
-  unsigned RegSize = Ty.getSizeInBits();
+  unsigned RegSize = static_cast<unsigned>(Ty.getSizeInBits());
   uint64_t MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
   uint64_t AlignBits = Query.MMODescrs[0].AlignInBits;
   unsigned AS = Query.Types[1].getAddressSpace();
@@ -500,8 +504,8 @@ static bool isLoadStoreSizeLegal(const GCNSubtarget &ST,
 
   if (AlignBits < MemSize) {
     const SITargetLowering *TLI = ST.getTargetLowering();
-    if (!TLI->allowsMisalignedMemoryAccessesImpl(MemSize, AS,
-                                                 Align(AlignBits / 8)))
+    if (!TLI->allowsMisalignedMemoryAccessesImpl(static_cast<unsigned>(MemSize),
+                                                 AS, Align(AlignBits / 8)))
       return false;
   }
 
@@ -531,7 +535,7 @@ static bool loadStoreBitcastWorkaround(const LLT Ty) {
   if (EnableNewLegality)
     return false;
 
-  const unsigned Size = Ty.getSizeInBits();
+  const unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
   if (Ty.isPointerVector())
     return true;
   if (Size <= 64)
@@ -556,8 +560,8 @@ static bool isLoadStoreLegal(const GCNSubtarget &ST, const LegalityQuery &Query)
 /// to a different type.
 static bool shouldBitcastLoadStoreType(const GCNSubtarget &ST, const LLT Ty,
                                        const LLT MemTy) {
-  const unsigned MemSizeInBits = MemTy.getSizeInBits();
-  const unsigned Size = Ty.getSizeInBits();
+  const unsigned MemSizeInBits = static_cast<unsigned>(MemTy.getSizeInBits());
+  const unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
   if (Size != MemSizeInBits)
     return Size <= 32 && Ty.isVector();
 
@@ -576,7 +580,7 @@ static bool shouldBitcastLoadStoreType(const GCNSubtarget &ST, const LLT Ty,
 static bool shouldWidenLoad(const GCNSubtarget &ST, LLT MemoryTy,
                             uint64_t AlignInBits, unsigned AddrSpace,
                             unsigned Opcode) {
-  unsigned SizeInBits = MemoryTy.getSizeInBits();
+  unsigned SizeInBits = static_cast<unsigned>(MemoryTy.getSizeInBits());
   // We don't want to widen cases that are naturally legal.
   if (isPowerOf2_32(SizeInBits))
     return false;
@@ -594,7 +598,7 @@ static bool shouldWidenLoad(const GCNSubtarget &ST, LLT MemoryTy,
   // to it.
   //
   // TODO: Could check dereferenceable for less aligned cases.
-  unsigned RoundedSize = NextPowerOf2(SizeInBits);
+  unsigned RoundedSize = static_cast<unsigned>(NextPowerOf2(SizeInBits));
   if (AlignInBits < RoundedSize)
     return false;
 
@@ -634,7 +638,8 @@ static LLT castBufferRsrcFromV4I32(MachineInstr &MI, MachineIRBuilder &B,
   const LLT VectorTy = getBufferRsrcRegisterType(PointerTy);
   if (!PointerTy.isVector()) {
     // Happy path: (4 x s32) -> (s32, s32, s32, s32) -> (p8)
-    const unsigned NumParts = PointerTy.getSizeInBits() / 32;
+    const unsigned NumParts =
+        static_cast<unsigned>(PointerTy.getSizeInBits() / 32);
     const LLT I32 = LLT::integer(32);
 
     Register VectorReg = MRI.createGenericVirtualRegister(VectorTy);
@@ -670,7 +675,8 @@ static Register castBufferRsrcToV4I32(Register Pointer, MachineIRBuilder &B) {
   if (!PointerTy.isVector()) {
     // Special case: p8 -> (s32, s32, s32, s32) -> (4xs32)
     SmallVector<Register, 4> PointerParts;
-    const unsigned NumParts = PointerTy.getSizeInBits() / 32;
+    const unsigned NumParts =
+        static_cast<unsigned>(PointerTy.getSizeInBits() / 32);
     auto Unmerged = B.buildUnmerge(LLT::integer(32), Pointer);
     for (unsigned I = 0; I < NumParts; ++I)
       PointerParts.push_back(Unmerged.getReg(I));
@@ -1548,11 +1554,13 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
       .legalIf(sameSize(0, 1))
       .widenScalarIf(smallerThan(1, 0),
                      [](const LegalityQuery &Query) {
-                       return std::pair(
-                           1, LLT::scalar(Query.Types[0].getSizeInBits()));
+                       return std::pair(1,
+                                        LLT::scalar(static_cast<unsigned>(
+                                            Query.Types[0].getSizeInBits())));
                      })
       .narrowScalarIf(largerThan(1, 0), [](const LegalityQuery &Query) {
-        return std::pair(1, LLT::scalar(Query.Types[0].getSizeInBits()));
+        return std::pair(1, LLT::scalar(static_cast<unsigned>(
+                                Query.Types[0].getSizeInBits())));
       });
 
   getActionDefinitionsBuilder(G_PTRTOINT)
@@ -1564,11 +1572,13 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
       .legalIf(sameSize(0, 1))
       .widenScalarIf(smallerThan(0, 1),
                      [](const LegalityQuery &Query) {
-                       return std::pair(
-                           0, LLT::scalar(Query.Types[1].getSizeInBits()));
+                       return std::pair(0,
+                                        LLT::scalar(static_cast<unsigned>(
+                                            Query.Types[1].getSizeInBits())));
                      })
       .narrowScalarIf(largerThan(0, 1), [](const LegalityQuery &Query) {
-        return std::pair(0, LLT::scalar(Query.Types[1].getSizeInBits()));
+        return std::pair(0, LLT::scalar(static_cast<unsigned>(
+                                Query.Types[1].getSizeInBits())));
       });
 
   getActionDefinitionsBuilder(G_ADDRSPACE_CAST)
@@ -1580,7 +1590,8 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
     const LLT DstTy = Query.Types[0];
 
     // Split vector extloads.
-    unsigned MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
+    unsigned MemSize =
+        static_cast<unsigned>(Query.MMODescrs[0].MemoryTy.getSizeInBits());
 
     if (DstTy.isVector() && DstTy.getSizeInBits() > MemSize)
       return true;
@@ -1712,8 +1723,10 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
               const LLT DstTy = Query.Types[0];
               const LLT PtrTy = Query.Types[1];
 
-              const unsigned DstSize = DstTy.getSizeInBits();
-              unsigned MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
+              const unsigned DstSize =
+                  static_cast<unsigned>(DstTy.getSizeInBits());
+              unsigned MemSize = static_cast<unsigned>(
+                  Query.MMODescrs[0].MemoryTy.getSizeInBits());
 
               // Split extloads.
               if (DstSize > MemSize)
@@ -1726,7 +1739,7 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
                 return std::pair(0, LLT::scalar(MaxSize));
 
               uint64_t Align = Query.MMODescrs[0].AlignInBits;
-              return std::pair(0, LLT::scalar(Align));
+              return std::pair(0, LLT::scalar(static_cast<unsigned>(Align)));
             })
         .fewerElementsIf(
             [=](const LegalityQuery &Query) -> bool {
@@ -1747,10 +1760,11 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
               // FIXME: 3 element stores scalarized on SI
 
               // Split if it's too large for the address space.
-              unsigned MemSize = Query.MMODescrs[0].MemoryTy.getSizeInBits();
+              unsigned MemSize = static_cast<unsigned>(
+                  Query.MMODescrs[0].MemoryTy.getSizeInBits());
               if (MemSize > MaxSize) {
                 unsigned NumElts = DstTy.getNumElements();
-                unsigned EltSize = EltTy.getSizeInBits();
+                unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
 
                 if (MaxSize % EltSize == 0) {
                   return std::pair(
@@ -1774,8 +1788,8 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
               if (DstTy.getSizeInBits() > MemSize)
                 return std::pair(0, EltTy);
 
-              unsigned EltSize = EltTy.getSizeInBits();
-              unsigned DstSize = DstTy.getSizeInBits();
+              unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
+              unsigned DstSize = static_cast<unsigned>(DstTy.getSizeInBits());
               if (!isPowerOf2_32(DstSize)) {
                 // We're probably decomposing an odd sized store. Try to split
                 // to the widest type. TODO: Account for alignment. As-is it
@@ -1990,9 +2004,9 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
           const LLT EltTy = Query.Types[EltTypeIdx];
           const LLT VecTy = Query.Types[VecTypeIdx];
           const LLT IdxTy = Query.Types[IdxTypeIdx];
-          const unsigned EltSize = EltTy.getSizeInBits();
-          const bool isLegalVecType =
-              !!SIRegisterInfo::getSGPRClassForBitWidth(VecTy.getSizeInBits());
+          const unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
+          const bool isLegalVecType = !!SIRegisterInfo::getSGPRClassForBitWidth(
+              static_cast<unsigned>(VecTy.getSizeInBits()));
           // Address space 8 pointers are 128-bit wide values, but the logic
           // below will try to bitcast them to 2N x s64, which will fail.
           // Therefore, as an intermediate step, wrap extracts/insertions from a
@@ -2019,8 +2033,10 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
                      // indexing if this is scalar. If not, fall back to 32.
                      const LLT EltTy = Query.Types[EltTypeIdx];
                      const LLT VecTy = Query.Types[VecTypeIdx];
-                     const unsigned DstEltSize = EltTy.getSizeInBits();
-                     const unsigned VecSize = VecTy.getSizeInBits();
+                     const unsigned DstEltSize =
+                         static_cast<unsigned>(EltTy.getSizeInBits());
+                     const unsigned VecSize =
+                         static_cast<unsigned>(VecTy.getSizeInBits());
 
                      const unsigned TargetEltSize =
                          DstEltSize % 64 == 0 ? 64 : 32;
@@ -2130,7 +2146,8 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
         const LLT &EltTy = Ty.getElementType();
         if (EltTy.getSizeInBits() < 8 || EltTy.getSizeInBits() > 512)
           return true;
-        if (!llvm::has_single_bit<uint32_t>(EltTy.getSizeInBits()))
+        if (!llvm::has_single_bit<uint32_t>(
+                static_cast<unsigned>(EltTy.getSizeInBits())))
           return true;
       }
       return false;
@@ -2190,9 +2207,11 @@ AMDGPULegalizerInfo::AMDGPULegalizerInfo(const GCNSubtarget &ST_,
         // Pick the next power of 2, or a multiple of 64 over 128.
         // Whichever is smaller.
         const LLT &Ty = Query.Types[BigTyIdx];
-        unsigned NewSizeInBits = 1 << Log2_32_Ceil(Ty.getSizeInBits() + 1);
+        unsigned NewSizeInBits =
+            1 << Log2_32_Ceil(static_cast<unsigned>(Ty.getSizeInBits() + 1));
         if (NewSizeInBits >= 256) {
-          unsigned RoundedTo = alignTo<64>(Ty.getSizeInBits() + 1);
+          unsigned RoundedTo =
+              static_cast<unsigned>(alignTo<64>(Ty.getSizeInBits() + 1));
           if (RoundedTo < NewSizeInBits)
             NewSizeInBits = RoundedTo;
         }
@@ -2622,7 +2641,8 @@ bool AMDGPULegalizerInfo::legalizeAddrSpaceCast(
       return true;
     }
 
-    unsigned NullVal = AMDGPU::getNullPointerValue(DestAS);
+    unsigned NullVal =
+        static_cast<unsigned>(AMDGPU::getNullPointerValue(DestAS));
 
     auto SegmentNull = B.buildConstant(DstTy, NullVal);
     auto FlatNull = B.buildConstant(SrcTy, 0);
@@ -3030,8 +3050,8 @@ bool AMDGPULegalizerInfo::legalizeExtract(LegalizerHelper &Helper,
     return Helper.lowerExtract(MI) == LegalizerHelper::Legalized;
 
   const LLT DstTy = MRI.getType(DstReg);
-  unsigned StartIdx = Offset / 32;
-  unsigned DstCount = DstTy.getSizeInBits() / 32;
+  unsigned StartIdx = static_cast<unsigned>(Offset / 32);
+  unsigned DstCount = static_cast<unsigned>(DstTy.getSizeInBits() / 32);
   auto Unmerge = B.buildUnmerge(LLT::integer(32), SrcReg);
 
   if (DstCount == 1) {
@@ -3062,9 +3082,9 @@ bool AMDGPULegalizerInfo::legalizeInsert(LegalizerHelper &Helper,
   Register InsertSrc = MI.getOperand(2).getReg();
   uint64_t Offset = MI.getOperand(3).getImm();
 
-  unsigned DstSize = MRI.getType(DstReg).getSizeInBits();
+  unsigned DstSize = static_cast<unsigned>(MRI.getType(DstReg).getSizeInBits());
   const LLT InsertTy = MRI.getType(InsertSrc);
-  unsigned InsertSize = InsertTy.getSizeInBits();
+  unsigned InsertSize = static_cast<unsigned>(InsertTy.getSizeInBits());
 
   // Fall back to generic lowering for non-32-bit-aligned cases which
   // require shift+mask sequences that generic code handles correctly.
@@ -3074,7 +3094,7 @@ bool AMDGPULegalizerInfo::legalizeInsert(LegalizerHelper &Helper,
   const LLT I32 = LLT::integer(32);
   unsigned DstCount = DstSize / 32;
   unsigned InsertCount = InsertSize / 32;
-  unsigned StartIdx = Offset / 32;
+  unsigned StartIdx = static_cast<unsigned>(Offset / 32);
 
   auto SrcUnmerge = B.buildUnmerge(I32, SrcReg);
 
@@ -3123,7 +3143,7 @@ bool AMDGPULegalizerInfo::legalizeExtractVectorElt(
   // vector of integers using ptrtoint (and inttoptr on the output) in order to
   // drive the legalization forward.
   if (EltTy.isPointer() && EltTy.getSizeInBits() > 64) {
-    LLT IntTy = LLT::integer(EltTy.getSizeInBits());
+    LLT IntTy = LLT::integer(static_cast<unsigned>(EltTy.getSizeInBits()));
     LLT IntVecTy = VecTy.changeElementType(IntTy);
 
     auto IntVec = B.buildPtrToInt(IntVecTy, Vec);
@@ -3145,7 +3165,7 @@ bool AMDGPULegalizerInfo::legalizeExtractVectorElt(
 
   if (IdxVal < VecTy.getNumElements()) {
     auto Unmerge = B.buildUnmerge(EltTy, Vec);
-    B.buildCopy(Dst, Unmerge.getReg(IdxVal));
+    B.buildCopy(Dst, Unmerge.getReg(static_cast<unsigned>(IdxVal)));
   } else {
     B.buildUndef(Dst);
   }
@@ -3176,7 +3196,7 @@ bool AMDGPULegalizerInfo::legalizeInsertVectorElt(
   // new value, and then inttoptr the result vector back. This will then allow
   // the rest of legalization to take over.
   if (EltTy.isPointer() && EltTy.getSizeInBits() > 64) {
-    LLT IntTy = LLT::integer(EltTy.getSizeInBits());
+    LLT IntTy = LLT::integer(static_cast<unsigned>(EltTy.getSizeInBits()));
     LLT IntVecTy = VecTy.changeElementType(IntTy);
 
     auto IntVecSource = B.buildPtrToInt(IntVecTy, Vec);
@@ -3476,9 +3496,10 @@ bool AMDGPULegalizerInfo::legalizeGlobalValue(
 
 static LLT widenToNextPowerOf2(LLT Ty) {
   if (Ty.isVector())
-    return Ty.changeElementCount(
-        ElementCount::getFixed(PowerOf2Ceil(Ty.getNumElements())));
-  return Ty.changeElementSize(PowerOf2Ceil(Ty.getSizeInBits()));
+    return Ty.changeElementCount(ElementCount::getFixed(
+        static_cast<unsigned>(PowerOf2Ceil(Ty.getNumElements()))));
+  return Ty.changeElementSize(
+      static_cast<unsigned>(PowerOf2Ceil(Ty.getSizeInBits())));
 }
 
 bool AMDGPULegalizerInfo::legalizeLoad(LegalizerHelper &Helper,
@@ -3514,15 +3535,15 @@ bool AMDGPULegalizerInfo::legalizeLoad(LegalizerHelper &Helper,
   }
 
   MachineMemOperand *MMO = *MI.memoperands_begin();
-  const unsigned ValSize = ValTy.getSizeInBits();
+  const unsigned ValSize = static_cast<unsigned>(ValTy.getSizeInBits());
   const LLT MemTy = MMO->getMemoryType();
   const Align MemAlign = MMO->getAlign();
-  const unsigned MemSize = MemTy.getSizeInBits();
+  const unsigned MemSize = static_cast<unsigned>(MemTy.getSizeInBits());
   const uint64_t AlignInBits = 8 * MemAlign.value();
 
   // Widen non-power-of-2 loads to the alignment if needed
   if (shouldWidenLoad(ST, MemTy, AlignInBits, AddrSpace, MI.getOpcode())) {
-    const unsigned WideMemSize = PowerOf2Ceil(MemSize);
+    const unsigned WideMemSize = static_cast<unsigned>(PowerOf2Ceil(MemSize));
 
     // This was already the correct extending load result type, so just adjust
     // the memory type.
@@ -4777,7 +4798,7 @@ bool AMDGPULegalizerInfo::legalizeMul(LegalizerHelper &Helper,
   LLT Ty = MRI.getType(DstReg);
   assert(Ty.isScalar());
 
-  unsigned Size = Ty.getSizeInBits();
+  unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
   if (ST.useVMulU64Inst() && Size == 64)
     return true;
 
@@ -4862,7 +4883,7 @@ bool AMDGPULegalizerInfo::legalizeCTLS(MachineInstr &MI,
   LLT SrcTy = MRI.getType(Src);
   const LLT I32 = LLT::integer(32);
   assert(SrcTy == I32 && "legalizeCTLS only supports i32");
-  unsigned BitWidth = SrcTy.getSizeInBits();
+  unsigned BitWidth = static_cast<unsigned>(SrcTy.getSizeInBits());
 
   auto Sffbh = B.buildIntrinsic(Intrinsic::amdgcn_sffbh, {I32}).addUse(Src);
   auto Clamped = B.buildUMin(I32, Sffbh, B.buildConstant(I32, BitWidth));
@@ -5004,7 +5025,8 @@ bool AMDGPULegalizerInfo::legalizeWorkGroupId(
   }
   case AMDGPU::ClusterDimsAttr::Kind::Unknown: {
     using namespace AMDGPU::Hwreg;
-    unsigned ClusterIdField = HwregEncoding::encode(ID_IB_STS2, 6, 4);
+    unsigned ClusterIdField =
+        static_cast<unsigned>(HwregEncoding::encode(ID_IB_STS2, 6, 4));
     Register ClusterId = MRI.createGenericVirtualRegister(I32);
     MRI.setRegClass(ClusterId, &AMDGPU::SReg_32RegClass);
     B.buildInstr(AMDGPU::S_GETREG_B32_const)
@@ -5586,7 +5608,7 @@ bool AMDGPULegalizerInfo::legalizeFastUnsafeFDIV(MachineInstr &MI,
   Register Res = MI.getOperand(0).getReg();
   Register LHS = MI.getOperand(1).getReg();
   Register RHS = MI.getOperand(2).getReg();
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
   LLT ResTy = MRI.getType(Res);
 
   bool AllowInaccurateRcp = MI.getFlag(MachineInstr::FmAfn);
@@ -5646,7 +5668,7 @@ bool AMDGPULegalizerInfo::legalizeFastUnsafeFDIV64(MachineInstr &MI,
   Register Res = MI.getOperand(0).getReg();
   Register X = MI.getOperand(1).getReg();
   Register Y = MI.getOperand(2).getReg();
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
   LLT ResTy = MRI.getType(Res);
 
   bool AllowInaccurateRcp = MI.getFlag(MachineInstr::FmAfn);
@@ -5701,7 +5723,7 @@ bool AMDGPULegalizerInfo::legalizeFDIV16(MachineInstr &MI,
   Register LHS = MI.getOperand(1).getReg();
   Register RHS = MI.getOperand(2).getReg();
 
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
 
   LLT I32 = LLT::integer(32);
 
@@ -5751,8 +5773,8 @@ bool AMDGPULegalizerInfo::legalizeFDIV16(MachineInstr &MI,
   return true;
 }
 
-static constexpr unsigned SPDenormModeBitField =
-    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_MODE, 4, 2);
+static constexpr unsigned SPDenormModeBitField = static_cast<unsigned>(
+    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_MODE, 4, 2));
 
 // Enable or disable FP32 denorm mode. When 'Enable' is true, emit instructions
 // to enable denorm mode. When 'Enable' is false, disable denorm mode.
@@ -5790,7 +5812,7 @@ bool AMDGPULegalizerInfo::legalizeFDIV32(MachineInstr &MI,
   const SIMachineFunctionInfo *MFI = B.getMF().getInfo<SIMachineFunctionInfo>();
   SIModeRegisterDefaults Mode = MFI->getMode();
 
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
 
   LLT S1 = LLT::scalar(1);
 
@@ -5874,7 +5896,7 @@ bool AMDGPULegalizerInfo::legalizeFDIV64(MachineInstr &MI,
   Register LHS = MI.getOperand(1).getReg();
   Register RHS = MI.getOperand(2).getReg();
 
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
 
   LLT S1 = LLT::scalar(1);
 
@@ -5951,7 +5973,7 @@ bool AMDGPULegalizerInfo::legalizeFFREXP(MachineInstr &MI,
   Register Res0 = MI.getOperand(0).getReg();
   Register Res1 = MI.getOperand(1).getReg();
   Register Val = MI.getOperand(2).getReg();
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
 
   LLT Ty = MRI.getType(Res0);
   LLT InstrExpTy = Ty == F16 ? LLT::integer(16) : LLT::integer(32);
@@ -5986,7 +6008,7 @@ bool AMDGPULegalizerInfo::legalizeFDIVFastIntrin(MachineInstr &MI,
   Register Res = MI.getOperand(0).getReg();
   Register LHS = MI.getOperand(2).getReg();
   Register RHS = MI.getOperand(3).getReg();
-  uint16_t Flags = MI.getFlags();
+  uint16_t Flags = static_cast<uint16_t>(MI.getFlags());
 
   LLT S1 = LLT::scalar(1);
 
@@ -6339,12 +6361,13 @@ bool AMDGPULegalizerInfo::legalizeLaneOp(LegalizerHelper &Helper,
   }
 
   LLT Ty = MRI.getType(DstReg);
-  unsigned Size = Ty.getSizeInBits();
+  unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
 
   unsigned SplitSize = 32;
   if (IID == Intrinsic::amdgcn_update_dpp && (Size % 64 == 0) &&
       ST.hasDPALU_DPP() &&
-      AMDGPU::isLegalDPALU_DPPControl(ST, MI.getOperand(4).getImm()))
+      AMDGPU::isLegalDPALU_DPPControl(
+          ST, static_cast<unsigned>(MI.getOperand(4).getImm())))
     SplitSize = 64;
 
   if (Size == SplitSize) {
@@ -6390,7 +6413,7 @@ bool AMDGPULegalizerInfo::legalizeLaneOp(LegalizerHelper &Helper,
   bool NeedsBitcast = false;
   if (IntTy.isVector()) {
     LLT EltTy = IntTy.getElementType();
-    unsigned EltSize = EltTy.getSizeInBits();
+    unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
     if (EltSize == SplitSize) {
       PartialResTy = EltTy;
     } else if (EltSize == 16 || EltSize == 32) {
@@ -6426,8 +6449,9 @@ bool AMDGPULegalizerInfo::legalizeLaneOp(LegalizerHelper &Helper,
 
   if (NeedsBitcast || IsFloat)
     B.buildBitcast(
-        DstReg,
-        B.buildMergeLikeInstr(LLT::integer(IntTy.getSizeInBits()), PartialRes));
+        DstReg, B.buildMergeLikeInstr(
+                    LLT::integer(static_cast<unsigned>(IntTy.getSizeInBits())),
+                    PartialRes));
   else
     B.buildMergeLikeInstr(DstReg, PartialRes);
 
@@ -6442,7 +6466,7 @@ bool AMDGPULegalizerInfo::getImplicitArgPtr(Register DstReg,
     ST.getTargetLowering()->getImplicitParameterOffset(
       B.getMF(), AMDGPUTargetLowering::FIRST_IMPLICIT);
   LLT DstTy = MRI.getType(DstReg);
-  LLT IdxTy = LLT::integer(DstTy.getSizeInBits());
+  LLT IdxTy = LLT::integer(static_cast<unsigned>(DstTy.getSizeInBits()));
 
   Register KernargPtrReg = MRI.createGenericVirtualRegister(DstTy);
   if (!loadInputValue(KernargPtrReg, B,
@@ -6481,7 +6505,8 @@ bool AMDGPULegalizerInfo::legalizePointerAsRsrcIntrin(
     Register Zero = B.buildConstant(I32, 0).getReg(0);
     // Build the lower 64-bit value, which has a 57-bit base and the lower 7-bit
     // num_records.
-    LLT PtrIntTy = LLT::integer(MRI.getType(Pointer).getSizeInBits());
+    LLT PtrIntTy = LLT::integer(
+        static_cast<unsigned>(MRI.getType(Pointer).getSizeInBits()));
     auto PointerInt = B.buildPtrToInt(PtrIntTy, Pointer);
     auto ExtPointer = B.buildAnyExtOrTrunc(I64, PointerInt);
     auto NumRecordsLHS = B.buildShl(I64, NumRecords, B.buildConstant(I32, 57));
@@ -6762,7 +6787,7 @@ bool AMDGPULegalizerInfo::legalizeBufferStore(MachineInstr &MI,
   const LLT I32 = LLT::integer(32);
 
   MachineMemOperand *MMO = *MI.memoperands_begin();
-  const int MemSize = MMO->getSize().getValue();
+  const int MemSize = static_cast<int>(MMO->getSize().getValue());
   LLT MemTy = MMO->getMemoryType();
 
   if (IsFormat && !IsTyped && !IsD16 && MemTy.getSizeInBits() < 32) {
@@ -6799,11 +6824,12 @@ bool AMDGPULegalizerInfo::legalizeBufferStore(MachineInstr &MI,
 
   unsigned Format = 0;
   if (IsTyped) {
-    Format = MI.getOperand(5 + OpOffset).getImm();
+    Format = static_cast<unsigned>(MI.getOperand(5 + OpOffset).getImm());
     ++OpOffset;
   }
 
-  unsigned AuxiliaryData = MI.getOperand(5 + OpOffset).getImm();
+  unsigned AuxiliaryData =
+      static_cast<unsigned>(MI.getOperand(5 + OpOffset).getImm());
 
   std::tie(VOffset, ImmOffset) = splitBufferOffsets(B, VOffset);
 
@@ -6929,11 +6955,12 @@ bool AMDGPULegalizerInfo::legalizeBufferLoad(MachineInstr &MI,
 
   unsigned Format = 0;
   if (IsTyped) {
-    Format = MI.getOperand(5 + OpOffset).getImm();
+    Format = static_cast<unsigned>(MI.getOperand(5 + OpOffset).getImm());
     ++OpOffset;
   }
 
-  unsigned AuxiliaryData = MI.getOperand(5 + OpOffset).getImm();
+  unsigned AuxiliaryData =
+      static_cast<unsigned>(MI.getOperand(5 + OpOffset).getImm());
   unsigned ImmOffset;
 
   LLT Ty = MRI.getType(Dst);
@@ -7048,11 +7075,13 @@ bool AMDGPULegalizerInfo::legalizeBufferLoad(MachineInstr &MI,
       }
     }
   } else if (IsTFE) {
-    const unsigned NumValueDWords = divideCeil(Ty.getSizeInBits(), 32);
+    const unsigned NumValueDWords =
+        static_cast<unsigned>(divideCeil(Ty.getSizeInBits(), 32));
     Register DstInt =
-        EltTy.isFloat() ? MRI.createGenericVirtualRegister(Ty.changeElementType(
-                              LLT::integer(EltTy.getSizeInBits())))
-                        : Dst;
+        EltTy.isFloat()
+            ? MRI.createGenericVirtualRegister(Ty.changeElementType(
+                  LLT::integer(static_cast<unsigned>(EltTy.getSizeInBits()))))
+            : Dst;
     if (MemTy.getSizeInBits() < 32) {
       Register ExtDst = MRI.createGenericVirtualRegister(I32);
       buildTFEBufferLoad(Opc, ExtDst, StatusDst, RSrc, VIndex, VOffset, SOffset,
@@ -7237,7 +7266,8 @@ bool AMDGPULegalizerInfo::legalizeBufferAtomic(MachineInstr &MI,
 
   Register VOffset = MI.getOperand(4 + OpOffset).getReg();
   Register SOffset = MI.getOperand(5 + OpOffset).getReg();
-  unsigned AuxiliaryData = MI.getOperand(6 + OpOffset).getImm();
+  unsigned AuxiliaryData =
+      static_cast<unsigned>(MI.getOperand(6 + OpOffset).getImm());
 
   MachineMemOperand *MMO = *MI.memoperands_begin();
 
@@ -7343,7 +7373,7 @@ static void convertImageAddrToPacked(MachineIRBuilder &B, MachineInstr &MI,
     }
   }
 
-  int NumAddrRegs = AddrRegs.size();
+  int NumAddrRegs = static_cast<int>(AddrRegs.size());
   if (NumAddrRegs != 1) {
     LLT EltTy = B.getMRI()->getType(AddrRegs[0]);
     auto VAddr =
@@ -7420,7 +7450,8 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
 
   int DMaskLanes = 0;
   if (!BaseOpcode->Atomic) {
-    DMask = MI.getOperand(ArgOffset + Intr->DMaskIndex).getImm();
+    DMask = static_cast<unsigned>(
+        MI.getOperand(ArgOffset + Intr->DMaskIndex).getImm());
     if (BaseOpcode->Gather4) {
       DMaskLanes = 4;
     } else if (DMask != 0) {
@@ -7508,20 +7539,21 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
 
     if (UsePartialNSA) {
       // Pack registers that would go over NSAMaxSize into last VAddr register
-      LLT PackedAddrTy =
-          LLT::fixed_vector(2 * (PackedRegs.size() - NSAMaxSize + 1), F16);
+      LLT PackedAddrTy = LLT::fixed_vector(
+          static_cast<unsigned>(2 * (PackedRegs.size() - NSAMaxSize + 1)), F16);
       auto Concat = B.buildConcatVectors(
           PackedAddrTy, ArrayRef(PackedRegs).slice(NSAMaxSize - 1));
       PackedRegs[NSAMaxSize - 1] = Concat.getReg(0);
       PackedRegs.resize(NSAMaxSize);
     } else if (!UseNSA && PackedRegs.size() > 1) {
-      LLT PackedAddrTy = LLT::fixed_vector(2 * PackedRegs.size(), F16);
+      LLT PackedAddrTy =
+          LLT::fixed_vector(static_cast<unsigned>(2 * PackedRegs.size()), F16);
       auto Concat = B.buildConcatVectors(PackedAddrTy, PackedRegs);
       PackedRegs[0] = Concat.getReg(0);
       PackedRegs.resize(1);
     }
 
-    const unsigned NumPacked = PackedRegs.size();
+    const unsigned NumPacked = static_cast<unsigned>(PackedRegs.size());
     for (unsigned I = Intr->VAddrStart; I < Intr->VAddrEnd; I++) {
       MachineOperand &SrcOp = MI.getOperand(ArgOffset + I);
       if (!SrcOp.isReg()) {
@@ -7629,8 +7661,9 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
     TFETy = LLT::fixed_vector(AdjustedNumElts + 1, I32);
     RegTy = I32;
   } else {
-    unsigned EltSize = EltTy.getSizeInBits();
-    unsigned RoundedElts = (AdjustedTy.getSizeInBits() + 31) / 32;
+    unsigned EltSize = static_cast<unsigned>(EltTy.getSizeInBits());
+    unsigned RoundedElts =
+        static_cast<unsigned>((AdjustedTy.getSizeInBits() + 31) / 32);
     unsigned RoundedSize = 32 * RoundedElts;
     RoundedTy = LLT::scalarOrVector(
         ElementCount::getFixed(RoundedSize / EltSize), EltTy);
@@ -7651,7 +7684,7 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
   // TODO: For TFE with d16, if we used a TFE type that was a multiple of <2 x
   // f16> instead of i32, we would only need 1 bitcast instead of multiple.
   const LLT LoadResultTy = IsTFE ? TFETy : RoundedTy;
-  const int ResultNumRegs = LoadResultTy.getSizeInBits() / 32;
+  const int ResultNumRegs = static_cast<int>(LoadResultTy.getSizeInBits() / 32);
 
   Register NewResultReg = MRI->createGenericVirtualRegister(LoadResultTy);
 
@@ -7742,13 +7775,13 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
   // Pad out any elements eliminated due to the dmask.
   LLT ResTy = MRI->getType(ResultRegs[0]);
   if (!ResTy.isVector()) {
-    padWithUndef(ResTy, NumElts - ResultRegs.size());
+    padWithUndef(ResTy, static_cast<int>(NumElts - ResultRegs.size()));
     B.buildBuildVector(DstReg, ResultRegs);
     return true;
   }
 
   assert(!ST.hasUnpackedD16VMem() && (ResTy == V2I16 || ResTy == V2F16));
-  const int RegsToCover = (Ty.getSizeInBits() + 31) / 32;
+  const int RegsToCover = static_cast<int>((Ty.getSizeInBits() + 31) / 32);
 
   // Deal with the one annoying legal case.
   const LLT V3I16 = LLT::fixed_vector(3, I16);
@@ -7783,7 +7816,7 @@ bool AMDGPULegalizerInfo::legalizeImageIntrinsic(
     return true;
   }
 
-  padWithUndef(ResTy, RegsToCover - ResultRegs.size());
+  padWithUndef(ResTy, static_cast<int>(RegsToCover - ResultRegs.size()));
   B.buildConcatVectors(DstReg, ResultRegs);
   return true;
 }
@@ -7796,7 +7829,7 @@ bool AMDGPULegalizerInfo::legalizeSBufferLoad(LegalizerHelper &Helper,
   Register OrigDst = MI.getOperand(0).getReg();
   Register Dst;
   LLT Ty = B.getMRI()->getType(OrigDst);
-  unsigned Size = Ty.getSizeInBits();
+  unsigned Size = static_cast<unsigned>(Ty.getSizeInBits());
   MachineFunction &MF = B.getMF();
   bool HasMMO = !MI.memoperands_empty();
 
@@ -8173,7 +8206,7 @@ bool AMDGPULegalizerInfo::legalizeBVHIntersectRayIntrinsic(
 
   if (!UseNSA) {
     // Build a single vector containing all the operands so far prepared.
-    LLT OpTy = LLT::fixed_vector(Ops.size(), I32);
+    LLT OpTy = LLT::fixed_vector(static_cast<unsigned>(Ops.size()), I32);
     Register MergedOps = B.buildMergeLikeInstr(OpTy, Ops).getReg(0);
     Ops.clear();
     Ops.push_back(MergedOps);
@@ -8285,11 +8318,11 @@ bool AMDGPULegalizerInfo::legalizeConstHwRegRead(MachineInstr &MI,
   return true;
 }
 
-static constexpr unsigned FPEnvModeBitField =
-    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_MODE, 0, 23);
+static constexpr unsigned FPEnvModeBitField = static_cast<unsigned>(
+    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_MODE, 0, 23));
 
-static constexpr unsigned FPEnvTrapBitField =
-    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_TRAPSTS, 0, 5);
+static constexpr unsigned FPEnvTrapBitField = static_cast<unsigned>(
+    AMDGPU::Hwreg::HwregEncoding::encode(AMDGPU::Hwreg::ID_TRAPSTS, 0, 5));
 
 bool AMDGPULegalizerInfo::legalizeGetFPEnv(MachineInstr &MI,
                                            MachineRegisterInfo &MRI,



More information about the llvm-commits mailing list