[llvm] [AMDGPU][NFC] Explicitly narrow conversions in the IR-level passes (PR #215207)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 06:49:45 PDT 2026
https://github.com/gretay-amd updated https://github.com/llvm/llvm-project/pull/215207
>From 02c6767cc5aed6bc98ff0e88736d3cfcbb8db579 Mon Sep 17 00:00:00 2001
From: Greta Y <Greta.Yorsh at amd.com>
Date: Thu, 6 Aug 2026 16:15:30 +0100
Subject: [PATCH] [AMDGPU][NFC] Explicitly narrow conversions in the IR-level
passes
This patch handles the following cases:
DataLayout::getTypeAllocSize, getTypeStoreSize and getTypeSizeInBits return
TypeSize, and ConstantInt::getZExtValue returns uint64_t; both are assigned to
32-bit locals holding element sizes, offsets and address spaces. Add a
static_cast to make the existing narrowing conversion explicit. The address
spaces are AMDGPUAS enumerators; the sizes are those of the buffer-fat-pointer
and LDS objects these passes rewrite, which the LDS limit and the alloca
promotion thresholds already bound well below 2^32.
Container size() returns size_t and is assigned to unsigned or int locals, or
passed to an unsigned parameter, holding element and operand counts. Add a
static_cast to make the existing narrowing conversion explicit. One of these is
IRBuilder::CreatePHI, whose NumReservedValues parameter is only a capacity hint
and is bounded here by the number of returning blocks in a function.
FixedVectorType::getNumElements() returns unsigned but the element counts
reaching some of these sites come from a wider ElementCount or uint64_t. Add a
static_cast to make the existing narrowing conversion explicit.
In AMDGPURewriteAGPRCopyMFMA, a register is passed to tryReassigningMFMAChain,
whose third parameter is MCPhysReg. Add a static_cast on that argument only. The
second parameter is Register and must not be cast: it carries a virtual register
number, whose high bit distinguishes virtual from physical, and truncating it to
MCPhysReg (uint16_t) destroys that. Casting the wrong argument here silences the
warning just as well and breaks 12 CodeGen/AMDGPU tests, which is how this was
caught.
This fixes 105 instances of MSVC warning C4244 and 30 of C4267 ("possible loss
of data") across 20 files in llvm/lib/Target/AMDGPU.
Assisted-by: Claude <noreply at anthropic.com>
---
.../AMDGPU/AMDGPUAsanInstrumentation.cpp | 8 +-
.../Target/AMDGPU/AMDGPUAtomicOptimizer.cpp | 5 +-
llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp | 19 ++--
.../Target/AMDGPU/AMDGPUCodeGenPrepare.cpp | 28 +++---
.../AMDGPU/AMDGPUImageIntrinsicOptimizer.cpp | 2 +-
.../AMDGPU/AMDGPULateCodeGenPrepare.cpp | 31 ++++---
.../AMDGPU/AMDGPULowerBufferFatPointers.cpp | 92 +++++++++++--------
.../lib/Target/AMDGPU/AMDGPULowerExecSync.cpp | 2 +-
.../Target/AMDGPU/AMDGPULowerIntrinsics.cpp | 3 +-
.../AMDGPU/AMDGPULowerKernelArguments.cpp | 2 +-
.../AMDGPU/AMDGPULowerKernelAttributes.cpp | 3 +-
.../AMDGPU/AMDGPULowerModuleLDSPass.cpp | 15 +--
.../AMDGPU/AMDGPUPreloadKernelArguments.cpp | 11 ++-
.../AMDGPU/AMDGPUPrintfRuntimeBinding.cpp | 13 ++-
.../lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp | 68 ++++++++------
.../AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp | 12 ++-
.../AMDGPU/AMDGPURewriteOutArguments.cpp | 8 +-
llvm/lib/Target/AMDGPU/AMDGPUSplitModule.cpp | 20 ++--
llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp | 24 ++---
.../AMDGPU/AMDGPUUnifyDivergentExitNodes.cpp | 3 +-
20 files changed, 221 insertions(+), 148 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsanInstrumentation.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsanInstrumentation.cpp
index 991995a0c3d485..96fcff6e656604 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsanInstrumentation.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsanInstrumentation.cpp
@@ -197,8 +197,9 @@ void instrumentAddress(Module &M, IRBuilder<> &IRB, Instruction *OrigIns,
if (Alignment.value() >= Granularity ||
Alignment.value() >= FixedSize / 8)
return instrumentAddressImpl(
- M, IRB, OrigIns, InsertBefore, Addr, Alignment, FixedSize, IsWrite,
- SizeArgument, UseCalls, Recover, AsanScale, AsanOffset);
+ M, IRB, OrigIns, InsertBefore, Addr, Alignment,
+ static_cast<uint32_t>(FixedSize), IsWrite, SizeArgument, UseCalls,
+ Recover, AsanScale, AsanOffset);
}
}
// Instrument unusual size or unusual alignment.
@@ -287,7 +288,8 @@ void getInterestingMemoryOperands(
// Use the pointer alignment as the element alignment if the stride is a
// mutiple of the pointer alignment. Otherwise, the element alignment
// should be Align(1).
- unsigned PointerAlign = Alignment.valueOrOne().value();
+ unsigned PointerAlign =
+ static_cast<unsigned>(Alignment.valueOrOne().value());
if (!isa<ConstantInt>(Stride) ||
cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
Alignment = Align(1);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAtomicOptimizer.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAtomicOptimizer.cpp
index aa822c13dd8fe1..d0e9e4b2291998 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAtomicOptimizer.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAtomicOptimizer.cpp
@@ -621,7 +621,7 @@ std::pair<Value *, Value *> AMDGPUAtomicOptimizerImpl::buildScanIteratively(
static Constant *getIdentityValueForAtomicOp(Type *const Ty,
AtomicRMWInst::BinOp Op) {
LLVMContext &C = Ty->getContext();
- const unsigned BitWidth = Ty->getPrimitiveSizeInBits();
+ const unsigned BitWidth = static_cast<unsigned>(Ty->getPrimitiveSizeInBits());
switch (Op) {
default:
llvm_unreachable("Unhandled atomic op");
@@ -744,7 +744,8 @@ void AMDGPUAtomicOptimizerImpl::optimizeAtomic(Instruction &I,
Type *const Ty = I.getType();
Type *Int32Ty = B.getInt32Ty();
bool isAtomicFloatingPointTy = Ty->isFloatingPointTy();
- [[maybe_unused]] const unsigned TyBitWidth = DL.getTypeSizeInBits(Ty);
+ [[maybe_unused]] const unsigned TyBitWidth =
+ static_cast<unsigned>(DL.getTypeSizeInBits(Ty));
// This is the value in the atomic operation we need to combine in order to
// reduce the number of atomic operations.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
index 2ed1be7c2ade05..eac4027f12dd80 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
@@ -833,8 +833,10 @@ struct AAAMDSizeRangeAttribute
emitAttributeIfNotDefaultAfterClamp(Attributor &A,
std::pair<unsigned, unsigned> Default) {
auto [Min, Max] = Default;
- unsigned Lower = getAssumed().getLower().getZExtValue();
- unsigned Upper = getAssumed().getUpper().getZExtValue();
+ unsigned Lower =
+ static_cast<unsigned>(getAssumed().getLower().getZExtValue());
+ unsigned Upper =
+ static_cast<unsigned>(getAssumed().getUpper().getZExtValue());
// Clamp the range to the default value.
if (Lower < Min)
@@ -1125,10 +1127,12 @@ struct AAAMDWavesPerEU : public AAAMDSizeRangeAttribute {
return false;
ConstantRange Assumed = getAssumed();
- unsigned Min = std::max(Assumed.getLower().getZExtValue(),
- CallerAA->getAssumed().getLower().getZExtValue());
- unsigned Max = std::max(Assumed.getUpper().getZExtValue(),
- CallerAA->getAssumed().getUpper().getZExtValue());
+ unsigned Min = static_cast<unsigned>(
+ std::max(Assumed.getLower().getZExtValue(),
+ CallerAA->getAssumed().getLower().getZExtValue()));
+ unsigned Max = static_cast<unsigned>(
+ std::max(Assumed.getUpper().getZExtValue(),
+ CallerAA->getAssumed().getUpper().getZExtValue()));
ConstantRange Range(APInt(32, Min), APInt(32, Max));
IntegerRangeState RangeState(Range);
getState() = RangeState;
@@ -1219,7 +1223,8 @@ static unsigned inlineAsmGetNumRequiredAGPRs(const InlineAsm *IA,
//
// We ought to be going through TargetLowering to get the number of
// registers, but we should avoid the dependence on CodeGen here.
- RegCount = divideCeil(DL.getTypeSizeInBits(Ty), 32);
+ RegCount =
+ static_cast<unsigned>(divideCeil(DL.getTypeSizeInBits(Ty), 32));
} else {
// Physical register reference
auto [Kind, RegIdx, NumRegs] = AMDGPU::parseAsmConstraintPhysReg(Code);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
index 60fb91d58d2bc7..a97f151524504b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCodeGenPrepare.cpp
@@ -320,7 +320,7 @@ bool AMDGPUCodeGenPrepareImpl::isLegalFloatingTy(const Type *Ty) const {
bool AMDGPUCodeGenPrepareImpl::canWidenScalarExtLoad(LoadInst &I) const {
Type *Ty = I.getType();
- int TySize = DL.getTypeSizeInBits(Ty);
+ int TySize = static_cast<int>(DL.getTypeSizeInBits(Ty));
Align Alignment = DL.getValueOrABITypeAlignment(I.getAlign(), Ty);
return I.isSimple() && TySize < 32 && Alignment >= 4 && UA.isUniformAtDef(&I);
@@ -359,7 +359,7 @@ static Value *insertValues(IRBuilder<> &Builder,
}
Value *NewVal = PoisonValue::get(Ty);
- for (int I = 0, E = Values.size(); I != E; ++I)
+ for (int I = 0, E = static_cast<int>(Values.size()); I != E; ++I)
NewVal = Builder.CreateInsertElement(NewVal, Values[I], I);
return NewVal;
@@ -407,7 +407,7 @@ bool AMDGPUCodeGenPrepareImpl::replaceMulWithMul24(BinaryOperator &I) const {
IntegerType *IntrinTy = Size > 32 ? Builder.getInt64Ty() : I32Ty;
Type *DstTy = LHSVals[0]->getType();
- for (int I = 0, E = LHSVals.size(); I != E; ++I) {
+ for (int I = 0, E = static_cast<int>(LHSVals.size()); I != E; ++I) {
Value *LHS = IsSigned ? Builder.CreateSExtOrTrunc(LHSVals[I], I32Ty)
: Builder.CreateZExtOrTrunc(LHSVals[I], I32Ty);
Value *RHS = IsSigned ? Builder.CreateSExtOrTrunc(RHSVals[I], I32Ty)
@@ -972,7 +972,7 @@ bool AMDGPUCodeGenPrepareImpl::visitFDiv(BinaryOperator &FDiv) {
extractValues(Builder, RsqDenVals, RsqOp);
SmallVector<Value *, 4> ResultVals(NumVals.size());
- for (int I = 0, E = NumVals.size(); I != E; ++I) {
+ for (int I = 0, E = static_cast<int>(NumVals.size()); I != E; ++I) {
Value *NumElt = NumVals[I];
Value *DenElt = DenVals[I];
Value *RsqDenElt = RsqOp ? RsqDenVals[I] : nullptr;
@@ -1566,7 +1566,7 @@ bool AMDGPUCodeGenPrepareImpl::visitLoadInst(LoadInst &I) {
}
}
- int TySize = DL.getTypeSizeInBits(I.getType());
+ int TySize = static_cast<int>(DL.getTypeSizeInBits(I.getType()));
Type *IntNTy = Builder.getIntNTy(TySize);
Value *ValTrunc = Builder.CreateTrunc(WidenLoad, IntNTy);
Value *ValOrig = Builder.CreateBitCast(ValTrunc, I.getType());
@@ -1700,7 +1700,7 @@ static bool isInterestingPHIIncomingValue(const Value *V) {
return false;
CurVal = VecSrc;
- EltsCovered.set(Idx->getZExtValue());
+ EltsCovered.set(static_cast<unsigned>(Idx->getZExtValue()));
// All elements covered.
if (EltsCovered.all())
@@ -1900,7 +1900,7 @@ bool AMDGPUCodeGenPrepareImpl::visitPHINode(PHINode &I) {
unsigned Idx = 0;
// For 8/16 bits type, don't scalarize fully but break it up into as many
// 32-bit slices as we can, and scalarize the tail.
- const unsigned EltSize = DL.getTypeSizeInBits(EltTy);
+ const unsigned EltSize = static_cast<unsigned>(DL.getTypeSizeInBits(EltTy));
const unsigned NumElts = FVT->getNumElements();
if (EltSize == 8 || EltSize == 16) {
const unsigned SubVecSize = (32 / EltSize);
@@ -1931,10 +1931,11 @@ bool AMDGPUCodeGenPrepareImpl::visitPHINode(PHINode &I) {
S.NewPHI = B.CreatePHI(S.Ty, I.getNumIncomingValues());
for (const auto &[Idx, BB] : enumerate(I.blocks())) {
- S.NewPHI->addIncoming(S.getSlicedVal(BB, I.getIncomingValue(Idx),
- "largephi.extractslice" +
- std::to_string(IncNameSuffix++)),
- BB);
+ S.NewPHI->addIncoming(
+ S.getSlicedVal(BB, I.getIncomingValue(static_cast<unsigned>(Idx)),
+ "largephi.extractslice" +
+ std::to_string(IncNameSuffix++)),
+ BB);
}
}
@@ -2127,7 +2128,8 @@ Value *AMDGPUCodeGenPrepareImpl::applyFractPat(IRBuilder<> &Builder,
SmallVector<Value *, 4> ResultVals(FractVals.size());
Type *Ty = FractArg->getType()->getScalarType();
- for (unsigned I = 0, E = FractVals.size(); I != E; ++I) {
+ for (unsigned I = 0, E = static_cast<unsigned>(FractVals.size()); I != E;
+ ++I) {
ResultVals[I] =
Builder.CreateIntrinsic(Intrinsic::amdgcn_fract, {Ty}, {FractVals[I]});
}
@@ -2204,7 +2206,7 @@ bool AMDGPUCodeGenPrepareImpl::visitSqrt(IntrinsicInst &Sqrt) {
extractValues(Builder, SrcVals, SrcVal);
SmallVector<Value *, 4> ResultVals(SrcVals.size());
- for (int I = 0, E = SrcVals.size(); I != E; ++I) {
+ for (int I = 0, E = static_cast<int>(SrcVals.size()); I != E; ++I) {
if (CanTreatAsDAZ)
ResultVals[I] = Builder.CreateCall(getSqrtF32(), SrcVals[I]);
else
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUImageIntrinsicOptimizer.cpp b/llvm/lib/Target/AMDGPU/AMDGPUImageIntrinsicOptimizer.cpp
index ecfdfdf344257b..677a6cfe1a97f0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUImageIntrinsicOptimizer.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUImageIntrinsicOptimizer.cpp
@@ -207,7 +207,7 @@ bool optimizeSection(ArrayRef<SmallVector<IntrinsicInst *, 4>> MergeableInsts) {
// Number of instructions and the number of vaddr/vdata dword transfers
// should be reduced.
- unsigned NumLoads = IIList.size();
+ unsigned NumLoads = static_cast<unsigned>(IIList.size());
unsigned NumMsaas = NumElts;
unsigned NumVAddrLoads = 3 * NumLoads;
unsigned NumVDataLoads = divideCeil(NumElts, isD16 ? 2 : 1) * NumLoads;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
index 0f9fdfb46ce237..32c35a55fcd277 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULateCodeGenPrepare.cpp
@@ -243,14 +243,17 @@ Type *LiveRegOptimizer::calculateConvertType(Type *OriginalType) {
TypeSize OriginalSize = DL.getTypeSizeInBits(VTy);
TypeSize ConvertScalarSize = DL.getTypeSizeInBits(ConvertToScalar);
- unsigned ConvertEltCount =
- (OriginalSize + ConvertScalarSize - 1) / ConvertScalarSize;
+ unsigned ConvertEltCount = static_cast<unsigned>(
+ (OriginalSize + ConvertScalarSize - 1) / ConvertScalarSize);
if (OriginalSize <= ConvertScalarSize)
- return IntegerType::get(Mod.getContext(), ConvertScalarSize);
+ return IntegerType::get(Mod.getContext(),
+ static_cast<unsigned>(ConvertScalarSize));
- return VectorType::get(Type::getIntNTy(Mod.getContext(), ConvertScalarSize),
- ConvertEltCount, false);
+ return VectorType::get(
+ Type::getIntNTy(Mod.getContext(),
+ static_cast<unsigned>(ConvertScalarSize)),
+ ConvertEltCount, false);
}
Value *LiveRegOptimizer::convertToOptType(Instruction *V,
@@ -277,7 +280,7 @@ Value *LiveRegOptimizer::convertToOptType(Instruction *V,
ShuffleMask.push_back(I);
for (uint64_t I = OriginalElementCount; I < ExpandedVecElementCount; I++)
- ShuffleMask.push_back(OriginalElementCount);
+ ShuffleMask.push_back(static_cast<int>(OriginalElementCount));
Value *ExpandedVec = Builder.CreateShuffleVector(V, ShuffleMask);
return Builder.CreateBitCast(ExpandedVec, NewTy, V->getName() + ".bc");
@@ -301,8 +304,8 @@ Value *LiveRegOptimizer::convertFromOptType(Type *ConvertType, Instruction *V,
assert(OriginalSize > NewSize);
// For wide scalars, we can just truncate the value.
if (!V->getType()->isVectorTy()) {
- Instruction *Trunc = cast<Instruction>(
- Builder.CreateTrunc(V, IntegerType::get(Mod.getContext(), NewSize)));
+ Instruction *Trunc = cast<Instruction>(Builder.CreateTrunc(
+ V, IntegerType::get(Mod.getContext(), static_cast<unsigned>(NewSize))));
return cast<Instruction>(Builder.CreateBitCast(Trunc, NewVTy));
}
@@ -310,7 +313,8 @@ Value *LiveRegOptimizer::convertFromOptType(Type *ConvertType, Instruction *V,
// type.
VectorType *ExpandedVT = VectorType::get(
Type::getIntNTy(Mod.getContext(), NewVTy->getScalarSizeInBits()),
- (OriginalSize / NewVTy->getScalarSizeInBits()), false);
+ static_cast<unsigned>(OriginalSize / NewVTy->getScalarSizeInBits()),
+ false);
Instruction *Converted =
cast<Instruction>(Builder.CreateBitCast(V, ExpandedVT));
@@ -473,7 +477,7 @@ bool LiveRegOptimizer::optimizeLiveType(
}
}
assert(NewVal);
- U->setOperand(OpIdx, NewVal);
+ U->setOperand(static_cast<unsigned>(OpIdx), NewVal);
}
}
}
@@ -494,7 +498,7 @@ bool AMDGPULateCodeGenPrepare::canWidenScalarExtLoad(LoadInst &LI) const {
// Skip aggregate types.
if (Ty->isAggregateType())
return false;
- unsigned TySize = DL.getTypeStoreSize(Ty);
+ unsigned TySize = static_cast<unsigned>(DL.getTypeStoreSize(Ty));
// Only handle sub-DWORD loads.
if (TySize >= 4)
return false;
@@ -529,7 +533,8 @@ bool AMDGPULateCodeGenPrepare::visitLoadInst(LoadInst &LI) {
IRBuilder<> IRB(&LI);
IRB.SetCurrentDebugLocation(LI.getDebugLoc());
- unsigned LdBits = DL.getTypeStoreSizeInBits(LI.getType());
+ unsigned LdBits =
+ static_cast<unsigned>(DL.getTypeStoreSizeInBits(LI.getType()));
auto *IntNTy = Type::getIntNTy(LI.getContext(), LdBits);
auto *NewPtr = IRB.CreateConstGEP1_64(
@@ -540,7 +545,7 @@ bool AMDGPULateCodeGenPrepare::visitLoadInst(LoadInst &LI) {
LoadInst *NewLd = IRB.CreateAlignedLoad(IRB.getInt32Ty(), NewPtr, Align(4));
AMDGPU::copyMetadataForWidenedLoad(*NewLd, LI);
- unsigned ShAmt = Adjust * 8;
+ unsigned ShAmt = static_cast<unsigned>(Adjust * 8);
Value *Shifted = ShAmt ? IRB.CreateLShr(NewLd, ShAmt) : NewLd;
Value *NewVal = IRB.CreateBitCast(
IRB.CreateTrunc(Shifted, DL.typeSizeEqualsStoreSize(LI.getType())
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerBufferFatPointers.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerBufferFatPointers.cpp
index a00f3372c1f60e..cb4ff0427c464a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerBufferFatPointers.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerBufferFatPointers.cpp
@@ -503,7 +503,7 @@ void StoreFatPtrsAsIntsAndExpandMemcpyVisitor::forEachAggLeaf(
const StructLayout *Layout = DL.getStructLayout(ST);
for (auto [I, ElemTy, ElemOff] :
enumerate(ST->elements(), Layout->getMemberOffsets()))
- Recurse(I, ElemTy, ElemOff.getFixedValue());
+ Recurse(static_cast<unsigned>(I), ElemTy, ElemOff.getFixedValue());
return;
}
auto *AT = cast<ArrayType>(Ty);
@@ -850,7 +850,7 @@ Type *LegalizeBufferContentTypesVisitor::scalarArrayTypeAsVector(Type *T) {
if (!DL.typeSizeEqualsStoreSize(AT))
reportFatalUsageError(
"loading padded arrays from buffer fat pinters should have recursed");
- return FixedVectorType::get(ET, AT->getNumElements());
+ return FixedVectorType::get(ET, static_cast<unsigned>(AT->getNumElements()));
}
Value *LegalizeBufferContentTypesVisitor::arrayToVector(Value *V,
@@ -872,7 +872,7 @@ Value *LegalizeBufferContentTypesVisitor::vectorToArray(Value *V,
const Twine &Name) {
Value *ArrayRes = PoisonValue::get(OrigType);
ArrayType *AT = cast<ArrayType>(OrigType);
- unsigned EC = AT->getNumElements();
+ unsigned EC = static_cast<unsigned>(AT->getNumElements());
for (auto I : iota_range<unsigned>(0, EC, /*Inclusive=*/false)) {
Value *Elem = IRB.CreateExtractElement(V, I, Name + ".elem." + Twine(I));
ArrayRes = IRB.CreateInsertValue(ArrayRes, Elem, I,
@@ -895,7 +895,8 @@ LegalizeBufferContentTypesVisitor::analyzeOobProperties(Value *Ptr, Type *Ty,
return Result;
const SCEV *PtrOp = SE->getSCEV(Ptr);
if (ByteOffset > 0)
- PtrOp = SE->getAddExpr(PtrOp, SE->getConstant(IRB.getInt32(ByteOffset)));
+ PtrOp = SE->getAddExpr(PtrOp, SE->getConstant(IRB.getInt32(
+ static_cast<uint32_t>(ByteOffset))));
const auto *PtrBase = dyn_cast<SCEVUnknown>(SE->getPointerBase(PtrOp));
if (!PtrBase)
return Result;
@@ -910,7 +911,8 @@ LegalizeBufferContentTypesVisitor::analyzeOobProperties(Value *Ptr, Type *Ty,
if (NumRecordsIfKnown == ZeroBasePointerToNumRecords.end())
return Result;
- unsigned TypeSize = DL.getTypeStoreSize(Ty).getKnownMinValue();
+ unsigned TypeSize =
+ static_cast<unsigned>(DL.getTypeStoreSize(Ty).getKnownMinValue());
const SCEV *PtrDiff = SE->getMinusSCEV(PtrOp, PtrBase);
APInt MaxNoWrapOffset = APInt::getAllOnes(BufferOffsetWidth) - TypeSize;
if (SE->isKnownNonNegative(PtrDiff) ||
@@ -997,14 +999,15 @@ Type *LegalizeBufferContentTypesVisitor::legalNonAggregateForMemOp(
TypeSize Size = DL.getTypeStoreSizeInBits(T);
// Implicitly zero-extend to the next byte if needed.
if (!DL.typeSizeEqualsStoreSize(T))
- T = IRB.getIntNTy(Size.getFixedValue());
+ T = IRB.getIntNTy(static_cast<unsigned>(Size.getFixedValue()));
Type *ElemTy = T->getScalarType();
if (isa<PointerType, ScalableVectorType>(ElemTy)) {
// Pointers are always big enough, and we'll let scalable vectors through to
// fail in codegen.
return T;
}
- unsigned ElemSize = DL.getTypeSizeInBits(ElemTy).getFixedValue();
+ unsigned ElemSize =
+ static_cast<unsigned>(DL.getTypeSizeInBits(ElemTy).getFixedValue());
if (isPowerOf2_32(ElemSize) && ElemSize >= 16 && ElemSize <= MaxWidth) {
// [vectors of] anything that's 16/32/64/128 bits can be cast and split into
// legal buffer operations, except that we might need to cut them into
@@ -1018,8 +1021,8 @@ Type *LegalizeBufferContentTypesVisitor::legalNonAggregateForMemOp(
BestVectorElemType = IRB.getInt16Ty();
else
BestVectorElemType = IRB.getInt8Ty();
- unsigned NumCastElems =
- Size.getFixedValue() / BestVectorElemType->getIntegerBitWidth();
+ unsigned NumCastElems = static_cast<unsigned>(
+ Size.getFixedValue() / BestVectorElemType->getIntegerBitWidth());
if (NumCastElems == 1)
return BestVectorElemType;
return FixedVectorType::get(BestVectorElemType, NumCastElems);
@@ -1031,8 +1034,10 @@ Value *LegalizeBufferContentTypesVisitor::makeLegalNonAggregate(
TypeSize SourceSize = DL.getTypeSizeInBits(SourceType);
TypeSize TargetSize = DL.getTypeSizeInBits(TargetType);
if (SourceSize != TargetSize) {
- Type *ShortScalarTy = IRB.getIntNTy(SourceSize.getFixedValue());
- Type *ByteScalarTy = IRB.getIntNTy(TargetSize.getFixedValue());
+ Type *ShortScalarTy =
+ IRB.getIntNTy(static_cast<unsigned>(SourceSize.getFixedValue()));
+ Type *ByteScalarTy =
+ IRB.getIntNTy(static_cast<unsigned>(TargetSize.getFixedValue()));
Value *AsScalar = IRB.CreateBitCast(V, ShortScalarTy, Name + ".as.scalar");
Value *Zext = IRB.CreateZExt(AsScalar, ByteScalarTy, Name + ".zext");
V = Zext;
@@ -1047,8 +1052,10 @@ Value *LegalizeBufferContentTypesVisitor::makeIllegalNonAggregate(
TypeSize LegalSize = DL.getTypeSizeInBits(LegalType);
TypeSize OrigSize = DL.getTypeSizeInBits(OrigType);
if (LegalSize != OrigSize) {
- Type *ShortScalarTy = IRB.getIntNTy(OrigSize.getFixedValue());
- Type *ByteScalarTy = IRB.getIntNTy(LegalSize.getFixedValue());
+ Type *ShortScalarTy =
+ IRB.getIntNTy(static_cast<unsigned>(OrigSize.getFixedValue()));
+ Type *ByteScalarTy =
+ IRB.getIntNTy(static_cast<unsigned>(LegalSize.getFixedValue()));
Value *AsScalar = IRB.CreateBitCast(V, ByteScalarTy, Name + ".bytes.cast");
Value *Trunc = IRB.CreateTrunc(AsScalar, ShortScalarTy, Name + ".trunc");
return IRB.CreateBitCast(Trunc, OrigType, Name + ".orig");
@@ -1118,9 +1125,12 @@ void LegalizeBufferContentTypesVisitor::getVecSlices(
return false;
};
while (Index < TotalElems) {
- TrySlice(ElemsPer4Words, 128) || TrySlice(ElemsPer3Words, 96) ||
- TrySlice(ElemsPer2Words, 64) || TrySlice(ElemsPerWord, 32) ||
- TrySlice(ElemsPerShort, 16) || TrySlice(ElemsPerByte, 8);
+ TrySlice(static_cast<unsigned>(ElemsPer4Words), 128) ||
+ TrySlice(static_cast<unsigned>(ElemsPer3Words), 96) ||
+ TrySlice(static_cast<unsigned>(ElemsPer2Words), 64) ||
+ TrySlice(static_cast<unsigned>(ElemsPerWord), 32) ||
+ TrySlice(static_cast<unsigned>(ElemsPerShort), 16) ||
+ TrySlice(static_cast<unsigned>(ElemsPerByte), 8);
}
}
@@ -1134,8 +1144,9 @@ Value *LegalizeBufferContentTypesVisitor::extractSlice(Value *Vec, VecSlice S,
if (S.Length == 1)
return IRB.CreateExtractElement(Vec, S.Index,
Name + ".slice." + Twine(S.Index));
- SmallVector<int> Mask = llvm::to_vector(
- llvm::iota_range<int>(S.Index, S.Index + S.Length, /*Inclusive=*/false));
+ SmallVector<int> Mask = llvm::to_vector(llvm::iota_range<int>(
+ static_cast<int>(S.Index), static_cast<int>(S.Index + S.Length),
+ /*Inclusive=*/false));
return IRB.CreateShuffleVector(Vec, Mask, Name + ".slice." + Twine(S.Index));
}
@@ -1157,7 +1168,7 @@ Value *LegalizeBufferContentTypesVisitor::insertSlice(Value *Whole, Value *Part,
SmallVector<int> ExtPartMask(NumElems, -1);
for (auto [I, E] : llvm::enumerate(
MutableArrayRef<int>(ExtPartMask).take_front(S.Length))) {
- E = I;
+ E = static_cast<int>(I);
}
Value *ExtPart = IRB.CreateShuffleVector(Part, ExtPartMask,
Name + ".ext." + Twine(S.Index));
@@ -1166,7 +1177,7 @@ Value *LegalizeBufferContentTypesVisitor::insertSlice(Value *Whole, Value *Part,
llvm::to_vector(llvm::iota_range<int>(0, NumElems, /*Inclusive=*/false));
for (auto [I, E] :
llvm::enumerate(MutableArrayRef<int>(Mask).slice(S.Index, S.Length)))
- E = I + NumElems;
+ E = static_cast<int>(I + NumElems);
return IRB.CreateShuffleVector(Whole, ExtPart, Mask,
Name + ".parts." + Twine(S.Index));
}
@@ -1179,7 +1190,7 @@ bool LegalizeBufferContentTypesVisitor::visitLoadImpl(
bool Changed = false;
for (auto [I, ElemTy, Offset] :
llvm::enumerate(ST->elements(), Layout->getMemberOffsets())) {
- AggIdxs.push_back(I);
+ AggIdxs.push_back(static_cast<unsigned>(I));
Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
AggByteOff + Offset.getFixedValue(), Result,
Name + "." + Twine(I));
@@ -1193,8 +1204,9 @@ bool LegalizeBufferContentTypesVisitor::visitLoadImpl(
ElemTy->isVectorTy()) {
TypeSize ElemAllocSize = DL.getTypeAllocSize(ElemTy);
bool Changed = false;
- for (auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
- /*Inclusive=*/false)) {
+ for (auto I : llvm::iota_range<uint32_t>(
+ 0, static_cast<uint32_t>(AT->getNumElements()),
+ /*Inclusive=*/false)) {
AggIdxs.push_back(I);
Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
AggByteOff + I * ElemAllocSize.getFixedValue(),
@@ -1240,17 +1252,20 @@ bool LegalizeBufferContentTypesVisitor::visitLoadImpl(
// But if we're already a scalar (which can happen if we're splitting up a
// struct), the element type will be the legal type itself.
Type *ElemType = LegalType->getScalarType();
- unsigned ElemBytes = DL.getTypeStoreSize(ElemType);
+ unsigned ElemBytes = static_cast<unsigned>(DL.getTypeStoreSize(ElemType));
AAMDNodes AANodes = OrigLI.getAAMetadata();
if (IsAggPart && Slices.empty())
Slices.push_back(VecSlice{/*Index=*/0, /*Length=*/1});
for (VecSlice S : Slices) {
Type *SliceType =
- S.Length != 1 ? FixedVectorType::get(ElemType, S.Length) : ElemType;
+ S.Length != 1
+ ? FixedVectorType::get(ElemType, static_cast<unsigned>(S.Length))
+ : ElemType;
int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
// You can't reasonably expect loads to wrap around the edge of memory.
Value *NewPtr = IRB.CreateGEP(
- IRB.getInt8Ty(), OrigLI.getPointerOperand(), IRB.getInt32(ByteOffset),
+ IRB.getInt8Ty(), OrigLI.getPointerOperand(),
+ IRB.getInt32(static_cast<uint32_t>(ByteOffset)),
OrigPtr->getName() + ".off.ptr." + Twine(ByteOffset),
ST->hasRelaxedBufferOOBMode() ? GEPNoWrapFlags::noUnsignedWrap()
: GEPNoWrapFlags::none());
@@ -1304,7 +1319,7 @@ std::pair<bool, bool> LegalizeBufferContentTypesVisitor::visitStoreImpl(
bool Changed = false;
for (auto [I, ElemTy, Offset] :
llvm::enumerate(ST->elements(), Layout->getMemberOffsets())) {
- AggIdxs.push_back(I);
+ AggIdxs.push_back(static_cast<unsigned>(I));
Changed |= std::get<0>(visitStoreImpl(OrigSI, ElemTy, AggIdxs,
AggByteOff + Offset.getFixedValue(),
Name + "." + Twine(I)));
@@ -1318,8 +1333,9 @@ std::pair<bool, bool> LegalizeBufferContentTypesVisitor::visitStoreImpl(
ElemTy->isVectorTy()) {
TypeSize ElemAllocSize = DL.getTypeAllocSize(ElemTy);
bool Changed = false;
- for (auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
- /*Inclusive=*/false)) {
+ for (auto I : llvm::iota_range<uint32_t>(
+ 0, static_cast<uint32_t>(AT->getNumElements()),
+ /*Inclusive=*/false)) {
AggIdxs.push_back(I);
Changed |= std::get<0>(visitStoreImpl(
OrigSI, ElemTy, AggIdxs,
@@ -1367,14 +1383,17 @@ std::pair<bool, bool> LegalizeBufferContentTypesVisitor::visitStoreImpl(
Type *ElemType = LegalType->getScalarType();
if (IsAggPart && Slices.empty())
Slices.push_back(VecSlice{/*Index=*/0, /*Length=*/1});
- unsigned ElemBytes = DL.getTypeStoreSize(ElemType);
+ unsigned ElemBytes = static_cast<unsigned>(DL.getTypeStoreSize(ElemType));
AAMDNodes AANodes = OrigSI.getAAMetadata();
for (VecSlice S : Slices) {
Type *SliceType =
- S.Length != 1 ? FixedVectorType::get(ElemType, S.Length) : ElemType;
+ S.Length != 1
+ ? FixedVectorType::get(ElemType, static_cast<unsigned>(S.Length))
+ : ElemType;
int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
Value *NewPtr = IRB.CreateGEP(
- IRB.getInt8Ty(), OrigPtr, IRB.getInt32(ByteOffset),
+ IRB.getInt8Ty(), OrigPtr,
+ IRB.getInt32(static_cast<uint32_t>(ByteOffset)),
OrigPtr->getName() + ".part." + Twine(S.Index),
ST->hasRelaxedBufferOOBMode() ? GEPNoWrapFlags::noUnsignedWrap()
: GEPNoWrapFlags::none());
@@ -1856,10 +1875,11 @@ void SplitPtrStructs::killAndReplaceSplitInstructions(
std::optional<DIExpression *> RsrcExpr =
DIExpression::createFragmentExpression(Dbg->getExpression(), 0,
- RsrcSz);
+ static_cast<unsigned>(RsrcSz));
std::optional<DIExpression *> OffExpr =
- DIExpression::createFragmentExpression(Dbg->getExpression(), RsrcSz,
- OffSz);
+ DIExpression::createFragmentExpression(Dbg->getExpression(),
+ static_cast<unsigned>(RsrcSz),
+ static_cast<unsigned>(OffSz));
if (OffExpr) {
OffDbg->setExpression(*OffExpr);
OffDbg->replaceVariableLocationOp(I, Off);
@@ -2694,7 +2714,7 @@ static Function *moveFunctionAdaptingType(Function *OldF, FunctionType *NewTy,
OldArg.replaceAllUsesWith(&NewArg);
NewArg.mutateType(NewArgTy);
- AttributeSet ArgAttr = OldAttrs.getParamAttrs(I);
+ AttributeSet ArgAttr = OldAttrs.getParamAttrs(static_cast<unsigned>(I));
// Intrinsics get their attributes fixed later.
if (OldArgTy != NewArgTy && !IsIntrinsic)
ArgAttr = ArgAttr.removeAttributes(
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerExecSync.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerExecSync.cpp
index 1fad3489cf1a7f..8d483010e87ba5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerExecSync.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerExecSync.cpp
@@ -127,7 +127,7 @@ static bool lowerExecSyncGlobalVariables(Module &M, GVUsesInfoTy &GVUsesInfo) {
unsigned Offset;
if (TargetExtType *ExtTy = isNamedBarrier(*GV)) {
unsigned BarrierScope = ExtTy->getIntParameter(0);
- unsigned BarCnt = GV->getGlobalSize(DL) / 16;
+ unsigned BarCnt = static_cast<unsigned>(GV->getGlobalSize(DL) / 16);
unsigned BarID = allocateExecSyncID(KernelBarrierIDs, Kernels,
BarrierScope, NumBarScopes, BarCnt);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
index 6150e14f38c7df..c8f2413397b67d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerIntrinsics.cpp
@@ -159,7 +159,8 @@ bool AMDGPULowerIntrinsicsImpl::visitBarrier(IntrinsicInst &I) {
if (I.getIntrinsicID() == Intrinsic::amdgcn_s_barrier_wait ||
I.getIntrinsicID() == Intrinsic::amdgcn_s_barrier_signal ||
I.getIntrinsicID() == Intrinsic::amdgcn_s_barrier_signal_isfirst) {
- int BarrierID = cast<ConstantInt>(I.getArgOperand(0))->getSExtValue();
+ int BarrierID =
+ static_cast<int>(cast<ConstantInt>(I.getArgOperand(0))->getSExtValue());
if (BarrierID == AMDGPU::Barrier::TRAP ||
BarrierID == AMDGPU::Barrier::WORKGROUP ||
(BarrierID >= AMDGPU::Barrier::NAMED_BARRIER_FIRST &&
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerKernelArguments.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerKernelArguments.cpp
index f0d1f90ca5c2ca..3078f3a7d6aeee 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerKernelArguments.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerKernelArguments.cpp
@@ -369,7 +369,7 @@ static bool lowerKernelArguments(Function &F, const TargetMachine &TM,
Value *ExtractBits = OffsetDiff == 0 ?
Load : Builder.CreateLShr(Load, OffsetDiff * 8);
- IntegerType *ArgIntTy = Builder.getIntNTy(Size);
+ IntegerType *ArgIntTy = Builder.getIntNTy(static_cast<unsigned>(Size));
Value *Trunc = Builder.CreateTrunc(ExtractBits, ArgIntTy);
Value *NewVal = Builder.CreateBitCast(Trunc, ArgTy,
Arg.getName() + ".load");
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerKernelAttributes.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerKernelAttributes.cpp
index aa3d2d06024ccd..1244d7c16eef86 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerKernelAttributes.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerKernelAttributes.cpp
@@ -207,7 +207,8 @@ static bool processUse(CallInst *CI, bool IsV5OrAbove) {
if (!Load || !Load->isSimple())
continue;
- unsigned LoadSize = DL.getTypeStoreSize(Load->getType());
+ unsigned LoadSize =
+ static_cast<unsigned>(DL.getTypeStoreSize(Load->getType()));
// TODO: Handle merged loads.
if (IsV5OrAbove) { // Base is ImplicitArgPtr.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPULowerModuleLDSPass.cpp b/llvm/lib/Target/AMDGPU/AMDGPULowerModuleLDSPass.cpp
index def1c94f5d4129..331eacfbf33cf6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPULowerModuleLDSPass.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPULowerModuleLDSPass.cpp
@@ -565,7 +565,7 @@ class AMDGPULowerModuleLDS {
for (size_t i = 0; i < OrderedKernels.size(); i++) {
Metadata *AttrMDArgs[1] = {
- ConstantAsMetadata::get(Builder.getInt32(i)),
+ ConstantAsMetadata::get(Builder.getInt32(static_cast<uint32_t>(i))),
};
OrderedKernels[i]->setMetadata("llvm.amdgcn.lds.kernel.id",
MDNode::get(Ctx, AttrMDArgs));
@@ -1202,14 +1202,16 @@ class AMDGPULowerModuleLDS {
if (AllocateModuleScopeStruct) {
// Allocated at zero, recorded once on construction, not once per
// kernel
- Offset += MaybeModuleScopeStruct->getGlobalSize(DL);
+ Offset +=
+ static_cast<uint32_t>(MaybeModuleScopeStruct->getGlobalSize(DL));
}
if (AllocateKernelScopeStruct) {
GlobalVariable *KernelStruct = Replacement->second.SGV;
- Offset = alignTo(Offset, AMDGPU::getAlign(DL, KernelStruct));
+ Offset = static_cast<uint32_t>(
+ alignTo(Offset, AMDGPU::getAlign(DL, KernelStruct)));
recordLDSAbsoluteAddress(&M, KernelStruct, Offset);
- Offset += KernelStruct->getGlobalSize(DL);
+ Offset += static_cast<uint32_t>(KernelStruct->getGlobalSize(DL));
}
// If there is dynamic allocation, the alignment needed is included in
@@ -1218,7 +1220,8 @@ class AMDGPULowerModuleLDS {
// alignment padding could be missed.
if (AllocateDynamicVariable) {
GlobalVariable *DynamicVariable = KernelToCreatedDynamicLDS[&Func];
- Offset = alignTo(Offset, AMDGPU::getAlign(DL, DynamicVariable));
+ Offset = static_cast<uint32_t>(
+ alignTo(Offset, AMDGPU::getAlign(DL, DynamicVariable)));
recordLDSAbsoluteAddress(&M, DynamicVariable, Offset);
}
@@ -1395,7 +1398,7 @@ class AMDGPULowerModuleLDS {
Constant *GEPIdx[] = {ConstantInt::get(I32, 0), ConstantInt::get(I32, I)};
Constant *GEP = ConstantExpr::getGetElementPtr(
DL, LDSTy, SGV, GEPIdx, GEPNoWrapFlags::inBounds());
- if (IsPaddingField[I]) {
+ if (IsPaddingField[static_cast<unsigned>(I)]) {
assert(GV->use_empty());
GV->eraseFromParent();
} else {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPreloadKernelArguments.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPreloadKernelArguments.cpp
index caad246b4ff7f1..237bca0901bdd5 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPreloadKernelArguments.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPreloadKernelArguments.cpp
@@ -216,11 +216,12 @@ class PreloadKernelArgInfo {
// FIXME: Expand handle merged loads.
LLVMContext &Ctx = F.getContext();
Type *LoadTy = Load->getType();
- HiddenArg HA = getHiddenArgFromOffset(Offset);
+ HiddenArg HA = getHiddenArgFromOffset(static_cast<unsigned>(Offset));
if (HA == END_HIDDEN_ARGS || LoadTy != getHiddenArgType(Ctx, HA))
continue;
- ImplicitArgLoads.push_back(std::make_pair(Load, Offset));
+ ImplicitArgLoads.push_back(
+ std::make_pair(Load, static_cast<unsigned>(Offset)));
}
}
@@ -236,7 +237,8 @@ class PreloadKernelArgInfo {
// argument that we cannot preload.
auto *PreloadEnd = llvm::find_if(
ImplicitArgLoads, [&](const std::pair<LoadInst *, unsigned> &Load) {
- unsigned LoadSize = DL.getTypeStoreSize(Load.first->getType());
+ unsigned LoadSize =
+ static_cast<unsigned>(DL.getTypeStoreSize(Load.first->getType()));
unsigned LoadOffset = Load.second;
if (!canPreloadKernArgAtOffset(LoadOffset + LoadSize +
ImplicitArgsBaseOffset))
@@ -256,7 +258,8 @@ class PreloadKernelArgInfo {
LoadInst *LoadInst = I->first;
unsigned LoadOffset = I->second;
unsigned HiddenArgIndex = getHiddenArgFromOffset(LoadOffset);
- unsigned Index = NF->arg_size() - LastHiddenArgIndex + HiddenArgIndex - 1;
+ unsigned Index = static_cast<unsigned>(
+ NF->arg_size() - LastHiddenArgIndex + HiddenArgIndex - 1);
Argument *Arg = NF->getArg(Index);
LoadInst->replaceAllUsesWith(Arg);
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPrintfRuntimeBinding.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPrintfRuntimeBinding.cpp
index 98928e26f8de01..ac233e77335c4a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPrintfRuntimeBinding.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPrintfRuntimeBinding.cpp
@@ -176,7 +176,7 @@ bool AMDGPUPrintfRuntimeBindingImpl::lowerPrintfForGpu(Module &M) {
ArgCount++) {
Value *Arg = CI->getArgOperand(ArgCount);
Type *ArgType = Arg->getType();
- unsigned ArgSize = TD->getTypeAllocSize(ArgType);
+ unsigned ArgSize = static_cast<unsigned>(TD->getTypeAllocSize(ArgType));
//
// ArgSize by design should be a multiple of DWORD_ALIGN,
// expand the arguments that do not follow this rule.
@@ -197,7 +197,7 @@ bool AMDGPUPrintfRuntimeBindingImpl::lowerPrintfForGpu(Module &M) {
else
Arg = Builder.CreateSExt(Arg, ResType);
ArgType = Arg->getType();
- ArgSize = TD->getTypeAllocSize(ArgType);
+ ArgSize = static_cast<unsigned>(TD->getTypeAllocSize(ArgType));
CI->setOperand(ArgCount, Arg);
}
if (OpConvSpecifiers[ArgCount - 1] == 'f') {
@@ -215,7 +215,8 @@ bool AMDGPUPrintfRuntimeBindingImpl::lowerPrintfForGpu(Module &M) {
// Match the store loop below: ceil(len / 4) dwords, or one dword
// for the empty string. The trailing NUL is not stored.
StringRef Str = getAsConstantStr(Arg);
- ArgSize = Str.empty() ? 4 : alignTo(Str.size(), 4);
+ ArgSize =
+ static_cast<unsigned>(Str.empty() ? 4 : alignTo(Str.size(), 4));
}
LLVM_DEBUG(dbgs() << "Printf ArgSize (in buffer) = " << ArgSize
@@ -397,9 +398,11 @@ bool AMDGPUPrintfRuntimeBindingImpl::lowerPrintfForGpu(Module &M) {
} else {
WhatToStore.push_back(Arg);
}
- for (unsigned I = 0, E = WhatToStore.size(); I != E; ++I) {
+ for (unsigned I = 0, E = static_cast<unsigned>(WhatToStore.size());
+ I != E; ++I) {
Value *TheBtCast = WhatToStore[I];
- unsigned ArgSize = TD->getTypeAllocSize(TheBtCast->getType());
+ unsigned ArgSize =
+ static_cast<unsigned>(TD->getTypeAllocSize(TheBtCast->getType()));
StoreInst *StBuff = new StoreInst(TheBtCast, BufferIdx, BrnchPoint);
LLVM_DEBUG(dbgs() << "inserting store to printf buffer:\n"
<< *StBuff << '\n');
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
index aefcabd0d3bf17..c5d29283a7031b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUPromoteAlloca.cpp
@@ -350,14 +350,14 @@ void AMDGPUPromoteAllocaImpl::setFunctionLimits(const Function &F) {
// R600 register tuples/aliasing are fragile with large vector promotions so
// apply architecture specific limit here.
const int R600MaxVectorRegs = 16;
- MaxVectorRegs = F.getFnAttributeAsParsedInteger(
+ MaxVectorRegs = static_cast<unsigned>(F.getFnAttributeAsParsedInteger(
"amdgpu-promote-alloca-to-vector-max-regs",
- IsAMDGCN ? PromoteAllocaToVectorMaxRegs : R600MaxVectorRegs);
+ IsAMDGCN ? PromoteAllocaToVectorMaxRegs : R600MaxVectorRegs));
if (PromoteAllocaToVectorMaxRegs.getNumOccurrences())
MaxVectorRegs = PromoteAllocaToVectorMaxRegs;
- VGPRBudgetRatio = F.getFnAttributeAsParsedInteger(
+ VGPRBudgetRatio = static_cast<unsigned>(F.getFnAttributeAsParsedInteger(
"amdgpu-promote-alloca-to-vector-vgpr-ratio",
- PromoteAllocaToVectorVGPRRatio);
+ PromoteAllocaToVectorVGPRRatio));
if (PromoteAllocaToVectorVGPRRatio.getNumOccurrences())
VGPRBudgetRatio = PromoteAllocaToVectorVGPRRatio;
}
@@ -415,7 +415,8 @@ bool AMDGPUPromoteAllocaImpl::run(Function &F, bool PromoteToLDS) {
if (AA.Vector.Ty) {
std::optional<TypeSize> Size = AA.Alloca->getAllocationSize(DL);
assert(Size); // Expected to succeed on non-array alloca.
- const unsigned AllocaCost = Size->getFixedValue() * 8;
+ const unsigned AllocaCost =
+ static_cast<unsigned>(Size->getFixedValue() * 8);
// First, check if we have enough budget to vectorize this alloca.
if (AllocaCost <= VectorizationBudget) {
promoteAllocaToVector(AA);
@@ -458,7 +459,8 @@ static bool isSupportedMemset(MemSetInst *I, AllocaInst *AI,
// TODO: Now that we moved to PromoteAlloca we could handle any memsets
// (except maybe volatile ones?) - we just need to use shufflevector if it
// only affects a subset of the vector.
- const unsigned Size = DL.getTypeStoreSize(AI->getAllocatedType());
+ const unsigned Size =
+ static_cast<unsigned>(DL.getTypeStoreSize(AI->getAllocatedType()));
return I->getOperand(0) == AI &&
match(I->getOperand(2), m_SpecificInt(Size)) && !I->isVolatile();
}
@@ -656,7 +658,7 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
TypeSize EltSize = DL.getTypeStoreSize(VecEltTy);
assert(AccessSize.isKnownMultipleOf(EltSize) &&
"promotable access must cover a whole number of elements");
- const unsigned NumLoadedElts = AccessSize / EltSize;
+ const unsigned NumLoadedElts = static_cast<unsigned>(AccessSize / EltSize);
if (NumLoadedElts > 1) {
auto *SubVecTy = FixedVectorType::get(VecEltTy, NumLoadedElts);
assert(DL.getTypeStoreSize(SubVecTy) == DL.getTypeStoreSize(AccessTy));
@@ -676,9 +678,10 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
if (!isa<ConstantInt>(Index) &&
llvm::isPowerOf2_32(SubVecTy->getNumElements()) &&
IsProperlyDivisible && IsAlignedLoad) {
- IntegerType *NewElemTy = Builder.getIntNTy(NumBits);
+ IntegerType *NewElemTy =
+ Builder.getIntNTy(static_cast<unsigned>(NumBits));
const unsigned NewNumElts =
- DL.getTypeStoreSize(VectorTy) * 8u / NumBits;
+ static_cast<unsigned>(DL.getTypeStoreSize(VectorTy) * 8u / NumBits);
const unsigned LShrAmt = llvm::Log2_32(SubVecTy->getNumElements());
FixedVectorType *BitCastTy =
FixedVectorType::get(NewElemTy, NewNumElts);
@@ -745,7 +748,7 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
TypeSize EltSize = DL.getTypeStoreSize(VecEltTy);
assert(AccessSize.isKnownMultipleOf(EltSize) &&
"promotable access must cover a whole number of elements");
- const unsigned NumWrittenElts = AccessSize / EltSize;
+ const unsigned NumWrittenElts = static_cast<unsigned>(AccessSize / EltSize);
if (NumWrittenElts > 1) {
const unsigned NumVecElts = AA.Vector.Ty->getNumElements();
auto *SubVecTy = FixedVectorType::get(VecEltTy, NumWrittenElts);
@@ -771,10 +774,11 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
if (auto *MTI = dyn_cast<MemTransferInst>(Inst)) {
// For memcpy, we need to know curval.
ConstantInt *Length = cast<ConstantInt>(MTI->getLength());
- unsigned NumCopied = Length->getZExtValue() / ElementSize;
+ unsigned NumCopied =
+ static_cast<unsigned>(Length->getZExtValue() / ElementSize);
MemTransferInfo *TI = &AA.Vector.TransferInfo[MTI];
- unsigned SrcBegin = TI->SrcIndex->getZExtValue();
- unsigned DestBegin = TI->DestIndex->getZExtValue();
+ unsigned SrcBegin = static_cast<unsigned>(TI->SrcIndex->getZExtValue());
+ unsigned DestBegin = static_cast<unsigned>(TI->DestIndex->getZExtValue());
SmallVector<int> Mask;
for (unsigned Idx = 0; Idx < AA.Vector.Ty->getNumElements(); ++Idx) {
@@ -794,7 +798,8 @@ static Value *promoteAllocaUserToVector(Instruction *Inst, const DataLayout &DL,
// For memset, we don't need to know the previous value because we
// currently only allow memsets that cover the whole alloca.
Value *Elt = MSI->getOperand(1);
- const unsigned BytesPerElt = DL.getTypeStoreSize(VecEltTy);
+ const unsigned BytesPerElt =
+ static_cast<unsigned>(DL.getTypeStoreSize(VecEltTy));
if (BytesPerElt > 1) {
Value *EltBytes = Builder.CreateVectorSplat(BytesPerElt, Elt);
@@ -936,9 +941,11 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
if (VectorType::isValidElementType(ElemTy) && NumElems > 0) {
- unsigned ElementSize = DL.getTypeSizeInBits(ElemTy) / 8;
+ unsigned ElementSize =
+ static_cast<unsigned>(DL.getTypeSizeInBits(ElemTy) / 8);
if (ElementSize > 0) {
- unsigned AllocaSize = DL.getTypeStoreSize(AllocaTy);
+ unsigned AllocaSize =
+ static_cast<unsigned>(DL.getTypeStoreSize(AllocaTy));
// Expand vector if required to match padding of inner type,
// i.e. odd size subvectors.
// Storage size of new vector must match that of alloca for correct
@@ -946,7 +953,8 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
if (NumElems * ElementSize != AllocaSize)
NumElems = AllocaSize / ElementSize;
if (NumElems > 0 && (AllocaSize % ElementSize) == 0)
- VectorTy = FixedVectorType::get(ElemTy, NumElems);
+ VectorTy =
+ FixedVectorType::get(ElemTy, static_cast<unsigned>(NumElems));
}
}
}
@@ -955,8 +963,8 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
return nullptr;
}
- const unsigned MaxElements =
- (MaxVectorRegs * 32) / DL.getTypeSizeInBits(VectorTy->getElementType());
+ const unsigned MaxElements = static_cast<unsigned>(
+ (MaxVectorRegs * 32) / DL.getTypeSizeInBits(VectorTy->getElementType()));
if (VectorTy->getNumElements() > MaxElements ||
VectorTy->getNumElements() < 2) {
@@ -966,7 +974,8 @@ AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(Type *AllocaTy) const {
}
Type *VecEltTy = VectorTy->getElementType();
- unsigned ElementSizeInBits = DL.getTypeSizeInBits(VecEltTy);
+ unsigned ElementSizeInBits =
+ static_cast<unsigned>(DL.getTypeSizeInBits(VecEltTy));
if (ElementSizeInBits != DL.getTypeAllocSizeInBits(VecEltTy)) {
LLVM_DEBUG(dbgs() << " Cannot convert to vector if the allocation size "
"does not match the type's size\n");
@@ -994,7 +1003,8 @@ void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &AA) const {
};
Type *VecEltTy = AA.Vector.Ty->getElementType();
- unsigned ElementSize = DL.getTypeSizeInBits(VecEltTy) / 8;
+ unsigned ElementSize =
+ static_cast<unsigned>(DL.getTypeSizeInBits(VecEltTy) / 8);
assert(ElementSize > 0);
for (auto *U : AA.Uses) {
Instruction *Inst = cast<Instruction>(U->getUser());
@@ -1130,10 +1140,12 @@ void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &AA) {
LLVM_DEBUG(dbgs() << "Promoting to vectors: " << *AA.Alloca << '\n');
LLVM_DEBUG(dbgs() << " type conversion: " << *AA.Alloca->getAllocatedType()
<< " -> " << *AA.Vector.Ty << '\n');
- const unsigned VecStoreSize = DL.getTypeStoreSize(AA.Vector.Ty);
+ const unsigned VecStoreSize =
+ static_cast<unsigned>(DL.getTypeStoreSize(AA.Vector.Ty));
Type *VecEltTy = AA.Vector.Ty->getElementType();
- const unsigned ElementSize = DL.getTypeSizeInBits(VecEltTy) / 8;
+ const unsigned ElementSize =
+ static_cast<unsigned>(DL.getTypeSizeInBits(VecEltTy) / 8);
// Alloca is uninitialized memory. Imitate that by making the first value
// undef.
@@ -1576,8 +1588,9 @@ bool AMDGPUPromoteAllocaImpl::hasSufficientLocalMem(const Function &F) {
// legalizing, which could also potentially change. We try to estimate the
// worst case here, but we probably should fix the addresses earlier.
for (auto Alloc : AllocatedSizes) {
- CurrentLocalMemUsage = alignTo(CurrentLocalMemUsage, Alloc.second);
- CurrentLocalMemUsage += Alloc.first;
+ CurrentLocalMemUsage =
+ static_cast<uint32_t>(alignTo(CurrentLocalMemUsage, Alloc.second));
+ CurrentLocalMemUsage += static_cast<uint32_t>(Alloc.first);
}
unsigned MaxOccupancy =
@@ -1628,12 +1641,13 @@ bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
// FIXME: It is also possible that if we're allowed to use all of the memory
// could end up using more than the maximum due to alignment padding.
- uint32_t NewSize = alignTo(CurrentLocalMemUsage, Alignment);
+ uint32_t NewSize =
+ static_cast<uint32_t>(alignTo(CurrentLocalMemUsage, Alignment));
std::optional<TypeSize> ElemSize = AA.Alloca->getAllocationSize(DL);
if (!ElemSize || ElemSize->isScalable())
return false;
TypeSize AllocSize = WorkGroupSize * *ElemSize;
- NewSize += AllocSize.getFixedValue();
+ NewSize += static_cast<uint32_t>(AllocSize.getFixedValue());
if (NewSize > LocalMemLimit) {
LLVM_DEBUG(dbgs() << " " << AllocSize
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
index 5e27f39072ebb6..c099478447615f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteAGPRCopyMFMA.cpp
@@ -216,7 +216,7 @@ bool AMDGPURewriteAGPRCopyMFMAImpl::recomputeRegClassExceptRewritable(
continue;
}
- unsigned OpNo = &MO - &MI->getOperand(0);
+ unsigned OpNo = static_cast<unsigned>(&MO - &MI->getOperand(0));
NewRC = MI->getRegClassConstraintEffect(OpNo, NewRC, &TII, &TRI);
if (!NewRC || NewRC == OldRC) {
LLVM_DEBUG(dbgs() << "User of " << printReg(Reg, &TRI)
@@ -397,8 +397,9 @@ bool AMDGPURewriteAGPRCopyMFMAImpl::tryFoldCopiesToAGPR(
for (MachineInstr &CopySrcDefMI : MRI.def_instructions(CopySrcReg)) {
if (isRewriteCandidate(CopySrcDefMI) &&
- tryReassigningMFMAChain(
- CopySrcDefMI, CopySrcDefMI.getOperand(0).getReg(), AssignedAGPR))
+ tryReassigningMFMAChain(CopySrcDefMI,
+ CopySrcDefMI.getOperand(0).getReg(),
+ static_cast<MCPhysReg>(AssignedAGPR)))
MadeChange = true;
}
}
@@ -429,8 +430,9 @@ bool AMDGPURewriteAGPRCopyMFMAImpl::tryFoldCopiesFromAGPR(
MachineInstr &CopyUseMI = *CopyUseMO.getParent();
if (isRewriteCandidate(CopyUseMI)) {
- if (tryReassigningMFMAChain(CopyUseMI, CopyDstReg,
- VRM.getPhys(CopyDstReg)))
+ if (tryReassigningMFMAChain(
+ CopyUseMI, CopyDstReg,
+ static_cast<MCPhysReg>(VRM.getPhys(CopyDstReg))))
MadeChange = true;
}
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPURewriteOutArguments.cpp b/llvm/lib/Target/AMDGPU/AMDGPURewriteOutArguments.cpp
index 1b63467844a246..fdaf7721d5c19f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPURewriteOutArguments.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPURewriteOutArguments.cpp
@@ -234,7 +234,7 @@ bool AMDGPURewriteOutArguments::runOnFunction(Function &F) {
SmallVector<Type *, 4> ReturnTypes;
Type *RetTy = F.getReturnType();
if (!RetTy->isVoidTy()) {
- ReturnNumRegs = DL->getTypeStoreSize(RetTy) / 4;
+ ReturnNumRegs = static_cast<unsigned>(DL->getTypeStoreSize(RetTy) / 4);
if (ReturnNumRegs >= MaxNumRetRegs)
return false;
@@ -297,7 +297,8 @@ bool AMDGPURewriteOutArguments::runOnFunction(Function &F) {
// TODO: This is an approximation. When legalized this could be more. We
// can ask TLI for exactly how many.
- unsigned ArgNumRegs = DL->getTypeStoreSize(ArgTy) / 4;
+ unsigned ArgNumRegs =
+ static_cast<unsigned>(DL->getTypeStoreSize(ArgTy) / 4);
if (ArgNumRegs + ReturnNumRegs > MaxNumRetRegs)
continue;
@@ -338,7 +339,8 @@ bool AMDGPURewriteOutArguments::runOnFunction(Function &F) {
}
if (ThisReplaceable) {
- OutArgIndexes.insert({OutArg->getArgNo(), ReturnTypes.size()});
+ OutArgIndexes.insert(
+ {OutArg->getArgNo(), static_cast<unsigned>(ReturnTypes.size())});
ReturnTypes.push_back(ArgTy);
++NumOutArgumentsReplaced;
Changing = true;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSplitModule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSplitModule.cpp
index d119711edb18d4..2aabc0e30bb32f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSplitModule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSplitModule.cpp
@@ -302,8 +302,10 @@ class SplitGraph {
iterator_range<nodes_iterator> nodes() const { return Nodes; }
const Node &getNode(unsigned ID) const { return *Nodes[ID]; }
- unsigned getNumNodes() const { return Nodes.size(); }
- BitVector createNodesBitVector() const { return BitVector(Nodes.size()); }
+ unsigned getNumNodes() const { return static_cast<unsigned>(Nodes.size()); }
+ BitVector createNodesBitVector() const {
+ return BitVector(static_cast<unsigned>(Nodes.size()));
+ }
const Module &getModule() const { return M; }
@@ -714,7 +716,8 @@ SplitGraph::getNode(DenseMap<const GlobalValue *, Node *> &Cache,
NonCopyable = isNonCopyable(*Fn);
Cost = CostMap.at(Fn);
}
- N = new (NodesPool.Allocate()) Node(Nodes.size(), GV, Cost, NonCopyable);
+ N = new (NodesPool.Allocate())
+ Node(static_cast<unsigned>(Nodes.size()), GV, Cost, NonCopyable);
Nodes.push_back(N);
assert(&getNode(N->getID()) == N);
return *N;
@@ -840,7 +843,7 @@ unsigned SplitProposal::findCheapestPartition() const {
unsigned CurPID = InvalidPID;
for (const auto &[Idx, Part] : enumerate(Partitions)) {
if (Part.first <= CurCost) {
- CurPID = Idx;
+ CurPID = static_cast<unsigned>(Idx);
CurCost = Part.first;
}
}
@@ -1521,10 +1524,13 @@ static void splitAMDGPUModule(
}
if (SummariesOS)
- printPartitionSummary(*SummariesOS, PID, *MPart, PartCost, ModuleCost);
+ printPartitionSummary(*SummariesOS, PID, *MPart,
+ static_cast<unsigned>(PartCost),
+ static_cast<unsigned>(ModuleCost));
- LLVM_DEBUG(
- printPartitionSummary(dbgs(), PID, *MPart, PartCost, ModuleCost));
+ LLVM_DEBUG(printPartitionSummary(dbgs(), PID, *MPart,
+ static_cast<unsigned>(PartCost),
+ static_cast<unsigned>(ModuleCost)));
ModuleCallback(std::move(MPart));
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
index b11329e5ef6bc6..407a2d268aa29f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSwLowerLDS.cpp
@@ -240,7 +240,7 @@ SetVector<Function *> AMDGPUSwLowerLDS::getOrderedIndirectLDSAccessingKernels(
sortByName(std::vector<Function *>(Kernels.begin(), Kernels.end()));
for (size_t i = 0; i < Kernels.size(); i++) {
Metadata *AttrMDArgs[1] = {
- ConstantAsMetadata::get(IRB.getInt32(i)),
+ ConstantAsMetadata::get(IRB.getInt32(static_cast<uint32_t>(i))),
};
Function *Func = OrderedKernels[i];
Func->setMetadata("llvm.amdgcn.lds.kernel.id",
@@ -429,18 +429,18 @@ void AMDGPUSwLowerLDS::populateSwMetadataGlobal(Function *Func) {
const uint64_t RightRedzoneSize =
AMDGPU::getRedzoneSizeForGlobal(AsanScale, SizeInBytes);
// Update MallocSize with current size and redzone size.
- MallocSize += SizeInBytes;
+ MallocSize += static_cast<uint32_t>(SizeInBytes);
if (!AMDGPU::isDynamicLDS(*GV))
- LDSParams.RedzoneOffsetAndSizeVector.emplace_back(MallocSize,
- RightRedzoneSize);
- MallocSize += RightRedzoneSize;
+ LDSParams.RedzoneOffsetAndSizeVector.emplace_back(
+ MallocSize, static_cast<uint32_t>(RightRedzoneSize));
+ MallocSize += static_cast<uint32_t>(RightRedzoneSize);
// Align current size plus redzone.
uint64_t AlignedSize =
alignTo(SizeInBytes + RightRedzoneSize, MaxAlignment);
Constant *AlignedSizeInBytesConst =
ConstantInt::get(Int32Ty, AlignedSize);
// Align MallocSize
- MallocSize = alignTo(MallocSize, MaxAlignment);
+ MallocSize = static_cast<uint32_t>(alignTo(MallocSize, MaxAlignment));
Constant *InitItem =
ConstantStruct::get(LDSItemTy, {ItemStartOffset, SizeInBytesConst,
AlignedSizeInBytesConst});
@@ -459,7 +459,7 @@ void AMDGPUSwLowerLDS::populateSwMetadataGlobal(Function *Func) {
Type *Ty = LDSParams.SwLDS->getValueType();
const uint64_t SizeInBytes = DL.getTypeAllocSize(Ty);
uint64_t AlignedSize = alignTo(SizeInBytes, MaxAlignment);
- LDSParams.LDSSize = AlignedSize;
+ LDSParams.LDSSize = static_cast<uint32_t>(AlignedSize);
SmallString<128> MDTypeStr;
raw_svector_ostream MDTypeOS(MDTypeStr);
MDTypeOS << "llvm.amdgcn.sw.lds." << Func->getName() << ".md.type";
@@ -582,7 +582,8 @@ void AMDGPUSwLowerLDS::updateMallocSizeForDynamicLDS(
assert(SwLDS && SwLDSMetadata);
StructType *MetadataStructType =
cast<StructType>(SwLDSMetadata->getValueType());
- unsigned MaxAlignment = SwLDS->getAlign().valueOrOne().value();
+ unsigned MaxAlignment =
+ static_cast<unsigned>(SwLDS->getAlign().valueOrOne().value());
Value *MaxAlignValue = IRB.getInt32(MaxAlignment);
Value *MaxAlignValueMinusOne = IRB.getInt32(MaxAlignment - 1);
@@ -878,7 +879,7 @@ void AMDGPUSwLowerLDS::lowerKernelLDSAccesses(Function *Func,
GetUniqueLDSGlobals(LDSParams.IndirectAccess.StaticLDSGlobals);
// The metadata global always has an item for the SwLDS pointer itself, so
// there is at least one static item and the last one ends the static region.
- unsigned LastStaticLDSIdx = UniqueLDSGlobals.size();
+ unsigned LastStaticLDSIdx = static_cast<unsigned>(UniqueLDSGlobals.size());
UniqueLDSGlobals.clear();
auto *GEPForEndStaticLDSOffset =
@@ -1130,7 +1131,8 @@ void AMDGPUSwLowerLDS::lowerNonKernelLDSAccesses(
for (GlobalVariable *GV : LDSGlobals) {
const auto *GVIt = llvm::find(OrdereLDSGlobals, GV);
assert(GVIt != OrdereLDSGlobals.end());
- uint32_t GVOffset = std::distance(OrdereLDSGlobals.begin(), GVIt);
+ uint32_t GVOffset =
+ static_cast<uint32_t>(std::distance(OrdereLDSGlobals.begin(), GVIt));
Value *OffsetGEP = IRB.CreateInBoundsGEP(
LDSOffsetTable->getValueType(), LDSOffsetTable,
@@ -1177,7 +1179,7 @@ void AMDGPUSwLowerLDS::initAsanInfo() {
llvm::getAddressSanitizerParams(M.getTargetTriple(), LongSize, false, &Offset,
&Scale, &OrShadowOffset);
AsanInfo.Scale = Scale;
- AsanInfo.Offset = Offset;
+ AsanInfo.Offset = static_cast<uint32_t>(Offset);
}
static bool hasFnWithSanitizeAddressAttr(FunctionVariableMap &LDSAccesses) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUUnifyDivergentExitNodes.cpp b/llvm/lib/Target/AMDGPU/AMDGPUUnifyDivergentExitNodes.cpp
index 9978a6a7c99c99..35e9e7e5125e7a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUUnifyDivergentExitNodes.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUUnifyDivergentExitNodes.cpp
@@ -149,7 +149,8 @@ BasicBlock *AMDGPUUnifyDivergentExitNodesImpl::unifyReturnBlockSet(
B.CreateRetVoid();
} else {
// If the function doesn't return void... add a PHI node to the block...
- PN = B.CreatePHI(F.getReturnType(), ReturningBlocks.size(),
+ PN = B.CreatePHI(F.getReturnType(),
+ static_cast<unsigned>(ReturningBlocks.size()),
"UnifiedRetVal");
B.CreateRet(PN);
}
More information about the llvm-commits
mailing list