40#include "llvm/IR/IntrinsicsAMDGPU.h"
41#include "llvm/IR/IntrinsicsR600.h"
49#define DEBUG_TYPE "amdgpu-promote-alloca"
56 DisablePromoteAllocaToVector(
"disable-promote-alloca-to-vector",
57 cl::desc(
"Disable promote alloca to vector"),
61 DisablePromoteAllocaToLDS(
"disable-promote-alloca-to-lds",
62 cl::desc(
"Disable promote alloca to LDS"),
66 "amdgpu-promote-alloca-to-vector-limit",
67 cl::desc(
"Maximum byte size to consider promote alloca to vector"),
71 "amdgpu-promote-alloca-to-vector-max-regs",
73 "Maximum vector size (in 32b registers) to use when promoting alloca"),
79 "amdgpu-promote-alloca-to-vector-vgpr-ratio",
80 cl::desc(
"Ratio of VGPRs to budget for promoting alloca to vectors"),
84 LoopUserWeight(
"promote-alloca-vector-loop-user-weight",
85 cl::desc(
"The bonus weight of users of allocas within loop "
86 "when sorting profitable allocas"),
92struct GEPToVectorIndex {
100struct MemTransferInfo {
106struct AllocaAnalysis {
111 bool HaveSelectOrPHI =
false;
124 explicit AllocaAnalysis(
AllocaInst *Alloca) : Alloca(Alloca) {}
128class AMDGPUPromoteAllocaImpl {
139 unsigned VGPRBudgetRatio;
140 unsigned MaxVectorRegs;
142 bool IsAMDGCN =
false;
143 bool IsAMDHSA =
false;
145 std::pair<Value *, Value *> getLocalSizeYZ(
IRBuilder<> &Builder);
148 bool collectAllocaUses(AllocaAnalysis &
AA)
const;
154 bool binaryOpIsDerivedFromSameAlloca(
Value *Alloca,
Value *Val,
159 bool hasSufficientLocalMem(
const Function &
F);
162 void analyzePromoteToVector(AllocaAnalysis &
AA)
const;
163 void promoteAllocaToVector(AllocaAnalysis &
AA);
164 void analyzePromoteToLDS(AllocaAnalysis &
AA)
const;
165 bool tryPromoteAllocaToLDS(AllocaAnalysis &
AA,
bool SufficientLDS,
170 void scoreAlloca(AllocaAnalysis &
AA)
const;
172 void setFunctionLimits(
const Function &
F);
176 : TM(TM), LI(LI),
Mod(M),
DL(M.getDataLayout()) {
177 const Triple &TT = M.getTargetTriple();
178 IsAMDGCN = TT.isAMDGCN();
182 bool run(
Function &
F,
bool PromoteToLDS);
195 if (
auto *TPC = getAnalysisIfAvailable<TargetPassConfig>())
196 return AMDGPUPromoteAllocaImpl(
198 getAnalysis<LoopInfoWrapperPass>().getLoopInfo())
203 StringRef getPassName()
const override {
return "AMDGPU Promote Alloca"; }
212static unsigned getMaxVGPRs(
unsigned LDSBytes,
const TargetMachine &TM,
219 if (DynamicVGPRBlockSize == 0 && ST.isDynamicVGPREnabled())
220 DynamicVGPRBlockSize = ST.getDynamicVGPRBlockSize();
222 unsigned MaxVGPRs = ST.getMaxNumVGPRs(
223 ST.getWavesPerEU(ST.getFlatWorkGroupSizes(
F), LDSBytes,
F).first,
224 DynamicVGPRBlockSize);
229 if (!
F.hasFnAttribute(Attribute::AlwaysInline) &&
231 MaxVGPRs = std::min(MaxVGPRs, 32u);
237char AMDGPUPromoteAlloca::ID = 0;
240 "AMDGPU promote alloca to vector or LDS",
false,
false)
253 bool Changed = AMDGPUPromoteAllocaImpl(TM, *
F.getParent(), LI)
266 bool Changed = AMDGPUPromoteAllocaImpl(TM, *
F.getParent(), LI)
277 return new AMDGPUPromoteAlloca();
280bool AMDGPUPromoteAllocaImpl::collectAllocaUses(AllocaAnalysis &
AA)
const {
283 <<
" " << *Inst <<
"\n");
288 while (!WorkList.empty()) {
289 auto *Cur = WorkList.pop_back_val();
290 if (
find(
AA.Pointers, Cur) !=
AA.Pointers.end())
292 AA.Pointers.insert(Cur);
293 for (
auto &U : Cur->uses()) {
297 return RejectUser(Inst,
"pointer escapes via store");
300 AA.Uses.push_back(&U);
303 WorkList.push_back(Inst);
307 if (!binaryOpIsDerivedFromSameAlloca(
AA.Alloca, Cur,
SI, 1, 2))
308 return RejectUser(Inst,
"select from mixed objects");
309 WorkList.push_back(Inst);
310 AA.HaveSelectOrPHI =
true;
316 switch (
Phi->getNumIncomingValues()) {
320 if (!binaryOpIsDerivedFromSameAlloca(
AA.Alloca, Cur, Phi, 0, 1))
321 return RejectUser(Inst,
"phi from mixed objects");
324 return RejectUser(Inst,
"phi with too many operands");
327 WorkList.push_back(Inst);
328 AA.HaveSelectOrPHI =
true;
335void AMDGPUPromoteAllocaImpl::scoreAlloca(AllocaAnalysis &
AA)
const {
339 for (
auto *U :
AA.Uses) {
345 1 + (LoopUserWeight * LI.getLoopDepth(Inst->
getParent()));
346 LLVM_DEBUG(
dbgs() <<
" [+" << UserScore <<
"]:\t" << *Inst <<
"\n");
353void AMDGPUPromoteAllocaImpl::setFunctionLimits(
const Function &
F) {
357 const int R600MaxVectorRegs = 16;
358 MaxVectorRegs =
F.getFnAttributeAsParsedInteger(
359 "amdgpu-promote-alloca-to-vector-max-regs",
360 IsAMDGCN ? PromoteAllocaToVectorMaxRegs : R600MaxVectorRegs);
361 if (PromoteAllocaToVectorMaxRegs.getNumOccurrences())
362 MaxVectorRegs = PromoteAllocaToVectorMaxRegs;
363 VGPRBudgetRatio =
F.getFnAttributeAsParsedInteger(
364 "amdgpu-promote-alloca-to-vector-vgpr-ratio",
365 PromoteAllocaToVectorVGPRRatio);
366 if (PromoteAllocaToVectorVGPRRatio.getNumOccurrences())
367 VGPRBudgetRatio = PromoteAllocaToVectorVGPRRatio;
370bool AMDGPUPromoteAllocaImpl::run(
Function &
F,
bool PromoteToLDS) {
371 if (DisablePromoteAllocaToLDS && DisablePromoteAllocaToVector)
374 bool SufficientLDS = PromoteToLDS && hasSufficientLocalMem(
F);
375 MaxVGPRs = IsAMDGCN ? getMaxVGPRs(CurrentLocalMemUsage, TM,
F) : 128;
376 setFunctionLimits(
F);
378 unsigned VectorizationBudget =
379 (PromoteAllocaToVectorLimit ? PromoteAllocaToVectorLimit * 8
383 std::vector<AllocaAnalysis> Allocas;
388 if (!AI->isStaticAlloca() || AI->isArrayAllocation())
393 AllocaAnalysis
AA{AI};
394 if (collectAllocaUses(
AA)) {
395 analyzePromoteToVector(
AA);
397 analyzePromoteToLDS(
AA);
398 if (
AA.Vector.Ty ||
AA.LDS.Enable) {
400 Allocas.push_back(std::move(
AA));
407 [](
const auto &
A,
const auto &
B) {
return A.Score >
B.Score; });
411 dbgs() <<
"Sorted Worklist:\n";
412 for (
const auto &
AA : Allocas)
413 dbgs() <<
" " << *
AA.Alloca <<
"\n";
419 for (AllocaAnalysis &
AA : Allocas) {
421 std::optional<TypeSize>
Size =
AA.Alloca->getAllocationSize(
DL);
423 const unsigned AllocaCost =
Size->getFixedValue() * 8;
425 if (AllocaCost <= VectorizationBudget) {
426 promoteAllocaToVector(
AA);
428 assert((VectorizationBudget - AllocaCost) < VectorizationBudget &&
430 VectorizationBudget -= AllocaCost;
432 << VectorizationBudget <<
"\n");
436 << AllocaCost <<
", budget:" << VectorizationBudget
437 <<
"): " << *
AA.Alloca <<
"\n");
442 tryPromoteAllocaToLDS(
AA, SufficientLDS, DeferredIntrs))
445 finishDeferredAllocaToLDSPromotion(DeferredIntrs);
467 return I->getOperand(0) == AI &&
475 if (Ptr ==
AA.Alloca)
476 return B.getInt32(0);
479 auto I =
AA.Vector.GEPVectorIdx.find(
GEP);
480 assert(
I !=
AA.Vector.GEPVectorIdx.end() &&
"Must have entry for GEP!");
482 if (!
I->second.Full) {
483 Value *Result =
nullptr;
484 B.SetInsertPoint(
GEP);
486 if (
I->second.VarIndex) {
487 Result =
I->second.VarIndex;
488 Result =
B.CreateSExtOrTrunc(Result,
B.getInt32Ty());
490 if (
I->second.VarMul)
491 Result =
B.CreateMul(Result,
I->second.VarMul);
493 if (
I->second.VarShift)
494 Result =
B.CreateAShr(Result,
I->second.VarShift,
"",
true);
497 if (
I->second.ConstIndex) {
499 Result =
B.CreateAdd(Result,
I->second.ConstIndex);
501 Result =
I->second.ConstIndex;
505 Result =
B.getInt32(0);
507 I->second.Full = Result;
510 return I->second.Full;
513static std::optional<GEPToVectorIndex>
519 unsigned BW =
DL.getIndexTypeSizeInBits(
GEP->getType());
521 APInt ConstOffset(BW, 0);
542 if (!CurGEP->collectOffset(
DL, BW, VarOffsets, ConstOffset))
546 CurPtr = CurGEP->getPointerOperand();
549 assert(CurPtr == Alloca &&
"GEP not based on alloca");
551 int64_t VecElemSize =
DL.getTypeAllocSize(VecElemTy);
552 if (VarOffsets.
size() > 1)
558 if (ConstOffset.
srem(VecElemSize) != 0)
560 APInt IndexQuot = ConstOffset.
sdiv(VecElemSize);
562 GEPToVectorIndex Result;
564 if (!ConstOffset.
isZero())
565 Result.ConstIndex = ConstantInt::get(Ctx, IndexQuot.
sextOrTrunc(BW));
568 if (VarOffsets.
empty())
573 const auto &VarOffset = VarOffsets.
front();
574 auto ScaleOpt = VarOffset.second.tryZExtValue();
575 if (!ScaleOpt || *ScaleOpt == 0)
579 Result.VarIndex = VarOffset.first;
585 if (Scale >= (
uint64_t)VecElemSize) {
586 if (Scale % VecElemSize != 0)
591 uint64_t VarMul = Scale / VecElemSize;
594 Result.VarMul = ConstantInt::get(Ctx,
APInt(BW, VarMul));
596 if ((
uint64_t)VecElemSize % Scale != 0)
601 uint64_t Divisor = VecElemSize / Scale;
611 Result.VarShift = ConstantInt::get(Ctx,
APInt(BW,
Log2_64(Divisor)));
632 unsigned VecStoreSize,
633 unsigned ElementSize,
639 Builder.SetInsertPoint(Inst);
641 Type *VecEltTy =
AA.Vector.Ty->getElementType();
644 case Instruction::Load: {
645 Value *CurVal = GetCurVal();
651 TypeSize AccessSize =
DL.getTypeStoreSize(AccessTy);
653 if (CI->isNullValue() && AccessSize == VecStoreSize) {
655 Builder.CreateBitPreservingCastChain(
DL, CurVal, AccessTy));
663 const unsigned NumLoadedElts = AccessSize /
DL.getTypeStoreSize(VecEltTy);
665 assert(
DL.getTypeStoreSize(SubVecTy) ==
DL.getTypeStoreSize(AccessTy));
674 TypeSize NumBits =
DL.getTypeStoreSize(SubVecTy) * 8u;
676 bool IsAlignedLoad = NumBits <= (LoadAlign * 8u);
678 bool IsProperlyDivisible = TotalNumElts % NumLoadedElts == 0;
681 IsProperlyDivisible && IsAlignedLoad) {
683 const unsigned NewNumElts =
684 DL.getTypeStoreSize(VectorTy) * 8u / NumBits;
685 const unsigned LShrAmt =
llvm::Log2_32(SubVecTy->getNumElements());
689 Builder.CreateBitPreservingCastChain(
DL, CurVal, BitCastTy);
690 Value *NewIdx = Builder.CreateLShr(
691 Index, ConstantInt::get(Index->getType(), LShrAmt));
692 Value *ExtVal = Builder.CreateExtractElement(BCVal, NewIdx);
694 Builder.CreateBitPreservingCastChain(
DL, ExtVal, AccessTy);
700 for (
unsigned K = 0; K < NumLoadedElts; ++K) {
702 Builder.CreateAdd(Index, ConstantInt::get(Index->getType(), K));
703 SubVec = Builder.CreateInsertElement(
704 SubVec, Builder.CreateExtractElement(CurVal, CurIdx), K);
708 Builder.CreateBitPreservingCastChain(
DL, SubVec, AccessTy));
713 Value *ExtractElement = Builder.CreateExtractElement(CurVal, Index);
714 if (AccessTy != VecEltTy)
715 ExtractElement = Builder.CreateBitOrPointerCast(ExtractElement, AccessTy);
720 case Instruction::Store: {
727 Value *Val =
SI->getValueOperand();
731 TypeSize AccessSize =
DL.getTypeStoreSize(AccessTy);
733 if (CI->isNullValue() && AccessSize == VecStoreSize)
734 return Builder.CreateBitPreservingCastChain(
DL, Val,
AA.Vector.Ty);
739 const unsigned NumWrittenElts =
740 AccessSize /
DL.getTypeStoreSize(VecEltTy);
741 const unsigned NumVecElts =
AA.Vector.Ty->getNumElements();
743 assert(
DL.getTypeStoreSize(SubVecTy) ==
DL.getTypeStoreSize(AccessTy));
745 Val = Builder.CreateBitPreservingCastChain(
DL, Val, SubVecTy);
746 Value *CurVec = GetCurVal();
747 for (
unsigned K = 0, NumElts = std::min(NumWrittenElts, NumVecElts);
750 Builder.CreateAdd(Index, ConstantInt::get(Index->getType(), K));
751 CurVec = Builder.CreateInsertElement(
752 CurVec, Builder.CreateExtractElement(Val, K), CurIdx);
757 if (Val->
getType() != VecEltTy)
758 Val = Builder.CreateBitOrPointerCast(Val, VecEltTy);
759 return Builder.CreateInsertElement(GetCurVal(), Val, Index);
761 case Instruction::Call: {
765 unsigned NumCopied =
Length->getZExtValue() / ElementSize;
766 MemTransferInfo *TI = &
AA.Vector.TransferInfo[MTI];
771 for (
unsigned Idx = 0; Idx <
AA.Vector.Ty->getNumElements(); ++Idx) {
772 if (Idx >= DestBegin && Idx < DestBegin + NumCopied) {
773 Mask.push_back(SrcBegin < AA.Vector.Ty->getNumElements()
781 return Builder.CreateShuffleVector(GetCurVal(), Mask);
787 Value *Elt = MSI->getOperand(1);
788 const unsigned BytesPerElt =
DL.getTypeStoreSize(VecEltTy);
789 if (BytesPerElt > 1) {
790 Value *EltBytes = Builder.CreateVectorSplat(BytesPerElt, Elt);
796 Elt = Builder.CreateBitCast(EltBytes, PtrInt);
797 Elt = Builder.CreateIntToPtr(Elt, VecEltTy);
799 Elt = Builder.CreateBitCast(EltBytes, VecEltTy);
802 return Builder.CreateVectorSplat(
AA.Vector.Ty->getElementCount(), Elt);
806 if (Intr->getIntrinsicID() == Intrinsic::objectsize) {
807 Intr->replaceAllUsesWith(
808 Builder.getIntN(Intr->getType()->getIntegerBitWidth(),
809 DL.getTypeAllocSize(
AA.Vector.Ty)));
838 TypeSize AccTS =
DL.getTypeStoreSize(AccessTy);
842 if (AccTS * 8 !=
DL.getTypeSizeInBits(AccessTy))
854template <
typename InstContainer>
866 auto &BlockUses = UsesByBlock[BB];
869 if (BlockUses.empty())
873 if (BlockUses.size() == 1) {
880 if (!BlockUses.contains(&Inst))
901AMDGPUPromoteAllocaImpl::getVectorTypeForAlloca(
Type *AllocaTy)
const {
902 if (DisablePromoteAllocaToVector) {
909 uint64_t NumElems = 1;
912 NumElems *= ArrayTy->getNumElements();
913 ElemTy = ArrayTy->getElementType();
919 NumElems *= InnerVectorTy->getNumElements();
920 ElemTy = InnerVectorTy->getElementType();
924 unsigned ElementSize =
DL.getTypeSizeInBits(ElemTy) / 8;
925 if (ElementSize > 0) {
926 unsigned AllocaSize =
DL.getTypeStoreSize(AllocaTy);
931 if (NumElems * ElementSize != AllocaSize)
932 NumElems = AllocaSize / ElementSize;
933 if (NumElems > 0 && (AllocaSize % ElementSize) == 0)
943 const unsigned MaxElements =
944 (MaxVectorRegs * 32) /
DL.getTypeSizeInBits(VectorTy->getElementType());
946 if (VectorTy->getNumElements() > MaxElements ||
947 VectorTy->getNumElements() < 2) {
949 <<
" has an unsupported number of elements\n");
953 Type *VecEltTy = VectorTy->getElementType();
954 unsigned ElementSizeInBits =
DL.getTypeSizeInBits(VecEltTy);
955 if (ElementSizeInBits !=
DL.getTypeAllocSizeInBits(VecEltTy)) {
956 LLVM_DEBUG(
dbgs() <<
" Cannot convert to vector if the allocation size "
957 "does not match the type's size\n");
964void AMDGPUPromoteAllocaImpl::analyzePromoteToVector(AllocaAnalysis &
AA)
const {
965 if (
AA.HaveSelectOrPHI) {
966 LLVM_DEBUG(
dbgs() <<
" Cannot convert to vector due to select or phi\n");
970 Type *AllocaTy =
AA.Alloca->getAllocatedType();
971 AA.Vector.Ty = getVectorTypeForAlloca(AllocaTy);
977 <<
" " << *Inst <<
"\n");
978 AA.Vector.Ty =
nullptr;
981 Type *VecEltTy =
AA.Vector.Ty->getElementType();
982 unsigned ElementSize =
DL.getTypeSizeInBits(VecEltTy) / 8;
984 for (
auto *U :
AA.Uses) {
993 return RejectUser(Inst,
"unsupported load/store as aggregate");
1000 return RejectUser(Inst,
"not a simple load or store");
1002 Ptr = Ptr->stripPointerCasts();
1005 if (Ptr ==
AA.Alloca &&
1006 DL.getTypeStoreSize(
AA.Alloca->getAllocatedType()) ==
1007 DL.getTypeStoreSize(AccessTy)) {
1008 AA.Vector.Worklist.push_back(Inst);
1013 return RejectUser(Inst,
"not a supported access type");
1015 AA.Vector.Worklist.push_back(Inst);
1024 return RejectUser(Inst,
"cannot compute vector index for GEP");
1026 AA.Vector.GEPVectorIdx[
GEP] = std::move(
Index.value());
1027 AA.Vector.UsersToRemove.push_back(Inst);
1033 AA.Vector.Worklist.push_back(Inst);
1038 if (TransferInst->isVolatile())
1039 return RejectUser(Inst,
"mem transfer inst is volatile");
1042 if (!Len || (
Len->getZExtValue() % ElementSize))
1043 return RejectUser(Inst,
"mem transfer inst length is non-constant or "
1044 "not a multiple of the vector element size");
1047 if (Ptr ==
AA.Alloca)
1048 return ConstantInt::get(Ptr->getContext(),
APInt(32, 0));
1051 const auto &GEPI =
AA.Vector.GEPVectorIdx.find(
GEP)->second;
1054 if (GEPI.ConstIndex)
1055 return GEPI.ConstIndex;
1056 return ConstantInt::get(Ptr->getContext(),
APInt(32, 0));
1059 MemTransferInfo *TI =
1060 &
AA.Vector.TransferInfo.try_emplace(TransferInst).first->second;
1061 unsigned OpNum =
U->getOperandNo();
1063 Value *Dest = TransferInst->getDest();
1066 return RejectUser(Inst,
"could not calculate constant dest index");
1067 TI->DestIndex =
Index;
1070 Value *Src = TransferInst->getSource();
1073 return RejectUser(Inst,
"could not calculate constant src index");
1074 TI->SrcIndex =
Index;
1080 if (Intr->getIntrinsicID() == Intrinsic::objectsize) {
1081 AA.Vector.Worklist.push_back(Inst);
1089 return RejectUser(Inst,
"assume-like intrinsic cannot have any users");
1090 AA.Vector.UsersToRemove.push_back(Inst);
1095 return isAssumeLikeIntrinsic(cast<Instruction>(U));
1097 AA.Vector.UsersToRemove.push_back(Inst);
1101 return RejectUser(Inst,
"unhandled alloca user");
1105 for (
const auto &Entry :
AA.Vector.TransferInfo) {
1106 const MemTransferInfo &TI =
Entry.second;
1107 if (!TI.SrcIndex || !TI.DestIndex)
1108 return RejectUser(
Entry.first,
1109 "mem transfer inst between different objects");
1110 AA.Vector.Worklist.push_back(
Entry.first);
1114void AMDGPUPromoteAllocaImpl::promoteAllocaToVector(AllocaAnalysis &
AA) {
1116 LLVM_DEBUG(
dbgs() <<
" type conversion: " << *
AA.Alloca->getAllocatedType()
1117 <<
" -> " << *
AA.Vector.Ty <<
'\n');
1118 const unsigned VecStoreSize =
DL.getTypeStoreSize(
AA.Vector.Ty);
1120 Type *VecEltTy =
AA.Vector.Ty->getElementType();
1121 const unsigned ElementSize =
DL.getTypeSizeInBits(VecEltTy) / 8;
1143 BasicBlock *BB = I->getParent();
1144 auto GetCurVal = [&]() -> Value * {
1145 if (Value *CurVal = Updater.FindValueForBlock(BB))
1148 if (!Placeholders.empty() && Placeholders.back()->getParent() == BB)
1149 return Placeholders.back();
1153 IRBuilder<> Builder(I);
1154 auto *Placeholder = cast<Instruction>(Builder.CreateFreeze(
1155 PoisonValue::get(AA.Vector.Ty),
"promotealloca.placeholder"));
1156 Placeholders.insert(Placeholder);
1157 return Placeholders.back();
1161 ElementSize, GetCurVal);
1175 Placeholder->replaceAllUsesWith(
1177 Placeholder->eraseFromParent();
1183 I->eraseFromParent();
1188 I->dropDroppableUses();
1190 I->eraseFromParent();
1195 AA.Alloca->eraseFromParent();
1198std::pair<Value *, Value *>
1199AMDGPUPromoteAllocaImpl::getLocalSizeYZ(
IRBuilder<> &Builder) {
1205 Intrinsic::r600_read_local_size_y, {});
1207 Intrinsic::r600_read_local_size_z, {});
1209 ST.makeLIDRangeMetadata(LocalSizeY);
1210 ST.makeLIDRangeMetadata(LocalSizeZ);
1212 return std::pair(LocalSizeY, LocalSizeZ);
1253 F.removeFnAttr(
"amdgpu-no-dispatch-ptr");
1270 LoadXY->
setMetadata(LLVMContext::MD_invariant_load, MD);
1271 LoadZU->
setMetadata(LLVMContext::MD_invariant_load, MD);
1272 ST.makeLIDRangeMetadata(LoadZU);
1277 return std::pair(
Y, LoadZU);
1289 IntrID = IsAMDGCN ? (
Intrinsic::ID)Intrinsic::amdgcn_workitem_id_x
1291 AttrName =
"amdgpu-no-workitem-id-x";
1294 IntrID = IsAMDGCN ? (
Intrinsic::ID)Intrinsic::amdgcn_workitem_id_y
1296 AttrName =
"amdgpu-no-workitem-id-y";
1300 IntrID = IsAMDGCN ? (
Intrinsic::ID)Intrinsic::amdgcn_workitem_id_z
1302 AttrName =
"amdgpu-no-workitem-id-z";
1310 ST.makeLIDRangeMetadata(CI);
1311 F->removeFnAttr(AttrName);
1321 switch (
II->getIntrinsicID()) {
1322 case Intrinsic::memcpy:
1323 case Intrinsic::memmove:
1324 case Intrinsic::memset:
1325 case Intrinsic::lifetime_start:
1326 case Intrinsic::lifetime_end:
1327 case Intrinsic::invariant_start:
1328 case Intrinsic::invariant_end:
1329 case Intrinsic::launder_invariant_group:
1330 case Intrinsic::strip_invariant_group:
1331 case Intrinsic::objectsize:
1338bool AMDGPUPromoteAllocaImpl::binaryOpIsDerivedFromSameAlloca(
1360 if (OtherObj != BaseAlloca) {
1362 dbgs() <<
"Found a binary instruction with another alloca object\n");
1369void AMDGPUPromoteAllocaImpl::analyzePromoteToLDS(AllocaAnalysis &
AA)
const {
1370 if (DisablePromoteAllocaToLDS) {
1378 const Function &ContainingFunction = *
AA.Alloca->getFunction();
1388 <<
" promote alloca to LDS not supported with calling convention.\n");
1399 if (
find(
AA.LDS.Worklist,
User) ==
AA.LDS.Worklist.end())
1400 AA.LDS.Worklist.push_back(
User);
1405 if (UseInst->
getOpcode() == Instruction::PtrToInt)
1409 if (LI->isVolatile())
1415 if (
SI->isVolatile())
1421 if (RMW->isVolatile())
1427 if (CAS->isVolatile())
1435 if (!binaryOpIsDerivedFromSameAlloca(
AA.Alloca,
Use->get(), ICmp, 0, 1))
1439 if (
find(
AA.LDS.Worklist,
User) ==
AA.LDS.Worklist.end())
1440 AA.LDS.Worklist.push_back(ICmp);
1447 if (!
GEP->isInBounds())
1460 if (
find(
AA.LDS.Worklist,
User) ==
AA.LDS.Worklist.end())
1461 AA.LDS.Worklist.push_back(
User);
1464 AA.LDS.Enable =
true;
1467bool AMDGPUPromoteAllocaImpl::hasSufficientLocalMem(
const Function &
F) {
1475 for (
Type *ParamTy : FTy->params()) {
1479 LLVM_DEBUG(
dbgs() <<
"Function has local memory argument. Promoting to "
1480 "local memory disabled.\n");
1485 LocalMemLimit =
ST.getAddressableLocalMemorySize();
1486 if (LocalMemLimit == 0)
1496 if (
Use->getFunction() == &
F)
1500 if (VisitedConstants.
insert(
C).second)
1512 if (visitUsers(&GV, &GV)) {
1520 while (!
Stack.empty()) {
1522 if (visitUsers(&GV,
C)) {
1542 LLVM_DEBUG(
dbgs() <<
"Function has a reference to externally allocated "
1543 "local memory. Promoting to local memory "
1558 CurrentLocalMemUsage = 0;
1564 for (
auto Alloc : AllocatedSizes) {
1565 CurrentLocalMemUsage =
alignTo(CurrentLocalMemUsage,
Alloc.second);
1566 CurrentLocalMemUsage +=
Alloc.first;
1569 unsigned MaxOccupancy =
1570 ST.getWavesPerEU(
ST.getFlatWorkGroupSizes(
F), CurrentLocalMemUsage,
F)
1574 unsigned MaxSizeWithWaveCount =
1575 ST.getMaxLocalMemSizeWithWaveCount(MaxOccupancy,
F);
1578 if (CurrentLocalMemUsage > MaxSizeWithWaveCount)
1581 LocalMemLimit = MaxSizeWithWaveCount;
1584 <<
" bytes of LDS\n"
1585 <<
" Rounding size to " << MaxSizeWithWaveCount
1586 <<
" with a maximum occupancy of " << MaxOccupancy <<
'\n'
1587 <<
" and " << (LocalMemLimit - CurrentLocalMemUsage)
1588 <<
" available for promotion\n");
1594bool AMDGPUPromoteAllocaImpl::tryPromoteAllocaToLDS(
1595 AllocaAnalysis &
AA,
bool SufficientLDS,
1605 const Function &ContainingFunction = *
AA.Alloca->getParent()->getParent();
1607 unsigned WorkGroupSize =
ST.getFlatWorkGroupSizes(ContainingFunction).second;
1609 Align Alignment =
AA.Alloca->getAlign();
1617 uint32_t NewSize =
alignTo(CurrentLocalMemUsage, Alignment);
1618 std::optional<TypeSize> ElemSize =
AA.Alloca->getAllocationSize(
DL);
1619 if (!ElemSize || ElemSize->isScalable())
1621 TypeSize AllocSize = WorkGroupSize * *ElemSize;
1624 if (NewSize > LocalMemLimit) {
1626 <<
" bytes of local memory not available to promote\n");
1630 CurrentLocalMemUsage = NewSize;
1639 Twine(
F->getName()) +
Twine(
'.') +
AA.Alloca->getName(),
nullptr,
1644 Value *TCntY, *TCntZ;
1646 std::tie(TCntY, TCntZ) = getLocalSizeYZ(Builder);
1647 Value *TIdX = getWorkitemID(Builder, 0);
1648 Value *TIdY = getWorkitemID(Builder, 1);
1649 Value *TIdZ = getWorkitemID(Builder, 2);
1661 AA.Alloca->mutateType(
Offset->getType());
1662 AA.Alloca->replaceAllUsesWith(
Offset);
1663 AA.Alloca->eraseFromParent();
1667 for (
Value *V :
AA.LDS.Worklist) {
1689 assert(
V->getType()->isPtrOrPtrVectorTy());
1691 Type *NewTy =
V->getType()->getWithNewType(NewPtrTy);
1692 V->mutateType(NewTy);
1702 for (
unsigned I = 0,
E =
Phi->getNumIncomingValues();
I !=
E; ++
I) {
1704 Phi->getIncomingValue(
I)))
1715 case Intrinsic::lifetime_start:
1716 case Intrinsic::lifetime_end:
1720 case Intrinsic::memcpy:
1721 case Intrinsic::memmove:
1725 DeferredIntrs.
insert(Intr);
1727 case Intrinsic::memset: {
1735 case Intrinsic::invariant_start:
1736 case Intrinsic::invariant_end:
1737 case Intrinsic::launder_invariant_group:
1738 case Intrinsic::strip_invariant_group: {
1756 case Intrinsic::objectsize: {
1760 Intrinsic::objectsize,
1776void AMDGPUPromoteAllocaImpl::finishDeferredAllocaToLDSPromotion(
1783 assert(ID == Intrinsic::memcpy || ID == Intrinsic::memmove);
1787 ID,
MI->getRawDest(),
MI->getDestAlign(),
MI->getRawSource(),
1788 MI->getSourceAlign(),
MI->getLength(),
MI->isVolatile());
1790 for (
unsigned I = 0;
I != 2; ++
I) {
1792 B->addDereferenceableParamAttr(
I, Bytes);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool runOnFunction(Function &F, bool PostInlining)
AMD GCN specific subclass of TargetSubtarget.
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Remove Loads Into Fake Uses
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Target-Independent Code Generator Pass Configuration Options pass.
static const AMDGPUSubtarget & get(const MachineFunction &MF)
Class for arbitrary precision integers.
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
LLVM_ABI APInt sdiv(const APInt &RHS) const
Signed division function for APInt.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
LLVM_ABI APInt srem(const APInt &RHS) const
Function for signed remainder operation.
an instruction to allocate memory on the stack
Type * getAllocatedType() const
Return the type that is being allocated by the instruction.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
An instruction that atomically checks whether a specified value is in a memory location,...
an instruction that atomically reads a memory location, combines it with another value,...
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
InstListType::iterator iterator
Instruction iterators...
Represents analyses that only rely on functions' control flow.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
void addDereferenceableRetAttr(uint64_t Bytes)
adds the dereferenceable attribute to the list of attributes.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
This class represents a function call, abstracting a target machine's calling convention.
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
This is the shared class of boolean and integer constants.
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
This is an important base class in LLVM.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
FunctionPass class - This class is used to implement most global optimizations.
Class to represent function types.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
bool hasExternalLinkage() const
void setUnnamedAddr(UnnamedAddr Val)
unsigned getAddressSpace() const
@ InternalLinkage
Rename collisions when linking (static functions).
Type * getValueType() const
MaybeAlign getAlign() const
Returns the alignment of the given variable.
LLVM_ABI uint64_t getGlobalSize(const DataLayout &DL) const
Get the size of this global variable in bytes.
void setAlignment(Align Align)
Sets the alignment attribute of the GlobalVariable.
This instruction compares its operands according to the predicate given to the constructor.
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
BasicBlock * GetInsertBlock() const
Value * CreateInBoundsGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="")
CallInst * CreateMemSet(Value *Ptr, Value *Val, uint64_t Size, MaybeAlign Align, bool isVolatile=false, const AAMDNodes &AAInfo=AAMDNodes())
Create and insert a memset to the specified pointer and the specified value.
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
CallInst * CreateCall(FunctionType *FTy, Value *Callee, ArrayRef< Value * > Args={}, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateConstInBoundsGEP1_64(Type *Ty, Value *Ptr, uint64_t Idx0, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
LLVM_ABI CallInst * CreateMemTransferInst(Intrinsic::ID IntrID, Value *Dst, MaybeAlign DstAlign, Value *Src, MaybeAlign SrcAlign, Value *Size, bool isVolatile=false, const AAMDNodes &AAInfo=AAMDNodes())
Value * CreateMul(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Analysis pass that exposes the LoopInfo for a function.
The legacy pass manager's analysis pass to compute loop information.
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
This class implements a map that also provides access to all stored values in a deterministic order.
std::pair< KeyT, ValueT > & front()
Value * getLength() const
Value * getRawDest() const
MaybeAlign getDestAlign() const
This class wraps the llvm.memset and llvm.memset.inline intrinsics.
This class wraps the llvm.memcpy/memmove intrinsics.
A Module instance is used to store all the information related to an LLVM module.
virtual void getAnalysisUsage(AnalysisUsage &) const
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
Class to represent pointers.
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Helper class for SSA formation on a set of values defined in multiple blocks.
LLVM_ABI void Initialize(Type *Ty, StringRef Name)
Reset this object to get ready for a new set of SSA updates with type 'Ty'.
LLVM_ABI Value * GetValueInMiddleOfBlock(BasicBlock *BB)
Construct SSA form, materializing a value that is live in the middle of the specified block.
LLVM_ABI void AddAvailableValue(BasicBlock *BB, Value *V)
Indicate that a rewritten value is available in the specified block with the specified value.
This class represents the LLVM 'select' instruction.
A vector that has set insertion semantics.
bool contains(const_arg_type key) const
Check if the SetVector contains the given key.
bool insert(const value_type &X)
Insert a new element into the SetVector.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static unsigned getPointerOperandIndex()
Represent a constant reference to a string, i.e.
Primary interface to the complete machine description for the target machine.
const STC & getSubtarget(const Function &F) const
This method returns a pointer to the specified type of TargetSubtargetInfo.
Triple - Helper class for working with autoconf configuration names.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
The instances of the Type class are immutable: once they are created, they are never changed.
bool isArrayTy() const
True if this is an instance of ArrayType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
bool isAggregateType() const
Return true if the type is an aggregate type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI void print(raw_ostream &O, bool IsForDebug=false) const
Implement operator<< on Value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
iterator_range< user_iterator > users()
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
void mutateType(Type *Ty)
Mutate the type of this Value to be of the specified type.
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
static LLVM_ABI bool isValidElementType(Type *ElemTy)
Return true if the specified type is valid as a element type.
Type * getElementType() const
Value handle that is nullable, but tries to track the Value.
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
@ LOCAL_ADDRESS
Address space for local memory.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
unsigned getDynamicVGPRBlockSize(const Function &F)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
initializer< Ty > init(const Ty &Val)
NodeAddr< PhiNode * > Phi
This is an optimization pass for GlobalISel generic memory operations.
void stable_sort(R &&Range)
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isAssumeLikeIntrinsic(const Instruction *I)
Return true if it is an intrinsic that cannot be speculated but also cannot trap.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
const Value * getPointerOperand(const Value *V)
A helper function that returns the pointer operand of a load, store or GEP instruction.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
FunctionPass * createAMDGPUPromoteAlloca()
@ Mod
The access may modify the value stored in memory.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
char & AMDGPUPromoteAllocaID
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
This struct is a compact representation of a valid (non-zero power of two) alignment.
unsigned countMinTrailingZeros() const
Returns the minimum number of trailing zero bits.
A MapVector that performs no allocations if smaller than a certain size.
Function object to check whether the second component of a container supported by std::get (like std:...