243#include "llvm/IR/IntrinsicsAMDGPU.h"
261#define DEBUG_TYPE "amdgpu-lower-buffer-fat-pointers"
286 Type *remapType(
Type *SrcTy)
override;
287 void clear() { Map.clear(); }
293class BufferFatPtrToIntTypeMap :
public BufferFatPtrTypeLoweringBase {
294 using BufferFatPtrTypeLoweringBase::BufferFatPtrTypeLoweringBase;
304class BufferFatPtrToStructTypeMap :
public BufferFatPtrTypeLoweringBase {
305 using BufferFatPtrTypeLoweringBase::BufferFatPtrTypeLoweringBase;
314Type *BufferFatPtrTypeLoweringBase::remapTypeImpl(
Type *Ty) {
320 return *
Entry = remapScalar(PT);
326 return *
Entry = remapVector(VT);
334 bool IsUniqued = !TyAsStruct || TyAsStruct->
isLiteral();
343 Type *NewElem = remapTypeImpl(OldElem);
344 ElementTypes[
I] = NewElem;
345 Changed |= (OldElem != NewElem);
353 return *
Entry = ArrayType::get(ElementTypes[0], ArrTy->getNumElements());
355 return *
Entry = FunctionType::get(ElementTypes[0],
365 SmallString<16>
Name(STy->getName());
373Type *BufferFatPtrTypeLoweringBase::remapType(
Type *SrcTy) {
374 return remapTypeImpl(SrcTy);
377Type *BufferFatPtrToStructTypeMap::remapScalar(PointerType *PT) {
378 LLVMContext &Ctx = PT->getContext();
383Type *BufferFatPtrToStructTypeMap::remapVector(VectorType *VT) {
384 ElementCount
EC = VT->getElementCount();
385 LLVMContext &Ctx = VT->getContext();
404 if (!ST->isLiteral() || ST->getNumElements() != 2)
410 return MaybeRsrc && MaybeOff &&
419 return isBufferFatPtrOrVector(U.get()->getType());
432class StoreFatPtrsAsIntsAndExpandMemcpyVisitor
433 :
public InstVisitor<StoreFatPtrsAsIntsAndExpandMemcpyVisitor, bool> {
434 BufferFatPtrToIntTypeMap *TypeMap;
438 const DataLayout &
DL;
441 const TargetTransformInfo *
TTI;
450 function_ref<
void(
Type *LeafTy,
Type *IntLeafTy, ArrayRef<unsigned> Idxs,
455 StoreFatPtrsAsIntsAndExpandMemcpyVisitor(BufferFatPtrToIntTypeMap *TypeMap,
457 : TypeMap(TypeMap), IRB(
M, InstSimplifyFolder(
M.getDataLayout())),
458 DL(
M.getDataLayout()) {}
460 ScalarEvolution *SE);
462 bool visitInstruction(Instruction &
I) {
return false; }
463 bool visitAllocaInst(AllocaInst &
I);
464 bool visitLoadInst(LoadInst &LI);
465 bool visitStoreInst(StoreInst &SI);
466 bool visitGetElementPtrInst(GetElementPtrInst &
I);
468 bool visitMemCpyInst(MemCpyInst &MCI);
469 bool visitMemMoveInst(MemMoveInst &MMI);
470 bool visitMemSetInst(MemSetInst &MSI);
471 bool visitMemSetPatternInst(MemSetPatternInst &MSPI);
475Value *StoreFatPtrsAsIntsAndExpandMemcpyVisitor::applyOffset(
Value *Ptr,
478 return IRB.CreatePtrAdd(
479 Ptr, ConstantInt::get(
DL.getIndexType(Ptr->
getType()),
Off),
483void StoreFatPtrsAsIntsAndExpandMemcpyVisitor::forEachAggLeaf(
486 function_ref<
void(
Type *LeafTy,
Type *IntLeafTy, ArrayRef<unsigned> Idxs,
489 Type *IntTy = TypeMap->remapType(Ty);
492 if (
DL.getTypeStoreSize(Ty) != 0)
493 Visit(Ty, IntTy, AggIdxs,
Off, Name);
496 auto Recurse = [&](
unsigned I,
Type *ElemTy,
uint64_t ElemOff) {
498 forEachAggLeaf(ElemTy, AggIdxs,
Off + ElemOff, Name +
"." + Twine(
I),
503 const StructLayout *Layout =
DL.getStructLayout(ST);
504 for (
auto [
I, ElemTy, ElemOff] :
506 Recurse(
I, ElemTy, ElemOff.getFixedValue());
510 Type *ElemTy = AT->getElementType();
511 uint64_t Stride =
DL.getTypeAllocSize(ElemTy).getFixedValue();
513 Recurse(
I, ElemTy,
I * Stride);
516bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::processFunction(
517 Function &
F,
const TargetTransformInfo *
TTI, ScalarEvolution *SE) {
538bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitAllocaInst(AllocaInst &
I) {
539 Type *Ty =
I.getAllocatedType();
540 Type *NewTy = TypeMap->remapType(Ty);
545 TypeSize AllocSize =
DL.getTypeAllocSize(Ty);
546 if (AllocSize.
isFixed() &&
DL.getTypeAllocSize(NewTy) != AllocSize)
547 NewTy = ArrayType::get(IRB.getInt8Ty(), AllocSize.
getFixedValue());
548 I.setAllocatedType(NewTy);
552bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitGetElementPtrInst(
553 GetElementPtrInst &
I) {
554 Type *Ty =
I.getSourceElementType();
555 if (Ty == TypeMap->remapType(Ty))
559 IRB.SetInsertPoint(&
I);
561 Value *NewGEP = IRB.CreatePtrAdd(
I.getPointerOperand(),
Off,
I.getName(),
563 I.replaceAllUsesWith(NewGEP);
568bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitLoadInst(LoadInst &LI) {
570 Type *IntTy = TypeMap->remapType(Ty);
574 IRB.SetInsertPoint(&LI);
580 SmallVector<unsigned> AggIdxs;
583 [&](
Type *LeafTy,
Type *IntLeafTy, ArrayRef<unsigned> Idxs,
585 Value *Ptr = applyOffset(LI.getPointerOperand(), Off);
586 LoadInst *NewLI = IRB.CreateAlignedLoad(
587 IntLeafTy, Ptr, commonAlignment(LI.getAlign(), Off), Name);
588 NewLI->setVolatile(LI.isVolatile());
589 copyMetadataForLoad(*NewLI, LI);
590 NewLI->setAAMetadata(AATags.adjustForAccess(Off, IntLeafTy, DL));
592 if (LeafTy != IntLeafTy)
593 V = IRB.CreateIntToPtr(NewLI, LeafTy, Name +
".ptr");
594 Agg = IRB.CreateInsertValue(Agg, V, Idxs, Name +
".agg");
601 NLI->mutateType(IntTy);
602 NLI = IRB.Insert(NLI);
605 Value *CastBack = IRB.CreateIntToPtr(NLI, Ty, NLI->getName() +
".ptr");
611bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitStoreInst(StoreInst &SI) {
613 Type *Ty =
V->getType();
614 Type *IntTy = TypeMap->remapType(Ty);
618 IRB.SetInsertPoint(&SI);
622 AAMDNodes AATags =
SI.getAAMetadata();
623 SmallVector<unsigned> AggIdxs;
625 Ty, AggIdxs, 0,
V->getName(),
626 [&](
Type *LeafTy,
Type *IntLeafTy, ArrayRef<unsigned> Idxs,
628 Value *Leaf = IRB.CreateExtractValue(V, Idxs, Name);
629 if (LeafTy != IntLeafTy)
630 Leaf = IRB.CreatePtrToInt(Leaf, IntLeafTy, Name +
".int");
631 auto *NewSI = cast<StoreInst>(SI.clone());
632 NewSI->setAlignment(commonAlignment(SI.getAlign(), Off));
633 NewSI->setOperand(0, Leaf);
634 NewSI->setOperand(1, applyOffset(SI.getPointerOperand(), Off));
636 NewSI->setMetadata(LLVMContext::MD_DIAssignID, nullptr);
638 NewSI->setAAMetadata(AATags.adjustForAccess(Off, IntLeafTy, DL));
640 SI.eraseFromParent();
643 Value *IntV = IRB.CreatePtrToInt(V, IntTy,
V->getName() +
".int");
647 SI.setOperand(0, IntV);
651bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemCpyInst(
663bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemMoveInst(
669 "memmove() on buffer descriptors is not implemented because pointer "
670 "comparison on buffer descriptors isn't implemented\n");
673bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemSetInst(
682bool StoreFatPtrsAsIntsAndExpandMemcpyVisitor::visitMemSetPatternInst(
683 MemSetPatternInst &MSPI) {
712class LegalizeBufferContentTypesVisitor
713 :
public InstVisitor<LegalizeBufferContentTypesVisitor, bool> {
714 friend class InstVisitor<LegalizeBufferContentTypesVisitor, bool>;
718 const DataLayout &
DL;
720 ScalarEvolution *SE =
nullptr;
730 const TargetMachine *TM;
731 const GCNSubtarget *ST =
nullptr;
735 Type *scalarArrayTypeAsVector(
Type *MaybeArrayType);
736 Value *arrayToVector(
Value *V,
Type *TargetType,
const Twine &Name);
737 Value *vectorToArray(
Value *V,
Type *OrigType,
const Twine &Name);
741 struct OobProperties {
743 bool NoWrapFromMax =
false;
745 bool NoPartialOOB =
false;
747 OobProperties() =
delete;
749 OobProperties(
bool NoWrapFromMax,
bool NoPartialOOB)
750 : NoWrapFromMax(NoWrapFromMax), NoPartialOOB(NoPartialOOB) {}
780 uint64_t maxIntrinsicWidth(
Type *Ty, Align
A, OobProperties OobProps);
787 Value *makeLegalNonAggregate(
Value *V,
Type *TargetType,
const Twine &Name);
788 Value *makeIllegalNonAggregate(
Value *V,
Type *OrigType,
const Twine &Name);
802 SmallVectorImpl<VecSlice> &Slices);
804 Value *extractSlice(
Value *Vec, VecSlice S,
const Twine &Name);
805 Value *insertSlice(
Value *Whole,
Value *Part, VecSlice S,
const Twine &Name);
815 Type *intrinsicTypeFor(
Type *LegalType);
817 bool visitLoadImpl(LoadInst &OrigLI,
Type *PartType,
818 SmallVectorImpl<uint32_t> &AggIdxs,
uint64_t AggByteOffset,
819 Value *&Result,
const Twine &Name);
821 std::pair<bool, bool> visitStoreImpl(StoreInst &OrigSI,
Type *PartType,
822 SmallVectorImpl<uint32_t> &AggIdxs,
826 bool visitInstruction(Instruction &
I) {
return false; }
827 bool visitLoadInst(LoadInst &LI);
828 bool visitStoreInst(StoreInst &SI);
831 bool visitIntrinsicInst(IntrinsicInst &
II);
832 bool visitAddrSpaceCastInst(AddrSpaceCastInst &ASCI);
835 LegalizeBufferContentTypesVisitor(
Module &M,
const TargetMachine *TM)
836 : IRB(
M, InstSimplifyFolder(
M.getDataLayout())),
DL(
M.getDataLayout()),
842Type *LegalizeBufferContentTypesVisitor::scalarArrayTypeAsVector(
Type *
T) {
846 Type *ET = AT->getElementType();
849 "should have recursed");
850 if (!
DL.typeSizeEqualsStoreSize(AT))
852 "loading padded arrays from buffer fat pinters should have recursed");
856Value *LegalizeBufferContentTypesVisitor::arrayToVector(
Value *V,
861 unsigned EC = VT->getNumElements();
862 for (
auto I : iota_range<unsigned>(0, EC,
false)) {
863 Value *Elem = IRB.CreateExtractValue(V,
I, Name +
".elem." + Twine(
I));
864 VectorRes = IRB.CreateInsertElement(VectorRes, Elem,
I,
865 Name +
".as.vec." + Twine(
I));
870Value *LegalizeBufferContentTypesVisitor::vectorToArray(
Value *V,
875 unsigned EC = AT->getNumElements();
876 for (
auto I : iota_range<unsigned>(0, EC,
false)) {
877 Value *Elem = IRB.CreateExtractElement(V,
I, Name +
".elem." + Twine(
I));
878 ArrayRes = IRB.CreateInsertValue(ArrayRes, Elem,
I,
879 Name +
".as.array." + Twine(
I));
884LegalizeBufferContentTypesVisitor::OobProperties
885LegalizeBufferContentTypesVisitor::analyzeOobProperties(
Value *Ptr,
Type *Ty,
887 OobProperties
Result(
false,
false);
890 return OobProperties(
true,
true);
896 const SCEV *PtrOp = SE->
getSCEV(Ptr);
902 Value *PtrBaseVal = PtrBase->getValue();
909 auto NumRecordsIfKnown = ZeroBasePointerToNumRecords.
find(PtrBaseVal);
910 if (NumRecordsIfKnown == ZeroBasePointerToNumRecords.
end())
913 unsigned TypeSize =
DL.getTypeStoreSize(Ty).getKnownMinValue();
918 Result.NoWrapFromMax =
true;
922 if (!NumRecordsIfKnown->second)
924 const SCEV *NumRecords = SE->
getSCEV(NumRecordsIfKnown->second);
927 std::optional<unsigned> MaybeNumRecordsWidth =
929 if (!MaybeNumRecordsWidth)
931 unsigned NumRecordsWidth = *MaybeNumRecordsWidth;
932 Type *NumRecordsTy = IRB.getIntNTy(NumRecordsWidth);
934 Type *CompareTy = IRB.getInt64Ty();
942 Result.NoPartialOOB =
true;
944 const SCEV *BoundsDiff =
949 Result.NoPartialOOB =
true;
954LegalizeBufferContentTypesVisitor::maxIntrinsicWidth(
Type *
T, Align
A,
955 OobProperties OobProps) {
961 TypeSize ElemBits =
DL.getTypeSizeInBits(VT->getElementType());
968 if (!OobProps.NoWrapFromMax)
987 if (!OobProps.NoPartialOOB)
992 return Result.value() * 8;
995Type *LegalizeBufferContentTypesVisitor::legalNonAggregateForMemOp(
997 TypeSize
Size =
DL.getTypeStoreSizeInBits(
T);
999 if (!
DL.typeSizeEqualsStoreSize(
T))
1000 T = IRB.getIntNTy(
Size.getFixedValue());
1007 unsigned ElemSize =
DL.getTypeSizeInBits(ElemTy).getFixedValue();
1008 if (
isPowerOf2_32(ElemSize) && ElemSize >= 16 && ElemSize <= MaxWidth) {
1014 Type *BestVectorElemType =
nullptr;
1015 if (
Size.isKnownMultipleOf(32) && MaxWidth >= 32)
1016 BestVectorElemType = IRB.getInt32Ty();
1017 else if (
Size.isKnownMultipleOf(16) && MaxWidth >= 16)
1018 BestVectorElemType = IRB.getInt16Ty();
1020 BestVectorElemType = IRB.getInt8Ty();
1021 unsigned NumCastElems =
1023 if (NumCastElems == 1)
1024 return BestVectorElemType;
1028Value *LegalizeBufferContentTypesVisitor::makeLegalNonAggregate(
1029 Value *V,
Type *TargetType,
const Twine &Name) {
1030 Type *SourceType =
V->getType();
1031 TypeSize SourceSize =
DL.getTypeSizeInBits(SourceType);
1032 TypeSize TargetSize =
DL.getTypeSizeInBits(TargetType);
1033 if (SourceSize != TargetSize) {
1036 Value *AsScalar = IRB.CreateBitCast(V, ShortScalarTy, Name +
".as.scalar");
1037 Value *Zext = IRB.CreateZExt(AsScalar, ByteScalarTy, Name +
".zext");
1039 SourceType = ByteScalarTy;
1041 return IRB.CreateBitCast(V, TargetType, Name +
".legal");
1044Value *LegalizeBufferContentTypesVisitor::makeIllegalNonAggregate(
1045 Value *V,
Type *OrigType,
const Twine &Name) {
1046 Type *LegalType =
V->getType();
1047 TypeSize LegalSize =
DL.getTypeSizeInBits(LegalType);
1048 TypeSize OrigSize =
DL.getTypeSizeInBits(OrigType);
1049 if (LegalSize != OrigSize) {
1052 Value *AsScalar = IRB.CreateBitCast(V, ByteScalarTy, Name +
".bytes.cast");
1053 Value *Trunc = IRB.CreateTrunc(AsScalar, ShortScalarTy, Name +
".trunc");
1054 return IRB.CreateBitCast(Trunc, OrigType, Name +
".orig");
1056 return IRB.CreateBitCast(V, OrigType, Name +
".real.ty");
1059Type *LegalizeBufferContentTypesVisitor::intrinsicTypeFor(
Type *LegalType) {
1063 Type *ET = VT->getElementType();
1066 if (VT->getNumElements() == 1)
1068 if (
DL.getTypeSizeInBits(LegalType) == 96 &&
DL.getTypeSizeInBits(ET) < 32)
1071 switch (VT->getNumElements()) {
1075 return IRB.getInt8Ty();
1077 return IRB.getInt16Ty();
1079 return IRB.getInt32Ty();
1089void LegalizeBufferContentTypesVisitor::getVecSlices(
1090 Type *
T,
uint64_t MaxWidth, SmallVectorImpl<VecSlice> &Slices) {
1097 DL.getTypeSizeInBits(VT->getElementType()).getFixedValue();
1099 uint64_t ElemsPer4Words = 128 / ElemBitWidth;
1100 uint64_t ElemsPer2Words = ElemsPer4Words / 2;
1101 uint64_t ElemsPerWord = ElemsPer2Words / 2;
1102 uint64_t ElemsPerShort = ElemsPerWord / 2;
1103 uint64_t ElemsPerByte = ElemsPerShort / 2;
1107 uint64_t ElemsPer3Words = ElemsPerWord * 3;
1109 uint64_t TotalElems = VT->getNumElements();
1111 auto TrySlice = [&](
unsigned MaybeLen,
unsigned Width) {
1112 if (MaybeLen > 0 && Width <= MaxWidth && Index + MaybeLen <= TotalElems) {
1113 VecSlice Slice{
Index, MaybeLen};
1120 while (Index < TotalElems) {
1121 TrySlice(ElemsPer4Words, 128) || TrySlice(ElemsPer3Words, 96) ||
1122 TrySlice(ElemsPer2Words, 64) || TrySlice(ElemsPerWord, 32) ||
1123 TrySlice(ElemsPerShort, 16) || TrySlice(ElemsPerByte, 8);
1127Value *LegalizeBufferContentTypesVisitor::extractSlice(
Value *Vec, VecSlice S,
1128 const Twine &Name) {
1132 if (S.Length == VecVT->getNumElements() && S.Index == 0)
1135 return IRB.CreateExtractElement(Vec, S.Index,
1136 Name +
".slice." + Twine(S.Index));
1138 llvm::iota_range<int>(S.Index, S.Index + S.Length,
false));
1139 return IRB.CreateShuffleVector(Vec, Mask, Name +
".slice." + Twine(S.Index));
1142Value *LegalizeBufferContentTypesVisitor::insertSlice(
Value *Whole,
Value *Part,
1144 const Twine &Name) {
1148 if (S.Length == WholeVT->getNumElements() && S.Index == 0)
1150 if (S.Length == 1) {
1151 return IRB.CreateInsertElement(Whole, Part, S.Index,
1152 Name +
".slice." + Twine(S.Index));
1157 SmallVector<int> ExtPartMask(NumElems, -1);
1162 Value *ExtPart = IRB.CreateShuffleVector(Part, ExtPartMask,
1163 Name +
".ext." + Twine(S.Index));
1165 SmallVector<int>
Mask =
1170 return IRB.CreateShuffleVector(Whole, ExtPart, Mask,
1171 Name +
".parts." + Twine(S.Index));
1174bool LegalizeBufferContentTypesVisitor::visitLoadImpl(
1175 LoadInst &OrigLI,
Type *PartType, SmallVectorImpl<uint32_t> &AggIdxs,
1178 const StructLayout *Layout =
DL.getStructLayout(ST);
1180 for (
auto [
I, ElemTy,
Offset] :
1183 Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
1184 AggByteOff +
Offset.getFixedValue(), Result,
1185 Name +
"." + Twine(
I));
1191 Type *ElemTy = AT->getElementType();
1194 TypeSize ElemAllocSize =
DL.getTypeAllocSize(ElemTy);
1196 for (
auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
1199 Changed |= visitLoadImpl(OrigLI, ElemTy, AggIdxs,
1201 Result, Name + Twine(
I));
1211 Type *ArrayAsVecType = scalarArrayTypeAsVector(PartType);
1212 OobProperties OobProps =
1214 uint64_t MaxWidth = maxIntrinsicWidth(ArrayAsVecType, PartAlign, OobProps);
1215 Type *LegalType = legalNonAggregateForMemOp(ArrayAsVecType, MaxWidth);
1218 getVecSlices(LegalType, MaxWidth, Slices);
1219 bool HasSlices = Slices.
size() > 1;
1220 bool IsAggPart = !AggIdxs.
empty();
1222 if (!HasSlices && !IsAggPart) {
1223 Type *LoadableType = intrinsicTypeFor(LegalType);
1224 if (LoadableType == PartType)
1227 IRB.SetInsertPoint(&OrigLI);
1229 NLI->mutateType(LoadableType);
1230 NLI = IRB.Insert(NLI);
1231 NLI->setName(Name +
".loadable");
1233 LoadsRes = IRB.CreateBitCast(NLI, LegalType, Name +
".from.loadable");
1235 IRB.SetInsertPoint(&OrigLI);
1243 unsigned ElemBytes =
DL.getTypeStoreSize(ElemType);
1245 if (IsAggPart && Slices.
empty())
1247 for (VecSlice S : Slices) {
1250 int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
1252 Value *NewPtr = IRB.CreateGEP(
1254 OrigPtr->
getName() +
".off.ptr." + Twine(ByteOffset),
1257 Type *LoadableType = intrinsicTypeFor(SliceType);
1258 LoadInst *NewLI = IRB.CreateAlignedLoad(
1260 Name +
".off." + Twine(ByteOffset));
1266 Value *
Loaded = IRB.CreateBitCast(NewLI, SliceType,
1267 NewLI->
getName() +
".from.loadable");
1268 LoadsRes = insertSlice(LoadsRes, Loaded, S, Name);
1271 if (LegalType != ArrayAsVecType)
1272 LoadsRes = makeIllegalNonAggregate(LoadsRes, ArrayAsVecType, Name);
1273 if (ArrayAsVecType != PartType)
1274 LoadsRes = vectorToArray(LoadsRes, PartType, Name);
1277 Result = IRB.CreateInsertValue(Result, LoadsRes, AggIdxs, Name);
1283bool LegalizeBufferContentTypesVisitor::visitLoadInst(LoadInst &LI) {
1287 SmallVector<uint32_t> AggIdxs;
1290 bool Changed = visitLoadImpl(LI, OrigType, AggIdxs, 0, Result, LI.
getName());
1299std::pair<bool, bool> LegalizeBufferContentTypesVisitor::visitStoreImpl(
1300 StoreInst &OrigSI,
Type *PartType, SmallVectorImpl<uint32_t> &AggIdxs,
1301 uint64_t AggByteOff,
const Twine &Name) {
1303 const StructLayout *Layout =
DL.getStructLayout(ST);
1305 for (
auto [
I, ElemTy,
Offset] :
1308 Changed |= std::get<0>(visitStoreImpl(OrigSI, ElemTy, AggIdxs,
1309 AggByteOff +
Offset.getFixedValue(),
1310 Name +
"." + Twine(
I)));
1313 return std::make_pair(
Changed,
false);
1316 Type *ElemTy = AT->getElementType();
1319 TypeSize ElemAllocSize =
DL.getTypeAllocSize(ElemTy);
1321 for (
auto I : llvm::iota_range<uint32_t>(0, AT->getNumElements(),
1324 Changed |= std::get<0>(visitStoreImpl(
1325 OrigSI, ElemTy, AggIdxs,
1329 return std::make_pair(
Changed,
false);
1334 Value *NewData = OrigData;
1336 bool IsAggPart = !AggIdxs.
empty();
1338 NewData = IRB.CreateExtractValue(NewData, AggIdxs, Name);
1340 Type *ArrayAsVecType = scalarArrayTypeAsVector(PartType);
1341 if (ArrayAsVecType != PartType) {
1342 NewData = arrayToVector(NewData, ArrayAsVecType, Name);
1346 OobProperties OobProps =
1348 uint64_t MaxWidth = maxIntrinsicWidth(ArrayAsVecType, PartAlign, OobProps);
1349 Type *LegalType = legalNonAggregateForMemOp(ArrayAsVecType, MaxWidth);
1350 if (LegalType != ArrayAsVecType) {
1351 NewData = makeLegalNonAggregate(NewData, LegalType, Name);
1355 getVecSlices(LegalType, MaxWidth, Slices);
1356 bool NeedToSplit = Slices.
size() > 1 || IsAggPart;
1358 Type *StorableType = intrinsicTypeFor(LegalType);
1359 if (StorableType == PartType)
1360 return std::make_pair(
false,
false);
1361 NewData = IRB.CreateBitCast(NewData, StorableType, Name +
".storable");
1363 return std::make_pair(
true,
true);
1368 if (IsAggPart && Slices.
empty())
1370 unsigned ElemBytes =
DL.getTypeStoreSize(ElemType);
1372 for (VecSlice S : Slices) {
1375 int64_t ByteOffset = AggByteOff + S.Index * ElemBytes;
1376 Value *NewPtr = IRB.CreateGEP(
1377 IRB.getInt8Ty(), OrigPtr, IRB.getInt32(ByteOffset),
1378 OrigPtr->
getName() +
".part." + Twine(S.Index),
1381 Value *DataSlice = extractSlice(NewData, S, Name);
1382 Type *StorableType = intrinsicTypeFor(SliceType);
1383 DataSlice = IRB.CreateBitCast(DataSlice, StorableType,
1384 DataSlice->
getName() +
".storable");
1388 NewSI->setOperand(0, DataSlice);
1389 NewSI->setOperand(1, NewPtr);
1392 return std::make_pair(
true,
false);
1395bool LegalizeBufferContentTypesVisitor::visitStoreInst(StoreInst &SI) {
1398 IRB.SetInsertPoint(&SI);
1399 SmallVector<uint32_t> AggIdxs;
1400 Value *OrigData =
SI.getValueOperand();
1401 auto [
Changed, ModifiedInPlace] =
1402 visitStoreImpl(SI, OrigData->
getType(), AggIdxs, 0, OrigData->
getName());
1403 if (
Changed && !ModifiedInPlace)
1404 SI.eraseFromParent();
1408bool LegalizeBufferContentTypesVisitor::visitAddrSpaceCastInst(
1409 AddrSpaceCastInst &AI) {
1414 auto Record = ZeroBasePointerToNumRecords.
find(Src);
1415 if (Record != ZeroBasePointerToNumRecords.
end())
1416 ZeroBasePointerToNumRecords.
insert({&AI,
Record->second});
1418 ZeroBasePointerToNumRecords.
insert({&AI,
nullptr});
1422bool LegalizeBufferContentTypesVisitor::visitIntrinsicInst(IntrinsicInst &
II) {
1423 if (
II.getIntrinsicID() != Intrinsic::amdgcn_make_buffer_rsrc)
1425 ZeroBasePointerToNumRecords.
insert({&
II,
II.getOperand(2)});
1429bool LegalizeBufferContentTypesVisitor::processFunction(
Function &
F,
1430 ScalarEvolution *SE) {
1437 ZeroBasePointerToNumRecords.
clear();
1444static std::pair<Constant *, Constant *>
1447 return std::make_pair(
C->getAggregateElement(0u),
C->getAggregateElement(1u));
1452class FatPtrConstMaterializer final :
public ValueMaterializer {
1453 BufferFatPtrToStructTypeMap *TypeMap;
1459 ValueMapper InternalMapper;
1461 Constant *materializeBufferFatPtrConst(Constant *
C);
1465 FatPtrConstMaterializer(BufferFatPtrToStructTypeMap *TypeMap,
1468 InternalMapper(UnderlyingMap,
RF_None, TypeMap, this) {}
1469 ~FatPtrConstMaterializer() =
default;
1475Constant *FatPtrConstMaterializer::materializeBufferFatPtrConst(Constant *
C) {
1476 Type *SrcTy =
C->getType();
1478 if (
C->isNullValue())
1479 return ConstantAggregateZero::getNullValue(NewTy);
1492 if (Constant *S =
VC->getSplatValue()) {
1497 auto EC =
VC->getType()->getElementCount();
1503 for (
Value *
Op :
VC->operand_values()) {
1518 "fat pointer) values are not supported");
1522 "constant exprs containing ptr addrspace(7) (buffer "
1523 "fat pointer) values should have been expanded earlier");
1528Value *FatPtrConstMaterializer::materialize(
Value *V) {
1536 return materializeBufferFatPtrConst(
C);
1544class SplitPtrStructs :
public InstVisitor<SplitPtrStructs, PtrParts> {
1587 void processConditionals();
1636void SplitPtrStructs::copyMetadata(
Value *Dest,
Value *Src) {
1640 if (!DestI || !SrcI)
1643 DestI->copyMetadata(*SrcI);
1648 "of something that wasn't rewritten");
1649 auto *RsrcEntry = &RsrcParts[
V];
1650 auto *OffEntry = &OffParts[
V];
1651 if (*RsrcEntry && *OffEntry)
1652 return {*RsrcEntry, *OffEntry};
1656 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1659 IRBuilder<InstSimplifyFolder>::InsertPointGuard Guard(IRB);
1664 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1667 IRB.SetInsertPoint(*
I->getInsertionPointAfterDef());
1668 IRB.SetCurrentDebugLocation(
I->getDebugLoc());
1670 IRB.SetInsertPointPastAllocas(
A->getParent());
1671 IRB.SetCurrentDebugLocation(
DebugLoc());
1673 Value *Rsrc = IRB.CreateExtractValue(V, 0,
V->getName() +
".rsrc");
1674 Value *
Off = IRB.CreateExtractValue(V, 1,
V->getName() +
".off");
1675 return {*RsrcEntry = Rsrc, *OffEntry =
Off};
1688 V =
GEP->getPointerOperand();
1690 V = ASC->getPointerOperand();
1694void SplitPtrStructs::getPossibleRsrcRoots(Instruction *
I,
1695 SmallPtrSetImpl<Value *> &Roots,
1696 SmallPtrSetImpl<Value *> &Seen) {
1700 for (
Value *In :
PHI->incoming_values()) {
1707 if (!Seen.
insert(SI).second)
1722void SplitPtrStructs::processConditionals() {
1723 SmallDenseMap<Value *, Value *> FoundRsrcs;
1724 SmallPtrSet<Value *, 4> Roots;
1725 SmallPtrSet<Value *, 4> Seen;
1726 for (Instruction *
I : Conditionals) {
1728 Value *Rsrc = RsrcParts[
I];
1730 assert(Rsrc &&
Off &&
"must have visited conditionals by now");
1732 std::optional<Value *> MaybeRsrc;
1733 auto MaybeFoundRsrc = FoundRsrcs.
find(
I);
1734 if (MaybeFoundRsrc != FoundRsrcs.
end()) {
1735 MaybeRsrc = MaybeFoundRsrc->second;
1737 IRBuilder<InstSimplifyFolder>::InsertPointGuard Guard(IRB);
1740 getPossibleRsrcRoots(
I, Roots, Seen);
1743 for (
Value *V : Roots)
1745 for (
Value *V : Seen)
1757 if (Diff.size() == 1) {
1758 Value *RootVal = *Diff.begin();
1762 MaybeRsrc = std::get<0>(getPtrParts(RootVal));
1764 MaybeRsrc = RootVal;
1772 IRB.SetInsertPoint(*
PHI->getInsertionPointAfterDef());
1773 IRB.SetCurrentDebugLocation(
PHI->getDebugLoc());
1775 NewRsrc = *MaybeRsrc;
1778 auto *RsrcPHI = IRB.CreatePHI(RsrcTy,
PHI->getNumIncomingValues());
1779 RsrcPHI->takeName(Rsrc);
1780 for (
auto [V, BB] :
llvm::zip(
PHI->incoming_values(),
PHI->blocks())) {
1781 Value *VRsrc = std::get<0>(getPtrParts(V));
1782 RsrcPHI->addIncoming(VRsrc, BB);
1784 copyMetadata(RsrcPHI,
PHI);
1789 auto *NewOff = IRB.CreatePHI(OffTy,
PHI->getNumIncomingValues());
1790 NewOff->takeName(
Off);
1791 for (
auto [V, BB] :
llvm::zip(
PHI->incoming_values(),
PHI->blocks())) {
1792 assert(OffParts.
count(V) &&
"An offset part had to be created by now");
1793 Value *VOff = std::get<1>(getPtrParts(V));
1794 NewOff->addIncoming(VOff, BB);
1796 copyMetadata(NewOff,
PHI);
1806 RsrcInst->replaceAllUsesWith(NewRsrc);
1810 OffInst->replaceAllUsesWith(NewOff);
1815 for (
Value *V : Seen)
1816 FoundRsrcs[
V] = NewRsrc;
1821 if (RsrcInst != *MaybeRsrc) {
1823 RsrcInst->replaceAllUsesWith(*MaybeRsrc);
1826 for (
Value *V : Seen)
1827 FoundRsrcs[
V] = *MaybeRsrc;
1835void SplitPtrStructs::killAndReplaceSplitInstructions(
1836 SmallVectorImpl<Instruction *> &Origs) {
1837 for (Instruction *
I : ConditionalTemps)
1838 I->eraseFromParent();
1840 for (Instruction *
I : Origs) {
1846 for (DbgVariableRecord *Dbg : Dbgs) {
1847 auto &
DL =
I->getDataLayout();
1849 "We should've RAUW'd away loads, stores, etc. at this point");
1850 DbgVariableRecord *OffDbg =
Dbg->clone();
1851 auto [Rsrc,
Off] = getPtrParts(
I);
1853 int64_t RsrcSz =
DL.getTypeSizeInBits(Rsrc->
getType());
1854 int64_t OffSz =
DL.getTypeSizeInBits(
Off->getType());
1856 std::optional<DIExpression *> RsrcExpr =
1859 std::optional<DIExpression *> OffExpr =
1870 Dbg->setExpression(*RsrcExpr);
1871 Dbg->replaceVariableLocationOp(
I, Rsrc);
1878 I->replaceUsesWithIf(
Poison, [&](
const Use &U) ->
bool {
1884 if (
I->use_empty()) {
1885 I->eraseFromParent();
1888 IRB.SetInsertPoint(*
I->getInsertionPointAfterDef());
1889 IRB.SetCurrentDebugLocation(
I->getDebugLoc());
1890 auto [Rsrc,
Off] = getPtrParts(
I);
1892 Struct = IRB.CreateInsertValue(Struct, Rsrc, 0);
1893 Struct = IRB.CreateInsertValue(Struct,
Off, 1);
1894 copyMetadata(Struct,
I);
1896 I->replaceAllUsesWith(Struct);
1897 I->eraseFromParent();
1901void SplitPtrStructs::setAlign(CallInst *Intr, Align
A,
unsigned RsrcArgIdx) {
1903 Intr->
addParamAttr(RsrcArgIdx, Attribute::getWithAlignment(Ctx,
A));
1909 case AtomicOrdering::Release:
1910 case AtomicOrdering::AcquireRelease:
1911 case AtomicOrdering::SequentiallyConsistent:
1912 IRB.CreateFence(AtomicOrdering::Release, SSID);
1922 case AtomicOrdering::Acquire:
1923 case AtomicOrdering::AcquireRelease:
1924 case AtomicOrdering::SequentiallyConsistent:
1925 IRB.CreateFence(AtomicOrdering::Acquire, SSID);
1932Value *SplitPtrStructs::handleMemoryInst(Instruction *
I,
Value *Arg,
Value *Ptr,
1933 Type *Ty, Align Alignment,
1936 IRB.SetInsertPoint(
I);
1938 auto [Rsrc,
Off] = getPtrParts(Ptr);
1941 Args.push_back(Arg);
1942 Args.push_back(Rsrc);
1944 insertPreMemOpFence(Order, SSID);
1948 Args.push_back(IRB.getInt32(0));
1953 Args.push_back(IRB.getInt32(Aux));
1957 IID = Order == AtomicOrdering::NotAtomic
1958 ? Intrinsic::amdgcn_raw_ptr_buffer_load
1959 : Intrinsic::amdgcn_raw_ptr_atomic_buffer_load;
1961 IID = Intrinsic::amdgcn_raw_ptr_buffer_store;
1963 switch (RMW->getOperation()) {
1965 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_swap;
1968 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_add;
1971 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub;
1974 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_and;
1977 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_or;
1980 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_xor;
1983 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_smax;
1986 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_smin;
1989 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_umax;
1992 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_umin;
1995 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fadd;
1998 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmax;
2001 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmin;
2004 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_cond_sub_u32;
2007 IID = Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub_clamp_u32;
2011 "atomic floating point subtraction not supported for "
2012 "buffer resources and should've been expanded away");
2017 "atomic floating point fmaximum not supported for "
2018 "buffer resources and should've been expanded away");
2023 "atomic floating point fminimum not supported for "
2024 "buffer resources and should've been expanded away");
2029 "atomic floating point fmaximumnum not supported for "
2030 "buffer resources and should've been expanded away");
2035 "atomic floating point fminimumnum not supported for "
2036 "buffer resources and should've been expanded away");
2041 "atomic nand not supported for buffer resources and "
2042 "should've been expanded away");
2047 "wrapping increment/decrement not supported for "
2048 "buffer resources and should've been expanded away");
2055 CallInst *
Call = IRB.CreateIntrinsicWithoutFolding(IID, Ty, Args);
2056 copyMetadata(
Call,
I);
2057 setAlign(
Call, Alignment, Arg ? 1 : 0);
2060 insertPostMemOpFence(Order, SSID);
2064 I->replaceAllUsesWith(
Call);
2068PtrParts SplitPtrStructs::visitInstruction(Instruction &
I) {
2069 return {
nullptr,
nullptr};
2072PtrParts SplitPtrStructs::visitLoadInst(LoadInst &LI) {
2074 return {
nullptr,
nullptr};
2078 return {
nullptr,
nullptr};
2081PtrParts SplitPtrStructs::visitStoreInst(StoreInst &SI) {
2083 return {
nullptr,
nullptr};
2084 Value *Arg =
SI.getValueOperand();
2085 handleMemoryInst(&SI, Arg,
SI.getPointerOperand(), Arg->
getType(),
2086 SI.getAlign(),
SI.getOrdering(),
SI.isVolatile(),
2087 SI.getSyncScopeID());
2088 return {
nullptr,
nullptr};
2091PtrParts SplitPtrStructs::visitAtomicRMWInst(AtomicRMWInst &AI) {
2093 return {
nullptr,
nullptr};
2098 return {
nullptr,
nullptr};
2103PtrParts SplitPtrStructs::visitAtomicCmpXchgInst(AtomicCmpXchgInst &AI) {
2106 return {
nullptr,
nullptr};
2107 IRB.SetInsertPoint(&AI);
2112 bool IsNonTemporal = AI.
getMetadata(LLVMContext::MD_nontemporal);
2114 auto [Rsrc,
Off] = getPtrParts(Ptr);
2115 insertPreMemOpFence(Order, SSID);
2122 CallInst *
Call = IRB.CreateIntrinsicWithoutFolding(
2123 Intrinsic::amdgcn_raw_ptr_buffer_atomic_cmpswap, Ty,
2125 IRB.getInt32(0), IRB.getInt32(Aux)});
2126 copyMetadata(
Call, &AI);
2129 insertPostMemOpFence(Order, SSID);
2132 Res = IRB.CreateInsertValue(Res,
Call, 0);
2134 Res = IRB.CreateInsertValue(Res, Succeeded, 1);
2137 return {
nullptr,
nullptr};
2140PtrParts SplitPtrStructs::visitGetElementPtrInst(GetElementPtrInst &
GEP) {
2141 using namespace llvm::PatternMatch;
2142 Value *Ptr =
GEP.getPointerOperand();
2144 return {
nullptr,
nullptr};
2145 IRB.SetInsertPoint(&
GEP);
2147 auto [Rsrc,
Off] = getPtrParts(Ptr);
2148 const DataLayout &
DL =
GEP.getDataLayout();
2149 bool IsNUW =
GEP.hasNoUnsignedWrap();
2150 bool IsNUSW =
GEP.hasNoUnsignedSignedWrap();
2161 GEP.mutateType(FatPtrTy);
2163 GEP.mutateType(ResTy);
2165 if (BroadcastsPtr) {
2166 Rsrc = IRB.CreateVectorSplat(ResRsrcVecTy->getElementCount(), Rsrc,
2168 Off = IRB.CreateVectorSplat(ResRsrcVecTy->getElementCount(),
Off,
2176 bool HasNonNegativeOff =
false;
2178 HasNonNegativeOff = !CI->isNegative();
2184 NewOff = IRB.CreateAdd(
Off, OffAccum,
"",
2185 IsNUW || (IsNUSW && HasNonNegativeOff),
2188 copyMetadata(NewOff, &
GEP);
2191 return {Rsrc, NewOff};
2194PtrParts SplitPtrStructs::visitPtrToIntInst(PtrToIntInst &PI) {
2197 return {
nullptr,
nullptr};
2198 IRB.SetInsertPoint(&PI);
2203 auto [Rsrc,
Off] = getPtrParts(Ptr);
2209 Res = IRB.CreateIntCast(
Off, ResTy,
false,
2212 Value *RsrcInt = IRB.CreatePtrToInt(Rsrc, ResTy, PI.
getName() +
".rsrc");
2213 Value *Shl = IRB.CreateShl(
2216 "", Width >= FatPtrWidth, Width > FatPtrWidth);
2217 Value *OffCast = IRB.CreateIntCast(
Off, ResTy,
false,
2219 Res = IRB.CreateOr(Shl, OffCast);
2222 copyMetadata(Res, &PI);
2226 return {
nullptr,
nullptr};
2229PtrParts SplitPtrStructs::visitPtrToAddrInst(PtrToAddrInst &PA) {
2232 return {
nullptr,
nullptr};
2233 IRB.SetInsertPoint(&PA);
2235 auto [Rsrc,
Off] = getPtrParts(Ptr);
2237 copyMetadata(Res, &PA);
2241 return {
nullptr,
nullptr};
2244PtrParts SplitPtrStructs::visitIntToPtrInst(IntToPtrInst &IP) {
2246 return {
nullptr,
nullptr};
2247 IRB.SetInsertPoint(&IP);
2256 Type *RsrcTy = RetTy->getElementType(0);
2257 Type *OffTy = RetTy->getElementType(1);
2266 RsrcInt = IRB.CreateIntCast(RsrcPart, RsrcIntTy,
false);
2268 Value *Rsrc = IRB.CreateIntToPtr(RsrcInt, RsrcTy, IP.
getName() +
".rsrc");
2270 IRB.CreateIntCast(
Int, OffTy,
false, IP.
getName() +
".off");
2272 copyMetadata(Rsrc, &IP);
2277PtrParts SplitPtrStructs::visitAddrSpaceCastInst(AddrSpaceCastInst &
I) {
2281 return {
nullptr,
nullptr};
2282 IRB.SetInsertPoint(&
I);
2285 if (
In->getType() ==
I.getType()) {
2286 auto [Rsrc,
Off] = getPtrParts(In);
2292 Type *RsrcTy = ResTy->getElementType(0);
2293 Type *OffTy = ResTy->getElementType(1);
2299 if (InConst && InConst->isNullValue()) {
2302 return {NullRsrc, ZeroOff};
2308 return {PoisonRsrc, PoisonOff};
2314 return {UndefRsrc, UndefOff};
2319 "only buffer resources (addrspace 8) and null/poison pointers can be "
2320 "cast to buffer fat pointers (addrspace 7)");
2322 return {
In, ZeroOff};
2325PtrParts SplitPtrStructs::visitICmpInst(ICmpInst &Cmp) {
2328 return {
nullptr,
nullptr};
2330 IRB.SetInsertPoint(&Cmp);
2331 ICmpInst::Predicate Pred =
Cmp.getPredicate();
2333 assert((Pred == ICmpInst::ICMP_EQ || Pred == ICmpInst::ICMP_NE) &&
2334 "Pointer comparison is only equal or unequal");
2335 auto [LhsRsrc, LhsOff] = getPtrParts(Lhs);
2336 auto [RhsRsrc, RhsOff] = getPtrParts(Rhs);
2337 Value *Res = IRB.CreateICmp(Pred, LhsOff, RhsOff);
2338 copyMetadata(Res, &Cmp);
2341 Cmp.replaceAllUsesWith(Res);
2342 return {
nullptr,
nullptr};
2345PtrParts SplitPtrStructs::visitFreezeInst(FreezeInst &
I) {
2347 return {
nullptr,
nullptr};
2348 IRB.SetInsertPoint(&
I);
2349 auto [Rsrc,
Off] = getPtrParts(
I.getOperand(0));
2351 Value *RsrcRes = IRB.CreateFreeze(Rsrc,
I.getName() +
".rsrc");
2352 copyMetadata(RsrcRes, &
I);
2353 Value *OffRes = IRB.CreateFreeze(
Off,
I.getName() +
".off");
2354 copyMetadata(OffRes, &
I);
2356 return {RsrcRes, OffRes};
2359PtrParts SplitPtrStructs::visitExtractElementInst(ExtractElementInst &
I) {
2361 return {
nullptr,
nullptr};
2362 IRB.SetInsertPoint(&
I);
2363 Value *Vec =
I.getVectorOperand();
2364 Value *Idx =
I.getIndexOperand();
2365 auto [Rsrc,
Off] = getPtrParts(Vec);
2367 Value *RsrcRes = IRB.CreateExtractElement(Rsrc, Idx,
I.getName() +
".rsrc");
2368 copyMetadata(RsrcRes, &
I);
2369 Value *OffRes = IRB.CreateExtractElement(
Off, Idx,
I.getName() +
".off");
2370 copyMetadata(OffRes, &
I);
2372 return {RsrcRes, OffRes};
2375PtrParts SplitPtrStructs::visitInsertElementInst(InsertElementInst &
I) {
2379 return {
nullptr,
nullptr};
2380 IRB.SetInsertPoint(&
I);
2381 Value *Vec =
I.getOperand(0);
2382 Value *Elem =
I.getOperand(1);
2383 Value *Idx =
I.getOperand(2);
2384 auto [VecRsrc, VecOff] = getPtrParts(Vec);
2385 auto [ElemRsrc, ElemOff] = getPtrParts(Elem);
2388 IRB.CreateInsertElement(VecRsrc, ElemRsrc, Idx,
I.getName() +
".rsrc");
2389 copyMetadata(RsrcRes, &
I);
2391 IRB.CreateInsertElement(VecOff, ElemOff, Idx,
I.getName() +
".off");
2392 copyMetadata(OffRes, &
I);
2394 return {RsrcRes, OffRes};
2397PtrParts SplitPtrStructs::visitShuffleVectorInst(ShuffleVectorInst &
I) {
2400 return {
nullptr,
nullptr};
2401 IRB.SetInsertPoint(&
I);
2404 Value *V2 =
I.getOperand(1);
2405 ArrayRef<int>
Mask =
I.getShuffleMask();
2406 auto [V1Rsrc, V1Off] = getPtrParts(
V1);
2407 auto [V2Rsrc, V2Off] = getPtrParts(V2);
2410 IRB.CreateShuffleVector(V1Rsrc, V2Rsrc, Mask,
I.getName() +
".rsrc");
2411 copyMetadata(RsrcRes, &
I);
2413 IRB.CreateShuffleVector(V1Off, V2Off, Mask,
I.getName() +
".off");
2414 copyMetadata(OffRes, &
I);
2416 return {RsrcRes, OffRes};
2419PtrParts SplitPtrStructs::visitPHINode(PHINode &
PHI) {
2421 return {
nullptr,
nullptr};
2422 IRB.SetInsertPoint(*
PHI.getInsertionPointAfterDef());
2428 Value *TmpRsrc = IRB.CreateExtractValue(&
PHI, 0,
PHI.getName() +
".rsrc");
2429 Value *TmpOff = IRB.CreateExtractValue(&
PHI, 1,
PHI.getName() +
".off");
2430 Conditionals.push_back(&
PHI);
2432 return {TmpRsrc, TmpOff};
2435PtrParts SplitPtrStructs::visitSelectInst(SelectInst &SI) {
2437 return {
nullptr,
nullptr};
2438 IRB.SetInsertPoint(&SI);
2443 auto [TrueRsrc, TrueOff] = getPtrParts(
True);
2444 auto [FalseRsrc, FalseOff] = getPtrParts(
False);
2447 IRB.CreateSelect(
Cond, TrueRsrc, FalseRsrc,
SI.getName() +
".rsrc", &SI);
2448 copyMetadata(RsrcRes, &SI);
2449 Conditionals.push_back(&SI);
2451 IRB.CreateSelect(
Cond, TrueOff, FalseOff,
SI.getName() +
".off", &SI);
2452 copyMetadata(OffRes, &SI);
2454 return {RsrcRes, OffRes};
2465 case Intrinsic::amdgcn_make_buffer_rsrc:
2466 case Intrinsic::ptrmask:
2467 case Intrinsic::invariant_start:
2468 case Intrinsic::invariant_end:
2469 case Intrinsic::launder_invariant_group:
2470 case Intrinsic::memcpy:
2471 case Intrinsic::memcpy_inline:
2472 case Intrinsic::memmove:
2473 case Intrinsic::memset:
2474 case Intrinsic::memset_inline:
2475 case Intrinsic::experimental_memset_pattern:
2476 case Intrinsic::amdgcn_load_to_lds:
2477 case Intrinsic::amdgcn_load_async_to_lds:
2482PtrParts SplitPtrStructs::visitIntrinsicInst(IntrinsicInst &
I) {
2487 case Intrinsic::amdgcn_make_buffer_rsrc: {
2489 return {
nullptr,
nullptr};
2491 Value *Stride =
I.getArgOperand(1);
2492 Value *NumRecords =
I.getArgOperand(2);
2495 Type *RsrcType = SplitType->getElementType(0);
2496 Type *OffType = SplitType->getElementType(1);
2497 IRB.SetInsertPoint(&
I);
2498 Value *Rsrc = IRB.CreateIntrinsic(
2499 IID, {RsrcType,
Base->getType(), NumRecords->
getType()},
2501 copyMetadata(Rsrc, &
I);
2505 return {Rsrc,
Zero};
2507 case Intrinsic::ptrmask: {
2508 Value *Ptr =
I.getArgOperand(0);
2510 return {
nullptr,
nullptr};
2512 IRB.SetInsertPoint(&
I);
2513 auto [Rsrc,
Off] = getPtrParts(Ptr);
2514 if (
Mask->getType() !=
Off->getType())
2516 "pointer (data layout not set up correctly?)");
2517 Value *OffRes = IRB.CreateAnd(
Off, Mask,
I.getName() +
".off");
2518 copyMetadata(OffRes, &
I);
2520 return {Rsrc, OffRes};
2524 case Intrinsic::invariant_start: {
2525 Value *Ptr =
I.getArgOperand(1);
2527 return {
nullptr,
nullptr};
2528 IRB.SetInsertPoint(&
I);
2529 auto [Rsrc,
Off] = getPtrParts(Ptr);
2531 auto *NewRsrc = IRB.CreateIntrinsic(IID, {NewTy}, {
I.getOperand(0), Rsrc});
2532 copyMetadata(NewRsrc, &
I);
2535 I.replaceAllUsesWith(NewRsrc);
2536 return {
nullptr,
nullptr};
2538 case Intrinsic::invariant_end: {
2539 Value *RealPtr =
I.getArgOperand(2);
2541 return {
nullptr,
nullptr};
2542 IRB.SetInsertPoint(&
I);
2543 Value *RealRsrc = getPtrParts(RealPtr).first;
2544 Value *InvPtr =
I.getArgOperand(0);
2546 Value *NewRsrc = IRB.CreateIntrinsic(IID, {RealRsrc->
getType()},
2547 {InvPtr,
Size, RealRsrc});
2548 copyMetadata(NewRsrc, &
I);
2551 I.replaceAllUsesWith(NewRsrc);
2552 return {
nullptr,
nullptr};
2554 case Intrinsic::launder_invariant_group: {
2555 Value *Ptr =
I.getArgOperand(0);
2557 return {
nullptr,
nullptr};
2558 IRB.SetInsertPoint(&
I);
2559 auto [Rsrc,
Off] = getPtrParts(Ptr);
2560 Value *NewRsrc = IRB.CreateIntrinsic(IID, {Rsrc->
getType()}, {Rsrc});
2561 copyMetadata(NewRsrc, &
I);
2564 return {NewRsrc,
Off};
2566 case Intrinsic::amdgcn_load_to_lds:
2567 case Intrinsic::amdgcn_load_async_to_lds: {
2568 Value *Ptr =
I.getArgOperand(0);
2570 return {
nullptr,
nullptr};
2571 IRB.SetInsertPoint(&
I);
2572 auto [Rsrc,
Off] = getPtrParts(Ptr);
2573 Value *LDSPtr =
I.getArgOperand(1);
2574 Value *LoadSize =
I.getArgOperand(2);
2575 Value *ImmOff =
I.getArgOperand(3);
2576 Value *Aux =
I.getArgOperand(4);
2577 Value *SOffset = IRB.getInt32(0);
2579 IID == Intrinsic::amdgcn_load_to_lds
2580 ? Intrinsic::amdgcn_raw_ptr_buffer_load_lds
2581 : Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds;
2582 Instruction *NewLoad = IRB.CreateIntrinsicWithoutFolding(
2583 NewIntr, {}, {Rsrc, LDSPtr, LoadSize,
Off, SOffset, ImmOff, Aux});
2584 copyMetadata(NewLoad, &
I);
2586 I.replaceAllUsesWith(NewLoad);
2587 return {
nullptr,
nullptr};
2590 return {
nullptr,
nullptr};
2593void SplitPtrStructs::processFunction(
Function &
F) {
2595 SmallVector<Instruction *, 0> Originals(
2597 LLVM_DEBUG(
dbgs() <<
"Splitting pointer structs in function: " <<
F.getName()
2599 for (Instruction *
I : Originals) {
2608 "Can't have a resource but no offset");
2610 RsrcParts[
I] = Rsrc;
2614 processConditionals();
2615 killAndReplaceSplitInstructions(Originals);
2621 Conditionals.clear();
2622 ConditionalTemps.clear();
2626class AMDGPULowerBufferFatPointers :
public ModulePass {
2630 AMDGPULowerBufferFatPointers() : ModulePass(
ID) {}
2633 bool runOnModule(
Module &M)
override;
2635 void getAnalysisUsage(AnalysisUsage &AU)
const override;
2643 BufferFatPtrToStructTypeMap *TypeMap) {
2644 bool HasFatPointers =
false;
2647 HasFatPointers |= (
I.getType() != TypeMap->remapType(
I.getType()));
2649 for (
const Value *V :
I.operand_values())
2650 HasFatPointers |= (V->getType() != TypeMap->remapType(V->getType()));
2652 return HasFatPointers;
2656 BufferFatPtrToStructTypeMap *TypeMap) {
2657 Type *Ty =
F.getFunctionType();
2658 return Ty != TypeMap->remapType(Ty);
2674 while (!OldF->
empty()) {
2688 CloneMap[&NewArg] = &OldArg;
2689 NewArg.takeName(&OldArg);
2690 Type *OldArgTy = OldArg.getType(), *NewArgTy = NewArg.getType();
2692 NewArg.mutateType(OldArgTy);
2693 OldArg.replaceAllUsesWith(&NewArg);
2694 NewArg.mutateType(NewArgTy);
2698 if (OldArgTy != NewArgTy && !IsIntrinsic)
2701 AttributeFuncs::typeIncompatible(NewArgTy, ArgAttr));
2708 AttributeFuncs::typeIncompatible(NewF->
getReturnType(), RetAttrs));
2710 NewF->
getContext(), OldAttrs.getFnAttrs(), RetAttrs, ArgAttrs));
2718 CloneMap[&BB] = &BB;
2724bool AMDGPULowerBufferFatPointers::run(
Module &M,
const TargetMachine &TM,
2727 const DataLayout &
DL =
M.getDataLayout();
2733 LLVMContext &Ctx =
M.getContext();
2735 BufferFatPtrToStructTypeMap StructTM(
DL);
2736 BufferFatPtrToIntTypeMap IntTM(
DL);
2740 Ctx.
emitError(
"global variables with a buffer fat pointer address "
2741 "space (7) are not supported");
2743 GV.eraseFromParent();
2748 Type *VT = GV.getValueType();
2749 if (VT != StructTM.remapType(VT)) {
2751 Ctx.
emitError(
"global variables that contain buffer fat pointers "
2752 "(address space 7 pointers) are unsupported. Use "
2753 "buffer resource pointers (address space 8) instead");
2755 GV.eraseFromParent();
2771 SmallPtrSet<Constant *, 8> Visited;
2772 SetVector<Constant *> BufferFatPtrConsts;
2773 while (!Worklist.
empty()) {
2775 if (!Visited.
insert(
C).second)
2791 StoreFatPtrsAsIntsAndExpandMemcpyVisitor MemOpsRewrite(&IntTM, M);
2792 LegalizeBufferContentTypesVisitor BufferContentsTypeRewrite(M, &TM);
2796 const TargetTransformInfo *
TTI = GetTTI(
F);
2797 ScalarEvolution *SE = GetSE(
F);
2798 Changed |= MemOpsRewrite.processFunction(
F,
TTI, SE);
2799 if (InterfaceChange || BodyChanges) {
2800 NeedsRemap.
push_back(std::make_pair(&
F, InterfaceChange));
2801 Changed |= BufferContentsTypeRewrite.processFunction(
F, SE);
2804 if (NeedsRemap.
empty())
2811 FatPtrConstMaterializer Materializer(&StructTM, CloneMap);
2813 ValueMapper LowerInFuncs(CloneMap,
RF_None, &StructTM, &Materializer);
2814 for (
auto [
F, InterfaceChange] : NeedsRemap) {
2816 if (InterfaceChange)
2822 LowerInFuncs.remapFunction(*NewF);
2827 if (InterfaceChange) {
2828 F->replaceAllUsesWith(NewF);
2829 F->eraseFromParent();
2837 SplitPtrStructs Splitter(M, &TM);
2839 Splitter.processFunction(*
F);
2844 F->eraseFromParent();
2848 F->replaceAllUsesWith(*NewF);
2854bool AMDGPULowerBufferFatPointers::runOnModule(
Module &M) {
2855 TargetPassConfig &TPC = getAnalysis<TargetPassConfig>();
2856 const TargetMachine &TM = TPC.
getTM<TargetMachine>();
2857 auto GetTTI = [&](
Function &
F) ->
const TargetTransformInfo * {
2858 if (
F.isDeclaration())
2860 return &getAnalysis<TargetTransformInfoWrapperPass>().getTTI(
F);
2862 auto GetSE = [&](
Function &
F) -> ScalarEvolution * {
2863 if (
F.isDeclaration())
2865 return &getAnalysis<ScalarEvolutionWrapperPass>(
F).getSE();
2867 return run(M, TM, GetTTI, GetSE);
2870char AMDGPULowerBufferFatPointers::ID = 0;
2874void AMDGPULowerBufferFatPointers::getAnalysisUsage(
AnalysisUsage &AU)
const {
2880#define PASS_DESC "Lower buffer fat pointer operations to buffer resources"
2891 return new AMDGPULowerBufferFatPointers();
2898 if (
F.isDeclaration())
2903 if (
F.isDeclaration())
2907 return AMDGPULowerBufferFatPointers().run(M, TM, GetTTI, GetSE)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
function_ref< const TargetTransformInfo *(Function &)> GetTTIFn
static Function * moveFunctionAdaptingType(Function *OldF, FunctionType *NewTy, ValueToValueMapTy &CloneMap)
Move the body of OldF into a new function, returning it.
static void makeCloneInPraceMap(Function *F, ValueToValueMapTy &CloneMap)
static bool isBufferFatPtrOrVector(Type *Ty)
static bool isSplitFatPtr(Type *Ty)
std::pair< Value *, Value * > PtrParts
static bool hasFatPointerInterface(const Function &F, BufferFatPtrToStructTypeMap *TypeMap)
static bool isRemovablePointerIntrinsic(Intrinsic::ID IID)
Returns true if this intrinsic needs to be removed when it is applied to ptr addrspace(7) values.
static bool containsBufferFatPointers(const Function &F, BufferFatPtrToStructTypeMap *TypeMap)
Returns true if there are values that have a buffer fat pointer in them, which means we'll need to pe...
static Value * rsrcPartRoot(Value *V)
Returns the instruction that defines the resource part of the value V.
static constexpr unsigned BufferOffsetWidth
function_ref< ScalarEvolution *(Function &)> GetSEFn
static bool isBufferFatPtrConst(Constant *C)
static std::pair< Constant *, Constant * > splitLoweredFatBufferConst(Constant *C)
Return the ptr addrspace(8) and i32 (resource and offset parts) in a lowered buffer fat pointer const...
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Expand Atomic instructions
Atomic ordering constants.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
AMD GCN specific subclass of TargetSubtarget.
This header defines various interfaces for pass management in LLVM.
Machine Check Debug Module
static bool processFunction(Function &F, NVPTXTargetMachine &TM)
uint64_t IntrinsicInst * II
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
const SmallVectorImpl< MachineOperand > & Cond
static void visit(BasicBlock &Start, std::function< bool(BasicBlock *)> op)
This file defines generic set operations that may be used on set's of different types,...
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Target-Independent Code Generator Pass Configuration Options pass.
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
bool ule(const APInt &RHS) const
Unsigned less or equal comparison.
bool sge(const APInt &RHS) const
Signed greater or equal comparison.
This class represents a conversion between pointers from one address space to another.
Value * getPointerOperand()
Gets the pointer operand.
unsigned getSrcAddressSpace() const
Returns the address space of the pointer operand.
unsigned getDestAddressSpace() const
Returns the address space of the result.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
This class represents an incoming formal argument to a Function.
An instruction that atomically checks whether a specified value is in a memory location,...
Value * getNewValOperand()
AtomicOrdering getMergedOrdering() const
Returns a single ordering which is at least as strong as both the success and failure orderings for t...
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
Value * getCompareOperand()
Value * getPointerOperand()
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this cmpxchg instruction.
an instruction that atomically reads a memory location, combines it with another value,...
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isVolatile() const
Return true if this is a RMW on a volatile memory location.
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
Value * getPointerOperand()
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this rmw instruction.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI AttributeSet removeAttributes(LLVMContext &C, const AttributeMask &AttrsToRemove) const
Remove the specified attributes from this set.
LLVM Basic Block Representation.
LLVM_ABI void removeFromParent()
Unlink 'this' from the containing function, but do not delete it.
LLVM_ABI void insertInto(Function *Parent, BasicBlock *InsertBefore=nullptr)
Insert unlinked basic block into a function.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
Adds the attribute to the indicated argument.
This class represents a function call, abstracting a target machine's calling convention.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
static LLVM_ABI std::optional< DIExpression * > createFragmentExpression(const DIExpression *Expr, unsigned OffsetInBits, unsigned SizeInBits)
Create a DIExpression to describe one part of an aggregate variable that is fragmented across multipl...
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI void insertBefore(DbgRecord *InsertBefore)
LLVM_ABI void eraseFromParent()
LLVM_ABI void replaceVariableLocationOp(Value *OldValue, Value *NewValue, bool AllowEmpty=false)
void setExpression(DIExpression *NewExpr)
iterator find(const_arg_type_t< KeyT > Val)
Implements a dense probed hash-table based set.
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
This class represents a freeze function that returns random concrete value if an operand is either a ...
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
const BasicBlock & front() const
iterator_range< arg_iterator > args()
AttributeList getAttributes() const
Return the attribute list for this Function.
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
void setAttributes(AttributeList Attrs)
Set the attribute list for this Function.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void updateAfterNameChange()
Update internal caches that depend on the function name (such as the intrinsic ID and libcall cache).
Type * getReturnType() const
Returns the type of the ret val.
void copyAttributesFrom(const Function *Src)
copyAttributesFrom - copy all additional attributes (those not needed to create a Function) from the ...
bool hasRelaxedBufferOOBMode() const
bool hasUnalignedBufferAccessEnabled() const
std::optional< unsigned > getBufferResourceNumRecordsWidth() const
Return the width, in bits, of the num_records field of a buffer resource (V#) on this subtarget,...
static GEPNoWrapFlags noUnsignedWrap()
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
LLVM_ABI void copyMetadata(const GlobalObject *Src, unsigned Offset)
Copy metadata from Src, adjusting offsets by Offset.
LinkageTypes getLinkage() const
void setDLLStorageClass(DLLStorageClassTypes C)
unsigned getAddressSpace() const
Module * getParent()
Get the module that this global value is contained inside of...
DLLStorageClassTypes getDLLStorageClass() const
This instruction compares its operands according to the predicate given to the constructor.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
Base class for instruction visitors.
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
This class represents a cast from an integer to a pointer.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
A wrapper class for inspecting calls to intrinsic functions.
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this load instruction.
Type * getPointerOperandType() const
void setVolatile(bool V)
Specify whether this is a volatile load or not.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
unsigned getDestAddressSpace() const
unsigned getSourceAddressSpace() const
ModulePass class - This class is used to implement unstructured interprocedural optimizations and ana...
A Module instance is used to store all the information related to an LLVM module.
const FunctionListType & getFunctionList() const
Get the Module's list of functions (constant).
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
This class represents a cast from a pointer to an address (non-capturing ptrtoint).
Value * getPointerOperand()
Gets the pointer operand.
This class represents a cast from a pointer to an integer.
Value * getPointerOperand()
Gets the pointer operand.
Analysis pass that exposes the ScalarEvolution for a function.
The main scalar evolution driver.
LLVM_ABI bool isKnownNonNegative(const SCEV *S)
Test if the given expression is known to be non-negative.
LLVM_ABI bool isKnownNonPositive(const SCEV *S)
Test if the given expression is known to be non-positive.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEVFlags Flags=SCEV::FlagNone, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI SCEVUse getAddExpr(SmallVectorImpl< SCEVUse > &Ops, SCEVFlagsPair Flags={}, unsigned Depth=0)
Get a canonical add expression, or something simpler if possible.
LLVM_ABI bool isSCEVable(Type *Ty) const
Test if values of the given type are analyzable within the SCEV framework.
APInt getSignedRangeMin(const SCEV *S)
Determine the min of the signed range for a particular SCEV.
LLVM_ABI const SCEV * getNoopOrZeroExtend(const SCEV *V, Type *Ty)
Return a SCEV corresponding to a conversion of the input value to the specified type.
LLVM_ABI const SCEV * getPointerBase(const SCEV *V)
Transitively follow the chain of pointer-type operands until reaching a SCEV that does not have a sin...
APInt getUnsignedRangeMax(const SCEV *S)
Determine the max of the unsigned range for a particular SCEV.
LLVM_ABI const SCEV * getTruncateOrZeroExtend(const SCEV *V, Type *Ty, unsigned Depth=0)
Return a SCEV corresponding to a conversion of the input value to the specified type.
This class represents the LLVM 'select' instruction.
ArrayRef< value_type > getArrayRef() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
This instruction constructs a fixed permutation of two input vectors.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Value * getValueOperand()
Value * getPointerOperand()
MutableArrayRef< TypeSize > getMemberOffsets()
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
bool isLiteral() const
Return true if this type is uniqued by structural equivalence, false if it is a struct definition.
Type * getElementType(unsigned N) const
Analysis pass providing the TargetTransformInfo.
Primary interface to the complete machine description for the target machine.
const STC & getSubtarget(const Function &F) const
This method returns a pointer to the specified type of TargetSubtargetInfo.
Target-Independent Code Generator Pass Configuration Options.
TMC & getTM() const
Get the right type of TargetMachine for this target.
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
bool isVectorTy() const
True if this is an instance of VectorType.
bool isSingleValueType() const
Return true if the type is a valid type for a register in codegen.
unsigned getNumContainedTypes() const
Return the number of types in the derived type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
This is a class that can be implemented by clients to remap types when cloning constants and instruct...
size_type count(const KeyT &Val) const
Return 1 if the specified key is in the map, 0 otherwise.
iterator find(const KeyT &Val)
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
LLVM_ABI Constant * mapConstant(const Constant &C)
LLVM_ABI Value * mapValue(const Value &V)
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
std::pair< iterator, bool > insert(const ValueT &V)
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
iterator insertAfter(iterator where, pointer New)
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ BUFFER_FAT_POINTER
Address space for 160-bit buffer fat pointers.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
LLVM_ABI std::optional< Function * > remangleIntrinsicFunction(Function *F)
bool match(Val *V, const Pattern &P)
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
PointerTypeMap run(const Module &M)
Compute the PointerTypeMap for the module M.
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
LLVM_ABI void findDbgValues(Value *V, SmallVectorImpl< DbgVariableRecord * > &DbgVariableRecords)
Finds the dbg.values describing a value.
ModulePass * createAMDGPULowerBufferFatPointersPass()
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI void copyMetadataForLoad(LoadInst &Dest, const LoadInst &Source)
Copy the metadata from the source instruction to the destination (the replacement for the source inst...
bool set_is_subset(const S1Ty &S1, const S2Ty &S2)
set_is_subset(A, B) - Return true iff A in B
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
InnerAnalysisManagerProxy< FunctionAnalysisManager, Module > FunctionAnalysisManagerModuleProxy
Provide the FunctionAnalysisManager to Module proxy.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
LLVM_ABI bool convertUsersOfConstantsToInstructions(ArrayRef< Constant * > Consts, Function *RestrictToFunc=nullptr, bool RemoveDeadConstants=true, bool IncludeSelf=false)
Replace constant expressions users of the given constants with instructions.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI Value * emitGEPOffset(IRBuilderBase *Builder, const DataLayout &DL, User *GEP, bool NoAssumptions=false)
Given a getelementptr instruction/constantexpr, emit the code necessary to compute the offset from th...
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
char & AMDGPULowerBufferFatPointersID
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
DWARFExpression::Operation Op
S1Ty set_difference(const S1Ty &S1, const S2Ty &S2)
set_difference(A, B) - Return A - B
ArrayRef(const T &OneElt) -> ArrayRef< T >
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI void expandMemSetAsLoop(MemSetInst *MemSet, const TargetTransformInfo *TTI=nullptr)
Expand MemSet as a loop.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
LLVM_ABI void expandMemSetPatternAsLoop(MemSetPatternInst *MemSet, const TargetTransformInfo *TTI=nullptr)
Expand MemSetPattern as a loop.
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
LLVM_ABI void expandMemCpyAsLoop(MemCpyInst *MemCpy, const TargetTransformInfo &TTI, ScalarEvolution *SE=nullptr)
Expand MemCpy as a loop. MemCpy is not deleted.
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
LLVM_ABI AAMDNodes adjustForAccess(unsigned AccessSize)
Create a new AAMDNode for accessing AccessSize bytes of this AAMDNode.
PreservedAnalyses run(Module &M, ModuleAnalysisManager &AM)
This struct is a compact representation of a valid (non-zero power of two) alignment.