29#include "llvm/IR/IntrinsicsAMDGPU.h"
36#define DEBUG_TYPE "AMDGPUtti"
39 "amdgpu-unroll-threshold-private",
40 cl::desc(
"Unroll threshold for AMDGPU if private memory used in a loop"),
44 "amdgpu-unroll-threshold-local",
45 cl::desc(
"Unroll threshold for AMDGPU if local memory used in a loop"),
49 "amdgpu-unroll-threshold-if",
50 cl::desc(
"Unroll threshold increment for AMDGPU for each if statement inside loop"),
54 "amdgpu-unroll-runtime-local",
55 cl::desc(
"Allow runtime unroll for AMDGPU if local memory used in a loop"),
59 "amdgpu-unroll-max-block-to-analyze",
60 cl::desc(
"Inner loop block size threshold to analyze in unroll for AMDGPU"),
65 cl::desc(
"Cost of alloca argument"));
73 cl::desc(
"Maximum alloca size to use for inline cost"));
78 cl::desc(
"Maximum number of BBs allowed in a function after inlining"
79 " (compile time constraint)"));
83 "amdgpu-memcpy-loop-unroll",
84 cl::desc(
"Unroll factor (affecting 4x32-bit operations) to use for memory "
85 "operations when lowering statically-sized memcpy, memmove, or"
97 for (
const Value *V :
I->operand_values()) {
100 return SubLoop->contains(PHI); }))
110 TargetTriple(TM->getTargetTriple()),
112 TLI(ST->getTargetLowering()) {}
117 const Function &
F = *L->getHeader()->getParent();
119 F.getFnAttributeAsParsedInteger(
"amdgpu-unroll-threshold", 300);
120 UP.
MaxCount = std::numeric_limits<unsigned>::max();
135 const unsigned MaxAlloca = (256 - 16) * 4;
141 if (
MDNode *LoopUnrollThreshold =
143 if (LoopUnrollThreshold->getNumOperands() == 2) {
145 LoopUnrollThreshold->getOperand(1));
146 if (MetaThresholdValue) {
152 ThresholdPrivate = std::min(ThresholdPrivate, UP.
Threshold);
153 ThresholdLocal = std::min(ThresholdLocal, UP.
Threshold);
158 unsigned MaxBoost = std::max(ThresholdPrivate, ThresholdLocal);
161 unsigned LocalGEPsSeen = 0;
164 return SubLoop->contains(BB); }))
177 if ((L->contains(Succ0) && L->isLoopExiting(Succ0)) ||
178 (L->contains(Succ1) && L->isLoopExiting(Succ1)))
184 << *L <<
" due to " << *Br <<
'\n');
196 unsigned AS =
GEP->getAddressSpace();
197 unsigned Threshold = 0;
199 Threshold = ThresholdPrivate;
201 Threshold = ThresholdLocal;
209 const Value *Ptr =
GEP->getPointerOperand();
215 if (!AllocaSize || AllocaSize->getFixedValue() > MaxAlloca)
224 if (LocalGEPsSeen > 1 || L->getLoopDepth() > 2 ||
229 << *L <<
" due to LDS use.\n");
234 bool HasLoopDef =
false;
237 if (!Inst || L->isLoopInvariant(
Op))
241 return SubLoop->contains(Inst); }))
265 << *L <<
" due to " << *
GEP <<
'\n');
289 TLI(ST->getTargetLowering()), CommonTTI(TM,
F),
290 IsGraphics(
AMDGPU::isGraphics(
F.getCallingConv())) {
296 return !
F || !ST->isSingleLaneExecution(*
F);
317 (ST->hasAnyPackedFP64Ops() || ST->hasAnyPackedU64Ops()) ? 128
318 : ST->hasAnyPackedFP32Ops() ? 64
331 if (Opcode == Instruction::Load || Opcode == Instruction::Store)
332 return 32 * 4 / ElemWidth;
335 return (ElemWidth == 8 && ST->has16BitInsts()) ? 4
336 : (ElemWidth == 16 && ST->has16BitInsts()) ? 2
337 : (ElemWidth == 32 && ST->hasAnyPackedFP32Ops()) ? 2
338 : (ElemWidth == 64 &&
339 (ST->hasAnyPackedFP64Ops() || ST->hasAnyPackedU64Ops()))
350 return !ST->hasGFX940Insts() && !ST->hasGFX950Insts();
354 unsigned ChainSizeInBytes,
356 unsigned VecRegBitWidth = VF * LoadSize;
359 return 128 / LoadSize;
365 unsigned ChainSizeInBytes,
367 unsigned VecRegBitWidth = VF * StoreSize;
368 if (VecRegBitWidth > 128)
369 return 128 / StoreSize;
385 return 8 * ST->getMaxPrivateElementSize();
393 unsigned AddrSpace)
const {
398 return (Alignment >= 4 || ST->hasUnalignedScratchAccessEnabled()) &&
399 ChainSizeInBytes <= ST->getMaxPrivateElementSize();
406 unsigned AddrSpace)
const {
412 unsigned AddrSpace)
const {
422 unsigned DestAddrSpace,
Align SrcAlign,
Align DestAlign,
423 std::optional<uint32_t> AtomicElementSize)
const {
425 if (AtomicElementSize)
439 unsigned I32EltsInVector = 4;
449 unsigned RemainingBytes,
unsigned SrcAddrSpace,
unsigned DestAddrSpace,
451 std::optional<uint32_t> AtomicCpySize)
const {
455 OpsOut, Context, RemainingBytes, SrcAddrSpace, DestAddrSpace, SrcAlign,
456 DestAlign, AtomicCpySize);
459 while (RemainingBytes >= 16) {
461 RemainingBytes -= 16;
465 while (RemainingBytes >= 8) {
471 while (RemainingBytes >= 4) {
477 while (RemainingBytes >= 2) {
483 while (RemainingBytes) {
490 bool HasUnorderedReductions)
const {
502 case Intrinsic::amdgcn_ds_ordered_add:
503 case Intrinsic::amdgcn_ds_ordered_swap: {
506 if (!Ordering || !Volatile)
509 unsigned OrderingVal = Ordering->getZExtValue();
516 Info.WriteMem =
true;
517 Info.IsVolatile = !Volatile->isZero();
532 FAddSub->
getOpcode() == Instruction::FSub) &&
533 "Expected an fadd or an fsub");
540 if (!HasFMAD && !HasFMA)
553 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
554 int ISD = TLI->InstructionOpcodeToISD(Opcode);
558 unsigned NElts = LT.second.isVector() ?
559 LT.second.getVectorNumElements() : 1;
568 return get64BitInstrCost(
CostKind) * LT.first * NElts;
570 if (ST->has16BitInsts() && SLT == MVT::i16)
571 NElts = (NElts + 1) / 2;
574 return getFullRateInstrCost() * LT.first * NElts;
577 if (SLT == MVT::i64 && ST->hasAnyPackedU64Ops())
578 NElts = (NElts + 1) / 2;
583 if (SLT == MVT::i64) {
585 return 2 * getFullRateInstrCost() * LT.first * NElts;
588 if (ST->has16BitInsts() && SLT == MVT::i16)
589 NElts = (NElts + 1) / 2;
591 return LT.first * NElts * getFullRateInstrCost();
593 const int QuarterRateCost = getQuarterRateInstrCost(
CostKind);
594 if (SLT == MVT::i64) {
595 const int FullRateCost = getFullRateInstrCost();
596 return (4 * QuarterRateCost + (2 * 2) * FullRateCost) * LT.first * NElts;
599 if (ST->has16BitInsts() && SLT == MVT::i16)
600 NElts = (NElts + 1) / 2;
603 return QuarterRateCost * NElts * LT.first;
612 (FAddSub->getOpcode() == Instruction::FAdd ||
613 FAddSub->getOpcode() == Instruction::FSub) &&
620 if (ST->hasAnyPackedFP32Ops() && SLT == MVT::f32)
621 NElts = (NElts + 1) / 2;
622 if (ST->hasBF16PackedInsts() && SLT == MVT::bf16)
623 NElts = (NElts + 1) / 2;
624 if (SLT == MVT::f64) {
625 if (ST->hasAnyPackedFP64Ops())
626 NElts = (NElts + 1) / 2;
627 return LT.first * NElts * get64BitInstrCost(
CostKind);
630 if (ST->has16BitInsts() && SLT == MVT::f16)
631 NElts = (NElts + 1) / 2;
633 if (SLT == MVT::f32 || SLT == MVT::f16 || SLT == MVT::bf16)
634 return LT.first * NElts * getFullRateInstrCost();
640 if (SLT == MVT::f64) {
645 if (!ST->hasUsableDivScaleConditionOutput())
646 Cost += 3 * getFullRateInstrCost();
648 return LT.first *
Cost * NElts;
653 if ((SLT == MVT::f32 && !HasFP32Denormals) ||
654 (SLT == MVT::f16 && ST->has16BitInsts())) {
655 return LT.first * getTransInstrCost(
CostKind) * NElts;
659 if (SLT == MVT::f16 && ST->has16BitInsts()) {
665 int Cost = 4 * getFullRateInstrCost() + 2 * getTransInstrCost(
CostKind);
666 return LT.first *
Cost * NElts;
673 int Cost = getTransInstrCost(
CostKind) + getFullRateInstrCost();
674 return LT.first *
Cost * NElts;
677 if (SLT == MVT::f32 || SLT == MVT::f16) {
679 int Cost = (SLT == MVT::f16 ? 14 : 10) * getFullRateInstrCost() +
682 if (!HasFP32Denormals) {
684 Cost += 2 * getFullRateInstrCost();
687 return LT.first * NElts *
Cost;
693 return TLI->isFNegFree(SLT) ? 0 : NElts;
707 case Intrinsic::fmuladd:
708 case Intrinsic::copysign:
709 case Intrinsic::minimumnum:
710 case Intrinsic::maximumnum:
711 case Intrinsic::canonicalize:
713 case Intrinsic::round:
714 case Intrinsic::uadd_sat:
715 case Intrinsic::usub_sat:
716 case Intrinsic::sadd_sat:
717 case Intrinsic::ssub_sat:
728 switch (ICA.
getID()) {
729 case Intrinsic::fabs:
732 case Intrinsic::amdgcn_workitem_id_x:
733 case Intrinsic::amdgcn_workitem_id_y:
734 case Intrinsic::amdgcn_workitem_id_z:
738 case Intrinsic::amdgcn_workgroup_id_x:
739 case Intrinsic::amdgcn_workgroup_id_y:
740 case Intrinsic::amdgcn_workgroup_id_z:
741 case Intrinsic::amdgcn_lds_kernel_id:
742 case Intrinsic::amdgcn_dispatch_ptr:
743 case Intrinsic::amdgcn_dispatch_id:
744 case Intrinsic::amdgcn_implicitarg_ptr:
745 case Intrinsic::amdgcn_queue_ptr:
757 case Intrinsic::exp2:
758 case Intrinsic::exp10: {
760 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(RetTy);
763 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
765 if (SLT == MVT::f64) {
767 if (IID == Intrinsic::exp)
769 else if (IID == Intrinsic::exp10)
775 if (SLT == MVT::f32) {
776 unsigned NumFullRateOps = 0;
778 unsigned NumTransOps = 1;
784 NumFullRateOps = ST->hasFastFMAF32() ? 13 : 17;
786 if (IID == Intrinsic::exp) {
789 }
else if (IID == Intrinsic::exp10) {
795 if (HasFP32Denormals)
800 NumTransOps * getTransInstrCost(
CostKind);
801 return LT.first * NElts *
Cost;
807 case Intrinsic::log2:
808 case Intrinsic::log10: {
809 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(RetTy);
812 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
814 if (SLT == MVT::f32) {
815 unsigned NumFullRateOps = 0;
817 if (IID == Intrinsic::log2) {
825 NumFullRateOps = ST->hasFastFMAF32() ? 8 : 11;
828 if (HasFP32Denormals)
832 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(
CostKind);
833 return LT.first * NElts *
Cost;
839 case Intrinsic::cos: {
840 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(RetTy);
843 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
845 if (SLT == MVT::f32) {
847 unsigned NumFullRateOps = ST->hasTrigReducedRange() ? 2 : 1;
850 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(
CostKind);
851 return LT.first * NElts *
Cost;
856 case Intrinsic::sqrt: {
857 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(RetTy);
860 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
862 if (SLT == MVT::f32) {
863 unsigned NumFullRateOps = 0;
867 NumFullRateOps = HasFP32Denormals ? 17 : 16;
871 NumFullRateOps * getFullRateInstrCost() + getTransInstrCost(
CostKind);
872 return LT.first * NElts *
Cost;
884 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(RetTy);
886 unsigned NElts = LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
888 if ((ST->hasVOP3PInsts() &&
889 (SLT == MVT::f16 || SLT == MVT::i16 ||
890 (SLT == MVT::bf16 && ST->hasBF16PackedInsts()))) ||
891 (ST->hasAnyPackedFP64Ops() && SLT == MVT::f64) ||
892 (ST->hasAnyPackedU64Ops() && SLT == MVT::i64)) {
893 NElts = (NElts + 1) / 2;
894 }
else if (SLT == MVT::f32) {
895 bool HasPk2FP32Op = ST->hasAnyPackedFP32Ops() &&
896 IID != Intrinsic::minimumnum &&
897 IID != Intrinsic::maximumnum;
898 NElts = HasPk2FP32Op ? (NElts + 1) / 2 : NElts;
902 unsigned InstRate = getQuarterRateInstrCost(
CostKind);
904 switch (ICA.
getID()) {
906 case Intrinsic::fmuladd:
907 if (SLT == MVT::f64) {
908 InstRate = get64BitInstrCost(
CostKind);
912 if ((SLT == MVT::f32 && ST->hasFastFMAF32()) || SLT == MVT::f16)
913 InstRate = getFullRateInstrCost();
915 InstRate = ST->hasFastFMAF32() ? getHalfRateInstrCost(
CostKind)
916 : getQuarterRateInstrCost(
CostKind);
919 case Intrinsic::copysign:
920 return NElts * getFullRateInstrCost();
921 case Intrinsic::minimumnum:
922 case Intrinsic::maximumnum: {
934 SLT == MVT::f64 ? get64BitInstrCost(
CostKind) : getFullRateInstrCost();
935 InstRate = BaseRate *
NumOps;
938 case Intrinsic::canonicalize: {
940 SLT == MVT::f64 ? get64BitInstrCost(
CostKind) : getFullRateInstrCost();
943 case Intrinsic::uadd_sat:
944 case Intrinsic::usub_sat:
945 case Intrinsic::sadd_sat:
946 case Intrinsic::ssub_sat: {
947 if (SLT == MVT::i16 || SLT == MVT::i32)
948 InstRate = getFullRateInstrCost();
950 static const auto ValidSatTys = {MVT::v2i16, MVT::v4i16};
957 if (SLT == MVT::i16 || SLT == MVT::i32)
958 InstRate = 2 * getFullRateInstrCost();
964 return LT.first * NElts * InstRate;
970 assert((
I ==
nullptr ||
I->getOpcode() == Opcode) &&
971 "Opcode should reflect passed instruction.");
974 const int CBrCost = SCost ? 5 : 7;
976 case Instruction::UncondBr:
978 return SCost ? 1 : 4;
979 case Instruction::CondBr:
983 case Instruction::Switch: {
987 return (
SI ? (
SI->getNumCases() + 1) : 4) * (CBrCost + 1);
989 case Instruction::Ret:
990 return SCost ? 1 : 10;
1002 if (FVT && FVT->getElementType()->isIntegerTy(1) && FVT->getNumElements() > 1)
1003 return FVT->getNumElements();
1004 return std::nullopt;
1013 if (Opcode == Instruction::BitCast) {
1015 Elts && Dst->isIntegerTy(*Elts))
1017 getFullRateInstrCost();
1019 Elts && Src->isIntegerTy(*Elts))
1021 getFullRateInstrCost();
1029 std::optional<FastMathFlags> FMF,
1036 if (Opcode == Instruction::Add || Opcode == Instruction::Xor) {
1039 getFullRateInstrCost();
1042 EVT OrigTy = TLI->getValueType(
DL, Ty);
1049 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1050 return LT.first * getFullRateInstrCost();
1057 EVT OrigTy = TLI->getValueType(
DL, Ty);
1064 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1065 return LT.first * getHalfRateInstrCost(
CostKind);
1072 case Instruction::ExtractElement:
1073 case Instruction::InsertElement: {
1080 if (EltSize == 16 && Index == 0 && ST->has16BitInsts())
1084 if (EltSize == 1 && Opcode == Instruction::InsertElement)
1090 if (Opcode == Instruction::ExtractElement && EltSize == 8) {
1092 unsigned NumElts = FVTy->getNumElements();
1119 if (Indices.
size() > 1)
1125 TLI->ParseConstraints(
DL, ST->getRegisterInfo(), *CI);
1127 const int TargetOutputIdx = Indices.
empty() ? -1 : Indices[0];
1130 for (
auto &TC : TargetConstraints) {
1135 if (TargetOutputIdx != -1 && TargetOutputIdx != OutputIdx++)
1138 TLI->ComputeConstraintToUse(TC,
SDValue());
1141 TRI, TC.ConstraintCode, TC.ConstraintVT).second;
1145 if (!RC || !
TRI->isSGPRClass(RC))
1175bool GCNTTIImpl::isSourceOfDivergence(
const Value *V)
const {
1199 case Intrinsic::read_register:
1201 case Intrinsic::amdgcn_addrspacecast_nonnull: {
1203 Intrinsic->getOperand(0)->getType()->getPointerAddressSpace();
1204 unsigned DstAS =
Intrinsic->getType()->getPointerAddressSpace();
1207 ST->hasGloballyAddressableScratch();
1209 case Intrinsic::amdgcn_workitem_id_y:
1210 case Intrinsic::amdgcn_workitem_id_z: {
1215 *
F, IID == Intrinsic::amdgcn_workitem_id_y ? 1 : 2);
1216 return !HasUniformYZ && (!ThisDimSize || *ThisDimSize != 1);
1225 if (CI->isInlineAsm())
1240 ST->hasGloballyAddressableScratch();
1246bool GCNTTIImpl::isAlwaysUniform(
const Value *V)
const {
1251 if (CI->isInlineAsm())
1269 bool XDimDoesntResetWithinWaves =
false;
1272 XDimDoesntResetWithinWaves = ST->hasWavefrontsEvenlySplittingXDim(*
F);
1274 using namespace llvm::PatternMatch;
1280 return C >= ST->getWavefrontSizeLog2() && XDimDoesntResetWithinWaves;
1287 ST->getWavefrontSizeLog2() &&
1288 XDimDoesntResetWithinWaves;
1303 case Intrinsic::amdgcn_if:
1304 case Intrinsic::amdgcn_else: {
1305 ArrayRef<unsigned> Indices = ExtValue->
getIndices();
1306 return Indices.
size() == 1 && Indices[0] == 1;
1323 case Intrinsic::amdgcn_is_shared:
1324 case Intrinsic::amdgcn_is_private:
1325 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1326 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1327 case Intrinsic::amdgcn_load_to_lds:
1328 case Intrinsic::amdgcn_make_buffer_rsrc:
1338 Value *NewV)
const {
1339 auto IntrID =
II->getIntrinsicID();
1341 case Intrinsic::amdgcn_is_shared:
1342 case Intrinsic::amdgcn_is_private: {
1343 unsigned TrueAS = IntrID == Intrinsic::amdgcn_is_shared ?
1351 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1352 case Intrinsic::amdgcn_flat_atomic_fmin_num: {
1353 Type *DestTy =
II->getType();
1360 M,
II->getIntrinsicID(), {DestTy, SrcTy, DestTy});
1361 II->setArgOperand(0, NewV);
1362 II->setCalledFunction(NewDecl);
1365 case Intrinsic::amdgcn_load_to_lds: {
1370 II->setArgOperand(0, NewV);
1371 II->setCalledFunction(NewDecl);
1374 case Intrinsic::amdgcn_make_buffer_rsrc: {
1376 Type *DstTy =
II->getType();
1377 Type *NumRecordsTy =
II->getArgOperand(2)->getType();
1380 M,
II->getIntrinsicID(), {DstTy, SrcTy, NumRecordsTy});
1381 II->setArgOperand(0, NewV);
1382 II->setCalledFunction(NewDecl);
1403 unsigned ScalarSize =
DL.getTypeSizeInBits(SrcTy->getElementType());
1405 (ScalarSize == 16 || ScalarSize == 8)) {
1418 unsigned NumSrcElts = SrcVecTy->getNumElements();
1419 if (ST->hasVOP3PInsts() && ScalarSize == 16 && NumSrcElts == 2 &&
1425 unsigned EltsPerReg = 32 / ScalarSize;
1433 return divideCeil(DstVecTy->getNumElements(), EltsPerReg);
1436 if (Index % EltsPerReg == 0)
1439 return divideCeil(DstVecTy->getNumElements(), EltsPerReg);
1445 unsigned NumDstElts = DstVecTy->getNumElements();
1447 unsigned EndIndex = Index + NumInsertElts;
1448 unsigned BeginSubIdx = Index % EltsPerReg;
1449 unsigned EndSubIdx = EndIndex % EltsPerReg;
1452 if (BeginSubIdx != 0) {
1460 if (EndIndex < NumDstElts && BeginSubIdx < EndSubIdx)
1469 unsigned NumElts = DstVecTy->getNumElements();
1473 unsigned EltsFromLHS = NumElts - Index;
1474 bool LHSIsAligned = (Index % EltsPerReg) == 0;
1475 bool RHSIsAligned = (EltsFromLHS % EltsPerReg) == 0;
1476 if (LHSIsAligned && RHSIsAligned)
1478 if (LHSIsAligned && !RHSIsAligned)
1479 return divideCeil(NumElts, EltsPerReg) - (EltsFromLHS / EltsPerReg);
1480 if (!LHSIsAligned && RHSIsAligned)
1488 if (!Mask.empty()) {
1498 for (
unsigned DstIdx = 0; DstIdx < Mask.size(); DstIdx += EltsPerReg) {
1501 for (
unsigned I = 0;
I < EltsPerReg && DstIdx +
I < Mask.size(); ++
I) {
1502 int SrcIdx = Mask[DstIdx +
I];
1506 if (SrcIdx < (
int)NumSrcElts) {
1507 Reg = SrcIdx / EltsPerReg;
1508 if (SrcIdx % EltsPerReg !=
I)
1511 Reg = NumSrcElts + (SrcIdx - NumSrcElts) / EltsPerReg;
1512 if ((SrcIdx - NumSrcElts) % EltsPerReg !=
I)
1518 if (Regs.
size() >= 2)
1542 if (
I->getOpcode() == Instruction::FAdd ||
1543 I->getOpcode() == Instruction::FSub) {
1544 for (
Use &
Op :
I->operands()) {
1546 if (!
FMul ||
FMul->getOpcode() != Instruction::FMul ||
1547 !
FMul->hasOneUse() ||
1551 if (
FMul->getParent() !=
I->getParent())
1557 for (
auto &
Op :
I->operands()) {
1570 if (OpInst->getType()->isVectorTy() && OpInst->getNumOperands() > 1) {
1572 if (VecOpInst && VecOpInst->
hasOneUse())
1577 OpInst->getOperand(0),
1578 OpInst->getOperand(1)) == 0) {
1587 unsigned EltSize =
DL.getTypeSizeInBits(
1592 if (EltSize < 16 || !ST->has16BitInsts())
1595 int NumSubElts, SubIndex;
1596 if (Shuffle->changesLength()) {
1597 if (Shuffle->increasesLength() && Shuffle->isIdentityWithPadding()) {
1602 if ((Shuffle->isExtractSubvectorMask(SubIndex) ||
1603 Shuffle->isInsertSubvectorMask(NumSubElts, SubIndex)) &&
1604 !(SubIndex & 0x1)) {
1610 if (Shuffle->isReverse() || Shuffle->isZeroEltSplat() ||
1611 Shuffle->isSingleSource()) {
1618 return !
Ops.empty();
1639 if (Callee->hasFnAttribute(Attribute::AlwaysInline) ||
1640 Callee->hasFnAttribute(Attribute::InlineHint))
1646 if (Callee->size() == 1)
1648 size_t BBSize = Caller->size() + Callee->size() - 1;
1651 << Callee->getName() <<
" into " << Caller->getName()
1652 <<
": caller BBs=" << Caller->size() <<
", callee BBs="
1653 << Callee->size() <<
", combined BBs=" << BBSize
1665 const int NrOfSGPRUntilSpill = 26;
1666 const int NrOfVGPRUntilSpill = 32;
1670 unsigned adjustThreshold = 0;
1676 for (
auto ArgVT : ValueVTs) {
1680 SGPRsInUse += CCRegNum;
1682 VGPRsInUse += CCRegNum;
1692 ArgStackCost +=
const_cast<GCNTTIImpl *
>(TTIImpl)->getMemoryOpCost(
1695 ArgStackCost +=
const_cast<GCNTTIImpl *
>(TTIImpl)->getMemoryOpCost(
1701 adjustThreshold += std::max(0, SGPRsInUse - NrOfSGPRUntilSpill) *
1703 adjustThreshold += std::max(0, VGPRsInUse - NrOfVGPRUntilSpill) *
1705 return adjustThreshold;
1714 unsigned AllocaSize = 0;
1721 unsigned AddrSpace = Ty->getAddressSpace();
1731 AllocaSize +=
Size->getFixedValue();
1775 static_assert(InlinerVectorBonusPercent == 0,
"vector bonus assumed to be 0");
1779 return BB.getTerminator()->getNumSuccessors() > 1;
1782 Threshold += Threshold / 2;
1790 unsigned AllocaThresholdBonus =
1791 (Threshold * ArgAllocaSize->getFixedValue()) / AllocaSize;
1793 return AllocaThresholdBonus;
1799 CommonTTI.getUnrollingPreferences(L, SE, UP, ORE);
1804 CommonTTI.getPeelingPreferences(L, SE, PP);
1808 return getQuarterRateInstrCost(
CostKind);
1812 return ST->hasFullRate64Ops()
1813 ? getFullRateInstrCost()
1814 : ST->hasHalfRate64Ops() ? getHalfRateInstrCost(
CostKind)
1815 : getQuarterRateInstrCost(
CostKind);
1818std::pair<InstructionCost, MVT>
1819GCNTTIImpl::getTypeLegalizationCost(
Type *Ty)
const {
1821 auto Size =
DL.getTypeSizeInBits(Ty);
1833 if (ST->hasVmemPrefInsts() || ST->hasSmemPrefetchInsts())
1834 return ST->getDataCacheLineSize();
1839 return ST->hasPrefetch() ? 128 : 0;
1850 LB.push_back({
"amdgpu-max-num-workgroups[0]", MaxNumWorkgroups[0]});
1851 LB.push_back({
"amdgpu-max-num-workgroups[1]", MaxNumWorkgroups[1]});
1852 LB.push_back({
"amdgpu-max-num-workgroups[2]", MaxNumWorkgroups[2]});
1853 std::pair<unsigned, unsigned> FlatWorkGroupSize =
1854 ST->getFlatWorkGroupSizes(
F);
1855 LB.push_back({
"amdgpu-flat-work-group-size[0]", FlatWorkGroupSize.first});
1856 LB.push_back({
"amdgpu-flat-work-group-size[1]", FlatWorkGroupSize.second});
1857 std::pair<unsigned, unsigned> WavesPerEU = ST->getWavesPerEU(
F);
1858 LB.push_back({
"amdgpu-waves-per-eu[0]", WavesPerEU.first});
1859 LB.push_back({
"amdgpu-waves-per-eu[1]", WavesPerEU.second});
1864 if (!ST->hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
1871 Attribute IEEEAttr =
F->getFnAttribute(
"amdgpu-ieee");
1886 if ((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
1888 VecTy->getElementType()->isIntegerTy(8)) {
1899 if (VecTy->getElementType()->isIntegerTy(8)) {
1910 case Intrinsic::amdgcn_wave_shuffle:
1917 if (isAlwaysUniform(V))
1920 if (isSourceOfDivergence(V))
1928 bool HasBaseReg, int64_t Scale,
1929 unsigned AddrSpace)
const {
1930 if (HasBaseReg && Scale != 0) {
1934 if (getST()->hasScaleOffset() && Ty && Ty->isSized() &&
1954 unsigned EffInsnsA =
A.Insns +
A.ScaleCost;
1955 unsigned EffInsnsB =
B.Insns +
B.ScaleCost;
1957 return std::tie(EffInsnsA,
A.NumIVMuls,
A.AddRecCost,
A.NumBaseAdds,
1958 A.SetupCost,
A.ImmCost,
A.NumRegs) <
1959 std::tie(EffInsnsB,
B.NumIVMuls,
B.AddRecCost,
B.NumBaseAdds,
1960 B.SetupCost,
B.ImmCost,
B.NumRegs);
1977 case Intrinsic::amdgcn_wave_shuffle:
1980 return UniformArgs[0] || UniformArgs[1];
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Provides AMDGPU specific target descriptions.
Base class for AMDGPU specific classes of TargetSubtarget.
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Register const TargetRegisterInfo * TRI
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
This file implements the SmallBitVector class.
std::optional< unsigned > getReqdWorkGroupSize(const Function &F, unsigned Dim) const
bool hasWavefrontsEvenlySplittingXDim(const Function &F, bool REquiresUniformYZ=false) const
uint64_t getMaxMemIntrinsicInlineSizeThreshold() const override
AMDGPUTTIImpl(const AMDGPUTargetMachine *TM, const Function &F)
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
an instruction to allocate memory on the stack
LLVM_ABI bool isStaticAlloca() const
Return true if this alloca is in the entry block of the function and is a constant size.
LLVM_ABI std::optional< TypeSize > getAllocationSize(const DataLayout &DL) const
Get allocation size in bytes.
This class represents an incoming formal argument to a Function.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool getValueAsBool() const
Return the attribute's value as a boolean.
bool isValid() const
Return true if the attribute is any kind of attribute.
LLVM Basic Block Representation.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
unsigned getNumberOfParts(Type *Tp) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
bool isInlineAsm() const
Check if this call is an inline asm statement.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
CallingConv::ID getCallingConv() const
Value * getArgOperand(unsigned i) const
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
unsigned getArgOperandNo(const Use *U) const
Given a use for a arg operand, get the arg operand number that corresponds to it.
This class represents a function call, abstracting a target machine's calling convention.
Conditional Branch instruction.
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
GCNTTIImpl(const AMDGPUTargetMachine *TM, const Function &F)
unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Account for loads of i8 vector types to have reduced cost.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
void collectKernelLaunchBounds(const Function &F, SmallVectorImpl< std::pair< StringRef, int64_t > > &LB) const override
bool isUniform(const Instruction *I, const SmallBitVector &UniformArgs) const override
bool isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes, Align Alignment, unsigned AddrSpace) const override
bool isInlineAsmSourceOfDivergence(const CallInst *CI, ArrayRef< unsigned > Indices={}) const
Analyze if the results of inline asm are divergent.
bool isReadRegisterSourceOfDivergence(const IntrinsicInst *ReadReg) const
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
unsigned getNumberOfRegisters(unsigned RCID) const override
bool isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes, Align Alignment, unsigned AddrSpace) const override
unsigned getCacheLineSize() const override
Data cache line size for LoopDataPrefetch pass. Has no use before GFX12.
unsigned getStoreVectorFactor(unsigned VF, unsigned StoreSize, unsigned ChainSizeInBytes, VectorType *VecTy) const override
bool isLegalToVectorizeMemChain(unsigned ChainSizeInBytes, Align Alignment, unsigned AddrSpace) const
bool isLSRCostLess(const TTI::LSRCost &A, const TTI::LSRCost &B) const override
bool shouldPrefetchAddressSpace(unsigned AS) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *ValTy, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool hasBranchDivergence(const Function *F=nullptr) const override
Value * rewriteIntrinsicWithAddressSpace(IntrinsicInst *II, Value *OldV, Value *NewV) const override
unsigned getCallerAllocaCost(const CallBase *CB, const AllocaInst *AI) const override
void getMemcpyLoopResidualLoweringType(SmallVectorImpl< Type * > &OpsOut, LLVMContext &Context, unsigned RemainingBytes, unsigned SrcAddrSpace, unsigned DestAddrSpace, Align SrcAlign, Align DestAlign, std::optional< uint32_t > AtomicCpySize) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
unsigned getInliningThresholdMultiplier() const override
unsigned getLoadVectorFactor(unsigned VF, unsigned LoadSize, unsigned ChainSizeInBytes, VectorType *VecTy) const override
unsigned getPrefetchDistance() const override
How much before a load we should place the prefetch instruction.
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
KnownIEEEMode fpenvIEEEMode(const Instruction &I) const
Return KnownIEEEMode::On if we know if the use context can assume "amdgpu-ieee"="true" and KnownIEEEM...
unsigned adjustInliningThreshold(const CallBase *CB) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Whether it is profitable to sink the operands of an Instruction I to the basic block of I.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool shouldDropLSRSolutionIfLessProfitable() const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
int getInliningLastCallToStaticBonus() const override
bool collectFlatAddressOperands(SmallVectorImpl< int > &OpIndexes, Intrinsic::ID IID) const override
ValueUniformity getValueUniformity(const Value *V) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
unsigned getNumberOfParts(Type *Tp) const override
When counting parts on AMD GPUs, account for i8s being grouped together under a single i32 value.
bool preferSLPInstCountCheck() const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
unsigned getMinVectorRegisterBitWidth() const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind Vector) const override
bool isNumRegsMajorCostOfLSR() const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
Type * getMemcpyLoopLoweringType(LLVMContext &Context, Value *Length, unsigned SrcAddrSpace, unsigned DestAddrSpace, Align SrcAlign, Align DestAlign, std::optional< uint32_t > AtomicElementSize) const override
uint64_t getMaxMemIntrinsicInlineSizeThreshold() const override
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI bool hasApproxFunc() const LLVM_READONLY
Determine whether the approximate-math-functions flag is set.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI bool hasAllowContract() const LLVM_READONLY
Determine whether the allow-contract flag is set.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
FastMathFlags getFlags() const
Type * getReturnType() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Represents a single loop in the control flow graph.
static LLVM_ABI MVT getVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
A Module instance is used to store all the information related to an LLVM module.
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
bool isFMADLegal(const SelectionDAG &DAG, const SDNode *N) const override
Returns true if be combined with to form an ISD::FMAD.
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
The main scalar evolution driver.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Represent a constant reference to a string, i.e.
std::vector< AsmOperandInfo > AsmOperandInfoVector
Primary interface to the complete machine description for the target machine.
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
Base class of all SIMD vector types.
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ BUFFER_STRIDED_POINTER
Address space for 192-bit fat buffer pointers with an additional index.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ BUFFER_FAT_POINTER
Address space for 160-bit buffer fat pointers.
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
LLVM_READNONE constexpr bool isShader(CallingConv::ID CC)
bool isFlatGlobalAddrSpace(unsigned AS)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
SmallVector< unsigned > getMaxNumWorkGroups(const Function &F)
bool isExtendedGlobalAddrSpace(unsigned AS)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ FADD
Simple binary floating point operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ AND
Bitwise operators - logical and, logical or, logical xor.
LLVM_ABI int getInstrCost()
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
specific_fpval m_FPOne()
Match a float 1.0 or vector with all elements equal to 1.0.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_FAbs(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract_or_null(Y &&MD)
Extract a Value from Metadata, allowing null.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI void ComputeValueVTs(const TargetLowering &TLI, const DataLayout &DL, Type *Ty, SmallVectorImpl< EVT > &ValueVTs, SmallVectorImpl< EVT > *MemVTs=nullptr, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
ComputeValueVTs - Given an LLVM IR type, compute a sequence of EVTs that represent all the individual...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ Load
The value being inserted comes from a load (InsertElement only).
LLVM_ABI MDNode * findOptionMDForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for a loop.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
@ AlwaysUniform
The result value is always uniform.
@ NeverUniform
The result value can never be assumed to be uniform.
@ Default
The result value is uniform if and only if all operands are uniform.
@ Custom
The result value requires a custom uniformity check.
MCRegisterClass TargetRegisterClass
This struct is a compact representation of a valid (non-zero power of two) alignment.
static constexpr DenormalMode getPreserveSign()
uint64_t getScalarSizeInBits() const
Information about a load/store intrinsic defined by the target.
bool isInlineCompatible(SIModeRegisterDefaults CalleeMode) const