18#include "llvm/IR/IntrinsicsRISCV.h"
26#define DEBUG_TYPE "riscvtti"
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
44 cl::desc(
"Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
57 size_t NumInstr = OpCodes.size();
62 return LMULCost * NumInstr;
64 for (
auto Op : OpCodes) {
66 case RISCV::VRGATHER_VI:
69 case RISCV::VRGATHER_VV:
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
97 case RISCV::VFREDOSUM_VS: {
105 case RISCV::VFMV_F_S:
110 case RISCV::VFMV_S_F:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
117 case RISCV::VFIRST_M:
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
160 if (!BO || !BO->hasOneUse())
163 if (BO->getOpcode() != Instruction::Shl)
174 if (ShAmt == Trailing)
191 if (!Cmp || !Cmp->isEquality())
207 if ((CmpC & Mask) != CmpC)
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
230 bool Takes12BitImm =
false;
231 unsigned ImmArgIdx = ~0U;
234 case Instruction::GetElementPtr:
239 case Instruction::Store: {
244 if (Idx == 1 || !Inst)
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(),
DL, getTLI()->getValueType(
DL, Ty),
257 case Instruction::Load:
260 case Instruction::And:
262 if (
Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
265 if (
Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
268 if (ST->hasStdExtZbs() && (~
Imm).isPowerOf2())
270 if (Inst && Idx == 1 &&
Imm.getBitWidth() <= ST->getXLen() &&
273 if (Inst && Idx == 1 &&
Imm.getBitWidth() == 64 &&
276 Takes12BitImm =
true;
278 case Instruction::Add:
279 Takes12BitImm =
true;
281 case Instruction::Or:
282 case Instruction::Xor:
284 if (ST->hasStdExtZbs() &&
Imm.isPowerOf2())
286 Takes12BitImm =
true;
288 case Instruction::Mul:
290 if (
Imm.isPowerOf2() ||
Imm.isNegatedPowerOf2())
293 if ((
Imm + 1).isPowerOf2() || (
Imm - 1).isPowerOf2())
296 Takes12BitImm =
true;
298 case Instruction::Sub:
299 case Instruction::Shl:
300 case Instruction::LShr:
301 case Instruction::AShr:
302 Takes12BitImm =
true;
313 if (
Imm.getSignificantBits() <= 64 &&
336 return ST->hasVInstructions();
346 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
350 if (Opcode == Instruction::FAdd)
359 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
360 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
361 InputTypeA != InputTypeB || !InputTypeA->
isIntegerTy(8) ||
377 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second,
CostKind);
386 std::pair<InstructionCost, MVT> AccLT =
394 bool WidenFirst =
false;
395 if (VF.
isScalable() && AccLT.second.isScalableVector()) {
396 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
409 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
412 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second,
CostKind);
416 std::pair<InstructionCost, MVT> RedLT =
418 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
420 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
424 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
436 switch (
II->getIntrinsicID()) {
440 case Intrinsic::vector_reduce_mul:
441 case Intrinsic::vector_reduce_fmul:
447 if (ST->hasVInstructions())
448 if (
unsigned MinVLen = ST->getRealMinVLen();
463 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
466 (ST->hasVInstructions() &&
489 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
495RISCVTTIImpl::getConstantPoolLoadCost(
Type *Ty,
500 return getStaticDataAddrGenerationCost(
CostKind) +
506 unsigned Size = Mask.size();
509 for (
unsigned I = 0;
I !=
Size; ++
I) {
510 if (
static_cast<unsigned>(Mask[
I]) ==
I)
516 for (
unsigned J =
I + 1; J !=
Size; ++J)
518 if (
static_cast<unsigned>(Mask[J]) != J %
I)
546 "Expected fixed vector type and non-empty mask");
549 unsigned NumOfDests =
divideCeil(Mask.size(), LegalNumElts);
553 if (NumOfDests <= 1 ||
555 Tp->getElementType()->getPrimitiveSizeInBits() ||
556 LegalNumElts >= Tp->getElementCount().getFixedValue())
559 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
562 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
566 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
567 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
568 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
570 assert(NormalizedVF >= Mask.size() &&
571 "Normalized mask expected to be not shorter than original mask.");
576 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
577 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
580 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
583 Cost +=
TTI.getShuffleCost(
586 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
588 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
589 Cost +=
TTI.getShuffleCost(
592 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
615 if (!VLen || Mask.empty())
619 LegalVT =
TTI.getTypeLegalizationCost(
625 if (NumOfDests <= 1 ||
627 Tp->getElementType()->getPrimitiveSizeInBits() ||
631 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
634 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
640 unsigned NormalizedVF =
645 assert(NormalizedVF >= Mask.size() &&
646 "Normalized mask expected to be not shorter than original mask.");
652 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
653 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
656 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
661 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
663 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
665 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
672 if ((NumOfDestRegs > 2 && NumShuffles <=
static_cast<int>(NumOfDestRegs)) ||
673 (NumOfDestRegs <= 2 && NumShuffles < 4))
688 if (!
LT.second.isFixedLengthVector())
696 auto GetSlideOpcode = [&](
int SlideAmt) {
698 bool IsVI =
isUInt<5>(std::abs(SlideAmt));
700 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
701 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
704 std::array<std::pair<int, int>, 2> SrcInfo;
708 if (SrcInfo[1].second == 0)
712 if (SrcInfo[0].second != 0) {
713 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
714 FirstSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
717 if (SrcInfo[1].first == -1)
718 return FirstSlideCost;
721 if (SrcInfo[1].second != 0) {
722 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
723 SecondSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
726 getRISCVInstructionCost(RISCV::VMERGE_VVM,
LT.second,
CostKind);
733 return FirstSlideCost + SecondSlideCost + MaskCost;
743 "Expected the Mask to match the return size if given");
745 "Expected the same scalar types");
748 if (VIC == TTI::VectorInstrContext::SplatOpFolded &&
764 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
766 *
this, LT.second, ST->getRealVLen(),
768 if (VRegSplittingCost.
isValid())
769 return VRegSplittingCost;
774 if (Mask.size() >= 2) {
775 MVT EltTp = LT.second.getVectorElementType();
786 return 2 * LT.first * TLI->getLMULCost(LT.second);
788 if (Mask[0] == 0 || Mask[0] == 1) {
792 if (
equal(DeinterleaveMask, Mask))
793 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
798 if (LT.second.getScalarSizeInBits() != 1 &&
801 unsigned NumSlides =
Log2_32(Mask.size() / SubVectorSize);
803 for (
unsigned I = 0;
I != NumSlides; ++
I) {
804 unsigned InsertIndex = SubVectorSize * (1 <<
I);
809 std::pair<InstructionCost, MVT> DestLT =
814 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
828 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
829 LT.second.getVectorNumElements() <= 256)) {
834 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second,
CostKind);
848 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
849 LT.second.getVectorNumElements() <= 256)) {
850 auto &
C = SrcTy->getContext();
851 auto EC = SrcTy->getElementCount();
856 return 2 * IndexCost +
857 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
876 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
904 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
905 if (std::optional<unsigned> VLen = ST->getRealVLen();
906 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
907 SubLT.second.getSizeInBits() <= *VLen)
915 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second,
CostKind);
922 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second,
CostKind);
934 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
941 if (IsLoad && LT.second.isVector() &&
943 LT.second.getVectorElementCount()))
947 Instruction::InsertElement);
948 if (LT.second.getScalarSizeInBits() == 1) {
956 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
969 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
970 RISCV::VMV_X_S, RISCV::VMV_V_X,
979 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second,
CostKind);
985 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second,
CostKind);
991 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
992 if (Index >= 0 && Index < 32)
993 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
994 else if (Index < 0 && Index > -32)
995 Opcodes[1] = RISCV::VSLIDEUP_VI;
996 return LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1000 if (!LT.second.isVector())
1006 if (SrcTy->getElementType()->isIntegerTy(1)) {
1018 MVT ContainerVT = LT.second;
1019 if (LT.second.isFixedLengthVector())
1020 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1022 if (ContainerVT.
bitsLE(M1VT)) {
1032 if (LT.second.isFixedLengthVector())
1034 LenCost =
isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1035 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1036 if (LT.second.isFixedLengthVector() &&
1037 isInt<5>(LT.second.getVectorNumElements() - 1))
1038 Opcodes[1] = RISCV::VRSUB_VI;
1040 getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1041 return LT.first * (LenCost + GatherCost);
1048 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1050 getRISCVInstructionCost(M1Opcodes, M1VT,
CostKind) + 3;
1054 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT,
CostKind) * Ratio;
1056 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second,
CostKind);
1057 return FixedCost + LT.first * (GatherCost + SlideCost);
1091 Ty, DemandedElts, Insert, Extract,
CostKind);
1093 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1094 if (Ty->getScalarSizeInBits() == 1) {
1104 assert(LT.second.isFixedLengthVector());
1105 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1109 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second,
CostKind);
1122 switch (MICA.
getID()) {
1123 case Intrinsic::vp_load_ff: {
1124 EVT DataTypeVT = TLI->getValueType(
DL, DataTy);
1125 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1132 case Intrinsic::experimental_vp_strided_load:
1133 case Intrinsic::experimental_vp_strided_store:
1135 case Intrinsic::masked_compressstore:
1136 case Intrinsic::masked_expandload:
1138 case Intrinsic::vp_scatter:
1139 case Intrinsic::vp_gather:
1140 case Intrinsic::masked_scatter:
1141 case Intrinsic::masked_gather:
1143 case Intrinsic::vp_load:
1144 case Intrinsic::vp_store:
1145 case Intrinsic::masked_load:
1146 case Intrinsic::masked_store:
1155 unsigned Opcode = MICA.
getID() == Intrinsic::masked_load ? Instruction::Load
1156 : Instruction::Store;
1167 if (MICA.
getID() == Intrinsic::vp_load ||
1168 MICA.
getID() == Intrinsic::vp_store) {
1180 bool UseMaskForCond,
bool UseMaskForGaps)
const {
1186 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1190 if (LT.second.isVector()) {
1196 VTy->getElementCount().divideCoefficientBy(Factor));
1197 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1198 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1203 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1204 unsigned VecSizeInBits =
1205 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1206 unsigned VLENForTuning =
1208 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1210 MVT SubVecVT = getTLI()->getValueType(
DL, SubVecTy).getSimpleVT();
1211 Cost += Factor * TLI->getLMULCost(SubVecVT);
1217 unsigned NumLoads = getEstimatedVLFor(VTy);
1233 if (UseMaskForGaps) {
1236 "Indices should not contain duplicate elements");
1237 unsigned NumOfFields = Indices.
size();
1238 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.
back() + 1);
1239 if (IsTailGapOnly &&
1240 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1242 if (LT.second.isVector() &&
1243 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1245 FVTy->getElementType(),
1246 FVTy->getElementCount().divideCoefficientBy(Factor));
1247 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1250 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1259 unsigned VF = FVTy->getNumElements() / Factor;
1266 if (Opcode == Instruction::Load) {
1268 for (
unsigned Index : Indices) {
1272 Mask.resize(VF * Factor, -1);
1276 Cost += ShuffleCost;
1294 UseMaskForCond, UseMaskForGaps);
1296 assert(Opcode == Instruction::Store &&
"Opcode must be a store");
1303 return MemCost + ShuffleCost;
1310 bool IsLoad = MICA.
getID() == Intrinsic::masked_gather ||
1311 MICA.
getID() == Intrinsic::vp_gather;
1312 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1320 if ((Opcode == Instruction::Load &&
1322 (Opcode == Instruction::Store &&
1328 if (MICA.
getID() == Intrinsic::vp_gather ||
1329 MICA.
getID() == Intrinsic::vp_scatter) {
1332 if (DataLT.first > 1)
1334 if (PtrLT.first > 1)
1342 unsigned NumLoads = getEstimatedVLFor(&VTy);
1349 unsigned Opcode = MICA.
getID() == Intrinsic::masked_expandload
1351 : Instruction::Store;
1355 bool IsLegal = (Opcode == Instruction::Store &&
1357 (Opcode == Instruction::Load &&
1381 if (Opcode == Instruction::Store)
1382 Opcodes.
append({RISCV::VCOMPRESS_VM});
1384 Opcodes.
append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1386 LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1411 unsigned NumLoads = getEstimatedVLFor(&VTy);
1422 for (
auto *Ty : Tys) {
1423 if (!Ty->isVectorTy())
1437 {Intrinsic::floor, MVT::f32, 9},
1438 {Intrinsic::floor, MVT::f64, 9},
1439 {Intrinsic::ceil, MVT::f32, 9},
1440 {Intrinsic::ceil, MVT::f64, 9},
1441 {Intrinsic::trunc, MVT::f32, 7},
1442 {Intrinsic::trunc, MVT::f64, 7},
1443 {Intrinsic::round, MVT::f32, 9},
1444 {Intrinsic::round, MVT::f64, 9},
1445 {Intrinsic::roundeven, MVT::f32, 9},
1446 {Intrinsic::roundeven, MVT::f64, 9},
1447 {Intrinsic::rint, MVT::f32, 7},
1448 {Intrinsic::rint, MVT::f64, 7},
1449 {Intrinsic::nearbyint, MVT::f32, 9},
1450 {Intrinsic::nearbyint, MVT::f64, 9},
1451 {Intrinsic::bswap, MVT::i16, 3},
1452 {Intrinsic::bswap, MVT::i32, 12},
1453 {Intrinsic::bswap, MVT::i64, 31},
1454 {Intrinsic::bitreverse, MVT::i8, 17},
1455 {Intrinsic::bitreverse, MVT::i16, 24},
1456 {Intrinsic::bitreverse, MVT::i32, 33},
1457 {Intrinsic::bitreverse, MVT::i64, 52},
1458 {Intrinsic::ctpop, MVT::i8, 12},
1459 {Intrinsic::ctpop, MVT::i16, 19},
1460 {Intrinsic::ctpop, MVT::i32, 20},
1461 {Intrinsic::ctpop, MVT::i64, 21},
1462 {Intrinsic::ctlz, MVT::i8, 19},
1463 {Intrinsic::ctlz, MVT::i16, 28},
1464 {Intrinsic::ctlz, MVT::i32, 31},
1465 {Intrinsic::ctlz, MVT::i64, 35},
1466 {Intrinsic::cttz, MVT::i8, 16},
1467 {Intrinsic::cttz, MVT::i16, 23},
1468 {Intrinsic::cttz, MVT::i32, 24},
1469 {Intrinsic::cttz, MVT::i64, 25},
1476 switch (ICA.
getID()) {
1477 case Intrinsic::lrint:
1478 case Intrinsic::llrint:
1479 case Intrinsic::lround:
1480 case Intrinsic::llround: {
1484 if (ST->hasVInstructions() && LT.second.isVector()) {
1486 unsigned SrcEltSz =
DL.getTypeSizeInBits(SrcTy->getScalarType());
1487 unsigned DstEltSz =
DL.getTypeSizeInBits(RetTy->getScalarType());
1488 if (LT.second.getVectorElementType() == MVT::bf16) {
1489 if (!ST->hasVInstructionsBF16Minimal())
1492 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1494 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1495 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1496 !ST->hasVInstructionsF16()) {
1497 if (!ST->hasVInstructionsF16Minimal())
1500 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1502 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1504 }
else if (SrcEltSz > DstEltSz) {
1505 Ops = {RISCV::VFNCVT_X_F_W};
1506 }
else if (SrcEltSz < DstEltSz) {
1507 Ops = {RISCV::VFWCVT_X_F_V};
1509 Ops = {RISCV::VFCVT_X_F_V};
1514 if (SrcEltSz > DstEltSz)
1515 return SrcLT.first *
1516 getRISCVInstructionCost(
Ops, SrcLT.second,
CostKind);
1517 return LT.first * getRISCVInstructionCost(
Ops, LT.second,
CostKind);
1521 case Intrinsic::ceil:
1522 case Intrinsic::floor:
1523 case Intrinsic::trunc:
1524 case Intrinsic::rint:
1525 case Intrinsic::round:
1526 case Intrinsic::roundeven: {
1529 if (!LT.second.isVector() && TLI->isOperationCustom(
ISD::FCEIL, LT.second))
1530 return LT.first * 8;
1533 case Intrinsic::umin:
1534 case Intrinsic::umax:
1535 case Intrinsic::smin:
1536 case Intrinsic::smax: {
1538 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1541 if (ST->hasVInstructions() && LT.second.isVector()) {
1543 switch (ICA.
getID()) {
1544 case Intrinsic::umin:
1545 Op = RISCV::VMINU_VV;
1547 case Intrinsic::umax:
1548 Op = RISCV::VMAXU_VV;
1550 case Intrinsic::smin:
1551 Op = RISCV::VMIN_VV;
1553 case Intrinsic::smax:
1554 Op = RISCV::VMAX_VV;
1557 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1561 case Intrinsic::sadd_sat:
1562 case Intrinsic::ssub_sat:
1563 case Intrinsic::uadd_sat:
1564 case Intrinsic::usub_sat: {
1566 if (ST->hasVInstructions() && LT.second.isVector()) {
1568 switch (ICA.
getID()) {
1569 case Intrinsic::sadd_sat:
1570 Op = RISCV::VSADD_VV;
1572 case Intrinsic::ssub_sat:
1573 Op = RISCV::VSSUB_VV;
1575 case Intrinsic::uadd_sat:
1576 Op = RISCV::VSADDU_VV;
1578 case Intrinsic::usub_sat:
1579 Op = RISCV::VSSUBU_VV;
1582 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1586 case Intrinsic::fma:
1587 case Intrinsic::fmuladd: {
1590 if (ST->hasVInstructions() && LT.second.isVector())
1592 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second,
CostKind);
1595 case Intrinsic::fabs: {
1597 if (ST->hasVInstructions() && LT.second.isVector()) {
1603 if (LT.second.getVectorElementType() == MVT::bf16 ||
1604 (LT.second.getVectorElementType() == MVT::f16 &&
1605 !ST->hasVInstructionsF16()))
1606 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1611 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second,
CostKind);
1615 case Intrinsic::sqrt: {
1617 if (ST->hasVInstructions() && LT.second.isVector()) {
1620 MVT ConvType = LT.second;
1621 MVT FsqrtType = LT.second;
1624 if (LT.second.getVectorElementType() == MVT::bf16) {
1625 if (LT.second == MVT::nxv32bf16) {
1626 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1627 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1628 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1629 ConvType = MVT::nxv16f16;
1630 FsqrtType = MVT::nxv16f32;
1632 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1633 FsqrtOp = {RISCV::VFSQRT_V};
1634 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1636 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1637 !ST->hasVInstructionsF16()) {
1638 if (LT.second == MVT::nxv32f16) {
1639 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1640 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1641 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1642 ConvType = MVT::nxv16f16;
1643 FsqrtType = MVT::nxv16f32;
1645 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1646 FsqrtOp = {RISCV::VFSQRT_V};
1647 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1650 FsqrtOp = {RISCV::VFSQRT_V};
1653 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType,
CostKind) +
1654 getRISCVInstructionCost(ConvOp, ConvType,
CostKind));
1658 case Intrinsic::cttz:
1659 case Intrinsic::ctlz:
1660 case Intrinsic::ctpop: {
1662 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1664 switch (ICA.
getID()) {
1665 case Intrinsic::cttz:
1668 case Intrinsic::ctlz:
1671 case Intrinsic::ctpop:
1672 Op = RISCV::VCPOP_V;
1675 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1679 case Intrinsic::abs: {
1681 if (ST->hasVInstructions() && LT.second.isVector()) {
1683 if (ST->hasStdExtZvabd())
1685 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second,
CostKind);
1690 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1695 case Intrinsic::fshl:
1696 case Intrinsic::fshr: {
1703 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1705 (RetTy->getIntegerBitWidth() == 32 ||
1706 RetTy->getIntegerBitWidth() == 64) &&
1707 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1712 case Intrinsic::clmul: {
1714 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbkc()) {
1717 if (!ST->is64Bit() || LT.second != MVT::i64)
1723 return LT.first * getRISCVInstructionCost(
1724 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1729 case Intrinsic::masked_udiv:
1732 case Intrinsic::masked_sdiv:
1735 case Intrinsic::masked_urem:
1738 case Intrinsic::masked_srem:
1741 case Intrinsic::get_active_lane_mask: {
1742 if (ST->hasVInstructions()) {
1751 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1757 case Intrinsic::stepvector: {
1761 if (ST->hasVInstructions())
1762 return getRISCVInstructionCost(RISCV::VID_V, LT.second,
CostKind) +
1764 getRISCVInstructionCost(RISCV::VADD_VX, LT.second,
CostKind);
1765 return 1 + (LT.first - 1);
1767 case Intrinsic::vector_splice_left:
1768 case Intrinsic::vector_splice_right: {
1773 if (ST->hasVInstructions() && LT.second.isVector()) {
1775 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1780 case Intrinsic::experimental_cttz_elts: {
1781 if (!ST->hasVInstructions())
1786 if (!LT.second.isVector())
1790 if (LT.second.getVectorElementType() != MVT::i1)
1791 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second,
CostKind);
1793 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind);
1805 return LT.first *
Cost;
1807 case Intrinsic::experimental_vp_splice: {
1815 case Intrinsic::vp_merge: {
1823 case Intrinsic::fptoui_sat:
1824 case Intrinsic::fptosi_sat: {
1826 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1831 if (!SrcTy->isVectorTy())
1834 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1851 case Intrinsic::experimental_vector_extract_last_active: {
1873 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1874 TLI->getVectorIdxTy(
getDataLayout()), MaskTy->getElementCount(),
1875 true, &VScaleRange);
1876 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1884 if (StepLT.first > 1)
1888 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1890 Cost += MaskLT.first *
1891 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second,
CostKind);
1893 Cost += StepLT.first *
1894 getRISCVInstructionCost(Opcodes, StepLT.second,
CostKind);
1898 Cost += ValLT.first *
1899 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1905 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1907 LT.second.isVector()) {
1908 MVT EltTy = LT.second.getVectorElementType();
1910 ICA.
getID(), EltTy))
1911 return LT.first * Entry->Cost;
1924 if (ST->hasVInstructions() && PtrTy->
isVectorTy())
1942 if (ST->hasStdExtP() &&
1950 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1951 Dst->getScalarSizeInBits() > ST->getELen())
1954 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1969 if (Src->getScalarSizeInBits() == 1) {
1974 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second,
CostKind) +
1975 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1981 if (Dst->getScalarSizeInBits() == 1) {
1987 return SrcLT.first *
1988 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
2000 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
2001 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2003 SrcLT.second.getSizeInBits()) ||
2005 DstLT.second.getSizeInBits()) ||
2006 SrcLT.first > 1 || DstLT.first > 1)
2010 assert((SrcLT.first == 1) && (DstLT.first == 1) &&
"Illegal type");
2012 int PowDiff = (int)
Log2_32(DstLT.second.getScalarSizeInBits()) -
2013 (int)
Log2_32(SrcLT.second.getScalarSizeInBits());
2017 if ((PowDiff < 1) || (PowDiff > 3))
2019 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2020 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2023 return getRISCVInstructionCost(
Op, DstLT.second,
CostKind);
2029 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2030 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2034 : RISCV::VFNCVT_F_F_W;
2036 for (; SrcEltSize != DstEltSize;) {
2040 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2042 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2050 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2052 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2054 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2055 unsigned SrcEltSize = Src->getScalarSizeInBits();
2056 unsigned DstEltSize = Dst->getScalarSizeInBits();
2058 if ((SrcEltSize == 16) &&
2059 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2065 std::pair<InstructionCost, MVT> VecF32LT =
2068 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2073 if (DstEltSize == SrcEltSize)
2074 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2075 else if (DstEltSize > SrcEltSize)
2076 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2081 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2082 Cost += getRISCVInstructionCost(FNCVT, VecVT,
CostKind);
2083 if ((SrcEltSize / 2) > DstEltSize) {
2094 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2095 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2096 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2097 unsigned SrcEltSize = Src->getScalarSizeInBits();
2098 unsigned DstEltSize = Dst->getScalarSizeInBits();
2101 if ((DstEltSize == 16) &&
2102 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2108 std::pair<InstructionCost, MVT> VecF32LT =
2111 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2116 if (DstEltSize == SrcEltSize)
2117 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2118 else if (DstEltSize > SrcEltSize) {
2119 if ((DstEltSize / 2) > SrcEltSize) {
2123 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2126 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2128 Cost += getRISCVInstructionCost(FNCVT, DstLT.second,
CostKind);
2135unsigned RISCVTTIImpl::getEstimatedVLFor(
VectorType *Ty)
const {
2137 const unsigned EltSize =
DL.getTypeSizeInBits(Ty->getElementType());
2138 const unsigned MinSize =
DL.getTypeSizeInBits(Ty).getKnownMinValue();
2153 if (Ty->getScalarSizeInBits() > ST->getELen())
2157 if (Ty->getElementType()->isIntegerTy(1)) {
2161 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2167 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2171 case Intrinsic::maximum:
2173 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2175 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2190 case Intrinsic::minimum:
2192 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2194 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2200 const unsigned EltTyBits =
DL.getTypeSizeInBits(DstTy);
2209 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2218 case Intrinsic::smax:
2219 SplitOp = RISCV::VMAX_VV;
2220 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2222 case Intrinsic::smin:
2223 SplitOp = RISCV::VMIN_VV;
2224 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2226 case Intrinsic::umax:
2227 SplitOp = RISCV::VMAXU_VV;
2228 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2230 case Intrinsic::umin:
2231 SplitOp = RISCV::VMINU_VV;
2232 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2234 case Intrinsic::maxnum:
2235 SplitOp = RISCV::VFMAX_VV;
2236 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2238 case Intrinsic::minnum:
2239 SplitOp = RISCV::VFMIN_VV;
2240 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2245 (LT.first > 1) ? (LT.first - 1) *
2246 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2248 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2253 std::optional<FastMathFlags> FMF,
2259 if (Ty->getScalarSizeInBits() > ST->getELen())
2262 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2270 Type *ElementTy = Ty->getElementType();
2275 if (LT.second == MVT::v1i1)
2276 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind) +
2294 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2295 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second,
CostKind) +
2296 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second,
CostKind) +
2297 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2306 return (LT.first - 1) *
2307 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind) +
2308 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) + 1;
2316 return (LT.first - 1) *
2317 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second,
CostKind) +
2318 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2331 SplitOp = RISCV::VADD_VV;
2332 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2335 SplitOp = RISCV::VOR_VV;
2336 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2339 SplitOp = RISCV::VXOR_VV;
2340 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2343 SplitOp = RISCV::VAND_VV;
2344 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2348 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2349 LT.second.getScalarType() == MVT::bf16)
2353 for (
unsigned i = 0; i < LT.first.getValue(); i++)
2356 return getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2358 SplitOp = RISCV::VFADD_VV;
2359 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2364 (LT.first > 1) ? (LT.first - 1) *
2365 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2367 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2371 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *ValTy,
2382 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2388 if (IsUnsigned && Opcode == Instruction::Add &&
2389 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2393 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind);
2400 return (LT.first - 1) +
2407 assert(OpInfo.isConstant() &&
"non constant operand?");
2414 if (OpInfo.isUniform())
2420 return getConstantPoolLoadCost(Ty,
CostKind);
2429 EVT VT = TLI->getValueType(
DL, Src,
true);
2431 if (VT == MVT::Other ||
2437 if (Opcode == Instruction::Store && OpInfo.isConstant())
2452 if (Src->
isVectorTy() && LT.second.isVector() &&
2454 LT.second.getSizeInBits()))
2464 if (ST->hasVInstructions() && LT.second.isVector() &&
2466 BaseCost *= TLI->getLMULCost(LT.second);
2467 return Cost + BaseCost;
2476 Op1Info, Op2Info,
I);
2480 Op1Info, Op2Info,
I);
2485 Op1Info, Op2Info,
I);
2487 auto GetConstantMatCost =
2489 if (OpInfo.isUniform())
2494 return getConstantPoolLoadCost(ValTy,
CostKind);
2499 ConstantMatCost += GetConstantMatCost(Op1Info);
2501 ConstantMatCost += GetConstantMatCost(Op2Info);
2504 if (Opcode == Instruction::Select && LT.second.isVector()) {
2505 if (CondTy->isVectorTy()) {
2510 return ConstantMatCost +
2512 getRISCVInstructionCost(
2513 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2517 return ConstantMatCost +
2518 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2528 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2529 return ConstantMatCost +
2531 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2533 LT.first * getRISCVInstructionCost(
2534 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2541 return ConstantMatCost +
2542 LT.first * getRISCVInstructionCost(
2543 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2547 if ((Opcode == Instruction::ICmp) && ValTy->
isVectorTy() &&
2551 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2556 if ((Opcode == Instruction::FCmp) && ValTy->
isVectorTy() &&
2561 return ConstantMatCost +
2562 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind);
2572 Op1Info, Op2Info,
I);
2581 return ConstantMatCost +
2582 LT.first * getRISCVInstructionCost(
2583 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2590 return ConstantMatCost +
2592 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2601 return ConstantMatCost +
2603 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second,
CostKind);
2616 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2617 U->getType()->isIntegerTy() &&
2618 !isa<ConstantData>(U->getOperand(1)) &&
2619 !isa<ConstantData>(U->getOperand(2));
2627 Op1Info, Op2Info,
I);
2634 return Opcode == Instruction::PHI ? 0 : 1;
2651 if (Opcode != Instruction::ExtractElement &&
2652 Opcode != Instruction::InsertElement)
2658 if (Opcode == Instruction::InsertElement &&
2659 VIC == TTI::VectorInstrContext::SplatOpFolded &&
2660 ST->sinkSplatOperands() && Index == 0)
2667 if (!LT.second.isVector()) {
2677 auto NumElems = FixedVecTy->getNumElements();
2683 return Opcode == Instruction::ExtractElement
2684 ? StoreCost * NumElems + LoadCost
2685 : (StoreCost + LoadCost) * NumElems + StoreCost;
2689 if (LT.second.isScalableVector() && !LT.first.isValid())
2697 if (Opcode == Instruction::ExtractElement) {
2703 return ExtendCost + ExtractCost;
2713 return ExtendCost + InsertCost + TruncCost;
2720 if (LT.second.isFloatingPoint())
2721 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2725 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2727 getRISCVInstructionCost(MoveOpc, LT.second,
CostKind);
2729 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2734 if (LT.second.isFixedLengthVector()) {
2735 unsigned Width = LT.second.getVectorNumElements();
2736 Index = Index % Width;
2741 if (
auto VLEN = ST->getRealVLen()) {
2742 unsigned EltSize = LT.second.getScalarSizeInBits();
2743 unsigned M1Max = *VLEN / EltSize;
2744 Index = Index % M1Max;
2750 else if (Opcode == Instruction::InsertElement)
2758 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2759 LT.second.isScalableVector()))) {
2761 Align VecAlign =
DL.getPrefTypeAlign(Val);
2762 Align SclAlign =
DL.getPrefTypeAlign(ScalarType);
2767 if (Opcode == Instruction::ExtractElement)
2803 Opcode == Instruction::InsertElement
2804 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2805 RISCV::VSLIDE1DOWN_VX,
2806 RISCV::VSLIDEUP_VX},
2808 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2809 RISCV::VSRL_VX, RISCV::VMV_X_S},
2812 return BaseCost + SlideCost;
2818 unsigned Index)
const {
2827 assert(Index < EC.getKnownMinValue() &&
"Unexpected reverse index");
2829 EC.getKnownMinValue() - 1 - Index,
nullptr,
2838std::optional<InstructionCost>
2844 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2846 if (Opcode == Instruction::UDiv)
2853 return std::nullopt;
2875 if (std::optional<InstructionCost> CombinedCost =
2877 Op2Info, Args, CxtI))
2878 return *CombinedCost;
2882 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2885 if (!LT.second.isVector()) {
2895 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2896 if (
const auto *Entry =
CostTableLookup(DivTbl, ISDOpcode, LT.second))
2897 return Entry->Cost * LT.first;
2906 if ((LT.second.getVectorElementType() == MVT::f16 ||
2907 LT.second.getVectorElementType() == MVT::bf16) &&
2908 TLI->getOperationAction(ISDOpcode, LT.second) ==
2910 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2914 CastCost += LT.first * Args.size() *
2922 LT.second = PromotedVT;
2925 auto getConstantMatCost =
2935 return getConstantPoolLoadCost(Ty,
CostKind);
2941 ConstantMatCost += getConstantMatCost(0, Op1Info);
2943 ConstantMatCost += getConstantMatCost(1, Op2Info);
2946 switch (ISDOpcode) {
2949 Op = RISCV::VADD_VV;
2954 Op = RISCV::VSLL_VV;
2959 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2964 Op = RISCV::VMUL_VV;
2968 Op = RISCV::VDIV_VV;
2972 Op = RISCV::VREM_VV;
2976 Op = RISCV::VFADD_VV;
2979 Op = RISCV::VFMUL_VV;
2982 Op = RISCV::VFDIV_VV;
2985 Op = RISCV::VFSGNJN_VV;
2990 return CastCost + ConstantMatCost +
2999 if (Ty->isFPOrFPVectorTy())
3001 return CastCost + ConstantMatCost + LT.first *
InstrCost;
3024 if (Info.isSameBase() && V !=
Base) {
3025 if (
GEP->hasAllConstantIndices())
3031 unsigned Stride =
DL.getTypeStoreSize(AccessTy);
3032 if (Info.isUnitStride() &&
3038 GEP->getType()->getPointerAddressSpace()))
3041 {TTI::OK_AnyValue, TTI::OP_None},
3042 {TTI::OK_AnyValue, TTI::OP_None}, {});
3059 if (ST->enableDefaultUnroll())
3069 if (L->getHeader()->getParent()->hasOptSize())
3073 L->getExitingBlocks(ExitingBlocks);
3075 <<
"Blocks: " << L->getNumBlocks() <<
"\n"
3076 <<
"Exit blocks: " << ExitingBlocks.
size() <<
"\n");
3080 if (ExitingBlocks.
size() > 2)
3085 if (L->getNumBlocks() > 4)
3093 for (
auto *BB : L->getBlocks()) {
3094 for (
auto &
I : *BB) {
3098 if (IsVectorized && (
I.getType()->isVectorTy() ||
3100 return V->getType()->isVectorTy();
3139 bool HasMask =
false;
3142 bool IsWrite) -> int64_t {
3143 if (
auto *TarExtTy =
3145 return TarExtTy->getIntParameter(0);
3151 case Intrinsic::riscv_vle_mask:
3152 case Intrinsic::riscv_vse_mask:
3153 case Intrinsic::riscv_vlseg2_mask:
3154 case Intrinsic::riscv_vlseg3_mask:
3155 case Intrinsic::riscv_vlseg4_mask:
3156 case Intrinsic::riscv_vlseg5_mask:
3157 case Intrinsic::riscv_vlseg6_mask:
3158 case Intrinsic::riscv_vlseg7_mask:
3159 case Intrinsic::riscv_vlseg8_mask:
3160 case Intrinsic::riscv_vsseg2_mask:
3161 case Intrinsic::riscv_vsseg3_mask:
3162 case Intrinsic::riscv_vsseg4_mask:
3163 case Intrinsic::riscv_vsseg5_mask:
3164 case Intrinsic::riscv_vsseg6_mask:
3165 case Intrinsic::riscv_vsseg7_mask:
3166 case Intrinsic::riscv_vsseg8_mask:
3169 case Intrinsic::riscv_vle:
3170 case Intrinsic::riscv_vse:
3171 case Intrinsic::riscv_vlseg2:
3172 case Intrinsic::riscv_vlseg3:
3173 case Intrinsic::riscv_vlseg4:
3174 case Intrinsic::riscv_vlseg5:
3175 case Intrinsic::riscv_vlseg6:
3176 case Intrinsic::riscv_vlseg7:
3177 case Intrinsic::riscv_vlseg8:
3178 case Intrinsic::riscv_vsseg2:
3179 case Intrinsic::riscv_vsseg3:
3180 case Intrinsic::riscv_vsseg4:
3181 case Intrinsic::riscv_vsseg5:
3182 case Intrinsic::riscv_vsseg6:
3183 case Intrinsic::riscv_vsseg7:
3184 case Intrinsic::riscv_vsseg8: {
3201 Ty = TarExtTy->getTypeParameter(0U);
3206 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3207 unsigned VLIndex = RVVIInfo->VLOperand;
3208 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3216 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3219 unsigned ElemSize = Ty->getScalarSizeInBits();
3223 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3224 Alignment, Mask, EVL);
3227 case Intrinsic::riscv_vlse_mask:
3228 case Intrinsic::riscv_vsse_mask:
3229 case Intrinsic::riscv_vlsseg2_mask:
3230 case Intrinsic::riscv_vlsseg3_mask:
3231 case Intrinsic::riscv_vlsseg4_mask:
3232 case Intrinsic::riscv_vlsseg5_mask:
3233 case Intrinsic::riscv_vlsseg6_mask:
3234 case Intrinsic::riscv_vlsseg7_mask:
3235 case Intrinsic::riscv_vlsseg8_mask:
3236 case Intrinsic::riscv_vssseg2_mask:
3237 case Intrinsic::riscv_vssseg3_mask:
3238 case Intrinsic::riscv_vssseg4_mask:
3239 case Intrinsic::riscv_vssseg5_mask:
3240 case Intrinsic::riscv_vssseg6_mask:
3241 case Intrinsic::riscv_vssseg7_mask:
3242 case Intrinsic::riscv_vssseg8_mask:
3245 case Intrinsic::riscv_vlse:
3246 case Intrinsic::riscv_vsse:
3247 case Intrinsic::riscv_vlsseg2:
3248 case Intrinsic::riscv_vlsseg3:
3249 case Intrinsic::riscv_vlsseg4:
3250 case Intrinsic::riscv_vlsseg5:
3251 case Intrinsic::riscv_vlsseg6:
3252 case Intrinsic::riscv_vlsseg7:
3253 case Intrinsic::riscv_vlsseg8:
3254 case Intrinsic::riscv_vssseg2:
3255 case Intrinsic::riscv_vssseg3:
3256 case Intrinsic::riscv_vssseg4:
3257 case Intrinsic::riscv_vssseg5:
3258 case Intrinsic::riscv_vssseg6:
3259 case Intrinsic::riscv_vssseg7:
3260 case Intrinsic::riscv_vssseg8: {
3277 Ty = TarExtTy->getTypeParameter(0U);
3282 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3283 unsigned VLIndex = RVVIInfo->VLOperand;
3284 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3293 unsigned PointerAlign = Alignment.valueOrOne().value();
3296 Alignment =
Align(1);
3303 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3306 unsigned ElemSize = Ty->getScalarSizeInBits();
3310 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3311 Alignment, Mask, EVL, Stride);
3314 case Intrinsic::riscv_vloxei_mask:
3315 case Intrinsic::riscv_vluxei_mask:
3316 case Intrinsic::riscv_vsoxei_mask:
3317 case Intrinsic::riscv_vsuxei_mask:
3318 case Intrinsic::riscv_vloxseg2_mask:
3319 case Intrinsic::riscv_vloxseg3_mask:
3320 case Intrinsic::riscv_vloxseg4_mask:
3321 case Intrinsic::riscv_vloxseg5_mask:
3322 case Intrinsic::riscv_vloxseg6_mask:
3323 case Intrinsic::riscv_vloxseg7_mask:
3324 case Intrinsic::riscv_vloxseg8_mask:
3325 case Intrinsic::riscv_vluxseg2_mask:
3326 case Intrinsic::riscv_vluxseg3_mask:
3327 case Intrinsic::riscv_vluxseg4_mask:
3328 case Intrinsic::riscv_vluxseg5_mask:
3329 case Intrinsic::riscv_vluxseg6_mask:
3330 case Intrinsic::riscv_vluxseg7_mask:
3331 case Intrinsic::riscv_vluxseg8_mask:
3332 case Intrinsic::riscv_vsoxseg2_mask:
3333 case Intrinsic::riscv_vsoxseg3_mask:
3334 case Intrinsic::riscv_vsoxseg4_mask:
3335 case Intrinsic::riscv_vsoxseg5_mask:
3336 case Intrinsic::riscv_vsoxseg6_mask:
3337 case Intrinsic::riscv_vsoxseg7_mask:
3338 case Intrinsic::riscv_vsoxseg8_mask:
3339 case Intrinsic::riscv_vsuxseg2_mask:
3340 case Intrinsic::riscv_vsuxseg3_mask:
3341 case Intrinsic::riscv_vsuxseg4_mask:
3342 case Intrinsic::riscv_vsuxseg5_mask:
3343 case Intrinsic::riscv_vsuxseg6_mask:
3344 case Intrinsic::riscv_vsuxseg7_mask:
3345 case Intrinsic::riscv_vsuxseg8_mask:
3348 case Intrinsic::riscv_vloxei:
3349 case Intrinsic::riscv_vluxei:
3350 case Intrinsic::riscv_vsoxei:
3351 case Intrinsic::riscv_vsuxei:
3352 case Intrinsic::riscv_vloxseg2:
3353 case Intrinsic::riscv_vloxseg3:
3354 case Intrinsic::riscv_vloxseg4:
3355 case Intrinsic::riscv_vloxseg5:
3356 case Intrinsic::riscv_vloxseg6:
3357 case Intrinsic::riscv_vloxseg7:
3358 case Intrinsic::riscv_vloxseg8:
3359 case Intrinsic::riscv_vluxseg2:
3360 case Intrinsic::riscv_vluxseg3:
3361 case Intrinsic::riscv_vluxseg4:
3362 case Intrinsic::riscv_vluxseg5:
3363 case Intrinsic::riscv_vluxseg6:
3364 case Intrinsic::riscv_vluxseg7:
3365 case Intrinsic::riscv_vluxseg8:
3366 case Intrinsic::riscv_vsoxseg2:
3367 case Intrinsic::riscv_vsoxseg3:
3368 case Intrinsic::riscv_vsoxseg4:
3369 case Intrinsic::riscv_vsoxseg5:
3370 case Intrinsic::riscv_vsoxseg6:
3371 case Intrinsic::riscv_vsoxseg7:
3372 case Intrinsic::riscv_vsoxseg8:
3373 case Intrinsic::riscv_vsuxseg2:
3374 case Intrinsic::riscv_vsuxseg3:
3375 case Intrinsic::riscv_vsuxseg4:
3376 case Intrinsic::riscv_vsuxseg5:
3377 case Intrinsic::riscv_vsuxseg6:
3378 case Intrinsic::riscv_vsuxseg7:
3379 case Intrinsic::riscv_vsuxseg8: {
3396 Ty = TarExtTy->getTypeParameter(0U);
3401 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3402 unsigned VLIndex = RVVIInfo->VLOperand;
3403 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3416 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3419 unsigned ElemSize = Ty->getScalarSizeInBits();
3424 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3425 Align(1), Mask, EVL,
3434 if (Ty->isVectorTy()) {
3437 if ((EltTy->
isHalfTy() && !ST->hasVInstructionsF16()) ||
3443 if (
Size.isScalable() && ST->hasVInstructions())
3446 if (ST->useRVVForFixedLengthVectors())
3466 return std::max<unsigned>(1U, RegWidth.
getFixedValue() / ElemWidth);
3474 return ST->enableUnalignedVectorMem();
3480 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3502 Align Alignment)
const {
3504 if (!VTy || VTy->isScalableTy())
3512 if (VTy->getElementType()->isIntegerTy(8))
3513 if (VTy->getElementCount().getFixedValue() > 256)
3514 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3515 ST->getMaxLMULForFixedLengthVectors();
3520 Align Alignment)
const {
3522 if (!VTy || VTy->isScalableTy())
3533 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3536 return TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, ElementTy));
3545 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
3546 bool Considerable =
false;
3547 AllowPromotionWithoutCommonHeader =
false;
3550 Type *ConsideredSExtType =
3552 if (
I.getType() != ConsideredSExtType)
3556 for (
const User *U :
I.users()) {
3558 Considerable =
true;
3562 if (GEPInst->getNumOperands() > 2) {
3563 AllowPromotionWithoutCommonHeader =
true;
3568 return Considerable;
3573 case Instruction::Add:
3574 case Instruction::Sub:
3575 case Instruction::Mul:
3576 case Instruction::And:
3577 case Instruction::Or:
3578 case Instruction::Xor:
3579 case Instruction::FAdd:
3580 case Instruction::FSub:
3581 case Instruction::FMul:
3582 case Instruction::FDiv:
3583 case Instruction::ICmp:
3584 case Instruction::FCmp:
3586 case Instruction::Shl:
3587 case Instruction::LShr:
3588 case Instruction::AShr:
3589 case Instruction::UDiv:
3590 case Instruction::SDiv:
3591 case Instruction::URem:
3592 case Instruction::SRem:
3593 case Instruction::Select:
3594 return Operand == 1;
3601 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3611 switch (
II->getIntrinsicID()) {
3612 case Intrinsic::fma:
3613 case Intrinsic::fmuladd:
3614 return Operand == 0 || Operand == 1;
3615 case Intrinsic::vp_udiv:
3616 case Intrinsic::vp_sdiv:
3617 case Intrinsic::vp_urem:
3618 case Intrinsic::vp_srem:
3619 case Intrinsic::ssub_sat:
3620 case Intrinsic::usub_sat:
3621 return Operand == 1;
3623 case Intrinsic::smin:
3624 case Intrinsic::umin:
3625 case Intrinsic::smax:
3626 case Intrinsic::umax:
3627 case Intrinsic::sadd_sat:
3628 case Intrinsic::uadd_sat:
3629 return Operand == 0 || Operand == 1;
3638 GatherUseOps)
const {
3639 if (Scalars.
empty() || !ST->hasVInstructions() || !ST->sinkSplatOperands() ||
3644 if (SplatIt == Scalars.
end() || (*SplatIt)->getType()->isIntegerTy(1) ||
3650 if (!GatherUseOps(UserOps) || UserOps.
empty())
3669 if (
I->isBitwiseLogicOp()) {
3670 if (!
I->getType()->isVectorTy()) {
3671 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3672 for (
auto &
Op :
I->operands()) {
3680 }
else if (
I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3681 for (
auto &
Op :
I->operands()) {
3693 Ops.push_back(&Not);
3694 Ops.push_back(&InsertElt);
3702 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3710 if (!ST->sinkSplatOperands())
3713 for (
auto OpIdx :
enumerate(
I->operands())) {
3733 for (
Use &U :
Op->uses()) {
3740 Use *InsertEltUse = &
Op->getOperandUse(0);
3743 Ops.push_back(&InsertElt->getOperandUse(1));
3744 Ops.push_back(InsertEltUse);
3745 Ops.push_back(&OpIdx.value());
3754 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3757 Options.AllowOverlappingLoads =
true;
3758 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3760 if (ST->is64Bit()) {
3761 Options.LoadSizes = {8, 4, 2, 1};
3762 Options.AllowedTailExpansions = {3, 5, 6};
3764 Options.LoadSizes = {4, 2, 1};
3765 Options.AllowedTailExpansions = {3};
3768 if (IsZeroCmp && ST->hasVInstructions()) {
3769 unsigned VLenB = ST->getRealMinVLen() / 8;
3772 unsigned MinSize = ST->getXLen() / 8 + 1;
3773 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3787 if (
I->getOpcode() == Instruction::Or &&
3791 if (
I->getOpcode() == Instruction::Add ||
3792 I->getOpcode() == Instruction::Sub)
3810std::optional<Instruction *>
3816 if (
is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3817 II.getIntrinsicID())) {
3820 if (!ST->hasVInstructions())
3823 bool HasAVL =
II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3824 unsigned Offset = HasAVL ? 1 : 0;
3825 unsigned BitWidth =
II.getType()->getIntegerBitWidth();
3850 Value *AVL =
II.getArgOperand(0);
3879 II.getRange().value_or(ConstantRange::getFull(
BitWidth));
3881 if (NewRange != OldRange) {
3882 II.addRangeRetAttr(NewRange);
3892 if (
II.user_empty())
3897 const APInt *Scalar;
3902 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3906 unsigned TargetEltBW =
DL.getTypeSizeInBits(TargetVecTy->getElementType());
3907 unsigned SourceEltBW =
DL.getTypeSizeInBits(SourceVecTy->getElementType());
3908 if (TargetEltBW % SourceEltBW)
3910 unsigned TargetScale = TargetEltBW / SourceEltBW;
3911 if (VL % TargetScale || TargetScale == 1)
3913 Type *VLTy =
II.getOperand(2)->getType();
3914 ElementCount SourceEC = SourceVecTy->getElementCount();
3915 unsigned NewEltBW = SourceEltBW * TargetScale;
3917 !
DL.fitsInLegalInteger(NewEltBW))
3920 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, NewEltTy)))
3924 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3925 "Lossless bitcast between types expected");
3931 RetTy, Intrinsic::riscv_vmv_v_x,
3932 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3933 ConstantInt::get(VLTy, VL / TargetScale)}),
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
uint64_t IntrinsicInst * II
This file describes how to lower LLVM code to machine code.
Class for arbitrary precision integers.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
@ ICMP_SLT
signed less than
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Convenience struct for specifying and reasoning about fast-math flags.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
Intrinsic::ID getID() const
bool isTypeBasedOnly() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Represents a single loop in the control flow graph.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
Align getAlignment() const
unsigned getAddressSpace() const
Type * getDataType() const
bool getVariableMask() const
Intrinsic::ID getID() const
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
TargetTransformInfo::VectorInstrContext getBuildVectorContextHint(ArrayRef< int > Mask, ArrayRef< Value * > Scalars, function_ref< bool(SmallVectorImpl< TargetTransformInfo::BuildVectorUseOp > &)> GatherUseOps) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
An efficient, type-erasing, non-owning reference to a callable.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
@ SplatOpFolded
All of the value's users support splatting the value.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
auto find_if_not(R &&Range, UnaryPredicate P)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const