18#include "llvm/IR/IntrinsicsRISCV.h"
26#define DEBUG_TYPE "riscvtti"
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
44 cl::desc(
"Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
57 size_t NumInstr = OpCodes.size();
62 return LMULCost * NumInstr;
64 for (
auto Op : OpCodes) {
66 case RISCV::VRGATHER_VI:
69 case RISCV::VRGATHER_VV:
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
97 case RISCV::VFREDOSUM_VS: {
105 case RISCV::VFMV_F_S:
110 case RISCV::VFMV_S_F:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
117 case RISCV::VFIRST_M:
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
160 if (!BO || !BO->hasOneUse())
163 if (BO->getOpcode() != Instruction::Shl)
174 if (ShAmt == Trailing)
191 if (!Cmp || !Cmp->isEquality())
207 if ((CmpC & Mask) != CmpC)
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
230 bool Takes12BitImm =
false;
231 unsigned ImmArgIdx = ~0U;
234 case Instruction::GetElementPtr:
239 case Instruction::Store: {
244 if (Idx == 1 || !Inst)
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(),
DL, getTLI()->getValueType(
DL, Ty),
257 case Instruction::Load:
260 case Instruction::And:
262 if (
Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
265 if (
Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
268 if (ST->hasStdExtZbs() && (~
Imm).isPowerOf2())
270 if (Inst && Idx == 1 &&
Imm.getBitWidth() <= ST->getXLen() &&
273 if (Inst && Idx == 1 &&
Imm.getBitWidth() == 64 &&
276 Takes12BitImm =
true;
278 case Instruction::Add:
279 Takes12BitImm =
true;
281 case Instruction::Or:
282 case Instruction::Xor:
284 if (ST->hasStdExtZbs() &&
Imm.isPowerOf2())
286 Takes12BitImm =
true;
288 case Instruction::Mul:
290 if (
Imm.isPowerOf2() ||
Imm.isNegatedPowerOf2())
293 if ((
Imm + 1).isPowerOf2() || (
Imm - 1).isPowerOf2())
296 Takes12BitImm =
true;
298 case Instruction::Sub:
299 case Instruction::Shl:
300 case Instruction::LShr:
301 case Instruction::AShr:
302 Takes12BitImm =
true;
313 if (
Imm.getSignificantBits() <= 64 &&
336 return ST->hasVInstructions();
346 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
350 if (Opcode == Instruction::FAdd)
359 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
360 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
361 InputTypeA != InputTypeB || !InputTypeA->
isIntegerTy(8) ||
377 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second,
CostKind);
386 std::pair<InstructionCost, MVT> AccLT =
394 bool WidenFirst =
false;
395 if (VF.
isScalable() && AccLT.second.isScalableVector()) {
396 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
409 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
412 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second,
CostKind);
416 std::pair<InstructionCost, MVT> RedLT =
418 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
420 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
424 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
436 switch (
II->getIntrinsicID()) {
440 case Intrinsic::vector_reduce_mul:
441 case Intrinsic::vector_reduce_fmul:
447 if (ST->hasVInstructions())
448 if (
unsigned MinVLen = ST->getRealMinVLen();
463 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
466 (ST->hasVInstructions() &&
489 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
495RISCVTTIImpl::getConstantPoolLoadCost(
Type *Ty,
500 return getStaticDataAddrGenerationCost(
CostKind) +
506 unsigned Size = Mask.size();
509 for (
unsigned I = 0;
I !=
Size; ++
I) {
510 if (
static_cast<unsigned>(Mask[
I]) ==
I)
516 for (
unsigned J =
I + 1; J !=
Size; ++J)
518 if (
static_cast<unsigned>(Mask[J]) != J %
I)
546 "Expected fixed vector type and non-empty mask");
549 unsigned NumOfDests =
divideCeil(Mask.size(), LegalNumElts);
553 if (NumOfDests <= 1 ||
555 Tp->getElementType()->getPrimitiveSizeInBits() ||
556 LegalNumElts >= Tp->getElementCount().getFixedValue())
559 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
562 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
566 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
567 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
568 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
570 assert(NormalizedVF >= Mask.size() &&
571 "Normalized mask expected to be not shorter than original mask.");
576 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
577 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
580 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
583 Cost +=
TTI.getShuffleCost(
586 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
588 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
589 Cost +=
TTI.getShuffleCost(
592 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
615 if (!VLen || Mask.empty())
619 LegalVT =
TTI.getTypeLegalizationCost(
625 if (NumOfDests <= 1 ||
627 Tp->getElementType()->getPrimitiveSizeInBits() ||
631 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
634 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
640 unsigned NormalizedVF =
645 assert(NormalizedVF >= Mask.size() &&
646 "Normalized mask expected to be not shorter than original mask.");
652 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
653 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
656 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
661 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
663 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
665 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
672 if ((NumOfDestRegs > 2 && NumShuffles <=
static_cast<int>(NumOfDestRegs)) ||
673 (NumOfDestRegs <= 2 && NumShuffles < 4))
688 if (!
LT.second.isFixedLengthVector())
696 auto GetSlideOpcode = [&](
int SlideAmt) {
698 bool IsVI =
isUInt<5>(std::abs(SlideAmt));
700 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
701 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
704 std::array<std::pair<int, int>, 2> SrcInfo;
708 if (SrcInfo[1].second == 0)
712 if (SrcInfo[0].second != 0) {
713 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
714 FirstSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
717 if (SrcInfo[1].first == -1)
718 return FirstSlideCost;
721 if (SrcInfo[1].second != 0) {
722 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
723 SecondSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
726 getRISCVInstructionCost(RISCV::VMERGE_VVM,
LT.second,
CostKind);
733 return FirstSlideCost + SecondSlideCost + MaskCost;
744 "Expected the Mask to match the return size if given");
746 "Expected the same scalar types");
762 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
764 *
this, LT.second, ST->getRealVLen(),
766 if (VRegSplittingCost.
isValid())
767 return VRegSplittingCost;
772 if (Mask.size() >= 2) {
773 MVT EltTp = LT.second.getVectorElementType();
784 return 2 * LT.first * TLI->getLMULCost(LT.second);
786 if (Mask[0] == 0 || Mask[0] == 1) {
790 if (
equal(DeinterleaveMask, Mask))
791 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
796 if (LT.second.getScalarSizeInBits() != 1 &&
799 unsigned NumSlides =
Log2_32(Mask.size() / SubVectorSize);
801 for (
unsigned I = 0;
I != NumSlides; ++
I) {
802 unsigned InsertIndex = SubVectorSize * (1 <<
I);
807 std::pair<InstructionCost, MVT> DestLT =
812 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
826 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
827 LT.second.getVectorNumElements() <= 256)) {
832 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second,
CostKind);
846 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
847 LT.second.getVectorNumElements() <= 256)) {
848 auto &
C = SrcTy->getContext();
849 auto EC = SrcTy->getElementCount();
854 return 2 * IndexCost +
855 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
874 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
902 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
903 if (std::optional<unsigned> VLen = ST->getRealVLen();
904 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
905 SubLT.second.getSizeInBits() <= *VLen)
913 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second,
CostKind);
920 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second,
CostKind);
932 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
939 if (IsLoad && LT.second.isVector() &&
941 LT.second.getVectorElementCount()))
945 Instruction::InsertElement);
946 if (LT.second.getScalarSizeInBits() == 1) {
954 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
967 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
968 RISCV::VMV_X_S, RISCV::VMV_V_X,
977 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second,
CostKind);
983 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second,
CostKind);
989 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
990 if (Index >= 0 && Index < 32)
991 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
992 else if (Index < 0 && Index > -32)
993 Opcodes[1] = RISCV::VSLIDEUP_VI;
994 return LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
998 if (!LT.second.isVector())
1004 if (SrcTy->getElementType()->isIntegerTy(1)) {
1016 MVT ContainerVT = LT.second;
1017 if (LT.second.isFixedLengthVector())
1018 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1020 if (ContainerVT.
bitsLE(M1VT)) {
1030 if (LT.second.isFixedLengthVector())
1032 LenCost =
isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1033 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1034 if (LT.second.isFixedLengthVector() &&
1035 isInt<5>(LT.second.getVectorNumElements() - 1))
1036 Opcodes[1] = RISCV::VRSUB_VI;
1038 getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1039 return LT.first * (LenCost + GatherCost);
1046 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1048 getRISCVInstructionCost(M1Opcodes, M1VT,
CostKind) + 3;
1052 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT,
CostKind) * Ratio;
1054 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second,
CostKind);
1055 return FixedCost + LT.first * (GatherCost + SlideCost);
1089 Ty, DemandedElts, Insert, Extract,
CostKind);
1091 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1092 if (Ty->getScalarSizeInBits() == 1) {
1102 assert(LT.second.isFixedLengthVector());
1103 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1107 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second,
CostKind);
1120 switch (MICA.
getID()) {
1121 case Intrinsic::vp_load_ff: {
1122 EVT DataTypeVT = TLI->getValueType(
DL, DataTy);
1123 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1130 case Intrinsic::experimental_vp_strided_load:
1131 case Intrinsic::experimental_vp_strided_store:
1133 case Intrinsic::masked_compressstore:
1134 case Intrinsic::masked_expandload:
1136 case Intrinsic::vp_scatter:
1137 case Intrinsic::vp_gather:
1138 case Intrinsic::masked_scatter:
1139 case Intrinsic::masked_gather:
1141 case Intrinsic::vp_load:
1142 case Intrinsic::vp_store:
1143 case Intrinsic::masked_load:
1144 case Intrinsic::masked_store:
1153 unsigned Opcode = MICA.
getID() == Intrinsic::masked_load ? Instruction::Load
1154 : Instruction::Store;
1165 if (MICA.
getID() == Intrinsic::vp_load ||
1166 MICA.
getID() == Intrinsic::vp_store) {
1178 bool UseMaskForCond,
bool UseMaskForGaps)
const {
1184 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1188 if (LT.second.isVector()) {
1194 VTy->getElementCount().divideCoefficientBy(Factor));
1195 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1196 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1201 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1202 unsigned VecSizeInBits =
1203 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1204 unsigned VLENForTuning =
1206 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1208 MVT SubVecVT = getTLI()->getValueType(
DL, SubVecTy).getSimpleVT();
1209 Cost += Factor * TLI->getLMULCost(SubVecVT);
1215 unsigned NumLoads = getEstimatedVLFor(VTy);
1231 if (UseMaskForGaps) {
1234 "Indices should not contain duplicate elements");
1235 unsigned NumOfFields = Indices.
size();
1236 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.
back() + 1);
1237 if (IsTailGapOnly &&
1238 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1240 if (LT.second.isVector() &&
1241 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1243 FVTy->getElementType(),
1244 FVTy->getElementCount().divideCoefficientBy(Factor));
1245 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1248 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1257 unsigned VF = FVTy->getNumElements() / Factor;
1264 if (Opcode == Instruction::Load) {
1266 for (
unsigned Index : Indices) {
1270 Mask.resize(VF * Factor, -1);
1274 Cost += ShuffleCost;
1292 UseMaskForCond, UseMaskForGaps);
1294 assert(Opcode == Instruction::Store &&
"Opcode must be a store");
1301 return MemCost + ShuffleCost;
1308 bool IsLoad = MICA.
getID() == Intrinsic::masked_gather ||
1309 MICA.
getID() == Intrinsic::vp_gather;
1310 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1318 if ((Opcode == Instruction::Load &&
1320 (Opcode == Instruction::Store &&
1326 if (MICA.
getID() == Intrinsic::vp_gather ||
1327 MICA.
getID() == Intrinsic::vp_scatter) {
1330 if (DataLT.first > 1)
1332 if (PtrLT.first > 1)
1340 unsigned NumLoads = getEstimatedVLFor(&VTy);
1347 unsigned Opcode = MICA.
getID() == Intrinsic::masked_expandload
1349 : Instruction::Store;
1353 bool IsLegal = (Opcode == Instruction::Store &&
1355 (Opcode == Instruction::Load &&
1379 if (Opcode == Instruction::Store)
1380 Opcodes.
append({RISCV::VCOMPRESS_VM});
1382 Opcodes.
append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1384 LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1409 unsigned NumLoads = getEstimatedVLFor(&VTy);
1420 for (
auto *Ty : Tys) {
1421 if (!Ty->isVectorTy())
1435 {Intrinsic::floor, MVT::f32, 9},
1436 {Intrinsic::floor, MVT::f64, 9},
1437 {Intrinsic::ceil, MVT::f32, 9},
1438 {Intrinsic::ceil, MVT::f64, 9},
1439 {Intrinsic::trunc, MVT::f32, 7},
1440 {Intrinsic::trunc, MVT::f64, 7},
1441 {Intrinsic::round, MVT::f32, 9},
1442 {Intrinsic::round, MVT::f64, 9},
1443 {Intrinsic::roundeven, MVT::f32, 9},
1444 {Intrinsic::roundeven, MVT::f64, 9},
1445 {Intrinsic::rint, MVT::f32, 7},
1446 {Intrinsic::rint, MVT::f64, 7},
1447 {Intrinsic::nearbyint, MVT::f32, 9},
1448 {Intrinsic::nearbyint, MVT::f64, 9},
1449 {Intrinsic::bswap, MVT::i16, 3},
1450 {Intrinsic::bswap, MVT::i32, 12},
1451 {Intrinsic::bswap, MVT::i64, 31},
1452 {Intrinsic::bitreverse, MVT::i8, 17},
1453 {Intrinsic::bitreverse, MVT::i16, 24},
1454 {Intrinsic::bitreverse, MVT::i32, 33},
1455 {Intrinsic::bitreverse, MVT::i64, 52},
1456 {Intrinsic::ctpop, MVT::i8, 12},
1457 {Intrinsic::ctpop, MVT::i16, 19},
1458 {Intrinsic::ctpop, MVT::i32, 20},
1459 {Intrinsic::ctpop, MVT::i64, 21},
1460 {Intrinsic::ctlz, MVT::i8, 19},
1461 {Intrinsic::ctlz, MVT::i16, 28},
1462 {Intrinsic::ctlz, MVT::i32, 31},
1463 {Intrinsic::ctlz, MVT::i64, 35},
1464 {Intrinsic::cttz, MVT::i8, 16},
1465 {Intrinsic::cttz, MVT::i16, 23},
1466 {Intrinsic::cttz, MVT::i32, 24},
1467 {Intrinsic::cttz, MVT::i64, 25},
1474 switch (ICA.
getID()) {
1475 case Intrinsic::lrint:
1476 case Intrinsic::llrint:
1477 case Intrinsic::lround:
1478 case Intrinsic::llround: {
1482 if (ST->hasVInstructions() && LT.second.isVector()) {
1484 unsigned SrcEltSz =
DL.getTypeSizeInBits(SrcTy->getScalarType());
1485 unsigned DstEltSz =
DL.getTypeSizeInBits(RetTy->getScalarType());
1486 if (LT.second.getVectorElementType() == MVT::bf16) {
1487 if (!ST->hasVInstructionsBF16Minimal())
1490 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1492 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1493 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1494 !ST->hasVInstructionsF16()) {
1495 if (!ST->hasVInstructionsF16Minimal())
1498 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1500 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1502 }
else if (SrcEltSz > DstEltSz) {
1503 Ops = {RISCV::VFNCVT_X_F_W};
1504 }
else if (SrcEltSz < DstEltSz) {
1505 Ops = {RISCV::VFWCVT_X_F_V};
1507 Ops = {RISCV::VFCVT_X_F_V};
1512 if (SrcEltSz > DstEltSz)
1513 return SrcLT.first *
1514 getRISCVInstructionCost(
Ops, SrcLT.second,
CostKind);
1515 return LT.first * getRISCVInstructionCost(
Ops, LT.second,
CostKind);
1519 case Intrinsic::ceil:
1520 case Intrinsic::floor:
1521 case Intrinsic::trunc:
1522 case Intrinsic::rint:
1523 case Intrinsic::round:
1524 case Intrinsic::roundeven: {
1527 if (!LT.second.isVector() && TLI->isOperationCustom(
ISD::FCEIL, LT.second))
1528 return LT.first * 8;
1531 case Intrinsic::umin:
1532 case Intrinsic::umax:
1533 case Intrinsic::smin:
1534 case Intrinsic::smax: {
1536 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1539 if (ST->hasVInstructions() && LT.second.isVector()) {
1541 switch (ICA.
getID()) {
1542 case Intrinsic::umin:
1543 Op = RISCV::VMINU_VV;
1545 case Intrinsic::umax:
1546 Op = RISCV::VMAXU_VV;
1548 case Intrinsic::smin:
1549 Op = RISCV::VMIN_VV;
1551 case Intrinsic::smax:
1552 Op = RISCV::VMAX_VV;
1555 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1559 case Intrinsic::sadd_sat:
1560 case Intrinsic::ssub_sat:
1561 case Intrinsic::uadd_sat:
1562 case Intrinsic::usub_sat: {
1564 if (ST->hasVInstructions() && LT.second.isVector()) {
1566 switch (ICA.
getID()) {
1567 case Intrinsic::sadd_sat:
1568 Op = RISCV::VSADD_VV;
1570 case Intrinsic::ssub_sat:
1571 Op = RISCV::VSSUB_VV;
1573 case Intrinsic::uadd_sat:
1574 Op = RISCV::VSADDU_VV;
1576 case Intrinsic::usub_sat:
1577 Op = RISCV::VSSUBU_VV;
1580 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1584 case Intrinsic::fma:
1585 case Intrinsic::fmuladd: {
1588 if (ST->hasVInstructions() && LT.second.isVector())
1590 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second,
CostKind);
1593 case Intrinsic::fabs: {
1595 if (ST->hasVInstructions() && LT.second.isVector()) {
1601 if (LT.second.getVectorElementType() == MVT::bf16 ||
1602 (LT.second.getVectorElementType() == MVT::f16 &&
1603 !ST->hasVInstructionsF16()))
1604 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1609 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second,
CostKind);
1613 case Intrinsic::sqrt: {
1615 if (ST->hasVInstructions() && LT.second.isVector()) {
1618 MVT ConvType = LT.second;
1619 MVT FsqrtType = LT.second;
1622 if (LT.second.getVectorElementType() == MVT::bf16) {
1623 if (LT.second == MVT::nxv32bf16) {
1624 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1625 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1626 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1627 ConvType = MVT::nxv16f16;
1628 FsqrtType = MVT::nxv16f32;
1630 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1631 FsqrtOp = {RISCV::VFSQRT_V};
1632 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1634 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1635 !ST->hasVInstructionsF16()) {
1636 if (LT.second == MVT::nxv32f16) {
1637 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1638 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1639 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1640 ConvType = MVT::nxv16f16;
1641 FsqrtType = MVT::nxv16f32;
1643 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1644 FsqrtOp = {RISCV::VFSQRT_V};
1645 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1648 FsqrtOp = {RISCV::VFSQRT_V};
1651 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType,
CostKind) +
1652 getRISCVInstructionCost(ConvOp, ConvType,
CostKind));
1656 case Intrinsic::cttz:
1657 case Intrinsic::ctlz:
1658 case Intrinsic::ctpop: {
1660 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1662 switch (ICA.
getID()) {
1663 case Intrinsic::cttz:
1666 case Intrinsic::ctlz:
1669 case Intrinsic::ctpop:
1670 Op = RISCV::VCPOP_V;
1673 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1677 case Intrinsic::abs: {
1679 if (ST->hasVInstructions() && LT.second.isVector()) {
1681 if (ST->hasStdExtZvabd())
1683 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second,
CostKind);
1688 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1693 case Intrinsic::fshl:
1694 case Intrinsic::fshr: {
1701 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1703 (RetTy->getIntegerBitWidth() == 32 ||
1704 RetTy->getIntegerBitWidth() == 64) &&
1705 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1710 case Intrinsic::clmul: {
1712 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbc() &&
1713 !ST->hasStdExtZbkc()) {
1716 if (!ST->is64Bit() || LT.second != MVT::i64)
1722 return LT.first * getRISCVInstructionCost(
1723 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1728 case Intrinsic::masked_udiv:
1731 case Intrinsic::masked_sdiv:
1734 case Intrinsic::masked_urem:
1737 case Intrinsic::masked_srem:
1740 case Intrinsic::get_active_lane_mask: {
1741 if (ST->hasVInstructions()) {
1750 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1756 case Intrinsic::stepvector: {
1760 if (ST->hasVInstructions())
1761 return getRISCVInstructionCost(RISCV::VID_V, LT.second,
CostKind) +
1763 getRISCVInstructionCost(RISCV::VADD_VX, LT.second,
CostKind);
1764 return 1 + (LT.first - 1);
1766 case Intrinsic::vector_splice_left:
1767 case Intrinsic::vector_splice_right: {
1772 if (ST->hasVInstructions() && LT.second.isVector()) {
1774 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1779 case Intrinsic::experimental_cttz_elts: {
1780 if (!ST->hasVInstructions())
1787 if (LT.second.getVectorElementType() != MVT::i1)
1788 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second,
CostKind);
1790 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind);
1802 return LT.first *
Cost;
1804 case Intrinsic::experimental_vp_splice: {
1812 case Intrinsic::vp_merge: {
1820 case Intrinsic::fptoui_sat:
1821 case Intrinsic::fptosi_sat: {
1823 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1828 if (!SrcTy->isVectorTy())
1831 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1848 case Intrinsic::experimental_vector_extract_last_active: {
1870 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1871 TLI->getVectorIdxTy(
getDataLayout()), MaskTy->getElementCount(),
1872 true, &VScaleRange);
1873 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1881 if (StepLT.first > 1)
1885 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1887 Cost += MaskLT.first *
1888 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second,
CostKind);
1890 Cost += StepLT.first *
1891 getRISCVInstructionCost(Opcodes, StepLT.second,
CostKind);
1895 Cost += ValLT.first *
1896 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1902 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1904 LT.second.isVector()) {
1905 MVT EltTy = LT.second.getVectorElementType();
1907 ICA.
getID(), EltTy))
1908 return LT.first * Entry->Cost;
1921 if (ST->hasVInstructions() && PtrTy->
isVectorTy())
1939 if (ST->hasStdExtP() &&
1947 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1948 Dst->getScalarSizeInBits() > ST->getELen())
1951 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1966 if (Src->getScalarSizeInBits() == 1) {
1971 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second,
CostKind) +
1972 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1978 if (Dst->getScalarSizeInBits() == 1) {
1984 return SrcLT.first *
1985 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
1997 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
1998 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2000 SrcLT.second.getSizeInBits()) ||
2002 DstLT.second.getSizeInBits()) ||
2003 SrcLT.first > 1 || DstLT.first > 1)
2007 assert((SrcLT.first == 1) && (DstLT.first == 1) &&
"Illegal type");
2009 int PowDiff = (int)
Log2_32(DstLT.second.getScalarSizeInBits()) -
2010 (int)
Log2_32(SrcLT.second.getScalarSizeInBits());
2014 if ((PowDiff < 1) || (PowDiff > 3))
2016 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2017 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2020 return getRISCVInstructionCost(
Op, DstLT.second,
CostKind);
2026 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2027 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2031 : RISCV::VFNCVT_F_F_W;
2033 for (; SrcEltSize != DstEltSize;) {
2037 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2039 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2047 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2049 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2051 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2052 unsigned SrcEltSize = Src->getScalarSizeInBits();
2053 unsigned DstEltSize = Dst->getScalarSizeInBits();
2055 if ((SrcEltSize == 16) &&
2056 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2062 std::pair<InstructionCost, MVT> VecF32LT =
2065 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2070 if (DstEltSize == SrcEltSize)
2071 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2072 else if (DstEltSize > SrcEltSize)
2073 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2078 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2079 Cost += getRISCVInstructionCost(FNCVT, VecVT,
CostKind);
2080 if ((SrcEltSize / 2) > DstEltSize) {
2091 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2092 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2093 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2094 unsigned SrcEltSize = Src->getScalarSizeInBits();
2095 unsigned DstEltSize = Dst->getScalarSizeInBits();
2098 if ((DstEltSize == 16) &&
2099 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2105 std::pair<InstructionCost, MVT> VecF32LT =
2108 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2113 if (DstEltSize == SrcEltSize)
2114 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2115 else if (DstEltSize > SrcEltSize) {
2116 if ((DstEltSize / 2) > SrcEltSize) {
2120 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2123 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2125 Cost += getRISCVInstructionCost(FNCVT, DstLT.second,
CostKind);
2132unsigned RISCVTTIImpl::getEstimatedVLFor(
VectorType *Ty)
const {
2134 const unsigned EltSize =
DL.getTypeSizeInBits(Ty->getElementType());
2135 const unsigned MinSize =
DL.getTypeSizeInBits(Ty).getKnownMinValue();
2150 if (Ty->getScalarSizeInBits() > ST->getELen())
2154 if (Ty->getElementType()->isIntegerTy(1)) {
2158 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2164 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2168 case Intrinsic::maximum:
2170 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2172 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2187 case Intrinsic::minimum:
2189 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2191 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2197 const unsigned EltTyBits =
DL.getTypeSizeInBits(DstTy);
2206 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2215 case Intrinsic::smax:
2216 SplitOp = RISCV::VMAX_VV;
2217 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2219 case Intrinsic::smin:
2220 SplitOp = RISCV::VMIN_VV;
2221 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2223 case Intrinsic::umax:
2224 SplitOp = RISCV::VMAXU_VV;
2225 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2227 case Intrinsic::umin:
2228 SplitOp = RISCV::VMINU_VV;
2229 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2231 case Intrinsic::maxnum:
2232 SplitOp = RISCV::VFMAX_VV;
2233 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2235 case Intrinsic::minnum:
2236 SplitOp = RISCV::VFMIN_VV;
2237 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2242 (LT.first > 1) ? (LT.first - 1) *
2243 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2245 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2250 std::optional<FastMathFlags> FMF,
2256 if (Ty->getScalarSizeInBits() > ST->getELen())
2259 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2267 Type *ElementTy = Ty->getElementType();
2272 if (LT.second == MVT::v1i1)
2273 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind) +
2291 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2292 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second,
CostKind) +
2293 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second,
CostKind) +
2294 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2303 return (LT.first - 1) *
2304 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind) +
2305 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) + 1;
2313 return (LT.first - 1) *
2314 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second,
CostKind) +
2315 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2328 SplitOp = RISCV::VADD_VV;
2329 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2332 SplitOp = RISCV::VOR_VV;
2333 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2336 SplitOp = RISCV::VXOR_VV;
2337 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2340 SplitOp = RISCV::VAND_VV;
2341 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2345 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2346 LT.second.getScalarType() == MVT::bf16)
2350 for (
unsigned i = 0; i < LT.first.getValue(); i++)
2353 return getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2355 SplitOp = RISCV::VFADD_VV;
2356 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2361 (LT.first > 1) ? (LT.first - 1) *
2362 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2364 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2368 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *ValTy,
2379 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2385 if (IsUnsigned && Opcode == Instruction::Add &&
2386 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2390 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind);
2397 return (LT.first - 1) +
2404 assert(OpInfo.isConstant() &&
"non constant operand?");
2411 if (OpInfo.isUniform())
2417 return getConstantPoolLoadCost(Ty,
CostKind);
2426 EVT VT = TLI->getValueType(
DL, Src,
true);
2428 if (VT == MVT::Other ||
2434 if (Opcode == Instruction::Store && OpInfo.isConstant())
2449 if (Src->
isVectorTy() && LT.second.isVector() &&
2451 LT.second.getSizeInBits()))
2461 if (ST->hasVInstructions() && LT.second.isVector() &&
2463 BaseCost *= TLI->getLMULCost(LT.second);
2464 return Cost + BaseCost;
2473 Op1Info, Op2Info,
I);
2477 Op1Info, Op2Info,
I);
2482 Op1Info, Op2Info,
I);
2484 auto GetConstantMatCost =
2486 if (OpInfo.isUniform())
2491 return getConstantPoolLoadCost(ValTy,
CostKind);
2496 ConstantMatCost += GetConstantMatCost(Op1Info);
2498 ConstantMatCost += GetConstantMatCost(Op2Info);
2501 if (Opcode == Instruction::Select && LT.second.isVector()) {
2502 if (CondTy->isVectorTy()) {
2507 return ConstantMatCost +
2509 getRISCVInstructionCost(
2510 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2514 return ConstantMatCost +
2515 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2525 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2526 return ConstantMatCost +
2528 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2530 LT.first * getRISCVInstructionCost(
2531 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2538 return ConstantMatCost +
2539 LT.first * getRISCVInstructionCost(
2540 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2544 if ((Opcode == Instruction::ICmp) && ValTy->
isVectorTy() &&
2548 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2553 if ((Opcode == Instruction::FCmp) && ValTy->
isVectorTy() &&
2558 return ConstantMatCost +
2559 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind);
2569 Op1Info, Op2Info,
I);
2578 return ConstantMatCost +
2579 LT.first * getRISCVInstructionCost(
2580 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2587 return ConstantMatCost +
2589 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2598 return ConstantMatCost +
2600 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second,
CostKind);
2613 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2614 U->getType()->isIntegerTy() &&
2615 !isa<ConstantData>(U->getOperand(1)) &&
2616 !isa<ConstantData>(U->getOperand(2));
2624 Op1Info, Op2Info,
I);
2631 return Opcode == Instruction::PHI ? 0 : 1;
2648 if (Opcode != Instruction::ExtractElement &&
2649 Opcode != Instruction::InsertElement)
2657 if (!LT.second.isVector()) {
2667 auto NumElems = FixedVecTy->getNumElements();
2673 return Opcode == Instruction::ExtractElement
2674 ? StoreCost * NumElems + LoadCost
2675 : (StoreCost + LoadCost) * NumElems + StoreCost;
2679 if (LT.second.isScalableVector() && !LT.first.isValid())
2687 if (Opcode == Instruction::ExtractElement) {
2693 return ExtendCost + ExtractCost;
2703 return ExtendCost + InsertCost + TruncCost;
2710 if (LT.second.isFloatingPoint())
2711 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2715 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2717 getRISCVInstructionCost(MoveOpc, LT.second,
CostKind);
2719 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2724 if (LT.second.isFixedLengthVector()) {
2725 unsigned Width = LT.second.getVectorNumElements();
2726 Index = Index % Width;
2731 if (
auto VLEN = ST->getRealVLen()) {
2732 unsigned EltSize = LT.second.getScalarSizeInBits();
2733 unsigned M1Max = *VLEN / EltSize;
2734 Index = Index % M1Max;
2740 else if (Opcode == Instruction::InsertElement)
2748 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2749 LT.second.isScalableVector()))) {
2751 Align VecAlign =
DL.getPrefTypeAlign(Val);
2752 Align SclAlign =
DL.getPrefTypeAlign(ScalarType);
2757 if (Opcode == Instruction::ExtractElement)
2793 Opcode == Instruction::InsertElement
2794 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2795 RISCV::VSLIDE1DOWN_VX,
2796 RISCV::VSLIDEUP_VX},
2798 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2799 RISCV::VSRL_VX, RISCV::VMV_X_S},
2802 return BaseCost + SlideCost;
2808 unsigned Index)
const {
2817 assert(Index < EC.getKnownMinValue() &&
"Unexpected reverse index");
2819 EC.getKnownMinValue() - 1 - Index,
nullptr,
2828std::optional<InstructionCost>
2834 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2836 if (Opcode == Instruction::UDiv)
2843 return std::nullopt;
2865 if (std::optional<InstructionCost> CombinedCost =
2867 Op2Info, Args, CxtI))
2868 return *CombinedCost;
2872 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2875 if (!LT.second.isVector()) {
2885 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2886 if (
const auto *Entry =
CostTableLookup(DivTbl, ISDOpcode, LT.second))
2887 return Entry->Cost * LT.first;
2896 if ((LT.second.getVectorElementType() == MVT::f16 ||
2897 LT.second.getVectorElementType() == MVT::bf16) &&
2898 TLI->getOperationAction(ISDOpcode, LT.second) ==
2900 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2904 CastCost += LT.first * Args.size() *
2912 LT.second = PromotedVT;
2915 auto getConstantMatCost =
2925 return getConstantPoolLoadCost(Ty,
CostKind);
2931 ConstantMatCost += getConstantMatCost(0, Op1Info);
2933 ConstantMatCost += getConstantMatCost(1, Op2Info);
2936 switch (ISDOpcode) {
2939 Op = RISCV::VADD_VV;
2944 Op = RISCV::VSLL_VV;
2949 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2954 Op = RISCV::VMUL_VV;
2958 Op = RISCV::VDIV_VV;
2962 Op = RISCV::VREM_VV;
2966 Op = RISCV::VFADD_VV;
2969 Op = RISCV::VFMUL_VV;
2972 Op = RISCV::VFDIV_VV;
2975 Op = RISCV::VFSGNJN_VV;
2980 return CastCost + ConstantMatCost +
2989 if (Ty->isFPOrFPVectorTy())
2991 return CastCost + ConstantMatCost + LT.first *
InstrCost;
3014 if (Info.isSameBase() && V !=
Base) {
3015 if (
GEP->hasAllConstantIndices())
3021 unsigned Stride =
DL.getTypeStoreSize(AccessTy);
3022 if (Info.isUnitStride() &&
3028 GEP->getType()->getPointerAddressSpace()))
3031 {TTI::OK_AnyValue, TTI::OP_None},
3032 {TTI::OK_AnyValue, TTI::OP_None}, {});
3049 if (ST->enableDefaultUnroll())
3059 if (L->getHeader()->getParent()->hasOptSize())
3063 L->getExitingBlocks(ExitingBlocks);
3065 <<
"Blocks: " << L->getNumBlocks() <<
"\n"
3066 <<
"Exit blocks: " << ExitingBlocks.
size() <<
"\n");
3070 if (ExitingBlocks.
size() > 2)
3075 if (L->getNumBlocks() > 4)
3083 for (
auto *BB : L->getBlocks()) {
3084 for (
auto &
I : *BB) {
3088 if (IsVectorized && (
I.getType()->isVectorTy() ||
3090 return V->getType()->isVectorTy();
3129 bool HasMask =
false;
3132 bool IsWrite) -> int64_t {
3133 if (
auto *TarExtTy =
3135 return TarExtTy->getIntParameter(0);
3141 case Intrinsic::riscv_vle_mask:
3142 case Intrinsic::riscv_vse_mask:
3143 case Intrinsic::riscv_vlseg2_mask:
3144 case Intrinsic::riscv_vlseg3_mask:
3145 case Intrinsic::riscv_vlseg4_mask:
3146 case Intrinsic::riscv_vlseg5_mask:
3147 case Intrinsic::riscv_vlseg6_mask:
3148 case Intrinsic::riscv_vlseg7_mask:
3149 case Intrinsic::riscv_vlseg8_mask:
3150 case Intrinsic::riscv_vsseg2_mask:
3151 case Intrinsic::riscv_vsseg3_mask:
3152 case Intrinsic::riscv_vsseg4_mask:
3153 case Intrinsic::riscv_vsseg5_mask:
3154 case Intrinsic::riscv_vsseg6_mask:
3155 case Intrinsic::riscv_vsseg7_mask:
3156 case Intrinsic::riscv_vsseg8_mask:
3159 case Intrinsic::riscv_vle:
3160 case Intrinsic::riscv_vse:
3161 case Intrinsic::riscv_vlseg2:
3162 case Intrinsic::riscv_vlseg3:
3163 case Intrinsic::riscv_vlseg4:
3164 case Intrinsic::riscv_vlseg5:
3165 case Intrinsic::riscv_vlseg6:
3166 case Intrinsic::riscv_vlseg7:
3167 case Intrinsic::riscv_vlseg8:
3168 case Intrinsic::riscv_vsseg2:
3169 case Intrinsic::riscv_vsseg3:
3170 case Intrinsic::riscv_vsseg4:
3171 case Intrinsic::riscv_vsseg5:
3172 case Intrinsic::riscv_vsseg6:
3173 case Intrinsic::riscv_vsseg7:
3174 case Intrinsic::riscv_vsseg8: {
3191 Ty = TarExtTy->getTypeParameter(0U);
3196 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3197 unsigned VLIndex = RVVIInfo->VLOperand;
3198 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3206 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3209 unsigned ElemSize = Ty->getScalarSizeInBits();
3213 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3214 Alignment, Mask, EVL);
3217 case Intrinsic::riscv_vlse_mask:
3218 case Intrinsic::riscv_vsse_mask:
3219 case Intrinsic::riscv_vlsseg2_mask:
3220 case Intrinsic::riscv_vlsseg3_mask:
3221 case Intrinsic::riscv_vlsseg4_mask:
3222 case Intrinsic::riscv_vlsseg5_mask:
3223 case Intrinsic::riscv_vlsseg6_mask:
3224 case Intrinsic::riscv_vlsseg7_mask:
3225 case Intrinsic::riscv_vlsseg8_mask:
3226 case Intrinsic::riscv_vssseg2_mask:
3227 case Intrinsic::riscv_vssseg3_mask:
3228 case Intrinsic::riscv_vssseg4_mask:
3229 case Intrinsic::riscv_vssseg5_mask:
3230 case Intrinsic::riscv_vssseg6_mask:
3231 case Intrinsic::riscv_vssseg7_mask:
3232 case Intrinsic::riscv_vssseg8_mask:
3235 case Intrinsic::riscv_vlse:
3236 case Intrinsic::riscv_vsse:
3237 case Intrinsic::riscv_vlsseg2:
3238 case Intrinsic::riscv_vlsseg3:
3239 case Intrinsic::riscv_vlsseg4:
3240 case Intrinsic::riscv_vlsseg5:
3241 case Intrinsic::riscv_vlsseg6:
3242 case Intrinsic::riscv_vlsseg7:
3243 case Intrinsic::riscv_vlsseg8:
3244 case Intrinsic::riscv_vssseg2:
3245 case Intrinsic::riscv_vssseg3:
3246 case Intrinsic::riscv_vssseg4:
3247 case Intrinsic::riscv_vssseg5:
3248 case Intrinsic::riscv_vssseg6:
3249 case Intrinsic::riscv_vssseg7:
3250 case Intrinsic::riscv_vssseg8: {
3267 Ty = TarExtTy->getTypeParameter(0U);
3272 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3273 unsigned VLIndex = RVVIInfo->VLOperand;
3274 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3283 unsigned PointerAlign = Alignment.valueOrOne().value();
3286 Alignment =
Align(1);
3293 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3296 unsigned ElemSize = Ty->getScalarSizeInBits();
3300 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3301 Alignment, Mask, EVL, Stride);
3304 case Intrinsic::riscv_vloxei_mask:
3305 case Intrinsic::riscv_vluxei_mask:
3306 case Intrinsic::riscv_vsoxei_mask:
3307 case Intrinsic::riscv_vsuxei_mask:
3308 case Intrinsic::riscv_vloxseg2_mask:
3309 case Intrinsic::riscv_vloxseg3_mask:
3310 case Intrinsic::riscv_vloxseg4_mask:
3311 case Intrinsic::riscv_vloxseg5_mask:
3312 case Intrinsic::riscv_vloxseg6_mask:
3313 case Intrinsic::riscv_vloxseg7_mask:
3314 case Intrinsic::riscv_vloxseg8_mask:
3315 case Intrinsic::riscv_vluxseg2_mask:
3316 case Intrinsic::riscv_vluxseg3_mask:
3317 case Intrinsic::riscv_vluxseg4_mask:
3318 case Intrinsic::riscv_vluxseg5_mask:
3319 case Intrinsic::riscv_vluxseg6_mask:
3320 case Intrinsic::riscv_vluxseg7_mask:
3321 case Intrinsic::riscv_vluxseg8_mask:
3322 case Intrinsic::riscv_vsoxseg2_mask:
3323 case Intrinsic::riscv_vsoxseg3_mask:
3324 case Intrinsic::riscv_vsoxseg4_mask:
3325 case Intrinsic::riscv_vsoxseg5_mask:
3326 case Intrinsic::riscv_vsoxseg6_mask:
3327 case Intrinsic::riscv_vsoxseg7_mask:
3328 case Intrinsic::riscv_vsoxseg8_mask:
3329 case Intrinsic::riscv_vsuxseg2_mask:
3330 case Intrinsic::riscv_vsuxseg3_mask:
3331 case Intrinsic::riscv_vsuxseg4_mask:
3332 case Intrinsic::riscv_vsuxseg5_mask:
3333 case Intrinsic::riscv_vsuxseg6_mask:
3334 case Intrinsic::riscv_vsuxseg7_mask:
3335 case Intrinsic::riscv_vsuxseg8_mask:
3338 case Intrinsic::riscv_vloxei:
3339 case Intrinsic::riscv_vluxei:
3340 case Intrinsic::riscv_vsoxei:
3341 case Intrinsic::riscv_vsuxei:
3342 case Intrinsic::riscv_vloxseg2:
3343 case Intrinsic::riscv_vloxseg3:
3344 case Intrinsic::riscv_vloxseg4:
3345 case Intrinsic::riscv_vloxseg5:
3346 case Intrinsic::riscv_vloxseg6:
3347 case Intrinsic::riscv_vloxseg7:
3348 case Intrinsic::riscv_vloxseg8:
3349 case Intrinsic::riscv_vluxseg2:
3350 case Intrinsic::riscv_vluxseg3:
3351 case Intrinsic::riscv_vluxseg4:
3352 case Intrinsic::riscv_vluxseg5:
3353 case Intrinsic::riscv_vluxseg6:
3354 case Intrinsic::riscv_vluxseg7:
3355 case Intrinsic::riscv_vluxseg8:
3356 case Intrinsic::riscv_vsoxseg2:
3357 case Intrinsic::riscv_vsoxseg3:
3358 case Intrinsic::riscv_vsoxseg4:
3359 case Intrinsic::riscv_vsoxseg5:
3360 case Intrinsic::riscv_vsoxseg6:
3361 case Intrinsic::riscv_vsoxseg7:
3362 case Intrinsic::riscv_vsoxseg8:
3363 case Intrinsic::riscv_vsuxseg2:
3364 case Intrinsic::riscv_vsuxseg3:
3365 case Intrinsic::riscv_vsuxseg4:
3366 case Intrinsic::riscv_vsuxseg5:
3367 case Intrinsic::riscv_vsuxseg6:
3368 case Intrinsic::riscv_vsuxseg7:
3369 case Intrinsic::riscv_vsuxseg8: {
3386 Ty = TarExtTy->getTypeParameter(0U);
3391 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3392 unsigned VLIndex = RVVIInfo->VLOperand;
3393 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3406 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3409 unsigned ElemSize = Ty->getScalarSizeInBits();
3414 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3415 Align(1), Mask, EVL,
3424 if (Ty->isVectorTy()) {
3427 if ((EltTy->
isHalfTy() && !ST->hasVInstructionsF16()) ||
3433 if (
Size.isScalable() && ST->hasVInstructions())
3436 if (ST->useRVVForFixedLengthVectors())
3456 return std::max<unsigned>(1U, RegWidth.
getFixedValue() / ElemWidth);
3464 return ST->enableUnalignedVectorMem();
3470 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3492 Align Alignment)
const {
3494 if (!VTy || VTy->isScalableTy())
3502 if (VTy->getElementType()->isIntegerTy(8))
3503 if (VTy->getElementCount().getFixedValue() > 256)
3504 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3505 ST->getMaxLMULForFixedLengthVectors();
3510 Align Alignment)
const {
3512 if (!VTy || VTy->isScalableTy())
3523 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3526 return TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, ElementTy));
3535 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
3536 bool Considerable =
false;
3537 AllowPromotionWithoutCommonHeader =
false;
3540 Type *ConsideredSExtType =
3542 if (
I.getType() != ConsideredSExtType)
3546 for (
const User *U :
I.users()) {
3548 Considerable =
true;
3552 if (GEPInst->getNumOperands() > 2) {
3553 AllowPromotionWithoutCommonHeader =
true;
3558 return Considerable;
3563 case Instruction::Add:
3564 case Instruction::Sub:
3565 case Instruction::Mul:
3566 case Instruction::And:
3567 case Instruction::Or:
3568 case Instruction::Xor:
3569 case Instruction::FAdd:
3570 case Instruction::FSub:
3571 case Instruction::FMul:
3572 case Instruction::FDiv:
3573 case Instruction::ICmp:
3574 case Instruction::FCmp:
3576 case Instruction::Shl:
3577 case Instruction::LShr:
3578 case Instruction::AShr:
3579 case Instruction::UDiv:
3580 case Instruction::SDiv:
3581 case Instruction::URem:
3582 case Instruction::SRem:
3583 case Instruction::Select:
3584 return Operand == 1;
3591 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3601 switch (
II->getIntrinsicID()) {
3602 case Intrinsic::fma:
3603 case Intrinsic::fmuladd:
3604 return Operand == 0 || Operand == 1;
3605 case Intrinsic::vp_udiv:
3606 case Intrinsic::vp_sdiv:
3607 case Intrinsic::vp_urem:
3608 case Intrinsic::vp_srem:
3609 case Intrinsic::ssub_sat:
3610 case Intrinsic::usub_sat:
3611 return Operand == 1;
3613 case Intrinsic::smin:
3614 case Intrinsic::umin:
3615 case Intrinsic::smax:
3616 case Intrinsic::umax:
3617 case Intrinsic::sadd_sat:
3618 case Intrinsic::uadd_sat:
3619 return Operand == 0 || Operand == 1;
3632 if (
I->isBitwiseLogicOp()) {
3633 if (!
I->getType()->isVectorTy()) {
3634 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3635 for (
auto &
Op :
I->operands()) {
3643 }
else if (
I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3644 for (
auto &
Op :
I->operands()) {
3656 Ops.push_back(&Not);
3657 Ops.push_back(&InsertElt);
3665 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3673 if (!ST->sinkSplatOperands())
3676 for (
auto OpIdx :
enumerate(
I->operands())) {
3696 for (
Use &U :
Op->uses()) {
3703 Use *InsertEltUse = &
Op->getOperandUse(0);
3706 Ops.push_back(&InsertElt->getOperandUse(1));
3707 Ops.push_back(InsertEltUse);
3708 Ops.push_back(&OpIdx.value());
3717 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3720 Options.AllowOverlappingLoads =
true;
3721 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3723 if (ST->is64Bit()) {
3724 Options.LoadSizes = {8, 4, 2, 1};
3725 Options.AllowedTailExpansions = {3, 5, 6};
3727 Options.LoadSizes = {4, 2, 1};
3728 Options.AllowedTailExpansions = {3};
3731 if (IsZeroCmp && ST->hasVInstructions()) {
3732 unsigned VLenB = ST->getRealMinVLen() / 8;
3735 unsigned MinSize = ST->getXLen() / 8 + 1;
3736 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3750 if (
I->getOpcode() == Instruction::Or &&
3754 if (
I->getOpcode() == Instruction::Add ||
3755 I->getOpcode() == Instruction::Sub)
3773std::optional<Instruction *>
3779 if (
is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3780 II.getIntrinsicID())) {
3783 if (!ST->hasVInstructions())
3786 bool HasAVL =
II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3787 unsigned Offset = HasAVL ? 1 : 0;
3788 unsigned BitWidth =
II.getType()->getIntegerBitWidth();
3813 Value *AVL =
II.getArgOperand(0);
3842 II.getRange().value_or(ConstantRange::getFull(
BitWidth));
3844 if (NewRange != OldRange) {
3845 II.addRangeRetAttr(NewRange);
3855 if (
II.user_empty())
3860 const APInt *Scalar;
3865 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3869 unsigned TargetEltBW =
DL.getTypeSizeInBits(TargetVecTy->getElementType());
3870 unsigned SourceEltBW =
DL.getTypeSizeInBits(SourceVecTy->getElementType());
3871 if (TargetEltBW % SourceEltBW)
3873 unsigned TargetScale = TargetEltBW / SourceEltBW;
3874 if (VL % TargetScale || TargetScale == 1)
3876 Type *VLTy =
II.getOperand(2)->getType();
3877 ElementCount SourceEC = SourceVecTy->getElementCount();
3878 unsigned NewEltBW = SourceEltBW * TargetScale;
3880 !
DL.fitsInLegalInteger(NewEltBW))
3883 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, NewEltTy)))
3887 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3888 "Lossless bitcast between types expected");
3894 RetTy, Intrinsic::riscv_vmv_v_x,
3895 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3896 ConstantInt::get(VLTy, VL / TargetScale)}),
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
uint64_t IntrinsicInst * II
This file describes how to lower LLVM code to machine code.
Class for arbitrary precision integers.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
size_t size() const
Get the array size.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
@ ICMP_SLT
signed less than
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Convenience struct for specifying and reasoning about fast-math flags.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
Intrinsic::ID getID() const
bool isTypeBasedOnly() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Represents a single loop in the control flow graph.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
Align getAlignment() const
unsigned getAddressSpace() const
Type * getDataType() const
bool getVariableMask() const
Intrinsic::ID getID() const
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ BinaryOp
One of the operands is a binary op.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const