19#include "llvm/IR/IntrinsicsRISCV.h"
28#define DEBUG_TYPE "riscvtti"
31 "riscv-v-register-bit-width-lmul",
33 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
34 "by autovectorized code. Fractional LMULs are not supported."),
40 "Overrides result used for getMaximumVF query which is used "
41 "exclusively by SLP vectorizer."),
46 cl::desc(
"Set the lower bound of a trip count to decide on "
47 "vectorization while tail-folding."),
59 size_t NumInstr = OpCodes.size();
64 return LMULCost * NumInstr;
66 for (
auto Op : OpCodes) {
68 case RISCV::VRGATHER_VI:
71 case RISCV::VRGATHER_VV:
74 case RISCV::VSLIDEUP_VI:
75 case RISCV::VSLIDEDOWN_VI:
78 case RISCV::VSLIDEUP_VX:
79 case RISCV::VSLIDEDOWN_VX:
82 case RISCV::VREDMAX_VS:
83 case RISCV::VREDMIN_VS:
84 case RISCV::VREDMAXU_VS:
85 case RISCV::VREDMINU_VS:
86 case RISCV::VREDSUM_VS:
87 case RISCV::VREDAND_VS:
88 case RISCV::VREDOR_VS:
89 case RISCV::VREDXOR_VS:
90 case RISCV::VFREDMAX_VS:
91 case RISCV::VFREDMIN_VS:
92 case RISCV::VFREDUSUM_VS: {
99 case RISCV::VFREDOSUM_VS: {
107 case RISCV::VFMV_F_S:
112 case RISCV::VFMV_S_F:
114 case RISCV::VMXOR_MM:
115 case RISCV::VMAND_MM:
116 case RISCV::VMANDN_MM:
117 case RISCV::VMNAND_MM:
119 case RISCV::VFIRST_M:
138 assert(Ty->isIntegerTy() &&
139 "getIntImmCost can only estimate cost of materialising integers");
162 if (!BO || !BO->hasOneUse())
165 if (BO->getOpcode() != Instruction::Shl)
176 if (ShAmt == Trailing)
193 if (!Cmp || !Cmp->isEquality())
209 if ((CmpC & Mask) != CmpC)
216 return NewCmpC >= -2048 && NewCmpC <= 2048;
223 assert(Ty->isIntegerTy() &&
224 "getIntImmCost can only estimate cost of materialising integers");
232 bool Takes12BitImm =
false;
233 unsigned ImmArgIdx = ~0U;
236 case Instruction::GetElementPtr:
241 case Instruction::Store: {
246 if (Idx == 1 || !Inst)
251 if (!getTLI()->allowsMemoryAccessForAlignment(
252 Ty->getContext(),
DL, getTLI()->getValueType(
DL, Ty),
259 case Instruction::Load:
262 case Instruction::And:
264 if (
Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
267 if (
Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
270 if (ST->hasStdExtZbs() && (~
Imm).isPowerOf2())
272 if (Inst && Idx == 1 &&
Imm.getBitWidth() <= ST->getXLen() &&
275 if (Inst && Idx == 1 &&
Imm.getBitWidth() == 64 &&
278 Takes12BitImm =
true;
280 case Instruction::Add:
281 Takes12BitImm =
true;
283 case Instruction::Or:
284 case Instruction::Xor:
286 if (ST->hasStdExtZbs() &&
Imm.isPowerOf2())
288 Takes12BitImm =
true;
290 case Instruction::Mul:
292 if (
Imm.isPowerOf2() ||
Imm.isNegatedPowerOf2())
295 if ((
Imm + 1).isPowerOf2() || (
Imm - 1).isPowerOf2())
298 Takes12BitImm =
true;
300 case Instruction::Sub:
301 case Instruction::Shl:
302 case Instruction::LShr:
303 case Instruction::AShr:
304 Takes12BitImm =
true;
315 if (
Imm.getSignificantBits() <= 64 &&
338 return ST->hasVInstructions();
348 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
352 if (Opcode == Instruction::FAdd)
361 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
362 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
363 InputTypeA != InputTypeB || !InputTypeA->
isIntegerTy(8) ||
379 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second,
CostKind);
388 std::pair<InstructionCost, MVT> AccLT =
396 bool WidenFirst =
false;
397 if (VF.
isScalable() && AccLT.second.isScalableVector()) {
398 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
411 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
414 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second,
CostKind);
418 std::pair<InstructionCost, MVT> RedLT =
420 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
422 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
426 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
438 switch (
II->getIntrinsicID()) {
442 case Intrinsic::vector_reduce_mul:
443 case Intrinsic::vector_reduce_fmul:
449 if (ST->hasVInstructions())
450 if (
unsigned MinVLen = ST->getRealMinVLen();
465 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
468 (ST->hasVInstructions() &&
491 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
497RISCVTTIImpl::getConstantPoolLoadCost(
Type *Ty,
514 unsigned Size = Mask.size();
517 for (
unsigned I = 0;
I !=
Size; ++
I) {
518 if (
static_cast<unsigned>(Mask[
I]) ==
I)
524 for (
unsigned J =
I + 1; J !=
Size; ++J)
526 if (
static_cast<unsigned>(Mask[J]) != J %
I)
554 "Expected fixed vector type and non-empty mask");
557 unsigned NumOfDests =
divideCeil(Mask.size(), LegalNumElts);
561 if (NumOfDests <= 1 ||
563 Tp->getElementType()->getPrimitiveSizeInBits() ||
564 LegalNumElts >= Tp->getElementCount().getFixedValue())
567 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
570 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
574 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
575 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
576 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
578 assert(NormalizedVF >= Mask.size() &&
579 "Normalized mask expected to be not shorter than original mask.");
584 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
585 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
588 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
591 Cost +=
TTI.getShuffleCost(
594 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
596 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
597 Cost +=
TTI.getShuffleCost(
600 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
623 if (!VLen || Mask.empty())
627 LegalVT =
TTI.getTypeLegalizationCost(
633 if (NumOfDests <= 1 ||
635 Tp->getElementType()->getPrimitiveSizeInBits() ||
639 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
642 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
648 unsigned NormalizedVF =
653 assert(NormalizedVF >= Mask.size() &&
654 "Normalized mask expected to be not shorter than original mask.");
660 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
661 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
664 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
669 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
671 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
673 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
680 if ((NumOfDestRegs > 2 && NumShuffles <=
static_cast<int>(NumOfDestRegs)) ||
681 (NumOfDestRegs <= 2 && NumShuffles < 4))
696 if (!
LT.second.isFixedLengthVector())
704 auto GetSlideOpcode = [&](
int SlideAmt) {
706 bool IsVI =
isUInt<5>(std::abs(SlideAmt));
708 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
709 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
712 std::array<std::pair<int, int>, 2> SrcInfo;
716 if (SrcInfo[1].second == 0)
719 if (ST->hasStdExtZvzip() &&
LT.second.getScalarSizeInBits() != 1) {
721 if (
isPairEven(SrcInfo, Mask, Factor) && Factor == 1)
722 return getRISCVInstructionCost(RISCV::VPAIRE_VV,
LT.second,
CostKind);
723 if (
isPairOdd(SrcInfo, Mask, Factor) && Factor == 1)
724 return getRISCVInstructionCost(RISCV::VPAIRO_VV,
LT.second,
CostKind);
728 if (SrcInfo[0].second != 0) {
729 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
730 FirstSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
733 if (SrcInfo[1].first == -1)
734 return FirstSlideCost;
737 if (SrcInfo[1].second != 0) {
738 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
739 SecondSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
742 getRISCVInstructionCost(RISCV::VMERGE_VVM,
LT.second,
CostKind);
749 return FirstSlideCost + SecondSlideCost + MaskCost;
752std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(
MVT InterleavedVT)
const {
762 LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
764 return InterleavedVT;
767std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(
MVT InterleavedVT)
const {
775 return InterleavedVT;
785 "Expected the Mask to match the return size if given");
787 "Expected the same scalar types");
790 if (VIC == TTI::VectorInstrContext::SplatOpFolded &&
806 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
808 *
this, LT.second, ST->getRealVLen(),
810 if (VRegSplittingCost.
isValid())
811 return VRegSplittingCost;
816 if (Mask.size() >= 2) {
817 MVT EltTp = LT.second.getVectorElementType();
828 return 2 * LT.first * TLI->getLMULCost(LT.second);
830 if (Mask[0] == 0 || Mask[0] == 1) {
834 if (
equal(DeinterleaveMask, Mask))
835 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
840 if (LT.second.getScalarSizeInBits() != 1 &&
843 unsigned NumSlides =
Log2_32(Mask.size() / SubVectorSize);
845 for (
unsigned I = 0;
I != NumSlides; ++
I) {
846 unsigned InsertIndex = SubVectorSize * (1 <<
I);
851 std::pair<InstructionCost, MVT> DestLT =
856 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
870 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
871 LT.second.getVectorNumElements() <= 256)) {
876 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second,
CostKind);
890 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
891 LT.second.getVectorNumElements() <= 256)) {
892 auto &
C = SrcTy->getContext();
893 auto EC = SrcTy->getElementCount();
898 return 2 * IndexCost +
899 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
918 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
946 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
947 if (std::optional<unsigned> VLen = ST->getRealVLen();
948 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
949 SubLT.second.getSizeInBits() <= *VLen)
957 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second,
CostKind);
964 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second,
CostKind);
976 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
983 if (IsLoad && LT.second.isVector() &&
985 LT.second.getVectorElementCount()))
989 Instruction::InsertElement);
990 if (LT.second.getScalarSizeInBits() == 1) {
998 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
1011 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
1012 RISCV::VMV_X_S, RISCV::VMV_V_X,
1021 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second,
CostKind);
1027 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second,
CostKind);
1033 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
1034 if (Index >= 0 && Index < 32)
1035 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
1036 else if (Index < 0 && Index > -32)
1037 Opcodes[1] = RISCV::VSLIDEUP_VI;
1038 return LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1042 if (!LT.second.isVector())
1048 if (SrcTy->getElementType()->isIntegerTy(1)) {
1060 MVT ContainerVT = LT.second;
1061 if (LT.second.isFixedLengthVector())
1062 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1064 if (ContainerVT.
bitsLE(M1VT)) {
1074 if (LT.second.isFixedLengthVector())
1076 LenCost =
isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1077 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1078 if (LT.second.isFixedLengthVector() &&
1079 isInt<5>(LT.second.getVectorNumElements() - 1))
1080 Opcodes[1] = RISCV::VRSUB_VI;
1082 getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1083 return LT.first * (LenCost + GatherCost);
1090 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1092 getRISCVInstructionCost(M1Opcodes, M1VT,
CostKind) + 3;
1096 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT,
CostKind) * Ratio;
1098 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second,
CostKind);
1099 return FixedCost + LT.first * (GatherCost + SlideCost);
1133 Ty, DemandedElts, Insert, Extract,
CostKind);
1135 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1136 if (Ty->getScalarSizeInBits() == 1) {
1146 assert(LT.second.isFixedLengthVector());
1147 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1151 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second,
CostKind);
1164 switch (MICA.
getID()) {
1165 case Intrinsic::vp_load_ff: {
1166 EVT DataTypeVT = TLI->getValueType(
DL, DataTy);
1167 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1174 case Intrinsic::experimental_vp_strided_load:
1175 case Intrinsic::experimental_vp_strided_store:
1177 case Intrinsic::masked_compressstore:
1178 case Intrinsic::masked_expandload:
1180 case Intrinsic::vp_scatter:
1181 case Intrinsic::vp_gather:
1182 case Intrinsic::masked_scatter:
1183 case Intrinsic::masked_gather:
1185 case Intrinsic::vp_load:
1186 case Intrinsic::vp_store:
1187 case Intrinsic::masked_load:
1188 case Intrinsic::masked_store:
1197 unsigned Opcode = MICA.
getID() == Intrinsic::masked_load ? Instruction::Load
1198 : Instruction::Store;
1209 if (MICA.
getID() == Intrinsic::vp_load ||
1210 MICA.
getID() == Intrinsic::vp_store) {
1222 bool UseMaskForCond,
bool UseMaskForGaps)
const {
1228 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1232 if (LT.second.isVector()) {
1238 VTy->getElementCount().divideCoefficientBy(Factor));
1239 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1240 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1245 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1246 unsigned VecSizeInBits =
1247 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1248 unsigned VLENForTuning =
1250 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1252 MVT SubVecVT = getTLI()->getValueType(
DL, SubVecTy).getSimpleVT();
1253 Cost += Factor * TLI->getLMULCost(SubVecVT);
1259 unsigned NumLoads = getEstimatedVLFor(VTy);
1275 if (UseMaskForGaps) {
1278 "Indices should not contain duplicate elements");
1279 unsigned NumOfFields = Indices.
size();
1280 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.
back() + 1);
1281 if (IsTailGapOnly &&
1282 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1284 if (LT.second.isVector() &&
1285 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1287 FVTy->getElementType(),
1288 FVTy->getElementCount().divideCoefficientBy(Factor));
1289 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1292 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1301 unsigned VF = FVTy->getNumElements() / Factor;
1308 if (Opcode == Instruction::Load) {
1310 for (
unsigned Index : Indices) {
1314 Mask.resize(VF * Factor, -1);
1318 Cost += ShuffleCost;
1336 UseMaskForCond, UseMaskForGaps);
1338 assert(Opcode == Instruction::Store &&
"Opcode must be a store");
1345 return MemCost + ShuffleCost;
1352 bool IsLoad = MICA.
getID() == Intrinsic::masked_gather ||
1353 MICA.
getID() == Intrinsic::vp_gather;
1354 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1362 if ((Opcode == Instruction::Load &&
1364 (Opcode == Instruction::Store &&
1370 if (MICA.
getID() == Intrinsic::vp_gather ||
1371 MICA.
getID() == Intrinsic::vp_scatter) {
1374 if (DataLT.first > 1)
1376 if (PtrLT.first > 1)
1384 unsigned NumLoads = getEstimatedVLFor(&VTy);
1391 unsigned Opcode = MICA.
getID() == Intrinsic::masked_expandload
1393 : Instruction::Store;
1397 bool IsLegal = (Opcode == Instruction::Store &&
1399 (Opcode == Instruction::Load &&
1423 if (Opcode == Instruction::Store)
1424 Opcodes.
append({RISCV::VCOMPRESS_VM});
1426 Opcodes.
append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1428 LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1453 unsigned NumLoads = getEstimatedVLFor(&VTy);
1456 uint64_t CacheLineBytes = ST->getCacheLineSize();
1457 if (!CacheLineBytes)
1458 CacheLineBytes = 64;
1461 int64_t Stride = StrideCI->getSExtValue();
1463 if (Stride != std::numeric_limits<int64_t>::min() && Stride != 0) {
1464 uint64_t AbsStride = (uint64_t)std::abs(Stride);
1465 if (AbsStride < CacheLineBytes) {
1466 uint64_t MaxCombines = ST->getMaxVectorCoalesceElts();
1467 if ((CacheLineBytes / AbsStride) >= MaxCombines)
1468 NumLoads =
divideCeil(NumLoads, MaxCombines);
1472 NumLoads =
divideCeil((NumLoads * AbsStride), CacheLineBytes);
1486 for (
auto *Ty : Tys) {
1487 if (!Ty->isVectorTy())
1501 {Intrinsic::floor, MVT::f32, 9},
1502 {Intrinsic::floor, MVT::f64, 9},
1503 {Intrinsic::ceil, MVT::f32, 9},
1504 {Intrinsic::ceil, MVT::f64, 9},
1505 {Intrinsic::trunc, MVT::f32, 7},
1506 {Intrinsic::trunc, MVT::f64, 7},
1507 {Intrinsic::round, MVT::f32, 9},
1508 {Intrinsic::round, MVT::f64, 9},
1509 {Intrinsic::roundeven, MVT::f32, 9},
1510 {Intrinsic::roundeven, MVT::f64, 9},
1511 {Intrinsic::rint, MVT::f32, 7},
1512 {Intrinsic::rint, MVT::f64, 7},
1513 {Intrinsic::nearbyint, MVT::f32, 9},
1514 {Intrinsic::nearbyint, MVT::f64, 9},
1515 {Intrinsic::bswap, MVT::i16, 3},
1516 {Intrinsic::bswap, MVT::i32, 12},
1517 {Intrinsic::bswap, MVT::i64, 31},
1518 {Intrinsic::bitreverse, MVT::i8, 17},
1519 {Intrinsic::bitreverse, MVT::i16, 24},
1520 {Intrinsic::bitreverse, MVT::i32, 33},
1521 {Intrinsic::bitreverse, MVT::i64, 52},
1522 {Intrinsic::ctpop, MVT::i8, 12},
1523 {Intrinsic::ctpop, MVT::i16, 19},
1524 {Intrinsic::ctpop, MVT::i32, 20},
1525 {Intrinsic::ctpop, MVT::i64, 21},
1526 {Intrinsic::ctlz, MVT::i8, 19},
1527 {Intrinsic::ctlz, MVT::i16, 28},
1528 {Intrinsic::ctlz, MVT::i32, 31},
1529 {Intrinsic::ctlz, MVT::i64, 35},
1530 {Intrinsic::cttz, MVT::i8, 16},
1531 {Intrinsic::cttz, MVT::i16, 23},
1532 {Intrinsic::cttz, MVT::i32, 24},
1533 {Intrinsic::cttz, MVT::i64, 25},
1540 switch (ICA.
getID()) {
1541 case Intrinsic::lrint:
1542 case Intrinsic::llrint:
1543 case Intrinsic::lround:
1544 case Intrinsic::llround: {
1548 if (ST->hasVInstructions() && LT.second.isVector()) {
1550 unsigned SrcEltSz =
DL.getTypeSizeInBits(SrcTy->getScalarType());
1551 unsigned DstEltSz =
DL.getTypeSizeInBits(RetTy->getScalarType());
1552 if (LT.second.getVectorElementType() == MVT::bf16) {
1553 if (!ST->hasVInstructionsBF16Minimal())
1556 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1558 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1559 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1560 !ST->hasVInstructionsF16()) {
1561 if (!ST->hasVInstructionsF16Minimal())
1564 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1566 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1568 }
else if (SrcEltSz > DstEltSz) {
1569 Ops = {RISCV::VFNCVT_X_F_W};
1570 }
else if (SrcEltSz < DstEltSz) {
1571 Ops = {RISCV::VFWCVT_X_F_V};
1573 Ops = {RISCV::VFCVT_X_F_V};
1578 if (SrcEltSz > DstEltSz)
1579 return SrcLT.first *
1580 getRISCVInstructionCost(
Ops, SrcLT.second,
CostKind);
1581 return LT.first * getRISCVInstructionCost(
Ops, LT.second,
CostKind);
1585 case Intrinsic::ceil:
1586 case Intrinsic::floor:
1587 case Intrinsic::trunc:
1588 case Intrinsic::rint:
1589 case Intrinsic::round:
1590 case Intrinsic::roundeven: {
1593 if (!LT.second.isVector() && TLI->isOperationCustom(
ISD::FCEIL, LT.second))
1594 return LT.first * 8;
1597 case Intrinsic::umin:
1598 case Intrinsic::umax:
1599 case Intrinsic::smin:
1600 case Intrinsic::smax: {
1602 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1605 if (ST->hasVInstructions() && LT.second.isVector()) {
1607 switch (ICA.
getID()) {
1608 case Intrinsic::umin:
1609 Op = RISCV::VMINU_VV;
1611 case Intrinsic::umax:
1612 Op = RISCV::VMAXU_VV;
1614 case Intrinsic::smin:
1615 Op = RISCV::VMIN_VV;
1617 case Intrinsic::smax:
1618 Op = RISCV::VMAX_VV;
1621 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1625 case Intrinsic::sadd_sat:
1626 case Intrinsic::ssub_sat:
1627 case Intrinsic::uadd_sat:
1628 case Intrinsic::usub_sat: {
1630 if (ST->hasVInstructions() && LT.second.isVector()) {
1632 switch (ICA.
getID()) {
1633 case Intrinsic::sadd_sat:
1634 Op = RISCV::VSADD_VV;
1636 case Intrinsic::ssub_sat:
1637 Op = RISCV::VSSUB_VV;
1639 case Intrinsic::uadd_sat:
1640 Op = RISCV::VSADDU_VV;
1642 case Intrinsic::usub_sat:
1643 Op = RISCV::VSSUBU_VV;
1646 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1650 case Intrinsic::fma:
1651 case Intrinsic::fmuladd: {
1654 if (ST->hasVInstructions() && LT.second.isVector())
1656 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second,
CostKind);
1659 case Intrinsic::fabs: {
1661 if (ST->hasVInstructions() && LT.second.isVector()) {
1667 if (LT.second.getVectorElementType() == MVT::bf16 ||
1668 (LT.second.getVectorElementType() == MVT::f16 &&
1669 !ST->hasVInstructionsF16()))
1670 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1675 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second,
CostKind);
1679 case Intrinsic::sqrt: {
1681 if (ST->hasVInstructions() && LT.second.isVector()) {
1684 MVT ConvType = LT.second;
1685 MVT FsqrtType = LT.second;
1688 if (LT.second.getVectorElementType() == MVT::bf16) {
1689 if (LT.second == MVT::nxv32bf16) {
1690 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1691 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1692 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1693 ConvType = MVT::nxv16f16;
1694 FsqrtType = MVT::nxv16f32;
1696 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1697 FsqrtOp = {RISCV::VFSQRT_V};
1698 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1700 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1701 !ST->hasVInstructionsF16()) {
1702 if (LT.second == MVT::nxv32f16) {
1703 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1704 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1705 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1706 ConvType = MVT::nxv16f16;
1707 FsqrtType = MVT::nxv16f32;
1709 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1710 FsqrtOp = {RISCV::VFSQRT_V};
1711 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1714 FsqrtOp = {RISCV::VFSQRT_V};
1717 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType,
CostKind) +
1718 getRISCVInstructionCost(ConvOp, ConvType,
CostKind));
1722 case Intrinsic::cttz:
1723 case Intrinsic::ctlz:
1724 case Intrinsic::ctpop: {
1726 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1728 switch (ICA.
getID()) {
1729 case Intrinsic::cttz:
1732 case Intrinsic::ctlz:
1735 case Intrinsic::ctpop:
1736 Op = RISCV::VCPOP_V;
1739 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1743 case Intrinsic::abs: {
1745 if (ST->hasVInstructions() && LT.second.isVector()) {
1747 if (ST->hasStdExtZvabd())
1749 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second,
CostKind);
1754 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1759 case Intrinsic::fshl:
1760 case Intrinsic::fshr: {
1767 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1769 (RetTy->getIntegerBitWidth() == 32 ||
1770 RetTy->getIntegerBitWidth() == 64) &&
1771 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1776 case Intrinsic::clmul: {
1778 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbkc()) {
1781 if (!ST->is64Bit() || LT.second != MVT::i64)
1787 return LT.first * getRISCVInstructionCost(
1788 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1793 case Intrinsic::masked_udiv:
1796 case Intrinsic::masked_sdiv:
1799 case Intrinsic::masked_urem:
1802 case Intrinsic::masked_srem:
1805 case Intrinsic::get_active_lane_mask: {
1806 if (ST->hasVInstructions()) {
1815 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1821 case Intrinsic::stepvector: {
1825 if (ST->hasVInstructions())
1826 return getRISCVInstructionCost(RISCV::VID_V, LT.second,
CostKind) +
1828 getRISCVInstructionCost(RISCV::VADD_VX, LT.second,
CostKind);
1829 return 1 + (LT.first - 1);
1831 case Intrinsic::vector_splice_left:
1832 case Intrinsic::vector_splice_right: {
1837 if (ST->hasVInstructions() && LT.second.isVector()) {
1839 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1844 case Intrinsic::experimental_cttz_elts: {
1845 if (!ST->hasVInstructions())
1850 if (!LT.second.isVector())
1854 if (LT.second.getVectorElementType() != MVT::i1)
1855 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second,
CostKind);
1857 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind);
1869 return LT.first *
Cost;
1871 case Intrinsic::experimental_vp_splice: {
1879 case Intrinsic::vp_merge: {
1887 case Intrinsic::fptoui_sat:
1888 case Intrinsic::fptosi_sat: {
1890 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1895 if (!SrcTy->isVectorTy())
1898 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1915 case Intrinsic::experimental_vector_extract_last_active: {
1937 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1938 TLI->getVectorIdxTy(
getDataLayout()), MaskTy->getElementCount(),
1939 true, &VScaleRange);
1940 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1948 if (StepLT.first > 1)
1952 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1954 Cost += MaskLT.first *
1955 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second,
CostKind);
1957 Cost += StepLT.first *
1958 getRISCVInstructionCost(Opcodes, StepLT.second,
CostKind);
1962 Cost += ValLT.first *
1963 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1967 case Intrinsic::vector_interleave2:
1968 case Intrinsic::vector_deinterleave2: {
1969 if (!ST->hasStdExtZvzip())
1972 bool IsInterleave = ICA.
getID() == Intrinsic::vector_interleave2;
1973 Type *InterleavedTy = IsInterleave ? RetTy : ICA.
getArgTypes().front();
1977 [](
const Value *Arg) { return isa<UndefValue>(Arg); }))
1984 unsigned HalfVF = HalfFVT->getNumElements();
1989 for (
unsigned Start = 0; Start != 2; ++Start)
1996 if (!LT.second.isScalableVector())
1999 if (std::optional<MVT> CostVT = getZvzipVZIPCostVT(LT.second))
2001 getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT,
CostKind);
2002 }
else if (std::optional<MVT> CostVT = getZvzipVUNZIPCostVT(LT.second)) {
2004 getRISCVInstructionCost({RISCV::VUNZIPE_V, RISCV::VUNZIPO_V},
2011 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
2013 LT.second.isVector()) {
2014 MVT EltTy = LT.second.getVectorElementType();
2016 ICA.
getID(), EltTy))
2017 return LT.first * Entry->Cost;
2030 if (ST->hasVInstructions() && PtrTy->
isVectorTy())
2048 if (ST->hasStdExtP() &&
2056 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
2057 Dst->getScalarSizeInBits() > ST->getELen())
2060 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2075 if (Src->getScalarSizeInBits() == 1) {
2080 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second,
CostKind) +
2081 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
2087 if (Dst->getScalarSizeInBits() == 1) {
2093 return SrcLT.first *
2094 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
2106 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
2107 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2109 SrcLT.second.getSizeInBits()) ||
2111 DstLT.second.getSizeInBits()) ||
2112 SrcLT.first > 1 || DstLT.first > 1)
2116 assert((SrcLT.first == 1) && (DstLT.first == 1) &&
"Illegal type");
2118 int PowDiff = (int)
Log2_32(DstLT.second.getScalarSizeInBits()) -
2119 (int)
Log2_32(SrcLT.second.getScalarSizeInBits());
2123 if ((PowDiff < 1) || (PowDiff > 3))
2125 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2126 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2129 return getRISCVInstructionCost(
Op, DstLT.second,
CostKind);
2135 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2136 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2140 : RISCV::VFNCVT_F_F_W;
2142 for (; SrcEltSize != DstEltSize;) {
2146 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2148 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2156 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2158 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2160 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2161 unsigned SrcEltSize = Src->getScalarSizeInBits();
2162 unsigned DstEltSize = Dst->getScalarSizeInBits();
2164 if ((SrcEltSize == 16) &&
2165 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2171 std::pair<InstructionCost, MVT> VecF32LT =
2174 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2179 if (DstEltSize == SrcEltSize)
2180 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2181 else if (DstEltSize > SrcEltSize)
2182 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2187 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2188 Cost += getRISCVInstructionCost(FNCVT, VecVT,
CostKind);
2189 if ((SrcEltSize / 2) > DstEltSize) {
2200 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2201 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2202 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2203 unsigned SrcEltSize = Src->getScalarSizeInBits();
2204 unsigned DstEltSize = Dst->getScalarSizeInBits();
2207 if ((DstEltSize == 16) &&
2208 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2214 std::pair<InstructionCost, MVT> VecF32LT =
2217 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2222 if (DstEltSize == SrcEltSize)
2223 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2224 else if (DstEltSize > SrcEltSize) {
2225 if ((DstEltSize / 2) > SrcEltSize) {
2229 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2232 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2234 Cost += getRISCVInstructionCost(FNCVT, DstLT.second,
CostKind);
2241unsigned RISCVTTIImpl::getEstimatedVLFor(
VectorType *Ty)
const {
2243 const unsigned EltSize =
DL.getTypeSizeInBits(Ty->getElementType());
2244 const unsigned MinSize =
DL.getTypeSizeInBits(Ty).getKnownMinValue();
2259 if (Ty->getScalarSizeInBits() > ST->getELen())
2263 if (Ty->getElementType()->isIntegerTy(1)) {
2267 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2273 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2277 case Intrinsic::maximum:
2279 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2281 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2296 case Intrinsic::minimum:
2298 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2300 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2306 const unsigned EltTyBits =
DL.getTypeSizeInBits(DstTy);
2315 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2324 case Intrinsic::smax:
2325 SplitOp = RISCV::VMAX_VV;
2326 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2328 case Intrinsic::smin:
2329 SplitOp = RISCV::VMIN_VV;
2330 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2332 case Intrinsic::umax:
2333 SplitOp = RISCV::VMAXU_VV;
2334 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2336 case Intrinsic::umin:
2337 SplitOp = RISCV::VMINU_VV;
2338 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2340 case Intrinsic::maxnum:
2341 SplitOp = RISCV::VFMAX_VV;
2342 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2344 case Intrinsic::minnum:
2345 SplitOp = RISCV::VFMIN_VV;
2346 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2351 (LT.first > 1) ? (LT.first - 1) *
2352 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2354 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2359 std::optional<FastMathFlags> FMF,
2365 if (Ty->getScalarSizeInBits() > ST->getELen())
2368 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2376 Type *ElementTy = Ty->getElementType();
2381 if (LT.second == MVT::v1i1)
2382 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind) +
2400 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2401 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second,
CostKind) +
2402 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second,
CostKind) +
2403 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2412 return (LT.first - 1) *
2413 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind) +
2414 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) + 1;
2422 return (LT.first - 1) *
2423 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second,
CostKind) +
2424 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2437 SplitOp = RISCV::VADD_VV;
2438 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2441 SplitOp = RISCV::VOR_VV;
2442 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2445 SplitOp = RISCV::VXOR_VV;
2446 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2449 SplitOp = RISCV::VAND_VV;
2450 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2454 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2455 LT.second.getScalarType() == MVT::bf16)
2459 for (
unsigned i = 0; i < LT.first.getValue(); i++)
2462 return getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2464 SplitOp = RISCV::VFADD_VV;
2465 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2470 (LT.first > 1) ? (LT.first - 1) *
2471 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2473 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2477 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *ValTy,
2488 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2494 if (IsUnsigned && Opcode == Instruction::Add &&
2495 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2499 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind);
2506 return (LT.first - 1) +
2513 assert(OpInfo.isConstant() &&
"non constant operand?");
2520 if (OpInfo.isUniform())
2526 return getConstantPoolLoadCost(Ty,
CostKind);
2535 EVT VT = TLI->getValueType(
DL, Src,
true);
2537 if (VT == MVT::Other ||
2543 if (Opcode == Instruction::Store && OpInfo.isConstant())
2558 if (Src->
isVectorTy() && LT.second.isVector() &&
2560 LT.second.getSizeInBits()))
2570 if (ST->hasVInstructions() && LT.second.isVector() &&
2572 BaseCost *= TLI->getLMULCost(LT.second);
2573 return Cost + BaseCost;
2582 Op1Info, Op2Info,
I);
2586 Op1Info, Op2Info,
I);
2591 Op1Info, Op2Info,
I);
2593 auto GetConstantMatCost =
2595 if (OpInfo.isUniform())
2600 return getConstantPoolLoadCost(ValTy,
CostKind);
2605 ConstantMatCost += GetConstantMatCost(Op1Info);
2607 ConstantMatCost += GetConstantMatCost(Op2Info);
2610 if (Opcode == Instruction::Select && LT.second.isVector()) {
2611 if (CondTy->isVectorTy()) {
2616 return ConstantMatCost +
2618 getRISCVInstructionCost(
2619 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2623 return ConstantMatCost +
2624 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2634 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2635 return ConstantMatCost +
2637 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2639 LT.first * getRISCVInstructionCost(
2640 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2647 return ConstantMatCost +
2648 LT.first * getRISCVInstructionCost(
2649 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2653 if ((Opcode == Instruction::ICmp) && ValTy->
isVectorTy() &&
2657 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2662 if ((Opcode == Instruction::FCmp) && ValTy->
isVectorTy() &&
2667 return ConstantMatCost +
2668 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind);
2678 Op1Info, Op2Info,
I);
2687 return ConstantMatCost +
2688 LT.first * getRISCVInstructionCost(
2689 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2696 return ConstantMatCost +
2698 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2707 return ConstantMatCost +
2709 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second,
CostKind);
2722 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2723 U->getType()->isIntegerTy() &&
2724 !isa<ConstantData>(U->getOperand(1)) &&
2725 !isa<ConstantData>(U->getOperand(2));
2733 Op1Info, Op2Info,
I);
2740 return Opcode == Instruction::PHI ? 0 : 1;
2757 if (Opcode != Instruction::ExtractElement &&
2758 Opcode != Instruction::InsertElement)
2764 if (Opcode == Instruction::InsertElement &&
2765 VIC == TTI::VectorInstrContext::SplatOpFolded &&
2766 ST->sinkSplatOperands() && Index == 0)
2773 if (!LT.second.isVector()) {
2783 auto NumElems = FixedVecTy->getNumElements();
2789 return Opcode == Instruction::ExtractElement
2790 ? StoreCost * NumElems + LoadCost
2791 : (StoreCost + LoadCost) * NumElems + StoreCost;
2795 if (LT.second.isScalableVector() && !LT.first.isValid())
2803 if (Opcode == Instruction::ExtractElement) {
2809 return ExtendCost + ExtractCost;
2819 return ExtendCost + InsertCost + TruncCost;
2826 if (LT.second.isFloatingPoint())
2827 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2831 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2833 getRISCVInstructionCost(MoveOpc, LT.second,
CostKind);
2835 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2840 if (LT.second.isFixedLengthVector()) {
2841 unsigned Width = LT.second.getVectorNumElements();
2842 Index = Index % Width;
2847 if (
auto VLEN = ST->getRealVLen()) {
2848 unsigned EltSize = LT.second.getScalarSizeInBits();
2849 unsigned M1Max = *VLEN / EltSize;
2850 Index = Index % M1Max;
2856 else if (Opcode == Instruction::InsertElement)
2864 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2865 LT.second.isScalableVector()))) {
2867 Align VecAlign =
DL.getPrefTypeAlign(Val);
2868 Align SclAlign =
DL.getPrefTypeAlign(ScalarType);
2873 if (Opcode == Instruction::ExtractElement)
2909 Opcode == Instruction::InsertElement
2910 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2911 RISCV::VSLIDE1DOWN_VX,
2912 RISCV::VSLIDEUP_VX},
2914 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2915 RISCV::VSRL_VX, RISCV::VMV_X_S},
2918 return BaseCost + SlideCost;
2924 unsigned Index)
const {
2933 assert(Index < EC.getKnownMinValue() &&
"Unexpected reverse index");
2935 EC.getKnownMinValue() - 1 - Index,
nullptr,
2944std::optional<InstructionCost>
2950 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2952 if (Opcode == Instruction::UDiv)
2959 return std::nullopt;
2981 if (std::optional<InstructionCost> CombinedCost =
2983 Op2Info, Args, CtxI))
2984 return *CombinedCost;
2988 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2991 if (!LT.second.isVector()) {
3001 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
3002 if (
const auto *Entry =
CostTableLookup(DivTbl, ISDOpcode, LT.second))
3003 return Entry->Cost * LT.first;
3012 if ((LT.second.getVectorElementType() == MVT::f16 ||
3013 LT.second.getVectorElementType() == MVT::bf16) &&
3014 TLI->getOperationAction(ISDOpcode, LT.second) ==
3016 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
3020 CastCost += LT.first * Args.size() *
3028 LT.second = PromotedVT;
3031 auto getConstantMatCost =
3041 return getConstantPoolLoadCost(Ty,
CostKind);
3047 ConstantMatCost += getConstantMatCost(0, Op1Info);
3049 ConstantMatCost += getConstantMatCost(1, Op2Info);
3052 switch (ISDOpcode) {
3055 Op = RISCV::VADD_VV;
3060 Op = RISCV::VSLL_VV;
3065 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
3070 Op = RISCV::VMUL_VV;
3074 Op = RISCV::VDIV_VV;
3078 Op = RISCV::VREM_VV;
3082 Op = RISCV::VFADD_VV;
3085 Op = RISCV::VFMUL_VV;
3088 Op = RISCV::VFDIV_VV;
3091 Op = RISCV::VFSGNJN_VV;
3096 return CastCost + ConstantMatCost +
3105 if (Ty->isFPOrFPVectorTy())
3107 return CastCost + ConstantMatCost + LT.first *
InstrCost;
3130 if (Info.isSameBase() && V !=
Base) {
3131 if (
GEP->hasAllConstantIndices())
3137 unsigned Stride =
DL.getTypeStoreSize(AccessTy);
3138 if (Info.isUnitStride() &&
3144 GEP->getType()->getPointerAddressSpace()))
3147 {TTI::OK_AnyValue, TTI::OP_None},
3148 {TTI::OK_AnyValue, TTI::OP_None}, {});
3165 if (ST->enableDefaultUnroll())
3175 if (L->getHeader()->getParent()->hasOptSize())
3179 L->getExitingBlocks(ExitingBlocks);
3181 <<
"Blocks: " << L->getNumBlocks() <<
"\n"
3182 <<
"Exit blocks: " << ExitingBlocks.
size() <<
"\n");
3186 if (ExitingBlocks.
size() > 2)
3191 if (L->getNumBlocks() > 4)
3199 for (
auto *BB : L->getBlocks()) {
3200 for (
auto &
I : *BB) {
3204 if (IsVectorized && (
I.getType()->isVectorTy() ||
3206 return V->getType()->isVectorTy();
3245 bool HasMask =
false;
3248 bool IsWrite) -> int64_t {
3249 if (
auto *TarExtTy =
3251 return TarExtTy->getIntParameter(0);
3257 case Intrinsic::riscv_vle_mask:
3258 case Intrinsic::riscv_vse_mask:
3259 case Intrinsic::riscv_vlseg2_mask:
3260 case Intrinsic::riscv_vlseg3_mask:
3261 case Intrinsic::riscv_vlseg4_mask:
3262 case Intrinsic::riscv_vlseg5_mask:
3263 case Intrinsic::riscv_vlseg6_mask:
3264 case Intrinsic::riscv_vlseg7_mask:
3265 case Intrinsic::riscv_vlseg8_mask:
3266 case Intrinsic::riscv_vsseg2_mask:
3267 case Intrinsic::riscv_vsseg3_mask:
3268 case Intrinsic::riscv_vsseg4_mask:
3269 case Intrinsic::riscv_vsseg5_mask:
3270 case Intrinsic::riscv_vsseg6_mask:
3271 case Intrinsic::riscv_vsseg7_mask:
3272 case Intrinsic::riscv_vsseg8_mask:
3275 case Intrinsic::riscv_vle:
3276 case Intrinsic::riscv_vse:
3277 case Intrinsic::riscv_vlseg2:
3278 case Intrinsic::riscv_vlseg3:
3279 case Intrinsic::riscv_vlseg4:
3280 case Intrinsic::riscv_vlseg5:
3281 case Intrinsic::riscv_vlseg6:
3282 case Intrinsic::riscv_vlseg7:
3283 case Intrinsic::riscv_vlseg8:
3284 case Intrinsic::riscv_vsseg2:
3285 case Intrinsic::riscv_vsseg3:
3286 case Intrinsic::riscv_vsseg4:
3287 case Intrinsic::riscv_vsseg5:
3288 case Intrinsic::riscv_vsseg6:
3289 case Intrinsic::riscv_vsseg7:
3290 case Intrinsic::riscv_vsseg8: {
3307 Ty = TarExtTy->getTypeParameter(0U);
3312 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3313 unsigned VLIndex = RVVIInfo->VLOperand;
3314 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3322 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3325 unsigned ElemSize = Ty->getScalarSizeInBits();
3329 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3330 Alignment, Mask, EVL);
3333 case Intrinsic::riscv_vlse_mask:
3334 case Intrinsic::riscv_vsse_mask:
3335 case Intrinsic::riscv_vlsseg2_mask:
3336 case Intrinsic::riscv_vlsseg3_mask:
3337 case Intrinsic::riscv_vlsseg4_mask:
3338 case Intrinsic::riscv_vlsseg5_mask:
3339 case Intrinsic::riscv_vlsseg6_mask:
3340 case Intrinsic::riscv_vlsseg7_mask:
3341 case Intrinsic::riscv_vlsseg8_mask:
3342 case Intrinsic::riscv_vssseg2_mask:
3343 case Intrinsic::riscv_vssseg3_mask:
3344 case Intrinsic::riscv_vssseg4_mask:
3345 case Intrinsic::riscv_vssseg5_mask:
3346 case Intrinsic::riscv_vssseg6_mask:
3347 case Intrinsic::riscv_vssseg7_mask:
3348 case Intrinsic::riscv_vssseg8_mask:
3351 case Intrinsic::riscv_vlse:
3352 case Intrinsic::riscv_vsse:
3353 case Intrinsic::riscv_vlsseg2:
3354 case Intrinsic::riscv_vlsseg3:
3355 case Intrinsic::riscv_vlsseg4:
3356 case Intrinsic::riscv_vlsseg5:
3357 case Intrinsic::riscv_vlsseg6:
3358 case Intrinsic::riscv_vlsseg7:
3359 case Intrinsic::riscv_vlsseg8:
3360 case Intrinsic::riscv_vssseg2:
3361 case Intrinsic::riscv_vssseg3:
3362 case Intrinsic::riscv_vssseg4:
3363 case Intrinsic::riscv_vssseg5:
3364 case Intrinsic::riscv_vssseg6:
3365 case Intrinsic::riscv_vssseg7:
3366 case Intrinsic::riscv_vssseg8: {
3383 Ty = TarExtTy->getTypeParameter(0U);
3388 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3389 unsigned VLIndex = RVVIInfo->VLOperand;
3390 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3399 unsigned PointerAlign = Alignment.valueOrOne().value();
3402 Alignment =
Align(1);
3409 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3412 unsigned ElemSize = Ty->getScalarSizeInBits();
3416 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3417 Alignment, Mask, EVL, Stride);
3420 case Intrinsic::riscv_vloxei_mask:
3421 case Intrinsic::riscv_vluxei_mask:
3422 case Intrinsic::riscv_vsoxei_mask:
3423 case Intrinsic::riscv_vsuxei_mask:
3424 case Intrinsic::riscv_vloxseg2_mask:
3425 case Intrinsic::riscv_vloxseg3_mask:
3426 case Intrinsic::riscv_vloxseg4_mask:
3427 case Intrinsic::riscv_vloxseg5_mask:
3428 case Intrinsic::riscv_vloxseg6_mask:
3429 case Intrinsic::riscv_vloxseg7_mask:
3430 case Intrinsic::riscv_vloxseg8_mask:
3431 case Intrinsic::riscv_vluxseg2_mask:
3432 case Intrinsic::riscv_vluxseg3_mask:
3433 case Intrinsic::riscv_vluxseg4_mask:
3434 case Intrinsic::riscv_vluxseg5_mask:
3435 case Intrinsic::riscv_vluxseg6_mask:
3436 case Intrinsic::riscv_vluxseg7_mask:
3437 case Intrinsic::riscv_vluxseg8_mask:
3438 case Intrinsic::riscv_vsoxseg2_mask:
3439 case Intrinsic::riscv_vsoxseg3_mask:
3440 case Intrinsic::riscv_vsoxseg4_mask:
3441 case Intrinsic::riscv_vsoxseg5_mask:
3442 case Intrinsic::riscv_vsoxseg6_mask:
3443 case Intrinsic::riscv_vsoxseg7_mask:
3444 case Intrinsic::riscv_vsoxseg8_mask:
3445 case Intrinsic::riscv_vsuxseg2_mask:
3446 case Intrinsic::riscv_vsuxseg3_mask:
3447 case Intrinsic::riscv_vsuxseg4_mask:
3448 case Intrinsic::riscv_vsuxseg5_mask:
3449 case Intrinsic::riscv_vsuxseg6_mask:
3450 case Intrinsic::riscv_vsuxseg7_mask:
3451 case Intrinsic::riscv_vsuxseg8_mask:
3454 case Intrinsic::riscv_vloxei:
3455 case Intrinsic::riscv_vluxei:
3456 case Intrinsic::riscv_vsoxei:
3457 case Intrinsic::riscv_vsuxei:
3458 case Intrinsic::riscv_vloxseg2:
3459 case Intrinsic::riscv_vloxseg3:
3460 case Intrinsic::riscv_vloxseg4:
3461 case Intrinsic::riscv_vloxseg5:
3462 case Intrinsic::riscv_vloxseg6:
3463 case Intrinsic::riscv_vloxseg7:
3464 case Intrinsic::riscv_vloxseg8:
3465 case Intrinsic::riscv_vluxseg2:
3466 case Intrinsic::riscv_vluxseg3:
3467 case Intrinsic::riscv_vluxseg4:
3468 case Intrinsic::riscv_vluxseg5:
3469 case Intrinsic::riscv_vluxseg6:
3470 case Intrinsic::riscv_vluxseg7:
3471 case Intrinsic::riscv_vluxseg8:
3472 case Intrinsic::riscv_vsoxseg2:
3473 case Intrinsic::riscv_vsoxseg3:
3474 case Intrinsic::riscv_vsoxseg4:
3475 case Intrinsic::riscv_vsoxseg5:
3476 case Intrinsic::riscv_vsoxseg6:
3477 case Intrinsic::riscv_vsoxseg7:
3478 case Intrinsic::riscv_vsoxseg8:
3479 case Intrinsic::riscv_vsuxseg2:
3480 case Intrinsic::riscv_vsuxseg3:
3481 case Intrinsic::riscv_vsuxseg4:
3482 case Intrinsic::riscv_vsuxseg5:
3483 case Intrinsic::riscv_vsuxseg6:
3484 case Intrinsic::riscv_vsuxseg7:
3485 case Intrinsic::riscv_vsuxseg8: {
3502 Ty = TarExtTy->getTypeParameter(0U);
3507 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3508 unsigned VLIndex = RVVIInfo->VLOperand;
3509 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3522 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3525 unsigned ElemSize = Ty->getScalarSizeInBits();
3530 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3531 Align(1), Mask, EVL,
3540 if (Ty->isVectorTy()) {
3543 if ((EltTy->
isHalfTy() && !ST->hasVInstructionsF16()) ||
3549 if (
Size.isScalable() && ST->hasVInstructions())
3552 if (ST->useRVVForFixedLengthVectors())
3572 return std::max<unsigned>(1U, RegWidth.
getFixedValue() / ElemWidth);
3580 return ST->enableUnalignedVectorMem();
3586 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3608 Align Alignment)
const {
3618 if (VTy->getElementType()->isIntegerTy(8)) {
3619 uint64_t MaxEltCount = VTy->getElementCount().getKnownMinValue();
3620 if (VTy->isScalableTy())
3623 if (MaxEltCount > 256)
3632 Align Alignment)
const {
3639 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3642 return TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, ElementTy));
3651 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
3652 bool Considerable =
false;
3653 AllowPromotionWithoutCommonHeader =
false;
3656 Type *ConsideredSExtType =
3658 if (
I.getType() != ConsideredSExtType)
3662 for (
const User *U :
I.users()) {
3664 Considerable =
true;
3668 if (GEPInst->getNumOperands() > 2) {
3669 AllowPromotionWithoutCommonHeader =
true;
3674 return Considerable;
3679 case Instruction::Add:
3680 case Instruction::Sub:
3681 case Instruction::Mul:
3682 case Instruction::And:
3683 case Instruction::Or:
3684 case Instruction::Xor:
3685 case Instruction::FAdd:
3686 case Instruction::FSub:
3687 case Instruction::FMul:
3688 case Instruction::FDiv:
3689 case Instruction::ICmp:
3690 case Instruction::FCmp:
3692 case Instruction::Shl:
3693 case Instruction::LShr:
3694 case Instruction::AShr:
3695 case Instruction::UDiv:
3696 case Instruction::SDiv:
3697 case Instruction::URem:
3698 case Instruction::SRem:
3699 case Instruction::Select:
3700 return Operand == 1;
3707 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3717 switch (
II->getIntrinsicID()) {
3718 case Intrinsic::fma:
3719 case Intrinsic::fmuladd:
3720 return Operand == 0 || Operand == 1;
3721 case Intrinsic::vp_udiv:
3722 case Intrinsic::vp_sdiv:
3723 case Intrinsic::vp_urem:
3724 case Intrinsic::vp_srem:
3725 case Intrinsic::ssub_sat:
3726 case Intrinsic::usub_sat:
3727 return Operand == 1;
3729 case Intrinsic::smin:
3730 case Intrinsic::umin:
3731 case Intrinsic::smax:
3732 case Intrinsic::umax:
3733 case Intrinsic::sadd_sat:
3734 case Intrinsic::uadd_sat:
3735 return Operand == 0 || Operand == 1;
3744 GatherUseOps)
const {
3745 if (Scalars.
empty() || !ST->hasVInstructions() || !ST->sinkSplatOperands() ||
3750 if (SplatIt == Scalars.
end() || (*SplatIt)->getType()->isIntegerTy(1) ||
3756 if (!GatherUseOps(UserOps) || UserOps.
empty())
3775 if (
I->isBitwiseLogicOp()) {
3776 if (!
I->getType()->isVectorTy()) {
3777 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3778 for (
auto &
Op :
I->operands()) {
3786 }
else if (
I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3787 for (
auto &
Op :
I->operands()) {
3799 Ops.push_back(&Not);
3800 Ops.push_back(&InsertElt);
3808 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3816 if (!ST->sinkSplatOperands())
3819 for (
auto OpIdx :
enumerate(
I->operands())) {
3839 for (
Use &U :
Op->uses()) {
3846 Use *InsertEltUse = &
Op->getOperandUse(0);
3849 Ops.push_back(&InsertElt->getOperandUse(1));
3850 Ops.push_back(InsertEltUse);
3851 Ops.push_back(&OpIdx.value());
3860 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3863 Options.AllowOverlappingLoads =
true;
3864 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3866 if (ST->is64Bit()) {
3867 Options.LoadSizes = {8, 4, 2, 1};
3868 Options.AllowedTailExpansions = {3, 5, 6};
3870 Options.LoadSizes = {4, 2, 1};
3871 Options.AllowedTailExpansions = {3};
3874 if (IsZeroCmp && ST->hasVInstructions()) {
3875 unsigned VLenB = ST->getRealMinVLen() / 8;
3878 unsigned MinSize = ST->getXLen() / 8 + 1;
3879 unsigned MaxSize = VLenB * 8;
3893 if (
I->getOpcode() == Instruction::Or &&
3897 if (
I->getOpcode() == Instruction::Add ||
3898 I->getOpcode() == Instruction::Sub)
3916std::optional<Instruction *>
3922 if (
is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3923 II.getIntrinsicID())) {
3926 if (!ST->hasVInstructions())
3929 bool HasAVL =
II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3930 unsigned Offset = HasAVL ? 1 : 0;
3931 unsigned BitWidth =
II.getType()->getIntegerBitWidth();
3956 Value *AVL =
II.getArgOperand(0);
3985 II.getRange().value_or(ConstantRange::getFull(
BitWidth));
3987 if (NewRange != OldRange) {
3988 II.addRangeRetAttr(NewRange);
3998 if (
II.user_empty())
4003 const APInt *Scalar;
4008 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
4012 unsigned TargetEltBW =
DL.getTypeSizeInBits(TargetVecTy->getElementType());
4013 unsigned SourceEltBW =
DL.getTypeSizeInBits(SourceVecTy->getElementType());
4014 if (TargetEltBW % SourceEltBW)
4016 unsigned TargetScale = TargetEltBW / SourceEltBW;
4017 if (VL % TargetScale || TargetScale == 1)
4019 Type *VLTy =
II.getOperand(2)->getType();
4020 ElementCount SourceEC = SourceVecTy->getElementCount();
4021 unsigned NewEltBW = SourceEltBW * TargetScale;
4023 !
DL.fitsInLegalInteger(NewEltBW))
4026 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, NewEltTy)))
4030 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
4031 "Lossless bitcast between types expected");
4037 RetTy, Intrinsic::riscv_vmv_v_x,
4038 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
4039 ConstantInt::get(VLTy, VL / TargetScale)}),
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
uint64_t IntrinsicInst * II
This file describes how to lower LLVM code to machine code.
Class for arbitrary precision integers.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
@ ICMP_SLT
signed less than
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Convenience struct for specifying and reasoning about fast-math flags.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
static FixedVectorType * getHalfElementsVectorType(FixedVectorType *VTy)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
Intrinsic::ID getID() const
bool isTypeBasedOnly() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Represents a single loop in the control flow graph.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
ElementCount getVectorElementCount() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
Align getAlignment() const
unsigned getAddressSpace() const
Type * getDataType() const
bool getVariableMask() const
const Value * getStrideVal() const
Intrinsic::ID getID() const
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
TargetTransformInfo::VectorInstrContext getBuildVectorContextHint(ArrayRef< int > Mask, ArrayRef< Value * > Scalars, function_ref< bool(SmallVectorImpl< TargetTransformInfo::BuildVectorUseOp > &)> GatherUseOps) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CtxI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
An efficient, type-erasing, non-owning reference to a callable.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
@ SplatOpFolded
All of the value's users support splatting the value.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
bool isPairEven(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-even idiom.
bool isPairOdd(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-odd idiom.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
auto find_if_not(R &&Range, UnaryPredicate P)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const