18#include "llvm/IR/IntrinsicsRISCV.h"
26#define DEBUG_TYPE "riscvtti"
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
44 cl::desc(
"Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
57 size_t NumInstr = OpCodes.size();
62 return LMULCost * NumInstr;
64 for (
auto Op : OpCodes) {
66 case RISCV::VRGATHER_VI:
69 case RISCV::VRGATHER_VV:
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
97 case RISCV::VFREDOSUM_VS: {
105 case RISCV::VFMV_F_S:
110 case RISCV::VFMV_S_F:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
117 case RISCV::VFIRST_M:
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
160 if (!BO || !BO->hasOneUse())
163 if (BO->getOpcode() != Instruction::Shl)
174 if (ShAmt == Trailing)
191 if (!Cmp || !Cmp->isEquality())
207 if ((CmpC & Mask) != CmpC)
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
230 bool Takes12BitImm =
false;
231 unsigned ImmArgIdx = ~0U;
234 case Instruction::GetElementPtr:
239 case Instruction::Store: {
244 if (Idx == 1 || !Inst)
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(),
DL, getTLI()->getValueType(
DL, Ty),
257 case Instruction::Load:
260 case Instruction::And:
262 if (
Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
265 if (
Imm == UINT64_C(0xffffffff) &&
266 ((ST->hasStdExtZba() && ST->isRV64()) || ST->isRV32()))
269 if (ST->hasStdExtZbs() && (~
Imm).isPowerOf2())
271 if (Inst && Idx == 1 &&
Imm.getBitWidth() <= ST->getXLen() &&
274 if (Inst && Idx == 1 &&
Imm.getBitWidth() == 64 &&
277 Takes12BitImm =
true;
279 case Instruction::Add:
280 Takes12BitImm =
true;
282 case Instruction::Or:
283 case Instruction::Xor:
285 if (ST->hasStdExtZbs() &&
Imm.isPowerOf2())
287 Takes12BitImm =
true;
289 case Instruction::Mul:
291 if (
Imm.isPowerOf2() ||
Imm.isNegatedPowerOf2())
294 if ((
Imm + 1).isPowerOf2() || (
Imm - 1).isPowerOf2())
297 Takes12BitImm =
true;
299 case Instruction::Sub:
300 case Instruction::Shl:
301 case Instruction::LShr:
302 case Instruction::AShr:
303 Takes12BitImm =
true;
314 if (
Imm.getSignificantBits() <= 64 &&
337 return ST->hasVInstructions();
347 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
351 if (Opcode == Instruction::FAdd)
360 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
361 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
362 InputTypeA != InputTypeB || !InputTypeA->
isIntegerTy(8) ||
378 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second,
CostKind);
387 std::pair<InstructionCost, MVT> AccLT =
395 bool WidenFirst =
false;
396 if (VF.
isScalable() && AccLT.second.isScalableVector()) {
397 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
410 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
413 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second,
CostKind);
417 std::pair<InstructionCost, MVT> RedLT =
419 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
421 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
425 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
437 switch (
II->getIntrinsicID()) {
441 case Intrinsic::vector_reduce_mul:
442 case Intrinsic::vector_reduce_fmul:
448 if (ST->hasVInstructions())
449 if (
unsigned MinVLen = ST->getRealMinVLen();
464 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
467 (ST->hasVInstructions() &&
490 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
496RISCVTTIImpl::getConstantPoolLoadCost(
Type *Ty,
501 return getStaticDataAddrGenerationCost(
CostKind) +
507 unsigned Size = Mask.size();
510 for (
unsigned I = 0;
I !=
Size; ++
I) {
511 if (
static_cast<unsigned>(Mask[
I]) ==
I)
517 for (
unsigned J =
I + 1; J !=
Size; ++J)
519 if (
static_cast<unsigned>(Mask[J]) != J %
I)
547 "Expected fixed vector type and non-empty mask");
550 unsigned NumOfDests =
divideCeil(Mask.size(), LegalNumElts);
554 if (NumOfDests <= 1 ||
556 Tp->getElementType()->getPrimitiveSizeInBits() ||
557 LegalNumElts >= Tp->getElementCount().getFixedValue())
560 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
563 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
567 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
568 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
569 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
571 assert(NormalizedVF >= Mask.size() &&
572 "Normalized mask expected to be not shorter than original mask.");
577 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
578 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
581 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
584 Cost +=
TTI.getShuffleCost(
587 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
589 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
590 Cost +=
TTI.getShuffleCost(
593 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
616 if (!VLen || Mask.empty())
620 LegalVT =
TTI.getTypeLegalizationCost(
626 if (NumOfDests <= 1 ||
628 Tp->getElementType()->getPrimitiveSizeInBits() ||
632 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
635 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
641 unsigned NormalizedVF =
646 assert(NormalizedVF >= Mask.size() &&
647 "Normalized mask expected to be not shorter than original mask.");
653 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
654 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
657 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
662 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
664 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
666 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
673 if ((NumOfDestRegs > 2 && NumShuffles <=
static_cast<int>(NumOfDestRegs)) ||
674 (NumOfDestRegs <= 2 && NumShuffles < 4))
689 if (!
LT.second.isFixedLengthVector())
697 auto GetSlideOpcode = [&](
int SlideAmt) {
699 bool IsVI =
isUInt<5>(std::abs(SlideAmt));
701 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
702 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
705 std::array<std::pair<int, int>, 2> SrcInfo;
709 if (SrcInfo[1].second == 0)
713 if (SrcInfo[0].second != 0) {
714 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
715 FirstSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
718 if (SrcInfo[1].first == -1)
719 return FirstSlideCost;
722 if (SrcInfo[1].second != 0) {
723 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
724 SecondSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
727 getRISCVInstructionCost(RISCV::VMERGE_VVM,
LT.second,
CostKind);
734 return FirstSlideCost + SecondSlideCost + MaskCost;
745 "Expected the Mask to match the return size if given");
747 "Expected the same scalar types");
763 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
765 *
this, LT.second, ST->getRealVLen(),
767 if (VRegSplittingCost.
isValid())
768 return VRegSplittingCost;
773 if (Mask.size() >= 2) {
774 MVT EltTp = LT.second.getVectorElementType();
785 return 2 * LT.first * TLI->getLMULCost(LT.second);
787 if (Mask[0] == 0 || Mask[0] == 1) {
791 if (
equal(DeinterleaveMask, Mask))
792 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
797 if (LT.second.getScalarSizeInBits() != 1 &&
800 unsigned NumSlides =
Log2_32(Mask.size() / SubVectorSize);
802 for (
unsigned I = 0;
I != NumSlides; ++
I) {
803 unsigned InsertIndex = SubVectorSize * (1 <<
I);
808 std::pair<InstructionCost, MVT> DestLT =
813 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
827 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
828 LT.second.getVectorNumElements() <= 256)) {
833 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second,
CostKind);
847 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
848 LT.second.getVectorNumElements() <= 256)) {
849 auto &
C = SrcTy->getContext();
850 auto EC = SrcTy->getElementCount();
855 return 2 * IndexCost +
856 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
875 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
903 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
904 if (std::optional<unsigned> VLen = ST->getRealVLen();
905 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
906 SubLT.second.getSizeInBits() <= *VLen)
914 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second,
CostKind);
921 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second,
CostKind);
933 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
940 if (IsLoad && LT.second.isVector() &&
942 LT.second.getVectorElementCount()))
946 Instruction::InsertElement);
947 if (LT.second.getScalarSizeInBits() == 1) {
955 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
968 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
969 RISCV::VMV_X_S, RISCV::VMV_V_X,
978 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second,
CostKind);
984 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second,
CostKind);
990 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
991 if (Index >= 0 && Index < 32)
992 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
993 else if (Index < 0 && Index > -32)
994 Opcodes[1] = RISCV::VSLIDEUP_VI;
995 return LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
999 if (!LT.second.isVector())
1005 if (SrcTy->getElementType()->isIntegerTy(1)) {
1017 MVT ContainerVT = LT.second;
1018 if (LT.second.isFixedLengthVector())
1019 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1021 if (ContainerVT.
bitsLE(M1VT)) {
1031 if (LT.second.isFixedLengthVector())
1033 LenCost =
isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1034 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1035 if (LT.second.isFixedLengthVector() &&
1036 isInt<5>(LT.second.getVectorNumElements() - 1))
1037 Opcodes[1] = RISCV::VRSUB_VI;
1039 getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1040 return LT.first * (LenCost + GatherCost);
1047 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1049 getRISCVInstructionCost(M1Opcodes, M1VT,
CostKind) + 3;
1053 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT,
CostKind) * Ratio;
1055 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second,
CostKind);
1056 return FixedCost + LT.first * (GatherCost + SlideCost);
1090 Ty, DemandedElts, Insert, Extract,
CostKind);
1092 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1093 if (Ty->getScalarSizeInBits() == 1) {
1103 assert(LT.second.isFixedLengthVector());
1104 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1108 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second,
CostKind);
1121 switch (MICA.
getID()) {
1122 case Intrinsic::vp_load_ff: {
1123 EVT DataTypeVT = TLI->getValueType(
DL, DataTy);
1124 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1131 case Intrinsic::experimental_vp_strided_load:
1132 case Intrinsic::experimental_vp_strided_store:
1134 case Intrinsic::masked_compressstore:
1135 case Intrinsic::masked_expandload:
1137 case Intrinsic::vp_scatter:
1138 case Intrinsic::vp_gather:
1139 case Intrinsic::masked_scatter:
1140 case Intrinsic::masked_gather:
1142 case Intrinsic::vp_load:
1143 case Intrinsic::vp_store:
1144 case Intrinsic::masked_load:
1145 case Intrinsic::masked_store:
1154 unsigned Opcode = MICA.
getID() == Intrinsic::masked_load ? Instruction::Load
1155 : Instruction::Store;
1170 bool UseMaskForCond,
bool UseMaskForGaps)
const {
1176 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1180 if (LT.second.isVector()) {
1186 VTy->getElementCount().divideCoefficientBy(Factor));
1187 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1188 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1193 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1194 unsigned VecSizeInBits =
1195 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1196 unsigned VLENForTuning =
1198 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1200 MVT SubVecVT = getTLI()->getValueType(
DL, SubVecTy).getSimpleVT();
1201 Cost += Factor * TLI->getLMULCost(SubVecVT);
1207 unsigned NumLoads = getEstimatedVLFor(VTy);
1223 if (UseMaskForGaps) {
1226 "Indices should not contain duplicate elements");
1227 unsigned NumOfFields = Indices.
size();
1228 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.
back() + 1);
1229 if (IsTailGapOnly &&
1230 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1232 if (LT.second.isVector() &&
1233 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1235 FVTy->getElementType(),
1236 FVTy->getElementCount().divideCoefficientBy(Factor));
1237 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1240 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1249 unsigned VF = FVTy->getNumElements() / Factor;
1256 if (Opcode == Instruction::Load) {
1258 for (
unsigned Index : Indices) {
1262 Mask.resize(VF * Factor, -1);
1266 Cost += ShuffleCost;
1284 UseMaskForCond, UseMaskForGaps);
1286 assert(Opcode == Instruction::Store &&
"Opcode must be a store");
1293 return MemCost + ShuffleCost;
1300 bool IsLoad = MICA.
getID() == Intrinsic::masked_gather ||
1301 MICA.
getID() == Intrinsic::vp_gather;
1302 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1308 if ((Opcode == Instruction::Load &&
1310 (Opcode == Instruction::Store &&
1318 unsigned NumLoads = getEstimatedVLFor(&VTy);
1325 unsigned Opcode = MICA.
getID() == Intrinsic::masked_expandload
1327 : Instruction::Store;
1331 bool IsLegal = (Opcode == Instruction::Store &&
1333 (Opcode == Instruction::Load &&
1357 if (Opcode == Instruction::Store)
1358 Opcodes.
append({RISCV::VCOMPRESS_VM});
1360 Opcodes.
append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1362 LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1381 unsigned NumLoads = getEstimatedVLFor(&VTy);
1392 for (
auto *Ty : Tys) {
1393 if (!Ty->isVectorTy())
1407 {Intrinsic::floor, MVT::f32, 9},
1408 {Intrinsic::floor, MVT::f64, 9},
1409 {Intrinsic::ceil, MVT::f32, 9},
1410 {Intrinsic::ceil, MVT::f64, 9},
1411 {Intrinsic::trunc, MVT::f32, 7},
1412 {Intrinsic::trunc, MVT::f64, 7},
1413 {Intrinsic::round, MVT::f32, 9},
1414 {Intrinsic::round, MVT::f64, 9},
1415 {Intrinsic::roundeven, MVT::f32, 9},
1416 {Intrinsic::roundeven, MVT::f64, 9},
1417 {Intrinsic::rint, MVT::f32, 7},
1418 {Intrinsic::rint, MVT::f64, 7},
1419 {Intrinsic::nearbyint, MVT::f32, 9},
1420 {Intrinsic::nearbyint, MVT::f64, 9},
1421 {Intrinsic::bswap, MVT::i16, 3},
1422 {Intrinsic::bswap, MVT::i32, 12},
1423 {Intrinsic::bswap, MVT::i64, 31},
1424 {Intrinsic::bitreverse, MVT::i8, 17},
1425 {Intrinsic::bitreverse, MVT::i16, 24},
1426 {Intrinsic::bitreverse, MVT::i32, 33},
1427 {Intrinsic::bitreverse, MVT::i64, 52},
1428 {Intrinsic::ctpop, MVT::i8, 12},
1429 {Intrinsic::ctpop, MVT::i16, 19},
1430 {Intrinsic::ctpop, MVT::i32, 20},
1431 {Intrinsic::ctpop, MVT::i64, 21},
1432 {Intrinsic::ctlz, MVT::i8, 19},
1433 {Intrinsic::ctlz, MVT::i16, 28},
1434 {Intrinsic::ctlz, MVT::i32, 31},
1435 {Intrinsic::ctlz, MVT::i64, 35},
1436 {Intrinsic::cttz, MVT::i8, 16},
1437 {Intrinsic::cttz, MVT::i16, 23},
1438 {Intrinsic::cttz, MVT::i32, 24},
1439 {Intrinsic::cttz, MVT::i64, 25},
1446 switch (ICA.
getID()) {
1447 case Intrinsic::lrint:
1448 case Intrinsic::llrint:
1449 case Intrinsic::lround:
1450 case Intrinsic::llround: {
1454 if (ST->hasVInstructions() && LT.second.isVector()) {
1456 unsigned SrcEltSz =
DL.getTypeSizeInBits(SrcTy->getScalarType());
1457 unsigned DstEltSz =
DL.getTypeSizeInBits(RetTy->getScalarType());
1458 if (LT.second.getVectorElementType() == MVT::bf16) {
1459 if (!ST->hasVInstructionsBF16Minimal())
1462 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1464 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1465 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1466 !ST->hasVInstructionsF16()) {
1467 if (!ST->hasVInstructionsF16Minimal())
1470 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1472 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1474 }
else if (SrcEltSz > DstEltSz) {
1475 Ops = {RISCV::VFNCVT_X_F_W};
1476 }
else if (SrcEltSz < DstEltSz) {
1477 Ops = {RISCV::VFWCVT_X_F_V};
1479 Ops = {RISCV::VFCVT_X_F_V};
1484 if (SrcEltSz > DstEltSz)
1485 return SrcLT.first *
1486 getRISCVInstructionCost(
Ops, SrcLT.second,
CostKind);
1487 return LT.first * getRISCVInstructionCost(
Ops, LT.second,
CostKind);
1491 case Intrinsic::ceil:
1492 case Intrinsic::floor:
1493 case Intrinsic::trunc:
1494 case Intrinsic::rint:
1495 case Intrinsic::round:
1496 case Intrinsic::roundeven: {
1499 if (!LT.second.isVector() && TLI->isOperationCustom(
ISD::FCEIL, LT.second))
1500 return LT.first * 8;
1503 case Intrinsic::umin:
1504 case Intrinsic::umax:
1505 case Intrinsic::smin:
1506 case Intrinsic::smax: {
1508 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1511 if (ST->hasVInstructions() && LT.second.isVector()) {
1513 switch (ICA.
getID()) {
1514 case Intrinsic::umin:
1515 Op = RISCV::VMINU_VV;
1517 case Intrinsic::umax:
1518 Op = RISCV::VMAXU_VV;
1520 case Intrinsic::smin:
1521 Op = RISCV::VMIN_VV;
1523 case Intrinsic::smax:
1524 Op = RISCV::VMAX_VV;
1527 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1531 case Intrinsic::sadd_sat:
1532 case Intrinsic::ssub_sat:
1533 case Intrinsic::uadd_sat:
1534 case Intrinsic::usub_sat: {
1536 if (ST->hasVInstructions() && LT.second.isVector()) {
1538 switch (ICA.
getID()) {
1539 case Intrinsic::sadd_sat:
1540 Op = RISCV::VSADD_VV;
1542 case Intrinsic::ssub_sat:
1543 Op = RISCV::VSSUB_VV;
1545 case Intrinsic::uadd_sat:
1546 Op = RISCV::VSADDU_VV;
1548 case Intrinsic::usub_sat:
1549 Op = RISCV::VSSUBU_VV;
1552 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1556 case Intrinsic::fma:
1557 case Intrinsic::fmuladd: {
1560 if (ST->hasVInstructions() && LT.second.isVector())
1562 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second,
CostKind);
1565 case Intrinsic::fabs: {
1567 if (ST->hasVInstructions() && LT.second.isVector()) {
1573 if (LT.second.getVectorElementType() == MVT::bf16 ||
1574 (LT.second.getVectorElementType() == MVT::f16 &&
1575 !ST->hasVInstructionsF16()))
1576 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1581 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second,
CostKind);
1585 case Intrinsic::sqrt: {
1587 if (ST->hasVInstructions() && LT.second.isVector()) {
1590 MVT ConvType = LT.second;
1591 MVT FsqrtType = LT.second;
1594 if (LT.second.getVectorElementType() == MVT::bf16) {
1595 if (LT.second == MVT::nxv32bf16) {
1596 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1597 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1598 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1599 ConvType = MVT::nxv16f16;
1600 FsqrtType = MVT::nxv16f32;
1602 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1603 FsqrtOp = {RISCV::VFSQRT_V};
1604 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1606 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1607 !ST->hasVInstructionsF16()) {
1608 if (LT.second == MVT::nxv32f16) {
1609 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1610 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1611 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1612 ConvType = MVT::nxv16f16;
1613 FsqrtType = MVT::nxv16f32;
1615 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1616 FsqrtOp = {RISCV::VFSQRT_V};
1617 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1620 FsqrtOp = {RISCV::VFSQRT_V};
1623 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType,
CostKind) +
1624 getRISCVInstructionCost(ConvOp, ConvType,
CostKind));
1628 case Intrinsic::cttz:
1629 case Intrinsic::ctlz:
1630 case Intrinsic::ctpop: {
1632 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1634 switch (ICA.
getID()) {
1635 case Intrinsic::cttz:
1638 case Intrinsic::ctlz:
1641 case Intrinsic::ctpop:
1642 Op = RISCV::VCPOP_V;
1645 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1649 case Intrinsic::abs: {
1651 if (ST->hasVInstructions() && LT.second.isVector()) {
1653 if (ST->hasStdExtZvabd())
1655 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second,
CostKind);
1660 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1665 case Intrinsic::fshl:
1666 case Intrinsic::fshr: {
1673 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1675 (RetTy->getIntegerBitWidth() == 32 ||
1676 RetTy->getIntegerBitWidth() == 64) &&
1677 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1682 case Intrinsic::clmul: {
1684 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbc() &&
1685 !ST->hasStdExtZbkc()) {
1688 if (!ST->is64Bit() || LT.second != MVT::i64)
1694 return LT.first * getRISCVInstructionCost(
1695 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1700 case Intrinsic::masked_udiv:
1703 case Intrinsic::masked_sdiv:
1706 case Intrinsic::masked_urem:
1709 case Intrinsic::masked_srem:
1712 case Intrinsic::get_active_lane_mask: {
1713 if (ST->hasVInstructions()) {
1722 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1728 case Intrinsic::stepvector: {
1732 if (ST->hasVInstructions())
1733 return getRISCVInstructionCost(RISCV::VID_V, LT.second,
CostKind) +
1735 getRISCVInstructionCost(RISCV::VADD_VX, LT.second,
CostKind);
1736 return 1 + (LT.first - 1);
1738 case Intrinsic::vector_splice_left:
1739 case Intrinsic::vector_splice_right: {
1744 if (ST->hasVInstructions() && LT.second.isVector()) {
1746 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1751 case Intrinsic::experimental_cttz_elts: {
1752 if (!ST->hasVInstructions())
1759 if (LT.second.getVectorElementType() != MVT::i1)
1760 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second,
CostKind);
1762 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind);
1774 return LT.first *
Cost;
1776 case Intrinsic::experimental_vp_splice: {
1784 case Intrinsic::vp_merge: {
1792 case Intrinsic::fptoui_sat:
1793 case Intrinsic::fptosi_sat: {
1795 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1800 if (!SrcTy->isVectorTy())
1803 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1820 case Intrinsic::experimental_vector_extract_last_active: {
1842 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1843 TLI->getVectorIdxTy(
getDataLayout()), MaskTy->getElementCount(),
1844 true, &VScaleRange);
1845 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1853 if (StepLT.first > 1)
1857 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1859 Cost += MaskLT.first *
1860 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second,
CostKind);
1862 Cost += StepLT.first *
1863 getRISCVInstructionCost(Opcodes, StepLT.second,
CostKind);
1867 Cost += ValLT.first *
1868 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1874 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1876 LT.second.isVector()) {
1877 MVT EltTy = LT.second.getVectorElementType();
1879 ICA.
getID(), EltTy))
1880 return LT.first * Entry->Cost;
1893 if (ST->hasVInstructions() && PtrTy->
isVectorTy())
1911 if (ST->hasStdExtP() &&
1919 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1920 Dst->getScalarSizeInBits() > ST->getELen())
1923 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1938 if (Src->getScalarSizeInBits() == 1) {
1943 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second,
CostKind) +
1944 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1950 if (Dst->getScalarSizeInBits() == 1) {
1956 return SrcLT.first *
1957 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
1969 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
1970 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
1972 SrcLT.second.getSizeInBits()) ||
1974 DstLT.second.getSizeInBits()) ||
1975 SrcLT.first > 1 || DstLT.first > 1)
1979 assert((SrcLT.first == 1) && (DstLT.first == 1) &&
"Illegal type");
1981 int PowDiff = (int)
Log2_32(DstLT.second.getScalarSizeInBits()) -
1982 (int)
Log2_32(SrcLT.second.getScalarSizeInBits());
1986 if ((PowDiff < 1) || (PowDiff > 3))
1988 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
1989 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
1992 return getRISCVInstructionCost(
Op, DstLT.second,
CostKind);
1998 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
1999 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2003 : RISCV::VFNCVT_F_F_W;
2005 for (; SrcEltSize != DstEltSize;) {
2009 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2011 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2019 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2021 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2023 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2024 unsigned SrcEltSize = Src->getScalarSizeInBits();
2025 unsigned DstEltSize = Dst->getScalarSizeInBits();
2027 if ((SrcEltSize == 16) &&
2028 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2034 std::pair<InstructionCost, MVT> VecF32LT =
2037 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2042 if (DstEltSize == SrcEltSize)
2043 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2044 else if (DstEltSize > SrcEltSize)
2045 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2050 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2051 Cost += getRISCVInstructionCost(FNCVT, VecVT,
CostKind);
2052 if ((SrcEltSize / 2) > DstEltSize) {
2063 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2064 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2065 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2066 unsigned SrcEltSize = Src->getScalarSizeInBits();
2067 unsigned DstEltSize = Dst->getScalarSizeInBits();
2070 if ((DstEltSize == 16) &&
2071 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2077 std::pair<InstructionCost, MVT> VecF32LT =
2080 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2085 if (DstEltSize == SrcEltSize)
2086 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2087 else if (DstEltSize > SrcEltSize) {
2088 if ((DstEltSize / 2) > SrcEltSize) {
2092 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2095 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2097 Cost += getRISCVInstructionCost(FNCVT, DstLT.second,
CostKind);
2104unsigned RISCVTTIImpl::getEstimatedVLFor(
VectorType *Ty)
const {
2106 const unsigned EltSize =
DL.getTypeSizeInBits(Ty->getElementType());
2107 const unsigned MinSize =
DL.getTypeSizeInBits(Ty).getKnownMinValue();
2122 if (Ty->getScalarSizeInBits() > ST->getELen())
2126 if (Ty->getElementType()->isIntegerTy(1)) {
2130 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2136 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2140 case Intrinsic::maximum:
2142 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2144 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2159 case Intrinsic::minimum:
2161 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2163 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2169 const unsigned EltTyBits =
DL.getTypeSizeInBits(DstTy);
2178 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2187 case Intrinsic::smax:
2188 SplitOp = RISCV::VMAX_VV;
2189 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2191 case Intrinsic::smin:
2192 SplitOp = RISCV::VMIN_VV;
2193 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2195 case Intrinsic::umax:
2196 SplitOp = RISCV::VMAXU_VV;
2197 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2199 case Intrinsic::umin:
2200 SplitOp = RISCV::VMINU_VV;
2201 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2203 case Intrinsic::maxnum:
2204 SplitOp = RISCV::VFMAX_VV;
2205 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2207 case Intrinsic::minnum:
2208 SplitOp = RISCV::VFMIN_VV;
2209 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2214 (LT.first > 1) ? (LT.first - 1) *
2215 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2217 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2222 std::optional<FastMathFlags> FMF,
2228 if (Ty->getScalarSizeInBits() > ST->getELen())
2231 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2239 Type *ElementTy = Ty->getElementType();
2244 if (LT.second == MVT::v1i1)
2245 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind) +
2263 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2264 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second,
CostKind) +
2265 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second,
CostKind) +
2266 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2275 return (LT.first - 1) *
2276 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind) +
2277 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) + 1;
2285 return (LT.first - 1) *
2286 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second,
CostKind) +
2287 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2300 SplitOp = RISCV::VADD_VV;
2301 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2304 SplitOp = RISCV::VOR_VV;
2305 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2308 SplitOp = RISCV::VXOR_VV;
2309 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2312 SplitOp = RISCV::VAND_VV;
2313 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2317 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2318 LT.second.getScalarType() == MVT::bf16)
2322 for (
unsigned i = 0; i < LT.first.getValue(); i++)
2325 return getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2327 SplitOp = RISCV::VFADD_VV;
2328 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2333 (LT.first > 1) ? (LT.first - 1) *
2334 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2336 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2340 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *ValTy,
2351 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2357 if (IsUnsigned && Opcode == Instruction::Add &&
2358 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2362 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind);
2369 return (LT.first - 1) +
2376 assert(OpInfo.isConstant() &&
"non constant operand?");
2383 if (OpInfo.isUniform())
2389 return getConstantPoolLoadCost(Ty,
CostKind);
2398 EVT VT = TLI->getValueType(
DL, Src,
true);
2400 if (VT == MVT::Other ||
2406 if (Opcode == Instruction::Store && OpInfo.isConstant())
2421 if (Src->
isVectorTy() && LT.second.isVector() &&
2423 LT.second.getSizeInBits()))
2433 if (ST->hasVInstructions() && LT.second.isVector() &&
2435 BaseCost *= TLI->getLMULCost(LT.second);
2436 return Cost + BaseCost;
2445 Op1Info, Op2Info,
I);
2449 Op1Info, Op2Info,
I);
2454 Op1Info, Op2Info,
I);
2456 auto GetConstantMatCost =
2458 if (OpInfo.isUniform())
2463 return getConstantPoolLoadCost(ValTy,
CostKind);
2468 ConstantMatCost += GetConstantMatCost(Op1Info);
2470 ConstantMatCost += GetConstantMatCost(Op2Info);
2473 if (Opcode == Instruction::Select && LT.second.isVector()) {
2474 if (CondTy->isVectorTy()) {
2479 return ConstantMatCost +
2481 getRISCVInstructionCost(
2482 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2486 return ConstantMatCost +
2487 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2497 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2498 return ConstantMatCost +
2500 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2502 LT.first * getRISCVInstructionCost(
2503 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2510 return ConstantMatCost +
2511 LT.first * getRISCVInstructionCost(
2512 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2516 if ((Opcode == Instruction::ICmp) && ValTy->
isVectorTy() &&
2520 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2525 if ((Opcode == Instruction::FCmp) && ValTy->
isVectorTy() &&
2530 return ConstantMatCost +
2531 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind);
2541 Op1Info, Op2Info,
I);
2550 return ConstantMatCost +
2551 LT.first * getRISCVInstructionCost(
2552 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2559 return ConstantMatCost +
2561 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2570 return ConstantMatCost +
2572 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second,
CostKind);
2585 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2586 U->getType()->isIntegerTy() &&
2587 !isa<ConstantData>(U->getOperand(1)) &&
2588 !isa<ConstantData>(U->getOperand(2));
2596 Op1Info, Op2Info,
I);
2603 return Opcode == Instruction::PHI ? 0 : 1;
2620 if (Opcode != Instruction::ExtractElement &&
2621 Opcode != Instruction::InsertElement)
2629 if (!LT.second.isVector()) {
2638 Type *ElemTy = FixedVecTy->getElementType();
2639 auto NumElems = FixedVecTy->getNumElements();
2640 auto Align =
DL.getPrefTypeAlign(ElemTy);
2645 return Opcode == Instruction::ExtractElement
2646 ? StoreCost * NumElems + LoadCost
2647 : (StoreCost + LoadCost) * NumElems + StoreCost;
2651 if (LT.second.isScalableVector() && !LT.first.isValid())
2659 if (Opcode == Instruction::ExtractElement) {
2665 return ExtendCost + ExtractCost;
2675 return ExtendCost + InsertCost + TruncCost;
2682 if (LT.second.isFloatingPoint())
2683 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2687 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2689 getRISCVInstructionCost(MoveOpc, LT.second,
CostKind);
2691 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2696 if (LT.second.isFixedLengthVector()) {
2697 unsigned Width = LT.second.getVectorNumElements();
2698 Index = Index % Width;
2703 if (
auto VLEN = ST->getRealVLen()) {
2704 unsigned EltSize = LT.second.getScalarSizeInBits();
2705 unsigned M1Max = *VLEN / EltSize;
2706 Index = Index % M1Max;
2712 else if (Opcode == Instruction::InsertElement)
2720 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2721 LT.second.isScalableVector()))) {
2723 Align VecAlign =
DL.getPrefTypeAlign(Val);
2724 Align SclAlign =
DL.getPrefTypeAlign(ScalarType);
2729 if (Opcode == Instruction::ExtractElement)
2765 Opcode == Instruction::InsertElement
2766 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2767 RISCV::VSLIDE1DOWN_VX,
2768 RISCV::VSLIDEUP_VX},
2770 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2771 RISCV::VSRL_VX, RISCV::VMV_X_S},
2774 return BaseCost + SlideCost;
2780 unsigned Index)
const {
2789 assert(Index < EC.getKnownMinValue() &&
"Unexpected reverse index");
2791 EC.getKnownMinValue() - 1 - Index,
nullptr,
2800std::optional<InstructionCost>
2806 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2808 if (Opcode == Instruction::UDiv)
2815 return std::nullopt;
2837 if (std::optional<InstructionCost> CombinedCost =
2839 Op2Info, Args, CxtI))
2840 return *CombinedCost;
2844 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2847 if (!LT.second.isVector()) {
2857 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2858 if (
const auto *Entry =
CostTableLookup(DivTbl, ISDOpcode, LT.second))
2859 return Entry->Cost * LT.first;
2868 if ((LT.second.getVectorElementType() == MVT::f16 ||
2869 LT.second.getVectorElementType() == MVT::bf16) &&
2870 TLI->getOperationAction(ISDOpcode, LT.second) ==
2872 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2876 CastCost += LT.first * Args.size() *
2884 LT.second = PromotedVT;
2887 auto getConstantMatCost =
2897 return getConstantPoolLoadCost(Ty,
CostKind);
2903 ConstantMatCost += getConstantMatCost(0, Op1Info);
2905 ConstantMatCost += getConstantMatCost(1, Op2Info);
2908 switch (ISDOpcode) {
2911 Op = RISCV::VADD_VV;
2916 Op = RISCV::VSLL_VV;
2921 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2926 Op = RISCV::VMUL_VV;
2930 Op = RISCV::VDIV_VV;
2934 Op = RISCV::VREM_VV;
2938 Op = RISCV::VFADD_VV;
2941 Op = RISCV::VFMUL_VV;
2944 Op = RISCV::VFDIV_VV;
2947 Op = RISCV::VFSGNJN_VV;
2952 return CastCost + ConstantMatCost +
2961 if (Ty->isFPOrFPVectorTy())
2963 return CastCost + ConstantMatCost + LT.first *
InstrCost;
2986 if (Info.isSameBase() && V !=
Base) {
2987 if (
GEP->hasAllConstantIndices())
2993 unsigned Stride =
DL.getTypeStoreSize(AccessTy);
2994 if (Info.isUnitStride() &&
3000 GEP->getType()->getPointerAddressSpace()))
3003 {TTI::OK_AnyValue, TTI::OP_None},
3004 {TTI::OK_AnyValue, TTI::OP_None}, {});
3021 if (ST->enableDefaultUnroll())
3031 if (L->getHeader()->getParent()->hasOptSize())
3035 L->getExitingBlocks(ExitingBlocks);
3037 <<
"Blocks: " << L->getNumBlocks() <<
"\n"
3038 <<
"Exit blocks: " << ExitingBlocks.
size() <<
"\n");
3042 if (ExitingBlocks.
size() > 2)
3047 if (L->getNumBlocks() > 4)
3055 for (
auto *BB : L->getBlocks()) {
3056 for (
auto &
I : *BB) {
3060 if (IsVectorized && (
I.getType()->isVectorTy() ||
3062 return V->getType()->isVectorTy();
3103 bool HasMask =
false;
3106 bool IsWrite) -> int64_t {
3107 if (
auto *TarExtTy =
3109 return TarExtTy->getIntParameter(0);
3115 case Intrinsic::riscv_vle_mask:
3116 case Intrinsic::riscv_vse_mask:
3117 case Intrinsic::riscv_vlseg2_mask:
3118 case Intrinsic::riscv_vlseg3_mask:
3119 case Intrinsic::riscv_vlseg4_mask:
3120 case Intrinsic::riscv_vlseg5_mask:
3121 case Intrinsic::riscv_vlseg6_mask:
3122 case Intrinsic::riscv_vlseg7_mask:
3123 case Intrinsic::riscv_vlseg8_mask:
3124 case Intrinsic::riscv_vsseg2_mask:
3125 case Intrinsic::riscv_vsseg3_mask:
3126 case Intrinsic::riscv_vsseg4_mask:
3127 case Intrinsic::riscv_vsseg5_mask:
3128 case Intrinsic::riscv_vsseg6_mask:
3129 case Intrinsic::riscv_vsseg7_mask:
3130 case Intrinsic::riscv_vsseg8_mask:
3133 case Intrinsic::riscv_vle:
3134 case Intrinsic::riscv_vse:
3135 case Intrinsic::riscv_vlseg2:
3136 case Intrinsic::riscv_vlseg3:
3137 case Intrinsic::riscv_vlseg4:
3138 case Intrinsic::riscv_vlseg5:
3139 case Intrinsic::riscv_vlseg6:
3140 case Intrinsic::riscv_vlseg7:
3141 case Intrinsic::riscv_vlseg8:
3142 case Intrinsic::riscv_vsseg2:
3143 case Intrinsic::riscv_vsseg3:
3144 case Intrinsic::riscv_vsseg4:
3145 case Intrinsic::riscv_vsseg5:
3146 case Intrinsic::riscv_vsseg6:
3147 case Intrinsic::riscv_vsseg7:
3148 case Intrinsic::riscv_vsseg8: {
3165 Ty = TarExtTy->getTypeParameter(0U);
3170 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3171 unsigned VLIndex = RVVIInfo->VLOperand;
3172 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3180 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3183 unsigned ElemSize = Ty->getScalarSizeInBits();
3187 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3188 Alignment, Mask, EVL);
3191 case Intrinsic::riscv_vlse_mask:
3192 case Intrinsic::riscv_vsse_mask:
3193 case Intrinsic::riscv_vlsseg2_mask:
3194 case Intrinsic::riscv_vlsseg3_mask:
3195 case Intrinsic::riscv_vlsseg4_mask:
3196 case Intrinsic::riscv_vlsseg5_mask:
3197 case Intrinsic::riscv_vlsseg6_mask:
3198 case Intrinsic::riscv_vlsseg7_mask:
3199 case Intrinsic::riscv_vlsseg8_mask:
3200 case Intrinsic::riscv_vssseg2_mask:
3201 case Intrinsic::riscv_vssseg3_mask:
3202 case Intrinsic::riscv_vssseg4_mask:
3203 case Intrinsic::riscv_vssseg5_mask:
3204 case Intrinsic::riscv_vssseg6_mask:
3205 case Intrinsic::riscv_vssseg7_mask:
3206 case Intrinsic::riscv_vssseg8_mask:
3209 case Intrinsic::riscv_vlse:
3210 case Intrinsic::riscv_vsse:
3211 case Intrinsic::riscv_vlsseg2:
3212 case Intrinsic::riscv_vlsseg3:
3213 case Intrinsic::riscv_vlsseg4:
3214 case Intrinsic::riscv_vlsseg5:
3215 case Intrinsic::riscv_vlsseg6:
3216 case Intrinsic::riscv_vlsseg7:
3217 case Intrinsic::riscv_vlsseg8:
3218 case Intrinsic::riscv_vssseg2:
3219 case Intrinsic::riscv_vssseg3:
3220 case Intrinsic::riscv_vssseg4:
3221 case Intrinsic::riscv_vssseg5:
3222 case Intrinsic::riscv_vssseg6:
3223 case Intrinsic::riscv_vssseg7:
3224 case Intrinsic::riscv_vssseg8: {
3241 Ty = TarExtTy->getTypeParameter(0U);
3246 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3247 unsigned VLIndex = RVVIInfo->VLOperand;
3248 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3257 unsigned PointerAlign = Alignment.valueOrOne().value();
3260 Alignment =
Align(1);
3267 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3270 unsigned ElemSize = Ty->getScalarSizeInBits();
3274 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3275 Alignment, Mask, EVL, Stride);
3278 case Intrinsic::riscv_vloxei_mask:
3279 case Intrinsic::riscv_vluxei_mask:
3280 case Intrinsic::riscv_vsoxei_mask:
3281 case Intrinsic::riscv_vsuxei_mask:
3282 case Intrinsic::riscv_vloxseg2_mask:
3283 case Intrinsic::riscv_vloxseg3_mask:
3284 case Intrinsic::riscv_vloxseg4_mask:
3285 case Intrinsic::riscv_vloxseg5_mask:
3286 case Intrinsic::riscv_vloxseg6_mask:
3287 case Intrinsic::riscv_vloxseg7_mask:
3288 case Intrinsic::riscv_vloxseg8_mask:
3289 case Intrinsic::riscv_vluxseg2_mask:
3290 case Intrinsic::riscv_vluxseg3_mask:
3291 case Intrinsic::riscv_vluxseg4_mask:
3292 case Intrinsic::riscv_vluxseg5_mask:
3293 case Intrinsic::riscv_vluxseg6_mask:
3294 case Intrinsic::riscv_vluxseg7_mask:
3295 case Intrinsic::riscv_vluxseg8_mask:
3296 case Intrinsic::riscv_vsoxseg2_mask:
3297 case Intrinsic::riscv_vsoxseg3_mask:
3298 case Intrinsic::riscv_vsoxseg4_mask:
3299 case Intrinsic::riscv_vsoxseg5_mask:
3300 case Intrinsic::riscv_vsoxseg6_mask:
3301 case Intrinsic::riscv_vsoxseg7_mask:
3302 case Intrinsic::riscv_vsoxseg8_mask:
3303 case Intrinsic::riscv_vsuxseg2_mask:
3304 case Intrinsic::riscv_vsuxseg3_mask:
3305 case Intrinsic::riscv_vsuxseg4_mask:
3306 case Intrinsic::riscv_vsuxseg5_mask:
3307 case Intrinsic::riscv_vsuxseg6_mask:
3308 case Intrinsic::riscv_vsuxseg7_mask:
3309 case Intrinsic::riscv_vsuxseg8_mask:
3312 case Intrinsic::riscv_vloxei:
3313 case Intrinsic::riscv_vluxei:
3314 case Intrinsic::riscv_vsoxei:
3315 case Intrinsic::riscv_vsuxei:
3316 case Intrinsic::riscv_vloxseg2:
3317 case Intrinsic::riscv_vloxseg3:
3318 case Intrinsic::riscv_vloxseg4:
3319 case Intrinsic::riscv_vloxseg5:
3320 case Intrinsic::riscv_vloxseg6:
3321 case Intrinsic::riscv_vloxseg7:
3322 case Intrinsic::riscv_vloxseg8:
3323 case Intrinsic::riscv_vluxseg2:
3324 case Intrinsic::riscv_vluxseg3:
3325 case Intrinsic::riscv_vluxseg4:
3326 case Intrinsic::riscv_vluxseg5:
3327 case Intrinsic::riscv_vluxseg6:
3328 case Intrinsic::riscv_vluxseg7:
3329 case Intrinsic::riscv_vluxseg8:
3330 case Intrinsic::riscv_vsoxseg2:
3331 case Intrinsic::riscv_vsoxseg3:
3332 case Intrinsic::riscv_vsoxseg4:
3333 case Intrinsic::riscv_vsoxseg5:
3334 case Intrinsic::riscv_vsoxseg6:
3335 case Intrinsic::riscv_vsoxseg7:
3336 case Intrinsic::riscv_vsoxseg8:
3337 case Intrinsic::riscv_vsuxseg2:
3338 case Intrinsic::riscv_vsuxseg3:
3339 case Intrinsic::riscv_vsuxseg4:
3340 case Intrinsic::riscv_vsuxseg5:
3341 case Intrinsic::riscv_vsuxseg6:
3342 case Intrinsic::riscv_vsuxseg7:
3343 case Intrinsic::riscv_vsuxseg8: {
3360 Ty = TarExtTy->getTypeParameter(0U);
3365 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3366 unsigned VLIndex = RVVIInfo->VLOperand;
3367 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3380 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3383 unsigned ElemSize = Ty->getScalarSizeInBits();
3388 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3389 Align(1), Mask, EVL,
3398 if (Ty->isVectorTy()) {
3401 if ((EltTy->
isHalfTy() && !ST->hasVInstructionsF16()) ||
3407 if (
Size.isScalable() && ST->hasVInstructions())
3410 if (ST->useRVVForFixedLengthVectors())
3430 return std::max<unsigned>(1U, RegWidth.
getFixedValue() / ElemWidth);
3438 return ST->enableUnalignedVectorMem();
3444 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3466 Align Alignment)
const {
3468 if (!VTy || VTy->isScalableTy())
3476 if (VTy->getElementType()->isIntegerTy(8))
3477 if (VTy->getElementCount().getFixedValue() > 256)
3478 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3479 ST->getMaxLMULForFixedLengthVectors();
3484 Align Alignment)
const {
3486 if (!VTy || VTy->isScalableTy())
3497 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3500 return TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, ElementTy));
3509 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
3510 bool Considerable =
false;
3511 AllowPromotionWithoutCommonHeader =
false;
3514 Type *ConsideredSExtType =
3516 if (
I.getType() != ConsideredSExtType)
3520 for (
const User *U :
I.users()) {
3522 Considerable =
true;
3526 if (GEPInst->getNumOperands() > 2) {
3527 AllowPromotionWithoutCommonHeader =
true;
3532 return Considerable;
3537 case Instruction::Add:
3538 case Instruction::Sub:
3539 case Instruction::Mul:
3540 case Instruction::And:
3541 case Instruction::Or:
3542 case Instruction::Xor:
3543 case Instruction::FAdd:
3544 case Instruction::FSub:
3545 case Instruction::FMul:
3546 case Instruction::FDiv:
3547 case Instruction::ICmp:
3548 case Instruction::FCmp:
3550 case Instruction::Shl:
3551 case Instruction::LShr:
3552 case Instruction::AShr:
3553 case Instruction::UDiv:
3554 case Instruction::SDiv:
3555 case Instruction::URem:
3556 case Instruction::SRem:
3557 case Instruction::Select:
3558 return Operand == 1;
3565 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3575 switch (
II->getIntrinsicID()) {
3576 case Intrinsic::fma:
3577 case Intrinsic::fmuladd:
3578 return Operand == 0 || Operand == 1;
3579 case Intrinsic::vp_udiv:
3580 case Intrinsic::vp_sdiv:
3581 case Intrinsic::vp_urem:
3582 case Intrinsic::vp_srem:
3583 case Intrinsic::ssub_sat:
3584 case Intrinsic::usub_sat:
3585 return Operand == 1;
3587 case Intrinsic::smin:
3588 case Intrinsic::umin:
3589 case Intrinsic::smax:
3590 case Intrinsic::umax:
3591 case Intrinsic::sadd_sat:
3592 case Intrinsic::uadd_sat:
3593 return Operand == 0 || Operand == 1;
3606 if (
I->isBitwiseLogicOp()) {
3607 if (!
I->getType()->isVectorTy()) {
3608 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3609 for (
auto &
Op :
I->operands()) {
3617 }
else if (
I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3618 for (
auto &
Op :
I->operands()) {
3630 Ops.push_back(&Not);
3631 Ops.push_back(&InsertElt);
3639 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3647 if (!ST->sinkSplatOperands())
3650 for (
auto OpIdx :
enumerate(
I->operands())) {
3670 for (
Use &U :
Op->uses()) {
3677 Use *InsertEltUse = &
Op->getOperandUse(0);
3680 Ops.push_back(&InsertElt->getOperandUse(1));
3681 Ops.push_back(InsertEltUse);
3682 Ops.push_back(&OpIdx.value());
3691 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3694 Options.AllowOverlappingLoads =
true;
3695 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3697 if (ST->is64Bit()) {
3698 Options.LoadSizes = {8, 4, 2, 1};
3699 Options.AllowedTailExpansions = {3, 5, 6};
3701 Options.LoadSizes = {4, 2, 1};
3702 Options.AllowedTailExpansions = {3};
3705 if (IsZeroCmp && ST->hasVInstructions()) {
3706 unsigned VLenB = ST->getRealMinVLen() / 8;
3709 unsigned MinSize = ST->getXLen() / 8 + 1;
3710 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3724 if (
I->getOpcode() == Instruction::Or &&
3728 if (
I->getOpcode() == Instruction::Add ||
3729 I->getOpcode() == Instruction::Sub)
3747std::optional<Instruction *>
3753 if (
is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3754 II.getIntrinsicID())) {
3757 if (!ST->hasVInstructions())
3760 bool HasAVL =
II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3761 unsigned Offset = HasAVL ? 1 : 0;
3762 unsigned BitWidth =
II.getType()->getIntegerBitWidth();
3787 Value *AVL =
II.getArgOperand(0);
3816 II.getRange().value_or(ConstantRange::getFull(
BitWidth));
3818 if (NewRange != OldRange) {
3819 II.addRangeRetAttr(NewRange);
3829 if (
II.user_empty())
3834 const APInt *Scalar;
3839 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3843 unsigned TargetEltBW =
DL.getTypeSizeInBits(TargetVecTy->getElementType());
3844 unsigned SourceEltBW =
DL.getTypeSizeInBits(SourceVecTy->getElementType());
3845 if (TargetEltBW % SourceEltBW)
3847 unsigned TargetScale = TargetEltBW / SourceEltBW;
3848 if (VL % TargetScale || TargetScale == 1)
3850 Type *VLTy =
II.getOperand(2)->getType();
3851 ElementCount SourceEC = SourceVecTy->getElementCount();
3852 unsigned NewEltBW = SourceEltBW * TargetScale;
3854 !
DL.fitsInLegalInteger(NewEltBW))
3857 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, NewEltTy)))
3861 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3862 "Lossless bitcast between types expected");
3868 RetTy, Intrinsic::riscv_vmv_v_x,
3869 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3870 ConstantInt::get(VLTy, VL / TargetScale)}),
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
This file describes how to lower LLVM code to machine code.
Class for arbitrary precision integers.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
size_t size() const
Get the array size.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
@ ICMP_SLT
signed less than
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Convenience struct for specifying and reasoning about fast-math flags.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
Intrinsic::ID getID() const
bool isTypeBasedOnly() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Represents a single loop in the control flow graph.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
Align getAlignment() const
unsigned getAddressSpace() const
Type * getDataType() const
bool getVariableMask() const
Intrinsic::ID getID() const
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
LLVM_ABI bool isScalableTy(SmallPtrSetImpl< const Type * > &Visited) const
Return true if this is a type whose size is a known multiple of vscale.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ BinaryOp
One of the operands is a binary op.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const