24#include "llvm/IR/IntrinsicsAArch64.h"
36#define DEBUG_TYPE "aarch64tti"
42 "sve-prefer-fixed-over-scalable-if-equal",
cl::Hidden);
60 "Penalty of calling a function that requires a change to PSTATE.SM"));
64 cl::desc(
"Penalty of inlining a call that requires a change to PSTATE.SM"));
75 cl::desc(
"The cost of a histcnt instruction"));
79 cl::desc(
"The number of instructions to search for a redundant dmb"));
83 cl::desc(
"Threshold for forced unrolling of small loops in AArch64"));
86class TailFoldingOption {
101 bool NeedsDefault =
true;
105 void setNeedsDefault(
bool V) { NeedsDefault =
V; }
120 assert((InitialBits == TailFoldingOpts::Disabled || !NeedsDefault) &&
121 "Initial bits should only include one of "
122 "(disabled|all|simple|default)");
123 Bits = NeedsDefault ? DefaultBits : InitialBits;
125 Bits &= ~DisableBits;
131 errs() <<
"invalid argument '" << Opt
132 <<
"' to -sve-tail-folding=; the option should be of the form\n"
133 " (disabled|all|default|simple)[+(reductions|recurrences"
134 "|reverse|noreductions|norecurrences|noreverse)]\n";
140 void operator=(
const std::string &Val) {
149 setNeedsDefault(
false);
152 StringRef(Val).split(TailFoldTypes,
'+', -1,
false);
154 unsigned StartIdx = 1;
155 if (TailFoldTypes[0] ==
"disabled")
156 setInitialBits(TailFoldingOpts::Disabled);
157 else if (TailFoldTypes[0] ==
"all")
158 setInitialBits(TailFoldingOpts::All);
159 else if (TailFoldTypes[0] ==
"default")
160 setNeedsDefault(
true);
161 else if (TailFoldTypes[0] ==
"simple")
162 setInitialBits(TailFoldingOpts::Simple);
165 setInitialBits(TailFoldingOpts::Disabled);
168 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
169 if (TailFoldTypes[
I] ==
"reductions")
170 setEnableBit(TailFoldingOpts::Reductions);
171 else if (TailFoldTypes[
I] ==
"recurrences")
172 setEnableBit(TailFoldingOpts::Recurrences);
173 else if (TailFoldTypes[
I] ==
"reverse")
174 setEnableBit(TailFoldingOpts::Reverse);
175 else if (TailFoldTypes[
I] ==
"noreductions")
176 setDisableBit(TailFoldingOpts::Reductions);
177 else if (TailFoldTypes[
I] ==
"norecurrences")
178 setDisableBit(TailFoldingOpts::Recurrences);
179 else if (TailFoldTypes[
I] ==
"noreverse")
180 setDisableBit(TailFoldingOpts::Reverse);
197 "Control the use of vectorisation using tail-folding for SVE where the"
198 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
199 "\ndisabled (Initial) No loop types will vectorize using "
201 "\ndefault (Initial) Uses the default tail-folding settings for "
203 "\nall (Initial) All legal loop types will vectorize using "
205 "\nsimple (Initial) Use tail-folding for simple loops (not "
206 "reductions or recurrences)"
207 "\nreductions Use tail-folding for loops containing reductions"
208 "\nnoreductions Inverse of above"
209 "\nrecurrences Use tail-folding for loops containing fixed order "
211 "\nnorecurrences Inverse of above"
212 "\nreverse Use tail-folding for loops requiring reversed "
214 "\nnoreverse Inverse of above"),
259 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
260 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
261 FeatureStr.
split(Features,
",");
277 return F.hasFnAttribute(
"fmv-features");
287 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
288 CallAttrs.callee().hasStreamingInterfaceOrBody())
293 if (
CallAttrs.callee().hasStreamingBody()) {
303 CallAttrs.requiresPreservingAllZAState()) {
326 auto FVTy = dyn_cast<FixedVectorType>(Ty);
328 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
337 unsigned DefaultCallPenalty)
const {
362 if (
F ==
Call.getCaller())
368 return DefaultCallPenalty;
379 ST->isSVEorStreamingSVEAvailable() &&
380 !ST->disableMaximizeScalableBandwidth();
404 assert(Ty->isIntegerTy());
406 unsigned BitSize = Ty->getPrimitiveSizeInBits();
413 ImmVal =
Imm.sext((BitSize + 63) & ~0x3fU);
418 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
424 return std::max<InstructionCost>(1,
Cost);
431 assert(Ty->isIntegerTy());
433 unsigned BitSize = Ty->getPrimitiveSizeInBits();
439 unsigned ImmIdx = ~0U;
443 case Instruction::GetElementPtr:
448 case Instruction::Store:
451 case Instruction::Add:
452 case Instruction::Sub:
453 case Instruction::Mul:
454 case Instruction::UDiv:
455 case Instruction::SDiv:
456 case Instruction::URem:
457 case Instruction::SRem:
458 case Instruction::And:
459 case Instruction::Or:
460 case Instruction::Xor:
461 case Instruction::ICmp:
465 case Instruction::Shl:
466 case Instruction::LShr:
467 case Instruction::AShr:
471 case Instruction::Trunc:
472 case Instruction::ZExt:
473 case Instruction::SExt:
474 case Instruction::IntToPtr:
475 case Instruction::PtrToInt:
476 case Instruction::BitCast:
477 case Instruction::PHI:
478 case Instruction::Call:
479 case Instruction::Select:
480 case Instruction::Ret:
481 case Instruction::Load:
486 int NumConstants = (BitSize + 63) / 64;
499 assert(Ty->isIntegerTy());
501 unsigned BitSize = Ty->getPrimitiveSizeInBits();
510 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
516 case Intrinsic::sadd_with_overflow:
517 case Intrinsic::uadd_with_overflow:
518 case Intrinsic::ssub_with_overflow:
519 case Intrinsic::usub_with_overflow:
520 case Intrinsic::smul_with_overflow:
521 case Intrinsic::umul_with_overflow:
523 int NumConstants = (BitSize + 63) / 64;
530 case Intrinsic::experimental_stackmap:
531 if ((Idx < 2) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
534 case Intrinsic::experimental_patchpoint_void:
535 case Intrinsic::experimental_patchpoint:
536 if ((Idx < 4) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
539 case Intrinsic::experimental_gc_statepoint:
540 if ((Idx < 5) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
550 if (TyWidth == 32 || TyWidth == 64)
559 return ST->getMispredictionPenalty();
580 unsigned TotalHistCnts = 1;
590 unsigned EC = VTy->getElementCount().getKnownMinValue();
595 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
597 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
601 TotalHistCnts = EC / NaturalVectorWidth;
621 switch (ICA.
getID()) {
622 case Intrinsic::experimental_vector_histogram_add: {
629 case Intrinsic::clmul: {
634 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
638 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
643 -1,
nullptr,
nullptr) *
646 -1,
nullptr,
nullptr);
650 if (LT.second.SimpleTy == MVT::nxv2i64)
651 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
654 if (ST->hasSVE2() || ST->hasSME()) {
655 switch (LT.second.SimpleTy) {
670 if (LT.second.SimpleTy == MVT::nxv2i64)
674 switch (LT.second.SimpleTy) {
684 -1,
nullptr,
nullptr) *
687 -1,
nullptr,
nullptr));
696 return LT.first * 11;
698 return LT.first * 14;
705 case Intrinsic::umin:
706 case Intrinsic::umax:
707 case Intrinsic::smin:
708 case Intrinsic::smax: {
709 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
710 MVT::v8i16, MVT::v2i32, MVT::v4i32,
711 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
718 ICA.
getID() == Intrinsic::smin || ICA.
getID() == Intrinsic::smax;
719 EVT VT = TLI->getValueType(
DL, RetTy,
true);
720 if (VT == MVT::v2i8 || VT == MVT::v2i16 || VT == MVT::v4i8)
721 return LT.first * (IsSigned ? 5 : 3);
723 if (LT.second == MVT::v2i64)
729 case Intrinsic::scmp:
730 case Intrinsic::ucmp: {
732 {Intrinsic::scmp, MVT::i32, 3},
733 {Intrinsic::scmp, MVT::i64, 3},
734 {Intrinsic::scmp, MVT::v8i8, 3},
735 {Intrinsic::scmp, MVT::v16i8, 3},
736 {Intrinsic::scmp, MVT::v4i16, 3},
737 {Intrinsic::scmp, MVT::v8i16, 3},
738 {Intrinsic::scmp, MVT::v2i32, 3},
739 {Intrinsic::scmp, MVT::v4i32, 3},
740 {Intrinsic::scmp, MVT::v1i64, 3},
741 {Intrinsic::scmp, MVT::v2i64, 3},
747 return Entry->Cost * LT.first;
750 case Intrinsic::sadd_sat:
751 case Intrinsic::ssub_sat:
752 case Intrinsic::uadd_sat:
753 case Intrinsic::usub_sat: {
754 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
755 MVT::v8i16, MVT::v2i32, MVT::v4i32,
761 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
763 return LT.first * Instrs;
768 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
769 return LT.first * Instrs;
773 case Intrinsic::abs: {
774 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
775 MVT::v8i16, MVT::v2i32, MVT::v4i32,
776 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
777 MVT::nxv4i32, MVT::nxv2i64};
783 case Intrinsic::bswap: {
784 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
785 MVT::v4i32, MVT::v2i64};
788 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
793 case Intrinsic::fmuladd: {
798 (EltTy->
isHalfTy() && ST->hasFullFP16()))
802 case Intrinsic::stepvector: {
811 Cost += AddCost * (LT.first - 1);
815 case Intrinsic::vector_extract:
816 case Intrinsic::vector_insert: {
829 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
830 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
838 getTLI()->getTypeConversion(
C, SubVecVT);
840 getTLI()->getTypeConversion(
C, VecVT);
848 case Intrinsic::bitreverse: {
850 {Intrinsic::bitreverse, MVT::i32, 1},
851 {Intrinsic::bitreverse, MVT::i64, 1},
852 {Intrinsic::bitreverse, MVT::v8i8, 1},
853 {Intrinsic::bitreverse, MVT::v16i8, 1},
854 {Intrinsic::bitreverse, MVT::v4i16, 2},
855 {Intrinsic::bitreverse, MVT::v8i16, 2},
856 {Intrinsic::bitreverse, MVT::v2i32, 2},
857 {Intrinsic::bitreverse, MVT::v4i32, 2},
858 {Intrinsic::bitreverse, MVT::v1i64, 2},
859 {Intrinsic::bitreverse, MVT::v2i64, 2},
867 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
868 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
869 return LegalisationCost.first * Entry->Cost + 1;
871 return LegalisationCost.first * Entry->Cost;
875 case Intrinsic::ctpop: {
879 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
882 return LT.first + ExtraCost;
884 if (!ST->hasNEON()) {
914 RetTy->getScalarSizeInBits()
917 return LT.first * Entry->Cost + ExtraCost;
921 case Intrinsic::sadd_with_overflow:
922 case Intrinsic::uadd_with_overflow:
923 case Intrinsic::ssub_with_overflow:
924 case Intrinsic::usub_with_overflow:
925 case Intrinsic::smul_with_overflow:
926 case Intrinsic::umul_with_overflow: {
928 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
929 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
930 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
931 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
932 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
933 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
934 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
935 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
936 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
937 {Intrinsic::usub_with_overflow, MVT::i8, 3},
938 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
939 {Intrinsic::usub_with_overflow, MVT::i16, 3},
940 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
941 {Intrinsic::usub_with_overflow, MVT::i32, 1},
942 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
943 {Intrinsic::usub_with_overflow, MVT::i64, 1},
944 {Intrinsic::smul_with_overflow, MVT::i8, 5},
945 {Intrinsic::umul_with_overflow, MVT::i8, 4},
946 {Intrinsic::smul_with_overflow, MVT::i16, 5},
947 {Intrinsic::umul_with_overflow, MVT::i16, 4},
948 {Intrinsic::smul_with_overflow, MVT::i32, 2},
949 {Intrinsic::umul_with_overflow, MVT::i32, 2},
950 {Intrinsic::smul_with_overflow, MVT::i64, 3},
951 {Intrinsic::umul_with_overflow, MVT::i64, 3},
953 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
960 case Intrinsic::fptosi_sat:
961 case Intrinsic::fptoui_sat: {
964 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
966 EVT MTy = TLI->getValueType(
DL, RetTy);
969 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
970 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
971 LT.second == MVT::v2f64)) {
973 (LT.second == MVT::f64 && MTy == MVT::i32) ||
974 (LT.second == MVT::f32 && MTy == MVT::i64)))
983 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
990 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
991 (LT.second == MVT::f16 && MTy == MVT::i64) ||
992 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
1006 if ((LT.second.getScalarType() == MVT::f32 ||
1007 LT.second.getScalarType() == MVT::f64 ||
1008 LT.second.getScalarType() == MVT::f16) &&
1011 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1012 if (LT.second.isVector())
1013 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1017 LegalTy, {LegalTy, LegalTy});
1021 LegalTy, {LegalTy, LegalTy});
1023 return LT.first *
Cost +
1024 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1030 RetTy = RetTy->getScalarType();
1031 if (LT.second.isVector()) {
1049 return LT.first *
Cost;
1051 case Intrinsic::fshl:
1052 case Intrinsic::fshr: {
1061 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1062 (RetTy->getPrimitiveSizeInBits() == 32 ||
1063 RetTy->getPrimitiveSizeInBits() == 64)) {
1076 {Intrinsic::fshl, MVT::v4i32, 2},
1077 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1078 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1079 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1085 return LegalisationCost.first * Entry->Cost;
1089 if (!RetTy->isIntegerTy())
1094 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1095 RetTy->getScalarSizeInBits() < 64) ||
1096 (RetTy->getScalarSizeInBits() % 64 != 0);
1097 unsigned ExtraCost = HigherCost ? 1 : 0;
1098 if (RetTy->getScalarSizeInBits() == 32 ||
1099 RetTy->getScalarSizeInBits() == 64)
1102 else if (HigherCost)
1106 return TyL.first + ExtraCost;
1108 case Intrinsic::get_active_lane_mask: {
1110 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1112 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1115 if (RetTy->isScalableTy()) {
1116 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1126 if (ST->hasSVE2p1() || ST->hasSME2()) {
1138 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1141 return Cost + (SplitCost * (
Cost - 1));
1156 case Intrinsic::experimental_vector_match: {
1157 if (!ST->hasSVE2() || !ST->isSVEAvailable())
1163 unsigned SearchSize = NeedleTy->getNumElements();
1164 if (SearchSize <= 2)
1169 {MVT::nxv8i16, MVT::nxv16i8, MVT::v8i16, MVT::v16i8, MVT::v8i8},
1173 unsigned ElementSizeInBits = SearchVT.getScalarSizeInBits();
1179 unsigned MatchesRequiredForNeedle =
1191 return Cost * LegalParts * MatchesRequiredForNeedle;
1193 case Intrinsic::cttz: {
1195 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
1196 return LT.first * 2;
1197 if (LT.second == MVT::v4i16 || LT.second == MVT::v8i16 ||
1198 LT.second == MVT::v2i32 || LT.second == MVT::v4i32)
1199 return LT.first * 3;
1202 case Intrinsic::experimental_cttz_elts: {
1212 case Intrinsic::loop_dependence_raw_mask:
1213 case Intrinsic::loop_dependence_war_mask: {
1215 if (ST->hasSVE2() || ST->hasSME()) {
1216 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1217 unsigned EltSizeInBytes =
1227 case Intrinsic::experimental_vector_extract_last_active:
1228 if (ST->isSVEorStreamingSVEAvailable()) {
1234 case Intrinsic::pow: {
1237 EVT VT = getTLI()->getValueType(
DL, RetTy);
1238 RTLIB::Libcall LC = RTLIB::getPOW(VT);
1239 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1254 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1255 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1265 return (Sqrt * 2) +
FMul;
1276 case Intrinsic::sqrt:
1277 case Intrinsic::fabs:
1278 case Intrinsic::ceil:
1279 case Intrinsic::floor:
1280 case Intrinsic::nearbyint:
1281 case Intrinsic::round:
1282 case Intrinsic::rint:
1283 case Intrinsic::roundeven:
1284 case Intrinsic::trunc:
1285 case Intrinsic::minnum:
1286 case Intrinsic::maxnum:
1287 case Intrinsic::minimum:
1288 case Intrinsic::maximum: {
1306 auto RequiredType =
II.getType();
1309 assert(PN &&
"Expected Phi Node!");
1312 if (!PN->hasOneUse())
1313 return std::nullopt;
1315 for (
Value *IncValPhi : PN->incoming_values()) {
1318 Reinterpret->getIntrinsicID() !=
1319 Intrinsic::aarch64_sve_convert_to_svbool ||
1320 RequiredType != Reinterpret->getArgOperand(0)->getType())
1321 return std::nullopt;
1329 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1331 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1404 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1409 return GoverningPredicateIdx;
1414 GoverningPredicateIdx = Index;
1436 return UndefIntrinsic;
1441 UndefIntrinsic = IID;
1468 return CmpPredicate;
1473 CmpPredicate = Pred;
1489 return ResultLanes == InactiveLanesTakenFromOperand;
1494 return OperandIdxForInactiveLanes;
1498 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1499 ResultLanes = InactiveLanesTakenFromOperand;
1500 OperandIdxForInactiveLanes = Index;
1505 return ResultLanes == InactiveLanesAreNotDefined;
1509 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1510 ResultLanes = InactiveLanesAreNotDefined;
1515 return ResultLanes == InactiveLanesAreUnused;
1519 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1520 ResultLanes = InactiveLanesAreUnused;
1530 ResultIsZeroInitialized =
true;
1541 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1546 return OperandIdxWithNoActiveLanes;
1551 OperandIdxWithNoActiveLanes = Index;
1556 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1559 unsigned IROpcode = 0;
1562 enum PredicationStyle {
1564 InactiveLanesTakenFromOperand,
1565 InactiveLanesAreNotDefined,
1566 InactiveLanesAreUnused
1569 bool ResultIsZeroInitialized =
false;
1570 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1571 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1579 return !isa<ScalableVectorType>(V->getType());
1587 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1588 case Intrinsic::aarch64_sve_fcvt_f16f32:
1589 case Intrinsic::aarch64_sve_fcvt_f16f64:
1590 case Intrinsic::aarch64_sve_fcvt_f32f16:
1591 case Intrinsic::aarch64_sve_fcvt_f32f64:
1592 case Intrinsic::aarch64_sve_fcvt_f64f16:
1593 case Intrinsic::aarch64_sve_fcvt_f64f32:
1594 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1595 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1596 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1597 case Intrinsic::aarch64_sve_fcvtzs:
1598 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1599 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1600 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1601 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1602 case Intrinsic::aarch64_sve_fcvtzu:
1603 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1604 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1605 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1606 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1607 case Intrinsic::aarch64_sve_revb:
1608 case Intrinsic::aarch64_sve_revh:
1609 case Intrinsic::aarch64_sve_revw:
1610 case Intrinsic::aarch64_sve_revd:
1611 case Intrinsic::aarch64_sve_scvtf:
1612 case Intrinsic::aarch64_sve_scvtf_f16i32:
1613 case Intrinsic::aarch64_sve_scvtf_f16i64:
1614 case Intrinsic::aarch64_sve_scvtf_f32i64:
1615 case Intrinsic::aarch64_sve_scvtf_f64i32:
1616 case Intrinsic::aarch64_sve_ucvtf:
1617 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1618 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1619 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1620 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1623 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1624 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1625 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1626 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1629 case Intrinsic::aarch64_sve_fabd:
1631 case Intrinsic::aarch64_sve_fadd:
1634 case Intrinsic::aarch64_sve_fdiv:
1637 case Intrinsic::aarch64_sve_fmax:
1639 case Intrinsic::aarch64_sve_fmaxnm:
1641 case Intrinsic::aarch64_sve_fmin:
1643 case Intrinsic::aarch64_sve_fminnm:
1645 case Intrinsic::aarch64_sve_fmla:
1647 case Intrinsic::aarch64_sve_fmls:
1649 case Intrinsic::aarch64_sve_fmul:
1652 case Intrinsic::aarch64_sve_fmulx:
1654 case Intrinsic::aarch64_sve_fnmla:
1656 case Intrinsic::aarch64_sve_fnmls:
1658 case Intrinsic::aarch64_sve_fsub:
1661 case Intrinsic::aarch64_sve_add:
1664 case Intrinsic::aarch64_sve_mla:
1666 case Intrinsic::aarch64_sve_mls:
1668 case Intrinsic::aarch64_sve_mul:
1671 case Intrinsic::aarch64_sve_sabd:
1673 case Intrinsic::aarch64_sve_sdiv:
1676 case Intrinsic::aarch64_sve_smax:
1678 case Intrinsic::aarch64_sve_smin:
1680 case Intrinsic::aarch64_sve_smulh:
1682 case Intrinsic::aarch64_sve_sub:
1685 case Intrinsic::aarch64_sve_uabd:
1687 case Intrinsic::aarch64_sve_udiv:
1690 case Intrinsic::aarch64_sve_umax:
1692 case Intrinsic::aarch64_sve_umin:
1694 case Intrinsic::aarch64_sve_umulh:
1696 case Intrinsic::aarch64_sve_asr:
1699 case Intrinsic::aarch64_sve_lsl:
1702 case Intrinsic::aarch64_sve_lsr:
1705 case Intrinsic::aarch64_sve_and:
1708 case Intrinsic::aarch64_sve_bic:
1710 case Intrinsic::aarch64_sve_eor:
1713 case Intrinsic::aarch64_sve_orr:
1716 case Intrinsic::aarch64_sve_shsub:
1718 case Intrinsic::aarch64_sve_shsubr:
1720 case Intrinsic::aarch64_sve_sqrshl:
1722 case Intrinsic::aarch64_sve_sqshl:
1724 case Intrinsic::aarch64_sve_sqsub:
1726 case Intrinsic::aarch64_sve_srshl:
1728 case Intrinsic::aarch64_sve_uhsub:
1730 case Intrinsic::aarch64_sve_uhsubr:
1732 case Intrinsic::aarch64_sve_uqrshl:
1734 case Intrinsic::aarch64_sve_uqshl:
1736 case Intrinsic::aarch64_sve_uqsub:
1738 case Intrinsic::aarch64_sve_urshl:
1741 case Intrinsic::aarch64_sve_add_u:
1744 case Intrinsic::aarch64_sve_and_u:
1747 case Intrinsic::aarch64_sve_asr_u:
1750 case Intrinsic::aarch64_sve_eor_u:
1753 case Intrinsic::aarch64_sve_fadd_u:
1756 case Intrinsic::aarch64_sve_fdiv_u:
1759 case Intrinsic::aarch64_sve_fmul_u:
1762 case Intrinsic::aarch64_sve_fsub_u:
1765 case Intrinsic::aarch64_sve_lsl_u:
1768 case Intrinsic::aarch64_sve_lsr_u:
1771 case Intrinsic::aarch64_sve_mul_u:
1774 case Intrinsic::aarch64_sve_orr_u:
1777 case Intrinsic::aarch64_sve_sdiv_u:
1780 case Intrinsic::aarch64_sve_sub_u:
1783 case Intrinsic::aarch64_sve_udiv_u:
1787 case Intrinsic::aarch64_sve_addqv:
1788 case Intrinsic::aarch64_sve_bic_z:
1789 case Intrinsic::aarch64_sve_brka_z:
1790 case Intrinsic::aarch64_sve_brkb_z:
1791 case Intrinsic::aarch64_sve_brkn_z:
1792 case Intrinsic::aarch64_sve_brkpa_z:
1793 case Intrinsic::aarch64_sve_brkpb_z:
1794 case Intrinsic::aarch64_sve_cntp:
1795 case Intrinsic::aarch64_sve_compact:
1796 case Intrinsic::aarch64_sve_eorv:
1797 case Intrinsic::aarch64_sve_eorqv:
1798 case Intrinsic::aarch64_sve_nand_z:
1799 case Intrinsic::aarch64_sve_nor_z:
1800 case Intrinsic::aarch64_sve_orn_z:
1801 case Intrinsic::aarch64_sve_orv:
1802 case Intrinsic::aarch64_sve_orqv:
1803 case Intrinsic::aarch64_sve_pnext:
1804 case Intrinsic::aarch64_sve_rdffr_z:
1805 case Intrinsic::aarch64_sve_saddv:
1806 case Intrinsic::aarch64_sve_uaddv:
1807 case Intrinsic::aarch64_sve_umaxv:
1808 case Intrinsic::aarch64_sve_umaxqv:
1809 case Intrinsic::aarch64_sve_facge:
1810 case Intrinsic::aarch64_sve_facgt:
1811 case Intrinsic::aarch64_sve_ld1:
1812 case Intrinsic::aarch64_sve_ld1_gather:
1813 case Intrinsic::aarch64_sve_ld1_gather_index:
1814 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1815 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1816 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1817 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1818 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1819 case Intrinsic::aarch64_sve_ld1q_gather_index:
1820 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1821 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1822 case Intrinsic::aarch64_sve_ld1ro:
1823 case Intrinsic::aarch64_sve_ld1rq:
1824 case Intrinsic::aarch64_sve_ld1udq:
1825 case Intrinsic::aarch64_sve_ld1uwq:
1826 case Intrinsic::aarch64_sve_ld2_sret:
1827 case Intrinsic::aarch64_sve_ld2q_sret:
1828 case Intrinsic::aarch64_sve_ld3_sret:
1829 case Intrinsic::aarch64_sve_ld3q_sret:
1830 case Intrinsic::aarch64_sve_ld4_sret:
1831 case Intrinsic::aarch64_sve_ld4q_sret:
1832 case Intrinsic::aarch64_sve_ldff1:
1833 case Intrinsic::aarch64_sve_ldff1_gather:
1834 case Intrinsic::aarch64_sve_ldff1_gather_index:
1835 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1836 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1837 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1838 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1839 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1840 case Intrinsic::aarch64_sve_ldnf1:
1841 case Intrinsic::aarch64_sve_ldnt1:
1842 case Intrinsic::aarch64_sve_ldnt1_gather:
1843 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1844 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1845 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1848 case Intrinsic::aarch64_sve_and_z:
1851 case Intrinsic::aarch64_sve_orr_z:
1854 case Intrinsic::aarch64_sve_eor_z:
1858 case Intrinsic::aarch64_sve_cmpeq:
1859 case Intrinsic::aarch64_sve_cmpeq_wide:
1862 case Intrinsic::aarch64_sve_cmpge:
1863 case Intrinsic::aarch64_sve_cmpge_wide:
1866 case Intrinsic::aarch64_sve_cmpgt:
1867 case Intrinsic::aarch64_sve_cmpgt_wide:
1870 case Intrinsic::aarch64_sve_cmphi:
1871 case Intrinsic::aarch64_sve_cmphi_wide:
1874 case Intrinsic::aarch64_sve_cmphs:
1875 case Intrinsic::aarch64_sve_cmphs_wide:
1878 case Intrinsic::aarch64_sve_cmple_wide:
1881 case Intrinsic::aarch64_sve_cmplo_wide:
1884 case Intrinsic::aarch64_sve_cmpls_wide:
1887 case Intrinsic::aarch64_sve_cmplt_wide:
1890 case Intrinsic::aarch64_sve_cmpne:
1891 case Intrinsic::aarch64_sve_cmpne_wide:
1894 case Intrinsic::aarch64_sve_fcmpeq:
1897 case Intrinsic::aarch64_sve_fcmpge:
1900 case Intrinsic::aarch64_sve_fcmpgt:
1903 case Intrinsic::aarch64_sve_fcmpne:
1906 case Intrinsic::aarch64_sve_fcmpuo:
1910 case Intrinsic::aarch64_sve_prf:
1911 case Intrinsic::aarch64_sve_prfb_gather_index:
1912 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
1913 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
1914 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
1915 case Intrinsic::aarch64_sve_prfd_gather_index:
1916 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
1917 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
1918 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
1919 case Intrinsic::aarch64_sve_prfh_gather_index:
1920 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
1921 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
1922 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
1923 case Intrinsic::aarch64_sve_prfw_gather_index:
1924 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
1925 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
1926 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
1929 case Intrinsic::aarch64_sve_st1_scatter:
1930 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
1931 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
1932 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
1933 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
1934 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
1935 case Intrinsic::aarch64_sve_st1dq:
1936 case Intrinsic::aarch64_sve_st1q_scatter_index:
1937 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
1938 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
1939 case Intrinsic::aarch64_sve_st1wq:
1940 case Intrinsic::aarch64_sve_stnt1:
1941 case Intrinsic::aarch64_sve_stnt1_scatter:
1942 case Intrinsic::aarch64_sve_stnt1_scatter_index:
1943 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
1944 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
1946 case Intrinsic::aarch64_sve_st2:
1947 case Intrinsic::aarch64_sve_st2q:
1949 case Intrinsic::aarch64_sve_st3:
1950 case Intrinsic::aarch64_sve_st3q:
1952 case Intrinsic::aarch64_sve_st4:
1953 case Intrinsic::aarch64_sve_st4q:
1961 Value *UncastedPred;
1967 Pred = UncastedPred;
1973 if (OrigPredTy->getMinNumElements() <=
1975 ->getMinNumElements())
1976 Pred = UncastedPred;
1980 return C &&
C->isAllOnesValue();
1987 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
1988 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
1996static std::optional<Instruction *>
2003 Value *Op1 =
II.getOperand(1);
2004 Value *Op2 =
II.getOperand(2);
2029 Value *NarrowOp1, *NarrowOp2;
2040 else if (SimpleNarrow == NarrowOp1)
2042 else if (SimpleNarrow == NarrowOp2)
2047 SimpleNarrow->
getType(), SimpleNarrow);
2056 return std::nullopt;
2067 if (SimpleII == Inactive)
2075static std::optional<Instruction *>
2079 assert((
Opc == Instruction::ICmp ||
Opc == Instruction::FCmp) &&
2080 "Expected a compare operation!");
2087 Opc == Instruction::ICmp &&
LHS->getType() !=
RHS->getType();
2088 assert((IsWideICmp ||
LHS->getType() ==
RHS->getType()) &&
2089 "Unexpected wide compare!");
2105 const APInt *LHSVal, *RHSVal;
2107 return std::nullopt;
2130 return std::nullopt;
2144static std::optional<Instruction *>
2148 return std::nullopt;
2177 II.setCalledFunction(NewDecl);
2183 return std::nullopt;
2194 if (
Opc == Instruction::FCmp ||
Opc == Instruction::ICmp)
2197 return std::nullopt;
2209static std::optional<Instruction *>
2211 auto m_ConvertToSVBool = [](
auto P) {
2215 Intrinsic::aarch64_sve_convert_from_svbool;
2238 return std::nullopt;
2242 case Intrinsic::aarch64_sve_and_z:
2243 case Intrinsic::aarch64_sve_bic_z:
2244 case Intrinsic::aarch64_sve_eor_z:
2245 case Intrinsic::aarch64_sve_nand_z:
2246 case Intrinsic::aarch64_sve_nor_z:
2247 case Intrinsic::aarch64_sve_orn_z:
2248 case Intrinsic::aarch64_sve_orr_z:
2251 return std::nullopt;
2254 Value *BinOpPred = BinOp->getOperand(0);
2255 Value *BinOpOp1 = BinOp->getOperand(1);
2256 Value *BinOpOp2 = BinOp->getOperand(2);
2258 Value *NarrowBinOpPred;
2260 return std::nullopt;
2262 Value *NarrowBinOpOp1 =
2264 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2265 if (BinOpOp1 != BinOpOp2)
2269 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2273static std::optional<Instruction *>
2280 return BinOpCombine;
2285 return std::nullopt;
2288 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2297 if (CursorVTy->getElementCount().getKnownMinValue() <
2298 IVTy->getElementCount().getKnownMinValue())
2302 if (Cursor->getType() == IVTy)
2303 EarliestReplacement = Cursor;
2308 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2309 Intrinsic::aarch64_sve_convert_to_svbool ||
2310 IntrinsicCursor->getIntrinsicID() ==
2311 Intrinsic::aarch64_sve_convert_from_svbool))
2314 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2315 Cursor = IntrinsicCursor->getOperand(0);
2320 if (!EarliestReplacement)
2321 return std::nullopt;
2329 auto *OpPredicate =
II.getOperand(0);
2346 II.getArgOperand(2));
2352 return std::nullopt;
2356 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2365 II.getArgOperand(0));
2374 if (!
II.hasOneUse())
2375 return std::nullopt;
2378 return std::nullopt;
2381 switch (
II.getIntrinsicID()) {
2382 case Intrinsic::aarch64_sve_cmpne:
2383 IID = Intrinsic::aarch64_sve_cmpeq;
2385 case Intrinsic::aarch64_sve_cmpne_wide:
2386 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2388 case Intrinsic::aarch64_sve_cmpeq:
2389 IID = Intrinsic::aarch64_sve_cmpne;
2391 case Intrinsic::aarch64_sve_cmpeq_wide:
2392 IID = Intrinsic::aarch64_sve_cmpne_wide;
2395 return std::nullopt;
2400 IID,
II.getOperand(1)->getType(),
2401 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2413 return std::nullopt;
2415 for (
auto *U :
II.users()) {
2418 Type *Ty =
II.getOperand(1)->getType();
2423 Intrinsic::aarch64_sve_umin, Ty,
2424 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2430 return std::nullopt;
2444 return std::nullopt;
2449 if (!SplatValue || !SplatValue->isZero())
2450 return std::nullopt;
2455 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2456 return std::nullopt;
2460 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2461 return std::nullopt;
2464 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2465 return std::nullopt;
2470 return std::nullopt;
2473 return std::nullopt;
2477 return std::nullopt;
2481 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2482 return std::nullopt;
2484 unsigned NumElts = VecTy->getNumElements();
2485 unsigned PredicateBits = 0;
2488 for (
unsigned I = 0;
I < NumElts; ++
I) {
2491 return std::nullopt;
2493 PredicateBits |= 1 << (
I * (16 / NumElts));
2497 if (PredicateBits == 0) {
2499 PFalse->takeName(&
II);
2505 for (
unsigned I = 0;
I < 16; ++
I)
2506 if ((PredicateBits & (1 <<
I)) != 0)
2509 unsigned PredSize = Mask & -Mask;
2514 for (
unsigned I = 0;
I < 16;
I += PredSize)
2515 if ((PredicateBits & (1 <<
I)) == 0)
2516 return std::nullopt;
2518 auto *ConvertToSVBool =
2521 auto *ConvertFromSVBool =
2523 II.getType(), ConvertToSVBool);
2531 Value *Pg =
II.getArgOperand(0);
2532 Value *Vec =
II.getArgOperand(1);
2533 auto IntrinsicID =
II.getIntrinsicID();
2534 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2546 auto OpC = OldBinOp->getOpcode();
2552 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2558 if (IsAfter &&
C &&
C->isNullValue()) {
2562 Extract->insertBefore(
II.getIterator());
2563 Extract->takeName(&
II);
2569 return std::nullopt;
2571 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2572 return std::nullopt;
2574 const auto PTruePattern =
2580 return std::nullopt;
2582 unsigned Idx = MinNumElts - 1;
2592 if (Idx >= PgVTy->getMinNumElements())
2593 return std::nullopt;
2598 Extract->insertBefore(
II.getIterator());
2599 Extract->takeName(&
II);
2612 Value *Pg =
II.getArgOperand(0);
2614 Value *Vec =
II.getArgOperand(2);
2617 if (!Ty->isIntegerTy())
2618 return std::nullopt;
2623 return std::nullopt;
2640 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2655static std::optional<Instruction *>
2659 if (
Pattern == AArch64SVEPredPattern::all) {
2668 return MinNumElts && NumElts >= MinNumElts
2670 II, ConstantInt::get(
II.getType(), MinNumElts)))
2674static std::optional<Instruction *>
2677 if (!ST->isStreaming())
2678 return std::nullopt;
2690 Value *PgVal =
II.getArgOperand(0);
2691 Value *OpVal =
II.getArgOperand(1);
2695 if (PgVal == OpVal &&
2696 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2697 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2712 return std::nullopt;
2716 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2717 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2731 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2732 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2733 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2734 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2735 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2736 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2737 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2738 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2739 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2740 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2741 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2742 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2743 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2753 return std::nullopt;
2756template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2757static std::optional<Instruction *>
2759 bool MergeIntoAddendOp) {
2761 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2762 if (MergeIntoAddendOp) {
2763 AddendOp =
II.getOperand(1);
2764 Mul =
II.getOperand(2);
2766 AddendOp =
II.getOperand(2);
2767 Mul =
II.getOperand(1);
2772 return std::nullopt;
2774 if (!
Mul->hasOneUse())
2775 return std::nullopt;
2778 if (
II.getType()->isFPOrFPVectorTy()) {
2783 return std::nullopt;
2785 return std::nullopt;
2790 if (MergeIntoAddendOp)
2800static std::optional<Instruction *>
2802 Value *Pred =
II.getOperand(0);
2803 Value *PtrOp =
II.getOperand(1);
2804 Type *VecTy =
II.getType();
2819static std::optional<Instruction *>
2821 Value *VecOp =
II.getOperand(0);
2822 Value *Pred =
II.getOperand(1);
2823 Value *PtrOp =
II.getOperand(2);
2839 case Intrinsic::aarch64_sve_fmul_u:
2840 return Instruction::BinaryOps::FMul;
2841 case Intrinsic::aarch64_sve_fadd_u:
2842 return Instruction::BinaryOps::FAdd;
2843 case Intrinsic::aarch64_sve_fsub_u:
2844 return Instruction::BinaryOps::FSub;
2846 return Instruction::BinaryOpsEnd;
2850static std::optional<Instruction *>
2853 if (
II.isStrictFP())
2854 return std::nullopt;
2856 auto *OpPredicate =
II.getOperand(0);
2858 if (BinOpCode == Instruction::BinaryOpsEnd ||
2860 return std::nullopt;
2862 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
2866static std::optional<Instruction *>
2868 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
2869 "Expected MLA_U intrinsic");
2870 Value *Acc =
II.getArgOperand(1);
2871 Value *MulOp0 =
II.getArgOperand(2);
2872 Value *MulOp1 =
II.getArgOperand(3);
2887 II.setArgOperand(2, MulOp1);
2888 II.setArgOperand(3, MulOp0);
2892 return std::nullopt;
2895static std::optional<Instruction *>
2897 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
2898 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
2899 "Expected SADALP or UADALP intrinsic");
2905 return std::nullopt;
2909 return std::nullopt;
2913 II.getIntrinsicID(), {II.getType()},
2914 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
2924 Intrinsic::aarch64_sve_mla>(
2928 Intrinsic::aarch64_sve_mad>(
2931 return std::nullopt;
2934static std::optional<Instruction *>
2938 Intrinsic::aarch64_sve_fmla>(IC,
II,
2943 Intrinsic::aarch64_sve_fmad>(IC,
II,
2948 Intrinsic::aarch64_sve_fmla>(IC,
II,
2951 return std::nullopt;
2954static std::optional<Instruction *>
2958 Intrinsic::aarch64_sve_fmla>(IC,
II,
2963 Intrinsic::aarch64_sve_fmad>(IC,
II,
2968 Intrinsic::aarch64_sve_fmla_u>(
2974static std::optional<Instruction *>
2978 Intrinsic::aarch64_sve_fmls>(IC,
II,
2983 Intrinsic::aarch64_sve_fnmsb>(
2988 Intrinsic::aarch64_sve_fmls>(IC,
II,
2991 return std::nullopt;
2994static std::optional<Instruction *>
2998 Intrinsic::aarch64_sve_fmls>(IC,
II,
3003 Intrinsic::aarch64_sve_fnmsb>(
3008 Intrinsic::aarch64_sve_fmls_u>(
3017 Intrinsic::aarch64_sve_mls>(
3020 return std::nullopt;
3025 Value *UnpackArg =
II.getArgOperand(0);
3027 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
3028 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
3041 return std::nullopt;
3045 auto *OpVal =
II.getOperand(0);
3046 auto *OpIndices =
II.getOperand(1);
3053 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
3054 return std::nullopt;
3069 Type *RetTy =
II.getType();
3070 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
3071 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
3075 if ((
match(
II.getArgOperand(0),
3082 if (TyA ==
B->getType() &&
3087 TyA->getMinNumElements());
3093 return std::nullopt;
3101 if (
match(
II.getArgOperand(0),
3106 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
3108 return std::nullopt;
3111static std::optional<Instruction *>
3113 Value *Mask =
II.getOperand(0);
3114 Value *BasePtr =
II.getOperand(1);
3115 Value *Index =
II.getOperand(2);
3126 BasePtr->getPointerAlignment(
II.getDataLayout());
3129 BasePtr, IndexBase);
3136 return std::nullopt;
3139static std::optional<Instruction *>
3141 Value *Val =
II.getOperand(0);
3142 Value *Mask =
II.getOperand(1);
3143 Value *BasePtr =
II.getOperand(2);
3144 Value *Index =
II.getOperand(3);
3154 BasePtr->getPointerAlignment(
II.getDataLayout());
3157 BasePtr, IndexBase);
3163 return std::nullopt;
3169 Value *Pred =
II.getOperand(0);
3170 Value *Vec =
II.getOperand(1);
3171 Value *DivVec =
II.getOperand(2);
3175 if (!SplatConstantInt)
3176 return std::nullopt;
3180 if (DivisorValue == -1)
3181 return std::nullopt;
3182 if (DivisorValue == 1)
3188 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3195 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3197 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3201 return std::nullopt;
3205 size_t VecSize = Vec.
size();
3210 size_t HalfVecSize = VecSize / 2;
3214 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3222 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3240 return std::nullopt;
3247 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3248 CurrentInsertElt = InsertElt->getOperand(0);
3254 return std::nullopt;
3258 for (
size_t I = 0;
I < Elts.
size();
I++) {
3259 if (Elts[
I] ==
nullptr)
3264 if (InsertEltChain ==
nullptr)
3265 return std::nullopt;
3271 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3272 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3273 IIScalableTy->getMinNumElements() /
3278 auto *WideShuffleMaskTy =
3289 auto NarrowBitcast =
3302 return std::nullopt;
3307 Value *Pred =
II.getOperand(0);
3308 Value *Vec =
II.getOperand(1);
3309 Value *Shift =
II.getOperand(2);
3312 Value *AbsPred, *MergedValue;
3318 return std::nullopt;
3326 return std::nullopt;
3331 return std::nullopt;
3334 {
II.getType()}, {Pred, Vec, Shift});
3341 Value *Vec =
II.getOperand(0);
3346 return std::nullopt;
3352 auto *NI =
II.getNextNode();
3355 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3357 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3358 auto *NIBB = NI->getParent();
3359 NI = NI->getNextNode();
3361 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3362 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3368 if (NextII &&
II.isIdenticalTo(NextII))
3371 return std::nullopt;
3379 {II.getType(), II.getOperand(0)->getType()},
3380 {II.getOperand(0), II.getOperand(1)}));
3387 if (PredPattern == AArch64SVEPredPattern::all ||
3388 PredPattern == AArch64SVEPredPattern::pow2)
3390 return std::nullopt;
3396 Value *Passthru =
II.getOperand(0);
3404 auto *Mask = ConstantInt::get(Ty, MaskValue);
3410 return std::nullopt;
3413static std::optional<Instruction *>
3420 return std::nullopt;
3426 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3436 UMinID,
II.getType(), {Pg, NewUMin, ConstantInt::get(II.getType(), 1)});
3446 return std::nullopt;
3452 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3460 return std::nullopt;
3463 II.getType(), {Pg, A, B});
3465 UMinID,
II.getType(), {Pg, NewOrr, ConstantInt::get(II.getType(), 1)});
3474 constexpr Intrinsic::ID CmphsID = Intrinsic::aarch64_sve_cmphs;
3479 Value *
A, *PgLHS, *PgRHS;
3485 !
LHS->hasOneUser() || !
RHS->hasOneUser())
3486 return std::nullopt;
3489 if (ConstB > ConstA)
3495 if (PgLHS != PgRHS || (Pg !=
LHS && Pg !=
RHS && Pg != PgLHS))
3496 return std::nullopt;
3498 Type *VecTy =
A->getType();
3502 Constant *Limit = ConstantInt::get(VecTy, ConstA - ConstB);
3509std::optional<Instruction *>
3520 case Intrinsic::aarch64_dmb:
3522 case Intrinsic::aarch64_neon_fmaxnm:
3523 case Intrinsic::aarch64_neon_fminnm:
3525 case Intrinsic::aarch64_sve_convert_from_svbool:
3527 case Intrinsic::aarch64_sve_dup:
3529 case Intrinsic::aarch64_sve_dup_x:
3531 case Intrinsic::aarch64_sve_cmpeq:
3532 case Intrinsic::aarch64_sve_cmpeq_wide:
3534 case Intrinsic::aarch64_sve_cmpne:
3535 case Intrinsic::aarch64_sve_cmpne_wide:
3537 case Intrinsic::aarch64_sve_rdffr:
3539 case Intrinsic::aarch64_sve_lasta:
3540 case Intrinsic::aarch64_sve_lastb:
3542 case Intrinsic::aarch64_sve_clasta_n:
3543 case Intrinsic::aarch64_sve_clastb_n:
3545 case Intrinsic::aarch64_sve_cntd:
3547 case Intrinsic::aarch64_sve_cntw:
3549 case Intrinsic::aarch64_sve_cnth:
3551 case Intrinsic::aarch64_sve_cntb:
3553 case Intrinsic::aarch64_sme_cntsd:
3555 case Intrinsic::aarch64_sve_ptest_any:
3556 case Intrinsic::aarch64_sve_ptest_first:
3557 case Intrinsic::aarch64_sve_ptest_last:
3559 case Intrinsic::aarch64_sve_fadd:
3561 case Intrinsic::aarch64_sve_fadd_u:
3563 case Intrinsic::aarch64_sve_fmul_u:
3565 case Intrinsic::aarch64_sve_fsub:
3567 case Intrinsic::aarch64_sve_fsub_u:
3569 case Intrinsic::aarch64_sve_add:
3571 case Intrinsic::aarch64_sve_add_u:
3573 Intrinsic::aarch64_sve_mla_u>(
3575 case Intrinsic::aarch64_sve_mla_u:
3577 case Intrinsic::aarch64_sve_sadalp:
3578 case Intrinsic::aarch64_sve_uadalp:
3580 case Intrinsic::aarch64_sve_sub:
3582 case Intrinsic::aarch64_sve_sub_u:
3584 Intrinsic::aarch64_sve_mls_u>(
3586 case Intrinsic::aarch64_sve_tbl:
3588 case Intrinsic::aarch64_sve_uunpkhi:
3589 case Intrinsic::aarch64_sve_uunpklo:
3590 case Intrinsic::aarch64_sve_sunpkhi:
3591 case Intrinsic::aarch64_sve_sunpklo:
3593 case Intrinsic::aarch64_sve_uzp1:
3595 case Intrinsic::aarch64_sve_zip1:
3596 case Intrinsic::aarch64_sve_zip2:
3598 case Intrinsic::aarch64_sve_ld1_gather_index:
3600 case Intrinsic::aarch64_sve_st1_scatter_index:
3602 case Intrinsic::aarch64_sve_ld1:
3604 case Intrinsic::aarch64_sve_st1:
3606 case Intrinsic::aarch64_sve_sdiv:
3608 case Intrinsic::aarch64_sve_sel:
3610 case Intrinsic::aarch64_sve_srshl:
3612 case Intrinsic::aarch64_sve_dupq_lane:
3614 case Intrinsic::aarch64_sve_insr:
3616 case Intrinsic::aarch64_sve_whilelo:
3618 case Intrinsic::aarch64_sve_ptrue:
3620 case Intrinsic::aarch64_sve_uxtb:
3622 case Intrinsic::aarch64_sve_uxth:
3624 case Intrinsic::aarch64_sve_uxtw:
3626 case Intrinsic::aarch64_sme_in_streaming_mode:
3628 case Intrinsic::aarch64_sve_umin_u:
3630 case Intrinsic::aarch64_sve_orr_u:
3632 case Intrinsic::aarch64_sve_and_z:
3636 return std::nullopt;
3643 SimplifyAndSetOp)
const {
3644 switch (
II.getIntrinsicID()) {
3647 case Intrinsic::aarch64_neon_fcvtxn:
3648 case Intrinsic::aarch64_neon_rshrn:
3649 case Intrinsic::aarch64_neon_sqrshrn:
3650 case Intrinsic::aarch64_neon_sqrshrun:
3651 case Intrinsic::aarch64_neon_sqshrn:
3652 case Intrinsic::aarch64_neon_sqshrun:
3653 case Intrinsic::aarch64_neon_sqxtn:
3654 case Intrinsic::aarch64_neon_sqxtun:
3655 case Intrinsic::aarch64_neon_uqrshrn:
3656 case Intrinsic::aarch64_neon_uqshrn:
3657 case Intrinsic::aarch64_neon_uqxtn:
3658 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3662 return std::nullopt;
3666 return ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3676 if (ST->useSVEForFixedLengthVectors() &&
3679 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3680 else if (ST->isNeonAvailable())
3685 if (ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3694bool AArch64TTIImpl::isSingleExtWideningInstruction(
3696 Type *SrcOverrideTy)
const {
3711 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3714 Type *SrcTy = SrcOverrideTy;
3716 case Instruction::Add:
3717 case Instruction::Sub: {
3726 if (Opcode == Instruction::Sub)
3750 assert(SrcTy &&
"Expected some SrcTy");
3752 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3758 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3760 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3764 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3767Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3769 Type *SrcOverrideTy)
const {
3770 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3771 Opcode != Instruction::Mul)
3781 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3784 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3790 ->getScalarSizeInBits();
3793 unsigned MaxEltSize = 0;
3796 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3797 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3798 MaxEltSize = std::max(EltSize0, EltSize1);
3801 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3802 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3805 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3807 MaxEltSize = DstEltSize / 2;
3808 }
else if (Opcode == Instruction::Mul &&
3816 Known.Zero.countLeadingOnes() >
3821 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3825 if (MaxEltSize * 2 > DstEltSize)
3843 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3844 (Src->isScalableTy() && !ST->hasSVE2()))
3854 if (AddUser && AddUser->getOpcode() == Instruction::Add)
3858 if (!Shr || Shr->getOpcode() != Instruction::LShr)
3862 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
3863 Src->getScalarSizeInBits() !=
3887 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3891 if (
I &&
I->hasOneUser()) {
3894 if (
Type *ExtTy = isBinExtWideningInstruction(
3895 SingleUser->getOpcode(), Dst,
Operands,
3896 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3909 if (isSingleExtWideningInstruction(
3910 SingleUser->getOpcode(), Dst,
Operands,
3911 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3915 if (SingleUser->getOpcode() == Instruction::Add) {
3916 if (
I == SingleUser->getOperand(1) ||
3918 cast<CastInst>(SingleUser->getOperand(1))->getOpcode() == Opcode))
3933 EVT SrcTy = TLI->getValueType(
DL, Src);
3934 EVT DstTy = TLI->getValueType(
DL, Dst);
3942 Instruction::ExtractElement, Src,
CostKind, -1,
nullptr,
nullptr);
3944 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH,
CostKind);
3948 if (!SrcTy.isSimple() || !DstTy.
isSimple())
3953 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
3982 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
3985 ST->useSVEForFixedLengthVectors(WiderTy)) {
3986 std::pair<InstructionCost, MVT> LT =
3988 unsigned NumElements =
4004 const unsigned int SVE_EXT_COST = 1;
4005 const unsigned int SVE_FCVT_COST = 1;
4006 const unsigned int SVE_UNPACK_ONCE = 4;
4007 const unsigned int SVE_UNPACK_TWICE = 16;
4136 SVE_EXT_COST + SVE_FCVT_COST},
4141 SVE_EXT_COST + SVE_FCVT_COST},
4148 SVE_EXT_COST + SVE_FCVT_COST},
4152 SVE_EXT_COST + SVE_FCVT_COST},
4158 SVE_EXT_COST + SVE_FCVT_COST},
4161 SVE_EXT_COST + SVE_FCVT_COST},
4166 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4168 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4178 SVE_EXT_COST + SVE_FCVT_COST},
4183 SVE_EXT_COST + SVE_FCVT_COST},
4196 SVE_EXT_COST + SVE_FCVT_COST},
4200 SVE_EXT_COST + SVE_FCVT_COST},
4212 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4214 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4216 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4218 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4222 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4224 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4240 SVE_EXT_COST + SVE_FCVT_COST},
4245 SVE_EXT_COST + SVE_FCVT_COST},
4256 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4258 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4260 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4262 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4264 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4266 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4270 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4272 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4274 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4276 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4501 if (ST->hasFullFP16())
4513 Src->getScalarType(), CCH,
CostKind) +
4521 ST->isSVEorStreamingSVEAvailable() &&
4522 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4524 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4533 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4536 return Part1 + Part2;
4543 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4555 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4568 CostKind, Index,
nullptr,
nullptr);
4572 auto DstVT = TLI->getValueType(
DL, Dst);
4573 auto SrcVT = TLI->getValueType(
DL, Src);
4578 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4584 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4594 case Instruction::SExt:
4599 case Instruction::ZExt:
4600 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4613 return Opcode == Instruction::PHI ? 0 : 1;
4622 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4624 assert(Ty->isVectorTy() &&
"This must be a vector type");
4631 if (!LT.second.isVector())
4636 if (LT.second.isFixedLengthVector()) {
4637 unsigned Width = LT.second.getVectorNumElements();
4638 Index = Index % Width;
4645 if (Index == 0 && !Ty->getScalarType()->isIntegerTy())
4657 if (Index * Ty->getScalarSizeInBits() < 128)
4659 if (Index * Ty->getScalarSizeInBits() < 512 &&
4660 Opcode == Instruction::ExtractElement)
4662 return Ty->getScalarType()->isIntegerTy() ? Cost + 1 : Cost;
4663 if (Opcode == Instruction::ExtractElement)
4665 if (Opcode == Instruction::InsertElement)
4674 if (VIC == TTI::VectorInstrContext::Load) {
4675 if (ST->hasFastLD1Single())
4679 : ST->getVectorInsertExtractBaseCost() + 1;
4687 : ST->getVectorInsertExtractBaseCost() + 1;
4711 auto ExtractCanFuseWithFmul = [&]() {
4718 auto IsAllowedScalarTy = [&](
const Type *
T) {
4719 return T->isFloatTy() ||
T->isDoubleTy() ||
4720 (
T->isHalfTy() && ST->hasFullFP16());
4724 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4727 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4728 !BO->getType()->isVectorTy();
4733 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4737 return Idx == 0 || (RegWidth != 0 && (Idx * EltSz) % RegWidth == 0);
4746 DenseMap<User *, unsigned> UserToExtractIdx;
4747 for (
auto *U :
Scalar->users()) {
4748 if (!IsUserFMulScalarTy(U))
4752 UserToExtractIdx[
U];
4754 if (UserToExtractIdx.
empty())
4756 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4757 for (
auto *U : S->users()) {
4758 if (UserToExtractIdx.
contains(U)) {
4760 auto *Op0 =
FMul->getOperand(0);
4761 auto *Op1 =
FMul->getOperand(1);
4762 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4763 UserToExtractIdx[
U] =
L;
4769 for (
auto &[U, L] : UserToExtractIdx) {
4781 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4782 if (!IsUserFMulScalarTy(U))
4787 const auto *BO = cast<BinaryOperator>(U);
4788 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4789 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4791 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4794 return IsExtractLaneEquivalentToZero(
4795 cast<ConstantInt>(OtherEE->getIndexOperand())
4798 OtherEE->getType()->getScalarSizeInBits());
4806 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4807 ExtractCanFuseWithFmul())
4812 :
ST->getVectorInsertExtractBaseCost();
4821 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
4824 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr,
nullptr,
4830 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4832 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr, Scalar,
4833 ScalarUserAndIdx, VIC);
4840 return getVectorInstrCostHelper(
I.getOpcode(), Ty,
CostKind, Index, &
I,
4847 unsigned Index)
const {
4858 : ST->getVectorInsertExtractBaseCost() + 1;
4867 if (Ty->getElementType()->isFloatingPointTy())
4870 unsigned VecInstCost =
4872 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
4879 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
4880 return std::nullopt;
4881 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
4882 return std::nullopt;
4884 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
4885 return std::nullopt;
4892 Cost += InstCost(PromotedTy);
4914 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4921 Op2Info, Args, CxtI);
4928 Ty,
CostKind, Op1Info, Op2Info,
true,
4931 [&](
Type *PromotedTy) {
4935 return *PromotedCost;
4938 if (Ty->getScalarType()->isFP128Ty())
4946 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
4966 ST->hasLimited64bitVectorMulBandwidth())
4969 if (Ty->getScalarSizeInBits() > 64) {
4974 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
4977 if (LT.second == MVT::v2i64) {
4981 return LT.first * Mul64CostFactor;
5002 if (LT.second == MVT::nxv2i64)
5003 return LT.first * Mul64CostFactor;
5062 auto VT = TLI->getValueType(
DL, Ty);
5063 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
5067 : (3 * AsrCost + AddCost);
5069 return MulCost + AsrCost + 2 * AddCost;
5071 }
else if (VT.isVector()) {
5081 if (Ty->isScalableTy() && ST->hasSVE())
5082 Cost += 2 * AsrCost;
5087 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
5091 }
else if (LT.second == MVT::v2i64) {
5092 return VT.getVectorNumElements() *
5099 if (Ty->isScalableTy() && ST->hasSVE())
5100 return MulCost + 2 * AddCost + 2 * AsrCost;
5101 return 2 * MulCost + AddCost + AsrCost + UsraCost;
5106 LT.second.isFixedLengthVector()) {
5116 return ExtractCost + InsertCost +
5124 auto VT = TLI->getValueType(
DL, Ty);
5140 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
5141 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
5142 LT.second == MVT::nxv16i8;
5143 bool Is128bit = LT.second.is128BitVector();
5155 (HasMULH ? 0 : ShrCost) +
5156 AddCost * 2 + ShrCost;
5157 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
5164 if (!VT.isVector() && VT.getSizeInBits() > 64)
5168 Opcode, Ty,
CostKind, Op1Info, Op2Info);
5170 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
5174 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
5184 if (
nullptr != Entry)
5192 FVTy && LT.second.isFixedLengthVector()) {
5193 unsigned NumElts = FVTy->getNumElements();
5194 unsigned RegElts = LT.second.getVectorNumElements();
5196 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
5200 if (LT.second.getScalarType() == MVT::i8)
5202 else if (LT.second.getScalarType() == MVT::i16)
5214 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
5215 return (4 + DivCost) * VTy->getNumElements();
5221 -1,
nullptr,
nullptr);
5248 LT.second.isFixedLengthVector())
5249 return 2 * LT.first + 1;
5258 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
5259 (Ty->isHalfTy() && ST->hasFullFP16())) &&
5268 if (!Ty->getScalarType()->isFP128Ty())
5275 if (!Ty->getScalarType()->isFP128Ty())
5276 return 2 * LT.first;
5283 if (!Ty->isVectorTy())
5299 int MaxMergeDistance = 64;
5303 return NumVectorInstToHideOverhead;
5313 unsigned Opcode1,
unsigned Opcode2)
const {
5316 if (!
Sched.hasInstrSchedModel())
5320 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
5322 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5328 "Cannot handle variant scheduling classes without an MI");
5344 const int AmortizationCost = 20;
5352 VecPred = CurrentPred;
5360 static const auto ValidMinMaxTys = {
5361 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5362 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5363 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5367 (ST->hasFullFP16() &&
5373 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5374 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5375 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5376 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5377 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5378 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5379 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5380 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5381 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5382 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5383 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5385 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5386 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5395 if (Opcode == Instruction::FCmp) {
5397 ValTy,
CostKind, Op1Info, Op2Info,
false,
5399 false, [&](
Type *PromotedTy) {
5411 return *PromotedCost;
5415 if (LT.second.getScalarType() != MVT::f64 &&
5416 LT.second.getScalarType() != MVT::f32 &&
5417 LT.second.getScalarType() != MVT::f16)
5422 unsigned Factor = 1;
5423 if (!CondTy->isVectorTy() &&
5437 AArch64::FCMEQv4f32))
5449 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5468 Op1Info, Op2Info,
I);
5474 if (ST->requiresStrictAlign()) {
5479 Options.AllowOverlappingLoads =
true;
5480 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5485 Options.LoadSizes = {8, 4, 2, 1};
5486 Options.AllowedTailExpansions = {3, 5, 6};
5491 return ST->hasSVE();
5497 switch (MICA.
getID()) {
5498 case Intrinsic::masked_scatter:
5499 case Intrinsic::masked_gather:
5501 case Intrinsic::masked_load:
5502 case Intrinsic::masked_store:
5503 case Intrinsic::masked_expandload:
5504 case Intrinsic::masked_compressstore:
5518 if (!LT.first.isValid())
5523 if (VT->getElementType()->isIntegerTy(1))
5534 if (MICA.
getID() == Intrinsic::masked_expandload) {
5542 if (MICA.
getID() == Intrinsic::masked_compressstore) {
5563 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5564 return MemOpCost * 2;
5573 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5574 "Should be called on only load or stores.");
5576 case Instruction::Load:
5579 return ST->getGatherOverhead();
5581 case Instruction::Store:
5584 return ST->getScatterOverhead();
5595 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5596 MICA.
getID() == Intrinsic::vp_gather)
5598 : Instruction::Store;
5608 if (!LT.first.isValid())
5612 if (!LT.second.isVector() ||
5614 VT->getElementType()->isIntegerTy(1))
5624 ElementCount LegalVF = LT.second.getVectorElementCount();
5627 {TTI::OK_AnyValue, TTI::OP_None},
I);
5643 EVT VT = TLI->getValueType(
DL, Ty,
true);
5645 if (VT == MVT::Other)
5650 if (!LT.first.isValid())
5660 (VTy->getElementType()->isIntegerTy(1) &&
5661 !VTy->getElementCount().isKnownMultipleOf(
5671 if (Opcode == Instruction::Store)
5675 if (ST->getFixedLoadLatency())
5676 return (LT.first - 1) + ST->getFixedLoadLatency();
5685 if (LT.second.isScalableVector() ||
5686 ST->useSVEForFixedLengthVectors(LT.second)) {
5687 Inst = AArch64::LDR_ZXI;
5688 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5689 switch (LT.second.getSizeInBits()) {
5691 Inst = AArch64::LDRBui;
5694 Inst = AArch64::LDRHui;
5697 Inst = AArch64::LDRSui;
5700 Inst = AArch64::LDRDui;
5703 Inst = AArch64::LDRQui;
5709 switch (LT.second.getSizeInBits()) {
5711 Inst = AArch64::LDRBBui;
5714 Inst = AArch64::LDRHHui;
5717 Inst = AArch64::LDRWui;
5720 Inst = AArch64::LDRXui;
5728 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5730 ?
Sched.getSchedClassDesc(SchedClass)
5736 return (LT.first - 1) + ST->getLoadLatency();
5739 float NumLoads = (LT.first - 1).
getValue();
5740 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5741 Sched.computeInstrLatency(*ST, *SCD);
5744 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5745 LT.second.is128BitVector() && Alignment <
Align(16)) {
5751 const int AmortizationCost = 6;
5753 return LT.first * 2 * AmortizationCost;
5757 if (Ty->isPtrOrPtrVectorTy())
5762 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5764 if (VT == MVT::v4i8)
5771 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5772 Alignment !=
Align(1))
5784 if (Remainder != 0) {
5787 while (!TypeWorklist.
empty()) {
5797 TypeWorklist.
push_back({CurrNumElements - PrevPow2,
Offset + PrevPow2});
5809 bool UseMaskForCond,
bool UseMaskForGaps)
const {
5810 assert(Factor >= 2 &&
"Invalid interleave factor");
5819 if (Factor > TLI->getMaxSupportedInterleaveFactor())
5823 DL.getTypeSizeInBits(VecTy).getKnownMinValue() != (3 * 128))
5828 unsigned MaxNativeInterleaveFactor = TLI->getMaxSupportedInterleaveFactor();
5833 (UseMaskForCond || UseMaskForGaps ||
5834 (Factor > MaxNativeInterleaveFactor &&
5835 TLI->useSVEForFixedLengthVectorVT(LT.second))))
5838 if (!UseMaskForGaps && Factor <= MaxNativeInterleaveFactor) {
5841 EC.divideCoefficientBy(Factor));
5847 if (EC.isKnownMultipleOf(Factor) &&
5848 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
5849 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
5854 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
5860 if (UseMaskForCond) {
5861 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5862 : Intrinsic::masked_store;
5882 if (Opcode == Instruction::Store && Factor == 4 &&
5883 SubVecCost.second.getScalarSizeInBits() ==
5884 (4 * ResultCost.second.getScalarSizeInBits()))
5885 LegalizationCost *= 4;
5887 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
5893 UseMaskForCond, UseMaskForGaps);
5900 for (
auto *
I : Tys) {
5901 if (!
I->isVectorTy())
5912 Align Alignment)
const {
5919 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
5920 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
5925 bool HasUnorderedReductions)
const {
5928 return ST->getMaxInterleaveFactor();
5938 enum { MaxStridedLoads = 7 };
5940 int StridedLoads = 0;
5943 for (
const auto BB : L->blocks()) {
5944 for (
auto &
I : *BB) {
5950 if (L->isLoopInvariant(PtrValue))
5955 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
5964 if (StridedLoads > MaxStridedLoads / 2)
5965 return StridedLoads;
5968 return StridedLoads;
5971 int StridedLoads = countStridedLoads(L, SE);
5973 <<
" strided loads\n");
5989 unsigned *FinalSize) {
5993 for (
auto *BB : L->getBlocks()) {
5994 for (
auto &
I : *BB) {
6000 if (!Cost.isValid())
6004 if (LoopCost > Budget)
6026 if (MaxTC > 0 && MaxTC <= 32)
6037 if (Blocks.
size() != 2)
6059 if (!L->isInnermost() || L->getNumBlocks() > 8)
6063 if (!L->getExitBlock())
6069 bool HasParellelizableReductions =
6070 L->getNumBlocks() == 1 &&
6071 any_of(L->getHeader()->phis(),
6073 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
6076 if (HasParellelizableReductions &&
6098 if (HasParellelizableReductions) {
6109 if (Header == Latch) {
6112 unsigned Width = 10;
6118 unsigned MaxInstsPerLine = 16;
6120 unsigned BestUC = 1;
6121 unsigned SizeWithBestUC = BestUC *
Size;
6123 unsigned SizeWithUC = UC *
Size;
6124 if (SizeWithUC > 48)
6126 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
6127 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
6129 SizeWithBestUC = BestUC *
Size;
6139 for (
auto *BB : L->blocks()) {
6140 for (
auto &
I : *BB) {
6150 for (
auto *U :
I.users())
6152 LoadedValuesPlus.
insert(U);
6159 return LoadedValuesPlus.
contains(
SI->getOperand(0));
6185 auto *I = dyn_cast<Instruction>(V);
6186 return I && DependsOnLoopLoad(I, Depth + 1);
6193 DependsOnLoopLoad(
I, 0)) {
6225 if (L->getLoopDepth() > 1)
6236 for (
auto *BB : L->getBlocks()) {
6237 for (
auto &
I : *BB) {
6241 if (IsVectorized &&
I.getType()->isVectorTy())
6258 if (ST->isAppleMLike())
6260 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
6282 !ST->getSchedModel().isOutOfOrder()) {
6305 bool CanCreate)
const {
6309 case Intrinsic::aarch64_neon_st1x2:
6310 case Intrinsic::aarch64_neon_st1x3:
6311 case Intrinsic::aarch64_neon_st1x4:
6312 case Intrinsic::aarch64_neon_st2:
6313 case Intrinsic::aarch64_neon_st3:
6314 case Intrinsic::aarch64_neon_st4: {
6317 if (!CanCreate || !ST)
6319 unsigned NumElts = Inst->
arg_size() - 1;
6320 if (ST->getNumElements() != NumElts)
6322 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6328 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6330 Res = Builder.CreateInsertValue(Res, L, i);
6334 case Intrinsic::aarch64_neon_ld1x2:
6335 case Intrinsic::aarch64_neon_ld1x3:
6336 case Intrinsic::aarch64_neon_ld1x4:
6337 case Intrinsic::aarch64_neon_ld2:
6338 case Intrinsic::aarch64_neon_ld3:
6339 case Intrinsic::aarch64_neon_ld4:
6340 if (Inst->
getType() == ExpectedType)
6351 case Intrinsic::aarch64_neon_ld1x2:
6352 case Intrinsic::aarch64_neon_ld1x3:
6353 case Intrinsic::aarch64_neon_ld1x4:
6354 case Intrinsic::aarch64_neon_ld2:
6355 case Intrinsic::aarch64_neon_ld3:
6356 case Intrinsic::aarch64_neon_ld4:
6357 Info.ReadMem =
true;
6358 Info.WriteMem =
false;
6361 case Intrinsic::aarch64_neon_st1x2:
6362 case Intrinsic::aarch64_neon_st1x3:
6363 case Intrinsic::aarch64_neon_st1x4:
6364 case Intrinsic::aarch64_neon_st2:
6365 case Intrinsic::aarch64_neon_st3:
6366 case Intrinsic::aarch64_neon_st4:
6367 Info.ReadMem =
false;
6368 Info.WriteMem =
true;
6377 case Intrinsic::aarch64_neon_ld1x2:
6378 case Intrinsic::aarch64_neon_st1x2:
6379 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6381 case Intrinsic::aarch64_neon_ld1x3:
6382 case Intrinsic::aarch64_neon_st1x3:
6383 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6385 case Intrinsic::aarch64_neon_ld1x4:
6386 case Intrinsic::aarch64_neon_st1x4:
6387 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6389 case Intrinsic::aarch64_neon_ld2:
6390 case Intrinsic::aarch64_neon_st2:
6391 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6393 case Intrinsic::aarch64_neon_ld3:
6394 case Intrinsic::aarch64_neon_st3:
6395 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6397 case Intrinsic::aarch64_neon_ld4:
6398 case Intrinsic::aarch64_neon_st4:
6399 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6411 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6412 bool Considerable =
false;
6413 AllowPromotionWithoutCommonHeader =
false;
6416 Type *ConsideredSExtType =
6418 if (
I.getType() != ConsideredSExtType)
6422 for (
const User *U :
I.users()) {
6424 Considerable =
true;
6428 if (GEPInst->getNumOperands() > 2) {
6429 AllowPromotionWithoutCommonHeader =
true;
6434 return Considerable;
6485 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6495 return LegalizationCost + 2;
6505 LegalizationCost *= LT.first - 1;
6508 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6517 return LegalizationCost + 2;
6525 std::optional<FastMathFlags> FMF,
6541 return BaseCost + FixedVTy->getNumElements();
6555 MVT MTy = LT.second;
6560 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6608 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6609 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6621 return (LT.first - 1) +
Log2_32(NElts);
6626 return (LT.first - 1) + Entry->Cost;
6638 if (LT.first != 1) {
6644 ExtraCost *= LT.first - 1;
6647 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6648 return Cost + ExtraCost;
6656 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6658 EVT VecVT = TLI->getValueType(
DL, VecTy);
6659 EVT ResVT = TLI->getValueType(
DL, ResTy);
6669 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6671 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6673 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6675 return (LT.first - 1) * 2 + 2;
6686 EVT VecVT = TLI->getValueType(
DL, VecTy);
6687 EVT ResVT = TLI->getValueType(
DL, ResTy);
6690 RedOpcode == Instruction::Add) {
6696 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6698 return LT.first + 2;
6733 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6734 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6748 if (LT.second.getScalarType() == MVT::i1) {
6757 assert(Entry &&
"Illegal Type for Splice");
6758 LegalizationCost += Entry->Cost;
6759 return LegalizationCost * LT.first;
6763 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
6772 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
6773 Opcode != Instruction::FAdd && Opcode != Instruction::FSub))
6779 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
6780 if (!FMF->allowReassoc() || !FMF->allowContract())
6784 "FastMathFlags only apply to floating-point partial reductions");
6788 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
6789 "Unexpected values for OpBExtend or InputTypeB");
6793 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
6794 InputTypeA != InputTypeB))
6805 assert(!OpBExtend &&
"Extended second operand without extended first.");
6806 assert(InputTypeA == AccumType &&
"Type mismatch with no extensions.");
6812 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
6815 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
6828 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
6837 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
6843 std::pair<InstructionCost, MVT> AccumLT =
6845 std::pair<InstructionCost, MVT> InputLT =
6849 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
6850 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
6851 (AccumLT.second.isFixedLengthVector() &&
6852 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
6856 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
6864 if (AccumLT.second.getScalarType() == MVT::i32 &&
6865 InputLT.second.getScalarType() == MVT::i8) {
6867 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
6868 return Cost + INegCost;
6870 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
6871 return Cost + INegCost;
6876 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
6877 return Cost * 3 + INegCost;
6880 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
6882 if (AccumLT.second.getScalarType() == MVT::i64 &&
6883 InputLT.second.getScalarType() == MVT::i16)
6884 return Cost + INegCost;
6887 if (AccumLT.second.getScalarType() == MVT::i32 &&
6888 InputLT.second.getScalarType() == MVT::i16 &&
6889 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
6892 if (AccumLT.second.getScalarType() == MVT::i64 &&
6893 InputLT.second.getScalarType() == MVT::i8)
6899 return Cost + INegCost;
6902 if (AccumLT.second.getScalarType() == MVT::i16 &&
6903 InputLT.second.getScalarType() == MVT::i8 &&
6904 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
6910 if (Opcode == Instruction::FAdd && !IsSub &&
6911 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
6912 AccumLT.second.getScalarType() == MVT::f32 &&
6913 InputLT.second.getScalarType() == MVT::f16)
6917 if (Ratio == 2 && !IsUSDot) {
6918 MVT InVT = InputLT.second.getScalarType();
6922 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
6924 return (BinOp || IsSub) ?
Cost * 2 :
Cost;
6927 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
6931 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
6932 InVT == MVT::bf16 && IsSub)
6942 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
6943 return Cost * 2 + FNegCost;
6947 AccumType, VF, OpAExtend, OpBExtend,
6959 "Expected the Mask to match the return size if given");
6961 "Expected the same scalar types");
6967 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
6968 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
6969 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
6977 return std::max<InstructionCost>(1, LT.first / 4);
6985 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
6987 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
6990 unsigned TpNumElts = Mask.size();
6991 unsigned LTNumElts = LT.second.getVectorNumElements();
6992 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
6994 LT.second.getVectorElementCount());
6996 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
6998 for (
unsigned N = 0;
N < NumVecs;
N++) {
7002 unsigned Source1 = -1U, Source2 = -1U;
7003 unsigned NumSources = 0;
7004 for (
unsigned E = 0; E < LTNumElts; E++) {
7005 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
7014 unsigned Source = MaskElt / LTNumElts;
7015 if (NumSources == 0) {
7018 }
else if (NumSources == 1 && Source != Source1) {
7021 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
7027 if (Source == Source1)
7029 else if (Source == Source2)
7030 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
7039 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
7050 NTp, NTp,
CostKind, NMask, 0,
nullptr, Args,
7053 Result.first->second = NCost;
7067 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
7068 if (LT.second.getFixedSizeInBits() >= 128 &&
7070 LT.second.getVectorNumElements() / 2) {
7073 if (Index == (
int)LT.second.getVectorNumElements() / 2)
7087 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
7090 return M.value() < 0 || M.value() == (int)M.index();
7096 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
7097 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
7106 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
7107 ST->isSVEorStreamingSVEAvailable() &&
7112 if (ST->isSVEorStreamingSVEAvailable() &&
7126 if (IsLoad && LT.second.isVector() &&
7128 LT.second.getVectorElementCount()))
7134 if (Mask.size() == 4 &&
7136 (SrcTy->getScalarSizeInBits() == 16 ||
7137 SrcTy->getScalarSizeInBits() == 32) &&
7138 all_of(Mask, [](
int E) {
return E < 8; }))
7144 if (LT.second.isFixedLengthVector() &&
7145 LT.second.getVectorNumElements() == Mask.size() &&
7151 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7152 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7153 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
7154 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7155 LT.second.getVectorNumElements(), 16) ||
7156 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7157 LT.second.getVectorNumElements(), 32) ||
7158 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7159 LT.second.getVectorNumElements(), 64) ||
7162 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
7291 return LT.first * Entry->Cost;
7300 LT.second.getSizeInBits() <= 128 && SubTp) {
7302 if (SubLT.second.isVector()) {
7303 int NumElts = LT.second.getVectorNumElements();
7304 int NumSubElts = SubLT.second.getVectorNumElements();
7305 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
7311 if (IsExtractSubvector)
7332 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
7345 return ST->useFixedOverScalableIfEqualCost();
7349 return ST->getEpilogueVectorizationMinVF();
7384 unsigned NumInsns = 0;
7386 NumInsns += BB->size();
7396 int64_t Scale,
unsigned AddrSpace)
const {
7424 if (
I->getOpcode() == Instruction::Or &&
7428 if (
I->getOpcode() == Instruction::Add ||
7429 I->getOpcode() == Instruction::Sub)
7454 return all_equal(Shuf->getShuffleMask());
7461 bool AllowSplat =
false) {
7466 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7467 auto *FullTy = FullV->
getType();
7468 auto *HalfTy = HalfV->getType();
7470 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7473 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7476 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7480 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7494 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7495 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7509 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7510 (M2Start != 0 && M2Start != (NumElements / 2)))
7512 if (S1Op1 && S2Op1 && M1Start != M2Start)
7522 return Ext->getType()->getScalarSizeInBits() ==
7523 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7537 Value *VectorOperand =
nullptr;
7554 if (!
GEP ||
GEP->getNumOperands() != 2)
7558 Value *Offsets =
GEP->getOperand(1);
7561 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7567 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7568 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7569 Ops.push_back(&
GEP->getOperandUse(1));
7605 switch (
II->getIntrinsicID()) {
7606 case Intrinsic::aarch64_neon_smull:
7607 case Intrinsic::aarch64_neon_umull:
7610 Ops.push_back(&
II->getOperandUse(0));
7611 Ops.push_back(&
II->getOperandUse(1));
7616 case Intrinsic::fma:
7617 case Intrinsic::fmuladd:
7624 Ops.push_back(&
II->getOperandUse(0));
7626 Ops.push_back(&
II->getOperandUse(1));
7629 case Intrinsic::aarch64_neon_sqdmull:
7630 case Intrinsic::aarch64_neon_sqdmulh:
7631 case Intrinsic::aarch64_neon_sqrdmulh:
7634 Ops.push_back(&
II->getOperandUse(0));
7636 Ops.push_back(&
II->getOperandUse(1));
7637 return !
Ops.empty();
7638 case Intrinsic::aarch64_neon_fmlal:
7639 case Intrinsic::aarch64_neon_fmlal2:
7640 case Intrinsic::aarch64_neon_fmlsl:
7641 case Intrinsic::aarch64_neon_fmlsl2:
7644 Ops.push_back(&
II->getOperandUse(1));
7646 Ops.push_back(&
II->getOperandUse(2));
7647 return !
Ops.empty();
7648 case Intrinsic::aarch64_sve_ptest_first:
7649 case Intrinsic::aarch64_sve_ptest_last:
7651 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7652 Ops.push_back(&
II->getOperandUse(0));
7653 return !
Ops.empty();
7654 case Intrinsic::aarch64_sme_write_horiz:
7655 case Intrinsic::aarch64_sme_write_vert:
7656 case Intrinsic::aarch64_sme_writeq_horiz:
7657 case Intrinsic::aarch64_sme_writeq_vert: {
7659 if (!Idx || Idx->getOpcode() != Instruction::Add)
7661 Ops.push_back(&
II->getOperandUse(1));
7664 case Intrinsic::aarch64_sme_read_horiz:
7665 case Intrinsic::aarch64_sme_read_vert:
7666 case Intrinsic::aarch64_sme_readq_horiz:
7667 case Intrinsic::aarch64_sme_readq_vert:
7668 case Intrinsic::aarch64_sme_ld1b_vert:
7669 case Intrinsic::aarch64_sme_ld1h_vert:
7670 case Intrinsic::aarch64_sme_ld1w_vert:
7671 case Intrinsic::aarch64_sme_ld1d_vert:
7672 case Intrinsic::aarch64_sme_ld1q_vert:
7673 case Intrinsic::aarch64_sme_st1b_vert:
7674 case Intrinsic::aarch64_sme_st1h_vert:
7675 case Intrinsic::aarch64_sme_st1w_vert:
7676 case Intrinsic::aarch64_sme_st1d_vert:
7677 case Intrinsic::aarch64_sme_st1q_vert:
7678 case Intrinsic::aarch64_sme_ld1b_horiz:
7679 case Intrinsic::aarch64_sme_ld1h_horiz:
7680 case Intrinsic::aarch64_sme_ld1w_horiz:
7681 case Intrinsic::aarch64_sme_ld1d_horiz:
7682 case Intrinsic::aarch64_sme_ld1q_horiz:
7683 case Intrinsic::aarch64_sme_st1b_horiz:
7684 case Intrinsic::aarch64_sme_st1h_horiz:
7685 case Intrinsic::aarch64_sme_st1w_horiz:
7686 case Intrinsic::aarch64_sme_st1d_horiz:
7687 case Intrinsic::aarch64_sme_st1q_horiz: {
7689 if (!Idx || Idx->getOpcode() != Instruction::Add)
7691 Ops.push_back(&
II->getOperandUse(3));
7694 case Intrinsic::aarch64_neon_pmull:
7697 Ops.push_back(&
II->getOperandUse(0));
7698 Ops.push_back(&
II->getOperandUse(1));
7700 case Intrinsic::aarch64_neon_pmull64:
7702 II->getArgOperand(1)))
7704 Ops.push_back(&
II->getArgOperandUse(0));
7705 Ops.push_back(&
II->getArgOperandUse(1));
7707 case Intrinsic::masked_gather:
7710 Ops.push_back(&
II->getArgOperandUse(0));
7712 case Intrinsic::masked_scatter:
7715 Ops.push_back(&
II->getArgOperandUse(1));
7722 auto ShouldSinkCondition = [](
Value *
Cond,
7727 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
7731 Ops.push_back(&
II->getOperandUse(0));
7735 switch (
I->getOpcode()) {
7736 case Instruction::GetElementPtr:
7737 case Instruction::Add:
7738 case Instruction::Sub:
7740 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
7742 Ops.push_back(&
I->getOperandUse(
Op));
7747 case Instruction::Select: {
7748 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
7751 Ops.push_back(&
I->getOperandUse(0));
7754 case Instruction::UncondBr:
7756 case Instruction::CondBr: {
7760 Ops.push_back(&
I->getOperandUse(0));
7763 case Instruction::FMul:
7768 Ops.push_back(&
I->getOperandUse(0));
7770 Ops.push_back(&
I->getOperandUse(1));
7780 case Instruction::Xor:
7783 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
7785 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
7790 case Instruction::And:
7791 case Instruction::Or:
7794 if (
I->getOpcode() == Instruction::Or &&
7799 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
7802 for (
auto &
Op :
I->operands()) {
7814 Ops.push_back(&Not);
7815 Ops.push_back(&InsertElt);
7825 if (!
I->getType()->isVectorTy())
7826 return !
Ops.empty();
7828 switch (
I->getOpcode()) {
7829 case Instruction::Sub:
7830 case Instruction::Add: {
7839 Ops.push_back(&Ext1->getOperandUse(0));
7840 Ops.push_back(&Ext2->getOperandUse(0));
7843 Ops.push_back(&
I->getOperandUse(0));
7844 Ops.push_back(&
I->getOperandUse(1));
7848 case Instruction::Or: {
7851 if (ST->hasNEON()) {
7865 if (
I->getParent() != MainAnd->
getParent() ||
7870 if (
I->getParent() != IA->getParent() ||
7871 I->getParent() != IB->getParent())
7876 Ops.push_back(&
I->getOperandUse(0));
7877 Ops.push_back(&
I->getOperandUse(1));
7886 case Instruction::Mul: {
7887 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
7890 if (Ty->isScalableTy())
7894 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
7897 int NumZExts = 0, NumSExts = 0;
7898 for (
auto &
Op :
I->operands()) {
7905 auto *ExtOp = Ext->getOperand(0);
7906 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
7907 Ops.push_back(&Ext->getOperandUse(0));
7915 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
7916 I->getType()->getScalarSizeInBits())
7953 if (!ElementConstant || !ElementConstant->
isZero())
7956 unsigned Opcode = OperandInstr->
getOpcode();
7957 if (Opcode == Instruction::SExt)
7959 else if (Opcode == Instruction::ZExt)
7964 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
7974 Ops.push_back(&Insert->getOperandUse(1));
7980 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
7984 if (!ShouldSinkSplatForIndexedVariant(
I))
7989 Ops.push_back(&
I->getOperandUse(0));
7991 Ops.push_back(&
I->getOperandUse(1));
7993 return !
Ops.empty();
7995 case Instruction::FMul: {
7997 if (
I->getType()->isScalableTy())
7998 return !
Ops.empty();
8002 return !
Ops.empty();
8006 Ops.push_back(&
I->getOperandUse(0));
8008 Ops.push_back(&
I->getOperandUse(1));
8009 return !
Ops.empty();
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
bool useSVEForFixedLengthVectors() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index) const override
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
bool preferFixedOverScalableIfEqualCost() const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
static bool isCommutative(Predicate Pred)
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static bool isCommutative(Predicate P)
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
user_iterator user_begin()
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
bool isSMEABIRoutine() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ CTTZ_ELTS
Returns the number of number of trailing (least significant) zero elements in a vector.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const SymbolicStrideMap &StridesMap=SymbolicStrideMap(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
CmpInst::Predicate getCmpPredicate() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
bool hasCmpPredicate() const
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
SVEIntrinsicInfo & setCmpPredicate(CmpInst::Predicate Pred)
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...