25#include "llvm/IR/IntrinsicsAArch64.h"
37#define DEBUG_TYPE "aarch64tti"
40class TailFoldingOption {
55 bool NeedsDefault =
true;
59 void setNeedsDefault(
bool V) { NeedsDefault = V; }
75 "Initial bits should only include one of "
76 "(disabled|all|simple|default)");
77 Bits = NeedsDefault ? DefaultBits : InitialBits;
85 errs() <<
"invalid argument '" << Opt
86 <<
"' to -sve-tail-folding=; the option should be of the form\n"
87 " (disabled|all|default|simple)[+(reductions|recurrences"
88 "|reverse|noreductions|norecurrences|noreverse)]\n";
94 void operator=(
const std::string &Val) {
103 setNeedsDefault(
false);
108 unsigned StartIdx = 1;
109 if (TailFoldTypes[0] ==
"disabled")
111 else if (TailFoldTypes[0] ==
"all")
113 else if (TailFoldTypes[0] ==
"default")
114 setNeedsDefault(
true);
115 else if (TailFoldTypes[0] ==
"simple")
122 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
123 if (TailFoldTypes[
I] ==
"reductions")
125 else if (TailFoldTypes[
I] ==
"recurrences")
127 else if (TailFoldTypes[
I] ==
"reverse")
129 else if (TailFoldTypes[
I] ==
"noreductions")
131 else if (TailFoldTypes[
I] ==
"norecurrences")
133 else if (TailFoldTypes[
I] ==
"noreverse")
145 return (
getBits(DefaultBits) & Required) == Required;
155 "Control the use of vectorisation using tail-folding for SVE where the"
156 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
157 "\ndisabled (Initial) No loop types will vectorize using "
159 "\ndefault (Initial) Uses the default tail-folding settings for "
161 "\nall (Initial) All legal loop types will vectorize using "
163 "\nsimple (Initial) Use tail-folding for simple loops (not "
164 "reductions or recurrences)"
165 "\nreductions Use tail-folding for loops containing reductions"
166 "\nnoreductions Inverse of above"
167 "\nrecurrences Use tail-folding for loops containing fixed order "
169 "\nnorecurrences Inverse of above"
170 "\nreverse Use tail-folding for loops requiring reversed "
172 "\nnoreverse Inverse of above"),
190 unsigned IID =
II->getIntrinsicID();
197 case Intrinsic::vscale:
199 case Intrinsic::masked_gather:
203 if (
I->getOperand(0)->getType()->isScalableTy())
206 case Intrinsic::masked_scatter:
207 case Intrinsic::masked_compressstore:
208 case Intrinsic::experimental_vector_histogram_add:
212 if (
I->getOperand(0)->getType()->isScalableTy())
220 if (
I->getType()->isScalableTy() ||
222 [](
const Value *V) { return V->getType()->isScalableTy(); }) ||
236 bool ConsiderZA,
bool ConsiderZT,
238 if (!ConsiderZA && !ConsiderZT && !ConsiderSM)
241 bool IsAlwaysInline =
F->hasFnAttribute(Attribute::AlwaysInline);
248 bool AssumeVScaleIsEquivalent =
249 F->getReturnType()->isScalableTy() ||
250 any_of(
F->getFunctionType()->params(),
251 [](
const Type *
T) { return T->isScalableTy(); });
259 if (CB->isInlineAsm())
262 if (
CallAttrs.callee().isSMEABIRoutine())
271 if (ConsiderSM &&
CallAttrs.callee().hasStreamingCompatibleInterface())
283 return isa<FixedVectorType>(V->getType());
289 (!AssumeVScaleIsEquivalent &&
301 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
302 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
303 FeatureStr.
split(Features,
",");
319 return F.hasFnAttribute(
"fmv-features");
371 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
372 CallAttrs.callee().hasStreamingInterfaceOrBody())
380 if (
CallAttrs.callee().hasStreamingBody()) {
385 bool ConsiderZA =
CallAttrs.requiresZASave();
386 bool ConsiderZT =
CallAttrs.requiresPreservingZT0() ||
387 CallAttrs.requiresPreservingAllZAState();
388 bool ConsiderSM =
CallAttrs.requiresSMChange();
393 if (ConsiderSM && (Caller->isStrictFP() || Callee->isStrictFP()))
418 auto FVTy = dyn_cast<FixedVectorType>(Ty);
420 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
429 unsigned DefaultCallPenalty)
const {
454 if (
F ==
Call.getCaller())
455 return ST->getCLOpts().call_penalty_sm_change * DefaultCallPenalty;
457 return ST->getCLOpts().inline_call_penalty_sm_change * DefaultCallPenalty;
460 return DefaultCallPenalty;
471 ST->isSVEorStreamingSVEAvailable() &&
472 !ST->disableMaximizeScalableBandwidth();
496 assert(Ty->isIntegerTy());
498 unsigned BitSize = Ty->getPrimitiveSizeInBits();
505 ImmVal =
Imm.sext((BitSize + 63) & ~0x3fU);
510 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
516 return std::max<InstructionCost>(1,
Cost);
523 assert(Ty->isIntegerTy());
525 unsigned BitSize = Ty->getPrimitiveSizeInBits();
531 unsigned ImmIdx = ~0U;
535 case Instruction::GetElementPtr:
540 case Instruction::Store:
543 case Instruction::Add:
544 case Instruction::Sub:
545 case Instruction::Mul:
546 case Instruction::UDiv:
547 case Instruction::SDiv:
548 case Instruction::URem:
549 case Instruction::SRem:
550 case Instruction::And:
551 case Instruction::Or:
552 case Instruction::Xor:
553 case Instruction::ICmp:
557 case Instruction::Shl:
558 case Instruction::LShr:
559 case Instruction::AShr:
563 case Instruction::Trunc:
564 case Instruction::ZExt:
565 case Instruction::SExt:
566 case Instruction::IntToPtr:
567 case Instruction::PtrToInt:
568 case Instruction::BitCast:
569 case Instruction::PHI:
570 case Instruction::Call:
571 case Instruction::Select:
572 case Instruction::Ret:
573 case Instruction::Load:
578 int NumConstants = (BitSize + 63) / 64;
591 assert(Ty->isIntegerTy());
593 unsigned BitSize = Ty->getPrimitiveSizeInBits();
602 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
608 case Intrinsic::sadd_with_overflow:
609 case Intrinsic::uadd_with_overflow:
610 case Intrinsic::ssub_with_overflow:
611 case Intrinsic::usub_with_overflow:
612 case Intrinsic::smul_with_overflow:
613 case Intrinsic::umul_with_overflow:
615 int NumConstants = (BitSize + 63) / 64;
622 case Intrinsic::experimental_stackmap:
623 if ((Idx < 2) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
626 case Intrinsic::experimental_patchpoint_void:
627 case Intrinsic::experimental_patchpoint:
628 if ((Idx < 4) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
631 case Intrinsic::experimental_gc_statepoint:
632 if ((Idx < 5) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
642 if (TyWidth == 32 || TyWidth == 64)
651 return ST->getMispredictionPenalty();
672 unsigned TotalHistCnts = 1;
682 unsigned EC = VTy->getElementCount().getKnownMinValue();
687 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
689 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
693 TotalHistCnts = EC / NaturalVectorWidth;
695 return InstructionCost(ST->getCLOpts().base_histcnt_cost * TotalHistCnts);
712 !
is_contained({Intrinsic::masked_load, Intrinsic::masked_store},
716 switch (ICA.
getID()) {
717 case Intrinsic::experimental_vector_histogram_add: {
724 case Intrinsic::clmul: {
729 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
733 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
738 -1,
nullptr,
nullptr) *
741 -1,
nullptr,
nullptr);
745 if (LT.second.SimpleTy == MVT::nxv2i64)
746 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
749 if (ST->hasSVE2() || ST->hasSME()) {
750 switch (LT.second.SimpleTy) {
765 if (LT.second.SimpleTy == MVT::nxv2i64)
769 switch (LT.second.SimpleTy) {
779 -1,
nullptr,
nullptr) *
782 -1,
nullptr,
nullptr));
791 return LT.first * 11;
793 return LT.first * 14;
800 case Intrinsic::smulh:
801 case Intrinsic::umulh: {
807 if (RetTy->getScalarSizeInBits() > 64)
821 return MulCost + ExtraCost;
826 return LT.first + ExtraCost;
842 return LT.first * Entry->Cost + ExtraCost;
846 case Intrinsic::umin:
847 case Intrinsic::umax:
848 case Intrinsic::smin:
849 case Intrinsic::smax: {
850 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
851 MVT::v8i16, MVT::v2i32, MVT::v4i32,
852 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
859 ICA.
getID() == Intrinsic::smin || ICA.
getID() == Intrinsic::smax;
860 EVT VT = TLI->getValueType(
DL, RetTy,
true);
861 if (VT == MVT::v2i8 || VT == MVT::v2i16 || VT == MVT::v4i8)
862 return LT.first * (IsSigned ? 5 : 3);
864 if (LT.second == MVT::v2i64)
870 case Intrinsic::scmp:
871 case Intrinsic::ucmp: {
873 {Intrinsic::scmp, MVT::i32, 3},
874 {Intrinsic::scmp, MVT::i64, 3},
875 {Intrinsic::scmp, MVT::v8i8, 3},
876 {Intrinsic::scmp, MVT::v16i8, 3},
877 {Intrinsic::scmp, MVT::v4i16, 3},
878 {Intrinsic::scmp, MVT::v8i16, 3},
879 {Intrinsic::scmp, MVT::v2i32, 3},
880 {Intrinsic::scmp, MVT::v4i32, 3},
881 {Intrinsic::scmp, MVT::v1i64, 3},
882 {Intrinsic::scmp, MVT::v2i64, 3},
888 return Entry->Cost * LT.first;
891 case Intrinsic::sadd_sat:
892 case Intrinsic::ssub_sat:
893 case Intrinsic::uadd_sat:
894 case Intrinsic::usub_sat: {
895 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
896 MVT::v8i16, MVT::v2i32, MVT::v4i32,
902 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
904 return LT.first * Instrs;
909 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
910 return LT.first * Instrs;
914 case Intrinsic::abs: {
915 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
916 MVT::v8i16, MVT::v2i32, MVT::v4i32,
917 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
918 MVT::nxv4i32, MVT::nxv2i64};
924 case Intrinsic::bswap: {
925 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
926 MVT::v4i32, MVT::v2i64};
929 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
934 case Intrinsic::fmuladd: {
939 (EltTy->
isHalfTy() && ST->hasFullFP16()))
943 case Intrinsic::stepvector: {
952 Cost += AddCost * (LT.first - 1);
956 case Intrinsic::vector_extract:
957 case Intrinsic::vector_insert: {
970 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
971 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
979 getTLI()->getTypeConversion(
C, SubVecVT);
981 getTLI()->getTypeConversion(
C, VecVT);
989 case Intrinsic::bitreverse: {
991 {Intrinsic::bitreverse, MVT::i32, 1},
992 {Intrinsic::bitreverse, MVT::i64, 1},
993 {Intrinsic::bitreverse, MVT::v8i8, 1},
994 {Intrinsic::bitreverse, MVT::v16i8, 1},
995 {Intrinsic::bitreverse, MVT::v4i16, 2},
996 {Intrinsic::bitreverse, MVT::v8i16, 2},
997 {Intrinsic::bitreverse, MVT::v2i32, 2},
998 {Intrinsic::bitreverse, MVT::v4i32, 2},
999 {Intrinsic::bitreverse, MVT::v1i64, 2},
1000 {Intrinsic::bitreverse, MVT::v2i64, 2},
1008 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
1009 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
1010 return LegalisationCost.first * Entry->Cost + 1;
1012 return LegalisationCost.first * Entry->Cost;
1016 case Intrinsic::ctpop: {
1018 MVT MTy = LT.second;
1020 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
1023 return LT.first + ExtraCost;
1025 if (!ST->hasNEON()) {
1055 RetTy->getScalarSizeInBits()
1058 return LT.first * Entry->Cost + ExtraCost;
1062 case Intrinsic::sadd_with_overflow:
1063 case Intrinsic::uadd_with_overflow:
1064 case Intrinsic::ssub_with_overflow:
1065 case Intrinsic::usub_with_overflow:
1066 case Intrinsic::smul_with_overflow:
1067 case Intrinsic::umul_with_overflow: {
1069 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
1070 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
1071 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
1072 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
1073 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
1074 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
1075 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
1076 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
1077 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
1078 {Intrinsic::usub_with_overflow, MVT::i8, 3},
1079 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
1080 {Intrinsic::usub_with_overflow, MVT::i16, 3},
1081 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
1082 {Intrinsic::usub_with_overflow, MVT::i32, 1},
1083 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
1084 {Intrinsic::usub_with_overflow, MVT::i64, 1},
1085 {Intrinsic::smul_with_overflow, MVT::i8, 5},
1086 {Intrinsic::umul_with_overflow, MVT::i8, 4},
1087 {Intrinsic::smul_with_overflow, MVT::i16, 5},
1088 {Intrinsic::umul_with_overflow, MVT::i16, 4},
1089 {Intrinsic::smul_with_overflow, MVT::i32, 2},
1090 {Intrinsic::umul_with_overflow, MVT::i32, 2},
1091 {Intrinsic::smul_with_overflow, MVT::i64, 3},
1092 {Intrinsic::umul_with_overflow, MVT::i64, 3},
1094 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
1101 case Intrinsic::fptosi_sat:
1102 case Intrinsic::fptoui_sat: {
1105 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1107 EVT MTy = TLI->getValueType(
DL, RetTy);
1110 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
1111 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
1112 LT.second == MVT::v2f64)) {
1114 (LT.second == MVT::f64 && MTy == MVT::i32) ||
1115 (LT.second == MVT::f32 && MTy == MVT::i64)))
1124 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
1131 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
1132 (LT.second == MVT::f16 && MTy == MVT::i64) ||
1133 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
1147 if ((LT.second.getScalarType() == MVT::f32 ||
1148 LT.second.getScalarType() == MVT::f64 ||
1149 LT.second.getScalarType() == MVT::f16) &&
1152 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1153 if (LT.second.isVector())
1154 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1158 LegalTy, {LegalTy, LegalTy});
1162 LegalTy, {LegalTy, LegalTy});
1164 return LT.first *
Cost +
1165 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1171 RetTy = RetTy->getScalarType();
1172 if (LT.second.isVector()) {
1190 return LT.first *
Cost;
1192 case Intrinsic::fshl:
1193 case Intrinsic::fshr: {
1202 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1203 (RetTy->getPrimitiveSizeInBits() == 32 ||
1204 RetTy->getPrimitiveSizeInBits() == 64)) {
1217 {Intrinsic::fshl, MVT::v4i32, 2},
1218 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1219 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1220 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1226 return LegalisationCost.first * Entry->Cost;
1230 if (!RetTy->isIntegerTy())
1235 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1236 RetTy->getScalarSizeInBits() < 64) ||
1237 (RetTy->getScalarSizeInBits() % 64 != 0);
1238 unsigned ExtraCost = HigherCost ? 1 : 0;
1239 if (RetTy->getScalarSizeInBits() == 32 ||
1240 RetTy->getScalarSizeInBits() == 64)
1243 else if (HigherCost)
1247 return TyL.first + ExtraCost;
1249 case Intrinsic::get_active_lane_mask: {
1251 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1253 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1256 if (RetTy->isScalableTy()) {
1257 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1267 if (ST->hasSVE2p1() || ST->hasSME2()) {
1279 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1282 return Cost + (SplitCost * (
Cost - 1));
1297 case Intrinsic::experimental_vector_match: {
1298 if (!ST->hasSVE2() || !ST->isSVEAvailable())
1304 unsigned SearchSize = NeedleTy->getNumElements();
1305 if (SearchSize <= 2)
1310 {MVT::nxv8i16, MVT::nxv16i8, MVT::v8i16, MVT::v16i8, MVT::v8i8},
1314 unsigned ElementSizeInBits = SearchVT.getScalarSizeInBits();
1320 unsigned MatchesRequiredForNeedle =
1332 return Cost * LegalParts * MatchesRequiredForNeedle;
1334 case Intrinsic::cttz: {
1336 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
1337 return LT.first * 2;
1338 if (LT.second == MVT::v4i16 || LT.second == MVT::v8i16 ||
1339 LT.second == MVT::v2i32 || LT.second == MVT::v4i32)
1340 return LT.first * 3;
1343 case Intrinsic::experimental_cttz_elts: {
1353 case Intrinsic::loop_dependence_raw_mask:
1354 case Intrinsic::loop_dependence_war_mask: {
1356 if (ST->hasSVE2() || ST->hasSME()) {
1357 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1358 unsigned EltSizeInBytes =
1368 case Intrinsic::experimental_vector_extract_last_active:
1369 if (ST->isSVEorStreamingSVEAvailable()) {
1375 case Intrinsic::pow: {
1378 EVT VT = getTLI()->getValueType(
DL, RetTy);
1379 RTLIB::Libcall LC = RTLIB::getPOW(VT);
1380 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1395 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1396 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1406 return (Sqrt * 2) +
FMul;
1417 case Intrinsic::sqrt:
1418 case Intrinsic::fabs:
1419 case Intrinsic::ceil:
1420 case Intrinsic::floor:
1421 case Intrinsic::nearbyint:
1422 case Intrinsic::round:
1423 case Intrinsic::rint:
1424 case Intrinsic::roundeven:
1425 case Intrinsic::trunc:
1426 case Intrinsic::minnum:
1427 case Intrinsic::maxnum:
1428 case Intrinsic::minimum:
1429 case Intrinsic::maximum: {
1447 auto RequiredType =
II.getType();
1450 assert(PN &&
"Expected Phi Node!");
1453 if (!PN->hasOneUse())
1454 return std::nullopt;
1456 for (
Value *IncValPhi : PN->incoming_values()) {
1459 Reinterpret->getIntrinsicID() !=
1460 Intrinsic::aarch64_sve_convert_to_svbool ||
1461 RequiredType != Reinterpret->getArgOperand(0)->getType())
1462 return std::nullopt;
1470 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1472 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1545 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1550 return GoverningPredicateIdx;
1555 GoverningPredicateIdx = Index;
1577 return UndefIntrinsic;
1582 UndefIntrinsic = IID;
1609 return CmpPredicate;
1614 CmpPredicate = Pred;
1630 return ResultLanes == InactiveLanesTakenFromOperand;
1635 return OperandIdxForInactiveLanes;
1639 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1640 ResultLanes = InactiveLanesTakenFromOperand;
1641 OperandIdxForInactiveLanes = Index;
1646 return ResultLanes == InactiveLanesAreNotDefined;
1650 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1651 ResultLanes = InactiveLanesAreNotDefined;
1656 return ResultLanes == InactiveLanesAreUnused;
1660 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1661 ResultLanes = InactiveLanesAreUnused;
1671 ResultIsZeroInitialized =
true;
1682 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1687 return OperandIdxWithNoActiveLanes;
1692 OperandIdxWithNoActiveLanes = Index;
1697 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1700 unsigned IROpcode = 0;
1703 enum PredicationStyle {
1705 InactiveLanesTakenFromOperand,
1706 InactiveLanesAreNotDefined,
1707 InactiveLanesAreUnused
1710 bool ResultIsZeroInitialized =
false;
1711 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1712 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1720 return !isa<ScalableVectorType>(V->getType());
1728 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1729 case Intrinsic::aarch64_sve_fcvt_f16f32:
1730 case Intrinsic::aarch64_sve_fcvt_f16f64:
1731 case Intrinsic::aarch64_sve_fcvt_f32f16:
1732 case Intrinsic::aarch64_sve_fcvt_f32f64:
1733 case Intrinsic::aarch64_sve_fcvt_f64f16:
1734 case Intrinsic::aarch64_sve_fcvt_f64f32:
1735 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1736 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1737 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1738 case Intrinsic::aarch64_sve_fcvtzs:
1739 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1740 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1741 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1742 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1743 case Intrinsic::aarch64_sve_fcvtzu:
1744 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1745 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1746 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1747 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1748 case Intrinsic::aarch64_sve_revb:
1749 case Intrinsic::aarch64_sve_revh:
1750 case Intrinsic::aarch64_sve_revw:
1751 case Intrinsic::aarch64_sve_revd:
1752 case Intrinsic::aarch64_sve_scvtf:
1753 case Intrinsic::aarch64_sve_scvtf_f16i32:
1754 case Intrinsic::aarch64_sve_scvtf_f16i64:
1755 case Intrinsic::aarch64_sve_scvtf_f32i64:
1756 case Intrinsic::aarch64_sve_scvtf_f64i32:
1757 case Intrinsic::aarch64_sve_ucvtf:
1758 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1759 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1760 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1761 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1764 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1765 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1766 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1767 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1770 case Intrinsic::aarch64_sve_fabd:
1772 case Intrinsic::aarch64_sve_fadd:
1775 case Intrinsic::aarch64_sve_fdiv:
1778 case Intrinsic::aarch64_sve_fmax:
1780 case Intrinsic::aarch64_sve_fmaxnm:
1782 case Intrinsic::aarch64_sve_fmin:
1784 case Intrinsic::aarch64_sve_fminnm:
1786 case Intrinsic::aarch64_sve_fmla:
1788 case Intrinsic::aarch64_sve_fmls:
1790 case Intrinsic::aarch64_sve_fmul:
1793 case Intrinsic::aarch64_sve_fmulx:
1795 case Intrinsic::aarch64_sve_fnmla:
1797 case Intrinsic::aarch64_sve_fnmls:
1799 case Intrinsic::aarch64_sve_fsub:
1802 case Intrinsic::aarch64_sve_add:
1805 case Intrinsic::aarch64_sve_mla:
1807 case Intrinsic::aarch64_sve_mls:
1809 case Intrinsic::aarch64_sve_mul:
1812 case Intrinsic::aarch64_sve_sabd:
1814 case Intrinsic::aarch64_sve_sdiv:
1817 case Intrinsic::aarch64_sve_smax:
1819 case Intrinsic::aarch64_sve_smin:
1821 case Intrinsic::aarch64_sve_smulh:
1823 case Intrinsic::aarch64_sve_sub:
1826 case Intrinsic::aarch64_sve_uabd:
1828 case Intrinsic::aarch64_sve_udiv:
1831 case Intrinsic::aarch64_sve_umax:
1833 case Intrinsic::aarch64_sve_umin:
1835 case Intrinsic::aarch64_sve_umulh:
1837 case Intrinsic::aarch64_sve_asr:
1840 case Intrinsic::aarch64_sve_lsl:
1843 case Intrinsic::aarch64_sve_lsr:
1846 case Intrinsic::aarch64_sve_and:
1849 case Intrinsic::aarch64_sve_bic:
1851 case Intrinsic::aarch64_sve_eor:
1854 case Intrinsic::aarch64_sve_orr:
1857 case Intrinsic::aarch64_sve_shsub:
1859 case Intrinsic::aarch64_sve_shsubr:
1861 case Intrinsic::aarch64_sve_sqrshl:
1863 case Intrinsic::aarch64_sve_sqshl:
1865 case Intrinsic::aarch64_sve_sqsub:
1867 case Intrinsic::aarch64_sve_srshl:
1869 case Intrinsic::aarch64_sve_uhsub:
1871 case Intrinsic::aarch64_sve_uhsubr:
1873 case Intrinsic::aarch64_sve_uqrshl:
1875 case Intrinsic::aarch64_sve_uqshl:
1877 case Intrinsic::aarch64_sve_uqsub:
1879 case Intrinsic::aarch64_sve_urshl:
1882 case Intrinsic::aarch64_sve_add_u:
1885 case Intrinsic::aarch64_sve_and_u:
1888 case Intrinsic::aarch64_sve_asr_u:
1891 case Intrinsic::aarch64_sve_eor_u:
1894 case Intrinsic::aarch64_sve_fadd_u:
1897 case Intrinsic::aarch64_sve_fdiv_u:
1900 case Intrinsic::aarch64_sve_fmul_u:
1903 case Intrinsic::aarch64_sve_fsub_u:
1906 case Intrinsic::aarch64_sve_lsl_u:
1909 case Intrinsic::aarch64_sve_lsr_u:
1912 case Intrinsic::aarch64_sve_mul_u:
1915 case Intrinsic::aarch64_sve_orr_u:
1918 case Intrinsic::aarch64_sve_sdiv_u:
1921 case Intrinsic::aarch64_sve_sub_u:
1924 case Intrinsic::aarch64_sve_udiv_u:
1928 case Intrinsic::aarch64_sve_addqv:
1929 case Intrinsic::aarch64_sve_bic_z:
1930 case Intrinsic::aarch64_sve_brka_z:
1931 case Intrinsic::aarch64_sve_brkb_z:
1932 case Intrinsic::aarch64_sve_brkn_z:
1933 case Intrinsic::aarch64_sve_brkpa_z:
1934 case Intrinsic::aarch64_sve_brkpb_z:
1935 case Intrinsic::aarch64_sve_cntp:
1936 case Intrinsic::aarch64_sve_compact:
1937 case Intrinsic::aarch64_sve_eorv:
1938 case Intrinsic::aarch64_sve_eorqv:
1939 case Intrinsic::aarch64_sve_nand_z:
1940 case Intrinsic::aarch64_sve_nor_z:
1941 case Intrinsic::aarch64_sve_orn_z:
1942 case Intrinsic::aarch64_sve_orv:
1943 case Intrinsic::aarch64_sve_orqv:
1944 case Intrinsic::aarch64_sve_pnext:
1945 case Intrinsic::aarch64_sve_rdffr_z:
1946 case Intrinsic::aarch64_sve_saddv:
1947 case Intrinsic::aarch64_sve_uaddv:
1948 case Intrinsic::aarch64_sve_umaxv:
1949 case Intrinsic::aarch64_sve_umaxqv:
1950 case Intrinsic::aarch64_sve_facge:
1951 case Intrinsic::aarch64_sve_facgt:
1952 case Intrinsic::aarch64_sve_ld1:
1953 case Intrinsic::aarch64_sve_ld1_gather:
1954 case Intrinsic::aarch64_sve_ld1_gather_index:
1955 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1956 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1957 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1958 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1959 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1960 case Intrinsic::aarch64_sve_ld1q_gather_index:
1961 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1962 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1963 case Intrinsic::aarch64_sve_ld1ro:
1964 case Intrinsic::aarch64_sve_ld1rq:
1965 case Intrinsic::aarch64_sve_ld1udq:
1966 case Intrinsic::aarch64_sve_ld1uwq:
1967 case Intrinsic::aarch64_sve_ld2_sret:
1968 case Intrinsic::aarch64_sve_ld2q_sret:
1969 case Intrinsic::aarch64_sve_ld3_sret:
1970 case Intrinsic::aarch64_sve_ld3q_sret:
1971 case Intrinsic::aarch64_sve_ld4_sret:
1972 case Intrinsic::aarch64_sve_ld4q_sret:
1973 case Intrinsic::aarch64_sve_ldff1:
1974 case Intrinsic::aarch64_sve_ldff1_gather:
1975 case Intrinsic::aarch64_sve_ldff1_gather_index:
1976 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1977 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1978 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1979 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1980 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1981 case Intrinsic::aarch64_sve_ldnf1:
1982 case Intrinsic::aarch64_sve_ldnt1:
1983 case Intrinsic::aarch64_sve_ldnt1_gather:
1984 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1985 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1986 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1989 case Intrinsic::aarch64_sve_and_z:
1992 case Intrinsic::aarch64_sve_orr_z:
1995 case Intrinsic::aarch64_sve_eor_z:
1999 case Intrinsic::aarch64_sve_cmpeq:
2000 case Intrinsic::aarch64_sve_cmpeq_wide:
2003 case Intrinsic::aarch64_sve_cmpge:
2004 case Intrinsic::aarch64_sve_cmpge_wide:
2007 case Intrinsic::aarch64_sve_cmpgt:
2008 case Intrinsic::aarch64_sve_cmpgt_wide:
2011 case Intrinsic::aarch64_sve_cmphi:
2012 case Intrinsic::aarch64_sve_cmphi_wide:
2015 case Intrinsic::aarch64_sve_cmphs:
2016 case Intrinsic::aarch64_sve_cmphs_wide:
2019 case Intrinsic::aarch64_sve_cmple_wide:
2022 case Intrinsic::aarch64_sve_cmplo_wide:
2025 case Intrinsic::aarch64_sve_cmpls_wide:
2028 case Intrinsic::aarch64_sve_cmplt_wide:
2031 case Intrinsic::aarch64_sve_cmpne:
2032 case Intrinsic::aarch64_sve_cmpne_wide:
2035 case Intrinsic::aarch64_sve_fcmpeq:
2038 case Intrinsic::aarch64_sve_fcmpge:
2041 case Intrinsic::aarch64_sve_fcmpgt:
2044 case Intrinsic::aarch64_sve_fcmpne:
2047 case Intrinsic::aarch64_sve_fcmpuo:
2051 case Intrinsic::aarch64_sve_prf:
2052 case Intrinsic::aarch64_sve_prfb_gather_index:
2053 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
2054 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
2055 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
2056 case Intrinsic::aarch64_sve_prfd_gather_index:
2057 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
2058 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
2059 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
2060 case Intrinsic::aarch64_sve_prfh_gather_index:
2061 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
2062 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
2063 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
2064 case Intrinsic::aarch64_sve_prfw_gather_index:
2065 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
2066 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
2067 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
2070 case Intrinsic::aarch64_sve_st1_scatter:
2071 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
2072 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
2073 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
2074 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
2075 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
2076 case Intrinsic::aarch64_sve_st1dq:
2077 case Intrinsic::aarch64_sve_st1q_scatter_index:
2078 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
2079 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
2080 case Intrinsic::aarch64_sve_st1wq:
2081 case Intrinsic::aarch64_sve_stnt1:
2082 case Intrinsic::aarch64_sve_stnt1_scatter:
2083 case Intrinsic::aarch64_sve_stnt1_scatter_index:
2084 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
2085 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
2087 case Intrinsic::aarch64_sve_st2:
2088 case Intrinsic::aarch64_sve_st2q:
2090 case Intrinsic::aarch64_sve_st3:
2091 case Intrinsic::aarch64_sve_st3q:
2093 case Intrinsic::aarch64_sve_st4:
2094 case Intrinsic::aarch64_sve_st4q:
2102 Value *UncastedPred;
2108 Pred = UncastedPred;
2114 if (OrigPredTy->getMinNumElements() <=
2116 ->getMinNumElements())
2117 Pred = UncastedPred;
2121 return C &&
C->isAllOnesValue();
2128 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
2129 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
2137static std::optional<Instruction *>
2144 Value *Op1 =
II.getOperand(1);
2145 Value *Op2 =
II.getOperand(2);
2170 Value *NarrowOp1, *NarrowOp2;
2181 else if (SimpleNarrow == NarrowOp1)
2183 else if (SimpleNarrow == NarrowOp2)
2188 SimpleNarrow->
getType(), SimpleNarrow);
2197 return std::nullopt;
2208 if (SimpleII == Inactive)
2216static std::optional<Instruction *>
2220 assert((
Opc == Instruction::ICmp ||
Opc == Instruction::FCmp) &&
2221 "Expected a compare operation!");
2228 Opc == Instruction::ICmp &&
LHS->getType() !=
RHS->getType();
2229 assert((IsWideICmp ||
LHS->getType() ==
RHS->getType()) &&
2230 "Unexpected wide compare!");
2246 const APInt *LHSVal, *RHSVal;
2248 return std::nullopt;
2271 return std::nullopt;
2285static std::optional<Instruction *>
2289 return std::nullopt;
2318 II.setCalledFunction(NewDecl);
2324 return std::nullopt;
2335 if (
Opc == Instruction::FCmp ||
Opc == Instruction::ICmp)
2338 return std::nullopt;
2350static std::optional<Instruction *>
2352 auto m_ConvertToSVBool = [](
auto P) {
2356 Intrinsic::aarch64_sve_convert_from_svbool;
2379 return std::nullopt;
2383 case Intrinsic::aarch64_sve_and_z:
2384 case Intrinsic::aarch64_sve_bic_z:
2385 case Intrinsic::aarch64_sve_eor_z:
2386 case Intrinsic::aarch64_sve_nand_z:
2387 case Intrinsic::aarch64_sve_nor_z:
2388 case Intrinsic::aarch64_sve_orn_z:
2389 case Intrinsic::aarch64_sve_orr_z:
2392 return std::nullopt;
2395 Value *BinOpPred = BinOp->getOperand(0);
2396 Value *BinOpOp1 = BinOp->getOperand(1);
2397 Value *BinOpOp2 = BinOp->getOperand(2);
2399 Value *NarrowBinOpPred;
2401 return std::nullopt;
2403 Value *NarrowBinOpOp1 =
2405 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2406 if (BinOpOp1 != BinOpOp2)
2410 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2414static std::optional<Instruction *>
2421 return BinOpCombine;
2426 return std::nullopt;
2429 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2438 if (CursorVTy->getElementCount().getKnownMinValue() <
2439 IVTy->getElementCount().getKnownMinValue())
2443 if (Cursor->getType() == IVTy)
2444 EarliestReplacement = Cursor;
2449 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2450 Intrinsic::aarch64_sve_convert_to_svbool ||
2451 IntrinsicCursor->getIntrinsicID() ==
2452 Intrinsic::aarch64_sve_convert_from_svbool))
2455 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2456 Cursor = IntrinsicCursor->getOperand(0);
2461 if (!EarliestReplacement)
2462 return std::nullopt;
2470 auto *OpPredicate =
II.getOperand(0);
2487 II.getArgOperand(2));
2493 return std::nullopt;
2497 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2506 II.getArgOperand(0));
2515 if (!
II.hasOneUse())
2516 return std::nullopt;
2519 return std::nullopt;
2522 switch (
II.getIntrinsicID()) {
2523 case Intrinsic::aarch64_sve_cmpne:
2524 IID = Intrinsic::aarch64_sve_cmpeq;
2526 case Intrinsic::aarch64_sve_cmpne_wide:
2527 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2529 case Intrinsic::aarch64_sve_cmpeq:
2530 IID = Intrinsic::aarch64_sve_cmpne;
2532 case Intrinsic::aarch64_sve_cmpeq_wide:
2533 IID = Intrinsic::aarch64_sve_cmpne_wide;
2536 return std::nullopt;
2541 IID,
II.getOperand(1)->getType(),
2542 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2554 return std::nullopt;
2556 for (
auto *U :
II.users()) {
2559 Type *Ty =
II.getOperand(1)->getType();
2564 Intrinsic::aarch64_sve_umin, Ty,
2565 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2571 return std::nullopt;
2585 return std::nullopt;
2590 if (!SplatValue || !SplatValue->isZero())
2591 return std::nullopt;
2596 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2597 return std::nullopt;
2601 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2602 return std::nullopt;
2605 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2606 return std::nullopt;
2611 return std::nullopt;
2614 return std::nullopt;
2618 return std::nullopt;
2622 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2623 return std::nullopt;
2625 unsigned NumElts = VecTy->getNumElements();
2626 unsigned PredicateBits = 0;
2629 for (
unsigned I = 0;
I < NumElts; ++
I) {
2632 return std::nullopt;
2634 PredicateBits |= 1 << (
I * (16 / NumElts));
2638 if (PredicateBits == 0) {
2640 PFalse->takeName(&
II);
2646 for (
unsigned I = 0;
I < 16; ++
I)
2647 if ((PredicateBits & (1 <<
I)) != 0)
2650 unsigned PredSize = Mask & -Mask;
2655 for (
unsigned I = 0;
I < 16;
I += PredSize)
2656 if ((PredicateBits & (1 <<
I)) == 0)
2657 return std::nullopt;
2659 auto *ConvertToSVBool =
2662 auto *ConvertFromSVBool =
2664 II.getType(), ConvertToSVBool);
2672 Value *Pg =
II.getArgOperand(0);
2673 Value *Vec =
II.getArgOperand(1);
2674 auto IntrinsicID =
II.getIntrinsicID();
2675 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2687 auto OpC = OldBinOp->getOpcode();
2693 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2699 if (IsAfter &&
C &&
C->isNullValue()) {
2703 Extract->insertBefore(
II.getIterator());
2704 Extract->takeName(&
II);
2710 return std::nullopt;
2712 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2713 return std::nullopt;
2715 const auto PTruePattern =
2721 return std::nullopt;
2723 unsigned Idx = MinNumElts - 1;
2733 if (Idx >= PgVTy->getMinNumElements())
2734 return std::nullopt;
2739 Extract->insertBefore(
II.getIterator());
2740 Extract->takeName(&
II);
2753 Value *Pg =
II.getArgOperand(0);
2755 Value *Vec =
II.getArgOperand(2);
2758 if (!Ty->isIntegerTy())
2759 return std::nullopt;
2764 return std::nullopt;
2781 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2796static std::optional<Instruction *>
2800 if (
Pattern == AArch64SVEPredPattern::all) {
2809 return MinNumElts && NumElts >= MinNumElts
2811 II, ConstantInt::get(
II.getType(), MinNumElts)))
2815static std::optional<Instruction *>
2818 if (!ST->isStreaming())
2819 return std::nullopt;
2831 Value *PgVal =
II.getArgOperand(0);
2832 Value *OpVal =
II.getArgOperand(1);
2836 if (PgVal == OpVal &&
2837 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2838 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2853 return std::nullopt;
2857 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2858 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2872 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2873 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2874 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2875 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2876 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2877 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2878 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2879 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2880 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2881 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2882 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2883 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2884 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2894 return std::nullopt;
2897template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2898static std::optional<Instruction *>
2900 bool MergeIntoAddendOp) {
2902 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2903 if (MergeIntoAddendOp) {
2904 AddendOp =
II.getOperand(1);
2905 Mul =
II.getOperand(2);
2907 AddendOp =
II.getOperand(2);
2908 Mul =
II.getOperand(1);
2913 return std::nullopt;
2915 if (!
Mul->hasOneUse())
2916 return std::nullopt;
2919 if (
II.getType()->isFPOrFPVectorTy()) {
2924 return std::nullopt;
2926 return std::nullopt;
2931 if (MergeIntoAddendOp)
2941static std::optional<Instruction *>
2943 Value *Pred =
II.getOperand(0);
2944 Value *PtrOp =
II.getOperand(1);
2945 Type *VecTy =
II.getType();
2960static std::optional<Instruction *>
2962 Value *VecOp =
II.getOperand(0);
2963 Value *Pred =
II.getOperand(1);
2964 Value *PtrOp =
II.getOperand(2);
2980 case Intrinsic::aarch64_sve_fmul_u:
2981 return Instruction::BinaryOps::FMul;
2982 case Intrinsic::aarch64_sve_fadd_u:
2983 return Instruction::BinaryOps::FAdd;
2984 case Intrinsic::aarch64_sve_fsub_u:
2985 return Instruction::BinaryOps::FSub;
2987 return Instruction::BinaryOpsEnd;
2991static std::optional<Instruction *>
2994 if (
II.isStrictFP())
2995 return std::nullopt;
2997 auto *OpPredicate =
II.getOperand(0);
2999 if (BinOpCode == Instruction::BinaryOpsEnd ||
3001 return std::nullopt;
3003 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
3007static std::optional<Instruction *>
3009 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
3010 "Expected MLA_U intrinsic");
3011 Value *Acc =
II.getArgOperand(1);
3012 Value *MulOp0 =
II.getArgOperand(2);
3013 Value *MulOp1 =
II.getArgOperand(3);
3028 II.setArgOperand(2, MulOp1);
3029 II.setArgOperand(3, MulOp0);
3033 return std::nullopt;
3036static std::optional<Instruction *>
3038 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
3039 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
3040 "Expected SADALP or UADALP intrinsic");
3046 return std::nullopt;
3050 return std::nullopt;
3054 II.getIntrinsicID(), {II.getType()},
3055 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
3065 Intrinsic::aarch64_sve_mla>(
3069 Intrinsic::aarch64_sve_mad>(
3072 return std::nullopt;
3075static std::optional<Instruction *>
3079 Intrinsic::aarch64_sve_fmla>(IC,
II,
3084 Intrinsic::aarch64_sve_fmad>(IC,
II,
3089 Intrinsic::aarch64_sve_fmla>(IC,
II,
3092 return std::nullopt;
3095static std::optional<Instruction *>
3099 Intrinsic::aarch64_sve_fmla>(IC,
II,
3104 Intrinsic::aarch64_sve_fmad>(IC,
II,
3109 Intrinsic::aarch64_sve_fmla_u>(
3115static std::optional<Instruction *>
3119 Intrinsic::aarch64_sve_fmls>(IC,
II,
3124 Intrinsic::aarch64_sve_fnmsb>(
3129 Intrinsic::aarch64_sve_fmls>(IC,
II,
3132 return std::nullopt;
3135static std::optional<Instruction *>
3139 Intrinsic::aarch64_sve_fmls>(IC,
II,
3144 Intrinsic::aarch64_sve_fnmsb>(
3149 Intrinsic::aarch64_sve_fmls_u>(
3158 Intrinsic::aarch64_sve_mls>(
3161 return std::nullopt;
3166 Value *UnpackArg =
II.getArgOperand(0);
3168 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
3169 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
3182 return std::nullopt;
3186 auto *OpVal =
II.getOperand(0);
3187 auto *OpIndices =
II.getOperand(1);
3194 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
3195 return std::nullopt;
3210 Type *RetTy =
II.getType();
3211 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
3212 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
3216 if ((
match(
II.getArgOperand(0),
3223 if (TyA ==
B->getType() &&
3228 TyA->getMinNumElements());
3234 return std::nullopt;
3242 if (
match(
II.getArgOperand(0),
3247 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
3249 return std::nullopt;
3252static std::optional<Instruction *>
3254 Value *Mask =
II.getOperand(0);
3255 Value *BasePtr =
II.getOperand(1);
3256 Value *Index =
II.getOperand(2);
3267 BasePtr->getPointerAlignment(
II.getDataLayout());
3270 BasePtr, IndexBase);
3277 return std::nullopt;
3280static std::optional<Instruction *>
3282 Value *Val =
II.getOperand(0);
3283 Value *Mask =
II.getOperand(1);
3284 Value *BasePtr =
II.getOperand(2);
3285 Value *Index =
II.getOperand(3);
3295 BasePtr->getPointerAlignment(
II.getDataLayout());
3298 BasePtr, IndexBase);
3304 return std::nullopt;
3310 Value *Pred =
II.getOperand(0);
3311 Value *Vec =
II.getOperand(1);
3312 Value *DivVec =
II.getOperand(2);
3316 if (!SplatConstantInt)
3317 return std::nullopt;
3321 if (DivisorValue == -1)
3322 return std::nullopt;
3323 if (DivisorValue == 1)
3329 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3336 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3338 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3342 return std::nullopt;
3346 size_t VecSize = Vec.
size();
3351 size_t HalfVecSize = VecSize / 2;
3355 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3363 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3381 return std::nullopt;
3388 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3389 CurrentInsertElt = InsertElt->getOperand(0);
3395 return std::nullopt;
3399 for (
size_t I = 0;
I < Elts.
size();
I++) {
3400 if (Elts[
I] ==
nullptr)
3405 if (InsertEltChain ==
nullptr)
3406 return std::nullopt;
3412 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3413 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3414 IIScalableTy->getMinNumElements() /
3419 auto *WideShuffleMaskTy =
3430 auto NarrowBitcast =
3443 return std::nullopt;
3448 Value *Pred =
II.getOperand(0);
3449 Value *Vec =
II.getOperand(1);
3450 Value *Shift =
II.getOperand(2);
3453 Value *AbsPred, *MergedValue;
3459 return std::nullopt;
3467 return std::nullopt;
3472 return std::nullopt;
3475 {
II.getType()}, {Pred, Vec, Shift});
3482 Value *Vec =
II.getOperand(0);
3487 return std::nullopt;
3490static std::optional<Instruction *>
3492 unsigned LookaheadThreshold) {
3494 auto *NI =
II.getNextNode();
3496 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3498 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3499 auto *NIBB = NI->getParent();
3500 NI = NI->getNextNode();
3502 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3503 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3509 if (NextII &&
II.isIdenticalTo(NextII))
3512 return std::nullopt;
3520 {II.getType(), II.getOperand(0)->getType()},
3521 {II.getOperand(0), II.getOperand(1)}));
3528 if (PredPattern == AArch64SVEPredPattern::all ||
3529 PredPattern == AArch64SVEPredPattern::pow2)
3531 return std::nullopt;
3537 Value *Passthru =
II.getOperand(0);
3545 auto *Mask = ConstantInt::get(Ty, MaskValue);
3551 return std::nullopt;
3554static std::optional<Instruction *>
3561 return std::nullopt;
3567 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3577 UMinID,
II.getType(), {Pg, NewUMin, ConstantInt::get(II.getType(), 1)});
3587 return std::nullopt;
3593 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3601 return std::nullopt;
3604 II.getType(), {Pg, A, B});
3606 UMinID,
II.getType(), {Pg, NewOrr, ConstantInt::get(II.getType(), 1)});
3615 constexpr Intrinsic::ID CmphsID = Intrinsic::aarch64_sve_cmphs;
3620 Value *
A, *PgLHS, *PgRHS;
3626 !
LHS->hasOneUser() || !
RHS->hasOneUser())
3627 return std::nullopt;
3630 if (ConstB > ConstA)
3636 if (PgLHS != PgRHS || (Pg !=
LHS && Pg !=
RHS && Pg != PgLHS))
3637 return std::nullopt;
3639 Type *VecTy =
A->getType();
3643 Constant *Limit = ConstantInt::get(VecTy, ConstA - ConstB);
3650std::optional<Instruction *>
3661 case Intrinsic::aarch64_dmb:
3663 case Intrinsic::aarch64_neon_fmaxnm:
3664 case Intrinsic::aarch64_neon_fminnm:
3666 case Intrinsic::aarch64_sve_convert_from_svbool:
3668 case Intrinsic::aarch64_sve_dup:
3670 case Intrinsic::aarch64_sve_dup_x:
3672 case Intrinsic::aarch64_sve_cmpeq:
3673 case Intrinsic::aarch64_sve_cmpeq_wide:
3675 case Intrinsic::aarch64_sve_cmpne:
3676 case Intrinsic::aarch64_sve_cmpne_wide:
3678 case Intrinsic::aarch64_sve_rdffr:
3680 case Intrinsic::aarch64_sve_lasta:
3681 case Intrinsic::aarch64_sve_lastb:
3683 case Intrinsic::aarch64_sve_clasta_n:
3684 case Intrinsic::aarch64_sve_clastb_n:
3686 case Intrinsic::aarch64_sve_cntd:
3688 case Intrinsic::aarch64_sve_cntw:
3690 case Intrinsic::aarch64_sve_cnth:
3692 case Intrinsic::aarch64_sve_cntb:
3694 case Intrinsic::aarch64_sme_cntsd:
3696 case Intrinsic::aarch64_sve_ptest_any:
3697 case Intrinsic::aarch64_sve_ptest_first:
3698 case Intrinsic::aarch64_sve_ptest_last:
3700 case Intrinsic::aarch64_sve_fadd:
3702 case Intrinsic::aarch64_sve_fadd_u:
3704 case Intrinsic::aarch64_sve_fmul_u:
3706 case Intrinsic::aarch64_sve_fsub:
3708 case Intrinsic::aarch64_sve_fsub_u:
3710 case Intrinsic::aarch64_sve_add:
3712 case Intrinsic::aarch64_sve_add_u:
3714 Intrinsic::aarch64_sve_mla_u>(
3716 case Intrinsic::aarch64_sve_mla_u:
3718 case Intrinsic::aarch64_sve_sadalp:
3719 case Intrinsic::aarch64_sve_uadalp:
3721 case Intrinsic::aarch64_sve_sub:
3723 case Intrinsic::aarch64_sve_sub_u:
3725 Intrinsic::aarch64_sve_mls_u>(
3727 case Intrinsic::aarch64_sve_tbl:
3729 case Intrinsic::aarch64_sve_uunpkhi:
3730 case Intrinsic::aarch64_sve_uunpklo:
3731 case Intrinsic::aarch64_sve_sunpkhi:
3732 case Intrinsic::aarch64_sve_sunpklo:
3734 case Intrinsic::aarch64_sve_uzp1:
3736 case Intrinsic::aarch64_sve_zip1:
3737 case Intrinsic::aarch64_sve_zip2:
3739 case Intrinsic::aarch64_sve_ld1_gather_index:
3741 case Intrinsic::aarch64_sve_st1_scatter_index:
3743 case Intrinsic::aarch64_sve_ld1:
3745 case Intrinsic::aarch64_sve_st1:
3747 case Intrinsic::aarch64_sve_sdiv:
3749 case Intrinsic::aarch64_sve_sel:
3751 case Intrinsic::aarch64_sve_srshl:
3753 case Intrinsic::aarch64_sve_dupq_lane:
3755 case Intrinsic::aarch64_sve_insr:
3757 case Intrinsic::aarch64_sve_whilelo:
3759 case Intrinsic::aarch64_sve_ptrue:
3761 case Intrinsic::aarch64_sve_uxtb:
3763 case Intrinsic::aarch64_sve_uxth:
3765 case Intrinsic::aarch64_sve_uxtw:
3767 case Intrinsic::aarch64_sme_in_streaming_mode:
3769 case Intrinsic::aarch64_sve_umin_u:
3771 case Intrinsic::aarch64_sve_orr_u:
3773 case Intrinsic::aarch64_sve_and_z:
3777 return std::nullopt;
3784 SimplifyAndSetOp)
const {
3785 switch (
II.getIntrinsicID()) {
3788 case Intrinsic::aarch64_neon_fcvtxn:
3789 case Intrinsic::aarch64_neon_rshrn:
3790 case Intrinsic::aarch64_neon_sqrshrn:
3791 case Intrinsic::aarch64_neon_sqrshrun:
3792 case Intrinsic::aarch64_neon_sqshrn:
3793 case Intrinsic::aarch64_neon_sqshrun:
3794 case Intrinsic::aarch64_neon_sqxtn:
3795 case Intrinsic::aarch64_neon_sqxtun:
3796 case Intrinsic::aarch64_neon_uqrshrn:
3797 case Intrinsic::aarch64_neon_uqshrn:
3798 case Intrinsic::aarch64_neon_uqxtn:
3799 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3803 return std::nullopt;
3807 return ST->isSVEAvailable() ||
3808 (ST->isSVEorStreamingSVEAvailable() &&
3809 ST->getCLOpts().enable_scalable_autovec_in_streaming_mode);
3818 if (ST->useSVEForFixedLengthVectors() &&
3819 (ST->isSVEAvailable() ||
3820 ST->getCLOpts().enable_fixedwidth_autovec_in_streaming_mode))
3822 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3823 else if (ST->isNeonAvailable())
3828 if (ST->isSVEAvailable() ||
3829 (ST->isSVEorStreamingSVEAvailable() &&
3830 ST->getCLOpts().enable_scalable_autovec_in_streaming_mode))
3838bool AArch64TTIImpl::isSingleExtWideningInstruction(
3840 Type *SrcOverrideTy)
const {
3855 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3858 Type *SrcTy = SrcOverrideTy;
3860 case Instruction::Add:
3861 case Instruction::Sub: {
3870 if (Opcode == Instruction::Sub)
3894 assert(SrcTy &&
"Expected some SrcTy");
3896 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3902 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3904 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3908 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3911Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3913 Type *SrcOverrideTy)
const {
3914 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3915 Opcode != Instruction::Mul)
3925 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3928 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3934 ->getScalarSizeInBits();
3937 unsigned MaxEltSize = 0;
3940 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3941 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3942 MaxEltSize = std::max(EltSize0, EltSize1);
3945 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3946 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3949 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3951 MaxEltSize = DstEltSize / 2;
3952 }
else if (Opcode == Instruction::Mul &&
3960 Known.Zero.countLeadingOnes() >
3965 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3969 if (MaxEltSize * 2 > DstEltSize)
3987 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3988 (Src->isScalableTy() && !ST->hasSVE2()))
3998 if (AddUser && AddUser->getOpcode() == Instruction::Add)
4002 if (!Shr || Shr->getOpcode() != Instruction::LShr)
4006 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
4007 Src->getScalarSizeInBits() !=
4031 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4045 if (
I && !
I->users().empty()) {
4049 auto GetUserAbsorbedCastCost =
4050 [&](
const Instruction *Usr) -> std::optional<InstructionCost> {
4053 if (
Type *ExtTy = isBinExtWideningInstruction(
4055 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
4068 if (isSingleExtWideningInstruction(
4070 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
4074 if (Usr->getOpcode() == Instruction::Add) {
4075 if (
I == Usr->getOperand(1) ||
4091 return std::nullopt;
4095 bool AllUsersAbsorbCast =
true;
4096 for (
const User *U :
I->users()) {
4098 std::optional<InstructionCost> UserCost = GetUserAbsorbedCastCost(Usr);
4100 AllUsersAbsorbCast =
false;
4103 MaxAbsorbedCost = std::max(MaxAbsorbedCost, *UserCost);
4106 if (AllUsersAbsorbCast)
4107 return MaxAbsorbedCost;
4110 EVT SrcTy = TLI->getValueType(
DL, Src);
4111 EVT DstTy = TLI->getValueType(
DL, Dst);
4115 if ((SrcTy.isScalableVector() && SrcTy.getScalarSizeInBits() > 64) ||
4125 Instruction::ExtractElement, Src,
CostKind, -1,
nullptr,
nullptr);
4127 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH,
CostKind);
4131 if (!SrcTy.isSimple() || !DstTy.
isSimple())
4136 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
4165 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
4168 ST->useSVEForFixedLengthVectors(WiderTy)) {
4169 std::pair<InstructionCost, MVT> LT =
4171 unsigned NumElements =
4187 const unsigned int SVE_EXT_COST = 1;
4188 const unsigned int SVE_FCVT_COST = 1;
4189 const unsigned int SVE_UNPACK_ONCE = 4;
4190 const unsigned int SVE_UNPACK_TWICE = 16;
4319 SVE_EXT_COST + SVE_FCVT_COST},
4324 SVE_EXT_COST + SVE_FCVT_COST},
4331 SVE_EXT_COST + SVE_FCVT_COST},
4335 SVE_EXT_COST + SVE_FCVT_COST},
4341 SVE_EXT_COST + SVE_FCVT_COST},
4344 SVE_EXT_COST + SVE_FCVT_COST},
4349 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4351 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4361 SVE_EXT_COST + SVE_FCVT_COST},
4366 SVE_EXT_COST + SVE_FCVT_COST},
4379 SVE_EXT_COST + SVE_FCVT_COST},
4383 SVE_EXT_COST + SVE_FCVT_COST},
4395 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4397 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4399 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4401 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4405 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4407 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4423 SVE_EXT_COST + SVE_FCVT_COST},
4428 SVE_EXT_COST + SVE_FCVT_COST},
4439 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4441 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4443 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4445 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4447 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4449 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4453 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4455 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4457 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4459 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4684 if (ST->hasFullFP16())
4696 Src->getScalarType(), CCH,
CostKind) +
4704 ST->isSVEorStreamingSVEAvailable() &&
4705 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4707 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4716 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4719 return Part1 + Part2;
4726 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4738 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4751 CostKind, Index,
nullptr,
nullptr);
4755 auto DstVT = TLI->getValueType(
DL, Dst);
4756 auto SrcVT = TLI->getValueType(
DL, Src);
4761 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4767 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4777 case Instruction::SExt:
4782 case Instruction::ZExt:
4783 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4796 return Opcode == Instruction::PHI ? 0 : 1;
4805 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4807 assert(Ty->isVectorTy() &&
"This must be a vector type");
4814 if (!LT.second.isVector())
4819 if (LT.second.isFixedLengthVector()) {
4820 unsigned Width = LT.second.getVectorNumElements();
4821 Index = Index % Width;
4828 if (Index == 0 && !Ty->getScalarType()->isIntegerTy())
4840 if (Index * Ty->getScalarSizeInBits() < 128)
4842 if (Index * Ty->getScalarSizeInBits() < 512 &&
4843 Opcode == Instruction::ExtractElement)
4845 return Ty->getScalarType()->isIntegerTy() ? Cost + 1 : Cost;
4846 if (Opcode == Instruction::ExtractElement)
4848 if (Opcode == Instruction::InsertElement)
4857 if (VIC == TTI::VectorInstrContext::Load) {
4858 if (ST->hasFastLD1Single())
4862 : ST->getVectorInsertExtractBaseCost() + 1;
4870 : ST->getVectorInsertExtractBaseCost() + 1;
4894 auto ExtractCanFuseWithFmul = [&]() {
4901 auto IsAllowedScalarTy = [&](
const Type *
T) {
4902 return T->isFloatTy() ||
T->isDoubleTy() ||
4903 (
T->isHalfTy() && ST->hasFullFP16());
4907 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4910 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4911 !BO->getType()->isVectorTy();
4916 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4920 return Idx == 0 || (RegWidth != 0 && (
Idx * EltSz) % RegWidth == 0);
4929 DenseMap<User *, unsigned> UserToExtractIdx;
4930 for (
auto *U :
Scalar->users()) {
4931 if (!IsUserFMulScalarTy(U))
4935 UserToExtractIdx[
U];
4937 if (UserToExtractIdx.
empty())
4939 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4940 for (
auto *U : S->users()) {
4941 if (UserToExtractIdx.
contains(U)) {
4943 auto *Op0 =
FMul->getOperand(0);
4944 auto *Op1 =
FMul->getOperand(1);
4945 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4946 UserToExtractIdx[
U] =
L;
4952 for (
auto &[U, L] : UserToExtractIdx) {
4964 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4965 if (!IsUserFMulScalarTy(U))
4970 const auto *BO = cast<BinaryOperator>(U);
4971 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4972 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4974 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4977 return IsExtractLaneEquivalentToZero(
4978 cast<ConstantInt>(OtherEE->getIndexOperand())
4981 OtherEE->getType()->getScalarSizeInBits());
4989 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4990 ExtractCanFuseWithFmul())
4995 :
ST->getVectorInsertExtractBaseCost();
5004 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
5007 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr,
nullptr,
5013 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
5015 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr, Scalar,
5016 ScalarUserAndIdx, VIC);
5023 return getVectorInstrCostHelper(
I.getOpcode(), Ty,
CostKind, Index, &
I,
5030 unsigned Index)
const {
5041 : ST->getVectorInsertExtractBaseCost() + 1;
5053 if (ST->hasFastLD1Single()) {
5054 if ((VIC == TTI::VectorInstrContext::Store && Extract) ||
5055 (VIC == TTI::VectorInstrContext::Load && Insert))
5059 if (Ty->getElementType()->isFloatingPointTy())
5062 unsigned VecInstCost =
5064 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
5071 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
5072 return std::nullopt;
5073 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
5074 return std::nullopt;
5076 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
5077 return std::nullopt;
5084 Cost += InstCost(PromotedTy);
5099 int ISD = TLI->InstructionOpcodeToISD(Opcode);
5113 Op2Info, Args, CtxI);
5120 Ty,
CostKind, Op1Info, Op2Info,
true,
5123 [&](
Type *PromotedTy) {
5127 return *PromotedCost;
5130 if (Ty->getScalarType()->isFP128Ty())
5138 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
5158 ST->hasLimited64bitVectorMulBandwidth())
5161 if (Ty->getScalarSizeInBits() > 64) {
5166 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
5169 if (LT.second == MVT::v2i64) {
5173 return LT.first * Mul64CostFactor;
5194 if (LT.second == MVT::nxv2i64)
5195 return LT.first * Mul64CostFactor;
5254 auto VT = TLI->getValueType(
DL, Ty);
5255 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
5259 : (3 * AsrCost + AddCost);
5261 return MulCost + AsrCost + 2 * AddCost;
5263 }
else if (VT.isVector()) {
5273 if (Ty->isScalableTy() && ST->hasSVE())
5274 Cost += 2 * AsrCost;
5279 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
5283 }
else if (LT.second == MVT::v2i64) {
5284 return VT.getVectorNumElements() *
5291 if (Ty->isScalableTy() && ST->hasSVE())
5292 return MulCost + 2 * AddCost + 2 * AsrCost;
5293 return 2 * MulCost + AddCost + AsrCost + UsraCost;
5298 LT.second.isFixedLengthVector()) {
5308 return ExtractCost + InsertCost +
5316 auto VT = TLI->getValueType(
DL, Ty);
5332 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
5333 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
5334 LT.second == MVT::nxv16i8;
5335 bool Is128bit = LT.second.is128BitVector();
5347 (HasMULH ? 0 : ShrCost) +
5348 AddCost * 2 + ShrCost;
5349 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
5356 if (!VT.isVector() && VT.getSizeInBits() > 64)
5360 Opcode, Ty,
CostKind, Op1Info, Op2Info);
5362 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
5366 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
5376 if (
nullptr != Entry)
5384 FVTy && LT.second.isFixedLengthVector()) {
5385 unsigned NumElts = FVTy->getNumElements();
5386 unsigned RegElts = LT.second.getVectorNumElements();
5388 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
5392 if (LT.second.getScalarType() == MVT::i8)
5394 else if (LT.second.getScalarType() == MVT::i16)
5406 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
5407 return (4 + DivCost) * VTy->getNumElements();
5413 -1,
nullptr,
nullptr);
5440 LT.second.isFixedLengthVector())
5441 return 2 * LT.first + 1;
5450 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
5451 (Ty->isHalfTy() && ST->hasFullFP16())) &&
5460 if (!Ty->getScalarType()->isFP128Ty())
5467 if (!Ty->getScalarType()->isFP128Ty())
5468 return 2 * LT.first;
5475 if (!Ty->isVectorTy())
5490 unsigned NumVectorInstToHideOverhead =
5491 ST->getCLOpts().neon_nonconst_stride_overhead;
5492 int MaxMergeDistance = 64;
5496 return NumVectorInstToHideOverhead;
5506 unsigned Opcode1,
unsigned Opcode2)
const {
5509 if (!
Sched.hasInstrSchedModel())
5513 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
5515 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5521 "Cannot handle variant scheduling classes without an MI");
5537 const int AmortizationCost = 20;
5545 VecPred = CurrentPred;
5553 static const auto ValidMinMaxTys = {
5554 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5555 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5556 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5560 (ST->hasFullFP16() &&
5566 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5567 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5568 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5569 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5570 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5571 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5572 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5573 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5574 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5575 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5576 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5578 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5579 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5588 if (Opcode == Instruction::FCmp) {
5590 ValTy,
CostKind, Op1Info, Op2Info,
false,
5592 false, [&](
Type *PromotedTy) {
5604 return *PromotedCost;
5608 if (LT.second.getScalarType() != MVT::f64 &&
5609 LT.second.getScalarType() != MVT::f32 &&
5610 LT.second.getScalarType() != MVT::f16)
5615 unsigned Factor = 1;
5616 if (!CondTy->isVectorTy() &&
5630 AArch64::FCMEQv4f32))
5642 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5661 Op1Info, Op2Info,
I);
5667 if (ST->requiresStrictAlign()) {
5672 Options.AllowOverlappingLoads =
true;
5673 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5678 Options.LoadSizes = {8, 4, 2, 1};
5679 Options.AllowedTailExpansions = {3, 5, 6};
5684 return ST->hasSVE();
5690 switch (MICA.
getID()) {
5691 case Intrinsic::masked_scatter:
5692 case Intrinsic::masked_gather:
5694 case Intrinsic::masked_load:
5695 case Intrinsic::masked_store:
5696 case Intrinsic::masked_expandload:
5697 case Intrinsic::masked_compressstore:
5711 if (!LT.first.isValid())
5716 if (VT->getElementType()->isIntegerTy(1))
5722 return is_contained({Intrinsic::masked_load, Intrinsic::masked_store},
5728 if (MICA.
getID() == Intrinsic::masked_expandload) {
5741 if (MICA.
getID() == Intrinsic::masked_compressstore) {
5761 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5762 return MemOpCost * 2;
5771 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5772 "Should be called on only load or stores.");
5774 case Instruction::Load:
5775 return ST->getCLOpts().sve_gather_overhead.value_or(
5776 ST->getGatherOverhead());
5778 case Instruction::Store:
5779 return ST->getCLOpts().sve_scatter_overhead.value_or(
5780 ST->getScatterOverhead());
5791 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5792 MICA.
getID() == Intrinsic::vp_gather)
5794 : Instruction::Store;
5804 if (!LT.first.isValid())
5808 if (!LT.second.isVector() ||
5810 VT->getElementType()->isIntegerTy(1))
5820 ElementCount LegalVF = LT.second.getVectorElementCount();
5823 {TTI::OK_AnyValue, TTI::OP_None},
I);
5839 EVT VT = TLI->getValueType(
DL, Ty,
true);
5841 if (VT == MVT::Other)
5846 if (!LT.first.isValid())
5852 if (VTy->getElementType()->isIntegerTy(1) &&
5853 !VTy->getElementCount().isKnownMultipleOf(
5859 Intrinsic::ID IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5860 : Intrinsic::masked_store;
5874 if (Opcode == Instruction::Store)
5878 if (ST->getFixedLoadLatency())
5879 return (LT.first - 1) + ST->getFixedLoadLatency();
5888 if (LT.second.isScalableVector() ||
5889 ST->useSVEForFixedLengthVectors(LT.second)) {
5890 Inst = AArch64::LDR_ZXI;
5891 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5892 switch (LT.second.getSizeInBits()) {
5894 Inst = AArch64::LDRBui;
5897 Inst = AArch64::LDRHui;
5900 Inst = AArch64::LDRSui;
5903 Inst = AArch64::LDRDui;
5906 Inst = AArch64::LDRQui;
5912 switch (LT.second.getSizeInBits()) {
5914 Inst = AArch64::LDRBBui;
5917 Inst = AArch64::LDRHHui;
5920 Inst = AArch64::LDRWui;
5923 Inst = AArch64::LDRXui;
5931 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5933 ?
Sched.getSchedClassDesc(SchedClass)
5939 return (LT.first - 1) + ST->getLoadLatency();
5942 float NumLoads = (LT.first - 1).
getValue();
5943 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5944 Sched.computeInstrLatency(*ST, *SCD);
5947 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5948 LT.second.is128BitVector() && Alignment <
Align(16)) {
5954 const int AmortizationCost = 6;
5956 return LT.first * 2 * AmortizationCost;
5960 if (Ty->isPtrOrPtrVectorTy())
5965 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5967 if (VT == MVT::v4i8)
5974 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5975 Alignment !=
Align(1))
5987 if (Remainder != 0) {
5990 while (!TypeWorklist.
empty()) {
6000 TypeWorklist.
push_back({CurrNumElements - PrevPow2,
Offset + PrevPow2});
6012 bool UseMaskForCond,
bool UseMaskForGaps)
const {
6013 assert(Factor >= 2 &&
"Invalid interleave factor");
6022 if (Factor > TLI->getMaxSupportedInterleaveFactor())
6026 DL.getTypeSizeInBits(VecTy).getKnownMinValue() != (3 * 128))
6031 unsigned MaxNativeInterleaveFactor = TLI->getMaxSupportedInterleaveFactor();
6036 (UseMaskForCond || UseMaskForGaps ||
6037 (Factor > MaxNativeInterleaveFactor &&
6038 TLI->useSVEForFixedLengthVectorVT(LT.second))))
6041 if (!UseMaskForGaps && Factor <= MaxNativeInterleaveFactor) {
6044 EC.divideCoefficientBy(Factor));
6050 if (EC.isKnownMultipleOf(Factor) &&
6051 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
6052 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
6057 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
6063 if (UseMaskForCond) {
6064 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
6065 : Intrinsic::masked_store;
6085 if (Opcode == Instruction::Store && Factor == 4 &&
6086 SubVecCost.second.getScalarSizeInBits() ==
6087 (4 * ResultCost.second.getScalarSizeInBits()))
6088 LegalizationCost *= 4;
6090 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
6096 UseMaskForCond, UseMaskForGaps);
6103 for (
auto *
I : Tys) {
6104 if (!
I->isVectorTy())
6115 Align Alignment)
const {
6122 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
6123 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
6135 Size.getFixedValue() <= 16;
6140 bool IsStore, std::optional<Instruction::CastOps> CastHint)
const {
6141 if (NumVectors <= 1 || !ST->enableSubRegLiveness() || !ST->hasSVE2p1())
6160 (CastHint == Instruction::ZExt || CastHint == Instruction::SExt))
6167 DL.getTypeSizeInBits(VectorTy).isKnownMultipleOf(
6173 bool HasUnorderedReductions)
const {
6176 return ST->getMaxInterleaveFactor();
6186 enum { MaxStridedLoads = 7 };
6188 int StridedLoads = 0;
6191 for (
const auto BB : L->blocks()) {
6192 for (
auto &
I : *BB) {
6198 if (L->isLoopInvariant(PtrValue))
6203 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
6212 if (StridedLoads > MaxStridedLoads / 2)
6213 return StridedLoads;
6216 return StridedLoads;
6219 int StridedLoads = countStridedLoads(L, SE);
6221 <<
" strided loads\n");
6237 unsigned *FinalSize) {
6241 for (
auto *BB : L->getBlocks()) {
6242 for (
auto &
I : *BB) {
6248 if (!Cost.isValid())
6252 if (LoopCost > Budget)
6274 if (MaxTC > 0 && MaxTC <= 32)
6285 if (Blocks.
size() != 2)
6307 if (!L->isInnermost() || L->getNumBlocks() > 8)
6311 if (!L->getExitBlock())
6317 bool HasParellelizableReductions =
6318 L->getNumBlocks() == 1 &&
6319 any_of(L->getHeader()->phis(),
6321 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
6324 if (HasParellelizableReductions &&
6346 if (HasParellelizableReductions) {
6357 if (Header == Latch) {
6360 unsigned Width = 10;
6366 unsigned MaxInstsPerLine = 16;
6368 unsigned BestUC = 1;
6369 unsigned SizeWithBestUC = BestUC *
Size;
6371 unsigned SizeWithUC = UC *
Size;
6372 if (SizeWithUC > 48)
6374 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
6375 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
6377 SizeWithBestUC = BestUC *
Size;
6387 for (
auto *BB : L->blocks()) {
6388 for (
auto &
I : *BB) {
6398 for (
auto *U :
I.users())
6400 LoadedValuesPlus.
insert(U);
6407 return LoadedValuesPlus.
contains(
SI->getOperand(0));
6433 auto *I = dyn_cast<Instruction>(V);
6434 return I && DependsOnLoopLoad(I, Depth + 1);
6441 DependsOnLoopLoad(
I, 0)) {
6473 if (L->getLoopDepth() > 1)
6484 for (
auto *BB : L->getBlocks()) {
6485 for (
auto &
I : *BB) {
6489 if (IsVectorized &&
I.getType()->isVectorTy())
6500 if (
Cost >= ST->getCLOpts().force_unroll_threshold)
6509 if (ST->isAppleMLike())
6511 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
6512 ST->getCLOpts().enable_falkor_hwpf_unroll_fix)
6533 !ST->getSchedModel().isOutOfOrder()) {
6545 if (
Cost < ST->getCLOpts().force_unroll_threshold)
6556 bool CanCreate)
const {
6560 case Intrinsic::aarch64_neon_st1x2:
6561 case Intrinsic::aarch64_neon_st1x3:
6562 case Intrinsic::aarch64_neon_st1x4:
6563 case Intrinsic::aarch64_neon_st2:
6564 case Intrinsic::aarch64_neon_st3:
6565 case Intrinsic::aarch64_neon_st4: {
6568 if (!CanCreate || !ST)
6570 unsigned NumElts = Inst->
arg_size() - 1;
6571 if (ST->getNumElements() != NumElts)
6573 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6579 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6581 Res = Builder.CreateInsertValue(Res, L, i);
6585 case Intrinsic::aarch64_neon_ld1x2:
6586 case Intrinsic::aarch64_neon_ld1x3:
6587 case Intrinsic::aarch64_neon_ld1x4:
6588 case Intrinsic::aarch64_neon_ld2:
6589 case Intrinsic::aarch64_neon_ld3:
6590 case Intrinsic::aarch64_neon_ld4:
6591 if (Inst->
getType() == ExpectedType)
6602 case Intrinsic::aarch64_neon_ld1x2:
6603 case Intrinsic::aarch64_neon_ld1x3:
6604 case Intrinsic::aarch64_neon_ld1x4:
6605 case Intrinsic::aarch64_neon_ld2:
6606 case Intrinsic::aarch64_neon_ld3:
6607 case Intrinsic::aarch64_neon_ld4:
6608 Info.ReadMem =
true;
6609 Info.WriteMem =
false;
6612 case Intrinsic::aarch64_neon_st1x2:
6613 case Intrinsic::aarch64_neon_st1x3:
6614 case Intrinsic::aarch64_neon_st1x4:
6615 case Intrinsic::aarch64_neon_st2:
6616 case Intrinsic::aarch64_neon_st3:
6617 case Intrinsic::aarch64_neon_st4:
6618 Info.ReadMem =
false;
6619 Info.WriteMem =
true;
6628 case Intrinsic::aarch64_neon_ld1x2:
6629 case Intrinsic::aarch64_neon_st1x2:
6630 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6632 case Intrinsic::aarch64_neon_ld1x3:
6633 case Intrinsic::aarch64_neon_st1x3:
6634 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6636 case Intrinsic::aarch64_neon_ld1x4:
6637 case Intrinsic::aarch64_neon_st1x4:
6638 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6640 case Intrinsic::aarch64_neon_ld2:
6641 case Intrinsic::aarch64_neon_st2:
6642 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6644 case Intrinsic::aarch64_neon_ld3:
6645 case Intrinsic::aarch64_neon_st3:
6646 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6648 case Intrinsic::aarch64_neon_ld4:
6649 case Intrinsic::aarch64_neon_st4:
6650 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6662 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6663 bool Considerable =
false;
6664 AllowPromotionWithoutCommonHeader =
false;
6667 Type *ConsideredSExtType =
6669 if (
I.getType() != ConsideredSExtType)
6673 for (
const User *U :
I.users()) {
6675 Considerable =
true;
6679 if (GEPInst->getNumOperands() > 2) {
6680 AllowPromotionWithoutCommonHeader =
true;
6685 return Considerable;
6736 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6746 return LegalizationCost + 2;
6756 LegalizationCost *= LT.first - 1;
6759 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6768 return LegalizationCost + 2;
6776 std::optional<FastMathFlags> FMF,
6792 return BaseCost + FixedVTy->getNumElements();
6806 MVT MTy = LT.second;
6811 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6859 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6860 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6872 return (LT.first - 1) +
Log2_32(NElts);
6877 return (LT.first - 1) + Entry->Cost;
6889 if (LT.first != 1) {
6895 ExtraCost *= LT.first - 1;
6898 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6899 return Cost + ExtraCost;
6907 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6909 EVT VecVT = TLI->getValueType(
DL, VecTy);
6910 EVT ResVT = TLI->getValueType(
DL, ResTy);
6920 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6922 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6924 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6926 return (LT.first - 1) * 2 + 2;
6937 EVT VecVT = TLI->getValueType(
DL, VecTy);
6938 EVT ResVT = TLI->getValueType(
DL, ResTy);
6941 RedOpcode == Instruction::Add) {
6947 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6949 return LT.first + 2;
6984 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6985 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6999 if (LT.second.getScalarType() == MVT::i1) {
7008 assert(Entry &&
"Illegal Type for Splice");
7009 LegalizationCost += Entry->Cost;
7010 return LegalizationCost * LT.first;
7014 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
7023 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
7024 Opcode != Instruction::FAdd && Opcode != Instruction::FSub))
7030 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
7031 if (!FMF->allowReassoc() || !FMF->allowContract())
7035 "FastMathFlags only apply to floating-point partial reductions");
7039 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
7040 "Unexpected values for OpBExtend or InputTypeB");
7044 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
7045 InputTypeA != InputTypeB))
7056 assert(!OpBExtend &&
"Extended second operand without extended first.");
7057 assert(InputTypeA == AccumType &&
"Type mismatch with no extensions.");
7063 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
7066 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
7079 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
7088 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
7094 std::pair<InstructionCost, MVT> AccumLT =
7096 std::pair<InstructionCost, MVT> InputLT =
7100 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
7101 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
7102 (AccumLT.second.isFixedLengthVector() &&
7103 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
7107 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
7115 if (AccumLT.second.getScalarType() == MVT::i32 &&
7116 InputLT.second.getScalarType() == MVT::i8) {
7118 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
7119 return Cost + INegCost;
7121 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
7122 return Cost + INegCost;
7127 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
7128 return Cost * 3 + INegCost;
7131 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
7133 if (AccumLT.second.getScalarType() == MVT::i64 &&
7134 InputLT.second.getScalarType() == MVT::i16)
7135 return Cost + INegCost;
7138 if (AccumLT.second.getScalarType() == MVT::i32 &&
7139 InputLT.second.getScalarType() == MVT::i16 &&
7140 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
7143 if (AccumLT.second.getScalarType() == MVT::i64 &&
7144 InputLT.second.getScalarType() == MVT::i8)
7150 return Cost + INegCost;
7153 if (AccumLT.second.getScalarType() == MVT::i16 &&
7154 InputLT.second.getScalarType() == MVT::i8 &&
7155 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
7161 if (Opcode == Instruction::FAdd && !IsSub &&
7162 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
7163 AccumLT.second.getScalarType() == MVT::f32 &&
7164 InputLT.second.getScalarType() == MVT::f16)
7168 if (Ratio == 2 && !IsUSDot) {
7169 MVT InVT = InputLT.second.getScalarType();
7173 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
7175 return (BinOp || IsSub) ?
Cost * 2 :
Cost;
7178 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
7182 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
7183 InVT == MVT::bf16 && IsSub)
7193 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
7194 return Cost * 2 + FNegCost;
7198 AccumType, VF, OpAExtend, OpBExtend,
7209 "Expected the Mask to match the return size if given");
7211 "Expected the same scalar types");
7217 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
7218 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
7219 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
7227 return std::max<InstructionCost>(1, LT.first / 4);
7235 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
7237 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
7240 unsigned TpNumElts = Mask.size();
7241 unsigned LTNumElts = LT.second.getVectorNumElements();
7242 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
7244 LT.second.getVectorElementCount());
7246 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
7248 for (
unsigned N = 0;
N < NumVecs;
N++) {
7252 unsigned Source1 = -1U, Source2 = -1U;
7253 unsigned NumSources = 0;
7254 for (
unsigned E = 0; E < LTNumElts; E++) {
7255 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
7264 unsigned Source = MaskElt / LTNumElts;
7265 if (NumSources == 0) {
7268 }
else if (NumSources == 1 && Source != Source1) {
7271 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
7277 if (Source == Source1)
7279 else if (Source == Source2)
7280 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
7289 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
7300 NTp, NTp,
CostKind, NMask, 0,
nullptr, Args,
7303 Result.first->second = NCost;
7317 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
7318 if (LT.second.getFixedSizeInBits() >= 128 &&
7320 LT.second.getVectorNumElements() / 2) {
7323 if (Index == (
int)LT.second.getVectorNumElements() / 2)
7337 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
7340 return M.value() < 0 || M.value() == (int)M.index();
7346 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
7347 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
7356 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
7357 ST->isSVEorStreamingSVEAvailable() &&
7362 if (ST->isSVEorStreamingSVEAvailable() &&
7376 if (IsLoad && LT.second.isVector() &&
7378 LT.second.getVectorElementCount()))
7384 if (Mask.size() == 4 &&
7386 (SrcTy->getScalarSizeInBits() == 16 ||
7387 SrcTy->getScalarSizeInBits() == 32) &&
7388 all_of(Mask, [](
int E) {
return E < 8; }))
7394 if (LT.second.isFixedLengthVector() &&
7395 LT.second.getVectorNumElements() == Mask.size() &&
7401 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7402 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7403 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
7404 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7405 LT.second.getVectorNumElements(), 16) ||
7406 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7407 LT.second.getVectorNumElements(), 32) ||
7408 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7409 LT.second.getVectorNumElements(), 64) ||
7412 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
7541 return LT.first * Entry->Cost;
7550 LT.second.getSizeInBits() <= 128 && SubTp) {
7552 if (SubLT.second.isVector()) {
7553 int NumElts = LT.second.getVectorNumElements();
7554 int NumSubElts = SubLT.second.getVectorNumElements();
7555 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
7561 if (IsExtractSubvector)
7582 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
7593 return valueOr(ST->getCLOpts().sve_prefer_fixed_over_scalable_if_equal,
7594 ST->useFixedOverScalableIfEqualCost());
7598 return ST->getEpilogueVectorizationMinVF();
7633 unsigned NumInsns = 0;
7635 NumInsns += BB->size();
7639 return NumInsns >= ST->getCLOpts().sve_tail_folding_insn_threshold;
7645 int64_t Scale,
unsigned AddrSpace)
const {
7667 int64_t BaseOffset,
bool HasBaseReg,
7668 int64_t Scale,
unsigned AddrSpace,
7670 int64_t ScalableOffset)
const {
7678 EVT MemVT = TLI->getValueType(
DL, Ty);
7683 EVT LegalVT = TLI->getLegalTypeToTransformTo(Ctx, MemVT);
7688 "expected vector split");
7695 if (ScalableOffset && ScalableOffset % LegalNumBytes == 0)
7701 AddrSpace,
I, ScalableOffset);
7706 if (ST->getCLOpts().enable_aarch64_or_like_select) {
7711 if (
I->getOpcode() == Instruction::Or &&
7715 if (
I->getOpcode() == Instruction::Add ||
7716 I->getOpcode() == Instruction::Sub)
7730 if (ST->getCLOpts().enable_aarch64_lsr_cost_opt)
7741 return all_equal(Shuf->getShuffleMask());
7748 bool AllowSplat =
false) {
7753 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7754 auto *FullTy = FullV->
getType();
7755 auto *HalfTy = HalfV->getType();
7757 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7760 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7763 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7767 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7781 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7782 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7796 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7797 (M2Start != 0 && M2Start != (NumElements / 2)))
7799 if (S1Op1 && S2Op1 && M1Start != M2Start)
7809 return Ext->getType()->getScalarSizeInBits() ==
7810 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7824 Value *VectorOperand =
nullptr;
7841 if (!
GEP ||
GEP->getNumOperands() != 2)
7845 Value *Offsets =
GEP->getOperand(1);
7848 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7854 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7855 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7856 Ops.push_back(&
GEP->getOperandUse(1));
7892 switch (
II->getIntrinsicID()) {
7893 case Intrinsic::aarch64_neon_smull:
7894 case Intrinsic::aarch64_neon_umull:
7897 Ops.push_back(&
II->getOperandUse(0));
7898 Ops.push_back(&
II->getOperandUse(1));
7903 case Intrinsic::fma:
7904 case Intrinsic::fmuladd:
7911 Ops.push_back(&
II->getOperandUse(0));
7913 Ops.push_back(&
II->getOperandUse(1));
7916 case Intrinsic::aarch64_neon_sqdmull:
7917 case Intrinsic::aarch64_neon_sqdmulh:
7918 case Intrinsic::aarch64_neon_sqrdmulh:
7921 Ops.push_back(&
II->getOperandUse(0));
7923 Ops.push_back(&
II->getOperandUse(1));
7924 return !
Ops.empty();
7925 case Intrinsic::aarch64_neon_fmlal:
7926 case Intrinsic::aarch64_neon_fmlal2:
7927 case Intrinsic::aarch64_neon_fmlsl:
7928 case Intrinsic::aarch64_neon_fmlsl2:
7931 Ops.push_back(&
II->getOperandUse(1));
7933 Ops.push_back(&
II->getOperandUse(2));
7934 return !
Ops.empty();
7935 case Intrinsic::aarch64_sve_ptest_first:
7936 case Intrinsic::aarch64_sve_ptest_last:
7938 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7939 Ops.push_back(&
II->getOperandUse(0));
7940 return !
Ops.empty();
7941 case Intrinsic::aarch64_sme_write_horiz:
7942 case Intrinsic::aarch64_sme_write_vert:
7943 case Intrinsic::aarch64_sme_writeq_horiz:
7944 case Intrinsic::aarch64_sme_writeq_vert: {
7946 if (!Idx || Idx->getOpcode() != Instruction::Add)
7948 Ops.push_back(&
II->getOperandUse(1));
7951 case Intrinsic::aarch64_sme_read_horiz:
7952 case Intrinsic::aarch64_sme_read_vert:
7953 case Intrinsic::aarch64_sme_readq_horiz:
7954 case Intrinsic::aarch64_sme_readq_vert:
7955 case Intrinsic::aarch64_sme_ld1b_vert:
7956 case Intrinsic::aarch64_sme_ld1h_vert:
7957 case Intrinsic::aarch64_sme_ld1w_vert:
7958 case Intrinsic::aarch64_sme_ld1d_vert:
7959 case Intrinsic::aarch64_sme_ld1q_vert:
7960 case Intrinsic::aarch64_sme_st1b_vert:
7961 case Intrinsic::aarch64_sme_st1h_vert:
7962 case Intrinsic::aarch64_sme_st1w_vert:
7963 case Intrinsic::aarch64_sme_st1d_vert:
7964 case Intrinsic::aarch64_sme_st1q_vert:
7965 case Intrinsic::aarch64_sme_ld1b_horiz:
7966 case Intrinsic::aarch64_sme_ld1h_horiz:
7967 case Intrinsic::aarch64_sme_ld1w_horiz:
7968 case Intrinsic::aarch64_sme_ld1d_horiz:
7969 case Intrinsic::aarch64_sme_ld1q_horiz:
7970 case Intrinsic::aarch64_sme_st1b_horiz:
7971 case Intrinsic::aarch64_sme_st1h_horiz:
7972 case Intrinsic::aarch64_sme_st1w_horiz:
7973 case Intrinsic::aarch64_sme_st1d_horiz:
7974 case Intrinsic::aarch64_sme_st1q_horiz: {
7976 if (!Idx || Idx->getOpcode() != Instruction::Add)
7978 Ops.push_back(&
II->getOperandUse(3));
7981 case Intrinsic::aarch64_neon_pmull:
7984 Ops.push_back(&
II->getOperandUse(0));
7985 Ops.push_back(&
II->getOperandUse(1));
7987 case Intrinsic::aarch64_neon_pmull64:
7989 II->getArgOperand(1)))
7991 Ops.push_back(&
II->getArgOperandUse(0));
7992 Ops.push_back(&
II->getArgOperandUse(1));
7994 case Intrinsic::masked_gather:
7997 Ops.push_back(&
II->getArgOperandUse(0));
7999 case Intrinsic::masked_scatter:
8002 Ops.push_back(&
II->getArgOperandUse(1));
8009 auto ShouldSinkCondition = [](
Value *
Cond,
8014 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
8018 Ops.push_back(&
II->getOperandUse(0));
8022 switch (
I->getOpcode()) {
8023 case Instruction::GetElementPtr:
8024 case Instruction::Add:
8025 case Instruction::Sub:
8027 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
8029 Ops.push_back(&
I->getOperandUse(
Op));
8034 case Instruction::Select: {
8035 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
8038 Ops.push_back(&
I->getOperandUse(0));
8041 case Instruction::UncondBr:
8043 case Instruction::CondBr: {
8047 Ops.push_back(&
I->getOperandUse(0));
8050 case Instruction::FMul:
8055 Ops.push_back(&
I->getOperandUse(0));
8057 Ops.push_back(&
I->getOperandUse(1));
8067 case Instruction::Xor:
8070 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
8072 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
8077 case Instruction::And:
8078 case Instruction::Or:
8081 if (
I->getOpcode() == Instruction::Or &&
8086 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
8089 for (
auto &
Op :
I->operands()) {
8101 Ops.push_back(&Not);
8102 Ops.push_back(&InsertElt);
8112 if (!
I->getType()->isVectorTy())
8113 return !
Ops.empty();
8115 switch (
I->getOpcode()) {
8116 case Instruction::Sub:
8117 case Instruction::Add: {
8126 Ops.push_back(&Ext1->getOperandUse(0));
8127 Ops.push_back(&Ext2->getOperandUse(0));
8130 Ops.push_back(&
I->getOperandUse(0));
8131 Ops.push_back(&
I->getOperandUse(1));
8135 case Instruction::Or: {
8138 if (ST->hasNEON()) {
8152 if (
I->getParent() != MainAnd->
getParent() ||
8157 if (
I->getParent() != IA->getParent() ||
8158 I->getParent() != IB->getParent())
8163 Ops.push_back(&
I->getOperandUse(0));
8164 Ops.push_back(&
I->getOperandUse(1));
8173 case Instruction::Mul: {
8174 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
8177 if (Ty->isScalableTy())
8181 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
8184 int NumZExts = 0, NumSExts = 0;
8185 for (
auto &
Op :
I->operands()) {
8192 auto *ExtOp = Ext->getOperand(0);
8193 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
8194 Ops.push_back(&Ext->getOperandUse(0));
8202 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
8203 I->getType()->getScalarSizeInBits())
8240 if (!ElementConstant || !ElementConstant->
isZero())
8243 unsigned Opcode = OperandInstr->
getOpcode();
8244 if (Opcode == Instruction::SExt)
8246 else if (Opcode == Instruction::ZExt)
8251 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
8261 Ops.push_back(&Insert->getOperandUse(1));
8267 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
8271 if (!ShouldSinkSplatForIndexedVariant(
I))
8276 Ops.push_back(&
I->getOperandUse(0));
8278 Ops.push_back(&
I->getOperandUse(1));
8280 return !
Ops.empty();
8282 case Instruction::FMul: {
8284 if (
I->getType()->isScalableTy())
8285 return !
Ops.empty();
8289 return !
Ops.empty();
8293 Ops.push_back(&
I->getOperandUse(0));
8295 Ops.push_back(&
I->getOperandUse(1));
8296 return !
Ops.empty();
8305 Align Alignment)
const {
8306 if (!(ST->isSVEAvailable() ||
8307 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
8311 DataType->getPrimitiveSizeInBits().getFixedValue() < 128)
8318 if (!LT.first.isValid())
8323 switch (LT.second.SimpleTy) {
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
bool useSVEForFixedLengthVectors() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index) const override
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
bool preferFixedOverScalableIfEqualCost() const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
bool hasMultiVectorLoadStore(unsigned NumVectors, TTI::MaskSource Mask, VectorType *VectorTy, bool IsStore, std::optional< Instruction::CastOps > CastHint) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalSpeculativeLoad(Type *DataType, unsigned AddressSpace) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
bool isElementTypeLegalForCompressStore(Type *Ty) const
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
The compiler must not inline when that may alter the behavior of the program.
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
static bool isCommutative(Predicate Pred)
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static bool isCommutative(Predicate P)
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
user_iterator user_begin()
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ CTTZ_ELTS
Returns the number of number of trailing (least significant) zero elements in a vector.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI bool isTargetIntrinsic(ID IID)
isTargetIntrinsic - Returns true if IID is an intrinsic specific to a certain target.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
@ LLVM_MARK_AS_BITMASK_ENUM
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const SymbolicStrideMap &StridesMap=SymbolicStrideMap(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
constexpr bool valueOr(BoolOrDefault X, bool Default)
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
CmpInst::Predicate getCmpPredicate() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
bool hasCmpPredicate() const
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
SVEIntrinsicInfo & setCmpPredicate(CmpInst::Predicate Pred)
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isScalableVT() const
Return true if the type is a scalable type.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...