24#include "llvm/IR/IntrinsicsAArch64.h"
36#define DEBUG_TYPE "aarch64tti"
42 "sve-prefer-fixed-over-scalable-if-equal",
cl::Hidden);
60 "Penalty of calling a function that requires a change to PSTATE.SM"));
64 cl::desc(
"Penalty of inlining a call that requires a change to PSTATE.SM"));
75 cl::desc(
"The cost of a histcnt instruction"));
79 cl::desc(
"The number of instructions to search for a redundant dmb"));
83 cl::desc(
"Threshold for forced unrolling of small loops in AArch64"));
86class TailFoldingOption {
101 bool NeedsDefault =
true;
105 void setNeedsDefault(
bool V) { NeedsDefault =
V; }
120 assert((InitialBits == TailFoldingOpts::Disabled || !NeedsDefault) &&
121 "Initial bits should only include one of "
122 "(disabled|all|simple|default)");
123 Bits = NeedsDefault ? DefaultBits : InitialBits;
125 Bits &= ~DisableBits;
131 errs() <<
"invalid argument '" << Opt
132 <<
"' to -sve-tail-folding=; the option should be of the form\n"
133 " (disabled|all|default|simple)[+(reductions|recurrences"
134 "|reverse|noreductions|norecurrences|noreverse)]\n";
140 void operator=(
const std::string &Val) {
149 setNeedsDefault(
false);
152 StringRef(Val).split(TailFoldTypes,
'+', -1,
false);
154 unsigned StartIdx = 1;
155 if (TailFoldTypes[0] ==
"disabled")
156 setInitialBits(TailFoldingOpts::Disabled);
157 else if (TailFoldTypes[0] ==
"all")
158 setInitialBits(TailFoldingOpts::All);
159 else if (TailFoldTypes[0] ==
"default")
160 setNeedsDefault(
true);
161 else if (TailFoldTypes[0] ==
"simple")
162 setInitialBits(TailFoldingOpts::Simple);
165 setInitialBits(TailFoldingOpts::Disabled);
168 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
169 if (TailFoldTypes[
I] ==
"reductions")
170 setEnableBit(TailFoldingOpts::Reductions);
171 else if (TailFoldTypes[
I] ==
"recurrences")
172 setEnableBit(TailFoldingOpts::Recurrences);
173 else if (TailFoldTypes[
I] ==
"reverse")
174 setEnableBit(TailFoldingOpts::Reverse);
175 else if (TailFoldTypes[
I] ==
"noreductions")
176 setDisableBit(TailFoldingOpts::Reductions);
177 else if (TailFoldTypes[
I] ==
"norecurrences")
178 setDisableBit(TailFoldingOpts::Recurrences);
179 else if (TailFoldTypes[
I] ==
"noreverse")
180 setDisableBit(TailFoldingOpts::Reverse);
197 "Control the use of vectorisation using tail-folding for SVE where the"
198 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
199 "\ndisabled (Initial) No loop types will vectorize using "
201 "\ndefault (Initial) Uses the default tail-folding settings for "
203 "\nall (Initial) All legal loop types will vectorize using "
205 "\nsimple (Initial) Use tail-folding for simple loops (not "
206 "reductions or recurrences)"
207 "\nreductions Use tail-folding for loops containing reductions"
208 "\nnoreductions Inverse of above"
209 "\nrecurrences Use tail-folding for loops containing fixed order "
211 "\nnorecurrences Inverse of above"
212 "\nreverse Use tail-folding for loops requiring reversed "
214 "\nnoreverse Inverse of above"),
259 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
260 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
261 FeatureStr.
split(Features,
",");
277 return F.hasFnAttribute(
"fmv-features");
287 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
288 CallAttrs.callee().hasStreamingInterfaceOrBody())
293 if (
CallAttrs.callee().hasStreamingBody()) {
303 CallAttrs.requiresPreservingAllZAState()) {
326 auto FVTy = dyn_cast<FixedVectorType>(Ty);
328 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
337 unsigned DefaultCallPenalty)
const {
362 if (
F ==
Call.getCaller())
368 return DefaultCallPenalty;
379 ST->isSVEorStreamingSVEAvailable() &&
380 !ST->disableMaximizeScalableBandwidth();
404 assert(Ty->isIntegerTy());
406 unsigned BitSize = Ty->getPrimitiveSizeInBits();
413 ImmVal = Imm.sext((BitSize + 63) & ~0x3fU);
418 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
424 return std::max<InstructionCost>(1,
Cost);
431 assert(Ty->isIntegerTy());
433 unsigned BitSize = Ty->getPrimitiveSizeInBits();
439 unsigned ImmIdx = ~0U;
443 case Instruction::GetElementPtr:
448 case Instruction::Store:
451 case Instruction::Add:
452 case Instruction::Sub:
453 case Instruction::Mul:
454 case Instruction::UDiv:
455 case Instruction::SDiv:
456 case Instruction::URem:
457 case Instruction::SRem:
458 case Instruction::And:
459 case Instruction::Or:
460 case Instruction::Xor:
461 case Instruction::ICmp:
465 case Instruction::Shl:
466 case Instruction::LShr:
467 case Instruction::AShr:
471 case Instruction::Trunc:
472 case Instruction::ZExt:
473 case Instruction::SExt:
474 case Instruction::IntToPtr:
475 case Instruction::PtrToInt:
476 case Instruction::BitCast:
477 case Instruction::PHI:
478 case Instruction::Call:
479 case Instruction::Select:
480 case Instruction::Ret:
481 case Instruction::Load:
486 int NumConstants = (BitSize + 63) / 64;
499 assert(Ty->isIntegerTy());
501 unsigned BitSize = Ty->getPrimitiveSizeInBits();
510 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
516 case Intrinsic::sadd_with_overflow:
517 case Intrinsic::uadd_with_overflow:
518 case Intrinsic::ssub_with_overflow:
519 case Intrinsic::usub_with_overflow:
520 case Intrinsic::smul_with_overflow:
521 case Intrinsic::umul_with_overflow:
523 int NumConstants = (BitSize + 63) / 64;
530 case Intrinsic::experimental_stackmap:
531 if ((Idx < 2) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
534 case Intrinsic::experimental_patchpoint_void:
535 case Intrinsic::experimental_patchpoint:
536 if ((Idx < 4) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
539 case Intrinsic::experimental_gc_statepoint:
540 if ((Idx < 5) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
550 if (TyWidth == 32 || TyWidth == 64)
559 return ST->getMispredictionPenalty();
580 unsigned TotalHistCnts = 1;
590 unsigned EC = VTy->getElementCount().getKnownMinValue();
595 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
597 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
601 TotalHistCnts = EC / NaturalVectorWidth;
621 switch (ICA.
getID()) {
622 case Intrinsic::experimental_vector_histogram_add: {
629 case Intrinsic::clmul: {
634 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
638 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
643 -1,
nullptr,
nullptr) *
646 -1,
nullptr,
nullptr);
650 if (LT.second.SimpleTy == MVT::nxv2i64)
651 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
654 if (ST->hasSVE2() || ST->hasSME()) {
655 switch (LT.second.SimpleTy) {
670 if (LT.second.SimpleTy == MVT::nxv2i64)
674 switch (LT.second.SimpleTy) {
684 -1,
nullptr,
nullptr) *
687 -1,
nullptr,
nullptr));
696 return LT.first * 11;
698 return LT.first * 14;
705 case Intrinsic::umin:
706 case Intrinsic::umax:
707 case Intrinsic::smin:
708 case Intrinsic::smax: {
709 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
710 MVT::v8i16, MVT::v2i32, MVT::v4i32,
711 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
715 if (LT.second == MVT::v2i64)
721 case Intrinsic::scmp:
722 case Intrinsic::ucmp: {
724 {Intrinsic::scmp, MVT::i32, 3},
725 {Intrinsic::scmp, MVT::i64, 3},
726 {Intrinsic::scmp, MVT::v8i8, 3},
727 {Intrinsic::scmp, MVT::v16i8, 3},
728 {Intrinsic::scmp, MVT::v4i16, 3},
729 {Intrinsic::scmp, MVT::v8i16, 3},
730 {Intrinsic::scmp, MVT::v2i32, 3},
731 {Intrinsic::scmp, MVT::v4i32, 3},
732 {Intrinsic::scmp, MVT::v1i64, 3},
733 {Intrinsic::scmp, MVT::v2i64, 3},
739 return Entry->Cost * LT.first;
742 case Intrinsic::sadd_sat:
743 case Intrinsic::ssub_sat:
744 case Intrinsic::uadd_sat:
745 case Intrinsic::usub_sat: {
746 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
747 MVT::v8i16, MVT::v2i32, MVT::v4i32,
753 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
755 return LT.first * Instrs;
760 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
761 return LT.first * Instrs;
765 case Intrinsic::abs: {
766 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
767 MVT::v8i16, MVT::v2i32, MVT::v4i32,
768 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
769 MVT::nxv4i32, MVT::nxv2i64};
775 case Intrinsic::bswap: {
776 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
777 MVT::v4i32, MVT::v2i64};
780 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
785 case Intrinsic::fmuladd: {
790 (EltTy->
isHalfTy() && ST->hasFullFP16()))
794 case Intrinsic::stepvector: {
803 Cost += AddCost * (LT.first - 1);
807 case Intrinsic::vector_extract:
808 case Intrinsic::vector_insert: {
821 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
822 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
830 getTLI()->getTypeConversion(
C, SubVecVT);
832 getTLI()->getTypeConversion(
C, VecVT);
840 case Intrinsic::bitreverse: {
842 {Intrinsic::bitreverse, MVT::i32, 1},
843 {Intrinsic::bitreverse, MVT::i64, 1},
844 {Intrinsic::bitreverse, MVT::v8i8, 1},
845 {Intrinsic::bitreverse, MVT::v16i8, 1},
846 {Intrinsic::bitreverse, MVT::v4i16, 2},
847 {Intrinsic::bitreverse, MVT::v8i16, 2},
848 {Intrinsic::bitreverse, MVT::v2i32, 2},
849 {Intrinsic::bitreverse, MVT::v4i32, 2},
850 {Intrinsic::bitreverse, MVT::v1i64, 2},
851 {Intrinsic::bitreverse, MVT::v2i64, 2},
859 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
860 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
861 return LegalisationCost.first * Entry->Cost + 1;
863 return LegalisationCost.first * Entry->Cost;
867 case Intrinsic::ctpop: {
871 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
874 return LT.first + ExtraCost;
876 if (!ST->hasNEON()) {
906 RetTy->getScalarSizeInBits()
909 return LT.first * Entry->Cost + ExtraCost;
913 case Intrinsic::sadd_with_overflow:
914 case Intrinsic::uadd_with_overflow:
915 case Intrinsic::ssub_with_overflow:
916 case Intrinsic::usub_with_overflow:
917 case Intrinsic::smul_with_overflow:
918 case Intrinsic::umul_with_overflow: {
920 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
921 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
922 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
923 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
924 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
925 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
926 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
927 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
928 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
929 {Intrinsic::usub_with_overflow, MVT::i8, 3},
930 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
931 {Intrinsic::usub_with_overflow, MVT::i16, 3},
932 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
933 {Intrinsic::usub_with_overflow, MVT::i32, 1},
934 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
935 {Intrinsic::usub_with_overflow, MVT::i64, 1},
936 {Intrinsic::smul_with_overflow, MVT::i8, 5},
937 {Intrinsic::umul_with_overflow, MVT::i8, 4},
938 {Intrinsic::smul_with_overflow, MVT::i16, 5},
939 {Intrinsic::umul_with_overflow, MVT::i16, 4},
940 {Intrinsic::smul_with_overflow, MVT::i32, 2},
941 {Intrinsic::umul_with_overflow, MVT::i32, 2},
942 {Intrinsic::smul_with_overflow, MVT::i64, 3},
943 {Intrinsic::umul_with_overflow, MVT::i64, 3},
945 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
952 case Intrinsic::fptosi_sat:
953 case Intrinsic::fptoui_sat: {
956 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
958 EVT MTy = TLI->getValueType(
DL, RetTy);
961 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
962 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
963 LT.second == MVT::v2f64)) {
965 (LT.second == MVT::f64 && MTy == MVT::i32) ||
966 (LT.second == MVT::f32 && MTy == MVT::i64)))
975 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
982 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
983 (LT.second == MVT::f16 && MTy == MVT::i64) ||
984 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
998 if ((LT.second.getScalarType() == MVT::f32 ||
999 LT.second.getScalarType() == MVT::f64 ||
1000 LT.second.getScalarType() == MVT::f16) &&
1003 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1004 if (LT.second.isVector())
1005 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1009 LegalTy, {LegalTy, LegalTy});
1013 LegalTy, {LegalTy, LegalTy});
1015 return LT.first *
Cost +
1016 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1022 RetTy = RetTy->getScalarType();
1023 if (LT.second.isVector()) {
1041 return LT.first *
Cost;
1043 case Intrinsic::fshl:
1044 case Intrinsic::fshr: {
1053 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1054 (RetTy->getPrimitiveSizeInBits() == 32 ||
1055 RetTy->getPrimitiveSizeInBits() == 64)) {
1068 {Intrinsic::fshl, MVT::v4i32, 2},
1069 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1070 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1071 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1077 return LegalisationCost.first * Entry->Cost;
1081 if (!RetTy->isIntegerTy())
1086 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1087 RetTy->getScalarSizeInBits() < 64) ||
1088 (RetTy->getScalarSizeInBits() % 64 != 0);
1089 unsigned ExtraCost = HigherCost ? 1 : 0;
1090 if (RetTy->getScalarSizeInBits() == 32 ||
1091 RetTy->getScalarSizeInBits() == 64)
1094 else if (HigherCost)
1098 return TyL.first + ExtraCost;
1100 case Intrinsic::get_active_lane_mask: {
1102 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1104 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1107 if (RetTy->isScalableTy()) {
1108 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1118 if (ST->hasSVE2p1() || ST->hasSME2()) {
1130 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1133 return Cost + (SplitCost * (
Cost - 1));
1148 case Intrinsic::experimental_vector_match: {
1151 unsigned SearchSize = NeedleTy->getNumElements();
1152 auto IsSupportedTypeAndSearchSize = [&]() {
1153 if (SearchVT == MVT::nxv8i16 || SearchVT == MVT::v8i16)
1154 return SearchSize == 8;
1156 if (SearchVT == MVT::nxv16i8 || SearchVT == MVT::v16i8 ||
1157 SearchVT == MVT::v8i8)
1158 return SearchSize == 8 || SearchSize == 16;
1163 if (!ST->hasSVE2() || !ST->isSVEAvailable() ||
1164 !IsSupportedTypeAndSearchSize())
1177 case Intrinsic::cttz: {
1179 if (LT.second == MVT::i32 || LT.second == MVT::i64) {
1182 LT.second.getSizeInBits() > RetTy->getScalarSizeInBits() ? 1 : 0;
1184 if (LT.second.getSizeInBits() < RetTy->getScalarSizeInBits())
1185 ExtraCost += (LT.first - 1) * 3;
1187 return LT.first * (ST->hasCSSC() ? 1 : 2) + ExtraCost;
1191 {Intrinsic::cttz, MVT::v8i8, 2},
1192 {Intrinsic::cttz, MVT::v16i8, 2},
1193 {Intrinsic::cttz, MVT::v4i16, 3},
1194 {Intrinsic::cttz, MVT::v8i16, 3},
1195 {Intrinsic::cttz, MVT::v2i32, 3},
1196 {Intrinsic::cttz, MVT::v4i32, 3},
1197 {Intrinsic::cttz, MVT::v1i64, 6},
1198 {Intrinsic::cttz, MVT::v2i64, 6}};
1202 return LT.first * Entry->Cost;
1205 case Intrinsic::experimental_cttz_elts: {
1207 if (!getTLI()->shouldExpandCttzElements(ArgVT)) {
1215 case Intrinsic::loop_dependence_raw_mask:
1216 case Intrinsic::loop_dependence_war_mask: {
1218 if (ST->hasSVE2() || ST->hasSME()) {
1219 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1220 unsigned EltSizeInBytes =
1230 case Intrinsic::experimental_vector_extract_last_active:
1231 if (ST->isSVEorStreamingSVEAvailable()) {
1237 case Intrinsic::pow: {
1240 EVT VT = getTLI()->getValueType(
DL, RetTy);
1241 RTLIB::Libcall LC = RTLIB::getPOW(VT);
1242 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1257 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1258 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1268 return (Sqrt * 2) +
FMul;
1279 case Intrinsic::sqrt:
1280 case Intrinsic::fabs:
1281 case Intrinsic::ceil:
1282 case Intrinsic::floor:
1283 case Intrinsic::nearbyint:
1284 case Intrinsic::round:
1285 case Intrinsic::rint:
1286 case Intrinsic::roundeven:
1287 case Intrinsic::trunc:
1288 case Intrinsic::minnum:
1289 case Intrinsic::maxnum:
1290 case Intrinsic::minimum:
1291 case Intrinsic::maximum: {
1309 auto RequiredType =
II.getType();
1312 assert(PN &&
"Expected Phi Node!");
1315 if (!PN->hasOneUse())
1316 return std::nullopt;
1318 for (
Value *IncValPhi : PN->incoming_values()) {
1321 Reinterpret->getIntrinsicID() !=
1322 Intrinsic::aarch64_sve_convert_to_svbool ||
1323 RequiredType != Reinterpret->getArgOperand(0)->getType())
1324 return std::nullopt;
1332 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1334 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1407 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1412 return GoverningPredicateIdx;
1417 GoverningPredicateIdx = Index;
1439 return UndefIntrinsic;
1444 UndefIntrinsic = IID;
1471 return CmpPredicate;
1476 CmpPredicate = Pred;
1492 return ResultLanes == InactiveLanesTakenFromOperand;
1497 return OperandIdxForInactiveLanes;
1501 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1502 ResultLanes = InactiveLanesTakenFromOperand;
1503 OperandIdxForInactiveLanes = Index;
1508 return ResultLanes == InactiveLanesAreNotDefined;
1512 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1513 ResultLanes = InactiveLanesAreNotDefined;
1518 return ResultLanes == InactiveLanesAreUnused;
1522 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1523 ResultLanes = InactiveLanesAreUnused;
1533 ResultIsZeroInitialized =
true;
1544 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1549 return OperandIdxWithNoActiveLanes;
1554 OperandIdxWithNoActiveLanes = Index;
1559 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1562 unsigned IROpcode = 0;
1565 enum PredicationStyle {
1567 InactiveLanesTakenFromOperand,
1568 InactiveLanesAreNotDefined,
1569 InactiveLanesAreUnused
1572 bool ResultIsZeroInitialized =
false;
1573 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1574 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1582 return !isa<ScalableVectorType>(V->getType());
1590 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1591 case Intrinsic::aarch64_sve_fcvt_f16f32:
1592 case Intrinsic::aarch64_sve_fcvt_f16f64:
1593 case Intrinsic::aarch64_sve_fcvt_f32f16:
1594 case Intrinsic::aarch64_sve_fcvt_f32f64:
1595 case Intrinsic::aarch64_sve_fcvt_f64f16:
1596 case Intrinsic::aarch64_sve_fcvt_f64f32:
1597 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1598 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1599 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1600 case Intrinsic::aarch64_sve_fcvtzs:
1601 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1602 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1603 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1604 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1605 case Intrinsic::aarch64_sve_fcvtzu:
1606 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1607 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1608 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1609 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1610 case Intrinsic::aarch64_sve_revb:
1611 case Intrinsic::aarch64_sve_revh:
1612 case Intrinsic::aarch64_sve_revw:
1613 case Intrinsic::aarch64_sve_revd:
1614 case Intrinsic::aarch64_sve_scvtf:
1615 case Intrinsic::aarch64_sve_scvtf_f16i32:
1616 case Intrinsic::aarch64_sve_scvtf_f16i64:
1617 case Intrinsic::aarch64_sve_scvtf_f32i64:
1618 case Intrinsic::aarch64_sve_scvtf_f64i32:
1619 case Intrinsic::aarch64_sve_ucvtf:
1620 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1621 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1622 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1623 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1626 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1627 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1628 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1629 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1632 case Intrinsic::aarch64_sve_fabd:
1634 case Intrinsic::aarch64_sve_fadd:
1637 case Intrinsic::aarch64_sve_fdiv:
1640 case Intrinsic::aarch64_sve_fmax:
1642 case Intrinsic::aarch64_sve_fmaxnm:
1644 case Intrinsic::aarch64_sve_fmin:
1646 case Intrinsic::aarch64_sve_fminnm:
1648 case Intrinsic::aarch64_sve_fmla:
1650 case Intrinsic::aarch64_sve_fmls:
1652 case Intrinsic::aarch64_sve_fmul:
1655 case Intrinsic::aarch64_sve_fmulx:
1657 case Intrinsic::aarch64_sve_fnmla:
1659 case Intrinsic::aarch64_sve_fnmls:
1661 case Intrinsic::aarch64_sve_fsub:
1664 case Intrinsic::aarch64_sve_add:
1667 case Intrinsic::aarch64_sve_mla:
1669 case Intrinsic::aarch64_sve_mls:
1671 case Intrinsic::aarch64_sve_mul:
1674 case Intrinsic::aarch64_sve_sabd:
1676 case Intrinsic::aarch64_sve_sdiv:
1679 case Intrinsic::aarch64_sve_smax:
1681 case Intrinsic::aarch64_sve_smin:
1683 case Intrinsic::aarch64_sve_smulh:
1685 case Intrinsic::aarch64_sve_sub:
1688 case Intrinsic::aarch64_sve_uabd:
1690 case Intrinsic::aarch64_sve_udiv:
1693 case Intrinsic::aarch64_sve_umax:
1695 case Intrinsic::aarch64_sve_umin:
1697 case Intrinsic::aarch64_sve_umulh:
1699 case Intrinsic::aarch64_sve_asr:
1702 case Intrinsic::aarch64_sve_lsl:
1705 case Intrinsic::aarch64_sve_lsr:
1708 case Intrinsic::aarch64_sve_and:
1711 case Intrinsic::aarch64_sve_bic:
1713 case Intrinsic::aarch64_sve_eor:
1716 case Intrinsic::aarch64_sve_orr:
1719 case Intrinsic::aarch64_sve_shsub:
1721 case Intrinsic::aarch64_sve_shsubr:
1723 case Intrinsic::aarch64_sve_sqrshl:
1725 case Intrinsic::aarch64_sve_sqshl:
1727 case Intrinsic::aarch64_sve_sqsub:
1729 case Intrinsic::aarch64_sve_srshl:
1731 case Intrinsic::aarch64_sve_uhsub:
1733 case Intrinsic::aarch64_sve_uhsubr:
1735 case Intrinsic::aarch64_sve_uqrshl:
1737 case Intrinsic::aarch64_sve_uqshl:
1739 case Intrinsic::aarch64_sve_uqsub:
1741 case Intrinsic::aarch64_sve_urshl:
1744 case Intrinsic::aarch64_sve_add_u:
1747 case Intrinsic::aarch64_sve_and_u:
1750 case Intrinsic::aarch64_sve_asr_u:
1753 case Intrinsic::aarch64_sve_eor_u:
1756 case Intrinsic::aarch64_sve_fadd_u:
1759 case Intrinsic::aarch64_sve_fdiv_u:
1762 case Intrinsic::aarch64_sve_fmul_u:
1765 case Intrinsic::aarch64_sve_fsub_u:
1768 case Intrinsic::aarch64_sve_lsl_u:
1771 case Intrinsic::aarch64_sve_lsr_u:
1774 case Intrinsic::aarch64_sve_mul_u:
1777 case Intrinsic::aarch64_sve_orr_u:
1780 case Intrinsic::aarch64_sve_sdiv_u:
1783 case Intrinsic::aarch64_sve_sub_u:
1786 case Intrinsic::aarch64_sve_udiv_u:
1790 case Intrinsic::aarch64_sve_addqv:
1791 case Intrinsic::aarch64_sve_bic_z:
1792 case Intrinsic::aarch64_sve_brka_z:
1793 case Intrinsic::aarch64_sve_brkb_z:
1794 case Intrinsic::aarch64_sve_brkn_z:
1795 case Intrinsic::aarch64_sve_brkpa_z:
1796 case Intrinsic::aarch64_sve_brkpb_z:
1797 case Intrinsic::aarch64_sve_cntp:
1798 case Intrinsic::aarch64_sve_compact:
1799 case Intrinsic::aarch64_sve_eorv:
1800 case Intrinsic::aarch64_sve_eorqv:
1801 case Intrinsic::aarch64_sve_nand_z:
1802 case Intrinsic::aarch64_sve_nor_z:
1803 case Intrinsic::aarch64_sve_orn_z:
1804 case Intrinsic::aarch64_sve_orv:
1805 case Intrinsic::aarch64_sve_orqv:
1806 case Intrinsic::aarch64_sve_pnext:
1807 case Intrinsic::aarch64_sve_rdffr_z:
1808 case Intrinsic::aarch64_sve_saddv:
1809 case Intrinsic::aarch64_sve_uaddv:
1810 case Intrinsic::aarch64_sve_umaxv:
1811 case Intrinsic::aarch64_sve_umaxqv:
1812 case Intrinsic::aarch64_sve_facge:
1813 case Intrinsic::aarch64_sve_facgt:
1814 case Intrinsic::aarch64_sve_ld1:
1815 case Intrinsic::aarch64_sve_ld1_gather:
1816 case Intrinsic::aarch64_sve_ld1_gather_index:
1817 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1818 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1819 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1820 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1821 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1822 case Intrinsic::aarch64_sve_ld1q_gather_index:
1823 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1824 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1825 case Intrinsic::aarch64_sve_ld1ro:
1826 case Intrinsic::aarch64_sve_ld1rq:
1827 case Intrinsic::aarch64_sve_ld1udq:
1828 case Intrinsic::aarch64_sve_ld1uwq:
1829 case Intrinsic::aarch64_sve_ld2_sret:
1830 case Intrinsic::aarch64_sve_ld2q_sret:
1831 case Intrinsic::aarch64_sve_ld3_sret:
1832 case Intrinsic::aarch64_sve_ld3q_sret:
1833 case Intrinsic::aarch64_sve_ld4_sret:
1834 case Intrinsic::aarch64_sve_ld4q_sret:
1835 case Intrinsic::aarch64_sve_ldff1:
1836 case Intrinsic::aarch64_sve_ldff1_gather:
1837 case Intrinsic::aarch64_sve_ldff1_gather_index:
1838 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1839 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1840 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1841 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1842 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1843 case Intrinsic::aarch64_sve_ldnf1:
1844 case Intrinsic::aarch64_sve_ldnt1:
1845 case Intrinsic::aarch64_sve_ldnt1_gather:
1846 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1847 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1848 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1851 case Intrinsic::aarch64_sve_and_z:
1854 case Intrinsic::aarch64_sve_orr_z:
1857 case Intrinsic::aarch64_sve_eor_z:
1861 case Intrinsic::aarch64_sve_cmpeq:
1862 case Intrinsic::aarch64_sve_cmpeq_wide:
1865 case Intrinsic::aarch64_sve_cmpge:
1866 case Intrinsic::aarch64_sve_cmpge_wide:
1869 case Intrinsic::aarch64_sve_cmpgt:
1870 case Intrinsic::aarch64_sve_cmpgt_wide:
1873 case Intrinsic::aarch64_sve_cmphi:
1874 case Intrinsic::aarch64_sve_cmphi_wide:
1877 case Intrinsic::aarch64_sve_cmphs:
1878 case Intrinsic::aarch64_sve_cmphs_wide:
1881 case Intrinsic::aarch64_sve_cmple_wide:
1884 case Intrinsic::aarch64_sve_cmplo_wide:
1887 case Intrinsic::aarch64_sve_cmpls_wide:
1890 case Intrinsic::aarch64_sve_cmplt_wide:
1893 case Intrinsic::aarch64_sve_cmpne:
1894 case Intrinsic::aarch64_sve_cmpne_wide:
1897 case Intrinsic::aarch64_sve_fcmpeq:
1900 case Intrinsic::aarch64_sve_fcmpge:
1903 case Intrinsic::aarch64_sve_fcmpgt:
1906 case Intrinsic::aarch64_sve_fcmpne:
1909 case Intrinsic::aarch64_sve_fcmpuo:
1913 case Intrinsic::aarch64_sve_prf:
1914 case Intrinsic::aarch64_sve_prfb_gather_index:
1915 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
1916 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
1917 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
1918 case Intrinsic::aarch64_sve_prfd_gather_index:
1919 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
1920 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
1921 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
1922 case Intrinsic::aarch64_sve_prfh_gather_index:
1923 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
1924 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
1925 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
1926 case Intrinsic::aarch64_sve_prfw_gather_index:
1927 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
1928 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
1929 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
1932 case Intrinsic::aarch64_sve_st1_scatter:
1933 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
1934 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
1935 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
1936 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
1937 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
1938 case Intrinsic::aarch64_sve_st1dq:
1939 case Intrinsic::aarch64_sve_st1q_scatter_index:
1940 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
1941 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
1942 case Intrinsic::aarch64_sve_st1wq:
1943 case Intrinsic::aarch64_sve_stnt1:
1944 case Intrinsic::aarch64_sve_stnt1_scatter:
1945 case Intrinsic::aarch64_sve_stnt1_scatter_index:
1946 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
1947 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
1949 case Intrinsic::aarch64_sve_st2:
1950 case Intrinsic::aarch64_sve_st2q:
1952 case Intrinsic::aarch64_sve_st3:
1953 case Intrinsic::aarch64_sve_st3q:
1955 case Intrinsic::aarch64_sve_st4:
1956 case Intrinsic::aarch64_sve_st4q:
1964 Value *UncastedPred;
1970 Pred = UncastedPred;
1976 if (OrigPredTy->getMinNumElements() <=
1978 ->getMinNumElements())
1979 Pred = UncastedPred;
1983 return C &&
C->isAllOnesValue();
1990 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
1991 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
1999static std::optional<Instruction *>
2006 Value *Op1 =
II.getOperand(1);
2007 Value *Op2 =
II.getOperand(2);
2033 return std::nullopt;
2044 if (SimpleII == Inactive)
2052static std::optional<Instruction *>
2056 assert((
Opc == Instruction::ICmp ||
Opc == Instruction::FCmp) &&
2057 "Expected a compare operation!");
2064 Opc == Instruction::ICmp &&
LHS->getType() !=
RHS->getType();
2065 assert((IsWideICmp ||
LHS->getType() ==
RHS->getType()) &&
2066 "Unexpected wide compare!");
2082 const APInt *LHSVal, *RHSVal;
2084 return std::nullopt;
2107 return std::nullopt;
2121static std::optional<Instruction *>
2125 return std::nullopt;
2154 II.setCalledFunction(NewDecl);
2160 return std::nullopt;
2171 if (
Opc == Instruction::FCmp ||
Opc == Instruction::ICmp)
2174 return std::nullopt;
2186static std::optional<Instruction *>
2188 auto m_ConvertToSVBool = [](
auto P) {
2192 Intrinsic::aarch64_sve_convert_from_svbool;
2215 return std::nullopt;
2219 case Intrinsic::aarch64_sve_and_z:
2220 case Intrinsic::aarch64_sve_bic_z:
2221 case Intrinsic::aarch64_sve_eor_z:
2222 case Intrinsic::aarch64_sve_nand_z:
2223 case Intrinsic::aarch64_sve_nor_z:
2224 case Intrinsic::aarch64_sve_orn_z:
2225 case Intrinsic::aarch64_sve_orr_z:
2228 return std::nullopt;
2231 Value *BinOpPred = BinOp->getOperand(0);
2232 Value *BinOpOp1 = BinOp->getOperand(1);
2233 Value *BinOpOp2 = BinOp->getOperand(2);
2235 Value *NarrowBinOpPred;
2237 return std::nullopt;
2239 Value *NarrowBinOpOp1 =
2241 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2242 if (BinOpOp1 != BinOpOp2)
2246 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2250static std::optional<Instruction *>
2257 return BinOpCombine;
2262 return std::nullopt;
2265 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2274 if (CursorVTy->getElementCount().getKnownMinValue() <
2275 IVTy->getElementCount().getKnownMinValue())
2279 if (Cursor->getType() == IVTy)
2280 EarliestReplacement = Cursor;
2285 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2286 Intrinsic::aarch64_sve_convert_to_svbool ||
2287 IntrinsicCursor->getIntrinsicID() ==
2288 Intrinsic::aarch64_sve_convert_from_svbool))
2291 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2292 Cursor = IntrinsicCursor->getOperand(0);
2297 if (!EarliestReplacement)
2298 return std::nullopt;
2306 auto *OpPredicate =
II.getOperand(0);
2323 II.getArgOperand(2));
2329 return std::nullopt;
2333 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2342 II.getArgOperand(0));
2351 if (!
II.hasOneUse())
2352 return std::nullopt;
2355 return std::nullopt;
2358 switch (
II.getIntrinsicID()) {
2359 case Intrinsic::aarch64_sve_cmpne:
2360 IID = Intrinsic::aarch64_sve_cmpeq;
2362 case Intrinsic::aarch64_sve_cmpne_wide:
2363 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2365 case Intrinsic::aarch64_sve_cmpeq:
2366 IID = Intrinsic::aarch64_sve_cmpne;
2368 case Intrinsic::aarch64_sve_cmpeq_wide:
2369 IID = Intrinsic::aarch64_sve_cmpne_wide;
2372 return std::nullopt;
2377 IID,
II.getOperand(1)->getType(),
2378 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2390 return std::nullopt;
2392 for (
auto *U :
II.users()) {
2395 Type *Ty =
II.getOperand(1)->getType();
2400 Intrinsic::aarch64_sve_umin, Ty,
2401 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2407 return std::nullopt;
2421 return std::nullopt;
2426 if (!SplatValue || !SplatValue->isZero())
2427 return std::nullopt;
2432 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2433 return std::nullopt;
2437 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2438 return std::nullopt;
2441 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2442 return std::nullopt;
2447 return std::nullopt;
2450 return std::nullopt;
2454 return std::nullopt;
2458 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2459 return std::nullopt;
2461 unsigned NumElts = VecTy->getNumElements();
2462 unsigned PredicateBits = 0;
2465 for (
unsigned I = 0;
I < NumElts; ++
I) {
2468 return std::nullopt;
2470 PredicateBits |= 1 << (
I * (16 / NumElts));
2474 if (PredicateBits == 0) {
2476 PFalse->takeName(&
II);
2482 for (
unsigned I = 0;
I < 16; ++
I)
2483 if ((PredicateBits & (1 <<
I)) != 0)
2486 unsigned PredSize = Mask & -Mask;
2491 for (
unsigned I = 0;
I < 16;
I += PredSize)
2492 if ((PredicateBits & (1 <<
I)) == 0)
2493 return std::nullopt;
2495 auto *ConvertToSVBool =
2498 auto *ConvertFromSVBool =
2500 II.getType(), ConvertToSVBool);
2508 Value *Pg =
II.getArgOperand(0);
2509 Value *Vec =
II.getArgOperand(1);
2510 auto IntrinsicID =
II.getIntrinsicID();
2511 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2523 auto OpC = OldBinOp->getOpcode();
2529 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2535 if (IsAfter &&
C &&
C->isNullValue()) {
2539 Extract->insertBefore(
II.getIterator());
2540 Extract->takeName(&
II);
2546 return std::nullopt;
2548 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2549 return std::nullopt;
2551 const auto PTruePattern =
2557 return std::nullopt;
2559 unsigned Idx = MinNumElts - 1;
2569 if (Idx >= PgVTy->getMinNumElements())
2570 return std::nullopt;
2575 Extract->insertBefore(
II.getIterator());
2576 Extract->takeName(&
II);
2589 Value *Pg =
II.getArgOperand(0);
2591 Value *Vec =
II.getArgOperand(2);
2594 if (!Ty->isIntegerTy())
2595 return std::nullopt;
2600 return std::nullopt;
2617 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2632static std::optional<Instruction *>
2636 if (
Pattern == AArch64SVEPredPattern::all) {
2645 return MinNumElts && NumElts >= MinNumElts
2647 II, ConstantInt::get(
II.getType(), MinNumElts)))
2651static std::optional<Instruction *>
2654 if (!ST->isStreaming())
2655 return std::nullopt;
2667 Value *PgVal =
II.getArgOperand(0);
2668 Value *OpVal =
II.getArgOperand(1);
2672 if (PgVal == OpVal &&
2673 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2674 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2689 return std::nullopt;
2693 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2694 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2708 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2709 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2710 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2711 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2712 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2713 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2714 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2715 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2716 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2717 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2718 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2719 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2720 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2730 return std::nullopt;
2733template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2734static std::optional<Instruction *>
2736 bool MergeIntoAddendOp) {
2738 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2739 if (MergeIntoAddendOp) {
2740 AddendOp =
II.getOperand(1);
2741 Mul =
II.getOperand(2);
2743 AddendOp =
II.getOperand(2);
2744 Mul =
II.getOperand(1);
2749 return std::nullopt;
2751 if (!
Mul->hasOneUse())
2752 return std::nullopt;
2755 if (
II.getType()->isFPOrFPVectorTy()) {
2760 return std::nullopt;
2762 return std::nullopt;
2767 if (MergeIntoAddendOp)
2777static std::optional<Instruction *>
2779 Value *Pred =
II.getOperand(0);
2780 Value *PtrOp =
II.getOperand(1);
2781 Type *VecTy =
II.getType();
2796static std::optional<Instruction *>
2798 Value *VecOp =
II.getOperand(0);
2799 Value *Pred =
II.getOperand(1);
2800 Value *PtrOp =
II.getOperand(2);
2816 case Intrinsic::aarch64_sve_fmul_u:
2817 return Instruction::BinaryOps::FMul;
2818 case Intrinsic::aarch64_sve_fadd_u:
2819 return Instruction::BinaryOps::FAdd;
2820 case Intrinsic::aarch64_sve_fsub_u:
2821 return Instruction::BinaryOps::FSub;
2823 return Instruction::BinaryOpsEnd;
2827static std::optional<Instruction *>
2830 if (
II.isStrictFP())
2831 return std::nullopt;
2833 auto *OpPredicate =
II.getOperand(0);
2835 if (BinOpCode == Instruction::BinaryOpsEnd ||
2837 return std::nullopt;
2839 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
2843static std::optional<Instruction *>
2845 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
2846 "Expected MLA_U intrinsic");
2847 Value *Acc =
II.getArgOperand(1);
2848 Value *MulOp0 =
II.getArgOperand(2);
2849 Value *MulOp1 =
II.getArgOperand(3);
2864 II.setArgOperand(2, MulOp1);
2865 II.setArgOperand(3, MulOp0);
2869 return std::nullopt;
2872static std::optional<Instruction *>
2874 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
2875 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
2876 "Expected SADALP or UADALP intrinsic");
2882 return std::nullopt;
2886 return std::nullopt;
2890 II.getIntrinsicID(), {II.getType()},
2891 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
2901 Intrinsic::aarch64_sve_mla>(
2905 Intrinsic::aarch64_sve_mad>(
2908 return std::nullopt;
2911static std::optional<Instruction *>
2915 Intrinsic::aarch64_sve_fmla>(IC,
II,
2920 Intrinsic::aarch64_sve_fmad>(IC,
II,
2925 Intrinsic::aarch64_sve_fmla>(IC,
II,
2928 return std::nullopt;
2931static std::optional<Instruction *>
2935 Intrinsic::aarch64_sve_fmla>(IC,
II,
2940 Intrinsic::aarch64_sve_fmad>(IC,
II,
2945 Intrinsic::aarch64_sve_fmla_u>(
2951static std::optional<Instruction *>
2955 Intrinsic::aarch64_sve_fmls>(IC,
II,
2960 Intrinsic::aarch64_sve_fnmsb>(
2965 Intrinsic::aarch64_sve_fmls>(IC,
II,
2968 return std::nullopt;
2971static std::optional<Instruction *>
2975 Intrinsic::aarch64_sve_fmls>(IC,
II,
2980 Intrinsic::aarch64_sve_fnmsb>(
2985 Intrinsic::aarch64_sve_fmls_u>(
2994 Intrinsic::aarch64_sve_mls>(
2997 return std::nullopt;
3002 Value *UnpackArg =
II.getArgOperand(0);
3004 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
3005 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
3018 return std::nullopt;
3022 auto *OpVal =
II.getOperand(0);
3023 auto *OpIndices =
II.getOperand(1);
3030 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
3031 return std::nullopt;
3046 Type *RetTy =
II.getType();
3047 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
3048 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
3052 if ((
match(
II.getArgOperand(0),
3059 if (TyA ==
B->getType() &&
3064 TyA->getMinNumElements());
3070 return std::nullopt;
3078 if (
match(
II.getArgOperand(0),
3083 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
3085 return std::nullopt;
3088static std::optional<Instruction *>
3090 Value *Mask =
II.getOperand(0);
3091 Value *BasePtr =
II.getOperand(1);
3092 Value *Index =
II.getOperand(2);
3103 BasePtr->getPointerAlignment(
II.getDataLayout());
3106 BasePtr, IndexBase);
3113 return std::nullopt;
3116static std::optional<Instruction *>
3118 Value *Val =
II.getOperand(0);
3119 Value *Mask =
II.getOperand(1);
3120 Value *BasePtr =
II.getOperand(2);
3121 Value *Index =
II.getOperand(3);
3131 BasePtr->getPointerAlignment(
II.getDataLayout());
3134 BasePtr, IndexBase);
3140 return std::nullopt;
3146 Value *Pred =
II.getOperand(0);
3147 Value *Vec =
II.getOperand(1);
3148 Value *DivVec =
II.getOperand(2);
3152 if (!SplatConstantInt)
3153 return std::nullopt;
3157 if (DivisorValue == -1)
3158 return std::nullopt;
3159 if (DivisorValue == 1)
3165 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3172 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3174 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3178 return std::nullopt;
3182 size_t VecSize = Vec.
size();
3187 size_t HalfVecSize = VecSize / 2;
3191 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3199 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3217 return std::nullopt;
3224 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3225 CurrentInsertElt = InsertElt->getOperand(0);
3231 return std::nullopt;
3235 for (
size_t I = 0;
I < Elts.
size();
I++) {
3236 if (Elts[
I] ==
nullptr)
3241 if (InsertEltChain ==
nullptr)
3242 return std::nullopt;
3248 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3249 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3250 IIScalableTy->getMinNumElements() /
3255 auto *WideShuffleMaskTy =
3266 auto NarrowBitcast =
3279 return std::nullopt;
3284 Value *Pred =
II.getOperand(0);
3285 Value *Vec =
II.getOperand(1);
3286 Value *Shift =
II.getOperand(2);
3289 Value *AbsPred, *MergedValue;
3295 return std::nullopt;
3303 return std::nullopt;
3308 return std::nullopt;
3311 {
II.getType()}, {Pred, Vec, Shift});
3318 Value *Vec =
II.getOperand(0);
3323 return std::nullopt;
3329 auto *NI =
II.getNextNode();
3332 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3334 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3335 auto *NIBB = NI->getParent();
3336 NI = NI->getNextNode();
3338 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3339 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3345 if (NextII &&
II.isIdenticalTo(NextII))
3348 return std::nullopt;
3356 {II.getType(), II.getOperand(0)->getType()},
3357 {II.getOperand(0), II.getOperand(1)}));
3364 if (PredPattern == AArch64SVEPredPattern::all ||
3365 PredPattern == AArch64SVEPredPattern::pow2)
3367 return std::nullopt;
3373 Value *Passthru =
II.getOperand(0);
3381 auto *Mask = ConstantInt::get(Ty, MaskValue);
3387 return std::nullopt;
3390static std::optional<Instruction *>
3397 return std::nullopt;
3403 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3413 UMinID,
II.getType(), {Pg, NewUMin, ConstantInt::get(II.getType(), 1)});
3423 return std::nullopt;
3429 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3437 return std::nullopt;
3440 II.getType(), {Pg, A, B});
3442 UMinID,
II.getType(), {Pg, NewOrr, ConstantInt::get(II.getType(), 1)});
3451 constexpr Intrinsic::ID CmphsID = Intrinsic::aarch64_sve_cmphs;
3456 Value *
A, *PgLHS, *PgRHS;
3462 !
LHS->hasOneUser() || !
RHS->hasOneUser())
3463 return std::nullopt;
3466 if (ConstB > ConstA)
3472 if (PgLHS != PgRHS || (Pg !=
LHS && Pg !=
RHS && Pg != PgLHS))
3473 return std::nullopt;
3475 Type *VecTy =
A->getType();
3479 Constant *Limit = ConstantInt::get(VecTy, ConstA - ConstB);
3486std::optional<Instruction *>
3497 case Intrinsic::aarch64_dmb:
3499 case Intrinsic::aarch64_neon_fmaxnm:
3500 case Intrinsic::aarch64_neon_fminnm:
3502 case Intrinsic::aarch64_sve_convert_from_svbool:
3504 case Intrinsic::aarch64_sve_dup:
3506 case Intrinsic::aarch64_sve_dup_x:
3508 case Intrinsic::aarch64_sve_cmpeq:
3509 case Intrinsic::aarch64_sve_cmpeq_wide:
3511 case Intrinsic::aarch64_sve_cmpne:
3512 case Intrinsic::aarch64_sve_cmpne_wide:
3514 case Intrinsic::aarch64_sve_rdffr:
3516 case Intrinsic::aarch64_sve_lasta:
3517 case Intrinsic::aarch64_sve_lastb:
3519 case Intrinsic::aarch64_sve_clasta_n:
3520 case Intrinsic::aarch64_sve_clastb_n:
3522 case Intrinsic::aarch64_sve_cntd:
3524 case Intrinsic::aarch64_sve_cntw:
3526 case Intrinsic::aarch64_sve_cnth:
3528 case Intrinsic::aarch64_sve_cntb:
3530 case Intrinsic::aarch64_sme_cntsd:
3532 case Intrinsic::aarch64_sve_ptest_any:
3533 case Intrinsic::aarch64_sve_ptest_first:
3534 case Intrinsic::aarch64_sve_ptest_last:
3536 case Intrinsic::aarch64_sve_fadd:
3538 case Intrinsic::aarch64_sve_fadd_u:
3540 case Intrinsic::aarch64_sve_fmul_u:
3542 case Intrinsic::aarch64_sve_fsub:
3544 case Intrinsic::aarch64_sve_fsub_u:
3546 case Intrinsic::aarch64_sve_add:
3548 case Intrinsic::aarch64_sve_add_u:
3550 Intrinsic::aarch64_sve_mla_u>(
3552 case Intrinsic::aarch64_sve_mla_u:
3554 case Intrinsic::aarch64_sve_sadalp:
3555 case Intrinsic::aarch64_sve_uadalp:
3557 case Intrinsic::aarch64_sve_sub:
3559 case Intrinsic::aarch64_sve_sub_u:
3561 Intrinsic::aarch64_sve_mls_u>(
3563 case Intrinsic::aarch64_sve_tbl:
3565 case Intrinsic::aarch64_sve_uunpkhi:
3566 case Intrinsic::aarch64_sve_uunpklo:
3567 case Intrinsic::aarch64_sve_sunpkhi:
3568 case Intrinsic::aarch64_sve_sunpklo:
3570 case Intrinsic::aarch64_sve_uzp1:
3572 case Intrinsic::aarch64_sve_zip1:
3573 case Intrinsic::aarch64_sve_zip2:
3575 case Intrinsic::aarch64_sve_ld1_gather_index:
3577 case Intrinsic::aarch64_sve_st1_scatter_index:
3579 case Intrinsic::aarch64_sve_ld1:
3581 case Intrinsic::aarch64_sve_st1:
3583 case Intrinsic::aarch64_sve_sdiv:
3585 case Intrinsic::aarch64_sve_sel:
3587 case Intrinsic::aarch64_sve_srshl:
3589 case Intrinsic::aarch64_sve_dupq_lane:
3591 case Intrinsic::aarch64_sve_insr:
3593 case Intrinsic::aarch64_sve_whilelo:
3595 case Intrinsic::aarch64_sve_ptrue:
3597 case Intrinsic::aarch64_sve_uxtb:
3599 case Intrinsic::aarch64_sve_uxth:
3601 case Intrinsic::aarch64_sve_uxtw:
3603 case Intrinsic::aarch64_sme_in_streaming_mode:
3605 case Intrinsic::aarch64_sve_umin_u:
3607 case Intrinsic::aarch64_sve_orr_u:
3609 case Intrinsic::aarch64_sve_and_z:
3613 return std::nullopt;
3620 SimplifyAndSetOp)
const {
3621 switch (
II.getIntrinsicID()) {
3624 case Intrinsic::aarch64_neon_fcvtxn:
3625 case Intrinsic::aarch64_neon_rshrn:
3626 case Intrinsic::aarch64_neon_sqrshrn:
3627 case Intrinsic::aarch64_neon_sqrshrun:
3628 case Intrinsic::aarch64_neon_sqshrn:
3629 case Intrinsic::aarch64_neon_sqshrun:
3630 case Intrinsic::aarch64_neon_sqxtn:
3631 case Intrinsic::aarch64_neon_sqxtun:
3632 case Intrinsic::aarch64_neon_uqrshrn:
3633 case Intrinsic::aarch64_neon_uqshrn:
3634 case Intrinsic::aarch64_neon_uqxtn:
3635 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3639 return std::nullopt;
3643 return ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3653 if (ST->useSVEForFixedLengthVectors() &&
3656 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3657 else if (ST->isNeonAvailable())
3662 if (ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3671bool AArch64TTIImpl::isSingleExtWideningInstruction(
3673 Type *SrcOverrideTy)
const {
3688 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3691 Type *SrcTy = SrcOverrideTy;
3693 case Instruction::Add:
3694 case Instruction::Sub: {
3703 if (Opcode == Instruction::Sub)
3727 assert(SrcTy &&
"Expected some SrcTy");
3729 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3735 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3737 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3741 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3744Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3746 Type *SrcOverrideTy)
const {
3747 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3748 Opcode != Instruction::Mul)
3758 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3761 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3767 ->getScalarSizeInBits();
3770 unsigned MaxEltSize = 0;
3773 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3774 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3775 MaxEltSize = std::max(EltSize0, EltSize1);
3778 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3779 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3782 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3784 MaxEltSize = DstEltSize / 2;
3785 }
else if (Opcode == Instruction::Mul &&
3793 Known.Zero.countLeadingOnes() >
3798 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3802 if (MaxEltSize * 2 > DstEltSize)
3820 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3821 (Src->isScalableTy() && !ST->hasSVE2()))
3831 if (AddUser && AddUser->getOpcode() == Instruction::Add)
3835 if (!Shr || Shr->getOpcode() != Instruction::LShr)
3839 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
3840 Src->getScalarSizeInBits() !=
3864 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3868 if (
I &&
I->hasOneUser()) {
3871 if (
Type *ExtTy = isBinExtWideningInstruction(
3872 SingleUser->getOpcode(), Dst,
Operands,
3873 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3886 if (isSingleExtWideningInstruction(
3887 SingleUser->getOpcode(), Dst,
Operands,
3888 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3892 if (SingleUser->getOpcode() == Instruction::Add) {
3893 if (
I == SingleUser->getOperand(1) ||
3895 cast<CastInst>(SingleUser->getOperand(1))->getOpcode() == Opcode))
3910 EVT SrcTy = TLI->getValueType(
DL, Src);
3911 EVT DstTy = TLI->getValueType(
DL, Dst);
3913 if (!SrcTy.isSimple() || !DstTy.
isSimple())
3918 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
3947 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
3950 ST->useSVEForFixedLengthVectors(WiderTy)) {
3951 std::pair<InstructionCost, MVT> LT =
3953 unsigned NumElements =
3969 const unsigned int SVE_EXT_COST = 1;
3970 const unsigned int SVE_FCVT_COST = 1;
3971 const unsigned int SVE_UNPACK_ONCE = 4;
3972 const unsigned int SVE_UNPACK_TWICE = 16;
4101 SVE_EXT_COST + SVE_FCVT_COST},
4106 SVE_EXT_COST + SVE_FCVT_COST},
4113 SVE_EXT_COST + SVE_FCVT_COST},
4117 SVE_EXT_COST + SVE_FCVT_COST},
4123 SVE_EXT_COST + SVE_FCVT_COST},
4126 SVE_EXT_COST + SVE_FCVT_COST},
4131 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4133 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4143 SVE_EXT_COST + SVE_FCVT_COST},
4148 SVE_EXT_COST + SVE_FCVT_COST},
4161 SVE_EXT_COST + SVE_FCVT_COST},
4165 SVE_EXT_COST + SVE_FCVT_COST},
4177 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4179 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4181 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4183 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4187 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4189 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4205 SVE_EXT_COST + SVE_FCVT_COST},
4210 SVE_EXT_COST + SVE_FCVT_COST},
4221 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4223 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4225 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4227 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4229 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4231 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4235 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4237 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4239 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4241 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4466 if (ST->hasFullFP16())
4478 Src->getScalarType(), CCH,
CostKind) +
4486 ST->isSVEorStreamingSVEAvailable() &&
4487 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4489 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4498 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4501 return Part1 + Part2;
4508 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4520 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4533 CostKind, Index,
nullptr,
nullptr);
4537 auto DstVT = TLI->getValueType(
DL, Dst);
4538 auto SrcVT = TLI->getValueType(
DL, Src);
4543 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4549 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4559 case Instruction::SExt:
4564 case Instruction::ZExt:
4565 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4578 return Opcode == Instruction::PHI ? 0 : 1;
4587 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4596 if (!LT.second.isVector())
4601 if (LT.second.isFixedLengthVector()) {
4602 unsigned Width = LT.second.getVectorNumElements();
4603 Index = Index % Width;
4617 if (VIC == TTI::VectorInstrContext::Load) {
4618 if (ST->hasFastLD1Single())
4630 : ST->getVectorInsertExtractBaseCost() + 1;
4654 auto ExtractCanFuseWithFmul = [&]() {
4661 auto IsAllowedScalarTy = [&](
const Type *
T) {
4662 return T->isFloatTy() ||
T->isDoubleTy() ||
4663 (
T->isHalfTy() && ST->hasFullFP16());
4667 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4670 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4671 !BO->getType()->isVectorTy();
4676 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4680 return Idx == 0 || (RegWidth != 0 && (Idx * EltSz) % RegWidth == 0);
4689 DenseMap<User *, unsigned> UserToExtractIdx;
4690 for (
auto *U :
Scalar->users()) {
4691 if (!IsUserFMulScalarTy(U))
4695 UserToExtractIdx[
U];
4697 if (UserToExtractIdx.
empty())
4699 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4700 for (
auto *U : S->users()) {
4701 if (UserToExtractIdx.
contains(U)) {
4703 auto *Op0 =
FMul->getOperand(0);
4704 auto *Op1 =
FMul->getOperand(1);
4705 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4706 UserToExtractIdx[
U] =
L;
4712 for (
auto &[U, L] : UserToExtractIdx) {
4724 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4725 if (!IsUserFMulScalarTy(U))
4730 const auto *BO = cast<BinaryOperator>(U);
4731 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4732 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4734 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4737 return IsExtractLaneEquivalentToZero(
4738 cast<ConstantInt>(OtherEE->getIndexOperand())
4741 OtherEE->getType()->getScalarSizeInBits());
4749 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4750 ExtractCanFuseWithFmul())
4755 :
ST->getVectorInsertExtractBaseCost();
4764 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
4767 return getVectorInstrCostHelper(Opcode, Val,
CostKind, Index,
nullptr,
4773 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4775 return getVectorInstrCostHelper(Opcode, Val,
CostKind, Index,
nullptr, Scalar,
4776 ScalarUserAndIdx, VIC);
4783 return getVectorInstrCostHelper(
I.getOpcode(), Val,
CostKind, Index, &
I,
4790 unsigned Index)
const {
4802 : ST->getVectorInsertExtractBaseCost() + 1;
4811 if (Ty->getElementType()->isFloatingPointTy())
4814 unsigned VecInstCost =
4816 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
4823 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
4824 return std::nullopt;
4825 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
4826 return std::nullopt;
4828 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
4829 return std::nullopt;
4836 Cost += InstCost(PromotedTy);
4858 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4865 Op2Info, Args, CxtI);
4872 Ty,
CostKind, Op1Info, Op2Info,
true,
4875 [&](
Type *PromotedTy) {
4879 return *PromotedCost;
4882 if (Ty->getScalarType()->isFP128Ty())
4890 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
4910 ST->hasLimited64bitVectorMulBandwidth())
4913 if (Ty->getScalarSizeInBits() > 64) {
4918 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
4921 if (LT.second == MVT::v2i64) {
4925 return LT.first * Mul64CostFactor;
4946 if (LT.second == MVT::nxv2i64)
4947 return LT.first * Mul64CostFactor;
5006 auto VT = TLI->getValueType(
DL, Ty);
5007 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
5011 : (3 * AsrCost + AddCost);
5013 return MulCost + AsrCost + 2 * AddCost;
5015 }
else if (VT.isVector()) {
5025 if (Ty->isScalableTy() && ST->hasSVE())
5026 Cost += 2 * AsrCost;
5031 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
5035 }
else if (LT.second == MVT::v2i64) {
5036 return VT.getVectorNumElements() *
5043 if (Ty->isScalableTy() && ST->hasSVE())
5044 return MulCost + 2 * AddCost + 2 * AsrCost;
5045 return 2 * MulCost + AddCost + AsrCost + UsraCost;
5050 LT.second.isFixedLengthVector()) {
5060 return ExtractCost + InsertCost +
5068 auto VT = TLI->getValueType(
DL, Ty);
5084 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
5085 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
5086 LT.second == MVT::nxv16i8;
5087 bool Is128bit = LT.second.is128BitVector();
5099 (HasMULH ? 0 : ShrCost) +
5100 AddCost * 2 + ShrCost;
5101 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
5108 if (!VT.isVector() && VT.getSizeInBits() > 64)
5112 Opcode, Ty,
CostKind, Op1Info, Op2Info);
5114 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
5118 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
5128 if (
nullptr != Entry)
5136 FVTy && LT.second.isFixedLengthVector()) {
5137 unsigned NumElts = FVTy->getNumElements();
5138 unsigned RegElts = LT.second.getVectorNumElements();
5140 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
5144 if (LT.second.getScalarType() == MVT::i8)
5146 else if (LT.second.getScalarType() == MVT::i16)
5158 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
5159 return (4 + DivCost) * VTy->getNumElements();
5165 -1,
nullptr,
nullptr);
5192 LT.second.isFixedLengthVector())
5193 return 2 * LT.first + 1;
5202 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
5203 (Ty->isHalfTy() && ST->hasFullFP16())) &&
5212 if (!Ty->getScalarType()->isFP128Ty())
5219 if (!Ty->getScalarType()->isFP128Ty())
5220 return 2 * LT.first;
5227 if (!Ty->isVectorTy())
5243 int MaxMergeDistance = 64;
5247 return NumVectorInstToHideOverhead;
5257 unsigned Opcode1,
unsigned Opcode2)
const {
5260 if (!
Sched.hasInstrSchedModel())
5264 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
5266 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5272 "Cannot handle variant scheduling classes without an MI");
5288 const int AmortizationCost = 20;
5296 VecPred = CurrentPred;
5304 static const auto ValidMinMaxTys = {
5305 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5306 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5307 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5311 (ST->hasFullFP16() &&
5317 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5318 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5319 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5320 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5321 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5322 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5323 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5324 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5325 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5326 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5327 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5329 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5330 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5339 if (Opcode == Instruction::FCmp) {
5341 ValTy,
CostKind, Op1Info, Op2Info,
false,
5343 false, [&](
Type *PromotedTy) {
5355 return *PromotedCost;
5359 if (LT.second.getScalarType() != MVT::f64 &&
5360 LT.second.getScalarType() != MVT::f32 &&
5361 LT.second.getScalarType() != MVT::f16)
5366 unsigned Factor = 1;
5367 if (!CondTy->isVectorTy() &&
5381 AArch64::FCMEQv4f32))
5393 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5412 Op1Info, Op2Info,
I);
5418 if (ST->requiresStrictAlign()) {
5423 Options.AllowOverlappingLoads =
true;
5424 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5429 Options.LoadSizes = {8, 4, 2, 1};
5430 Options.AllowedTailExpansions = {3, 5, 6};
5435 return ST->hasSVE();
5441 switch (MICA.
getID()) {
5442 case Intrinsic::masked_scatter:
5443 case Intrinsic::masked_gather:
5445 case Intrinsic::masked_load:
5446 case Intrinsic::masked_store:
5447 case Intrinsic::masked_expandload:
5448 case Intrinsic::masked_compressstore:
5462 if (!LT.first.isValid())
5467 if (VT->getElementType()->isIntegerTy(1))
5478 if (MICA.
getID() == Intrinsic::masked_expandload) {
5486 if (MICA.
getID() == Intrinsic::masked_compressstore) {
5507 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5508 return MemOpCost * 2;
5517 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5518 "Should be called on only load or stores.");
5520 case Instruction::Load:
5523 return ST->getGatherOverhead();
5525 case Instruction::Store:
5528 return ST->getScatterOverhead();
5539 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5540 MICA.
getID() == Intrinsic::vp_gather)
5542 : Instruction::Store;
5552 if (!LT.first.isValid())
5556 if (!LT.second.isVector() ||
5558 VT->getElementType()->isIntegerTy(1))
5568 ElementCount LegalVF = LT.second.getVectorElementCount();
5571 {TTI::OK_AnyValue, TTI::OP_None},
I);
5587 EVT VT = TLI->getValueType(
DL, Ty,
true);
5589 if (VT == MVT::Other)
5594 if (!LT.first.isValid())
5604 (VTy->getElementType()->isIntegerTy(1) &&
5605 !VTy->getElementCount().isKnownMultipleOf(
5615 if (Opcode == Instruction::Store)
5619 if (ST->getFixedLoadLatency())
5620 return (LT.first - 1) + ST->getFixedLoadLatency();
5629 if (LT.second.isScalableVector() ||
5630 ST->useSVEForFixedLengthVectors(LT.second)) {
5631 Inst = AArch64::LDR_ZXI;
5632 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5633 switch (LT.second.getSizeInBits()) {
5635 Inst = AArch64::LDRBui;
5638 Inst = AArch64::LDRHui;
5641 Inst = AArch64::LDRSui;
5644 Inst = AArch64::LDRDui;
5647 Inst = AArch64::LDRQui;
5653 switch (LT.second.getSizeInBits()) {
5655 Inst = AArch64::LDRBBui;
5658 Inst = AArch64::LDRHHui;
5661 Inst = AArch64::LDRWui;
5664 Inst = AArch64::LDRXui;
5672 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5676 float NumLoads = (LT.first - 1).
getValue();
5677 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5678 Sched.computeInstrLatency(*ST, *SCD);
5681 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5682 LT.second.is128BitVector() && Alignment <
Align(16)) {
5688 const int AmortizationCost = 6;
5690 return LT.first * 2 * AmortizationCost;
5694 if (Ty->isPtrOrPtrVectorTy())
5699 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5701 if (VT == MVT::v4i8)
5708 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5723 while (!TypeWorklist.
empty()) {
5745 bool UseMaskForCond,
bool UseMaskForGaps)
const {
5746 assert(Factor >= 2 &&
"Invalid interleave factor");
5761 if (!VecTy->
isScalableTy() && (UseMaskForCond || UseMaskForGaps))
5764 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
5767 EC.divideCoefficientBy(Factor));
5773 if (EC.isKnownMultipleOf(Factor) &&
5774 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
5775 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
5780 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
5786 if (UseMaskForCond) {
5787 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5788 : Intrinsic::masked_store;
5808 if (Opcode == Instruction::Store && Factor == 4 &&
5809 SubVecCost.second.getScalarSizeInBits() ==
5810 (4 * ResultCost.second.getScalarSizeInBits()))
5811 LegalizationCost *= 4;
5813 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
5819 UseMaskForCond, UseMaskForGaps);
5826 for (
auto *
I : Tys) {
5827 if (!
I->isVectorTy())
5838 Align Alignment)
const {
5845 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
5846 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
5851 bool HasUnorderedReductions)
const {
5854 return ST->getMaxInterleaveFactor();
5864 enum { MaxStridedLoads = 7 };
5866 int StridedLoads = 0;
5869 for (
const auto BB : L->blocks()) {
5870 for (
auto &
I : *BB) {
5876 if (L->isLoopInvariant(PtrValue))
5881 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
5890 if (StridedLoads > MaxStridedLoads / 2)
5891 return StridedLoads;
5894 return StridedLoads;
5897 int StridedLoads = countStridedLoads(L, SE);
5899 <<
" strided loads\n");
5915 unsigned *FinalSize) {
5919 for (
auto *BB : L->getBlocks()) {
5920 for (
auto &
I : *BB) {
5926 if (!Cost.isValid())
5930 if (LoopCost > Budget)
5952 if (MaxTC > 0 && MaxTC <= 32)
5963 if (Blocks.
size() != 2)
5985 if (!L->isInnermost() || L->getNumBlocks() > 8)
5989 if (!L->getExitBlock())
5995 bool HasParellelizableReductions =
5996 L->getNumBlocks() == 1 &&
5997 any_of(L->getHeader()->phis(),
5999 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
6002 if (HasParellelizableReductions &&
6024 if (HasParellelizableReductions) {
6035 if (Header == Latch) {
6038 unsigned Width = 10;
6044 unsigned MaxInstsPerLine = 16;
6046 unsigned BestUC = 1;
6047 unsigned SizeWithBestUC = BestUC *
Size;
6049 unsigned SizeWithUC = UC *
Size;
6050 if (SizeWithUC > 48)
6052 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
6053 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
6055 SizeWithBestUC = BestUC *
Size;
6065 for (
auto *BB : L->blocks()) {
6066 for (
auto &
I : *BB) {
6076 for (
auto *U :
I.users())
6078 LoadedValuesPlus.
insert(U);
6085 return LoadedValuesPlus.
contains(
SI->getOperand(0));
6111 auto *I = dyn_cast<Instruction>(V);
6112 return I && DependsOnLoopLoad(I, Depth + 1);
6119 DependsOnLoopLoad(
I, 0)) {
6151 if (L->getLoopDepth() > 1)
6162 for (
auto *BB : L->getBlocks()) {
6163 for (
auto &
I : *BB) {
6167 if (IsVectorized &&
I.getType()->isVectorTy())
6184 if (ST->isAppleMLike())
6186 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
6208 !ST->getSchedModel().isOutOfOrder()) {
6231 bool CanCreate)
const {
6235 case Intrinsic::aarch64_neon_st1x2:
6236 case Intrinsic::aarch64_neon_st1x3:
6237 case Intrinsic::aarch64_neon_st1x4:
6238 case Intrinsic::aarch64_neon_st2:
6239 case Intrinsic::aarch64_neon_st3:
6240 case Intrinsic::aarch64_neon_st4: {
6243 if (!CanCreate || !ST)
6245 unsigned NumElts = Inst->
arg_size() - 1;
6246 if (ST->getNumElements() != NumElts)
6248 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6254 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6256 Res = Builder.CreateInsertValue(Res, L, i);
6260 case Intrinsic::aarch64_neon_ld1x2:
6261 case Intrinsic::aarch64_neon_ld1x3:
6262 case Intrinsic::aarch64_neon_ld1x4:
6263 case Intrinsic::aarch64_neon_ld2:
6264 case Intrinsic::aarch64_neon_ld3:
6265 case Intrinsic::aarch64_neon_ld4:
6266 if (Inst->
getType() == ExpectedType)
6277 case Intrinsic::aarch64_neon_ld1x2:
6278 case Intrinsic::aarch64_neon_ld1x3:
6279 case Intrinsic::aarch64_neon_ld1x4:
6280 case Intrinsic::aarch64_neon_ld2:
6281 case Intrinsic::aarch64_neon_ld3:
6282 case Intrinsic::aarch64_neon_ld4:
6283 Info.ReadMem =
true;
6284 Info.WriteMem =
false;
6287 case Intrinsic::aarch64_neon_st1x2:
6288 case Intrinsic::aarch64_neon_st1x3:
6289 case Intrinsic::aarch64_neon_st1x4:
6290 case Intrinsic::aarch64_neon_st2:
6291 case Intrinsic::aarch64_neon_st3:
6292 case Intrinsic::aarch64_neon_st4:
6293 Info.ReadMem =
false;
6294 Info.WriteMem =
true;
6303 case Intrinsic::aarch64_neon_ld1x2:
6304 case Intrinsic::aarch64_neon_st1x2:
6305 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6307 case Intrinsic::aarch64_neon_ld1x3:
6308 case Intrinsic::aarch64_neon_st1x3:
6309 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6311 case Intrinsic::aarch64_neon_ld1x4:
6312 case Intrinsic::aarch64_neon_st1x4:
6313 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6315 case Intrinsic::aarch64_neon_ld2:
6316 case Intrinsic::aarch64_neon_st2:
6317 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6319 case Intrinsic::aarch64_neon_ld3:
6320 case Intrinsic::aarch64_neon_st3:
6321 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6323 case Intrinsic::aarch64_neon_ld4:
6324 case Intrinsic::aarch64_neon_st4:
6325 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6337 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6338 bool Considerable =
false;
6339 AllowPromotionWithoutCommonHeader =
false;
6342 Type *ConsideredSExtType =
6344 if (
I.getType() != ConsideredSExtType)
6348 for (
const User *U :
I.users()) {
6350 Considerable =
true;
6354 if (GEPInst->getNumOperands() > 2) {
6355 AllowPromotionWithoutCommonHeader =
true;
6360 return Considerable;
6411 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6421 return LegalizationCost + 2;
6431 LegalizationCost *= LT.first - 1;
6434 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6443 return LegalizationCost + 2;
6451 std::optional<FastMathFlags> FMF,
6467 return BaseCost + FixedVTy->getNumElements();
6481 MVT MTy = LT.second;
6486 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6534 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6535 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6547 return (LT.first - 1) +
Log2_32(NElts);
6552 return (LT.first - 1) + Entry->Cost;
6564 if (LT.first != 1) {
6570 ExtraCost *= LT.first - 1;
6573 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6574 return Cost + ExtraCost;
6582 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6584 EVT VecVT = TLI->getValueType(
DL, VecTy);
6585 EVT ResVT = TLI->getValueType(
DL, ResTy);
6595 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6597 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6599 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6601 return (LT.first - 1) * 2 + 2;
6612 EVT VecVT = TLI->getValueType(
DL, VecTy);
6613 EVT ResVT = TLI->getValueType(
DL, ResTy);
6616 RedOpcode == Instruction::Add) {
6622 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6624 return LT.first + 2;
6659 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6660 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6674 if (LT.second.getScalarType() == MVT::i1) {
6683 assert(Entry &&
"Illegal Type for Splice");
6684 LegalizationCost += Entry->Cost;
6685 return LegalizationCost * LT.first;
6689 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
6698 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
6699 Opcode != Instruction::FAdd && Opcode != Instruction::FSub) ||
6706 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
6707 if (!FMF->allowReassoc() || !FMF->allowContract())
6711 "FastMathFlags only apply to floating-point partial reductions");
6715 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
6716 "Unexpected values for OpBExtend or InputTypeB");
6720 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
6721 InputTypeA != InputTypeB))
6724 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
6727 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
6740 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
6749 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
6755 std::pair<InstructionCost, MVT> AccumLT =
6757 std::pair<InstructionCost, MVT> InputLT =
6761 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
6762 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
6763 (AccumLT.second.isFixedLengthVector() &&
6764 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
6768 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
6776 if (AccumLT.second.getScalarType() == MVT::i32 &&
6777 InputLT.second.getScalarType() == MVT::i8) {
6779 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
6780 return Cost + INegCost;
6782 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
6783 return Cost + INegCost;
6788 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
6789 return Cost * 3 + INegCost;
6792 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
6794 if (AccumLT.second.getScalarType() == MVT::i64 &&
6795 InputLT.second.getScalarType() == MVT::i16)
6796 return Cost + INegCost;
6799 if (AccumLT.second.getScalarType() == MVT::i32 &&
6800 InputLT.second.getScalarType() == MVT::i16 &&
6801 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
6804 if (AccumLT.second.getScalarType() == MVT::i64 &&
6805 InputLT.second.getScalarType() == MVT::i8)
6811 return Cost + INegCost;
6814 if (AccumLT.second.getScalarType() == MVT::i16 &&
6815 InputLT.second.getScalarType() == MVT::i8 &&
6816 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
6822 if (Opcode == Instruction::FAdd && !IsSub &&
6823 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
6824 AccumLT.second.getScalarType() == MVT::f32 &&
6825 InputLT.second.getScalarType() == MVT::f16)
6829 if (Ratio == 2 && !IsUSDot) {
6830 MVT InVT = InputLT.second.getScalarType();
6833 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
6838 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
6842 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
6843 InVT == MVT::bf16 && IsSub)
6853 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
6854 return Cost * 2 + FNegCost;
6858 AccumType, VF, OpAExtend, OpBExtend,
6870 "Expected the Mask to match the return size if given");
6872 "Expected the same scalar types");
6878 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
6879 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
6880 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
6888 return std::max<InstructionCost>(1, LT.first / 4);
6896 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
6898 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
6901 unsigned TpNumElts = Mask.size();
6902 unsigned LTNumElts = LT.second.getVectorNumElements();
6903 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
6905 LT.second.getVectorElementCount());
6907 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
6909 for (
unsigned N = 0;
N < NumVecs;
N++) {
6913 unsigned Source1 = -1U, Source2 = -1U;
6914 unsigned NumSources = 0;
6915 for (
unsigned E = 0; E < LTNumElts; E++) {
6916 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
6925 unsigned Source = MaskElt / LTNumElts;
6926 if (NumSources == 0) {
6929 }
else if (NumSources == 1 && Source != Source1) {
6932 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
6938 if (Source == Source1)
6940 else if (Source == Source2)
6941 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
6950 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
6961 NTp, NTp, NMask,
CostKind, 0,
nullptr, Args,
6964 Result.first->second = NCost;
6978 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
6979 if (LT.second.getFixedSizeInBits() >= 128 &&
6981 LT.second.getVectorNumElements() / 2) {
6984 if (Index == (
int)LT.second.getVectorNumElements() / 2)
6998 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
7001 return M.value() < 0 || M.value() == (int)M.index();
7007 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
7008 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
7017 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
7018 ST->isSVEorStreamingSVEAvailable() &&
7023 if (ST->isSVEorStreamingSVEAvailable() &&
7037 if (IsLoad && LT.second.isVector() &&
7039 LT.second.getVectorElementCount()))
7045 if (Mask.size() == 4 &&
7047 (SrcTy->getScalarSizeInBits() == 16 ||
7048 SrcTy->getScalarSizeInBits() == 32) &&
7049 all_of(Mask, [](
int E) {
return E < 8; }))
7055 if (LT.second.isFixedLengthVector() &&
7056 LT.second.getVectorNumElements() == Mask.size() &&
7062 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7063 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7064 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
7065 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7066 LT.second.getVectorNumElements(), 16) ||
7067 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7068 LT.second.getVectorNumElements(), 32) ||
7069 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7070 LT.second.getVectorNumElements(), 64) ||
7073 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
7202 return LT.first * Entry->Cost;
7211 LT.second.getSizeInBits() <= 128 && SubTp) {
7213 if (SubLT.second.isVector()) {
7214 int NumElts = LT.second.getVectorNumElements();
7215 int NumSubElts = SubLT.second.getVectorNumElements();
7216 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
7222 if (IsExtractSubvector)
7239 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
7252 return ST->useFixedOverScalableIfEqualCost();
7256 return ST->getEpilogueVectorizationMinVF();
7291 unsigned NumInsns = 0;
7293 NumInsns += BB->size();
7303 int64_t Scale,
unsigned AddrSpace)
const {
7331 if (
I->getOpcode() == Instruction::Or &&
7335 if (
I->getOpcode() == Instruction::Add ||
7336 I->getOpcode() == Instruction::Sub)
7361 return all_equal(Shuf->getShuffleMask());
7368 bool AllowSplat =
false) {
7373 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7374 auto *FullTy = FullV->
getType();
7375 auto *HalfTy = HalfV->getType();
7377 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7380 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7383 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7387 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7401 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7402 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7416 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7417 (M2Start != 0 && M2Start != (NumElements / 2)))
7419 if (S1Op1 && S2Op1 && M1Start != M2Start)
7429 return Ext->getType()->getScalarSizeInBits() ==
7430 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7444 Value *VectorOperand =
nullptr;
7461 if (!
GEP ||
GEP->getNumOperands() != 2)
7465 Value *Offsets =
GEP->getOperand(1);
7468 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7474 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7475 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7476 Ops.push_back(&
GEP->getOperandUse(1));
7512 switch (
II->getIntrinsicID()) {
7513 case Intrinsic::aarch64_neon_smull:
7514 case Intrinsic::aarch64_neon_umull:
7517 Ops.push_back(&
II->getOperandUse(0));
7518 Ops.push_back(&
II->getOperandUse(1));
7523 case Intrinsic::fma:
7524 case Intrinsic::fmuladd:
7531 Ops.push_back(&
II->getOperandUse(0));
7533 Ops.push_back(&
II->getOperandUse(1));
7536 case Intrinsic::aarch64_neon_sqdmull:
7537 case Intrinsic::aarch64_neon_sqdmulh:
7538 case Intrinsic::aarch64_neon_sqrdmulh:
7541 Ops.push_back(&
II->getOperandUse(0));
7543 Ops.push_back(&
II->getOperandUse(1));
7544 return !
Ops.empty();
7545 case Intrinsic::aarch64_neon_fmlal:
7546 case Intrinsic::aarch64_neon_fmlal2:
7547 case Intrinsic::aarch64_neon_fmlsl:
7548 case Intrinsic::aarch64_neon_fmlsl2:
7551 Ops.push_back(&
II->getOperandUse(1));
7553 Ops.push_back(&
II->getOperandUse(2));
7554 return !
Ops.empty();
7555 case Intrinsic::aarch64_sve_ptest_first:
7556 case Intrinsic::aarch64_sve_ptest_last:
7558 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7559 Ops.push_back(&
II->getOperandUse(0));
7560 return !
Ops.empty();
7561 case Intrinsic::aarch64_sme_write_horiz:
7562 case Intrinsic::aarch64_sme_write_vert:
7563 case Intrinsic::aarch64_sme_writeq_horiz:
7564 case Intrinsic::aarch64_sme_writeq_vert: {
7566 if (!Idx || Idx->getOpcode() != Instruction::Add)
7568 Ops.push_back(&
II->getOperandUse(1));
7571 case Intrinsic::aarch64_sme_read_horiz:
7572 case Intrinsic::aarch64_sme_read_vert:
7573 case Intrinsic::aarch64_sme_readq_horiz:
7574 case Intrinsic::aarch64_sme_readq_vert:
7575 case Intrinsic::aarch64_sme_ld1b_vert:
7576 case Intrinsic::aarch64_sme_ld1h_vert:
7577 case Intrinsic::aarch64_sme_ld1w_vert:
7578 case Intrinsic::aarch64_sme_ld1d_vert:
7579 case Intrinsic::aarch64_sme_ld1q_vert:
7580 case Intrinsic::aarch64_sme_st1b_vert:
7581 case Intrinsic::aarch64_sme_st1h_vert:
7582 case Intrinsic::aarch64_sme_st1w_vert:
7583 case Intrinsic::aarch64_sme_st1d_vert:
7584 case Intrinsic::aarch64_sme_st1q_vert:
7585 case Intrinsic::aarch64_sme_ld1b_horiz:
7586 case Intrinsic::aarch64_sme_ld1h_horiz:
7587 case Intrinsic::aarch64_sme_ld1w_horiz:
7588 case Intrinsic::aarch64_sme_ld1d_horiz:
7589 case Intrinsic::aarch64_sme_ld1q_horiz:
7590 case Intrinsic::aarch64_sme_st1b_horiz:
7591 case Intrinsic::aarch64_sme_st1h_horiz:
7592 case Intrinsic::aarch64_sme_st1w_horiz:
7593 case Intrinsic::aarch64_sme_st1d_horiz:
7594 case Intrinsic::aarch64_sme_st1q_horiz: {
7596 if (!Idx || Idx->getOpcode() != Instruction::Add)
7598 Ops.push_back(&
II->getOperandUse(3));
7601 case Intrinsic::aarch64_neon_pmull:
7604 Ops.push_back(&
II->getOperandUse(0));
7605 Ops.push_back(&
II->getOperandUse(1));
7607 case Intrinsic::aarch64_neon_pmull64:
7609 II->getArgOperand(1)))
7611 Ops.push_back(&
II->getArgOperandUse(0));
7612 Ops.push_back(&
II->getArgOperandUse(1));
7614 case Intrinsic::masked_gather:
7617 Ops.push_back(&
II->getArgOperandUse(0));
7619 case Intrinsic::masked_scatter:
7622 Ops.push_back(&
II->getArgOperandUse(1));
7629 auto ShouldSinkCondition = [](
Value *
Cond,
7634 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
7638 Ops.push_back(&
II->getOperandUse(0));
7642 switch (
I->getOpcode()) {
7643 case Instruction::GetElementPtr:
7644 case Instruction::Add:
7645 case Instruction::Sub:
7647 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
7649 Ops.push_back(&
I->getOperandUse(
Op));
7654 case Instruction::Select: {
7655 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
7658 Ops.push_back(&
I->getOperandUse(0));
7661 case Instruction::UncondBr:
7663 case Instruction::CondBr: {
7667 Ops.push_back(&
I->getOperandUse(0));
7670 case Instruction::FMul:
7675 Ops.push_back(&
I->getOperandUse(0));
7677 Ops.push_back(&
I->getOperandUse(1));
7687 case Instruction::Xor:
7690 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
7692 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
7697 case Instruction::And:
7698 case Instruction::Or:
7701 if (
I->getOpcode() == Instruction::Or &&
7706 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
7709 for (
auto &
Op :
I->operands()) {
7721 Ops.push_back(&Not);
7722 Ops.push_back(&InsertElt);
7732 if (!
I->getType()->isVectorTy())
7733 return !
Ops.empty();
7735 switch (
I->getOpcode()) {
7736 case Instruction::Sub:
7737 case Instruction::Add: {
7746 Ops.push_back(&Ext1->getOperandUse(0));
7747 Ops.push_back(&Ext2->getOperandUse(0));
7750 Ops.push_back(&
I->getOperandUse(0));
7751 Ops.push_back(&
I->getOperandUse(1));
7755 case Instruction::Or: {
7758 if (ST->hasNEON()) {
7772 if (
I->getParent() != MainAnd->
getParent() ||
7777 if (
I->getParent() != IA->getParent() ||
7778 I->getParent() != IB->getParent())
7783 Ops.push_back(&
I->getOperandUse(0));
7784 Ops.push_back(&
I->getOperandUse(1));
7793 case Instruction::Mul: {
7794 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
7797 if (Ty->isScalableTy())
7801 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
7804 int NumZExts = 0, NumSExts = 0;
7805 for (
auto &
Op :
I->operands()) {
7812 auto *ExtOp = Ext->getOperand(0);
7813 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
7814 Ops.push_back(&Ext->getOperandUse(0));
7822 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
7823 I->getType()->getScalarSizeInBits())
7860 if (!ElementConstant || !ElementConstant->
isZero())
7863 unsigned Opcode = OperandInstr->
getOpcode();
7864 if (Opcode == Instruction::SExt)
7866 else if (Opcode == Instruction::ZExt)
7871 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
7881 Ops.push_back(&Insert->getOperandUse(1));
7887 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
7891 if (!ShouldSinkSplatForIndexedVariant(
I))
7896 Ops.push_back(&
I->getOperandUse(0));
7898 Ops.push_back(&
I->getOperandUse(1));
7900 return !
Ops.empty();
7902 case Instruction::FMul: {
7904 if (
I->getType()->isScalableTy())
7905 return !
Ops.empty();
7909 return !
Ops.empty();
7913 Ops.push_back(&
I->getOperandUse(0));
7915 Ops.push_back(&
I->getOperandUse(1));
7916 return !
Ops.empty();
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
bool preferFixedOverScalableIfEqualCost() const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
static bool isCommutative(Predicate Pred)
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static bool isCommutative(Predicate P)
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
bool isSMEABIRoutine() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
LLVM_ABI bool isScalableTy(SmallPtrSetImpl< const Type * > &Visited) const
Return true if this is a type whose size is a known multiple of vscale.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const DenseMap< Value *, const SCEV * > &StridesMap=DenseMap< Value *, const SCEV * >(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
CmpInst::Predicate getCmpPredicate() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
bool hasCmpPredicate() const
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
SVEIntrinsicInfo & setCmpPredicate(CmpInst::Predicate Pred)
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...