24#include "llvm/IR/IntrinsicsAArch64.h"
36#define DEBUG_TYPE "aarch64tti"
42 "sve-prefer-fixed-over-scalable-if-equal",
cl::Hidden);
60 "Penalty of calling a function that requires a change to PSTATE.SM"));
64 cl::desc(
"Penalty of inlining a call that requires a change to PSTATE.SM"));
75 cl::desc(
"The cost of a histcnt instruction"));
79 cl::desc(
"The number of instructions to search for a redundant dmb"));
83 cl::desc(
"Threshold for forced unrolling of small loops in AArch64"));
86class TailFoldingOption {
101 bool NeedsDefault =
true;
105 void setNeedsDefault(
bool V) { NeedsDefault =
V; }
120 assert((InitialBits == TailFoldingOpts::Disabled || !NeedsDefault) &&
121 "Initial bits should only include one of "
122 "(disabled|all|simple|default)");
123 Bits = NeedsDefault ? DefaultBits : InitialBits;
125 Bits &= ~DisableBits;
131 errs() <<
"invalid argument '" << Opt
132 <<
"' to -sve-tail-folding=; the option should be of the form\n"
133 " (disabled|all|default|simple)[+(reductions|recurrences"
134 "|reverse|noreductions|norecurrences|noreverse)]\n";
140 void operator=(
const std::string &Val) {
149 setNeedsDefault(
false);
152 StringRef(Val).split(TailFoldTypes,
'+', -1,
false);
154 unsigned StartIdx = 1;
155 if (TailFoldTypes[0] ==
"disabled")
156 setInitialBits(TailFoldingOpts::Disabled);
157 else if (TailFoldTypes[0] ==
"all")
158 setInitialBits(TailFoldingOpts::All);
159 else if (TailFoldTypes[0] ==
"default")
160 setNeedsDefault(
true);
161 else if (TailFoldTypes[0] ==
"simple")
162 setInitialBits(TailFoldingOpts::Simple);
165 setInitialBits(TailFoldingOpts::Disabled);
168 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
169 if (TailFoldTypes[
I] ==
"reductions")
170 setEnableBit(TailFoldingOpts::Reductions);
171 else if (TailFoldTypes[
I] ==
"recurrences")
172 setEnableBit(TailFoldingOpts::Recurrences);
173 else if (TailFoldTypes[
I] ==
"reverse")
174 setEnableBit(TailFoldingOpts::Reverse);
175 else if (TailFoldTypes[
I] ==
"noreductions")
176 setDisableBit(TailFoldingOpts::Reductions);
177 else if (TailFoldTypes[
I] ==
"norecurrences")
178 setDisableBit(TailFoldingOpts::Recurrences);
179 else if (TailFoldTypes[
I] ==
"noreverse")
180 setDisableBit(TailFoldingOpts::Reverse);
197 "Control the use of vectorisation using tail-folding for SVE where the"
198 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
199 "\ndisabled (Initial) No loop types will vectorize using "
201 "\ndefault (Initial) Uses the default tail-folding settings for "
203 "\nall (Initial) All legal loop types will vectorize using "
205 "\nsimple (Initial) Use tail-folding for simple loops (not "
206 "reductions or recurrences)"
207 "\nreductions Use tail-folding for loops containing reductions"
208 "\nnoreductions Inverse of above"
209 "\nrecurrences Use tail-folding for loops containing fixed order "
211 "\nnorecurrences Inverse of above"
212 "\nreverse Use tail-folding for loops requiring reversed "
214 "\nnoreverse Inverse of above"),
259 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
260 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
261 FeatureStr.
split(Features,
",");
277 return F.hasFnAttribute(
"fmv-features");
287 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
288 CallAttrs.callee().hasStreamingInterfaceOrBody())
293 if (
CallAttrs.callee().hasStreamingBody()) {
303 CallAttrs.requiresPreservingAllZAState()) {
326 auto FVTy = dyn_cast<FixedVectorType>(Ty);
328 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
337 unsigned DefaultCallPenalty)
const {
362 if (
F ==
Call.getCaller())
368 return DefaultCallPenalty;
379 ST->isSVEorStreamingSVEAvailable() &&
380 !ST->disableMaximizeScalableBandwidth();
404 assert(Ty->isIntegerTy());
406 unsigned BitSize = Ty->getPrimitiveSizeInBits();
413 ImmVal = Imm.sext((BitSize + 63) & ~0x3fU);
418 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
424 return std::max<InstructionCost>(1,
Cost);
431 assert(Ty->isIntegerTy());
433 unsigned BitSize = Ty->getPrimitiveSizeInBits();
439 unsigned ImmIdx = ~0U;
443 case Instruction::GetElementPtr:
448 case Instruction::Store:
451 case Instruction::Add:
452 case Instruction::Sub:
453 case Instruction::Mul:
454 case Instruction::UDiv:
455 case Instruction::SDiv:
456 case Instruction::URem:
457 case Instruction::SRem:
458 case Instruction::And:
459 case Instruction::Or:
460 case Instruction::Xor:
461 case Instruction::ICmp:
465 case Instruction::Shl:
466 case Instruction::LShr:
467 case Instruction::AShr:
471 case Instruction::Trunc:
472 case Instruction::ZExt:
473 case Instruction::SExt:
474 case Instruction::IntToPtr:
475 case Instruction::PtrToInt:
476 case Instruction::BitCast:
477 case Instruction::PHI:
478 case Instruction::Call:
479 case Instruction::Select:
480 case Instruction::Ret:
481 case Instruction::Load:
486 int NumConstants = (BitSize + 63) / 64;
499 assert(Ty->isIntegerTy());
501 unsigned BitSize = Ty->getPrimitiveSizeInBits();
510 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
516 case Intrinsic::sadd_with_overflow:
517 case Intrinsic::uadd_with_overflow:
518 case Intrinsic::ssub_with_overflow:
519 case Intrinsic::usub_with_overflow:
520 case Intrinsic::smul_with_overflow:
521 case Intrinsic::umul_with_overflow:
523 int NumConstants = (BitSize + 63) / 64;
530 case Intrinsic::experimental_stackmap:
531 if ((Idx < 2) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
534 case Intrinsic::experimental_patchpoint_void:
535 case Intrinsic::experimental_patchpoint:
536 if ((Idx < 4) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
539 case Intrinsic::experimental_gc_statepoint:
540 if ((Idx < 5) || (Imm.getBitWidth() <= 64 &&
isInt<64>(Imm.getSExtValue())))
550 if (TyWidth == 32 || TyWidth == 64)
559 return ST->getMispredictionPenalty();
580 unsigned TotalHistCnts = 1;
590 unsigned EC = VTy->getElementCount().getKnownMinValue();
595 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
597 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
601 TotalHistCnts = EC / NaturalVectorWidth;
621 switch (ICA.
getID()) {
622 case Intrinsic::experimental_vector_histogram_add: {
629 case Intrinsic::clmul: {
634 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
638 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
643 -1,
nullptr,
nullptr) *
646 -1,
nullptr,
nullptr);
650 if (LT.second.SimpleTy == MVT::nxv2i64)
651 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
654 if (ST->hasSVE2() || ST->hasSME()) {
655 switch (LT.second.SimpleTy) {
670 if (LT.second.SimpleTy == MVT::nxv2i64)
674 switch (LT.second.SimpleTy) {
684 -1,
nullptr,
nullptr) *
687 -1,
nullptr,
nullptr));
696 return LT.first * 11;
698 return LT.first * 14;
705 case Intrinsic::umin:
706 case Intrinsic::umax:
707 case Intrinsic::smin:
708 case Intrinsic::smax: {
709 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
710 MVT::v8i16, MVT::v2i32, MVT::v4i32,
711 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
715 if (LT.second == MVT::v2i64)
721 case Intrinsic::scmp:
722 case Intrinsic::ucmp: {
724 {Intrinsic::scmp, MVT::i32, 3},
725 {Intrinsic::scmp, MVT::i64, 3},
726 {Intrinsic::scmp, MVT::v8i8, 3},
727 {Intrinsic::scmp, MVT::v16i8, 3},
728 {Intrinsic::scmp, MVT::v4i16, 3},
729 {Intrinsic::scmp, MVT::v8i16, 3},
730 {Intrinsic::scmp, MVT::v2i32, 3},
731 {Intrinsic::scmp, MVT::v4i32, 3},
732 {Intrinsic::scmp, MVT::v1i64, 3},
733 {Intrinsic::scmp, MVT::v2i64, 3},
739 return Entry->Cost * LT.first;
742 case Intrinsic::sadd_sat:
743 case Intrinsic::ssub_sat:
744 case Intrinsic::uadd_sat:
745 case Intrinsic::usub_sat: {
746 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
747 MVT::v8i16, MVT::v2i32, MVT::v4i32,
753 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
755 return LT.first * Instrs;
760 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
761 return LT.first * Instrs;
765 case Intrinsic::abs: {
766 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
767 MVT::v8i16, MVT::v2i32, MVT::v4i32,
768 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
769 MVT::nxv4i32, MVT::nxv2i64};
775 case Intrinsic::bswap: {
776 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
777 MVT::v4i32, MVT::v2i64};
780 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
785 case Intrinsic::fmuladd: {
790 (EltTy->
isHalfTy() && ST->hasFullFP16()))
794 case Intrinsic::stepvector: {
803 Cost += AddCost * (LT.first - 1);
807 case Intrinsic::vector_extract:
808 case Intrinsic::vector_insert: {
821 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
822 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
830 getTLI()->getTypeConversion(
C, SubVecVT);
832 getTLI()->getTypeConversion(
C, VecVT);
840 case Intrinsic::bitreverse: {
842 {Intrinsic::bitreverse, MVT::i32, 1},
843 {Intrinsic::bitreverse, MVT::i64, 1},
844 {Intrinsic::bitreverse, MVT::v8i8, 1},
845 {Intrinsic::bitreverse, MVT::v16i8, 1},
846 {Intrinsic::bitreverse, MVT::v4i16, 2},
847 {Intrinsic::bitreverse, MVT::v8i16, 2},
848 {Intrinsic::bitreverse, MVT::v2i32, 2},
849 {Intrinsic::bitreverse, MVT::v4i32, 2},
850 {Intrinsic::bitreverse, MVT::v1i64, 2},
851 {Intrinsic::bitreverse, MVT::v2i64, 2},
859 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
860 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
861 return LegalisationCost.first * Entry->Cost + 1;
863 return LegalisationCost.first * Entry->Cost;
867 case Intrinsic::ctpop: {
871 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
874 return LT.first + ExtraCost;
876 if (!ST->hasNEON()) {
906 RetTy->getScalarSizeInBits()
909 return LT.first * Entry->Cost + ExtraCost;
913 case Intrinsic::sadd_with_overflow:
914 case Intrinsic::uadd_with_overflow:
915 case Intrinsic::ssub_with_overflow:
916 case Intrinsic::usub_with_overflow:
917 case Intrinsic::smul_with_overflow:
918 case Intrinsic::umul_with_overflow: {
920 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
921 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
922 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
923 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
924 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
925 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
926 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
927 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
928 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
929 {Intrinsic::usub_with_overflow, MVT::i8, 3},
930 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
931 {Intrinsic::usub_with_overflow, MVT::i16, 3},
932 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
933 {Intrinsic::usub_with_overflow, MVT::i32, 1},
934 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
935 {Intrinsic::usub_with_overflow, MVT::i64, 1},
936 {Intrinsic::smul_with_overflow, MVT::i8, 5},
937 {Intrinsic::umul_with_overflow, MVT::i8, 4},
938 {Intrinsic::smul_with_overflow, MVT::i16, 5},
939 {Intrinsic::umul_with_overflow, MVT::i16, 4},
940 {Intrinsic::smul_with_overflow, MVT::i32, 2},
941 {Intrinsic::umul_with_overflow, MVT::i32, 2},
942 {Intrinsic::smul_with_overflow, MVT::i64, 3},
943 {Intrinsic::umul_with_overflow, MVT::i64, 3},
945 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
952 case Intrinsic::fptosi_sat:
953 case Intrinsic::fptoui_sat: {
956 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
958 EVT MTy = TLI->getValueType(
DL, RetTy);
961 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
962 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
963 LT.second == MVT::v2f64)) {
965 (LT.second == MVT::f64 && MTy == MVT::i32) ||
966 (LT.second == MVT::f32 && MTy == MVT::i64)))
975 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
982 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
983 (LT.second == MVT::f16 && MTy == MVT::i64) ||
984 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
998 if ((LT.second.getScalarType() == MVT::f32 ||
999 LT.second.getScalarType() == MVT::f64 ||
1000 LT.second.getScalarType() == MVT::f16) &&
1003 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1004 if (LT.second.isVector())
1005 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1009 LegalTy, {LegalTy, LegalTy});
1013 LegalTy, {LegalTy, LegalTy});
1015 return LT.first *
Cost +
1016 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1022 RetTy = RetTy->getScalarType();
1023 if (LT.second.isVector()) {
1041 return LT.first *
Cost;
1043 case Intrinsic::fshl:
1044 case Intrinsic::fshr: {
1053 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1054 (RetTy->getPrimitiveSizeInBits() == 32 ||
1055 RetTy->getPrimitiveSizeInBits() == 64)) {
1068 {Intrinsic::fshl, MVT::v4i32, 2},
1069 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1070 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1071 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1077 return LegalisationCost.first * Entry->Cost;
1081 if (!RetTy->isIntegerTy())
1086 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1087 RetTy->getScalarSizeInBits() < 64) ||
1088 (RetTy->getScalarSizeInBits() % 64 != 0);
1089 unsigned ExtraCost = HigherCost ? 1 : 0;
1090 if (RetTy->getScalarSizeInBits() == 32 ||
1091 RetTy->getScalarSizeInBits() == 64)
1094 else if (HigherCost)
1098 return TyL.first + ExtraCost;
1100 case Intrinsic::get_active_lane_mask: {
1102 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1104 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1107 if (RetTy->isScalableTy()) {
1108 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1118 if (ST->hasSVE2p1() || ST->hasSME2()) {
1130 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1133 return Cost + (SplitCost * (
Cost - 1));
1148 case Intrinsic::experimental_vector_match: {
1151 unsigned SearchSize = NeedleTy->getNumElements();
1152 if (!getTLI()->shouldExpandVectorMatch(SearchVT, SearchSize)) {
1165 case Intrinsic::cttz: {
1167 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
1168 return LT.first * 2;
1169 if (LT.second == MVT::v4i16 || LT.second == MVT::v8i16 ||
1170 LT.second == MVT::v2i32 || LT.second == MVT::v4i32)
1171 return LT.first * 3;
1174 case Intrinsic::experimental_cttz_elts: {
1176 if (!getTLI()->shouldExpandCttzElements(ArgVT)) {
1184 case Intrinsic::loop_dependence_raw_mask:
1185 case Intrinsic::loop_dependence_war_mask: {
1187 if (ST->hasSVE2() || ST->hasSME()) {
1188 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1189 unsigned EltSizeInBytes =
1199 case Intrinsic::experimental_vector_extract_last_active:
1200 if (ST->isSVEorStreamingSVEAvailable()) {
1206 case Intrinsic::pow: {
1209 EVT VT = getTLI()->getValueType(
DL, RetTy);
1211 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1226 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1227 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1237 return (Sqrt * 2) +
FMul;
1248 case Intrinsic::sqrt:
1249 case Intrinsic::fabs:
1250 case Intrinsic::ceil:
1251 case Intrinsic::floor:
1252 case Intrinsic::nearbyint:
1253 case Intrinsic::round:
1254 case Intrinsic::rint:
1255 case Intrinsic::roundeven:
1256 case Intrinsic::trunc:
1257 case Intrinsic::minnum:
1258 case Intrinsic::maxnum:
1259 case Intrinsic::minimum:
1260 case Intrinsic::maximum: {
1278 auto RequiredType =
II.getType();
1281 assert(PN &&
"Expected Phi Node!");
1284 if (!PN->hasOneUse())
1285 return std::nullopt;
1287 for (
Value *IncValPhi : PN->incoming_values()) {
1290 Reinterpret->getIntrinsicID() !=
1291 Intrinsic::aarch64_sve_convert_to_svbool ||
1292 RequiredType != Reinterpret->getArgOperand(0)->getType())
1293 return std::nullopt;
1301 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1303 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1376 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1381 return GoverningPredicateIdx;
1386 GoverningPredicateIdx = Index;
1408 return UndefIntrinsic;
1413 UndefIntrinsic = IID;
1439 return ResultLanes == InactiveLanesTakenFromOperand;
1444 return OperandIdxForInactiveLanes;
1448 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1449 ResultLanes = InactiveLanesTakenFromOperand;
1450 OperandIdxForInactiveLanes = Index;
1455 return ResultLanes == InactiveLanesAreNotDefined;
1459 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1460 ResultLanes = InactiveLanesAreNotDefined;
1465 return ResultLanes == InactiveLanesAreUnused;
1469 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1470 ResultLanes = InactiveLanesAreUnused;
1480 ResultIsZeroInitialized =
true;
1491 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1496 return OperandIdxWithNoActiveLanes;
1501 OperandIdxWithNoActiveLanes = Index;
1506 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1509 unsigned IROpcode = 0;
1511 enum PredicationStyle {
1513 InactiveLanesTakenFromOperand,
1514 InactiveLanesAreNotDefined,
1515 InactiveLanesAreUnused
1518 bool ResultIsZeroInitialized =
false;
1519 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1520 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1528 return !isa<ScalableVectorType>(V->getType());
1536 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1537 case Intrinsic::aarch64_sve_fcvt_f16f32:
1538 case Intrinsic::aarch64_sve_fcvt_f16f64:
1539 case Intrinsic::aarch64_sve_fcvt_f32f16:
1540 case Intrinsic::aarch64_sve_fcvt_f32f64:
1541 case Intrinsic::aarch64_sve_fcvt_f64f16:
1542 case Intrinsic::aarch64_sve_fcvt_f64f32:
1543 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1544 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1545 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1546 case Intrinsic::aarch64_sve_fcvtzs:
1547 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1548 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1549 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1550 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1551 case Intrinsic::aarch64_sve_fcvtzu:
1552 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1553 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1554 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1555 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1556 case Intrinsic::aarch64_sve_revb:
1557 case Intrinsic::aarch64_sve_revh:
1558 case Intrinsic::aarch64_sve_revw:
1559 case Intrinsic::aarch64_sve_revd:
1560 case Intrinsic::aarch64_sve_scvtf:
1561 case Intrinsic::aarch64_sve_scvtf_f16i32:
1562 case Intrinsic::aarch64_sve_scvtf_f16i64:
1563 case Intrinsic::aarch64_sve_scvtf_f32i64:
1564 case Intrinsic::aarch64_sve_scvtf_f64i32:
1565 case Intrinsic::aarch64_sve_ucvtf:
1566 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1567 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1568 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1569 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1572 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1573 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1574 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1575 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1578 case Intrinsic::aarch64_sve_fabd:
1580 case Intrinsic::aarch64_sve_fadd:
1583 case Intrinsic::aarch64_sve_fdiv:
1586 case Intrinsic::aarch64_sve_fmax:
1588 case Intrinsic::aarch64_sve_fmaxnm:
1590 case Intrinsic::aarch64_sve_fmin:
1592 case Intrinsic::aarch64_sve_fminnm:
1594 case Intrinsic::aarch64_sve_fmla:
1596 case Intrinsic::aarch64_sve_fmls:
1598 case Intrinsic::aarch64_sve_fmul:
1601 case Intrinsic::aarch64_sve_fmulx:
1603 case Intrinsic::aarch64_sve_fnmla:
1605 case Intrinsic::aarch64_sve_fnmls:
1607 case Intrinsic::aarch64_sve_fsub:
1610 case Intrinsic::aarch64_sve_add:
1613 case Intrinsic::aarch64_sve_mla:
1615 case Intrinsic::aarch64_sve_mls:
1617 case Intrinsic::aarch64_sve_mul:
1620 case Intrinsic::aarch64_sve_sabd:
1622 case Intrinsic::aarch64_sve_sdiv:
1625 case Intrinsic::aarch64_sve_smax:
1627 case Intrinsic::aarch64_sve_smin:
1629 case Intrinsic::aarch64_sve_smulh:
1631 case Intrinsic::aarch64_sve_sub:
1634 case Intrinsic::aarch64_sve_uabd:
1636 case Intrinsic::aarch64_sve_udiv:
1639 case Intrinsic::aarch64_sve_umax:
1641 case Intrinsic::aarch64_sve_umin:
1643 case Intrinsic::aarch64_sve_umulh:
1645 case Intrinsic::aarch64_sve_asr:
1648 case Intrinsic::aarch64_sve_lsl:
1651 case Intrinsic::aarch64_sve_lsr:
1654 case Intrinsic::aarch64_sve_and:
1657 case Intrinsic::aarch64_sve_bic:
1659 case Intrinsic::aarch64_sve_eor:
1662 case Intrinsic::aarch64_sve_orr:
1665 case Intrinsic::aarch64_sve_shsub:
1667 case Intrinsic::aarch64_sve_shsubr:
1669 case Intrinsic::aarch64_sve_sqrshl:
1671 case Intrinsic::aarch64_sve_sqshl:
1673 case Intrinsic::aarch64_sve_sqsub:
1675 case Intrinsic::aarch64_sve_srshl:
1677 case Intrinsic::aarch64_sve_uhsub:
1679 case Intrinsic::aarch64_sve_uhsubr:
1681 case Intrinsic::aarch64_sve_uqrshl:
1683 case Intrinsic::aarch64_sve_uqshl:
1685 case Intrinsic::aarch64_sve_uqsub:
1687 case Intrinsic::aarch64_sve_urshl:
1690 case Intrinsic::aarch64_sve_add_u:
1693 case Intrinsic::aarch64_sve_and_u:
1696 case Intrinsic::aarch64_sve_asr_u:
1699 case Intrinsic::aarch64_sve_eor_u:
1702 case Intrinsic::aarch64_sve_fadd_u:
1705 case Intrinsic::aarch64_sve_fdiv_u:
1708 case Intrinsic::aarch64_sve_fmul_u:
1711 case Intrinsic::aarch64_sve_fsub_u:
1714 case Intrinsic::aarch64_sve_lsl_u:
1717 case Intrinsic::aarch64_sve_lsr_u:
1720 case Intrinsic::aarch64_sve_mul_u:
1723 case Intrinsic::aarch64_sve_orr_u:
1726 case Intrinsic::aarch64_sve_sdiv_u:
1729 case Intrinsic::aarch64_sve_sub_u:
1732 case Intrinsic::aarch64_sve_udiv_u:
1736 case Intrinsic::aarch64_sve_addqv:
1737 case Intrinsic::aarch64_sve_bic_z:
1738 case Intrinsic::aarch64_sve_brka_z:
1739 case Intrinsic::aarch64_sve_brkb_z:
1740 case Intrinsic::aarch64_sve_brkn_z:
1741 case Intrinsic::aarch64_sve_brkpa_z:
1742 case Intrinsic::aarch64_sve_brkpb_z:
1743 case Intrinsic::aarch64_sve_cntp:
1744 case Intrinsic::aarch64_sve_compact:
1745 case Intrinsic::aarch64_sve_eorv:
1746 case Intrinsic::aarch64_sve_eorqv:
1747 case Intrinsic::aarch64_sve_nand_z:
1748 case Intrinsic::aarch64_sve_nor_z:
1749 case Intrinsic::aarch64_sve_orn_z:
1750 case Intrinsic::aarch64_sve_orv:
1751 case Intrinsic::aarch64_sve_orqv:
1752 case Intrinsic::aarch64_sve_pnext:
1753 case Intrinsic::aarch64_sve_rdffr_z:
1754 case Intrinsic::aarch64_sve_saddv:
1755 case Intrinsic::aarch64_sve_uaddv:
1756 case Intrinsic::aarch64_sve_umaxv:
1757 case Intrinsic::aarch64_sve_umaxqv:
1758 case Intrinsic::aarch64_sve_cmpeq:
1759 case Intrinsic::aarch64_sve_cmpeq_wide:
1760 case Intrinsic::aarch64_sve_cmpge:
1761 case Intrinsic::aarch64_sve_cmpge_wide:
1762 case Intrinsic::aarch64_sve_cmpgt:
1763 case Intrinsic::aarch64_sve_cmpgt_wide:
1764 case Intrinsic::aarch64_sve_cmphi:
1765 case Intrinsic::aarch64_sve_cmphi_wide:
1766 case Intrinsic::aarch64_sve_cmphs:
1767 case Intrinsic::aarch64_sve_cmphs_wide:
1768 case Intrinsic::aarch64_sve_cmple_wide:
1769 case Intrinsic::aarch64_sve_cmplo_wide:
1770 case Intrinsic::aarch64_sve_cmpls_wide:
1771 case Intrinsic::aarch64_sve_cmplt_wide:
1772 case Intrinsic::aarch64_sve_cmpne:
1773 case Intrinsic::aarch64_sve_cmpne_wide:
1774 case Intrinsic::aarch64_sve_facge:
1775 case Intrinsic::aarch64_sve_facgt:
1776 case Intrinsic::aarch64_sve_fcmpeq:
1777 case Intrinsic::aarch64_sve_fcmpge:
1778 case Intrinsic::aarch64_sve_fcmpgt:
1779 case Intrinsic::aarch64_sve_fcmpne:
1780 case Intrinsic::aarch64_sve_fcmpuo:
1781 case Intrinsic::aarch64_sve_ld1:
1782 case Intrinsic::aarch64_sve_ld1_gather:
1783 case Intrinsic::aarch64_sve_ld1_gather_index:
1784 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1785 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1786 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1787 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1788 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1789 case Intrinsic::aarch64_sve_ld1q_gather_index:
1790 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1791 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1792 case Intrinsic::aarch64_sve_ld1ro:
1793 case Intrinsic::aarch64_sve_ld1rq:
1794 case Intrinsic::aarch64_sve_ld1udq:
1795 case Intrinsic::aarch64_sve_ld1uwq:
1796 case Intrinsic::aarch64_sve_ld2_sret:
1797 case Intrinsic::aarch64_sve_ld2q_sret:
1798 case Intrinsic::aarch64_sve_ld3_sret:
1799 case Intrinsic::aarch64_sve_ld3q_sret:
1800 case Intrinsic::aarch64_sve_ld4_sret:
1801 case Intrinsic::aarch64_sve_ld4q_sret:
1802 case Intrinsic::aarch64_sve_ldff1:
1803 case Intrinsic::aarch64_sve_ldff1_gather:
1804 case Intrinsic::aarch64_sve_ldff1_gather_index:
1805 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1806 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1807 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1808 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1809 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1810 case Intrinsic::aarch64_sve_ldnf1:
1811 case Intrinsic::aarch64_sve_ldnt1:
1812 case Intrinsic::aarch64_sve_ldnt1_gather:
1813 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1814 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1815 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1818 case Intrinsic::aarch64_sve_and_z:
1821 case Intrinsic::aarch64_sve_orr_z:
1824 case Intrinsic::aarch64_sve_eor_z:
1828 case Intrinsic::aarch64_sve_prf:
1829 case Intrinsic::aarch64_sve_prfb_gather_index:
1830 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
1831 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
1832 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
1833 case Intrinsic::aarch64_sve_prfd_gather_index:
1834 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
1835 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
1836 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
1837 case Intrinsic::aarch64_sve_prfh_gather_index:
1838 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
1839 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
1840 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
1841 case Intrinsic::aarch64_sve_prfw_gather_index:
1842 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
1843 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
1844 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
1847 case Intrinsic::aarch64_sve_st1_scatter:
1848 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
1849 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
1850 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
1851 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
1852 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
1853 case Intrinsic::aarch64_sve_st1dq:
1854 case Intrinsic::aarch64_sve_st1q_scatter_index:
1855 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
1856 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
1857 case Intrinsic::aarch64_sve_st1wq:
1858 case Intrinsic::aarch64_sve_stnt1:
1859 case Intrinsic::aarch64_sve_stnt1_scatter:
1860 case Intrinsic::aarch64_sve_stnt1_scatter_index:
1861 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
1862 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
1864 case Intrinsic::aarch64_sve_st2:
1865 case Intrinsic::aarch64_sve_st2q:
1867 case Intrinsic::aarch64_sve_st3:
1868 case Intrinsic::aarch64_sve_st3q:
1870 case Intrinsic::aarch64_sve_st4:
1871 case Intrinsic::aarch64_sve_st4q:
1879 Value *UncastedPred;
1885 Pred = UncastedPred;
1891 if (OrigPredTy->getMinNumElements() <=
1893 ->getMinNumElements())
1894 Pred = UncastedPred;
1898 return C &&
C->isAllOnesValue();
1905 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
1906 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
1914static std::optional<Instruction *>
1921 Value *Op1 =
II.getOperand(1);
1922 Value *Op2 =
II.getOperand(2);
1948 return std::nullopt;
1959 if (SimpleII == Inactive)
1969static std::optional<Instruction *>
1973 return std::nullopt;
2002 II.setCalledFunction(NewDecl);
2012 return std::nullopt;
2024static std::optional<Instruction *>
2026 auto m_ConvertToSVBool = [](
auto P) {
2030 Intrinsic::aarch64_sve_convert_from_svbool;
2053 return std::nullopt;
2057 case Intrinsic::aarch64_sve_and_z:
2058 case Intrinsic::aarch64_sve_bic_z:
2059 case Intrinsic::aarch64_sve_eor_z:
2060 case Intrinsic::aarch64_sve_nand_z:
2061 case Intrinsic::aarch64_sve_nor_z:
2062 case Intrinsic::aarch64_sve_orn_z:
2063 case Intrinsic::aarch64_sve_orr_z:
2066 return std::nullopt;
2069 Value *BinOpPred = BinOp->getOperand(0);
2070 Value *BinOpOp1 = BinOp->getOperand(1);
2071 Value *BinOpOp2 = BinOp->getOperand(2);
2073 Value *NarrowBinOpPred;
2075 return std::nullopt;
2077 Value *NarrowBinOpOp1 =
2079 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2080 if (BinOpOp1 != BinOpOp2)
2084 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2088static std::optional<Instruction *>
2095 return BinOpCombine;
2100 return std::nullopt;
2103 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2112 if (CursorVTy->getElementCount().getKnownMinValue() <
2113 IVTy->getElementCount().getKnownMinValue())
2117 if (Cursor->getType() == IVTy)
2118 EarliestReplacement = Cursor;
2123 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2124 Intrinsic::aarch64_sve_convert_to_svbool ||
2125 IntrinsicCursor->getIntrinsicID() ==
2126 Intrinsic::aarch64_sve_convert_from_svbool))
2129 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2130 Cursor = IntrinsicCursor->getOperand(0);
2135 if (!EarliestReplacement)
2136 return std::nullopt;
2144 auto *OpPredicate =
II.getOperand(0);
2161 II.getArgOperand(2));
2167 return std::nullopt;
2171 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2180 II.getArgOperand(0));
2189 if (!
II.hasOneUse())
2190 return std::nullopt;
2193 return std::nullopt;
2196 switch (
II.getIntrinsicID()) {
2197 case Intrinsic::aarch64_sve_cmpne:
2198 IID = Intrinsic::aarch64_sve_cmpeq;
2200 case Intrinsic::aarch64_sve_cmpne_wide:
2201 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2203 case Intrinsic::aarch64_sve_cmpeq:
2204 IID = Intrinsic::aarch64_sve_cmpne;
2206 case Intrinsic::aarch64_sve_cmpeq_wide:
2207 IID = Intrinsic::aarch64_sve_cmpne_wide;
2210 return std::nullopt;
2215 IID,
II.getOperand(1)->getType(),
2216 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2228 return std::nullopt;
2230 for (
auto *U :
II.users()) {
2233 Type *Ty =
II.getOperand(1)->getType();
2238 Intrinsic::aarch64_sve_umin, Ty,
2239 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2245 return std::nullopt;
2259 return std::nullopt;
2264 if (!SplatValue || !SplatValue->isZero())
2265 return std::nullopt;
2270 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2271 return std::nullopt;
2275 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2276 return std::nullopt;
2279 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2280 return std::nullopt;
2285 return std::nullopt;
2288 return std::nullopt;
2292 return std::nullopt;
2296 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2297 return std::nullopt;
2299 unsigned NumElts = VecTy->getNumElements();
2300 unsigned PredicateBits = 0;
2303 for (
unsigned I = 0;
I < NumElts; ++
I) {
2306 return std::nullopt;
2308 PredicateBits |= 1 << (
I * (16 / NumElts));
2312 if (PredicateBits == 0) {
2314 PFalse->takeName(&
II);
2320 for (
unsigned I = 0;
I < 16; ++
I)
2321 if ((PredicateBits & (1 <<
I)) != 0)
2324 unsigned PredSize = Mask & -Mask;
2329 for (
unsigned I = 0;
I < 16;
I += PredSize)
2330 if ((PredicateBits & (1 <<
I)) == 0)
2331 return std::nullopt;
2333 auto *ConvertToSVBool =
2336 auto *ConvertFromSVBool =
2338 II.getType(), ConvertToSVBool);
2346 Value *Pg =
II.getArgOperand(0);
2347 Value *Vec =
II.getArgOperand(1);
2348 auto IntrinsicID =
II.getIntrinsicID();
2349 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2361 auto OpC = OldBinOp->getOpcode();
2367 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2373 if (IsAfter &&
C &&
C->isNullValue()) {
2377 Extract->insertBefore(
II.getIterator());
2378 Extract->takeName(&
II);
2384 return std::nullopt;
2386 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2387 return std::nullopt;
2389 const auto PTruePattern =
2395 return std::nullopt;
2397 unsigned Idx = MinNumElts - 1;
2407 if (Idx >= PgVTy->getMinNumElements())
2408 return std::nullopt;
2413 Extract->insertBefore(
II.getIterator());
2414 Extract->takeName(&
II);
2427 Value *Pg =
II.getArgOperand(0);
2429 Value *Vec =
II.getArgOperand(2);
2432 if (!Ty->isIntegerTy())
2433 return std::nullopt;
2438 return std::nullopt;
2455 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2470static std::optional<Instruction *>
2474 if (
Pattern == AArch64SVEPredPattern::all) {
2483 return MinNumElts && NumElts >= MinNumElts
2485 II, ConstantInt::get(
II.getType(), MinNumElts)))
2489static std::optional<Instruction *>
2492 if (!ST->isStreaming())
2493 return std::nullopt;
2505 Value *PgVal =
II.getArgOperand(0);
2506 Value *OpVal =
II.getArgOperand(1);
2510 if (PgVal == OpVal &&
2511 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2512 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2527 return std::nullopt;
2531 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2532 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2546 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2547 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2548 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2549 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2550 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2551 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2552 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2553 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2554 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2555 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2556 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2557 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2558 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2568 return std::nullopt;
2571template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2572static std::optional<Instruction *>
2574 bool MergeIntoAddendOp) {
2576 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2577 if (MergeIntoAddendOp) {
2578 AddendOp =
II.getOperand(1);
2579 Mul =
II.getOperand(2);
2581 AddendOp =
II.getOperand(2);
2582 Mul =
II.getOperand(1);
2587 return std::nullopt;
2589 if (!
Mul->hasOneUse())
2590 return std::nullopt;
2593 if (
II.getType()->isFPOrFPVectorTy()) {
2598 return std::nullopt;
2600 return std::nullopt;
2605 if (MergeIntoAddendOp)
2615static std::optional<Instruction *>
2617 Value *Pred =
II.getOperand(0);
2618 Value *PtrOp =
II.getOperand(1);
2619 Type *VecTy =
II.getType();
2634static std::optional<Instruction *>
2636 Value *VecOp =
II.getOperand(0);
2637 Value *Pred =
II.getOperand(1);
2638 Value *PtrOp =
II.getOperand(2);
2654 case Intrinsic::aarch64_sve_fmul_u:
2655 return Instruction::BinaryOps::FMul;
2656 case Intrinsic::aarch64_sve_fadd_u:
2657 return Instruction::BinaryOps::FAdd;
2658 case Intrinsic::aarch64_sve_fsub_u:
2659 return Instruction::BinaryOps::FSub;
2661 return Instruction::BinaryOpsEnd;
2665static std::optional<Instruction *>
2668 if (
II.isStrictFP())
2669 return std::nullopt;
2671 auto *OpPredicate =
II.getOperand(0);
2673 if (BinOpCode == Instruction::BinaryOpsEnd ||
2675 return std::nullopt;
2677 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
2681static std::optional<Instruction *>
2683 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
2684 "Expected MLA_U intrinsic");
2685 Value *Acc =
II.getArgOperand(1);
2686 Value *MulOp0 =
II.getArgOperand(2);
2687 Value *MulOp1 =
II.getArgOperand(3);
2702 II.setArgOperand(2, MulOp1);
2703 II.setArgOperand(3, MulOp0);
2707 return std::nullopt;
2710static std::optional<Instruction *>
2712 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
2713 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
2714 "Expected SADALP or UADALP intrinsic");
2720 return std::nullopt;
2724 return std::nullopt;
2728 II.getIntrinsicID(), {II.getType()},
2729 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
2739 Intrinsic::aarch64_sve_mla>(
2743 Intrinsic::aarch64_sve_mad>(
2746 return std::nullopt;
2749static std::optional<Instruction *>
2753 Intrinsic::aarch64_sve_fmla>(IC,
II,
2758 Intrinsic::aarch64_sve_fmad>(IC,
II,
2763 Intrinsic::aarch64_sve_fmla>(IC,
II,
2766 return std::nullopt;
2769static std::optional<Instruction *>
2773 Intrinsic::aarch64_sve_fmla>(IC,
II,
2778 Intrinsic::aarch64_sve_fmad>(IC,
II,
2783 Intrinsic::aarch64_sve_fmla_u>(
2789static std::optional<Instruction *>
2793 Intrinsic::aarch64_sve_fmls>(IC,
II,
2798 Intrinsic::aarch64_sve_fnmsb>(
2803 Intrinsic::aarch64_sve_fmls>(IC,
II,
2806 return std::nullopt;
2809static std::optional<Instruction *>
2813 Intrinsic::aarch64_sve_fmls>(IC,
II,
2818 Intrinsic::aarch64_sve_fnmsb>(
2823 Intrinsic::aarch64_sve_fmls_u>(
2832 Intrinsic::aarch64_sve_mls>(
2835 return std::nullopt;
2840 Value *UnpackArg =
II.getArgOperand(0);
2842 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
2843 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
2856 return std::nullopt;
2860 auto *OpVal =
II.getOperand(0);
2861 auto *OpIndices =
II.getOperand(1);
2868 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
2869 return std::nullopt;
2884 Type *RetTy =
II.getType();
2885 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
2886 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
2890 if ((
match(
II.getArgOperand(0),
2897 if (TyA ==
B->getType() &&
2902 TyA->getMinNumElements());
2908 return std::nullopt;
2916 if (
match(
II.getArgOperand(0),
2921 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
2923 return std::nullopt;
2926static std::optional<Instruction *>
2928 Value *Mask =
II.getOperand(0);
2929 Value *BasePtr =
II.getOperand(1);
2930 Value *Index =
II.getOperand(2);
2941 BasePtr->getPointerAlignment(
II.getDataLayout());
2944 BasePtr, IndexBase);
2951 return std::nullopt;
2954static std::optional<Instruction *>
2956 Value *Val =
II.getOperand(0);
2957 Value *Mask =
II.getOperand(1);
2958 Value *BasePtr =
II.getOperand(2);
2959 Value *Index =
II.getOperand(3);
2969 BasePtr->getPointerAlignment(
II.getDataLayout());
2972 BasePtr, IndexBase);
2978 return std::nullopt;
2984 Value *Pred =
II.getOperand(0);
2985 Value *Vec =
II.getOperand(1);
2986 Value *DivVec =
II.getOperand(2);
2990 if (!SplatConstantInt)
2991 return std::nullopt;
2995 if (DivisorValue == -1)
2996 return std::nullopt;
2997 if (DivisorValue == 1)
3003 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3010 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3012 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3016 return std::nullopt;
3020 size_t VecSize = Vec.
size();
3025 size_t HalfVecSize = VecSize / 2;
3029 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3037 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3055 return std::nullopt;
3062 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3063 CurrentInsertElt = InsertElt->getOperand(0);
3069 return std::nullopt;
3073 for (
size_t I = 0;
I < Elts.
size();
I++) {
3074 if (Elts[
I] ==
nullptr)
3079 if (InsertEltChain ==
nullptr)
3080 return std::nullopt;
3086 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3087 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3088 IIScalableTy->getMinNumElements() /
3093 auto *WideShuffleMaskTy =
3104 auto NarrowBitcast =
3117 return std::nullopt;
3122 Value *Pred =
II.getOperand(0);
3123 Value *Vec =
II.getOperand(1);
3124 Value *Shift =
II.getOperand(2);
3127 Value *AbsPred, *MergedValue;
3133 return std::nullopt;
3141 return std::nullopt;
3146 return std::nullopt;
3149 {
II.getType()}, {Pred, Vec, Shift});
3156 Value *Vec =
II.getOperand(0);
3161 return std::nullopt;
3167 auto *NI =
II.getNextNode();
3170 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3172 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3173 auto *NIBB = NI->getParent();
3174 NI = NI->getNextNode();
3176 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3177 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3183 if (NextII &&
II.isIdenticalTo(NextII))
3186 return std::nullopt;
3194 {II.getType(), II.getOperand(0)->getType()},
3195 {II.getOperand(0), II.getOperand(1)}));
3202 if (PredPattern == AArch64SVEPredPattern::all ||
3203 PredPattern == AArch64SVEPredPattern::pow2)
3205 return std::nullopt;
3211 Value *Passthru =
II.getOperand(0);
3219 auto *Mask = ConstantInt::get(Ty, MaskValue);
3225 return std::nullopt;
3228static std::optional<Instruction *>
3235 return std::nullopt;
3238std::optional<Instruction *>
3249 case Intrinsic::aarch64_dmb:
3251 case Intrinsic::aarch64_neon_fmaxnm:
3252 case Intrinsic::aarch64_neon_fminnm:
3254 case Intrinsic::aarch64_sve_convert_from_svbool:
3256 case Intrinsic::aarch64_sve_dup:
3258 case Intrinsic::aarch64_sve_dup_x:
3260 case Intrinsic::aarch64_sve_cmpeq:
3261 case Intrinsic::aarch64_sve_cmpeq_wide:
3263 case Intrinsic::aarch64_sve_cmpne:
3264 case Intrinsic::aarch64_sve_cmpne_wide:
3266 case Intrinsic::aarch64_sve_rdffr:
3268 case Intrinsic::aarch64_sve_lasta:
3269 case Intrinsic::aarch64_sve_lastb:
3271 case Intrinsic::aarch64_sve_clasta_n:
3272 case Intrinsic::aarch64_sve_clastb_n:
3274 case Intrinsic::aarch64_sve_cntd:
3276 case Intrinsic::aarch64_sve_cntw:
3278 case Intrinsic::aarch64_sve_cnth:
3280 case Intrinsic::aarch64_sve_cntb:
3282 case Intrinsic::aarch64_sme_cntsd:
3284 case Intrinsic::aarch64_sve_ptest_any:
3285 case Intrinsic::aarch64_sve_ptest_first:
3286 case Intrinsic::aarch64_sve_ptest_last:
3288 case Intrinsic::aarch64_sve_fadd:
3290 case Intrinsic::aarch64_sve_fadd_u:
3292 case Intrinsic::aarch64_sve_fmul_u:
3294 case Intrinsic::aarch64_sve_fsub:
3296 case Intrinsic::aarch64_sve_fsub_u:
3298 case Intrinsic::aarch64_sve_add:
3300 case Intrinsic::aarch64_sve_add_u:
3302 Intrinsic::aarch64_sve_mla_u>(
3304 case Intrinsic::aarch64_sve_mla_u:
3306 case Intrinsic::aarch64_sve_sadalp:
3307 case Intrinsic::aarch64_sve_uadalp:
3309 case Intrinsic::aarch64_sve_sub:
3311 case Intrinsic::aarch64_sve_sub_u:
3313 Intrinsic::aarch64_sve_mls_u>(
3315 case Intrinsic::aarch64_sve_tbl:
3317 case Intrinsic::aarch64_sve_uunpkhi:
3318 case Intrinsic::aarch64_sve_uunpklo:
3319 case Intrinsic::aarch64_sve_sunpkhi:
3320 case Intrinsic::aarch64_sve_sunpklo:
3322 case Intrinsic::aarch64_sve_uzp1:
3324 case Intrinsic::aarch64_sve_zip1:
3325 case Intrinsic::aarch64_sve_zip2:
3327 case Intrinsic::aarch64_sve_ld1_gather_index:
3329 case Intrinsic::aarch64_sve_st1_scatter_index:
3331 case Intrinsic::aarch64_sve_ld1:
3333 case Intrinsic::aarch64_sve_st1:
3335 case Intrinsic::aarch64_sve_sdiv:
3337 case Intrinsic::aarch64_sve_sel:
3339 case Intrinsic::aarch64_sve_srshl:
3341 case Intrinsic::aarch64_sve_dupq_lane:
3343 case Intrinsic::aarch64_sve_insr:
3345 case Intrinsic::aarch64_sve_whilelo:
3347 case Intrinsic::aarch64_sve_ptrue:
3349 case Intrinsic::aarch64_sve_uxtb:
3351 case Intrinsic::aarch64_sve_uxth:
3353 case Intrinsic::aarch64_sve_uxtw:
3355 case Intrinsic::aarch64_sme_in_streaming_mode:
3359 return std::nullopt;
3366 SimplifyAndSetOp)
const {
3367 switch (
II.getIntrinsicID()) {
3370 case Intrinsic::aarch64_neon_fcvtxn:
3371 case Intrinsic::aarch64_neon_rshrn:
3372 case Intrinsic::aarch64_neon_sqrshrn:
3373 case Intrinsic::aarch64_neon_sqrshrun:
3374 case Intrinsic::aarch64_neon_sqshrn:
3375 case Intrinsic::aarch64_neon_sqshrun:
3376 case Intrinsic::aarch64_neon_sqxtn:
3377 case Intrinsic::aarch64_neon_sqxtun:
3378 case Intrinsic::aarch64_neon_uqrshrn:
3379 case Intrinsic::aarch64_neon_uqshrn:
3380 case Intrinsic::aarch64_neon_uqxtn:
3381 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3385 return std::nullopt;
3389 return ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3399 if (ST->useSVEForFixedLengthVectors() &&
3402 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3403 else if (ST->isNeonAvailable())
3408 if (ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3417bool AArch64TTIImpl::isSingleExtWideningInstruction(
3419 Type *SrcOverrideTy)
const {
3434 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3437 Type *SrcTy = SrcOverrideTy;
3439 case Instruction::Add:
3440 case Instruction::Sub: {
3449 if (Opcode == Instruction::Sub)
3473 assert(SrcTy &&
"Expected some SrcTy");
3475 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3481 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3483 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3487 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3490Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3492 Type *SrcOverrideTy)
const {
3493 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3494 Opcode != Instruction::Mul)
3504 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3507 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3513 ->getScalarSizeInBits();
3516 unsigned MaxEltSize = 0;
3519 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3520 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3521 MaxEltSize = std::max(EltSize0, EltSize1);
3524 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3525 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3528 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3530 MaxEltSize = DstEltSize / 2;
3531 }
else if (Opcode == Instruction::Mul &&
3539 Known.Zero.countLeadingOnes() >
3544 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3548 if (MaxEltSize * 2 > DstEltSize)
3566 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3567 (Src->isScalableTy() && !ST->hasSVE2()))
3577 if (AddUser && AddUser->getOpcode() == Instruction::Add)
3581 if (!Shr || Shr->getOpcode() != Instruction::LShr)
3585 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
3586 Src->getScalarSizeInBits() !=
3610 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3614 if (
I &&
I->hasOneUser()) {
3617 if (
Type *ExtTy = isBinExtWideningInstruction(
3618 SingleUser->getOpcode(), Dst, Operands,
3619 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3632 if (isSingleExtWideningInstruction(
3633 SingleUser->getOpcode(), Dst, Operands,
3634 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3638 if (SingleUser->getOpcode() == Instruction::Add) {
3639 if (
I == SingleUser->getOperand(1) ||
3641 cast<CastInst>(SingleUser->getOperand(1))->getOpcode() == Opcode))
3656 EVT SrcTy = TLI->getValueType(
DL, Src);
3657 EVT DstTy = TLI->getValueType(
DL, Dst);
3659 if (!SrcTy.isSimple() || !DstTy.
isSimple())
3664 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
3693 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
3696 ST->useSVEForFixedLengthVectors(WiderTy)) {
3697 std::pair<InstructionCost, MVT> LT =
3699 unsigned NumElements =
3715 const unsigned int SVE_EXT_COST = 1;
3716 const unsigned int SVE_FCVT_COST = 1;
3717 const unsigned int SVE_UNPACK_ONCE = 4;
3718 const unsigned int SVE_UNPACK_TWICE = 16;
3847 SVE_EXT_COST + SVE_FCVT_COST},
3852 SVE_EXT_COST + SVE_FCVT_COST},
3859 SVE_EXT_COST + SVE_FCVT_COST},
3863 SVE_EXT_COST + SVE_FCVT_COST},
3869 SVE_EXT_COST + SVE_FCVT_COST},
3872 SVE_EXT_COST + SVE_FCVT_COST},
3877 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3879 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3889 SVE_EXT_COST + SVE_FCVT_COST},
3894 SVE_EXT_COST + SVE_FCVT_COST},
3907 SVE_EXT_COST + SVE_FCVT_COST},
3911 SVE_EXT_COST + SVE_FCVT_COST},
3923 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3925 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3927 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3929 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3933 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
3935 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
3951 SVE_EXT_COST + SVE_FCVT_COST},
3956 SVE_EXT_COST + SVE_FCVT_COST},
3967 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3969 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3971 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3973 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3975 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3977 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
3981 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
3983 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
3985 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
3987 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4212 if (ST->hasFullFP16())
4224 Src->getScalarType(), CCH,
CostKind) +
4232 ST->isSVEorStreamingSVEAvailable() &&
4233 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4235 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4244 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4247 return Part1 + Part2;
4254 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4266 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4279 CostKind, Index,
nullptr,
nullptr);
4283 auto DstVT = TLI->getValueType(
DL, Dst);
4284 auto SrcVT = TLI->getValueType(
DL, Src);
4289 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4295 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4305 case Instruction::SExt:
4310 case Instruction::ZExt:
4311 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4324 return Opcode == Instruction::PHI ? 0 : 1;
4333 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4342 if (!LT.second.isVector())
4347 if (LT.second.isFixedLengthVector()) {
4348 unsigned Width = LT.second.getVectorNumElements();
4349 Index = Index % Width;
4363 if (VIC == TTI::VectorInstrContext::Load) {
4364 if (ST->hasFastLD1Single())
4376 : ST->getVectorInsertExtractBaseCost() + 1;
4400 auto ExtractCanFuseWithFmul = [&]() {
4407 auto IsAllowedScalarTy = [&](
const Type *
T) {
4408 return T->isFloatTy() ||
T->isDoubleTy() ||
4409 (
T->isHalfTy() && ST->hasFullFP16());
4413 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4416 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4417 !BO->getType()->isVectorTy();
4422 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4426 return Idx == 0 || (RegWidth != 0 && (Idx * EltSz) % RegWidth == 0);
4435 DenseMap<User *, unsigned> UserToExtractIdx;
4436 for (
auto *U :
Scalar->users()) {
4437 if (!IsUserFMulScalarTy(U))
4441 UserToExtractIdx[
U];
4443 if (UserToExtractIdx.
empty())
4445 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4446 for (
auto *U : S->users()) {
4447 if (UserToExtractIdx.
contains(U)) {
4449 auto *Op0 =
FMul->getOperand(0);
4450 auto *Op1 =
FMul->getOperand(1);
4451 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4452 UserToExtractIdx[
U] =
L;
4458 for (
auto &[U, L] : UserToExtractIdx) {
4470 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4471 if (!IsUserFMulScalarTy(U))
4476 const auto *BO = cast<BinaryOperator>(U);
4477 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4478 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4480 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4483 return IsExtractLaneEquivalentToZero(
4484 cast<ConstantInt>(OtherEE->getIndexOperand())
4487 OtherEE->getType()->getScalarSizeInBits());
4495 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4496 ExtractCanFuseWithFmul())
4501 :
ST->getVectorInsertExtractBaseCost();
4510 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
4513 return getVectorInstrCostHelper(Opcode, Val,
CostKind, Index,
nullptr,
4519 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4521 return getVectorInstrCostHelper(Opcode, Val,
CostKind, Index,
nullptr, Scalar,
4522 ScalarUserAndIdx, VIC);
4529 return getVectorInstrCostHelper(
I.getOpcode(), Val,
CostKind, Index, &
I,
4536 unsigned Index)
const {
4548 : ST->getVectorInsertExtractBaseCost() + 1;
4557 if (Ty->getElementType()->isFloatingPointTy())
4560 unsigned VecInstCost =
4562 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
4569 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
4570 return std::nullopt;
4571 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
4572 return std::nullopt;
4574 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
4575 return std::nullopt;
4582 Cost += InstCost(PromotedTy);
4605 Op2Info, Args, CxtI);
4609 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4616 Ty,
CostKind, Op1Info, Op2Info,
true,
4619 [&](
Type *PromotedTy) {
4623 return *PromotedCost;
4626 if (Ty->getScalarType()->isFP128Ty())
4634 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
4654 ST->hasLimited64bitVectorMulBandwidth())
4657 if (Ty->getScalarSizeInBits() > 64) {
4662 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
4665 if (LT.second == MVT::v2i64) {
4669 return LT.first * Mul64CostFactor;
4690 if (LT.second == MVT::nxv2i64)
4691 return LT.first * Mul64CostFactor;
4750 auto VT = TLI->getValueType(
DL, Ty);
4751 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
4755 : (3 * AsrCost + AddCost);
4757 return MulCost + AsrCost + 2 * AddCost;
4759 }
else if (VT.isVector()) {
4769 if (Ty->isScalableTy() && ST->hasSVE())
4770 Cost += 2 * AsrCost;
4775 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
4779 }
else if (LT.second == MVT::v2i64) {
4780 return VT.getVectorNumElements() *
4787 if (Ty->isScalableTy() && ST->hasSVE())
4788 return MulCost + 2 * AddCost + 2 * AsrCost;
4789 return 2 * MulCost + AddCost + AsrCost + UsraCost;
4794 LT.second.isFixedLengthVector()) {
4804 return ExtractCost + InsertCost +
4812 auto VT = TLI->getValueType(
DL, Ty);
4828 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
4829 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
4830 LT.second == MVT::nxv16i8;
4831 bool Is128bit = LT.second.is128BitVector();
4843 (HasMULH ? 0 : ShrCost) +
4844 AddCost * 2 + ShrCost;
4845 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
4852 if (!VT.isVector() && VT.getSizeInBits() > 64)
4856 Opcode, Ty,
CostKind, Op1Info, Op2Info);
4858 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
4862 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
4872 if (
nullptr != Entry)
4880 FVTy && LT.second.isFixedLengthVector()) {
4881 unsigned NumElts = FVTy->getNumElements();
4882 unsigned RegElts = LT.second.getVectorNumElements();
4884 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
4888 if (LT.second.getScalarType() == MVT::i8)
4890 else if (LT.second.getScalarType() == MVT::i16)
4902 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
4903 return (4 + DivCost) * VTy->getNumElements();
4909 -1,
nullptr,
nullptr);
4932 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
4933 (Ty->isHalfTy() && ST->hasFullFP16())) &&
4942 if (!Ty->getScalarType()->isFP128Ty())
4949 if (!Ty->getScalarType()->isFP128Ty())
4950 return 2 * LT.first;
4957 if (!Ty->isVectorTy())
4973 int MaxMergeDistance = 64;
4977 return NumVectorInstToHideOverhead;
4987 unsigned Opcode1,
unsigned Opcode2)
const {
4990 if (!
Sched.hasInstrSchedModel())
4994 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
4996 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5002 "Cannot handle variant scheduling classes without an MI");
5018 const int AmortizationCost = 20;
5026 VecPred = CurrentPred;
5034 static const auto ValidMinMaxTys = {
5035 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5036 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5037 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5041 (ST->hasFullFP16() &&
5047 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5048 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5049 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5050 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5051 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5052 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5053 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5054 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5055 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5056 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5057 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5059 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5060 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5069 if (Opcode == Instruction::FCmp) {
5071 ValTy,
CostKind, Op1Info, Op2Info,
false,
5073 false, [&](
Type *PromotedTy) {
5085 return *PromotedCost;
5089 if (LT.second.getScalarType() != MVT::f64 &&
5090 LT.second.getScalarType() != MVT::f32 &&
5091 LT.second.getScalarType() != MVT::f16)
5096 unsigned Factor = 1;
5097 if (!CondTy->isVectorTy() &&
5111 AArch64::FCMEQv4f32))
5123 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5142 Op1Info, Op2Info,
I);
5148 if (ST->requiresStrictAlign()) {
5153 Options.AllowOverlappingLoads =
true;
5154 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5159 Options.LoadSizes = {8, 4, 2, 1};
5160 Options.AllowedTailExpansions = {3, 5, 6};
5165 return ST->hasSVE();
5171 switch (MICA.
getID()) {
5172 case Intrinsic::masked_scatter:
5173 case Intrinsic::masked_gather:
5175 case Intrinsic::masked_load:
5176 case Intrinsic::masked_expandload:
5177 case Intrinsic::masked_store:
5191 if (!LT.first.isValid())
5196 if (VT->getElementType()->isIntegerTy(1))
5207 if (MICA.
getID() == Intrinsic::masked_expandload) {
5223 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5224 return MemOpCost * 2;
5233 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5234 "Should be called on only load or stores.");
5236 case Instruction::Load:
5239 return ST->getGatherOverhead();
5241 case Instruction::Store:
5244 return ST->getScatterOverhead();
5255 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5256 MICA.
getID() == Intrinsic::vp_gather)
5258 : Instruction::Store;
5268 if (!LT.first.isValid())
5272 if (!LT.second.isVector() ||
5274 VT->getElementType()->isIntegerTy(1))
5284 ElementCount LegalVF = LT.second.getVectorElementCount();
5287 {TTI::OK_AnyValue, TTI::OP_None},
I);
5303 EVT VT = TLI->getValueType(
DL, Ty,
true);
5305 if (VT == MVT::Other)
5310 if (!LT.first.isValid())
5320 (VTy->getElementType()->isIntegerTy(1) &&
5321 !VTy->getElementCount().isKnownMultipleOf(
5331 if (Opcode == Instruction::Store)
5335 if (ST->getFixedLoadLatency())
5336 return (LT.first - 1) + ST->getFixedLoadLatency();
5345 if (LT.second.isScalableVector() ||
5346 ST->useSVEForFixedLengthVectors(LT.second)) {
5347 Inst = AArch64::LDR_ZXI;
5348 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5349 switch (LT.second.getSizeInBits()) {
5351 Inst = AArch64::LDRBui;
5354 Inst = AArch64::LDRHui;
5357 Inst = AArch64::LDRSui;
5360 Inst = AArch64::LDRDui;
5363 Inst = AArch64::LDRQui;
5369 switch (LT.second.getSizeInBits()) {
5371 Inst = AArch64::LDRBBui;
5374 Inst = AArch64::LDRHHui;
5377 Inst = AArch64::LDRWui;
5380 Inst = AArch64::LDRXui;
5388 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5392 float NumLoads = (LT.first - 1).
getValue();
5393 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5394 Sched.computeInstrLatency(*ST, *SCD);
5397 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5398 LT.second.is128BitVector() && Alignment <
Align(16)) {
5404 const int AmortizationCost = 6;
5406 return LT.first * 2 * AmortizationCost;
5410 if (Ty->isPtrOrPtrVectorTy())
5415 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5417 if (VT == MVT::v4i8)
5424 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5439 while (!TypeWorklist.
empty()) {
5461 bool UseMaskForCond,
bool UseMaskForGaps)
const {
5462 assert(Factor >= 2 &&
"Invalid interleave factor");
5477 if (!VecTy->
isScalableTy() && (UseMaskForCond || UseMaskForGaps))
5480 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
5483 EC.divideCoefficientBy(Factor));
5489 if (EC.isKnownMultipleOf(Factor) &&
5490 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
5491 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
5496 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
5502 if (UseMaskForCond) {
5503 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5504 : Intrinsic::masked_store;
5524 if (Opcode == Instruction::Store && Factor == 4 &&
5525 SubVecCost.second.getScalarSizeInBits() ==
5526 (4 * ResultCost.second.getScalarSizeInBits()))
5527 LegalizationCost *= 4;
5529 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
5535 UseMaskForCond, UseMaskForGaps);
5542 for (
auto *
I : Tys) {
5543 if (!
I->isVectorTy())
5554 Align Alignment)
const {
5561 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
5562 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
5567 bool HasUnorderedReductions)
const {
5570 return ST->getMaxInterleaveFactor();
5580 enum { MaxStridedLoads = 7 };
5582 int StridedLoads = 0;
5585 for (
const auto BB : L->blocks()) {
5586 for (
auto &
I : *BB) {
5592 if (L->isLoopInvariant(PtrValue))
5597 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
5606 if (StridedLoads > MaxStridedLoads / 2)
5607 return StridedLoads;
5610 return StridedLoads;
5613 int StridedLoads = countStridedLoads(L, SE);
5615 <<
" strided loads\n");
5631 unsigned *FinalSize) {
5635 for (
auto *BB : L->getBlocks()) {
5636 for (
auto &
I : *BB) {
5642 if (!Cost.isValid())
5646 if (LoopCost > Budget)
5668 if (MaxTC > 0 && MaxTC <= 32)
5679 if (Blocks.
size() != 2)
5701 if (!L->isInnermost() || L->getNumBlocks() > 8)
5705 if (!L->getExitBlock())
5711 bool HasParellelizableReductions =
5712 L->getNumBlocks() == 1 &&
5713 any_of(L->getHeader()->phis(),
5715 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
5718 if (HasParellelizableReductions &&
5740 if (HasParellelizableReductions) {
5751 if (Header == Latch) {
5754 unsigned Width = 10;
5760 unsigned MaxInstsPerLine = 16;
5762 unsigned BestUC = 1;
5763 unsigned SizeWithBestUC = BestUC *
Size;
5765 unsigned SizeWithUC = UC *
Size;
5766 if (SizeWithUC > 48)
5768 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
5769 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
5771 SizeWithBestUC = BestUC *
Size;
5781 for (
auto *BB : L->blocks()) {
5782 for (
auto &
I : *BB) {
5792 for (
auto *U :
I.users())
5794 LoadedValuesPlus.
insert(U);
5801 return LoadedValuesPlus.
contains(
SI->getOperand(0));
5827 auto *I = dyn_cast<Instruction>(V);
5828 return I && DependsOnLoopLoad(I, Depth + 1);
5835 DependsOnLoopLoad(
I, 0)) {
5851 if (L->getLoopDepth() > 1)
5862 for (
auto *BB : L->getBlocks()) {
5863 for (
auto &
I : *BB) {
5867 if (IsVectorized &&
I.getType()->isVectorTy())
5884 if (ST->isAppleMLike())
5886 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
5908 !ST->getSchedModel().isOutOfOrder()) {
5931 bool CanCreate)
const {
5935 case Intrinsic::aarch64_neon_st1x2:
5936 case Intrinsic::aarch64_neon_st1x3:
5937 case Intrinsic::aarch64_neon_st1x4:
5938 case Intrinsic::aarch64_neon_st2:
5939 case Intrinsic::aarch64_neon_st3:
5940 case Intrinsic::aarch64_neon_st4: {
5943 if (!CanCreate || !ST)
5945 unsigned NumElts = Inst->
arg_size() - 1;
5946 if (ST->getNumElements() != NumElts)
5948 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
5954 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
5956 Res = Builder.CreateInsertValue(Res, L, i);
5960 case Intrinsic::aarch64_neon_ld1x2:
5961 case Intrinsic::aarch64_neon_ld1x3:
5962 case Intrinsic::aarch64_neon_ld1x4:
5963 case Intrinsic::aarch64_neon_ld2:
5964 case Intrinsic::aarch64_neon_ld3:
5965 case Intrinsic::aarch64_neon_ld4:
5966 if (Inst->
getType() == ExpectedType)
5977 case Intrinsic::aarch64_neon_ld1x2:
5978 case Intrinsic::aarch64_neon_ld1x3:
5979 case Intrinsic::aarch64_neon_ld1x4:
5980 case Intrinsic::aarch64_neon_ld2:
5981 case Intrinsic::aarch64_neon_ld3:
5982 case Intrinsic::aarch64_neon_ld4:
5983 Info.ReadMem =
true;
5984 Info.WriteMem =
false;
5987 case Intrinsic::aarch64_neon_st1x2:
5988 case Intrinsic::aarch64_neon_st1x3:
5989 case Intrinsic::aarch64_neon_st1x4:
5990 case Intrinsic::aarch64_neon_st2:
5991 case Intrinsic::aarch64_neon_st3:
5992 case Intrinsic::aarch64_neon_st4:
5993 Info.ReadMem =
false;
5994 Info.WriteMem =
true;
6003 case Intrinsic::aarch64_neon_ld1x2:
6004 case Intrinsic::aarch64_neon_st1x2:
6005 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6007 case Intrinsic::aarch64_neon_ld1x3:
6008 case Intrinsic::aarch64_neon_st1x3:
6009 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6011 case Intrinsic::aarch64_neon_ld1x4:
6012 case Intrinsic::aarch64_neon_st1x4:
6013 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6015 case Intrinsic::aarch64_neon_ld2:
6016 case Intrinsic::aarch64_neon_st2:
6017 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6019 case Intrinsic::aarch64_neon_ld3:
6020 case Intrinsic::aarch64_neon_st3:
6021 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6023 case Intrinsic::aarch64_neon_ld4:
6024 case Intrinsic::aarch64_neon_st4:
6025 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6037 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6038 bool Considerable =
false;
6039 AllowPromotionWithoutCommonHeader =
false;
6042 Type *ConsideredSExtType =
6044 if (
I.getType() != ConsideredSExtType)
6048 for (
const User *U :
I.users()) {
6050 Considerable =
true;
6054 if (GEPInst->getNumOperands() > 2) {
6055 AllowPromotionWithoutCommonHeader =
true;
6060 return Considerable;
6111 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6121 return LegalizationCost + 2;
6131 LegalizationCost *= LT.first - 1;
6134 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6143 return LegalizationCost + 2;
6151 std::optional<FastMathFlags> FMF,
6167 return BaseCost + FixedVTy->getNumElements();
6184 MVT MTy = LT.second;
6185 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6233 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6234 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6246 return (LT.first - 1) +
Log2_32(NElts);
6251 return (LT.first - 1) + Entry->Cost;
6263 if (LT.first != 1) {
6269 ExtraCost *= LT.first - 1;
6272 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6273 return Cost + ExtraCost;
6281 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6283 EVT VecVT = TLI->getValueType(
DL, VecTy);
6284 EVT ResVT = TLI->getValueType(
DL, ResTy);
6294 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6296 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6298 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6300 return (LT.first - 1) * 2 + 2;
6311 EVT VecVT = TLI->getValueType(
DL, VecTy);
6312 EVT ResVT = TLI->getValueType(
DL, ResTy);
6315 RedOpcode == Instruction::Add) {
6321 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6323 return LT.first + 2;
6358 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6359 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6373 if (LT.second.getScalarType() == MVT::i1) {
6382 assert(Entry &&
"Illegal Type for Splice");
6383 LegalizationCost += Entry->Cost;
6384 return LegalizationCost * LT.first;
6388 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
6397 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
6398 Opcode != Instruction::FAdd && Opcode != Instruction::FSub) ||
6405 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
6406 if (!FMF->allowReassoc() || !FMF->allowContract())
6410 "FastMathFlags only apply to floating-point partial reductions");
6414 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
6415 "Unexpected values for OpBExtend or InputTypeB");
6419 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
6420 InputTypeA != InputTypeB))
6423 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
6426 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
6439 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
6448 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
6454 std::pair<InstructionCost, MVT> AccumLT =
6456 std::pair<InstructionCost, MVT> InputLT =
6460 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
6461 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
6462 (AccumLT.second.isFixedLengthVector() &&
6463 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
6467 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
6475 if (AccumLT.second.getScalarType() == MVT::i32 &&
6476 InputLT.second.getScalarType() == MVT::i8) {
6478 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
6479 return Cost + INegCost;
6481 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
6482 return Cost + INegCost;
6487 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
6488 return Cost * 3 + INegCost;
6491 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
6493 if (AccumLT.second.getScalarType() == MVT::i64 &&
6494 InputLT.second.getScalarType() == MVT::i16)
6495 return Cost + INegCost;
6498 if (AccumLT.second.getScalarType() == MVT::i32 &&
6499 InputLT.second.getScalarType() == MVT::i16 &&
6500 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
6503 if (AccumLT.second.getScalarType() == MVT::i64 &&
6504 InputLT.second.getScalarType() == MVT::i8)
6510 return Cost + INegCost;
6513 if (AccumLT.second.getScalarType() == MVT::i16 &&
6514 InputLT.second.getScalarType() == MVT::i8 &&
6515 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
6521 if (Opcode == Instruction::FAdd && !IsSub &&
6522 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
6523 AccumLT.second.getScalarType() == MVT::f32 &&
6524 InputLT.second.getScalarType() == MVT::f16)
6528 if (Ratio == 2 && !IsUSDot) {
6529 MVT InVT = InputLT.second.getScalarType();
6532 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
6537 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
6541 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
6542 InVT == MVT::bf16 && IsSub)
6552 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
6553 return Cost * 2 + FNegCost;
6557 AccumType, VF, OpAExtend, OpBExtend,
6569 "Expected the Mask to match the return size if given");
6571 "Expected the same scalar types");
6577 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
6578 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
6579 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
6587 return std::max<InstructionCost>(1, LT.first / 4);
6595 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
6597 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
6600 unsigned TpNumElts = Mask.size();
6601 unsigned LTNumElts = LT.second.getVectorNumElements();
6602 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
6604 LT.second.getVectorElementCount());
6606 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
6608 for (
unsigned N = 0;
N < NumVecs;
N++) {
6612 unsigned Source1 = -1U, Source2 = -1U;
6613 unsigned NumSources = 0;
6614 for (
unsigned E = 0; E < LTNumElts; E++) {
6615 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
6624 unsigned Source = MaskElt / LTNumElts;
6625 if (NumSources == 0) {
6628 }
else if (NumSources == 1 && Source != Source1) {
6631 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
6637 if (Source == Source1)
6639 else if (Source == Source2)
6640 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
6649 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
6660 NTp, NTp, NMask,
CostKind, 0,
nullptr, Args,
6663 Result.first->second = NCost;
6677 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
6678 if (LT.second.getFixedSizeInBits() >= 128 &&
6680 LT.second.getVectorNumElements() / 2) {
6683 if (Index == (
int)LT.second.getVectorNumElements() / 2)
6697 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
6700 return M.value() < 0 || M.value() == (int)M.index();
6706 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
6707 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
6716 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
6717 ST->isSVEorStreamingSVEAvailable() &&
6722 if (ST->isSVEorStreamingSVEAvailable() &&
6736 if (IsLoad && LT.second.isVector() &&
6738 LT.second.getVectorElementCount()))
6744 if (Mask.size() == 4 &&
6746 (SrcTy->getScalarSizeInBits() == 16 ||
6747 SrcTy->getScalarSizeInBits() == 32) &&
6748 all_of(Mask, [](
int E) {
return E < 8; }))
6754 if (LT.second.isFixedLengthVector() &&
6755 LT.second.getVectorNumElements() == Mask.size() &&
6761 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
6762 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
6763 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
6764 isREVMask(Mask, LT.second.getScalarSizeInBits(),
6765 LT.second.getVectorNumElements(), 16) ||
6766 isREVMask(Mask, LT.second.getScalarSizeInBits(),
6767 LT.second.getVectorNumElements(), 32) ||
6768 isREVMask(Mask, LT.second.getScalarSizeInBits(),
6769 LT.second.getVectorNumElements(), 64) ||
6772 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
6901 return LT.first * Entry->Cost;
6910 LT.second.getSizeInBits() <= 128 && SubTp) {
6912 if (SubLT.second.isVector()) {
6913 int NumElts = LT.second.getVectorNumElements();
6914 int NumSubElts = SubLT.second.getVectorNumElements();
6915 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
6921 if (IsExtractSubvector)
6938 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
6957 return ST->useFixedOverScalableIfEqualCost();
6961 return ST->getEpilogueVectorizationMinVF();
6996 unsigned NumInsns = 0;
6998 NumInsns += BB->size();
7008 int64_t Scale,
unsigned AddrSpace)
const {
7036 if (
I->getOpcode() == Instruction::Or &&
7040 if (
I->getOpcode() == Instruction::Add ||
7041 I->getOpcode() == Instruction::Sub)
7066 return all_equal(Shuf->getShuffleMask());
7073 bool AllowSplat =
false) {
7078 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7079 auto *FullTy = FullV->
getType();
7080 auto *HalfTy = HalfV->getType();
7082 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7085 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7088 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7092 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7106 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7107 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7121 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7122 (M2Start != 0 && M2Start != (NumElements / 2)))
7124 if (S1Op1 && S2Op1 && M1Start != M2Start)
7134 return Ext->getType()->getScalarSizeInBits() ==
7135 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7149 Value *VectorOperand =
nullptr;
7166 if (!
GEP ||
GEP->getNumOperands() != 2)
7170 Value *Offsets =
GEP->getOperand(1);
7173 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7179 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7180 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7181 Ops.push_back(&
GEP->getOperandUse(1));
7217 switch (
II->getIntrinsicID()) {
7218 case Intrinsic::aarch64_neon_smull:
7219 case Intrinsic::aarch64_neon_umull:
7222 Ops.push_back(&
II->getOperandUse(0));
7223 Ops.push_back(&
II->getOperandUse(1));
7228 case Intrinsic::fma:
7229 case Intrinsic::fmuladd:
7236 Ops.push_back(&
II->getOperandUse(0));
7238 Ops.push_back(&
II->getOperandUse(1));
7241 case Intrinsic::aarch64_neon_sqdmull:
7242 case Intrinsic::aarch64_neon_sqdmulh:
7243 case Intrinsic::aarch64_neon_sqrdmulh:
7246 Ops.push_back(&
II->getOperandUse(0));
7248 Ops.push_back(&
II->getOperandUse(1));
7249 return !
Ops.empty();
7250 case Intrinsic::aarch64_neon_fmlal:
7251 case Intrinsic::aarch64_neon_fmlal2:
7252 case Intrinsic::aarch64_neon_fmlsl:
7253 case Intrinsic::aarch64_neon_fmlsl2:
7256 Ops.push_back(&
II->getOperandUse(1));
7258 Ops.push_back(&
II->getOperandUse(2));
7259 return !
Ops.empty();
7260 case Intrinsic::aarch64_sve_ptest_first:
7261 case Intrinsic::aarch64_sve_ptest_last:
7263 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7264 Ops.push_back(&
II->getOperandUse(0));
7265 return !
Ops.empty();
7266 case Intrinsic::aarch64_sme_write_horiz:
7267 case Intrinsic::aarch64_sme_write_vert:
7268 case Intrinsic::aarch64_sme_writeq_horiz:
7269 case Intrinsic::aarch64_sme_writeq_vert: {
7271 if (!Idx || Idx->getOpcode() != Instruction::Add)
7273 Ops.push_back(&
II->getOperandUse(1));
7276 case Intrinsic::aarch64_sme_read_horiz:
7277 case Intrinsic::aarch64_sme_read_vert:
7278 case Intrinsic::aarch64_sme_readq_horiz:
7279 case Intrinsic::aarch64_sme_readq_vert:
7280 case Intrinsic::aarch64_sme_ld1b_vert:
7281 case Intrinsic::aarch64_sme_ld1h_vert:
7282 case Intrinsic::aarch64_sme_ld1w_vert:
7283 case Intrinsic::aarch64_sme_ld1d_vert:
7284 case Intrinsic::aarch64_sme_ld1q_vert:
7285 case Intrinsic::aarch64_sme_st1b_vert:
7286 case Intrinsic::aarch64_sme_st1h_vert:
7287 case Intrinsic::aarch64_sme_st1w_vert:
7288 case Intrinsic::aarch64_sme_st1d_vert:
7289 case Intrinsic::aarch64_sme_st1q_vert:
7290 case Intrinsic::aarch64_sme_ld1b_horiz:
7291 case Intrinsic::aarch64_sme_ld1h_horiz:
7292 case Intrinsic::aarch64_sme_ld1w_horiz:
7293 case Intrinsic::aarch64_sme_ld1d_horiz:
7294 case Intrinsic::aarch64_sme_ld1q_horiz:
7295 case Intrinsic::aarch64_sme_st1b_horiz:
7296 case Intrinsic::aarch64_sme_st1h_horiz:
7297 case Intrinsic::aarch64_sme_st1w_horiz:
7298 case Intrinsic::aarch64_sme_st1d_horiz:
7299 case Intrinsic::aarch64_sme_st1q_horiz: {
7301 if (!Idx || Idx->getOpcode() != Instruction::Add)
7303 Ops.push_back(&
II->getOperandUse(3));
7306 case Intrinsic::aarch64_neon_pmull:
7309 Ops.push_back(&
II->getOperandUse(0));
7310 Ops.push_back(&
II->getOperandUse(1));
7312 case Intrinsic::aarch64_neon_pmull64:
7314 II->getArgOperand(1)))
7316 Ops.push_back(&
II->getArgOperandUse(0));
7317 Ops.push_back(&
II->getArgOperandUse(1));
7319 case Intrinsic::masked_gather:
7322 Ops.push_back(&
II->getArgOperandUse(0));
7324 case Intrinsic::masked_scatter:
7327 Ops.push_back(&
II->getArgOperandUse(1));
7334 auto ShouldSinkCondition = [](
Value *
Cond,
7339 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
7343 Ops.push_back(&
II->getOperandUse(0));
7347 switch (
I->getOpcode()) {
7348 case Instruction::GetElementPtr:
7349 case Instruction::Add:
7350 case Instruction::Sub:
7352 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
7354 Ops.push_back(&
I->getOperandUse(
Op));
7359 case Instruction::Select: {
7360 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
7363 Ops.push_back(&
I->getOperandUse(0));
7366 case Instruction::UncondBr:
7368 case Instruction::CondBr: {
7372 Ops.push_back(&
I->getOperandUse(0));
7375 case Instruction::FMul:
7380 Ops.push_back(&
I->getOperandUse(0));
7382 Ops.push_back(&
I->getOperandUse(1));
7392 case Instruction::Xor:
7395 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
7397 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
7402 case Instruction::And:
7403 case Instruction::Or:
7406 if (
I->getOpcode() == Instruction::Or &&
7411 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
7414 for (
auto &
Op :
I->operands()) {
7426 Ops.push_back(&Not);
7427 Ops.push_back(&InsertElt);
7437 if (!
I->getType()->isVectorTy())
7438 return !
Ops.empty();
7440 switch (
I->getOpcode()) {
7441 case Instruction::Sub:
7442 case Instruction::Add: {
7451 Ops.push_back(&Ext1->getOperandUse(0));
7452 Ops.push_back(&Ext2->getOperandUse(0));
7455 Ops.push_back(&
I->getOperandUse(0));
7456 Ops.push_back(&
I->getOperandUse(1));
7460 case Instruction::Or: {
7463 if (ST->hasNEON()) {
7477 if (
I->getParent() != MainAnd->
getParent() ||
7482 if (
I->getParent() != IA->getParent() ||
7483 I->getParent() != IB->getParent())
7488 Ops.push_back(&
I->getOperandUse(0));
7489 Ops.push_back(&
I->getOperandUse(1));
7498 case Instruction::Mul: {
7499 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
7502 if (Ty->isScalableTy())
7506 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
7509 int NumZExts = 0, NumSExts = 0;
7510 for (
auto &
Op :
I->operands()) {
7517 auto *ExtOp = Ext->getOperand(0);
7518 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
7519 Ops.push_back(&Ext->getOperandUse(0));
7527 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
7528 I->getType()->getScalarSizeInBits())
7565 if (!ElementConstant || !ElementConstant->
isZero())
7568 unsigned Opcode = OperandInstr->
getOpcode();
7569 if (Opcode == Instruction::SExt)
7571 else if (Opcode == Instruction::ZExt)
7576 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
7586 Ops.push_back(&Insert->getOperandUse(1));
7592 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
7596 if (!ShouldSinkSplatForIndexedVariant(
I))
7601 Ops.push_back(&
I->getOperandUse(0));
7603 Ops.push_back(&
I->getOperandUse(1));
7605 return !
Ops.empty();
7607 case Instruction::FMul: {
7609 if (
I->getType()->isScalableTy())
7610 return !
Ops.empty();
7614 return !
Ops.empty();
7618 Ops.push_back(&
I->getOperandUse(0));
7620 Ops.push_back(&
I->getOperandUse(1));
7621 return !
Ops.empty();
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
MachineInstr unsigned OpIdx
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
bool preferFixedOverScalableIfEqualCost(bool IsEpilogue) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
bool isSMEABIRoutine() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
LLVM_ABI bool isScalableTy(SmallPtrSetImpl< const Type * > &Visited) const
Return true if this is a type whose size is a known multiple of vscale.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
@ C
The default llvm calling convention, compatible with C.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
LLVM_ABI Libcall getPOW(EVT RetVT)
getPOW - Return the POW_* value for the given types, or UNKNOWN_LIBCALL if there is none.
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const DenseMap< Value *, const SCEV * > &StridesMap=DenseMap< Value *, const SCEV * >(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...