68 if (!VPBB->getParent())
71 auto EndIter = Term ? Term->getIterator() : VPBB->end();
76 VPValue *VPV = Ingredient.getVPSingleValue();
100 nullptr , IsConsecutive,
101 *VPI, Ingredient.getDebugLoc());
105 VPI->getOperand(0)->getScalarType(), PSE,
108 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
109 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
112 Ingredient.operands(), *VPI,
113 Ingredient.getDebugLoc(),
GEP);
125 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
130 if (VectorID == Intrinsic::assume ||
131 VectorID == Intrinsic::lifetime_end ||
132 VectorID == Intrinsic::lifetime_start ||
133 VectorID == Intrinsic::sideeffect ||
134 VectorID == Intrinsic::pseudoprobe) {
139 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
140 VectorID != Intrinsic::pseudoprobe;
144 Ingredient.getDebugLoc());
147 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
148 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
152 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
156 *VPI, Ingredient.getDebugLoc());
160 "inductions must be created earlier");
169 "Only recpies with zero or one defined values expected");
170 Ingredient.eraseFromParent();
181 const Loop *L =
nullptr;
186 if (
A->getOpcode() != Instruction::Store ||
187 B->getOpcode() != Instruction::Store)
200 const APInt *Distance;
206 Type *TyA =
A->getOperand(0)->getScalarType();
208 Type *TyB =
B->getOperand(0)->getScalarType();
214 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
216 auto VFs =
B->getParent()->getPlan()->vectorFactors();
220 return Distance->
abs().
uge(
228 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
229 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
238 return ExcludeRecipes.contains(
Store) ||
239 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
252 std::optional<SinkStoreInfo> SinkInfo = {}) {
253 bool CheckReads = SinkInfo.has_value();
257 if (SinkInfo && SinkInfo->shouldSkip(R))
261 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
286template <
unsigned Opcode>
291 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
292 "Only Load and Store opcodes supported");
293 constexpr bool IsLoad = (Opcode == Instruction::Load);
296 RecipesByAddressAndType;
301 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
305 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
309 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
314 for (
auto &Group :
Groups) {
329 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
341 if (Candidate->getParent() == SinkTo ||
346 if (!ScalarVFOnly && RepR->isSingleScalar())
349 WorkList.
insert({SinkTo, Candidate});
361 for (
auto &Recipe : *VPBB)
363 InsertIfValidSinkCandidate(VPBB,
Op);
367 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
370 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
375 auto UsersOutsideSinkTo =
377 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
379 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
380 return !U->usesFirstLaneOnly(SinkCandidate);
383 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
385 if (NeedsDuplicating) {
389 if (
auto *SinkCandidateRepR =
394 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
395 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
399 Clone = SinkCandidate->
clone();
409 InsertIfValidSinkCandidate(SinkTo,
Op);
419 if (!EntryBB || EntryBB->size() != 1 ||
429 if (EntryBB->getNumSuccessors() != 2)
434 if (!Succ0 || !Succ1)
437 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
439 if (Succ0->getSingleSuccessor() == Succ1)
441 if (Succ1->getSingleSuccessor() == Succ0)
458 if (!Region1->isReplicator())
460 auto *MiddleBasicBlock =
462 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
467 if (!Region2 || !Region2->isReplicator())
472 if (!Mask1 || Mask1 != Mask2)
475 assert(Mask1 && Mask2 &&
"both region must have conditions");
481 if (TransformedRegions.
contains(Region1))
488 if (!Then1 || !Then2)
508 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
514 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
515 Phi1ToMove.eraseFromParent();
518 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
532 TransformedRegions.
insert(Region1);
535 return !TransformedRegions.
empty();
543 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
544 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
545 auto *BlockInMask = PredRecipe->
getMask();
566 Region->setParent(ParentRegion);
572 RecipeWithoutMask->getDebugLoc());
573 Exiting->appendRecipe(PHIRecipe);
586 if (RepR->isPredicated())
605 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
617 if (!VPBB->getParent())
621 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
630 R.moveBefore(*PredVPBB, PredVPBB->
end());
632 auto *ParentRegion = VPBB->getParent();
633 if (ParentRegion && ParentRegion->getExiting() == VPBB)
634 ParentRegion->setExiting(PredVPBB);
638 return !WorkList.
empty();
645 bool ShouldSimplify =
true;
646 while (ShouldSimplify) {
662 if (!
IV ||
IV->getTruncInst())
677 for (
auto *U : FindMyCast->
users()) {
679 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
680 FoundUserCast = UserCast;
687 FindMyCast = FoundUserCast;
689 if (FindMyCast !=
IV)
711 VPUser *PhiUser = PhiR->getSingleUser();
717 PhiR->replaceAllUsesWith(Start);
718 PhiR->eraseFromParent();
755 Def->user_empty() || !Def->getUnderlyingValue() ||
756 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
769 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
771 Def->getUnderlyingInstr());
772 Clone->insertAfter(Def);
773 Def->replaceAllUsesWith(Clone);
785 PtrIV->replaceAllUsesWith(PtrAdd);
792 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
793 return U->usesScalars(WideIV);
802 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
805 Plan,
ID.getKind(),
ID.getInductionOpcode(),
807 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
808 WideIV->getDebugLoc(), Builder, WrapFlags);
811 if (!HasOnlyVectorVFs) {
813 "plans containing a scalar VF cannot also include scalable VFs");
814 WideIV->replaceAllUsesWith(Steps);
817 WideIV->replaceUsesWithIf(Steps,
818 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
820 return U.usesFirstLaneOnly(WideIV);
821 return U.usesScalars(WideIV);
837 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
842 if (!Def || Def->getNumOperands() != 2)
850 auto IsWideIVInc = [&]() {
851 auto &
ID = WideIV->getInductionDescriptor();
854 VPValue *IVStep = WideIV->getStepValue();
855 switch (
ID.getInductionOpcode()) {
856 case Instruction::Add:
858 case Instruction::FAdd:
860 case Instruction::FSub:
863 case Instruction::Sub: {
883 return IsWideIVInc() ? WideIV :
nullptr;
900 if (WideIntOrFp && WideIntOrFp->getTruncInst())
911 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
913 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
914 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
919 if (Incoming != WideIV) {
921 EndValue =
B.createAdd(EndValue, One,
DL);
926 VPIRValue *Start = WideIV->getStartValue();
927 VPValue *Step = WideIV->getStepValue();
928 EndValue =
B.createDerivedIV(
930 Start, EndValue, Step);
944 if (WideIntOrFp && WideIntOrFp->getTruncInst())
954 Start, VectorTC, Step);
986 assert(EndValue &&
"Must have computed the end value up front");
991 if (Incoming != WideIV)
1003 auto *Zero = Plan.
getZero(StepTy);
1004 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1009 return B.createNaryOp(
1010 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1012 : Instruction::FAdd,
1013 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1028 const SCEV *Start, *Step;
1039 if (!StartVPV || !StepVPV)
1048 VPValue *ExitCount = Builder.createOverflowingOp(
1051 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1060 VPBuilder VectorPHBuilder(VectorPH, VectorPH->begin());
1070 EndValues[WideIV] = EndValue;
1080 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1081 R.eraseFromParent();
1090 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1092 if (PredVPBB == MiddleVPBB) {
1094 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1097 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1100 Plan, ExitIRI->getOperand(Idx), PSE);
1103 ExitIRI->setOperand(Idx, Escape);
1120 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1124 ExpR->replaceAllUsesWith(V->second);
1128 ExpR->eraseFromParent();
1134 bool CanCreateNewRecipe) {
1135 VPlan *Plan = Def->getParent()->getPlan();
1145 Def->replaceAllUsesWith(
X);
1146 Def->eraseFromParent();
1158 Def->replaceAllUsesWith(
X);
1170 Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1176 Def->replaceAllUsesWith(
X);
1182 Def->replaceAllUsesWith(Plan->
getFalse());
1188 Def->replaceAllUsesWith(
X);
1193 if (CanCreateNewRecipe &&
1198 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1199 !Def->getOperand(1)->hasMoreThanOneUniqueUser())) {
1200 Def->replaceAllUsesWith(
1201 Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z)));
1208 Def->replaceAllUsesWith(Def->getOperand(1));
1215 Def->replaceAllUsesWith(Builder.createLogicalAnd(
X,
Y));
1221 Def->replaceAllUsesWith(Plan->
getFalse());
1226 Def->replaceAllUsesWith(
X);
1232 if (CanCreateNewRecipe &&
1234 Def->replaceAllUsesWith(Builder.createNot(
C));
1240 Def->setOperand(0,
C);
1241 Def->setOperand(1,
Y);
1242 Def->setOperand(2,
X);
1247 if (CanCreateNewRecipe &&
1251 Y->getScalarType()->isIntegerTy(1)) {
1252 Def->replaceAllUsesWith(
1253 Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z)));
1259 if (CanCreateNewRecipe &&
1265 auto *
Select = Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
1266 X,
Y, Def->getDebugLoc());
1267 Def->replaceAllUsesWith(
Select);
1276 VPlan *Plan = Def->getParent()->getPlan();
1282 return Def->replaceAllUsesWith(V);
1288 PredPHI->replaceAllUsesWith(
Op);
1296 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1300 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1301 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1302 RepR->getDebugLoc());
1303 Unmasked->insertBefore(RepR);
1304 RepR->replaceAllUsesWith(Unmasked);
1305 RepR->eraseFromParent();
1319 bool CanCreateNewRecipe =
1324 Type *TruncTy = Def->getScalarType();
1325 Type *ATy =
A->getScalarType();
1326 if (TruncTy == ATy) {
1327 Def->replaceAllUsesWith(
A);
1336 : Instruction::ZExt;
1339 if (
auto *UnderlyingExt = Def->getOperand(0)->getUnderlyingValue()) {
1341 Ext->setUnderlyingValue(UnderlyingExt);
1343 Def->replaceAllUsesWith(Ext);
1345 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1346 Def->replaceAllUsesWith(Trunc);
1356 return Def->replaceAllUsesWith(
A);
1359 return Def->replaceAllUsesWith(
A);
1362 return Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1368 return Def->replaceAllUsesWith(Builder.createSub(
1369 Plan->
getZero(
A->getScalarType()),
A, Def->getDebugLoc(),
"", NW));
1372 if (CanCreateNewRecipe &&
1380 ->hasNoSignedWrap()};
1381 return Def->replaceAllUsesWith(
1382 Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW));
1391 MulR->hasNoSignedWrap() &&
1393 return Def->replaceAllUsesWith(Builder.createNaryOp(
1395 {A, Plan->getConstantInt(APC->getBitWidth(), ShiftAmt)}, NW,
1396 Def->getDebugLoc()));
1401 return Def->replaceAllUsesWith(Builder.createNaryOp(
1403 {A, Plan->getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
1408 return Def->replaceAllUsesWith(
A);
1423 R->setOperand(1,
Y);
1424 R->setOperand(2,
X);
1428 R->replaceAllUsesWith(Cmp);
1433 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1434 Cmp->setDebugLoc(Def->getDebugLoc());
1446 if (
Op->getNumUsers() > 1 ||
1450 }
else if (!UnpairedCmp) {
1451 UnpairedCmp =
Op->getDefiningRecipe();
1455 UnpairedCmp =
nullptr;
1462 if (NewOps.
size() < Def->getNumOperands()) {
1464 return Def->replaceAllUsesWith(NewAnyOf);
1471 if (CanCreateNewRecipe &&
1477 return Def->replaceAllUsesWith(NewCmp);
1483 Def->getOperand(1)->getScalarType() == Def->getScalarType())
1484 return Def->replaceAllUsesWith(Def->getOperand(1));
1488 Type *WideStepTy = Def->getScalarType();
1489 if (
X->getScalarType() != WideStepTy)
1490 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1491 Def->replaceAllUsesWith(
X);
1500 Def->getScalarType()->isIntegerTy(1)) {
1501 Def->setOperand(1, Def->getOperand(0));
1502 Def->setOperand(0,
Y);
1509 return Def->replaceAllUsesWith(Def->getOperand(0));
1515 Def->replaceAllUsesWith(
1516 BuildVector->getOperand(BuildVector->getNumOperands() - 1));
1521 return Def->replaceAllUsesWith(
X);
1524 return Def->replaceAllUsesWith(
A);
1527 return Def->replaceAllUsesWith(
A);
1533 Def->replaceAllUsesWith(
1534 BuildVector->getOperand(BuildVector->getNumOperands() - 2));
1541 Def->replaceAllUsesWith(BuildVector->getOperand(Idx));
1546 Def->replaceAllUsesWith(
1554 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1555 return U.usesFirstLaneOnly(Def);
1564 "broadcast operand must be single-scalar");
1565 Def->setOperand(0,
C);
1570 return Def->replaceUsesWithIf(
1571 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1574 if (Def->getNumOperands() == 1) {
1575 Def->replaceAllUsesWith(Def->getOperand(0));
1580 Phi->replaceAllUsesWith(Phi->getOperand(0));
1586 if (Def->getNumOperands() == 1 &&
1588 return Def->replaceAllUsesWith(IRV);
1601 return Def->replaceAllUsesWith(
A);
1608 return Def->replaceAllUsesWith(WidenIV->getRegion()->getCanonicalIV());
1611 Def->replaceAllUsesWith(Builder.createNaryOp(
1612 Instruction::ExtractElement, {A, LaneToExtract}, Def->getDebugLoc()));
1626 auto *IVInc = Def->getOperand(0);
1627 if (IVInc->getNumUsers() == 2) {
1632 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1633 Def->replaceAllUsesWith(IVInc);
1635 Inc->replaceAllUsesWith(Phi);
1636 Phi->setOperand(0,
Y);
1652 Steps->replaceAllUsesWith(Steps->getOperand(0));
1660 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1662 return PhiR && PhiR->isInLoop();
1668 return Def->replaceAllUsesWith(
A);
1694 R.getVPSingleValue()->replaceAllUsesWith(
X);
1710 while (!Worklist.
empty()) {
1719 R->replaceAllUsesWith(
1720 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1724static std::optional<Instruction::BinaryOps>
1727 case Intrinsic::masked_udiv:
1728 return Instruction::UDiv;
1729 case Intrinsic::masked_sdiv:
1730 return Instruction::SDiv;
1731 case Intrinsic::masked_urem:
1732 return Instruction::URem;
1733 case Intrinsic::masked_srem:
1734 return Instruction::SRem;
1751 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1755 if (RepR && RepR->getOpcode() == Instruction::Store &&
1758 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1759 true ,
nullptr , *RepR ,
1760 *RepR , RepR->getDebugLoc());
1761 Clone->insertBefore(RepOrWidenR);
1763 VPValue *ExtractOp = Clone->getOperand(0);
1769 Clone->setOperand(0, ExtractOp);
1770 RepR->eraseFromParent();
1782 VPValue *SafeDivisor = Builder.createSelect(
1783 IntrR->getOperand(2), IntrR->getOperand(1),
1785 VPValue *Clone = Builder.createNaryOp(
1786 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1789 IntrR->eraseFromParent();
1798 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1807 return !U->usesScalars(
Op);
1811 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1814 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1815 IntroducesBCastOf(Op)))
1819 bool LiveInNeedsBroadcast =
1820 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1821 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1822 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1829 RepOrWidenR->getUnderlyingInstr());
1830 Clone->insertBefore(RepOrWidenR);
1831 RepOrWidenR->replaceAllUsesWith(Clone);
1833 RepOrWidenR->eraseFromParent();
1869 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1870 UniqueValues.
insert(Blend->getIncomingValue(0));
1871 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1873 UniqueValues.
insert(Blend->getIncomingValue(
I));
1875 if (UniqueValues.
size() == 1) {
1876 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1877 Blend->eraseFromParent();
1881 if (Blend->isNormalized())
1887 unsigned StartIndex = 0;
1888 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1900 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1902 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1903 if (
I == StartIndex)
1905 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1906 OperandsWithMask.
push_back(Blend->getMask(
I));
1911 OperandsWithMask, *Blend, Blend->getDebugLoc());
1912 NewBlend->insertBefore(&R);
1914 VPValue *DeadMask = Blend->getMask(StartIndex);
1916 Blend->eraseFromParent();
1921 if (NewBlend->getNumOperands() == 3 &&
1923 VPValue *Inc0 = NewBlend->getOperand(0);
1924 VPValue *Inc1 = NewBlend->getOperand(1);
1925 VPValue *OldMask = NewBlend->getOperand(2);
1926 NewBlend->setOperand(0, Inc1);
1927 NewBlend->setOperand(1, Inc0);
1928 NewBlend->setOperand(2, NewMask);
1955 APInt MaxVal = AlignedTC - 1;
1958 unsigned NewBitWidth =
1964 bool MadeChange =
false;
1989 "canonical IV is not expected to have a truncation");
1994 NewWideIV->insertBefore(WideIV);
2001 Cmp->replaceAllUsesWith(
2002 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2016 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2018 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2032 const SCEV *VectorTripCount =
2037 "Trip count SCEV must be computable");
2058 auto *Term = &ExitingVPBB->
back();
2071 for (
unsigned Part = 0; Part < UF; ++Part) {
2077 Extracts[Part] = Ext;
2089 match(Phi->getBackedgeValue(),
2091 assert(Index &&
"Expected index from ActiveLaneMask instruction");
2108 "Expected one VPActiveLaneMaskPHIRecipe for each unroll part");
2115 "Expected incoming values of Phi to be ActiveLaneMasks");
2120 EntryALM->setOperand(2, ALMMultiplier);
2121 LoopALM->setOperand(2, ALMMultiplier);
2125 ExtractFromALM(EntryALM, EntryExtracts);
2130 ExtractFromALM(LoopALM, LoopExtracts);
2132 Not->setOperand(0, LoopExtracts[0]);
2135 for (
unsigned Part = 0; Part < UF; ++Part) {
2136 Phis[Part]->setStartValue(EntryExtracts[Part]);
2137 Phis[Part]->setBackedgeValue(LoopExtracts[Part]);
2150 auto *Term = &ExitingVPBB->
back();
2162 const SCEV *VectorTripCount =
2168 "Trip count SCEV must be computable");
2187 Term->setOperand(1, Plan.
getTrue());
2192 {}, Term->getDebugLoc());
2194 Term->eraseFromParent();
2202 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2203 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2221 RecurKind RK = PhiR->getRecurrenceKind();
2228 RecWithFlags->dropPoisonGeneratingFlags();
2234struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2243 return GEP->getSourceElementType();
2246 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2247 [](
auto *
I) {
return I->getSourceElementType(); })
2248 .
Default([](
auto *) {
return nullptr; });
2252 static bool canHandle(
const VPSingleDefRecipe *Def) {
2261 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2262 C->second == Instruction::ExtractValue)))
2266 return !
Def->mayReadOrWriteMemory();
2270 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2273 getGEPSourceElementType(Def),
Def->getScalarType(),
2276 if (RFlags->hasPredicate())
2279 return hash_combine(Result, SIVSteps->getInductionOpcode());
2284 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2285 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2288 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2290 !
equal(
L->operands(),
R->operands()))
2294 "must have valid opcode info for both recipes");
2296 if (LFlags->hasPredicate() &&
2297 LFlags->getPredicate() !=
2301 if (LSIV->getInductionOpcode() !=
2311 const VPRegionBlock *RegionL =
L->getRegion();
2312 const VPRegionBlock *RegionR =
R->getRegion();
2315 L->getParent() !=
R->getParent())
2317 return L->getScalarType() ==
R->getScalarType();
2333 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2337 if (!VPDT.
dominates(V->getParent(), VPBB))
2342 Def->replaceAllUsesWith(V);
2355 bool Sinking =
false) {
2384 "Expected vector prehader's successor to be the vector loop region");
2392 return !Op->isDefinedOutsideLoopRegions();
2395 R.moveBefore(*Preheader, Preheader->
end());
2415 assert(!RepR->isPredicated() &&
2416 "Expected prior transformation of predicated replicates to "
2417 "replicate regions");
2422 if (!RepR->isSingleScalar())
2426 if (RepR->getOpcode() == Instruction::Store &&
2427 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2432 assert((!R.mayWriteToMemory() ||
2433 (RepR && RepR->getOpcode() == Instruction::Store &&
2434 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2435 "The only recipes that may write to memory are expected to be "
2436 "stores with invariant pointer-operand");
2446 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2447 auto *UserR = cast<VPRecipeBase>(U);
2448 VPBasicBlock *Parent = UserR->getParent();
2450 if (SinkBB && SinkBB != Parent)
2455 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2456 Parent->getSinglePredecessor() != LoopRegion;
2466 "Defining block must dominate sink block");
2491 VPValue *ResultVPV = R.getVPSingleValue();
2493 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2494 if (!NewResSizeInBits)
2507 (void)OldResSizeInBits;
2515 VPW->dropPoisonGeneratingFlags();
2517 assert((OldResSizeInBits != NewResSizeInBits ||
2519 "Only ICmps should not need extending the result.");
2525 if (OldResSizeInBits != NewResSizeInBits) {
2527 Instruction::ZExt, ResultVPV, OldResTy);
2529 Ext->setOperand(0, ResultVPV);
2539 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2540 if (OpSizeInBits == NewResSizeInBits)
2542 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2543 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2549 Builder.setInsertPoint(&R);
2550 ProcessedIter->second =
2551 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2553 Op = ProcessedIter->second;
2557 NWR->insertBefore(&R);
2561 VPValue *Replacement = NWR->getVPSingleValue();
2562 if (OldResSizeInBits != NewResSizeInBits)
2568 R.eraseFromParent();
2574 std::optional<VPDominatorTree> VPDT;
2582 bool SimplifiedPhi =
false;
2592 assert(VPBB->getNumSuccessors() == 2 &&
2593 "Two successors expected for BranchOnCond");
2594 unsigned RemovedIdx;
2605 "There must be a single edge between VPBB and its successor");
2608 auto Phis = RemovedSucc->
phis();
2611 SimplifiedPhi |= !std::empty(Phis);
2615 VPBB->back().eraseFromParent();
2627 if (Reachable.contains(
B))
2638 for (
VPValue *Def : R.definedValues())
2639 Def->replaceAllUsesWith(&Tmp);
2640 R.eraseFromParent();
2644 return SimplifiedPhi;
2676 "expected to run before loop regions are created");
2678 auto CanUseVersionedStride = [&VPDT, Preheader](
VPUser &U,
unsigned) {
2681 return VPDT.
dominates(Preheader, Parent);
2684 for (
const SCEV *Stride : StridesMap.
values()) {
2687 const APInt *StrideConst;
2710 RewriteMap[StrideV] = PSE.
getSCEV(StrideV);
2717 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2720 if (NewSCEV != ScevExpr) {
2722 ExpSCEV->replaceAllUsesWith(NewExp);
2733 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2738 while (!Worklist.
empty()) {
2741 if (!Visited.
insert(CurRec).second)
2763 RecWithFlags->isDisjoint()) {
2766 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2767 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2768 RecWithFlags->replaceAllUsesWith(New);
2769 RecWithFlags->eraseFromParent();
2772 RecWithFlags->dropPoisonGeneratingFlags();
2777 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2778 "found instruction with poison generating flags not covered by "
2779 "VPRecipeWithIRFlags");
2784 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2795 auto IsNotHeaderMask = [](
VPValue *Mask) {
2808 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2809 if (AddrDef && WidenRec->isConsecutive() &&
2810 IsNotHeaderMask(WidenRec->getMask()))
2811 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2813 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2814 if (AddrDef && IsNotHeaderMask(InterleaveRec->getMask()))
2815 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2825 const bool &EpilogueAllowed) {
2826 if (InterleaveGroups.empty())
2837 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2844 for (
const auto *IG : InterleaveGroups) {
2847 for (
auto *Member : IG->members())
2849 StartMember = Member;
2857 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2863 StoredValues.
push_back(StoreR->getStoredValue());
2870 bool NeedsMaskForGaps =
2871 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2872 (!StoredValues.
empty() && !IG->isFull());
2875 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2879 "Dead member in non-load group?");
2884 InsertPos->getAsRecipe()))
2885 InsertPos = MemberR;
2886 IRInsertPos = &InsertPos->getIngredient();
2896 VPValue *Addr = Start->getAddr();
2898 if (IG->getIndex(StartMember) != 0 ||
2906 assert(IG->getIndex(IRInsertPos) != 0 &&
2907 "index of insert position shouldn't be zero");
2911 IG->getIndex(IRInsertPos),
2915 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2921 if (IG->isReverse()) {
2924 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2925 ReversePtr->insertBefore(InsertPosR);
2929 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2931 VPIG->insertBefore(InsertPosR);
2934 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2937 if (!Member->getType()->isVoidTy()) {
2955static std::optional<VPValue *>
3008 VPValue *UncountableCondition =
nullptr;
3012 return std::nullopt;
3015 Worklist.
push_back(UncountableCondition);
3016 while (!Worklist.
empty()) {
3020 if (V->isDefinedOutsideLoopRegions())
3026 if (V->getNumUsers() > 1)
3027 return std::nullopt;
3039 return std::nullopt;
3043 return std::nullopt;
3051 return std::nullopt;
3059 return std::nullopt;
3061 return UncountableCondition;
3117 for (
auto &Exit : Exits) {
3118 if (Exit.EarlyExitingVPBB == LatchVPBB)
3122 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3123 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3134 std::optional<VPValue *>
Cond =
3150 assert(
Load &&
"Couldn't find exactly one load");
3153 "Uncountable exit condition load is conditional.");
3167 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3191 while (InsertIt != HeaderVPBB->
end() &&
3193 erase(ConditionRecipes, &*InsertIt);
3196 for (
auto *Recipe :
reverse(ConditionRecipes))
3197 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3201 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3203 Type *IVScalarTy =
IV->getScalarType();
3209 {Zero, FirstActive, ALMMultiplier},
3210 DebugLoc(),
"uncountable.exit.mask");
3215 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3217 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3227 "Expected BranchOnCond terminator for MiddleVPBB");
3238 auto Phis = ScalarPH->
phis();
3248 "Continuing from different IV");
3264 if (Pred == MiddleVPBB)
3269 VPValue *CondOfEarlyExitingVPBB;
3270 [[maybe_unused]]
bool Matched =
3271 match(EarlyExitingVPBB->getTerminator(),
3273 assert(Matched &&
"Terminator must be BranchOnCond");
3277 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3278 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3280 TrueSucc == ExitBlock
3281 ? CondOfEarlyExitingVPBB
3282 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3288 "exit condition must dominate the latch");
3297 assert(!Exits.
empty() &&
"must have at least one early exit");
3304 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3307 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3313 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3314 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3316 Exits[
I].EarlyExitingVPBB) &&
3317 "RPO sort must place dominating exits before dominated ones");
3323 VPValue *Combined = Exits[0].CondToExit;
3336 "Unexpected terminator");
3337 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3338 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3339 LatchExitingBranch->eraseFromParent();
3342 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3348 LatchVPBB->
setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3352 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3357 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3361 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3369 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3372 LatchVPBB->
setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3404 for (
auto [Exit, VectorEarlyExitVPBB] :
3405 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3406 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3418 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3419 VPValue *NewIncoming = IncomingVal;
3421 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3426 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3427 ExitIRI->addIncoming(NewIncoming);
3430 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3464 bool IsLastDispatch = (
I + 2 == Exits.
size());
3466 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3472 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3475 CurrentBB = FalseBB;
3490 VPValue *VecOp = Red->getVecOp();
3492 assert(!Red->isPartialReduction() &&
3493 "This path does not support partial reductions");
3496 auto IsExtendedRedValidAndClampRange =
3509 "getExtendedReductionCost only supports integer types");
3510 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3511 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3512 Red->getFastMathFlagsOrNone(),
CostKind);
3513 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3521 IsExtendedRedValidAndClampRange(
3542 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3543 Opcode != Instruction::FAdd)
3546 assert(!Red->isPartialReduction() &&
3547 "This path does not support partial reductions");
3551 auto IsMulAccValidAndClampRange =
3563 (Ext0->getOpcode() != Ext1->getOpcode() ||
3564 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3568 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3570 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3577 ExtCost += Ext0->computeCost(VF, Ctx);
3579 ExtCost += Ext1->computeCost(VF, Ctx);
3581 ExtCost += OuterExt->computeCost(VF, Ctx);
3583 return MulAccCost.
isValid() &&
3584 MulAccCost < ExtCost + MulCost + RedCost;
3589 VPValue *VecOp = Red->getVecOp();
3627 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3629 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3630 Mul->setOperand(1, ExtB);
3640 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3645 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3652 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3669 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3678 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3679 Ext0->getOpcode() == Ext1->getOpcode() &&
3680 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3682 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3683 *Ext0, *Ext0, Ext0->getDebugLoc());
3684 NewExt0->insertBefore(Ext0);
3689 Ext->getScalarType(),
nullptr, *Ext1,
3690 *Ext1, Ext1->getDebugLoc());
3693 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3694 NewMul->insertBefore(
Mul);
3695 Ext->replaceAllUsesWith(NewMul);
3696 Ext->eraseFromParent();
3697 Mul->eraseFromParent();
3711 assert(!Red->isPartialReduction() &&
3712 "This path does not support partial reductions");
3715 auto IP = std::next(Red->getIterator());
3716 auto *VPBB = Red->getParent();
3726 Red->replaceAllUsesWith(AbstractR);
3746 return CommonMetadata;
3749template <
unsigned Opcode>
3754 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3755 "Only Load and Store opcodes supported");
3756 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3763 for (
auto Recipes :
Groups) {
3764 if (Recipes.size() < 2)
3769 "Expected all recipes in group to have the same load-store type");
3776 VPValue *MaskI = RecipeI->getMask();
3782 bool HasComplementaryMask =
false;
3787 VPValue *MaskJ = RecipeJ->getMask();
3796 if (HasComplementaryMask) {
3797 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3807template <
typename InstType>
3825 for (
auto &Group :
Groups) {
3845 return R->isSingleScalar() == IsSingleScalar;
3847 "all members in group must agree on IsSingleScalar");
3852 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3853 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3855 UnpredicatedLoad->insertBefore(EarliestLoad);
3859 Load->replaceAllUsesWith(UnpredicatedLoad);
3860 Load->eraseFromParent();
3869 if (!StoreLoc || !StoreLoc->AATags.Scope)
3876 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3888 for (
auto &Group :
Groups) {
3901 VPValue *SelectedValue = Group[0]->getOperand(0);
3904 bool IsSingleScalar = Group[0]->isSingleScalar();
3905 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3906 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3907 "all members in group must agree on IsSingleScalar");
3908 VPValue *Mask = Group[
I]->getMask();
3910 SelectedValue = Builder.createSelect(Mask,
Value, SelectedValue,
3919 StoreWithMinAlign->getUnderlyingInstr(),
3920 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3921 nullptr, *LastStore, CommonMetadata);
3922 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3926 Store->eraseFromParent();
3941 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
3946 if (Member0Op == OpV)
3956 return !IsScalable && !W->getMask() && W->isConsecutive() &&
3959 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
3974 if (R->getScalarType() != WideMember0->getScalarType())
3976 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
3980 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
3983 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
3988 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
3989 const auto &[
OpIdx, OpV] =
P;
4001static std::optional<ElementCount>
4005 if (!InterleaveR || InterleaveR->
getMask())
4006 return std::nullopt;
4008 Type *GroupElementTy =
nullptr;
4012 return Op->getScalarType() == GroupElementTy;
4014 return std::nullopt;
4018 return Op->getScalarType() == GroupElementTy;
4020 return std::nullopt;
4024 if (IG->getFactor() != IG->getNumMembers())
4025 return std::nullopt;
4031 assert(
Size.isScalable() == VF.isScalable() &&
4032 "if Size is scalable, VF must be scalable and vice versa");
4033 return Size.getKnownMinValue();
4037 unsigned MinVal = VF.getKnownMinValue();
4039 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4042 return std::nullopt;
4050 return RepR && RepR->isSingleScalar();
4064 if (V->isDefinedOutsideLoopRegions()) {
4067 return M->isDefinedOutsideLoopRegions() &&
4068 M->getScalarType() == V->getScalarType();
4070 "expected distinct loop-invariant values of matching scalar type");
4085 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4087 for (
VPValue *Member : Members)
4088 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4089 WideMember0->setOperand(
4098 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4100 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4101 *LoadGroup, LoadGroup->getDebugLoc());
4107 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4108 "must be a single scalar load");
4109 NarrowedOps.
insert(RepR);
4114 VPValue *PtrOp = WideLoad->getAddr();
4116 PtrOp = VecPtr->getOperand(0);
4121 nullptr, {}, *WideLoad);
4122 N->insertBefore(WideLoad);
4127std::unique_ptr<VPlan>
4147 "unexpected branch-on-count");
4150 std::optional<ElementCount> VFToOptimize;
4164 if (R.mayWriteToMemory() && !InterleaveR)
4170 return any_of(V->users(), [&](VPUser *U) {
4171 auto *UR = cast<VPRecipeBase>(U);
4172 return UR->getParent()->getParent() != VectorLoop;
4189 std::optional<ElementCount> NarrowedVF =
4191 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4193 VFToOptimize = NarrowedVF;
4196 if (InterleaveR->getStoredValues().empty())
4201 auto *Member0 = InterleaveR->getStoredValues()[0];
4211 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4214 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4215 return IR && IR->getInterleaveGroup()->isFull() &&
4216 IR->getVPValue(Op.index()) == Op.value();
4225 VFToOptimize->isScalable()))
4230 if (StoreGroups.empty())
4234 bool RequiresScalarEpilogue =
4245 std::unique_ptr<VPlan> NewPlan;
4247 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4248 Plan.
setVF(*VFToOptimize);
4249 NewPlan->removeVF(*VFToOptimize);
4256 for (
auto *StoreGroup : StoreGroups) {
4258 NarrowedOps, Preheader);
4264 StoreGroup->getDebugLoc());
4271 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4277 if (VFToOptimize->isScalable()) {
4280 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4288 materializeVectorTripCount(Plan, VectorPH,
false,
4289 RequiresScalarEpilogue, Step);
4294 removeDeadRecipes(Plan);
4297 "All VPVectorPointerRecipes should have been removed");
4317 "Cannot handle loops with uncountable early exits");
4324 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4331 if (
any_of(RecurSplice->users(),
4332 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4413 {},
"vector.recur.extract.for.phi");
4416 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4430 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4431 VPValue *InvariantCandidate = BinOp->getOperand(1);
4433 std::swap(WidenIVCandidate, InvariantCandidate);
4447 auto *ClonedOp = BinOp->
clone();
4448 if (ClonedOp->getOperand(0) == WidenIV) {
4449 ClonedOp->setOperand(0, ScalarIV);
4451 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4452 ClonedOp->setOperand(1, ScalarIV);
4467 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4468 bool UseMax) -> std::optional<APSInt> {
4470 for (
bool Signed : {
true,
false}) {
4479 return std::nullopt;
4487 PhiR->getRecurrenceKind()))
4496 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4510 !
match(FindLastSelect,
4519 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4525 "IVOfExpressionToSink not being an AddRec must imply "
4526 "FindLastExpression not being an AddRec.");
4537 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4538 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4545 if (IVOfExpressionToSink) {
4546 const SCEV *FindLastExpressionSCEV =
4548 if (
match(FindLastExpressionSCEV,
4551 if (
auto NewSentinel =
4552 CheckSentinel(FindLastExpressionSCEV, NewUseMax)) {
4555 SentinelVal = *NewSentinel;
4556 UseSigned = NewSentinel->isSigned();
4558 IVSCEV = FindLastExpressionSCEV;
4559 IVOfExpressionToSink =
nullptr;
4569 if (AR->hasNoSignedWrap())
4571 else if (AR->hasNoUnsignedWrap())
4581 VPValue *NewFindLastSelect = BackedgeVal;
4583 if (!SentinelVal || IVOfExpressionToSink) {
4586 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4587 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4588 if (FindLastSelect->getDefiningRecipe()->getOperand(1) == PhiR)
4589 SelectCond = LoopBuilder.
createNot(SelectCond);
4596 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4599 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4608 VPIRFlags Flags(MinMaxKind,
false,
false,
4614 NewFindLastSelect, Flags, ExitDL);
4617 VPValue *VectorRegionExitingVal = ReducedIV;
4618 if (IVOfExpressionToSink)
4619 VectorRegionExitingVal =
4621 ReducedIV, IVOfExpressionToSink);
4624 VPValue *StartVPV = PhiR->getStartValue();
4631 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4641 AnyOfPhi->insertAfter(PhiR);
4648 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4661 PhiR->hasUsesOutsideReductionChain());
4662 NewPhiR->insertBefore(PhiR);
4663 PhiR->replaceAllUsesWith(NewPhiR);
4664 PhiR->eraseFromParent();
4671struct ReductionExtend {
4672 Type *SrcType =
nullptr;
4673 ExtendKind Kind = ExtendKind::PR_None;
4679struct ExtendedReductionOperand {
4683 ReductionExtend ExtendA, ExtendB;
4691struct VPPartialReductionChain {
4694 VPWidenRecipe *ReductionBinOp =
nullptr;
4696 ExtendedReductionOperand ExtendedOp;
4703 unsigned AccumulatorOpIdx;
4704 unsigned ScaleFactor;
4707 VPBlendRecipe *Blend =
nullptr;
4712static std::optional<unsigned>
4716 "Expected a non-normalized blend with two incoming values");
4722 return std::nullopt;
4723 return FirstIncomingHasOneUse ? 0 : 1;
4735 if (!
Op->hasOneUse() ||
4741 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4742 Op->getOperand(1), NarrowTy);
4744 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4753 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4755 assert(Ext->getOpcode() ==
4757 "Expected both the LHS and RHS extends to be the same");
4758 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4761 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4762 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4763 auto *
Max = Builder.insert(
4765 {FreezeX, FreezeY}, SrcTy));
4766 auto *Min = Builder.insert(
4768 {FreezeX, FreezeY}, SrcTy));
4771 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4772 Op->getScalarType());
4784 if (!
Mul->hasOneUse() ||
4785 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4786 MulLHS->getOpcode() != MulRHS->getOpcode())
4789 auto *NewLHS = Builder.createWidenCast(
4790 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4791 auto *NewRHS = MulLHS == MulRHS
4793 : Builder.createWidenCast(MulRHS->getOpcode(),
4794 MulRHS->getOperand(0),
4795 Ext->getScalarType());
4796 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4797 Builder.insert(NewMul);
4798 Op->replaceAllUsesWith(NewMul);
4799 Op->eraseFromParent();
4800 Mul->eraseFromParent();
4809 VPValue *VecOp = Red->getVecOp();
4863static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4871 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4874 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4890 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4892 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4897 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4907 Builder.insert(NegRecipe);
4908 ExtendedOp = NegRecipe;
4923 std::optional<unsigned> BlendReductionIdx =
4924 getBlendReductionUpdateValueIdx(Chain.Blend);
4925 assert(BlendReductionIdx &&
4927 "Expected blend to contain the reduction update");
4938 assert((!ExitValue || IsLastInChain) &&
4939 "if we found ExitValue, it must match RdxPhi's backedge value");
4950 PartialRed->insertBefore(WidenRecipe);
4960 E->insertBefore(WidenRecipe);
4961 PartialRed->replaceAllUsesWith(
E);
4974 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
4975 StartInst->setOperand(2, NewScaleFactor);
4983 VPValue *OldStartValue = StartInst->getOperand(0);
4984 StartInst->setOperand(0, StartInst->getOperand(1));
4988 assert(RdxResult &&
"Could not find reduction result");
4991 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
4992 : Instruction::BinaryOps::Sub;
4998 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
5004 const VPPartialReductionChain &Link,
5007 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5008 std::optional<unsigned> BinOpc = std::nullopt;
5010 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5011 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5013 std::optional<llvm::FastMathFlags>
Flags;
5017 auto GetLinkOpcode = [&Link]() ->
unsigned {
5020 return Instruction::Add;
5022 return Instruction::FAdd;
5024 return Link.ReductionBinOp->
getOpcode();
5029 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5030 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5051static std::optional<ExtendedReductionOperand>
5054 "Op should be operand of UpdateR");
5062 if (
Op->hasOneUse() &&
5071 Type *RHSInputType =
Y->getScalarType();
5072 if (LHSInputType != RHSInputType ||
5073 LHSExt->getOpcode() != RHSExt->getOpcode())
5074 return std::nullopt;
5077 return ExtendedReductionOperand{
5079 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5083 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5086 VPValue *CastSource = CastRecipe->getOperand(0);
5087 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5097 return ExtendedReductionOperand{
5104 if (!
Op->hasOneUse())
5105 return std::nullopt;
5110 return std::nullopt;
5120 return std::nullopt;
5124 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5127 const APInt *RHSConst =
nullptr;
5133 return std::nullopt;
5137 if (Cast && OuterExtKind &&
5138 getPartialReductionExtendKind(Cast) != OuterExtKind)
5139 return std::nullopt;
5141 Type *RHSInputType = LHSInputType;
5142 ExtendKind RHSExtendKind = LHSExtendKind;
5145 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5148 return ExtendedReductionOperand{
5149 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5156static std::optional<SmallVector<VPPartialReductionChain>>
5163 return std::nullopt;
5173 VPValue *CurrentValue = ExitValue;
5174 while (CurrentValue != RedPhiR) {
5176 std::optional<unsigned> BlendReductionIdx;
5180 return std::nullopt;
5182 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5183 if (!BlendReductionIdx)
5184 return std::nullopt;
5191 return std::nullopt;
5198 std::optional<ExtendedReductionOperand> ExtendedOp =
5199 matchExtendedReductionOperand(UpdateR,
Op);
5201 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5203 return std::nullopt;
5211 return std::nullopt;
5213 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5216 return std::nullopt;
5218 VPPartialReductionChain Link(
5219 {UpdateR, *ExtendedOp, RK,
5224 CurrentValue = PrevValue;
5229 std::reverse(Chain.
begin(), Chain.
end());
5248 if (
auto Chains = getScaledReductions(RedPhiR))
5249 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5252 if (ChainsByPhi.
empty())
5260 for (
const auto &[
_, Chains] : ChainsByPhi)
5261 for (
const VPPartialReductionChain &Chain : Chains) {
5262 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5264 PartialReductionBlends.
insert(Chain.Blend);
5265 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5271 auto ExtendUsersValid = [&](
VPValue *Ext) {
5273 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5277 auto IsProfitablePartialReductionChainForVF =
5284 for (
const VPPartialReductionChain &Link : Chain) {
5285 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5286 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5290 PartialCost += LinkCost;
5291 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5293 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5294 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5297 RegularCost += Extend->computeCost(VF, CostCtx);
5299 return PartialCost.
isValid() && PartialCost < RegularCost;
5307 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5308 for (
const VPPartialReductionChain &Chain : Chains) {
5309 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5313 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5315 return PhiR == RedPhiR;
5319 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5321 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5327 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5336 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5337 return RepR && RepR->getOpcode() == Instruction::Store;
5348 return IsProfitablePartialReductionChainForVF(Chains, VF);
5354 for (
auto &[Phi, Chains] : ChainsByPhi)
5355 for (
const VPPartialReductionChain &Chain : Chains)
5356 transformToPartialReduction(Chain, Plan, Phi);
5371 if (VPI && VPI->getUnderlyingValue() &&
5382 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5385 if (!ProcessVPInst(VPI))
5394 assert(New->getParent() &&
"New recipe must have been inserted");
5395 if (VPI->
getOpcode() == Instruction::Load)
5404 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5411 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5413 VPI, FinalRedStoresBuilder))
5422 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5435 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5439 return Scalarize(VPI);
5446 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5448 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5458 const SCEV *PtrSCEV =
5460 bool IsSingleScalarLoad =
5466 I, Ptr, IsSingleScalarLoad,
5475 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5477 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5481 std::optional<int64_t> Stride =
5483 if (Stride != 1 && Stride != -1)
5514 return ReplaceWith(VPI,
Load);
5523 auto *StoreR = Builder.createWidenStore(
5526 return ReplaceWith(VPI, StoreR);
5533 return ReplaceWith(VPI, Recipe);
5535 return Scalarize(VPI);
5558 if (VPI->mayHaveSideEffects())
5562 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5567 if (VPI->getOpcode() == Instruction::Add &&
5576 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5577 *VPI, VPI->getDebugLoc(),
I);
5578 Recipe->insertBefore(VPI);
5579 VPI->replaceAllUsesWith(Recipe);
5580 VPI->eraseFromParent();
5590 switch (Param.ParamKind) {
5591 case VFParamKind::Vector:
5592 case VFParamKind::GlobalPredicate:
5594 case VFParamKind::OMP_Uniform:
5595 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5596 SE->isLoopInvariant(
5597 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5599 case VFParamKind::OMP_Linear:
5600 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5601 m_scev_AffineAddRec(
5602 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5603 m_SpecificLoop(L)));
5620 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5621 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5624 if (It == Mappings.end())
5631struct CallWideningDecision {
5632 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5633 CallWideningDecision(KindTy Kind, Function *Variant =
nullptr)
5656 return CallWideningDecision::KindTy::Scalarize;
5666 return CallWideningDecision::KindTy::Scalarize;
5670 false, VF, CostCtx);
5685 return CallWideningDecision::KindTy::Intrinsic;
5689 if (VecFunc && ScalarCost >= VecCallCost)
5690 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5692 return CallWideningDecision::KindTy::Scalarize;
5702 if (!VPI || !VPI->getUnderlyingValue() ||
5703 VPI->getOpcode() != Instruction::Call)
5708 VPI->op_begin() + CI->arg_size());
5710 CallWideningDecision Decision =
5719 switch (Decision.Kind) {
5720 case CallWideningDecision::KindTy::Intrinsic: {
5724 *VPI, VPI->getDebugLoc());
5727 case CallWideningDecision::KindTy::VectorVariant: {
5731 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5732 Ops.push_back(Mask);
5734 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5736 *VPI, VPI->getDebugLoc());
5739 case CallWideningDecision::KindTy::Scalarize:
5745 VPI->replaceAllUsesWith(Replacement);
5746 VPI->eraseFromParent();
5769 if (!LoadR || LoadR->isConsecutive())
5772 VPValue *Ptr = LoadR->getAddr();
5785 Align Alignment = LoadR->getAlign();
5788 if (!Ctx.TTI.isLegalStridedLoadStore(DataTy, Alignment))
5793 Intrinsic::experimental_vp_strided_load, DataTy,
5794 LoadR->isMasked(), Alignment, Ctx);
5795 return StridedLoadStoreCost < CurrentCost;
5806 Ctx.invalidateWideningDecision(&LoadR->getIngredient(), VF);
5811 I32VF = Builder.createScalarZExtOrTrunc(
5828 "Stride type from SCEV must match the index type");
5829 VPValue *CanIV = Builder.createScalarSExtOrTrunc(
5832 auto *
Offset = Builder.createOverflowingOp(
5833 Instruction::Mul, {CanIV, StrideInBytes},
5834 {AddRecPtr->hasNoUnsignedWrap(),
false});
5838 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5841 VPValue *NewPtr = Builder.createVectorPointer(
5843 LoadR->getDebugLoc());
5845 VPValue *Mask = LoadR->getMask();
5848 auto *StridedLoad = Builder.createWidenMemIntrinsic(
5849 Intrinsic::experimental_vp_strided_load,
5850 {NewPtr, StrideInBytes, Mask, I32VF}, LoadTy, Alignment, *LoadR,
5851 LoadR->getDebugLoc());
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
MachineInstr unsigned OpIdx
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
constexpr bool isVector() const
One or more elements.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This class represents an analyzed expression in the program.
LLVM_ABI Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNonZero(const SCEV *S)
Test if the given expression is known to be non-zero.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
void clearSuccessors()
Remove all the successors of this block.
VPBlockBase * getSinglePredecessor() const
void clearPredecessors()
Remove all the predecessor of this block.
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", const VPIRFlags &Flags={})
VPExpandSCEVRecipe * createExpandSCEV(const SCEV *Expr)
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
void setInsertPoint(VPBasicBlock *TheBB)
This specifies that created VPInstructions should be appended to the end of the specified block.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode)
Returns default flags for Opcode for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * tryToExpand(const SCEV *S)
Try to expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
unsigned getNumUsers() const
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPIRValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
@ C
The default llvm calling convention, compatible with C.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
VPInstruction_match< VPInstruction::ActiveLaneMask, Op0_t, Op1_t, Op2_t > m_ActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPIRValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
LLVM_ABI_FOR_TEST cl::opt< bool > EnableWideActiveLaneMask
UncountableExitStyle
Different methods of handling early exits.
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...