53 "should not try to widen irregular types");
68 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
77 if (!VPBB->getParent())
80 auto EndIter = Term ? Term->getIterator() : VPBB->end();
85 VPValue *VPV = Ingredient.getVPSingleValue();
106 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
108 nullptr , IsConsecutive,
109 *VPI, Ingredient.getDebugLoc());
111 bool IsConsecutive = IsConsecutiveAccess(
112 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
114 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
115 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
118 Ingredient.operands(), *VPI,
119 Ingredient.getDebugLoc(),
GEP);
131 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
136 if (VectorID == Intrinsic::assume ||
137 VectorID == Intrinsic::lifetime_end ||
138 VectorID == Intrinsic::lifetime_start ||
139 VectorID == Intrinsic::sideeffect ||
140 VectorID == Intrinsic::pseudoprobe) {
145 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
146 VectorID != Intrinsic::pseudoprobe;
150 Ingredient.getDebugLoc());
153 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
154 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
158 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
162 *VPI, Ingredient.getDebugLoc());
166 "inductions must be created earlier");
175 "Only recpies with zero or one defined values expected");
176 Ingredient.eraseFromParent();
187 const Loop *L =
nullptr;
192 if (
A->getOpcode() != Instruction::Store ||
193 B->getOpcode() != Instruction::Store)
206 const APInt *Distance;
212 Type *TyA =
A->getOperand(0)->getScalarType();
214 Type *TyB =
B->getOperand(0)->getScalarType();
220 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
222 auto VFs =
B->getParent()->getPlan()->vectorFactors();
226 return Distance->
abs().
uge(
234 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
235 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
244 return ExcludeRecipes.contains(
Store) ||
245 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
258 std::optional<SinkStoreInfo> SinkInfo = {}) {
259 bool CheckReads = SinkInfo.has_value();
263 if (SinkInfo && SinkInfo->shouldSkip(R))
267 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
292template <
unsigned Opcode>
297 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
298 "Only Load and Store opcodes supported");
299 constexpr bool IsLoad = (Opcode == Instruction::Load);
302 RecipesByAddressAndType;
307 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
311 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
315 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
320 for (
auto &Group :
Groups) {
335 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
347 if (Candidate->getParent() == SinkTo ||
352 if (!ScalarVFOnly && RepR->isSingleScalar())
355 WorkList.
insert({SinkTo, Candidate});
367 for (
auto &Recipe : *VPBB)
369 InsertIfValidSinkCandidate(VPBB,
Op);
373 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
376 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
381 auto UsersOutsideSinkTo =
383 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
385 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
386 return !U->usesFirstLaneOnly(SinkCandidate);
389 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
391 if (NeedsDuplicating) {
395 if (
auto *SinkCandidateRepR =
400 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
401 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
405 Clone = SinkCandidate->
clone();
415 InsertIfValidSinkCandidate(SinkTo,
Op);
424 if (EntryBB->getNumSuccessors() != 2)
429 if (!Succ0 || !Succ1)
432 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
434 if (Succ0->getSingleSuccessor() == Succ1)
436 if (Succ1->getSingleSuccessor() == Succ0)
453 if (!Region1->isReplicator())
455 auto *MiddleBasicBlock =
457 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
462 if (!Region2 || !Region2->isReplicator())
465 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
466 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
467 if (!Mask1 || Mask1 != Mask2)
470 assert(Mask1 && Mask2 &&
"both region must have conditions");
476 if (TransformedRegions.
contains(Region1))
483 if (!Then1 || !Then2)
503 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
509 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
510 Phi1ToMove.eraseFromParent();
513 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
527 TransformedRegions.
insert(Region1);
530 return !TransformedRegions.
empty();
538 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
539 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
540 auto *BlockInMask = PredRecipe->
getMask();
561 Region->setParent(ParentRegion);
567 RecipeWithoutMask->getDebugLoc());
568 Exiting->appendRecipe(PHIRecipe);
581 if (RepR->isPredicated())
600 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
612 if (!VPBB->getParent())
616 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
625 R.moveBefore(*PredVPBB, PredVPBB->
end());
627 auto *ParentRegion = VPBB->getParent();
628 if (ParentRegion && ParentRegion->getExiting() == VPBB)
629 ParentRegion->setExiting(PredVPBB);
633 return !WorkList.
empty();
640 bool ShouldSimplify =
true;
641 while (ShouldSimplify) {
657 if (!
IV ||
IV->getTruncInst())
672 for (
auto *U : FindMyCast->
users()) {
674 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
675 FoundUserCast = UserCast;
682 FindMyCast = FoundUserCast;
684 if (FindMyCast !=
IV)
706 VPUser *PhiUser = PhiR->getSingleUser();
712 PhiR->replaceAllUsesWith(Start);
713 PhiR->eraseFromParent();
750 Def->user_empty() || !Def->getUnderlyingValue() ||
751 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
764 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
766 Def->getUnderlyingInstr());
767 Clone->insertAfter(Def);
768 Def->replaceAllUsesWith(Clone);
780 PtrIV->replaceAllUsesWith(PtrAdd);
787 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
788 return U->usesScalars(WideIV);
797 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
800 Plan, ID.getKind(), ID.getInductionOpcode(),
802 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
803 WideIV->getDebugLoc(), Builder, WrapFlags);
806 if (!HasOnlyVectorVFs) {
808 "plans containing a scalar VF cannot also include scalable VFs");
809 WideIV->replaceAllUsesWith(Steps);
812 WideIV->replaceUsesWithIf(Steps,
813 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
815 return U.usesFirstLaneOnly(WideIV);
816 return U.usesScalars(WideIV);
832 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
837 if (!Def || Def->getNumOperands() != 2)
845 auto IsWideIVInc = [&]() {
846 auto &ID = WideIV->getInductionDescriptor();
849 VPValue *IVStep = WideIV->getStepValue();
850 switch (ID.getInductionOpcode()) {
851 case Instruction::Add:
853 case Instruction::FAdd:
855 case Instruction::FSub:
858 case Instruction::Sub: {
878 return IsWideIVInc() ? WideIV :
nullptr;
902 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
904 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
905 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
910 if (Incoming != WideIV) {
912 EndValue =
B.createAdd(EndValue, One,
DL);
917 VPIRValue *Start = WideIV->getStartValue();
918 VPValue *Step = WideIV->getStepValue();
919 EndValue =
B.createDerivedIV(
921 Start, EndValue, Step);
935 if (WideIntOrFp && WideIntOrFp->getTruncInst())
945 Start, VectorTC, Step);
977 assert(EndValue &&
"Must have computed the end value up front");
982 if (Incoming != WideIV)
994 auto *Zero = Plan.
getZero(StepTy);
995 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1000 return B.createNaryOp(
1001 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1003 : Instruction::FAdd,
1004 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1019 const SCEV *Start, *Step;
1030 if (!StartVPV || !StepVPV)
1039 VPValue *ExitCount = Builder.createOverflowingOp(
1042 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1051 VPBuilder VectorPHBuilder(VectorPH, VectorPH->begin());
1061 EndValues[WideIV] = EndValue;
1071 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1072 R.eraseFromParent();
1081 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1083 if (PredVPBB == MiddleVPBB) {
1085 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1088 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1091 Plan, ExitIRI->getOperand(Idx), PSE);
1094 ExitIRI->setOperand(Idx, Escape);
1111 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1115 ExpR->replaceAllUsesWith(V->second);
1119 ExpR->eraseFromParent();
1125 bool CanCreateNewRecipe) {
1126 VPlan *Plan = Def->getParent()->getPlan();
1136 Def->replaceAllUsesWith(
X);
1137 Def->eraseFromParent();
1149 Def->replaceAllUsesWith(
X);
1161 Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1167 Def->replaceAllUsesWith(
X);
1173 Def->replaceAllUsesWith(Plan->
getFalse());
1179 Def->replaceAllUsesWith(
X);
1184 if (CanCreateNewRecipe &&
1189 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1190 !Def->getOperand(1)->hasMoreThanOneUniqueUser())) {
1191 Def->replaceAllUsesWith(
1192 Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z)));
1199 Def->replaceAllUsesWith(Def->getOperand(1));
1206 Def->replaceAllUsesWith(Builder.createLogicalAnd(
X,
Y));
1212 Def->replaceAllUsesWith(Plan->
getFalse());
1217 Def->replaceAllUsesWith(
X);
1223 if (CanCreateNewRecipe &&
1225 Def->replaceAllUsesWith(Builder.createNot(
C));
1231 Def->setOperand(0,
C);
1232 Def->setOperand(1,
Y);
1233 Def->setOperand(2,
X);
1238 if (CanCreateNewRecipe &&
1242 Y->getScalarType()->isIntegerTy(1)) {
1243 Def->replaceAllUsesWith(
1244 Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z)));
1250 if (CanCreateNewRecipe &&
1256 auto *
Select = Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
1257 X,
Y, Def->getDebugLoc());
1258 Def->replaceAllUsesWith(
Select);
1267 VPlan *Plan = Def->getParent()->getPlan();
1273 return Def->replaceAllUsesWith(V);
1279 PredPHI->replaceAllUsesWith(
Op);
1287 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1291 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1292 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1293 RepR->getDebugLoc());
1294 Unmasked->insertBefore(RepR);
1295 RepR->replaceAllUsesWith(Unmasked);
1296 RepR->eraseFromParent();
1310 bool CanCreateNewRecipe =
1315 Type *TruncTy = Def->getScalarType();
1316 Type *ATy =
A->getScalarType();
1317 if (TruncTy == ATy) {
1318 Def->replaceAllUsesWith(
A);
1326 : Instruction::ZExt;
1329 if (
auto *UnderlyingExt = Z->getUnderlyingValue()) {
1331 Ext->setUnderlyingValue(UnderlyingExt);
1333 Def->replaceAllUsesWith(Ext);
1335 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1336 Def->replaceAllUsesWith(Trunc);
1346 return Def->replaceAllUsesWith(
A);
1349 return Def->replaceAllUsesWith(
A);
1352 return Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1358 return Def->replaceAllUsesWith(Builder.createSub(
1359 Plan->
getZero(
A->getScalarType()),
A, Def->getDebugLoc(),
"", NW));
1362 if (CanCreateNewRecipe &&
1370 return Def->replaceAllUsesWith(
1371 Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW));
1377 return Def->replaceAllUsesWith(Builder.createAnd(
1386 MulR->hasNoSignedWrap() &&
1388 return Def->replaceAllUsesWith(Builder.createNaryOp(
1390 {A, Plan->getConstantInt(APC->getBitWidth(), ShiftAmt)}, NW,
1391 Def->getDebugLoc()));
1396 return Def->replaceAllUsesWith(Builder.createNaryOp(
1398 {A, Plan->getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
1403 return Def->replaceAllUsesWith(
A);
1418 R->setOperand(1,
Y);
1419 R->setOperand(2,
X);
1423 R->replaceAllUsesWith(Cmp);
1428 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1429 Cmp->setDebugLoc(Def->getDebugLoc());
1441 if (
Op->getNumUsers() > 1 ||
1445 }
else if (!UnpairedCmp) {
1446 UnpairedCmp =
Op->getDefiningRecipe();
1450 UnpairedCmp =
nullptr;
1457 if (NewOps.
size() < Def->getNumOperands()) {
1459 return Def->replaceAllUsesWith(NewAnyOf);
1466 if (CanCreateNewRecipe &&
1472 return Def->replaceAllUsesWith(NewCmp);
1479 A->getScalarType() == Def->getScalarType())
1480 return Def->replaceAllUsesWith(
A);
1484 Type *WideStepTy = Def->getScalarType();
1485 if (
X->getScalarType() != WideStepTy)
1486 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1487 Def->replaceAllUsesWith(
X);
1496 Def->getScalarType()->isIntegerTy(1)) {
1497 Def->setOperand(1, Plan->
getTrue());
1498 Def->setOperand(0,
Y);
1505 return Def->replaceAllUsesWith(Def->getOperand(0));
1511 Def->replaceAllUsesWith(
1512 BuildVector->getOperand(BuildVector->getNumOperands() - 1));
1517 return Def->replaceAllUsesWith(
X);
1520 return Def->replaceAllUsesWith(
A);
1523 return Def->replaceAllUsesWith(
A);
1529 Def->replaceAllUsesWith(
1530 BuildVector->getOperand(BuildVector->getNumOperands() - 2));
1537 Def->replaceAllUsesWith(BuildVector->getOperand(Idx));
1542 Def->replaceAllUsesWith(
1550 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1551 return U.usesFirstLaneOnly(Def);
1560 "broadcast operand must be single-scalar");
1561 Def->setOperand(0, Z);
1566 return Def->replaceUsesWithIf(
1567 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1570 if (Def->getNumOperands() == 1) {
1571 Def->replaceAllUsesWith(Def->getOperand(0));
1576 Phi->replaceAllUsesWith(Phi->getOperand(0));
1582 if (Def->getNumOperands() == 1 &&
1584 return Def->replaceAllUsesWith(IRV);
1597 return Def->replaceAllUsesWith(
A);
1604 return Def->replaceAllUsesWith(WidenIV->getRegion()->getCanonicalIV());
1607 Def->replaceAllUsesWith(Builder.createNaryOp(
1608 Instruction::ExtractElement, {A, LaneToExtract}, Def->getDebugLoc()));
1623 if (IVInc->getNumUsers() == 2) {
1628 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1629 Def->replaceAllUsesWith(IVInc);
1631 Inc->replaceAllUsesWith(Phi);
1632 Phi->setOperand(0,
Y);
1648 Steps->replaceAllUsesWith(Steps->getOperand(0));
1656 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1658 return PhiR && PhiR->isInLoop();
1664 return Def->replaceAllUsesWith(
A);
1690 R.getVPSingleValue()->replaceAllUsesWith(
X);
1706 while (!Worklist.
empty()) {
1715 R->replaceAllUsesWith(
1716 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1720static std::optional<Instruction::BinaryOps>
1723 case Intrinsic::masked_udiv:
1724 return Instruction::UDiv;
1725 case Intrinsic::masked_sdiv:
1726 return Instruction::SDiv;
1727 case Intrinsic::masked_urem:
1728 return Instruction::URem;
1729 case Intrinsic::masked_srem:
1730 return Instruction::SRem;
1747 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1751 if (RepR && RepR->getOpcode() == Instruction::Store &&
1754 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1755 true ,
nullptr , *RepR ,
1756 *RepR , RepR->getDebugLoc());
1757 Clone->insertBefore(RepOrWidenR);
1759 VPValue *ExtractOp = Clone->getOperand(0);
1765 Clone->setOperand(0, ExtractOp);
1766 RepR->eraseFromParent();
1778 VPValue *SafeDivisor = Builder.createSelect(
1779 IntrR->getOperand(2), IntrR->getOperand(1),
1781 VPValue *Clone = Builder.createNaryOp(
1782 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1785 IntrR->eraseFromParent();
1794 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1803 return !U->usesScalars(
Op);
1807 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1810 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1811 IntroducesBCastOf(Op)))
1815 bool LiveInNeedsBroadcast =
1816 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1817 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1818 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1825 RepOrWidenR->getUnderlyingInstr());
1826 Clone->insertBefore(RepOrWidenR);
1827 RepOrWidenR->replaceAllUsesWith(Clone);
1829 RepOrWidenR->eraseFromParent();
1865 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1866 UniqueValues.
insert(Blend->getIncomingValue(0));
1867 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1869 UniqueValues.
insert(Blend->getIncomingValue(
I));
1871 if (UniqueValues.
size() == 1) {
1872 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1873 Blend->eraseFromParent();
1877 if (Blend->isNormalized())
1883 unsigned StartIndex = 0;
1884 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1896 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1898 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1899 if (
I == StartIndex)
1901 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1902 OperandsWithMask.
push_back(Blend->getMask(
I));
1907 OperandsWithMask, *Blend, Blend->getDebugLoc());
1908 NewBlend->insertBefore(&R);
1910 VPValue *DeadMask = Blend->getMask(StartIndex);
1912 Blend->eraseFromParent();
1917 if (NewBlend->getNumOperands() == 3 &&
1919 VPValue *Inc0 = NewBlend->getOperand(0);
1920 VPValue *Inc1 = NewBlend->getOperand(1);
1921 VPValue *OldMask = NewBlend->getOperand(2);
1922 NewBlend->setOperand(0, Inc1);
1923 NewBlend->setOperand(1, Inc0);
1924 NewBlend->setOperand(2, NewMask);
1951 APInt MaxVal = AlignedTC - 1;
1954 unsigned NewBitWidth =
1960 bool MadeChange =
false;
1985 "canonical IV is not expected to have a truncation");
1990 NewWideIV->insertBefore(WideIV);
1997 Cmp->replaceAllUsesWith(
1998 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2012 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2014 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2028 const SCEV *VectorTripCount =
2033 "Trip count SCEV must be computable");
2054 auto *Term = &ExitingVPBB->
back();
2067 for (
unsigned Part = 0; Part < UF; ++Part) {
2073 Extracts[Part] = Ext;
2085 match(Phi->getBackedgeValue(),
2087 assert(Index &&
"Expected index from ActiveLaneMask instruction");
2104 "Expected one VPActiveLaneMaskPHIRecipe for each unroll part");
2111 "Expected incoming values of Phi to be ActiveLaneMasks");
2116 EntryALM->setOperand(2, ALMMultiplier);
2117 LoopALM->setOperand(2, ALMMultiplier);
2121 ExtractFromALM(EntryALM, EntryExtracts);
2126 ExtractFromALM(LoopALM, LoopExtracts);
2128 Not->setOperand(0, LoopExtracts[0]);
2131 for (
unsigned Part = 0; Part < UF; ++Part) {
2132 Phis[Part]->setStartValue(EntryExtracts[Part]);
2133 Phis[Part]->setBackedgeValue(LoopExtracts[Part]);
2146 auto *Term = &ExitingVPBB->
back();
2158 const SCEV *VectorTripCount =
2164 "Trip count SCEV must be computable");
2183 Term->setOperand(1, Plan.
getTrue());
2188 {}, Term->getDebugLoc());
2190 Term->eraseFromParent();
2198 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2199 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2217 RecurKind RK = PhiR->getRecurrenceKind();
2224 RecWithFlags->dropPoisonGeneratingFlags();
2230struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2239 return GEP->getSourceElementType();
2242 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2243 [](
auto *
I) {
return I->getSourceElementType(); })
2244 .
Default([](
auto *) {
return nullptr; });
2248 static bool canHandle(
const VPSingleDefRecipe *Def) {
2257 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2258 C->second == Instruction::ExtractValue)))
2262 return !
Def->mayReadOrWriteMemory();
2266 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2269 getGEPSourceElementType(Def),
Def->getScalarType(),
2272 if (RFlags->hasPredicate())
2275 return hash_combine(Result, SIVSteps->getInductionOpcode());
2280 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2281 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2284 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2286 !
equal(
L->operands(),
R->operands()))
2290 "must have valid opcode info for both recipes");
2292 if (LFlags->hasPredicate() &&
2293 LFlags->getPredicate() !=
2297 if (LSIV->getInductionOpcode() !=
2307 const VPRegionBlock *RegionL =
L->getRegion();
2308 const VPRegionBlock *RegionR =
R->getRegion();
2311 L->getParent() !=
R->getParent())
2313 return L->getScalarType() ==
R->getScalarType();
2329 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2333 if (!VPDT.
dominates(V->getParent(), VPBB))
2338 Def->replaceAllUsesWith(V);
2351 bool Sinking =
false) {
2380 "Expected vector prehader's successor to be the vector loop region");
2388 return !Op->isDefinedOutsideLoopRegions();
2391 R.moveBefore(*Preheader, Preheader->
end());
2411 assert(!RepR->isPredicated() &&
2412 "Expected prior transformation of predicated replicates to "
2413 "replicate regions");
2418 if (!RepR->isSingleScalar())
2422 if (RepR->getOpcode() == Instruction::Store &&
2423 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2428 assert((!R.mayWriteToMemory() ||
2429 (RepR && RepR->getOpcode() == Instruction::Store &&
2430 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2431 "The only recipes that may write to memory are expected to be "
2432 "stores with invariant pointer-operand");
2442 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2443 auto *UserR = cast<VPRecipeBase>(U);
2444 VPBasicBlock *Parent = UserR->getParent();
2446 if (SinkBB && SinkBB != Parent)
2451 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2452 Parent->getSinglePredecessor() != LoopRegion;
2462 "Defining block must dominate sink block");
2487 VPValue *ResultVPV = R.getVPSingleValue();
2489 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2490 if (!NewResSizeInBits)
2503 (void)OldResSizeInBits;
2511 VPW->dropPoisonGeneratingFlags();
2513 assert((OldResSizeInBits != NewResSizeInBits ||
2515 "Only ICmps should not need extending the result.");
2521 if (OldResSizeInBits != NewResSizeInBits) {
2523 Instruction::ZExt, ResultVPV, OldResTy);
2525 Ext->setOperand(0, ResultVPV);
2535 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2536 if (OpSizeInBits == NewResSizeInBits)
2538 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2539 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2545 Builder.setInsertPoint(&R);
2546 ProcessedIter->second =
2547 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2549 Op = ProcessedIter->second;
2553 NWR->insertBefore(&R);
2557 VPValue *Replacement = NWR->getVPSingleValue();
2558 if (OldResSizeInBits != NewResSizeInBits)
2564 R.eraseFromParent();
2570 std::optional<VPDominatorTree> VPDT;
2578 bool SimplifiedPhi =
false;
2588 assert(VPBB->getNumSuccessors() == 2 &&
2589 "Two successors expected for BranchOnCond");
2590 unsigned RemovedIdx;
2601 "There must be a single edge between VPBB and its successor");
2604 auto Phis = RemovedSucc->
phis();
2607 SimplifiedPhi |= !std::empty(Phis);
2611 VPBB->back().eraseFromParent();
2623 if (Reachable.contains(
B))
2634 for (
VPValue *Def : R.definedValues())
2635 Def->replaceAllUsesWith(&Tmp);
2636 R.eraseFromParent();
2640 return SimplifiedPhi;
2672 "expected to run before loop regions are created");
2674 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2681 return VPDT.
dominates(Header, R->getParent());
2684 for (
const SCEV *Stride : StridesMap.
values()) {
2687 const APInt *StrideConst;
2710 RewriteMap[StrideV] = PSE.
getSCEV(StrideV);
2717 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2720 if (NewSCEV != ScevExpr) {
2722 ExpSCEV->replaceAllUsesWith(NewExp);
2733 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2738 while (!Worklist.
empty()) {
2741 if (!Visited.
insert(CurRec).second)
2763 RecWithFlags->isDisjoint()) {
2766 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2767 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2768 RecWithFlags->replaceAllUsesWith(New);
2769 RecWithFlags->eraseFromParent();
2772 RecWithFlags->dropPoisonGeneratingFlags();
2777 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2778 "found instruction with poison generating flags not covered by "
2779 "VPRecipeWithIRFlags");
2784 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2806 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2807 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2808 match(WidenRec->getMask(), m_UnlessHdrMask))
2809 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2811 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2812 if (AddrDef && InterleaveRec->getMask() &&
2813 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2814 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2824 const bool &EpilogueAllowed) {
2825 if (InterleaveGroups.empty())
2836 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2843 for (
const auto *IG : InterleaveGroups) {
2846 for (
auto *Member : IG->members())
2848 StartMember = Member;
2856 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2862 StoredValues.
push_back(StoreR->getStoredValue());
2869 bool NeedsMaskForGaps =
2870 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2871 (!StoredValues.
empty() && !IG->isFull());
2874 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2878 "Dead member in non-load group?");
2883 InsertPos->getAsRecipe()))
2884 InsertPos = MemberR;
2885 IRInsertPos = &InsertPos->getIngredient();
2895 VPValue *Addr = Start->getAddr();
2897 if (IG->getIndex(StartMember) != 0 ||
2905 assert(IG->getIndex(IRInsertPos) != 0 &&
2906 "index of insert position shouldn't be zero");
2910 IG->getIndex(IRInsertPos),
2914 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2920 if (IG->isReverse()) {
2923 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2924 ReversePtr->insertBefore(InsertPosR);
2928 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2930 VPIG->insertBefore(InsertPosR);
2933 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2936 if (!Member->getType()->isVoidTy()) {
2954static std::optional<VPValue *>
3007 VPValue *UncountableCondition =
nullptr;
3011 return std::nullopt;
3014 Worklist.
push_back(UncountableCondition);
3015 while (!Worklist.
empty()) {
3019 if (V->isDefinedOutsideLoopRegions())
3025 if (V->getNumUsers() > 1)
3026 return std::nullopt;
3038 return std::nullopt;
3042 return std::nullopt;
3050 return std::nullopt;
3055 if (Recipes.
empty() ||
3057 return std::nullopt;
3059 return UncountableCondition;
3115 for (
auto &Exit : Exits) {
3116 if (Exit.EarlyExitingVPBB == LatchVPBB)
3120 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3121 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3132 std::optional<VPValue *>
Cond =
3148 assert(
Load &&
"Couldn't find exactly one load");
3151 "Uncountable exit condition load is conditional.");
3165 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3189 while (InsertIt != HeaderVPBB->
end() &&
3191 erase(ConditionRecipes, &*InsertIt);
3194 for (
auto *Recipe :
reverse(ConditionRecipes))
3195 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3199 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3201 Type *IVScalarTy =
IV->getScalarType();
3207 {Zero, FirstActive, ALMMultiplier},
3208 DebugLoc(),
"uncountable.exit.mask");
3213 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3215 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3225 "Expected BranchOnCond terminator for MiddleVPBB");
3236 auto Phis = ScalarPH->
phis();
3246 "Continuing from different IV");
3260 for (
auto [EarlyExitingVPBB, ExitBlock] :
3264 VPValue *CondOfEarlyExitingVPBB;
3265 [[maybe_unused]]
bool Matched =
3266 match(EarlyExitingVPBB->getTerminator(),
3268 assert(Matched &&
"Terminator must be BranchOnCond");
3272 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3273 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3275 TrueSucc == ExitBlock
3276 ? CondOfEarlyExitingVPBB
3277 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3283 "exit condition must dominate the latch");
3291 assert(!Exits.
empty() &&
"must have at least one early exit");
3298 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3301 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3307 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3308 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3310 Exits[
I].EarlyExitingVPBB) &&
3311 "RPO sort must place dominating exits before dominated ones");
3317 VPValue *Combined = Exits[0].CondToExit;
3330 "Unexpected terminator");
3331 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3332 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3333 LatchExitingBranch->eraseFromParent();
3336 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3342 LatchVPBB->
setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3346 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3351 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3355 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3363 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3366 LatchVPBB->
setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3398 for (
auto [Exit, VectorEarlyExitVPBB] :
3399 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3400 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3412 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3413 VPValue *NewIncoming = IncomingVal;
3415 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3420 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3421 ExitIRI->addIncoming(NewIncoming);
3424 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3458 bool IsLastDispatch = (
I + 2 == Exits.
size());
3460 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3466 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3469 CurrentBB = FalseBB;
3484 VPValue *VecOp = Red->getVecOp();
3486 assert(!Red->isPartialReduction() &&
3487 "This path does not support partial reductions");
3490 auto IsExtendedRedValidAndClampRange =
3503 "getExtendedReductionCost only supports integer types");
3504 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3505 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3506 Red->getFastMathFlagsOrNone(),
CostKind);
3507 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3515 IsExtendedRedValidAndClampRange(
3536 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3537 Opcode != Instruction::FAdd)
3540 assert(!Red->isPartialReduction() &&
3541 "This path does not support partial reductions");
3545 auto IsMulAccValidAndClampRange =
3557 (Ext0->getOpcode() != Ext1->getOpcode() ||
3558 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3562 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3564 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3571 ExtCost += Ext0->computeCost(VF, Ctx);
3573 ExtCost += Ext1->computeCost(VF, Ctx);
3575 ExtCost += OuterExt->computeCost(VF, Ctx);
3577 return MulAccCost.
isValid() &&
3578 MulAccCost < ExtCost + MulCost + RedCost;
3583 VPValue *VecOp = Red->getVecOp();
3621 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3623 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3624 Mul->setOperand(1, ExtB);
3634 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3639 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3646 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3663 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3672 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3673 Ext0->getOpcode() == Ext1->getOpcode() &&
3674 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3676 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3677 *Ext0, *Ext0, Ext0->getDebugLoc());
3678 NewExt0->insertBefore(Ext0);
3683 Ext->getScalarType(),
nullptr, *Ext1,
3684 *Ext1, Ext1->getDebugLoc());
3687 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3688 NewMul->insertBefore(
Mul);
3689 Ext->replaceAllUsesWith(NewMul);
3690 Ext->eraseFromParent();
3691 Mul->eraseFromParent();
3705 assert(!Red->isPartialReduction() &&
3706 "This path does not support partial reductions");
3709 auto IP = std::next(Red->getIterator());
3710 auto *VPBB = Red->getParent();
3720 Red->replaceAllUsesWith(AbstractR);
3740 return CommonMetadata;
3743template <
unsigned Opcode>
3748 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3749 "Only Load and Store opcodes supported");
3750 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3757 for (
auto Recipes :
Groups) {
3758 if (Recipes.size() < 2)
3763 "Expected all recipes in group to have the same load-store type");
3770 VPValue *MaskI = RecipeI->getMask();
3776 bool HasComplementaryMask =
false;
3781 VPValue *MaskJ = RecipeJ->getMask();
3790 if (HasComplementaryMask) {
3791 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3801template <
typename InstType>
3819 for (
auto &Group :
Groups) {
3839 return R->isSingleScalar() == IsSingleScalar;
3841 "all members in group must agree on IsSingleScalar");
3846 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3847 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3849 UnpredicatedLoad->insertBefore(EarliestLoad);
3853 Load->replaceAllUsesWith(UnpredicatedLoad);
3854 Load->eraseFromParent();
3863 if (!StoreLoc || !StoreLoc->AATags.Scope)
3870 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3882 for (
auto &Group :
Groups) {
3895 VPValue *SelectedValue = Group[0]->getOperand(0);
3898 bool IsSingleScalar = Group[0]->isSingleScalar();
3899 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3900 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3901 "all members in group must agree on IsSingleScalar");
3902 VPValue *Mask = Group[
I]->getMask();
3904 SelectedValue = Builder.createSelect(
3907 Value->getScalarType()));
3915 StoreWithMinAlign->getUnderlyingInstr(),
3916 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3917 nullptr, *LastStore, CommonMetadata);
3918 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3922 Store->eraseFromParent();
3937 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
3942 if (Member0Op == OpV)
3952 return !IsScalable && !W->getMask() && W->isConsecutive() &&
3955 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
3970 if (R->getScalarType() != WideMember0->getScalarType())
3972 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
3976 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
3979 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
3984 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
3985 const auto &[
OpIdx, OpV] =
P;
3997static std::optional<ElementCount>
4001 if (!InterleaveR || InterleaveR->
getMask())
4002 return std::nullopt;
4004 Type *GroupElementTy =
nullptr;
4008 return Op->getScalarType() == GroupElementTy;
4010 return std::nullopt;
4014 return Op->getScalarType() == GroupElementTy;
4016 return std::nullopt;
4020 if (IG->getFactor() != IG->getNumMembers())
4021 return std::nullopt;
4027 assert(
Size.isScalable() == VF.isScalable() &&
4028 "if Size is scalable, VF must be scalable and vice versa");
4029 return Size.getKnownMinValue();
4033 unsigned MinVal = VF.getKnownMinValue();
4035 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4038 return std::nullopt;
4046 return RepR && RepR->isSingleScalar();
4060 if (V->isDefinedOutsideLoopRegions()) {
4063 return M->isDefinedOutsideLoopRegions() &&
4064 M->getScalarType() == V->getScalarType();
4066 "expected distinct loop-invariant values of matching scalar type");
4081 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4083 for (
VPValue *Member : Members)
4084 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4085 WideMember0->setOperand(
4094 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4096 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4097 *LoadGroup, LoadGroup->getDebugLoc());
4103 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4104 "must be a single scalar load");
4105 NarrowedOps.
insert(RepR);
4110 VPValue *PtrOp = WideLoad->getAddr();
4112 PtrOp = VecPtr->getOperand(0);
4117 nullptr, {}, *WideLoad);
4118 N->insertBefore(WideLoad);
4123std::unique_ptr<VPlan>
4143 "unexpected branch-on-count");
4146 std::optional<ElementCount> VFToOptimize;
4160 if (R.mayWriteToMemory() && !InterleaveR)
4166 return any_of(V->users(), [&](VPUser *U) {
4167 auto *UR = cast<VPRecipeBase>(U);
4168 return UR->getParent()->getParent() != VectorLoop;
4185 std::optional<ElementCount> NarrowedVF =
4187 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4189 VFToOptimize = NarrowedVF;
4192 if (InterleaveR->getStoredValues().empty())
4197 auto *Member0 = InterleaveR->getStoredValues()[0];
4207 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4210 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4211 return IR && IR->getInterleaveGroup()->isFull() &&
4212 IR->getVPValue(Op.index()) == Op.value();
4221 VFToOptimize->isScalable()))
4226 if (StoreGroups.empty())
4230 bool RequiresScalarEpilogue =
4241 std::unique_ptr<VPlan> NewPlan;
4243 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4244 Plan.
setVF(*VFToOptimize);
4245 NewPlan->removeVF(*VFToOptimize);
4252 for (
auto *StoreGroup : StoreGroups) {
4254 NarrowedOps, Preheader);
4260 StoreGroup->getDebugLoc());
4267 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4273 if (VFToOptimize->isScalable()) {
4276 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4284 materializeVectorTripCount(Plan, VectorPH,
false,
4285 RequiresScalarEpilogue, Step);
4290 removeDeadRecipes(Plan);
4293 "All VPVectorPointerRecipes should have been removed");
4313 "Cannot handle loops with uncountable early exits");
4320 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4327 if (
any_of(RecurSplice->users(),
4328 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4409 {},
"vector.recur.extract.for.phi");
4412 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4426 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4427 VPValue *InvariantCandidate = BinOp->getOperand(1);
4429 std::swap(WidenIVCandidate, InvariantCandidate);
4443 auto *ClonedOp = BinOp->
clone();
4444 if (ClonedOp->getOperand(0) == WidenIV) {
4445 ClonedOp->setOperand(0, ScalarIV);
4447 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4448 ClonedOp->setOperand(1, ScalarIV);
4462 return std::nullopt;
4467 return std::nullopt;
4479 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4480 bool UseMax) -> std::optional<APSInt> {
4482 for (
bool Signed : {
true,
false}) {
4491 return std::nullopt;
4499 PhiR->getRecurrenceKind()))
4508 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4522 !
match(FindLastSelect,
4531 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4536 "IVOfExpressionToSink not being an AddRec must imply "
4537 "FindLastExpression not being an AddRec.");
4546 bool UseMax = *StepDirection;
4547 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4548 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4555 if (IVOfExpressionToSink) {
4556 const SCEV *FindLastExpressionSCEV =
4558 if (std::optional<bool> NewUseMax =
4560 if (
auto NewSentinel =
4561 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4564 SentinelVal = *NewSentinel;
4565 UseSigned = NewSentinel->isSigned();
4566 UseMax = *NewUseMax;
4567 IVSCEV = FindLastExpressionSCEV;
4568 IVOfExpressionToSink =
nullptr;
4578 if (AR->hasNoSignedWrap())
4580 else if (AR->hasNoUnsignedWrap())
4590 VPValue *NewFindLastSelect = BackedgeVal;
4592 if (!SentinelVal || IVOfExpressionToSink) {
4595 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4596 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4597 if (
match(FindLastSelect,
4599 SelectCond = LoopBuilder.
createNot(SelectCond);
4606 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4609 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4618 VPIRFlags Flags(MinMaxKind,
false,
false,
4624 NewFindLastSelect, Flags, ExitDL);
4627 VPValue *VectorRegionExitingVal = ReducedIV;
4628 if (IVOfExpressionToSink)
4629 VectorRegionExitingVal =
4631 ReducedIV, IVOfExpressionToSink);
4634 VPValue *StartVPV = PhiR->getStartValue();
4641 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4651 AnyOfPhi->insertAfter(PhiR);
4658 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4671 PhiR->hasUsesOutsideReductionChain());
4672 NewPhiR->insertBefore(PhiR);
4673 PhiR->replaceAllUsesWith(NewPhiR);
4674 PhiR->eraseFromParent();
4681struct ReductionExtend {
4682 Type *SrcType =
nullptr;
4683 ExtendKind Kind = ExtendKind::PR_None;
4689struct ExtendedReductionOperand {
4693 ReductionExtend ExtendA, ExtendB;
4701struct VPPartialReductionChain {
4704 VPWidenRecipe *ReductionBinOp =
nullptr;
4706 ExtendedReductionOperand ExtendedOp;
4713 unsigned AccumulatorOpIdx;
4714 unsigned ScaleFactor;
4717 VPBlendRecipe *Blend =
nullptr;
4722static std::optional<unsigned>
4726 "Expected a non-normalized blend with two incoming values");
4732 return std::nullopt;
4733 return FirstIncomingHasOneUse ? 0 : 1;
4745 if (!
Op->hasOneUse() ||
4751 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4752 Op->getOperand(1), NarrowTy);
4754 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4763 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4765 assert(Ext->getOpcode() ==
4767 "Expected both the LHS and RHS extends to be the same");
4768 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4771 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4772 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4773 auto *
Max = Builder.insert(
4775 {FreezeX, FreezeY}, SrcTy));
4776 auto *Min = Builder.insert(
4778 {FreezeX, FreezeY}, SrcTy));
4781 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4782 Op->getScalarType());
4794 if (!
Mul->hasOneUse() ||
4795 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4796 MulLHS->getOpcode() != MulRHS->getOpcode())
4799 auto *NewLHS = Builder.createWidenCast(
4800 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4801 auto *NewRHS = MulLHS == MulRHS
4803 : Builder.createWidenCast(MulRHS->getOpcode(),
4804 MulRHS->getOperand(0),
4805 Ext->getScalarType());
4806 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4807 Builder.insert(NewMul);
4808 Op->replaceAllUsesWith(NewMul);
4809 Op->eraseFromParent();
4810 Mul->eraseFromParent();
4819 VPValue *VecOp = Red->getVecOp();
4873static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4881 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4884 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4900 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4902 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4907 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4917 Builder.insert(NegRecipe);
4918 ExtendedOp = NegRecipe;
4933 std::optional<unsigned> BlendReductionIdx =
4934 getBlendReductionUpdateValueIdx(Chain.Blend);
4935 assert(BlendReductionIdx &&
4937 "Expected blend to contain the reduction update");
4948 assert((!ExitValue || IsLastInChain) &&
4949 "if we found ExitValue, it must match RdxPhi's backedge value");
4960 PartialRed->insertBefore(WidenRecipe);
4970 E->insertBefore(WidenRecipe);
4971 PartialRed->replaceAllUsesWith(
E);
4984 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
4985 StartInst->setOperand(2, NewScaleFactor);
4993 VPValue *OldStartValue = StartInst->getOperand(0);
4994 StartInst->setOperand(0, StartInst->getOperand(1));
4998 assert(RdxResult &&
"Could not find reduction result");
5001 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
5002 : Instruction::BinaryOps::Sub;
5008 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
5014 const VPPartialReductionChain &Link,
5017 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5018 std::optional<unsigned> BinOpc = std::nullopt;
5020 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5021 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5023 std::optional<llvm::FastMathFlags>
Flags;
5027 auto GetLinkOpcode = [&Link]() ->
unsigned {
5030 return Instruction::Add;
5032 return Instruction::FAdd;
5034 return Link.ReductionBinOp->
getOpcode();
5039 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5040 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5061static std::optional<ExtendedReductionOperand>
5064 "Op should be operand of UpdateR");
5072 if (
Op->hasOneUse() &&
5081 Type *RHSInputType =
Y->getScalarType();
5082 if (LHSInputType != RHSInputType ||
5083 LHSExt->getOpcode() != RHSExt->getOpcode())
5084 return std::nullopt;
5087 return ExtendedReductionOperand{
5089 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5093 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5096 VPValue *CastSource = CastRecipe->getOperand(0);
5097 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5107 return ExtendedReductionOperand{
5114 if (!
Op->hasOneUse())
5115 return std::nullopt;
5120 return std::nullopt;
5130 return std::nullopt;
5134 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5137 const APInt *RHSConst =
nullptr;
5143 return std::nullopt;
5147 if (Cast && OuterExtKind &&
5148 getPartialReductionExtendKind(Cast) != OuterExtKind)
5149 return std::nullopt;
5151 Type *RHSInputType = LHSInputType;
5152 ExtendKind RHSExtendKind = LHSExtendKind;
5155 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5158 return ExtendedReductionOperand{
5159 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5166static std::optional<SmallVector<VPPartialReductionChain>>
5173 return std::nullopt;
5183 VPValue *CurrentValue = ExitValue;
5184 while (CurrentValue != RedPhiR) {
5186 std::optional<unsigned> BlendReductionIdx;
5190 return std::nullopt;
5192 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5193 if (!BlendReductionIdx)
5194 return std::nullopt;
5201 return std::nullopt;
5208 std::optional<ExtendedReductionOperand> ExtendedOp =
5209 matchExtendedReductionOperand(UpdateR,
Op);
5211 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5213 return std::nullopt;
5221 return std::nullopt;
5223 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5226 return std::nullopt;
5228 VPPartialReductionChain Link(
5229 {UpdateR, *ExtendedOp, RK,
5234 CurrentValue = PrevValue;
5239 std::reverse(Chain.
begin(), Chain.
end());
5258 if (
auto Chains = getScaledReductions(RedPhiR))
5259 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5262 if (ChainsByPhi.
empty())
5270 for (
const auto &[
_, Chains] : ChainsByPhi)
5271 for (
const VPPartialReductionChain &Chain : Chains) {
5272 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5274 PartialReductionBlends.
insert(Chain.Blend);
5275 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5281 auto ExtendUsersValid = [&](
VPValue *Ext) {
5283 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5287 auto IsProfitablePartialReductionChainForVF =
5294 for (
const VPPartialReductionChain &Link : Chain) {
5295 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5296 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5300 PartialCost += LinkCost;
5301 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5303 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5304 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5307 RegularCost += Extend->computeCost(VF, CostCtx);
5309 return PartialCost.
isValid() && PartialCost < RegularCost;
5317 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5318 for (
const VPPartialReductionChain &Chain : Chains) {
5319 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5323 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5325 return PhiR == RedPhiR;
5329 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5331 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5337 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5346 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5347 return RepR && RepR->getOpcode() == Instruction::Store;
5358 return IsProfitablePartialReductionChainForVF(Chains, VF);
5364 for (
auto &[Phi, Chains] : ChainsByPhi)
5365 for (
const VPPartialReductionChain &Chain : Chains)
5366 transformToPartialReduction(Chain, Plan, Phi);
5381 if (VPI && VPI->getUnderlyingValue() &&
5392 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5395 if (!ProcessVPInst(VPI))
5404 assert(New->getParent() &&
"New recipe must have been inserted");
5405 if (VPI->
getOpcode() == Instruction::Load)
5414 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5421 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5423 VPI, FinalRedStoresBuilder))
5432 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5445 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5449 return Scalarize(VPI);
5456 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5458 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5468 const SCEV *PtrSCEV =
5470 bool IsSingleScalarLoad =
5476 I, Ptr, IsSingleScalarLoad,
5485 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5487 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5491 std::optional<int64_t> Stride =
5493 if (Stride != 1 && Stride != -1)
5524 return ReplaceWith(VPI,
Load);
5533 auto *StoreR = Builder.createWidenStore(
5536 return ReplaceWith(VPI, StoreR);
5543 return ReplaceWith(VPI, Recipe);
5545 return Scalarize(VPI);
5568 if (VPI->mayHaveSideEffects())
5572 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5577 if (VPI->getOpcode() == Instruction::Add &&
5586 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5587 *VPI, VPI->getDebugLoc(),
I);
5588 Recipe->insertBefore(VPI);
5589 VPI->replaceAllUsesWith(Recipe);
5590 VPI->eraseFromParent();
5600 switch (Param.ParamKind) {
5601 case VFParamKind::Vector:
5602 case VFParamKind::GlobalPredicate:
5604 case VFParamKind::OMP_Uniform:
5605 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5606 SE->isLoopInvariant(
5607 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5609 case VFParamKind::OMP_Linear:
5610 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5611 m_scev_AffineAddRec(
5612 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5613 m_SpecificLoop(L)));
5630 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5631 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5634 if (It == Mappings.end())
5641struct CallWideningDecision {
5642 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5643 CallWideningDecision(KindTy Kind, Function *Variant =
nullptr)
5666 return CallWideningDecision::KindTy::Scalarize;
5676 return CallWideningDecision::KindTy::Scalarize;
5680 false, VF, CostCtx);
5695 return CallWideningDecision::KindTy::Intrinsic;
5699 if (VecFunc && ScalarCost >= VecCallCost)
5700 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5702 return CallWideningDecision::KindTy::Scalarize;
5712 if (!VPI || !VPI->getUnderlyingValue() ||
5713 VPI->getOpcode() != Instruction::Call)
5718 VPI->op_begin() + CI->arg_size());
5720 CallWideningDecision Decision =
5729 switch (Decision.Kind) {
5730 case CallWideningDecision::KindTy::Intrinsic: {
5734 *VPI, VPI->getDebugLoc());
5737 case CallWideningDecision::KindTy::VectorVariant: {
5741 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5742 Ops.push_back(Mask);
5744 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5746 *VPI, VPI->getDebugLoc());
5749 case CallWideningDecision::KindTy::Scalarize:
5755 VPI->replaceAllUsesWith(Replacement);
5756 VPI->eraseFromParent();
5779 if (!LoadR || LoadR->isConsecutive())
5782 VPValue *Ptr = LoadR->getAddr();
5795 Align Alignment = LoadR->getAlign();
5798 if (!Ctx.TTI.isLegalStridedLoadStore(DataTy, Alignment))
5803 Intrinsic::experimental_vp_strided_load, DataTy,
5804 LoadR->isMasked(), Alignment, Ctx);
5805 return StridedLoadStoreCost < CurrentCost;
5816 Ctx.invalidateWideningDecision(&LoadR->getIngredient(), VF);
5821 I32VF = Builder.createScalarZExtOrTrunc(
5838 "Stride type from SCEV must match the index type");
5839 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5842 auto *
Offset = Builder.createOverflowingOp(
5843 Instruction::Mul, {CanIV, StrideInBytes},
5844 {AddRecPtr->hasNoUnsignedWrap(),
false});
5848 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5851 VPValue *NewPtr = Builder.createVectorPointer(
5853 LoadR->getDebugLoc());
5855 VPValue *Mask = LoadR->getMask();
5858 auto *StridedLoad = Builder.createWidenMemIntrinsic(
5859 Intrinsic::experimental_vp_strided_load,
5860 {NewPtr, StrideInBytes, Mask, I32VF}, LoadTy, Alignment, *LoadR,
5861 LoadR->getDebugLoc());
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
MachineInstr unsigned OpIdx
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
constexpr bool isVector() const
One or more elements.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
void clearSuccessors()
Remove all the successors of this block.
VPBlockBase * getSinglePredecessor() const
void clearPredecessors()
Remove all the predecessor of this block.
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", const VPIRFlags &Flags={})
VPExpandSCEVRecipe * createExpandSCEV(const SCEV *Expr)
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * tryToExpand(const SCEV *S)
Try to expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
unsigned getNumUsers() const
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPIRValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
VPInstruction_match< VPInstruction::ActiveLaneMask, Op0_t, Op1_t, Op2_t > m_ActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPIRValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
LLVM_ABI_FOR_TEST cl::opt< bool > EnableWideActiveLaneMask
UncountableExitStyle
Different methods of handling early exits.
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...