57 if (!VPBB->getParent())
60 auto EndIter = Term ? Term->getIterator() : VPBB->end();
65 VPValue *VPV = Ingredient.getVPSingleValue();
86 *
Load, Ingredient.getOperand(0),
nullptr ,
87 false , *VPI, Ingredient.getDebugLoc());
90 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
91 nullptr ,
false , *VPI,
92 Ingredient.getDebugLoc());
95 Ingredient.operands(), *VPI,
96 Ingredient.getDebugLoc(),
GEP);
108 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
113 if (VectorID == Intrinsic::assume ||
114 VectorID == Intrinsic::lifetime_end ||
115 VectorID == Intrinsic::lifetime_start ||
116 VectorID == Intrinsic::sideeffect ||
117 VectorID == Intrinsic::pseudoprobe) {
122 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
123 VectorID != Intrinsic::pseudoprobe;
127 Ingredient.getDebugLoc());
130 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
131 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
135 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
139 *VPI, Ingredient.getDebugLoc());
143 "inductions must be created earlier");
152 "Only recpies with zero or one defined values expected");
153 Ingredient.eraseFromParent();
164 const Loop *L =
nullptr;
169 if (
A->getOpcode() != Instruction::Store ||
170 B->getOpcode() != Instruction::Store)
183 const APInt *Distance;
189 Type *TyA =
A->getOperand(0)->getScalarType();
191 Type *TyB =
B->getOperand(0)->getScalarType();
197 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
199 auto VFs =
B->getParent()->getPlan()->vectorFactors();
203 return Distance->
abs().
uge(
211 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
212 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
221 return ExcludeRecipes.contains(
Store) ||
222 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
235 std::optional<SinkStoreInfo> SinkInfo = {}) {
236 bool CheckReads = SinkInfo.has_value();
240 if (SinkInfo && SinkInfo->shouldSkip(R))
244 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
269template <
unsigned Opcode>
274 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
275 "Only Load and Store opcodes supported");
276 constexpr bool IsLoad = (Opcode == Instruction::Load);
279 RecipesByAddressAndType;
284 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
288 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
292 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
297 for (
auto &Group :
Groups) {
312 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
324 if (Candidate->getParent() == SinkTo ||
329 if (!ScalarVFOnly && RepR->isSingleScalar())
332 WorkList.
insert({SinkTo, Candidate});
344 for (
auto &Recipe : *VPBB)
346 InsertIfValidSinkCandidate(VPBB,
Op);
350 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
353 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
358 auto UsersOutsideSinkTo =
360 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
362 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
363 return !U->usesFirstLaneOnly(SinkCandidate);
366 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
368 if (NeedsDuplicating) {
372 if (
auto *SinkCandidateRepR =
377 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
378 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
382 Clone = SinkCandidate->
clone();
392 InsertIfValidSinkCandidate(SinkTo,
Op);
402 if (!EntryBB || EntryBB->size() != 1 ||
412 if (EntryBB->getNumSuccessors() != 2)
417 if (!Succ0 || !Succ1)
420 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
422 if (Succ0->getSingleSuccessor() == Succ1)
424 if (Succ1->getSingleSuccessor() == Succ0)
441 if (!Region1->isReplicator())
443 auto *MiddleBasicBlock =
445 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
450 if (!Region2 || !Region2->isReplicator())
455 if (!Mask1 || Mask1 != Mask2)
458 assert(Mask1 && Mask2 &&
"both region must have conditions");
464 if (TransformedRegions.
contains(Region1))
471 if (!Then1 || !Then2)
491 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
497 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
498 Phi1ToMove.eraseFromParent();
501 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
515 TransformedRegions.
insert(Region1);
518 return !TransformedRegions.
empty();
526 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
527 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
528 auto *BlockInMask = PredRecipe->
getMask();
549 Region->setParent(ParentRegion);
555 RecipeWithoutMask->getDebugLoc());
556 Exiting->appendRecipe(PHIRecipe);
569 if (RepR->isPredicated())
588 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
600 if (!VPBB->getParent())
604 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
613 R.moveBefore(*PredVPBB, PredVPBB->
end());
615 auto *ParentRegion = VPBB->getParent();
616 if (ParentRegion && ParentRegion->getExiting() == VPBB)
617 ParentRegion->setExiting(PredVPBB);
621 return !WorkList.
empty();
628 bool ShouldSimplify =
true;
629 while (ShouldSimplify) {
645 if (!
IV ||
IV->getTruncInst())
660 for (
auto *U : FindMyCast->
users()) {
662 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
663 FoundUserCast = UserCast;
670 FindMyCast = FoundUserCast;
672 if (FindMyCast !=
IV)
687 Builder.createDerivedIV(Kind, FPBinOp, StartV, CanonicalIV, Step);
696 BaseIV = Builder.createScalarCast(Instruction::Trunc, BaseIV, TruncTy,
DL);
702 if (ResultTy != StepTy) {
709 Builder.setInsertPoint(VecPreheader);
710 Step = Builder.createScalarCast(Instruction::Trunc, Step, ResultTy,
DL);
712 return Builder.createScalarIVSteps(InductionOpcode, FPBinOp, BaseIV, Step,
738 WideCanIV->getDebugLoc(), Builder));
739 WideCanIV->eraseFromParent();
756 WideCanIV->replaceAllUsesWith(WidenIV);
757 WideCanIV->eraseFromParent();
766 if (PHICost > BroadcastCost)
775 unsigned RegClass =
TTI.getRegisterClassForType(
true, VecTy);
787 WideCanIV->getNoWrapFlags(), WideCanIV->getDebugLoc());
788 NewWideIV->insertBefore(&*Header->getFirstNonPhi());
789 WideCanIV->replaceAllUsesWith(NewWideIV);
790 WideCanIV->eraseFromParent();
810 VPUser *PhiUser = PhiR->getSingleUser();
816 PhiR->replaceAllUsesWith(Start);
817 PhiR->eraseFromParent();
834 nullptr, StartV, StepV, PtrIV->
getDebugLoc(), Builder);
871 Def->user_empty() || !Def->getUnderlyingValue() ||
872 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
885 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
887 Def->getUnderlyingInstr());
888 Clone->insertAfter(Def);
889 Def->replaceAllUsesWith(Clone);
900 PtrIV->replaceAllUsesWith(PtrAdd);
907 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
908 return U->usesScalars(WideIV);
914 Plan,
ID.getKind(),
ID.getInductionOpcode(),
916 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
917 WideIV->getDebugLoc(), Builder);
920 if (!HasOnlyVectorVFs) {
922 "plans containing a scalar VF cannot also include scalable VFs");
923 WideIV->replaceAllUsesWith(Steps);
926 WideIV->replaceUsesWithIf(Steps,
927 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
929 return U.usesFirstLaneOnly(WideIV);
930 return U.usesScalars(WideIV);
946 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
951 if (!Def || Def->getNumOperands() != 2)
959 auto IsWideIVInc = [&]() {
960 auto &
ID = WideIV->getInductionDescriptor();
963 VPValue *IVStep = WideIV->getStepValue();
964 switch (
ID.getInductionOpcode()) {
965 case Instruction::Add:
967 case Instruction::FAdd:
969 case Instruction::FSub:
972 case Instruction::Sub: {
992 return IsWideIVInc() ? WideIV :
nullptr;
1009 if (WideIntOrFp && WideIntOrFp->getTruncInst())
1020 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
1022 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
1023 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
1028 if (Incoming != WideIV) {
1030 EndValue =
B.createAdd(EndValue, One,
DL);
1035 VPIRValue *Start = WideIV->getStartValue();
1036 VPValue *Step = WideIV->getStepValue();
1037 EndValue =
B.createDerivedIV(
1039 Start, EndValue, Step);
1053 if (WideIntOrFp && WideIntOrFp->getTruncInst())
1063 Start, VectorTC, Step);
1095 assert(EndValue &&
"Must have computed the end value up front");
1100 if (Incoming != WideIV)
1112 auto *Zero = Plan.
getZero(StepTy);
1113 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1118 return B.createNaryOp(
1119 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1121 : Instruction::FAdd,
1122 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1133 VPBuilder VectorPHBuilder(VectorPH, VectorPH->begin());
1143 EndValues[WideIV] = EndValue;
1153 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1154 R.eraseFromParent();
1163 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1165 if (PredVPBB == MiddleVPBB)
1167 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1170 Plan, ExitIRI->getOperand(Idx), PSE);
1172 ExitIRI->setOperand(Idx, Escape);
1189 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1193 ExpR->replaceAllUsesWith(V->second);
1197 ExpR->eraseFromParent();
1203 bool CanCreateNewRecipe) {
1204 VPlan *Plan = Def->getParent()->getPlan();
1214 Def->replaceAllUsesWith(
X);
1215 Def->eraseFromParent();
1227 Def->replaceAllUsesWith(
X);
1239 Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1245 Def->replaceAllUsesWith(
X);
1251 Def->replaceAllUsesWith(Plan->
getFalse());
1257 Def->replaceAllUsesWith(
X);
1262 if (CanCreateNewRecipe &&
1267 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1268 !Def->getOperand(1)->hasMoreThanOneUniqueUser())) {
1269 Def->replaceAllUsesWith(
1270 Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z)));
1277 Def->replaceAllUsesWith(Def->getOperand(1));
1284 Def->replaceAllUsesWith(Builder.createLogicalAnd(
X,
Y));
1290 Def->replaceAllUsesWith(Plan->
getFalse());
1295 Def->replaceAllUsesWith(
X);
1301 if (CanCreateNewRecipe &&
1303 Def->replaceAllUsesWith(Builder.createNot(
C));
1309 Def->setOperand(0,
C);
1310 Def->setOperand(1,
Y);
1311 Def->setOperand(2,
X);
1316 if (CanCreateNewRecipe &&
1320 Y->getScalarType()->isIntegerTy(1)) {
1321 Def->replaceAllUsesWith(
1322 Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z)));
1331 VPlan *Plan = Def->getParent()->getPlan();
1337 return Def->replaceAllUsesWith(V);
1343 PredPHI->replaceAllUsesWith(
Op);
1351 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1355 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1356 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1357 RepR->getDebugLoc());
1358 Unmasked->insertBefore(RepR);
1359 RepR->replaceAllUsesWith(Unmasked);
1360 RepR->eraseFromParent();
1374 bool CanCreateNewRecipe =
1379 Type *TruncTy = Def->getScalarType();
1380 Type *ATy =
A->getScalarType();
1381 if (TruncTy == ATy) {
1382 Def->replaceAllUsesWith(
A);
1391 : Instruction::ZExt;
1394 if (
auto *UnderlyingExt = Def->getOperand(0)->getUnderlyingValue()) {
1396 Ext->setUnderlyingValue(UnderlyingExt);
1398 Def->replaceAllUsesWith(Ext);
1400 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1401 Def->replaceAllUsesWith(Trunc);
1411 return Def->replaceAllUsesWith(
A);
1414 return Def->replaceAllUsesWith(
A);
1417 return Def->replaceAllUsesWith(Plan->
getZero(Def->getScalarType()));
1423 return Def->replaceAllUsesWith(Builder.createSub(
1424 Plan->
getZero(
A->getScalarType()),
A, Def->getDebugLoc(),
"", NW));
1427 if (CanCreateNewRecipe &&
1435 ->hasNoSignedWrap()};
1436 return Def->replaceAllUsesWith(
1437 Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW));
1446 MulR->hasNoSignedWrap() &&
1448 return Def->replaceAllUsesWith(Builder.createNaryOp(
1450 {A, Plan->getConstantInt(APC->getBitWidth(), ShiftAmt)}, NW,
1451 Def->getDebugLoc()));
1456 return Def->replaceAllUsesWith(Builder.createNaryOp(
1458 {A, Plan->getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
1463 return Def->replaceAllUsesWith(
A);
1478 R->setOperand(1,
Y);
1479 R->setOperand(2,
X);
1483 R->replaceAllUsesWith(Cmp);
1488 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1489 Cmp->setDebugLoc(Def->getDebugLoc());
1501 if (
Op->getNumUsers() > 1 ||
1505 }
else if (!UnpairedCmp) {
1506 UnpairedCmp =
Op->getDefiningRecipe();
1510 UnpairedCmp =
nullptr;
1517 if (NewOps.
size() < Def->getNumOperands()) {
1519 return Def->replaceAllUsesWith(NewAnyOf);
1526 if (CanCreateNewRecipe &&
1532 return Def->replaceAllUsesWith(NewCmp);
1538 Def->getOperand(1)->getScalarType() == Def->getScalarType())
1539 return Def->replaceAllUsesWith(Def->getOperand(1));
1543 Type *WideStepTy = Def->getScalarType();
1544 if (
X->getScalarType() != WideStepTy)
1545 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1546 Def->replaceAllUsesWith(
X);
1555 Def->getScalarType()->isIntegerTy(1)) {
1556 Def->setOperand(1, Def->getOperand(0));
1557 Def->setOperand(0,
Y);
1564 return Def->replaceAllUsesWith(Def->getOperand(0));
1570 Def->replaceAllUsesWith(
1571 BuildVector->getOperand(BuildVector->getNumOperands() - 1));
1576 return Def->replaceAllUsesWith(
X);
1579 return Def->replaceAllUsesWith(
A);
1582 return Def->replaceAllUsesWith(
A);
1588 Def->replaceAllUsesWith(
1589 BuildVector->getOperand(BuildVector->getNumOperands() - 2));
1596 Def->replaceAllUsesWith(BuildVector->getOperand(Idx));
1601 Def->replaceAllUsesWith(
1609 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1610 return U.usesFirstLaneOnly(Def);
1619 "broadcast operand must be single-scalar");
1620 Def->setOperand(0,
C);
1625 return Def->replaceUsesWithIf(
1626 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1629 if (Def->getNumOperands() == 1) {
1630 Def->replaceAllUsesWith(Def->getOperand(0));
1635 Phi->replaceAllUsesWith(Phi->getOperand(0));
1641 if (Def->getNumOperands() == 1 &&
1643 return Def->replaceAllUsesWith(IRV);
1656 return Def->replaceAllUsesWith(
A);
1663 return Def->replaceAllUsesWith(WidenIV->getRegion()->getCanonicalIV());
1666 Def->replaceAllUsesWith(Builder.createNaryOp(
1667 Instruction::ExtractElement, {A, LaneToExtract}, Def->getDebugLoc()));
1681 auto *IVInc = Def->getOperand(0);
1682 if (IVInc->getNumUsers() == 2) {
1687 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1688 Def->replaceAllUsesWith(IVInc);
1690 Inc->replaceAllUsesWith(Phi);
1691 Phi->setOperand(0,
Y);
1707 Steps->replaceAllUsesWith(Steps->getOperand(0));
1715 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1717 return PhiR && PhiR->isInLoop();
1723 return Def->replaceAllUsesWith(
A);
1749 R.getVPSingleValue()->replaceAllUsesWith(
X);
1765 while (!Worklist.
empty()) {
1774 R->replaceAllUsesWith(
1775 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1779static std::optional<Instruction::BinaryOps>
1782 case Intrinsic::masked_udiv:
1783 return Instruction::UDiv;
1784 case Intrinsic::masked_sdiv:
1785 return Instruction::SDiv;
1786 case Intrinsic::masked_urem:
1787 return Instruction::URem;
1788 case Intrinsic::masked_srem:
1789 return Instruction::SRem;
1806 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1810 if (RepR && RepR->getOpcode() == Instruction::Store &&
1813 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1814 true ,
nullptr , *RepR ,
1815 *RepR , RepR->getDebugLoc());
1816 Clone->insertBefore(RepOrWidenR);
1818 VPValue *ExtractOp = Clone->getOperand(0);
1824 Clone->setOperand(0, ExtractOp);
1825 RepR->eraseFromParent();
1837 VPValue *SafeDivisor = Builder.createSelect(
1838 IntrR->getOperand(2), IntrR->getOperand(1),
1840 VPValue *Clone = Builder.createNaryOp(
1841 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1844 IntrR->eraseFromParent();
1853 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1862 return !U->usesScalars(
Op);
1866 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1869 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1870 IntroducesBCastOf(Op)))
1874 bool LiveInNeedsBroadcast =
1875 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1876 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1877 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1884 RepOrWidenR->getUnderlyingInstr());
1885 Clone->insertBefore(RepOrWidenR);
1886 RepOrWidenR->replaceAllUsesWith(Clone);
1888 RepOrWidenR->eraseFromParent();
1924 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1925 UniqueValues.
insert(Blend->getIncomingValue(0));
1926 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1928 UniqueValues.
insert(Blend->getIncomingValue(
I));
1930 if (UniqueValues.
size() == 1) {
1931 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1932 Blend->eraseFromParent();
1936 if (Blend->isNormalized())
1942 unsigned StartIndex = 0;
1943 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1955 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1957 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1958 if (
I == StartIndex)
1960 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1961 OperandsWithMask.
push_back(Blend->getMask(
I));
1966 OperandsWithMask, *Blend, Blend->getDebugLoc());
1967 NewBlend->insertBefore(&R);
1969 VPValue *DeadMask = Blend->getMask(StartIndex);
1971 Blend->eraseFromParent();
1976 if (NewBlend->getNumOperands() == 3 &&
1978 VPValue *Inc0 = NewBlend->getOperand(0);
1979 VPValue *Inc1 = NewBlend->getOperand(1);
1980 VPValue *OldMask = NewBlend->getOperand(2);
1981 NewBlend->setOperand(0, Inc1);
1982 NewBlend->setOperand(1, Inc0);
1983 NewBlend->setOperand(2, NewMask);
2010 APInt MaxVal = AlignedTC - 1;
2013 unsigned NewBitWidth =
2019 bool MadeChange =
false;
2044 "canonical IV is not expected to have a truncation");
2049 NewWideIV->insertBefore(WideIV);
2056 Cmp->replaceAllUsesWith(
2057 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2071 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2073 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2087 const SCEV *VectorTripCount =
2092 "Trip count SCEV must be computable");
2113 auto *Term = &ExitingVPBB->
back();
2126 for (
unsigned Part = 0; Part < UF; ++Part) {
2132 Extracts[Part] = Ext;
2144 match(Phi->getBackedgeValue(),
2146 assert(Index &&
"Expected index from ActiveLaneMask instruction");
2163 "Expected one VPActiveLaneMaskPHIRecipe for each unroll part");
2170 "Expected incoming values of Phi to be ActiveLaneMasks");
2175 EntryALM->setOperand(2, ALMMultiplier);
2176 LoopALM->setOperand(2, ALMMultiplier);
2180 ExtractFromALM(EntryALM, EntryExtracts);
2185 ExtractFromALM(LoopALM, LoopExtracts);
2187 Not->setOperand(0, LoopExtracts[0]);
2190 for (
unsigned Part = 0; Part < UF; ++Part) {
2191 Phis[Part]->setStartValue(EntryExtracts[Part]);
2192 Phis[Part]->setBackedgeValue(LoopExtracts[Part]);
2205 auto *Term = &ExitingVPBB->
back();
2217 const SCEV *VectorTripCount =
2223 "Trip count SCEV must be computable");
2242 Term->setOperand(1, Plan.
getTrue());
2247 {}, Term->getDebugLoc());
2249 Term->eraseFromParent();
2257 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2258 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2276 RecurKind RK = PhiR->getRecurrenceKind();
2283 RecWithFlags->dropPoisonGeneratingFlags();
2289struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2298 return GEP->getSourceElementType();
2301 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2302 [](
auto *
I) {
return I->getSourceElementType(); })
2303 .
Default([](
auto *) {
return nullptr; });
2307 static bool canHandle(
const VPSingleDefRecipe *Def) {
2316 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2317 C->second == Instruction::ExtractValue)))
2321 return !
Def->mayReadOrWriteMemory();
2325 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2328 getGEPSourceElementType(Def),
Def->getScalarType(),
2331 if (RFlags->hasPredicate())
2334 return hash_combine(Result, SIVSteps->getInductionOpcode());
2339 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2340 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2343 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2345 !
equal(
L->operands(),
R->operands()))
2349 "must have valid opcode info for both recipes");
2351 if (LFlags->hasPredicate() &&
2352 LFlags->getPredicate() !=
2356 if (LSIV->getInductionOpcode() !=
2366 const VPRegionBlock *RegionL =
L->getRegion();
2367 const VPRegionBlock *RegionR =
R->getRegion();
2370 L->getParent() !=
R->getParent())
2372 return L->getScalarType() ==
R->getScalarType();
2388 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2392 if (!VPDT.
dominates(V->getParent(), VPBB))
2397 Def->replaceAllUsesWith(V);
2410 bool Sinking =
false) {
2439 "Expected vector prehader's successor to be the vector loop region");
2447 return !Op->isDefinedOutsideLoopRegions();
2450 R.moveBefore(*Preheader, Preheader->
end());
2470 assert(!RepR->isPredicated() &&
2471 "Expected prior transformation of predicated replicates to "
2472 "replicate regions");
2477 if (!RepR->isSingleScalar())
2481 if (RepR->getOpcode() == Instruction::Store &&
2482 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2487 assert((!R.mayWriteToMemory() ||
2488 (RepR && RepR->getOpcode() == Instruction::Store &&
2489 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2490 "The only recipes that may write to memory are expected to be "
2491 "stores with invariant pointer-operand");
2501 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2502 auto *UserR = cast<VPRecipeBase>(U);
2503 VPBasicBlock *Parent = UserR->getParent();
2505 if (SinkBB && SinkBB != Parent)
2510 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2511 Parent->getSinglePredecessor() != LoopRegion;
2521 "Defining block must dominate sink block");
2546 VPValue *ResultVPV = R.getVPSingleValue();
2548 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2549 if (!NewResSizeInBits)
2562 (void)OldResSizeInBits;
2570 VPW->dropPoisonGeneratingFlags();
2572 assert((OldResSizeInBits != NewResSizeInBits ||
2574 "Only ICmps should not need extending the result.");
2580 if (OldResSizeInBits != NewResSizeInBits) {
2582 Instruction::ZExt, ResultVPV, OldResTy);
2584 Ext->setOperand(0, ResultVPV);
2594 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2595 if (OpSizeInBits == NewResSizeInBits)
2597 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2598 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2604 Builder.setInsertPoint(&R);
2605 ProcessedIter->second =
2606 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2608 Op = ProcessedIter->second;
2612 NWR->insertBefore(&R);
2616 VPValue *Replacement = NWR->getVPSingleValue();
2617 if (OldResSizeInBits != NewResSizeInBits)
2623 R.eraseFromParent();
2629 std::optional<VPDominatorTree> VPDT;
2637 bool SimplifiedPhi =
false;
2647 assert(VPBB->getNumSuccessors() == 2 &&
2648 "Two successors expected for BranchOnCond");
2649 unsigned RemovedIdx;
2660 "There must be a single edge between VPBB and its successor");
2663 auto Phis = RemovedSucc->
phis();
2666 SimplifiedPhi |= !std::empty(Phis);
2670 VPBB->back().eraseFromParent();
2682 if (Reachable.contains(
B))
2693 for (
VPValue *Def : R.definedValues())
2694 Def->replaceAllUsesWith(&Tmp);
2695 R.eraseFromParent();
2699 return SimplifiedPhi;
2754 DebugLoc DL = CanonicalIVIncrement->getDebugLoc();
2765 auto *EntryIncrement =
2767 {StartV, VF}, {},
DL,
"index.part.next");
2773 {EntryIncrement, TC, ALMMultiplier},
DL,
2774 "active.lane.mask.entry");
2781 LaneMaskPhi->insertBefore(*HeaderVPBB, HeaderVPBB->begin());
2786 Builder.setInsertPoint(OriginalTerminator);
2787 auto *InLoopIncrement = Builder.createOverflowingOp(
2789 {CanonicalIVIncrement, &Plan.
getVF()}, {},
DL);
2791 {InLoopIncrement, TC, ALMMultiplier},
DL,
2792 "active.lane.mask.next");
2793 LaneMaskPhi->addBackedgeValue(ALM);
2797 auto *NotMask = Builder.createNot(ALM,
DL);
2804 VPlan &Plan,
bool UseActiveLaneMask,
bool UseActiveLaneMaskForControlFlow) {
2810 if (UseActiveLaneMaskForControlFlow) {
2816 VPBuilder Builder(Header, Header->getFirstNonPhi());
2821 if (UseActiveLaneMask) {
2824 Mask = Builder.createNaryOp(
2826 {WideCanonicalIV, Plan.
getTripCount(), ALMMultiplier},
nullptr,
2827 "active.lane.mask");
2843 "expected to run before loop regions are created");
2845 auto CanUseVersionedStride = [&VPDT, Preheader](
VPUser &U,
unsigned) {
2848 return VPDT.
dominates(Preheader, Parent);
2851 for (
const SCEV *Stride : StridesMap.
values()) {
2854 const APInt *StrideConst;
2877 RewriteMap[StrideV] = PSE.
getSCEV(StrideV);
2884 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2887 if (NewSCEV != ScevExpr) {
2889 ExpSCEV->replaceAllUsesWith(NewExp);
2900 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2905 while (!Worklist.
empty()) {
2908 if (!Visited.
insert(CurRec).second)
2930 RecWithFlags->isDisjoint()) {
2933 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2934 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2935 RecWithFlags->replaceAllUsesWith(New);
2936 RecWithFlags->eraseFromParent();
2939 RecWithFlags->dropPoisonGeneratingFlags();
2944 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2945 "found instruction with poison generating flags not covered by "
2946 "VPRecipeWithIRFlags");
2951 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2962 auto IsNotHeaderMask = [](
VPValue *Mask) {
2975 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2976 if (AddrDef && WidenRec->isConsecutive() &&
2977 IsNotHeaderMask(WidenRec->getMask()))
2978 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2980 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2981 if (AddrDef && IsNotHeaderMask(InterleaveRec->getMask()))
2982 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2992 const bool &EpilogueAllowed) {
2993 if (InterleaveGroups.empty())
3004 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
3011 for (
const auto *IG : InterleaveGroups) {
3014 for (
auto *Member : IG->members())
3016 StartMember = Member;
3024 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
3030 StoredValues.
push_back(StoreR->getStoredValue());
3037 bool NeedsMaskForGaps =
3038 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
3039 (!StoredValues.
empty() && !IG->isFull());
3042 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
3046 "Dead member in non-load group?");
3051 InsertPos->getAsRecipe()))
3052 InsertPos = MemberR;
3053 IRInsertPos = &InsertPos->getIngredient();
3063 VPValue *Addr = Start->getAddr();
3065 if (IG->getIndex(StartMember) != 0 ||
3073 assert(IG->getIndex(IRInsertPos) != 0 &&
3074 "index of insert position shouldn't be zero");
3078 IG->getIndex(IRInsertPos),
3082 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
3088 if (IG->isReverse()) {
3091 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
3092 ReversePtr->insertBefore(InsertPosR);
3096 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
3098 VPIG->insertBefore(InsertPosR);
3101 for (
unsigned i = 0; i < IG->getFactor(); ++i)
3104 if (!Member->getType()->isVoidTy()) {
3165 AddOp = Instruction::Add;
3166 MulOp = Instruction::Mul;
3168 AddOp =
ID.getInductionOpcode();
3169 MulOp = Instruction::FMul;
3177 Step = Builder.createScalarCast(Instruction::Trunc, Step, Ty,
DL);
3178 Start = Builder.createScalarCast(Instruction::Trunc, Start, Ty,
DL);
3187 Init = Builder.createWidenCast(Instruction::UIToFP,
Init, StepTy);
3192 Init = Builder.createNaryOp(MulOp, {
Init, SplatStep}, Flags);
3193 Init = Builder.createNaryOp(AddOp, {SplatStart,
Init}, Flags,
3211 if (R->getParent()->getEnclosingLoopRegion())
3212 Builder.setInsertPoint(R->getParent(), std::next(R->getIterator()));
3217 VF = Builder.createScalarCast(Instruction::CastOps::UIToFP, VF, StepTy,
3220 VF = Builder.createScalarZExtOrTrunc(VF, StepTy,
DL);
3222 Inc = Builder.createNaryOp(MulOp, {Step, VF}, Flags);
3229 auto *
Next = Builder.createNaryOp(AddOp, {Prev, Inc}, Flags,
3232 WidePHI->addIncoming(
Next);
3259 VPlan *Plan = R->getParent()->getPlan();
3260 VPValue *Start = R->getStartValue();
3261 VPValue *Step = R->getStepValue();
3262 VPValue *VF = R->getVFValue();
3264 assert(R->getInductionDescriptor().getKind() ==
3266 "Not a pointer induction according to InductionDescriptor!");
3267 assert(R->getScalarType()->isPointerTy() &&
"Unexpected type.");
3269 "Recipe should have been replaced");
3275 VPPhi *ScalarPtrPhi = Builder.createScalarPhi(Start,
DL,
"pointer.phi");
3279 Builder.setInsertPoint(R->getParent(), R->getParent()->getFirstNonPhi());
3282 Offset = Builder.createOverflowingOp(Instruction::Mul, {
Offset, Step});
3284 Builder.createWidePtrAdd(ScalarPtrPhi,
Offset,
DL,
"vector.gep");
3285 R->replaceAllUsesWith(PtrAdd);
3290 VF = Builder.createScalarZExtOrTrunc(VF, StepTy,
DL);
3291 VPValue *Inc = Builder.createOverflowingOp(Instruction::Mul, {Step, VF});
3294 Builder.createPtrAdd(ScalarPtrPhi, Inc,
DL,
"ptr.ind");
3301 VPValue *Start = R->getStartValue();
3302 VPValue *Step = R->getStepValue();
3303 VPValue *Index = R->getIndex();
3306 ? Builder.createScalarSExtOrTrunc(
3308 : Builder.createScalarCast(Instruction::SIToFP, Index, StepTy,
3310 switch (R->getInductionKind()) {
3312 assert(Index->getScalarType() == Start->getScalarType() &&
3313 "Index type does not match StartValue type");
3314 return R->replaceAllUsesWith(Builder.createAdd(
3315 Start, Builder.createOverflowingOp(Instruction::Mul, {Index, Step})));
3318 return R->replaceAllUsesWith(Builder.createPtrAdd(
3319 Start, Builder.createOverflowingOp(Instruction::Mul, {Index, Step})));
3324 (FPBinOp->
getOpcode() == Instruction::FAdd ||
3325 FPBinOp->
getOpcode() == Instruction::FSub) &&
3326 "Original BinOp should be defined for FP induction");
3328 VPValue *
FMul = Builder.createNaryOp(Instruction::FMul, {Step, Index}, FMF);
3329 return R->replaceAllUsesWith(
3330 Builder.createNaryOp(FPBinOp->
getOpcode(), {Start, FMul}, FMF));
3343 if (!R->isReplicator())
3347 R->dissolveToCFGLoop();
3368 assert(Br->getNumOperands() == 2 &&
3369 "BranchOnTwoConds must have exactly 2 conditions");
3373 assert(Successors.size() == 3 &&
3374 "BranchOnTwoConds must have exactly 3 successors");
3379 VPValue *Cond0 = Br->getOperand(0);
3380 VPValue *Cond1 = Br->getOperand(1);
3387 if (Succ0 == Succ1) {
3389 VPValue *Combined = Builder.createOr(Cond0, Cond1,
DL);
3393 Br->eraseFromParent();
3398 !BrOnTwoCondsBB->
getParent() &&
"regions must already be dissolved");
3411 Br->eraseFromParent();
3422 WidenIVR->eraseFromParent();
3432 WidenIVR->replaceAllUsesWith(PtrAdd);
3433 WidenIVR->eraseFromParent();
3437 WidenIVR->eraseFromParent();
3443 DerivedIVR->eraseFromParent();
3448 VPValue *CanIV = WideCanIV->getCanonicalIV();
3450 VPValue *Step = WideCanIV->getStepValue();
3453 "Expected unroller to have materialized step for UF != 1");
3458 Step = Builder.createAdd(
3461 Builder.createAdd(CanIV, Step, WideCanIV->getDebugLoc(),
"vec.iv",
3462 WideCanIV->getNoWrapFlags());
3464 WideCanIV->eraseFromParent();
3471 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
3472 Select = Builder.createSelect(Blend->getMask(
I),
3473 Blend->getIncomingValue(
I),
Select,
3474 R.getDebugLoc(),
"predphi", *Blend);
3475 Blend->replaceAllUsesWith(
Select);
3476 Blend->eraseFromParent();
3481 if (!VEPR->getOffset()) {
3483 "Expected unroller to have materialized offset for UF != 1");
3484 VEPR->materializeOffset();
3491 Expr->eraseFromParent();
3501 for (
VPValue *
Op : LastActiveL->operands()) {
3502 VPValue *NotMask = Builder.createNot(
Op, LastActiveL->getDebugLoc());
3507 VPValue *FirstInactiveLane = Builder.createFirstActiveLane(
3508 NotMasks, LastActiveL->getDebugLoc(),
"first.inactive.lane");
3514 Builder.createSub(FirstInactiveLane, One,
3515 LastActiveL->getDebugLoc(),
"last.active.lane");
3518 LastActiveL->eraseFromParent();
3525 assert(VPI->isMasked() &&
3526 "Unmasked MaskedCond should be simplified earlier");
3527 VPI->replaceAllUsesWith(Builder.createNaryOp(
3529 VPI->eraseFromParent();
3539 Instruction::Add, VPI->operands(), VPI->getNoWrapFlags(),
3540 VPI->getDebugLoc());
3541 VPI->replaceAllUsesWith(
Add);
3542 VPI->eraseFromParent();
3550 DebugLoc DL = BranchOnCountInst->getDebugLoc();
3553 BranchOnCountInst->eraseFromParent();
3568 ? Instruction::UIToFP
3569 : Instruction::Trunc;
3570 VectorStep = Builder.createWidenCast(CastOp, VectorStep, IVTy);
3576 Builder.createWidenCast(Instruction::Trunc, ScalarStep, IVTy);
3582 MulOpc = Instruction::FMul;
3583 Flags = VPI->getFastMathFlagsOrNone();
3585 MulOpc = Instruction::Mul;
3590 MulOpc, {VectorStep, ScalarStep}, Flags, R.getDebugLoc());
3592 VPI->replaceAllUsesWith(VectorStep);
3593 VPI->eraseFromParent();
3603static std::optional<VPValue *>
3656 VPValue *UncountableCondition =
nullptr;
3660 return std::nullopt;
3663 Worklist.
push_back(UncountableCondition);
3664 while (!Worklist.
empty()) {
3668 if (V->isDefinedOutsideLoopRegions())
3674 if (V->getNumUsers() > 1)
3675 return std::nullopt;
3687 return std::nullopt;
3691 return std::nullopt;
3699 return std::nullopt;
3707 return std::nullopt;
3709 return UncountableCondition;
3765 for (
auto &Exit : Exits) {
3766 if (Exit.EarlyExitingVPBB == LatchVPBB)
3770 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3771 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3782 std::optional<VPValue *>
Cond =
3798 assert(
Load &&
"Couldn't find exactly one load");
3801 "Uncountable exit condition load is conditional.");
3815 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3839 while (InsertIt != HeaderVPBB->
end() &&
3841 erase(ConditionRecipes, &*InsertIt);
3844 for (
auto *Recipe :
reverse(ConditionRecipes))
3845 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3849 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3851 Type *IVScalarTy =
IV->getScalarType();
3857 {Zero, FirstActive, ALMMultiplier},
3858 DebugLoc(),
"uncountable.exit.mask");
3863 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3865 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3875 "Expected BranchOnCond terminator for MiddleVPBB");
3886 auto Phis = ScalarPH->
phis();
3896 "Continuing from different IV");
3912 if (Pred == MiddleVPBB)
3917 VPValue *CondOfEarlyExitingVPBB;
3918 [[maybe_unused]]
bool Matched =
3919 match(EarlyExitingVPBB->getTerminator(),
3921 assert(Matched &&
"Terminator must be BranchOnCond");
3925 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3926 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3928 TrueSucc == ExitBlock
3929 ? CondOfEarlyExitingVPBB
3930 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3936 "exit condition must dominate the latch");
3945 assert(!Exits.
empty() &&
"must have at least one early exit");
3952 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3955 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3961 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3962 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3964 Exits[
I].EarlyExitingVPBB) &&
3965 "RPO sort must place dominating exits before dominated ones");
3971 VPValue *Combined = Exits[0].CondToExit;
3984 "Unexpected terminator");
3985 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3986 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3987 LatchExitingBranch->eraseFromParent();
3990 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3996 LatchVPBB->
setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
4000 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
4005 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
4009 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
4017 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
4020 LatchVPBB->
setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
4052 for (
auto [Exit, VectorEarlyExitVPBB] :
4053 zip_equal(Exits, VectorEarlyExitVPBBs)) {
4054 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
4066 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
4067 VPValue *NewIncoming = IncomingVal;
4069 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
4074 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
4075 ExitIRI->addIncoming(NewIncoming);
4078 EarlyExitingVPBB->getTerminator()->eraseFromParent();
4112 bool IsLastDispatch = (
I + 2 == Exits.
size());
4114 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
4120 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
4123 CurrentBB = FalseBB;
4138 VPValue *VecOp = Red->getVecOp();
4140 assert(!Red->isPartialReduction() &&
4141 "This path does not support partial reductions");
4144 auto IsExtendedRedValidAndClampRange =
4157 "getExtendedReductionCost only supports integer types");
4158 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
4159 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
4160 Red->getFastMathFlagsOrNone(),
CostKind);
4161 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
4169 IsExtendedRedValidAndClampRange(
4190 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
4191 Opcode != Instruction::FAdd)
4194 assert(!Red->isPartialReduction() &&
4195 "This path does not support partial reductions");
4199 auto IsMulAccValidAndClampRange =
4211 (Ext0->getOpcode() != Ext1->getOpcode() ||
4212 Ext0->getOpcode() == Instruction::CastOps::FPExt))
4216 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
4218 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
4225 ExtCost += Ext0->computeCost(VF, Ctx);
4227 ExtCost += Ext1->computeCost(VF, Ctx);
4229 ExtCost += OuterExt->computeCost(VF, Ctx);
4231 return MulAccCost.
isValid() &&
4232 MulAccCost < ExtCost + MulCost + RedCost;
4237 VPValue *VecOp = Red->getVecOp();
4275 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
4277 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
4278 Mul->setOperand(1, ExtB);
4288 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
4293 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
4300 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
4317 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
4326 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
4327 Ext0->getOpcode() == Ext1->getOpcode() &&
4328 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
4330 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
4331 *Ext0, *Ext0, Ext0->getDebugLoc());
4332 NewExt0->insertBefore(Ext0);
4337 Ext->getScalarType(),
nullptr, *Ext1,
4338 *Ext1, Ext1->getDebugLoc());
4341 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
4342 NewMul->insertBefore(
Mul);
4343 Ext->replaceAllUsesWith(NewMul);
4344 Ext->eraseFromParent();
4345 Mul->eraseFromParent();
4359 assert(!Red->isPartialReduction() &&
4360 "This path does not support partial reductions");
4363 auto IP = std::next(Red->getIterator());
4364 auto *VPBB = Red->getParent();
4374 Red->replaceAllUsesWith(AbstractR);
4404 for (
VPValue *VPV : VPValues) {
4412 if (
User->usesScalars(VPV))
4415 HoistPoint = HoistBlock->
begin();
4419 "All users must be in the vector preheader or dominated by it");
4424 VPV->replaceUsesWithIf(Broadcast,
4425 [VPV, Broadcast](
VPUser &U,
unsigned Idx) {
4426 return Broadcast != &U && !U.usesScalars(VPV);
4437 return CommonMetadata;
4440template <
unsigned Opcode>
4445 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
4446 "Only Load and Store opcodes supported");
4447 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
4454 for (
auto Recipes :
Groups) {
4455 if (Recipes.size() < 2)
4460 "Expected all recipes in group to have the same load-store type");
4467 VPValue *MaskI = RecipeI->getMask();
4473 bool HasComplementaryMask =
false;
4478 VPValue *MaskJ = RecipeJ->getMask();
4487 if (HasComplementaryMask) {
4488 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
4498template <
typename InstType>
4516 for (
auto &Group :
Groups) {
4536 return R->isSingleScalar() == IsSingleScalar;
4538 "all members in group must agree on IsSingleScalar");
4543 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
4544 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
4546 UnpredicatedLoad->insertBefore(EarliestLoad);
4550 Load->replaceAllUsesWith(UnpredicatedLoad);
4551 Load->eraseFromParent();
4560 if (!StoreLoc || !StoreLoc->AATags.Scope)
4567 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
4579 for (
auto &Group :
Groups) {
4592 VPValue *SelectedValue = Group[0]->getOperand(0);
4595 bool IsSingleScalar = Group[0]->isSingleScalar();
4596 for (
unsigned I = 1;
I < Group.size(); ++
I) {
4597 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
4598 "all members in group must agree on IsSingleScalar");
4599 VPValue *Mask = Group[
I]->getMask();
4601 SelectedValue = Builder.createSelect(Mask,
Value, SelectedValue,
4610 StoreWithMinAlign->getUnderlyingInstr(),
4611 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
4612 nullptr, *LastStore, CommonMetadata);
4613 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
4617 Store->eraseFromParent();
4624 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
4625 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
4688 auto UsesVectorOrInsideReplicateRegion = [DefR, LoopRegion](
VPUser *U) {
4690 return !U->usesScalars(DefR) || ParentRegion != LoopRegion;
4692 if (
none_of(DefR->users(), UsesVectorOrInsideReplicateRegion))
4702 DefR->replaceUsesWithIf(
4703 BuildVector, [BuildVector, &UsesVectorOrInsideReplicateRegion](
4705 return &U != BuildVector && UsesVectorOrInsideReplicateRegion(&U);
4719 for (
VPValue *Def : R.definedValues()) {
4729 unsigned NumFirstLaneUsers =
count_if(Def->users(), [&Def](
VPUser *U) {
4730 return U->usesFirstLaneOnly(Def);
4732 if (!NumFirstLaneUsers || NumFirstLaneUsers == Def->getNumUsers())
4739 Unpack->insertAfter(&R);
4740 Def->replaceUsesWithIf(Unpack, [&Def](
VPUser &U,
unsigned) {
4741 return U.usesFirstLaneOnly(Def);
4750 bool RequiresScalarEpilogue,
VPValue *Step,
4751 std::optional<uint64_t> MaxRuntimeStep) {
4763 "Step VPBB must dominate VectorPHVPBB");
4765 InsertPt = std::next(StepR->getIterator());
4767 VPBuilder Builder(VectorPHVPBB, InsertPt);
4773 if (!RequiresScalarEpilogue &&
match(TC,
m_APInt(TCVal)) && MaxRuntimeStep &&
4774 TCVal->
urem(*MaxRuntimeStep) == 0) {
4785 if (TailByMasking) {
4786 TC = Builder.createAdd(
4797 Builder.createNaryOp(Instruction::URem, {TC, Step},
4806 if (RequiresScalarEpilogue) {
4808 "requiring scalar epilogue is not supported with fail folding");
4811 R = Builder.createSelect(IsZero, Step, R);
4825 "VF and VFxUF must be materialized together");
4837 Builder.createElementCount(TCTy, VFEC * Plan.
getConcreteUF());
4844 VPValue *RuntimeVF = Builder.createElementCount(TCTy, VFEC);
4848 BC, [&VF](
VPUser &U,
unsigned) {
return !U.usesScalars(&VF); });
4852 VPValue *MulByUF = Builder.createOverflowingOp(
4865 auto *AliasMask = Builder.createNaryOp(
4870 Builder =
VPBuilder(Header, Header->getFirstNonPhi());
4873 auto *ClampedHeaderMask = Builder.createAnd(HeaderMask, AliasMask);
4875 return &U != ClampedHeaderMask;
4886 assert(IncomingAliasMask &&
"Expected an alias mask!");
4896 if (
Check.NeedsFreeze) {
4906 Intrinsic::loop_dependence_war_mask,
4910 AliasMask = Builder.createAnd(AliasMask, WARMask);
4912 AliasMask = WARMask;
4917 VPValue *NumActive = Builder.createNaryOp(
4920 VPValue *ClampedVF = Builder.createScalarZExtOrTrunc(
4946 VPValue *DistanceToMax = Builder.createSub(MaxUIntTripCount, TripCount);
4954 VPValue *TripCountCheck = Builder.createICmp(
4957 VPValue *
Cond = Builder.createOr(IsScalar, TripCountCheck,
DL);
4968 "Clamped VF not supported with interleaving");
4976 VPBuilder Builder(Entry, Entry->begin());
4988 if (!ExpSCEV || ExpSCEV->user_empty())
4990 Builder.setInsertPoint(ExpSCEV);
4999 ExpSCEV->eraseFromParent();
5008 BasicBlock *EntryBB = Entry->getIRBasicBlock();
5015 const SCEV *Expr = ExpSCEV->getSCEV();
5018 ExpandedSCEVs[Expr] = Res;
5023 ExpSCEV->eraseFromParent();
5026 "all VPExpandSCEVRecipes must have been expanded");
5029 auto EI = Entry->begin();
5039 return ExpandedSCEVs;
5053 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
5058 if (Member0Op == OpV)
5068 return !IsScalable && !W->getMask() && W->isConsecutive() &&
5071 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
5086 if (R->getScalarType() != WideMember0->getScalarType())
5088 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
5092 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
5095 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
5100 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
5101 const auto &[
OpIdx, OpV] =
P;
5113static std::optional<ElementCount>
5117 if (!InterleaveR || InterleaveR->
getMask())
5118 return std::nullopt;
5120 Type *GroupElementTy =
nullptr;
5124 return Op->getScalarType() == GroupElementTy;
5126 return std::nullopt;
5130 return Op->getScalarType() == GroupElementTy;
5132 return std::nullopt;
5136 if (IG->getFactor() != IG->getNumMembers())
5137 return std::nullopt;
5143 assert(
Size.isScalable() == VF.isScalable() &&
5144 "if Size is scalable, VF must be scalable and vice versa");
5145 return Size.getKnownMinValue();
5149 unsigned MinVal = VF.getKnownMinValue();
5151 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
5154 return std::nullopt;
5162 return RepR && RepR->isSingleScalar();
5176 if (V->isDefinedOutsideLoopRegions()) {
5179 return M->isDefinedOutsideLoopRegions() &&
5180 M->getScalarType() == V->getScalarType();
5182 "expected distinct loop-invariant values of matching scalar type");
5197 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
5199 for (
VPValue *Member : Members)
5200 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
5201 WideMember0->setOperand(
5210 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
5212 LoadGroup->getMask(),
true,
5213 *LoadGroup, LoadGroup->getDebugLoc());
5214 L->insertBefore(LoadGroup);
5220 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
5221 "must be a single scalar load");
5222 NarrowedOps.
insert(RepR);
5227 VPValue *PtrOp = WideLoad->getAddr();
5229 PtrOp = VecPtr->getOperand(0);
5234 nullptr, {}, *WideLoad);
5235 N->insertBefore(WideLoad);
5240std::unique_ptr<VPlan>
5260 "unexpected branch-on-count");
5263 std::optional<ElementCount> VFToOptimize;
5277 if (R.mayWriteToMemory() && !InterleaveR)
5283 return any_of(V->users(), [&](VPUser *U) {
5284 auto *UR = cast<VPRecipeBase>(U);
5285 return UR->getParent()->getParent() != VectorLoop;
5302 std::optional<ElementCount> NarrowedVF =
5304 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
5306 VFToOptimize = NarrowedVF;
5309 if (InterleaveR->getStoredValues().empty())
5314 auto *Member0 = InterleaveR->getStoredValues()[0];
5324 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
5327 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
5328 return IR && IR->getInterleaveGroup()->isFull() &&
5329 IR->getVPValue(Op.index()) == Op.value();
5338 VFToOptimize->isScalable()))
5343 if (StoreGroups.empty())
5347 bool RequiresScalarEpilogue =
5358 std::unique_ptr<VPlan> NewPlan;
5360 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
5361 Plan.
setVF(*VFToOptimize);
5362 NewPlan->removeVF(*VFToOptimize);
5369 for (
auto *StoreGroup : StoreGroups) {
5371 NarrowedOps, Preheader);
5376 StoreGroup->getDebugLoc());
5377 S->insertBefore(StoreGroup);
5378 StoreGroup->eraseFromParent();
5384 Type *CanIVTy = VectorLoop->getCanonicalIVType();
5390 if (VFToOptimize->isScalable()) {
5393 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
5401 materializeVectorTripCount(Plan, VectorPH,
false,
5402 RequiresScalarEpilogue, Step);
5407 removeDeadRecipes(Plan);
5410 "All VPVectorPointerRecipes should have been removed");
5426 "must have a BranchOnCond");
5429 if (VF.
isScalable() && VScaleForTuning.has_value())
5430 VectorStep *= *VScaleForTuning;
5431 assert(VectorStep > 0 &&
"trip count should not be zero");
5435 MiddleTerm->setMetadata(LLVMContext::MD_prof, BranchWeights);
5454 "Cannot handle loops with uncountable early exits");
5461 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
5468 if (
any_of(RecurSplice->users(),
5469 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
5550 {},
"vector.recur.extract.for.phi");
5553 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
5567 VPValue *WidenIVCandidate = BinOp->getOperand(0);
5568 VPValue *InvariantCandidate = BinOp->getOperand(1);
5570 std::swap(WidenIVCandidate, InvariantCandidate);
5584 auto *ClonedOp = BinOp->
clone();
5585 if (ClonedOp->getOperand(0) == WidenIV) {
5586 ClonedOp->setOperand(0, ScalarIV);
5588 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
5589 ClonedOp->setOperand(1, ScalarIV);
5604 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
5605 bool UseMax) -> std::optional<APSInt> {
5607 for (
bool Signed : {
true,
false}) {
5616 return std::nullopt;
5624 PhiR->getRecurrenceKind()))
5633 VPValue *BackedgeVal = PhiR->getBackedgeValue();
5647 !
match(FindLastSelect,
5656 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
5662 "IVOfExpressionToSink not being an AddRec must imply "
5663 "FindLastExpression not being an AddRec.");
5674 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
5675 bool UseSigned = SentinelVal && SentinelVal->isSigned();
5682 if (IVOfExpressionToSink) {
5683 const SCEV *FindLastExpressionSCEV =
5685 if (
match(FindLastExpressionSCEV,
5688 if (
auto NewSentinel =
5689 CheckSentinel(FindLastExpressionSCEV, NewUseMax)) {
5692 SentinelVal = *NewSentinel;
5693 UseSigned = NewSentinel->isSigned();
5695 IVSCEV = FindLastExpressionSCEV;
5696 IVOfExpressionToSink =
nullptr;
5706 if (AR->hasNoSignedWrap())
5708 else if (AR->hasNoUnsignedWrap())
5718 VPValue *NewFindLastSelect = BackedgeVal;
5720 if (!SentinelVal || IVOfExpressionToSink) {
5723 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
5724 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
5725 if (FindLastSelect->getDefiningRecipe()->getOperand(1) == PhiR)
5726 SelectCond = LoopBuilder.
createNot(SelectCond);
5733 if (SelectCond !=
Cond || IVOfExpressionToSink) {
5736 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
5745 VPIRFlags Flags(MinMaxKind,
false,
false,
5751 NewFindLastSelect, Flags, ExitDL);
5754 VPValue *VectorRegionExitingVal = ReducedIV;
5755 if (IVOfExpressionToSink)
5756 VectorRegionExitingVal =
5758 ReducedIV, IVOfExpressionToSink);
5761 VPValue *StartVPV = PhiR->getStartValue();
5768 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
5778 AnyOfPhi->insertAfter(PhiR);
5785 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
5798 PhiR->hasUsesOutsideReductionChain());
5799 NewPhiR->insertBefore(PhiR);
5800 PhiR->replaceAllUsesWith(NewPhiR);
5801 PhiR->eraseFromParent();
5808struct ReductionExtend {
5809 Type *SrcType =
nullptr;
5810 ExtendKind Kind = ExtendKind::PR_None;
5816struct ExtendedReductionOperand {
5820 ReductionExtend ExtendA, ExtendB;
5828struct VPPartialReductionChain {
5831 VPWidenRecipe *ReductionBinOp =
nullptr;
5833 ExtendedReductionOperand ExtendedOp;
5840 unsigned AccumulatorOpIdx;
5841 unsigned ScaleFactor;
5844 VPBlendRecipe *Blend =
nullptr;
5849static std::optional<unsigned>
5853 "Expected a non-normalized blend with two incoming values");
5859 return std::nullopt;
5860 return FirstIncomingHasOneUse ? 0 : 1;
5872 if (!
Op->hasOneUse() ||
5878 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
5879 Op->getOperand(1), NarrowTy);
5881 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
5890 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
5892 assert(Ext->getOpcode() ==
5894 "Expected both the LHS and RHS extends to be the same");
5895 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
5898 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
5899 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
5900 auto *
Max = Builder.insert(
5902 {FreezeX, FreezeY}, SrcTy));
5903 auto *Min = Builder.insert(
5905 {FreezeX, FreezeY}, SrcTy));
5908 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
5909 Op->getScalarType());
5921 if (!
Mul->hasOneUse() ||
5922 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
5923 MulLHS->getOpcode() != MulRHS->getOpcode())
5926 auto *NewLHS = Builder.createWidenCast(
5927 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
5928 auto *NewRHS = MulLHS == MulRHS
5930 : Builder.createWidenCast(MulRHS->getOpcode(),
5931 MulRHS->getOperand(0),
5932 Ext->getScalarType());
5933 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
5934 Builder.insert(NewMul);
5935 Op->replaceAllUsesWith(NewMul);
5936 Op->eraseFromParent();
5937 Mul->eraseFromParent();
5946 VPValue *VecOp = Red->getVecOp();
6000static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
6008 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
6011 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
6027 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
6029 (WidenRecipe->
getOpcode() == Instruction::FSub &&
6034 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
6044 Builder.insert(NegRecipe);
6045 ExtendedOp = NegRecipe;
6060 std::optional<unsigned> BlendReductionIdx =
6061 getBlendReductionUpdateValueIdx(Chain.Blend);
6062 assert(BlendReductionIdx &&
6064 "Expected blend to contain the reduction update");
6075 assert((!ExitValue || IsLastInChain) &&
6076 "if we found ExitValue, it must match RdxPhi's backedge value");
6087 PartialRed->insertBefore(WidenRecipe);
6097 E->insertBefore(WidenRecipe);
6098 PartialRed->replaceAllUsesWith(
E);
6111 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
6112 StartInst->setOperand(2, NewScaleFactor);
6120 VPValue *OldStartValue = StartInst->getOperand(0);
6121 StartInst->setOperand(0, StartInst->getOperand(1));
6125 assert(RdxResult &&
"Could not find reduction result");
6128 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
6129 : Instruction::BinaryOps::Sub;
6135 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
6141 const VPPartialReductionChain &Link,
6144 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
6145 std::optional<unsigned> BinOpc = std::nullopt;
6147 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
6148 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
6150 std::optional<llvm::FastMathFlags>
Flags;
6154 auto GetLinkOpcode = [&Link]() ->
unsigned {
6157 return Instruction::Add;
6159 return Instruction::FAdd;
6161 return Link.ReductionBinOp->
getOpcode();
6166 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
6167 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
6188static std::optional<ExtendedReductionOperand>
6191 "Op should be operand of UpdateR");
6199 if (
Op->hasOneUse() &&
6208 Type *RHSInputType =
Y->getScalarType();
6209 if (LHSInputType != RHSInputType ||
6210 LHSExt->getOpcode() != RHSExt->getOpcode())
6211 return std::nullopt;
6214 return ExtendedReductionOperand{
6216 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
6220 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
6223 VPValue *CastSource = CastRecipe->getOperand(0);
6224 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
6234 return ExtendedReductionOperand{
6241 if (!
Op->hasOneUse())
6242 return std::nullopt;
6247 return std::nullopt;
6257 return std::nullopt;
6261 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
6264 const APInt *RHSConst =
nullptr;
6270 return std::nullopt;
6274 if (Cast && OuterExtKind &&
6275 getPartialReductionExtendKind(Cast) != OuterExtKind)
6276 return std::nullopt;
6278 Type *RHSInputType = LHSInputType;
6279 ExtendKind RHSExtendKind = LHSExtendKind;
6282 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
6285 return ExtendedReductionOperand{
6286 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
6293static std::optional<SmallVector<VPPartialReductionChain>>
6300 return std::nullopt;
6310 VPValue *CurrentValue = ExitValue;
6311 while (CurrentValue != RedPhiR) {
6313 std::optional<unsigned> BlendReductionIdx;
6317 return std::nullopt;
6319 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
6320 if (!BlendReductionIdx)
6321 return std::nullopt;
6328 return std::nullopt;
6335 std::optional<ExtendedReductionOperand> ExtendedOp =
6336 matchExtendedReductionOperand(UpdateR,
Op);
6338 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
6340 return std::nullopt;
6348 return std::nullopt;
6350 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
6353 return std::nullopt;
6355 VPPartialReductionChain Link(
6356 {UpdateR, *ExtendedOp, RK,
6361 CurrentValue = PrevValue;
6366 std::reverse(Chain.
begin(), Chain.
end());
6385 if (
auto Chains = getScaledReductions(RedPhiR))
6386 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
6389 if (ChainsByPhi.
empty())
6397 for (
const auto &[
_, Chains] : ChainsByPhi)
6398 for (
const VPPartialReductionChain &Chain : Chains) {
6399 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
6401 PartialReductionBlends.
insert(Chain.Blend);
6402 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
6408 auto ExtendUsersValid = [&](
VPValue *Ext) {
6410 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
6414 auto IsProfitablePartialReductionChainForVF =
6421 for (
const VPPartialReductionChain &Link : Chain) {
6422 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
6423 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
6427 PartialCost += LinkCost;
6428 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
6430 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
6431 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
6434 RegularCost += Extend->computeCost(VF, CostCtx);
6436 return PartialCost.
isValid() && PartialCost < RegularCost;
6444 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
6445 for (
const VPPartialReductionChain &Chain : Chains) {
6446 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
6450 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
6452 return PhiR == RedPhiR;
6456 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
6458 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
6464 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
6473 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
6474 return RepR && RepR->getOpcode() == Instruction::Store;
6485 return IsProfitablePartialReductionChainForVF(Chains, VF);
6491 for (
auto &[Phi, Chains] : ChainsByPhi)
6492 for (
const VPPartialReductionChain &Chain : Chains)
6493 transformToPartialReduction(Chain, Plan, Phi);
6522 if (VPI && VPI->getUnderlyingValue() &&
6533 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
6536 if (!ProcessVPInst(VPI))
6545 New->insertBefore(VPI);
6546 if (VPI->
getOpcode() == Instruction::Load)
6561 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
6563 VPI, FinalRedStoresBuilder))
6572 return ReplaceWith(VPI, Histogram);
6585 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
6589 return Scalarize(VPI);
6596 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
6598 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
6608 const SCEV *PtrSCEV =
6610 bool IsSingleScalarLoad =
6616 I, Ptr, IsSingleScalarLoad,
6624 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
6629 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
6642 VectorPtr->insertBefore(VPI);
6653 return ReplaceWith(VPI, WidenedR);
6660 return ReplaceWith(VPI, Recipe);
6662 return Scalarize(VPI);
6685 if (VPI->mayHaveSideEffects())
6689 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
6694 if (VPI->getOpcode() == Instruction::Add &&
6703 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
6704 *VPI, VPI->getDebugLoc(),
I);
6705 Recipe->insertBefore(VPI);
6706 VPI->replaceAllUsesWith(Recipe);
6707 VPI->eraseFromParent();
6717 switch (Param.ParamKind) {
6718 case VFParamKind::Vector:
6719 case VFParamKind::GlobalPredicate:
6721 case VFParamKind::OMP_Uniform:
6722 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
6723 SE->isLoopInvariant(
6724 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
6726 case VFParamKind::OMP_Linear:
6727 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
6728 m_scev_AffineAddRec(
6729 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
6730 m_SpecificLoop(L)));
6747 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
6748 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
6751 if (It == Mappings.end())
6758struct CallWideningDecision {
6759 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
6760 CallWideningDecision(KindTy Kind, Function *Variant =
nullptr)
6783 return CallWideningDecision::KindTy::Scalarize;
6793 return CallWideningDecision::KindTy::Scalarize;
6797 false, VF, CostCtx);
6812 return CallWideningDecision::KindTy::Intrinsic;
6816 if (VecFunc && ScalarCost >= VecCallCost)
6817 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
6819 return CallWideningDecision::KindTy::Scalarize;
6829 if (!VPI || !VPI->getUnderlyingValue() ||
6830 VPI->getOpcode() != Instruction::Call)
6835 VPI->op_begin() + CI->arg_size());
6837 CallWideningDecision Decision =
6846 switch (Decision.Kind) {
6847 case CallWideningDecision::KindTy::Intrinsic: {
6851 *VPI, VPI->getDebugLoc());
6854 case CallWideningDecision::KindTy::VectorVariant: {
6858 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
6859 Ops.push_back(Mask);
6861 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
6863 *VPI, VPI->getDebugLoc());
6866 case CallWideningDecision::KindTy::Scalarize:
6872 VPI->replaceAllUsesWith(Replacement);
6873 VPI->eraseFromParent();
6896 if (!LoadR || LoadR->isConsecutive())
6899 VPValue *Ptr = LoadR->getAddr();
6912 Align Alignment = LoadR->getAlign();
6915 if (!Ctx.TTI.isLegalStridedLoadStore(DataTy, Alignment))
6920 Intrinsic::experimental_vp_strided_load, DataTy,
6921 LoadR->isMasked(), Alignment, Ctx);
6922 return StridedLoadStoreCost < CurrentCost;
6933 Ctx.invalidateWideningDecision(&LoadR->getIngredient(), VF);
6938 I32VF = Builder.createScalarZExtOrTrunc(
6955 "Stride type from SCEV must match the index type");
6956 VPValue *CanIV = Builder.createScalarSExtOrTrunc(
6959 auto *
Offset = Builder.createOverflowingOp(
6960 Instruction::Mul, {CanIV, StrideInBytes},
6961 {AddRecPtr->hasNoUnsignedWrap(),
false});
6965 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
6968 VPValue *NewPtr = Builder.createVectorPointer(
6970 LoadR->getDebugLoc());
6972 VPValue *Mask = LoadR->getMask();
6975 auto *StridedLoad = Builder.createWidenMemIntrinsic(
6976 Intrinsic::experimental_vp_strided_load,
6977 {NewPtr, StrideInBytes, Mask, I32VF}, LoadTy, Alignment, *LoadR,
6978 LoadR->getDebugLoc());
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static const Function * getParent(const Value *V)
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
static std::pair< Value *, APInt > getMask(Value *WideMask, unsigned Factor, ElementCount LeafValueEC)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
MachineInstr unsigned OpIdx
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
static bool dominates(InstrPosIndexes &PosIndexes, const MachineInstr &A, const MachineInstr &B)
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file declares the class VPlanVerifier, which contains utility functions to check the consistency...
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getCompilerGenerated()
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
constexpr bool isVector() const
One or more elements.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Utility class for floating point operations which can have information about relaxed accuracy require...
FastMathFlags getFastMathFlags() const
Convenience function for getting all the fast-math flags.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
static LLVM_ABI InductionDescriptor getCanonicalIntInduction(Type *Ty, ScalarEvolution &SE)
Returns the canonical integer induction for type Ty with start = 0 and step = 1.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_NoInduction
Not an induction variable.
@ IK_FpInduction
Floating point induction variable.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
LLVM_ABI MDNode * createBranchWeights(uint32_t TrueWeight, uint32_t FalseWeight, bool IsExpected=false)
Return metadata containing two branch weights.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
This class uses information about analyze scalars to rewrite expressions in canonical form.
LLVM_ABI Value * expandCodeFor(SCEVUse SH, Type *Ty, BasicBlock::iterator I)
Insert code to directly compute the specified SCEV expression into the program.
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This class represents an analyzed expression in the program.
LLVM_ABI Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
LLVM_ABI const SCEV * getUDivExpr(SCEVUse LHS, SCEVUse RHS)
Get a canonical unsigned division expression, or something simpler if possible.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNonZero(const SCEV *S)
Test if the given expression is known to be non-zero.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI const SCEV * getMulExpr(SmallVectorImpl< SCEVUse > &Ops, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Get a canonical multiply expression, or something simpler if possible.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
bool isStructTy() const
True if this is an instance of StructType.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
A recipe for generating the active lane mask for the vector loop that is used to predicate the vector...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
RecipeListTy::iterator iterator
Instruction iterators...
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
const std::string & getName() const
void clearSuccessors()
Remove all the successors of this block.
VPBlockBase * getSinglePredecessor() const
void clearPredecessors()
Remove all the predecessor of this block.
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleHierarchicalPredecessor()
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
RAII object that stores the current insertion point and restores it when the object is destroyed.
VPlan-based builder utility analogous to IRBuilder.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step)
Convert the input value Current to the corresponding value of an induction with Start and Step values...
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, const VPIRMetadata &Metadata={})
VPWidenPHIRecipe * createWidenPhi(ArrayRef< VPValue * > IncomingValues, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", const VPIRFlags &Flags={})
VPExpandSCEVRecipe * createExpandSCEV(const SCEV *Expr)
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
void setInsertPoint(VPBasicBlock *TheBB)
This specifies that created VPInstructions should be appended to the end of the specified block.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
A recipe for converting the input value IV value to the corresponding value of an IV with different s...
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode)
Returns default flags for Opcode for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
void dropPoisonGeneratingFlags()
Drop all poison-generating flags.
static LLVM_ABI_FOR_TEST VPIRInstruction * create(Instruction &I)
Create a new VPIRPhi for \I , if it is a PHINode, otherwise create a VPIRInstruction.
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ Unpack
Extracts all lanes from its (non-scalable) vector operand.
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ BuildStructVector
Given operands of (the same) struct type, creates a struct of fixed- width vectors each containing a ...
@ CanonicalIVIncrementForPart
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
void addIncoming(VPValue *IncomingV)
Append IncomingV as an incoming value to the phi-like recipe.
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
VPRegionValue * getUsedHeaderMask() const
Return the header mask if it exists and is used, or null otherwise.
VPInstruction * getOrCreateCanonicalIVIncrement()
Get the canonical IV increment instruction if it exists.
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
void clearCanonicalIVNUW(VPInstruction *Increment)
Unsets NUW for the canonical IV increment Increment, for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * tryToExpand(const SCEV *S)
Try to expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
bool isMaterialized() const
Returns true if this value has been materialized.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
void setUnderlyingValue(Value *Val)
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
unsigned getNumUsers() const
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe to compute the pointers for widened memory accesses of SourceElementTy, with the Stride expr...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
A Recipe for widening the canonical induction variable of the vector loop.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPIRValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
VPValue * getSplatVFValue() const
If the recipe has been unrolled, return the VPValue for the induction increment, otherwise return nul...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
VPValue * getLastUnrolledPartOperand()
Returns the VPValue representing the value of this induction at the last unrolled part,...
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
VPIRValue * getPoison(Type *Ty)
Return a VPIRValue wrapping a poison value of type Ty.
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
bool hasScalarTail() const
Returns true if the scalar tail may execute after the vector loop, i.e.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
@ C
The default llvm calling convention, compatible with C.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
VPInstruction_match< VPInstruction::ActiveLaneMask, Op0_t, Op1_t, Op2_t > m_ActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
VPValue * findIncomingAliasMask(const VPlan &Plan)
Finds the incoming alias-mask within the vector preheader.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool onlyScalarValuesUsed(const VPValue *Def)
Returns true if only scalar values of Def are used by all users.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
GEPNoWrapFlags getGEPFlagsForPtr(VPValue *Ptr)
Returns the GEP nowrap flags for Ptr, looking through pointer casts mirroring Value::stripPointerCast...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
SmallVector< VPRegisterUsage, 8 > calculateRegisterUsageForPlan(VPlan &Plan, ArrayRef< ElementCount > VFs, const TargetTransformInfo &TTI, const SmallPtrSetImpl< const Value * > &ValuesToIgnore)
Estimate the register usage for Plan and vectorization factors in VFs by calculating the highest numb...
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
detail::concat_range< ValueT, RangeTs... > concat(RangeTs &&...Ranges)
Returns a concatenated range across two or more ranges.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
LLVM_ABI_FOR_TEST cl::opt< bool > EnableWideActiveLaneMask
UncountableExitStyle
Different methods of handling early exits.
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto make_second_range(ContainerTy &&c)
Given a container of pairs, return a range over the second elements.
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A struct that represents some properties of the register usage of a loop.
SmallMapVector< unsigned, unsigned, 4 > MaxLocalUsers
Holds the maximum number of concurrent live intervals in the loop.
InstructionCost spillCost(const TargetTransformInfo &TTI, TargetTransformInfo::TargetCostKind CostKind, unsigned OverrideMaxNumRegs=0) const
Calculate the estimated cost of any spills due to using more registers than the number available for ...
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...