62 cl::desc(
"Use partial reduction intrinsics for "
63 "all supported unordered reductions."));
71 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
80 if (!VPBB->getParent())
83 auto EndIter = Term ? Term->getIterator() : VPBB->end();
88 VPValue *VPV = Ingredient.getVPSingleValue();
104 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
106 nullptr , IsConsecutive,
107 *VPI, Ingredient.getDebugLoc());
109 bool IsConsecutive = IsConsecutiveAccess(
110 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
112 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
113 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
116 Ingredient.operands(), *VPI,
117 Ingredient.getDebugLoc(),
GEP);
129 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
134 if (VectorID == Intrinsic::assume ||
135 VectorID == Intrinsic::lifetime_end ||
136 VectorID == Intrinsic::lifetime_start ||
137 VectorID == Intrinsic::sideeffect ||
138 VectorID == Intrinsic::pseudoprobe) {
143 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
144 VectorID != Intrinsic::pseudoprobe;
148 Ingredient.getDebugLoc());
151 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
152 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
156 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
160 *VPI, Ingredient.getDebugLoc());
164 "inductions must be created earlier");
173 "Only recpies with zero or one defined values expected");
174 Ingredient.eraseFromParent();
185 const Loop *L =
nullptr;
190 if (
A->getOpcode() != Instruction::Store ||
191 B->getOpcode() != Instruction::Store)
204 const APInt *Distance;
210 Type *TyA =
A->getOperand(0)->getScalarType();
211 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
212 Type *TyB =
B->getOperand(0)->getScalarType();
213 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
218 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
220 auto VFs =
B->getParent()->getPlan()->vectorFactors();
231 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
232 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
241 return ExcludeRecipes.contains(
Store) ||
242 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
255 std::optional<SinkStoreInfo> SinkInfo = {}) {
256 bool CheckReads = SinkInfo.has_value();
260 if (SinkInfo && SinkInfo->shouldSkip(R))
264 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
289template <
unsigned Opcode>
294 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
295 "Only Load and Store opcodes supported");
296 constexpr bool IsLoad = (Opcode == Instruction::Load);
299 RecipesByAddressAndType;
303 if (RepR.getOpcode() != Opcode || !FilterFn(&RepR))
307 VPValue *Addr = RepR.getOperand(IsLoad ? 0 : 1);
311 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(&RepR);
316 for (
auto &Group :
Groups) {
331 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
338 if (Candidate->getParent() == SinkTo ||
339 all_of(Candidate->operands(),
340 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
352 WorkList.
insert({SinkTo, Candidate});
364 for (
auto &Recipe : *VPBB)
366 InsertIfValidSinkCandidate(VPBB,
Op);
370 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
373 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
378 auto UsersOutsideSinkTo =
380 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
382 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
383 return !U->usesFirstLaneOnly(SinkCandidate);
386 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
388 if (NeedsDuplicating) {
392 if (
auto *SinkCandidateRepR =
397 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
398 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
403 Clone = SinkCandidate->
clone();
413 InsertIfValidSinkCandidate(SinkTo,
Op);
422 if (EntryBB->getNumSuccessors() != 2)
427 if (!Succ0 || !Succ1)
430 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
432 if (Succ0->getSingleSuccessor() == Succ1)
434 if (Succ1->getSingleSuccessor() == Succ0)
451 if (!Region1->isReplicator())
453 auto *MiddleBasicBlock =
455 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
460 if (!Region2 || !Region2->isReplicator())
463 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
464 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
465 if (!Mask1 || Mask1 != Mask2)
468 assert(Mask1 && Mask2 &&
"both region must have conditions");
474 if (TransformedRegions.
contains(Region1))
481 if (!Then1 || !Then2)
489 std::optional<VPExecutionFrequency> Freq1 =
492 if (Freq1 && Freq2) {
493 if (Freq2->Freq < Freq1->Freq) {
496 Freq1.emplace(Freq1->Freq, Freq1->IsEstimated || Freq2->IsEstimated);
520 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
526 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
527 Phi1ToMove.eraseFromParent();
530 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
544 TransformedRegions.
insert(Region1);
547 return !TransformedRegions.
empty();
555 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
556 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
557 auto *BlockInMask = PredRecipe->
getMask();
572 BOMRecipe->setExecutionFrequency(RecipeWithoutMask->getExecutionFrequency(),
574 RecipeWithoutMask->clearExecutionFrequency();
583 Region->setParent(ParentRegion);
589 RecipeWithoutMask->getDebugLoc());
590 Exiting->appendRecipe(PHIRecipe);
602 if (RepR.isPredicated())
620 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
632 if (!VPBB->getParent())
636 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
645 R.moveBefore(*PredVPBB, PredVPBB->
end());
647 auto *ParentRegion = VPBB->getParent();
648 if (ParentRegion && ParentRegion->getExiting() == VPBB)
649 ParentRegion->setExiting(PredVPBB);
653 return !WorkList.
empty();
660 bool ShouldSimplify =
true;
661 while (ShouldSimplify) {
678 if (
IV.getTruncInst())
693 for (
auto *U : FindMyCast->
users()) {
695 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
696 FoundUserCast = UserCast;
703 FindMyCast = FoundUserCast;
705 if (FindMyCast != &
IV)
730 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
732 PhiR->eraseFromParent();
798 Def->user_empty() || !Def->getUnderlyingValue() ||
799 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
812 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
814 Def->getScalarType(), Def->getUnderlyingInstr());
815 Clone->insertAfter(Def);
816 Def->replaceAllUsesWith(Clone);
817 Def->eraseFromParent();
832 PtrIV->replaceAllUsesWith(PtrAdd);
839 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
840 return U->usesScalars(WideIV);
851 Plan, ID.getKind(), ID.getInductionOpcode(),
853 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
854 WideIV->getDebugLoc(), Builder, WrapFlags);
857 if (!HasOnlyVectorVFs) {
859 "plans containing a scalar VF cannot also include scalable VFs");
860 WideIV->replaceAllUsesWith(Steps);
863 WideIV->replaceUsesWithIf(Steps, [WideIV, HasScalableVF](
VPUser &U) {
865 return U.usesFirstLaneOnly(WideIV);
866 return U.usesScalars(WideIV);
882 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
887 if (!Def || Def->getNumOperands() != 2)
895 auto IsWideIVInc = [&]() {
896 auto &ID = WideIV->getInductionDescriptor();
899 VPValue *IVStep = WideIV->getStepValue();
900 switch (ID.getInductionOpcode()) {
901 case Instruction::Add:
903 case Instruction::FAdd:
905 case Instruction::FSub:
908 case Instruction::Sub: {
928 return IsWideIVInc() ? WideIV :
nullptr;
952 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
954 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
955 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
960 if (Incoming != WideIV) {
962 EndValue =
B.createAdd(EndValue, One,
DL);
967 VPValue *Start = WideIV->getStartValue();
968 VPValue *Step = WideIV->getStepValue();
969 EndValue =
B.createDerivedIV(
971 Start, EndValue, Step);
985 if (WideIntOrFp && WideIntOrFp->getTruncInst())
995 Start, VectorTC, Step);
1027 assert(EndValue &&
"Must have computed the end value up front");
1032 if (Incoming != WideIV)
1044 auto *Zero = Plan.
getZero(StepTy);
1045 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1050 return B.createNaryOp(
1051 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1053 : Instruction::FAdd,
1054 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1071 const SCEV *Start, *Step;
1089 VPValue *ExitCount = Builder.createOverflowingOp(
1092 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1101 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1108 &WideIV, VectorPHBuilder, ResumeTC))
1109 EndValues[&WideIV] = EndValue;
1119 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1120 R.eraseFromParent();
1129 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1131 if (PredVPBB == MiddleVPBB) {
1133 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1136 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1139 Plan, ExitIRI->getOperand(Idx), PSE);
1142 ExitIRI->setOperand(Idx, Escape);
1156 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR.getSCEV(), &ExpR);
1160 ExpR.replaceAllUsesWith(V->second);
1164 ExpR.eraseFromParent();
1186 return Plan.
getZero(Def->getScalarType());
1203 return Def->getOperand(1);
1214 assert(
X->getScalarType()->isIntegerTy(1) &&
"must have boolean operands");
1230 assert(Weights.
size() == 2 &&
"unexpected branch weights");
1232 LLVMContext::MD_prof,
1264 return Plan.
getZero(Def->getScalarType());
1268 Def->getScalarType() ==
A->getScalarType())
1278 if (Def->getScalarType() ==
A->getScalarType())
1288 A->getScalarType() == Def->getScalarType())
1294 return Def->getOperand(0);
1300 return BuildVector->getLastOperand();
1316 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1322 return BuildVector->getOperand(Idx);
1326 if (Def->getNumOperands() == 1) {
1327 return Def->getOperand(0);
1331 return Phi->getOperand(0);
1337 if (Def->getNumOperands() == 1 &&
1343 A->getScalarType() == Def->getScalarType())
1355 return Def->getOperand(1);
1367 return VPR->getOperand(0);
1373 return Steps->getOperand(0);
1391struct VPCombineInserter {
1392 SmallVectorImpl<VPSingleDefRecipe *> &Worklist;
1394 void insertHelper(VPRecipeBase *R, VPBasicBlock *VPBB,
1408 VPCombineBuilder &Builder) {
1410 Def->replaceAllUsesWith(V);
1419 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1423 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1424 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1425 RepR->getDebugLoc());
1426 Builder.insert(Unmasked);
1438 bool CanCreateNewRecipe =
1444 if (CanCreateNewRecipe &&
1447 return Builder.createLogicalAnd(
X,
Y);
1450 if (CanCreateNewRecipe &&
1455 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1456 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1457 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1460 if (CanCreateNewRecipe &&
1464 return Builder.createLogicalOr(Z,
Y);
1468 if (CanCreateNewRecipe &&
1470 return Builder.createNot(
C);
1474 Def->setOperand(0,
C);
1475 Def->setOperand(1,
Y);
1476 Def->setOperand(2,
X);
1482 if (CanCreateNewRecipe &&
1486 Y->getScalarType()->isIntegerTy(1))
1487 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1491 if (CanCreateNewRecipe &&
1497 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1498 Def->getDebugLoc());
1504 Type *TruncTy = Def->getScalarType();
1505 Type *XTy =
X->getScalarType();
1508 unsigned ExtOpcode =
1512 if (
auto *UnderlyingExt =
Y->getUnderlyingValue()) {
1514 Ext->setUnderlyingValue(UnderlyingExt);
1518 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
X, TruncTy);
1527 return Builder.createSub(Plan.
getZero(
X->getScalarType()),
X,
1528 Def->getDebugLoc(),
"", NW);
1531 if (CanCreateNewRecipe &&
1539 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1546 Def->getDebugLoc());
1553 MulR->hasNoSignedWrap() &&
1555 return Builder.createNaryOp(
1558 Def->getDebugLoc());
1563 return Builder.createNaryOp(
1569 if (CanCreateNewRecipe &&
1572 return Builder.createAnd(
1574 Def->getDebugLoc());
1584 return match(U, m_Not(m_Specific(Cmp))) ||
1585 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1587 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1594 R->setOperand(1,
Y);
1595 R->setOperand(2,
X);
1600 R->replaceAllUsesWith(Cmp);
1605 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1606 Cmp->setDebugLoc(Def->getDebugLoc());
1619 if (
Op->getNumUsers() > 1 ||
1623 }
else if (!UnpairedCmp) {
1624 UnpairedCmp =
Op->getDefiningRecipe();
1628 UnpairedCmp =
nullptr;
1635 if (NewOps.
size() < Def->getNumOperands())
1642 if (CanCreateNewRecipe &&
1651 X->getScalarType() != Def->getScalarType())
1652 return Builder.createWidenCast(Instruction::Trunc,
X, Def->getScalarType());
1659 Def->getScalarType()->isIntegerTy(1)) {
1660 Def->setOperand(1, Plan.
getTrue());
1661 Def->setOperand(0,
Y);
1671 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U) {
1672 return U.usesFirstLaneOnly(Def);
1682 "broadcast operand must be single-scalar");
1683 Def->setOperand(0, Z);
1688 Def->replaceUsesWithIf(
1689 X, [Def](
const VPUser &U) {
return U.usesScalars(Def); });
1701 return Builder.createNaryOp(Instruction::ExtractElement, {
X, LaneToExtract},
1702 Def->getDebugLoc());
1714 IVInc->getNumUsers() == 2) {
1720 if ((Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) &&
1722 Def->replaceAllUsesWith(IVInc);
1724 Inc->replaceAllUsesWith(Phi);
1725 Phi->setOperand(0,
Y);
1734 Def->replaceUsesWithIf(StartV, [](
const VPUser &U) {
1736 return PhiR && PhiR->isInLoop();
1753 [[maybe_unused]]
unsigned InitWorklistSize = Worklist.
size();
1755 VPCombineBuilder Builder({Worklist});
1756 while (!Worklist.
empty()) {
1757 assert(Worklist.
size() < InitWorklistSize * 2 &&
1758 "Worklist is growing large, possible cycle?");
1760 Builder.setInsertPoint(Def);
1766 Def->replaceAllUsesWith(New);
1767 Def->eraseFromParent();
1771 Def->eraseFromParent();
1789 R.getVPSingleValue()->replaceAllUsesWith(
X);
1805 while (!Worklist.
empty()) {
1814 R->replaceAllUsesWith(
1815 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1819static std::optional<Instruction::BinaryOps>
1822 case Intrinsic::masked_udiv:
1823 return Instruction::UDiv;
1824 case Intrinsic::masked_sdiv:
1825 return Instruction::SDiv;
1826 case Intrinsic::masked_urem:
1827 return Instruction::URem;
1828 case Intrinsic::masked_srem:
1829 return Instruction::SRem;
1846 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1850 if (RepR && RepR->getOpcode() == Instruction::Store &&
1853 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1854 true ,
nullptr , *RepR ,
1855 *RepR , RepR->getDebugLoc());
1856 Clone->insertBefore(RepOrWidenR);
1858 VPValue *ExtractOp = Clone->getOperand(0);
1864 Clone->setOperand(0, ExtractOp);
1865 RepR->eraseFromParent();
1877 VPValue *SafeDivisor = Builder.createSelect(
1878 IntrR->getOperand(2), IntrR->getOperand(1),
1880 VPValue *Clone = Builder.createNaryOp(
1881 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1884 IntrR->eraseFromParent();
1893 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1902 return !U->usesScalars(
Op);
1906 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1909 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1910 IntroducesBCastOf(Op)))
1914 bool LiveInNeedsBroadcast =
1915 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1916 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1917 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1925 RepOrWidenR->getUnderlyingInstr());
1926 Clone->insertBefore(RepOrWidenR);
1927 RepOrWidenR->replaceAllUsesWith(Clone);
1929 RepOrWidenR->eraseFromParent();
1962 if (Blend.isNormalized() || !
match(Blend.getMask(0),
m_False()))
1963 UniqueValues.
insert(Blend.getIncomingValue(0));
1964 for (
unsigned I = 1;
I != Blend.getNumIncomingValues(); ++
I)
1966 UniqueValues.
insert(Blend.getIncomingValue(
I));
1968 if (UniqueValues.
size() == 1) {
1969 Blend.replaceAllUsesWith(*UniqueValues.
begin());
1970 Blend.eraseFromParent();
1974 if (Blend.isNormalized())
1980 unsigned StartIndex = 0;
1981 for (
unsigned I = 0;
I != Blend.getNumIncomingValues(); ++
I) {
1993 OperandsWithMask.
push_back(Blend.getIncomingValue(StartIndex));
1995 for (
unsigned I = 0;
I != Blend.getNumIncomingValues(); ++
I) {
1996 if (
I == StartIndex)
1998 OperandsWithMask.
push_back(Blend.getIncomingValue(
I));
1999 OperandsWithMask.
push_back(Blend.getMask(
I));
2004 OperandsWithMask, Blend, Blend.getDebugLoc());
2005 NewBlend->insertBefore(&Blend);
2007 VPValue *DeadMask = Blend.getMask(StartIndex);
2009 Blend.eraseFromParent();
2014 if (NewBlend->getNumOperands() == 3 &&
2016 VPValue *Inc0 = NewBlend->getOperand(0);
2017 VPValue *Inc1 = NewBlend->getOperand(1);
2018 VPValue *OldMask = NewBlend->getOperand(2);
2019 NewBlend->setOperand(0, Inc1);
2020 NewBlend->setOperand(1, Inc0);
2021 NewBlend->setOperand(2, NewMask);
2048 APInt MaxVal = AlignedTC - 1;
2051 unsigned NewBitWidth =
2057 bool MadeChange =
false;
2082 "canonical IV is not expected to have a truncation");
2087 NewWideIV->insertBefore(WideIV);
2094 Cmp->replaceAllUsesWith(
2095 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2109 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2111 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2125 const SCEV *VectorTripCount =
2130 "Trip count SCEV must be computable");
2145 bool MadeChange =
false;
2153 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2162 Builder.setInsertPoint(Extract);
2165 Start = Builder.createAdd(
2170 Extract->eraseFromParent();
2185 auto *Term = &ExitingVPBB->
back();
2191 bool MatchedCanIVInc =
2197 if (MatchedCanIVInc ||
2205 const SCEV *VectorTripCount =
2211 "Trip count SCEV must be computable");
2230 Term->setOperand(1, Plan.
getTrue());
2235 {}, Term->getDebugLoc());
2237 Term->eraseFromParent();
2245 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2246 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2262 RecurKind RK = PhiR.getRecurrenceKind();
2269 RecWithFlags->dropPoisonGeneratingFlags();
2275struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2284 return GEP->getSourceElementType();
2287 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2288 [](
auto *
I) {
return I->getSourceElementType(); })
2289 .
Default([](
auto *) {
return nullptr; });
2293 static bool canHandle(
const VPSingleDefRecipe *Def) {
2302 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2303 C->second == Instruction::ExtractValue ||
2304 C->second == Instruction::Alloca)))
2310 if (
Def->mayWriteToMemory())
2312 return !
Def->mayReadFromMemory() ||
2317 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2320 getGEPSourceElementType(Def),
Def->getScalarType(),
2323 if (RFlags->hasPredicate())
2326 return hash_combine(Result, SIVSteps->getInductionOpcode());
2335 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2336 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2339 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2341 !
equal(
L->operands(),
R->operands()))
2345 "must have valid opcode info for both recipes");
2347 if (LFlags->hasPredicate() &&
2348 LFlags->getPredicate() !=
2352 if (LSIV->getInductionOpcode() !=
2367 const VPRegionBlock *RegionL =
L->getRegion();
2368 const VPRegionBlock *RegionR =
R->getRegion();
2371 L->getParent() !=
R->getParent())
2373 return L->getScalarType() ==
R->getScalarType();
2392 if (R.mayWriteToMemory())
2395 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2398 auto [It, Inserted] =
2399 (IsLoad ? LoadCSEMap : CSEMap).try_emplace(Def, Def);
2404 if (!VPDT.
dominates(V->getParent(), VPBB))
2409 if (EarlierLoad->getAlign() <
Load->getAlign()) {
2416 EarlierLoad->intersect(*
Load);
2421 Def->replaceAllUsesWith(V);
2432 bool Sinking =
false) {
2461 "Expected vector prehader's successor to be the vector loop region");
2466 return !Op->isDefinedOutsideLoopRegions();
2472 R.moveBefore(*Preheader, Preheader->
end());
2493 assert(!RepR->isPredicated() &&
2494 "Expected prior transformation of predicated replicates to "
2495 "replicate regions");
2500 if (!RepR->isSingleScalar())
2504 if (RepR->getOpcode() == Instruction::Store &&
2505 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2513 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2514 auto *UserR = cast<VPRecipeBase>(U);
2515 VPBasicBlock *Parent = UserR->getParent();
2517 if (SinkBB && SinkBB != Parent)
2522 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2523 Parent->getSinglePredecessor() != LoopRegion;
2533 assert((!R.mayWriteToMemory() ||
2534 (RepR && RepR->getOpcode() == Instruction::Store &&
2535 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2536 "The only recipes that may write to memory are expected to be "
2537 "stores with invariant pointer-operand");
2545 "Defining block must dominate sink block");
2570 VPValue *ResultVPV = R.getVPSingleValue();
2572 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2573 if (!NewResSizeInBits)
2586 (void)OldResSizeInBits;
2594 VPW->dropPoisonGeneratingFlags();
2596 assert((OldResSizeInBits != NewResSizeInBits ||
2598 "Only ICmps should not need extending the result.");
2611 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2612 if (OpSizeInBits == NewResSizeInBits)
2614 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2615 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2619 Builder.setInsertPoint(PH);
2621 Builder.setInsertPoint(&R);
2622 ProcessedIter->second =
2623 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2625 Op = ProcessedIter->second;
2629 NWR->insertBefore(&R);
2634 VPValue *Replacement = NWR->getVPSingleValue();
2641 R.eraseFromParent();
2647 std::optional<VPDominatorTree> VPDT;
2655 bool SimplifiedPhi =
false;
2666 "Two successors expected for BranchOnCond");
2667 unsigned RemovedIdx;
2678 "There must be a single edge between VPBB and its successor");
2683 SimplifiedPhi =
true;
2687 if (!PhiR || PhiR->getNumIncoming() != 1)
2689 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2690 PhiR->eraseFromParent();
2707 if (Reachable.contains(
B))
2718 for (
VPValue *Def : R.definedValues())
2719 Def->replaceAllUsesWith(&Tmp);
2720 R.eraseFromParent();
2724 return SimplifiedPhi;
2750 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2759 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2760 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2771 "expected to run before loop regions are created");
2773 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U) {
2779 return VPDT.
dominates(Header, R->getParent());
2783 Value *StrideV = Stride->getValue();
2784 const APInt *StrideConst;
2791 CanUseVersionedStride);
2805 CanUseVersionedStride);
2807 RewriteMap[StrideV] = StrideExpr;
2812 const SCEV *ScevExpr = ExpSCEV.getSCEV();
2815 if (NewSCEV != ScevExpr) {
2817 ExpSCEV.replaceAllUsesWith(NewExp);
2828 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2833 while (!Worklist.
empty()) {
2836 if (!Visited.
insert(CurRec).second)
2858 RecWithFlags->isDisjoint()) {
2861 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2862 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2863 RecWithFlags->replaceAllUsesWith(New);
2864 RecWithFlags->eraseFromParent();
2867 RecWithFlags->dropPoisonGeneratingFlags();
2872 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2873 "found instruction with poison generating flags not covered by "
2874 "VPRecipeWithIRFlags");
2879 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2901 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2902 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2903 match(WidenRec->getMask(), m_UnlessHdrMask))
2904 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2906 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2907 if (AddrDef && InterleaveRec->getMask() &&
2908 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2909 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2919 const bool &EpilogueAllowed) {
2920 if (InterleaveGroups.empty())
2931 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2938 for (
const auto *IG : InterleaveGroups) {
2941 for (
auto *Member : IG->members())
2943 StartMember = Member;
2951 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2957 StoredValues.
push_back(StoreR->getStoredValue());
2964 bool NeedsMaskForGaps =
2965 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2966 (!StoredValues.
empty() && !IG->isFull());
2969 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2973 "Dead member in non-load group?");
2978 InsertPos->getAsRecipe()))
2979 InsertPos = MemberR;
2980 IRInsertPos = &InsertPos->getIngredient();
2990 VPValue *Addr = Start->getAddr();
2992 if (IG->getIndex(StartMember) != 0 ||
3000 assert(IG->getIndex(IRInsertPos) != 0 &&
3001 "index of insert position shouldn't be zero");
3005 IG->getIndex(IRInsertPos),
3009 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
3015 if (IG->isReverse()) {
3018 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
3019 ReversePtr->insertBefore(InsertPosR);
3023 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
3025 VPIG->insertBefore(InsertPosR);
3028 for (
unsigned i = 0; i < IG->getFactor(); ++i)
3031 if (!Member->getType()->isVoidTy()) {
3053struct CountableConditionMatch {
3055 PredicatedScalarEvolution &PSE;
3058 CountableConditionMatch(VPValue *&Cmp, PredicatedScalarEvolution &PSE,
3062 template <
typename ITy>
bool match(ITy *V)
const {
3082 return CountableConditionMatch(Cmp, PSE, L);
3101 VPValue *Uncountable =
nullptr;
3119 if (ExitBlocks.
size() != 1)
3126 if (!ExitBlocks.
front()->phis().empty())
3142 LatchVPBB->clearSuccessors();
3152 Term->setOperand(0, Countable);
3219 VPValue *UncountableCondition =
nullptr;
3226 Worklist.
push_back(UncountableCondition);
3227 while (!Worklist.
empty()) {
3231 if (V->isDefinedOutsideLoopRegions())
3237 if (V->getNumUsers() > 1)
3270 if (Recipes.
empty() ||
3274 return UncountableCondition;
3331 for (
auto &Exit : Exits) {
3332 if (Exit.EarlyExitingVPBB == LatchVPBB)
3336 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3337 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3351 "loop with side effects",
3352 "EarlyExitSideEffectsCond", ORE, TheLoop);
3367 assert(
Load &&
"Couldn't find exactly one load");
3370 "Uncountable exit condition load is conditional.");
3384 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3390 "load used by the exit condition that may "
3392 "EarlyExitSideEffectsFaultingLoad", ORE,
3405 "load used by the exit condition with an "
3406 "unsupported memory access pattern",
3407 "EarlyExitSideEffectsBadLoadAccessPattern", ORE,
3415 "load used by the exit condition with an "
3416 "unsupported memory access pattern",
3417 "EarlyExitSideEffectsBadLoadAccessPattern", ORE,
3427 while (InsertIt != HeaderVPBB->
end() &&
3429 erase(ConditionRecipes, &*InsertIt);
3432 for (
auto *Recipe :
reverse(ConditionRecipes))
3433 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3437 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3439 Type *IVScalarTy =
IV->getScalarType();
3445 "uncountable.exit.mask");
3450 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3452 if (!VPDT.
dominates(R.getParent(), LatchVPBB)) {
3454 "Early exit loop with side effects contains unsupported "
3455 "conditional memory operations",
3456 "EarlyExitSideEffectsUnsupportedConditionalMemOps", ORE, TheLoop);
3467 "Expected BranchOnCond terminator for MiddleVPBB");
3478 auto Phis = ScalarPH->
phis();
3483 "Early exit loop with side effects contains "
3484 "unsupported reductions, inductions or recurrences",
3485 "EarlyExitSideEffectsReductions", ORE, TheLoop);
3493 "Continuing from different IV");
3515 "Auto-vectorization of early exit loops with potentially "
3516 "faulting loads is not supported",
3517 "EarlyExitFaultingLoads", ORE, TheLoop);
3521 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3523 for (
auto [EarlyExitingVPBB, ExitBlock] :
3527 VPValue *CondOfEarlyExitingVPBB;
3528 [[maybe_unused]]
bool Matched =
3529 match(EarlyExitingVPBB->getTerminator(),
3531 assert(Matched &&
"Terminator must be BranchOnCond");
3535 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3536 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3538 TrueSucc == ExitBlock
3539 ? CondOfEarlyExitingVPBB
3540 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3546 "exit condition must dominate the latch");
3554 assert(!Exits.
empty() &&
"must have at least one early exit");
3561 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3564 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3570 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3571 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3573 Exits[
I].EarlyExitingVPBB) &&
3574 "RPO sort must place dominating exits before dominated ones");
3580 VPValue *Combined = Exits[0].CondToExit;
3602 "Unexpected terminator");
3603 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3604 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3605 LatchExitingBranch->eraseFromParent();
3608 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3609 LatchVPBB->clearSuccessors();
3614 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3615 MiddleVPBB->clearPredecessors();
3616 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3618 LatchVPBB, MiddleVPBB, ORE,
3619 TheLoop, PSE, DT, AC);
3624 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3628 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3636 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3639 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3671 for (
auto [Exit, VectorEarlyExitVPBB] :
3672 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3673 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3685 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3686 VPValue *NewIncoming = IncomingVal;
3688 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3693 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3694 ExitIRI->addIncoming(NewIncoming);
3697 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3732 bool IsLastDispatch = (
I + 2 == Exits.
size());
3734 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3740 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3743 CurrentBB = FalseBB;
3758 VPValue *VecOp = Red->getVecOp();
3761 if (Red->isPartialReduction())
3765 auto IsExtendedRedValidAndClampRange =
3778 "getExtendedReductionCost only supports integer types");
3779 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3780 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3781 Red->getFastMathFlagsOrNone(),
CostKind);
3782 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3790 IsExtendedRedValidAndClampRange(
3811 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3812 Opcode != Instruction::FAdd)
3816 if (Red->isPartialReduction())
3822 auto IsMulAccValidAndClampRange =
3834 (Ext0->getOpcode() != Ext1->getOpcode() ||
3835 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3839 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3841 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3848 ExtCost += Ext0->computeCost(VF, Ctx);
3850 ExtCost += Ext1->computeCost(VF, Ctx);
3852 ExtCost += OuterExt->computeCost(VF, Ctx);
3854 return MulAccCost.
isValid() &&
3855 MulAccCost < ExtCost + MulCost + RedCost;
3860 VPValue *VecOp = Red->getVecOp();
3898 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3900 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3901 Mul->setOperand(1, ExtB);
3911 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3916 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3923 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3940 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3949 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3950 Ext0->getOpcode() == Ext1->getOpcode() &&
3951 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3953 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3954 *Ext0, *Ext0, Ext0->getDebugLoc());
3955 NewExt0->insertBefore(Ext0);
3960 Ext->getScalarType(),
nullptr, *Ext1,
3961 *Ext1, Ext1->getDebugLoc());
3964 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3965 NewMul->insertBefore(
Mul);
3966 Ext->replaceAllUsesWith(NewMul);
3967 Ext->eraseFromParent();
3968 Mul->eraseFromParent();
3982 if (Red->isPartialReduction())
3986 auto IP = std::next(Red->getIterator());
3997 Red->replaceAllUsesWith(AbstractR);
4019 return CommonMetadata;
4022template <
unsigned Opcode>
4027 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
4028 "Only Load and Store opcodes supported");
4029 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
4036 for (
auto Recipes :
Groups) {
4037 if (Recipes.size() < 2)
4042 "Expected all recipes in group to have the same load-store type");
4049 VPValue *MaskI = RecipeI->getMask();
4055 bool HasComplementaryMask =
false;
4060 VPValue *MaskJ = RecipeJ->getMask();
4069 if (HasComplementaryMask) {
4070 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
4080template <
typename InstType>
4098 for (
auto &Group :
Groups) {
4118 return R->isSingleScalar() == IsSingleScalar;
4120 "all members in group must agree on IsSingleScalar");
4125 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
4126 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
4128 UnpredicatedLoad->insertBefore(EarliestLoad);
4132 Load->replaceAllUsesWith(UnpredicatedLoad);
4133 Load->eraseFromParent();
4142 if (!StoreLoc || !StoreLoc->AATags.Scope)
4149 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
4161 for (
auto &Group :
Groups) {
4174 VPValue *SelectedValue = Group[0]->getOperand(0);
4177 bool IsSingleScalar = Group[0]->isSingleScalar();
4178 for (
unsigned I = 1;
I < Group.size(); ++
I) {
4179 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
4180 "all members in group must agree on IsSingleScalar");
4181 VPValue *Mask = Group[
I]->getMask();
4183 SelectedValue = Builder.createSelect(
4186 Value->getScalarType()));
4194 StoreWithMinAlign->getUnderlyingInstr(),
4195 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
4196 nullptr, *LastStore, CommonMetadata);
4197 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
4201 Store->eraseFromParent();
4208 assert(UF > 1 &&
"Expected plan to have an UF > 1");
4214 VPValue *StoredValue =
nullptr;
4220 assert(
MemOp->isConsecutive() &&
"Expected consecutive load/store");
4224 assert(!
MemOp->isMasked() &&
"Masked accesses are not supported yet");
4227 : R.getVPSingleValue()->getScalarType();
4229 std::optional<Instruction::CastOps> CastHint;
4231 : R.getVPSingleValue()->getSingleUser();
4233 CastHint = Cast->getOpcode();
4236 if (!
TTI.hasMultiVectorLoadStore(
4238 VectorAccessType, IsStore, CastHint))
4246 VPBuilder Builder(VPBB, R.getIterator());
4248 VPValue *WideStoredValue = Builder.createNaryOp(
4251 {Multiplier, Ptr,
Align, WideStoredValue},
nullptr,
4252 {}, *
MemOp, R.getDebugLoc());
4254 VPValue *OldLoad = R.getVPSingleValue();
4264 R.eraseFromParent();
4280 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
4285 if (Member0Op == OpV)
4295 return !IsScalable && !W->getMask() && W->isConsecutive() &&
4298 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
4318 if (R->getScalarType() != WideMember0->getScalarType())
4320 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
4328 if (V->isDefinedOutsideLoopRegions())
4330 auto [It, Inserted] = FirstMembersOf.try_emplace(V,
Ops);
4331 if (!Inserted && (It->second.front() == V || V == WideMember0) &&
4336 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
4339 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
4344 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
4345 const auto &[OpIdx, OpV] =
P;
4346 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
4357static std::optional<ElementCount>
4361 if (!InterleaveR || InterleaveR->
getMask())
4362 return std::nullopt;
4364 Type *GroupElementTy =
nullptr;
4368 return Op->getScalarType() == GroupElementTy;
4370 return std::nullopt;
4374 return Op->getScalarType() == GroupElementTy;
4376 return std::nullopt;
4380 if (IG->getFactor() != IG->getNumMembers())
4381 return std::nullopt;
4387 assert(
Size.isScalable() == VF.isScalable() &&
4388 "if Size is scalable, VF must be scalable and vice versa");
4389 return Size.getKnownMinValue();
4393 unsigned MinVal = VF.getKnownMinValue();
4395 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4398 return std::nullopt;
4406 return RepR && RepR->isSingleScalar();
4420 if (V->isDefinedOutsideLoopRegions()) {
4423 return M->isDefinedOutsideLoopRegions() &&
4424 M->getScalarType() == V->getScalarType();
4426 "expected distinct loop-invariant values of matching scalar type");
4441 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4443 for (
VPValue *Member : Members)
4444 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4445 WideMember0->setOperand(
4454 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4456 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4457 *LoadGroup, LoadGroup->getDebugLoc());
4463 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4464 "must be a single scalar load");
4465 NarrowedOps.
insert(RepR);
4470 VPValue *PtrOp = WideLoad->getAddr();
4472 PtrOp = VecPtr->getOperand(0);
4477 nullptr, {}, *WideLoad);
4478 N->insertBefore(WideLoad);
4483std::unique_ptr<VPlan>
4503 "unexpected branch-on-count");
4507 std::optional<ElementCount> VFToOptimize;
4521 if (R.mayWriteToMemory() && !InterleaveR)
4527 return any_of(V->users(), [&](VPUser *U) {
4528 auto *UR = cast<VPRecipeBase>(U);
4529 return UR->getParent()->getParent() != VectorLoop;
4546 std::optional<ElementCount> NarrowedVF =
4548 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4550 VFToOptimize = NarrowedVF;
4553 if (InterleaveR->getStoredValues().empty())
4558 auto *Member0 = InterleaveR->getStoredValues()[0];
4568 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4571 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4572 return IR && IR->getInterleaveGroup()->isFull() &&
4573 IR->getVPValue(Op.index()) == Op.value();
4582 VFToOptimize->isScalable(), FirstMembersOf))
4587 if (StoreGroups.empty())
4591 bool RequiresScalarEpilogue =
4602 std::unique_ptr<VPlan> NewPlan;
4604 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4605 Plan.
setVF(*VFToOptimize);
4606 NewPlan->removeVF(*VFToOptimize);
4613 for (
auto *StoreGroup : StoreGroups) {
4615 NarrowedOps, Preheader);
4621 StoreGroup->getDebugLoc());
4628 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4634 if (VFToOptimize->isScalable()) {
4637 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4645 materializeVectorTripCount(Plan, VectorPH,
false,
4646 RequiresScalarEpilogue, Step);
4651 removeDeadRecipes(Plan);
4654 "All VPVectorPointerRecipes should have been removed");
4672 "Cannot handle loops with uncountable early exits");
4679 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4686 if (
any_of(RecurSplice->users(),
4687 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4768 {},
"vector.recur.extract.for.phi");
4771 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4785 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4786 VPValue *InvariantCandidate = BinOp->getOperand(1);
4788 std::swap(WidenIVCandidate, InvariantCandidate);
4802 auto *ClonedOp = BinOp->
clone();
4803 if (ClonedOp->getOperand(0) == WidenIV) {
4804 ClonedOp->setOperand(0, ScalarIV);
4806 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4807 ClonedOp->setOperand(1, ScalarIV);
4821 return std::nullopt;
4826 return std::nullopt;
4838 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4839 bool UseMax) -> std::optional<APSInt> {
4841 for (
bool Signed : {
true,
false}) {
4850 return std::nullopt;
4858 PhiR->getRecurrenceKind()))
4867 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4881 !
match(FindLastSelect,
4890 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4895 "IVOfExpressionToSink not being an AddRec must imply "
4896 "FindLastExpression not being an AddRec.");
4905 bool UseMax = *StepDirection;
4906 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4907 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4914 if (IVOfExpressionToSink) {
4915 const SCEV *FindLastExpressionSCEV =
4917 if (std::optional<bool> NewUseMax =
4919 if (
auto NewSentinel =
4920 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4923 SentinelVal = *NewSentinel;
4924 UseSigned = NewSentinel->isSigned();
4925 UseMax = *NewUseMax;
4926 IVSCEV = FindLastExpressionSCEV;
4927 IVOfExpressionToSink =
nullptr;
4937 if (AR->hasNoSignedWrap())
4939 else if (AR->hasNoUnsignedWrap())
4949 VPValue *NewFindLastSelect = BackedgeVal;
4951 if (!SentinelVal || IVOfExpressionToSink) {
4954 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4955 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4956 if (
match(FindLastSelect,
4958 SelectCond = LoopBuilder.
createNot(SelectCond);
4965 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4968 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4977 VPIRFlags Flags(MinMaxKind,
false,
false,
4983 NewFindLastSelect, Flags, ExitDL);
4986 VPValue *VectorRegionExitingVal = ReducedIV;
4987 bool SunkExpression =
false;
4988 if (IVOfExpressionToSink) {
4989 VectorRegionExitingVal =
4991 ReducedIV, IVOfExpressionToSink);
4992 SunkExpression =
true;
4996 VPValue *StartVPV = PhiR->getStartValue();
5003 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
5013 AnyOfPhi->insertAfter(PhiR);
5020 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
5033 PhiR->hasUsesOutsideReductionChain());
5035 NewPhiR->setExpressionSunk();
5036 NewPhiR->insertBefore(PhiR);
5037 PhiR->replaceAllUsesWith(NewPhiR);
5038 PhiR->eraseFromParent();
5045struct ReductionExtend {
5046 Type *SrcType =
nullptr;
5047 ExtendKind Kind = ExtendKind::PR_None;
5053struct ExtendedReductionOperand {
5057 ReductionExtend ExtendA, ExtendB;
5065struct PartialReductionDescriptor {
5068 VPWidenRecipe *ReductionBinOp =
nullptr;
5070 ExtendedReductionOperand ExtendedOp;
5077 unsigned AccumulatorOpIdx;
5078 unsigned ScaleFactor;
5081 VPBlendRecipe *Blend =
nullptr;
5086static std::optional<unsigned>
5090 "Expected a non-normalized blend with two incoming values");
5096 return std::nullopt;
5097 return FirstIncomingHasOneUse ? 0 : 1;
5109 if (!
Op->hasOneUse() ||
5115 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
5116 Op->getOperand(1), NarrowTy);
5118 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
5127 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
5129 assert(Ext->getOpcode() ==
5131 "Expected both the LHS and RHS extends to be the same");
5132 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
5135 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
5136 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
5137 auto *
Max = Builder.insert(
5139 {FreezeX, FreezeY}, SrcTy));
5140 auto *Min = Builder.insert(
5142 {FreezeX, FreezeY}, SrcTy));
5143 auto *AbsDiff = Builder.insert(
5146 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
5147 Op->getScalarType());
5159 if (!
Mul->hasOneUse() ||
5160 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
5161 MulLHS->getOpcode() != MulRHS->getOpcode())
5164 auto *NewLHS = Builder.createWidenCast(
5165 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
5166 auto *NewRHS = MulLHS == MulRHS
5168 : Builder.createWidenCast(MulRHS->getOpcode(),
5169 MulRHS->getOperand(0),
5170 Ext->getScalarType());
5171 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
5172 Builder.insert(NewMul);
5173 Op->replaceAllUsesWith(NewMul);
5174 Op->eraseFromParent();
5175 Mul->eraseFromParent();
5184 VPValue *VecOp = Red->getVecOp();
5237static void transformToPartialReduction(
const PartialReductionDescriptor &Link,
5245 WidenRecipe->
getOperand(1 - Link.AccumulatorOpIdx));
5248 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
5264 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
5266 (WidenRecipe->
getOpcode() == Instruction::FSub &&
5271 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
5283 Builder.insert(NegRecipe);
5284 ExtendedOp = NegRecipe;
5299 std::optional<unsigned> BlendReductionIdx =
5300 getBlendReductionUpdateValueIdx(Link.Blend);
5301 assert(BlendReductionIdx &&
5303 "Expected blend to contain the reduction update");
5317 [[maybe_unused]]
bool IsLastInChain =
5321 assert((!ExitValue || IsLastInChain) &&
5322 "if we found ExitValue, it must match RdxPhi's backedge value");
5333 PartialRed->insertBefore(WidenRecipe);
5343 E->insertBefore(WidenRecipe);
5344 PartialRed->replaceAllUsesWith(
E);
5350 const PartialReductionDescriptor &Link,
5353 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5354 std::optional<unsigned> BinOpc = std::nullopt;
5356 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5357 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5359 std::optional<llvm::FastMathFlags>
Flags;
5363 auto GetLinkOpcode = [&Link]() ->
unsigned {
5366 return Instruction::Add;
5368 return Instruction::FAdd;
5370 return Link.ReductionBinOp->
getOpcode();
5375 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5376 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5397static std::optional<ExtendedReductionOperand>
5400 "Op should be operand of UpdateR");
5408 if (
Op->hasOneUse() &&
5417 Type *RHSInputType =
Y->getScalarType();
5418 if (LHSInputType != RHSInputType ||
5419 LHSExt->getOpcode() != RHSExt->getOpcode())
5420 return std::nullopt;
5423 return ExtendedReductionOperand{
5425 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5429 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5432 VPValue *CastSource = CastRecipe->getOperand(0);
5433 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5443 return ExtendedReductionOperand{
5450 if (!
Op->hasOneUse())
5451 return std::nullopt;
5456 return std::nullopt;
5466 return std::nullopt;
5470 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5473 const APInt *RHSConst =
nullptr;
5479 return std::nullopt;
5483 if (Cast && OuterExtKind &&
5484 getPartialReductionExtendKind(Cast) != OuterExtKind)
5485 return std::nullopt;
5487 Type *RHSInputType = LHSInputType;
5488 ExtendKind RHSExtendKind = LHSExtendKind;
5491 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5494 return ExtendedReductionOperand{
5495 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5502static std::optional<SmallVector<PartialReductionDescriptor>>
5509 return std::nullopt;
5510 VPValue *ExitValue = RdxResult->getOperand(0);
5519 VPValue *CurrentValue = ExitValue;
5520 while (CurrentValue != RedPhiR) {
5522 std::optional<unsigned> BlendReductionIdx;
5526 return std::nullopt;
5528 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5529 if (!BlendReductionIdx)
5530 return std::nullopt;
5537 return std::nullopt;
5544 std::optional<ExtendedReductionOperand> ExtendedOp =
5545 matchExtendedReductionOperand(UpdateR,
Op);
5547 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5549 return std::nullopt;
5557 return std::nullopt;
5559 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5562 return std::nullopt;
5564 PartialReductionDescriptor Link(
5565 {UpdateR, *ExtendedOp, RK,
5570 CurrentValue = PrevValue;
5575 std::reverse(Chain.
begin(), Chain.
end());
5585 assert(Phi->getVFScaleFactor() == 1 &&
"scale factor must not be set");
5586 Phi->setVFScaleFactor(Factor);
5591 StartInst->setOperand(2, NewScaleFactor);
5598 VPValue *OldStartValue = StartInst->getOperand(0);
5599 StartInst->setOperand(0, StartInst->getOperand(1));
5603 assert(RdxResult &&
"Could not find reduction result");
5606 unsigned SubOpc = RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
5607 : Instruction::BinaryOps::Sub;
5610 Phi->getDebugLoc());
5612 NewResult, [&NewResult](
VPUser &U) {
return &U != NewResult; });
5627 if (
auto Chain = getScaledReductionChain(&RedPhiR))
5628 PhiToChain.
try_emplace(&RedPhiR, std::move(*Chain));
5633 UnorderedReductions.
push_back(&RedPhiR);
5639 for (
auto *Rdx : UnorderedReductions) {
5655 ? std::make_optional(Rdx->getFastMathFlagsOrNone())
5659 Backedge->getOpcode(), ScalarTy,
nullptr,
5661 std::nullopt, CostCtx.
CostKind, FMF);
5662 return PRCost <= CurrentCost;
5668 Rdx->getRecurrenceKind(), Rdx->getFastMathFlagsOrNone(),
5669 Backedge->getUnderlyingInstr(), Rdx, OtherOp,
nullptr,
5672 Partial->insertBefore(Backedge);
5673 Backedge->replaceAllUsesWith(Partial);
5674 Backedge->eraseFromParent();
5677 if (PhiToChain.
empty())
5685 for (
auto &[
_, Chain] : PhiToChain)
5686 for (
const PartialReductionDescriptor &Link : Chain) {
5687 PartialReductionOps.
insert(Link.ExtendedOp.ExtendsUser);
5689 PartialReductionBlends.
insert(Link.Blend);
5690 ScaledReductionMap[Link.ReductionBinOp] = Link.ScaleFactor;
5696 auto ExtendUsersValid = [&](
VPValue *Ext) {
5698 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5702 auto IsProfitablePartialReductionChainForVF =
5709 for (
const PartialReductionDescriptor &Link : Chain) {
5710 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5711 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5715 PartialCost += LinkCost;
5716 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5718 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5719 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5722 RegularCost += Extend->computeCost(VF, CostCtx);
5724 return PartialCost.
isValid() && PartialCost < RegularCost;
5732 for (
auto &[RedPhiR, Chain] : PhiToChain) {
5733 for (
const PartialReductionDescriptor &Link : Chain) {
5734 if (!
all_of(Link.ExtendedOp.ExtendsUser->
operands(), ExtendUsersValid)) {
5738 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5740 return PhiR == RedPhiR;
5744 return Blend == Link.Blend || PartialReductionBlends.
contains(Blend);
5746 return Link.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5752 if (!
all_of(Link.ReductionBinOp->
users(), UseIsValid)) {
5761 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5762 return RepR && RepR->getOpcode() == Instruction::Store;
5773 return IsProfitablePartialReductionChainForVF(Chain, VF);
5779 for (
auto &[Phi, Chain] : PhiToChain) {
5783 for (
const PartialReductionDescriptor &Link : Chain)
5784 transformToPartialReduction(Link, Plan, Phi);
5789 const PartialReductionDescriptor &Link = Chain[0];
5805 if (VPI.getUnderlyingValue() &&
5816 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5819 if (!ProcessVPInst(VPI))
5828 assert(New->getParent() &&
"New recipe must have been inserted");
5829 if (VPI->
getOpcode() == Instruction::Load)
5838 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5845 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5847 VPI, FinalRedStoresBuilder))
5856 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5869 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5873 return Scalarize(VPI);
5880 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5882 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5892 const SCEV *PtrSCEV =
5894 bool IsSingleScalarLoad =
5900 I, Ptr, IsSingleScalarLoad,
5909 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5911 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5915 std::optional<int64_t> Stride =
5917 if (Stride != 1 && Stride != -1)
5948 return ReplaceWith(VPI,
Load);
5957 auto *StoreR = Builder.createWidenStore(
5960 return ReplaceWith(VPI, StoreR);
5967 return ReplaceWith(VPI, Recipe);
5969 return Scalarize(VPI);
5989 if (VPI.mayHaveSideEffects())
5993 if (VPI.isMasked() && !VPI.isSafeToSpeculativelyExecute())
5998 if (VPI.getOpcode() == Instruction::Add &&
6007 VPI.getOpcode(), VPI.operandsWithoutMask(),
nullptr, VPI,
6008 VPI, VPI.getDebugLoc(), VPI.getScalarType(),
I);
6009 Recipe->insertBefore(&VPI);
6010 VPI.replaceAllUsesWith(Recipe);
6011 VPI.eraseFromParent();
6021 switch (Param.ParamKind) {
6022 case VFParamKind::Vector:
6023 case VFParamKind::GlobalPredicate:
6025 case VFParamKind::OMP_Uniform:
6026 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
6027 SE->isLoopInvariant(
6028 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
6030 case VFParamKind::OMP_Linear:
6031 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
6032 m_scev_AffineAddRec(
6033 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
6034 m_SpecificLoop(L)));
6051 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
6052 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
6055 if (It == Mappings.end())
6062struct CallWideningDecision {
6063 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
6064 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
6087 return CallWideningDecision::KindTy::Scalarize;
6097 return CallWideningDecision::KindTy::Scalarize;
6101 false, VF, CostCtx);
6116 return CallWideningDecision::KindTy::Intrinsic;
6120 if (VecFunc && ScalarCost >= VecCallCost)
6121 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
6123 return CallWideningDecision::KindTy::Scalarize;
6129 bool Widened =
false;
6134 if (!VPI.getUnderlyingValue() || VPI.getOpcode() != Instruction::Call)
6139 VPI.op_begin() + CI->arg_size());
6141 CallWideningDecision Decision =
6150 switch (Decision.Kind) {
6151 case CallWideningDecision::KindTy::Intrinsic: {
6155 VPI, VPI.getDebugLoc());
6159 case CallWideningDecision::KindTy::VectorVariant: {
6164 Ops.push_back(Mask);
6166 Ops.push_back(VPI.getOperand(VPI.getNumOperandsWithoutMask() - 1));
6172 case CallWideningDecision::KindTy::Scalarize:
6178 VPI.replaceAllUsesWith(Replacement);
6179 VPI.eraseFromParent();
6196 if (VPI.getOpcode() != Instruction::Trunc)
6225 !
TTI.isTruncateFree(
6226 toVectorTy(VPI.getOperand(0)->getScalarType(), VF),
6230 IsNarrowingProfitable,
Range))
6236 WideIV->getPHINode(), WideIV->getStartValue(), WideIV->getStepValue(),
6237 WideIV->getVFValue(), WideIV->getInductionDescriptor(), Trunc,
6239 NarrowIV->insertBefore(*HeaderVPBB, HeaderVPBB->
getFirstNonPhi());
6240 VPI.replaceAllUsesWith(NarrowIV);
6241 VPI.eraseFromParent();
6263 if (!MemR || MemR->isConsecutive())
6266 VPValue *Ptr = MemR->getAddr();
6278 VPValue *StoredValue =
nullptr;
6282 StoredValue = StoreR->getStoredValue();
6284 IntrinID = Intrinsic::experimental_vp_strided_store;
6288 IntrinID = Intrinsic::experimental_vp_strided_load;
6291 Align Alignment = MemR->getAlign();
6294 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
6299 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
6300 return StridedLoadStoreCost < CurrentCost;
6311 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
6316 I32VF = Builder.createScalarZExtOrTrunc(
6330 "Stride type from SCEV must match the index type");
6331 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
6334 auto *
Offset = Builder.createOverflowingOp(
6335 Instruction::Mul, {CanIV, StrideInBytes},
6336 {AddRecPtr->hasNoUnsignedWrap(),
false});
6340 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
6343 VPValue *NewPtr = Builder.createVectorPointer(
6347 VPValue *Mask = MemR->getMask();
6352 Ops.push_back(StoredValue);
6353 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
6355 auto *StridedR = Builder.createWidenMemIntrinsic(
6358 *MemR, R.getDebugLoc());
6361 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
size_t size() const
Get the array size.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
LLVM_ABI MDNode * createBranchWeights(uint32_t TrueWeight, uint32_t FalseWeight, bool IsExpected=false)
Return metadata containing two branch weights.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEVFlags Flags=SCEV::FlagNone)
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEVFlags Flags=SCEV::FlagNone, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEVFlags Flags=SCEV::FlagNone)
Return the SCEV object corresponding to -V.
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
RecipeListTy::iterator iterator
Instruction iterators...
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
void insert(VPRecipeBase *Recipe, iterator InsertPt)
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
void clearPredecessors()
Remove all the predecessor of this block.
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility similar to IRBuilder.
VPInstruction * createFreeze(VPValue *Op, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
static VPBuilderBase getToInsertAfter(VPRecipeBase *R)
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Type *ResultTy, Instruction *UV)
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
LLVM_ABI_FOR_TEST bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
Recipe to expand a SCEV expression.
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Base class of all SIMD vector types.
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS*X will result in a value whose quantity matches our ...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS*X will result in a value whose quantity matches our own.
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
void reportVectorizationFailure(const StringRef DebugMsg, const StringRef OREMsg, const StringRef ORETag, OptimizationRemarkEmitter *ORE, const Loop *TheLoop, Instruction *I=nullptr)
Reports a vectorization failure: print DebugMsg for debugging purposes along with the corresponding o...
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
OneOps_match< OpTy, Instruction::Freeze > m_Freeze(const OpTy &Op)
Matches FreezeInst.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
CmpClass_match< LHS, RHS, CmpInst, true > m_c_Cmp(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst, true > m_c_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
Matches an ICmp with a predicate over LHS and RHS in either order.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
LogicalOp_match< LHS, RHS, Instruction::Or, true > m_c_LogicalOr(const LHS &L, const RHS &R)
Matches L || R with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
cst_pred_ty< is_one > m_scev_One()
Match an integer 1.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
VPWidenStoreRecipe_match< Op0_t, Op1_t > m_WidenStore(const Op0_t &Op0, const Op1_t &Op1)
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPWidenLoadRecipe_match< Op0_t > m_WidenLoad(const Op0_t &Op0)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
VectorPointerRecipe_match< Op0_t, Op1_t > m_VecPtr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
bool match(Val *V, const Pattern &P)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
initializer< Ty > init(const Ty &Val)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
std::optional< int64_t > getConstantStride(VPValue *Addr, Type *AccessTy, PredicatedScalarEvolution &PSE, const Loop *L)
If the pointer operand Addr of a memory access is an affine AddRec w.r.t.
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
LLVM_ABI_FOR_TEST bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
LLVM_ABI_FOR_TEST const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
VPBuilderBase<> VPBuilder
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto make_isa_range(RangeT &&Range)
Return a range over Range containing only elements for which isa<T> holds, casting each of them to T.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A recipe for handling first-order recurrence phis.
WrapFlagsTy withoutNoSignedWrap()
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...