73#define DEBUG_TYPE "loop-unroll"
77 cl::desc(
"Forget everything in SCEV when doing LoopUnroll, instead of just"
78 " the current top-most loop. This is sometimes preferred to reduce"
83 cl::desc(
"The cost threshold for loop unrolling"));
88 cl::desc(
"The cost threshold for loop unrolling when optimizing for "
93 cl::desc(
"The cost threshold for partial loop unrolling"));
97 cl::desc(
"The maximum 'boost' (represented as a percentage >= 100) applied "
98 "to the threshold when aggressively unrolling a loop due to the "
99 "dynamic cost savings. If completely unrolling a loop will reduce "
100 "the total runtime from X to Y, we boost the loop unroll "
101 "threshold to DefaultThreshold*std::min(MaxPercentThresholdBoost, "
102 "X/Y). This limit avoids excessive code bloat."));
106 cl::desc(
"Don't allow loop unrolling to simulate more than this number of "
107 "iterations when checking full unroll profitability"));
111 cl::desc(
"Use this unroll count for all loops including those with "
112 "unroll_count pragma values, for testing purposes"));
116 cl::desc(
"Set the max unroll count for partial and runtime unrolling, for"
117 "testing purposes"));
122 "Set the max unroll count for full unrolling, for testing purposes"));
126 cl::desc(
"Allows loops to be partially unrolled until "
127 "-unroll-threshold loop size is reached."));
131 cl::desc(
"Allow generation of a loop remainder (extra iterations) "
132 "when unrolling a loop."));
136 cl::desc(
"Unroll loops with run-time trip counts"));
141 "The max of trip count upper bound that is considered in unrolling"));
145 cl::desc(
"Unrolled size limit for loops with unroll metadata "
146 "(full, enable, or count)."));
150 cl::desc(
"If the runtime tripcount for the loop is lower than the "
151 "threshold, the loop is considered as flat and will be less "
152 "aggressively unrolled."));
156 cl::desc(
"Allow the loop remainder to be unrolled."));
163 cl::desc(
"Enqueue and re-visit child loops in the loop PM after unrolling. "
164 "This shouldn't typically be needed as child loops (or their "
165 "clones) were already visited."));
169 cl::desc(
"Threshold (max size of unrolled loop) to use in aggressive (O3) "
174 cl::desc(
"Default threshold (max size of unrolled "
175 "loop), used in all but O3 optimizations"));
179 cl::desc(
"Maximum allowed iterations to unroll under pragma unroll full."));
184static const unsigned NoThreshold = std::numeric_limits<unsigned>::max();
192 std::optional<unsigned> UserThreshold, std::optional<unsigned> UserCount,
193 std::optional<bool> UserAllowPartial, std::optional<bool> UserRuntime,
194 std::optional<bool> UserUpperBound,
195 std::optional<unsigned> UserFullUnrollMaxCount) {
207 UP.
MaxCount = std::numeric_limits<unsigned>::max();
226 TTI.getUnrollingPreferences(L, SE, UP, &ORE);
229 bool OptForSize = L->getHeader()->getParent()->hasOptSize() ||
272 UP.
Count = *UserCount;
273 if (UserAllowPartial)
274 UP.
Partial = *UserAllowPartial;
279 if (UserFullUnrollMaxCount)
293struct UnrolledInstState {
297 unsigned IsCounted : 1;
301struct UnrolledInstStateKeyInfo {
302 using PtrInfo = DenseMapInfo<Instruction *>;
303 using PairInfo = DenseMapInfo<std::pair<Instruction *, int>>;
305 static inline UnrolledInstState getEmptyKey() {
306 return {PtrInfo::getEmptyKey(), 0, 0, 0};
309 static inline UnrolledInstState getTombstoneKey() {
310 return {PtrInfo::getTombstoneKey(), 0, 0, 0};
313 static inline unsigned getHashValue(
const UnrolledInstState &S) {
314 return PairInfo::getHashValue({S.I, S.Iteration});
317 static inline bool isEqual(
const UnrolledInstState &
LHS,
318 const UnrolledInstState &
RHS) {
319 return PairInfo::isEqual({
LHS.I,
LHS.Iteration}, {
RHS.I,
RHS.Iteration});
323struct EstimatedUnrollCost {
325 unsigned UnrolledCost;
329 unsigned RolledDynamicCost;
333 PragmaInfo(
bool UUC,
bool PFU,
unsigned PC,
bool PEU)
334 : UserUnrollCount(UUC), PragmaFullUnroll(PFU), PragmaCount(PC),
335 PragmaEnableUnroll(PEU) {}
336 const bool UserUnrollCount;
337 const bool PragmaFullUnroll;
338 const unsigned PragmaCount;
339 const bool PragmaEnableUnroll;
361 unsigned MaxIterationsCountToAnalyze) {
365 assert(MaxIterationsCountToAnalyze <
366 (
unsigned)(std::numeric_limits<int>::max() / 2) &&
367 "The unroll iterations max is too large!");
371 if (!L->isInnermost()) {
373 <<
"Not analyzing loop cost: not an innermost loop.\n");
378 if (!TripCount || TripCount > MaxIterationsCountToAnalyze) {
380 <<
"Not analyzing loop cost: trip count "
381 << (TripCount ?
"too large" :
"unknown") <<
".\n");
415 auto AddCostRecursively = [&](
Instruction &RootI,
int Iteration) {
416 assert(Iteration >= 0 &&
"Cannot have a negative iteration!");
417 assert(CostWorklist.
empty() &&
"Must start with an empty cost list");
418 assert(PHIUsedList.
empty() &&
"Must start with an empty phi used list");
424 for (;; --Iteration) {
430 auto CostIter = InstCostMap.
find({
I, Iteration, 0, 0});
431 if (CostIter == InstCostMap.
end())
436 auto &Cost = *CostIter;
442 Cost.IsCounted =
true;
446 if (PhiI->getParent() == L->getHeader()) {
447 assert(Cost.IsFree &&
"Loop PHIs shouldn't be evaluated as they "
448 "inherently simplify during unrolling.");
456 PhiI->getIncomingValueForBlock(L->getLoopLatch())))
457 if (L->contains(OpI))
466 transform(
I->operands(), std::back_inserter(Operands),
468 if (auto Res = SimplifiedValues.lookup(Op))
472 UnrolledCost +=
TTI.getInstructionCost(
I, Operands,
CostKind);
474 <<
"Adding cost of instruction (iteration " << Iteration
486 if (!OpI || !L->contains(OpI))
492 }
while (!CostWorklist.
empty());
494 if (PHIUsedList.
empty())
499 "Cannot track PHI-used values past the first iteration!");
507 assert(L->isLoopSimplifyForm() &&
"Must put loop into normal form first.");
508 assert(L->isLCSSAForm(DT) &&
509 "Must have loops in LCSSA form to track live-out values.");
512 <<
"Starting LoopUnroll profitability analysis...\n");
515 L->getHeader()->getParent()->hasMinSize() ?
521 for (
unsigned Iteration = 0; Iteration < TripCount; ++Iteration) {
534 PHI->getNumIncomingValues() == 2 &&
535 "Must have an incoming value only for the preheader and the latch.");
537 Value *V =
PHI->getIncomingValueForBlock(
538 Iteration == 0 ? L->getLoopPreheader() : L->getLoopLatch());
539 if (Iteration != 0 && SimplifiedValues.
count(V))
540 V = SimplifiedValues.
lookup(V);
545 SimplifiedValues.
clear();
546 while (!SimplifiedInputValues.
empty())
552 BBWorklist.
insert(L->getHeader());
554 for (
unsigned Idx = 0; Idx != BBWorklist.
size(); ++Idx) {
568 RolledDynamicCost +=
TTI.getInstructionCost(&
I,
CostKind);
573 bool IsFree = Analyzer.
visit(
I);
574 bool Inserted = InstCostMap.
insert({&
I, (int)Iteration,
578 assert(Inserted &&
"Cannot have a state for an unvisited instruction!");
586 const Function *Callee = CI->getCalledFunction();
587 if (!Callee ||
TTI.isLoweredToCall(Callee)) {
589 <<
"Can't analyze cost of loop with call\n");
596 if (
I.mayHaveSideEffects())
597 AddCostRecursively(
I, Iteration);
600 if (UnrolledCost > MaxUnrolledLoopSize) {
602 dbgs().
indent(3) <<
"Exceeded threshold.. exiting.\n";
604 <<
"UnrolledCost: " << UnrolledCost
605 <<
", MaxUnrolledLoopSize: " << MaxUnrolledLoopSize <<
"\n";
614 if (SimplifiedValues.
count(V))
615 V = SimplifiedValues.
lookup(V);
623 if (BI->isConditional()) {
624 if (
auto *SimpleCond = getSimplifiedConstant(BI->getCondition())) {
627 KnownSucc = BI->getSuccessor(0);
630 KnownSucc = BI->getSuccessor(SimpleCondVal->isZero() ? 1 : 0);
634 if (
auto *SimpleCond = getSimplifiedConstant(
SI->getCondition())) {
637 KnownSucc =
SI->getSuccessor(0);
640 KnownSucc =
SI->findCaseValue(SimpleCondVal)->getCaseSuccessor();
644 if (L->contains(KnownSucc))
645 BBWorklist.
insert(KnownSucc);
647 ExitWorklist.
insert({BB, KnownSucc});
653 if (L->contains(Succ))
656 ExitWorklist.
insert({BB, Succ});
657 AddCostRecursively(*TI, Iteration);
662 if (UnrolledCost == RolledDynamicCost) {
664 dbgs().
indent(3) <<
"No opportunities found.. exiting.\n";
665 dbgs().
indent(3) <<
"UnrolledCost: " << UnrolledCost <<
"\n";
671 while (!ExitWorklist.
empty()) {
673 std::tie(ExitingBB, ExitBB) = ExitWorklist.
pop_back_val();
680 Value *
Op = PN->getIncomingValueForBlock(ExitingBB);
682 if (L->contains(OpI))
683 AddCostRecursively(*OpI, TripCount - 1);
688 "All instructions must have a valid cost, whether the "
689 "loop is rolled or unrolled.");
693 dbgs().
indent(3) <<
"UnrolledCost: " << UnrolledCost
694 <<
", RolledDynamicCost: " << RolledDynamicCost <<
"\n";
705 Metrics.analyzeBasicBlock(BB,
TTI, EphValues,
false,
708 NotDuplicatable =
Metrics.notDuplicatable;
721 if (LoopSize.isValid() && LoopSize < BEInsns + 1)
723 LoopSize = BEInsns + 1;
729 <<
"Not unrolling: contains convergent operations.\n");
732 if (!LoopSize.isValid()) {
734 <<
"Not unrolling: loop size could not be computed.\n");
737 if (NotDuplicatable) {
739 <<
"Not unrolling: contains non-duplicatable instructions.\n");
747 unsigned CountOverwrite)
const {
748 unsigned LS = LoopSize.getValue();
749 assert(LS >= UP.
BEInsns &&
"LoopSize should not be less than BEInsns!");
760 if (
MDNode *LoopID = L->getLoopID())
787 "Unroll count hint metadata should have two operands.");
790 assert(
Count >= 1 &&
"Unroll count must be positive.");
802 unsigned MaxPercentThresholdBoost) {
803 if (Cost.RolledDynamicCost >= std::numeric_limits<unsigned>::max() / 100)
805 else if (Cost.UnrolledCost != 0)
807 return std::min(100 * Cost.RolledDynamicCost / Cost.UnrolledCost,
808 MaxPercentThresholdBoost);
810 return MaxPercentThresholdBoost;
813static std::optional<unsigned>
815 const unsigned TripMultiple,
const unsigned TripCount,
822 if (PInfo.UserUnrollCount) {
830 <<
"Not unrolling with user count " <<
UnrollCount <<
": "
832 :
"remainder not allowed")
837 if (PInfo.PragmaCount > 0) {
838 if ((UP.
AllowRemainder || (TripMultiple % PInfo.PragmaCount == 0))) {
840 << PInfo.PragmaCount <<
".\n");
841 return PInfo.PragmaCount;
844 <<
"Not unrolling with pragma count " << PInfo.PragmaCount
845 <<
": remainder not allowed, count does not divide trip "
846 <<
"multiple " << TripMultiple <<
".\n");
849 if (PInfo.PragmaFullUnroll) {
850 if (TripCount != 0) {
856 <<
"Won't unroll; trip count is too large.\n");
861 <<
"Fully unrolling with trip count: " << TripCount <<
".\n");
865 <<
"Not fully unrolling: unknown trip count.\n");
868 if (PInfo.PragmaEnableUnroll && !TripCount && MaxTripCount &&
871 <<
"Unrolling with max trip count: " << MaxTripCount <<
".\n");
883 assert(FullUnrollTripCount &&
"should be non-zero!");
887 <<
"Not unrolling: trip count " << FullUnrollTripCount
897 <<
" < threshold " << UP.
Threshold <<
".\n");
898 return FullUnrollTripCount;
902 <<
"Unrolled size " << UnrolledSize <<
" exceeds threshold "
903 << UP.
Threshold <<
"; checking for cost benefit.\n");
909 L, FullUnrollTripCount, DT, SE, EphValues,
TTI,
914 unsigned BoostedThreshold = UP.
Threshold * Boost / 100;
915 if (Cost->UnrolledCost < BoostedThreshold) {
917 return FullUnrollTripCount;
920 <<
"Not unrolling: cost " << Cost->UnrolledCost
921 <<
" >= boosted threshold " << BoostedThreshold <<
".\n");
927static std::optional<unsigned>
937 <<
"-unroll-allow-partial not given\n");
950 <<
"Unrolled size exceeds threshold; reducing count "
951 <<
"from " <<
count <<
" to " << NewCount <<
".\n");
972 <<
"Will not partially unroll: no profitable count.\n");
982 <<
"Partially unrolling with count: " <<
count <<
"\n");
999 const unsigned TripCount,
1000 const unsigned MaxTripCount,
const bool MaxOrZero,
1001 const unsigned TripMultiple,
1009 << TripCount <<
", MaxTripCount=" << MaxTripCount
1010 << (MaxOrZero ?
" (MaxOrZero)" :
"")
1011 <<
", TripMultiple=" << TripMultiple <<
"\n");
1013 const bool UserUnrollCount =
UnrollCount.getNumOccurrences() > 0;
1018 const bool ExplicitUnroll = PragmaCount > 0 || PragmaFullUnroll ||
1019 PragmaEnableUnroll || UserUnrollCount;
1022 if (ExplicitUnroll) {
1023 dbgs().
indent(1) <<
"Explicit unroll requested:";
1024 if (UserUnrollCount)
1025 dbgs() <<
" user-count";
1026 if (PragmaFullUnroll)
1027 dbgs() <<
" pragma-full";
1028 if (PragmaCount > 0)
1029 dbgs() <<
" pragma-count(" << PragmaCount <<
")";
1030 if (PragmaEnableUnroll)
1031 dbgs() <<
" pragma-enable";
1036 PragmaInfo PInfo(UserUnrollCount, PragmaFullUnroll, PragmaCount,
1037 PragmaEnableUnroll);
1043 "explicit unroll count");
1046 <<
"Using explicit peel count: " << PP.
PeelCount <<
".\n");
1056 MaxTripCount, UCE, UP)) {
1057 UP.
Count = *UnrollFactor;
1059 if (UserUnrollCount || (PragmaCount > 0)) {
1063 UP.
Runtime |= (PragmaCount > 0);
1064 return ExplicitUnroll;
1066 if (ExplicitUnroll && TripCount != 0) {
1081 UP.
Count = TripCount;
1083 TripCount, UCE, UP)) {
1084 UP.
Count = *UnrollFactor;
1085 return ExplicitUnroll;
1102 if (!TripCount && MaxTripCount && (UP.
UpperBound || MaxOrZero) &&
1104 UP.
Count = MaxTripCount;
1106 MaxTripCount, UCE, UP)) {
1107 UP.
Count = *UnrollFactor;
1108 return ExplicitUnroll;
1117 <<
"Peeling with count: " << PP.
PeelCount <<
".\n");
1120 return ExplicitUnroll;
1132 UP.
Count = *UnrollFactor;
1134 if ((PragmaFullUnroll || PragmaEnableUnroll) && TripCount &&
1135 UP.
Count != TripCount)
1138 "FullUnrollAsDirectedTooLarge",
1139 L->getStartLoc(), L->getHeader())
1140 <<
"unable to fully unroll loop as directed by unroll metadata "
1141 "because unrolled size is too large";
1145 if (UP.
Count == 0) {
1146 if (PragmaEnableUnroll)
1149 "UnrollAsDirectedTooLarge",
1150 L->getStartLoc(), L->getHeader())
1151 <<
"unable to unroll loop as directed by "
1152 "llvm.loop.unroll.enable metadata because unrolled size "
1157 return ExplicitUnroll;
1160 "All cases when TripCount is constant should be covered here.");
1161 if (PragmaFullUnroll)
1164 DEBUG_TYPE,
"CantFullUnrollAsDirectedRuntimeTripCount",
1165 L->getStartLoc(), L->getHeader())
1166 <<
"unable to fully unroll loop as directed by "
1167 "llvm.loop.unroll.full metadata because loop has a runtime "
1176 <<
"Not runtime unrolling: disabled by pragma.\n");
1184 <<
"Not runtime unrolling: max trip count " << MaxTripCount
1185 <<
" is small (< " << UP.
MaxUpperBound <<
") and not forced.\n");
1191 if (L->getHeader()->getParent()->hasProfileData()) {
1199 UP.
Runtime |= PragmaEnableUnroll || PragmaCount > 0 || UserUnrollCount;
1202 <<
"Will not try to unroll loop with runtime trip count "
1203 <<
"because -unroll-runtime not given\n");
1212 while (UP.
Count != 0 &&
1217 unsigned OrigCount = UP.
Count;
1221 while (UP.
Count != 0 && TripMultiple % UP.
Count != 0)
1224 <<
"Remainder loop is restricted (that could be architecture "
1225 "specific or because the loop contains a convergent "
1226 "instruction), so unroll count must divide the trip "
1228 << TripMultiple <<
". Reducing unroll count from " << OrigCount
1229 <<
" to " << UP.
Count <<
".\n");
1231 using namespace ore;
1236 "DifferentUnrollCountFromDirected",
1237 L->getStartLoc(), L->getHeader())
1238 <<
"Unable to unroll loop the number of times directed by "
1239 "llvm.loop.unroll.count metadata because remainder loop is "
1240 "restricted (that could be architecture specific or because "
1241 "the loop contains a convergent instruction) and so must "
1242 "have an unroll count that divides the loop trip multiple of "
1243 << NV(
"TripMultiple", TripMultiple) <<
". Unrolling instead "
1244 << NV(
"UnrollCount", UP.
Count) <<
" time(s).";
1251 if (MaxTripCount && UP.
Count > MaxTripCount)
1252 UP.
Count = MaxTripCount;
1255 <<
"Runtime unrolling with count: " << UP.
Count <<
"\n");
1258 return ExplicitUnroll;
1266 bool OnlyFullUnroll,
bool OnlyWhenForced,
bool ForgetAllSCEV,
1267 std::optional<unsigned> ProvidedCount,
1268 std::optional<unsigned> ProvidedThreshold,
1269 std::optional<bool> ProvidedAllowPartial,
1270 std::optional<bool> ProvidedRuntime,
1271 std::optional<bool> ProvidedUpperBound,
1272 std::optional<bool> ProvidedAllowPeeling,
1273 std::optional<bool> ProvidedAllowProfileBasedPeeling,
1274 std::optional<unsigned> ProvidedFullUnrollMaxCount,
1278 << L->getHeader()->getParent()->getName() <<
"] Loop %"
1279 << L->getHeader()->getName()
1280 <<
" (depth=" << L->getLoopDepth() <<
")\n");
1292 Loop *ParentL = L->getParentLoop();
1293 if (ParentL !=
nullptr &&
1297 <<
" llvm.loop.unroll_and_jam.\n");
1308 <<
"Not unrolling loop since it has llvm.loop.unroll_and_jam.\n");
1312 if (!L->isLoopSimplifyForm()) {
1314 <<
"Not unrolling loop which is not in loop-simplify form.\n");
1320 if (OnlyWhenForced && !(TM &
TM_Enable)) {
1322 <<
"disabled and loop not explicitly "
1327 bool OptForSize = L->getHeader()->getParent()->hasOptSize();
1329 L, SE,
TTI, BFI, PSI, ORE, OptLevel, ProvidedThreshold, ProvidedCount,
1330 ProvidedAllowPartial, ProvidedRuntime, ProvidedUpperBound,
1331 ProvidedFullUnrollMaxCount);
1333 L, SE,
TTI, ProvidedAllowPeeling, ProvidedAllowProfileBasedPeeling,
true);
1360 <<
"Not unrolling loop with inlinable calls.\n");
1369 unsigned TripCount = 0;
1370 unsigned TripMultiple = 1;
1372 L->getExitingBlocks(ExitingBlocks);
1373 for (
BasicBlock *ExitingBlock : ExitingBlocks)
1375 if (!TripCount || TC < TripCount)
1376 TripCount = TripMultiple = TC;
1382 BasicBlock *ExitingBlock = L->getLoopLatch();
1383 if (!ExitingBlock || !L->isLoopExiting(ExitingBlock))
1384 ExitingBlock = L->getExitingBlock();
1400 unsigned MaxTripCount = 0;
1401 bool MaxOrZero =
false;
1409 bool IsCountSetExplicitly =
1411 MaxTripCount, MaxOrZero, TripMultiple, UCE, UP, PP);
1414 <<
"Not unrolling: no viable strategy found.\n");
1421 assert(UP.
Count == 1 &&
"Cannot perform peel and unroll in the same step");
1422 LLVM_DEBUG(
dbgs() <<
"PEELING loop %" << L->getHeader()->getName()
1423 <<
" with iteration count " << PP.
PeelCount <<
"!\n");
1438 L->setLoopAlreadyUnrolled();
1443 if (OnlyFullUnroll && ((!TripCount && !MaxTripCount) ||
1444 UP.
Count < TripCount || UP.
Count < MaxTripCount)) {
1446 <<
"Not attempting partial/runtime unroll in FullLoopUnroll.\n");
1455 UP.
Runtime &= TripCount == 0 && TripMultiple % UP.
Count != 0;
1458 MDNode *OrigLoopID = L->getLoopID();
1461 Loop *RemainderLoop =
nullptr;
1474 L, ULO, LI, &SE, &DT, &AC, &
TTI, &ORE, PreserveLCSSA, &RemainderLoop,
AA);
1478 if (RemainderLoop) {
1479 std::optional<MDNode *> RemainderLoopID =
1482 if (RemainderLoopID)
1483 RemainderLoop->
setLoopID(*RemainderLoopID);
1487 std::optional<MDNode *> NewLoopID =
1491 L->setLoopID(*NewLoopID);
1495 return UnrollResult;
1502 L->setLoopAlreadyUnrolled();
1504 return UnrollResult;
1509class LoopUnroll :
public LoopPass {
1518 bool OnlyWhenForced;
1525 std::optional<unsigned> ProvidedCount;
1526 std::optional<unsigned> ProvidedThreshold;
1527 std::optional<bool> ProvidedAllowPartial;
1528 std::optional<bool> ProvidedRuntime;
1529 std::optional<bool> ProvidedUpperBound;
1530 std::optional<bool> ProvidedAllowPeeling;
1531 std::optional<bool> ProvidedAllowProfileBasedPeeling;
1532 std::optional<unsigned> ProvidedFullUnrollMaxCount;
1534 LoopUnroll(
int OptLevel = 2,
bool OnlyWhenForced =
false,
1535 bool ForgetAllSCEV =
false,
1536 std::optional<unsigned> Threshold = std::nullopt,
1537 std::optional<unsigned>
Count = std::nullopt,
1538 std::optional<bool> AllowPartial = std::nullopt,
1539 std::optional<bool>
Runtime = std::nullopt,
1540 std::optional<bool> UpperBound = std::nullopt,
1541 std::optional<bool> AllowPeeling = std::nullopt,
1542 std::optional<bool> AllowProfileBasedPeeling = std::nullopt,
1543 std::optional<unsigned> ProvidedFullUnrollMaxCount = std::nullopt)
1544 : LoopPass(
ID), OptLevel(OptLevel), OnlyWhenForced(OnlyWhenForced),
1545 ForgetAllSCEV(ForgetAllSCEV), ProvidedCount(std::
move(
Count)),
1546 ProvidedThreshold(Threshold), ProvidedAllowPartial(AllowPartial),
1547 ProvidedRuntime(
Runtime), ProvidedUpperBound(UpperBound),
1548 ProvidedAllowPeeling(AllowPeeling),
1549 ProvidedAllowProfileBasedPeeling(AllowProfileBasedPeeling),
1550 ProvidedFullUnrollMaxCount(ProvidedFullUnrollMaxCount) {
1554 bool runOnLoop(Loop *L, LPPassManager &LPM)
override {
1560 auto &DT = getAnalysis<DominatorTreeWrapperPass>().getDomTree();
1561 LoopInfo *LI = &getAnalysis<LoopInfoWrapperPass>().getLoopInfo();
1562 ScalarEvolution &SE = getAnalysis<ScalarEvolutionWrapperPass>().getSE();
1563 const TargetTransformInfo &
TTI =
1564 getAnalysis<TargetTransformInfoWrapperPass>().getTTI(
F);
1565 auto &AC = getAnalysis<AssumptionCacheTracker>().getAssumptionCache(
F);
1569 OptimizationRemarkEmitter ORE(&
F);
1570 bool PreserveLCSSA = mustPreserveAnalysisID(
LCSSAID);
1573 L, DT, LI, SE,
TTI, AC, ORE,
nullptr,
nullptr, PreserveLCSSA, OptLevel,
1574 false, OnlyWhenForced, ForgetAllSCEV, ProvidedCount,
1575 ProvidedThreshold, ProvidedAllowPartial, ProvidedRuntime,
1576 ProvidedUpperBound, ProvidedAllowPeeling,
1577 ProvidedAllowProfileBasedPeeling, ProvidedFullUnrollMaxCount);
1579 if (Result == LoopUnrollResult::FullyUnrolled)
1582 return Result != LoopUnrollResult::Unmodified;
1587 void getAnalysisUsage(AnalysisUsage &AU)
const override {
1598char LoopUnroll::ID = 0;
1607 bool ForgetAllSCEV,
int Threshold,
int Count,
1608 int AllowPartial,
int Runtime,
int UpperBound,
1613 return new LoopUnroll(
1614 OptLevel, OnlyWhenForced, ForgetAllSCEV,
1615 Threshold == -1 ? std::nullopt : std::optional<unsigned>(Threshold),
1616 Count == -1 ? std::nullopt : std::optional<unsigned>(
Count),
1617 AllowPartial == -1 ? std::nullopt : std::optional<bool>(AllowPartial),
1619 UpperBound == -1 ? std::nullopt : std::optional<bool>(UpperBound),
1620 AllowPeeling == -1 ? std::nullopt : std::optional<bool>(AllowPeeling));
1633 Loop *ParentL = L.getParentLoop();
1640 std::string LoopName = std::string(L.getName());
1645 true, OptLevel,
true,
1646 OnlyWhenForced, ForgetSCEV, std::nullopt,
1647 std::nullopt,
false,
1678 bool IsCurrentLoopValid =
false;
1685 if (SibLoop == &L) {
1686 IsCurrentLoopValid =
true;
1695 if (!IsCurrentLoopValid) {
1724 if (
auto *LAMProxy = AM.
getCachedResult<LoopAnalysisManagerFunctionProxy>(
F))
1725 LAM = &LAMProxy->getManager();
1730 auto *BFI = (PSI && PSI->hasProfileSummary()) ?
1740 for (
const auto &L : LI) {
1751 while (!Worklist.
empty()) {
1758 Loop *ParentL = L.getParentLoop();
1764 std::optional<bool> LocalAllowPeeling = UnrollOpts.AllowPeeling;
1765 if (PSI && PSI->hasHugeWorkingSetSize())
1766 LocalAllowPeeling =
false;
1767 std::string LoopName = std::string(L.getName());
1771 &L, DT, &LI, SE,
TTI, AC, ORE, BFI, PSI,
1772 true, UnrollOpts.OptLevel,
false,
1773 UnrollOpts.OnlyWhenForced, UnrollOpts.ForgetSCEV,
1775 std::nullopt, UnrollOpts.AllowPartial,
1776 UnrollOpts.AllowRuntime, UnrollOpts.AllowUpperBound, LocalAllowPeeling,
1777 UnrollOpts.AllowProfileBasedPeeling, UnrollOpts.FullUnrollMaxCount,
1789 LAM->clear(L, LoopName);
1801 OS, MapClassName2PassName);
1803 if (UnrollOpts.AllowPartial != std::nullopt)
1804 OS << (*UnrollOpts.AllowPartial ?
"" :
"no-") <<
"partial;";
1805 if (UnrollOpts.AllowPeeling != std::nullopt)
1806 OS << (*UnrollOpts.AllowPeeling ?
"" :
"no-") <<
"peeling;";
1807 if (UnrollOpts.AllowRuntime != std::nullopt)
1808 OS << (*UnrollOpts.AllowRuntime ?
"" :
"no-") <<
"runtime;";
1809 if (UnrollOpts.AllowUpperBound != std::nullopt)
1810 OS << (*UnrollOpts.AllowUpperBound ?
"" :
"no-") <<
"upperbound;";
1811 if (UnrollOpts.AllowProfileBasedPeeling != std::nullopt)
1812 OS << (*UnrollOpts.AllowProfileBasedPeeling ?
"" :
"no-")
1813 <<
"profile-peeling;";
1814 if (UnrollOpts.FullUnrollMaxCount != std::nullopt)
1815 OS <<
"full-unroll-max=" << UnrollOpts.FullUnrollMaxCount <<
';';
1816 OS <<
'O' << UnrollOpts.OptLevel;
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
This file defines DenseMapInfo traits for DenseMap.
This file defines the DenseMap class.
This file defines the DenseSet and SmallDenseSet classes.
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
This header defines various interfaces for pass management in LLVM.
This header provides classes for managing per-loop analyses.
This header provides classes for managing a pipeline of passes over loops in LLVM IR.
static MDNode * getUnrollMetadataForLoop(const Loop *L, StringRef Name)
static cl::opt< unsigned > UnrollMaxCount("unroll-max-count", cl::Hidden, cl::desc("Set the max unroll count for partial and runtime unrolling, for" "testing purposes"))
static cl::opt< unsigned > UnrollCount("unroll-count", cl::Hidden, cl::desc("Use this unroll count for all loops including those with " "unroll_count pragma values, for testing purposes"))
static cl::opt< unsigned > UnrollThresholdDefault("unroll-threshold-default", cl::init(150), cl::Hidden, cl::desc("Default threshold (max size of unrolled " "loop), used in all but O3 optimizations"))
static cl::opt< unsigned > FlatLoopTripCountThreshold("flat-loop-tripcount-threshold", cl::init(5), cl::Hidden, cl::desc("If the runtime tripcount for the loop is lower than the " "threshold, the loop is considered as flat and will be less " "aggressively unrolled."))
static cl::opt< unsigned > UnrollOptSizeThreshold("unroll-optsize-threshold", cl::init(0), cl::Hidden, cl::desc("The cost threshold for loop unrolling when optimizing for " "size"))
static bool hasUnrollFullPragma(const Loop *L)
static cl::opt< bool > UnrollUnrollRemainder("unroll-remainder", cl::Hidden, cl::desc("Allow the loop remainder to be unrolled."))
static unsigned unrollCountPragmaValue(const Loop *L)
static bool hasUnrollEnablePragma(const Loop *L)
static cl::opt< unsigned > PragmaUnrollThreshold("pragma-unroll-threshold", cl::init(16 *1024), cl::Hidden, cl::desc("Unrolled size limit for loops with unroll metadata " "(full, enable, or count)."))
static cl::opt< unsigned > UnrollFullMaxCount("unroll-full-max-count", cl::Hidden, cl::desc("Set the max unroll count for full unrolling, for testing purposes"))
static cl::opt< unsigned > UnrollMaxUpperBound("unroll-max-upperbound", cl::init(8), cl::Hidden, cl::desc("The max of trip count upper bound that is considered in unrolling"))
static std::optional< unsigned > shouldFullUnroll(Loop *L, const TargetTransformInfo &TTI, DominatorTree &DT, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, const unsigned FullUnrollTripCount, const UnrollCostEstimator UCE, const TargetTransformInfo::UnrollingPreferences &UP)
static std::optional< EstimatedUnrollCost > analyzeLoopUnrollCost(const Loop *L, unsigned TripCount, DominatorTree &DT, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, const TargetTransformInfo &TTI, unsigned MaxUnrolledLoopSize, unsigned MaxIterationsCountToAnalyze)
Figure out if the loop is worth full unrolling.
static cl::opt< unsigned > UnrollPartialThreshold("unroll-partial-threshold", cl::Hidden, cl::desc("The cost threshold for partial loop unrolling"))
static cl::opt< bool > UnrollAllowRemainder("unroll-allow-remainder", cl::Hidden, cl::desc("Allow generation of a loop remainder (extra iterations) " "when unrolling a loop."))
static std::optional< unsigned > shouldPartialUnroll(const unsigned LoopSize, const unsigned TripCount, const UnrollCostEstimator UCE, const TargetTransformInfo::UnrollingPreferences &UP)
static cl::opt< unsigned > PragmaUnrollFullMaxIterations("pragma-unroll-full-max-iterations", cl::init(1 '000 '000), cl::Hidden, cl::desc("Maximum allowed iterations to unroll under pragma unroll full."))
static const unsigned NoThreshold
A magic value for use with the Threshold parameter to indicate that the loop unroll should be perform...
static std::optional< unsigned > shouldPragmaUnroll(Loop *L, const PragmaInfo &PInfo, const unsigned TripMultiple, const unsigned TripCount, unsigned MaxTripCount, const UnrollCostEstimator UCE, const TargetTransformInfo::UnrollingPreferences &UP)
static cl::opt< bool > UnrollRevisitChildLoops("unroll-revisit-child-loops", cl::Hidden, cl::desc("Enqueue and re-visit child loops in the loop PM after unrolling. " "This shouldn't typically be needed as child loops (or their " "clones) were already visited."))
static cl::opt< unsigned > UnrollThreshold("unroll-threshold", cl::Hidden, cl::desc("The cost threshold for loop unrolling"))
static cl::opt< bool > UnrollRuntime("unroll-runtime", cl::Hidden, cl::desc("Unroll loops with run-time trip counts"))
static LoopUnrollResult tryToUnrollLoop(Loop *L, DominatorTree &DT, LoopInfo *LI, ScalarEvolution &SE, const TargetTransformInfo &TTI, AssumptionCache &AC, OptimizationRemarkEmitter &ORE, BlockFrequencyInfo *BFI, ProfileSummaryInfo *PSI, bool PreserveLCSSA, int OptLevel, bool OnlyFullUnroll, bool OnlyWhenForced, bool ForgetAllSCEV, std::optional< unsigned > ProvidedCount, std::optional< unsigned > ProvidedThreshold, std::optional< bool > ProvidedAllowPartial, std::optional< bool > ProvidedRuntime, std::optional< bool > ProvidedUpperBound, std::optional< bool > ProvidedAllowPeeling, std::optional< bool > ProvidedAllowProfileBasedPeeling, std::optional< unsigned > ProvidedFullUnrollMaxCount, AAResults *AA=nullptr)
static bool hasRuntimeUnrollDisablePragma(const Loop *L)
static unsigned getFullUnrollBoostingFactor(const EstimatedUnrollCost &Cost, unsigned MaxPercentThresholdBoost)
static cl::opt< unsigned > UnrollThresholdAggressive("unroll-threshold-aggressive", cl::init(300), cl::Hidden, cl::desc("Threshold (max size of unrolled loop) to use in aggressive (O3) " "optimizations"))
static cl::opt< unsigned > UnrollMaxIterationsCountToAnalyze("unroll-max-iteration-count-to-analyze", cl::init(10), cl::Hidden, cl::desc("Don't allow loop unrolling to simulate more than this number of " "iterations when checking full unroll profitability"))
static cl::opt< unsigned > UnrollMaxPercentThresholdBoost("unroll-max-percent-threshold-boost", cl::init(400), cl::Hidden, cl::desc("The maximum 'boost' (represented as a percentage >= 100) applied " "to the threshold when aggressively unrolling a loop due to the " "dynamic cost savings. If completely unrolling a loop will reduce " "the total runtime from X to Y, we boost the loop unroll " "threshold to DefaultThreshold*std::min(MaxPercentThresholdBoost, " "X/Y). This limit avoids excessive code bloat."))
static cl::opt< bool > UnrollAllowPartial("unroll-allow-partial", cl::Hidden, cl::desc("Allows loops to be partially unrolled until " "-unroll-threshold loop size is reached."))
This file exposes an interface to building/using memory SSA to walk memory instructions using a use/d...
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
A manager for alias analyses.
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
A function analysis which provides an AssumptionCache.
An immutable pass that tracks lazily created AssumptionCache objects.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction if the block is well formed or null if the block is not well forme...
Analysis pass which computes BlockFrequencyInfo.
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Conditional or Unconditional Branch instruction.
This is the shared class of boolean and integer constants.
This is an important base class in LLVM.
ValueT lookup(const_arg_type_t< KeyT > Val) const
lookup - Return the entry for the specified key, or a default constructed value if no such entry exis...
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Implements a dense probed hash-table based set.
Analysis pass which computes a DominatorTree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
This class provides an interface for updating the loop pass manager based on mutations to the loop ne...
void addChildLoops(ArrayRef< Loop * > NewChildLoops)
Loop passes should use this method to indicate they have added new child loops of the current loop.
void markLoopAsDeleted(Loop &L, llvm::StringRef Name)
Loop passes should use this method to indicate they have deleted a loop from the nest.
void addSiblingLoops(ArrayRef< Loop * > NewSibLoops)
Loop passes should use this method to indicate they have added new sibling loops to the current loop.
void markLoopAsDeleted(Loop &L)
Analysis pass that exposes the LoopInfo for a function.
void verifyLoop() const
Verify loop structure.
PreservedAnalyses run(Loop &L, LoopAnalysisManager &AM, LoopStandardAnalysisResults &AR, LPMUpdater &U)
PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
void printPipeline(raw_ostream &OS, function_ref< StringRef(StringRef)> MapClassName2PassName)
Represents a single loop in the control flow graph.
void setLoopID(MDNode *LoopID) const
Set the llvm.loop loop id metadata for this loop.
const MDOperand & getOperand(unsigned I) const
unsigned getNumOperands() const
Return number of MDNode operands.
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
Pass interface - Implemented by all 'passes'.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
bool empty() const
Determine if the PriorityWorklist is empty or not.
An analysis pass based on the new PM to deliver ProfileSummaryInfo.
Analysis providing profile information.
Analysis pass that exposes the ScalarEvolution for a function.
The main scalar evolution driver.
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI unsigned getSmallConstantTripCount(const Loop *L)
Returns the exact trip count of the loop if we can compute it, and the result is a small constant.
size_type size() const
Determine the number of elements in the SetVector.
void clear()
Completely clear the SetVector.
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
value_type pop_back_val()
A version of PriorityWorklist that selects small size optimized data structures for the vector and ma...
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
void insert_range(Range &&R)
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StringRef - Represent a constant reference to a string, i.e.
Analysis pass providing the TargetTransformInfo.
Produce an estimate of the unrolled cost of the specified loop.
ConvergenceKind Convergence
bool ConvergenceAllowsRuntime
LLVM_ABI uint64_t getUnrolledLoopSize(const TargetTransformInfo::UnrollingPreferences &UP, unsigned CountOverwrite=0) const
Returns loop size estimation for unrolled loop, given the unrolling configuration specified by UP.
LLVM_ABI bool canUnroll() const
Whether it is legal to unroll this loop.
unsigned NumInlineCandidates
LLVM_ABI UnrollCostEstimator(const Loop *L, const TargetTransformInfo &TTI, const SmallPtrSetImpl< const Value * > &EphValues, unsigned BEInsns)
uint64_t getRolledLoopSize() const
void visit(Iterator Start, Iterator End)
LLVM Value Representation.
std::pair< iterator, bool > insert(const ValueT &V)
iterator find(const_arg_type_t< ValueT > V)
An efficient, type-erasing, non-owning reference to a callable.
This class implements an extremely fast bulk output stream that can only output to a stream.
raw_ostream & indent(unsigned NumSpaces)
indent - Insert 'NumSpaces' spaces.
Abstract Attribute helper functions.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Add a small namespace to avoid name clashes with the classes used in the streaming interface.
DiagnosticInfoOptimizationBase::Argument NV
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool simplifyLoop(Loop *L, DominatorTree *DT, LoopInfo *LI, ScalarEvolution *SE, AssumptionCache *AC, MemorySSAUpdater *MSSAU, bool PreserveLCSSA)
Simplify each loop in a loop nest recursively.
LLVM_ABI std::optional< unsigned > getLoopEstimatedTripCount(Loop *L, unsigned *EstimatedLoopInvocationWeight=nullptr)
Return either:
bool isEqual(const GCNRPTracker::LiveRegSet &S1, const GCNRPTracker::LiveRegSet &S2)
LLVM_ABI void simplifyLoopAfterUnroll(Loop *L, bool SimplifyIVs, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT, AssumptionCache *AC, const TargetTransformInfo *TTI, AAResults *AA=nullptr)
Perform some cleanup and simplifications on loops after unrolling.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
auto successors(const MachineBasicBlock *BB)
@ Runtime
Detect stack use after return if not disabled runtime with (ASAN_OPTIONS=detect_stack_use_after_retur...
OuterAnalysisManagerProxy< ModuleAnalysisManager, Function > ModuleAnalysisManagerFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
LLVM_ABI bool computeUnrollCount(Loop *L, const TargetTransformInfo &TTI, DominatorTree &DT, LoopInfo *LI, AssumptionCache *AC, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, OptimizationRemarkEmitter *ORE, unsigned TripCount, unsigned MaxTripCount, bool MaxOrZero, unsigned TripMultiple, const UnrollCostEstimator &UCE, TargetTransformInfo::UnrollingPreferences &UP, TargetTransformInfo::PeelingPreferences &PP)
LLVM_ABI bool formLCSSARecursively(Loop &L, const DominatorTree &DT, const LoopInfo *LI, ScalarEvolution *SE)
Put a loop nest into LCSSA form.
LLVM_ABI std::optional< MDNode * > makeFollowupLoopID(MDNode *OrigLoopID, ArrayRef< StringRef > FollowupAttrs, const char *InheritOptionsAttrsPrefix="", bool AlwaysNew=false)
Create a new loop identifier for a loop created from a loop transformation.
LLVM_ABI bool shouldOptimizeForSize(const MachineFunction *MF, ProfileSummaryInfo *PSI, const MachineBlockFrequencyInfo *BFI, PGSOQueryType QueryType=PGSOQueryType::Other)
Returns true if machine function MF is suggested to be size-optimized based on the profile.
LLVM_ABI Pass * createLoopUnrollPass(int OptLevel=2, bool OnlyWhenForced=false, bool ForgetAllSCEV=false, int Threshold=-1, int Count=-1, int AllowPartial=-1, int Runtime=-1, int UpperBound=-1, int AllowPeeling=-1)
AnalysisManager< Loop, LoopStandardAnalysisResults & > LoopAnalysisManager
The loop analysis manager.
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
LLVM_ABI void initializeLoopUnrollPass(PassRegistry &)
TargetTransformInfo::PeelingPreferences gatherPeelingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, std::optional< bool > UserAllowPeeling, std::optional< bool > UserAllowProfileBasedPeeling, bool UnrollingSpecficValues=false)
LLVM_ABI CallBase * getLoopConvergenceHeart(const Loop *TheLoop)
Find the convergence heart of the loop.
LLVM_ABI TransformationMode hasUnrollAndJamTransformation(const Loop *L)
cl::opt< bool > ForgetSCEVInLoopUnroll
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
void computePeelCount(Loop *L, unsigned LoopSize, TargetTransformInfo::PeelingPreferences &PP, unsigned TripCount, DominatorTree &DT, ScalarEvolution &SE, const TargetTransformInfo &TTI, AssumptionCache *AC=nullptr, unsigned Threshold=UINT_MAX)
LLVM_TEMPLATE_ABI void appendLoopsToWorklist(RangeT &&, SmallPriorityWorklist< Loop *, 4 > &)
Utility that implements appending of loops onto a worklist given a range.
LLVM_ABI cl::opt< unsigned > SCEVCheapExpansionBudget
FunctionAddr VTableAddr Count
LLVM_ABI TransformationMode hasUnrollTransformation(const Loop *L)
LoopUnrollResult
Represents the result of a UnrollLoop invocation.
@ PartiallyUnrolled
The loop was partially unrolled – we still have a loop, but with a smaller trip count.
@ Unmodified
The loop was not modified.
@ FullyUnrolled
The loop was fully unrolled into straight-line code.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ABI void getLoopAnalysisUsage(AnalysisUsage &AU)
Helper to consistently add the set of standard passes to a loop pass's AnalysisUsage.
void peelLoop(Loop *L, unsigned PeelCount, bool PeelLast, LoopInfo *LI, ScalarEvolution *SE, DominatorTree &DT, AssumptionCache *AC, bool PreserveLCSSA, ValueToValueMapTy &VMap)
VMap is the value-map that maps instructions from the original loop to instructions in the last peele...
const char *const LLVMLoopUnrollFollowupAll
TransformationMode
The mode sets how eager a transformation should be applied.
@ TM_ForcedByUser
The transformation was directed by the user, e.g.
@ TM_Disable
The transformation should not be applied.
@ TM_Enable
The transformation should be applied without considering a cost model.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
LLVM_ABI TargetTransformInfo::UnrollingPreferences gatherUnrollingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, BlockFrequencyInfo *BFI, ProfileSummaryInfo *PSI, llvm::OptimizationRemarkEmitter &ORE, int OptLevel, std::optional< unsigned > UserThreshold, std::optional< unsigned > UserCount, std::optional< bool > UserAllowPartial, std::optional< bool > UserRuntime, std::optional< bool > UserUpperBound, std::optional< unsigned > UserFullUnrollMaxCount)
Gather the various unrolling parameters based on the defaults, compiler flags, TTI overrides and user...
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
OutputIt move(R &&Range, OutputIt Out)
Provide wrappers to std::move which take ranges instead of having to pass begin/end explicitly.
const char *const LLVMLoopUnrollFollowupRemainder
LLVM_ABI PreservedAnalyses getLoopPassPreservedAnalyses()
Returns the minimum set of Analyses that all loop passes must preserve.
const char *const LLVMLoopUnrollFollowupUnrolled
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI MDNode * GetUnrollMetadata(MDNode *LoopID, StringRef Name)
Given an llvm.loop loop id metadata node, returns the loop hint metadata node with the given name (fo...
LLVM_ABI LoopUnrollResult UnrollLoop(Loop *L, UnrollLoopOptions ULO, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT, AssumptionCache *AC, const llvm::TargetTransformInfo *TTI, OptimizationRemarkEmitter *ORE, bool PreserveLCSSA, Loop **RemainderLoop=nullptr, AAResults *AA=nullptr)
Unroll the given loop by Count.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Utility to calculate the size and a few similar metrics for a set of basic blocks.
static LLVM_ABI void collectEphemeralValues(const Loop *L, AssumptionCache *AC, SmallPtrSetImpl< const Value * > &EphValues)
Collect a loop's ephemeral values (those used only by an assume or similar intrinsics in the loop).
The adaptor from a function pass to a loop pass computes these analyses and makes them available to t...
TargetTransformInfo & TTI
A CRTP mix-in to automatically provide informational APIs needed for passes.
const Instruction * Heart
bool RuntimeUnrollMultiExit
bool AllowExpensiveTripCount
bool AddAdditionalAccumulators
unsigned SCEVExpansionBudget