LLVM 24.0.0git
MLInlineAdvisor.cpp
Go to the documentation of this file.
1//===- MLInlineAdvisor.cpp - machine learned InlineAdvisor ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the interface between the inliner and a learned model.
10// It delegates model evaluation to either the AOT compiled model (the
11// 'release' mode) or a runtime-loaded model (the 'development' case).
12//
13//===----------------------------------------------------------------------===//
31#include "llvm/IR/Dominators.h"
33#include "llvm/IR/Module.h"
34#include "llvm/IR/PassManager.h"
36
37using namespace llvm;
38
40 "inliner-interactive-channel-base", cl::Hidden,
42 "Base file path for the interactive mode. The incoming filename should "
43 "have the name <inliner-interactive-channel-base>.in, while the "
44 "outgoing name should be <inliner-interactive-channel-base>.out"));
45static const std::string InclDefaultMsg =
46 (Twine("In interactive mode, also send the default policy decision: ") +
48 .str();
49static cl::opt<bool>
50 InteractiveIncludeDefault("inliner-interactive-include-default", cl::Hidden,
52
54
56 "ml-inliner-skip-policy", cl::Hidden, cl::init(SkipMLPolicyCriteria::Never),
59 "if-caller-not-cold", "if the caller is not cold")));
60
61static cl::opt<std::string> ModelSelector("ml-inliner-model-selector",
62 cl::Hidden, cl::init(""));
63
64static cl::opt<bool> StopImmediatelyForTest("ml-inliner-stop-immediately",
66
67#if defined(LLVM_HAVE_TF_AOT_INLINERSIZEMODEL)
68// codegen-ed file
69#include "InlinerSizeModel.h" // NOLINT
70using CompiledModelType = llvm::InlinerSizeModel;
71#else
73#endif
74
75#if defined(LLVM_HAVE_MLIR_LOWERING_INLINER)
76constexpr bool HaveMLIRLoweringInliner = true;
78#include "llvm/Analysis/InlinerModels.h"
79
80enum class EmitCModelChoice {
81 Default,
82#define MLGO_MODEL(CLASS_NAME, CLI_FLAG) CLASS_NAME,
83#include "llvm/Analysis/InlinerModels.def"
84};
85
87 "mlgo-model", llvm::cl::desc("Select the MLGO model to execute:"),
90 "Use standard heuristic")
91#define MLGO_MODEL(CLASS_NAME, CLI_FLAG) \
92 , clEnumValN(EmitCModelChoice::CLASS_NAME, CLI_FLAG, \
93 "Use the " CLI_FLAG " MLGO model")
94#include "llvm/Analysis/InlinerModels.def"
95 ));
96
97static std::unique_ptr<MLModelRunner>
99 const std::vector<TensorSpec> &InputFeatures) {
100 switch (SelectedMLGOModel) {
102 return nullptr;
103#define MLGO_MODEL(CLASS_NAME, CLI_FLAG) \
104 case EmitCModelChoice::CLASS_NAME: \
105 return std::make_unique<EmitCModelRunner<CLASS_NAME>>(Ctx, InputFeatures);
106#include "llvm/Analysis/InlinerModels.def"
107 }
108 llvm_unreachable("Unknown MLGO model type!");
109}
110#else
111constexpr bool HaveMLIRLoweringInliner = false;
114static inline std::unique_ptr<MLModelRunner>
115createEmitCModelRunner(LLVMContext &, const std::vector<TensorSpec> &) {
116 return nullptr;
117}
118#endif
119
120std::unique_ptr<InlineAdvisor>
122 std::function<bool(CallBase &)> GetDefaultAdvice) {
125 return nullptr;
126 auto RunnerFactory = [&](const std::vector<TensorSpec> &InputFeatures)
127 -> std::unique_ptr<MLModelRunner> {
132 EmbeddedModelRunnerOptions().setModelSelector(ModelSelector));
133 };
134 return std::make_unique<MLInlineAdvisor>(M, MAM, RunnerFactory,
135 GetDefaultAdvice);
136}
137
138#define DEBUG_TYPE "inline-ml"
139
141 "ml-advisor-size-increase-threshold", cl::Hidden,
142 cl::desc("Maximum factor by which expected native size may increase before "
143 "blocking any further inlining."),
144 cl::init(2.0));
145
147 "ml-advisor-keep-fpi-cache", cl::Hidden,
148 cl::desc(
149 "For test - keep the ML Inline advisor's FunctionPropertiesInfo cache"),
150 cl::init(false));
151
152const std::vector<TensorSpec> &MLInlineAdvisor::getInitialFeatureMap() {
153 // clang-format off
154static std::vector<TensorSpec> FeatureMap{
155#define POPULATE_NAMES(DTYPE, SHAPE, NAME, __) TensorSpec::createSpec<DTYPE>(#NAME, SHAPE),
156// InlineCost features - these must come first
158
159// Non-cost features
161#undef POPULATE_NAMES
162};
163 // clang-format on
164 return FeatureMap;
165}
166
167const char *const llvm::DecisionName = "inlining_decision";
170const char *const llvm::DefaultDecisionName = "inlining_default";
173const char *const llvm::RewardName = "delta_size";
174
176 if (auto *CS = dyn_cast<CallBase>(&I))
177 if (Function *Callee = CS->getCalledFunction()) {
178 if (!Callee->isDeclaration()) {
179 return CS;
180 }
181 }
182 return nullptr;
183}
184
187 std::function<
188 std::unique_ptr<MLModelRunner>(const std::vector<TensorSpec> &)>
189 GetModelRunner,
190 std::function<bool(CallBase &)> GetDefaultAdvice)
192 M, MAM.getResult<FunctionAnalysisManagerModuleProxy>(M).getManager()),
194 CG(MAM.getResult<LazyCallGraphAnalysis>(M)),
195 UseIR2Vec(MAM.getCachedResult<IR2VecVocabAnalysis>(M) != nullptr),
196 InitialIRSize(getModuleIRSize()), CurrentIRSize(InitialIRSize),
197 PSI(MAM.getResult<ProfileSummaryAnalysis>(M)) {
198 // Extract the 'call site height' feature - the position of a call site
199 // relative to the farthest statically reachable SCC node. We don't mutate
200 // this value while inlining happens. Empirically, this feature proved
201 // critical in behavioral cloning - i.e. training a model to mimic the manual
202 // heuristic's decisions - and, thus, equally important for training for
203 // improvement.
204 CallGraph CGraph(M);
205 for (auto I = scc_begin(&CGraph); !I.isAtEnd(); ++I) {
206 const std::vector<CallGraphNode *> &CGNodes = *I;
207 unsigned Level = 0;
208 for (auto *CGNode : CGNodes) {
209 Function *F = CGNode->getFunction();
210 if (!F || F->isDeclaration())
211 continue;
212 for (auto &I : instructions(F)) {
213 if (auto *CS = getInlinableCS(I)) {
214 auto *Called = CS->getCalledFunction();
215 auto Pos = FunctionLevels.find(&CG.get(*Called));
216 // In bottom up traversal, an inlinable callee is either in the
217 // same SCC, or to a function in a visited SCC. So not finding its
218 // level means we haven't visited it yet, meaning it's in this SCC.
219 if (Pos == FunctionLevels.end())
220 continue;
221 Level = std::max(Level, Pos->second + 1);
222 }
223 }
224 }
225 for (auto *CGNode : CGNodes) {
226 Function *F = CGNode->getFunction();
227 if (F && !F->isDeclaration())
228 FunctionLevels[&CG.get(*F)] = Level;
229 }
230 }
231 for (auto KVP : FunctionLevels) {
232 AllNodes.insert(KVP.first);
233 EdgeCount += getLocalCalls(KVP.first->getFunction());
234 }
235 NodeCount = AllNodes.size();
236
237 if (auto *IR2VecVocabResult = MAM.getCachedResult<IR2VecVocabAnalysis>(M)) {
238 if (!IR2VecVocabResult->isValid()) {
239 M.getContext().emitError("IR2VecVocabAnalysis is not valid");
240 return;
241 }
242 // Add the IR2Vec features to the feature map
243 auto IR2VecDim = IR2VecVocabResult->getDimension();
244 FeatureMap.push_back(
245 TensorSpec::createSpec<float>("callee_embedding", {IR2VecDim}));
246 FeatureMap.push_back(
247 TensorSpec::createSpec<float>("caller_embedding", {IR2VecDim}));
248 }
251
252 ModelRunner = GetModelRunner(getFeatureMap());
253 if (!ModelRunner) {
254 M.getContext().emitError("Could not create model runner");
255 return;
256 }
257 ModelRunner->switchContext("");
258 ForceStop = StopImmediatelyForTest;
259}
260
262 return CG.lookup(F) ? FunctionLevels.at(CG.lookup(F)) : 0;
263}
264
266 if (!CurSCC || ForceStop)
267 return;
268 FPICache.clear();
269 // Function passes executed between InlinerPass runs may have changed the
270 // module-wide features.
271 // The cgscc pass manager rules are such that:
272 // - if a pass leads to merging SCCs, then the pipeline is restarted on the
273 // merged SCC
274 // - if a pass leads to splitting the SCC, then we continue with one of the
275 // splits
276 // This means that the NodesInLastSCC is a superset (not strict) of the nodes
277 // that subsequent passes would have processed
278 // - in addition, if new Nodes were created by a pass (e.g. CoroSplit),
279 // they'd be adjacent to Nodes in the last SCC. So we just need to check the
280 // boundary of Nodes in NodesInLastSCC for Nodes we haven't seen. We don't
281 // care about the nature of the Edge (call or ref). `FunctionLevels`-wise, we
282 // record them at the same level as the original node (this is a choice, may
283 // need revisiting).
284 // - nodes are only deleted at the end of a call graph walk where they are
285 // batch deleted, so we shouldn't see any dead nodes here.
286 while (!NodesInLastSCC.empty()) {
287 const auto *N = *NodesInLastSCC.begin();
288 assert(!N->isDead());
289 NodesInLastSCC.erase(N);
290 EdgeCount += getLocalCalls(N->getFunction());
291 const auto NLevel = FunctionLevels.at(N);
292 for (const auto &E : *(*N)) {
293 const auto *AdjNode = &E.getNode();
294 assert(!AdjNode->isDead() && !AdjNode->getFunction().isDeclaration());
295 auto I = AllNodes.insert(AdjNode);
296 // We've discovered a new function.
297 if (I.second) {
298 ++NodeCount;
299 NodesInLastSCC.insert(AdjNode);
300 FunctionLevels[AdjNode] = NLevel;
301 }
302 }
303 }
304
305 EdgeCount -= EdgesOfLastSeenNodes;
306 EdgesOfLastSeenNodes = 0;
307
308 // (Re)use NodesInLastSCC to remember the nodes in the SCC right now,
309 // in case the SCC is split before onPassExit and some nodes are split out
310 assert(NodesInLastSCC.empty());
311 for (const auto &N : *CurSCC)
312 NodesInLastSCC.insert(&N);
313}
314
316 // No need to keep this around - function passes will invalidate it.
317 if (!KeepFPICache)
318 FPICache.clear();
319 if (!CurSCC || ForceStop)
320 return;
321 // Keep track of the nodes and edges we last saw. Then, in onPassEntry,
322 // we update the node count and edge count from the subset of these nodes that
323 // survived.
324 EdgesOfLastSeenNodes = 0;
325
326 // Check on nodes that were in SCC onPassEntry
327 for (const LazyCallGraph::Node *N : NodesInLastSCC) {
328 assert(!N->isDead());
329 EdgesOfLastSeenNodes += getLocalCalls(N->getFunction());
330 }
331
332 // Check on nodes that may have got added to SCC
333 for (const auto &N : *CurSCC) {
334 assert(!N.isDead());
335 auto I = NodesInLastSCC.insert(&N);
336 if (I.second)
337 EdgesOfLastSeenNodes += getLocalCalls(N.getFunction());
338 }
339 assert(NodeCount >= NodesInLastSCC.size());
340 assert(EdgeCount >= EdgesOfLastSeenNodes);
341}
342
346
347// Update the internal state of the advisor, and force invalidate feature
348// analysis. Currently, we maintain minimal (and very simple) global state - the
349// number of functions and the number of static calls. We also keep track of the
350// total IR size in this module, to stop misbehaving policies at a certain bloat
351// factor (SizeIncreaseThreshold)
353 bool CalleeWasDeleted) {
354 assert(!ForceStop);
355 Function *Caller = Advice.getCaller();
356 Function *Callee = Advice.getCallee();
357 // The caller features aren't valid anymore.
358 {
361 PA.abandon<LoopAnalysis>();
362 FAM.invalidate(*Caller, PA);
363 }
365 if (Caller == Callee) {
366 assert(!CalleeWasDeleted);
367 // We double-counted CallerAndCalleeEdges - since the caller and callee
368 // would be the same
369 assert(Advice.CallerAndCalleeEdges % 2 == 0);
370 CurrentIRSize += getIRSize(*Caller) - Advice.CallerIRSize;
371 EdgeCount += getCachedFPI(*Caller).DirectCallsToDefinedFunctions -
372 Advice.CallerAndCalleeEdges / 2;
373 // The NodeCount would stay the same.
374 } else {
375 int64_t IRSizeAfter =
376 getIRSize(*Caller) + (CalleeWasDeleted ? 0 : Advice.CalleeIRSize);
377 CurrentIRSize += IRSizeAfter - (Advice.CallerIRSize + Advice.CalleeIRSize);
378
379 // We can delta-update module-wide features. We know the inlining only
380 // changed the caller, and maybe the callee (by deleting the latter). Nodes
381 // are simple to update. For edges, we 'forget' the edges that the caller
382 // and callee used to have before inlining, and add back what they currently
383 // have together.
384 int64_t NewCallerAndCalleeEdges =
386
387 // A dead function's node is not actually removed from the call graph until
388 // the end of the call graph walk, but the node no longer belongs to any
389 // valid SCC.
390 if (CalleeWasDeleted) {
391 --NodeCount;
392 NodesInLastSCC.erase(CG.lookup(*Callee));
393 DeadFunctions.insert(Callee);
394 } else {
395 NewCallerAndCalleeEdges +=
397 }
398 EdgeCount += (NewCallerAndCalleeEdges - Advice.CallerAndCalleeEdges);
399 }
400 if (CurrentIRSize > SizeIncreaseThreshold * InitialIRSize)
401 ForceStop = true;
402
403 assert(CurrentIRSize >= 0 && EdgeCount >= 0 && NodeCount >= 0);
404}
405
406int64_t MLInlineAdvisor::getModuleIRSize() const {
407 int64_t Ret = 0;
408 for (auto &F : M)
409 if (!F.isDeclaration())
410 Ret += getIRSize(F);
411 return Ret;
412}
413
415 auto InsertPair = FPICache.try_emplace(&F);
416 if (!InsertPair.second)
417 return InsertPair.first->second;
418 InsertPair.first->second = FAM.getResult<FunctionPropertiesAnalysis>(F);
419 return InsertPair.first->second;
420}
421
422std::unique_ptr<InlineAdvice> MLInlineAdvisor::getAdviceImpl(CallBase &CB) {
423 if (auto Skip = getSkipAdviceIfUnreachableCallsite(CB))
424 return Skip;
425
426 auto &Caller = *CB.getCaller();
427 auto &Callee = *CB.getCalledFunction();
428
429 auto GetAssumptionCache = [&](Function &F) -> AssumptionCache & {
430 return FAM.getResult<AssumptionAnalysis>(F);
431 };
432 auto &TIR = FAM.getResult<TargetIRAnalysis>(Callee);
433 auto &ORE = FAM.getResult<OptimizationRemarkEmitterAnalysis>(Caller);
434
436 if (!PSI.isFunctionEntryCold(&Caller)) {
437 // Return a MLInlineAdvice, despite delegating to the default advice,
438 // because we need to keep track of the internal state. This is different
439 // from the other instances where we return a "default" InlineAdvice,
440 // which happen at points we won't come back to the MLAdvisor for
441 // decisions requiring that state.
442 return ForceStop ? std::make_unique<InlineAdvice>(this, CB, ORE,
444 : std::make_unique<MLInlineAdvice>(this, CB, ORE,
445 GetDefaultAdvice(CB));
446 }
447 }
448 auto MandatoryKind = InlineAdvisor::getMandatoryKind(CB, FAM, ORE);
449 // If this is a "never inline" case, there won't be any changes to internal
450 // state we need to track, so we can just return the base InlineAdvice, which
451 // will do nothing interesting.
452 // Same thing if this is a recursive case.
453 if (MandatoryKind == InlineAdvisor::MandatoryInliningKind::Never ||
454 &Caller == &Callee)
455 return getMandatoryAdvice(CB, false);
456
457 bool Mandatory =
459
460 // If we need to stop, we won't want to track anymore any state changes, so
461 // we just return the base InlineAdvice, which acts as a noop.
462 if (ForceStop) {
463 ORE.emit([&] {
464 return OptimizationRemarkMissed(DEBUG_TYPE, "ForceStop", &CB)
465 << "Won't attempt inlining because module size grew too much.";
466 });
467 return std::make_unique<InlineAdvice>(this, CB, ORE, Mandatory);
468 }
469
470 int CostEstimate = 0;
471 if (!Mandatory) {
472 auto IsCallSiteInlinable =
473 llvm::getInliningCostEstimate(CB, TIR, GetAssumptionCache);
474 if (!IsCallSiteInlinable) {
475 // We can't inline this for correctness reasons, so return the base
476 // InlineAdvice, as we don't care about tracking any state changes (which
477 // won't happen).
478 return std::make_unique<InlineAdvice>(this, CB, ORE, false);
479 }
480 CostEstimate = *IsCallSiteInlinable;
481 }
482
483 const auto CostFeatures =
484 llvm::getInliningCostFeatures(CB, TIR, GetAssumptionCache);
485 if (!CostFeatures) {
486 return std::make_unique<InlineAdvice>(this, CB, ORE, false);
487 }
488
489 if (Mandatory)
490 return getMandatoryAdvice(CB, true);
491
492 auto NumCtantParams = 0;
493 for (auto I = CB.arg_begin(), E = CB.arg_end(); I != E; ++I) {
494 NumCtantParams += (isa<Constant>(*I));
495 }
496
497 auto &CallerBefore = getCachedFPI(Caller);
498 auto &CalleeBefore = getCachedFPI(Callee);
499
500 *ModelRunner->getTensor<int64_t>(FeatureIndex::callee_basic_block_count) =
501 CalleeBefore.BasicBlockCount;
502 *ModelRunner->getTensor<int64_t>(FeatureIndex::callsite_height) =
504 *ModelRunner->getTensor<int64_t>(FeatureIndex::node_count) = NodeCount;
505 *ModelRunner->getTensor<int64_t>(FeatureIndex::nr_ctant_params) =
506 NumCtantParams;
507 *ModelRunner->getTensor<int64_t>(FeatureIndex::edge_count) = EdgeCount;
508 *ModelRunner->getTensor<int64_t>(FeatureIndex::caller_users) =
509 CallerBefore.Uses;
510 *ModelRunner->getTensor<int64_t>(
511 FeatureIndex::caller_conditionally_executed_blocks) =
512 CallerBefore.BlocksReachedFromConditionalInstruction;
513 *ModelRunner->getTensor<int64_t>(FeatureIndex::caller_basic_block_count) =
514 CallerBefore.BasicBlockCount;
515 *ModelRunner->getTensor<int64_t>(
516 FeatureIndex::callee_conditionally_executed_blocks) =
517 CalleeBefore.BlocksReachedFromConditionalInstruction;
518 *ModelRunner->getTensor<int64_t>(FeatureIndex::callee_users) =
519 CalleeBefore.Uses;
520 *ModelRunner->getTensor<int64_t>(FeatureIndex::cost_estimate) = CostEstimate;
521 *ModelRunner->getTensor<int64_t>(FeatureIndex::is_callee_avail_external) =
522 Callee.hasAvailableExternallyLinkage();
523 *ModelRunner->getTensor<int64_t>(FeatureIndex::is_caller_avail_external) =
524 Caller.hasAvailableExternallyLinkage();
525
526 if (UseIR2Vec) {
527 // Python side expects float embeddings. The IR2Vec embeddings are doubles
528 // as of now due to the restriction of fromJSON method used by the
529 // readVocabulary method in ir2vec::Embeddings.
530 auto setEmbedding = [&](const ir2vec::Embedding &Embedding,
531 FeatureIndex Index) {
532 llvm::transform(Embedding, ModelRunner->getTensor<float>(Index),
533 [](double Val) { return static_cast<float>(Val); });
534 };
535
536 setEmbedding(CalleeBefore.getFunctionEmbedding(),
538 setEmbedding(CallerBefore.getFunctionEmbedding(),
540 }
541
542 // Add the cost features
543 for (size_t I = 0;
545 *ModelRunner->getTensor<int64_t>(inlineCostFeatureToMlFeature(
546 static_cast<InlineCostFeatureIndex>(I))) = CostFeatures->at(I);
547 }
548 // This one would have been set up to be right at the end.
550 *ModelRunner->getTensor<int64_t>(getFeatureMap().size() - 1) =
552 return getAdviceFromModel(CB, ORE);
553}
554
555std::unique_ptr<MLInlineAdvice>
558 return std::make_unique<MLInlineAdvice>(
559 this, CB, ORE, static_cast<bool>(ModelRunner->evaluate<int64_t>()));
560}
561
562std::unique_ptr<InlineAdvice>
563MLInlineAdvisor::getSkipAdviceIfUnreachableCallsite(CallBase &CB) {
564 if (!FAM.getResult<DominatorTreeAnalysis>(*CB.getCaller())
565 .isReachableFromEntry(CB.getParent()))
566 return std::make_unique<InlineAdvice>(this, CB, getCallerORE(CB), false);
567 return nullptr;
568}
569
570std::unique_ptr<InlineAdvice> MLInlineAdvisor::getMandatoryAdvice(CallBase &CB,
571 bool Advice) {
572 // Make sure we track inlinings in all cases - mandatory or not.
573 if (auto Skip = getSkipAdviceIfUnreachableCallsite(CB))
574 return Skip;
575 if (Advice && !ForceStop)
576 return getMandatoryAdviceImpl(CB);
577
578 // If this is a "never inline" case, there won't be any changes to internal
579 // state we need to track, so we can just return the base InlineAdvice, which
580 // will do nothing interesting.
581 // Same if we are forced to stop - we don't track anymore.
582 return std::make_unique<InlineAdvice>(this, CB, getCallerORE(CB), Advice);
583}
584
585std::unique_ptr<MLInlineAdvice>
587 return std::make_unique<MLInlineAdvice>(this, CB, getCallerORE(CB), true);
588}
589
590void MLInlineAdvisor::print(raw_ostream &OS) const {
591 OS << "[MLInlineAdvisor] Nodes: " << NodeCount << " Edges: " << EdgeCount
592 << " EdgesOfLastSeenNodes: " << EdgesOfLastSeenNodes << "\n";
593 OS << "[MLInlineAdvisor] FPI:\n";
594 for (auto I : FPICache) {
595 OS << I.first->getName() << ":\n";
596 I.second.print(OS);
597 OS << "\n";
598 }
599 OS << "\n";
600 OS << "[MLInlineAdvisor] FuncLevels:\n";
601 for (auto I : FunctionLevels)
602 OS << (DeadFunctions.contains(&I.first->getFunction())
603 ? "<deleted>"
604 : I.first->getFunction().getName())
605 << " : " << I.second << "\n";
606
607 OS << "\n";
608}
609
612 bool Recommendation)
613 : InlineAdvice(Advisor, CB, ORE, Recommendation),
614 CallerIRSize(Advisor->isForcedToStop() ? 0 : Advisor->getIRSize(*Caller)),
615 CalleeIRSize(Advisor->isForcedToStop() ? 0 : Advisor->getIRSize(*Callee)),
616 CallerAndCalleeEdges(Advisor->isForcedToStop()
617 ? 0
618 : (Advisor->getLocalCalls(*Caller) +
619 Advisor->getLocalCalls(*Callee))),
620 PreInlineCallerFPI(Advisor->getCachedFPI(*Caller)) {
621 if (Recommendation)
622 FPU.emplace(Advisor->getCachedFPI(*getCaller()), CB);
623}
624
625void MLInlineAdvice::reportContextForRemark(
627 using namespace ore;
628 OR << NV("Callee", Callee->getName());
629 for (size_t I = 0; I < getAdvisor()->getFeatureMap().size(); ++I)
630 OR << NV(getAdvisor()->getFeatureMap()[I].name(),
631 *getAdvisor()->getModelRunner().getTensor<int64_t>(I));
632 OR << NV("ShouldInline", isInliningRecommended());
633}
634
638
640 ORE.emit([&]() {
641 OptimizationRemark R(DEBUG_TYPE, "InliningSuccess", DLoc, Block);
642 reportContextForRemark(R);
643 return R;
644 });
645 getAdvisor()->onSuccessfulInlining(*this, /*CalleeWasDeleted*/ false);
646}
647
649 ORE.emit([&]() {
650 OptimizationRemark R(DEBUG_TYPE, "InliningSuccessWithCalleeDeleted", DLoc,
651 Block);
652 reportContextForRemark(R);
653 return R;
654 });
655 getAdvisor()->onSuccessfulInlining(*this, /*CalleeWasDeleted*/ true);
656}
657
659 const InlineResult &Result) {
660 getAdvisor()->getCachedFPI(*Caller) = PreInlineCallerFPI;
661 ORE.emit([&]() {
662 OptimizationRemarkMissed R(DEBUG_TYPE, "InliningAttemptedAndUnsuccessful",
663 DLoc, Block);
664 reportContextForRemark(R);
665 return R;
666 });
667}
669 assert(!FPU);
670 ORE.emit([&]() {
671 OptimizationRemarkMissed R(DEBUG_TYPE, "IniningNotAttempted", DLoc, Block);
672 reportContextForRemark(R);
673 return R;
674 });
675}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Expand Atomic instructions
This file provides interfaces used to build and manipulate a call graph, which is a very useful tool ...
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
@ Default
This file implements a model runner wrapping an EmitC compiled ML model.
#define DEBUG_TYPE
Module.h This file contains the declarations for the Module class.
This header defines various interfaces for pass management in LLVM.
#define INLINE_COST_FEATURE_ITERATOR(M)
#define INLINE_FEATURE_ITERATOR(M)
Implements a lazy call graph analysis and related passes for the new pass manager.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file provides helper functions for creating MLModelRunners and checking model validity in releas...
static cl::opt< bool > KeepFPICache("ml-advisor-keep-fpi-cache", cl::Hidden, cl::desc("For test - keep the ML Inline advisor's FunctionPropertiesInfo cache"), cl::init(false))
static cl::opt< std::string > ModelSelector("ml-inliner-model-selector", cl::Hidden, cl::init(""))
static const EmitCModelChoice SelectedMLGOModel
CallBase * getInlinableCS(Instruction &I)
constexpr bool HaveMLIRLoweringInliner
NoopSavedModelImpl CompiledModelType
SkipMLPolicyCriteria
EmitCModelChoice
static cl::opt< std::string > InteractiveChannelBaseName("inliner-interactive-channel-base", cl::Hidden, cl::desc("Base file path for the interactive mode. The incoming filename should " "have the name <inliner-interactive-channel-base>.in, while the " "outgoing name should be <inliner-interactive-channel-base>.out"))
#define POPULATE_NAMES(DTYPE, SHAPE, NAME, __)
static std::unique_ptr< MLModelRunner > createEmitCModelRunner(LLVMContext &, const std::vector< TensorSpec > &)
static cl::opt< bool > StopImmediatelyForTest("ml-inliner-stop-immediately", cl::Hidden)
static cl::opt< float > SizeIncreaseThreshold("ml-advisor-size-increase-threshold", cl::Hidden, cl::desc("Maximum factor by which expected native size may increase before " "blocking any further inlining."), cl::init(2.0))
static const std::string InclDefaultMsg
static cl::opt< SkipMLPolicyCriteria > SkipPolicy("ml-inliner-skip-policy", cl::Hidden, cl::init(SkipMLPolicyCriteria::Never), cl::values(clEnumValN(SkipMLPolicyCriteria::Never, "never", "never"), clEnumValN(SkipMLPolicyCriteria::IfCallerIsNotCold, "if-caller-not-cold", "if the caller is not cold")))
static cl::opt< bool > InteractiveIncludeDefault("inliner-interactive-include-default", cl::Hidden, cl::desc(InclDefaultMsg))
FunctionAnalysisManager FAM
ModuleAnalysisManager MAM
This builds on the llvm/ADT/GraphTraits.h file to find the strongly connected components (SCCs) of a ...
static const char * name
This pass exposes codegen information to IR-level passes.
A function analysis which provides an AssumptionCache.
A cache of @llvm.assume calls within a function.
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
The basic data container for the call graph of a Module of IR.
Definition CallGraph.h:72
Common features for diagnostics dealing with optimization remarks that are used by both IR and MIR pa...
Analysis pass which computes a DominatorTree.
Definition Dominators.h:241
int64_t DirectCallsToDefinedFunctions
Number of direct calls made from this function to other functions defined in this module.
This analysis provides the vocabulary for IR2Vec.
Definition IR2Vec.h:637
Function *const Callee
Function *const Caller
Caller and Callee are pre-inlining.
const BasicBlock *const Block
OptimizationRemarkEmitter & ORE
InlineAdvisor *const Advisor
LLVM_ABI InlineAdvice(InlineAdvisor *Advisor, CallBase &CB, OptimizationRemarkEmitter &ORE, bool IsInliningRecommended)
const DebugLoc DLoc
bool isInliningRecommended() const
Get the inlining recommendation.
OptimizationRemarkEmitter & getCallerORE(CallBase &CB)
FunctionAnalysisManager & FAM
static MandatoryInliningKind getMandatoryKind(CallBase &CB, FunctionAnalysisManager &FAM, OptimizationRemarkEmitter &ORE)
InlineAdvisor(InlineAdvisor &&)=delete
InlineResult is basically true or false.
Definition InlineCost.h:181
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An analysis pass which computes the call graph for a module.
A node in the call graph.
An SCC of the call graph.
Analysis pass that exposes the LoopInfo for a function.
Definition LoopInfo.h:594
InlineAdvice that tracks changes post inlining.
void updateCachedCallerFPI(FunctionAnalysisManager &FAM) const
const int64_t CallerIRSize
MLInlineAdvice(MLInlineAdvisor *Advisor, CallBase &CB, OptimizationRemarkEmitter &ORE, bool Recommendation)
const int64_t CalleeIRSize
void recordInliningImpl() override
Function * getCaller() const
const int64_t CallerAndCalleeEdges
void recordUnsuccessfulInliningImpl(const InlineResult &Result) override
Function * getCallee() const
void recordInliningWithCalleeDeletedImpl() override
void recordUnattemptedInliningImpl() override
const std::vector< TensorSpec > & getFeatureMap() const
std::unique_ptr< MLModelRunner > ModelRunner
FunctionPropertiesInfo & getCachedFPI(Function &) const
void onPassExit(LazyCallGraph::SCC *SCC) override
This must be called when the Inliner pass is exited, as function passes may be run subsequently.
void onSuccessfulInlining(const MLInlineAdvice &Advice, bool CalleeWasDeleted)
static const std::vector< TensorSpec > & getInitialFeatureMap()
virtual std::unique_ptr< MLInlineAdvice > getMandatoryAdviceImpl(CallBase &CB)
void onPassEntry(LazyCallGraph::SCC *SCC) override
This must be called when the Inliner pass is entered, to allow the InlineAdvisor update internal stat...
MLInlineAdvisor(Module &M, ModuleAnalysisManager &MAM, std::function< std::unique_ptr< MLModelRunner >(const std::vector< TensorSpec > &)> GetModelRunner, std::function< bool(CallBase &)> GetDefaultAdvice)
int64_t getLocalCalls(Function &F)
std::vector< TensorSpec > FeatureMap
virtual std::unique_ptr< MLInlineAdvice > getAdviceFromModel(CallBase &CB, OptimizationRemarkEmitter &ORE)
int64_t getIRSize(Function &F) const
std::function< bool(CallBase &)> GetDefaultAdvice
std::unique_ptr< InlineAdvice > getAdviceImpl(CallBase &CB) override
std::unique_ptr< InlineAdvice > getMandatoryAdvice(CallBase &CB, bool Advice) override
unsigned getInitialFunctionLevel(const Function &F) const
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
A mock class satisfying the interface expected by ReleaseModeModelRunner for its TGen parameter.
The optimization diagnostic interface.
Diagnostic information for missed-optimization remarks.
Diagnostic information for applied optimization remarks.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & abandon()
Mark an analysis as abandoned.
Definition Analysis.h:171
An analysis pass based on the new PM to deliver ProfileSummaryInfo.
Analysis pass providing the TargetTransformInfo.
static TensorSpec createSpec(const std::string &Name, const std::vector< int64_t > &Shape, int Port=0)
Definition TensorSpec.h:65
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
Definition DenseSet.h:182
const ParentTy * getParent() const
Definition ilist_node.h:34
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
Add a small namespace to avoid name clashes with the classes used in the streaming interface.
This is an optimization pass for GlobalISel generic memory operations.
constexpr FeatureIndex inlineCostFeatureToMlFeature(InlineCostFeatureIndex Feature)
LLVM_ABI const char *const DefaultDecisionName
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
scc_iterator< T > scc_begin(const T &G)
Construct the begin iterator for a deduced graph type T.
LLVM_ABI std::unique_ptr< InlineAdvisor > getReleaseModeAdvisor(Module &M, ModuleAnalysisManager &MAM, std::function< bool(CallBase &)> GetDefaultAdvice)
InnerAnalysisManagerProxy< FunctionAnalysisManager, Module > FunctionAnalysisManagerModuleProxy
Provide the FunctionAnalysisManager to Module proxy.
bool isReleaseModelValid(StringRef InteractiveChannelBaseName, const cl::opt< EnumType, ExternalStorage, ParserClass > &SelectedModel, EnumType DefaultModelVal=EnumType::Default)
Helper to check if a release-mode ML advisor has a valid model to execute.
Definition MLGOUtils.h:35
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
Definition STLExtras.h:2026
LLVM_ABI const TensorSpec DefaultDecisionSpec
LLVM_ABI const char *const DecisionName
static const std::vector< TensorSpec > InputFeatures
LLVM_ABI std::optional< InlineCostFeatures > getInliningCostFeatures(CallBase &Call, TargetTransformInfo &CalleeTTI, function_ref< AssumptionCache &(Function &)> GetAssumptionCache, function_ref< BlockFrequencyInfo &(Function &)> GetBFI=nullptr, function_ref< const TargetLibraryInfo &(Function &)> GetTLI=nullptr, ProfileSummaryInfo *PSI=nullptr, OptimizationRemarkEmitter *ORE=nullptr)
Get the expanded cost features.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI const TensorSpec InlineDecisionSpec
LLVM_ABI const char *const RewardName
LLVM_ABI std::optional< int > getInliningCostEstimate(CallBase &Call, TargetTransformInfo &CalleeTTI, function_ref< AssumptionCache &(Function &)> GetAssumptionCache, function_ref< BlockFrequencyInfo &(Function &)> GetBFI=nullptr, function_ref< const TargetLibraryInfo &(Function &)> GetTLI=nullptr, ProfileSummaryInfo *PSI=nullptr, OptimizationRemarkEmitter *ORE=nullptr)
Get the cost estimate ignoring thresholds.
std::unique_ptr< MLModelRunner > createReleaseModeModelRunner(LLVMContext &Ctx, const std::vector< TensorSpec > &InputFeatures, StringRef DecisionName, const std::string &InteractiveChannelBaseName, const TensorSpec &InteractiveDecisionSpec, CreateEmitCFunc &&CreateEmitCModelRunner, const EmbeddedModelRunnerOptions &Options={})
Helper to construct the appropriate MLModelRunner in release mode:
Definition MLGOUtils.h:60
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
Definition MIRParser.h:39
#define N
ReleaseModeModelRunner - production mode implementation of the MLModelRunner.
Embedding is a datatype that wraps std::vector<double>.
Definition IR2Vec.h:88