LLVM 24.0.0git
VPlanTransforms.h
Go to the documentation of this file.
1//===- VPlanTransforms.h - Utility VPlan to VPlan transforms --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file provides utility VPlan to VPlan transformations.
11//===----------------------------------------------------------------------===//
12
13#ifndef LLVM_TRANSFORMS_VECTORIZE_VPLANTRANSFORMS_H
14#define LLVM_TRANSFORMS_VECTORIZE_VPLANTRANSFORMS_H
15
16#include "VPlan.h"
17#include "VPlanVerifier.h"
19#include "llvm/ADT/ScopeExit.h"
23#include "llvm/Support/Regex.h"
24
25namespace llvm {
26
29class Instruction;
30class Loop;
31class LoopVersioning;
33class PHINode;
34class ScalarEvolution;
38class VPRecipeBuilder;
39struct VFRange;
40
42
43#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
49#endif
50
52 /// Helper to run a VPlan pass \p Pass on \p VPlan, forwarding extra arguments
53 /// to the pass. Performs verification/printing after each VPlan pass if
54 /// requested via command line options.
55 template <bool EnableVerify = true, typename PassTy, typename... ArgsTy>
56 static decltype(auto) runPass(StringRef PassName, PassTy &&Pass, VPlan &Plan,
57 ArgsTy &&...Args) {
58#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
59 static DenseMap<std::pair<Function *, StringRef /* Pass */>, unsigned>
60 PassCounter;
62 // Computing these is expensive, so only do it if any VPlan printing has
63 // been requested.
64 unsigned Instance;
65 std::string NumberedPassName;
66
68 !VPlanPrintBeforePasses.empty() || !VPlanPrintAfterPasses.empty()) {
69 Instance = ++PassCounter[{Fn, PassName}];
70
71 NumberedPassName = Instance == 1
72 ? PassName.str()
73 : (PassName + "@" + Twine(Instance)).str();
74 }
75
76 auto PrintPlan = [&](StringRef BeforeOrAfterStr) {
77 dbgs() << "VPlan for loop in '" << Fn->getName() << "' "
78 << BeforeOrAfterStr << " " << NumberedPassName << '\n';
81 else
82 dbgs() << Plan << '\n';
83 };
84
85 auto MatchesPassListOption = [&](const cl::list<std::string> &ListOpt) {
86 return (ListOpt.getNumOccurrences() > 0 &&
87 any_of(ListOpt, [&](StringRef Entry) {
88 return Regex(Entry).match(NumberedPassName);
89 }));
90 };
91
92 if (VPlanPrintBeforeAll || MatchesPassListOption(VPlanPrintBeforePasses))
93 PrintPlan("before");
94#endif
95
96 scope_exit PostTransformActions{[&]() {
97#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
98 // Make sure to print before verification, so that output is more useful
99 // in case of failures:
100 if (VPlanPrintAfterAll || MatchesPassListOption(VPlanPrintAfterPasses))
101 PrintPlan("after");
102#endif
103 if (VerifyEachVPlan && EnableVerify) {
104 if (!verifyVPlanIsValid(Plan))
105 report_fatal_error("Broken VPlan found, compilation aborted!");
106 }
107 }};
108
109 return std::forward<PassTy>(Pass)(Plan, std::forward<ArgsTy>(Args)...);
110 }
111#define RUN_VPLAN_PASS(PASS, ...) \
112 llvm::VPlanTransforms::runPass(#PASS, PASS, __VA_ARGS__)
113#define RUN_VPLAN_PASS_NO_VERIFY(PASS, ...) \
114 llvm::VPlanTransforms::runPass<false>(#PASS, PASS, __VA_ARGS__)
115
116 /// Create a base VPlan0, serving as the common starting point for all later
117 /// candidates. It consists of an initial plain CFG loop with loop blocks from
118 /// \p TheLoop being directly translated to VPBasicBlocks with VPInstruction
119 /// corresponding to the input IR.
120 ///
121 /// The created loop is wrapped in an initial skeleton to facilitate
122 /// vectorization, consisting of a vector pre-header, an exit block for the
123 /// main vector loop (middle.block) and a new block as preheader of the scalar
124 /// loop (scalar.ph). See below for an illustration. It also creates a
125 /// VPValue expression for the original trip count.
126 ///
127 /// [ ] <-- Plan's entry VPIRBasicBlock, wrapping the original loop's
128 /// / \ old preheader. Will contain iteration number check and SCEV
129 /// | | expansions.
130 /// | |
131 /// / v
132 /// | [ ] <-- vector loop bypass (may consist of multiple blocks) will be
133 /// | / | added later.
134 /// | / v
135 /// || [ ] <-- vector pre header.
136 /// |/ |
137 /// | v
138 /// | [ ] \ <-- plain CFG loop wrapping original loop to be vectorized.
139 /// | [ ]_|
140 /// | |
141 /// | v
142 /// | [ ] <--- middle-block with the branch to successors
143 /// | / |
144 /// | / |
145 /// | | v
146 /// \--->[ ] <--- scalar preheader (initial a VPBasicBlock, which will be
147 /// | | replaced later by a VPIRBasicBlock wrapping the scalar
148 /// | | preheader basic block.
149 /// | |
150 /// v <-- edge from middle to exit iff epilogue is not required.
151 /// | [ ] \
152 /// | [ ]_| <-- old scalar loop to handle remainder (scalar epilogue,
153 /// | | header wrapped in VPIRBasicBlock).
154 /// \ |
155 /// \ v
156 /// >[ ] <-- original loop exit block(s), wrapped in VPIRBasicBlocks.
157 LLVM_ABI_FOR_TEST static std::unique_ptr<VPlan>
158 buildVPlan0(Loop *TheLoop, LoopInfo &LI, Type *InductionTy,
159 PredicatedScalarEvolution &PSE, LoopVersioning *LVer = nullptr,
160 function_ref<const BranchProbabilityInfo &()> GetBPI = nullptr);
161
162 /// Add execution frequencies to each recipe in the loop body of \p Plan.
163 /// Frequencies are computed from the branch weights in \p Plan.
164 static void recordExecutionFrequencies(VPlan &Plan);
165
166 /// Replace VPPhi recipes in \p Plan's header with corresponding
167 /// VPHeaderPHIRecipe subclasses for inductions, reductions, and
168 /// fixed-order recurrences. This processes all header phis and creates
169 /// the appropriate widened recipe for each one. For fixed-order
170 /// recurrences, also creates FirstOrderRecurrenceSplice instructions and
171 /// sinks/hoists users as needed. Returns false if any fixed-order
172 /// recurrence cannot be handled.
174 VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &OrigLoop,
175 OptimizationRemarkEmitter *ORE, const VPDominatorTree &VPDT,
176 const MapVector<PHINode *, InductionDescriptor> &Inductions,
177 const MapVector<PHINode *, RecurrenceDescriptor> &Reductions,
178 const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
179 const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering);
180
181 /// Finalize SCEV predicates by adding induction predicates from \p Plan to
182 /// \p PSE and checking constraints. Returns false if predicated IVs have
183 /// outside-loop uses via ExitingIVValue, if SCEV predicate complexity exceeds
184 /// \p SCEVCheckThreshold, or if predicates are needed but \p OptForSize is
185 /// true.
186 static bool
187 finalizeSCEVPredicates(VPlan &Plan, PredicatedScalarEvolution &PSE,
188 bool OptForSize, unsigned SCEVCheckThreshold,
189 OptimizationRemarkEmitter *ORE, Loop *TheLoop);
190
191 /// Create VPReductionRecipes for in-loop reductions. This processes chains
192 /// of operations contributing to in-loop reductions and creates appropriate
193 /// VPReductionRecipe instances.
194 static void createInLoopReductionRecipes(VPlan &Plan, ElementCount MinVF);
195
196 /// If a check is needed to guard executing the scalar epilogue loop, it will
197 /// be added to the middle block.
198 LLVM_ABI_FOR_TEST static void addMiddleCheck(VPlan &Plan);
199
200 // Create a check in \p CheckBlock to see if the vector loop should be
201 // executed. May create VPExpandSCEV recipes in the plan's entry block.
202 static void addMinimumIterationCheck(
203 VPlan &Plan, ElementCount VF, unsigned UF,
204 ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue,
205 bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights,
206 DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock);
207
208 /// Add a new check block before the vector preheader to \p Plan to check if
209 /// the main vector loop should be executed (TC >= VF * UF).
210 static void
211 addIterationCountCheckBlock(VPlan &Plan, ElementCount VF, unsigned UF,
212 bool RequiresScalarEpilogue, Loop *OrigLoop,
214 DebugLoc DL, PredicatedScalarEvolution &PSE);
215
216 /// Add a check to \p Plan to see if the epilogue vector loop should be
217 /// executed.
219 VPlan &Plan, VPValue *MainVectorTripCount, bool RequiresScalarEpilogue,
220 ElementCount EpilogueVF, unsigned MainLoopStep, unsigned EpilogueLoopStep,
221 ScalarEvolution &SE);
222
223 /// Replace loops in \p Plan's flat CFG with VPRegionBlocks, turning \p Plan's
224 /// flat CFG into a hierarchical CFG. For the outermost loop, also create the
225 /// canonical IV's increment and adjust the latch terminator: replace
226 /// BranchOnCond with BranchOnCount, using \p DL for the canonical IV.
227 LLVM_ABI_FOR_TEST static void createLoopRegions(VPlan &Plan, DebugLoc DL);
228
229 /// Connect \p CheckBlock to \p Plan, branching on \p Cond.
230 static void attachVPCheckBlock(VPlan &Plan, VPValue *Cond,
231 VPBasicBlock *CheckBlock,
232 bool AddBranchWeights);
233 static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock,
234 bool AddBranchWeights);
235
236 /// Generate \p Checks as recipes and attach the check block to \p Plan.
237 static void attachMemoryChecks(VPlan &Plan,
239 ScalarEvolution &SE, DebugLoc DL,
240 bool AddBranchWeights);
241
242 /// Model the blocks the executed \p MainPlan generated for the main vector
243 /// loop in \p EpiPlan during epilogue vectorization, wrapping each in a
244 /// VPIRBasicBlock, with \p EnteredFrom the block \p EpiPlan is entered from.
245 /// Edges from blocks bypassing both vector loops are redirected to \p
246 /// EpiPlan's scalar preheader, all others are mirrored.
247 static void modelGeneratedMainLoopBlocks(VPlan &EpiPlan, VPlan &MainPlan,
248 VPIRBasicBlock *EnteredFrom);
249
250 /// Replaces the VPInstructions in \p Plan with corresponding
251 /// widen recipes. Returns false if any VPInstructions could not be converted
252 /// to a wide recipe if needed. Uses \p PSE to detect contiguous memory
253 /// accesses w.r.t. the \p OuterLoop induction variable.
255 VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE,
256 Loop *OuterLoop);
257
258 /// Try to legalize reductions with multiple in-loop uses. Currently only
259 /// strict and non-strict min/max reductions used by FindLastIV reductions are
260 /// supported, corresponding to computing the first and last argmin/argmax,
261 /// respectively. Otherwise return false.
262 static bool handleMultiUseReductions(VPlan &Plan,
263 OptimizationRemarkEmitter *ORE,
264 Loop *TheLoop);
265
266 /// Check if \p Plan contains any FMaxNum or FMinNum reductions. If they do,
267 /// try to update the vector loop to exit early if any input is NaN and resume
268 /// executing in the scalar loop to handle the NaNs there. Return false if
269 /// this attempt was unsuccessful.
270 static bool handleMaxMinNumReductions(VPlan &Plan);
271
272 /// Check if \p Plan contains any FindLast reductions. If it does, try to
273 /// update the vector loop to save the appropriate state using selects
274 /// for entire vectors for both the latest mask containing at least one active
275 /// element and the corresponding data vector. Return false if this attempt
276 /// was unsuccessful.
277 static bool handleFindLastReductions(VPlan &Plan);
278
279 /// Clear NSW/NUW flags from reduction instructions if necessary.
280 static void clearReductionWrapFlags(VPlan &Plan);
281
282 /// Explicitly unroll \p Plan by \p UF.
283 static void unrollByUF(VPlan &Plan, unsigned UF);
284
285 /// Replace replicating VPReplicateRecipe, VPScalarIVStepsRecipe and
286 /// VPInstruction in \p Plan with \p VF single-scalar recipes. Replicate
287 /// regions are dissolved by replicating their blocks and their recipes \p VF
288 /// times.
289 /// TODO: Also dissolve replicate regions with live outs.
290 static void replicateByVF(VPlan &Plan, ElementCount VF);
291
292 /// Optimize \p Plan based on \p BestVF and \p BestUF. This may restrict the
293 /// resulting plan to \p BestVF and \p BestUF.
294 static void optimizeForVFAndUF(VPlan &Plan, ElementCount BestVF,
295 unsigned BestUF,
296 PredicatedScalarEvolution &PSE);
297
298 /// Try to simplify VPInstruction::ExplicitVectorLength recipes when the AVL
299 /// is known to be <= VF, replacing them with the AVL directly.
300 static bool simplifyKnownEVL(VPlan &Plan, ElementCount VF,
301 PredicatedScalarEvolution &PSE);
302
303 /// Apply VPlan-to-VPlan optimizations to \p Plan, including induction recipe
304 /// optimizations, dead recipe removal, replicate region optimizations and
305 /// block merging.
306 LLVM_ABI_FOR_TEST static void optimize(VPlan &Plan);
307
308 /// Remove redundant VPBasicBlocks by merging them into their single
309 /// predecessor if the latter has a single successor.
310 static bool mergeBlocksIntoPredecessors(VPlan &Plan);
311
312 /// Wrap predicated VPReplicateRecipes with a mask operand in an if-then
313 /// region block and remove the mask operand. Optimize the created regions by
314 /// iteratively sinking scalar operands into the region, followed by merging
315 /// regions until no improvements are remaining.
316 static void createAndOptimizeReplicateRegions(VPlan &Plan);
317
318 /// Materialize the abstract header mask of the loop region into concrete
319 /// recipes: an active-lane-mask if \p UseActiveLaneMask (with a PHI if \p
320 /// UseActiveLaneMaskForControlFlow), else (WideCanonicalIV icmp ule BTC).
321 static void materializeHeaderMask(VPlan &Plan, bool UseActiveLaneMask,
322 bool UseActiveLaneMaskForControlFlow);
323
324 /// Insert truncates and extends for any truncated recipe. Redundant casts
325 /// will be folded later.
326 static void
327 truncateToMinimalBitwidths(VPlan &Plan,
328 const MapVector<Instruction *, uint64_t> &MinBWs);
329
330 /// Check \p Plan's live-ins and replace them with constants, if they can be
331 /// simplified via SCEV.
332 static void simplifyLiveInsWithSCEV(VPlan &Plan,
333 PredicatedScalarEvolution &PSE);
334
335 /// Replace symbolic strides from \p StridesMap in \p Plan with constants when
336 /// possible.
337 static void replaceSymbolicStrides(VPlan &Plan,
338 PredicatedScalarEvolution &PSE,
339 const SymbolicStrideMap &StridesMap,
340 const VPDominatorTree &VPDT);
341
342 /// Drop poison flags from recipes that may generate a poison value that is
343 /// used after vectorization, even when their operands are not poison. Those
344 /// recipes meet the following conditions:
345 /// * Contribute to the address computation of a recipe generating a widen
346 /// memory load/store (VPWidenMemoryInstructionRecipe or
347 /// VPInterleaveRecipe).
348 /// * Such a widen memory load/store is masked, but not with the header mask.
349 static void dropPoisonGeneratingRecipes(VPlan &Plan);
350
351 /// Add a VPCurrentIterationPHIRecipe and related recipes to \p Plan and
352 /// replaces all uses of the canonical IV except for the canonical IV
353 /// increment with a VPCurrentIterationPHIRecipe. The canonical IV is only
354 /// used to control the loop after this transformation.
355 static void
356 addExplicitVectorLength(VPlan &Plan,
357 const std::optional<unsigned> &MaxEVLSafeElements);
358
359 /// Optimize recipes which use an EVL-based header mask to VP intrinsics, for
360 /// example:
361 ///
362 /// %mask = icmp ult step-vector, EVL
363 /// %load = load %ptr, %mask
364 /// -->
365 /// %load = vp.load %ptr, EVL
366 static void optimizeEVLMasks(VPlan &Plan);
367
368 // For each Interleave Group in \p InterleaveGroups replace the Recipes
369 // widening its memory instructions with a single VPInterleaveRecipe at its
370 // insertion point.
371 static void createInterleaveGroups(
372 VPlan &Plan,
373 const SmallPtrSetImpl<const InterleaveGroup<Instruction> *>
374 &InterleaveGroups,
375 const bool &EpilogueAllowed);
376
377 /// Transform widen memory recipes into strided access recipes when legal
378 /// and profitable. Clamps \p Range to maintain consistency with widen
379 /// decisions of \p Plan, and uses \p Ctx to evaluate the cost.
380 static void convertToStridedAccesses(VPlan &Plan,
381 PredicatedScalarEvolution &PSE, Loop &L,
382 VPCostContext &Ctx, VFRange &Range);
383
384 /// Remove dead recipes from \p Plan.
385 static void removeDeadRecipes(VPlan &Plan);
386
387 /// Check if all loads in the loop are dereferenceable. Iterates over the
388 /// loop body blocks reachable from \p HeaderVPBB. Returns false if any
389 /// non-dereferenceable load is found.
390 static bool areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB,
391 Loop *TheLoop,
392 PredicatedScalarEvolution &PSE,
393 DominatorTree &DT,
394 AssumptionCache *AC);
395
396 /// If a single exit has multiple conditions combined together, split them
397 /// and create new exiting blocks. Currently limited to a single exit in the
398 /// latch block.
399 static bool splitCombinedExits(VPlan &Plan, PredicatedScalarEvolution &PSE,
400 Loop *TheLoop);
401
402 /// Update \p Plan to account for uncountable early exits by introducing
403 /// appropriate branching logic in the latch that handles early exits and the
404 /// latch exit condition. Multiple exits are handled with a dispatch block
405 /// that determines which exit to take based on lane-by-lane semantics.
406 LLVM_ABI_FOR_TEST static bool
407 handleUncountableEarlyExits(VPlan &Plan, OptimizationRemarkEmitter *ORE,
408 Loop *TheLoop, PredicatedScalarEvolution &PSE,
409 DominatorTree &DT, AssumptionCache *AC,
411
412 /// Disconnect countable early exits from the loop.
413 LLVM_ABI_FOR_TEST static void handleCountableEarlyExits(VPlan &Plan);
414
415 /// Replaces the exit condition from
416 /// (branch-on-cond eq CanonicalIVInc, VectorTripCount)
417 /// to
418 /// (branch-on-cond eq AVLNext, 0)
419 static void convertEVLExitCond(VPlan &Plan);
420
421 /// Replace loop regions with explicit CFG.
422 static void dissolveLoopRegions(VPlan &Plan);
423
424 /// Expand BranchOnTwoConds instructions into explicit CFG with
425 /// BranchOnCond instructions. Should be called after dissolveLoopRegions.
426 static void expandBranchOnTwoConds(VPlan &Plan);
427
428 /// Transform loops with variable-length stepping after region
429 /// dissolution.
430 ///
431 /// Once loop regions are replaced with explicit CFG, loops can step with
432 /// variable vector lengths instead of fixed lengths. This transformation:
433 /// * Makes CurrentIteration-Phi concrete.
434 // * Removes CanonicalIV and increment.
435 static void convertToVariableLengthStep(VPlan &Plan);
436
437 /// Lower abstract recipes to concrete ones, that can be codegen'd.
438 static void convertToConcreteRecipes(VPlan &Plan);
439
440 /// This function converts initial recipes to the abstract recipes and clamps
441 /// \p Range based on cost model for following optimizations and cost
442 /// estimations. The converted abstract recipes will lower to concrete
443 /// recipes before codegen.
444 static void convertToAbstractRecipes(VPlan &Plan, VPCostContext &Ctx,
445 VFRange &Range);
446
447 /// Perform instcombine-like simplifications on recipes in \p Plan.
448 static void combineRecipes(VPlan &Plan);
449
450 /// Cancel out redundant reverses in \p Plan, e.g. reverse(reverse(x)) -> x.
451 static void simplifyReverses(VPlan &Plan);
452
453 /// Remove BranchOnCond recipes with true or false conditions together with
454 /// removing dead edges to their successors. If \p OnlyLatches is true, only
455 /// process loop latches. Returns true if incoming values from any phi-like
456 /// recipe have been removed.
457 static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches = false);
458
459 /// Perform common-subexpression-elimination on \p Plan.
460 static void cse(VPlan &Plan);
461
462 /// If there's a single exit block, optimize its phi recipes that use exiting
463 /// IV values by feeding them precomputed end values instead, possibly taken
464 /// one step backwards.
465 static void optimizeInductionLiveOutUsers(VPlan &Plan,
466 PredicatedScalarEvolution &PSE,
467 const Loop *L);
468
469 /// Add explicit broadcasts for live-ins and VPValues defined in \p Plan's entry block if they are used as vectors.
470 static void materializeBroadcasts(VPlan &Plan);
471
472 /// Hoist predicated loads from the same address to the loop entry block, if
473 /// they are guaranteed to execute on both paths (i.e., in replicate regions
474 /// with complementary masks P and NOT P).
475 static void hoistPredicatedLoads(VPlan &Plan, PredicatedScalarEvolution &PSE,
476 const Loop *L);
477
478 /// Sink predicated stores to the same address with complementary predicates
479 /// (P and NOT P) to an unconditional store with select recipes for the
480 /// stored values. This eliminates branching overhead when all paths
481 /// unconditionally store to the same location.
482 static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE,
483 const Loop *L);
484
485 /// Widens memory operations by a factor of UF based on a target hook.
486 /// This allows targets to use wider memory operations when profitable.
487 static void widenMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF,
488 const TargetTransformInfo &TTI);
489
490 // Materialize vector trip counts for constants early if it can simply be
491 // computed as (Original TC / VF * UF) * VF * UF.
492 static void
493 materializeConstantVectorTripCount(VPlan &Plan, ElementCount BestVF,
494 unsigned BestUF,
495 PredicatedScalarEvolution &PSE);
496
497 /// Materialize vector trip count computations to a set of VPInstructions.
498 /// \p Step is used as the step value for the trip count computation.
499 /// \p MaxRuntimeStep is the maximum possible runtime value of Step, used to
500 /// prove the trip count is divisible by the step for scalable VFs.
501 static void materializeVectorTripCount(
502 VPlan &Plan, VPBasicBlock *VectorPHVPBB, bool TailByMasking,
503 bool RequiresScalarEpilogue, VPValue *Step,
504 std::optional<uint64_t> MaxRuntimeStep = std::nullopt);
505
506 /// Materialize the backedge-taken count to be computed explicitly using
507 /// VPInstructions.
508 static void materializeBackedgeTakenCount(VPlan &Plan,
509 VPBasicBlock *VectorPH);
510
511 /// Add explicit Build[Struct]Vector recipes to Pack multiple scalar values
512 /// into vectors and Unpack recipes to extract scalars from vectors as
513 /// needed.
514 static void materializePacksAndUnpacks(VPlan &Plan);
515
516 /// Materialize UF, VF and VFxUF to be computed explicitly using
517 /// VPInstructions.
518 static void materializeFactors(VPlan &Plan, VPBasicBlock *VectorPH,
519 ElementCount VF);
520
521 /// Attaches the alias-mask to the existing header-mask.
522 static void attachAliasMaskToHeaderMask(VPlan &Plan);
523
524 /// Materializes within the \p AliasCheckVPBB block. Updates the header mask
525 /// of the loop to use the alias mask. Returns the clamped VF.
526 static VPValue *materializeAliasMask(VPlan &Plan,
527 VPBasicBlock *AliasCheckVPBB,
528 ArrayRef<PointerDiffInfo> DiffChecks);
529
530 /// Materializes the alias mask within a check block before the loop. The
531 /// vector loop will only be entered if the clamped VF from the alias mask
532 /// is not scalar.
534 VPlan &Plan, ArrayRef<PointerDiffInfo> DiffChecks, bool HasBranchWeights);
535
536 /// Expand VPExpandSCEVRecipes in \p Plan's entry block to VPInstructions.
537 /// Recipes wrapping a SCEVAddRecExpr are kept for later IR-level expansion.
538 static void expandSCEVsToVPInstructions(VPlan &Plan, ScalarEvolution &SE);
539
540 /// Expand remaining VPExpandSCEVRecipes in \p Plan's entry block using
541 /// SCEVExpander. Each VPExpandSCEVRecipe is replaced with a live-in wrapping
542 /// the expanded IR value. A mapping from SCEV expressions to their expanded
543 /// IR value is returned.
544 static DenseMap<const SCEV *, Value *> expandSCEVs(VPlan &Plan,
545 ScalarEvolution &SE);
546
547 /// Try to find a single VF among \p Plan's VFs for which all interleave
548 /// groups (with known minimum VF elements) can be replaced by wide loads and
549 /// stores processing VF elements, if all transformed interleave groups access
550 /// the full vector width (checked via the maximum vector register width). If
551 /// the transformation can be applied, the original \p Plan will be split in
552 /// 2:
553 /// 1. The original Plan with the single VF containing the optimized recipes
554 /// using wide loads instead of interleave groups.
555 /// 2. A new clone which contains all VFs of Plan except the optimized VF.
556 ///
557 /// This effectively is a very simple form of loop-aware SLP, where we use
558 /// interleave groups to identify candidates.
559 static std::unique_ptr<VPlan>
560 narrowInterleaveGroups(VPlan &Plan, const TargetTransformInfo &TTI);
561
562 /// Adapts the vector loop region for tail folding by introducing a header
563 /// mask and conditionally executing the content of the region:
564 ///
565 /// Vector loop region before:
566 /// +-------------------------------------------+
567 /// |%iv = ... |
568 /// |... |
569 /// |%iv.next = add %iv, vfxuf |
570 /// |branch-on-count %iv.next, vector-trip-count|
571 /// +-------------------------------------------+
572 ///
573 /// Vector loop region after:
574 /// +-------------------------------------------+
575 /// |%iv = ... |
576 /// |%wide.iv = widen-canonical-iv ... |
577 /// |%header-mask = icmp ule %wide.iv, BTC |
578 /// |branch-on-cond %header-mask |---+
579 /// +-------------------------------------------+ |
580 /// | |
581 /// v |
582 /// +-------------------------------------------+ |
583 /// | ... | |
584 /// +-------------------------------------------+ |
585 /// | |
586 /// v |
587 /// +-------------------------------------------+ |
588 /// |<phis> = phi [..., ...], [poison, header] |
589 /// |%iv.next = add %iv, vfxuf |<--+
590 /// |branch-on-count %iv.next, vector-trip-count|
591 /// +-------------------------------------------+
592 ///
593 /// Any VPInstruction::ExtractLastLanes are also updated to extract from the
594 /// last active lane of the header mask.
595 static void foldTailByMasking(VPlan &Plan);
596
597 /// Predicate and linearize the control-flow in the only loop region of
598 /// \p Plan.
599 static void introduceMasksAndLinearize(VPlan &Plan);
600
601 /// Replace a VPWidenCanonicalIVRecipe if it is present in \p Plan, with a
602 /// VPWidenIntOrFpInductionRecipe, provided it would not cause additional
603 /// spills for \p VF at unroll factor \p UF.
604 static void
605 replaceWideCanonicalIVWithWideIV(VPlan &Plan, ScalarEvolution &SE,
606 const TargetTransformInfo &TTI,
608 ElementCount VF, unsigned UF);
609
610 /// Add branch weight metadata, if the \p Plan's middle block is terminated by
611 /// a BranchOnCond recipe.
612 static void
613 addBranchWeightToMiddleTerminator(VPlan &Plan, ElementCount VF,
614 std::optional<unsigned> VScaleForTuning);
615
616 /// Adjust first-order recurrence users in the middle block: create
617 /// penultimate element extracts for LCSSA phi users, and handle penultimate
618 /// extracts of the last active lane edge.
619 static void adjustFirstOrderRecurrenceMiddleUsers(VPlan &Plan,
620 VFRange &Range);
621
622 /// Optimize FindLast reductions selecting IVs (or expressions of IVs) by
623 /// converting them to FindIV reductions, if their IV range excludes a
624 /// suitable sentinel value. For expressions of IVs, the expression is sunk
625 /// to the middle block. The decision is based on SCEV expressions for \p L,
626 /// so this must run before any transform that changes the plan's iteration
627 /// space relative to \p L.
628 static void optimizeFindIVReductions(VPlan &Plan,
629 PredicatedScalarEvolution &PSE, Loop &L);
630
631 /// Detect and create partial reduction recipes for scaled or unordered
632 /// reductions in \p Plan. Must be called after recipe construction. If
633 /// partial reductions are only valid for a subset of VFs in Range, Range.End
634 /// is updated.
635 static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx,
636 VFRange &Range);
637
638 /// Convert load/store VPInstructions in \p Plan into widened or replicate
639 /// recipes. Non load/store input instructions are left unchanged.
640 static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
641 VPRecipeBuilder &RecipeBuilder,
642 VPCostContext &CostCtx);
643
644 /// Make VPlan-based scalarization decision prior to delegating to the ones
645 /// made by the legacy CM. Only transforms "usesFirstLaneOnly` def-use chains
646 /// enabled by prior widening of consecutive memory operations and calls for
647 /// now.
648 static void makeScalarizationDecisions(VPlan &Plan, VFRange &Range);
649
650 /// Convert call VPInstructions in \p Plan into widened call, vector
651 /// intrinsic or replicate recipes based on a cost comparison via \p CostCtx.
652 /// Returns true if any call was widened.
653 static bool makeCallWideningDecisions(VPlan &Plan, VFRange &Range,
654 VPRecipeBuilder &RecipeBuilder,
655 VPCostContext &CostCtx);
656
657 /// Replace truncates of a wide induction, or of that induction's increment,
658 /// by a VPWidenIntOrFpInductionRecipe producing the truncated type directly.
659 /// The canonical induction is narrowed even when the target reports the
660 /// truncate as free. If narrowing is only profitable for a subset of VFs in
661 /// \p Range, Range.End is updated.
662 static void narrowInductionTruncates(VPlan &Plan, VFRange &Range,
663 const TargetTransformInfo &TTI,
664 PredicatedScalarEvolution &PSE);
665};
666
667} // namespace llvm
668
669#endif // LLVM_TRANSFORMS_VECTORIZE_VPLANTRANSFORMS_H
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
#define LLVM_ABI_FOR_TEST
Definition Compiler.h:220
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static constexpr uint32_t MinItersBypassWeights[]
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
const SmallVectorImpl< MachineOperand > & Cond
This file defines the scope_exit class, which executes user-defined cleanup logic at scope exit.
This pass exposes codegen information to IR-level passes.
This file declares the class VPlanVerifier, which contains utility functions to check the consistency...
This file contains the declarations of the Vectorization Plan base classes:
static const char PassName[]
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
Analysis providing branch probability information.
A struct for saving information about induction variables.
This class emits a version of the loop where run-time checks ensure that may-alias pointers can't ove...
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
The optimization diagnostic interface.
Pass interface - Implemented by all 'passes'.
Definition Pass.h:99
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
LLVM_ABI bool match(StringRef String, SmallVectorImpl< StringRef > *Matches=nullptr, std::string *Error=nullptr) const
matches - Match the regex against a given String.
Definition Regex.cpp:84
The main scalar evolution driver.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Provides information about what library functions are available for the current target.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
TargetCostKind
The kind of cost model.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
BasicBlock * getIRBasicBlock() const
Definition VPlan.h:4609
Helper class to create VPRecipies from IR instructions.
void print(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print this VPRegionBlock to O (recursively), prefixing all lines with Indent.
Definition VPlan.cpp:811
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4844
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1042
VPIRBasicBlock * getScalarHeader() const
Return the VPIRBasicBlock wrapping the header of the scalar loop.
Definition VPlan.h:5002
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI_FOR_TEST cl::opt< bool > VerifyEachVPlan
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintAfterAll
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:83
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
LLVM_ABI_FOR_TEST cl::list< std::string > VPlanPrintAfterPasses
TargetTransformInfo TTI
LLVM_ABI_FOR_TEST cl::list< std::string > VPlanPrintBeforePasses
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintBeforeAll
LLVM_ABI_FOR_TEST bool verifyVPlanIsValid(const VPlan &Plan)
Verify invariants for general VPlans.
LLVM_ABI_FOR_TEST cl::opt< bool > VPlanPrintVectorRegionScope
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
static void simplifyLiveInsWithSCEV(VPlan &Plan, PredicatedScalarEvolution &PSE)
Check Plan's live-ins and replace them with constants, if they can be simplified via SCEV.
static VPValue * materializeAliasMask(VPlan &Plan, VPBasicBlock *AliasCheckVPBB, ArrayRef< PointerDiffInfo > DiffChecks)
Materializes within the AliasCheckVPBB block.
static void addMinimumVectorEpilogueIterationCheck(VPlan &Plan, VPValue *MainVectorTripCount, bool RequiresScalarEpilogue, ElementCount EpilogueVF, unsigned MainLoopStep, unsigned EpilogueLoopStep, ScalarEvolution &SE)
Add a check to Plan to see if the epilogue vector loop should be executed.
static bool makeCallWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert call VPInstructions in Plan into widened call, vector intrinsic or replicate recipes based on...
static decltype(auto) runPass(StringRef PassName, PassTy &&Pass, VPlan &Plan, ArgsTy &&...Args)
Helper to run a VPlan pass Pass on VPlan, forwarding extra arguments to the pass.
static void expandSCEVsToVPInstructions(VPlan &Plan, ScalarEvolution &SE)
Expand VPExpandSCEVRecipes in Plan's entry block to VPInstructions.
static void materializeBroadcasts(VPlan &Plan)
Add explicit broadcasts for live-ins and VPValues defined in Plan's entry block if they are used as v...
static void materializePacksAndUnpacks(VPlan &Plan)
Add explicit Build[Struct]Vector recipes to Pack multiple scalar values into vectors and Unpack recip...
static void createInterleaveGroups(VPlan &Plan, const SmallPtrSetImpl< const InterleaveGroup< Instruction > * > &InterleaveGroups, const bool &EpilogueAllowed)
static LLVM_ABI_FOR_TEST bool handleUncountableEarlyExits(VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style)
Update Plan to account for uncountable early exits by introducing appropriate branching logic in the ...
static bool simplifyKnownEVL(VPlan &Plan, ElementCount VF, PredicatedScalarEvolution &PSE)
Try to simplify VPInstruction::ExplicitVectorLength recipes when the AVL is known to be <= VF,...
static void introduceMasksAndLinearize(VPlan &Plan)
Predicate and linearize the control-flow in the only loop region of Plan.
static void materializeFactors(VPlan &Plan, VPBasicBlock *VectorPH, ElementCount VF)
Materialize UF, VF and VFxUF to be computed explicitly using VPInstructions.
static void foldTailByMasking(VPlan &Plan)
Adapts the vector loop region for tail folding by introducing a header mask and conditionally executi...
static void materializeBackedgeTakenCount(VPlan &Plan, VPBasicBlock *VectorPH)
Materialize the backedge-taken count to be computed explicitly using VPInstructions.
static LLVM_ABI_FOR_TEST bool tryToConvertVPInstructionsToVPRecipes(VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE, Loop *OuterLoop)
Replaces the VPInstructions in Plan with corresponding widen recipes.
static bool handleMultiUseReductions(VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop)
Try to legalize reductions with multiple in-loop uses.
static void createAndOptimizeReplicateRegions(VPlan &Plan)
Wrap predicated VPReplicateRecipes with a mask operand in an if-then region block and remove the mask...
static void convertToVariableLengthStep(VPlan &Plan)
Transform loops with variable-length stepping after region dissolution.
static void materializeHeaderMask(VPlan &Plan, bool UseActiveLaneMask, bool UseActiveLaneMaskForControlFlow)
Materialize the abstract header mask of the loop region into concrete recipes: an active-lane-mask if...
static void recordExecutionFrequencies(VPlan &Plan)
Add execution frequencies to each recipe in the loop body of Plan.
static void addBranchWeightToMiddleTerminator(VPlan &Plan, ElementCount VF, std::optional< unsigned > VScaleForTuning)
Add branch weight metadata, if the Plan's middle block is terminated by a BranchOnCond recipe.
static std::unique_ptr< VPlan > narrowInterleaveGroups(VPlan &Plan, const TargetTransformInfo &TTI)
Try to find a single VF among Plan's VFs for which all interleave groups (with known minimum VF eleme...
static bool handleFindLastReductions(VPlan &Plan)
Check if Plan contains any FindLast reductions.
static void createInLoopReductionRecipes(VPlan &Plan, ElementCount MinVF)
Create VPReductionRecipes for in-loop reductions.
static void materializeAliasMaskCheckBlock(VPlan &Plan, ArrayRef< PointerDiffInfo > DiffChecks, bool HasBranchWeights)
Materializes the alias mask within a check block before the loop.
static void modelGeneratedMainLoopBlocks(VPlan &EpiPlan, VPlan &MainPlan, VPIRBasicBlock *EnteredFrom)
Model the blocks the executed MainPlan generated for the main vector loop in EpiPlan during epilogue ...
static void unrollByUF(VPlan &Plan, unsigned UF)
Explicitly unroll Plan by UF.
static DenseMap< const SCEV *, Value * > expandSCEVs(VPlan &Plan, ScalarEvolution &SE)
Expand remaining VPExpandSCEVRecipes in Plan's entry block using SCEVExpander.
static void convertToConcreteRecipes(VPlan &Plan)
Lower abstract recipes to concrete ones, that can be codegen'd.
static LLVM_ABI_FOR_TEST void createLoopRegions(VPlan &Plan, DebugLoc DL)
Replace loops in Plan's flat CFG with VPRegionBlocks, turning Plan's flat CFG into a hierarchical CFG...
static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert load/store VPInstructions in Plan into widened or replicate recipes.
static LLVM_ABI_FOR_TEST void addMiddleCheck(VPlan &Plan)
If a check is needed to guard executing the scalar epilogue loop, it will be added to the middle bloc...
static void narrowInductionTruncates(VPlan &Plan, VFRange &Range, const TargetTransformInfo &TTI, PredicatedScalarEvolution &PSE)
Replace truncates of a wide induction, or of that induction's increment, by a VPWidenIntOrFpInduction...
static void expandBranchOnTwoConds(VPlan &Plan)
Expand BranchOnTwoConds instructions into explicit CFG with BranchOnCond instructions.
static void materializeVectorTripCount(VPlan &Plan, VPBasicBlock *VectorPHVPBB, bool TailByMasking, bool RequiresScalarEpilogue, VPValue *Step, std::optional< uint64_t > MaxRuntimeStep=std::nullopt)
Materialize vector trip count computations to a set of VPInstructions.
static void hoistPredicatedLoads(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Hoist predicated loads from the same address to the loop entry block, if they are guaranteed to execu...
static bool mergeBlocksIntoPredecessors(VPlan &Plan)
Remove redundant VPBasicBlocks by merging them into their single predecessor if the latter has a sing...
static void attachAliasMaskToHeaderMask(VPlan &Plan)
Attaches the alias-mask to the existing header-mask.
static void optimizeFindIVReductions(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L)
Optimize FindLast reductions selecting IVs (or expressions of IVs) by converting them to FindIV reduc...
static void convertToAbstractRecipes(VPlan &Plan, VPCostContext &Ctx, VFRange &Range)
This function converts initial recipes to the abstract recipes and clamps Range based on cost model f...
static void materializeConstantVectorTripCount(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
static void makeScalarizationDecisions(VPlan &Plan, VFRange &Range)
Make VPlan-based scalarization decision prior to delegating to the ones made by the legacy CM.
static bool areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC)
Check if all loads in the loop are dereferenceable.
static void replaceWideCanonicalIVWithWideIV(VPlan &Plan, ScalarEvolution &SE, const TargetTransformInfo &TTI, TargetTransformInfo::TargetCostKind CostKind, ElementCount VF, unsigned UF)
Replace a VPWidenCanonicalIVRecipe if it is present in Plan, with a VPWidenIntOrFpInductionRecipe,...
static LLVM_ABI_FOR_TEST std::unique_ptr< VPlan > buildVPlan0(Loop *TheLoop, LoopInfo &LI, Type *InductionTy, PredicatedScalarEvolution &PSE, LoopVersioning *LVer=nullptr, function_ref< const BranchProbabilityInfo &()> GetBPI=nullptr)
Create a base VPlan0, serving as the common starting point for all later candidates.
static void optimizeInductionLiveOutUsers(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
If there's a single exit block, optimize its phi recipes that use exiting IV values by feeding them p...
static void addExplicitVectorLength(VPlan &Plan, const std::optional< unsigned > &MaxEVLSafeElements)
Add a VPCurrentIterationPHIRecipe and related recipes to Plan and replaces all uses of the canonical ...
static void simplifyReverses(VPlan &Plan)
Cancel out redundant reverses in Plan, e.g. reverse(reverse(x)) -> x.
static void adjustFirstOrderRecurrenceMiddleUsers(VPlan &Plan, VFRange &Range)
Adjust first-order recurrence users in the middle block: create penultimate element extracts for LCSS...
static void optimizeEVLMasks(VPlan &Plan)
Optimize recipes which use an EVL-based header mask to VP intrinsics, for example:
static bool handleMaxMinNumReductions(VPlan &Plan)
Check if Plan contains any FMaxNum or FMinNum reductions.
static void removeDeadRecipes(VPlan &Plan)
Remove dead recipes from Plan.
static void attachCheckBlock(VPlan &Plan, Value *Cond, BasicBlock *CheckBlock, bool AddBranchWeights)
static LLVM_ABI_FOR_TEST void handleCountableEarlyExits(VPlan &Plan)
Disconnect countable early exits from the loop.
static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Sink predicated stores to the same address with complementary predicates (P and NOT P) to an uncondit...
static bool finalizeSCEVPredicates(VPlan &Plan, PredicatedScalarEvolution &PSE, bool OptForSize, unsigned SCEVCheckThreshold, OptimizationRemarkEmitter *ORE, Loop *TheLoop)
Finalize SCEV predicates by adding induction predicates from Plan to PSE and checking constraints.
static void replicateByVF(VPlan &Plan, ElementCount VF)
Replace replicating VPReplicateRecipe, VPScalarIVStepsRecipe and VPInstruction in Plan with VF single...
static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches=false)
Remove BranchOnCond recipes with true or false conditions together with removing dead edges to their ...
static void convertToStridedAccesses(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L, VPCostContext &Ctx, VFRange &Range)
Transform widen memory recipes into strided access recipes when legal and profitable.
static void addIterationCountCheckBlock(VPlan &Plan, ElementCount VF, unsigned UF, bool RequiresScalarEpilogue, Loop *OrigLoop, const uint32_t *MinItersBypassWeights, DebugLoc DL, PredicatedScalarEvolution &PSE)
Add a new check block before the vector preheader to Plan to check if the main vector loop should be ...
static void clearReductionWrapFlags(VPlan &Plan)
Clear NSW/NUW flags from reduction instructions if necessary.
static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx, VFRange &Range)
Detect and create partial reduction recipes for scaled or unordered reductions in Plan.
static void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF, ElementCount MinProfitableTripCount, bool RequiresScalarEpilogue, bool TailFolded, Loop *OrigLoop, const uint32_t *MinItersBypassWeights, DebugLoc DL, PredicatedScalarEvolution &PSE, VPBasicBlock *CheckBlock)
static void cse(VPlan &Plan)
Perform common-subexpression-elimination on Plan.
static void replaceSymbolicStrides(VPlan &Plan, PredicatedScalarEvolution &PSE, const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT)
Replace symbolic strides from StridesMap in Plan with constants when possible.
static void attachVPCheckBlock(VPlan &Plan, VPValue *Cond, VPBasicBlock *CheckBlock, bool AddBranchWeights)
Connect CheckBlock to Plan, branching on Cond.
static LLVM_ABI_FOR_TEST void optimize(VPlan &Plan)
Apply VPlan-to-VPlan optimizations to Plan, including induction recipe optimizations,...
static void dissolveLoopRegions(VPlan &Plan)
Replace loop regions with explicit CFG.
static void truncateToMinimalBitwidths(VPlan &Plan, const MapVector< Instruction *, uint64_t > &MinBWs)
Insert truncates and extends for any truncated recipe.
static LLVM_ABI_FOR_TEST bool createHeaderPhiRecipes(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &OrigLoop, OptimizationRemarkEmitter *ORE, const VPDominatorTree &VPDT, const MapVector< PHINode *, InductionDescriptor > &Inductions, const MapVector< PHINode *, RecurrenceDescriptor > &Reductions, const SmallPtrSetImpl< const PHINode * > &FixedOrderRecurrences, const SmallPtrSetImpl< PHINode * > &InLoopReductions, bool AllowReordering)
Replace VPPhi recipes in Plan's header with corresponding VPHeaderPHIRecipe subclasses for inductions...
static void dropPoisonGeneratingRecipes(VPlan &Plan)
Drop poison flags from recipes that may generate a poison value that is used after vectorization,...
static void optimizeForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Optimize Plan based on BestVF and BestUF.
static void widenMemoryAccessesByUF(VPlan &Plan, ElementCount VF, unsigned UF, const TargetTransformInfo &TTI)
Widens memory operations by a factor of UF based on a target hook.
static void convertEVLExitCond(VPlan &Plan)
Replaces the exit condition from (branch-on-cond eq CanonicalIVInc, VectorTripCount) to (branch-on-co...
static bool splitCombinedExits(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop *TheLoop)
If a single exit has multiple conditions combined together, split them and create new exiting blocks.
static void attachMemoryChecks(VPlan &Plan, ArrayRef< RuntimePointerCheck > Checks, ScalarEvolution &SE, DebugLoc DL, bool AddBranchWeights)
Generate Checks as recipes and attach the check block to Plan.
static void combineRecipes(VPlan &Plan)
Perform instcombine-like simplifications on recipes in Plan.