LLVM 24.0.0git
VPlan.h
Go to the documentation of this file.
1//===- VPlan.h - Represent A Vectorizer Plan --------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file contains the declarations of the Vectorization Plan base classes:
11/// 1. VPBasicBlock and VPRegionBlock that inherit from a common pure virtual
12/// VPBlockBase, together implementing a Hierarchical CFG;
13/// 2. Pure virtual VPRecipeBase serving as the base class for recipes contained
14/// within VPBasicBlocks;
15/// 3. Pure virtual VPSingleDefRecipe serving as a base class for recipes that
16/// also inherit from VPValue.
17/// 4. VPInstruction, a concrete Recipe and VPUser modeling a single planned
18/// instruction;
19/// 5. The VPlan class holding a candidate for vectorization;
20/// These are documented in docs/VectorizationPlan.rst.
21//
22//===----------------------------------------------------------------------===//
23
24#ifndef LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
25#define LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
26
27#include "VPlanValue.h"
28#include "llvm/ADT/Bitfields.h"
29#include "llvm/ADT/MapVector.h"
32#include "llvm/ADT/Twine.h"
33#include "llvm/ADT/ilist.h"
34#include "llvm/ADT/ilist_node.h"
38#include "llvm/IR/DebugLoc.h"
39#include "llvm/IR/FMF.h"
40#include "llvm/IR/Operator.h"
44#include <cassert>
45#include <cstddef>
46#include <functional>
47#include <optional>
48#include <string>
49#include <utility>
50#include <variant>
51
52namespace llvm {
53
54class BasicBlock;
55class DominatorTree;
57class IRBuilderBase;
58struct VPTransformState;
59class raw_ostream;
61class SCEV;
62class SCEVPredicate;
63class Type;
64class VPBasicBlock;
66template <typename InserterTy = VPBuilderDefaultInserter> class VPBuilderBase;
68class VPDominatorTree;
69class VPRegionBlock;
70class VPlan;
71class VPLane;
73class Value;
75
76struct VPCostContext;
77
78using VPlanPtr = std::unique_ptr<VPlan>;
79
80/// \enum UncountableExitStyle
81/// Different methods of handling early exits.
82///
84 /// No side effects to worry about, so we can process any uncountable exits
85 /// in the loop and branch either to the middle block if the trip count was
86 /// reached, or an early exitblock to determine which exit was taken.
88 /// All memory operations other than the load(s) required to determine whether
89 /// an uncountable exit occurre will be masked based on that condition. If an
90 /// uncountable exit is taken, then all lanes before the exiting lane will
91 /// complete, leaving just the final lane to execute in the scalar tail.
93};
94
95/// VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
96/// A VPBlockBase can be either a VPBasicBlock or a VPRegionBlock.
98 friend class VPBlockUtils;
99
100protected:
101 /// An enumeration for keeping track of the concrete subclass of VPBlockBase
102 /// that are actually instantiated. Values of this enumeration are kept in the
103 /// SubclassID field of the VPBlockBase objects. They are used for concrete
104 /// type identification.
105 using VPBlockTy = enum : unsigned char {
106 VPRegionBlockSC,
107 VPBasicBlockSC,
108 VPIRBasicBlockSC
109 };
110
111private:
112 /// An optional name for the block.
113 std::string Name;
114
115 /// The immediate VPRegionBlock which this VPBlockBase belongs to, or null if
116 /// it is a topmost VPBlockBase.
117 VPRegionBlock *Parent = nullptr;
118
119 /// List of predecessor blocks.
121
122 /// List of successor blocks.
124
125 /// VPlan containing the block. Set when the block is created via VPlan
126 /// helpers.
127 VPlan *Plan = nullptr;
128
129 /// Subclass identifier (for isa/dyn_cast).
130 const VPBlockTy SubclassID;
131
132 /// Unique number, used as node number in the dominator tree.
133 unsigned Number;
134
135 /// Add \p Successor as the last successor to this block.
136 void appendSuccessor(VPBlockBase *Successor) {
137 assert(Successor && "Cannot add nullptr successor!");
138 Successors.push_back(Successor);
139 }
140
141 /// Add \p Predecessor as the last predecessor to this block.
142 void appendPredecessor(VPBlockBase *Predecessor) {
143 assert(Predecessor && "Cannot add nullptr predecessor!");
144 Predecessors.push_back(Predecessor);
145 }
146
147 /// Remove \p Predecessor from the predecessors of this block.
148 void removePredecessor(VPBlockBase *Predecessor) {
149 auto Pos = find(Predecessors, Predecessor);
150 assert(Pos && "Predecessor does not exist");
151 Predecessors.erase(Pos);
152 }
153
154 /// Remove \p Successor from the successors of this block.
155 void removeSuccessor(VPBlockBase *Successor) {
156 auto Pos = find(Successors, Successor);
157 assert(Pos && "Successor does not exist");
158 Successors.erase(Pos);
159 }
160
161 /// This function replaces one predecessor with another, useful when
162 /// trying to replace an old block in the CFG with a new one.
163 void replacePredecessor(VPBlockBase *Old, VPBlockBase *New) {
164 auto I = find(Predecessors, Old);
165 assert(I != Predecessors.end());
166 assert(Old->getParent() == New->getParent() &&
167 "replaced predecessor must have the same parent");
168 *I = New;
169 }
170
171 /// This function replaces one successor with another, useful when
172 /// trying to replace an old block in the CFG with a new one.
173 void replaceSuccessor(VPBlockBase *Old, VPBlockBase *New) {
174 auto I = find(Successors, Old);
175 assert(I != Successors.end());
176 assert(Old->getParent() == New->getParent() &&
177 "replaced successor must have the same parent");
178 *I = New;
179 }
180
181public:
183
184 virtual ~VPBlockBase() = default;
185
186 const std::string &getName() const { return Name; }
187
188 void setName(const Twine &newName) { Name = newName.str(); }
189
190 /// \return an ID for the concrete type of this object.
191 /// This is used to implement the classof checks. This should not be used
192 /// for any other purpose, as the values may change as LLVM evolves.
193 unsigned getVPBlockID() const { return SubclassID; }
194
195 VPRegionBlock *getParent() { return Parent; }
196 const VPRegionBlock *getParent() const { return Parent; }
197
198 /// \return A pointer to the plan containing the current block.
199 VPlan *getPlan() { return Plan; }
200 const VPlan *getPlan() const { return Plan; }
201
202 /// Sets the pointer of the plan containing the block.
203 void setPlan(VPlan *ParentPlan) { Plan = ParentPlan; }
204
205 void setParent(VPRegionBlock *P) { Parent = P; }
206
207 /// \return the VPBasicBlock that is the entry of this VPBlockBase,
208 /// recursively, if the latter is a VPRegionBlock. Otherwise, if this
209 /// VPBlockBase is a VPBasicBlock, it is returned.
210 const VPBasicBlock *getEntryBasicBlock() const;
211 VPBasicBlock *getEntryBasicBlock();
212
213 /// \return the VPBasicBlock that is the exiting this VPBlockBase,
214 /// recursively, if the latter is a VPRegionBlock. Otherwise, if this
215 /// VPBlockBase is a VPBasicBlock, it is returned.
216 const VPBasicBlock *getExitingBasicBlock() const;
217 VPBasicBlock *getExitingBasicBlock();
218
219 const VPBlocksTy &getSuccessors() const { return Successors; }
220 VPBlocksTy &getSuccessors() { return Successors; }
221
222 /// Returns true if this block has any successors.
223 bool hasSuccessors() const { return !Successors.empty(); }
224 /// Returns true if this block has any predecessors.
225 bool hasPredecessors() const { return !Predecessors.empty(); }
226
229
230 const VPBlocksTy &getPredecessors() const { return Predecessors; }
231 VPBlocksTy &getPredecessors() { return Predecessors; }
232
233 /// \return the successor of this VPBlockBase if it has a single successor.
234 /// Otherwise return a null pointer.
236 return (Successors.size() == 1 ? *Successors.begin() : nullptr);
237 }
238
239 /// \return the predecessor of this VPBlockBase if it has a single
240 /// predecessor. Otherwise return a null pointer.
242 return (Predecessors.size() == 1 ? *Predecessors.begin() : nullptr);
243 }
244
245 size_t getNumSuccessors() const { return Successors.size(); }
246 size_t getNumPredecessors() const { return Predecessors.size(); }
247
248 /// An Enclosing Block of a block B is any block containing B, including B
249 /// itself. \return the closest enclosing block starting from "this", which
250 /// has successors. \return the root enclosing block if all enclosing blocks
251 /// have no successors.
252 VPBlockBase *getEnclosingBlockWithSuccessors();
253
254 /// \return the closest enclosing block starting from "this", which has
255 /// predecessors. \return the root enclosing block if all enclosing blocks
256 /// have no predecessors.
257 VPBlockBase *getEnclosingBlockWithPredecessors();
258
259 /// \return the successors either attached directly to this VPBlockBase or, if
260 /// this VPBlockBase is the exit block of a VPRegionBlock and has no
261 /// successors of its own, search recursively for the first enclosing
262 /// VPRegionBlock that has successors and return them. If no such
263 /// VPRegionBlock exists, return the (empty) successors of the topmost
264 /// VPBlockBase reached.
266 return getEnclosingBlockWithSuccessors()->getSuccessors();
267 }
268
269 /// \return the hierarchical predecessor of this VPBlockBase if it has a
270 /// single hierarchical predecessor. Otherwise return a null pointer.
274
275 /// Set a given VPBlockBase \p Successor as the single successor of this
276 /// VPBlockBase. This VPBlockBase is not added as predecessor of \p Successor.
277 /// This VPBlockBase must have no successors.
279 assert(Successors.empty() && "Setting one successor when others exist.");
280 assert(Successor->getParent() == getParent() &&
281 "connected blocks must have the same parent");
282 appendSuccessor(Successor);
283 }
284
285 /// Set two given VPBlockBases \p IfTrue and \p IfFalse to be the two
286 /// successors of this VPBlockBase. This VPBlockBase is not added as
287 /// predecessor of \p IfTrue or \p IfFalse. This VPBlockBase must have no
288 /// successors.
289 void setTwoSuccessors(VPBlockBase *IfTrue, VPBlockBase *IfFalse) {
290 assert(Successors.empty() && "Setting two successors when others exist.");
291 appendSuccessor(IfTrue);
292 appendSuccessor(IfFalse);
293 }
294
295 /// Set each VPBasicBlock in \p NewPreds as predecessor of this VPBlockBase.
296 /// This VPBlockBase must have no predecessors. This VPBlockBase is not added
297 /// as successor of any VPBasicBlock in \p NewPreds.
299 assert(Predecessors.empty() && "Block predecessors already set.");
300 for (auto *Pred : NewPreds)
301 appendPredecessor(Pred);
302 }
303
304 /// Set each VPBasicBlock in \p NewSuccss as successor of this VPBlockBase.
305 /// This VPBlockBase must have no successors. This VPBlockBase is not added
306 /// as predecessor of any VPBasicBlock in \p NewSuccs.
308 assert(Successors.empty() && "Block successors already set.");
309 for (auto *Succ : NewSuccs)
310 appendSuccessor(Succ);
311 }
312
313 /// Remove all the predecessor of this block.
314 void clearPredecessors() { Predecessors.clear(); }
315
316 /// Remove all the successors of this block.
317 void clearSuccessors() { Successors.clear(); }
318
319 /// Swap predecessors of the block. The block must have exactly 2
320 /// predecessors.
322 assert(Predecessors.size() == 2 && "must have 2 predecessors to swap");
323 std::swap(Predecessors[0], Predecessors[1]);
324 }
325
326 /// Swap successors of the block. The block must have exactly 2 successors.
327 // TODO: This should be part of introducing conditional branch recipes rather
328 // than being independent.
330 assert(Successors.size() == 2 && "must have 2 successors to swap");
331 std::swap(Successors[0], Successors[1]);
332 }
333
334 /// Returns the index for \p Pred in the blocks predecessors list.
335 unsigned getIndexForPredecessor(const VPBlockBase *Pred) const {
336 assert(count(Predecessors, Pred) == 1 &&
337 "must have Pred exactly once in Predecessors");
338 return std::distance(Predecessors.begin(), find(Predecessors, Pred));
339 }
340
341 /// Returns the index for \p Succ in the blocks successor list.
342 unsigned getIndexForSuccessor(const VPBlockBase *Succ) const {
343 assert(count(Successors, Succ) == 1 &&
344 "must have Succ exactly once in Successors");
345 return std::distance(Successors.begin(), find(Successors, Succ));
346 }
347
348 /// Return the unique number of the block.
349 unsigned getNumber() const { return Number; }
350
351 /// Set the unique number of the block, used for dominator tree.
352 void setNumber(unsigned N) { Number = N; }
353
354 /// The method which generates the output IR that correspond to this
355 /// VPBlockBase, thereby "executing" the VPlan.
356 virtual void execute(VPTransformState *State) = 0;
357
358 /// Return the cost of the block.
360
361 void printAsOperand(raw_ostream &OS, bool PrintType = false) const {
362 OS << getName();
363 }
364
365#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
366 /// Print plain-text dump of this VPBlockBase to \p O, prefixing all lines
367 /// with \p Indent. \p SlotTracker is used to print unnamed VPValue's using
368 /// consequtive numbers.
369 ///
370 /// Note that the numbering is applied to the whole VPlan, so printing
371 /// individual blocks is consistent with the whole VPlan printing.
372 virtual void print(raw_ostream &O, const Twine &Indent,
373 VPSlotTracker &SlotTracker) const = 0;
374
375 /// Print plain-text dump of this VPlan to \p O.
376 void print(raw_ostream &O) const;
377
378 /// Print the successors of this block to \p O, prefixing all lines with \p
379 /// Indent.
380 void printSuccessors(raw_ostream &O, const Twine &Indent) const;
381
382 /// Dump this VPBlockBase to dbgs().
383 LLVM_DUMP_METHOD void dump() const { print(dbgs()); }
384#endif
385
386 /// Clone the current block and it's recipes without updating the operands of
387 /// the cloned recipes, including all blocks in the single-entry single-exit
388 /// region for VPRegionBlocks.
389 virtual VPBlockBase *clone() = 0;
390
391protected:
392 VPBlockBase(VPBlockTy SC, const std::string &N) : Name(N), SubclassID(SC) {}
393};
394
395/// VPRecipeBase is a base class modeling a sequence of one or more output IR
396/// instructions. VPRecipeBase owns the VPValues it defines through VPDef
397/// and is responsible for deleting its defined values. Single-value
398/// recipes must inherit from VPSingleDef instead of inheriting from both
399/// VPRecipeBase and VPValue separately.
401 : public ilist_node_with_parent<VPRecipeBase, VPBasicBlock>,
402 public VPDef,
403 public VPUser {
404 friend VPBasicBlock;
405 friend class VPBlockUtils;
406
407 /// Each VPRecipe belongs to a single VPBasicBlock.
408 VPBasicBlock *Parent = nullptr;
409
410 /// The debug location for the recipe.
411 DebugLoc DL;
412
413public:
414 /// An enumeration for keeping track of the concrete subclass of VPRecipeBase
415 /// that is actually instantiated. Values of this enumeration are kept in the
416 /// SubclassID field of the VPRecipeBase objects. They are used for concrete
417 /// type identification.
418 using VPRecipeTy = enum : unsigned char {
419 VPBranchOnMaskSC,
420 VPDerivedIVSC,
421 VPExpandSCEVSC,
422 VPExpressionSC,
423 VPIRInstructionSC,
424 VPInstructionSC,
425 VPInterleaveEVLSC,
426 VPInterleaveSC,
427 VPReductionEVLSC,
428 VPReductionSC,
429 VPReplicateSC,
430 VPScalarIVStepsSC,
431 VPVectorPointerSC,
432 VPVectorEndPointerSC,
433 VPWidenCallSC,
434 VPWidenCanonicalIVSC,
435 VPWidenCastSC,
436 VPWidenGEPSC,
437 VPWidenIntrinsicSC,
438 VPWidenMemIntrinsicSC,
439 VPWidenLoadEVLSC,
440 VPWidenLoadSC,
441 VPWidenStoreEVLSC,
442 VPWidenStoreSC,
443 VPWidenSC,
444 VPBlendSC,
445 VPHistogramSC,
446 // START: Phi-like recipes. Need to be kept together.
447 VPWidenPHISC,
448 VPPredInstPHISC,
449 // START: SubclassID for recipes that inherit VPHeaderPHIRecipe.
450 // VPHeaderPHIRecipe need to be kept together.
451 VPCurrentIterationPHISC,
452 VPActiveLaneMaskPHISC,
453 VPFirstOrderRecurrencePHISC,
454 VPWidenIntOrFpInductionSC,
455 VPWidenPointerInductionSC,
456 VPReductionPHISC,
457 // END: SubclassID for recipes that inherit VPHeaderPHIRecipe
458 // END: Phi-like recipes
459 VPFirstPHISC = VPWidenPHISC,
460 VPFirstHeaderPHISC = VPCurrentIterationPHISC,
461 VPLastHeaderPHISC = VPReductionPHISC,
462 VPLastPHISC = VPReductionPHISC,
463 };
464
467 : VPDef(), VPUser(Operands), DL(DL), SubclassID(SC) {}
468
469 ~VPRecipeBase() override = default;
470
471 /// Clone the current recipe.
472 virtual VPRecipeBase *clone() = 0;
473
474 /// \return the VPBasicBlock which this VPRecipe belongs to.
475 VPBasicBlock *getParent() { return Parent; }
476 const VPBasicBlock *getParent() const { return Parent; }
477
478 /// \return the VPRegionBlock which the recipe belongs to.
479 VPRegionBlock *getRegion();
480 const VPRegionBlock *getRegion() const;
481
482 /// The method which generates the output IR instructions that correspond to
483 /// this VPRecipe, thereby "executing" the VPlan.
484 virtual void execute(VPTransformState &State) = 0;
485
486 /// Return the cost of this recipe, taking into account if the cost
487 /// computation should be skipped and the ForceTargetInstructionCost flag.
488 /// Also takes care of printing the cost for debugging.
490
491 /// Insert an unlinked recipe into a basic block immediately before
492 /// the specified recipe.
493 void insertBefore(VPRecipeBase *InsertPos);
494 /// Insert an unlinked recipe into \p BB immediately before the insertion
495 /// point \p IP;
496 void insertBefore(VPBasicBlock &BB, iplist<VPRecipeBase>::iterator IP);
497
498 /// Insert an unlinked Recipe into a basic block immediately after
499 /// the specified Recipe.
500 void insertAfter(VPRecipeBase *InsertPos);
501
502 /// Unlink this recipe from its current VPBasicBlock and insert it into
503 /// the VPBasicBlock that MovePos lives in, right after MovePos.
504 void moveAfter(VPRecipeBase *MovePos);
505
506 /// Unlink this recipe and insert into BB before I.
507 ///
508 /// \pre I is a valid iterator into BB.
509 void moveBefore(VPBasicBlock &BB, iplist<VPRecipeBase>::iterator I);
510
511 /// This method unlinks 'this' from the containing basic block, but does not
512 /// delete it.
513 void removeFromParent();
514
515 /// This method unlinks 'this' from the containing basic block and deletes it.
516 ///
517 /// \returns an iterator pointing to the element after the erased one
519
520 /// \return an ID for the concrete type of this object.
521 VPRecipeTy getVPRecipeID() const { return SubclassID; }
522
523 /// Method to support type inquiry through isa, cast, and dyn_cast.
524 static inline bool classof(const VPDef *D) {
525 // All VPDefs are also VPRecipeBases.
526 return true;
527 }
528
529 static inline bool classof(const VPUser *U) { return true; }
530
531 /// Returns true if the recipe may have side-effects.
532 bool mayHaveSideEffects() const;
533
534 /// Return true if we can safely execute this recipe unconditionally even if
535 /// it is masked originally.
536 bool isSafeToSpeculativelyExecute() const;
537
538 /// Returns true for PHI-like recipes.
539 bool isPhi() const;
540
541 /// Returns true if the recipe may read from memory.
542 bool mayReadFromMemory() const;
543
544 /// Returns true if the recipe may write to memory.
545 bool mayWriteToMemory() const;
546
547 /// Returns true if the recipe may read from or write to memory.
548 bool mayReadOrWriteMemory() const {
550 }
551
552 /// Returns the debug location of the recipe.
553 DebugLoc getDebugLoc() const { return DL; }
554
555 /// Set the recipe's debug location to \p NewDL.
556 void setDebugLoc(DebugLoc NewDL) { DL = NewDL; }
557
558#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
559 /// Dump the recipe to stderr (for debugging).
560 void dump() const;
561
562 /// Print the recipe, delegating to printRecipe().
563 void print(raw_ostream &O, const Twine &Indent,
565#endif
566
567private:
568 /// Subclass identifier (for isa/dyn_cast).
569 const VPRecipeTy SubclassID;
570
571protected:
572 /// Compute the cost of this recipe either using a recipe's specialized
573 /// implementation or using the legacy cost model and the underlying
574 /// instructions.
575 virtual InstructionCost computeCost(ElementCount VF,
576 VPCostContext &Ctx) const;
577
578#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
579 /// Each concrete VPRecipe prints itself, without printing common information,
580 /// like debug info or metadata.
581 virtual void printRecipe(raw_ostream &O, const Twine &Indent,
582 VPSlotTracker &SlotTracker) const = 0;
583#endif
584};
585
586// Helper macro to define common classof implementations for recipes.
587#define VP_CLASSOF_IMPL(VPRecipeID) \
588 static inline bool classof(const VPRecipeBase *R) { \
589 return R->getVPRecipeID() == VPRecipeID; \
590 } \
591 static inline bool classof(const VPValue *V) { \
592 auto *R = V->getDefiningRecipe(); \
593 return R && R->getVPRecipeID() == VPRecipeID; \
594 } \
595 static inline bool classof(const VPUser *U) { \
596 auto *R = dyn_cast<VPRecipeBase>(U); \
597 return R && R->getVPRecipeID() == VPRecipeID; \
598 } \
599 static inline bool classof(const VPSingleDefRecipe *R) { \
600 return R->getVPRecipeID() == VPRecipeID; \
601 }
602
603/// Compute the scalar result type for an IR \p Opcode given \p Operands.
604LLVM_ABI Type *computeScalarTypeForInstruction(unsigned Opcode,
606
607/// VPSingleDefRecipe is a base class for recipes that model a sequence of one
608/// or more output IR that define a single result VPValue. Note that
609/// VPSingleDefRecipe must inherit from VPRecipeBase before VPSingleDefValue.
611 public VPSingleDefValue {
612public:
616
619 : VPRecipeBase(SC, Operands, DL), VPSingleDefValue(this, UV) {}
620
622 Value *UV = nullptr, DebugLoc DL = DebugLoc::getUnknown())
623 : VPRecipeBase(SC, Operands, DL), VPSingleDefValue(this, UV, ResultTy) {}
624
625 static inline bool classof(const VPRecipeBase *R) {
626 switch (R->getVPRecipeID()) {
627 case VPRecipeBase::VPDerivedIVSC:
628 case VPRecipeBase::VPExpandSCEVSC:
629 case VPRecipeBase::VPExpressionSC:
630 case VPRecipeBase::VPInstructionSC:
631 case VPRecipeBase::VPReductionEVLSC:
632 case VPRecipeBase::VPReductionSC:
633 case VPRecipeBase::VPReplicateSC:
634 case VPRecipeBase::VPScalarIVStepsSC:
635 case VPRecipeBase::VPVectorPointerSC:
636 case VPRecipeBase::VPVectorEndPointerSC:
637 case VPRecipeBase::VPWidenCallSC:
638 case VPRecipeBase::VPWidenCanonicalIVSC:
639 case VPRecipeBase::VPWidenCastSC:
640 case VPRecipeBase::VPWidenGEPSC:
641 case VPRecipeBase::VPWidenIntrinsicSC:
642 case VPRecipeBase::VPWidenMemIntrinsicSC:
643 case VPRecipeBase::VPWidenSC:
644 case VPRecipeBase::VPBlendSC:
645 case VPRecipeBase::VPPredInstPHISC:
646 case VPRecipeBase::VPCurrentIterationPHISC:
647 case VPRecipeBase::VPActiveLaneMaskPHISC:
648 case VPRecipeBase::VPFirstOrderRecurrencePHISC:
649 case VPRecipeBase::VPWidenPHISC:
650 case VPRecipeBase::VPWidenIntOrFpInductionSC:
651 case VPRecipeBase::VPWidenPointerInductionSC:
652 case VPRecipeBase::VPReductionPHISC:
653 case VPRecipeBase::VPWidenLoadEVLSC:
654 case VPRecipeBase::VPWidenLoadSC:
655 return true;
656 case VPRecipeBase::VPBranchOnMaskSC:
657 case VPRecipeBase::VPInterleaveEVLSC:
658 case VPRecipeBase::VPInterleaveSC:
659 case VPRecipeBase::VPIRInstructionSC:
660 case VPRecipeBase::VPWidenStoreEVLSC:
661 case VPRecipeBase::VPWidenStoreSC:
662 case VPRecipeBase::VPHistogramSC:
663 return false;
664 }
665 llvm_unreachable("Unhandled VPRecipeID");
666 }
667
668 static inline bool classof(const VPValue *V) {
669 auto *R = V->getDefiningRecipe();
670 return R && classof(R);
671 }
672
673 static inline bool classof(const VPUser *U) {
674 auto *R = dyn_cast<VPRecipeBase>(U);
675 return R && classof(R);
676 }
677
678 VPSingleDefRecipe *clone() override = 0;
679
680 /// Returns the underlying instruction.
687
688#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
689 /// Print this VPSingleDefRecipe to dbgs() (for debugging).
690 LLVM_DUMP_METHOD void dump() const;
691#endif
692};
693
694/// Class to record and manage LLVM IR flags.
697 enum class OperationType : unsigned char {
698 Cmp,
699 FCmp,
700 OverflowingBinOp,
701 Trunc,
702 DisjointOp,
703 PossiblyExactOp,
704 GEPOp,
705 FPMathOp,
706 NonNegOp,
707 ReductionOp,
708 Other
709 };
710
711public:
712 struct WrapFlagsTy {
713 char HasNUW : 1;
714 char HasNSW : 1;
715
719 return {static_cast<bool>(HasNUW), false};
720 }
721 };
722
724 char HasNUW : 1;
725 char HasNSW : 1;
726
728 };
729
734
736 char NonNeg : 1;
737 NonNegFlagsTy(bool IsNonNeg) : NonNeg(IsNonNeg) {}
738 };
739
740private:
741 struct ExactFlagsTy {
742 char IsExact : 1;
743 ExactFlagsTy(bool Exact) : IsExact(Exact) {}
744 };
745 struct FastMathFlagsTy {
746 char AllowReassoc : 1;
747 char NoNaNs : 1;
748 char NoInfs : 1;
749 char NoSignedZeros : 1;
750 char AllowReciprocal : 1;
751 char AllowContract : 1;
752 char ApproxFunc : 1;
753
754 LLVM_ABI_FOR_TEST FastMathFlagsTy(const FastMathFlags &FMF);
755 };
756 /// Holds both the predicate and fast-math flags for floating-point
757 /// comparisons.
758 struct FCmpFlagsTy {
759 uint8_t CmpPredStorage;
760 FastMathFlagsTy FMFs;
761 };
762 /// Holds reduction-specific flags: RecurKind, IsOrdered, IsInLoop, and FMFs.
763 struct ReductionFlagsTy {
764 // RecurKind has ~26 values, needs 5 bits but uses 6 bits to account for
765 // additional kinds.
766 unsigned char Kind : 6;
767 // TODO: Derive order/in-loop from plan and remove here.
768 unsigned char IsOrdered : 1;
769 unsigned char IsInLoop : 1;
770 FastMathFlagsTy FMFs;
771
772 ReductionFlagsTy(RecurKind Kind, bool IsOrdered, bool IsInLoop,
773 FastMathFlags FMFs)
774 : Kind(static_cast<unsigned char>(Kind)), IsOrdered(IsOrdered),
775 IsInLoop(IsInLoop), FMFs(FMFs) {}
776 };
777
778 OperationType OpType;
779
780 union {
785 ExactFlagsTy ExactFlags;
788 FastMathFlagsTy FMFs;
789 FCmpFlagsTy FCmpFlags;
790 ReductionFlagsTy ReductionFlags;
792 };
793
794public:
795 VPIRFlags() : OpType(OperationType::Other), AllFlags() {}
796
798 if (auto *FCmp = dyn_cast<FCmpInst>(&I)) {
799 OpType = OperationType::FCmp;
801 FCmp->getPredicate());
802 assert(getPredicate() == FCmp->getPredicate() && "predicate truncated");
803 FCmpFlags.FMFs = FCmp->getFastMathFlags();
804 } else if (auto *Op = dyn_cast<CmpInst>(&I)) {
805 OpType = OperationType::Cmp;
807 Op->getPredicate());
808 assert(getPredicate() == Op->getPredicate() && "predicate truncated");
809 } else if (auto *Op = dyn_cast<PossiblyDisjointInst>(&I)) {
810 OpType = OperationType::DisjointOp;
811 DisjointFlags.IsDisjoint = Op->isDisjoint();
812 } else if (auto *Op = dyn_cast<OverflowingBinaryOperator>(&I)) {
813 OpType = OperationType::OverflowingBinOp;
814 WrapFlags = {Op->hasNoUnsignedWrap(), Op->hasNoSignedWrap()};
815 } else if (auto *Op = dyn_cast<TruncInst>(&I)) {
816 OpType = OperationType::Trunc;
817 TruncFlags = {Op->hasNoUnsignedWrap(), Op->hasNoSignedWrap()};
818 } else if (auto *Op = dyn_cast<PossiblyExactOperator>(&I)) {
819 OpType = OperationType::PossiblyExactOp;
820 ExactFlags.IsExact = Op->isExact();
821 } else if (auto *GEP = dyn_cast<GetElementPtrInst>(&I)) {
822 OpType = OperationType::GEPOp;
823 GEPFlagsStorage = GEP->getNoWrapFlags().getRaw();
824 assert(getGEPNoWrapFlags() == GEP->getNoWrapFlags() &&
825 "wrap flags truncated");
826 } else if (auto *PNNI = dyn_cast<PossiblyNonNegInst>(&I)) {
827 OpType = OperationType::NonNegOp;
828 NonNegFlags.NonNeg = PNNI->hasNonNeg();
829 } else if (auto *Op = dyn_cast<FPMathOperator>(&I)) {
830 OpType = OperationType::FPMathOp;
831 FMFs = Op->getFastMathFlags();
832 }
833 }
834
835 VPIRFlags(CmpInst::Predicate Pred) : OpType(OperationType::Cmp), AllFlags() {
837 assert(getPredicate() == Pred && "predicate truncated");
838 }
839
841 : OpType(OperationType::FCmp), AllFlags() {
843 assert(getPredicate() == Pred && "predicate truncated");
844 FCmpFlags.FMFs = FMFs;
845 }
846
848 : OpType(OperationType::OverflowingBinOp), AllFlags() {
849 this->WrapFlags = WrapFlags;
850 }
851
853 : OpType(OperationType::Trunc), AllFlags() {
854 this->TruncFlags = TruncFlags;
855 }
856
857 VPIRFlags(FastMathFlags FMFs) : OpType(OperationType::FPMathOp), AllFlags() {
858 this->FMFs = FMFs;
859 }
860
862 : OpType(OperationType::DisjointOp), AllFlags() {
863 this->DisjointFlags = DisjointFlags;
864 }
865
867 : OpType(OperationType::NonNegOp), AllFlags() {
868 this->NonNegFlags = NonNegFlags;
869 }
870
871 VPIRFlags(ExactFlagsTy ExactFlags)
872 : OpType(OperationType::PossiblyExactOp), AllFlags() {
873 this->ExactFlags = ExactFlags;
874 }
875
877 : OpType(OperationType::GEPOp), AllFlags() {
878 GEPFlagsStorage = GEPFlags.getRaw();
879 }
880
881 VPIRFlags(RecurKind Kind, bool IsOrdered, bool IsInLoop, FastMathFlags FMFs)
882 : OpType(OperationType::ReductionOp), AllFlags() {
883 ReductionFlags = ReductionFlagsTy(Kind, IsOrdered, IsInLoop, FMFs);
884 }
885
887 OpType = Other.OpType;
888 AllFlags[0] = Other.AllFlags[0];
889 AllFlags[1] = Other.AllFlags[1];
890 }
891
892 /// Only keep flags also present in \p Other. \p Other must have the same
893 /// OpType as the current object.
894 void intersectFlags(const VPIRFlags &Other);
895
896 /// Drop all poison-generating flags.
898 // NOTE: This needs to be kept in-sync with
899 // Instruction::dropPoisonGeneratingFlags.
900 switch (OpType) {
901 case OperationType::OverflowingBinOp:
902 WrapFlags.HasNUW = false;
903 WrapFlags.HasNSW = false;
904 break;
905 case OperationType::Trunc:
906 TruncFlags.HasNUW = false;
907 TruncFlags.HasNSW = false;
908 break;
909 case OperationType::DisjointOp:
910 DisjointFlags.IsDisjoint = false;
911 break;
912 case OperationType::PossiblyExactOp:
913 ExactFlags.IsExact = false;
914 break;
915 case OperationType::GEPOp:
916 GEPFlagsStorage = 0;
917 break;
918 case OperationType::FPMathOp:
919 case OperationType::FCmp:
920 case OperationType::ReductionOp:
921 getFMFsRef().NoNaNs = false;
922 getFMFsRef().NoInfs = false;
923 break;
924 case OperationType::NonNegOp:
925 NonNegFlags.NonNeg = false;
926 break;
927 case OperationType::Cmp:
928 case OperationType::Other:
929 break;
930 }
931 }
932
933 /// Apply the IR flags to \p I.
934 void applyFlags(Instruction &I) const {
935 switch (OpType) {
936 case OperationType::OverflowingBinOp:
937 I.setHasNoUnsignedWrap(WrapFlags.HasNUW);
938 I.setHasNoSignedWrap(WrapFlags.HasNSW);
939 break;
940 case OperationType::Trunc:
941 I.setHasNoUnsignedWrap(TruncFlags.HasNUW);
942 I.setHasNoSignedWrap(TruncFlags.HasNSW);
943 break;
944 case OperationType::DisjointOp:
945 cast<PossiblyDisjointInst>(&I)->setIsDisjoint(DisjointFlags.IsDisjoint);
946 break;
947 case OperationType::PossiblyExactOp:
948 I.setIsExact(ExactFlags.IsExact);
949 break;
950 case OperationType::GEPOp:
951 cast<GetElementPtrInst>(&I)->setNoWrapFlags(
953 break;
954 case OperationType::FPMathOp:
955 case OperationType::FCmp: {
956 const FastMathFlagsTy &F = getFMFsRef();
957 I.setHasAllowReassoc(F.AllowReassoc);
958 I.setHasNoNaNs(F.NoNaNs);
959 I.setHasNoInfs(F.NoInfs);
960 I.setHasNoSignedZeros(F.NoSignedZeros);
961 I.setHasAllowReciprocal(F.AllowReciprocal);
962 I.setHasAllowContract(F.AllowContract);
963 I.setHasApproxFunc(F.ApproxFunc);
964 break;
965 }
966 case OperationType::NonNegOp:
967 I.setNonNeg(NonNegFlags.NonNeg);
968 break;
969 case OperationType::ReductionOp:
970 llvm_unreachable("reduction ops should not use applyFlags");
971 case OperationType::Cmp:
972 case OperationType::Other:
973 break;
974 }
975 }
976
978 assert((OpType == OperationType::Cmp || OpType == OperationType::FCmp) &&
979 "recipe doesn't have a compare predicate");
980 uint8_t Storage = OpType == OperationType::FCmp ? FCmpFlags.CmpPredStorage
983 }
984
986 assert((OpType == OperationType::Cmp || OpType == OperationType::FCmp) &&
987 "recipe doesn't have a compare predicate");
988 if (OpType == OperationType::FCmp)
990 else
992 assert(getPredicate() == Pred && "predicate truncated");
993 }
994
998
999 /// Returns true if the recipe has a comparison predicate.
1000 bool hasPredicate() const {
1001 return OpType == OperationType::Cmp || OpType == OperationType::FCmp;
1002 }
1003
1004 /// Returns true if the recipe has fast-math flags.
1005 bool hasFastMathFlags() const {
1006 return OpType == OperationType::FPMathOp || OpType == OperationType::FCmp ||
1007 OpType == OperationType::ReductionOp;
1008 }
1009
1011
1012 bool hasNoUnsignedWrap() const {
1013 switch (OpType) {
1014 case OperationType::OverflowingBinOp:
1015 return WrapFlags.HasNUW;
1016 case OperationType::Trunc:
1017 return TruncFlags.HasNUW;
1018 default:
1019 llvm_unreachable("recipe doesn't have a NUW flag");
1020 }
1021 }
1022
1023 bool hasNoSignedWrap() const {
1024 switch (OpType) {
1025 case OperationType::OverflowingBinOp:
1026 return WrapFlags.HasNSW;
1027 case OperationType::Trunc:
1028 return TruncFlags.HasNSW;
1029 default:
1030 llvm_unreachable("recipe doesn't have a NSW flag");
1031 }
1032 }
1033
1035 switch (OpType) {
1036 case OperationType::OverflowingBinOp:
1037 case OperationType::Trunc:
1038 return {hasNoUnsignedWrap(), hasNoSignedWrap()};
1039 default:
1040 return {};
1041 }
1042 }
1043
1045 return {hasNoUnsignedWrap(), hasNoSignedWrap()};
1046 }
1047
1048 bool isDisjoint() const {
1049 assert(OpType == OperationType::DisjointOp &&
1050 "recipe cannot have a disjoing flag");
1051 return DisjointFlags.IsDisjoint;
1052 }
1053
1055 assert(OpType == OperationType::ReductionOp &&
1056 "recipe doesn't have reduction flags");
1057 return static_cast<RecurKind>(ReductionFlags.Kind);
1058 }
1059
1060 bool isReductionOrdered() const {
1061 assert(OpType == OperationType::ReductionOp &&
1062 "recipe doesn't have reduction flags");
1063 return ReductionFlags.IsOrdered;
1064 }
1065
1066 bool isReductionInLoop() const {
1067 assert(OpType == OperationType::ReductionOp &&
1068 "recipe doesn't have reduction flags");
1069 return ReductionFlags.IsInLoop;
1070 }
1071
1072private:
1073 /// Get a reference to the fast-math flags for FPMathOp, FCmp or ReductionOp.
1074 FastMathFlagsTy &getFMFsRef() {
1075 if (OpType == OperationType::FCmp)
1076 return FCmpFlags.FMFs;
1077 if (OpType == OperationType::ReductionOp)
1078 return ReductionFlags.FMFs;
1079 return FMFs;
1080 }
1081 const FastMathFlagsTy &getFMFsRef() const {
1082 if (OpType == OperationType::FCmp)
1083 return FCmpFlags.FMFs;
1084 if (OpType == OperationType::ReductionOp)
1085 return ReductionFlags.FMFs;
1086 return FMFs;
1087 }
1088
1089public:
1090 /// Returns default flags for \p Opcode and scalar \p ResultTy for opcodes
1091 /// that support it, asserts otherwise. Opcodes not supporting default flags
1092 /// include compares and ComputeReductionResult.
1093 LLVM_ABI_FOR_TEST static VPIRFlags getDefaultFlags(unsigned Opcode,
1094 Type *ResultTy = nullptr);
1095
1096#if !defined(NDEBUG)
1097 /// Returns true if the set flags are valid for \p Opcode.
1098 LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const;
1099
1100 /// Returns true if \p Opcode with scalar result type \p ResultTy has its
1101 /// required flags set.
1102 LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode,
1103 Type *ResultTy) const;
1104#endif
1105
1106#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1107 void printFlags(raw_ostream &O) const;
1108#endif
1109};
1111
1112static_assert(sizeof(VPIRFlags) <= 3, "VPIRFlags should not grow");
1113
1114/// A pure-virtual common base class for recipes defining a single VPValue and
1115/// using IR flags.
1118 const VPIRFlags &Flags,
1120 : VPSingleDefRecipe(SC, Operands, DL), VPIRFlags(Flags) {}
1121
1123 Type *ResultTy, const VPIRFlags &Flags,
1125 : VPSingleDefRecipe(SC, Operands, ResultTy, /*UV=*/nullptr, DL),
1126 VPIRFlags(Flags) {}
1127
1128 static inline bool classof(const VPRecipeBase *R) {
1129 return R->getVPRecipeID() == VPRecipeBase::VPBlendSC ||
1130 R->getVPRecipeID() == VPRecipeBase::VPInstructionSC ||
1131 R->getVPRecipeID() == VPRecipeBase::VPWidenSC ||
1132 R->getVPRecipeID() == VPRecipeBase::VPWidenGEPSC ||
1133 R->getVPRecipeID() == VPRecipeBase::VPWidenCallSC ||
1134 R->getVPRecipeID() == VPRecipeBase::VPWidenCastSC ||
1135 R->getVPRecipeID() == VPRecipeBase::VPWidenIntrinsicSC ||
1136 R->getVPRecipeID() == VPRecipeBase::VPWidenMemIntrinsicSC ||
1137 R->getVPRecipeID() == VPRecipeBase::VPReductionSC ||
1138 R->getVPRecipeID() == VPRecipeBase::VPReductionEVLSC ||
1139 R->getVPRecipeID() == VPRecipeBase::VPReplicateSC ||
1140 R->getVPRecipeID() == VPRecipeBase::VPVectorEndPointerSC ||
1141 R->getVPRecipeID() == VPRecipeBase::VPVectorPointerSC ||
1142 R->getVPRecipeID() == VPRecipeBase::VPWidenCanonicalIVSC ||
1143 R->getVPRecipeID() == VPRecipeBase::VPDerivedIVSC;
1144 }
1145
1146 static inline bool classof(const VPUser *U) {
1147 auto *R = dyn_cast<VPRecipeBase>(U);
1148 return R && classof(R);
1149 }
1150
1151 static inline bool classof(const VPValue *V) {
1152 auto *R = V->getDefiningRecipe();
1153 return R && classof(R);
1154 }
1155
1157
1158 static inline bool classof(const VPSingleDefRecipe *R) {
1159 return classof(static_cast<const VPRecipeBase *>(R));
1160 }
1161
1162 void execute(VPTransformState &State) override = 0;
1163
1164 /// Compute the cost for this recipe for \p VF, using \p Opcode and \p Ctx.
1166 VPCostContext &Ctx) const;
1167};
1168
1169/// The frequency with which a recipe executes, relative to the entry of the
1170/// loop region. IsEstimated is set if any branch weight it was composed from
1171/// was estimated from static heuristics.
1174 const bool IsEstimated;
1175
1178 assert(Freq > BlockFrequency() && "execution frequency must be non-zero");
1179 }
1180};
1181
1182/// Helper to manage IR metadata for recipes. It filters out metadata that
1183/// cannot be propagated.
1186
1187 /// Name of the VPlan-internal metadata kind holding the execution frequency.
1188 static constexpr StringLiteral ExecutionFrequencyMDName =
1189 "vplan.execution.frequency";
1190
1191 /// Name of the VPlan-internal metadata kind holding estimated branch weights.
1192 static constexpr StringLiteral EstimatedProfileMDName =
1193 "vplan.prof.estimated";
1194
1195 /// Returns the ID of the metadata kind named \p Kind, taking the context from
1196 /// any attached node; all belong to the context of the VPlan's function.
1197 unsigned getMDKindID(StringRef Kind) const {
1198 assert(!Metadata.empty() && "no node to take the context from");
1199 return Metadata.front().second->getContext().getMDKindID(Kind);
1200 }
1201
1202 /// Returns the node attached under the VPlan-internal metadata kind named
1203 /// \p Kind, or nullptr if there is none.
1204 MDNode *getInternalMetadata(StringRef Kind) const {
1205 return Metadata.empty() ? nullptr : getMetadata(getMDKindID(Kind));
1206 }
1207
1208public:
1209 VPIRMetadata() = default;
1210
1211 /// Adds metatadata that can be preserved from the original instruction
1212 /// \p I.
1214 getMetadataToPropagate(&I, Metadata);
1215 // Retain the branch weights of terminators. They are used to compute the
1216 // frequencies with which the blocks of the original loop execute. Also
1217 // retain !prof on selects.
1218 if (I.isTerminator() || isa<SelectInst>(&I))
1219 if (MDNode *BW = I.getMetadata(LLVMContext::MD_prof))
1220 Metadata.emplace_back(LLVMContext::MD_prof, BW);
1221 }
1222
1223 /// Copy constructor for cloning.
1225
1227
1228 /// Add all metadata to \p I.
1229 void applyMetadata(Instruction &I) const;
1230
1231 /// Set metadata with kind \p Kind to \p Node. If metadata with \p Kind
1232 /// already exists, it will be replaced. Otherwise, it will be added.
1233 void setMetadata(unsigned Kind, MDNode *Node) {
1234 auto It =
1235 llvm::find_if(Metadata, [Kind](const std::pair<unsigned, MDNode *> &P) {
1236 return P.first == Kind;
1237 });
1238 if (It != Metadata.end())
1239 It->second = Node;
1240 else
1241 Metadata.emplace_back(Kind, Node);
1242 }
1243
1244 /// Remove the metadata of kind \p Kind, if present.
1245 void eraseMetadata(unsigned Kind) {
1246 erase_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
1247 }
1248
1249 /// Intersect this VPIRMetadata object with \p MD, keeping only metadata
1250 /// nodes that are common to both.
1251 void intersect(const VPIRMetadata &MD);
1252
1253 /// Get metadata of kind \p Kind. Returns nullptr if not found.
1254 MDNode *getMetadata(unsigned Kind) const {
1255 auto It =
1256 find_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
1257 return It != Metadata.end() ? It->second : nullptr;
1258 }
1259
1260 /// Record that the recipe executes with frequency \p Freq, relative to the
1261 /// entry of the loop region.
1262 void setExecutionFrequency(std::optional<VPExecutionFrequency> Freq,
1263 LLVMContext &Ctx);
1264
1265 /// Returns the frequency recorded by setExecutionFrequency, if any.
1266 std::optional<VPExecutionFrequency> getExecutionFrequency() const;
1267
1268 /// Drop the frequency recorded by setExecutionFrequency, if any.
1269 void clearExecutionFrequency();
1270
1271 /// Returns the branch weights recorded for this terminator, preferring real
1272 /// profile data over an estimate, or nullptr if there are none.
1274 MDNode *Node = getMetadata(LLVMContext::MD_prof);
1275 return Node ? Node : getInternalMetadata(EstimatedProfileMDName);
1276 }
1277
1278 /// Returns true if the weights returned by getBranchWeights are estimated.
1280 return getInternalMetadata(EstimatedProfileMDName);
1281 }
1282
1283 /// Set estimated branch weights to \p Node.
1285 assert(!getMetadata(LLVMContext::MD_prof) &&
1286 "real profile data takes precedence over an estimate");
1287 setMetadata(Node->getContext().getMDKindID(EstimatedProfileMDName), Node);
1288 }
1289
1290#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1291 /// Print metadata with node IDs.
1292 void print(raw_ostream &O, VPSlotTracker &SlotTracker) const;
1293#endif
1294};
1295
1296/// This is a concrete Recipe that models a single VPlan-level instruction.
1297/// While as any Recipe it may generate a sequence of IR instructions when
1298/// executed, these instructions would always form a single-def expression as
1299/// the VPInstruction is also a single def-use vertex. Most VPInstruction
1300/// opcodes can take an optional mask. Masks may be assigned during
1301/// predication.
1303 public VPIRMetadata {
1304public:
1305 /// VPlan opcodes, extending LLVM IR with idiomatics instructions.
1306 enum {
1307 FirstOrderRecurrenceSplice = Instruction::OtherOpsEnd +
1308 1, // Combines the incoming and previous
1309 // values of a first-order recurrence.
1311 // Creates a mask where each lane is active (true) whilst the current
1312 // counter (first operand + index) is less than the second operand. i.e.
1313 // mask[i] = icmpt ult (op0 + i), op1
1314 // ActiveLaneMask is used for early-exit loops with stores, plus tail
1315 // folding for all styles except DataAndControlFlow. The size of the
1316 // mask returned is VF. When unrolled, ActiveLaneMask is duplicated.
1318 // As above, but takes an additional operand (Multiplier). The size of
1319 // the mask returned is VF * Multiplier (UF, op2).
1320 // WideActiveLaneMask is used for control flow and is unrolled by widening,
1321 // with one extract vector created per unroll part.
1323 // Signature: Vectors... -> WideVector
1324 // Concatenates all vector operands to a single wide vector.
1326 // Signature: (Multiplier, Address, Align) -> Vector
1327 // Loads a single wide vector of `Multiplier * VF` elements.
1329 // Signature: (Multiplier, Address, Alignment, Vector)
1330 // Stores a single wide vector of `Multiplier * VF` elements.
1332 // Extracts each unrolled part of a (VF * UF) widened vector/mask.
1335 // Represents the incoming loop-invariant alias-mask. All memory accesses
1336 // in the loop must stay within the active lanes.
1338 // Increment the canonical IV separately for each unrolled part.
1340 // Abstract instruction that compares two values and branches. This is
1341 // lowered to ICmp + BranchOnCond during VPlan to VPlan transformation.
1344 // Branch with 2 boolean condition operands and 3 successors. If condition
1345 // 0 is true, branches to successor 0; if condition 1 is true, branches to
1346 // successor 1; otherwise branches to successor 2. Expanded after region
1347 // dissolution into: (1) an OR of the two conditions branching to
1348 // middle.split or successor 2, and (2) middle.split branching to successor
1349 // 0 or successor 1 based on condition 0.
1352 /// Given operands of (the same) struct type, creates a struct of fixed-
1353 /// width vectors each containing a struct field of all operands. The
1354 /// number of operands matches the element count of every vector.
1356 /// Creates a fixed-width vector containing all operands. The number of
1357 /// operands matches the vector element count.
1359 /// Extracts all lanes from its (non-scalable) vector operand. This is an
1360 /// abstract VPInstruction whose single defined VPValue represents VF
1361 /// scalars extracted from a vector, to be replaced by VF ExtractElement
1362 /// VPInstructions.
1364 /// Reduce the operands to the final reduction result using the operation
1365 /// specified via the operation's VPIRFlags.
1367 // Extracts the last part of its operand. Removed during unrolling.
1369 // Extracts the last lane of its vector operand, per part.
1371 // Extracts the second-to-last lane from its operand or the second-to-last
1372 // part if it is scalar. In the latter case, the recipe will be removed
1373 // during unrolling.
1375 LogicalAnd, // Non-poison propagating logical And.
1376 LogicalOr, // Non-poison propagating logical Or.
1377 NumActiveLanes, // Counts the number of active lanes in a mask.
1378 // Add an offset in bytes (second operand) to a base pointer (first
1379 // operand). Only generates scalar values (either for the first lane only or
1380 // for all lanes, depending on its uses).
1382 // Add a vector offset in bytes (second operand) to a scalar base pointer
1383 // (first operand).
1385 // Returns a scalar boolean value, which is true if any lane of its
1386 // (boolean) vector operands is true. It produces the reduced value across
1387 // all unrolled iterations. Unrolling will add all copies of its original
1388 // operand as additional operands. Note does not block poison propagation.
1390 // Calculates the first active lane index of the vector predicate operands.
1391 // It produces the lane index across all unrolled iterations. Unrolling will
1392 // add all copies of its original operand as additional operands.
1393 // Implemented with @llvm.experimental.cttz.elts, but returns the expected
1394 // result even with operands that are all zeroes.
1396 // Calculates the last active lane index of the vector predicate operands.
1397 // The predicates must be prefix-masks (all 1s before all 0s). Used when
1398 // tail-folding to extract the correct live-out value from the last active
1399 // iteration. It produces the lane index across all unrolled iterations.
1400 // Unrolling will add all copies of its original operand as additional
1401 // operands.
1403 // Returns a reversed vector for the operand.
1405 /// Start vector for reductions with 3 operands: the original start value,
1406 /// the identity value for the reduction and an integer indicating the
1407 /// scaling factor.
1409 /// Extracts a single lane (first operand) from a set of vector operands.
1410 /// The lane specifies an index into a vector formed by combining all vector
1411 /// operands (all operands after the first one).
1413 /// Explicit user for values in the main VPlan, used by the epilogue vector
1414 /// loop.
1416 /// Extracts the last active lane from a set of vectors. The first operand
1417 /// is the default value if no lanes in the masks are active. Conceptually,
1418 /// this concatenates all data vectors (odd operands), concatenates all
1419 /// masks (even operands -- ignoring the default value), and returns the
1420 /// last active value from the combined data vector using the combined mask.
1422 /// Compute the exiting value of a wide induction after vectorization, that
1423 /// is the value of the last lane of the induction increment (i.e. its
1424 /// backedge value). Has the wide induction recipe as operand.
1427 /// Scale the first operand (vector step) by the second operand
1428 /// (scalar-step). Casts both operands to the result type if needed.
1430 // Creates a step vector starting from 0 to VF with a step of 1.
1432 /// Calls a scalar intrinsic. The intrinsic ID is the last operand.
1434
1436 };
1437
1438 /// Returns true if this recipe produces scalar values for all VF lanes.
1439 bool doesGeneratePerAllLanes() const;
1440
1441 /// Return the number of operands determined by the opcode of the
1442 /// VPInstruction, excluding mask. Returns -1u if the number of operands
1443 /// cannot be determined directly by the opcode.
1444 unsigned getNumOperandsForOpcode() const;
1445
1446private:
1447 typedef unsigned char OpcodeTy;
1448 OpcodeTy Opcode;
1449
1450 /// An optional name that can be used for the generated IR instruction.
1451 std::string Name;
1452
1453 /// Returns true if we can generate a scalar for the first lane only if
1454 /// needed.
1455 bool doesGenerateSingleScalar() const;
1456
1457 /// Utility method serving execute: Generates either a single-scalar or vector
1458 /// value. \p GenerateSingleScalar determines whether to generate a
1459 /// single-scalar value.
1460 Value *generate(VPTransformState &State, bool GenerateSingleScalar);
1461
1462 /// Returns true if the VPInstruction does not need masking.
1463 bool alwaysUnmasked() const {
1464 if (Opcode == VPInstruction::MaskedCond)
1465 return false;
1466
1467 // For now only VPInstructions with underlying values use masks.
1468 // TODO: provide masks to VPInstructions w/o underlying values.
1469 if (!getUnderlyingValue())
1470 return true;
1471
1472 return Instruction::isCast(Opcode) || Opcode == Instruction::PHI ||
1473 Opcode == Instruction::GetElementPtr;
1474 }
1475
1476public:
1477 VPInstruction(unsigned Opcode, ArrayRef<VPValue *> Operands,
1478 const VPIRFlags &Flags = {}, const VPIRMetadata &MD = {},
1479 DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "",
1480 Type *ResultTy = nullptr);
1481
1482 VP_CLASSOF_IMPL(VPRecipeBase::VPInstructionSC)
1483
1484 VPInstruction *clone() override {
1486 }
1487
1489 Type *ResultTy = nullptr) {
1490 auto *New = new VPInstruction(Opcode, NewOperands, *this, *this,
1491 getDebugLoc(), Name, ResultTy);
1492 if (getUnderlyingValue())
1493 New->setUnderlyingValue(getUnderlyingInstr());
1494 return New;
1495 }
1496
1497 unsigned getOpcode() const { return Opcode; }
1498
1499 /// Add \p Op as operand of this VPInstruction. Only supported for AnyOf,
1500 /// ComputeReductionResult, BuildVector, BuildStructVector, ExtractLane,
1501 /// ExtractLastActive, FirstActiveLane, LastActiveLane.
1502 void addOperand(VPValue *Op);
1503
1504 /// Generate the instruction.
1505 /// TODO: We currently execute only per-part unless a specific instance is
1506 /// provided.
1507 void execute(VPTransformState &State) override;
1508
1509 /// Return the cost of this VPInstruction.
1510 InstructionCost computeCost(ElementCount VF,
1511 VPCostContext &Ctx) const override;
1512
1513#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1514 /// Print the VPInstruction to dbgs() (for debugging).
1515 LLVM_DUMP_METHOD void dump() const;
1516#endif
1517
1518 bool hasResult() const {
1519 // CallInst may or may not have a result, depending on the called function.
1520 // Conservatively return calls have results for now.
1521 switch (getOpcode()) {
1522 case Instruction::Ret:
1523 case Instruction::UncondBr:
1524 case Instruction::CondBr:
1525 case Instruction::Store:
1526 case Instruction::Switch:
1527 case Instruction::IndirectBr:
1528 case Instruction::Resume:
1529 case Instruction::CatchRet:
1530 case Instruction::Unreachable:
1531 case Instruction::Fence:
1532 case Instruction::AtomicRMW:
1537 return false;
1538 default:
1539 return true;
1540 }
1541 }
1542
1543 /// Returns true if the VPInstruction has a mask operand.
1544 bool isMasked() const {
1545 unsigned NumOpsForOpcode = getNumOperandsForOpcode();
1546 // VPInstructions without a fixed number of operands cannot be masked.
1547 if (NumOpsForOpcode == -1u)
1548 return false;
1549 return NumOpsForOpcode + 1 == getNumOperands();
1550 }
1551
1552 /// Returns the number of operands, excluding the mask if the VPInstruction is
1553 /// masked.
1554 unsigned getNumOperandsWithoutMask() const {
1555 return getNumOperands() - isMasked();
1556 }
1557
1558 /// Add mask \p Mask to an unmasked VPInstruction, if it needs masking.
1559 void addMask(VPValue *Mask) {
1560 assert(!isMasked() && "recipe is already masked");
1561 if (alwaysUnmasked())
1562 return;
1563 assert(Mask->getScalarType()->isIntegerTy(1) &&
1564 "Mask must be an i1 (vector)");
1565 VPUser::addOperand(Mask);
1566 }
1567
1568 /// Returns the mask for the VPInstruction. Returns nullptr for unmasked
1569 /// VPInstructions.
1570 VPValue *getMask() const { return isMasked() ? getLastOperand() : nullptr; }
1571
1572 /// Returns an iterator range over the operands excluding the mask operand
1573 /// if present.
1580
1581 /// Returns true if the underlying opcode may read from or write to memory.
1582 bool opcodeMayReadOrWriteFromMemory() const;
1583
1584 /// Returns true if the recipe only uses the first lane of operand \p Op.
1585 bool usesFirstLaneOnly(const VPValue *Op) const override;
1586
1587 /// Returns true if the recipe only uses scalars of operand \p Op.
1588 bool usesScalars(const VPValue *Op) const override {
1589 return isSingleScalar() || usesFirstLaneOnly(Op);
1590 }
1591
1592 /// Returns true if the recipe only uses the first part of operand \p Op.
1593 bool usesFirstPartOnly(const VPValue *Op) const override;
1594
1595 /// Returns true if this VPInstruction produces a scalar value from a vector,
1596 /// e.g. by performing a reduction or extracting a lane.
1597 bool isVectorToScalar() const;
1598
1599 /// Returns true if the recipe produces a single scalar value.
1600 bool isSingleScalar() const;
1601
1602 /// Returns the symbolic name assigned to the VPInstruction.
1603 StringRef getName() const { return Name; }
1604
1605 /// Set the symbolic name for the VPInstruction.
1606 void setName(StringRef NewName) { Name = NewName.str(); }
1607
1608protected:
1609#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1610 /// Print the VPInstruction to \p O.
1611 void printRecipe(raw_ostream &O, const Twine &Indent,
1612 VPSlotTracker &SlotTracker) const override;
1613#endif
1614};
1615
1616/// Helper type to provide functions to access incoming values and blocks for
1617/// phi-like recipes.
1619protected:
1620 /// Return a VPRecipeBase* to the current object.
1621 virtual const VPRecipeBase *getAsRecipe() const = 0;
1622
1623public:
1624 virtual ~VPPhiAccessors() = default;
1625
1626 /// Returns the incoming VPValue with index \p Idx.
1627 VPValue *getIncomingValue(unsigned Idx) const {
1628 return getAsRecipe()->getOperand(Idx);
1629 }
1630
1631 /// Returns the incoming block with index \p Idx.
1632 const VPBasicBlock *getIncomingBlock(unsigned Idx) const;
1633
1634 /// Returns the incoming value for \p VPBB. \p VPBB must be an incoming block.
1636 getIncomingValueForBlock(const VPBasicBlock *VPBB) const;
1637
1638 /// Sets the incoming value for \p VPBB to \p V. \p VPBB must be an incoming
1639 /// block.
1640 void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const;
1641
1642 /// Returns the number of incoming values, also number of incoming blocks.
1643 virtual unsigned getNumIncoming() const {
1644 return getAsRecipe()->getNumOperands();
1645 }
1646
1647 /// Returns an interator range over the incoming values.
1649 return make_range(getAsRecipe()->op_begin(),
1650 getAsRecipe()->op_begin() + getNumIncoming());
1651 }
1652
1654 detail::index_iterator, std::function<const VPBasicBlock *(size_t)>>>;
1655
1656 /// Returns an iterator range over the incoming blocks.
1658 std::function<const VPBasicBlock *(size_t)> GetBlock = [this](size_t Idx) {
1659 return getIncomingBlock(Idx);
1660 };
1661 return map_range(index_range(0, getNumIncoming()), GetBlock);
1662 }
1663
1664 /// Returns an iterator range over pairs of incoming values and corresponding
1665 /// incoming blocks.
1671
1672 /// Removes the incoming value for \p IncomingBlock, which must be a
1673 /// predecessor.
1674 void removeIncomingValueFor(VPBlockBase *IncomingBlock) const;
1675
1676 /// Append \p IncomingV as an incoming value to the phi-like recipe.
1677 void addIncoming(VPValue *IncomingV) {
1678 auto *R = const_cast<VPRecipeBase *>(getAsRecipe());
1679 assert((R->getNumOperands() == 0 ||
1680 IncomingV->getScalarType() == R->getOperand(0)->getScalarType()) &&
1681 "all incoming values must have the same type");
1682 R->addOperand(IncomingV);
1683 }
1684
1685#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1686 /// Print the recipe.
1688#endif
1689};
1690
1693 const Twine &Name = "", Type *ResultTy = nullptr)
1694 : VPInstruction(Instruction::PHI, Operands, Flags, {}, DL, Name,
1695 ResultTy) {}
1696
1697 static inline bool classof(const VPUser *U) {
1698 auto *VPI = dyn_cast<VPInstruction>(U);
1699 return VPI && VPI->getOpcode() == Instruction::PHI;
1700 }
1701
1702 static inline bool classof(const VPValue *V) {
1703 auto *VPI = dyn_cast<VPInstruction>(V);
1704 return VPI && VPI->getOpcode() == Instruction::PHI;
1705 }
1706
1707 static inline bool classof(const VPSingleDefRecipe *SDR) {
1708 auto *VPI = dyn_cast<VPInstruction>(SDR);
1709 return VPI && VPI->getOpcode() == Instruction::PHI;
1710 }
1711
1712 VPPhi *clone() override {
1713 auto *PhiR = new VPPhi(operands(), *this, getDebugLoc(), getName());
1714 PhiR->setUnderlyingValue(getUnderlyingValue());
1715 return PhiR;
1716 }
1717
1718 void execute(VPTransformState &State) override;
1719
1720protected:
1721#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1722 /// Print the recipe.
1723 void printRecipe(raw_ostream &O, const Twine &Indent,
1724 VPSlotTracker &SlotTracker) const override;
1725#endif
1726
1727 const VPRecipeBase *getAsRecipe() const override { return this; }
1728};
1729
1730/// A recipe to wrap on original IR instruction not to be modified during
1731/// execution, except for PHIs. PHIs are modeled via the VPIRPhi subclass.
1732/// Expect PHIs, VPIRInstructions cannot have any operands.
1734 Instruction &I;
1735
1736protected:
1737 /// VPIRInstruction::create() should be used to create VPIRInstructions, as
1738 /// subclasses may need to be created, e.g. VPIRPhi.
1740 : VPRecipeBase(VPRecipeBase::VPIRInstructionSC, {}), I(I) {}
1741
1742public:
1743 ~VPIRInstruction() override = default;
1744
1745 /// Create a new VPIRPhi for \p \I, if it is a PHINode, otherwise create a
1746 /// VPIRInstruction.
1748
1749 VP_CLASSOF_IMPL(VPRecipeBase::VPIRInstructionSC)
1750
1752 auto *R = create(I);
1753 for (auto *Op : operands())
1754 R->addOperand(Op);
1755 return R;
1756 }
1757
1758 void execute(VPTransformState &State) override;
1759
1760 /// Return the cost of this VPIRInstruction.
1762 computeCost(ElementCount VF, VPCostContext &Ctx) const override;
1763
1764 Instruction &getInstruction() const { return I; }
1765
1766 bool usesScalars(const VPValue *Op) const override {
1768 "Op must be an operand of the recipe");
1769 return true;
1770 }
1771
1772 bool usesFirstPartOnly(const VPValue *Op) const override {
1774 "Op must be an operand of the recipe");
1775 return true;
1776 }
1777
1778 bool usesFirstLaneOnly(const VPValue *Op) const override {
1780 "Op must be an operand of the recipe");
1781 return true;
1782 }
1783
1784protected:
1785#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1786 /// Print the recipe.
1787 void printRecipe(raw_ostream &O, const Twine &Indent,
1788 VPSlotTracker &SlotTracker) const override;
1789#endif
1790};
1791
1792/// An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use
1793/// cast/dyn_cast/isa and execute() implementation. A single VPValue operand is
1794/// allowed, and it is used to add a new incoming value for the single
1795/// predecessor VPBB.
1797 public VPPhiAccessors {
1799
1800 static inline bool classof(const VPRecipeBase *U) {
1801 auto *R = dyn_cast<VPIRInstruction>(U);
1802 return R && isa<PHINode>(R->getInstruction());
1803 }
1804
1805 static inline bool classof(const VPUser *U) {
1806 auto *R = dyn_cast<VPRecipeBase>(U);
1807 return R && classof(R);
1808 }
1809
1811
1812 void execute(VPTransformState &State) override;
1813
1814protected:
1815#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1816 /// Print the recipe.
1817 void printRecipe(raw_ostream &O, const Twine &Indent,
1818 VPSlotTracker &SlotTracker) const override;
1819#endif
1820
1821 const VPRecipeBase *getAsRecipe() const override { return this; }
1822};
1823
1824/// VPWidenRecipe is a recipe for producing a widened instruction using the
1825/// opcode and operands of the recipe. This recipe covers most of the
1826/// traditional vectorization cases where each recipe transforms into a
1827/// vectorized version of itself.
1829 public VPIRMetadata {
1830 unsigned Opcode;
1831
1832public:
1834 const VPIRFlags &Flags = {}, const VPIRMetadata &Metadata = {},
1835 DebugLoc DL = {})
1836 : VPWidenRecipe(I.getOpcode(), Operands, Flags, Metadata, DL) {
1837 setUnderlyingValue(&I);
1838 }
1839
1841 const VPIRFlags &Flags = {}, const VPIRMetadata &Metadata = {},
1842 DebugLoc DL = {})
1843 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenSC, Operands,
1845 Flags, DL),
1846 VPIRMetadata(Metadata), Opcode(Opcode) {
1847 assert(flagsValidForOpcode(Opcode) &&
1848 "Set flags not supported for the provided opcode");
1849 assert(hasRequiredFlagsForOpcode(Opcode, getScalarType()) &&
1850 "Opcode requires specific flags to be set");
1851 }
1852
1853 ~VPWidenRecipe() override = default;
1854
1856
1858 if (auto *UV = getUnderlyingValue())
1859 return new VPWidenRecipe(*cast<Instruction>(UV), NewOperands, *this,
1860 *this, getDebugLoc());
1861 return new VPWidenRecipe(Opcode, NewOperands, *this, *this, getDebugLoc());
1862 }
1863
1864 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenSC)
1865
1866 /// Produce a widened instruction using the opcode and operands of the recipe,
1867 /// processing State.VF elements.
1868 void execute(VPTransformState &State) override;
1869
1870 /// Return the cost of this VPWidenRecipe.
1871 InstructionCost computeCost(ElementCount VF,
1872 VPCostContext &Ctx) const override;
1873
1874 unsigned getOpcode() const { return Opcode; }
1875
1876protected:
1877#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1878 /// Print the recipe.
1879 void printRecipe(raw_ostream &O, const Twine &Indent,
1880 VPSlotTracker &SlotTracker) const override;
1881#endif
1882
1883 /// Returns true if the recipe only uses the first lane of operand \p Op.
1884 bool usesFirstLaneOnly(const VPValue *Op) const override {
1886 "Op must be an operand of the recipe");
1887 return Opcode == Instruction::Select && Op == getOperand(0) &&
1889 }
1890};
1891
1892/// VPWidenCastRecipe is a recipe to create vector cast instructions.
1893/// TODO: Merge with VPWidenRecipe now that type is associated to every
1894/// VPRecipeValue.
1896 public VPIRMetadata {
1897 /// Cast instruction opcode.
1898 Instruction::CastOps Opcode;
1899
1900public:
1902 CastInst *CI = nullptr, const VPIRFlags &Flags = {},
1903 const VPIRMetadata &Metadata = {},
1904 DebugLoc DL = DebugLoc::getUnknown())
1905 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCastSC, Op, ResultTy, Flags,
1906 DL),
1907 VPIRMetadata(Metadata), Opcode(Opcode) {
1908 assert(flagsValidForOpcode(Opcode) &&
1909 "Set flags not supported for the provided opcode");
1910 assert(hasRequiredFlagsForOpcode(Opcode, ResultTy) &&
1911 "Opcode requires specific flags to be set");
1912 setUnderlyingValue(CI);
1913 }
1914
1915 ~VPWidenCastRecipe() override = default;
1916
1918 return new VPWidenCastRecipe(Opcode, getOperand(0), getScalarType(),
1920 *this, *this, getDebugLoc());
1921 }
1922
1923 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCastSC)
1924
1925 /// Produce widened copies of the cast.
1926 void execute(VPTransformState &State) override;
1927
1928 /// Return the cost of this VPWidenCastRecipe.
1929 InstructionCost computeCost(ElementCount VF,
1930 VPCostContext &Ctx) const override;
1931
1932 Instruction::CastOps getOpcode() const { return Opcode; }
1933
1934protected:
1935#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1936 /// Print the recipe.
1937 void printRecipe(raw_ostream &O, const Twine &Indent,
1938 VPSlotTracker &SlotTracker) const override;
1939#endif
1940};
1941
1942/// A recipe for widening vector intrinsics.
1944 public VPIRMetadata {
1945 /// ID of the vector intrinsic to widen.
1946 Intrinsic::ID VectorIntrinsicID;
1947
1948 /// True if the intrinsic may read from memory.
1949 bool MayReadFromMemory;
1950
1951 /// True if the intrinsic may read write to memory.
1952 bool MayWriteToMemory;
1953
1954 /// True if the intrinsic may have side-effects.
1955 bool MayHaveSideEffects;
1956
1957protected:
1959 ArrayRef<VPValue *> CallArguments, Type *Ty,
1960 const VPIRFlags &Flags = {},
1961 const VPIRMetadata &MD = {},
1962 DebugLoc DL = DebugLoc::getUnknown())
1963 : VPRecipeWithIRFlags(SC, CallArguments, Ty, Flags, DL), VPIRMetadata(MD),
1964 VectorIntrinsicID(VectorIntrinsicID) {
1965 LLVMContext &Ctx = Ty->getContext();
1966 AttributeSet Attrs = Intrinsic::getFnAttributes(Ctx, VectorIntrinsicID);
1967 MemoryEffects ME = Attrs.getMemoryEffects();
1968 MayReadFromMemory = !ME.onlyWritesMemory();
1969 MayWriteToMemory = !ME.onlyReadsMemory();
1970 MayHaveSideEffects = MayWriteToMemory ||
1971 !Attrs.hasAttribute(Attribute::NoUnwind) ||
1972 !Attrs.hasAttribute(Attribute::WillReturn);
1973 }
1974
1975 /// Helper function to produce the widened intrinsic call.
1976 CallInst *createVectorCall(VPTransformState &State);
1977
1978public:
1980 ArrayRef<VPValue *> CallArguments, Type *Ty,
1981 const VPIRFlags &Flags = {},
1982 const VPIRMetadata &MD = {},
1983 DebugLoc DL = DebugLoc::getUnknown())
1984 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenIntrinsicSC, CallArguments, Ty,
1985 Flags, DL),
1986 VPIRMetadata(MD), VectorIntrinsicID(VectorIntrinsicID),
1987 MayReadFromMemory(CI.mayReadFromMemory()),
1988 MayWriteToMemory(CI.mayWriteToMemory()),
1989 MayHaveSideEffects(CI.mayHaveSideEffects()) {
1990 setUnderlyingValue(&CI);
1991 }
1992
1994 ArrayRef<VPValue *> CallArguments, Type *Ty,
1995 const VPIRFlags &Flags = {},
1996 const VPIRMetadata &Metadata = {},
1997 DebugLoc DL = DebugLoc::getUnknown())
1998 : VPWidenIntrinsicRecipe(VPRecipeBase::VPWidenIntrinsicSC,
1999 VectorIntrinsicID, CallArguments, Ty, Flags,
2000 Metadata, DL) {}
2001
2002 ~VPWidenIntrinsicRecipe() override = default;
2003
2005 if (Value *CI = getUnderlyingValue())
2006 return new VPWidenIntrinsicRecipe(*cast<CallInst>(CI), VectorIntrinsicID,
2007 operands(), getScalarType(), *this,
2008 *this, getDebugLoc());
2009 return new VPWidenIntrinsicRecipe(VectorIntrinsicID, operands(),
2010 getScalarType(), *this, *this,
2011 getDebugLoc());
2012 }
2013
2014 static inline bool classof(const VPRecipeBase *R) {
2015 return R->getVPRecipeID() == VPRecipeBase::VPWidenIntrinsicSC ||
2016 R->getVPRecipeID() == VPRecipeBase::VPWidenMemIntrinsicSC;
2017 }
2018
2019 static inline bool classof(const VPUser *U) {
2020 auto *R = dyn_cast<VPRecipeBase>(U);
2021 return R && classof(R);
2022 }
2023
2024 static inline bool classof(const VPValue *V) {
2025 auto *R = V->getDefiningRecipe();
2026 return R && classof(R);
2027 }
2028
2029 static inline bool classof(const VPSingleDefRecipe *R) {
2030 return classof(static_cast<const VPRecipeBase *>(R));
2031 }
2032
2033 /// Produce a widened version of the vector intrinsic.
2034 void execute(VPTransformState &State) override;
2035
2036 /// Compute the cost of a vector intrinsic with \p ID and \p Operands.
2037 static InstructionCost computeCallCost(Intrinsic::ID ID,
2039 const VPRecipeWithIRFlags &R,
2040 ElementCount VF, VPCostContext &Ctx);
2041
2042 /// Return the cost of this vector intrinsic.
2043 InstructionCost computeCost(ElementCount VF,
2044 VPCostContext &Ctx) const override;
2045
2046 /// Return the ID of the intrinsic.
2047 Intrinsic::ID getVectorIntrinsicID() const { return VectorIntrinsicID; }
2048
2049 /// Return to name of the intrinsic as string.
2050 StringRef getIntrinsicName() const;
2051
2052 /// Returns true if the intrinsic may read from memory.
2053 bool mayReadFromMemory() const { return MayReadFromMemory; }
2054
2055 /// Returns true if the intrinsic may write to memory.
2056 bool mayWriteToMemory() const { return MayWriteToMemory; }
2057
2058 /// Returns true if the intrinsic may have side-effects.
2059 bool mayHaveSideEffects() const { return MayHaveSideEffects; }
2060
2061 bool usesFirstLaneOnly(const VPValue *Op) const override;
2062
2063protected:
2064#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2065 /// Print the recipe.
2066 void printRecipe(raw_ostream &O, const Twine &Indent,
2067 VPSlotTracker &SlotTracker) const override;
2068#endif
2069};
2070
2071/// A recipe for widening vector memory intrinsics.
2073 /// Alignment information for this memory access.
2074 Align Alignment;
2075
2076public:
2078 ArrayRef<VPValue *> CallArguments, Type *Ty,
2079 Align Alignment, const VPIRMetadata &MD = {},
2081 : VPWidenIntrinsicRecipe(VPRecipeBase::VPWidenMemIntrinsicSC,
2082 VectorIntrinsicID, CallArguments, Ty, {}, MD,
2083 DL),
2084 Alignment(Alignment) {
2085 assert((VectorIntrinsicID == Intrinsic::experimental_vp_strided_load ||
2086 VectorIntrinsicID == Intrinsic::experimental_vp_strided_store) &&
2087 "Unexpected intrinsic");
2088 }
2089
2090 ~VPWidenMemIntrinsicRecipe() override = default;
2091
2094 getScalarType(), Alignment, *this,
2095 getDebugLoc());
2096 }
2097
2098 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenMemIntrinsicSC)
2099
2100 /// Produce a widened version of the vector memory intrinsic.
2101 void execute(VPTransformState &State) override;
2102
2103 /// Helper function for computing the cost of vector memory intrinsic.
2105 bool IsMasked, Align Alignment,
2106 VPCostContext &Ctx);
2107
2108 /// Return the cost of this vector memory intrinsic.
2110 VPCostContext &Ctx) const override;
2111};
2112
2113/// A recipe for widening Call instructions using library calls.
2115 public VPIRMetadata {
2116 /// Variant stores a pointer to the chosen function. There is a 1:1 mapping
2117 /// between a given VF and the chosen vectorized variant, so there will be a
2118 /// different VPlan for each VF with a valid variant.
2119 Function *Variant;
2120
2121public:
2123 ArrayRef<VPValue *> CallArguments,
2124 const VPIRFlags &Flags = {},
2125 const VPIRMetadata &Metadata = {}, DebugLoc DL = {})
2126 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCallSC, CallArguments,
2127 toScalarizedTy(Variant->getReturnType()), Flags,
2128 DL),
2129 VPIRMetadata(Metadata), Variant(Variant) {
2130 setUnderlyingValue(UV);
2131 assert(isa<Function>(getLastOperand()->getLiveInIRValue()) &&
2132 "last operand must be the called function");
2133 assert(cast<Function>(CallArguments.back()->getLiveInIRValue())
2134 ->getReturnType() == getScalarType() &&
2135 "Scalar type must match return type of called scalar function");
2136 }
2137
2138 ~VPWidenCallRecipe() override = default;
2139
2141 return new VPWidenCallRecipe(getUnderlyingValue(), Variant, operands(),
2142 *this, *this, getDebugLoc());
2143 }
2144
2145 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCallSC)
2146
2147 /// Produce a widened version of the call instruction.
2148 void execute(VPTransformState &State) override;
2149
2150 /// Return the cost of this VPWidenCallRecipe.
2151 InstructionCost computeCost(ElementCount VF,
2152 VPCostContext &Ctx) const override;
2153
2154 /// Return the cost of widening a call using the vector function \p Variant.
2155 static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx);
2156
2160
2163
2164 /// Returns true if the recipe only uses the first lane of operand \p Op.
2165 bool usesFirstLaneOnly(const VPValue *Op) const override;
2166
2167protected:
2168#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2169 /// Print the recipe.
2170 void printRecipe(raw_ostream &O, const Twine &Indent,
2171 VPSlotTracker &SlotTracker) const override;
2172#endif
2173};
2174
2175/// A recipe representing a sequence of load -> update -> store as part of
2176/// a histogram operation. This means there may be aliasing between vector
2177/// lanes, which is handled by the llvm.experimental.vector.histogram family
2178/// of intrinsics. The only update operations currently supported are
2179/// 'add' and 'sub' where the other term is loop-invariant.
2181 /// Opcode of the update operation, currently either add or sub.
2182 unsigned Opcode;
2183
2184public:
2185 VPHistogramRecipe(unsigned Opcode, ArrayRef<VPValue *> Operands,
2186 const VPIRMetadata &Metadata = {},
2188 : VPRecipeBase(VPRecipeBase::VPHistogramSC, Operands, DL),
2189 VPIRMetadata(Metadata), Opcode(Opcode) {}
2190
2191 ~VPHistogramRecipe() override = default;
2192
2194 return new VPHistogramRecipe(Opcode, operands(), *this, getDebugLoc());
2195 }
2196
2197 VP_CLASSOF_IMPL(VPRecipeBase::VPHistogramSC);
2198
2199 /// Produce a vectorized histogram operation.
2200 void execute(VPTransformState &State) override;
2201
2202 /// Return the cost of this VPHistogramRecipe.
2204 VPCostContext &Ctx) const override;
2205
2206 /// Return the mask operand if one was provided, or a null pointer if all
2207 /// lanes should be executed unconditionally.
2208 VPValue *getMask() const {
2209 return getNumOperands() == 3 ? getOperand(2) : nullptr;
2210 }
2211
2212 /// Returns true if the recipe only uses the first lane of operand \p Op.
2213 bool usesFirstLaneOnly(const VPValue *Op) const override {
2215 "Op must be an operand of the recipe");
2216 return Op == getOperand(1);
2217 }
2218
2219protected:
2220#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2221 /// Print the recipe
2222 void printRecipe(raw_ostream &O, const Twine &Indent,
2223 VPSlotTracker &SlotTracker) const override;
2224#endif
2225};
2226
2227/// A recipe for handling GEP instructions.
2229 Type *SourceElementTy;
2230
2231public:
2233 const VPIRFlags &Flags = {},
2235 GetElementPtrInst *UV = nullptr)
2236 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenGEPSC, Operands,
2237 Operands[0]->getScalarType(), Flags, DL),
2238 SourceElementTy(SourceElementTy) {
2239 if (UV) {
2240 setUnderlyingValue(UV);
2243 assert(Metadata.empty() && "unexpected metadata on GEP");
2244 }
2245 }
2246
2247 ~VPWidenGEPRecipe() override = default;
2248
2254
2255 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenGEPSC)
2256
2257 /// This recipe generates a GEP instruction.
2258 unsigned getOpcode() const { return Instruction::GetElementPtr; }
2259
2260 /// Generate the gep nodes.
2261 void execute(VPTransformState &State) override;
2262
2263 Type *getSourceElementType() const { return SourceElementTy; }
2264
2265 /// Return the cost of this VPWidenGEPRecipe.
2267 VPCostContext &Ctx) const override {
2268 // TODO: Compute accurate cost after retiring the legacy cost model.
2269 return 0;
2270 }
2271
2272 /// Returns true if the recipe only uses the first lane of operand \p Op.
2273 bool usesFirstLaneOnly(const VPValue *Op) const override;
2274
2275protected:
2276#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2277 /// Print the recipe.
2278 void printRecipe(raw_ostream &O, const Twine &Indent,
2279 VPSlotTracker &SlotTracker) const override;
2280#endif
2281};
2282
2283/// A recipe to compute a pointer to the last element of each part of a widened
2284/// memory access for widened memory accesses of SourceElementTy. Used for
2285/// VPWidenMemoryRecipes or VPInterleaveRecipes that are reversed. An extra
2286/// Offset operand is added by convertToConcreteRecipes when UF = 1, and by the
2287/// unroller otherwise.
2289 Type *SourceElementTy;
2290
2291 /// The constant stride of the pointer computed by this recipe, expressed in
2292 /// units of SourceElementTy.
2293 int64_t Stride;
2294
2295public:
2296 VPVectorEndPointerRecipe(VPValue *Ptr, VPValue *VF, Type *SourceElementTy,
2297 int64_t Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
2298 : VPRecipeWithIRFlags(VPRecipeBase::VPVectorEndPointerSC, {Ptr, VF},
2299 Ptr->getScalarType(), GEPFlags, DL),
2300 SourceElementTy(SourceElementTy), Stride(Stride) {
2301 assert(Stride < 0 && "Stride must be negative");
2302 }
2303
2304 VP_CLASSOF_IMPL(VPRecipeBase::VPVectorEndPointerSC)
2305
2306 Type *getSourceElementType() const { return SourceElementTy; }
2307 int64_t getStride() const { return Stride; }
2308 VPValue *getPointer() const { return getOperand(0); }
2309 VPValue *getVFValue() const { return getOperand(1); }
2311 return getNumOperands() == 3 ? getOperand(2) : nullptr;
2312 }
2313
2314 /// Adds the offset operand to the recipe.
2315 /// Offset = Stride * (VF - 1) + Part * Stride * VF.
2316 void materializeOffset(unsigned Part = 0);
2317
2318 /// Append \p Offset as the offset operand. The offset is an integer index
2319 /// expressed in units of SourceElementTy.
2321 assert(Offset->getScalarType()->isIntegerTy() &&
2322 "offset must be an integer index");
2324 }
2325
2326 void execute(VPTransformState &State) override;
2327
2328 bool usesFirstLaneOnly(const VPValue *Op) const override {
2330 "Op must be an operand of the recipe");
2331 return true;
2332 }
2333
2334 /// Return the cost of this VPVectorPointerRecipe.
2336 VPCostContext &Ctx) const override {
2337 // TODO: Compute accurate cost after retiring the legacy cost model.
2338 return 0;
2339 }
2340
2341 /// Returns true if the recipe only uses the first part of operand \p Op.
2342 bool usesFirstPartOnly(const VPValue *Op) const override {
2344 "Op must be an operand of the recipe");
2345 assert(getNumOperands() <= 2 && "must have at most two operands");
2346 return true;
2347 }
2348
2350 auto *VEPR = new VPVectorEndPointerRecipe(
2353 if (auto *Offset = getOffset())
2354 VEPR->addOffset(Offset);
2355 return VEPR;
2356 }
2357
2358protected:
2359#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2360 /// Print the recipe.
2361 void printRecipe(raw_ostream &O, const Twine &Indent,
2362 VPSlotTracker &SlotTracker) const override;
2363#endif
2364};
2365
2366/// A recipe to compute the pointers for widened memory accesses of \p
2367/// SourceElementTy, with the \p Stride expressed in units of \p
2368/// SourceElementTy. Unrolling adds an extra \p VFxPart operand for unrolled
2369/// parts > 0 and it produces `GEP SourceElementTy Ptr, VFxPart * Stride`.
2371 Type *SourceElementTy;
2372
2373public:
2374 VPVectorPointerRecipe(VPValue *Ptr, Type *SourceElementTy, VPValue *Stride,
2375 GEPNoWrapFlags GEPFlags, DebugLoc DL)
2376 : VPRecipeWithIRFlags(VPRecipeBase::VPVectorPointerSC,
2377 ArrayRef<VPValue *>({Ptr, Stride}),
2378 Ptr->getScalarType(), GEPFlags, DL),
2379 SourceElementTy(SourceElementTy) {}
2380
2381 VP_CLASSOF_IMPL(VPRecipeBase::VPVectorPointerSC)
2382
2383 VPValue *getStride() const { return getOperand(1); }
2384
2386 return getNumOperands() > 2 ? getOperand(2) : nullptr;
2387 }
2388
2389 /// Add the per-part offset (VFxPart) used for unrolled parts > 0.
2390 void addPerPartOffset(VPValue *VFxPart) {
2391 assert(VFxPart->getScalarType()->isIntegerTy() &&
2392 "per-part offset must be an integer index");
2393 VPUser::addOperand(VFxPart);
2394 }
2395
2396 void execute(VPTransformState &State) override;
2397
2398 Type *getSourceElementType() const { return SourceElementTy; }
2399
2400 bool usesFirstLaneOnly(const VPValue *Op) const override {
2402 "Op must be an operand of the recipe");
2403 return true;
2404 }
2405
2406 /// Returns true if the recipe only uses the first part of operand \p Op.
2407 bool usesFirstPartOnly(const VPValue *Op) const override {
2409 "Op must be an operand of the recipe");
2410 assert(getNumOperands() <= 2 && "must have at most two operands");
2411 return true;
2412 }
2413
2415 auto *Clone =
2416 new VPVectorPointerRecipe(getOperand(0), SourceElementTy, getStride(),
2418 if (auto *VFxPart = getVFxPart())
2419 Clone->addPerPartOffset(VFxPart);
2420 return Clone;
2421 }
2422
2423 /// Return the cost of this VPHeaderPHIRecipe.
2425 VPCostContext &Ctx) const override {
2426 // TODO: Compute accurate cost after retiring the legacy cost model.
2427 return 0;
2428 }
2429
2430protected:
2431#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2432 /// Print the recipe.
2433 void printRecipe(raw_ostream &O, const Twine &Indent,
2434 VPSlotTracker &SlotTracker) const override;
2435#endif
2436};
2437
2438/// A pure virtual base class for all recipes modeling header phis, including
2439/// phis for first order recurrences, pointer inductions and reductions. The
2440/// start value is the first operand of the recipe and the incoming value from
2441/// the backedge is the second operand.
2442///
2443/// Inductions are modeled using the following sub-classes:
2444/// * VPWidenIntOrFpInductionRecipe: Generates vector values for integer and
2445/// floating point inductions with arbitrary start and step values. Produces
2446/// a vector PHI per-part.
2447/// * VPWidenPointerInductionRecipe: Generate vector and scalar values for a
2448/// pointer induction. Produces either a vector PHI per-part or scalar values
2449/// per-lane based on the canonical induction.
2450/// * VPFirstOrderRecurrencePHIRecipe
2451/// * VPReductionPHIRecipe
2452/// * VPActiveLaneMaskPHIRecipe
2453/// * VPEVLBasedIVPHIRecipe
2454///
2455/// Note that the canonical IV is modeled as a VPRegionValue associated with
2456/// its loop region.
2458 public VPPhiAccessors {
2459protected:
2460 VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr,
2461 VPValue *Start, Type *ResultTy,
2463 : VPSingleDefRecipe(VPRecipeID, Start, ResultTy, UnderlyingInstr, DL) {}
2464
2465 const VPRecipeBase *getAsRecipe() const override { return this; }
2466
2467public:
2468 ~VPHeaderPHIRecipe() override = default;
2469
2470 /// Method to support type inquiry through isa, cast, and dyn_cast.
2471 static inline bool classof(const VPRecipeBase *R) {
2472 return R->getVPRecipeID() >= VPRecipeBase::VPFirstHeaderPHISC &&
2473 R->getVPRecipeID() <= VPRecipeBase::VPLastHeaderPHISC;
2474 }
2475 static inline bool classof(const VPValue *V) {
2476 return isa<VPHeaderPHIRecipe>(V->getDefiningRecipe());
2477 }
2478 static inline bool classof(const VPSingleDefRecipe *R) {
2479 return isa<VPHeaderPHIRecipe>(static_cast<const VPRecipeBase *>(R));
2480 }
2481
2482 /// Generate the phi nodes.
2483 void execute(VPTransformState &State) override = 0;
2484
2485 /// Return the cost of this header phi recipe.
2487 VPCostContext &Ctx) const override;
2488
2489 /// Returns the start value of the phi, if one is set.
2491 return getNumOperands() == 0 ? nullptr : getOperand(0);
2492 }
2494 return getNumOperands() == 0 ? nullptr : getOperand(0);
2495 }
2496
2497 /// Update the start value of the recipe.
2499
2500 /// Returns the incoming value from the loop backedge.
2501 virtual VPValue *getBackedgeValue() { return getOperand(1); }
2502
2503 /// Update the incoming value from the loop backedge.
2505
2506 /// Add \p V as the incoming value from the loop backedge.
2508 assert(getNumOperands() == 1 &&
2509 "backedge value must be appended right after construction");
2510 assert(V->getScalarType() == getScalarType() &&
2511 "backedge value must have the same type as the start value");
2513 }
2514
2515protected:
2516#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2517 /// Print the recipe.
2518 void printRecipe(raw_ostream &O, const Twine &Indent,
2519 VPSlotTracker &SlotTracker) const override = 0;
2520#endif
2521};
2522
2523/// Base class for widened induction (VPWidenIntOrFpInductionRecipe and
2524/// VPWidenPointerInductionRecipe), providing shared functionality, including
2525/// retrieving the step value, induction descriptor and original phi node.
2527 InductionDescriptor IndDesc;
2528
2529public:
2531 VPValue *Step, const InductionDescriptor &IndDesc,
2532 Type *ResultTy, DebugLoc DL)
2533 : VPHeaderPHIRecipe(Kind, IV, Start, ResultTy, DL), IndDesc(IndDesc) {
2534 addOperand(Step);
2535 }
2536
2537 /// After unrolling, append the splat-VF step (`VF * step`) and the value of
2538 /// the induction at the last unrolled part.
2539 void addUnrolledPartOperands(VPValue *SplatVFStep, VPValue *LastPart) {
2540 assert(LastPart->getScalarType() == getScalarType() &&
2541 "last-part value must match the induction recipe's scalar type");
2543 ? SplatVFStep->getScalarType()->isIntegerTy()
2544 : SplatVFStep->getScalarType() == getScalarType()) &&
2545 "splat-step must match the induction type for non-pointer "
2546 "inductions, or be an integer index for pointer inductions");
2547 VPUser::addOperand(SplatVFStep);
2548 VPUser::addOperand(LastPart);
2549 }
2550
2551 static inline bool classof(const VPRecipeBase *R) {
2552 return R->getVPRecipeID() == VPRecipeBase::VPWidenIntOrFpInductionSC ||
2553 R->getVPRecipeID() == VPRecipeBase::VPWidenPointerInductionSC;
2554 }
2555
2556 static inline bool classof(const VPValue *V) {
2557 auto *R = V->getDefiningRecipe();
2558 return R && classof(R);
2559 }
2560
2561 static inline bool classof(const VPSingleDefRecipe *R) {
2562 return classof(static_cast<const VPRecipeBase *>(R));
2563 }
2564
2565 void execute(VPTransformState &State) override = 0;
2566
2567 /// Returns the step value of the induction.
2569 const VPValue *getStepValue() const { return getOperand(1); }
2570
2572 const VPValue *getVFValue() const { return getOperand(2); }
2573
2574 /// Returns the number of incoming values, also number of incoming blocks.
2575 /// Note that at the moment, VPWidenPointerInductionRecipe only has a single
2576 /// incoming value, its start value.
2577 unsigned getNumIncoming() const override { return 1; }
2578
2579 /// Returns the underlying PHINode if one exists, or null otherwise.
2583
2584 /// Returns the induction descriptor for the recipe.
2585 const InductionDescriptor &getInductionDescriptor() const { return IndDesc; }
2586
2587 /// Returns the SCEV predicates associated with this induction.
2589 return IndDesc.getNoWrapPredicates();
2590 }
2591
2593 // TODO: All operands of base recipe must exist and be at same index in
2594 // derived recipe.
2596 "VPWidenIntOrFpInductionRecipe generates its own backedge value");
2597 }
2598
2599 /// Returns true if the recipe only uses the first lane of operand \p Op.
2600 bool usesFirstLaneOnly(const VPValue *Op) const override {
2602 "Op must be an operand of the recipe");
2603 // The recipe creates its own wide start value, so it only requests the
2604 // first lane of the operand.
2605 // TODO: Remove once creating the start value is modeled separately.
2606 return Op == getStartValue() || Op == getStepValue();
2607 }
2608};
2609
2610/// A recipe for handling phi nodes of integer and floating-point inductions,
2611/// producing their vector values. This is an abstract recipe and must be
2612/// converted to concrete recipes before executing.
2614 public VPIRFlags {
2615 TruncInst *Trunc;
2616
2617 // If this recipe is unrolled it will have 2 additional operands.
2618 bool isUnrolled() const { return getNumOperands() == 5; }
2619
2620public:
2622 VPValue *VF, const InductionDescriptor &IndDesc,
2623 const VPIRFlags &Flags, DebugLoc DL)
2624 : VPWidenInductionRecipe(VPRecipeBase::VPWidenIntOrFpInductionSC, IV,
2625 Start, Step, IndDesc, Start->getScalarType(),
2626 DL),
2627 VPIRFlags(Flags), Trunc(nullptr) {
2628 addOperand(VF);
2629 }
2630
2632 VPValue *VF, const InductionDescriptor &IndDesc,
2633 TruncInst *Trunc, const VPIRFlags &Flags,
2634 DebugLoc DL)
2636 VPRecipeBase::VPWidenIntOrFpInductionSC, IV, Start, Step, IndDesc,
2637 Trunc ? Trunc->getType() : Start->getScalarType(), DL),
2638 VPIRFlags(Flags), Trunc(Trunc) {
2639 addOperand(VF);
2641 if (Trunc)
2643 assert(Metadata.empty() && "unexpected metadata on Trunc");
2644 }
2645
2647
2653
2654 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenIntOrFpInductionSC)
2655
2656 void execute(VPTransformState &State) override {
2657 llvm_unreachable("cannot execute this recipe, should be expanded via "
2658 "expandVPWidenIntOrFpInductionRecipe");
2659 }
2660
2661 /// If the recipe has been unrolled, return the VPValue for the induction
2662 /// increment, otherwise return null.
2664 return isUnrolled() ? getOperand(getNumOperands() - 2) : nullptr;
2665 }
2666
2667 /// Returns the number of incoming values, also number of incoming blocks.
2668 /// Note that at the moment, VPWidenIntOrFpInductionRecipes only have a single
2669 /// incoming value, its start value.
2670 unsigned getNumIncoming() const override { return 1; }
2671
2672 /// Returns the first defined value as TruncInst, if it is one or nullptr
2673 /// otherwise.
2674 TruncInst *getTruncInst() { return Trunc; }
2675 const TruncInst *getTruncInst() const { return Trunc; }
2676
2677 /// Return the cost of this VPWidenIntOrFpInductionRecipe.
2679 VPCostContext &Ctx) const override;
2680
2681 /// Returns true if the induction is canonical, i.e. starting at 0 and
2682 /// incremented by UF * VF (= the original IV is incremented by 1) and has the
2683 /// same type as the canonical induction.
2684 bool isCanonical() const;
2685
2686 /// Returns the VPValue representing the value of this induction at
2687 /// the last unrolled part, if it exists. Returns itself if unrolling did not
2688 /// take place.
2690 return isUnrolled() ? getLastOperand() : this;
2691 }
2692
2693protected:
2694#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2695 /// Print the recipe.
2696 void printRecipe(raw_ostream &O, const Twine &Indent,
2697 VPSlotTracker &SlotTracker) const override;
2698#endif
2699};
2700
2702public:
2703 /// Create a new VPWidenPointerInductionRecipe for \p Phi with start value \p
2704 /// Start and the number of elements unrolled \p NumUnrolledElems, typically
2705 /// VF*UF.
2707 VPValue *NumUnrolledElems,
2708 const InductionDescriptor &IndDesc, DebugLoc DL)
2709 : VPWidenInductionRecipe(VPRecipeBase::VPWidenPointerInductionSC, Phi,
2710 Start, Step, IndDesc, Start->getScalarType(),
2711 DL) {
2712 addOperand(NumUnrolledElems);
2713 }
2714
2716
2722
2723 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenPointerInductionSC)
2724
2725 /// Generate vector values for the pointer induction.
2726 void execute(VPTransformState &State) override {
2727 llvm_unreachable("cannot execute this recipe, should be expanded via "
2728 "expandVPWidenPointerInduction");
2729 };
2730
2731 /// Returns true if only scalar values will be generated.
2732 bool onlyScalarsGenerated(bool IsScalable);
2733
2734 /// Return the cost of this VPWidenPointerInductionRecipe.
2736 VPCostContext &Ctx) const override;
2737
2738protected:
2739#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2740 /// Print the recipe.
2741 void printRecipe(raw_ostream &O, const Twine &Indent,
2742 VPSlotTracker &SlotTracker) const override;
2743#endif
2744};
2745
2746/// A recipe for widened phis. Incoming values are operands of the recipe and
2747/// their operand index corresponds to the incoming predecessor block. If the
2748/// recipe is placed in an entry block to a (non-replicate) region, it must have
2749/// exactly 2 incoming values, the first from the predecessor of the region and
2750/// the second from the exiting block of the region.
2752 public VPPhiAccessors {
2753 /// Name to use for the generated IR instruction for the widened phi.
2754 std::string Name;
2755
2756public:
2757 /// Create a new VPWidenPHIRecipe with incoming values \p IncomingValues,
2758 /// debug location \p DL and \p Name.
2760 DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "")
2761 : VPSingleDefRecipe(VPRecipeBase::VPWidenPHISC, IncomingValues,
2762 IncomingValues[0]->getScalarType(),
2763 /*UV=*/nullptr, DL),
2764 Name(Name.str()) {
2765 assert(all_of(IncomingValues,
2766 [this](VPValue *VPV) {
2767 return VPV->getScalarType() == getScalarType();
2768 }) &&
2769 "all incoming values must have the same type");
2770 }
2771
2773 return new VPWidenPHIRecipe(operands(), getDebugLoc(), Name);
2774 }
2775
2776 ~VPWidenPHIRecipe() override = default;
2777
2778 /// This recipe generates a PHI.
2779 unsigned getOpcode() const { return Instruction::PHI; }
2780
2781 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenPHISC)
2782
2783 /// Generate the phi/select nodes.
2784 void execute(VPTransformState &State) override;
2785
2786 /// Return the cost of this VPWidenPHIRecipe.
2787 InstructionCost computeCost(ElementCount VF,
2788 VPCostContext &Ctx) const override;
2789
2790protected:
2791#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2792 /// Print the recipe.
2793 void printRecipe(raw_ostream &O, const Twine &Indent,
2794 VPSlotTracker &SlotTracker) const override;
2795#endif
2796
2797 const VPRecipeBase *getAsRecipe() const override { return this; }
2798};
2799
2800/// A recipe for handling first-order recurrence phis. The start value is the
2801/// first operand of the recipe and the incoming value from the backedge is the
2802/// second operand.
2805 VPValue &BackedgeValue)
2806 : VPHeaderPHIRecipe(VPRecipeBase::VPFirstOrderRecurrencePHISC, Phi,
2807 &Start, Start.getScalarType()) {
2808 addOperand(&BackedgeValue);
2809 }
2810
2811 VP_CLASSOF_IMPL(VPRecipeBase::VPFirstOrderRecurrencePHISC)
2812
2817
2818 void execute(VPTransformState &State) override;
2819
2820 /// Return the cost of this first-order recurrence phi recipe.
2822 VPCostContext &Ctx) const override;
2823
2824 /// Returns true if the recipe only uses the first lane of operand \p Op.
2825 bool usesFirstLaneOnly(const VPValue *Op) const override {
2827 "Op must be an operand of the recipe");
2828 return Op == getStartValue();
2829 }
2830
2831protected:
2832#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2833 /// Print the recipe.
2834 void printRecipe(raw_ostream &O, const Twine &Indent,
2835 VPSlotTracker &SlotTracker) const override;
2836#endif
2837};
2838
2839/// Possible variants of a reduction.
2840
2841/// This reduction is ordered and in-loop.
2842struct RdxOrdered {};
2843/// This reduction is in-loop.
2844struct RdxInLoop {};
2845/// This reduction is unordered with the partial result scaled down by some
2846/// factor.
2849};
2850using ReductionStyle = std::variant<RdxOrdered, RdxInLoop, RdxUnordered>;
2851
2852inline ReductionStyle getReductionStyle(bool InLoop, bool Ordered,
2853 unsigned ScaleFactor) {
2854 assert((!Ordered || InLoop) && "Ordered implies in-loop");
2855 if (Ordered)
2856 return RdxOrdered{};
2857 if (InLoop)
2858 return RdxInLoop{};
2859 return RdxUnordered{/*VFScaleFactor=*/ScaleFactor};
2860}
2861
2862/// A recipe for handling reduction phis. The start value is the first operand
2863/// of the recipe and the incoming value from the backedge is the second
2864/// operand.
2866 /// The recurrence kind of the reduction.
2867 const RecurKind Kind;
2868
2869 ReductionStyle Style;
2870
2871 /// The phi is part of a multi-use reduction (e.g., used in FindIV
2872 /// patterns for argmin/argmax).
2873 /// TODO: Also support cases where the phi itself has a single use, but its
2874 /// compare has multiple uses.
2875 bool HasUsesOutsideReductionChain;
2876
2877 /// Temporary flag indicating that the FindIV reduction expression has been
2878 /// sunk. While this is true, epilogue vectorization is disabled to avoid
2879 /// applying the sunk expression twice (once in the main vector loop and again
2880 /// in the epilogue), which can produce incorrect results by applying the sunk
2881 /// operation twice.
2882 /// TODO: Remove this flag once epilogue vectorization properly supports
2883 /// sunk FindIV expressions.
2884 bool ExpressionSunk = false;
2885
2886public:
2887 /// Create a new VPReductionPHIRecipe for the reduction \p Phi.
2889 VPValue &BackedgeValue, ReductionStyle Style,
2890 const VPIRFlags &Flags,
2891 bool HasUsesOutsideReductionChain = false)
2892 : VPHeaderPHIRecipe(VPRecipeBase::VPReductionPHISC, Phi, &Start,
2893 Start.getScalarType()),
2894 VPIRFlags(Flags), Kind(Kind), Style(Style),
2895 HasUsesOutsideReductionChain(HasUsesOutsideReductionChain) {
2896 addOperand(&BackedgeValue);
2897 }
2898
2899 ~VPReductionPHIRecipe() override = default;
2900
2902 VPValue *BackedgeValue) {
2903 auto *Clone = new VPReductionPHIRecipe(
2905 *Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
2906 Clone->ExpressionSunk = ExpressionSunk;
2907 return Clone;
2908 }
2909
2913
2914 VP_CLASSOF_IMPL(VPRecipeBase::VPReductionPHISC)
2915
2916 /// Generate the phi/select nodes.
2917 void execute(VPTransformState &State) override;
2918
2919 /// Get the factor that the VF of this recipe's output should be scaled by, or
2920 /// 1 if it isn't scaled.
2921 unsigned getVFScaleFactor() const {
2922 auto *Partial = std::get_if<RdxUnordered>(&Style);
2923 return Partial ? Partial->VFScaleFactor : 1;
2924 }
2925
2926 /// Set the VFScaleFactor for this reduction phi. Can only be set to a factor
2927 /// > 1.
2928 void setVFScaleFactor(unsigned ScaleFactor) {
2929 assert(ScaleFactor > 1 && "must set to scale factor > 1");
2930 Style = RdxUnordered{ScaleFactor};
2931 }
2932
2933 /// Returns the recurrence kind of the reduction.
2934 RecurKind getRecurrenceKind() const { return Kind; }
2935
2936 /// Returns true, if the phi is part of an ordered reduction.
2937 bool isOrdered() const { return std::holds_alternative<RdxOrdered>(Style); }
2938
2939 /// Returns true if the phi is part of an in-loop reduction.
2940 bool isInLoop() const {
2941 return std::holds_alternative<RdxInLoop>(Style) ||
2942 std::holds_alternative<RdxOrdered>(Style);
2943 }
2944
2945 /// Returns true, if the phi is part of a multi-use reduction.
2947 return HasUsesOutsideReductionChain;
2948 }
2949
2950 void setExpressionSunk() { ExpressionSunk = true; }
2951
2952 bool isExpressionSunk() const { return ExpressionSunk; }
2953
2954 /// Returns true if the recipe only uses the first lane of operand \p Op.
2955 bool usesFirstLaneOnly(const VPValue *Op) const override {
2957 "Op must be an operand of the recipe");
2958 return isOrdered() || isInLoop();
2959 }
2960
2961protected:
2962#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2963 /// Print the recipe.
2964 void printRecipe(raw_ostream &O, const Twine &Indent,
2965 VPSlotTracker &SlotTracker) const override;
2966#endif
2967};
2968
2969/// A recipe for vectorizing a phi-node as a sequence of mask-based select
2970/// instructions.
2972public:
2973 /// The blend operation is a User of the incoming values and of their
2974 /// respective masks, ordered [I0, M0, I1, M1, I2, M2, ...]. Note that M0 can
2975 /// be omitted (implied by passing an odd number of operands) in which case
2976 /// all other incoming values are merged into it.
2978 const VPIRFlags &Flags, DebugLoc DL)
2980 Operands[0]->getScalarType(), Flags, DL) {
2981 assert(Operands.size() >= 2 && "Expected at least two operands!");
2983 [this](unsigned I) {
2984 return getIncomingValue(I)->getScalarType() ==
2985 getScalarType();
2986 }) &&
2987 "all incoming values must have the same type");
2989 [this](unsigned I) {
2990 return getMask(I)->getScalarType()->isIntegerTy(1);
2991 }) &&
2992 "masks must be a bool");
2993 assert(hasRequiredFlagsForOpcode(Instruction::PHI, getScalarType()) &&
2994 "blends require the flags of the phi they replace");
2995 setUnderlyingValue(Phi);
2996 }
2997
2999
3002 NewOperands, *this, getDebugLoc());
3003 }
3004
3005 VP_CLASSOF_IMPL(VPRecipeBase::VPBlendSC)
3006
3007 /// A normalized blend is one that has an odd number of operands, whereby the
3008 /// first operand does not have an associated mask.
3009 bool isNormalized() const { return getNumOperands() % 2; }
3010
3011 /// Return the number of incoming values, taking into account when normalized
3012 /// the first incoming value will have no mask.
3013 unsigned getNumIncomingValues() const {
3014 return (getNumOperands() + isNormalized()) / 2;
3015 }
3016
3017 /// Return incoming value number \p Idx.
3018 VPValue *getIncomingValue(unsigned Idx) const {
3019 return Idx == 0 ? getOperand(0) : getOperand(Idx * 2 - isNormalized());
3020 }
3021
3022 /// Return mask number \p Idx.
3023 VPValue *getMask(unsigned Idx) const {
3024 assert((Idx > 0 || !isNormalized()) && "First index has no mask!");
3025 return Idx == 0 ? getOperand(1) : getOperand(Idx * 2 + !isNormalized());
3026 }
3027
3028 /// Set mask number \p Idx to \p V.
3029 void setMask(unsigned Idx, VPValue *V) {
3030 assert((Idx > 0 || !isNormalized()) && "First index has no mask!");
3031 assert(V->getScalarType()->isIntegerTy(1) && "Mask must be an i1 (vector)");
3032 Idx == 0 ? setOperand(1, V) : setOperand(Idx * 2 + !isNormalized(), V);
3033 }
3034
3035 void execute(VPTransformState &State) override {
3036 llvm_unreachable("VPBlendRecipe should be expanded by simplifyBlends");
3037 }
3038
3039 /// Return the cost of this VPWidenMemoryRecipe.
3040 InstructionCost computeCost(ElementCount VF,
3041 VPCostContext &Ctx) const override;
3042
3043 /// Returns true if the recipe only uses the first lane of operand \p Op.
3044 bool usesFirstLaneOnly(const VPValue *Op) const override;
3045
3046protected:
3047#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3048 /// Print the recipe.
3049 void printRecipe(raw_ostream &O, const Twine &Indent,
3050 VPSlotTracker &SlotTracker) const override;
3051#endif
3052};
3053
3054/// A common base class for interleaved memory operations.
3055/// An Interleaved memory operation is a memory access method that combines
3056/// multiple strided loads/stores into a single wide load/store with shuffles.
3057/// The first operand is the start address. The optional operands are, in order,
3058/// the stored values and the mask.
3060 public VPIRMetadata {
3062
3063 /// Indicates if the interleave group is in a conditional block and requires a
3064 /// mask.
3065 bool HasMask = false;
3066
3067 /// Indicates if gaps between members of the group need to be masked out or if
3068 /// unusued gaps can be loaded speculatively.
3069 bool NeedsMaskForGaps = false;
3070
3071protected:
3073 ArrayRef<VPValue *> Operands,
3074 ArrayRef<VPValue *> StoredValues, VPValue *Mask,
3075 bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
3076 : VPRecipeBase(SC, Operands, DL), VPIRMetadata(MD), IG(IG),
3077 NeedsMaskForGaps(NeedsMaskForGaps) {
3078 // TODO: extend the masked interleaved-group support to reversed access.
3079 assert((!Mask || !IG->isReverse()) &&
3080 "Reversed masked interleave-group not supported.");
3081 if (StoredValues.empty()) {
3082 for (Instruction *Inst : IG->members()) {
3083 assert(!Inst->getType()->isVoidTy() && "must have result");
3084 new VPMultiDefValue(this, Inst, Inst->getType());
3085 }
3086 } else {
3087 for (auto *SV : StoredValues)
3088 addOperand(SV);
3089 }
3090 if (Mask) {
3091 HasMask = true;
3092 addOperand(Mask);
3093 }
3094 }
3095
3096public:
3097 VPInterleaveBase *clone() override = 0;
3098
3099 static inline bool classof(const VPRecipeBase *R) {
3100 return R->getVPRecipeID() == VPRecipeBase::VPInterleaveSC ||
3101 R->getVPRecipeID() == VPRecipeBase::VPInterleaveEVLSC;
3102 }
3103
3104 static inline bool classof(const VPUser *U) {
3105 auto *R = dyn_cast<VPRecipeBase>(U);
3106 return R && classof(R);
3107 }
3108
3109 /// Return the address accessed by this recipe.
3110 VPValue *getAddr() const {
3111 return getOperand(0); // Address is the 1st, mandatory operand.
3112 }
3113
3114 /// Return the mask used by this recipe. Note that a full mask is represented
3115 /// by a nullptr.
3116 VPValue *getMask() const {
3117 // Mask is optional and the last operand.
3118 return HasMask ? getLastOperand() : nullptr;
3119 }
3120
3121 /// Return true if the access needs a mask because of the gaps.
3122 bool needsMaskForGaps() const { return NeedsMaskForGaps; }
3123
3125
3126 Instruction *getInsertPos() const { return IG->getInsertPos(); }
3127
3128 void execute(VPTransformState &State) override {
3129 llvm_unreachable("VPInterleaveBase should not be instantiated.");
3130 }
3131
3132 /// Return the cost of this recipe.
3133 InstructionCost computeCost(ElementCount VF,
3134 VPCostContext &Ctx) const override;
3135
3136 /// Returns true if the recipe only uses the first lane of operand \p Op.
3137 bool usesFirstLaneOnly(const VPValue *Op) const override = 0;
3138
3139 /// Returns the number of stored operands of this interleave group. Returns 0
3140 /// for load interleave groups.
3141 virtual unsigned getNumStoreOperands() const = 0;
3142
3143 /// Return the VPValues stored by this interleave group. If it is a load
3144 /// interleave group, return an empty ArrayRef.
3146 return {op_end() - (getNumStoreOperands() + (HasMask ? 1 : 0)),
3148 }
3149};
3150
3151/// VPInterleaveRecipe is a recipe for transforming an interleave group of load
3152/// or stores into one wide load/store and shuffles. The first operand of a
3153/// VPInterleave recipe is the address, followed by the stored values, followed
3154/// by an optional mask.
3156public:
3158 ArrayRef<VPValue *> StoredValues, VPValue *Mask,
3159 bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
3160 : VPInterleaveBase(VPRecipeBase::VPInterleaveSC, IG, Addr, StoredValues,
3161 Mask, NeedsMaskForGaps, MD, DL) {}
3162
3163 ~VPInterleaveRecipe() override = default;
3164
3168 needsMaskForGaps(), *this, getDebugLoc());
3169 }
3170
3171 VP_CLASSOF_IMPL(VPRecipeBase::VPInterleaveSC)
3172
3173 /// Generate the wide load or store, and shuffles.
3174 void execute(VPTransformState &State) override;
3175
3176 bool usesFirstLaneOnly(const VPValue *Op) const override {
3178 "Op must be an operand of the recipe");
3179 return Op == getAddr() && !llvm::is_contained(getStoredValues(), Op);
3180 }
3181
3182 unsigned getNumStoreOperands() const override {
3183 return getNumOperands() - (getMask() ? 2 : 1);
3184 }
3185
3186protected:
3187#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3188 /// Print the recipe.
3189 void printRecipe(raw_ostream &O, const Twine &Indent,
3190 VPSlotTracker &SlotTracker) const override;
3191#endif
3192};
3193
3194/// A recipe for interleaved memory operations with vector-predication
3195/// intrinsics. The first operand is the address, the second operand is the
3196/// explicit vector length. Stored values and mask are optional operands.
3198public:
3200 : VPInterleaveBase(VPRecipeBase::VPInterleaveEVLSC,
3201 R.getInterleaveGroup(), {R.getAddr(), &EVL},
3202 R.getStoredValues(), Mask, R.needsMaskForGaps(), R,
3203 R.getDebugLoc()) {
3204 assert(!getInterleaveGroup()->isReverse() &&
3205 "Reversed interleave-group with tail folding is not supported.");
3206 assert(!needsMaskForGaps() && "Interleaved access with gap mask is not "
3207 "supported for scalable vector.");
3208 }
3209
3210 ~VPInterleaveEVLRecipe() override = default;
3211
3213 llvm_unreachable("cloning not implemented yet");
3214 }
3215
3216 VP_CLASSOF_IMPL(VPRecipeBase::VPInterleaveEVLSC)
3217
3218 /// The VPValue of the explicit vector length.
3219 VPValue *getEVL() const { return getOperand(1); }
3220
3221 /// Generate the wide load or store, and shuffles.
3222 void execute(VPTransformState &State) override;
3223
3224 /// The recipe only uses the first lane of the address, and EVL operand.
3225 bool usesFirstLaneOnly(const VPValue *Op) const override {
3227 "Op must be an operand of the recipe");
3228 return (Op == getAddr() && !llvm::is_contained(getStoredValues(), Op)) ||
3229 Op == getEVL();
3230 }
3231
3232 unsigned getNumStoreOperands() const override {
3233 return getNumOperands() - (getMask() ? 3 : 2);
3234 }
3235
3236protected:
3237#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3238 /// Print the recipe.
3239 void printRecipe(raw_ostream &O, const Twine &Indent,
3240 VPSlotTracker &SlotTracker) const override;
3241#endif
3242};
3243
3244/// A recipe to represent inloop, ordered or partial reduction operations. It
3245/// performs a reduction on a vector operand into a scalar (vector in the case
3246/// of a partial reduction) value, and adds the result to a chain. The Operands
3247/// are {ChainOp, VecOp, [Condition]}.
3249
3250 /// The recurrence kind for the reduction in question.
3251 RecurKind RdxKind;
3252 /// Whether the reduction is conditional.
3253 bool IsConditional = false;
3254 ReductionStyle Style;
3255
3256protected:
3259 VPValue *CondOp, ReductionStyle Style, DebugLoc DL)
3261 DL),
3262 RdxKind(RdxKind), Style(Style) {
3264 [this](VPValue *VPV) {
3265 return VPV->getScalarType() == getScalarType() ||
3266 (isa<VPInstruction>(VPV) &&
3267 cast<VPInstruction>(VPV)->getOpcode() ==
3269 }) &&
3270 "all incoming values must have the same type");
3271 if (CondOp) {
3272 assert(CondOp->getScalarType()->isIntegerTy(1) &&
3273 "CondOp must be a bool");
3274 IsConditional = true;
3275 addOperand(CondOp);
3276 }
3278 }
3279
3280public:
3282 VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp,
3284 : VPReductionRecipe(VPRecipeBase::VPReductionSC, RdxKind, FMFs, I,
3285 {ChainOp, VecOp}, CondOp, Style, DL) {}
3286
3288 VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp,
3290 : VPReductionRecipe(VPRecipeBase::VPReductionSC, RdxKind, FMFs, nullptr,
3291 {ChainOp, VecOp}, CondOp, Style, DL) {}
3292
3293 ~VPReductionRecipe() override = default;
3294
3296 return new VPReductionRecipe(RdxKind, getFastMathFlagsOrNone(),
3298 getCondOp(), Style, getDebugLoc());
3299 }
3300
3301 static inline bool classof(const VPRecipeBase *R) {
3302 return R->getVPRecipeID() == VPRecipeBase::VPReductionSC ||
3303 R->getVPRecipeID() == VPRecipeBase::VPReductionEVLSC;
3304 }
3305
3306 static inline bool classof(const VPUser *U) {
3307 auto *R = dyn_cast<VPRecipeBase>(U);
3308 return R && classof(R);
3309 }
3310
3311 static inline bool classof(const VPValue *VPV) {
3312 const VPRecipeBase *R = VPV->getDefiningRecipe();
3313 return R && classof(R);
3314 }
3315
3316 static inline bool classof(const VPSingleDefRecipe *R) {
3317 return classof(static_cast<const VPRecipeBase *>(R));
3318 }
3319
3320 /// Generate the reduction in the loop.
3321 void execute(VPTransformState &State) override;
3322
3323 /// Return the cost of VPReductionRecipe.
3324 InstructionCost computeCost(ElementCount VF,
3325 VPCostContext &Ctx) const override;
3326
3327 /// Return the recurrence kind for the in-loop reduction.
3328 RecurKind getRecurrenceKind() const { return RdxKind; }
3329 /// Return true if the in-loop reduction is ordered.
3330 bool isOrdered() const { return std::holds_alternative<RdxOrdered>(Style); };
3331 /// Return true if the in-loop reduction is conditional.
3332 bool isConditional() const { return IsConditional; };
3333 /// Returns true if the reduction outputs a vector with a scaled down VF.
3334 bool isPartialReduction() const {
3335 return std::holds_alternative<RdxUnordered>(Style);
3336 }
3337 /// Returns true if the reduction is in-loop.
3338 bool isInLoop() const {
3339 return std::holds_alternative<RdxInLoop>(Style) ||
3340 std::holds_alternative<RdxOrdered>(Style);
3341 }
3342 /// The VPValue of the scalar Chain being accumulated.
3343 VPValue *getChainOp() const { return getOperand(0); }
3344 /// The VPValue of the vector value to be reduced.
3345 VPValue *getVecOp() const { return getOperand(1); }
3346 /// The VPValue of the condition for the block.
3348 return isConditional() ? getLastOperand() : nullptr;
3349 }
3350 /// Get the factor that the VF of this recipe's output should be scaled by, or
3351 /// 1 if it isn't scaled.
3352 unsigned getVFScaleFactor() const {
3353 auto *Partial = std::get_if<RdxUnordered>(&Style);
3354 return Partial ? Partial->VFScaleFactor : 1;
3355 }
3356
3357protected:
3358#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3359 /// Print the recipe.
3360 void printRecipe(raw_ostream &O, const Twine &Indent,
3361 VPSlotTracker &SlotTracker) const override;
3362#endif
3363};
3364
3365/// A recipe to represent inloop reduction operations with vector-predication
3366/// intrinsics, performing a reduction on a vector operand with the explicit
3367/// vector length (EVL) into a scalar value, and adding the result to a chain.
3368/// The Operands are {ChainOp, VecOp, EVL, [Condition]}.
3370public:
3373 : VPReductionRecipe(VPRecipeBase::VPReductionEVLSC, R.getRecurrenceKind(),
3376 {R.getChainOp(), R.getVecOp(), &EVL}, CondOp,
3377 getReductionStyle(R.isInLoop(), R.isOrdered(),
3378 R.getVFScaleFactor()),
3379 DL) {}
3380
3381 ~VPReductionEVLRecipe() override = default;
3382
3384 llvm_unreachable("cloning not implemented yet");
3385 }
3386
3387 VP_CLASSOF_IMPL(VPRecipeBase::VPReductionEVLSC)
3388
3389 /// Generate the reduction in the loop
3390 void execute(VPTransformState &State) override;
3391
3392 /// The VPValue of the explicit vector length.
3393 VPValue *getEVL() const { return getOperand(2); }
3394
3395 /// Returns true if the recipe only uses the first lane of operand \p Op.
3396 bool usesFirstLaneOnly(const VPValue *Op) const override {
3398 "Op must be an operand of the recipe");
3399 return Op == getEVL();
3400 }
3401
3402protected:
3403#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3404 /// Print the recipe.
3405 void printRecipe(raw_ostream &O, const Twine &Indent,
3406 VPSlotTracker &SlotTracker) const override;
3407#endif
3408};
3409
3410/// VPReplicateRecipe replicates a given instruction producing multiple scalar
3411/// copies of the original scalar type, one per lane, instead of producing a
3412/// single copy of widened type for all lanes. If the instruction is known to be
3413/// a single scalar, only one copy will be generated.
3415 public VPIRMetadata {
3416 /// Indicator if only a single replica per lane is needed.
3417 bool IsSingleScalar;
3418
3419 /// Indicator if the replicas are also predicated.
3420 bool IsPredicated;
3421
3422public:
3424 bool IsSingleScalar, VPValue *Mask = nullptr,
3425 const VPIRFlags &Flags = {}, VPIRMetadata Metadata = {},
3426 DebugLoc DL = DebugLoc::getUnknown())
3427 : VPRecipeWithIRFlags(VPRecipeBase::VPReplicateSC, Operands,
3428 computeScalarType(I, Operands), Flags, DL),
3429 VPIRMetadata(Metadata), IsSingleScalar(IsSingleScalar),
3430 IsPredicated(Mask) {
3431 assert((!IsSingleScalar || !I->isCast()) &&
3432 "Single-scalar casts should use VPInstruction");
3433 setUnderlyingValue(I);
3434 if (Mask)
3435 addOperand(Mask);
3436 }
3437
3438 ~VPReplicateRecipe() override = default;
3439
3440 /// Compute the scalar result type for a VPReplicateRecipe wrapping \p I with
3441 /// \p Operands (excluding any predicate mask).
3442 static Type *computeScalarType(const Instruction *I,
3444
3446
3448 auto *Copy = new VPReplicateRecipe(
3449 getUnderlyingInstr(), NewOperands, IsSingleScalar,
3450 isPredicated() ? getMask() : nullptr, *this, *this, getDebugLoc());
3451 Copy->transferFlags(*this);
3452 return Copy;
3453 }
3454
3455 VP_CLASSOF_IMPL(VPRecipeBase::VPReplicateSC)
3456
3457 /// Generate replicas of the desired Ingredient. Replicas will be generated
3458 /// for all parts and lanes unless a specific part and lane are specified in
3459 /// the \p State.
3460 void execute(VPTransformState &State) override;
3461
3462 /// Return the cost of this VPReplicateRecipe.
3463 InstructionCost computeCost(ElementCount VF,
3464 VPCostContext &Ctx) const override;
3465
3466 /// Return the cost of scalarizing a call to \p CalledFn with argument
3467 /// operands \p ArgOps for a given \p VF.
3468 static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy,
3470 bool IsSingleScalar, ElementCount VF,
3471 VPCostContext &Ctx);
3472
3473 /// Returns true if the recipe produces a single scalar value.
3474 bool isSingleScalar() const { return IsSingleScalar; }
3475
3476 /// Returns true if the recipe produces scalar values for all VF lanes.
3477 bool doesGeneratePerAllLanes() const { return !IsSingleScalar; }
3478
3479 bool isPredicated() const { return IsPredicated; }
3480
3481 /// Returns true if the recipe only uses the first lane of operand \p Op.
3482 bool usesFirstLaneOnly(const VPValue *Op) const override {
3484 "Op must be an operand of the recipe");
3485 return isSingleScalar();
3486 }
3487
3488 /// Returns true if the recipe uses scalars of operand \p Op.
3489 bool usesScalars(const VPValue *Op) const override {
3491 "Op must be an operand of the recipe");
3492 return true;
3493 }
3494
3495 /// Return the mask of a predicated VPReplicateRecipe.
3497 assert(isPredicated() && "Trying to get the mask of a unpredicated recipe");
3498 return getLastOperand();
3499 }
3500
3501 /// Return the recipe's operands, excluding the mask of a predicated recipe.
3505
3506 /// Returns the number of operands, excluding the mask if the recipe is
3507 /// predicated.
3508 unsigned getNumOperandsWithoutMask() const {
3509 return getNumOperands() - isPredicated();
3510 }
3511
3512 unsigned getOpcode() const { return getUnderlyingInstr()->getOpcode(); }
3513
3514protected:
3515#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3516 /// Print the recipe.
3517 void printRecipe(raw_ostream &O, const Twine &Indent,
3518 VPSlotTracker &SlotTracker) const override;
3519#endif
3520};
3521
3522/// A recipe for generating conditional branches on the bits of a mask.
3524 public VPIRMetadata {
3525public:
3527 const VPIRMetadata &Metadata = {})
3528 : VPRecipeBase(VPRecipeBase::VPBranchOnMaskSC, {BlockInMask}, DL),
3529 VPIRMetadata(Metadata) {}
3530
3532 return new VPBranchOnMaskRecipe(getOperand(0), getDebugLoc(), *this);
3533 }
3534
3535 VP_CLASSOF_IMPL(VPRecipeBase::VPBranchOnMaskSC)
3536
3537 /// Generate the extraction of the appropriate bit from the block mask and the
3538 /// conditional branch.
3539 void execute(VPTransformState &State) override;
3540
3541 /// Return the cost of this VPBranchOnMaskRecipe.
3542 InstructionCost computeCost(ElementCount VF,
3543 VPCostContext &Ctx) const override;
3544
3545#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3546 /// Print the recipe.
3547 void printRecipe(raw_ostream &O, const Twine &Indent,
3548 VPSlotTracker &SlotTracker) const override {
3549 O << Indent << "BRANCH-ON-MASK ";
3551 }
3552#endif
3553
3554 /// Returns true if the recipe uses scalars of operand \p Op.
3555 bool usesScalars(const VPValue *Op) const override {
3557 "Op must be an operand of the recipe");
3558 return true;
3559 }
3560};
3561
3562/// A recipe to combine multiple recipes into a single 'expression' recipe,
3563/// which should be considered a single entity for cost-modeling and transforms.
3564/// The recipe needs to be 'decomposed', i.e. replaced by its individual
3565/// expression recipes, before execute. The individual expression recipes are
3566/// completely disconnected from the def-use graph of other recipes not part of
3567/// the expression. Def-use edges between pairs of expression recipes remain
3568/// intact, whereas every edge between an expression recipe and a recipe outside
3569/// the expression is elevated to connect the non-expression recipe with the
3570/// VPExpressionRecipe itself.
3572 /// Recipes included in this VPExpressionRecipe. This could contain
3573 /// duplicates.
3574 SmallVector<VPSingleDefRecipe *> ExpressionRecipes;
3575
3576 /// Temporary VPValues used for external operands of the expression, i.e.
3577 /// operands not defined by recipes in the expression.
3578 SmallVector<VPValue *> LiveInPlaceholders;
3579
3580 enum class ExpressionTypes {
3581 /// Represents an inloop extended reduction operation, performing a
3582 /// reduction on an extended vector operand into a scalar value, and adding
3583 /// the result to a chain.
3584 ExtendedReduction,
3585 /// Represents an inloop extended reduction operation, which is negated,
3586 /// then reduced before adding the result to a chain.
3587 NegatedExtendedReduction,
3588 /// Represent an inloop multiply-accumulate reduction, multiplying the
3589 /// extended vector operands, performing a reduction.add on the result, and
3590 /// adding the scalar result to a chain.
3591 ExtMulAccReduction,
3592 /// Represent an inloop multiply-accumulate reduction, multiplying the
3593 /// vector operands, performing a reduction.add on the result, and adding
3594 /// the scalar result to a chain.
3595 MulAccReduction,
3596 /// Represent an inloop multiply-accumulate reduction, multiplying the
3597 /// extended vector operands, negating the multiplication, performing a
3598 /// reduction.add on the result, and adding the scalar result to a chain.
3599 ExtNegatedMulAccReduction,
3600 };
3601
3602 /// Type of the expression.
3603 ExpressionTypes ExpressionType;
3604
3605public:
3606 /// Construct a new VPExpressionRecipe by internalizing recipes in \p
3607 /// ExpressionRecipes. External operands (i.e. not defined by another recipe
3608 /// in the expression) are replaced by temporary VPValues and the original
3609 /// operands are transferred to the VPExpressionRecipe itself. Clone recipes
3610 /// as needed (excluding last) to ensure they are only used by other recipes
3611 /// in the expression.
3612 VPExpressionRecipe(ExpressionTypes ExpressionType,
3613 ArrayRef<VPSingleDefRecipe *> ExpressionRecipes);
3614
3616 : VPExpressionRecipe(ExpressionTypes::ExtendedReduction, {Ext, Red}) {}
3618 VPReductionRecipe *Red)
3619 : VPExpressionRecipe(ExpressionTypes::NegatedExtendedReduction,
3620 {Ext, Neg, Red}) {
3621 assert((Red->getRecurrenceKind() == RecurKind::Add ||
3622 Red->getRecurrenceKind() == RecurKind::FAdd ||
3623 Red->getRecurrenceKind() == RecurKind::AddChainWithSubs) &&
3624 "Expected an add or add-chain-with-subs reduction");
3625 if (Neg->getOpcode() == Instruction::Sub) {
3626 [[maybe_unused]] auto *SubConst = dyn_cast<VPConstantInt>(getOperand(1));
3627 assert(SubConst && SubConst->isZero() && "Expected a negating sub");
3628 } else
3629 assert(Neg->getOpcode() == Instruction::FNeg && "Unexpected opcode");
3630 }
3632 : VPExpressionRecipe(ExpressionTypes::MulAccReduction, {Mul, Red}) {}
3635 : VPExpressionRecipe(ExpressionTypes::ExtMulAccReduction,
3636 {Ext0, Ext1, Mul, Red}) {}
3639 VPReductionRecipe *Red)
3640 : VPExpressionRecipe(ExpressionTypes::ExtNegatedMulAccReduction,
3641 {Ext0, Ext1, Mul, Neg, Red}) {
3642 assert((Mul->getOpcode() == Instruction::Mul ||
3643 Mul->getOpcode() == Instruction::FMul) &&
3644 "Expected a mul");
3645 assert((Red->getRecurrenceKind() == RecurKind::Add ||
3646 Red->getRecurrenceKind() == RecurKind::FAdd ||
3647 Red->getRecurrenceKind() == RecurKind::AddChainWithSubs) &&
3648 "Expected an add or add-chain-with-subs reduction");
3649 assert(getNumOperands() >= 3 && "Expected at least three operands");
3650 if (Neg->getOpcode() == Instruction::Sub) {
3651 [[maybe_unused]] auto *SubConst = dyn_cast<VPConstantInt>(getOperand(2));
3652 assert(SubConst && SubConst->isZero() &&
3653 Neg->getOpcode() == Instruction::Sub && "Expected a negating sub");
3654 } else
3655 assert(Neg->getOpcode() == Instruction::FNeg && "Unexpected opcode");
3656 }
3657
3659 SmallPtrSet<VPSingleDefRecipe *, 4> ExpressionRecipesSeen;
3660 for (auto *R : reverse(ExpressionRecipes)) {
3661 if (ExpressionRecipesSeen.insert(R).second)
3662 delete R;
3663 }
3664 for (VPValue *T : LiveInPlaceholders)
3665 delete T;
3666 }
3667
3668 VP_CLASSOF_IMPL(VPRecipeBase::VPExpressionSC)
3669
3671 assert(!ExpressionRecipes.empty() && "empty expressions should be removed");
3672 SmallVector<VPSingleDefRecipe *> NewExpressiondRecipes;
3673 for (auto *R : ExpressionRecipes)
3674 NewExpressiondRecipes.push_back(R->clone());
3675 for (auto *New : NewExpressiondRecipes) {
3676 for (const auto &[Idx, Old] : enumerate(ExpressionRecipes))
3677 New->replaceUsesOfWith(Old, NewExpressiondRecipes[Idx]);
3678 // Update placeholder operands in the cloned recipe to use the external
3679 // operands, to be internalized when the cloned expression is constructed.
3680 for (const auto &[Placeholder, OutsideOp] :
3681 zip(LiveInPlaceholders, operands()))
3682 New->replaceUsesOfWith(Placeholder, OutsideOp);
3683 }
3684 return new VPExpressionRecipe(ExpressionType, NewExpressiondRecipes);
3685 }
3686
3687 /// Return and insert the recipes of the expression back into the VPlan,
3688 /// directly before the current recipe. Leaves the expression recipe empty,
3689 /// which must be removed before codegen.
3691
3692 /// Returns the expression type of this recipe.
3693 ExpressionTypes getExpressionType() const { return ExpressionType; }
3694
3695 unsigned getVFScaleFactor() const {
3696 auto *PR = dyn_cast<VPReductionRecipe>(ExpressionRecipes.back());
3697 return PR ? PR->getVFScaleFactor() : 1;
3698 }
3699
3700 /// Method for generating code, must not be called as this recipe is abstract.
3701 void execute(VPTransformState &State) override {
3702 llvm_unreachable("recipe must be removed before execute");
3703 }
3704
3706 VPCostContext &Ctx) const override;
3707
3708 /// Returns true if this expression contains recipes that may read from or
3709 /// write to memory.
3710 bool mayReadOrWriteMemory() const;
3711
3712 /// Returns true if this expression contains recipes that may have side
3713 /// effects.
3714 bool mayHaveSideEffects() const;
3715
3716 /// Returns true if this VPExpressionRecipe produces a single scalar.
3717 bool isVectorToScalar() const;
3718
3719protected:
3720#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3721 /// Print the recipe.
3722 void printRecipe(raw_ostream &O, const Twine &Indent,
3723 VPSlotTracker &SlotTracker) const override;
3724#endif
3725};
3726
3727/// VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when
3728/// control converges back from a Branch-on-Mask. The phi nodes are needed in
3729/// order to merge values that are set under such a branch and feed their uses.
3730/// The phi nodes can be scalar or vector depending on the users of the value.
3731/// This recipe works in concert with VPBranchOnMaskRecipe.
3733public:
3734 /// Construct a VPPredInstPHIRecipe given \p PredInst whose value needs a phi
3735 /// nodes after merging back from a Branch-on-Mask.
3737 : VPSingleDefRecipe(VPRecipeBase::VPPredInstPHISC, PredV,
3738 PredV->getScalarType(), /*UV=*/nullptr, DL) {}
3739 ~VPPredInstPHIRecipe() override = default;
3740
3742 return new VPPredInstPHIRecipe(getOperand(0), getDebugLoc());
3743 }
3744
3745 VP_CLASSOF_IMPL(VPRecipeBase::VPPredInstPHISC)
3746
3747 /// Generates phi nodes for live-outs (from a replicate region) as needed to
3748 /// retain SSA form.
3749 void execute(VPTransformState &State) override;
3750
3751 /// Return the cost of this VPPredInstPHIRecipe.
3753 VPCostContext &Ctx) const override {
3754 // TODO: Compute accurate cost after retiring the legacy cost model.
3755 return 0;
3756 }
3757
3758protected:
3759#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3760 /// Print the recipe.
3761 void printRecipe(raw_ostream &O, const Twine &Indent,
3762 VPSlotTracker &SlotTracker) const override;
3763#endif
3764};
3765
3766/// A common mixin class for widening memory operations. An optional mask can be
3767/// provided as the last operand.
3769protected:
3771
3772 /// Alignment information for this memory access.
3774
3775 /// Whether the accessed addresses are consecutive.
3777
3778 /// Whether the memory access is masked.
3779 bool IsMasked = false;
3780
3781 void setMask(VPValue *Mask) {
3782 assert(!IsMasked && "cannot re-set mask");
3783 if (!Mask)
3784 return;
3785 assert(Mask->getScalarType()->isIntegerTy(1) &&
3786 "Mask must be an i1 (vector)");
3787 getAsRecipe()->addOperand(Mask);
3788 IsMasked = true;
3789 }
3790
3795
3796public:
3797 virtual ~VPWidenMemoryRecipe() = default;
3798
3799 /// Return a VPRecipeBase* to the current object.
3801 virtual const VPRecipeBase *getAsRecipe() const = 0;
3802
3803 /// Return whether the loaded-from / stored-to addresses are consecutive.
3804 bool isConsecutive() const { return Consecutive; }
3805
3806 /// Return the address accessed by this recipe.
3807 VPValue *getAddr() const { return getAsRecipe()->getOperand(0); }
3808
3809 /// Returns true if the recipe is masked.
3810 bool isMasked() const { return IsMasked; }
3811
3812 /// Return the mask used by this recipe. Note that a full mask is represented
3813 /// by a nullptr.
3814 VPValue *getMask() const {
3815 // Mask is optional and therefore the last operand.
3816 const VPRecipeBase *R = getAsRecipe();
3817 return isMasked() ? R->getLastOperand() : nullptr;
3818 }
3819
3820 /// Returns the alignment of the memory access.
3821 Align getAlign() const { return Alignment; }
3822
3823 /// Return the cost of this VPWidenMemoryRecipe.
3824 InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
3825
3827};
3828
3829/// A recipe for widening load operations, using the address to load from and an
3830/// optional mask.
3832 public VPWidenMemoryRecipe {
3834 bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
3835 : VPSingleDefRecipe(VPRecipeBase::VPWidenLoadSC, {Addr}, Load.getType(),
3836 &Load, DL),
3837 VPWidenMemoryRecipe(Load, Consecutive, Metadata) {
3838 setMask(Mask);
3839 }
3840
3843 getMask(), Consecutive, *this, getDebugLoc());
3844 }
3845
3846 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
3847
3848 /// Returns the opcode of the widened load.
3849 unsigned getOpcode() const { return Instruction::Load; }
3850
3851 /// Generate a wide load or gather.
3852 void execute(VPTransformState &State) override;
3853
3854 /// Return the cost of this VPWidenLoadRecipe.
3856 VPCostContext &Ctx) const override {
3857 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
3858 }
3859
3860 /// Returns true if the recipe only uses the first lane of operand \p Op.
3861 bool usesFirstLaneOnly(const VPValue *Op) const override {
3863 "Op must be an operand of the recipe");
3864 // Widened, consecutive loads operations only demand the first lane of
3865 // their address.
3866 return Op == getAddr() && isConsecutive();
3867 }
3868
3869protected:
3870 VPRecipeBase *getAsRecipe() override;
3871 const VPRecipeBase *getAsRecipe() const override;
3872
3873#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3874 /// Print the recipe.
3875 void printRecipe(raw_ostream &O, const Twine &Indent,
3876 VPSlotTracker &SlotTracker) const override;
3877#endif
3878};
3879
3880/// A recipe for widening load operations with vector-predication intrinsics,
3881/// using the address to load from, the explicit vector length and an optional
3882/// mask.
3884 : public VPSingleDefRecipe,
3885 public VPWidenMemoryRecipe {
3887 VPValue *Mask)
3888 : VPSingleDefRecipe(VPRecipeBase::VPWidenLoadEVLSC, {Addr, &EVL},
3889 L.getIngredient().getType(), &L.getIngredient(),
3890 L.getDebugLoc()),
3891 VPWidenMemoryRecipe(L.getIngredient(), L.isConsecutive(), L) {
3892 setMask(Mask);
3893 }
3894
3896 llvm_unreachable("cloning not supported");
3897 }
3898
3899 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadEVLSC)
3900
3901 /// Returns the opcode of the widened load.
3902 unsigned getOpcode() const { return Instruction::Load; }
3903
3904 /// Return the EVL operand.
3905 VPValue *getEVL() const { return getOperand(1); }
3906
3907 /// Generate the wide load or gather.
3908 void execute(VPTransformState &State) override;
3909
3910 /// Return the cost of this VPWidenLoadEVLRecipe.
3911 InstructionCost computeCost(ElementCount VF,
3912 VPCostContext &Ctx) const override;
3913
3914 /// Returns true if the recipe only uses the first lane of operand \p Op.
3915 bool usesFirstLaneOnly(const VPValue *Op) const override {
3917 "Op must be an operand of the recipe");
3918 // Widened loads only demand the first lane of EVL and consecutive loads
3919 // only demand the first lane of their address.
3920 return Op == getEVL() || (Op == getAddr() && isConsecutive());
3921 }
3922
3923protected:
3924 VPRecipeBase *getAsRecipe() override;
3925 const VPRecipeBase *getAsRecipe() const override;
3926
3927#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3928 /// Print the recipe.
3929 void printRecipe(raw_ostream &O, const Twine &Indent,
3930 VPSlotTracker &SlotTracker) const override;
3931#endif
3932};
3933
3934/// A recipe for widening store operations, using the stored value, the address
3935/// to store to and an optional mask.
3937 public VPWidenMemoryRecipe {
3939 VPValue *Mask, bool Consecutive,
3940 const VPIRMetadata &Metadata, DebugLoc DL)
3941 : VPRecipeBase(VPRecipeBase::VPWidenStoreSC, {Addr, StoredVal}, DL),
3942 VPWidenMemoryRecipe(Store, Consecutive, Metadata) {
3943 setMask(Mask);
3944 }
3945
3949 *this, getDebugLoc());
3950 }
3951
3952 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
3953
3954 /// Return the value stored by this recipe.
3955 VPValue *getStoredValue() const { return getOperand(1); }
3956
3957 /// Generate a wide store or scatter.
3958 void execute(VPTransformState &State) override;
3959
3960 /// Return the cost of this VPWidenStoreRecipe.
3962 VPCostContext &Ctx) const override {
3963 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
3964 }
3965
3966 /// Returns true if the recipe only uses the first lane of operand \p Op.
3967 bool usesFirstLaneOnly(const VPValue *Op) const override {
3969 "Op must be an operand of the recipe");
3970 // Widened, consecutive stores only demand the first lane of their address,
3971 // unless the same operand is also stored.
3972 return Op == getAddr() && isConsecutive() && Op != getStoredValue();
3973 }
3974
3975protected:
3976 VPRecipeBase *getAsRecipe() override;
3977 const VPRecipeBase *getAsRecipe() const override;
3978
3979#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3980 /// Print the recipe.
3981 void printRecipe(raw_ostream &O, const Twine &Indent,
3982 VPSlotTracker &SlotTracker) const override;
3983#endif
3984};
3985
3986/// A recipe for widening store operations with vector-predication intrinsics,
3987/// using the value to store, the address to store to, the explicit vector
3988/// length and an optional mask.
3990 : public VPRecipeBase,
3991 public VPWidenMemoryRecipe {
3993 VPValue *StoredVal, VPValue &EVL, VPValue *Mask)
3994 : VPRecipeBase(VPRecipeBase::VPWidenStoreEVLSC, {Addr, StoredVal, &EVL},
3995 S.getDebugLoc()),
3996 VPWidenMemoryRecipe(S.getIngredient(), S.isConsecutive(), S) {
3997 setMask(Mask);
3998 }
3999
4001 llvm_unreachable("cloning not supported");
4002 }
4003
4004 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreEVLSC)
4005
4006 /// Return the address accessed by this recipe.
4007 VPValue *getStoredValue() const { return getOperand(1); }
4008
4009 /// Return the EVL operand.
4010 VPValue *getEVL() const { return getOperand(2); }
4011
4012 /// Generate the wide store or scatter.
4013 void execute(VPTransformState &State) override;
4014
4015 /// Return the cost of this VPWidenStoreEVLRecipe.
4016 InstructionCost computeCost(ElementCount VF,
4017 VPCostContext &Ctx) const override;
4018
4019 /// Returns true if the recipe only uses the first lane of operand \p Op.
4020 bool usesFirstLaneOnly(const VPValue *Op) const override {
4022 "Op must be an operand of the recipe");
4023 if (Op == getEVL()) {
4024 assert(getStoredValue() != Op && "unexpected store of EVL");
4025 return true;
4026 }
4027 // Widened, consecutive memory operations only demand the first lane of
4028 // their address, unless the same operand is also stored. That latter can
4029 // happen with opaque pointers.
4030 return Op == getAddr() && isConsecutive() && Op != getStoredValue();
4031 }
4032
4033protected:
4034 VPRecipeBase *getAsRecipe() override;
4035 const VPRecipeBase *getAsRecipe() const override;
4036
4037#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4038 /// Print the recipe.
4039 void printRecipe(raw_ostream &O, const Twine &Indent,
4040 VPSlotTracker &SlotTracker) const override;
4041#endif
4042};
4043
4044/// Recipe to expand a SCEV expression.
4046 const SCEV *Expr;
4047
4048public:
4049 VPExpandSCEVRecipe(const SCEV *Expr);
4050
4051 ~VPExpandSCEVRecipe() override = default;
4052
4053 VPExpandSCEVRecipe *clone() override { return new VPExpandSCEVRecipe(Expr); }
4054
4055 VP_CLASSOF_IMPL(VPRecipeBase::VPExpandSCEVSC)
4056
4057 void execute(VPTransformState &State) override {
4058 llvm_unreachable("SCEV expressions must be expanded before final execute");
4059 }
4060
4061 /// Return the cost of this VPExpandSCEVRecipe.
4063 VPCostContext &Ctx) const override {
4064 // TODO: Compute accurate cost after retiring the legacy cost model.
4065 return 0;
4066 }
4067
4068 const SCEV *getSCEV() const { return Expr; }
4069
4070protected:
4071#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4072 /// Print the recipe.
4073 void printRecipe(raw_ostream &O, const Twine &Indent,
4074 VPSlotTracker &SlotTracker) const override;
4075#endif
4076};
4077
4078/// A recipe for generating the active lane mask for the vector loop that is
4079/// used to predicate the vector operations.
4081public:
4083 : VPHeaderPHIRecipe(VPRecipeBase::VPActiveLaneMaskPHISC, nullptr,
4084 StartMask, StartMask->getScalarType(), DL) {}
4085
4086 ~VPActiveLaneMaskPHIRecipe() override = default;
4087
4090 if (getNumOperands() == 2)
4091 R->addBackedgeValue(getOperand(1));
4092 return R;
4093 }
4094
4095 VP_CLASSOF_IMPL(VPRecipeBase::VPActiveLaneMaskPHISC)
4096
4097 /// Generate the active lane mask phi of the vector loop.
4098 void execute(VPTransformState &State) override;
4099
4100protected:
4101#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4102 /// Print the recipe.
4103 void printRecipe(raw_ostream &O, const Twine &Indent,
4104 VPSlotTracker &SlotTracker) const override;
4105#endif
4106};
4107
4108/// A recipe for generating the phi node tracking the current scalar iteration
4109/// index. It starts at the start value of the canonical induction and gets
4110/// incremented by the number of scalar iterations processed by the vector loop
4111/// iteration. The increment does not have to be loop invariant.
4113public:
4115 : VPHeaderPHIRecipe(VPRecipeBase::VPCurrentIterationPHISC, nullptr,
4116 StartIV, StartIV->getScalarType(), DL) {}
4117
4118 ~VPCurrentIterationPHIRecipe() override = default;
4119
4121 llvm_unreachable("cloning not implemented yet");
4122 }
4123
4124 VP_CLASSOF_IMPL(VPRecipeBase::VPCurrentIterationPHISC)
4125
4126 void execute(VPTransformState &State) override {
4127 llvm_unreachable("cannot execute this recipe, should be replaced by a "
4128 "scalar phi recipe");
4129 }
4130
4131 /// Return the cost of this VPCurrentIterationPHIRecipe.
4133 VPCostContext &Ctx) const override {
4134 // For now, match the behavior of the legacy cost model.
4135 return 0;
4136 }
4137
4138 /// Returns true if the recipe only uses the first lane of operand \p Op.
4139 bool usesFirstLaneOnly(const VPValue *Op) const override {
4141 "Op must be an operand of the recipe");
4142 return true;
4143 }
4144
4145protected:
4146#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4147 /// Print the recipe.
4148 LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent,
4149 VPSlotTracker &SlotTracker) const override;
4150#endif
4151};
4152
4153/// A Recipe for widening the canonical induction variable of the vector loop.
4154/// First operand is the canonical IV recipe, a second step operand (VF * Part)
4155/// is added during unrolling.
4157public:
4159 const VPIRFlags::WrapFlagsTy &Flags = {})
4160 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCanonicalIVSC, CanonicalIV,
4161 CanonicalIV->getType(), Flags) {}
4162
4163 ~VPWidenCanonicalIVRecipe() override = default;
4164
4166 auto *WideCanIV =
4168 if (VPValue *Step = getStepValue())
4169 WideCanIV->addPerPartStep(Step);
4170 return WideCanIV;
4171 }
4172
4173 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCanonicalIVSC)
4174
4175 void execute(VPTransformState &State) override {
4176 llvm_unreachable("Expected prior expansion of WidenCanonicalIV recipes");
4177 }
4178
4179 /// Return the cost of this VPWidenCanonicalIVPHIRecipe.
4181 VPCostContext &Ctx) const override {
4182 // TODO: Compute accurate cost after retiring the legacy cost model.
4183 return 0;
4184 }
4185
4186 /// Return the canonical IV being widened.
4190
4192 return getNumOperands() == 2 ? getOperand(1) : nullptr;
4193 }
4194
4195 /// Add the per-part step (VF * Part) used for unrolled parts.
4197 assert(Step->getScalarType() == getScalarType() &&
4198 "per-part step must have the same type as the canonical IV");
4199 VPUser::addOperand(Step);
4200 }
4201
4202protected:
4203#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4204 /// Print the recipe.
4205 void printRecipe(raw_ostream &O, const Twine &Indent,
4206 VPSlotTracker &SlotTracker) const override;
4207#endif
4208};
4209
4210/// A recipe for converting \p Current into \p Start + \p Current * \p Step.
4211/// FastMathFlags are derived from the \p FPBinOp in the case of FP inductions,
4212/// and the passed NoWrap \p Flags apply in the case of Ptr and Int inductions.
4214 /// Kind of the induction.
4216 /// If not nullptr, the floating point induction binary operator. Must be set
4217 /// for floating point inductions.
4218 const FPMathOperator *FPBinOp;
4219
4220public:
4222 const FPMathOperator *FPBinOp, VPValue *Start,
4223 VPValue *Current, VPValue *Step,
4224 const VPIRFlags::WrapFlagsTy &Flags = {})
4225 : VPRecipeWithIRFlags(VPRecipeBase::VPDerivedIVSC, {Start, Current, Step},
4226 Start->getScalarType(), Flags),
4227 Kind(Kind), FPBinOp(FPBinOp) {}
4228
4229 ~VPDerivedIVRecipe() override = default;
4230
4232 return new VPDerivedIVRecipe(Kind, FPBinOp, getStartValue(), getOperand(1),
4234 }
4235
4236 VP_CLASSOF_IMPL(VPRecipeBase::VPDerivedIVSC)
4237
4238 void execute(VPTransformState &State) override {
4239 llvm_unreachable("Expected prior expansion of this recipe");
4240 }
4241
4242 /// Return the cost of this VPDerivedIVRecipe.
4243 InstructionCost computeCost(ElementCount VF,
4244 VPCostContext &Ctx) const override;
4245
4246 VPValue *getStartValue() const { return getOperand(0); }
4247 VPValue *getIndex() const { return getOperand(1); }
4248 VPValue *getStepValue() const { return getOperand(2); }
4249 const FPMathOperator *getFPBinOp() const { return FPBinOp; }
4251
4252 /// Returns true if the recipe only uses the first lane of operand \p Op.
4253 bool usesFirstLaneOnly(const VPValue *Op) const override {
4255 "Op must be an operand of the recipe");
4256 return true;
4257 }
4258
4259protected:
4260#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4261 /// Print the recipe.
4262 void printRecipe(raw_ostream &O, const Twine &Indent,
4263 VPSlotTracker &SlotTracker) const override;
4264#endif
4265};
4266
4267/// A recipe for handling phi nodes of integer and floating-point inductions,
4268/// producing their scalar values. Before unrolling by UF the recipe represents
4269/// the VF*UF scalar values to be produced, or UF scalar values if only first
4270/// lane is used, and has 3 operands: IV, step and VF. Unrolling adds one extra
4271/// operand StartIndex to all unroll parts except part 0, as the recipe
4272/// represents the VF scalar values (this number of values is taken from
4273/// State.VF rather than from the VF operand) starting at IV + StartIndex.
4275 Instruction::BinaryOps InductionOpcode;
4276
4277public:
4281 : VPRecipeWithIRFlags(VPRecipeBase::VPScalarIVStepsSC, {IV, Step, VF},
4282 IV->getScalarType(), FMFs, DL),
4283 InductionOpcode(Opcode) {}
4284
4285 ~VPScalarIVStepsRecipe() override = default;
4286
4288 auto *NewR = new VPScalarIVStepsRecipe(
4289 getOperand(0), getOperand(1), getOperand(2), InductionOpcode,
4291 if (VPValue *StartIndex = getStartIndex())
4292 NewR->setStartIndex(StartIndex);
4293 return NewR;
4294 }
4295
4296 VP_CLASSOF_IMPL(VPRecipeBase::VPScalarIVStepsSC)
4297
4298 /// Generate the scalarized versions of the phi node as needed by their users.
4299 void execute(VPTransformState &State) override;
4300
4301 /// Return the cost of this VPScalarIVStepsRecipe.
4302 InstructionCost computeCost(ElementCount VF,
4303 VPCostContext &Ctx) const override;
4304
4305 VPValue *getStepValue() const { return getOperand(1); }
4306
4307 /// Return the number of scalars to produce per unroll part, used to compute
4308 /// StartIndex during unrolling.
4309 VPValue *getVFValue() const { return getOperand(2); }
4310
4311 /// Return the StartIndex, or null if known to be zero, valid only after
4312 /// unrolling.
4314 return getNumOperands() == 4 ? getOperand(3) : nullptr;
4315 }
4316
4317 /// Set or add the StartIndex operand.
4318 void setStartIndex(VPValue *StartIndex) {
4319 if (getNumOperands() == 4)
4320 setOperand(3, StartIndex);
4321 else
4322 addOperand(StartIndex);
4323 }
4324
4325 /// Returns true if this recipe produces scalar values for all VF lanes.
4326 bool doesGeneratePerAllLanes() const;
4327
4328 /// Returns true if the recipe only uses the first lane of operand \p Op.
4329 bool usesFirstLaneOnly(const VPValue *Op) const override {
4331 "Op must be an operand of the recipe");
4332 return true;
4333 }
4334
4335 Instruction::BinaryOps getInductionOpcode() const { return InductionOpcode; }
4336
4337protected:
4338#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4339 /// Print the recipe.
4340 void printRecipe(raw_ostream &O, const Twine &Indent,
4341 VPSlotTracker &SlotTracker) const override;
4342#endif
4343};
4344
4345/// CastInfo helper for casting from VPRecipeBase to a mixin class that is not
4346/// part of the VPRecipeBase class hierarchy (e.g. VPPhiAccessors,
4347/// VPIRMetadata).
4348namespace vpdetail {
4349template <typename VPMixin, typename... RecipeTys>
4351 : public DefaultDoCastIfPossible<VPMixin *, VPRecipeBase *,
4352 CastInfoMixinImpl<VPMixin, RecipeTys...>> {
4353 static_assert((std::is_base_of_v<VPMixin, RecipeTys> && ...),
4354 "Each type in RecipeTys must derive from VPMixin");
4355
4356 /// Used by isa.
4357 static bool isPossible(VPRecipeBase *R) { return isa<RecipeTys...>(R); }
4358
4359 /// Used by cast.
4360 static VPMixin *doCast(VPRecipeBase *R) {
4361 VPMixin *Out = nullptr;
4362 ((Out = dyn_cast<RecipeTys>(R)) || ...);
4363 assert(Out && "Illegal recipe for cast");
4364 return Out;
4365 }
4366 static VPMixin *castFailed() { return nullptr; }
4367};
4368} // namespace vpdetail
4369
4370/// Support casting from VPRecipeBase -> VPPhiAccessors.
4371template <>
4375
4376template <>
4381template <>
4383 : public ForwardToPointerCast<VPPhiAccessors, VPRecipeBase *,
4384 CastInfo<VPPhiAccessors, VPRecipeBase *>> {};
4385
4386/// Support casting from VPRecipeBase / VPUser -> VPWidenMemoryRecipe.
4387template <>
4392template <>
4397
4398/// Support casting from VPSingleDefRecipe -> VPWidenMemoryRecipe (loads only).
4399template <>
4403template <>
4408
4409/// Support casting from VPRecipeBase -> VPIRMetadata.
4410template <>
4417
4418template <>
4423template <>
4425 : public ForwardToPointerCast<VPIRMetadata, VPRecipeBase *,
4426 CastInfo<VPIRMetadata, VPRecipeBase *>> {};
4427
4428/// VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph. It
4429/// holds a sequence of zero or more VPRecipe's each representing a sequence of
4430/// output IR instructions. All PHI-like recipes must come before any non-PHI
4431/// recipes.
4432class LLVM_ABI_FOR_TEST VPBasicBlock : public VPBlockBase {
4433 friend class VPlan;
4434
4435 /// Use VPlan::createVPBasicBlock to create VPBasicBlocks.
4436 VPBasicBlock(const Twine &Name = "", VPRecipeBase *Recipe = nullptr)
4437 : VPBlockBase(VPBasicBlockSC, Name.str()) {
4438 if (Recipe)
4439 appendRecipe(Recipe);
4440 }
4441
4442public:
4444
4445protected:
4446 /// The VPRecipes held in the order of output instructions to generate.
4448
4449 VPBasicBlock(VPBlockTy BlockSC, const Twine &Name = "")
4450 : VPBlockBase(BlockSC, Name.str()) {}
4451
4452public:
4453 ~VPBasicBlock() override {
4454 while (!Recipes.empty())
4455 Recipes.pop_back();
4456 }
4457
4458 /// Instruction iterators...
4463
4464 //===--------------------------------------------------------------------===//
4465 /// Recipe iterator methods
4466 ///
4467 inline iterator begin() { return Recipes.begin(); }
4468 inline const_iterator begin() const { return Recipes.begin(); }
4469 inline iterator end() { return Recipes.end(); }
4470 inline const_iterator end() const { return Recipes.end(); }
4471
4472 inline reverse_iterator rbegin() { return Recipes.rbegin(); }
4473 inline const_reverse_iterator rbegin() const { return Recipes.rbegin(); }
4474 inline reverse_iterator rend() { return Recipes.rend(); }
4475 inline const_reverse_iterator rend() const { return Recipes.rend(); }
4476
4477 inline size_t size() const { return Recipes.size(); }
4478 inline bool empty() const { return Recipes.empty(); }
4479 inline const VPRecipeBase &front() const { return Recipes.front(); }
4480 inline VPRecipeBase &front() { return Recipes.front(); }
4481 inline const VPRecipeBase &back() const { return Recipes.back(); }
4482 inline VPRecipeBase &back() { return Recipes.back(); }
4483
4484 /// Returns a reference to the list of recipes.
4486
4487 /// Returns a pointer to a member of the recipe list.
4488 static RecipeListTy VPBasicBlock::*getSublistAccess(VPRecipeBase *) {
4489 return &VPBasicBlock::Recipes;
4490 }
4491
4492 /// Method to support type inquiry through isa, cast, and dyn_cast.
4493 static inline bool classof(const VPBlockBase *V) {
4494 return V->getVPBlockID() == VPBlockBase::VPBasicBlockSC ||
4495 V->getVPBlockID() == VPBlockBase::VPIRBasicBlockSC;
4496 }
4497
4498 void insert(VPRecipeBase *Recipe, iterator InsertPt) {
4499 assert(Recipe && "No recipe to append.");
4500 assert(!Recipe->Parent && "Recipe already in VPlan");
4501 Recipe->Parent = this;
4502 Recipes.insert(InsertPt, Recipe);
4503 }
4504
4505 /// Augment the existing recipes of a VPBasicBlock with an additional
4506 /// \p Recipe as the last recipe.
4507 void appendRecipe(VPRecipeBase *Recipe) { insert(Recipe, end()); }
4508
4509 /// The method which generates the output IR instructions that correspond to
4510 /// this VPBasicBlock, thereby "executing" the VPlan.
4511 void execute(VPTransformState *State) override;
4512
4513 /// Return the cost of this VPBasicBlock.
4514 InstructionCost cost(ElementCount VF, VPCostContext &Ctx) override;
4515
4516 /// Return the position of the first non-phi node recipe in the block.
4517 iterator getFirstNonPhi();
4518
4519 /// Returns an iterator range over the PHI-like recipes in the block.
4523
4524 /// Split current block at \p SplitAt by inserting a new block between the
4525 /// current block and its successors and moving all recipes starting at
4526 /// SplitAt to the new block. Returns the new block.
4527 VPBasicBlock *splitAt(iterator SplitAt);
4528
4529 VPRegionBlock *getEnclosingLoopRegion();
4530 const VPRegionBlock *getEnclosingLoopRegion() const;
4531
4532#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4533 /// Print this VPBsicBlock to \p O, prefixing all lines with \p Indent. \p
4534 /// SlotTracker is used to print unnamed VPValue's using consequtive numbers.
4535 ///
4536 /// Note that the numbering is applied to the whole VPlan, so printing
4537 /// individual blocks is consistent with the whole VPlan printing.
4538 void print(raw_ostream &O, const Twine &Indent,
4539 VPSlotTracker &SlotTracker) const override;
4540 using VPBlockBase::print; // Get the print(raw_stream &O) version.
4541#endif
4542
4543 /// If the block has multiple successors, return the branch recipe terminating
4544 /// the block. If there are no or only a single successor, return nullptr;
4545 VPRecipeBase *getTerminator();
4546 const VPRecipeBase *getTerminator() const;
4547
4548 /// Returns true if the block is exiting it's parent region.
4549 bool isExiting() const;
4550
4551 /// Clone the current block and it's recipes, without updating the operands of
4552 /// the cloned recipes.
4553 VPBasicBlock *clone() override;
4554
4555 /// Returns the predecessor block at index \p Idx with the predecessors as per
4556 /// the corresponding plain CFG. If the block is an entry block to a region,
4557 /// the first predecessor is the single predecessor of a region, and the
4558 /// second predecessor is the exiting block of the region.
4559 const VPBasicBlock *getCFGPredecessor(unsigned Idx) const;
4560
4561protected:
4562 /// Execute the recipes in the IR basic block \p BB.
4563 void executeRecipes(VPTransformState *State, BasicBlock *BB);
4564
4565 /// Connect the VPBBs predecessors' in the VPlan CFG to the IR basic block
4566 /// generated for this VPBB.
4567 void connectToPredecessors(VPTransformState &State);
4568
4569private:
4570 /// Create an IR BasicBlock to hold the output instructions generated by this
4571 /// VPBasicBlock, and return it. Update the CFGState accordingly.
4572 BasicBlock *createEmptyBasicBlock(VPTransformState &State);
4573};
4574
4575inline const VPBasicBlock *
4577 return getAsRecipe()->getParent()->getCFGPredecessor(Idx);
4578}
4579
4580/// A special type of VPBasicBlock that wraps an existing IR basic block.
4581/// Recipes of the block get added before the first non-phi instruction in the
4582/// wrapped block.
4583/// Note: At the moment, VPIRBasicBlock can only be used to wrap VPlan's
4584/// preheader block.
4585class VPIRBasicBlock : public VPBasicBlock {
4586 friend class VPlan;
4587
4588 BasicBlock *IRBB;
4589
4590 /// Use VPlan::createVPIRBasicBlock to create VPIRBasicBlocks.
4591 VPIRBasicBlock(BasicBlock *IRBB)
4592 : VPBasicBlock(VPIRBasicBlockSC,
4593 (Twine("ir-bb<") + IRBB->getName() + Twine(">")).str()),
4594 IRBB(IRBB) {}
4595
4596public:
4597 ~VPIRBasicBlock() override = default;
4598
4599 static inline bool classof(const VPBlockBase *V) {
4600 return V->getVPBlockID() == VPBlockBase::VPIRBasicBlockSC;
4601 }
4602
4603 /// The method which generates the output IR instructions that correspond to
4604 /// this VPBasicBlock, thereby "executing" the VPlan.
4605 void execute(VPTransformState *State) override;
4606
4607 VPIRBasicBlock *clone() override;
4608
4609 BasicBlock *getIRBasicBlock() const { return IRBB; }
4610};
4611
4612/// Track information about the canonical IV and header mask of a loop region.
4613/// TODO: Have it also track the canonical IV increment, subject of NUW flag.
4615 /// VPRegionValue for the canonical IV, whose allocation is managed by
4616 /// VPCanonicalIVInfo.
4617 std::unique_ptr<VPRegionValue> CanIV;
4618
4619 /// Optional VPRegionValue for the header mask, set when tail folding.
4620 std::unique_ptr<VPRegionValue> HeaderMask;
4621
4622 /// Whether the increment of the canonical IV may unsigned wrap or not.
4623 bool HasNUW = true;
4624
4625public:
4627 : CanIV(std::make_unique<VPRegionValue>(Ty, DL, Region)) {}
4628
4629 VPRegionValue *getRegionValue() { return CanIV.get(); }
4630 const VPRegionValue *getRegionValue() const { return CanIV.get(); }
4631
4632 VPRegionValue *getHeaderMask() const { return HeaderMask.get(); }
4633
4634 /// Create the header mask for the region and return it. Must only be called
4635 /// when no header mask exists yet.
4637 assert(!HeaderMask && "Header mask already created");
4638 HeaderMask = std::make_unique<VPRegionValue>(
4639 Type::getInt1Ty(CanIV->getType()->getContext()), DebugLoc::getUnknown(),
4640 CanIV->getDefiningRegion());
4641 return HeaderMask.get();
4642 }
4643
4644 bool hasNUW() const { return HasNUW; }
4645
4646 void clearNUW() { HasNUW = false; }
4647};
4648
4649/// VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks
4650/// which form a Single-Entry-Single-Exiting subgraph of the output IR CFG.
4651/// A VPRegionBlock may indicate that its contents are to be replicated several
4652/// times. This is designed to support predicated scalarization, in which a
4653/// scalar if-then code structure needs to be generated VF * UF times. Having
4654/// this replication indicator helps to keep a single model for multiple
4655/// candidate VF's. The actual replication takes place only once the desired VF
4656/// and UF have been determined.
4657class LLVM_ABI_FOR_TEST VPRegionBlock : public VPBlockBase {
4658 friend class VPlan;
4659
4660 /// Hold the Single Entry of the SESE region modelled by the VPRegionBlock.
4661 VPBlockBase *Entry;
4662
4663 /// Hold the Single Exiting block of the SESE region modelled by the
4664 /// VPRegionBlock.
4665 VPBlockBase *Exiting;
4666
4667 /// Holds the Canonical IV of the loop region along with additional
4668 /// information. If CanIVInfo is nullptr, the region is a replicating region.
4669 /// Loop regions retain their canonical IVs until they are dissolved, even if
4670 /// the canonical IV has no users.
4671 std::unique_ptr<VPCanonicalIVInfo> CanIVInfo;
4672
4673 /// Use VPlan::createLoopRegion() and VPlan::createReplicateRegion() to create
4674 /// VPRegionBlocks.
4675 VPRegionBlock(VPBlockBase *Entry, VPBlockBase *Exiting,
4676 const std::string &Name = "")
4677 : VPBlockBase(VPRegionBlockSC, Name), Entry(Entry), Exiting(Exiting) {
4678 if (Entry) {
4679 assert(!Entry->hasPredecessors() && "Entry block has predecessors.");
4680 assert(Exiting && "Must also pass Exiting if Entry is passed.");
4681 assert(!Exiting->hasSuccessors() && "Exit block has successors.");
4682 Entry->setParent(this);
4683 Exiting->setParent(this);
4684 }
4685 }
4686
4687 VPRegionBlock(Type *CanIVTy, DebugLoc DL, VPBlockBase *Entry,
4688 VPBlockBase *Exiting, const std::string &Name = "")
4689 : VPRegionBlock(Entry, Exiting, Name) {
4690 CanIVInfo = std::make_unique<VPCanonicalIVInfo>(CanIVTy, DL, this);
4691 }
4692
4693public:
4694 ~VPRegionBlock() override = default;
4695
4696 /// Method to support type inquiry through isa, cast, and dyn_cast.
4697 static inline bool classof(const VPBlockBase *V) {
4698 return V->getVPBlockID() == VPBlockBase::VPRegionBlockSC;
4699 }
4700
4701 const VPBlockBase *getEntry() const { return Entry; }
4702 VPBlockBase *getEntry() { return Entry; }
4703
4704 /// Set \p EntryBlock as the entry VPBlockBase of this VPRegionBlock. \p
4705 /// EntryBlock must have no predecessors.
4706 void setEntry(VPBlockBase *EntryBlock) {
4707 assert(!EntryBlock->hasPredecessors() &&
4708 "Entry block cannot have predecessors.");
4709 Entry = EntryBlock;
4710 EntryBlock->setParent(this);
4711 }
4712
4713 const VPBlockBase *getExiting() const { return Exiting; }
4714 VPBlockBase *getExiting() { return Exiting; }
4715
4716 /// Set \p ExitingBlock as the exiting VPBlockBase of this VPRegionBlock. \p
4717 /// ExitingBlock must have no successors.
4718 void setExiting(VPBlockBase *ExitingBlock) {
4719 assert(!ExitingBlock->hasSuccessors() &&
4720 "Exit block cannot have successors.");
4721 Exiting = ExitingBlock;
4722 ExitingBlock->setParent(this);
4723 }
4724
4725 /// Returns the pre-header VPBasicBlock of the loop region.
4727 assert(!isReplicator() && "should only get pre-header of loop regions");
4728 return getSinglePredecessor()->getExitingBasicBlock();
4729 }
4730
4731 /// An indicator whether this region is to generate multiple replicated
4732 /// instances of output IR corresponding to its VPBlockBases.
4733 bool isReplicator() const { return !CanIVInfo; }
4734
4735 /// Return the VPBranchOnMaskRecipe from the entry block of this replicating
4736 /// region.
4737 const VPBranchOnMaskRecipe *getEntryBranchOnMask() const;
4739 return const_cast<VPBranchOnMaskRecipe *>(
4740 static_cast<const VPRegionBlock *>(this)->getEntryBranchOnMask());
4741 }
4742
4743 /// The method which generates the output IR instructions that correspond to
4744 /// this VPRegionBlock, thereby "executing" the VPlan.
4745 void execute(VPTransformState *State) override;
4746
4747 // Return the cost of this region.
4748 InstructionCost cost(ElementCount VF, VPCostContext &Ctx) override;
4749
4750#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4751 /// Print this VPRegionBlock to \p O (recursively), prefixing all lines with
4752 /// \p Indent. \p SlotTracker is used to print unnamed VPValue's using
4753 /// consequtive numbers.
4754 ///
4755 /// Note that the numbering is applied to the whole VPlan, so printing
4756 /// individual regions is consistent with the whole VPlan printing.
4757 void print(raw_ostream &O, const Twine &Indent,
4758 VPSlotTracker &SlotTracker) const override;
4759 using VPBlockBase::print; // Get the print(raw_stream &O) version.
4760#endif
4761
4762 /// Clone all blocks in the single-entry single-exit region of the block and
4763 /// their recipes without updating the operands of the cloned recipes.
4764 VPRegionBlock *clone() override;
4765
4766 /// Remove the current region from its VPlan, connecting its predecessor to
4767 /// its entry, and its exiting block to its successor.
4768 void dissolveToCFGLoop();
4769
4770 /// Get the canonical IV increment instruction if it exists. Otherwise, create
4771 /// a new increment before the terminator and return it. The canonical IV
4772 /// increment is subject to DCE if unused, unlike the canonical IV itself.
4773 VPInstruction *getOrCreateCanonicalIVIncrement();
4774
4775 /// Return the canonical induction variable of the region, null for
4776 /// replicating regions.
4778 return CanIVInfo ? CanIVInfo->getRegionValue() : nullptr;
4779 }
4781 return CanIVInfo ? CanIVInfo->getRegionValue() : nullptr;
4782 }
4783
4784 /// Return the type of the canonical IV for loop regions.
4786 return CanIVInfo->getRegionValue()->getType();
4787 }
4788
4789 /// Return the header mask of the region, or null if not set.
4791 return CanIVInfo ? CanIVInfo->getHeaderMask() : nullptr;
4792 }
4793
4794 /// Return the header mask if it exists and is used, or null otherwise. The
4795 /// mask is materialized into concrete recipes only after costing, so cost and
4796 /// codegen accounting sites use this to skip an unused mask.
4798 VPRegionValue *HeaderMask = getHeaderMask();
4799 return HeaderMask && HeaderMask->getNumUsers() > 0 ? HeaderMask : nullptr;
4800 }
4801
4802 /// Create the header mask for the region and return it. Must only be called
4803 /// on loop regions that don't already have a header mask.
4805 assert(CanIVInfo && "Can only create header mask for loop regions");
4806 return CanIVInfo->createHeaderMask();
4807 }
4808
4809 /// Return the region values of the loop region (canonical IV, header mask)
4810 /// or an empty vector for replicate regions.
4812 if (!CanIVInfo)
4813 return {};
4814 SmallVector<VPRegionValue *, 2> R = {CanIVInfo->getRegionValue()};
4815 if (auto *HM = CanIVInfo->getHeaderMask())
4816 R.push_back(HM);
4817 return R;
4818 }
4819
4820 /// Indicates if NUW is set for the canonical IV increment, for loop regions.
4821 bool hasCanonicalIVNUW() const { return CanIVInfo->hasNUW(); }
4822
4823 /// Unsets NUW for the canonical IV increment \p Increment, for loop regions.
4825 assert(Increment && "Must provide increment to clear");
4826 Increment->dropPoisonGeneratingFlags();
4827 CanIVInfo->clearNUW();
4828 }
4829};
4830
4832 return getParent()->getParent();
4833}
4834
4836 return getParent()->getParent();
4837}
4838
4839/// VPlan models a candidate for vectorization, encoding various decisions take
4840/// to produce efficient output IR, including which branches, basic-blocks and
4841/// output IR instructions to generate, and their cost. VPlan holds a
4842/// Hierarchical-CFG of VPBasicBlocks and VPRegionBlocks rooted at an Entry
4843/// VPBasicBlock.
4844class VPlan {
4845 friend class VPlanPrinter;
4846 friend class VPSlotTracker;
4847
4848 /// VPBasicBlock corresponding to the original preheader. Used to place
4849 /// VPExpandSCEV recipes for expressions used during skeleton creation and the
4850 /// rest of VPlan execution.
4851 /// When this VPlan is used for the epilogue vector loop, the entry will be
4852 /// replaced by a new entry block created during skeleton creation.
4853 VPBasicBlock *Entry;
4854
4855 /// VPIRBasicBlock wrapping the header of the original scalar loop.
4856 VPIRBasicBlock *ScalarHeader;
4857
4858 /// Immutable list of VPIRBasicBlocks wrapping the exit blocks of the original
4859 /// scalar loop. Note that some exit blocks may be unreachable at the moment,
4860 /// e.g. if the scalar epilogue always executes.
4862
4863 /// Holds the VFs applicable to this VPlan.
4865
4866 /// Holds the UFs applicable to this VPlan. If empty, the VPlan is valid for
4867 /// any UF.
4869
4870 /// Holds the name of the VPlan, for printing.
4871 std::string Name;
4872
4873 /// Represents the trip count of the original loop, for folding
4874 /// the tail.
4875 VPValue *TripCount = nullptr;
4876
4877 /// Represents the backedge taken count of the original loop, for folding
4878 /// the tail. It equals TripCount - 1.
4879 VPSymbolicValue *BackedgeTakenCount = nullptr;
4880
4881 /// Represents the vector trip count.
4882 VPSymbolicValue VectorTripCount;
4883
4884 /// Represents the vectorization factor of the loop.
4885 VPSymbolicValue VF;
4886
4887 /// Represents the unroll factor of the loop.
4888 VPSymbolicValue UF;
4889
4890 /// Represents the loop-invariant VF * UF of the vector loop region.
4891 VPSymbolicValue VFxUF;
4892
4893 /// Contains all the external definitions created for this VPlan, as a mapping
4894 /// from IR Values to VPIRValues.
4896
4897 /// Blocks allocated and owned by the VPlan. They will be deleted once the
4898 /// VPlan is destroyed.
4899 SmallVector<VPBlockBase *> CreatedBlocks;
4900
4901 /// Construct a VPlan with \p Entry to the plan and with \p ScalarHeader
4902 /// wrapping the original header of the scalar loop. The vector loop will have
4903 /// index type \p IdxTy.
4904 VPlan(VPBasicBlock *Entry, VPIRBasicBlock *ScalarHeader, Type *IdxTy)
4905 : Entry(Entry), ScalarHeader(ScalarHeader), VectorTripCount(IdxTy),
4906 VF(IdxTy), UF(IdxTy), VFxUF(IdxTy) {
4907 Entry->setPlan(this);
4908 assert(ScalarHeader->getNumSuccessors() == 0 &&
4909 "scalar header must be a leaf node");
4910 }
4911
4912public:
4913 /// Construct a VPlan for \p L. This will create VPIRBasicBlocks wrapping the
4914 /// original preheader and scalar header of \p L, to be used as entry and
4915 /// scalar header blocks of the new VPlan. The vector loop will have index
4916 /// type \p IdxTy.
4917 VPlan(Loop *L, Type *IdxTy);
4918
4919 /// Construct a VPlan with a new VPBasicBlock as entry, a VPIRBasicBlock
4920 /// wrapping \p ScalarHeaderBB and vector loop index of type \p IdxTy.
4921 VPlan(BasicBlock *ScalarHeaderBB, Type *IdxTy)
4922 : VectorTripCount(IdxTy), VF(IdxTy), UF(IdxTy), VFxUF(IdxTy) {
4923 setEntry(createVPBasicBlock("preheader"));
4924 ScalarHeader = createVPIRBasicBlock(ScalarHeaderBB);
4925 }
4926
4928
4930 Entry = VPBB;
4931 VPBB->setPlan(this);
4932 }
4933
4934 /// Generate the IR code for this VPlan.
4935 void execute(VPTransformState *State);
4936
4937 /// Return the cost of this plan.
4939
4940 VPBasicBlock *getEntry() { return Entry; }
4941 const VPBasicBlock *getEntry() const { return Entry; }
4942
4943 /// Returns the preheader of the vector loop region, if one exists, or null
4944 /// otherwise.
4946 const VPRegionBlock *VectorRegion = getVectorLoopRegion();
4947 return VectorRegion
4948 ? cast<VPBasicBlock>(VectorRegion->getSinglePredecessor())
4949 : nullptr;
4950 }
4951
4952 /// Returns the VPRegionBlock of the vector loop.
4955
4956 /// Returns true if this VPlan is for an outer loop, i.e., its vector
4957 /// loop region contains a nested loop region.
4958 LLVM_ABI_FOR_TEST bool isOuterLoop() const;
4959
4960 /// Returns true if the vector loop region is tail-folded.
4961 bool hasTailFolded() const {
4962 const VPRegionBlock *LoopRegion = getVectorLoopRegion();
4963 return LoopRegion && LoopRegion->getHeaderMask();
4964 }
4965
4966 /// Returns true if the plan requires a scalar epilogue after the vector
4967 /// loop. Must be called before removeBranchOnConst.
4969 const VPBasicBlock *MiddleVPBB = getMiddleBlock();
4970 return MiddleVPBB->getSingleSuccessor() == getScalarPreheader();
4971 }
4972
4973 /// Returns the 'middle' block of the plan, that is the block that selects
4974 /// whether to execute the scalar tail loop or the exit block from the loop
4975 /// latch. If there is an early exit from the vector loop, the middle block
4976 /// conceptully has the early exit block as third successor, split accross 2
4977 /// VPBBs. In that case, the second VPBB selects whether to execute the scalar
4978 /// tail loop or the exit block. If the scalar tail loop or exit block are
4979 /// known to always execute, the middle block may branch directly to that
4980 /// block. This function cannot be called once the vector loop region has been
4981 /// removed.
4983 VPRegionBlock *LoopRegion = getVectorLoopRegion();
4984 assert(
4985 LoopRegion &&
4986 "cannot call the function after vector loop region has been removed");
4987 // The middle block is always the last successor of the region.
4988 return cast<VPBasicBlock>(LoopRegion->getSuccessors().back());
4989 }
4990
4992 return const_cast<VPlan *>(this)->getMiddleBlock();
4993 }
4994
4995 /// Return the VPBasicBlock for the preheader of the scalar loop.
4998 getScalarHeader()->getSinglePredecessor());
4999 }
5000
5001 /// Return the VPIRBasicBlock wrapping the header of the scalar loop.
5002 VPIRBasicBlock *getScalarHeader() const { return ScalarHeader; }
5003
5004 /// Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of
5005 /// the original scalar loop.
5006 ArrayRef<VPIRBasicBlock *> getExitBlocks() const { return ExitBlocks; }
5007
5008 /// Returns true if \p VPBB is an exit block.
5009 bool isExitBlock(VPBlockBase *VPBB);
5010
5011 /// The trip count of the original loop.
5013 assert(TripCount && "trip count needs to be set before accessing it");
5014 return TripCount;
5015 }
5016
5017 /// Set the trip count assuming it is currently null; if it is not - use
5018 /// resetTripCount().
5019 void setTripCount(VPValue *NewTripCount) {
5020 assert(!TripCount && NewTripCount && "TripCount should not be set yet.");
5021 TripCount = NewTripCount;
5022 }
5023
5024 /// Resets the trip count for the VPlan. The caller must make sure all uses of
5025 /// the original trip count have been replaced.
5026 void resetTripCount(VPValue *NewTripCount) {
5027 assert(TripCount && NewTripCount && TripCount->user_empty() &&
5028 "TripCount must be set when resetting");
5029 TripCount = NewTripCount;
5030 }
5031
5032 /// The backedge taken count of the original loop.
5034 // BTC shares the canonical IV type with VectorTripCount.
5035 if (!BackedgeTakenCount)
5036 BackedgeTakenCount = new VPSymbolicValue(VectorTripCount.getType());
5037 return BackedgeTakenCount;
5038 }
5039 VPValue *getBackedgeTakenCount() const { return BackedgeTakenCount; }
5040
5041 /// The vector trip count.
5042 VPSymbolicValue &getVectorTripCount() { return VectorTripCount; }
5043
5044 /// Returns the VF of the vector loop region.
5045 VPSymbolicValue &getVF() { return VF; };
5046 const VPSymbolicValue &getVF() const { return VF; };
5047
5048 /// Returns the UF of the vector loop region.
5049 VPSymbolicValue &getUF() { return UF; };
5050
5051 /// Returns VF * UF of the vector loop region.
5052 VPSymbolicValue &getVFxUF() { return VFxUF; }
5053
5056 }
5057
5058 const DataLayout &getDataLayout() const {
5060 }
5061
5064 }
5065
5066 void addVF(ElementCount VF) { VFs.insert(VF); }
5067
5069 assert(hasVF(VF) && "Cannot set VF not already in plan");
5070 VFs.clear();
5071 VFs.insert(VF);
5072 }
5073
5074 /// Remove \p VF from the plan.
5076 assert(hasVF(VF) && "tried to remove VF not present in plan");
5077 VFs.remove(VF);
5078 }
5079
5080 bool hasVF(ElementCount VF) const { return VFs.count(VF); }
5081 bool hasScalableVF() const {
5082 return any_of(VFs, [](ElementCount VF) { return VF.isScalable(); });
5083 }
5084
5085 /// Returns an iterator range over all VFs of the plan.
5088 return VFs;
5089 }
5090
5091 /// Returns the single VF of the plan, asserting that the plan has exactly
5092 /// one VF.
5094 assert(VFs.size() == 1 && "expected plan with single VF");
5095 return VFs[0];
5096 }
5097
5098 bool hasScalarVFOnly() const {
5099 bool HasScalarVFOnly = VFs.size() == 1 && VFs[0].isScalar();
5100 assert(HasScalarVFOnly == hasVF(ElementCount::getFixed(1)) &&
5101 "Plan with scalar VF should only have a single VF");
5102 return HasScalarVFOnly;
5103 }
5104
5105 bool hasUF(unsigned UF) const { return UFs.empty() || UFs.contains(UF); }
5106
5107 /// Returns the concrete UF of the plan, after unrolling.
5108 unsigned getConcreteUF() const {
5109 assert(UFs.size() == 1 && "Expected a single UF");
5110 return UFs[0];
5111 }
5112
5113 void setUF(unsigned UF) {
5114 assert(hasUF(UF) && "Cannot set the UF not already in plan");
5115 UFs.clear();
5116 UFs.insert(UF);
5117 }
5118
5119 /// Returns true if the VPlan already has been unrolled, i.e. it has a single
5120 /// concrete UF.
5121 bool isUnrolled() const { return UFs.size() == 1; }
5122
5123 /// Return a string with the name of the plan and the applicable VFs and UFs.
5124 std::string getName() const;
5125
5126 void setName(const Twine &newName) { Name = newName.str(); }
5127
5128 /// Gets the live-in VPIRValue for \p V or adds a new live-in (if none exists
5129 /// yet) for \p V.
5131 assert(V && "Trying to get or add the VPIRValue of a null Value");
5132 auto [It, Inserted] = LiveIns.try_emplace(V);
5133 if (Inserted) {
5134 if (auto *CI = dyn_cast<ConstantInt>(V))
5135 It->second = new VPConstantInt(CI);
5136 else
5137 It->second = new VPIRValue(V);
5138 }
5139
5140 assert(isa<VPIRValue>(It->second) &&
5141 "Only VPIRValues should be in mapping");
5142 return It->second;
5143 }
5145 assert(V && "Trying to get or add the VPIRValue of a null VPIRValue");
5146 return getOrAddLiveIn(V->getValue());
5147 }
5148
5149 /// Return a VPIRValue wrapping i1 true.
5150 VPIRValue *getTrue() { return getConstantInt(1, 1); }
5151
5152 /// Return a VPIRValue wrapping i1 false.
5153 VPIRValue *getFalse() { return getConstantInt(1, 0); }
5154
5155 /// Return a VPIRValue wrapping the null value of type \p Ty.
5156 VPIRValue *getZero(Type *Ty) { return getConstantInt(Ty, 0); }
5157
5158 /// Return a VPIRValue wrapping the AllOnes value of type \p Ty.
5160 return getConstantInt(APInt::getAllOnes(Ty->getIntegerBitWidth()));
5161 }
5162
5163 /// Return a VPIRValue wrapping a ConstantInt with the given type and value.
5164 VPIRValue *getConstantInt(Type *Ty, uint64_t Val, bool IsSigned = false) {
5165 return getOrAddLiveIn(ConstantInt::get(Ty, Val, IsSigned));
5166 }
5167
5168 /// Return a VPIRValue wrapping a ConstantInt with the given bitwidth and
5169 /// value.
5171 bool IsSigned = false) {
5172 return getConstantInt(APInt(BitWidth, Val, IsSigned));
5173 }
5174
5175 /// Return a VPIRValue wrapping a ConstantInt with the given APInt value.
5177 return getOrAddLiveIn(ConstantInt::get(getContext(), Val));
5178 }
5179
5180 /// Return a VPIRValue wrapping a poison value of type \p Ty.
5182 return getOrAddLiveIn(PoisonValue::get(Ty));
5183 }
5184
5185 /// Return the live-in VPIRValue for \p V, if there is one or nullptr
5186 /// otherwise.
5187 VPIRValue *getLiveIn(Value *V) const { return LiveIns.lookup(V); }
5188
5189 /// Return the list of live-in VPValues available in the VPlan.
5190 auto getLiveIns() const { return LiveIns.values(); }
5191
5192#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5193 /// Print the live-ins of this VPlan to \p O.
5194 void printLiveIns(raw_ostream &O) const;
5195
5196 /// Print this VPlan to \p O.
5197 LLVM_ABI_FOR_TEST void print(raw_ostream &O) const;
5198
5199 /// Print this VPlan in DOT format to \p O.
5200 LLVM_ABI_FOR_TEST void printDOT(raw_ostream &O) const;
5201
5202 /// Dump the plan to stderr (for debugging).
5203 LLVM_DUMP_METHOD void dump() const;
5204#endif
5205
5206 /// Clone the current VPlan, update all VPValues of the new VPlan and cloned
5207 /// recipes to refer to the clones, and return it.
5209
5210 /// Create a new VPBasicBlock with \p Name and containing \p Recipe if
5211 /// present. The returned block is owned by the VPlan and deleted once the
5212 /// VPlan is destroyed.
5214 VPRecipeBase *Recipe = nullptr) {
5215 auto *VPB = new VPBasicBlock(Name, Recipe);
5216 VPB->setPlan(this);
5217 VPB->setNumber(CreatedBlocks.size());
5218 CreatedBlocks.push_back(VPB);
5219 return VPB;
5220 }
5221
5222 /// Create a new loop region with a canonical IV using \p CanIVTy and
5223 /// \p DL. Use \p Name as the region's name and set entry and exiting blocks
5224 /// to \p Entry and \p Exiting respectively, if provided. The returned block
5225 /// is owned by the VPlan and deleted once the VPlan is destroyed.
5227 const std::string &Name = "",
5228 VPBlockBase *Entry = nullptr,
5229 VPBlockBase *Exiting = nullptr) {
5230 auto *VPB = new VPRegionBlock(CanIVTy, DL, Entry, Exiting, Name);
5231 VPB->setPlan(this);
5232 VPB->setNumber(CreatedBlocks.size());
5233 CreatedBlocks.push_back(VPB);
5234 return VPB;
5235 }
5236
5237 /// Create a new replicate region with \p Entry, \p Exiting and \p Name. The
5238 /// returned block is owned by the VPlan and deleted once the VPlan is
5239 /// destroyed.
5241 const std::string &Name = "") {
5242 auto *VPB = new VPRegionBlock(Entry, Exiting, Name);
5243 VPB->setPlan(this);
5244 VPB->setNumber(CreatedBlocks.size());
5245 CreatedBlocks.push_back(VPB);
5246 return VPB;
5247 }
5248
5249 /// Create a VPIRBasicBlock wrapping \p IRBB, but do not create
5250 /// VPIRInstructions wrapping the instructions in t\p IRBB. The returned
5251 /// block is owned by the VPlan and deleted once the VPlan is destroyed.
5253
5254 /// Create a VPIRBasicBlock from \p IRBB containing VPIRInstructions for all
5255 /// instructions in \p IRBB, except its terminator which is managed by the
5256 /// successors of the block in VPlan. The returned block is owned by the VPlan
5257 /// and deleted once the VPlan is destroyed.
5259
5260 unsigned getMaxBlockNumber() const { return CreatedBlocks.size(); }
5261
5262 /// Returns true if the VPlan is based on a loop with an early exit.
5263 bool hasEarlyExit() const {
5264 unsigned NumExitPredecessors =
5265 sum_of(map_range(ExitBlocks, [](VPIRBasicBlock *EB) {
5266 return EB->getNumPredecessors();
5267 }));
5268
5269 // If the scalar preheader executes unconditionally, there's no branch from
5270 // middle block to any exit. If there is any edge to an exit block
5271 // remaining, it must be an early exit.
5272 VPBasicBlock *ScalarPH = getScalarPreheader();
5273 VPBlockBase *ScalarPHPred =
5274 ScalarPH ? ScalarPH->getSinglePredecessor() : nullptr;
5275 if (ScalarPHPred && ScalarPHPred->getNumSuccessors() == 1)
5276 return NumExitPredecessors >= 1;
5277
5278 // Otherwise there must be at least 2 edges to exit blocks (from the middle
5279 // block and the early exiting edge).
5280 return NumExitPredecessors > 1;
5281 }
5282
5283 /// Returns true if the scalar tail may execute after the vector loop, i.e.
5284 /// if the middle block is a predecessor of the scalar preheader. Note that
5285 /// this relies on unneeded branches to the scalar tail loop being removed.
5286 bool hasScalarTail() const {
5287 auto *ScalarPH = getScalarPreheader();
5288 return ScalarPH &&
5289 is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
5290 }
5291
5292 /// The type of the canonical induction variable of the vector loop.
5293 Type *getIndexType() const { return VF.getType(); }
5294};
5295
5296#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5297inline raw_ostream &operator<<(raw_ostream &OS, const VPlan &Plan) {
5298 Plan.print(OS);
5299 return OS;
5300}
5301#endif
5302
5303} // end namespace llvm
5304
5305#endif // LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned uint64_t
static MCDisassembler::DecodeStatus addOperand(MCInst &Inst, const MCOperand &Opnd)
Rewrite undef for PHI
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static void print(raw_ostream &Out, object::Archive::Kind Kind, T Val)
This file implements methods to test, set and extract typed bits from packed unsigned integers.
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:686
#define LLVM_ABI_FOR_TEST
Definition Compiler.h:220
#define LLVM_PACKED_START
Definition Compiler.h:579
dxil translate DXIL Translate Metadata
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
Hexagon Common GEP
This file defines an InstructionCost class that is used when calculating the cost of an instruction,...
static std::pair< Value *, APInt > getMask(Value *WideMask, unsigned Factor, ElementCount LeafValueEC)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file implements a map that provides insertion order iteration.
static Interval intersect(const Interval &I1, const Interval &I2)
This file provides utility analysis objects describing memory locations.
#define T
#define P(N)
static StringRef getName(Value *V)
static bool mayHaveSideEffects(MachineInstr &MI)
SI Fold Operands
Func MI getDebugLoc()))
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
static const BasicSubtargetSubTypeKV * find(StringRef S, ArrayRef< BasicSubtargetSubTypeKV > A)
Find KV in array using binary search.
This file contains the declarations of the entities induced by Vectorization Plans,...
#define VP_CLASSOF_IMPL(VPRecipeID)
Definition VPlan.h:587
static const uint32_t IV[8]
Definition blake3_impl.h:83
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this basic block belongs to.
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
This class represents a function call, abstracting a target machine's calling convention.
This is the base class for all instructions that perform data casts.
Definition InstrTypes.h:512
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
Utility class for floating point operations which can have information about relaxed accuracy require...
Definition Operator.h:202
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags fromRaw(unsigned Flags)
unsigned getRaw() const
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Common base class shared among various IRBuilders.
Definition IRBuilder.h:111
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
InnerLoopVectorizer vectorizes loops which contain only one basic block to a specified vectorization ...
bool isCast() const
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
LoopVectorizationCostModel - estimates the expected speedups due to vectorization.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Metadata node.
Definition Metadata.h:1081
Root of the metadata hierarchy.
Definition Metadata.h:64
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
This class represents an assumption made using SCEV expressions which can be checked at run-time.
This class represents an analyzed expression in the program.
This class provides computation of slot numbers for LLVM Assembly writing.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator erase(const_iterator CI)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::string str() const
Get the contents as an std::string.
Definition StringRef.h:222
This class represents a truncation of integer types.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
Definition Twine.cpp:17
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
void execute(VPTransformState &State) override
Generate the active lane mask phi of the vector loop.
VPActiveLaneMaskPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4088
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPActiveLaneMaskPHIRecipe(VPValue *StartMask, DebugLoc DL)
Definition VPlan.h:4082
~VPActiveLaneMaskPHIRecipe() override=default
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4432
RecipeListTy::const_iterator const_iterator
Definition VPlan.h:4460
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
Definition VPlan.h:4507
RecipeListTy::const_reverse_iterator const_reverse_iterator
Definition VPlan.h:4462
RecipeListTy::iterator iterator
Instruction iterators...
Definition VPlan.h:4459
RecipeListTy & getRecipeList()
Returns a reference to the list of recipes.
Definition VPlan.h:4485
iplist< VPRecipeBase > RecipeListTy
Definition VPlan.h:4443
iterator end()
Definition VPlan.h:4469
iterator begin()
Recipe iterator methods.
Definition VPlan.h:4467
RecipeListTy::reverse_iterator reverse_iterator
Definition VPlan.h:4461
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
Definition VPlan.h:4520
const VPBasicBlock * getCFGPredecessor(unsigned Idx) const
Returns the predecessor block at index Idx with the predecessors as per the corresponding plain CFG.
Definition VPlan.cpp:756
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
Definition VPlan.cpp:233
~VPBasicBlock() override
Definition VPlan.h:4453
const_reverse_iterator rbegin() const
Definition VPlan.h:4473
reverse_iterator rend()
Definition VPlan.h:4474
RecipeListTy Recipes
The VPRecipes held in the order of output instructions to generate.
Definition VPlan.h:4447
VPRecipeBase & back()
Definition VPlan.h:4482
const VPRecipeBase & front() const
Definition VPlan.h:4479
const_iterator begin() const
Definition VPlan.h:4468
VPRecipeBase & front()
Definition VPlan.h:4480
const VPRecipeBase & back() const
Definition VPlan.h:4481
void insert(VPRecipeBase *Recipe, iterator InsertPt)
Definition VPlan.h:4498
bool empty() const
Definition VPlan.h:4478
const_iterator end() const
Definition VPlan.h:4470
static bool classof(const VPBlockBase *V)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:4493
static RecipeListTy VPBasicBlock::* getSublistAccess(VPRecipeBase *)
Returns a pointer to a member of the recipe list.
Definition VPlan.h:4488
reverse_iterator rbegin()
Definition VPlan.h:4472
friend class VPlan
Definition VPlan.h:4433
size_t size() const
Definition VPlan.h:4477
const_reverse_iterator rend() const
Definition VPlan.h:4475
VPBasicBlock(VPBlockTy BlockSC, const Twine &Name="")
Definition VPlan.h:4449
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3018
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
Definition VPlan.h:3023
VPBlendRecipe(PHINode *Phi, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL)
The blend operation is a User of the incoming values and of their respective masks,...
Definition VPlan.h:2977
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:3013
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:3035
VPBlendRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:3000
VPBlendRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2998
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
Definition VPlan.h:3029
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:3009
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:97
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
Definition VPlan.h:307
VPRegionBlock * getParent()
Definition VPlan.h:195
const VPlan * getPlan() const
Definition VPlan.h:200
void setPlan(VPlan *ParentPlan)
Sets the pointer of the plan containing the block.
Definition VPlan.h:203
VPBlocksTy & getPredecessors()
Definition VPlan.h:231
iterator_range< VPBlockBase ** > predecessors()
Definition VPlan.h:228
LLVM_DUMP_METHOD void dump() const
Dump this VPBlockBase to dbgs().
Definition VPlan.h:383
void setName(const Twine &newName)
Definition VPlan.h:188
size_t getNumSuccessors() const
Definition VPlan.h:245
iterator_range< VPBlockBase ** > successors()
Definition VPlan.h:227
virtual void print(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Print plain-text dump of this VPBlockBase to O, prefixing all lines with Indent.
bool hasPredecessors() const
Returns true if this block has any predecessors.
Definition VPlan.h:225
void swapSuccessors()
Swap successors of the block. The block must have exactly 2 successors.
Definition VPlan.h:329
void printSuccessors(raw_ostream &O, const Twine &Indent) const
Print the successors of this block to O, prefixing all lines with Indent.
Definition VPlan.cpp:641
SmallVectorImpl< VPBlockBase * > VPBlocksTy
Definition VPlan.h:182
virtual ~VPBlockBase()=default
unsigned getNumber() const
Return the unique number of the block.
Definition VPlan.h:349
void setNumber(unsigned N)
Set the unique number of the block, used for dominator tree.
Definition VPlan.h:352
unsigned getIndexForSuccessor(const VPBlockBase *Succ) const
Returns the index for Succ in the blocks successor list.
Definition VPlan.h:342
size_t getNumPredecessors() const
Definition VPlan.h:246
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
Definition VPlan.h:298
VPBlockBase * getEnclosingBlockWithPredecessors()
Definition VPlan.cpp:225
unsigned getIndexForPredecessor(const VPBlockBase *Pred) const
Returns the index for Pred in the blocks predecessors list.
Definition VPlan.h:335
enum :unsigned char { VPRegionBlockSC, VPBasicBlockSC, VPIRBasicBlockSC } VPBlockTy
An enumeration for keeping track of the concrete subclass of VPBlockBase that are actually instantiat...
Definition VPlan.h:105
bool hasSuccessors() const
Returns true if this block has any successors.
Definition VPlan.h:223
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
virtual VPBlockBase * clone()=0
Clone the current block and it's recipes without updating the operands of the cloned recipes,...
virtual InstructionCost cost(ElementCount VF, VPCostContext &Ctx)=0
Return the cost of the block.
VPlan * getPlan()
Definition VPlan.h:199
const VPRegionBlock * getParent() const
Definition VPlan.h:196
const std::string & getName() const
Definition VPlan.h:186
void clearSuccessors()
Remove all the successors of this block.
Definition VPlan.h:317
void setTwoSuccessors(VPBlockBase *IfTrue, VPBlockBase *IfFalse)
Set two given VPBlockBases IfTrue and IfFalse to be the two successors of this VPBlockBase.
Definition VPlan.h:289
VPBlockBase * getSinglePredecessor() const
Definition VPlan.h:241
virtual void execute(VPTransformState *State)=0
The method which generates the output IR that correspond to this VPBlockBase, thereby "executing" the...
const VPBlocksTy & getHierarchicalSuccessors()
Definition VPlan.h:265
void clearPredecessors()
Remove all the predecessor of this block.
Definition VPlan.h:314
friend class VPBlockUtils
Definition VPlan.h:98
unsigned getVPBlockID() const
Definition VPlan.h:193
void printAsOperand(raw_ostream &OS, bool PrintType=false) const
Definition VPlan.h:361
void swapPredecessors()
Swap predecessors of the block.
Definition VPlan.h:321
VPBlocksTy & getSuccessors()
Definition VPlan.h:220
VPBlockBase * getEnclosingBlockWithSuccessors()
An Enclosing Block of a block B is any block containing B, including B itself.
Definition VPlan.cpp:217
void setOneSuccessor(VPBlockBase *Successor)
Set a given VPBlockBase Successor as the single successor of this VPBlockBase.
Definition VPlan.h:278
void setParent(VPRegionBlock *P)
Definition VPlan.h:205
VPBlockBase * getSingleHierarchicalPredecessor()
Definition VPlan.h:271
VPBlockBase * getSingleSuccessor() const
Definition VPlan.h:235
const VPBlocksTy & getSuccessors() const
Definition VPlan.h:219
VPBlockBase(VPBlockTy SC, const std::string &N)
Definition VPlan.h:392
A recipe for generating conditional branches on the bits of a mask.
Definition VPlan.h:3524
VPBranchOnMaskRecipe(VPValue *BlockInMask, DebugLoc DL, const VPIRMetadata &Metadata={})
Definition VPlan.h:3526
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Definition VPlan.h:3547
VPBranchOnMaskRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3531
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe uses scalars of operand Op.
Definition VPlan.h:3555
VPlan-based builder utility similar to IRBuilder.
VPRegionValue * createHeaderMask()
Create the header mask for the region and return it.
Definition VPlan.h:4636
VPRegionValue * getHeaderMask() const
Definition VPlan.h:4632
VPRegionValue * getRegionValue()
Definition VPlan.h:4629
VPCanonicalIVInfo(Type *Ty, DebugLoc DL, VPRegionBlock *Region)
Definition VPlan.h:4626
const VPRegionValue * getRegionValue() const
Definition VPlan.h:4630
bool hasNUW() const
Definition VPlan.h:4644
VPCurrentIterationPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4120
VPCurrentIterationPHIRecipe(VPValue *StartIV, DebugLoc DL)
Definition VPlan.h:4114
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPCurrentIterationPHIRecipe.
Definition VPlan.h:4132
LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi nodes.
Definition VPlan.h:4126
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4139
~VPCurrentIterationPHIRecipe() override=default
InductionDescriptor::InductionKind getInductionKind() const
Definition VPlan.h:4250
VPValue * getIndex() const
Definition VPlan.h:4247
const FPMathOperator * getFPBinOp() const
Definition VPlan.h:4249
VPDerivedIVRecipe(InductionDescriptor::InductionKind Kind, const FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Definition VPlan.h:4221
VPValue * getStepValue() const
Definition VPlan.h:4248
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4238
VPDerivedIVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4231
~VPDerivedIVRecipe() override=default
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4253
VPValue * getStartValue() const
Definition VPlan.h:4246
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4057
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPExpandSCEVRecipe.
Definition VPlan.h:4062
VPExpandSCEVRecipe(const SCEV *Expr)
const SCEV * getSCEV() const
Definition VPlan.h:4068
VPExpandSCEVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4053
~VPExpandSCEVRecipe() override=default
void execute(VPTransformState &State) override
Method for generating code, must not be called as this recipe is abstract.
Definition VPlan.h:3701
bool isVectorToScalar() const
Returns true if this VPExpressionRecipe produces a single scalar.
VPExpressionRecipe(VPWidenCastRecipe *Ext, VPWidenRecipe *Neg, VPReductionRecipe *Red)
Definition VPlan.h:3617
VPExpressionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3670
SmallVector< VPSingleDefRecipe * > decompose()
Return and insert the recipes of the expression back into the VPlan, directly before the current reci...
~VPExpressionRecipe() override
Definition VPlan.h:3658
ExpressionTypes getExpressionType() const
Returns the expression type of this recipe.
Definition VPlan.h:3693
VPExpressionRecipe(VPWidenCastRecipe *Ext, VPReductionRecipe *Red)
Definition VPlan.h:3615
bool mayHaveSideEffects() const
Returns true if this expression contains recipes that may have side effects.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
bool mayReadOrWriteMemory() const
Returns true if this expression contains recipes that may read from or write to memory.
VPExpressionRecipe(VPWidenCastRecipe *Ext0, VPWidenCastRecipe *Ext1, VPWidenRecipe *Mul, VPReductionRecipe *Red)
Definition VPlan.h:3633
VPExpressionRecipe(ExpressionTypes ExpressionType, ArrayRef< VPSingleDefRecipe * > ExpressionRecipes)
Construct a new VPExpressionRecipe by internalizing recipes in ExpressionRecipes.
VPExpressionRecipe(VPWidenCastRecipe *Ext0, VPWidenCastRecipe *Ext1, VPWidenRecipe *Mul, VPWidenRecipe *Neg, VPReductionRecipe *Red)
Definition VPlan.h:3637
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getVFScaleFactor() const
Definition VPlan.h:3695
VPExpressionRecipe(VPWidenRecipe *Mul, VPReductionRecipe *Red)
Definition VPlan.h:3631
A pure virtual base class for all recipes modeling header phis, including phis for first order recurr...
Definition VPlan.h:2458
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this header phi recipe.
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:2465
void addBackedgeValue(VPValue *V)
Add V as the incoming value from the loop backedge.
Definition VPlan.h:2507
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2478
static bool classof(const VPValue *V)
Definition VPlan.h:2475
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override=0
Print the recipe.
VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr, VPValue *Start, Type *ResultTy, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2460
virtual VPValue * getBackedgeValue()
Returns the incoming value from the loop backedge.
Definition VPlan.h:2501
void setBackedgeValue(VPValue *V)
Update the incoming value from the loop backedge.
Definition VPlan.h:2504
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2490
void setStartValue(VPValue *V)
Update the start value of the recipe.
Definition VPlan.h:2498
static bool classof(const VPRecipeBase *R)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:2471
VPValue * getStartValue() const
Definition VPlan.h:2493
void execute(VPTransformState &State) override=0
Generate the phi nodes.
~VPHeaderPHIRecipe() override=default
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
Definition VPlan.h:2180
void execute(VPTransformState &State) override
Produce a vectorized histogram operation.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2213
VPHistogramRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2193
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHistogramRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getMask() const
Return the mask operand if one was provided, or a null pointer if all lanes should be executed uncond...
Definition VPlan.h:2208
VP_CLASSOF_IMPL(VPRecipeBase::VPHistogramSC)
~VPHistogramRecipe() override=default
VPHistogramRecipe(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2185
A special type of VPBasicBlock that wraps an existing IR basic block.
Definition VPlan.h:4585
void execute(VPTransformState *State) override
The method which generates the output IR instructions that correspond to this VPBasicBlock,...
Definition VPlan.cpp:453
BasicBlock * getIRBasicBlock() const
Definition VPlan.h:4609
static bool classof(const VPBlockBase *V)
Definition VPlan.h:4599
~VPIRBasicBlock() override=default
friend class VPlan
Definition VPlan.h:4586
VPIRBasicBlock * clone() override
Clone the current block and it's recipes, without updating the operands of the cloned recipes.
Definition VPlan.cpp:478
Class to record and manage LLVM IR flags.
Definition VPlan.h:696
WrapFlagsTy getNoWrapFlagsOrNone() const
Definition VPlan.h:1034
FastMathFlagsTy FMFs
Definition VPlan.h:788
ReductionFlagsTy ReductionFlags
Definition VPlan.h:790
VPIRFlags(RecurKind Kind, bool IsOrdered, bool IsInLoop, FastMathFlags FMFs)
Definition VPlan.h:881
LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const
Returns true if the set flags are valid for Opcode.
VPIRFlags(DisjointFlagsTy DisjointFlags)
Definition VPlan.h:861
VPIRFlags(WrapFlagsTy WrapFlags)
Definition VPlan.h:847
WrapFlagsTy WrapFlags
Definition VPlan.h:782
void printFlags(raw_ostream &O) const
VPIRFlags(CmpInst::Predicate Pred, FastMathFlags FMFs)
Definition VPlan.h:840
bool hasFastMathFlags() const
Returns true if the recipe has fast-math flags.
Definition VPlan.h:1005
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
bool isReductionOrdered() const
Definition VPlan.h:1060
TruncFlagsTy TruncFlags
Definition VPlan.h:783
CmpInst::Predicate getPredicate() const
Definition VPlan.h:977
WrapFlagsTy getNoWrapFlags() const
Definition VPlan.h:1044
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
uint8_t AllFlags[2]
Definition VPlan.h:791
void transferFlags(VPIRFlags &Other)
Definition VPlan.h:886
ExactFlagsTy ExactFlags
Definition VPlan.h:785
bool hasNoSignedWrap() const
Definition VPlan.h:1023
void intersectFlags(const VPIRFlags &Other)
Only keep flags also present in Other.
bool isDisjoint() const
Definition VPlan.h:1048
VPIRFlags(TruncFlagsTy TruncFlags)
Definition VPlan.h:852
VPIRFlags(FastMathFlags FMFs)
Definition VPlan.h:857
VPIRFlags(NonNegFlagsTy NonNegFlags)
Definition VPlan.h:866
VPIRFlags(CmpInst::Predicate Pred)
Definition VPlan.h:835
uint8_t GEPFlagsStorage
Definition VPlan.h:786
VPIRFlags(ExactFlagsTy ExactFlags)
Definition VPlan.h:871
GEPNoWrapFlags getGEPNoWrapFlags() const
Definition VPlan.h:995
bool hasPredicate() const
Returns true if the recipe has a comparison predicate.
Definition VPlan.h:1000
LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode, Type *ResultTy) const
Returns true if Opcode with scalar result type ResultTy has its required flags set.
DisjointFlagsTy DisjointFlags
Definition VPlan.h:784
void setPredicate(CmpInst::Predicate Pred)
Definition VPlan.h:985
bool hasNoUnsignedWrap() const
Definition VPlan.h:1012
FCmpFlagsTy FCmpFlags
Definition VPlan.h:789
NonNegFlagsTy NonNegFlags
Definition VPlan.h:787
bool isReductionInLoop() const
Definition VPlan.h:1066
void dropPoisonGeneratingFlags()
Drop all poison-generating flags.
Definition VPlan.h:897
void applyFlags(Instruction &I) const
Apply the IR flags to I.
Definition VPlan.h:934
VPIRFlags(GEPNoWrapFlags GEPFlags)
Definition VPlan.h:876
uint8_t CmpPredStorage
Definition VPlan.h:781
RecurKind getRecurKind() const
Definition VPlan.h:1054
VPIRFlags(Instruction &I)
Definition VPlan.h:797
Instruction & getInstruction() const
Definition VPlan.h:1764
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first part of operand Op.
Definition VPlan.h:1772
~VPIRInstruction() override=default
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPIRInstruction * clone() override
Clone the current recipe.
Definition VPlan.h:1751
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:1778
static LLVM_ABI_FOR_TEST VPIRInstruction * create(Instruction &I)
Create a new VPIRPhi for \I , if it is a PHINode, otherwise create a VPIRInstruction.
LLVM_ABI_FOR_TEST InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPIRInstruction.
bool usesScalars(const VPValue *Op) const override
Returns true if the VPUser uses scalars of operand Op.
Definition VPlan.h:1766
VPIRInstruction(Instruction &I)
VPIRInstruction::create() should be used to create VPIRInstructions, as subclasses may need to be cre...
Definition VPlan.h:1739
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Helper to manage IR metadata for recipes.
Definition VPlan.h:1184
MDNode * getBranchWeights() const
Returns the branch weights recorded for this terminator, preferring real profile data over an estimat...
Definition VPlan.h:1273
VPIRMetadata & operator=(const VPIRMetadata &Other)=default
MDNode * getMetadata(unsigned Kind) const
Get metadata of kind Kind. Returns nullptr if not found.
Definition VPlan.h:1254
VPIRMetadata(Instruction &I)
Adds metatadata that can be preserved from the original instruction I.
Definition VPlan.h:1213
VPIRMetadata(const VPIRMetadata &Other)=default
Copy constructor for cloning.
VPIRMetadata()=default
void setEstimatedBranchWeights(MDNode *Node)
Set estimated branch weights to Node.
Definition VPlan.h:1284
void applyMetadata(Instruction &I) const
Add all metadata to I.
void setMetadata(unsigned Kind, MDNode *Node)
Set metadata with kind Kind to Node.
Definition VPlan.h:1233
void eraseMetadata(unsigned Kind)
Remove the metadata of kind Kind, if present.
Definition VPlan.h:1245
bool hasEstimatedBranchWeights() const
Returns true if the weights returned by getBranchWeights are estimated.
Definition VPlan.h:1279
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1303
VPInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
Definition VPlan.h:1554
iterator_range< operand_iterator > operandsWithoutMask()
Returns an iterator range over the operands excluding the mask operand if present.
Definition VPlan.h:1574
VPInstruction * clone() override
Clone the current recipe.
Definition VPlan.h:1484
@ ExtractLastActive
Extracts the last active lane from a set of vectors.
Definition VPlan.h:1421
@ Intrinsic
Calls a scalar intrinsic. The intrinsic ID is the last operand.
Definition VPlan.h:1433
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1412
@ ExitingIVValue
Compute the exiting value of a wide induction after vectorization, that is the value of the last lane...
Definition VPlan.h:1425
@ WideIVStep
Scale the first operand (vector step) by the second operand (scalar-step).
Definition VPlan.h:1429
@ ResumeForEpilogue
Explicit user for values in the main VPlan, used by the epilogue vector loop.
Definition VPlan.h:1415
@ Unpack
Extracts all lanes from its (non-scalable) vector operand.
Definition VPlan.h:1363
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1408
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1358
@ BuildStructVector
Given operands of (the same) struct type, creates a struct of fixed- width vectors each containing a ...
Definition VPlan.h:1355
@ CanonicalIVIncrementForPart
Definition VPlan.h:1339
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1366
bool hasResult() const
Definition VPlan.h:1518
iterator_range< const_operand_iterator > operandsWithoutMask() const
Definition VPlan.h:1577
void addMask(VPValue *Mask)
Add mask Mask to an unmasked VPInstruction, if it needs masking.
Definition VPlan.h:1559
StringRef getName() const
Returns the symbolic name assigned to the VPInstruction.
Definition VPlan.h:1603
unsigned getOpcode() const
Definition VPlan.h:1497
void setName(StringRef NewName)
Set the symbolic name for the VPInstruction.
Definition VPlan.h:1606
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe only uses scalars of operand Op.
Definition VPlan.h:1588
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
VPValue * getMask() const
Returns the mask for the VPInstruction.
Definition VPlan.h:1570
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
VPInstruction * cloneWithOperands(ArrayRef< VPValue * > NewOperands, Type *ResultTy=nullptr)
Definition VPlan.h:1488
unsigned getNumOperandsForOpcode() const
Return the number of operands determined by the opcode of the VPInstruction, excluding mask.
bool isMasked() const
Returns true if the VPInstruction has a mask operand.
Definition VPlan.h:1544
A common base class for interleaved memory operations.
Definition VPlan.h:3060
virtual unsigned getNumStoreOperands() const =0
Returns the number of stored operands of this interleave group.
VPInterleaveBase(VPRecipeTy SC, const InterleaveGroup< Instruction > *IG, ArrayRef< VPValue * > Operands, ArrayRef< VPValue * > StoredValues, VPValue *Mask, bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
Definition VPlan.h:3072
bool usesFirstLaneOnly(const VPValue *Op) const override=0
Returns true if the recipe only uses the first lane of operand Op.
bool needsMaskForGaps() const
Return true if the access needs a mask because of the gaps.
Definition VPlan.h:3122
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:3128
static bool classof(const VPUser *U)
Definition VPlan.h:3104
Instruction * getInsertPos() const
Definition VPlan.h:3126
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:3099
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3124
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3116
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3145
VPInterleaveBase * clone() override=0
Clone the current recipe.
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3110
bool usesFirstLaneOnly(const VPValue *Op) const override
The recipe only uses the first lane of the address, and EVL operand.
Definition VPlan.h:3225
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3219
~VPInterleaveEVLRecipe() override=default
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3232
VPInterleaveEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3212
VPInterleaveEVLRecipe(VPInterleaveRecipe &R, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3199
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
Definition VPlan.h:3155
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3182
~VPInterleaveRecipe() override=default
VPInterleaveRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3165
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3176
VPInterleaveRecipe(const InterleaveGroup< Instruction > *IG, VPValue *Addr, ArrayRef< VPValue * > StoredValues, VPValue *Mask, bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
Definition VPlan.h:3157
In what follows, the term "input IR" refers to code that is fed into the vectorizer whereas the term ...
A VPRecipeValue defined by a multi-def recipe, stores a pointer to it.
Definition VPlanValue.h:377
Helper type to provide functions to access incoming values and blocks for phi-like recipes.
Definition VPlan.h:1618
virtual const VPRecipeBase * getAsRecipe() const =0
Return a VPRecipeBase* to the current object.
LLVM_ABI_FOR_TEST VPValue * getIncomingValueForBlock(const VPBasicBlock *VPBB) const
Returns the incoming value for VPBB. VPBB must be an incoming block.
VPUser::const_operand_range incoming_values() const
Returns an interator range over the incoming values.
Definition VPlan.h:1648
void addIncoming(VPValue *IncomingV)
Append IncomingV as an incoming value to the phi-like recipe.
Definition VPlan.h:1677
virtual unsigned getNumIncoming() const
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:1643
void removeIncomingValueFor(VPBlockBase *IncomingBlock) const
Removes the incoming value for IncomingBlock, which must be a predecessor.
const VPBasicBlock * getIncomingBlock(unsigned Idx) const
Returns the incoming block with index Idx.
Definition VPlan.h:4576
detail::zippy< llvm::detail::zip_first, VPUser::const_operand_range, const_incoming_blocks_range > incoming_values_and_blocks() const
Returns an iterator range over pairs of incoming values and corresponding incoming blocks.
Definition VPlan.h:1668
VPValue * getIncomingValue(unsigned Idx) const
Returns the incoming VPValue with index Idx.
Definition VPlan.h:1627
virtual ~VPPhiAccessors()=default
void printPhiOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the recipe.
void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const
Sets the incoming value for VPBB to V.
iterator_range< mapped_iterator< detail::index_iterator, std::function< const VPBasicBlock *(size_t)> > > const_incoming_blocks_range
Definition VPlan.h:1653
const_incoming_blocks_range incoming_blocks() const
Returns an iterator range over the incoming blocks.
Definition VPlan.h:1657
~VPPredInstPHIRecipe() override=default
VPPredInstPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3741
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPPredInstPHIRecipe.
Definition VPlan.h:3752
VPPredInstPHIRecipe(VPValue *PredV, DebugLoc DL)
Construct a VPPredInstPHIRecipe given PredInst whose value needs a phi nodes after merging back from ...
Definition VPlan.h:3736
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
bool mayReadFromMemory() const
Returns true if the recipe may read from memory.
bool mayReadOrWriteMemory() const
Returns true if the recipe may read from or write to memory.
Definition VPlan.h:548
virtual void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Each concrete VPRecipe prints itself, without printing common information, like debug info or metadat...
VPRegionBlock * getRegion()
Definition VPlan.h:4831
void setDebugLoc(DebugLoc NewDL)
Set the recipe's debug location to NewDL.
Definition VPlan.h:556
bool mayWriteToMemory() const
Returns true if the recipe may write to memory.
VPRecipeTy getVPRecipeID() const
Definition VPlan.h:521
~VPRecipeBase() override=default
VPBasicBlock * getParent()
Definition VPlan.h:475
enum :unsigned char { VPBranchOnMaskSC, VPDerivedIVSC, VPExpandSCEVSC, VPExpressionSC, VPIRInstructionSC, VPInstructionSC, VPInterleaveEVLSC, VPInterleaveSC, VPReductionEVLSC, VPReductionSC, VPReplicateSC, VPScalarIVStepsSC, VPVectorPointerSC, VPVectorEndPointerSC, VPWidenCallSC, VPWidenCanonicalIVSC, VPWidenCastSC, VPWidenGEPSC, VPWidenIntrinsicSC, VPWidenMemIntrinsicSC, VPWidenLoadEVLSC, VPWidenLoadSC, VPWidenStoreEVLSC, VPWidenStoreSC, VPWidenSC, VPBlendSC, VPHistogramSC, VPWidenPHISC, VPPredInstPHISC, VPCurrentIterationPHISC, VPActiveLaneMaskPHISC, VPFirstOrderRecurrencePHISC, VPWidenIntOrFpInductionSC, VPWidenPointerInductionSC, VPReductionPHISC, VPFirstPHISC=VPWidenPHISC, VPFirstHeaderPHISC=VPCurrentIterationPHISC, VPLastHeaderPHISC=VPReductionPHISC, VPLastPHISC=VPReductionPHISC, } VPRecipeTy
An enumeration for keeping track of the concrete subclass of VPRecipeBase that is actually instantiat...
Definition VPlan.h:418
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
virtual void execute(VPTransformState &State)=0
The method which generates the output IR instructions that correspond to this VPRecipe,...
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
static bool classof(const VPDef *D)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:524
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
virtual VPRecipeBase * clone()=0
Clone the current recipe.
friend class VPBlockUtils
Definition VPlan.h:405
const VPBasicBlock * getParent() const
Definition VPlan.h:476
VPRecipeBase(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:465
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this recipe, taking into account if the cost computation should be skipped and the...
static bool classof(const VPUser *U)
Definition VPlan.h:529
void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
void moveAfter(VPRecipeBase *MovePos)
Unlink this recipe from its current VPBasicBlock and insert it into the VPBasicBlock that MovePos liv...
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:350
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3393
VPReductionEVLRecipe(VPReductionRecipe &R, VPValue &EVL, VPValue *CondOp, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3371
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3396
VPReductionEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3383
~VPReductionEVLRecipe() override=default
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
Definition VPlan.h:2937
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
Definition VPlan.h:2928
VPReductionPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2910
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2921
~VPReductionPHIRecipe() override=default
bool isExpressionSunk() const
Definition VPlan.h:2952
bool hasUsesOutsideReductionChain() const
Returns true, if the phi is part of a multi-use reduction.
Definition VPlan.h:2946
VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start, VPValue &BackedgeValue, ReductionStyle Style, const VPIRFlags &Flags, bool HasUsesOutsideReductionChain=false)
Create a new VPReductionPHIRecipe for the reduction Phi.
Definition VPlan.h:2888
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2940
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2955
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
VPReductionPHIRecipe * cloneWithOperands(VPValue *Start, VPValue *BackedgeValue)
Definition VPlan.h:2901
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
Definition VPlan.h:2934
A recipe to represent inloop, ordered or partial reduction operations.
Definition VPlan.h:3248
bool isConditional() const
Return true if the in-loop reduction is conditional.
Definition VPlan.h:3332
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:3301
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:3316
VPValue * getVecOp() const
The VPValue of the vector value to be reduced.
Definition VPlan.h:3345
VPValue * getCondOp() const
The VPValue of the condition for the block.
Definition VPlan.h:3347
RecurKind getRecurrenceKind() const
Return the recurrence kind for the in-loop reduction.
Definition VPlan.h:3328
VPReductionRecipe(RecurKind RdxKind, FastMathFlags FMFs, Instruction *I, VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp, ReductionStyle Style, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3281
bool isOrdered() const
Return true if the in-loop reduction is ordered.
Definition VPlan.h:3330
VPReductionRecipe(const RecurKind RdxKind, FastMathFlags FMFs, VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp, ReductionStyle Style, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3287
VPReductionRecipe(VPRecipeTy SC, RecurKind RdxKind, FastMathFlags FMFs, Instruction *I, ArrayRef< VPValue * > Operands, VPValue *CondOp, ReductionStyle Style, DebugLoc DL)
Definition VPlan.h:3257
bool isPartialReduction() const
Returns true if the reduction outputs a vector with a scaled down VF.
Definition VPlan.h:3334
~VPReductionRecipe() override=default
VPValue * getChainOp() const
The VPValue of the scalar Chain being accumulated.
Definition VPlan.h:3343
bool isInLoop() const
Returns true if the reduction is in-loop.
Definition VPlan.h:3338
VPReductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3295
static bool classof(const VPUser *U)
Definition VPlan.h:3306
static bool classof(const VPValue *VPV)
Definition VPlan.h:3311
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:3352
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4657
const VPBlockBase * getEntry() const
Definition VPlan.h:4701
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4733
~VPRegionBlock() override=default
VPRegionValue * createHeaderMask()
Create the header mask for the region and return it.
Definition VPlan.h:4804
VPRegionValue * getUsedHeaderMask() const
Return the header mask if it exists and is used, or null otherwise.
Definition VPlan.h:4797
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4718
VPBlockBase * getExiting()
Definition VPlan.h:4714
VPBranchOnMaskRecipe * getEntryBranchOnMask()
Definition VPlan.h:4738
const VPRegionValue * getCanonicalIV() const
Definition VPlan.h:4780
SmallVector< VPRegionValue *, 2 > getRegionValues() const
Return the region values of the loop region (canonical IV, header mask) or an empty vector for replic...
Definition VPlan.h:4811
void setEntry(VPBlockBase *EntryBlock)
Set EntryBlock as the entry VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4706
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
Definition VPlan.h:4785
bool hasCanonicalIVNUW() const
Indicates if NUW is set for the canonical IV increment, for loop regions.
Definition VPlan.h:4821
void clearCanonicalIVNUW(VPInstruction *Increment)
Unsets NUW for the canonical IV increment Increment, for loop regions.
Definition VPlan.h:4824
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
Definition VPlan.h:4777
const VPBlockBase * getExiting() const
Definition VPlan.h:4713
VPBlockBase * getEntry()
Definition VPlan.h:4702
VPBasicBlock * getPreheaderVPBB()
Returns the pre-header VPBasicBlock of the loop region.
Definition VPlan.h:4726
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
Definition VPlan.h:4790
friend class VPlan
Definition VPlan.h:4658
static bool classof(const VPBlockBase *V)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:4697
VPValues are defined by a VPRegionBlock, like the canonical IV.
Definition VPlanValue.h:248
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
Definition VPlan.h:3415
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3474
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the recipe is predicated.
Definition VPlan.h:3508
VPReplicateRecipe(Instruction *I, ArrayRef< VPValue * > Operands, bool IsSingleScalar, VPValue *Mask=nullptr, const VPIRFlags &Flags={}, VPIRMetadata Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3423
~VPReplicateRecipe() override=default
static Type * computeScalarType(const Instruction *I, ArrayRef< VPValue * > Operands)
Compute the scalar result type for a VPReplicateRecipe wrapping I with Operands (excluding any predic...
VPReplicateRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:3447
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe uses scalars of operand Op.
Definition VPlan.h:3489
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
Definition VPlan.h:3502
bool isPredicated() const
Definition VPlan.h:3479
VPReplicateRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3445
bool doesGeneratePerAllLanes() const
Returns true if the recipe produces scalar values for all VF lanes.
Definition VPlan.h:3477
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3482
unsigned getOpcode() const
Definition VPlan.h:3512
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Definition VPlan.h:3496
Instruction::BinaryOps getInductionOpcode() const
Definition VPlan.h:4335
VPValue * getStepValue() const
Definition VPlan.h:4305
void setStartIndex(VPValue *StartIndex)
Set or add the StartIndex operand.
Definition VPlan.h:4318
VPScalarIVStepsRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4287
VPValue * getStartIndex() const
Return the StartIndex, or null if known to be zero, valid only after unrolling.
Definition VPlan.h:4313
VPValue * getVFValue() const
Return the number of scalars to produce per unroll part, used to compute StartIndex during unrolling.
Definition VPlan.h:4309
VPScalarIVStepsRecipe(VPValue *IV, VPValue *Step, VPValue *VF, Instruction::BinaryOps Opcode, FastMathFlags FMFs={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:4278
~VPScalarIVStepsRecipe() override=default
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4329
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
static bool classof(const VPValue *V)
Definition VPlan.h:668
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:625
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Value *UV, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:617
const Instruction * getUnderlyingInstr() const
Definition VPlan.h:684
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Type *ResultTy, Value *UV=nullptr, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:621
static bool classof(const VPUser *U)
Definition VPlan.h:673
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:613
VPSingleDefValue(VPSingleDefRecipe *Def, Value *UV=nullptr, Type *Ty=nullptr)
Construct a VPSingleDefValue. Must only be used by VPSingleDefRecipe.
Definition VPlan.cpp:167
This class can be used to assign names to VPValues.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:213
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:397
void printOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the operands to O.
Definition VPlan.cpp:1509
operand_range operands()
Definition VPlanValue.h:473
void setOperand(unsigned I, VPValue *New)
Definition VPlanValue.h:446
unsigned getNumOperands() const
Definition VPlanValue.h:437
operand_iterator op_end()
Definition VPlanValue.h:471
operand_iterator op_begin()
Definition VPlanValue.h:469
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:438
VPUser(ArrayRef< VPValue * > Operands)
Definition VPlanValue.h:418
iterator_range< const_operand_iterator > const_operand_range
Definition VPlanValue.h:467
VPValue * getLastOperand() const
Returns the last operand.
Definition VPlanValue.h:444
iterator_range< operand_iterator > operand_range
Definition VPlanValue.h:466
void addOperand(VPValue *Operand)
Definition VPlanValue.h:423
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
bool user_empty() const
Definition VPlanValue.h:161
void setUnderlyingValue(Value *Val)
Definition VPlanValue.h:205
unsigned getNumUsers() const
Definition VPlanValue.h:115
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:2328
VPValue * getVFValue() const
Definition VPlan.h:2309
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Type * getSourceElementType() const
Definition VPlan.h:2306
int64_t getStride() const
Definition VPlan.h:2307
VPVectorEndPointerRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2349
VPValue * getOffset() const
Definition VPlan.h:2310
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
Definition VPlan.h:2342
void addOffset(VPValue *Offset)
Append Offset as the offset operand.
Definition VPlan.h:2320
VPVectorEndPointerRecipe(VPValue *Ptr, VPValue *VF, Type *SourceElementTy, int64_t Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
Definition VPlan.h:2296
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPVectorPointerRecipe.
Definition VPlan.h:2335
VPValue * getPointer() const
Definition VPlan.h:2308
void materializeOffset(unsigned Part=0)
Adds the offset operand to the recipe.
void addPerPartOffset(VPValue *VFxPart)
Add the per-part offset (VFxPart) used for unrolled parts > 0.
Definition VPlan.h:2390
VPValue * getStride() const
Definition VPlan.h:2383
Type * getSourceElementType() const
Definition VPlan.h:2398
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:2400
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
Definition VPlan.h:2407
VPVectorPointerRecipe(VPValue *Ptr, Type *SourceElementTy, VPValue *Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
Definition VPlan.h:2374
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHeaderPHIRecipe.
Definition VPlan.h:2424
VPVectorPointerRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2414
VPValue * getVFxPart() const
Definition VPlan.h:2385
A recipe for widening Call instructions using library calls.
Definition VPlan.h:2115
VPWidenCallRecipe(Value *UV, Function *Variant, ArrayRef< VPValue * > CallArguments, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:2122
const_operand_range args() const
Definition VPlan.h:2162
VPWidenCallRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2140
operand_range args()
Definition VPlan.h:2161
Function * getCalledScalarFunction() const
Definition VPlan.h:2157
~VPWidenCallRecipe() override=default
~VPWidenCanonicalIVRecipe() override=default
VPValue * getStepValue() const
Definition VPlan.h:4191
void addPerPartStep(VPValue *Step)
Add the per-part step (VF * Part) used for unrolled parts.
Definition VPlan.h:4196
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCanonicalIVPHIRecipe.
Definition VPlan.h:4180
VPRegionValue * getCanonicalIV() const
Return the canonical IV being widened.
Definition VPlan.h:4187
VPWidenCanonicalIVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4165
VPWidenCanonicalIVRecipe(VPRegionValue *CanonicalIV, const VPIRFlags::WrapFlagsTy &Flags={})
Definition VPlan.h:4158
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4175
VPWidenCastRecipe is a recipe to create vector cast instructions.
Definition VPlan.h:1896
Instruction::CastOps getOpcode() const
Definition VPlan.h:1932
~VPWidenCastRecipe() override=default
VPWidenCastRecipe(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, CastInst *CI=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1901
VPWidenCastRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1917
unsigned getOpcode() const
This recipe generates a GEP instruction.
Definition VPlan.h:2258
Type * getSourceElementType() const
Definition VPlan.h:2263
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenGEPRecipe.
Definition VPlan.h:2266
VPWidenGEPRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2249
~VPWidenGEPRecipe() override=default
VPWidenGEPRecipe(Type *SourceElementTy, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, DebugLoc DL=DebugLoc::getUnknown(), GetElementPtrInst *UV=nullptr)
Definition VPlan.h:2232
void execute(VPTransformState &State) override=0
Generate the phi nodes.
ArrayRef< const SCEVPredicate * > getNoWrapPredicates() const
Returns the SCEV predicates associated with this induction.
Definition VPlan.h:2588
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2600
static bool classof(const VPValue *V)
Definition VPlan.h:2556
VPValue * getBackedgeValue() override
Returns the incoming value from the loop backedge.
Definition VPlan.h:2592
unsigned getNumIncoming() const override
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:2577
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
Definition VPlan.h:2580
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2568
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2585
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:2551
VPWidenInductionRecipe(VPRecipeTy Kind, PHINode *IV, VPValue *Start, VPValue *Step, const InductionDescriptor &IndDesc, Type *ResultTy, DebugLoc DL)
Definition VPlan.h:2530
const VPValue * getVFValue() const
Definition VPlan.h:2572
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2561
const VPValue * getStepValue() const
Definition VPlan.h:2569
void addUnrolledPartOperands(VPValue *SplatVFStep, VPValue *LastPart)
After unrolling, append the splat-VF step (VF * step) and the value of the induction at the last unro...
Definition VPlan.h:2539
const TruncInst * getTruncInst() const
Definition VPlan.h:2675
void execute(VPTransformState &State) override
Generate the phi nodes.
Definition VPlan.h:2656
~VPWidenIntOrFpInductionRecipe() override=default
VPWidenIntOrFpInductionRecipe(PHINode *IV, VPValue *Start, VPValue *Step, VPValue *VF, const InductionDescriptor &IndDesc, TruncInst *Trunc, const VPIRFlags &Flags, DebugLoc DL)
Definition VPlan.h:2631
VPValue * getSplatVFValue() const
If the recipe has been unrolled, return the VPValue for the induction increment, otherwise return nul...
Definition VPlan.h:2663
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenIntOrFpInductionRecipe.
VPWidenIntOrFpInductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2648
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2674
VPWidenIntOrFpInductionRecipe(PHINode *IV, VPValue *Start, VPValue *Step, VPValue *VF, const InductionDescriptor &IndDesc, const VPIRFlags &Flags, DebugLoc DL)
Definition VPlan.h:2621
VPValue * getLastUnrolledPartOperand()
Returns the VPValue representing the value of this induction at the last unrolled part,...
Definition VPlan.h:2689
unsigned getNumIncoming() const override
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:2670
bool isCanonical() const
Returns true if the induction is canonical, i.e.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
A recipe for widening vector intrinsics.
Definition VPlan.h:1944
VPWidenIntrinsicRecipe(VPRecipeTy SC, Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1958
VPWidenIntrinsicRecipe(Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1993
Intrinsic::ID getVectorIntrinsicID() const
Return the ID of the intrinsic.
Definition VPlan.h:2047
bool mayReadFromMemory() const
Returns true if the intrinsic may read from memory.
Definition VPlan.h:2053
VPWidenIntrinsicRecipe(CallInst &CI, Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1979
bool mayHaveSideEffects() const
Returns true if the intrinsic may have side-effects.
Definition VPlan.h:2059
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2029
static bool classof(const VPValue *V)
Definition VPlan.h:2024
VPWidenIntrinsicRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2004
bool mayWriteToMemory() const
Returns true if the intrinsic may write to memory.
Definition VPlan.h:2056
~VPWidenIntrinsicRecipe() override=default
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:2014
static bool classof(const VPUser *U)
Definition VPlan.h:2019
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
void execute(VPTransformState &State) override
Produce a widened version of the vector memory intrinsic.
~VPWidenMemIntrinsicRecipe() override=default
VPWidenMemIntrinsicRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2092
VPWidenMemIntrinsicRecipe(Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, Align Alignment, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2077
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector memory intrinsic.
A common mixin class for widening memory operations.
Definition VPlan.h:3768
bool IsMasked
Whether the memory access is masked.
Definition VPlan.h:3779
bool isConsecutive() const
Return whether the loaded-from / stored-to addresses are consecutive.
Definition VPlan.h:3804
virtual ~VPWidenMemoryRecipe()=default
Instruction & Ingredient
Definition VPlan.h:3770
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Return the cost of this VPWidenMemoryRecipe.
Instruction & getIngredient() const
Definition VPlan.h:3826
bool Consecutive
Whether the accessed addresses are consecutive.
Definition VPlan.h:3776
virtual const VPRecipeBase * getAsRecipe() const =0
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3814
Align Alignment
Alignment information for this memory access.
Definition VPlan.h:3773
VPWidenMemoryRecipe(Instruction &I, bool Consecutive, const VPIRMetadata &Metadata)
Definition VPlan.h:3791
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
bool isMasked() const
Returns true if the recipe is masked.
Definition VPlan.h:3810
void setMask(VPValue *Mask)
Definition VPlan.h:3781
Align getAlign() const
Returns the alignment of the memory access.
Definition VPlan.h:3821
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3807
A recipe for widened phis.
Definition VPlan.h:2752
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:2797
unsigned getOpcode() const
This recipe generates a PHI.
Definition VPlan.h:2779
VPWidenPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2772
~VPWidenPHIRecipe() override=default
VPWidenPHIRecipe(ArrayRef< VPValue * > IncomingValues, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new VPWidenPHIRecipe with incoming values IncomingValues, debug location DL and Name.
Definition VPlan.h:2759
VPWidenPointerInductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2717
~VPWidenPointerInductionRecipe() override=default
bool onlyScalarsGenerated(bool IsScalable)
Returns true if only scalar values will be generated.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate vector values for the pointer induction.
Definition VPlan.h:2726
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPointerInductionRecipe.
VPWidenPointerInductionRecipe(PHINode *Phi, VPValue *Start, VPValue *Step, VPValue *NumUnrolledElems, const InductionDescriptor &IndDesc, DebugLoc DL)
Create a new VPWidenPointerInductionRecipe for Phi with start value Start and the number of elements ...
Definition VPlan.h:2706
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
Definition VPlan.h:1829
VPWidenRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1855
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:1884
VPWidenRecipe(Instruction &I, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:1833
VPWidenRecipe(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:1840
~VPWidenRecipe() override=default
VPWidenRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:1857
unsigned getOpcode() const
Definition VPlan.h:1874
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4844
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
Definition VPlan.h:5187
LLVM_ABI_FOR_TEST void printDOT(raw_ostream &O) const
Print this VPlan in DOT format to O.
Definition VPlan.cpp:1160
friend class VPSlotTracker
Definition VPlan.h:4846
std::string getName() const
Return a string with the name of the plan and the applicable VFs and UFs.
Definition VPlan.cpp:1136
bool hasVF(ElementCount VF) const
Definition VPlan.h:5080
ElementCount getSingleVF() const
Returns the single VF of the plan, asserting that the plan has exactly one VF.
Definition VPlan.h:5093
const DataLayout & getDataLayout() const
Definition VPlan.h:5058
LLVMContext & getContext() const
Definition VPlan.h:5054
VPBasicBlock * getEntry()
Definition VPlan.h:4940
Type * getIndexType() const
The type of the canonical induction variable of the vector loop.
Definition VPlan.h:5293
void setName(const Twine &newName)
Definition VPlan.h:5126
bool hasScalableVF() const
Definition VPlan.h:5081
VPValue * getTripCount() const
The trip count of the original loop.
Definition VPlan.h:5012
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
Definition VPlan.h:5033
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
Definition VPlan.h:5087
LLVM_ABI_FOR_TEST ~VPlan()
Definition VPlan.cpp:888
VPIRValue * getOrAddLiveIn(VPIRValue *V)
Definition VPlan.h:5144
bool isExitBlock(VPBlockBase *VPBB)
Returns true if VPBB is an exit block.
Definition VPlan.cpp:907
const VPBasicBlock * getEntry() const
Definition VPlan.h:4941
friend class VPlanPrinter
Definition VPlan.h:4845
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
Definition VPlan.h:5153
VPIRValue * getConstantInt(const APInt &Val)
Return a VPIRValue wrapping a ConstantInt with the given APInt value.
Definition VPlan.h:5176
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
Definition VPlan.h:5052
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
Definition VPlan.h:5159
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
Definition VPlan.h:5240
VPIRBasicBlock * createEmptyVPIRBasicBlock(BasicBlock *IRBB)
Create a VPIRBasicBlock wrapping IRBB, but do not create VPIRInstructions wrapping the instructions i...
Definition VPlan.cpp:1300
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
Definition VPlan.h:5190
bool hasUF(unsigned UF) const
Definition VPlan.h:5105
VPIRValue * getPoison(Type *Ty)
Return a VPIRValue wrapping a poison value of type Ty.
Definition VPlan.h:5181
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
Definition VPlan.h:5006
VPlan(BasicBlock *ScalarHeaderBB, Type *IdxTy)
Construct a VPlan with a new VPBasicBlock as entry, a VPIRBasicBlock wrapping ScalarHeaderBB and vect...
Definition VPlan.h:4921
VPSymbolicValue & getVectorTripCount()
The vector trip count.
Definition VPlan.h:5042
VPValue * getBackedgeTakenCount() const
Definition VPlan.h:5039
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
Definition VPlan.h:5130
VPRegionBlock * createLoopRegion(Type *CanIVTy, DebugLoc DL, const std::string &Name="", VPBlockBase *Entry=nullptr, VPBlockBase *Exiting=nullptr)
Create a new loop region with a canonical IV using CanIVTy and DL.
Definition VPlan.h:5226
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
Definition VPlan.h:5156
void setVF(ElementCount VF)
Definition VPlan.h:5068
unsigned getMaxBlockNumber() const
Definition VPlan.h:5260
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
Definition VPlan.h:5121
Function * getIRFunction() const
Definition VPlan.h:5062
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1042
bool hasEarlyExit() const
Returns true if the VPlan is based on a loop with an early exit.
Definition VPlan.h:5263
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this plan.
Definition VPlan.cpp:1024
LLVM_ABI_FOR_TEST bool isOuterLoop() const
Returns true if this VPlan is for an outer loop, i.e., its vector loop region contains a nested loop ...
Definition VPlan.cpp:1066
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
Definition VPlan.h:5108
VPIRValue * getConstantInt(unsigned BitWidth, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given bitwidth and value.
Definition VPlan.h:5170
const VPBasicBlock * getMiddleBlock() const
Definition VPlan.h:4991
void setTripCount(VPValue *NewTripCount)
Set the trip count assuming it is currently null; if it is not - use resetTripCount().
Definition VPlan.h:5019
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
Definition VPlan.h:5026
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
Definition VPlan.h:4982
void setEntry(VPBasicBlock *VPBB)
Definition VPlan.h:4929
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
Definition VPlan.h:5213
LLVM_ABI_FOR_TEST VPIRBasicBlock * createVPIRBasicBlock(BasicBlock *IRBB)
Create a VPIRBasicBlock from IRBB containing VPIRInstructions for all instructions in IRBB,...
Definition VPlan.cpp:1308
void removeVF(ElementCount VF)
Remove VF from the plan.
Definition VPlan.h:5075
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
Definition VPlan.h:5150
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
Definition VPlan.h:4945
bool requiresScalarEpilogue() const
Returns true if the plan requires a scalar epilogue after the vector loop.
Definition VPlan.h:4968
LLVM_DUMP_METHOD void dump() const
Dump the plan to stderr (for debugging).
Definition VPlan.cpp:1166
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
Definition VPlan.h:5049
bool hasScalarVFOnly() const
Definition VPlan.h:5098
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
Definition VPlan.h:4996
void execute(VPTransformState *State)
Generate the IR code for this VPlan.
Definition VPlan.cpp:917
LLVM_ABI_FOR_TEST void print(raw_ostream &O) const
Print this VPlan to O.
Definition VPlan.cpp:1119
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
Definition VPlan.h:4961
void addVF(ElementCount VF)
Definition VPlan.h:5066
VPIRBasicBlock * getScalarHeader() const
Return the VPIRBasicBlock wrapping the header of the scalar loop.
Definition VPlan.h:5002
void printLiveIns(raw_ostream &O) const
Print the live-ins of this VPlan to O.
Definition VPlan.cpp:1075
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
Definition VPlan.h:5045
void setUF(unsigned UF)
Definition VPlan.h:5113
const VPSymbolicValue & getVF() const
Definition VPlan.h:5046
bool hasScalarTail() const
Returns true if the scalar tail may execute after the vector loop, i.e.
Definition VPlan.h:5286
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
Definition VPlan.cpp:1207
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5164
LLVM Value Representation.
Definition Value.h:75
Increasing range of size_t indices.
Definition STLExtras.h:2523
typename base_list_type::const_reverse_iterator const_reverse_iterator
Definition ilist.h:124
typename base_list_type::reverse_iterator reverse_iterator
Definition ilist.h:123
typename base_list_type::const_iterator const_iterator
Definition ilist.h:122
An intrusive list with ownership and callbacks specified/controlled by ilist_traits,...
Definition ilist.h:328
A range adaptor for a pair of iterators.
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
This file defines classes to implement an intrusive doubly linked list class (i.e.
This file defines the ilist_node class template, which is a convenient base class for creating classe...
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
Definition DwarfDebug.h:190
CastInfo helper for casting from VPRecipeBase to a mixin class that is not part of the VPRecipeBase c...
Definition VPlan.h:4348
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
LLVM_PACKED_END
Definition VPlan.h:1112
auto cast_if_present(const Y &Val)
cast_if_present<X> - Functionally identical to cast, except that a null value is accepted.
Definition Casting.h:683
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
Definition VPlan.h:2852
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
Type * toScalarizedTy(Type *Ty)
A helper for converting vectorized types to scalarized (non-vector) types.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
LLVM_ABI void getMetadataToPropagate(Instruction *Inst, SmallVectorImpl< std::pair< unsigned, MDNode * > > &Metadata)
Add metadata from Inst to Metadata, if it can be preserved after vectorization.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
auto cast_or_null(const Y &Val)
Definition Casting.h:714
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:83
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
Definition VPlan.h:92
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool isPointerTy(const Type *T)
Definition SPIRVUtils.h:383
LLVM_ABI Type * computeScalarTypeForInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands)
Compute the scalar result type for an IR Opcode given Operands.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
@ Other
Any other memory.
Definition ModRef.h:68
RecurKind
These are the kinds of recurrences that we support.
@ Mul
Product of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2028
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
auto sum_of(R &&Range, E Init=E{0})
Returns the sum of all values in Range with Init initial value.
Definition STLExtras.h:1733
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
std::variant< RdxOrdered, RdxInLoop, RdxUnordered > ReductionStyle
Definition VPlan.h:2850
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
std::unique_ptr< VPlan > VPlanPtr
Definition VPlan.h:78
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static Bitfield::Type get(StorageType Packed)
Unpacks the field from the Packed value.
Definition Bitfields.h:207
static void set(StorageType &Packed, typename Bitfield::Type Value)
Sets the typed value in the provided Packed value.
Definition Bitfields.h:223
This struct provides a method for customizing the way a cast is performed.
Definition Casting.h:476
Provides a cast trait that strips const from types to make it easier to implement a const-version of ...
Definition Casting.h:388
This cast trait just provides the default implementation of doCastIfPossible to make CastInfo special...
Definition Casting.h:309
Provides a cast trait that uses a defined pointer to pointer cast as a base for reference-to-referenc...
Definition Casting.h:423
This reduction is in-loop.
Definition VPlan.h:2844
Possible variants of a reduction.
Definition VPlan.h:2842
This reduction is unordered with the partial result scaled down by some factor.
Definition VPlan.h:2847
unsigned VFScaleFactor
Definition VPlan.h:2848
A MapVector that performs no allocations if smaller than a certain size.
Definition MapVector.h:342
Default inserter for VPBuilderBase, inserting R at It in VPBB.
An overlay on VPConstant for VPValues that wrap a ConstantInt.
Definition VPlanValue.h:306
Struct to hold various analysis needed for cost computations.
const BlockFrequency Freq
Definition VPlan.h:1173
VPExecutionFrequency(BlockFrequency Freq, bool IsEstimated)
Definition VPlan.h:1176
void execute(VPTransformState &State) override
Generate the phi nodes.
VPFirstOrderRecurrencePHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2813
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this first-order recurrence phi recipe.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2825
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPFirstOrderRecurrencePHIRecipe(PHINode *Phi, VPValue &Start, VPValue &BackedgeValue)
Definition VPlan.h:2804
DisjointFlagsTy(bool IsDisjoint)
Definition VPlan.h:732
NonNegFlagsTy(bool IsNonNeg)
Definition VPlan.h:737
TruncFlagsTy(bool HasNUW, bool HasNSW)
Definition VPlan.h:727
WrapFlagsTy(bool HasNUW, bool HasNSW)
Definition VPlan.h:716
WrapFlagsTy withoutNoSignedWrap()
Definition VPlan.h:718
An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use cast/dyn_cast/isa and exec...
Definition VPlan.h:1797
VPIRPhi(PHINode &PN)
Definition VPlan.h:1798
static bool classof(const VPRecipeBase *U)
Definition VPlan.h:1800
static bool classof(const VPUser *U)
Definition VPlan.h:1805
PHINode & getIRPhi() const
Definition VPlan.h:1810
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:1821
A VPValue representing a live-in from the input IR or a constant.
Definition VPlanValue.h:275
static bool classof(const VPUser *U)
Definition VPlan.h:1697
VPPhi * clone() override
Clone the current recipe.
Definition VPlan.h:1712
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:1727
static bool classof(const VPSingleDefRecipe *SDR)
Definition VPlan.h:1707
static bool classof(const VPValue *V)
Definition VPlan.h:1702
VPPhi(ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL, const Twine &Name="", Type *ResultTy=nullptr)
Definition VPlan.h:1692
A pure-virtual common base class for recipes defining a single VPValue and using IR flags.
Definition VPlan.h:1116
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1117
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:1158
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:1128
InstructionCost getCostForRecipeWithOpcode(unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const
Compute the cost for this recipe for VF, using Opcode and Ctx.
static bool classof(const VPValue *V)
Definition VPlan.h:1151
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Type *ResultTy, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1122
void execute(VPTransformState &State) override=0
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPRecipeWithIRFlags * clone() override=0
Clone the current recipe.
static bool classof(const VPUser *U)
Definition VPlan.h:1146
VPTransformState holds information passed down when "executing" a VPlan, needed for generating the ou...
A recipe for widening load operations with vector-predication intrinsics, using the address to load f...
Definition VPlan.h:3885
VPWidenLoadEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3895
unsigned getOpcode() const
Returns the opcode of the widened load.
Definition VPlan.h:3902
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3905
VPWidenLoadEVLRecipe(VPWidenLoadRecipe &L, VPValue *Addr, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3886
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3915
A recipe for widening load operations, using the address to load from and an optional mask.
Definition VPlan.h:3832
VPWidenLoadRecipe(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Definition VPlan.h:3833
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3861
unsigned getOpcode() const
Returns the opcode of the widened load.
Definition VPlan.h:3849
VPWidenLoadRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3841
VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC)
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenLoadRecipe.
Definition VPlan.h:3855
A recipe for widening store operations with vector-predication intrinsics, using the value to store,...
Definition VPlan.h:3991
VPValue * getStoredValue() const
Return the address accessed by this recipe.
Definition VPlan.h:4007
VPWidenStoreEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4000
VPWidenStoreEVLRecipe(VPWidenStoreRecipe &S, VPValue *Addr, VPValue *StoredVal, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3992
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4020
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:4010
A recipe for widening store operations, using the stored value, the address to store to and an option...
Definition VPlan.h:3937
VPWidenStoreRecipe(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Definition VPlan.h:3938
VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC)
VPValue * getStoredValue() const
Return the value stored by this recipe.
Definition VPlan.h:3955
VPWidenStoreRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3946
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenStoreRecipe.
Definition VPlan.h:3961
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3967
static VPMixin * castFailed()
Definition VPlan.h:4366
static bool isPossible(VPRecipeBase *R)
Used by isa.
Definition VPlan.h:4357
static VPMixin * doCast(VPRecipeBase *R)
Used by cast.
Definition VPlan.h:4360