LLVM 24.0.0git
VPlanRecipes.cpp
Go to the documentation of this file.
1//===- VPlanRecipes.cpp - Implementations for VPlan recipes ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains implementations for different VPlan recipes.
11///
12//===----------------------------------------------------------------------===//
13
15#include "VPlan.h"
16#include "VPlanHelpers.h"
17#include "VPlanPatternMatch.h"
18#include "VPlanUtils.h"
19#include "llvm/ADT/APFloat.h"
20#include "llvm/ADT/STLExtras.h"
23#include "llvm/ADT/Twine.h"
29#include "llvm/IR/BasicBlock.h"
30#include "llvm/IR/IRBuilder.h"
31#include "llvm/IR/Instruction.h"
33#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/Type.h"
36#include "llvm/IR/Value.h"
39#include "llvm/Support/Debug.h"
43#include <cassert>
44
45using namespace llvm;
46using namespace llvm::VPlanPatternMatch;
47
48#define LV_NAME "loop-vectorize"
49#define DEBUG_TYPE LV_NAME
50
51namespace llvm {
53} // namespace llvm
54
56 switch (getVPRecipeID()) {
57 case VPExpressionSC:
58 return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
59 case VPInstructionSC: {
60 auto *VPI = cast<VPInstruction>(this);
61 // Loads read from memory but don't write to memory.
62 if (VPI->getOpcode() == Instruction::Load ||
63 VPI->getOpcode() == VPInstruction::WideVectorLoad)
64 return false;
65 return VPI->opcodeMayReadOrWriteFromMemory();
66 }
67 case VPInterleaveEVLSC:
68 case VPInterleaveSC:
69 return cast<VPInterleaveBase>(this)->getNumStoreOperands() > 0;
70 case VPWidenStoreEVLSC:
71 case VPWidenStoreSC:
72 return true;
73 case VPReplicateSC:
74 return cast<Instruction>(getVPSingleValue()->getUnderlyingValue())
75 ->mayWriteToMemory();
76 case VPWidenCallSC:
77 return !cast<VPWidenCallRecipe>(this)
78 ->getCalledScalarFunction()
79 ->onlyReadsMemory();
80 case VPWidenMemIntrinsicSC:
81 case VPWidenIntrinsicSC:
82 return cast<VPWidenIntrinsicRecipe>(this)->mayWriteToMemory();
83 case VPActiveLaneMaskPHISC:
84 case VPCurrentIterationPHISC:
85 case VPBranchOnMaskSC:
86 case VPDerivedIVSC:
87 case VPFirstOrderRecurrencePHISC:
88 case VPReductionPHISC:
89 case VPScalarIVStepsSC:
90 case VPPredInstPHISC:
91 case VPExpandSCEVSC:
92 return false;
93 case VPBlendSC:
94 case VPReductionEVLSC:
95 case VPReductionSC:
96 case VPVectorPointerSC:
97 case VPWidenCanonicalIVSC:
98 case VPWidenCastSC:
99 case VPWidenGEPSC:
100 case VPWidenIntOrFpInductionSC:
101 case VPWidenLoadEVLSC:
102 case VPWidenLoadSC:
103 case VPWidenPHISC:
104 case VPWidenPointerInductionSC:
105 case VPWidenSC: {
106 const Instruction *I =
107 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
108 (void)I;
109 assert((!I || !I->mayWriteToMemory()) &&
110 "underlying instruction may write to memory");
111 return false;
112 }
113 default:
114 return true;
115 }
116}
117
119 switch (getVPRecipeID()) {
120 case VPExpressionSC:
121 return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
122 case VPInstructionSC: {
123 auto *VPI = cast<VPInstruction>(this);
124 // Stores write to memory but don't read from memory.
125 if (VPI->getOpcode() == VPInstruction::WideVectorStore)
126 return false;
127 return VPI->opcodeMayReadOrWriteFromMemory();
128 }
129 case VPWidenLoadEVLSC:
130 case VPWidenLoadSC:
131 return true;
132 case VPReplicateSC:
133 return cast<Instruction>(getVPSingleValue()->getUnderlyingValue())
134 ->mayReadFromMemory();
135 case VPWidenCallSC:
136 return !cast<VPWidenCallRecipe>(this)
137 ->getCalledScalarFunction()
138 ->onlyWritesMemory();
139 case VPWidenMemIntrinsicSC:
140 case VPWidenIntrinsicSC:
141 return cast<VPWidenIntrinsicRecipe>(this)->mayReadFromMemory();
142 case VPBranchOnMaskSC:
143 case VPDerivedIVSC:
144 case VPCurrentIterationPHISC:
145 case VPFirstOrderRecurrencePHISC:
146 case VPReductionPHISC:
147 case VPPredInstPHISC:
148 case VPScalarIVStepsSC:
149 case VPWidenStoreEVLSC:
150 case VPWidenStoreSC:
151 case VPExpandSCEVSC:
152 return false;
153 case VPBlendSC:
154 case VPReductionEVLSC:
155 case VPReductionSC:
156 case VPVectorPointerSC:
157 case VPWidenCanonicalIVSC:
158 case VPWidenCastSC:
159 case VPWidenGEPSC:
160 case VPWidenIntOrFpInductionSC:
161 case VPWidenPHISC:
162 case VPWidenPointerInductionSC:
163 case VPWidenSC: {
164 const Instruction *I =
165 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
166 (void)I;
167 assert((!I || !I->mayReadFromMemory()) &&
168 "underlying instruction may read from memory");
169 return false;
170 }
171 default:
172 // FIXME: Return false if the recipe represents an interleaved store.
173 return true;
174 }
175}
176
178 switch (getVPRecipeID()) {
179 case VPExpressionSC:
180 return cast<VPExpressionRecipe>(this)->mayHaveSideEffects();
181 case VPActiveLaneMaskPHISC:
182 case VPDerivedIVSC:
183 case VPCurrentIterationPHISC:
184 case VPFirstOrderRecurrencePHISC:
185 case VPReductionPHISC:
186 case VPPredInstPHISC:
187 case VPVectorEndPointerSC:
188 case VPExpandSCEVSC:
189 return false;
190 case VPInstructionSC: {
191 auto *VPI = cast<VPInstruction>(this);
192 return mayWriteToMemory() ||
193 VPI->getOpcode() == VPInstruction::BranchOnCount ||
194 VPI->getOpcode() == VPInstruction::BranchOnCond ||
195 VPI->getOpcode() == VPInstruction::BranchOnTwoConds;
196 }
197 case VPWidenCallSC: {
198 Function *Fn = cast<VPWidenCallRecipe>(this)->getCalledScalarFunction();
199 return mayWriteToMemory() || !Fn->doesNotThrow() || !Fn->willReturn();
200 }
201 case VPWidenMemIntrinsicSC:
202 case VPWidenIntrinsicSC:
203 return cast<VPWidenIntrinsicRecipe>(this)->mayHaveSideEffects();
204 case VPBlendSC:
205 case VPReductionEVLSC:
206 case VPReductionSC:
207 case VPScalarIVStepsSC:
208 case VPVectorPointerSC:
209 case VPWidenCanonicalIVSC:
210 case VPWidenCastSC:
211 case VPWidenGEPSC:
212 case VPWidenIntOrFpInductionSC:
213 case VPWidenPHISC:
214 case VPWidenPointerInductionSC:
215 case VPWidenSC: {
216 const Instruction *I =
217 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
218 (void)I;
219 assert((!I || !I->mayHaveSideEffects()) &&
220 "underlying instruction has side-effects");
221 return false;
222 }
223 case VPInterleaveEVLSC:
224 case VPInterleaveSC:
225 return mayWriteToMemory();
226 case VPWidenLoadEVLSC:
227 case VPWidenLoadSC:
228 case VPWidenStoreEVLSC:
229 case VPWidenStoreSC:
230 assert(
231 cast<VPWidenMemoryRecipe>(this)->getIngredient().mayHaveSideEffects() ==
233 "mayHaveSideffects result for ingredient differs from this "
234 "implementation");
235 return mayWriteToMemory();
236 case VPReplicateSC: {
237 auto *R = cast<VPReplicateRecipe>(this);
238 return R->getUnderlyingInstr()->mayHaveSideEffects();
239 }
240 default:
241 return true;
242 }
243}
244
246 switch (getVPRecipeID()) {
247 default:
248 return false;
249 case VPInstructionSC: {
250 unsigned Opcode = cast<VPInstruction>(this)->getOpcode();
251 if (Instruction::isCast(Opcode))
252 return true;
253
254 switch (Opcode) {
255 default:
256 return false;
257 case Instruction::Add:
258 case Instruction::Sub:
259 case Instruction::Mul:
260 case Instruction::GetElementPtr:
261 return true;
262 }
263 }
264 }
265}
266
268 assert(!Parent && "Recipe already in some VPBasicBlock");
269 assert(InsertPos->getParent() &&
270 "Insertion position not in any VPBasicBlock");
271 InsertPos->getParent()->insert(this, InsertPos->getIterator());
272}
273
274void VPRecipeBase::insertBefore(VPBasicBlock &BB,
276 assert(!Parent && "Recipe already in some VPBasicBlock");
277 assert(I == BB.end() || I->getParent() == &BB);
278 BB.insert(this, I);
279}
280
282 assert(!Parent && "Recipe already in some VPBasicBlock");
283 assert(InsertPos->getParent() &&
284 "Insertion position not in any VPBasicBlock");
285 InsertPos->getParent()->insert(this, std::next(InsertPos->getIterator()));
286}
287
289 assert(getParent() && "Recipe not in any VPBasicBlock");
291 Parent = nullptr;
292}
293
295 assert(getParent() && "Recipe not in any VPBasicBlock");
297}
298
301 insertAfter(InsertPos);
302}
303
309
311 // Get the underlying instruction for the recipe, if there is one. It is used
312 // to
313 // * decide if cost computation should be skipped for this recipe,
314 // * apply forced target instruction cost.
315 Instruction *UI = nullptr;
316 if (auto *S = dyn_cast<VPSingleDefRecipe>(this))
317 UI = dyn_cast_or_null<Instruction>(S->getUnderlyingValue());
318 else if (auto *IG = dyn_cast<VPInterleaveBase>(this))
319 UI = IG->getInsertPos();
320 else if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(this))
321 UI = &WidenMem->getIngredient();
322
323 InstructionCost RecipeCost;
324 if (UI && Ctx.skipCostComputation(UI, VF.isVector())) {
325 RecipeCost = 0;
326 } else {
327 RecipeCost = computeCost(VF, Ctx);
328 if (ForceTargetInstructionCost.getNumOccurrences() > 0 &&
329 RecipeCost.isValid()) {
330 // VPDerivedIVRecipe and VPScalarIVStepsRecipe never have underlying
331 // instructions.
334 else
335 RecipeCost = InstructionCost(0);
336 }
337 }
338
339 LLVM_DEBUG({
340 dbgs() << "Cost of " << RecipeCost << " for VF " << VF << ": ";
341 if (VPSlotTracker *SlotTracker = Ctx.getSlotTracker()) {
342 print(dbgs(), "", *SlotTracker);
343 dbgs() << "\n";
344 } else {
345 dump();
346 }
347 });
348 return RecipeCost;
349}
350
352 VPCostContext &Ctx) const {
353 llvm_unreachable("subclasses should implement computeCost");
354}
355
357 return (getVPRecipeID() >= VPFirstPHISC && getVPRecipeID() <= VPLastPHISC) ||
359}
360
362 assert(OpType == Other.OpType && "OpType must match");
363 switch (OpType) {
364 case OperationType::OverflowingBinOp:
365 WrapFlags.HasNUW &= Other.WrapFlags.HasNUW;
366 WrapFlags.HasNSW &= Other.WrapFlags.HasNSW;
367 break;
368 case OperationType::Trunc:
369 TruncFlags.HasNUW &= Other.TruncFlags.HasNUW;
370 TruncFlags.HasNSW &= Other.TruncFlags.HasNSW;
371 break;
372 case OperationType::DisjointOp:
373 DisjointFlags.IsDisjoint &= Other.DisjointFlags.IsDisjoint;
374 break;
375 case OperationType::PossiblyExactOp:
376 ExactFlags.IsExact &= Other.ExactFlags.IsExact;
377 break;
378 case OperationType::GEPOp:
379 GEPFlagsStorage &= Other.GEPFlagsStorage;
380 break;
381 case OperationType::FPMathOp:
382 case OperationType::FCmp:
383 assert((OpType != OperationType::FCmp ||
384 FCmpFlags.CmpPredStorage == Other.FCmpFlags.CmpPredStorage) &&
385 "Cannot drop CmpPredicate");
386 getFMFsRef() = getFastMathFlagsOrNone() & Other.getFastMathFlagsOrNone();
387 break;
388 case OperationType::NonNegOp:
389 NonNegFlags.NonNeg &= Other.NonNegFlags.NonNeg;
390 break;
391 case OperationType::Cmp:
392 assert(CmpPredStorage == Other.CmpPredStorage &&
393 "Cannot drop CmpPredicate");
394 break;
395 case OperationType::ReductionOp:
396 assert(ReductionFlags.Kind == Other.ReductionFlags.Kind &&
397 "Cannot change RecurKind");
398 assert(ReductionFlags.IsOrdered == Other.ReductionFlags.IsOrdered &&
399 "Cannot change IsOrdered");
400 assert(ReductionFlags.IsInLoop == Other.ReductionFlags.IsInLoop &&
401 "Cannot change IsInLoop");
402 getFMFsRef() = getFastMathFlagsOrNone() & Other.getFastMathFlagsOrNone();
403 break;
404 case OperationType::Other:
405 break;
406 }
407}
408
410 if (!hasFastMathFlags())
411 return {};
412 const FastMathFlagsTy &F = getFMFsRef();
413 FastMathFlags Res;
414 Res.setAllowReassoc(F.AllowReassoc);
415 Res.setNoNaNs(F.NoNaNs);
416 Res.setNoInfs(F.NoInfs);
417 Res.setNoSignedZeros(F.NoSignedZeros);
418 Res.setAllowReciprocal(F.AllowReciprocal);
419 Res.setAllowContract(F.AllowContract);
420 Res.setApproxFunc(F.ApproxFunc);
421 return Res;
422}
423
424#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
426
427void VPRecipeBase::print(raw_ostream &O, const Twine &Indent,
428 VPSlotTracker &SlotTracker) const {
429 printRecipe(O, Indent, SlotTracker);
430 if (auto DL = getDebugLoc()) {
431 O << ", !dbg ";
432 DL.print(O);
433 }
434
435 if (auto *Metadata = dyn_cast<VPIRMetadata>(this))
437}
438#endif
439
441 : VPSingleDefRecipe(VPRecipeBase::VPExpandSCEVSC, {}, Expr->getType()),
442 Expr(Expr) {}
443
444/// For call VPInstruction operands, return the operand index of the called
445/// function. The function is either the last operand (for unmasked calls) or
446/// the second-to-last operand (for masked calls).
448 unsigned NumOps = Operands.size();
449 auto *LastOp = dyn_cast<VPIRValue>(Operands[NumOps - 1]);
450 if (LastOp && isa<Function>(LastOp->getValue()))
451 return NumOps - 1;
453 "expected function operand");
454 return NumOps - 2;
455}
456
457/// For call VPInstruction operands, return the called function.
462
465 assert(!Operands.empty() &&
466 "zero-operand VPInstruction opcodes must pass explicit ResultTy");
467 // Assert operand \p Idx (if present and typed) has type \p ExpectedTy.
468 [[maybe_unused]] auto AssertOperandType = [&Operands](unsigned Idx,
469 Type *ExpectedTy) {
470 if (!ExpectedTy || Operands.size() <= Idx)
471 return;
472 [[maybe_unused]] Type *OpTy = Operands[Idx]->getScalarType();
473 assert((!OpTy || OpTy == ExpectedTy) &&
474 "different types inferred for different operands");
475 };
476
477 Type *Op0Ty = Operands[0]->getScalarType();
478 LLVMContext &Ctx = Op0Ty->getContext();
479 switch (Opcode) {
481 assert(Op0Ty->isIntegerTy(1) && "expected bool condition");
482 return Type::getVoidTy(Ctx);
484 assert(Op0Ty->isIntegerTy(1) && "expected bool condition");
485 AssertOperandType(1, IntegerType::get(Ctx, 1));
486 return Type::getVoidTy(Ctx);
488 assert(Op0Ty->isIntegerTy() && "expected integer operand");
489 AssertOperandType(1, Op0Ty);
490 return Type::getVoidTy(Ctx);
492 assert(Op0Ty->isIntegerTy() && "expected integer operand");
493 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
494 AssertOperandType(Idx, Op0Ty);
495 return Op0Ty;
496 case Instruction::Switch:
497 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
498 AssertOperandType(Idx, Op0Ty);
499 return Type::getVoidTy(Ctx);
501 case Instruction::Store:
502 return Type::getVoidTy(Ctx);
503 case Instruction::ICmp:
504 assert(Op0Ty->isIntOrPtrTy() && "expected integer or pointer operand");
505 AssertOperandType(1, Op0Ty);
506 return IntegerType::get(Ctx, 1);
507 case Instruction::FCmp:
508 assert(Op0Ty->isFloatingPointTy() && "expected floating-point operand");
509 AssertOperandType(1, Op0Ty);
510 return IntegerType::get(Ctx, 1);
513 assert(Op0Ty->isIntegerTy() && "expected integer operand");
514 AssertOperandType(1, Op0Ty);
515 return IntegerType::get(Ctx, 1);
517 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
518 return IntegerType::get(Ctx, 1);
521 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
522 AssertOperandType(1, Op0Ty);
523 return IntegerType::get(Ctx, 1);
525 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
526 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
527 AssertOperandType(Idx, Op0Ty);
528 return IntegerType::get(Ctx, 1);
530 assert(Op0Ty->isIntegerTy() && "expected integer operand");
531 return IntegerType::get(Ctx, 32);
532 case Instruction::Select: {
533 assert((!Op0Ty || Op0Ty->isIntegerTy(1)) &&
534 "select condition must be bool");
535 Type *Op1Ty = Operands[1]->getScalarType();
536 AssertOperandType(2, Op1Ty);
537 return Op1Ty;
538 }
539 case Instruction::InsertElement:
540 // The inserted scalar (operand 1) must match the vector element type;
541 // operand 2 must be an integer.
542 AssertOperandType(1, Op0Ty);
543 assert(Operands[2]->getScalarType()->isIntegerTy() &&
544 "expected integer operand");
545 return Op0Ty;
547 // The start value and the identity value (operands 0 and 1) fill the same
548 // vector and must match in type; operand 2 is the scaling factor.
549 AssertOperandType(1, Op0Ty);
550 return Op0Ty;
552 assert(Operands.size() >= 2 && "ExtractLane requires a lane operand and "
553 "at least one source vector operand");
554 // Operand 0 is the lane index, used for integer arithmetic.
555 assert(Op0Ty->isIntegerTy() && "expected integer operand");
556 Type *Op1Ty = Operands[1]->getScalarType();
557 for (unsigned Idx = 2; Idx != Operands.size(); ++Idx)
558 AssertOperandType(Idx, Op1Ty);
559 return Op1Ty;
560 }
563 assert(Operands[0]->getScalarType()->isPointerTy() &&
564 "expected pointer operand");
565 assert(Operands[1]->getScalarType()->isIntegerTy() &&
566 "expected integer operand");
567 return Op0Ty;
568 case Instruction::ExtractValue: {
569 assert(Operands.size() == 2 && "expected single level extractvalue");
570 auto *StructTy = cast<StructType>(Op0Ty);
571 return StructTy->getTypeAtIndex(
572 cast<VPConstantInt>(Operands[1])->getZExtValue());
573 }
579 case Instruction::Load:
580 case Instruction::Alloca:
581 llvm_unreachable("type must be passed explicitly");
582 case Instruction::Call:
584 default:
585 if (Instruction::isCast(Opcode))
586 llvm_unreachable("type must be passed explicitly");
587 break;
588 }
589
590 // Opcodes that require all operands to share the same scalar type as the
591 // result.
592 bool AllOperandsSameType =
593 Instruction::isBinaryOp(Opcode) ||
598 Opcode);
599 if (AllOperandsSameType)
600 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
601 AssertOperandType(Idx, Op0Ty);
602
603 return Op0Ty;
604}
605
608 unsigned Opcode = I->getOpcode();
609 if (Instruction::isCast(Opcode) ||
610 is_contained(ArrayRef<unsigned>({Instruction::ExtractValue,
611 Instruction::Load, Instruction::Alloca}),
612 Opcode))
613 return I->getType();
615}
616
618 const VPIRFlags &Flags, const VPIRMetadata &MD,
619 DebugLoc DL, const Twine &Name, Type *ResultTy)
621 VPRecipeBase::VPInstructionSC, Operands,
622 ResultTy ? ResultTy
624 Flags, DL),
625 VPIRMetadata(MD), Opcode(Opcode), Name(Name.str()) {
627 "Set flags not supported for the provided opcode");
629 "Opcode requires specific flags to be set");
633 "number of operands does not match opcode");
634}
635
637 if (Instruction::isUnaryOp(Opcode) || Instruction::isCast(Opcode))
638 return 1;
639
640 if (Instruction::isBinaryOp(Opcode))
641 return 2;
642
643 switch (Opcode) {
646 return 0;
647 case Instruction::Alloca:
648 case Instruction::ExtractValue:
649 case Instruction::Freeze:
650 case Instruction::Load:
663 return 1;
664 case Instruction::ICmp:
665 case Instruction::FCmp:
666 case Instruction::ExtractElement:
667 case Instruction::Store:
679 return 2;
680 case Instruction::InsertElement:
681 case Instruction::Select:
685 return 3;
687 return 4;
688 case Instruction::Call:
689 return getCalledFnOperandIndex(operands()) + 1;
690 case Instruction::GetElementPtr:
691 case Instruction::PHI:
692 case Instruction::Switch:
693 case Instruction::AtomicRMW:
694 case Instruction::AtomicCmpXchg:
695 case Instruction::Fence:
707 // Cannot determine the number of operands from the opcode.
708 return -1u;
709 }
710 llvm_unreachable("all cases should be handled above");
711}
712
714 return Opcode == VPInstruction::Unpack ||
716}
717
718bool VPInstruction::doesGenerateSingleScalar() const {
720 return true;
721 switch (Opcode) {
722 case Instruction::Freeze:
723 case Instruction::ICmp:
724 case Instruction::PHI:
725 case Instruction::Select:
734 return vputils::onlyFirstLaneUsed(this);
735 default:
737 }
738}
739
741 if (Kind == RecurKind::Sub)
742 return Instruction::Add;
743 if (Kind == RecurKind::FSub)
744 return Instruction::FAdd;
745 llvm_unreachable("RecurKind should be Sub/FSub.");
746}
747
748Value *VPInstruction::generate(VPTransformState &State,
749 bool GenerateSingleScalar) {
750 IRBuilderBase &Builder = State.Builder;
751
753 Value *A = State.get(getOperand(0), GenerateSingleScalar);
754 Value *B = State.get(getOperand(1), GenerateSingleScalar);
755 auto *Res =
756 Builder.CreateBinOp((Instruction::BinaryOps)getOpcode(), A, B, Name);
757 if (auto *I = dyn_cast<Instruction>(Res))
758 applyFlags(*I);
759 return Res;
760 }
762 Value *Op = State.get(getOperand(0), VPLane(0));
764 getScalarType());
765 if (auto *CastOp = dyn_cast<Instruction>(Res)) {
766 applyFlags(*CastOp);
767 applyMetadata(*CastOp);
768 }
769 return Res;
770 }
771
772 switch (getOpcode()) {
773 case VPInstruction::Not: {
774 Value *A = State.get(getOperand(0), GenerateSingleScalar);
775 return Builder.CreateNot(A, Name);
776 }
778 // TODO: Use IsSingleScalar to produce a scalar value.
779 Value *A = State.get(getOperand(0));
780 Value *B = State.get(getOperand(1));
781 return Builder.CreateLogicalAnd(A, B, Name);
782 }
784 // TODO: Use IsSingleScalar to produce a scalar value.
785 Value *A = State.get(getOperand(0));
786 Value *B = State.get(getOperand(1));
787 return Builder.CreateLogicalOr(A, B, Name);
788 }
789 case Instruction::ExtractElement: {
790 assert(GenerateSingleScalar &&
791 "Can only generate first lane for ExtractElement");
792 assert(State.VF.isVector() && "Only extract elements from vectors");
793 if (auto *Idx = dyn_cast<VPConstantInt>(getOperand(1)))
794 return State.get(getOperand(0), VPLane(Idx->getZExtValue()));
795 Value *Vec = State.get(getOperand(0));
796 Value *Idx = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
797 return Builder.CreateExtractElement(Vec, Idx, Name);
798 }
799 case Instruction::InsertElement: {
800 assert(!GenerateSingleScalar &&
801 "Cannot generate scalar value for InsertElement");
802 assert(State.VF.isVector() && "Can only insert elements into vectors");
803 Value *Vec = State.get(getOperand(0), /*NeedsSingleScalar=*/false);
804 Value *Elt = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
805 Value *Idx = State.get(getOperand(2), /*NeedsSingleScalar=*/true);
806 return Builder.CreateInsertElement(Vec, Elt, Idx, Name);
807 }
808 case Instruction::Freeze: {
809 Value *Op = State.get(getOperand(0), GenerateSingleScalar);
810 return Builder.CreateFreeze(Op, Name);
811 }
812 case Instruction::FCmp:
813 case Instruction::ICmp: {
814 Value *A = State.get(getOperand(0), GenerateSingleScalar);
815 Value *B = State.get(getOperand(1), GenerateSingleScalar);
816 return Builder.CreateCmp(getPredicate(), A, B, Name);
817 }
818 case Instruction::PHI: {
819 llvm_unreachable("should be handled by VPPhi::execute");
820 }
821 case Instruction::Select: {
822 Value *Cond =
823 State.get(getOperand(0), GenerateSingleScalar ||
825 Value *Op1 = State.get(getOperand(1), GenerateSingleScalar);
826 Value *Op2 = State.get(getOperand(2), GenerateSingleScalar);
827 Value *Sel =
828 Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(), Name);
829 if (auto *I = dyn_cast<Instruction>(Sel))
831 return Sel;
832 }
835 // Can produce either a scalar value as a icmp of a phi, or a vector value,
836 // as a get.active.lane.mask intrinsic.
837 // Get first lane of vector induction variable.
838 Value *VIVElem0 = State.get(getOperand(0), VPLane(0));
839 // Get the original loop tripcount.
840 Value *ScalarTC = State.get(getOperand(1), VPLane(0));
841
842 uint64_t Multiplier =
844 ? cast<VPConstantInt>(getOperand(2))->getZExtValue()
845 : 1;
846
847 // If this part of the active lane mask is scalar, generate the CMP directly
848 // to avoid unnecessary extracts.
849 if (State.VF.isScalar() && Multiplier == 1)
850 return Builder.CreateCmp(CmpInst::Predicate::ICMP_ULT, VIVElem0, ScalarTC,
851 Name);
852
853 auto *PredTy = VectorType::get(Builder.getInt1Ty(), State.VF * Multiplier);
854 return Builder.CreateIntrinsic(Intrinsic::get_active_lane_mask,
855 {PredTy, ScalarTC->getType()},
856 {VIVElem0, ScalarTC}, nullptr, Name);
857 }
859 assert(GenerateSingleScalar &&
860 "Can only generate first lane for NumActiveLanes");
861 Value *Op = State.get(getOperand(0));
862 auto *VecTy = cast<VectorType>(Op->getType());
863 assert(VecTy->getScalarSizeInBits() == 1 &&
864 "NumActiveLanes only implemented for i1 vectors");
865
866 Type *Ty = getScalarType();
867 Value *ZExt = Builder.CreateCast(
868 Instruction::ZExt, Op, VectorType::get(Ty, VecTy->getElementCount()));
869 Value *NumActive =
870 Builder.CreateUnaryIntrinsic(Intrinsic::vector_reduce_add, ZExt);
871 return NumActive;
872 }
874 // Generate code to combine the previous and current values in vector v3.
875 //
876 // vector.ph:
877 // v_init = vector(..., ..., ..., a[-1])
878 // br vector.body
879 //
880 // vector.body
881 // i = phi [0, vector.ph], [i+4, vector.body]
882 // v1 = phi [v_init, vector.ph], [v2, vector.body]
883 // v2 = a[i, i+1, i+2, i+3];
884 // v3 = vector(v1(3), v2(0, 1, 2))
885
886 auto *V1 = State.get(getOperand(0));
887 if (!V1->getType()->isVectorTy())
888 return V1;
889 Value *V2 = State.get(getOperand(1));
890 return Builder.CreateVectorSpliceRight(V1, V2, 1, Name);
891 }
893 // TODO: Restructure this code with an explicit remainder loop, vsetvli can
894 // be outside of the main loop.
895 assert(GenerateSingleScalar &&
896 "Can only generate first lane for ExplicitVectorLength");
897 Value *AVL = State.get(getOperand(0), /*NeedsSingleScalar=*/true);
898 // Compute EVL
899 assert(AVL->getType()->isIntegerTy() &&
900 "Requested vector length should be an integer.");
901
902 assert(State.VF.isScalable() && "Expected scalable vector factor.");
903 Value *VFArg = Builder.getInt32(State.VF.getKnownMinValue());
904
905 Value *EVL = Builder.CreateIntrinsic(
906 Builder.getInt32Ty(), Intrinsic::experimental_get_vector_length,
907 {AVL, VFArg, Builder.getTrue()});
908 return EVL;
909 }
911 assert(GenerateSingleScalar &&
912 "Can only generate first lane for BranchOnCond");
913 Value *Cond = State.get(getOperand(0), VPLane(0));
914 // Replace the temporary unreachable terminator with a new conditional
915 // branch, hooking it up to backward destination for latch blocks now, and
916 // to forward destination(s) later when they are created.
917 // Second successor may be backwards - iff it is already in VPBB2IRBB.
918 VPBasicBlock *SecondVPSucc =
919 cast<VPBasicBlock>(getParent()->getSuccessors()[1]);
920 BasicBlock *SecondIRSucc = State.CFG.VPBB2IRBB.lookup(SecondVPSucc);
921 BasicBlock *IRBB = State.CFG.VPBB2IRBB[getParent()];
922 auto *Br = Builder.CreateCondBr(Cond, IRBB, SecondIRSucc);
923 // First successor is always forward, reset it to nullptr.
924 Br->setSuccessor(0, nullptr);
926 applyMetadata(*Br);
927 return Br;
928 }
930 assert(!GenerateSingleScalar &&
931 "Cannot generate scalar value for Broadcast");
932 return Builder.CreateVectorSplat(
933 State.VF, State.get(getOperand(0), /*NeedsSingleScalar=*/true),
934 "broadcast");
935 }
937 assert(!GenerateSingleScalar &&
938 "Cannot generate scalar value for BuildStructVector");
939 // For struct types, we need to build a new 'wide' struct type, where each
940 // element is widened, i.e., we create a struct of vectors.
941 auto *StructTy = cast<StructType>(getOperand(0)->getScalarType());
942 Value *Res = PoisonValue::get(toVectorizedTy(StructTy, State.VF));
943 for (const auto &[LaneIndex, Op] : enumerate(operands())) {
944 for (unsigned FieldIndex = 0; FieldIndex != StructTy->getNumElements();
945 FieldIndex++) {
946 Value *ScalarValue =
947 Builder.CreateExtractValue(State.get(Op, true), FieldIndex);
948 Value *VectorValue = Builder.CreateExtractValue(Res, FieldIndex);
949 VectorValue =
950 Builder.CreateInsertElement(VectorValue, ScalarValue, LaneIndex);
951 Res = Builder.CreateInsertValue(Res, VectorValue, FieldIndex);
952 }
953 }
954 return Res;
955 }
957 assert(!GenerateSingleScalar &&
958 "Cannot generate scalar value for BuildVector");
959 auto *ScalarTy = getOperand(0)->getScalarType();
960 auto NumOfElements = ElementCount::getFixed(getNumOperands());
961 Value *Res = PoisonValue::get(toVectorizedTy(ScalarTy, NumOfElements));
962 for (const auto &[Idx, Op] : enumerate(operands()))
963 Res = Builder.CreateInsertElement(Res, State.get(Op, true),
964 Builder.getInt64(Idx));
965 return Res;
966 }
968 Type *ScalarTy = getScalarType();
969 auto *WideTy = VectorType::get(ScalarTy, State.VF * getNumOperands());
970 Value *Res = PoisonValue::get(WideTy);
971 for (const auto &[Idx, Op] : enumerate(operands()))
972 Res = Builder.CreateInsertVector(WideTy, Res, State.get(Op),
973 Idx * State.VF.getKnownMinValue());
974 return Res;
975 }
977 if (State.VF.isScalar())
978 return State.get(getOperand(0), true);
979 IRBuilderBase::FastMathFlagGuard FMFG(Builder);
981 // If this start vector is scaled then it should produce a vector with fewer
982 // elements than the VF.
983 ElementCount VF = State.VF.divideCoefficientBy(
984 cast<VPConstantInt>(getOperand(2))->getZExtValue());
985 auto *Iden = Builder.CreateVectorSplat(VF, State.get(getOperand(1), true));
986 return Builder.CreateInsertElement(Iden, State.get(getOperand(0), true),
987 Builder.getInt64(0));
988 }
990 assert(GenerateSingleScalar &&
991 "Can only generate first lane for ComputeReductionResult");
992 RecurKind RK = getRecurKind();
993 bool IsOrdered = isReductionOrdered();
994 bool IsInLoop = isReductionInLoop();
996 "FindIV should use min/max reduction kinds");
997
998 // The recipe may have multiple operands to be reduced together.
999 unsigned NumOperandsToReduce = getNumOperands();
1000 SmallVector<Value *, 2> RdxParts(NumOperandsToReduce);
1001 for (unsigned Part = 0; Part < NumOperandsToReduce; ++Part)
1002 RdxParts[Part] = State.get(getOperand(Part), IsInLoop);
1003
1004 IRBuilderBase::FastMathFlagGuard FMFG(Builder);
1006
1007 // Reduce multiple operands into one.
1008 Value *ReducedPartRdx = RdxParts[0];
1009 if (IsOrdered) {
1010 ReducedPartRdx = RdxParts[NumOperandsToReduce - 1];
1011 } else {
1012 // Floating-point operations should have some FMF to enable the reduction.
1013 for (unsigned Part = 1; Part < NumOperandsToReduce; ++Part) {
1014 Value *RdxPart = RdxParts[Part];
1016 ReducedPartRdx = createMinMaxOp(Builder, RK, ReducedPartRdx, RdxPart);
1017 else {
1018 // For sub-recurrences, each part's reduction variable is already
1019 // negative, we need to do: reduce.add(-acc_uf0 + -acc_uf1)
1020 Instruction::BinaryOps Opcode =
1022 ? getSubRecurOpcode(RK)
1023 : (Instruction::BinaryOps)RecurrenceDescriptor::getOpcode(RK);
1024 ReducedPartRdx =
1025 Builder.CreateBinOp(Opcode, RdxPart, ReducedPartRdx, "bin.rdx");
1026 }
1027 }
1028 }
1029
1030 // Create the reduction after the loop. Note that inloop reductions create
1031 // the target reduction in the loop using a Reduction recipe.
1032 if (State.VF.isVector() && !IsInLoop) {
1033 // TODO: Support in-order reductions based on the recurrence descriptor.
1034 // All ops in the reduction inherit fast-math-flags from the recurrence
1035 // descriptor.
1036 ReducedPartRdx = createSimpleReduction(Builder, ReducedPartRdx, RK);
1037 }
1038
1039 return ReducedPartRdx;
1040 }
1043 assert(GenerateSingleScalar &&
1044 "Can only generate first lane for ExtractLane and "
1045 "ExtractPenultimateElement");
1046 unsigned Offset =
1048 Value *Res;
1049 if (State.VF.isVector()) {
1050 assert(Offset <= State.VF.getKnownMinValue() &&
1051 "invalid offset to extract from");
1052 // Extract lane VF - Offset from the operand.
1053 Res = State.get(getOperand(0), VPLane::getLaneFromEnd(State.VF, Offset));
1054 } else {
1055 // TODO: Remove ExtractLastLane for scalar VFs.
1056 assert(Offset <= 1 && "invalid offset to extract from");
1057 Res = State.get(getOperand(0));
1058 }
1059 if (isa<ExtractElementInst>(Res))
1060 Res->setName(Name);
1061 return Res;
1062 }
1063 case VPInstruction::PtrAdd: {
1064 assert(GenerateSingleScalar && "Can only generate first lane for PtrAdd");
1065 Value *Ptr = State.get(getOperand(0), VPLane(0));
1066 Value *Addend = State.get(getOperand(1), VPLane(0));
1067 return Builder.CreatePtrAdd(Ptr, Addend, Name, getGEPNoWrapFlags());
1068 }
1070 assert(!GenerateSingleScalar &&
1071 "Cannot generate scalar value for WidePtrAdd");
1072 Value *Ptr =
1074 Value *Addend = State.get(getOperand(1));
1075 return Builder.CreatePtrAdd(Ptr, Addend, Name, getGEPNoWrapFlags());
1076 }
1077 case VPInstruction::AnyOf: {
1078 assert(GenerateSingleScalar && "Can only generate first lane for AnyOf");
1079 Value *Res = State.get(getOperand(0));
1080 for (VPValue *Op : drop_begin(operands()))
1081 Res = Builder.CreateOr(Res, State.get(Op));
1082 return State.VF.isScalar() ? Res : Builder.CreateOrReduce(Res);
1083 }
1085 assert(GenerateSingleScalar &&
1086 "Can only generate first lane for ExtractLane");
1087 assert(getNumOperands() != 2 && "ExtractLane from single source should be "
1088 "simplified to ExtractElement.");
1089 Value *LaneToExtract = State.get(getOperand(0), true);
1090 Type *IdxTy = getOperand(0)->getScalarType();
1091 Value *Res = nullptr;
1092 Value *RuntimeVF = getRuntimeVF(Builder, IdxTy, State.VF);
1093
1094 for (unsigned Idx = 1; Idx != getNumOperands(); ++Idx) {
1095 Value *VectorStart =
1096 Builder.CreateMul(RuntimeVF, ConstantInt::get(IdxTy, Idx - 1));
1097 Value *VectorIdx = Idx == 1
1098 ? LaneToExtract
1099 : Builder.CreateSub(LaneToExtract, VectorStart);
1100 Value *Ext = State.VF.isScalar()
1101 ? State.get(getOperand(Idx))
1102 : Builder.CreateExtractElement(
1103 State.get(getOperand(Idx)), VectorIdx);
1104 if (Res) {
1105 Value *Cmp = Builder.CreateICmpUGE(LaneToExtract, VectorStart);
1106 Res = Builder.CreateSelect(Cmp, Ext, Res);
1107 } else {
1108 Res = Ext;
1109 }
1110 }
1111 return Res;
1112 }
1114 assert(GenerateSingleScalar &&
1115 "Can only generate first lane for FirstActiveLane");
1116 Type *Ty = this->getScalarType();
1117 if (getNumOperands() == 1) {
1118 Value *Mask = State.get(getOperand(0));
1119 return Builder.CreateCountTrailingZeroElems(Ty, Mask,
1120 /*ZeroIsPoison=*/false, Name);
1121 }
1122 // If there are multiple operands, create a chain of selects to pick the
1123 // first operand with an active lane and add the number of lanes of the
1124 // preceding operands.
1125 Value *RuntimeVF = getRuntimeVF(Builder, Ty, State.VF);
1126 unsigned LastOpIdx = getNumOperands() - 1;
1127 Value *Res = nullptr;
1128 for (int Idx = LastOpIdx; Idx >= 0; --Idx) {
1129 Value *TrailingZeros =
1130 State.VF.isScalar()
1131 ? Builder.CreateZExt(
1132 Builder.CreateICmpEQ(State.get(getOperand(Idx)),
1133 Builder.getFalse()),
1134 Ty)
1136 Ty, State.get(getOperand(Idx)),
1137 /*ZeroIsPoison=*/false, Name);
1138 Value *Current = Builder.CreateAdd(
1139 Builder.CreateMul(RuntimeVF, ConstantInt::get(Ty, Idx)),
1140 TrailingZeros);
1141 if (Res) {
1142 Value *Cmp = Builder.CreateICmpNE(TrailingZeros, RuntimeVF);
1143 Res = Builder.CreateSelect(Cmp, Current, Res);
1144 } else {
1145 Res = Current;
1146 }
1147 }
1148
1149 return Res;
1150 }
1152 assert(GenerateSingleScalar &&
1153 "Can only generate first lane for ResumeForEpilogue");
1154 return State.get(getOperand(0), true);
1156 assert(!GenerateSingleScalar && "Cannot generate scalar value for Reverse");
1157 return Builder.CreateVectorReverse(State.get(getOperand(0)), "reverse");
1159 assert(GenerateSingleScalar &&
1160 "Can only generate first lane for ExtractLastActive");
1161 Value *Result = State.get(getOperand(0), /*NeedsSingleScalar=*/true);
1162 for (unsigned Idx = 1; Idx < getNumOperands(); Idx += 2) {
1163 Value *Data = State.get(getOperand(Idx));
1164 Value *Mask = State.get(getOperand(Idx + 1));
1165 Type *VTy = Data->getType();
1166
1167 if (State.VF.isScalar())
1168 Result = Builder.CreateSelect(Mask, Data, Result);
1169 else
1170 Result = Builder.CreateIntrinsic(
1171 Intrinsic::experimental_vector_extract_last_active, {VTy},
1172 {Data, Mask, Result});
1173 }
1174
1175 return Result;
1176 }
1178 assert(!GenerateSingleScalar &&
1179 "Cannot generate scalar value for ExtractVectorForPart");
1180 Value *Src = State.get(getOperand(0));
1181 Type *DstTy = VectorType::get(getScalarType(), State.VF);
1182 uint64_t Part = cast<VPConstantInt>(getOperand(1))->getZExtValue();
1183
1184 if (Src->getType() == DstTy)
1185 return Src;
1186
1187 return Builder.CreateExtractVector(
1188 DstTy, Src, Builder.getInt64(State.VF.getKnownMinValue() * Part), Name);
1189 }
1191 assert(!GenerateSingleScalar &&
1192 "Cannot generate scalar value for StepVector");
1193 return State.Builder.CreateStepVector(
1194 VectorType::get(getScalarType(), State.VF));
1196 assert(GenerateSingleScalar &&
1197 "Can only generate first lane for Intrinsic");
1198 SmallVector<Value *, 2> Args;
1199 for (VPValue *Op : drop_end(operands()))
1200 Args.push_back(State.get(Op, /*NeedsSingleScalar=*/true));
1201 return State.Builder.CreateIntrinsic(getScalarType(),
1202 vputils::getIntrinsicID(this), Args,
1203 /*FMFSource=*/nullptr, getName());
1204 }
1206 unsigned Multiplier = cast<VPConstantInt>(getOperand(0))->getZExtValue();
1207 auto *WideDataTy = VectorType::get(getScalarType(), State.VF * Multiplier);
1208
1209 Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
1210 Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
1211 LoadInst *WideLI = Builder.CreateAlignedLoad(WideDataTy, Addr, Alignment);
1212 applyMetadata(*WideLI);
1213 return WideLI;
1214 }
1216 unsigned Multiplier = cast<VPConstantInt>(getOperand(0))->getZExtValue();
1217 Value *WideData = State.get(getOperand(3));
1218 assert(cast<VectorType>(WideData->getType())->getElementCount() ==
1219 State.VF * Multiplier &&
1220 "stored value does not match wide element count");
1221 (void)Multiplier;
1222
1223 Value *Addr = State.get(getOperand(1), /*IsScalar=*/true);
1224 Align Alignment = Align(cast<VPConstantInt>(getOperand(2))->getZExtValue());
1225 StoreInst *WideSI = Builder.CreateAlignedStore(WideData, Addr, Alignment);
1226 applyMetadata(*WideSI);
1227 return WideSI;
1228 }
1229 default:
1230 llvm_unreachable("Unsupported opcode for instruction");
1231 }
1232}
1233
1235 unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const {
1236 Type *ScalarTy = this->getScalarType();
1237 Type *ResultTy = VF.isVector() ? toVectorTy(ScalarTy, VF) : ScalarTy;
1238 switch (Opcode) {
1239 case Instruction::FNeg:
1240 return Ctx.TTI.getArithmeticInstrCost(Opcode, ResultTy, Ctx.CostKind);
1241 case Instruction::UDiv:
1242 case Instruction::SDiv:
1243 case Instruction::SRem:
1244 case Instruction::URem:
1245 case Instruction::Add:
1246 case Instruction::FAdd:
1247 case Instruction::Sub:
1248 case Instruction::FSub:
1249 case Instruction::Mul:
1250 case Instruction::FMul:
1251 case Instruction::FDiv:
1252 case Instruction::FRem:
1253 case Instruction::Shl:
1254 case Instruction::LShr:
1255 case Instruction::AShr:
1256 case Instruction::And:
1257 case Instruction::Or:
1258 case Instruction::Xor: {
1259 // Certain instructions can be cheaper if they have a constant second
1260 // operand. One example of this are shifts on x86.
1261 VPValue *RHS = getOperand(1);
1262 TargetTransformInfo::OperandValueInfo RHSInfo = Ctx.getOperandInfo(RHS);
1263
1264 if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
1267
1270 if (CtxI)
1271 Operands.append(CtxI->value_op_begin(), CtxI->value_op_end());
1272 return Ctx.TTI.getArithmeticInstrCost(
1273 Opcode, ResultTy, Ctx.CostKind,
1274 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
1275 RHSInfo, Operands, CtxI, &Ctx.TLI);
1276 }
1277 case Instruction::Freeze:
1278 // NOTE: The only way to ask for the cost is via getInstructionCost, which
1279 // requires the actual vector instruction. Instead, both here and in the
1280 // LoopVectorizationCostModel::getInstructionCost the costs mirror the
1281 // current behaviour in llvm/Analysis/TargetTransformInfoImpl.h to keep
1282 // them in sync.
1283 return TTI::TCC_Free;
1284 case Instruction::ExtractValue:
1285 return Ctx.TTI.getInsertExtractValueCost(Instruction::ExtractValue,
1286 Ctx.CostKind);
1287 case Instruction::ICmp:
1288 case Instruction::FCmp: {
1289 Type *ScalarOpTy = getOperand(0)->getScalarType();
1290 Type *OpTy = VF.isVector() ? toVectorTy(ScalarOpTy, VF) : ScalarOpTy;
1292 return Ctx.TTI.getCmpSelInstrCost(
1294 Ctx.CostKind, {TTI::OK_AnyValue, TTI::OP_None},
1295 {TTI::OK_AnyValue, TTI::OP_None}, CtxI);
1296 }
1297 case Instruction::BitCast: {
1298 Type *ScalarTy = this->getScalarType();
1299 if (ScalarTy->isPointerTy())
1300 return 0;
1301 [[fallthrough]];
1302 }
1303 case Instruction::SExt:
1304 case Instruction::ZExt:
1305 case Instruction::FPToUI:
1306 case Instruction::FPToSI:
1307 case Instruction::FPExt:
1308 case Instruction::PtrToInt:
1309 case Instruction::PtrToAddr:
1310 case Instruction::IntToPtr:
1311 case Instruction::SIToFP:
1312 case Instruction::UIToFP:
1313 case Instruction::Trunc:
1314 case Instruction::FPTrunc:
1315 case Instruction::AddrSpaceCast: {
1316 // Computes the CastContextHint from a recipe that may access memory.
1317 auto ComputeCCH = [&](const VPRecipeBase *R) -> TTI::CastContextHint {
1318 if (isa<VPInterleaveBase>(R))
1320 if (const auto *ReplicateRecipe = dyn_cast<VPReplicateRecipe>(R)) {
1321 // Only compute CCH for memory operations, matching the legacy model
1322 // which only considers loads/stores for cast context hints.
1323 auto *UI = cast<Instruction>(ReplicateRecipe->getUnderlyingValue());
1324 if (!isa<LoadInst, StoreInst>(UI))
1326 return ReplicateRecipe->isPredicated() ? TTI::CastContextHint::Masked
1328 }
1329 const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
1330 if (WidenMemoryRecipe == nullptr)
1332 if (VF.isScalar())
1334 if (!WidenMemoryRecipe->isConsecutive())
1336 if (WidenMemoryRecipe->isMasked())
1339 };
1340
1341 VPValue *Operand = getOperand(0);
1343 bool IsReverse = false;
1344 // For Trunc/FPTrunc, get the context from the only user.
1345 if (Opcode == Instruction::Trunc || Opcode == Instruction::FPTrunc) {
1346 if (auto *Recipe = cast_or_null<VPRecipeBase>(getSingleUser())) {
1347 if (match(Recipe,
1351 IsReverse = true;
1353 Recipe->getVPSingleValue()->getSingleUser());
1354 }
1355 if (Recipe)
1356 CCH = ComputeCCH(Recipe);
1357 }
1358 }
1359 // For Z/Sext, get the context from the operand.
1360 else if (Opcode == Instruction::ZExt || Opcode == Instruction::SExt ||
1361 Opcode == Instruction::FPExt) {
1362 if (auto *Recipe = Operand->getDefiningRecipe()) {
1363 VPValue *ReverseOp;
1364 if (match(Recipe,
1365 m_CombineOr(m_Reverse(m_VPValue(ReverseOp)),
1367 m_VPValue(ReverseOp))))) {
1368 Recipe = ReverseOp->getDefiningRecipe();
1369 IsReverse = true;
1370 }
1371 if (Recipe)
1372 CCH = ComputeCCH(Recipe);
1373 }
1374 }
1375 if (IsReverse && CCH != TTI::CastContextHint::None)
1377
1378 auto *ScalarSrcTy = Operand->getScalarType();
1379 Type *SrcTy = VF.isVector() ? toVectorTy(ScalarSrcTy, VF) : ScalarSrcTy;
1380 // Arm TTI will use the underlying instruction to determine the cost.
1381 return Ctx.TTI.getCastInstrCost(
1382 Opcode, ResultTy, SrcTy, CCH, Ctx.CostKind,
1384 }
1385 case Instruction::Select: {
1387 bool IsScalarCond = getOperand(0)->isDefinedOutsideLoopRegions();
1388 Type *ScalarTy = this->getScalarType();
1389
1390 VPValue *Op0, *Op1;
1391 bool IsLogicalAnd =
1392 match(this, m_c_LogicalAnd(m_VPValue(Op0), m_VPValue(Op1)));
1393 bool IsLogicalOr =
1394 match(this, m_c_LogicalOr(m_VPValue(Op0), m_VPValue(Op1)));
1395 // Also match the inverted forms:
1396 // select x, false, y --> !x & y (still AND)
1397 // select x, y, true --> !x | y (still OR)
1398 IsLogicalAnd |=
1399 match(this, m_Select(m_VPValue(Op0), m_False(), m_VPValue(Op1)));
1400 IsLogicalOr |=
1401 match(this, m_Select(m_VPValue(Op0), m_VPValue(Op1), m_True()));
1402
1403 if (!IsScalarCond && ScalarTy->getScalarSizeInBits() == 1 &&
1404 (IsLogicalAnd || IsLogicalOr)) {
1405 // select x, y, false --> x & y
1406 // select x, true, y --> x | y
1407 const auto [Op1VK, Op1VP] = Ctx.getOperandInfo(Op0);
1408 const auto [Op2VK, Op2VP] = Ctx.getOperandInfo(Op1);
1409
1411 if (SI && all_of(operands(),
1412 [](VPValue *Op) { return Op->getUnderlyingValue(); }))
1413 append_range(Operands, SI->operands());
1414 return Ctx.TTI.getArithmeticInstrCost(
1415 IsLogicalOr ? Instruction::Or : Instruction::And, ResultTy,
1416 Ctx.CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, Operands, SI);
1417 }
1418
1419 Type *CondTy = getOperand(0)->getScalarType();
1420 if (!IsScalarCond && VF.isVector())
1421 CondTy = VectorType::get(CondTy, VF);
1422
1423 llvm::CmpPredicate Pred;
1424 if (!match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue())))
1425 if (auto *CondIRV = dyn_cast<VPIRValue>(getOperand(0)))
1426 if (auto *Cmp = dyn_cast<CmpInst>(CondIRV->getValue()))
1427 Pred = Cmp->getPredicate();
1428 Type *VectorTy = toVectorTy(this->getScalarType(), VF);
1429 return Ctx.TTI.getCmpSelInstrCost(
1430 Instruction::Select, VectorTy, CondTy, Pred, Ctx.CostKind,
1431 {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, SI);
1432 }
1433 }
1434 llvm_unreachable("called for unsupported opcode");
1435}
1436
1438 VPCostContext &Ctx) const {
1439 // NOTE: At the moment it seems only possible to expose this path for
1440 // the trunc, zext and sext opcodes.
1441 // TODO: Update VF arg to use onlyFirstLaneUsed once WidenCast is unified.
1443 // A scalar zext/trunc that only adjusts the width of an
1444 // ExplicitVectorLength to the canonical IV type is free: it feeds only
1445 // the IV increment and AVL decrement, which are modeled as free below.
1446 if (match(this, m_ZExtOrTrunc(m_EVL(m_VPValue()))))
1447 return 0;
1449 Ctx);
1450 }
1451
1453 if (!getUnderlyingValue() && getOpcode() != Instruction::FMul) {
1454 // TODO: Compute cost for VPInstructions without underlying values once
1455 // the legacy cost model has been retired.
1456 return 0;
1457 }
1458
1460 "Should only generate a vector value or single scalar, not scalars "
1461 "for all lanes.");
1463 getOpcode(),
1465 }
1466
1467 switch (getOpcode()) {
1468 case Instruction::Select: {
1470 match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue()));
1471 auto *CondTy = getOperand(0)->getScalarType();
1472 auto *VecTy = getOperand(1)->getScalarType();
1473 if (!vputils::onlyFirstLaneUsed(this)) {
1474 CondTy = toVectorTy(CondTy, VF);
1475 VecTy = toVectorTy(VecTy, VF);
1476 }
1477 return Ctx.TTI.getCmpSelInstrCost(Instruction::Select, VecTy, CondTy, Pred,
1478 Ctx.CostKind);
1479 }
1480 case Instruction::ExtractElement:
1482 if (VF.isScalar()) {
1483 // ExtractLane with VF=1 takes care of handling extracting across multiple
1484 // parts.
1485 return 0;
1486 }
1487
1488 // Add on the cost of extracting the element.
1489 auto *VecTy = toVectorTy(getOperand(0)->getScalarType(), VF);
1490 return Ctx.TTI.getVectorInstrCost(Instruction::ExtractElement, VecTy,
1491 Ctx.CostKind);
1492 }
1493 case VPInstruction::AnyOf: {
1494 auto *VecTy = toVectorTy(this->getScalarType(), VF);
1495 return Ctx.TTI.getArithmeticReductionCost(
1496 Instruction::Or, cast<VectorType>(VecTy), std::nullopt, Ctx.CostKind);
1497 }
1499 Type *Ty = this->getScalarType();
1500 Type *ScalarTy = getOperand(0)->getScalarType();
1501 if (VF.isScalar())
1502 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ScalarTy,
1504 CmpInst::ICMP_EQ, Ctx.CostKind);
1505 // Calculate the cost of determining the lane index.
1506 auto *PredTy = toVectorTy(ScalarTy, VF);
1507 IntrinsicCostAttributes Attrs(Intrinsic::experimental_cttz_elts, Ty,
1508 {PredTy, Type::getInt1Ty(Ctx.LLVMCtx)});
1509 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1510 }
1512 Type *Ty = this->getScalarType();
1513 Type *ScalarTy = getOperand(0)->getScalarType();
1514 if (VF.isScalar())
1515 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ScalarTy,
1517 CmpInst::ICMP_EQ, Ctx.CostKind);
1518 // Calculate the cost of determining the lane index: NOT + cttz_elts + SUB.
1519 auto *PredTy = toVectorTy(ScalarTy, VF);
1520 IntrinsicCostAttributes Attrs(Intrinsic::experimental_cttz_elts, Ty,
1521 {PredTy, Type::getInt1Ty(Ctx.LLVMCtx)});
1522 InstructionCost Cost = Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1523 // Add cost of NOT operation on the predicate.
1524 Cost += Ctx.TTI.getArithmeticInstrCost(
1525 Instruction::Xor, PredTy, Ctx.CostKind,
1526 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
1527 {TargetTransformInfo::OK_UniformConstantValue,
1528 TargetTransformInfo::OP_None});
1529 // Add cost of SUB operation on the index.
1530 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Sub, Ty, Ctx.CostKind);
1531 return Cost;
1532 }
1534 Type *ScalarTy = this->getScalarType();
1535 Type *VecTy = toVectorTy(ScalarTy, VF);
1536 Type *MaskTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF);
1538 Intrinsic::experimental_vector_extract_last_active, ScalarTy,
1539 {VecTy, MaskTy, ScalarTy});
1540 return Ctx.TTI.getIntrinsicInstrCost(ICA, Ctx.CostKind);
1541 }
1543 assert(VF.isVector() && "Scalar FirstOrderRecurrenceSplice?");
1544 Type *VectorTy = toVectorTy(this->getScalarType(), VF);
1545 return Ctx.TTI.getShuffleCost(
1547 cast<VectorType>(VectorTy), Ctx.CostKind, {}, -1);
1548 }
1551 Type *ArgTy = getOperand(0)->getScalarType();
1552 uint64_t Multiplier =
1554 ? cast<VPConstantInt>(getOperand(2))->getZExtValue()
1555 : 1;
1556 Type *RetTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF * Multiplier);
1557 IntrinsicCostAttributes Attrs(Intrinsic::get_active_lane_mask, RetTy,
1558 {ArgTy, ArgTy});
1559 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1560 }
1562 Type *Arg0Ty = getOperand(0)->getScalarType();
1563 Type *I32Ty = Type::getInt32Ty(Ctx.LLVMCtx);
1564 Type *I1Ty = Type::getInt1Ty(Ctx.LLVMCtx);
1565 IntrinsicCostAttributes Attrs(Intrinsic::experimental_get_vector_length,
1566 I32Ty, {Arg0Ty, I32Ty, I1Ty});
1567 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1568 }
1570 assert(VF.isVector() && "Reverse operation must be vector type");
1571 Type *EltTy = this->getScalarType();
1572 // Skip the reverse operation cost for the mask.
1573 // FIXME: Remove this once redundant mask reverse operations can be
1574 // eliminated by VPlanTransforms::cse before cost computation.
1575 if (EltTy->isIntegerTy(1))
1576 return 0;
1577 auto *VectorTy = cast<VectorType>(toVectorTy(EltTy, VF));
1578 return Ctx.TTI.getShuffleCost(TargetTransformInfo::SK_Reverse, VectorTy,
1579 VectorTy, Ctx.CostKind, /*Mask=*/{},
1580 /*Index=*/0);
1581 }
1583 // Add on the cost of extracting the element.
1584 auto *VecTy = toVectorTy(getOperand(0)->getScalarType(), VF);
1585 return Ctx.TTI.getIndexedVectorInstrCostFromEnd(Instruction::ExtractElement,
1586 VecTy, Ctx.CostKind, 0);
1587 }
1588 case VPInstruction::Not: {
1589 Type *ValTy = this->getScalarType();
1590 // InstCombine will fold `xor` to the conditional branch.
1591 if (auto *U = const_cast<VPUser *>(getSingleUser()))
1592 if (match(U, m_BranchOnCond(m_VPValue())))
1593 return 0;
1594 if (!vputils::onlyFirstLaneUsed(this))
1595 ValTy = toVectorTy(ValTy, VF);
1596 return Ctx.TTI.getArithmeticInstrCost(Instruction::Xor, ValTy,
1597 Ctx.CostKind);
1598 }
1600 // If TC <= VF then this is just a branch.
1601 // FIXME: Removing the branch happens in simplifyBranchConditionForVFAndUF
1602 // where it checks TC <= VF * UF, but we don't know UF yet. This means in
1603 // some cases we get a cost that's too high due to counting a cmp that
1604 // later gets removed.
1605 // FIXME: The compare could also be removed if TC = M * vscale,
1606 // VF = N * vscale, and M <= N. Detecting that would require having the
1607 // trip count as a SCEV though.
1608 if (VPCostContext::executesAtMostOnce(*getParent()->getPlan(), VF))
1609 return 0;
1610 // Otherwise BranchOnCount generates ICmpEQ followed by a branch.
1611 Type *ValTy = getOperand(0)->getScalarType();
1612 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ValTy,
1614 CmpInst::ICMP_EQ, Ctx.CostKind);
1615 }
1617 Type *Ty = getScalarType();
1619 for (const VPValue *Op : drop_end(operands()))
1620 ArgTys.push_back(Op->getScalarType());
1621 IntrinsicCostAttributes Attrs(vputils::getIntrinsicID(this), Ty, ArgTys);
1622 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1623 }
1625 // TODO: This isn't quite right since even if the step-vector is hoisted
1626 // out of the loop it has a non-zero cost in the middle block, etc.
1627 // Once the stepvector is correctly hoisted out of the vector loop by the
1628 // licm transform we can add the cost here so that it doesn't incorrectly
1629 // affect the choice of VF.
1630 return 0;
1632 // It isn't currently possible to expose cases where WideIVStep's cost is
1633 // queried.
1634 llvm_unreachable("Unhandled opcode");
1635 case Instruction::FCmp:
1636 case Instruction::ICmp:
1638 getOpcode(),
1641 if (VF == ElementCount::getScalable(1))
1643 [[fallthrough]];
1644 default:
1645 // TODO: Compute cost other VPInstructions once the legacy cost model has
1646 // been retired.
1648 "unexpected VPInstruction witht underlying value");
1649 return 0;
1650 }
1651}
1652
1665
1667 switch (getOpcode()) {
1668 case Instruction::Load:
1669 case Instruction::PHI:
1673 return true;
1674 default:
1676 }
1677}
1678
1680#ifndef NDEBUG
1681 Type *Ty = Op->getScalarType();
1682 switch (getOpcode()) {
1686 assert(Ty == getOperand(0)->getScalarType() &&
1687 "types of operand 0 and new operand must match");
1688 break;
1693 assert(Ty == getOperand(0)->getScalarType() &&
1694 "appended operand must match operand 0's scalar type");
1695 break;
1697 assert(Ty == getOperand(1)->getScalarType() &&
1698 "appended operand must match operand 1's scalar type");
1699 break;
1701 // The recipe is constructed with 3 operands (result, data, mask). Extra
1702 // operands beyond that are appended in (data, mask) pairs.
1703 constexpr unsigned NumInitialOperands = 3;
1704 assert(getNumOperands() >= NumInitialOperands &&
1705 "ExtractLastActive must have at least the initial 3 operands");
1706 bool IsMaskSlot = ((getNumOperands() - NumInitialOperands) & 1u) == 1u;
1707 assert((IsMaskSlot ? Ty->isIntegerTy(1)
1708 : Ty == getOperand(1)->getScalarType()) &&
1709 "ExtractLastActive expects alternating data/mask operands "
1710 "matching operand 1's type and i1, respectively");
1711 break;
1712 }
1713 default:
1714 llvm_unreachable("opcode does not support growing the operand list "
1715 "outside of construction");
1716 }
1717#endif
1719}
1720
1722 assert(!isMasked() && "cannot execute masked VPInstruction");
1723 IRBuilderBase::FastMathFlagGuard FMFGuard(State.Builder);
1725 "Set flags not supported for the provided opcode");
1727 "Opcode requires specific flags to be set");
1728 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
1729 bool GenerateSingleScalar = State.VF.isScalar() || doesGenerateSingleScalar();
1730 Value *GeneratedValue = generate(State, GenerateSingleScalar);
1731 if (!hasResult())
1732 return;
1733 assert(GeneratedValue && "generate must produce a value");
1734 assert(((GeneratedValue->getType()->isVectorTy() ||
1735 GeneratedValue->getType()->isStructTy()) == !GenerateSingleScalar) &&
1736 "scalar value but not only first lane defined");
1737 State.set(this, GeneratedValue, GenerateSingleScalar);
1739 getOpcode() == Instruction::Freeze) {
1740 // FIXME: This is a workaround to enable reliable updates of the scalar loop
1741 // resume phis, and to let epilogue vectorization recover the frozen
1742 // reduction start from the main plan. Must be removed once epilogue
1743 // vectorization explicitly connects VPlans.
1744 setUnderlyingValue(GeneratedValue);
1745 }
1746}
1747
1751 return false;
1752 switch (getOpcode()) {
1753 case Instruction::ExtractValue:
1754 case Instruction::InsertValue:
1755 case Instruction::GetElementPtr:
1756 case Instruction::ExtractElement:
1757 case Instruction::InsertElement:
1758 case Instruction::Freeze:
1759 case Instruction::FCmp:
1760 case Instruction::ICmp:
1761 case Instruction::Select:
1762 case Instruction::PHI:
1790 case VPInstruction::Not:
1798 return false;
1801 AttributeSet Attrs =
1803 return !Attrs.getMemoryEffects().doesNotAccessMemory();
1804 }
1805 case Instruction::Call:
1807 default:
1808 return true;
1809 }
1810}
1811
1813 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
1815 return vputils::onlyFirstLaneUsed(this);
1816
1817 switch (getOpcode()) {
1818 default:
1819 return false;
1820 case Instruction::ExtractElement:
1821 return Op == getOperand(1);
1822 case Instruction::InsertElement:
1823 return Op == getOperand(1) || Op == getOperand(2);
1825 return Op == getOperand(0);
1826 case Instruction::PHI:
1827 return true;
1828 case Instruction::FCmp:
1829 case Instruction::ICmp:
1830 case Instruction::Select:
1831 case Instruction::Or:
1832 case Instruction::Freeze:
1833 case VPInstruction::Not:
1834 // TODO: Cover additional opcodes.
1835 return vputils::onlyFirstLaneUsed(this);
1836 case Instruction::Load:
1849 return true;
1852 // Before replicating by VF, Build(Struct)Vector uses all lanes of the
1853 // operand, after replicating its operands only the first lane is used.
1854 // Before replicating, it will have only a single operand.
1855 return getNumOperands() > 1;
1857 return Op == getOperand(0) || vputils::onlyFirstLaneUsed(this);
1859 // WidePtrAdd supports scalar and vector base addresses.
1860 return false;
1862 return Op == getOperand(0) || Op == getOperand(1) || Op == getOperand(2);
1865 return Op == getOperand(0);
1866 };
1867 llvm_unreachable("switch should return");
1868}
1869
1871 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
1873 return vputils::onlyFirstPartUsed(this);
1874
1875 switch (getOpcode()) {
1876 default:
1877 return false;
1878 case Instruction::FCmp:
1879 case Instruction::ICmp:
1880 case Instruction::Select:
1881 return vputils::onlyFirstPartUsed(this);
1886 return true;
1887 };
1888 llvm_unreachable("switch should return");
1889}
1890
1891#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1893 VPSlotTracker SlotTracker(getParent()->getPlan());
1895}
1896
1898 VPSlotTracker &SlotTracker) const {
1899 O << Indent << "EMIT" << (isSingleScalar() ? "-SCALAR" : "") << " ";
1900
1901 if (hasResult()) {
1903 O << " = ";
1904 }
1905
1906 switch (getOpcode()) {
1907 case VPInstruction::Not:
1908 O << "not";
1909 break;
1911 O << "active lane mask";
1912 break;
1914 O << "wide active lane mask";
1915 break;
1917 O << "wide vector load";
1918 break;
1920 O << "wide vector store";
1921 break;
1923 O << "concat-vectors";
1924 break;
1926 O << "incoming-alias-mask";
1927 break;
1929 O << "EXPLICIT-VECTOR-LENGTH";
1930 break;
1932 O << "first-order splice";
1933 break;
1935 O << "branch-on-cond";
1936 break;
1938 O << "branch-on-two-conds";
1939 break;
1941 O << "VF * Part +";
1942 break;
1944 O << "branch-on-count";
1945 break;
1947 O << "broadcast";
1948 break;
1950 O << "buildstructvector";
1951 break;
1953 O << "buildvector";
1954 break;
1956 O << "exiting-iv-value";
1957 break;
1959 O << "masked-cond";
1960 break;
1962 O << "extract-lane";
1963 break;
1965 O << "extract-last-lane";
1966 break;
1968 O << "extract-last-part";
1969 break;
1971 O << "extract-penultimate-element";
1972 break;
1974 O << "extract-vector-for-part";
1975 break;
1977 O << "compute-reduction-result";
1978 break;
1980 O << "logical-and";
1981 break;
1983 O << "logical-or";
1984 break;
1986 O << "ptradd";
1987 break;
1989 O << "wide-ptradd";
1990 break;
1992 O << "any-of";
1993 break;
1995 O << "first-active-lane";
1996 break;
1998 O << "last-active-lane";
1999 break;
2001 O << "reduction-start-vector";
2002 break;
2004 O << "resume-for-epilogue";
2005 break;
2007 O << "reverse";
2008 break;
2010 O << "unpack";
2011 break;
2013 O << "extract-last-active";
2014 break;
2016 O << "num-active-lanes";
2017 break;
2019 O << "wide-iv-step";
2020 break;
2022 O << "step-vector " << *getScalarType();
2023 break;
2025 O << "call " << *getScalarType() << " @"
2028 Op->printAsOperand(O, SlotTracker);
2029 });
2030 O << ")";
2031 return;
2032 }
2033 case Instruction::Load:
2034 O << "load";
2035 break;
2036 default:
2038 }
2039
2040 if (!operands_empty()) {
2041 printFlags(O);
2043 }
2045 O << " to " << *getScalarType();
2046}
2047#endif
2048
2049/// Shared execute logic for VPPhi and VPWidenPHIRecipe. Creates a PHI node,
2050/// adds incoming values, and stores the result in State. For header phis, only
2051/// the preheader incoming value is added; the backedge is fixed up later by
2052/// VPlan::execute().
2054 VPTransformState &State, bool IsScalar,
2055 const Twine &Name) {
2056 unsigned NumIncoming = VPBlockUtils::isHeader(R->getParent(), State.VPDT)
2057 ? 1
2058 : Phi.getNumIncoming();
2059 Value *FirstInc = State.get(Phi.getIncomingValue(0), IsScalar);
2060 PHINode *NewPhi = State.Builder.CreatePHI(FirstInc->getType(), 2, Name);
2061 NewPhi->addIncoming(FirstInc,
2062 State.CFG.VPBB2IRBB.at(Phi.getIncomingBlock(0)));
2063 for (unsigned Idx = 1; Idx != NumIncoming; ++Idx)
2064 NewPhi->addIncoming(State.get(Phi.getIncomingValue(Idx), IsScalar),
2065 State.CFG.VPBB2IRBB.at(Phi.getIncomingBlock(Idx)));
2066 State.set(R, NewPhi, IsScalar);
2067}
2068
2070 executePhiRecipe(this, *this, State, /*IsScalar=*/true, getName());
2071}
2072
2073#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2074void VPPhi::printRecipe(raw_ostream &O, const Twine &Indent,
2075 VPSlotTracker &SlotTracker) const {
2076 O << Indent << "EMIT" << (isSingleScalar() ? "-SCALAR" : "") << " ";
2078 O << " = phi";
2079 printFlags(O);
2081}
2082#endif
2083
2084VPIRInstruction *VPIRInstruction ::create(Instruction &I) {
2085 if (auto *Phi = dyn_cast<PHINode>(&I))
2086 return new VPIRPhi(*Phi);
2087 return new VPIRInstruction(I);
2088}
2089
2091 assert(!isa<VPIRPhi>(this) && getNumOperands() == 0 &&
2092 "PHINodes must be handled by VPIRPhi");
2093 // Advance the insert point after the wrapped IR instruction. This allows
2094 // interleaving VPIRInstructions and other recipes.
2095 State.Builder.SetInsertPoint(std::next(I.getIterator()));
2096}
2097
2099 VPCostContext &Ctx) const {
2100 // The recipe wraps an existing IR instruction on the border of VPlan's scope,
2101 // hence it does not contribute to the cost-modeling for the VPlan.
2102 return 0;
2103}
2104
2105#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2107 VPSlotTracker &SlotTracker) const {
2108 O << Indent << "IR " << I;
2109}
2110#endif
2111
2113 PHINode *Phi = &getIRPhi();
2114 for (const auto &[Idx, Op] : enumerate(operands())) {
2115 VPValue *ExitValue = Op;
2116 auto Lane = vputils::isSingleScalar(ExitValue)
2118 : VPLane::getLastLaneForVF(State.VF);
2119 VPBlockBase *Pred = getParent()->getPredecessors()[Idx];
2120 auto *PredVPBB = Pred->getExitingBasicBlock();
2121 BasicBlock *PredBB = State.CFG.VPBB2IRBB[PredVPBB];
2122 // Set insertion point in PredBB in case an extract needs to be generated.
2123 // TODO: Model extracts explicitly.
2124 State.Builder.SetInsertPoint(PredBB->getTerminator());
2125 Value *V = State.get(ExitValue, VPLane(Lane));
2126 // If there is no existing block for PredBB in the phi, add a new incoming
2127 // value. Otherwise update the existing incoming value for PredBB.
2128 if (Phi->getBasicBlockIndex(PredBB) == -1)
2129 Phi->addIncoming(V, PredBB);
2130 else
2131 Phi->setIncomingValueForBlock(PredBB, V);
2132 }
2133
2134 // Advance the insert point after the wrapped IR instruction. This allows
2135 // interleaving VPIRInstructions and other recipes.
2136 State.Builder.SetInsertPoint(std::next(Phi->getIterator()));
2137}
2138
2140 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2141 assert(R->getNumOperands() == R->getParent()->getNumPredecessors() &&
2142 "Number of phi operands must match number of predecessors");
2143 unsigned Position = R->getParent()->getIndexForPredecessor(IncomingBlock);
2144 R->removeOperand(Position);
2145}
2146
2147VPValue *
2149 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2150 return getIncomingValue(R->getParent()->getIndexForPredecessor(VPBB));
2151}
2152
2154 VPValue *V) const {
2155 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2156 R->setOperand(R->getParent()->getIndexForPredecessor(VPBB), V);
2157}
2158
2159#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2161 VPSlotTracker &SlotTracker) const {
2163 O << "[ ";
2164 std::get<0>(Op)->printAsOperand(O, SlotTracker);
2165 O << ", ";
2166 std::get<1>(Op)->printAsOperand(O);
2167 O << " ]";
2168 });
2169}
2170#endif
2171
2172#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2174 VPSlotTracker &SlotTracker) const {
2176
2177 if (getNumOperands() != 0) {
2178 O << " (extra operand" << (getNumOperands() > 1 ? "s" : "") << ": ";
2180 [&O, &SlotTracker](auto Op) {
2181 std::get<0>(Op)->printAsOperand(O, SlotTracker);
2182 O << " from ";
2183 std::get<1>(Op)->printAsOperand(O);
2184 });
2185 O << ")";
2186 }
2187}
2188#endif
2189
2191 if (Metadata.empty())
2192 return;
2193 // Frequencies and estimated branch weights are VPlan-internal and must not
2194 // reach IR.
2195 unsigned ExecFreqKind = getMDKindID(ExecutionFrequencyMDName);
2196 unsigned EstProfKind = getMDKindID(EstimatedProfileMDName);
2197 for (const auto &[Kind, Node] : Metadata)
2198 if (Kind != ExecFreqKind && Kind != EstProfKind)
2199 I.setMetadata(Kind, Node);
2200}
2201
2202/// Returns the execution frequency recorded in \p Node.
2204 assert(Node->getNumOperands() <= 2 && "unexpected frequency node shape");
2205 uint64_t Freq =
2206 mdconst::extract<ConstantInt>(Node->getOperand(0))->getZExtValue();
2208 "frequency cannot exceed the one of an always executing block");
2209 return {BlockFrequency(Freq), Node->getNumOperands() == 2};
2210}
2211
2213 std::optional<VPExecutionFrequency> Freq, LLVMContext &Ctx) {
2214 // A recipe that always executes needs no annotation.
2215 if (!Freq || vputils::getExecutionProbability(Freq->Freq).isOne())
2216 return;
2218 ConstantInt::get(Type::getInt64Ty(Ctx), Freq->Freq.getFrequency()))};
2219 if (Freq->IsEstimated)
2221 setMetadata(Ctx.getMDKindID(ExecutionFrequencyMDName), MDNode::get(Ctx, Ops));
2222}
2223
2224std::optional<VPExecutionFrequency>
2226 if (MDNode *Node = getInternalMetadata(ExecutionFrequencyMDName))
2228 return std::nullopt;
2229}
2230
2232 if (Metadata.empty())
2233 return;
2234 unsigned ID = getMDKindID(ExecutionFrequencyMDName);
2235 erase_if(Metadata, [ID](const auto &P) { return P.first == ID; });
2236}
2237
2239 SmallVector<std::pair<unsigned, MDNode *>> MetadataIntersection;
2240 for (const auto &[KindA, MDA] : Metadata) {
2241 for (const auto &[KindB, MDB] : Other.Metadata) {
2242 if (KindA == KindB && MDA == MDB) {
2243 MetadataIntersection.emplace_back(KindA, MDA);
2244 break;
2245 }
2246 }
2247 }
2248 Metadata = std::move(MetadataIntersection);
2249}
2250
2251#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2253 const Module *M = SlotTracker.getModule();
2254 if (Metadata.empty() || !M)
2255 return;
2256
2257 ArrayRef<StringRef> MDNames = SlotTracker.getMDNames();
2258 O << " (";
2259 interleaveComma(Metadata, O, [&](const auto &KindNodePair) {
2260 auto [Kind, Node] = KindNodePair;
2261 assert(Kind < MDNames.size() && !MDNames[Kind].empty() &&
2262 "Unexpected unnamed metadata kind");
2263 O << "!" << MDNames[Kind] << " ";
2264 // Print the values of branch weights, which are more informative than the
2265 // ID of the metadata node holding them.
2266 SmallVector<uint32_t> Weights;
2267 bool IsEstimatedProfile = MDNames[Kind] == EstimatedProfileMDName;
2268 if ((Kind == LLVMContext::MD_prof || IsEstimatedProfile) &&
2269 extractBranchWeights(Node, Weights)) {
2270 if (IsEstimatedProfile)
2271 O << "estimated ";
2272 O << "{";
2273 interleaveComma(Weights, O);
2274 O << "}";
2275 } else if (MDNames[Kind] == ExecutionFrequencyMDName) {
2276 // Print the frequency together with the probability it corresponds to.
2277 auto [Freq, IsEstimated] = getExecutionFrequencyFromMD(Node);
2278 const fltSemantics &Sem = APFloat::IEEEdouble();
2279 uint64_t Full =
2281 APFloat Percent = APFloat(Sem, Freq.getFrequency()) * APFloat(Sem, 100) /
2282 APFloat(Sem, Full);
2283 SmallString<16> PercentStr;
2284 Percent.toString(PercentStr, /*FormatPrecision=*/4);
2285 O << Freq.getFrequency() << " (" << PercentStr << "%"
2286 << (IsEstimated ? ", estimated" : "") << ")";
2287 } else {
2288 SlotTracker.printMetadataAsOperand(O, Node);
2289 }
2290 });
2291 O << ")";
2292}
2293#endif
2294
2296 assert(State.VF.isVector() && "not widening");
2297 assert(Variant != nullptr && "Can't create vector function.");
2298
2299 FunctionType *VFTy = Variant->getFunctionType();
2300 // Add return type if intrinsic is overloaded on it.
2302 for (const auto &I : enumerate(args())) {
2303 Value *Arg;
2304 // Some vectorized function variants may also take a scalar argument,
2305 // e.g. linear parameters for pointers. This needs to be the scalar value
2306 // from the start of the respective part when interleaving.
2307 if (!VFTy->getParamType(I.index())->isVectorTy())
2308 Arg = State.get(I.value(), VPLane(0));
2309 else
2310 Arg = State.get(I.value(), usesFirstLaneOnly(I.value()));
2311 Args.push_back(Arg);
2312 }
2313
2316 if (CI)
2317 CI->getOperandBundlesAsDefs(OpBundles);
2318
2319 CallInst *V = State.Builder.CreateCall(Variant, Args, OpBundles);
2320 applyFlags(*V);
2321 applyMetadata(*V);
2322 V->setCallingConv(Variant->getCallingConv());
2323
2324 if (!V->getType()->isVoidTy())
2325 State.set(this, V);
2326}
2327
2329 VPCostContext &Ctx) const {
2330 assert(getVectorizedTypeVF(Variant->getReturnType()) == VF &&
2331 "Variant return type must match VF");
2332 return computeCallCost(Variant, Ctx);
2333}
2334
2336 VPCostContext &Ctx) {
2337 return Ctx.TTI.getCallInstrCost(nullptr, Variant->getReturnType(),
2338 Variant->getFunctionType()->params(),
2339 Ctx.CostKind);
2340}
2341
2343 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
2344 assert(Variant && "Variant not set");
2345 FunctionType *VFTy = Variant->getFunctionType();
2346 return all_of(enumerate(args()), [VFTy, &Op](const auto &Arg) {
2347 auto [Idx, V] = Arg;
2348 Type *ArgTy = VFTy->getParamType(Idx);
2349 return V != Op || ArgTy->isIntegerTy() || ArgTy->isFloatingPointTy() ||
2350 ArgTy->isPointerTy() || ArgTy->isByteTy();
2351 });
2352}
2353
2354#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2356 VPSlotTracker &SlotTracker) const {
2357 O << Indent << "WIDEN-CALL ";
2358
2359 Function *CalledFn = getCalledScalarFunction();
2360 if (CalledFn->getReturnType()->isVoidTy())
2361 O << "void ";
2362 else {
2364 O << " = ";
2365 }
2366
2367 O << "call";
2368 printFlags(O);
2369 O << "@" << CalledFn->getName() << "(";
2370 interleaveComma(args(), O, [&O, &SlotTracker](VPValue *Op) {
2371 Op->printAsOperand(O, SlotTracker);
2372 });
2373 O << ")";
2374
2375 O << " (using library function";
2376 if (Variant->hasName())
2377 O << ": " << Variant->getName();
2378 O << ")";
2379}
2380#endif
2381
2383 assert(State.VF.isVector() && "not widening");
2384
2385 SmallVector<Type *, 2> TysForDecl;
2386 // Add return type if intrinsic is overloaded on it.
2387 if (isVectorIntrinsicWithOverloadTypeAtArg(VectorIntrinsicID, -1,
2388 State.TTI)) {
2389 Type *RetTy = toVectorizedTy(getScalarType(), State.VF);
2390 ArrayRef<Type *> ContainedTys = getContainedTypes(RetTy);
2391 for (auto [Idx, Ty] : enumerate(ContainedTys)) {
2393 Idx, State.TTI))
2394 TysForDecl.push_back(Ty);
2395 }
2396 }
2398 for (const auto &I : enumerate(operands())) {
2399 // Some intrinsics have a scalar argument - don't replace it with a
2400 // vector.
2401 Value *Arg;
2402 if (isVectorIntrinsicWithScalarOpAtArg(VectorIntrinsicID, I.index(),
2403 State.TTI))
2404 Arg = State.get(I.value(), VPLane(0));
2405 else
2406 Arg = State.get(I.value(), usesFirstLaneOnly(I.value()));
2407 if (isVectorIntrinsicWithOverloadTypeAtArg(VectorIntrinsicID, I.index(),
2408 State.TTI))
2409 TysForDecl.push_back(Arg->getType());
2410 Args.push_back(Arg);
2411 }
2412
2413 // Use vector version of the intrinsic.
2414 Module *M = State.Builder.getModule();
2415 Function *VectorF =
2416 Intrinsic::getOrInsertDeclaration(M, VectorIntrinsicID, TysForDecl);
2417 assert(VectorF &&
2418 "Can't retrieve vector intrinsic or vector-predication intrinsics.");
2419
2422 if (CI)
2423 CI->getOperandBundlesAsDefs(OpBundles);
2424
2425 CallInst *V = State.Builder.CreateCall(VectorF, Args, OpBundles);
2426
2427 applyFlags(*V);
2428 applyMetadata(*V);
2429
2430 return V;
2431}
2432
2434 CallInst *V = createVectorCall(State);
2435 if (!V->getType()->isVoidTy())
2436 State.set(this, V);
2437}
2438
2441 const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx) {
2442 Type *ScalarRetTy = R.getScalarType();
2443 // Skip the reverse operation cost for the mask.
2444 // FIXME: Remove this once redundant mask reverse operations can be eliminated
2445 // by VPlanTransforms::cse before cost computation.
2446 if (ID == Intrinsic::experimental_vp_reverse && ScalarRetTy->isIntegerTy(1))
2447 return InstructionCost(0);
2448
2449 // Some backends analyze intrinsic arguments to determine cost. Use the
2450 // underlying value for the operand if it has one. Otherwise try to use the
2451 // operand of the underlying call instruction, if there is one. Otherwise
2452 // clear Arguments.
2453 // TODO: Rework TTI interface to be independent of concrete IR values.
2455 for (const auto &[Idx, Op] : enumerate(Operands)) {
2456 auto *V = Op->getUnderlyingValue();
2457 if (!V) {
2458 if (auto *UI = dyn_cast_or_null<CallBase>(R.getUnderlyingValue())) {
2459 Arguments.push_back(UI->getArgOperand(Idx));
2460 continue;
2461 }
2462 Arguments.clear();
2463 break;
2464 }
2465 Arguments.push_back(V);
2466 }
2467
2468 Type *RetTy = VF.isVector() ? toVectorizedTy(ScalarRetTy, VF) : ScalarRetTy;
2469 SmallVector<Type *> ParamTys =
2470 map_to_vector(Operands, [&](const VPValue *Op) {
2471 return toVectorTy(Op->getScalarType(), VF);
2472 });
2473
2475 for (const VPValue *Op : Operands)
2476 if (isa<VPWidenRecipe>(Op) &&
2479 break;
2480 }
2481
2482 // TODO: Rework TTI interface to avoid reliance on underlying IntrinsicInst.
2483 IntrinsicCostAttributes CostAttrs(
2484 ID, RetTy, Arguments, ParamTys, R.getFastMathFlagsOrNone(),
2485 dyn_cast_or_null<IntrinsicInst>(R.getUnderlyingValue()),
2487 return Ctx.TTI.getIntrinsicInstrCost(CostAttrs, Ctx.CostKind);
2488}
2489
2491 VPCostContext &Ctx) const {
2492 return computeCallCost(VectorIntrinsicID, operands(), *this, VF, Ctx);
2493}
2494
2496 return Intrinsic::getBaseName(VectorIntrinsicID);
2497}
2498
2500 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
2501 return all_of(enumerate(operands()), [this, &Op](const auto &X) {
2502 auto [Idx, V] = X;
2504 Idx, nullptr);
2505 });
2506}
2507
2508#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2510 VPSlotTracker &SlotTracker) const {
2511 O << Indent << "WIDEN-INTRINSIC ";
2512 if (getScalarType()->isVoidTy()) {
2513 O << "void ";
2514 } else {
2516 O << " = ";
2517 }
2518
2519 O << "call";
2520 printFlags(O);
2521 O << getIntrinsicName() << "(";
2523 O << ")";
2524}
2525#endif
2526
2528 CallInst *MemI = createVectorCall(State);
2530 assert(PtrPos && "Expected a memory intrinsic with a valid pointer position");
2531 MemI->addParamAttr(
2532 *PtrPos, Attribute::getWithAlignment(MemI->getContext(), Alignment));
2533 if (!MemI->getType()->isVoidTy())
2534 State.set(this, MemI);
2535}
2536
2538 Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment,
2539 VPCostContext &Ctx) {
2540 return Ctx.TTI.getMemIntrinsicInstrCost(
2541 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr, IsMasked, Alignment),
2542 Ctx.CostKind);
2543}
2544
2547 VPCostContext &Ctx) const {
2548 Type *DataTy;
2550 DataTy = getOperand(*DataPos)->getScalarType();
2551 else
2552 DataTy = getScalarType();
2553 assert(!DataTy->isVoidTy() && "Expected a non-void data type");
2554 Type *Ty = toVectorTy(DataTy, VF);
2556 assert(MaskPos && "Expected a memory intrinsic with a valid mask position");
2558 !match(getOperand(*MaskPos), m_True()),
2559 Alignment, Ctx);
2560}
2561
2563 IRBuilderBase &Builder = State.Builder;
2564
2565 Value *Address = State.get(getOperand(0));
2566 Value *IncAmt = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
2567 VectorType *VTy = cast<VectorType>(Address->getType());
2568
2569 // The histogram intrinsic requires a mask even if the recipe doesn't;
2570 // if the mask operand was omitted then all lanes should be executed and
2571 // we just need to synthesize an all-true mask.
2572 Value *Mask = nullptr;
2573 if (VPValue *VPMask = getMask())
2574 Mask = State.get(VPMask);
2575 else
2576 Mask =
2577 Builder.CreateVectorSplat(VTy->getElementCount(), Builder.getInt1(1));
2578
2579 // If this is a subtract, we want to invert the increment amount. We may
2580 // add a separate intrinsic in future, but for now we'll try this.
2581 if (Opcode == Instruction::Sub)
2582 IncAmt = Builder.CreateNeg(IncAmt);
2583 else
2584 assert(Opcode == Instruction::Add && "only add or sub supported for now");
2585
2586 Instruction *HistogramInst = State.Builder.CreateIntrinsicWithoutFolding(
2587 Intrinsic::experimental_vector_histogram_add, {VTy, IncAmt->getType()},
2588 {Address, IncAmt, Mask});
2589 applyMetadata(*HistogramInst);
2590}
2591
2593 VPCostContext &Ctx) const {
2594 // FIXME: Take the gather and scatter into account as well. For now we're
2595 // generating the same cost as the fallback path, but we'll likely
2596 // need to create a new TTI method for determining the cost, including
2597 // whether we can use base + vec-of-smaller-indices or just
2598 // vec-of-pointers.
2599 assert(VF.isVector() && "Invalid VF for histogram cost");
2600 Type *AddressTy = getOperand(0)->getScalarType();
2601 VPValue *IncAmt = getOperand(1);
2602 Type *IncTy = IncAmt->getScalarType();
2603 VectorType *VTy = VectorType::get(IncTy, VF);
2604
2605 // Assume that a non-constant update value (or a constant != 1) requires
2606 // a multiply, and add that into the cost.
2607 InstructionCost MulCost =
2608 Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, VTy, Ctx.CostKind);
2609 if (match(IncAmt, m_One()))
2610 MulCost = TTI::TCC_Free;
2611
2612 // Find the cost of the histogram operation itself.
2613 Type *PtrTy = VectorType::get(AddressTy, VF);
2614 Type *MaskTy = VectorType::get(Type::getInt1Ty(Ctx.LLVMCtx), VF);
2615 IntrinsicCostAttributes ICA(Intrinsic::experimental_vector_histogram_add,
2616 Type::getVoidTy(Ctx.LLVMCtx),
2617 {PtrTy, IncTy, MaskTy});
2618
2619 // Add the costs together with the add/sub operation.
2620 return Ctx.TTI.getIntrinsicInstrCost(ICA, Ctx.CostKind) + MulCost +
2621 Ctx.TTI.getArithmeticInstrCost(Opcode, VTy, Ctx.CostKind);
2622}
2623
2624#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2626 VPSlotTracker &SlotTracker) const {
2627 O << Indent << "WIDEN-HISTOGRAM buckets: ";
2629
2630 if (Opcode == Instruction::Sub)
2631 O << ", dec: ";
2632 else {
2633 assert(Opcode == Instruction::Add);
2634 O << ", inc: ";
2635 }
2637
2638 if (VPValue *Mask = getMask()) {
2639 O << ", mask: ";
2640 Mask->printAsOperand(O, SlotTracker);
2641 }
2642}
2643#endif
2644
2645VPIRFlags::FastMathFlagsTy::FastMathFlagsTy(const FastMathFlags &FMF) {
2646 AllowReassoc = FMF.allowReassoc();
2647 NoNaNs = FMF.noNaNs();
2648 NoInfs = FMF.noInfs();
2649 NoSignedZeros = FMF.noSignedZeros();
2650 AllowReciprocal = FMF.allowReciprocal();
2651 AllowContract = FMF.allowContract();
2652 ApproxFunc = FMF.approxFunc();
2653}
2654
2655VPIRFlags VPIRFlags::getDefaultFlags(unsigned Opcode, Type *ResultTy) {
2656 switch (Opcode) {
2657 case Instruction::Add:
2658 case Instruction::Sub:
2659 case Instruction::Mul:
2660 case Instruction::Shl:
2662 return WrapFlagsTy(false, false);
2663 case Instruction::Trunc:
2664 return TruncFlagsTy(false, false);
2665 case Instruction::Or:
2666 return DisjointFlagsTy(false);
2667 case Instruction::AShr:
2668 case Instruction::LShr:
2669 case Instruction::UDiv:
2670 case Instruction::SDiv:
2671 return ExactFlagsTy(false);
2672 case Instruction::GetElementPtr:
2675 return GEPNoWrapFlags::none();
2676 case Instruction::ZExt:
2677 case Instruction::UIToFP:
2678 return NonNegFlagsTy(false);
2679 case Instruction::FAdd:
2680 case Instruction::FSub:
2681 case Instruction::FMul:
2682 case Instruction::FDiv:
2683 case Instruction::FRem:
2684 case Instruction::FNeg:
2685 case Instruction::FPExt:
2686 case Instruction::FPTrunc:
2687 return FastMathFlags();
2688 case Instruction::Select:
2689 case Instruction::PHI:
2690 case Instruction::Call:
2691 // Selects, phis and calls only have fast-math flags if they have a
2692 // supported floating-point result type.
2694 return FastMathFlags();
2695 return VPIRFlags();
2696 case Instruction::ICmp:
2697 case Instruction::FCmp:
2699 llvm_unreachable("opcode requires explicit flags");
2700 default:
2701 return VPIRFlags();
2702 }
2703}
2704
2705#if !defined(NDEBUG)
2706bool VPIRFlags::flagsValidForOpcode(unsigned Opcode) const {
2707 switch (OpType) {
2708 case OperationType::OverflowingBinOp:
2709 return Opcode == Instruction::Add || Opcode == Instruction::Sub ||
2710 Opcode == Instruction::Mul || Opcode == Instruction::Shl ||
2711 Opcode == VPInstruction::VPInstruction::CanonicalIVIncrementForPart;
2712 case OperationType::Trunc:
2713 return Opcode == Instruction::Trunc;
2714 case OperationType::DisjointOp:
2715 return Opcode == Instruction::Or;
2716 case OperationType::PossiblyExactOp:
2717 return Opcode == Instruction::AShr || Opcode == Instruction::LShr ||
2718 Opcode == Instruction::UDiv || Opcode == Instruction::SDiv;
2719 case OperationType::GEPOp:
2720 return Opcode == Instruction::GetElementPtr ||
2721 Opcode == VPInstruction::PtrAdd ||
2722 Opcode == VPInstruction::WidePtrAdd;
2723 case OperationType::FPMathOp:
2724 return Opcode == Instruction::Call || Opcode == Instruction::FAdd ||
2725 Opcode == Instruction::FMul || Opcode == Instruction::FSub ||
2726 Opcode == Instruction::FNeg || Opcode == Instruction::FDiv ||
2727 Opcode == Instruction::FRem || Opcode == Instruction::FPExt ||
2728 Opcode == Instruction::FPTrunc || Opcode == Instruction::PHI ||
2729 Opcode == Instruction::Select || Opcode == Instruction::SIToFP ||
2730 Opcode == Instruction::UIToFP ||
2731 Opcode == VPInstruction::WideIVStep ||
2733 case OperationType::FCmp:
2734 return Opcode == Instruction::FCmp;
2735 case OperationType::NonNegOp:
2736 return Opcode == Instruction::ZExt || Opcode == Instruction::UIToFP;
2737 case OperationType::Cmp:
2738 return Opcode == Instruction::FCmp || Opcode == Instruction::ICmp;
2739 case OperationType::ReductionOp:
2741 case OperationType::Other:
2742 return true;
2743 }
2744 llvm_unreachable("Unknown OperationType enum");
2745}
2746
2748 Type *ResultTy) const {
2749 // Handle opcodes without default flags.
2750 if (Opcode == Instruction::ICmp)
2751 return OpType == OperationType::Cmp;
2752 if (Opcode == Instruction::FCmp)
2753 return OpType == OperationType::FCmp;
2755 return OpType == OperationType::ReductionOp;
2756
2757 OperationType Required = getDefaultFlags(Opcode, ResultTy).OpType;
2758 return Required == OperationType::Other || Required == OpType;
2759}
2760#endif
2761
2762#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2763static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind) {
2764 switch (Kind) {
2765 case RecurKind::None:
2766 OS << "none";
2767 break;
2768 case RecurKind::Add:
2769 OS << "add";
2770 break;
2771 case RecurKind::Sub:
2772 OS << "sub";
2773 break;
2775 OS << "add-chain-with-subs";
2776 break;
2777 case RecurKind::Mul:
2778 OS << "mul";
2779 break;
2780 case RecurKind::Or:
2781 OS << "or";
2782 break;
2783 case RecurKind::And:
2784 OS << "and";
2785 break;
2786 case RecurKind::Xor:
2787 OS << "xor";
2788 break;
2789 case RecurKind::SMin:
2790 OS << "smin";
2791 break;
2792 case RecurKind::SMax:
2793 OS << "smax";
2794 break;
2795 case RecurKind::UMin:
2796 OS << "umin";
2797 break;
2798 case RecurKind::UMax:
2799 OS << "umax";
2800 break;
2801 case RecurKind::FAdd:
2802 OS << "fadd";
2803 break;
2805 OS << "fadd-chain-with-subs";
2806 break;
2807 case RecurKind::FSub:
2808 OS << "fsub";
2809 break;
2810 case RecurKind::FMul:
2811 OS << "fmul";
2812 break;
2813 case RecurKind::FMin:
2814 OS << "fmin";
2815 break;
2816 case RecurKind::FMax:
2817 OS << "fmax";
2818 break;
2819 case RecurKind::FMinNum:
2820 OS << "fminnum";
2821 break;
2822 case RecurKind::FMaxNum:
2823 OS << "fmaxnum";
2824 break;
2826 OS << "fminimum";
2827 break;
2829 OS << "fmaximum";
2830 break;
2832 OS << "fminimumnum";
2833 break;
2835 OS << "fmaximumnum";
2836 break;
2837 case RecurKind::FMulAdd:
2838 OS << "fmuladd";
2839 break;
2840 case RecurKind::AnyOf:
2841 OS << "any-of";
2842 break;
2843 case RecurKind::FindIV:
2844 OS << "find-iv";
2845 break;
2847 OS << "find-last";
2848 break;
2849 }
2850}
2851
2853 switch (OpType) {
2854 case OperationType::Cmp:
2856 break;
2857 case OperationType::FCmp:
2860 break;
2861 case OperationType::DisjointOp:
2862 if (DisjointFlags.IsDisjoint)
2863 O << " disjoint";
2864 break;
2865 case OperationType::PossiblyExactOp:
2866 if (ExactFlags.IsExact)
2867 O << " exact";
2868 break;
2869 case OperationType::OverflowingBinOp:
2870 if (WrapFlags.HasNUW)
2871 O << " nuw";
2872 if (WrapFlags.HasNSW)
2873 O << " nsw";
2874 break;
2875 case OperationType::Trunc:
2876 if (TruncFlags.HasNUW)
2877 O << " nuw";
2878 if (TruncFlags.HasNSW)
2879 O << " nsw";
2880 break;
2881 case OperationType::FPMathOp:
2883 break;
2884 case OperationType::GEPOp: {
2886 if (Flags.isInBounds())
2887 O << " inbounds";
2888 else if (Flags.hasNoUnsignedSignedWrap())
2889 O << " nusw";
2890 if (Flags.hasNoUnsignedWrap())
2891 O << " nuw";
2892 break;
2893 }
2894 case OperationType::NonNegOp:
2895 if (NonNegFlags.NonNeg)
2896 O << " nneg";
2897 break;
2898 case OperationType::ReductionOp: {
2899 O << " (";
2901 if (isReductionInLoop())
2902 O << ", in-loop";
2903 if (isReductionOrdered())
2904 O << ", ordered";
2905 O << ")";
2907 break;
2908 }
2909 case OperationType::Other:
2910 break;
2911 }
2912 O << " ";
2913}
2914#endif
2915
2917 auto &Builder = State.Builder;
2918 switch (Opcode) {
2919 case Instruction::Call:
2920 case Instruction::UncondBr:
2921 case Instruction::CondBr:
2922 case Instruction::PHI:
2923 case Instruction::GetElementPtr:
2924 llvm_unreachable("This instruction is handled by a different recipe.");
2925 case Instruction::UDiv:
2926 case Instruction::SDiv:
2927 case Instruction::SRem:
2928 case Instruction::URem:
2929 case Instruction::Add:
2930 case Instruction::FAdd:
2931 case Instruction::Sub:
2932 case Instruction::FSub:
2933 case Instruction::FNeg:
2934 case Instruction::Mul:
2935 case Instruction::FMul:
2936 case Instruction::FDiv:
2937 case Instruction::FRem:
2938 case Instruction::Shl:
2939 case Instruction::LShr:
2940 case Instruction::AShr:
2941 case Instruction::And:
2942 case Instruction::Or:
2943 case Instruction::Xor: {
2944 // Just widen unops and binops.
2946 for (VPValue *VPOp : operands())
2947 Ops.push_back(State.get(VPOp));
2948
2949 Value *V = Builder.CreateNAryOp(Opcode, Ops);
2950
2951 if (auto *VecOp = dyn_cast<Instruction>(V)) {
2952 applyFlags(*VecOp);
2953 applyMetadata(*VecOp);
2954 }
2955
2956 // Use this vector value for all users of the original instruction.
2957 State.set(this, V);
2958 break;
2959 }
2960 case Instruction::ExtractValue: {
2961 assert(getNumOperands() == 2 && "expected single level extractvalue");
2962 Value *Op = State.get(getOperand(0));
2963 Value *Extract = Builder.CreateExtractValue(
2964 Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
2965 State.set(this, Extract);
2966 break;
2967 }
2968 case Instruction::Freeze: {
2969 Value *Op = State.get(getOperand(0));
2970 Value *Freeze = Builder.CreateFreeze(Op);
2971 State.set(this, Freeze);
2972 break;
2973 }
2974 case Instruction::ICmp:
2975 case Instruction::FCmp: {
2976 // Widen compares. Generate vector compares.
2977 bool FCmp = Opcode == Instruction::FCmp;
2978 Value *A = State.get(getOperand(0));
2979 Value *B = State.get(getOperand(1));
2980 Value *C = nullptr;
2981 if (FCmp) {
2982 C = Builder.CreateFCmp(getPredicate(), A, B);
2983 } else {
2984 C = Builder.CreateICmp(getPredicate(), A, B);
2985 }
2986 if (auto *I = dyn_cast<Instruction>(C)) {
2987 applyFlags(*I);
2988 applyMetadata(*I);
2989 }
2990 State.set(this, C);
2991 break;
2992 }
2993 case Instruction::Select: {
2994 VPValue *CondOp = getOperand(0);
2995 Value *Cond = State.get(CondOp, vputils::isSingleScalar(CondOp));
2996 Value *Op0 = State.get(getOperand(1));
2997 Value *Op1 = State.get(getOperand(2));
2998 Value *Sel = State.Builder.CreateSelect(Cond, Op0, Op1);
2999 State.set(this, Sel);
3000 if (auto *I = dyn_cast<Instruction>(Sel)) {
3002 applyFlags(*I);
3003 applyMetadata(*I);
3004 }
3005 break;
3006 }
3007 default:
3008 // This instruction is not vectorized by simple widening.
3009 LLVM_DEBUG(dbgs() << "LV: Found an unhandled opcode : "
3010 << Instruction::getOpcodeName(Opcode));
3011 llvm_unreachable("Unhandled instruction!");
3012 } // end of switch.
3013
3014#if !defined(NDEBUG)
3015 // Verify that VPlan type inference results agree with the type of the
3016 // generated values.
3017 assert(VectorType::get(this->getScalarType(), State.VF) ==
3018 State.get(this)->getType() &&
3019 "inferred type and type from generated instructions do not match");
3020#endif
3021}
3022
3024 VPCostContext &Ctx) const {
3025 switch (Opcode) {
3026 case Instruction::UDiv:
3027 case Instruction::SDiv:
3028 case Instruction::SRem:
3029 case Instruction::URem:
3030 // If the div/rem operation isn't safe to speculate and requires
3031 // predication, then the only way we can even create a vplan is to insert
3032 // a select on the second input operand to ensure we use the value of 1
3033 // for the inactive lanes. The select will be costed separately.
3034 case Instruction::FNeg:
3035 case Instruction::Add:
3036 case Instruction::FAdd:
3037 case Instruction::Sub:
3038 case Instruction::FSub:
3039 case Instruction::Mul:
3040 case Instruction::FMul:
3041 case Instruction::FDiv:
3042 case Instruction::FRem:
3043 case Instruction::Shl:
3044 case Instruction::LShr:
3045 case Instruction::AShr:
3046 case Instruction::And:
3047 case Instruction::Or:
3048 case Instruction::Xor:
3049 case Instruction::Freeze:
3050 case Instruction::ExtractValue:
3051 case Instruction::ICmp:
3052 case Instruction::FCmp:
3053 case Instruction::Select:
3054 return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
3055 default:
3056 llvm_unreachable("Unsupported opcode for instruction");
3057 }
3058}
3059
3060#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3062 VPSlotTracker &SlotTracker) const {
3063 O << Indent << "WIDEN ";
3065 O << " = " << Instruction::getOpcodeName(Opcode);
3066 printFlags(O);
3068}
3069#endif
3070
3072 auto &Builder = State.Builder;
3073 /// Vectorize casts.
3074 assert(State.VF.isVector() && "Not vectorizing?");
3075 Type *DestTy = VectorType::get(getScalarType(), State.VF);
3076 VPValue *Op = getOperand(0);
3077 Value *A = State.get(Op);
3078 Value *Cast = Builder.CreateCast(Instruction::CastOps(Opcode), A, DestTy);
3079 State.set(this, Cast);
3080 if (auto *CastOp = dyn_cast<Instruction>(Cast)) {
3081 applyFlags(*CastOp);
3082 applyMetadata(*CastOp);
3083 }
3084}
3085
3090
3091#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3093 VPSlotTracker &SlotTracker) const {
3094 O << Indent << "WIDEN-CAST ";
3096 O << " = " << Instruction::getOpcodeName(Opcode);
3097 printFlags(O);
3099 O << " to " << *getScalarType();
3100}
3101#endif
3102
3104 VPCostContext &Ctx) const {
3105 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
3106}
3107
3108#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3110 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
3111 O << Indent;
3113 O << " = WIDEN-INDUCTION";
3114 printFlags(O);
3116
3117 if (auto *TI = getTruncInst())
3118 O << " (truncated to " << *TI->getType() << ")";
3119}
3120#endif
3121
3123 // The step may be defined by a recipe in the preheader (e.g. if it requires
3124 // SCEV expansion), but for the canonical induction the step is required to be
3125 // 1, which is represented as live-in.
3126 return match(getStartValue(), m_ZeroInt()) &&
3127 match(getStepValue(), m_One()) &&
3128 getScalarType() == getRegion()->getCanonicalIVType();
3129}
3130
3133 VPCostContext &Ctx) const {
3134 // A widened induction generates a vector phi and increments it by the
3135 // splatted step each iteration.
3137 InstructionCost Cost = Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
3138 Type *StepTy = getScalarType();
3139 unsigned IncOpc = ID.getKind() == InductionDescriptor::IK_IntInduction
3140 ? Instruction::Add
3141 : ID.getInductionOpcode();
3142 assert(IncOpc != Instruction::BinaryOpsEnd &&
3143 "induction must have a valid increment opcode");
3144 return Cost + Ctx.TTI.getArithmeticInstrCost(IncOpc, toVectorTy(StepTy, VF),
3145 Ctx.CostKind);
3146}
3147
3148/// Returns the ConstantFP \p V wraps, or nullptr if it does not wrap one.
3149static const ConstantFP *getConstantFP(const VPValue *V) {
3150 auto *C = dyn_cast<VPConstant>(V);
3151 return C ? dyn_cast<ConstantFP>(C->getConstant()) : nullptr;
3152}
3153
3155 VPCostContext &Ctx) const {
3156 // The cost model for this is modelled on expandVPDerivedIV in
3157 // VPlanTransforms.cpp. In order to avoid overly pessimistic costs that can
3158 // negatively affect vectorization it takes into account any expected
3159 // simplifications that happen in simplifyRecipe.
3160 switch (getInductionKind()) {
3161 default:
3162 // TODO: Compute cost for remaining kinds.
3163 break;
3165 // There are currently no tests that expose a path where all lanes are
3166 // used, so it's better to bail out for now.
3167 if (!vputils::onlyFirstLaneUsed(this))
3168 break;
3169
3170 // Start off by assuming we need both mul and add, then refine this.
3171 bool NeedsMul = true, NeedsAdd = true, NeedsShl = false;
3172
3173 // If the start value is zero the add gets folded away.
3174 if (auto *StartC = dyn_cast<VPConstantInt>(getStartValue()))
3175 NeedsAdd = !StartC->isZero();
3176
3177 // For some values of step the arithmetic changes:
3178 // 1. A step of 1 requires no operation.
3179 // 2. A step of -1 requires a negate.
3180 // 3. A power-of-2 step will use a shl, instead of a mul.
3181 Type *StepTy = getStepValue()->getScalarType();
3183 if (auto *StepC = dyn_cast<VPConstantInt>(getStepValue())) {
3184 if (StepC->isOne())
3185 NeedsMul = false;
3186 else if (StepC->getAPInt().isAllOnes()) {
3187 // This will most likely end up as a negate in simplifyRecipe, and
3188 // the negate will be combined with the add to make a sub.
3189 // NOTE: This is perhaps an invalid assumption that the cost of an
3190 // 'add' is the same as a 'sub'.
3191 NeedsMul = false;
3192 NeedsAdd = true;
3193 } else if (StepC->getAPInt().isPowerOf2()) {
3194 // This will most likely end up as a shift-left in simplifyRecipe
3195 NeedsMul = false;
3196 NeedsShl = true;
3197 }
3198 }
3199
3200 // Add the cost of the conversion from index to step type if the index
3201 // will be used.
3202 Type *IndexTy = getIndex()->getScalarType();
3203 unsigned StepTySize = StepTy->getScalarSizeInBits();
3204 unsigned IndexTySize = IndexTy->getScalarSizeInBits();
3205 if ((NeedsAdd || NeedsMul || NeedsShl) && StepTySize != IndexTySize) {
3206 unsigned CastOpc =
3207 StepTySize < IndexTySize ? Instruction::Trunc : Instruction::ZExt;
3208 Cost += Ctx.TTI.getCastInstrCost(
3209 CastOpc, StepTy, IndexTy, TTI::CastContextHint::None, Ctx.CostKind);
3210 }
3211
3212 if (NeedsMul)
3213 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, StepTy,
3214 Ctx.CostKind);
3215 if (NeedsShl)
3216 Cost += Ctx.TTI.getArithmeticInstrCost(
3217 Instruction::Shl, StepTy, Ctx.CostKind,
3218 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
3219 {TargetTransformInfo::OK_UniformConstantValue,
3220 TargetTransformInfo::OP_None});
3221 if (NeedsAdd)
3222 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Add, StepTy,
3223 Ctx.CostKind);
3224 return Cost;
3225 }
3227 // There are currently no tests that expose a path where all lanes are
3228 // used, so it's better to bail out for now.
3229 if (!vputils::onlyFirstLaneUsed(this))
3230 break;
3231
3232 // Unlike the integer case, converting the index to the FP step type is
3233 // unavoidable: the index is always the integer canonical IV, so this
3234 // cast is never folded away.
3235 Type *StepTy = getStepValue()->getScalarType();
3236 Type *IndexTy = getIndex()->getScalarType();
3238 Ctx.TTI.getCastInstrCost(Instruction::SIToFP, StepTy, IndexTy,
3239 TTI::CastContextHint::None, Ctx.CostKind);
3240
3241 // If the step is 1.0, the multiply is an exact identity and gets folded
3242 // away, independent of fast-math flags.
3243 const ConstantFP *StepC = getConstantFP(getStepValue());
3244 bool NeedsMul = !StepC || !StepC->isOne();
3245
3246 // "fadd -0.0, X" folds to X unconditionally, but "fadd 0.0, X" only folds
3247 // to X without nsz if X can be proven to never be -0.0, which we cannot, as
3248 // Step may be -0.0.
3249 // TODO: Consider fast-math flags when they are available in
3250 // VPDerivedIVRecipe.
3251 const ConstantFP *StartC = getConstantFP(getStartValue());
3252 bool AddFolds = getFPBinOp()->getOpcode() == Instruction::FAdd && StartC &&
3253 StartC->isZero() && (StartC->isNegZero() || !NeedsMul);
3254
3255 if (NeedsMul)
3256 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::FMul, StepTy,
3257 Ctx.CostKind);
3258 if (!AddFolds)
3259 Cost += Ctx.TTI.getArithmeticInstrCost(getFPBinOp()->getOpcode(), StepTy,
3260 Ctx.CostKind);
3261 return Cost;
3262 }
3263 }
3264
3265 return 0;
3266}
3267
3268#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3270 VPSlotTracker &SlotTracker) const {
3271 O << Indent;
3273 O << " = DERIVED-IV";
3274 printFlags(O);
3275 getStartValue()->printAsOperand(O, SlotTracker);
3276 O << " + ";
3277 getOperand(1)->printAsOperand(O, SlotTracker);
3278 O << " * ";
3279 getStepValue()->printAsOperand(O, SlotTracker);
3280}
3281#endif
3282
3286
3288 VPCostContext &Ctx) const {
3289 Type *BaseIVTy = getOperand(0)->getScalarType();
3290 assert((BaseIVTy->isIntegerTy() || BaseIVTy->isFloatingPointTy()) &&
3291 "VPScalarIVStepsRecipe is only created for integer and FP inductions");
3292
3293 // If only the first lane is used, then there won't be any code that remains
3294 // in the loop for the first unrolled part.
3296 return 0;
3297
3298 // If the vector body executes at most once, the canonical IV is a constant
3299 // and every lane's step folds away with it.
3300 if (VPCostContext::executesAtMostOnce(*getParent()->getPlan(), VF))
3301 return 0;
3302
3303 // Typically the operations are:
3304 // 1. Add the start index to each lane value.
3305 // 2. Multiply the start index by the step.
3306 // 3. Add the scaled start index to base IV.
3307 // Any code generated for 1 and 2 should be loop invariant and therefore
3308 // hoisted out of the loop. We only need to add on the cost of 3.
3310 if (BaseIVTy->isFloatingPointTy()) {
3311 // Unlike the integer case, the users of an FP induction cannot be re-based
3312 // on a common value, so each lane needs its own FAdd/FSub.
3313 assert(!VF.isScalable() &&
3314 "FP scalar steps for all lanes are only created for fixed VFs");
3315 Cost = Ctx.TTI.getArithmeticInstrCost(InductionOpcode, BaseIVTy,
3316 Ctx.CostKind) *
3317 (VF.getFixedValue() - 1);
3318 } else {
3319 // Given the users of VPScalarIVStepsRecipe tend to be scalarized GEPs, i.e.
3320 // %add1 = add i32 %iv, 0
3321 // %add2 = add i32 %iv, 1
3322 // %gep1 = getelementptr i8, ptr %p, i32 %add1
3323 // %gep2 = getelementptr i8, ptr %p, i32 %add2
3324 // it's very likely that these GEPs will all be rewritten to have a common
3325 // base such that what's left is just
3326 // %base_gep = getelementptr i8, ptr %p, i32 %iv
3327 // %gep1 = getelementptr i8, ptr %base_gep, i32 0
3328 // %gep2 = getelementptr i8, ptr %base_gep, i32 1
3329 // Therefore, in reality the cost is somewhere betwen 1*AddCost and
3330 // (NumLanes - 1) * AddCost. For now, assume the cost of a single add.
3331 Cost = Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy,
3332 Ctx.CostKind);
3333 }
3334
3335 // If the steps are generated inside a replicate region, scale by execution
3336 // probability.
3337 const VPRegionBlock *Region = getRegion();
3338 if (Region && Region->isReplicator())
3339 Cost /= Ctx.getCostDivisor(
3340 Region->getEntryBranchOnMask()->getExecutionFrequency());
3341 return Cost;
3342}
3343
3345 // Fast-math-flags propagate from the original induction instruction.
3346 IRBuilder<>::FastMathFlagGuard FMFG(State.Builder);
3347 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
3348
3349 /// Compute scalar induction steps. \p ScalarIV is the scalar induction
3350 /// variable on which to base the steps, \p Step is the size of the step.
3351
3352 Value *BaseIV = State.get(getOperand(0), VPLane(0));
3353 Value *Step = State.get(getStepValue(), VPLane(0));
3354 IRBuilderBase &Builder = State.Builder;
3355
3356 // Ensure step has the same type as that of scalar IV.
3357 Type *BaseIVTy = BaseIV->getType()->getScalarType();
3358 assert(BaseIVTy == Step->getType() && "Types of BaseIV and Step must match!");
3359
3360 // We build scalar steps for both integer and floating-point induction
3361 // variables. Here, we determine the kind of arithmetic we will perform.
3364 if (BaseIVTy->isIntegerTy()) {
3365 AddOp = Instruction::Add;
3366 MulOp = Instruction::Mul;
3367 } else {
3368 AddOp = InductionOpcode;
3369 MulOp = Instruction::FMul;
3370 }
3371
3372 // Lanes other than the first have been materialized as separate
3373 // single-scalar recipes by replicateByVF, each with its own start index.
3374 assert((vputils::onlyFirstLaneUsed(this) || State.VF.isScalar()) &&
3375 "must have been replicated by VF");
3376 Value *StartIdx = getStartIndex() ? State.get(getStartIndex(), true)
3377 : Constant::getNullValue(BaseIVTy);
3378 auto *Mul = Builder.CreateBinOp(MulOp, StartIdx, Step);
3379 auto *Add = Builder.CreateBinOp(AddOp, BaseIV, Mul);
3380 State.set(this, Add, VPLane(0));
3381}
3382
3383#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3385 VPSlotTracker &SlotTracker) const {
3386 O << Indent;
3388 O << " = SCALAR-STEPS ";
3390}
3391#endif
3392
3394 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
3396}
3397
3399 assert(State.VF.isVector() && "not widening");
3400 auto Ops = map_to_vector(operands(), [&](VPValue *Op) {
3401 return State.get(Op, vputils::isSingleScalar(Op));
3402 });
3403 auto *GEP =
3404 State.Builder.CreateGEP(getSourceElementType(), Ops.front(),
3405 drop_begin(Ops), "wide.gep", getGEPNoWrapFlags());
3406 State.set(this, GEP, vputils::isSingleScalar(this));
3407}
3408
3409#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3411 VPSlotTracker &SlotTracker) const {
3412 O << Indent << "WIDEN-GEP ";
3414 O << " = getelementptr";
3415 printFlags(O);
3417}
3418#endif
3419
3421 assert(!getOffset() && "Unexpected offset operand");
3422 VPBuilder Builder(this);
3423 VPlan &Plan = *getParent()->getPlan();
3424 VPValue *VFVal = getVFValue();
3425 const DataLayout &DL = Plan.getDataLayout();
3426 Type *IndexTy = DL.getIndexType(this->getScalarType());
3427 VPValue *Stride =
3428 Plan.getConstantInt(IndexTy, getStride(), /*IsSigned=*/true);
3429 VPValue *VF =
3430 Builder.createScalarZExtOrTrunc(VFVal, IndexTy, DebugLoc::getUnknown());
3431
3432 // Offset for Part0 = Offset0 = Stride * (VF - 1).
3433 VPInstruction *VFMinusOne =
3434 Builder.createSub(VF, Plan.getConstantInt(IndexTy, 1u),
3435 DebugLoc::getUnknown(), "", {true, true});
3436 VPInstruction *Offset0 =
3437 Builder.createOverflowingOp(Instruction::Mul, {VFMinusOne, Stride});
3438
3439 // Offset for PartN = Offset0 + Part * Stride * VF.
3440 VPValue *PartxStride =
3441 Plan.getConstantInt(IndexTy, Part * getStride(), /*IsSigned=*/true);
3442 VPValue *Offset = Builder.createAdd(
3443 Offset0,
3444 Builder.createOverflowingOp(Instruction::Mul, {PartxStride, VF}));
3446}
3447
3449 auto &Builder = State.Builder;
3450 assert(getOffset() && "Expected prior materialization of offset");
3451 Value *Ptr = State.get(getPointer(), true);
3452 Value *Offset = State.get(getOffset(), true);
3453 Value *ResultPtr = Builder.CreateGEP(getSourceElementType(), Ptr, Offset, "",
3455 State.set(this, ResultPtr, /*IsScalar*/ true);
3456}
3457
3458#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3460 VPSlotTracker &SlotTracker) const {
3461 O << Indent;
3463 O << " = vector-end-pointer";
3464 printFlags(O);
3465 getSourceElementType()->print(O);
3466 O << ", ";
3468}
3469#endif
3470
3472 assert(getVFxPart() &&
3473 "Expected prior simplification of recipe without VFxPart");
3474
3475 auto &Builder = State.Builder;
3476 Value *Ptr = State.get(getOperand(0), VPLane(0));
3477 Value *Offset = State.get(getVFxPart(), true);
3478 // TODO: Expand to VPInstruction to support constant folding.
3479 if (!match(getStride(), m_One())) {
3480 Value *Stride = Builder.CreateZExtOrTrunc(State.get(getStride(), true),
3481 Offset->getType());
3482 Offset = Builder.CreateMul(Offset, Stride);
3483 }
3484 Value *ResultPtr = Builder.CreateGEP(getSourceElementType(), Ptr, Offset, "",
3486 State.set(this, ResultPtr, /*IsScalar*/ true);
3487}
3488
3489#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3491 VPSlotTracker &SlotTracker) const {
3492 O << Indent;
3494 O << " = vector-pointer";
3495 printFlags(O);
3496 getSourceElementType()->print(O);
3497 O << ", ";
3499}
3500#endif
3501
3503 VPCostContext &Ctx) const {
3504 // A blend will be expanded to a select VPInstruction, which will generate a
3505 // scalar select if only the first lane is used.
3507 VF = ElementCount::getFixed(1);
3508
3509 Type *ResultTy = toVectorTy(this->getScalarType(), VF);
3510 Type *CmpTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF);
3511
3513 for (unsigned I = 1, E = getNumIncomingValues(); I != E; ++I) {
3514 CmpPredicate Pred;
3515 if (!match(getMask(I), m_Cmp(Pred, m_VPValue(), m_VPValue())))
3516 Pred = getScalarType()->isFloatingPointTy() ? CmpInst::BAD_FCMP_PREDICATE
3518 Cost += Ctx.TTI.getCmpSelInstrCost(Instruction::Select, ResultTy, CmpTy,
3519 Pred, Ctx.CostKind);
3520 }
3521 return Cost;
3522}
3523
3524#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3526 VPSlotTracker &SlotTracker) const {
3527 O << Indent << "BLEND ";
3529 O << " =";
3530 printFlags(O);
3531 if (getNumIncomingValues() == 1) {
3532 // Not a User of any mask: not really blending, this is a
3533 // single-predecessor phi.
3534 getIncomingValue(0)->printAsOperand(O, SlotTracker);
3535 } else {
3536 for (unsigned I = 0, E = getNumIncomingValues(); I < E; ++I) {
3537 if (I != 0)
3538 O << " ";
3539 getIncomingValue(I)->printAsOperand(O, SlotTracker);
3540 if (I == 0 && isNormalized())
3541 continue;
3542 O << "/";
3543 getMask(I)->printAsOperand(O, SlotTracker);
3544 }
3545 }
3546}
3547#endif
3548
3552 "In-loop AnyOf reductions aren't currently supported");
3553 // Propagate the fast-math flags carried by the underlying instruction.
3554 IRBuilderBase::FastMathFlagGuard FMFGuard(State.Builder);
3555 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
3556 Value *NewVecOp = State.get(getVecOp());
3557 if (VPValue *Cond = getCondOp()) {
3558 Value *NewCond = State.get(Cond, State.VF.isScalar());
3559 VectorType *VecTy = dyn_cast<VectorType>(NewVecOp->getType());
3560 Type *ElementTy = VecTy ? VecTy->getElementType() : NewVecOp->getType();
3561
3562 Value *Start =
3564 if (State.VF.isVector())
3565 Start = State.Builder.CreateVectorSplat(VecTy->getElementCount(), Start);
3566
3567 Value *Select = State.Builder.CreateSelect(NewCond, NewVecOp, Start);
3568 NewVecOp = Select;
3569 }
3570 Value *NewRed;
3571 Value *NextInChain;
3572 if (isOrdered()) {
3573 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/true);
3574 if (State.VF.isVector())
3575 NewRed =
3576 createOrderedReduction(State.Builder, Kind, NewVecOp, PrevInChain);
3577 else
3578 NewRed = State.Builder.CreateBinOp(
3580 PrevInChain, NewVecOp);
3581 PrevInChain = NewRed;
3582 NextInChain = NewRed;
3583 } else if (isPartialReduction()) {
3584 assert((Kind == RecurKind::Add || Kind == RecurKind::FAdd) &&
3585 "Unexpected partial reduction kind");
3586 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/false);
3587 NewRed = State.Builder.CreateIntrinsic(
3588 PrevInChain->getType(),
3589 Kind == RecurKind::Add ? Intrinsic::vector_partial_reduce_add
3590 : Intrinsic::vector_partial_reduce_fadd,
3591 {PrevInChain, NewVecOp}, State.Builder.getFastMathFlags(),
3592 "partial.reduce");
3593 PrevInChain = NewRed;
3594 NextInChain = NewRed;
3595 } else {
3596 assert(isInLoop() &&
3597 "The reduction must either be ordered, partial or in-loop");
3598 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/true);
3599 NewRed = createSimpleReduction(State.Builder, NewVecOp, Kind);
3601 NextInChain = createMinMaxOp(State.Builder, Kind, NewRed, PrevInChain);
3602 else
3603 NextInChain = State.Builder.CreateBinOp(
3605 PrevInChain, NewRed);
3606 }
3607 State.set(this, NextInChain, /*IsScalar*/ !isPartialReduction());
3608}
3609
3611
3612 assert(State.VF.isVector() &&
3613 "Shouldn't generate VPReductionEVLRecipe with scalar VF");
3614 auto &Builder = State.Builder;
3615 // Propagate the fast-math flags carried by the underlying instruction.
3616 IRBuilderBase::FastMathFlagGuard FMFGuard(Builder);
3617 Builder.setFastMathFlags(getFastMathFlagsOrNone());
3618
3620 Value *Prev =
3621 State.get(getChainOp(), /*NeedsSingleScalar=*/!isPartialReduction());
3622 Value *VecOp = State.get(getVecOp());
3623 Value *EVL = State.get(getEVL(), VPLane(0));
3624
3625 Value *Mask;
3626 if (VPValue *CondOp = getCondOp())
3627 Mask = State.get(CondOp);
3628 else
3629 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
3630
3631 Value *NewRed;
3632 if (isPartialReduction()) {
3633 // For partial reductions, we need to generate a predicated select
3634 // (vp.merge) since `@llvm.vector.partial.reduce()` doesn't have a vector
3635 // predicated version.
3636 VectorType *VecTy = cast<VectorType>(VecOp->getType());
3637 Value *Identity = getRecurrenceIdentity(Kind, VecTy->getElementType(),
3639 Identity =
3640 State.Builder.CreateVectorSplat(VecTy->getElementCount(), Identity);
3641
3642 // TODO: Calculate the predicate cost for the partial reduction.
3643 Value *NewVecOp = State.Builder.CreateIntrinsic(
3644 VecTy, Intrinsic::vp_merge, {Mask, VecOp, Identity, EVL});
3645 assert((Kind == RecurKind::Add || Kind == RecurKind::FAdd) &&
3646 "Unexpected partial reduction kind");
3647 NewRed = State.Builder.CreateIntrinsic(
3648 Prev->getType(),
3649 Kind == RecurKind::Add ? Intrinsic::vector_partial_reduce_add
3650 : Intrinsic::vector_partial_reduce_fadd,
3651 {Prev, NewVecOp}, State.Builder.getFastMathFlags(), "partial.reduce");
3652 } else if (isOrdered()) {
3653 NewRed = createOrderedReduction(Builder, Kind, VecOp, Prev, Mask, EVL);
3654 } else {
3655 NewRed = createSimpleReduction(Builder, VecOp, Kind, Mask, EVL);
3657 NewRed = createMinMaxOp(Builder, Kind, NewRed, Prev);
3658 else
3659 NewRed = Builder.CreateBinOp(
3661 Prev);
3662 }
3663 State.set(this, NewRed, !isPartialReduction());
3664}
3665
3667 VPCostContext &Ctx) const {
3668 RecurKind RdxKind = getRecurrenceKind();
3669 Type *ElementTy = this->getScalarType();
3670 auto *VectorTy = cast<VectorType>(toVectorTy(ElementTy, VF));
3671 unsigned Opcode = RecurrenceDescriptor::getOpcode(RdxKind);
3673 std::optional<FastMathFlags> OptionalFMF =
3674 ElementTy->isFloatingPointTy() ? std::make_optional(FMFs) : std::nullopt;
3675
3676 if (isPartialReduction()) {
3677 InstructionCost CondCost = 0;
3678 if (isConditional()) {
3680 auto *CondTy =
3682 CondCost = Ctx.TTI.getCmpSelInstrCost(Instruction::Select, VectorTy,
3683 CondTy, Pred, Ctx.CostKind);
3684 }
3685 return CondCost + Ctx.TTI.getPartialReductionCost(
3686 Opcode, ElementTy, nullptr, ElementTy, VF,
3687 TTI::PR_None, TTI::PR_None, {}, Ctx.CostKind,
3688 OptionalFMF);
3689 }
3690
3691 // TODO: Support any-of reductions.
3692 assert(
3694 ForceTargetInstructionCost.getNumOccurrences() > 0) &&
3695 "Any-of reduction not implemented in VPlan-based cost model currently.");
3696
3697 // Note that TTI should model the cost of moving result to the scalar register
3698 // and the BinOp cost in the getMinMaxReductionCost().
3701 return Ctx.TTI.getMinMaxReductionCost(Id, VectorTy, FMFs, Ctx.CostKind);
3702 }
3703
3704 // Note that TTI should model the cost of moving result to the scalar register
3705 // and the BinOp cost in the getArithmeticReductionCost().
3706 return Ctx.TTI.getArithmeticReductionCost(Opcode, VectorTy, OptionalFMF,
3707 Ctx.CostKind);
3708}
3709
3711 ExpressionTypes ExpressionType,
3712 ArrayRef<VPSingleDefRecipe *> ExpressionRecipes)
3713 : VPSingleDefRecipe(VPRecipeBase::VPExpressionSC, {},
3714 cast<VPReductionRecipe>(ExpressionRecipes.back())
3715 ->getChainOp()
3716 ->getScalarType()),
3717 ExpressionRecipes(ExpressionRecipes), ExpressionType(ExpressionType) {
3718 assert(!ExpressionRecipes.empty() && "Nothing to combine?");
3719 assert(
3720 none_of(ExpressionRecipes,
3721 [](VPSingleDefRecipe *R) { return R->mayHaveSideEffects(); }) &&
3722 "expression cannot contain recipes with side-effects");
3723
3724 // Maintain a copy of the expression recipes as a set of users.
3725 SmallPtrSet<VPUser *, 4> ExpressionRecipesAsSetOfUsers;
3726 for (auto *R : ExpressionRecipes)
3727 ExpressionRecipesAsSetOfUsers.insert(R);
3728
3729 // Recipes in the expression, except the last one, must only be used by
3730 // (other) recipes inside the expression. If there are other users, external
3731 // to the expression, use a clone of the recipe for external users.
3732 for (VPSingleDefRecipe *R : reverse(ExpressionRecipes)) {
3733 if (R != ExpressionRecipes.back() &&
3734 any_of(R->users(), [&ExpressionRecipesAsSetOfUsers](VPUser *U) {
3735 return !ExpressionRecipesAsSetOfUsers.contains(U);
3736 })) {
3737 // There are users outside of the expression. Clone the recipe and use the
3738 // clone those external users.
3739 VPSingleDefRecipe *CopyForExtUsers = R->clone();
3740 R->replaceUsesWithIf(CopyForExtUsers,
3741 [&ExpressionRecipesAsSetOfUsers](VPUser &U) {
3742 return !ExpressionRecipesAsSetOfUsers.contains(&U);
3743 });
3744 CopyForExtUsers->insertBefore(R);
3745 }
3746 if (R->getParent())
3747 R->removeFromParent();
3748 }
3749
3750 // Internalize all external operands to the expression recipes. To do so,
3751 // create new temporary VPValues for all operands defined by a recipe outside
3752 // the expression. The original operands are added as operands of the
3753 // VPExpressionRecipe itself.
3754 for (auto *R : ExpressionRecipes) {
3755 for (const auto &[Idx, Op] : enumerate(R->operands())) {
3756 auto *Def = Op->getDefiningRecipe();
3757 if (Def && ExpressionRecipesAsSetOfUsers.contains(Def))
3758 continue;
3759 addOperand(Op);
3760 LiveInPlaceholders.push_back(new VPSymbolicValue(Op->getScalarType()));
3761 }
3762 }
3763
3764 // Replace each external operand with the first one created for it in
3765 // LiveInPlaceholders.
3766 for (auto *R : ExpressionRecipes)
3767 for (auto const &[LiveIn, Tmp] : zip(operands(), LiveInPlaceholders))
3768 R->replaceUsesOfWith(LiveIn, Tmp);
3769}
3770
3772 for (auto *R : ExpressionRecipes)
3773 // Since the list could contain duplicates, make sure the recipe hasn't
3774 // already been inserted.
3775 if (!R->getParent())
3776 R->insertBefore(this);
3777
3778 for (const auto &[Idx, Op] : enumerate(operands()))
3779 LiveInPlaceholders[Idx]->replaceAllUsesWith(Op);
3780
3781 replaceAllUsesWith(ExpressionRecipes.back());
3782 SmallVector<VPSingleDefRecipe *> DecomposedRecipes(ExpressionRecipes);
3783 ExpressionRecipes.clear();
3784 return DecomposedRecipes;
3785}
3786
3788 VPCostContext &Ctx) const {
3789 Type *RedTy = this->getScalarType();
3790 auto *SrcVecTy =
3792 unsigned Opcode = RecurrenceDescriptor::getOpcode(
3793 cast<VPReductionRecipe>(ExpressionRecipes.back())->getRecurrenceKind());
3794 switch (ExpressionType) {
3795 case ExpressionTypes::NegatedExtendedReduction:
3796 assert((Opcode == Instruction::Add || Opcode == Instruction::FAdd) &&
3797 "Unexpected opcode");
3798 Opcode = Opcode == Instruction::Add ? Instruction::Sub : Instruction::FSub;
3799 [[fallthrough]];
3800 case ExpressionTypes::ExtendedReduction: {
3801 auto *RedR = cast<VPReductionRecipe>(ExpressionRecipes.back());
3802 auto *ExtR = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3803
3804 if (RedR->isPartialReduction())
3805 return Ctx.TTI.getPartialReductionCost(
3806 Opcode, getOperand(0)->getScalarType(), nullptr, RedTy, VF,
3808 TargetTransformInfo::PR_None, std::nullopt, Ctx.CostKind,
3809 RedTy->isFloatingPointTy()
3810 ? std::optional{RedR->getFastMathFlagsOrNone()}
3811 : std::nullopt);
3812 else if (!RedTy->isFloatingPointTy())
3813 // TTI::getExtendedReductionCost only supports integer types.
3814 return Ctx.TTI.getExtendedReductionCost(
3815 Opcode, ExtR->getOpcode() == Instruction::ZExt, RedTy, SrcVecTy,
3816 std::nullopt, Ctx.CostKind);
3817 else
3819 }
3820 case ExpressionTypes::MulAccReduction:
3821 return Ctx.TTI.getMulAccReductionCost(false, Opcode, RedTy, SrcVecTy,
3822 Ctx.CostKind);
3823
3824 case ExpressionTypes::ExtNegatedMulAccReduction:
3825 switch (Opcode) {
3826 case Instruction::Add:
3827 Opcode = Instruction::Sub;
3828 break;
3829 case Instruction::FAdd:
3830 Opcode = Instruction::FSub;
3831 break;
3832 default:
3833 llvm_unreachable("Unsupported opcode for ExtNegatedMulAccReduction");
3834 }
3835 [[fallthrough]];
3836 case ExpressionTypes::ExtMulAccReduction: {
3837 auto *RedR = cast<VPReductionRecipe>(ExpressionRecipes.back());
3838 if (RedR->isPartialReduction()) {
3839 auto *Ext0R = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3840 auto *Ext1R = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3841 auto *Mul = cast<VPWidenRecipe>(ExpressionRecipes[2]);
3842 return Ctx.TTI.getPartialReductionCost(
3843 Opcode, getOperand(0)->getScalarType(),
3844 getOperand(1)->getScalarType(), RedTy, VF,
3846 Ext0R->getOpcode()),
3848 Ext1R->getOpcode()),
3849 Mul->getOpcode(), Ctx.CostKind,
3850 RedTy->isFloatingPointTy()
3851 ? std::optional{RedR->getFastMathFlagsOrNone()}
3852 : std::nullopt);
3853 }
3854 assert(Opcode != Instruction::FSub && "Only integer types are supported");
3855 return Ctx.TTI.getMulAccReductionCost(
3856 cast<VPWidenCastRecipe>(ExpressionRecipes.front())->getOpcode() ==
3857 Instruction::ZExt,
3858 Opcode, RedTy, SrcVecTy, Ctx.CostKind);
3859 }
3860 }
3861 llvm_unreachable("Unknown VPExpressionRecipe::ExpressionTypes enum");
3862}
3863
3865 return any_of(ExpressionRecipes, [](VPSingleDefRecipe *R) {
3866 return R->mayReadFromMemory() || R->mayWriteToMemory();
3867 });
3868}
3869
3871 assert(
3872 none_of(ExpressionRecipes,
3873 [](VPSingleDefRecipe *R) { return R->mayHaveSideEffects(); }) &&
3874 "expression cannot contain recipes with side-effects");
3875 return false;
3876}
3877
3879 auto *RR = dyn_cast<VPReductionRecipe>(ExpressionRecipes.back());
3880 return RR && !RR->isPartialReduction();
3881}
3882
3883#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3884
3886 VPSlotTracker &SlotTracker) const {
3887 O << Indent << "EXPRESSION ";
3889 O << " = ";
3890 auto *Red = cast<VPReductionRecipe>(ExpressionRecipes.back());
3891 unsigned Opcode = RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind());
3892 VPValue *Mask = getLastOperand();
3893 VPValue *EVL =
3895 ? getOperand(getNumOperands() - (Red->isConditional() ? 2 : 1))
3896 : nullptr;
3897 VPValue *RdxStart = getOperand(
3898 getNumOperands() - (Red->isConditional() ? 2 : 1) - (EVL ? 1 : 0));
3899 auto PrintEVLAndMask = [&]() {
3900 if (EVL) {
3901 O << ", ";
3902 EVL->printAsOperand(O, SlotTracker);
3903 }
3904 if (Red->isConditional()) {
3905 O << ", ";
3906 Mask->printAsOperand(O, SlotTracker);
3907 }
3908 };
3909
3910 switch (ExpressionType) {
3911 case ExpressionTypes::NegatedExtendedReduction:
3912 case ExpressionTypes::ExtendedReduction: {
3913 bool Negated = ExpressionType == ExpressionTypes::NegatedExtendedReduction;
3915 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3916 O << Instruction::getOpcodeName(Opcode) << " (";
3917 if (Negated)
3918 O << (Opcode == Instruction::Add ? "sub (0, " : "fneg(");
3920 if (Negated)
3921 O << ")";
3922 Red->printFlags(O);
3923
3924 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3925 O << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3926 << *Ext0->getScalarType();
3927 PrintEVLAndMask();
3928 O << ")";
3929 break;
3930 }
3931 case ExpressionTypes::ExtNegatedMulAccReduction: {
3932 RdxStart->printAsOperand(O, SlotTracker);
3933 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3935 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()))
3936 << " (sub (0, mul";
3937 auto *Mul = cast<VPWidenRecipe>(ExpressionRecipes[2]);
3938 Mul->printFlags(O);
3939 O << "(";
3941 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3942 O << " " << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3943 << *Ext0->getScalarType() << "), (";
3945 auto *Ext1 = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3946 O << " " << Instruction::getOpcodeName(Ext1->getOpcode()) << " to "
3947 << *Ext1->getScalarType() << ")";
3948 PrintEVLAndMask();
3949 O << "))";
3950 break;
3951 }
3952 case ExpressionTypes::MulAccReduction:
3953 case ExpressionTypes::ExtMulAccReduction: {
3954 RdxStart->printAsOperand(O, SlotTracker);
3955 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3957 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()))
3958 << " (";
3959 O << "mul";
3960 bool IsExtended = ExpressionType == ExpressionTypes::ExtMulAccReduction;
3961 auto *Mul = cast<VPWidenRecipe>(IsExtended ? ExpressionRecipes[2]
3962 : ExpressionRecipes[0]);
3963 Mul->printFlags(O);
3964 if (IsExtended)
3965 O << "(";
3967 if (IsExtended) {
3968 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3969 O << " " << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3970 << *Ext0->getScalarType() << "), (";
3971 } else {
3972 O << ", ";
3973 }
3975 if (IsExtended) {
3976 auto *Ext1 = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3977 O << " " << Instruction::getOpcodeName(Ext1->getOpcode()) << " to "
3978 << *Ext1->getScalarType() << ")";
3979 }
3980 PrintEVLAndMask();
3981 O << ")";
3982 break;
3983 }
3984 }
3985}
3986
3988 VPSlotTracker &SlotTracker) const {
3989 if (isPartialReduction())
3990 O << Indent << "PARTIAL-REDUCE ";
3991 else
3992 O << Indent << "REDUCE ";
3994 O << " = ";
3996 O << " +";
3997 printFlags(O);
3998 O << " reduce.";
4000 O << " (";
4002 if (isConditional()) {
4003 O << ", ";
4005 }
4006 O << ")";
4007}
4008
4010 VPSlotTracker &SlotTracker) const {
4011 if (isPartialReduction())
4012 O << Indent << "PARTIAL-REDUCE ";
4013 else
4014 O << Indent << "REDUCE ";
4016 O << " = ";
4018 O << " +";
4019 printFlags(O);
4020 O << " vp.reduce."
4023 << " (";
4025 O << ", ";
4027 if (isConditional()) {
4028 O << ", ";
4030 }
4031 O << ")";
4032}
4033
4034#endif
4035
4037 assert(IsSingleScalar &&
4038 "VPReplicateRecipes must be unrolled before ::execute");
4039 auto *Instr = getUnderlyingInstr();
4040 Instruction *Cloned = Instr->clone();
4041 Type *ResultTy = getScalarType();
4042 if (!ResultTy->isVoidTy()) {
4043 Cloned->setName(Instr->getName() + ".cloned");
4044 // The operands of the replicate recipe may have been narrowed, resulting in
4045 // a narrower result type. Update the type of the cloned instruction to the
4046 // correct type.
4047 if (ResultTy != Cloned->getType())
4048 Cloned->mutateType(ResultTy);
4049 }
4050
4051 applyFlags(*Cloned);
4052 applyMetadata(*Cloned);
4053
4054 if (hasPredicate())
4055 cast<CmpInst>(Cloned)->setPredicate(getPredicate());
4056
4057 // Replace the operands of the cloned instructions with their scalar
4058 // equivalents in the new loop.
4059 for (const auto &[Idx, V] : enumerate(operands()))
4060 Cloned->setOperand(Idx, State.get(V, true));
4061
4062 // Place the cloned scalar in the new loop.
4063 State.Builder.Insert(Cloned);
4064
4065 State.set(this, Cloned, true);
4066
4067 // If we just cloned a new assumption, add it the assumption cache.
4068 if (auto *II = dyn_cast<AssumeInst>(Cloned))
4069 State.AC->registerAssumption(II);
4070}
4071
4072/// Returns a SCEV expression for \p Ptr if it is a pointer computation for
4073/// which the legacy cost model computes a SCEV expression when computing the
4074/// address cost. Computing SCEVs for VPValues is incomplete and returns
4075/// SCEVCouldNotCompute in cases the legacy cost model can compute SCEVs. In
4076/// those cases we fall back to the legacy cost model. Otherwise return nullptr.
4077static const SCEV *getAddressAccessSCEV(const VPValue *Ptr,
4079 const Loop *L) {
4080 const SCEV *Addr = vputils::getSCEVExprForVPValue(Ptr, PSE, L);
4081 if (isa<SCEVCouldNotCompute>(Addr))
4082 return Addr;
4083
4084 return vputils::isAddressSCEVForCost(Addr, *PSE.getSE(), L) ? Addr : nullptr;
4085}
4086
4088 VPCostContext &Ctx) const {
4090 // VPReplicateRecipe may be cloned as part of an existing VPlan-to-VPlan
4091 // transform, avoid computing their cost multiple times for now.
4092 Ctx.SkipCostComputation.insert(UI);
4093
4094 if (VF.isScalable() && !isSingleScalar())
4096
4097 switch (UI->getOpcode()) {
4098 case Instruction::Alloca:
4099 if (VF.isScalable())
4101 return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul,
4102 this->getScalarType(), Ctx.CostKind);
4103 case Instruction::GetElementPtr:
4104 // We mark this instruction as zero-cost because the cost of GEPs in
4105 // vectorized code depends on whether the corresponding memory instruction
4106 // is scalarized or not. Therefore, we handle GEPs with the memory
4107 // instruction cost.
4108 return 0;
4109 case Instruction::Call: {
4110 auto *CalledFn = cast<Function>(getLastOperand()->getLiveInIRValue());
4111 Type *ResultTy = this->getScalarType();
4112 return computeCallCost(CalledFn, ResultTy, drop_end(operands()),
4113 isSingleScalar(), VF, Ctx);
4114 }
4115 case Instruction::Add:
4116 case Instruction::Sub:
4117 case Instruction::FAdd:
4118 case Instruction::FSub:
4119 case Instruction::Mul:
4120 case Instruction::FMul:
4121 case Instruction::FDiv:
4122 case Instruction::FRem:
4123 case Instruction::Shl:
4124 case Instruction::LShr:
4125 case Instruction::AShr:
4126 case Instruction::And:
4127 case Instruction::Or:
4128 case Instruction::Xor:
4129 case Instruction::ICmp:
4130 case Instruction::FCmp:
4132 Ctx) *
4133 (isSingleScalar() ? 1 : VF.getFixedValue());
4134 case Instruction::SDiv:
4135 case Instruction::UDiv:
4136 case Instruction::SRem:
4137 case Instruction::URem: {
4138 InstructionCost ScalarCost =
4140 if (isSingleScalar())
4141 return ScalarCost;
4142
4143 // If any of the operands is from a different replicate region and has its
4144 // cost skipped, it may have been forced to scalar. Fall back to legacy cost
4145 // model to avoid cost mis-match.
4146 if (any_of(operands(), [&Ctx, VF](VPValue *Op) {
4147 auto *PredR = dyn_cast<VPPredInstPHIRecipe>(Op);
4148 if (!PredR)
4149 return false;
4150 return Ctx.skipCostComputation(
4152 PredR->getOperand(0)->getUnderlyingValue()),
4153 VF.isVector());
4154 }))
4155 break;
4156
4157 ScalarCost = ScalarCost * VF.getFixedValue() +
4158 Ctx.getScalarizationOverhead(this->getScalarType(),
4159 to_vector(operands()), VF);
4160 // If the recipe is not predicated (i.e. not in a replicate region), return
4161 // the scalar cost. Otherwise handle predicated cost.
4162 const VPRegionBlock *ParentRegion = getRegion();
4163 if (!ParentRegion || !ParentRegion->isReplicator())
4164 return ScalarCost;
4165
4166 // Account for the phi nodes that we will create.
4167 ScalarCost += VF.getFixedValue() *
4168 Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
4169 // Scale the cost by the probability of executing the predicated blocks.
4170 // This assumes the predicated block for each vector lane is equally
4171 // likely.
4172 ScalarCost /= Ctx.getCostDivisor(
4173 ParentRegion->getEntryBranchOnMask()->getExecutionFrequency());
4174 return ScalarCost;
4175 }
4176 case Instruction::Load:
4177 case Instruction::Store: {
4178 bool IsLoad = UI->getOpcode() == Instruction::Load;
4179 const VPValue *PtrOp = getOperand(!IsLoad);
4180 const SCEV *PtrSCEV = getAddressAccessSCEV(PtrOp, Ctx.PSE, Ctx.L);
4182 break;
4183
4184 Type *ValTy = (IsLoad ? this : getOperand(0))->getScalarType();
4185 Type *ScalarPtrTy = PtrOp->getScalarType();
4186 const Align Alignment = getLoadStoreAlignment(UI);
4187 unsigned AS = cast<PointerType>(ScalarPtrTy)->getAddressSpace();
4189 bool PreferVectorizedAddressing = Ctx.TTI.prefersVectorizedAddressing();
4190 bool UsedByLoadStoreAddress =
4191 !PreferVectorizedAddressing && vputils::isUsedByLoadStoreAddress(this);
4192 InstructionCost ScalarMemOpCost = Ctx.TTI.getMemoryOpCost(
4193 UI->getOpcode(), ValTy, Alignment, AS, Ctx.CostKind, OpInfo,
4194 UsedByLoadStoreAddress ? UI : nullptr);
4195
4196 Type *PtrTy = isSingleScalar() ? ScalarPtrTy : toVectorTy(ScalarPtrTy, VF);
4197 InstructionCost ScalarCost =
4198 ScalarMemOpCost +
4199 Ctx.TTI.getAddressComputationCost(
4200 PtrTy, UsedByLoadStoreAddress ? nullptr : Ctx.PSE.getSE(), PtrSCEV,
4201 Ctx.CostKind);
4202 if (isSingleScalar())
4203 return ScalarCost;
4204
4205 SmallVector<const VPValue *> OpsToScalarize;
4206 Type *ResultTy = Type::getVoidTy(PtrTy->getContext());
4207 // Set ResultTy and OpsToScalarize, if scalarization is needed. Currently we
4208 // don't assign scalarization overhead in general, if the target prefers
4209 // vectorized addressing or the loaded value is used as part of an address
4210 // of another load or store.
4211 if (!UsedByLoadStoreAddress) {
4212 bool EfficientVectorLoadStore =
4213 Ctx.TTI.supportsEfficientVectorElementLoadStore();
4214 if (!(IsLoad && !PreferVectorizedAddressing) &&
4215 !(!IsLoad && EfficientVectorLoadStore))
4216 append_range(OpsToScalarize, operands());
4217
4218 if (!EfficientVectorLoadStore)
4219 ResultTy = this->getScalarType();
4220 }
4221
4223 IsLoad ? TTI::VectorInstrContext::Load : TTI::VectorInstrContext::Store;
4225 (ScalarCost * VF.getFixedValue()) +
4226 Ctx.getScalarizationOverhead(ResultTy, OpsToScalarize, VF, VIC, true);
4227
4228 const VPRegionBlock *ParentRegion = getRegion();
4229 if (ParentRegion && ParentRegion->isReplicator()) {
4230 if (!PtrSCEV)
4231 break;
4232 Cost /= Ctx.getCostDivisor(
4233 ParentRegion->getEntryBranchOnMask()->getExecutionFrequency());
4234 Cost += Ctx.TTI.getCFInstrCost(Instruction::CondBr, Ctx.CostKind);
4235
4236 auto *VecI1Ty = VectorType::get(
4237 IntegerType::getInt1Ty(Ctx.L->getHeader()->getContext()), VF);
4238 Cost += Ctx.TTI.getScalarizationOverhead(
4239 VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
4240 /*Insert=*/false, /*Extract=*/true, Ctx.CostKind);
4241
4242 if (Ctx.useEmulatedMaskMemRefHack(this, VF)) {
4243 // Artificially setting to a high enough value to practically disable
4244 // vectorization with such operations.
4245 return 3000000;
4246 }
4247 }
4248 return Cost;
4249 }
4250 case Instruction::SExt:
4251 case Instruction::ZExt:
4252 case Instruction::FPToUI:
4253 case Instruction::FPToSI:
4254 case Instruction::FPExt:
4255 case Instruction::PtrToInt:
4256 case Instruction::PtrToAddr:
4257 case Instruction::IntToPtr:
4258 case Instruction::SIToFP:
4259 case Instruction::UIToFP:
4260 case Instruction::Trunc:
4261 case Instruction::FPTrunc:
4262 case Instruction::Select:
4263 case Instruction::AddrSpaceCast: {
4265 Ctx) *
4266 (isSingleScalar() ? 1 : VF.getFixedValue());
4267 }
4268 case Instruction::ExtractValue:
4269 case Instruction::InsertValue:
4270 return Ctx.TTI.getInsertExtractValueCost(getOpcode(), Ctx.CostKind);
4271 }
4272
4273 return Ctx.getLegacyCost(UI, VF);
4274}
4275
4277 Function *CalledFn, Type *ResultTy, ArrayRef<const VPValue *> ArgOps,
4278 bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx) {
4280 ArgOps, [&](const VPValue *Op) { return Op->getScalarType(); });
4281
4282 Intrinsic::ID IntrinID = CalledFn->getIntrinsicID();
4283 auto GetIntrinsicCost = [&] {
4284 if (!IntrinID)
4286 return Ctx.TTI.getIntrinsicInstrCost(
4287 IntrinsicCostAttributes(IntrinID, ResultTy, Tys), Ctx.CostKind);
4288 };
4289
4290 if (IntrinID && VPCostContext::isFreeScalarIntrinsic(IntrinID)) {
4291 assert(GetIntrinsicCost() == 0 && "scalarizing intrinsic should be free");
4292 return 0;
4293 }
4294
4295 InstructionCost ScalarCallCost =
4296 Ctx.TTI.getCallInstrCost(CalledFn, ResultTy, Tys, Ctx.CostKind);
4297 if (IsSingleScalar) {
4298 ScalarCallCost = std::min(ScalarCallCost, GetIntrinsicCost());
4299 return ScalarCallCost;
4300 }
4301
4302 // Scalarization overhead is undefined for scalable VFs.
4303 if (VF.isScalable())
4305
4306 return ScalarCallCost * VF.getFixedValue() +
4307 Ctx.getScalarizationOverhead(ResultTy, ArgOps, VF);
4308}
4309
4310#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4312 VPSlotTracker &SlotTracker) const {
4313 O << Indent << (IsSingleScalar ? "CLONE " : "REPLICATE ");
4314
4315 if (!getScalarType()->isVoidTy()) {
4317 O << " = ";
4318 }
4319 if (auto *CB = dyn_cast<CallBase>(getUnderlyingInstr())) {
4320 O << "call";
4321 printFlags(O);
4322 O << "@" << CB->getCalledFunction()->getName() << "(";
4324 Op->printAsOperand(O, SlotTracker);
4325 });
4326 O << ")";
4327 } else {
4329 printFlags(O);
4331 }
4332
4333 // Find if the recipe is used by a widened recipe via an intervening
4334 // VPPredInstPHIRecipe. In this case, also pack the scalar values in a vector.
4335 if (any_of(users(), [](const VPUser *U) {
4336 if (auto *PredR = dyn_cast<VPPredInstPHIRecipe>(U))
4337 return !vputils::onlyScalarValuesUsed(PredR);
4338 return false;
4339 }))
4340 O << " (S->V)";
4341}
4342#endif
4343
4345 llvm_unreachable("recipe must be removed when dissolving replicate region");
4346}
4347
4349 VPCostContext &Ctx) const {
4350 // The legacy cost model doesn't assign costs to branches for individual
4351 // replicate regions. Match the current behavior in the VPlan cost model for
4352 // now.
4353 return 0;
4354}
4355
4357 llvm_unreachable("recipe must be removed when dissolving replicate region");
4358}
4359
4360#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4362 VPSlotTracker &SlotTracker) const {
4363 O << Indent << "PHI-PREDICATED-INSTRUCTION ";
4365 O << " = ";
4367}
4368#endif
4369
4371const VPRecipeBase *VPWidenLoadRecipe::getAsRecipe() const { return this; }
4372
4375
4377const VPRecipeBase *VPWidenStoreRecipe::getAsRecipe() const { return this; }
4378
4381
4383 VPCostContext &Ctx) const {
4384 const VPRecipeBase *R = getAsRecipe();
4386 Type *ScalarTy = IsLoad ? cast<VPSingleDefRecipe>(R)->getScalarType()
4387 : R->getOperand(1)->getScalarType();
4388 Type *Ty = toVectorTy(ScalarTy, VF);
4389 unsigned AS =
4390 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4391 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
4392
4393 if (!Consecutive) {
4394 // TODO: Using the original IR may not be accurate.
4395 // Currently, ARM will use the underlying IR to calculate gather/scatter
4396 // instruction cost.
4397 Type *PtrTy = getAddr()->getScalarType();
4398 const Value *Ptr = getAddr()->getUnderlyingValue();
4399
4400 // If the address value is uniform across all lanes, then the address can be
4401 // calculated with scalar type and broadcast.
4403 PtrTy = toVectorTy(PtrTy, VF);
4404
4405 unsigned IID = isa<VPWidenLoadRecipe>(R) ? Intrinsic::masked_gather
4406 : isa<VPWidenStoreRecipe>(R) ? Intrinsic::masked_scatter
4407 : isa<VPWidenLoadEVLRecipe>(R) ? Intrinsic::vp_gather
4408 : Intrinsic::vp_scatter;
4409 return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
4410 Ctx.CostKind) +
4411 Ctx.TTI.getMemIntrinsicInstrCost(
4413 &Ingredient),
4414 Ctx.CostKind);
4415 }
4416
4418 if (IsMasked) {
4419 unsigned IID = isa<VPWidenLoadRecipe>(R) ? Intrinsic::masked_load
4420 : Intrinsic::masked_store;
4421 Cost += Ctx.TTI.getMemIntrinsicInstrCost(
4422 MemIntrinsicCostAttributes(IID, Ty, Alignment, AS), Ctx.CostKind);
4423 } else {
4424 TTI::OperandValueInfo OpInfo = Ctx.getOperandInfo(
4426 : R->getOperand(1));
4427 Cost += Ctx.TTI.getMemoryOpCost(Opcode, Ty, Alignment, AS, Ctx.CostKind,
4428 OpInfo, &Ingredient);
4429 }
4430 return Cost;
4431}
4432
4434 Type *ScalarDataTy = getScalarType();
4435 auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
4436 bool CreateGather = !isConsecutive();
4437
4438 auto &Builder = State.Builder;
4439 Value *Mask = nullptr;
4440 if (auto *VPMask = getMask())
4441 Mask = State.get(VPMask);
4442
4443 Value *Addr = State.get(getAddr(), /*NeedsSingleScalar=*/!CreateGather);
4444 Value *NewLI;
4445 if (CreateGather) {
4446 NewLI = Builder.CreateMaskedGather(DataTy, Addr, Alignment, Mask, nullptr,
4447 "wide.masked.gather");
4448 } else if (Mask) {
4449 NewLI =
4450 Builder.CreateMaskedLoad(DataTy, Addr, Alignment, Mask,
4451 PoisonValue::get(DataTy), "wide.masked.load");
4452 } else {
4453 NewLI = Builder.CreateAlignedLoad(DataTy, Addr, Alignment, "wide.load");
4454 }
4456 State.set(this, NewLI);
4457}
4458
4459#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4461 VPSlotTracker &SlotTracker) const {
4462 O << Indent << "WIDEN ";
4464 O << " = load ";
4466}
4467#endif
4468
4470 Type *ScalarDataTy = getScalarType();
4471 auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
4472 bool CreateGather = !isConsecutive();
4473
4474 auto &Builder = State.Builder;
4475 CallInst *NewLI;
4476 Value *EVL = State.get(getEVL(), VPLane(0));
4477 Value *Addr = State.get(getAddr(), !CreateGather);
4478 Value *Mask = nullptr;
4479 if (VPValue *VPMask = getMask())
4480 Mask = State.get(VPMask);
4481 else
4482 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
4483
4484 if (CreateGather) {
4485 NewLI = Builder.CreateIntrinsicWithoutFolding(DataTy, Intrinsic::vp_gather,
4486 {Addr, Mask, EVL}, nullptr,
4487 "wide.masked.gather");
4488 } else {
4489 NewLI = Builder.CreateIntrinsicWithoutFolding(
4490 DataTy, Intrinsic::vp_load, {Addr, Mask, EVL}, nullptr, "vp.op.load");
4491 }
4492 NewLI->addParamAttr(
4494 applyMetadata(*NewLI);
4495 State.set(this, NewLI);
4496}
4497
4499 VPCostContext &Ctx) const {
4500 if (!Consecutive || IsMasked)
4501 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
4502
4503 // We need to use the getMemIntrinsicInstrCost() instead of getMemoryOpCost()
4504 // here because the EVL recipes using EVL to replace the tail mask. But in the
4505 // legacy model, it will always calculate the cost of mask.
4506 // TODO: Using getMemoryOpCost() instead of getMemIntrinsicInstrCost when we
4507 // don't need to compare to the legacy cost model.
4508 Type *Ty = toVectorTy(getScalarType(), VF);
4509 unsigned AS =
4510 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4511 return Ctx.TTI.getMemIntrinsicInstrCost(
4512 MemIntrinsicCostAttributes(Intrinsic::vp_load, Ty, Alignment, AS),
4513 Ctx.CostKind);
4514}
4515
4516#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4518 VPSlotTracker &SlotTracker) const {
4519 O << Indent << "WIDEN ";
4521 O << " = vp.load ";
4523}
4524#endif
4525
4527 VPValue *StoredVPValue = getStoredValue();
4528 bool CreateScatter = !isConsecutive();
4529
4530 auto &Builder = State.Builder;
4531
4532 Value *Mask = nullptr;
4533 if (auto *VPMask = getMask())
4534 Mask = State.get(VPMask);
4535
4536 Value *StoredVal = State.get(StoredVPValue);
4537 Value *Addr = State.get(getAddr(), /*NeedsSingleScalar=*/!CreateScatter);
4538 Instruction *NewSI = nullptr;
4539 if (CreateScatter)
4540 NewSI = Builder.CreateMaskedScatter(StoredVal, Addr, Alignment, Mask);
4541 else if (Mask)
4542 NewSI = Builder.CreateMaskedStore(StoredVal, Addr, Alignment, Mask);
4543 else
4544 NewSI = Builder.CreateAlignedStore(StoredVal, Addr, Alignment);
4545 applyMetadata(*NewSI);
4546}
4547
4548#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4550 VPSlotTracker &SlotTracker) const {
4551 O << Indent << "WIDEN store ";
4553}
4554#endif
4555
4557 VPValue *StoredValue = getStoredValue();
4558 bool CreateScatter = !isConsecutive();
4559
4560 auto &Builder = State.Builder;
4561
4562 CallInst *NewSI = nullptr;
4563 Value *StoredVal = State.get(StoredValue);
4564 Value *EVL = State.get(getEVL(), VPLane(0));
4565 Value *Mask = nullptr;
4566 if (VPValue *VPMask = getMask())
4567 Mask = State.get(VPMask);
4568 else
4569 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
4570
4571 Value *Addr = State.get(getAddr(), !CreateScatter);
4572 if (CreateScatter) {
4573 NewSI = Builder.CreateIntrinsicWithoutFolding(
4574 Type::getVoidTy(EVL->getContext()), Intrinsic::vp_scatter,
4575 {StoredVal, Addr, Mask, EVL});
4576 } else {
4577 NewSI = Builder.CreateIntrinsicWithoutFolding(
4578 Type::getVoidTy(EVL->getContext()), Intrinsic::vp_store,
4579 {StoredVal, Addr, Mask, EVL});
4580 }
4581 NewSI->addParamAttr(
4583 applyMetadata(*NewSI);
4584}
4585
4587 VPCostContext &Ctx) const {
4588 if (!Consecutive || IsMasked)
4589 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
4590
4591 // We need to use the getMemIntrinsicInstrCost() instead of getMemoryOpCost()
4592 // here because the EVL recipes using EVL to replace the tail mask. But in the
4593 // legacy model, it will always calculate the cost of mask.
4594 // TODO: Using getMemoryOpCost() instead of getMemIntrinsicInstrCost when we
4595 // don't need to compare to the legacy cost model.
4596 Type *Ty = toVectorTy(getStoredValue()->getScalarType(), VF);
4597 unsigned AS =
4598 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4599 return Ctx.TTI.getMemIntrinsicInstrCost(
4600 MemIntrinsicCostAttributes(Intrinsic::vp_store, Ty, Alignment, AS),
4601 Ctx.CostKind);
4602}
4603
4604#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4606 VPSlotTracker &SlotTracker) const {
4607 O << Indent << "WIDEN vp.store ";
4609}
4610#endif
4611
4613 VectorType *DstVTy, const DataLayout &DL) {
4614 // Verify that V is a vector type with same number of elements as DstVTy.
4615 auto VF = DstVTy->getElementCount();
4616 auto *SrcVecTy = cast<VectorType>(V->getType());
4617 assert(VF == SrcVecTy->getElementCount() && "Vector dimensions do not match");
4618 Type *SrcElemTy = SrcVecTy->getElementType();
4619 Type *DstElemTy = DstVTy->getElementType();
4620 assert((DL.getTypeSizeInBits(SrcElemTy) == DL.getTypeSizeInBits(DstElemTy)) &&
4621 "Vector elements must have same size");
4622
4623 // Do a direct cast if element types are castable.
4624 if (CastInst::isBitOrNoopPointerCastable(SrcElemTy, DstElemTy, DL)) {
4625 return Builder.CreateBitOrPointerCast(V, DstVTy);
4626 }
4627 // V cannot be directly casted to desired vector type.
4628 // May happen when V is a floating point vector but DstVTy is a vector of
4629 // pointers or vice-versa. Handle this using a two-step bitcast using an
4630 // intermediate Integer type for the bitcast i.e. Ptr <-> Int <-> Float.
4631 assert((DstElemTy->isPointerTy() != SrcElemTy->isPointerTy()) &&
4632 "Only one type should be a pointer type");
4633 assert((DstElemTy->isFloatingPointTy() != SrcElemTy->isFloatingPointTy()) &&
4634 "Only one type should be a floating point type");
4635 Type *IntTy =
4636 IntegerType::getIntNTy(V->getContext(), DL.getTypeSizeInBits(SrcElemTy));
4637 auto *VecIntTy = VectorType::get(IntTy, VF);
4638 Value *CastVal = Builder.CreateBitOrPointerCast(V, VecIntTy);
4639 return Builder.CreateBitOrPointerCast(CastVal, DstVTy);
4640}
4641
4642/// Return a vector containing interleaved elements from multiple
4643/// smaller input vectors.
4645 const Twine &Name) {
4646 unsigned Factor = Vals.size();
4647 assert(Factor > 1 && "Tried to interleave invalid number of vectors");
4648
4649 VectorType *VecTy = cast<VectorType>(Vals[0]->getType());
4650#ifndef NDEBUG
4651 for (Value *Val : Vals)
4652 assert(Val->getType() == VecTy && "Tried to interleave mismatched types");
4653#endif
4654
4655 // Scalable vectors cannot use arbitrary shufflevectors (only splats), so
4656 // must use intrinsics to interleave.
4657 if (VecTy->isScalableTy()) {
4658 assert(Factor <= 8 && "Unsupported interleave factor for scalable vectors");
4659 return Builder.CreateVectorInterleave(Vals, Name);
4660 }
4661
4662 // Fixed length. Start by concatenating all vectors into a wide vector.
4663 Value *WideVec = concatenateVectors(Builder, Vals);
4664
4665 // Interleave the elements into the wide vector.
4666 const unsigned NumElts = VecTy->getElementCount().getFixedValue();
4667 return Builder.CreateShuffleVector(
4668 WideVec, createInterleaveMask(NumElts, Factor), Name);
4669}
4670
4671// Try to vectorize the interleave group that \p Instr belongs to.
4672//
4673// E.g. Translate following interleaved load group (factor = 3):
4674// for (i = 0; i < N; i+=3) {
4675// R = Pic[i]; // Member of index 0
4676// G = Pic[i+1]; // Member of index 1
4677// B = Pic[i+2]; // Member of index 2
4678// ... // do something to R, G, B
4679// }
4680// To:
4681// %wide.vec = load <12 x i32> ; Read 4 tuples of R,G,B
4682// %R.vec = shuffle %wide.vec, poison, <0, 3, 6, 9> ; R elements
4683// %G.vec = shuffle %wide.vec, poison, <1, 4, 7, 10> ; G elements
4684// %B.vec = shuffle %wide.vec, poison, <2, 5, 8, 11> ; B elements
4685//
4686// Or translate following interleaved store group (factor = 3):
4687// for (i = 0; i < N; i+=3) {
4688// ... do something to R, G, B
4689// Pic[i] = R; // Member of index 0
4690// Pic[i+1] = G; // Member of index 1
4691// Pic[i+2] = B; // Member of index 2
4692// }
4693// To:
4694// %R_G.vec = shuffle %R.vec, %G.vec, <0, 1, 2, ..., 7>
4695// %B_U.vec = shuffle %B.vec, poison, <0, 1, 2, 3, u, u, u, u>
4696// %interleaved.vec = shuffle %R_G.vec, %B_U.vec,
4697// <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11> ; Interleave R,G,B elements
4698// store <12 x i32> %interleaved.vec ; Write 4 tuples of R,G,B
4700 assert((!needsMaskForGaps() || !State.VF.isScalable()) &&
4701 "Masking gaps for scalable vectors is not yet supported.");
4703 Instruction *Instr = Group->getInsertPos();
4704
4705 // Prepare for the vector type of the interleaved load/store.
4706 Type *ScalarTy = getLoadStoreType(Instr);
4707 unsigned InterleaveFactor = Group->getFactor();
4708 auto *VecTy = VectorType::get(ScalarTy, State.VF * InterleaveFactor);
4709
4710 VPValue *BlockInMask = getMask();
4711 VPValue *Addr = getAddr();
4712 Value *ResAddr = State.get(Addr, VPLane(0));
4713
4714 auto CreateGroupMask = [&BlockInMask, &State,
4715 &InterleaveFactor](Value *MaskForGaps) -> Value * {
4716 if (State.VF.isScalable()) {
4717 assert(!MaskForGaps && "Interleaved groups with gaps are not supported.");
4718 assert(InterleaveFactor <= 8 &&
4719 "Unsupported deinterleave factor for scalable vectors");
4720 auto *ResBlockInMask = State.get(BlockInMask);
4721 SmallVector<Value *> Ops(InterleaveFactor, ResBlockInMask);
4722 return interleaveVectors(State.Builder, Ops, "interleaved.mask");
4723 }
4724
4725 if (!BlockInMask)
4726 return MaskForGaps;
4727
4728 Value *ResBlockInMask = State.get(BlockInMask);
4729 Value *ShuffledMask = State.Builder.CreateShuffleVector(
4730 ResBlockInMask,
4731 createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
4732 "interleaved.mask");
4733 return MaskForGaps ? State.Builder.CreateBinOp(Instruction::And,
4734 ShuffledMask, MaskForGaps)
4735 : ShuffledMask;
4736 };
4737
4738 const DataLayout &DL = Instr->getDataLayout();
4739 // Vectorize the interleaved load group.
4740 if (isa<LoadInst>(Instr)) {
4741 Value *MaskForGaps = nullptr;
4742 if (needsMaskForGaps()) {
4743 MaskForGaps =
4744 createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
4745 assert(MaskForGaps && "Mask for Gaps is required but it is null");
4746 }
4747
4748 Instruction *NewLoad;
4749 if (BlockInMask || MaskForGaps) {
4750 Value *GroupMask = CreateGroupMask(MaskForGaps);
4751 Value *PoisonVec = PoisonValue::get(VecTy);
4752 NewLoad = State.Builder.CreateMaskedLoad(VecTy, ResAddr,
4753 Group->getAlign(), GroupMask,
4754 PoisonVec, "wide.masked.vec");
4755 } else
4756 NewLoad = State.Builder.CreateAlignedLoad(VecTy, ResAddr,
4757 Group->getAlign(), "wide.vec");
4758 applyMetadata(*NewLoad);
4759 // TODO: Also manage existing metadata using VPIRMetadata.
4760 Group->addMetadata(NewLoad);
4761
4763 if (VecTy->isScalableTy()) {
4764 // Scalable vectors cannot use arbitrary shufflevectors (only splats),
4765 // so must use intrinsics to deinterleave.
4766 assert(InterleaveFactor <= 8 &&
4767 "Unsupported deinterleave factor for scalable vectors");
4768 NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4769 Intrinsic::getDeinterleaveIntrinsicID(InterleaveFactor),
4770 NewLoad->getType(), NewLoad,
4771 /*FMFSource=*/nullptr, "strided.vec");
4772 }
4773
4774 auto CreateStridedVector = [&InterleaveFactor, &State,
4775 &NewLoad](unsigned Index) -> Value * {
4776 assert(Index < InterleaveFactor && "Illegal group index");
4777 if (State.VF.isScalable())
4778 return State.Builder.CreateExtractValue(NewLoad, Index);
4779
4780 // For fixed length VF, use shuffle to extract the sub-vectors from the
4781 // wide load.
4782 auto StrideMask =
4783 createStrideMask(Index, InterleaveFactor, State.VF.getFixedValue());
4784 return State.Builder.CreateShuffleVector(NewLoad, StrideMask,
4785 "strided.vec");
4786 };
4787
4788 for (unsigned I = 0, J = 0; I < InterleaveFactor; ++I) {
4789 Instruction *Member = Group->getMember(I);
4790
4791 // Skip the gaps in the group.
4792 if (!Member)
4793 continue;
4794
4795 Value *StridedVec = CreateStridedVector(I);
4796
4797 // If this member has different type, cast the result type.
4798 if (Member->getType() != ScalarTy) {
4799 VectorType *OtherVTy = VectorType::get(Member->getType(), State.VF);
4800 StridedVec =
4801 createBitOrPointerCast(State.Builder, StridedVec, OtherVTy, DL);
4802 }
4803
4804 if (Group->isReverse())
4805 StridedVec = State.Builder.CreateVectorReverse(StridedVec, "reverse");
4806
4807 State.set(VPDefs[J], StridedVec);
4808 ++J;
4809 }
4810 return;
4811 }
4812
4813 // The sub vector type for current instruction.
4814 auto *SubVT = VectorType::get(ScalarTy, State.VF);
4815
4816 // Vectorize the interleaved store group.
4817 Value *MaskForGaps =
4818 createBitMaskForGaps(State.Builder, State.VF.getKnownMinValue(), *Group);
4819 assert(((MaskForGaps != nullptr) == needsMaskForGaps()) &&
4820 "Mismatch between NeedsMaskForGaps and MaskForGaps");
4821 ArrayRef<VPValue *> StoredValues = getStoredValues();
4822 // Collect the stored vector from each member.
4823 SmallVector<Value *, 4> StoredVecs;
4824 unsigned StoredIdx = 0;
4825 for (unsigned i = 0; i < InterleaveFactor; i++) {
4826 assert((Group->getMember(i) || MaskForGaps) &&
4827 "Fail to get a member from an interleaved store group");
4828 Instruction *Member = Group->getMember(i);
4829
4830 // Skip the gaps in the group.
4831 if (!Member) {
4832 Value *Undef = PoisonValue::get(SubVT);
4833 StoredVecs.push_back(Undef);
4834 continue;
4835 }
4836
4837 Value *StoredVec = State.get(StoredValues[StoredIdx]);
4838 ++StoredIdx;
4839
4840 if (Group->isReverse())
4841 StoredVec = State.Builder.CreateVectorReverse(StoredVec, "reverse");
4842
4843 // If this member has different type, cast it to a unified type.
4844
4845 if (StoredVec->getType() != SubVT)
4846 StoredVec = createBitOrPointerCast(State.Builder, StoredVec, SubVT, DL);
4847
4848 StoredVecs.push_back(StoredVec);
4849 }
4850
4851 // Interleave all the smaller vectors into one wider vector.
4852 Value *IVec = interleaveVectors(State.Builder, StoredVecs, "interleaved.vec");
4853 Instruction *NewStoreInstr;
4854 if (BlockInMask || MaskForGaps) {
4855 Value *GroupMask = CreateGroupMask(MaskForGaps);
4856 NewStoreInstr = State.Builder.CreateMaskedStore(
4857 IVec, ResAddr, Group->getAlign(), GroupMask);
4858 } else
4859 NewStoreInstr =
4860 State.Builder.CreateAlignedStore(IVec, ResAddr, Group->getAlign());
4861
4862 applyMetadata(*NewStoreInstr);
4863 // TODO: Also manage existing metadata using VPIRMetadata.
4864 Group->addMetadata(NewStoreInstr);
4865}
4866
4867#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4869 VPSlotTracker &SlotTracker) const {
4871 O << Indent << "INTERLEAVE-GROUP with factor " << IG->getFactor() << ", ";
4873 VPValue *Mask = getMask();
4874 if (Mask) {
4875 O << ", ";
4876 Mask->printAsOperand(O, SlotTracker);
4877 }
4878
4879 unsigned OpIdx = 0;
4880 for (unsigned i = 0; i < IG->getFactor(); ++i) {
4881 if (!IG->getMember(i))
4882 continue;
4883 if (getNumStoreOperands() > 0) {
4884 O << "\n" << Indent << " store ";
4885 getOperand(1 + OpIdx)->printAsOperand(O, SlotTracker);
4886 O << " to index " << i;
4887 } else {
4888 O << "\n" << Indent << " ";
4890 O << " = load from index " << i;
4891 }
4892 ++OpIdx;
4893 }
4894}
4895#endif
4896
4898 assert(State.VF.isScalable() &&
4899 "Only support scalable VF for EVL tail-folding.");
4901 "Masking gaps for scalable vectors is not yet supported.");
4903 Instruction *Instr = Group->getInsertPos();
4904
4905 // Prepare for the vector type of the interleaved load/store.
4906 Type *ScalarTy = getLoadStoreType(Instr);
4907 unsigned InterleaveFactor = Group->getFactor();
4908 assert(InterleaveFactor <= 8 &&
4909 "Unsupported deinterleave/interleave factor for scalable vectors");
4910 ElementCount WideVF = State.VF * InterleaveFactor;
4911 auto *VecTy = VectorType::get(ScalarTy, WideVF);
4912
4913 VPValue *Addr = getAddr();
4914 Value *ResAddr = State.get(Addr, VPLane(0));
4915 Value *EVL = State.get(getEVL(), VPLane(0));
4916 Value *InterleaveEVL = State.Builder.CreateMul(
4917 EVL, ConstantInt::get(EVL->getType(), InterleaveFactor), "interleave.evl",
4918 /* NUW= */ true, /* NSW= */ true);
4919 LLVMContext &Ctx = State.Builder.getContext();
4920
4921 Value *GroupMask = nullptr;
4922 if (VPValue *BlockInMask = getMask()) {
4923 SmallVector<Value *> Ops(InterleaveFactor, State.get(BlockInMask));
4924 GroupMask = interleaveVectors(State.Builder, Ops, "interleaved.mask");
4925 } else {
4926 GroupMask =
4927 State.Builder.CreateVectorSplat(WideVF, State.Builder.getTrue());
4928 }
4929
4930 // Vectorize the interleaved load group.
4931 if (isa<LoadInst>(Instr)) {
4932 CallInst *NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4933 VecTy, Intrinsic::vp_load, {ResAddr, GroupMask, InterleaveEVL}, nullptr,
4934 "wide.vp.load");
4935 NewLoad->addParamAttr(0,
4936 Attribute::getWithAlignment(Ctx, Group->getAlign()));
4937
4938 applyMetadata(*NewLoad);
4939 // TODO: Also manage existing metadata using VPIRMetadata.
4940 Group->addMetadata(NewLoad);
4941
4942 // Scalable vectors cannot use arbitrary shufflevectors (only splats),
4943 // so must use intrinsics to deinterleave.
4944 NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4945 Intrinsic::getDeinterleaveIntrinsicID(InterleaveFactor),
4946 NewLoad->getType(), NewLoad,
4947 /*FMFSource=*/nullptr, "strided.vec");
4948
4949 const DataLayout &DL = Instr->getDataLayout();
4950 for (unsigned I = 0, J = 0; I < InterleaveFactor; ++I) {
4951 Instruction *Member = Group->getMember(I);
4952 // Skip the gaps in the group.
4953 if (!Member)
4954 continue;
4955
4956 Value *StridedVec = State.Builder.CreateExtractValue(NewLoad, I);
4957 // If this member has different type, cast the result type.
4958 if (Member->getType() != ScalarTy) {
4959 VectorType *OtherVTy = VectorType::get(Member->getType(), State.VF);
4960 StridedVec =
4961 createBitOrPointerCast(State.Builder, StridedVec, OtherVTy, DL);
4962 }
4963
4964 State.set(getVPValue(J), StridedVec);
4965 ++J;
4966 }
4967 return;
4968 } // End for interleaved load.
4969
4970 // The sub vector type for current instruction.
4971 auto *SubVT = VectorType::get(ScalarTy, State.VF);
4972 // Vectorize the interleaved store group.
4973 ArrayRef<VPValue *> StoredValues = getStoredValues();
4974 // Collect the stored vector from each member.
4975 SmallVector<Value *, 4> StoredVecs;
4976 const DataLayout &DL = Instr->getDataLayout();
4977 for (unsigned I = 0, StoredIdx = 0; I < InterleaveFactor; I++) {
4978 Instruction *Member = Group->getMember(I);
4979 // Skip the gaps in the group.
4980 if (!Member) {
4981 StoredVecs.push_back(PoisonValue::get(SubVT));
4982 continue;
4983 }
4984
4985 Value *StoredVec = State.get(StoredValues[StoredIdx]);
4986 // If this member has different type, cast it to a unified type.
4987 if (StoredVec->getType() != SubVT)
4988 StoredVec = createBitOrPointerCast(State.Builder, StoredVec, SubVT, DL);
4989
4990 StoredVecs.push_back(StoredVec);
4991 ++StoredIdx;
4992 }
4993
4994 // Interleave all the smaller vectors into one wider vector.
4995 Value *IVec = interleaveVectors(State.Builder, StoredVecs, "interleaved.vec");
4996 CallInst *NewStore = State.Builder.CreateIntrinsicWithoutFolding(
4997 Type::getVoidTy(Ctx), Intrinsic::vp_store,
4998 {IVec, ResAddr, GroupMask, InterleaveEVL});
4999
5000 NewStore->addParamAttr(1,
5001 Attribute::getWithAlignment(Ctx, Group->getAlign()));
5002
5003 applyMetadata(*NewStore);
5004 // TODO: Also manage existing metadata using VPIRMetadata.
5005 Group->addMetadata(NewStore);
5006}
5007
5008#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5010 VPSlotTracker &SlotTracker) const {
5012 O << Indent << "INTERLEAVE-GROUP with factor " << IG->getFactor() << ", ";
5014 O << ", ";
5016 if (VPValue *Mask = getMask()) {
5017 O << ", ";
5018 Mask->printAsOperand(O, SlotTracker);
5019 }
5020
5021 unsigned OpIdx = 0;
5022 for (unsigned i = 0; i < IG->getFactor(); ++i) {
5023 if (!IG->getMember(i))
5024 continue;
5025 if (getNumStoreOperands() > 0) {
5026 O << "\n" << Indent << " vp.store ";
5027 getOperand(2 + OpIdx)->printAsOperand(O, SlotTracker);
5028 O << " to index " << i;
5029 } else {
5030 O << "\n" << Indent << " ";
5032 O << " = vp.load from index " << i;
5033 }
5034 ++OpIdx;
5035 }
5036}
5037#endif
5038
5040 VPCostContext &Ctx) const {
5041 Instruction *InsertPos = getInsertPos();
5042 // Find the VPValue index of the interleave group. We need to skip gaps.
5043 unsigned InsertPosIdx = 0;
5044 for (unsigned Idx = 0; IG->getFactor(); ++Idx)
5045 if (auto *Member = IG->getMember(Idx)) {
5046 if (Member == InsertPos)
5047 break;
5048 InsertPosIdx++;
5049 }
5050 const VPValue *ValV = getNumDefinedValues() > 0
5051 ? getVPValue(InsertPosIdx)
5052 : getStoredValues()[InsertPosIdx];
5053 Type *ValTy = ValV->getScalarType();
5054 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
5055 unsigned AS =
5056 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
5057
5058 unsigned InterleaveFactor = IG->getFactor();
5059 auto *WideVecTy = VectorType::get(ValTy, VF * InterleaveFactor);
5060
5061 // Holds the indices of existing members in the interleaved group.
5063 for (unsigned IF = 0; IF < InterleaveFactor; IF++)
5064 if (IG->getMember(IF))
5065 Indices.push_back(IF);
5066
5067 // Calculate the cost of the whole interleaved group.
5068 InstructionCost Cost = Ctx.TTI.getInterleavedMemoryOpCost(
5069 InsertPos->getOpcode(), WideVecTy, IG->getFactor(), Indices,
5070 IG->getAlign(), AS, Ctx.CostKind, getMask(), NeedsMaskForGaps);
5071
5072 if (!IG->isReverse())
5073 return Cost;
5074
5075 return Cost + IG->getNumMembers() *
5076 Ctx.TTI.getShuffleCost(TargetTransformInfo::SK_Reverse,
5077 VectorTy, VectorTy, Ctx.CostKind, {},
5078 0);
5079}
5080
5083 VPCostContext &Ctx) const {
5084 // The recipe creates a scalar phi, a GEP to increment the induction and
5085 // vector add to compute the vector of pointers.
5086 // TODO: Charge costs for induction increment and vector add as well.
5087 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5088}
5089
5091 return vputils::onlyScalarValuesUsed(this) &&
5092 (!IsScalable || vputils::onlyFirstLaneUsed(this));
5093}
5094
5095#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5097 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5098 assert((getNumOperands() == 3 || getNumOperands() == 5) &&
5099 "unexpected number of operands");
5100 O << Indent << "EMIT ";
5102 O << " = WIDEN-POINTER-INDUCTION ";
5104 O << ", ";
5106 O << ", ";
5108 if (getNumOperands() == 5) {
5109 O << ", ";
5111 O << ", ";
5113 }
5114}
5115
5117 VPSlotTracker &SlotTracker) const {
5118 O << Indent << "EMIT ";
5120 O << " = EXPAND SCEV " << *Expr;
5121}
5122#endif
5123
5124#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5126 VPSlotTracker &SlotTracker) const {
5127 O << Indent << "EMIT ";
5129 O << " = WIDEN-CANONICAL-INDUCTION";
5130 printFlags(O);
5132}
5133#endif
5134
5136 auto &Builder = State.Builder;
5137 // Create a vector from the initial value.
5138 auto *VectorInit = getStartValue()->getLiveInIRValue();
5139
5140 Type *VecTy = State.VF.isScalar()
5141 ? VectorInit->getType()
5142 : VectorType::get(VectorInit->getType(), State.VF);
5143
5144 BasicBlock *VectorPH =
5145 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5146 if (State.VF.isVector()) {
5147 auto *IdxTy = Builder.getInt32Ty();
5148 auto *One = ConstantInt::get(IdxTy, 1);
5149 IRBuilder<>::InsertPointGuard Guard(Builder);
5150 Builder.SetInsertPoint(VectorPH->getTerminator());
5151 auto *RuntimeVF = getRuntimeVF(Builder, IdxTy, State.VF);
5152 auto *LastIdx = Builder.CreateSub(RuntimeVF, One);
5153 VectorInit = Builder.CreateInsertElement(
5154 PoisonValue::get(VecTy), VectorInit, LastIdx, "vector.recur.init");
5155 }
5156
5157 // Create a phi node for the new recurrence.
5158 PHINode *Phi = PHINode::Create(VecTy, 2, "vector.recur");
5159 Phi->insertBefore(State.CFG.PrevBB->getFirstInsertionPt());
5160 Phi->addIncoming(VectorInit, VectorPH);
5161 State.set(this, Phi);
5162}
5163
5166 VPCostContext &Ctx) const {
5167 if (VF.isScalar())
5168 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5169
5170 return 0;
5171}
5172
5173#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5175 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5176 O << Indent << "FIRST-ORDER-RECURRENCE-PHI ";
5178 O << " = phi ";
5180}
5181#endif
5182
5184 // Reductions do not have to start at zero. They can start with
5185 // any loop invariant values.
5186 VPValue *StartVPV = getStartValue();
5187
5188 // In order to support recurrences we need to be able to vectorize Phi nodes.
5189 // Phi nodes have cycles, so we need to vectorize them in two stages. This is
5190 // stage #1: We create a new vector PHI node with no incoming edges. We'll use
5191 // this value when we vectorize all of the instructions that use the PHI.
5192 BasicBlock *VectorPH =
5193 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5194 bool ScalarPHI = State.VF.isScalar() || isInLoop();
5195 Value *StartV = State.get(StartVPV, ScalarPHI);
5196 Type *VecTy = StartV->getType();
5197
5198 BasicBlock *HeaderBB = State.CFG.PrevBB;
5199 assert(State.CurrentParentLoop->getHeader() == HeaderBB &&
5200 "recipe must be in the vector loop header");
5201 auto *Phi = PHINode::Create(VecTy, 2, "vec.phi");
5202 Phi->insertBefore(HeaderBB->getFirstInsertionPt());
5203 State.set(this, Phi, isInLoop());
5204
5205 Phi->addIncoming(StartV, VectorPH);
5206}
5207
5208#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5210 VPSlotTracker &SlotTracker) const {
5211 O << Indent << "WIDEN-REDUCTION-PHI ";
5212
5214 O << " = phi (";
5215 printRecurrenceKind(O, Kind);
5216 O << ")";
5217 printFlags(O);
5219 if (getVFScaleFactor() > 1)
5220 O << " (VF scaled by 1/" << getVFScaleFactor() << ")";
5221}
5222#endif
5223
5225 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
5226 return vputils::onlyFirstLaneUsed(this);
5227}
5228
5230 executePhiRecipe(this, *this, State, /*IsScalar=*/false, Name);
5231}
5232
5234 VPCostContext &Ctx) const {
5235 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5236}
5237
5238#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5240 VPSlotTracker &SlotTracker) const {
5241 O << Indent << "WIDEN-PHI ";
5242
5244 O << " = phi ";
5246}
5247#endif
5248
5250 BasicBlock *VectorPH =
5251 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5252 Value *StartMask = State.get(getOperand(0));
5253 PHINode *Phi =
5254 State.Builder.CreatePHI(StartMask->getType(), 2, "active.lane.mask");
5255 Phi->addIncoming(StartMask, VectorPH);
5256 State.set(this, Phi);
5257}
5258
5259#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5261 VPSlotTracker &SlotTracker) const {
5262 O << Indent << "ACTIVE-LANE-MASK-PHI ";
5263
5265 O << " = phi ";
5267}
5268#endif
5269
5270#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5272 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5273 O << Indent << "CURRENT-ITERATION-PHI ";
5274
5276 O << " = phi ";
5278}
5279#endif
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
static MCDisassembler::DecodeStatus addOperand(MCInst &Inst, const MCOperand &Opnd)
AMDGPU Lower Kernel Arguments
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static const Function * getParent(const Value *V)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static void replaceAllUsesWith(Value *Old, Value *New, SmallPtrSet< BasicBlock *, 32 > &FreshBBs, bool IsHuge)
Replace all old uses with new ones, and push the updated BBs into FreshBBs.
Hexagon Common GEP
Value * getPointer(Value *Ptr)
iv users
Definition IVUsers.cpp:48
static constexpr Value * getValue(Ty &ValueOrUse)
static std::pair< Value *, APInt > getMask(Value *WideMask, unsigned Factor, ElementCount LeafValueEC)
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file provides a LoopVectorizationPlanner class.
static const SCEV * getAddressAccessSCEV(Value *Ptr, PredicatedScalarEvolution &PSE, const Loop *TheLoop)
Gets the address access SCEV for Ptr, if it should be used for cost modeling according to isAddressSC...
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
static bool isOrdered(const Instruction *I)
uint64_t IntrinsicInst * II
#define P(N)
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
SI Fold Operands
static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue GlueChain, SDNodeFlags Flags)
This file contains some templates that are useful if you are working with the STL at all.
This file defines less commonly used SmallVector utilities.
This file defines the SmallVector class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This file contains the declarations of different VPlan-related auxiliary helpers.
static Value * interleaveVectors(IRBuilderBase &Builder, ArrayRef< Value * > Vals, const Twine &Name)
Return a vector containing interleaved elements from multiple smaller input vectors.
static const ConstantFP * getConstantFP(const VPValue *V)
Returns the ConstantFP V wraps, or nullptr if it does not wrap one.
static void executePhiRecipe(VPSingleDefRecipe *R, VPPhiAccessors &Phi, VPTransformState &State, bool IsScalar, const Twine &Name)
Shared execute logic for VPPhi and VPWidenPHIRecipe.
static Value * createBitOrPointerCast(IRBuilderBase &Builder, Value *V, VectorType *DstVTy, const DataLayout &DL)
static Instruction::BinaryOps getSubRecurOpcode(RecurKind Kind)
static VPExecutionFrequency getExecutionFrequencyFromMD(const MDNode *Node)
Returns the execution frequency recorded in Node.
static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind)
static unsigned getCalledFnOperandIndex(ArrayRef< VPValue * > Operands)
For call VPInstruction operands, return the operand index of the called function.
This file contains the declarations of the Vectorization Plan base classes:
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
Adds the attribute to the indicated argument.
This class represents a function call, abstracting a target machine's calling convention.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
static LLVM_ABI StringRef getPredicateName(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
void setSuccessor(unsigned idx, BasicBlock *NewSucc)
static ConstantAsMetadata * get(Constant *C)
Definition Metadata.h:548
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
bool isNegZero() const
Return true if the value is negative zero.
Definition Constants.h:473
bool isOne() const
Returns true if this value is exactly +1.0.
Definition Constants.h:485
bool isZero() const
Return true if the value is positive or negative zero.
Definition Constants.h:467
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:320
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
static bool isSupportedFloatingPointType(Type *Ty)
Returns true if Ty is a supported floating-point type for phi, select, or call FPMathOperators.
Definition Operator.h:302
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
LLVM_ABI void print(raw_ostream &O) const
Print fast-math flags to O.
Definition Operator.cpp:290
void setAllowContract(bool B=true)
Definition FMF.h:90
bool noSignedZeros() const
Definition FMF.h:67
bool noInfs() const
Definition FMF.h:66
void setAllowReciprocal(bool B=true)
Definition FMF.h:87
bool allowReciprocal() const
Definition FMF.h:68
void setNoSignedZeros(bool B=true)
Definition FMF.h:84
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
bool approxFunc() const
Definition FMF.h:70
void setNoNaNs(bool B=true)
Definition FMF.h:78
void setAllowReassoc(bool B=true)
Flag setters.
Definition FMF.h:75
bool noNaNs() const
Definition FMF.h:65
void setApproxFunc(bool B=true)
Definition FMF.h:93
void setNoInfs(bool B=true)
Definition FMF.h:81
bool allowContract() const
Definition FMF.h:69
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
bool willReturn() const
Determine if the function will return.
Definition Function.h:647
Intrinsic::ID getIntrinsicID() const LLVM_READONLY
getIntrinsicID - This method returns the ID number of the specified function, or Intrinsic::not_intri...
Definition Function.h:247
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:577
bool doesNotAccessMemory() const
Determine if the function does not access memory.
Definition Function.cpp:873
Type * getReturnType() const
Returns the type of the ret val.
Definition Function.h:217
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags none()
Common base class shared among various IRBuilders.
Definition IRBuilder.h:111
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Definition IRBuilder.h:2678
IntegerType * getInt1Ty()
Fetch the type representing a single bit.
Definition IRBuilder.h:516
Value * CreateInsertValue(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2732
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
Definition IRBuilder.h:2666
LLVM_ABI Value * CreateVectorSpliceRight(Value *V1, Value *V2, Value *Offset, const Twine &Name="")
Create a vector.splice.right intrinsic call, or a shufflevector that produces the same result if the ...
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
Definition IRBuilder.h:1943
CondBrInst * CreateCondBr(Value *Cond, BasicBlock *True, BasicBlock *False, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a conditional 'br Cond, TrueDest, FalseDest' instruction.
Definition IRBuilder.h:1221
LLVM_ABI Value * CreateSelectFMF(Value *C, Value *True, Value *False, FMFSource FMFSource, const Twine &Name="", Instruction *MDFrom=nullptr)
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2725
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateFreeze(Value *V, const Twine &Name="")
Definition IRBuilder.h:2744
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Definition IRBuilder.h:531
Value * CreateExtractVector(Type *DstType, Value *SrcVec, Value *Idx, const Twine &Name="")
Create a call to the vector.extract intrinsic.
Definition IRBuilder.h:1117
Value * CreatePtrAdd(Value *Ptr, Value *Offset, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
Definition IRBuilder.h:2101
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Definition IRBuilder.h:2293
void setFastMathFlags(FastMathFlags NewFMF)
Set the fast-math flags to be used with generated fp-math operators.
Definition IRBuilder.h:298
LLVM_ABI Value * CreateVectorReverse(Value *V, const Twine &Name="")
Return a vector value that contains the vector V reversed.
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2395
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Definition IRBuilder.h:479
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Definition IRBuilder.h:1775
LLVM_ABI Value * CreateOrReduce(Value *Src)
Create a vector int OR reduction intrinsic of the source vector.
ConstantInt * getInt32(uint32_t C)
Get a constant 32-bit value.
Definition IRBuilder.h:474
Value * CreateCmp(CmpInst::Predicate Pred, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:2525
Value * CreateNot(Value *V, const Twine &Name="")
Definition IRBuilder.h:1859
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2391
Value * CreateCountTrailingZeroElems(Type *ResTy, Value *Mask, bool ZeroIsPoison=true, const Twine &Name="")
Create a call to llvm.experimental_cttz_elts.
Definition IRBuilder.h:1159
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1444
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2130
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1427
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:459
Value * CreateBinOp(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:1736
Value * CreateICmpUGE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2403
Value * CreateLogicalOr(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Definition IRBuilder.h:1783
StoreInst * CreateAlignedStore(Value *Val, Value *Ptr, MaybeAlign Align, bool isVolatile=false)
Definition IRBuilder.h:1962
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
Definition IRBuilder.h:1597
LLVM_ABI Value * CreateStepVector(Type *DstType, const Twine &Name="")
Creates a vector of type DstType with the linear sequence <0, 1, ...>
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
Definition IRBuilder.h:1131
Value * CreateMul(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1461
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
A struct for saving information about induction variables.
@ IK_FpInduction
Floating point induction variable.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
bool isCast() const
bool isBinaryOp() const
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
const char * getOpcodeName() const
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
bool isUnaryOp() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
The group of interleaved loads/stores sharing the same stride and close to each other.
uint32_t getFactor() const
InstTy * getMember(uint32_t Index) const
Get the member with the given index Index.
bool isReverse() const
InstTy * getInsertPos() const
void addMetadata(InstTy *NewInst) const
Add metadata (e.g.
Align getAlign() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Metadata node.
Definition Metadata.h:1081
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
Information for memory intrinsic cost model.
Root of the metadata hierarchy.
Definition Metadata.h:64
LLVM_ABI void print(raw_ostream &OS, const Module *M=nullptr, bool IsForDebug=false) const
Print.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
static bool isAnyOfRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static LLVM_ABI bool isSubRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is for a sub operation.
static bool isFindIVRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static bool isMinMaxRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is any min/max kind.
This class represents an analyzed expression in the program.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
This class represents the LLVM 'select' instruction.
This class provides computation of slot numbers for LLVM Assembly writing.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static LLVM_ABI PartialReductionExtendKind getPartialReductionExtendKind(Instruction *I)
Get the kind of extension that an instruction represents.
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Reverse
Reverse the order of the vector.
CastContextHint
Represents a hint about the context in which a cast is used.
@ Reversed
The cast is used with a reversed load/store.
@ Masked
The cast is used with a masked load/store.
@ Normal
The cast is used with a normal load/store.
@ Interleave
The cast is used with an interleaved load/store.
@ GatherScatter
The cast is used with a gather/scatter.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isByteTy() const
True if this is an instance of ByteType.
Definition Type.h:237
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
Definition Type.h:265
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
value_op_iterator value_op_end()
Definition User.h:288
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
value_op_iterator value_op_begin()
Definition User.h:285
void execute(VPTransformState &State) override
Generate the active lane mask phi of the vector loop.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4432
RecipeListTy & getRecipeList()
Returns a reference to the list of recipes.
Definition VPlan.h:4485
iterator end()
Definition VPlan.h:4469
void insert(VPRecipeBase *Recipe, iterator InsertPt)
Definition VPlan.h:4498
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenMemoryRecipe.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3018
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:3013
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:3009
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:97
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
static bool isHeader(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop header, based on regions or VPDT in their absence.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPBranchOnMaskRecipe.
void execute(VPTransformState &State) override
Generate the extraction of the appropriate bit from the block mask and the conditional branch.
LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
Definition VPlanValue.h:578
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
Definition VPlanValue.h:551
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
Definition VPlanValue.h:563
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Definition VPlanValue.h:573
InductionDescriptor::InductionKind getInductionKind() const
Definition VPlan.h:4250
VPValue * getIndex() const
Definition VPlan.h:4247
VPValue * getStepValue() const
Definition VPlan.h:4248
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPDerivedIVRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getStartValue() const
Definition VPlan.h:4246
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPExpandSCEVRecipe(const SCEV *Expr)
bool isVectorToScalar() const
Returns true if this VPExpressionRecipe produces a single scalar.
SmallVector< VPSingleDefRecipe * > decompose()
Return and insert the recipes of the expression back into the VPlan, directly before the current reci...
bool mayHaveSideEffects() const
Returns true if this expression contains recipes that may have side effects.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
bool mayReadOrWriteMemory() const
Returns true if this expression contains recipes that may read from or write to memory.
VPExpressionRecipe(ExpressionTypes ExpressionType, ArrayRef< VPSingleDefRecipe * > ExpressionRecipes)
Construct a new VPExpressionRecipe by internalizing recipes in ExpressionRecipes.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this header phi recipe.
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2490
void execute(VPTransformState &State) override
Produce a vectorized histogram operation.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHistogramRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getMask() const
Return the mask operand if one was provided, or a null pointer if all lanes should be executed uncond...
Definition VPlan.h:2208
Class to record and manage LLVM IR flags.
Definition VPlan.h:696
FastMathFlagsTy FMFs
Definition VPlan.h:788
ReductionFlagsTy ReductionFlags
Definition VPlan.h:790
LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const
Returns true if the set flags are valid for Opcode.
WrapFlagsTy WrapFlags
Definition VPlan.h:782
void printFlags(raw_ostream &O) const
bool hasFastMathFlags() const
Returns true if the recipe has fast-math flags.
Definition VPlan.h:1005
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
bool isReductionOrdered() const
Definition VPlan.h:1060
TruncFlagsTy TruncFlags
Definition VPlan.h:783
CmpInst::Predicate getPredicate() const
Definition VPlan.h:977
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
ExactFlagsTy ExactFlags
Definition VPlan.h:785
void intersectFlags(const VPIRFlags &Other)
Only keep flags also present in Other.
uint8_t GEPFlagsStorage
Definition VPlan.h:786
GEPNoWrapFlags getGEPNoWrapFlags() const
Definition VPlan.h:995
bool hasPredicate() const
Returns true if the recipe has a comparison predicate.
Definition VPlan.h:1000
LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode, Type *ResultTy) const
Returns true if Opcode with scalar result type ResultTy has its required flags set.
DisjointFlagsTy DisjointFlags
Definition VPlan.h:784
FCmpFlagsTy FCmpFlags
Definition VPlan.h:789
NonNegFlagsTy NonNegFlags
Definition VPlan.h:787
bool isReductionInLoop() const
Definition VPlan.h:1066
void applyFlags(Instruction &I) const
Apply the IR flags to I.
Definition VPlan.h:934
uint8_t CmpPredStorage
Definition VPlan.h:781
RecurKind getRecurKind() const
Definition VPlan.h:1054
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
LLVM_ABI_FOR_TEST InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPIRInstruction.
VPIRInstruction(Instruction &I)
VPIRInstruction::create() should be used to create VPIRInstructions, as subclasses may need to be cre...
Definition VPlan.h:1739
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
std::optional< VPExecutionFrequency > getExecutionFrequency() const
Returns the frequency recorded by setExecutionFrequency, if any.
void intersect(const VPIRMetadata &MD)
Intersect this VPIRMetadata object with MD, keeping only metadata nodes that are common to both.
void clearExecutionFrequency()
Drop the frequency recorded by setExecutionFrequency, if any.
VPIRMetadata()=default
void print(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print metadata with node IDs.
void applyMetadata(Instruction &I) const
Add all metadata to I.
void setMetadata(unsigned Kind, MDNode *Node)
Set metadata with kind Kind to Node.
Definition VPlan.h:1233
void setExecutionFrequency(std::optional< VPExecutionFrequency > Freq, LLVMContext &Ctx)
Record that the recipe executes with frequency Freq, relative to the entry of the loop region.
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1303
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPInstruction.
VPInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
bool doesGeneratePerAllLanes() const
Returns true if this recipe produces scalar values for all VF lanes.
@ ExtractLastActive
Extracts the last active lane from a set of vectors.
Definition VPlan.h:1421
@ Intrinsic
Calls a scalar intrinsic. The intrinsic ID is the last operand.
Definition VPlan.h:1433
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1412
@ ExitingIVValue
Compute the exiting value of a wide induction after vectorization, that is the value of the last lane...
Definition VPlan.h:1425
@ WideIVStep
Scale the first operand (vector step) by the second operand (scalar-step).
Definition VPlan.h:1429
@ ResumeForEpilogue
Explicit user for values in the main VPlan, used by the epilogue vector loop.
Definition VPlan.h:1415
@ Unpack
Extracts all lanes from its (non-scalable) vector operand.
Definition VPlan.h:1363
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1408
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1358
@ BuildStructVector
Given operands of (the same) struct type, creates a struct of fixed- width vectors each containing a ...
Definition VPlan.h:1355
@ CanonicalIVIncrementForPart
Definition VPlan.h:1339
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1366
bool hasResult() const
Definition VPlan.h:1518
bool opcodeMayReadOrWriteFromMemory() const
Returns true if the underlying opcode may read from or write to memory.
LLVM_DUMP_METHOD void dump() const
Print the VPInstruction to dbgs() (for debugging).
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the VPInstruction to O.
StringRef getName() const
Returns the symbolic name assigned to the VPInstruction.
Definition VPlan.h:1603
unsigned getOpcode() const
Definition VPlan.h:1497
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
void addOperand(VPValue *Op)
Add Op as operand of this VPInstruction.
bool isVectorToScalar() const
Returns true if this VPInstruction produces a scalar value from a vector, e.g.
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
unsigned getNumOperandsForOpcode() const
Return the number of operands determined by the opcode of the VPInstruction, excluding mask.
bool isMasked() const
Returns true if the VPInstruction has a mask operand.
Definition VPlan.h:1544
void execute(VPTransformState &State) override
Generate the instruction.
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
bool needsMaskForGaps() const
Return true if the access needs a mask because of the gaps.
Definition VPlan.h:3122
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this recipe.
Instruction * getInsertPos() const
Definition VPlan.h:3126
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3124
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3116
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3145
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3110
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3219
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3232
void execute(VPTransformState &State) override
Generate the wide load or store, and shuffles.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3182
void execute(VPTransformState &State) override
Generate the wide load or store, and shuffles.
static LLVM_ABI std::optional< unsigned > getMaskParamPos(Intrinsic::ID IntrinsicID)
static LLVM_ABI std::optional< unsigned > getMemoryDataParamPos(Intrinsic::ID)
static LLVM_ABI std::optional< unsigned > getMemoryPointerParamPos(Intrinsic::ID)
In what follows, the term "input IR" refers to code that is fed into the vectorizer whereas the term ...
static VPLane getLastLaneForVF(const ElementCount &VF)
static VPLane getLaneFromEnd(const ElementCount &VF, unsigned Offset)
static VPLane getFirstLane()
Helper type to provide functions to access incoming values and blocks for phi-like recipes.
Definition VPlan.h:1618
virtual const VPRecipeBase * getAsRecipe() const =0
Return a VPRecipeBase* to the current object.
LLVM_ABI_FOR_TEST VPValue * getIncomingValueForBlock(const VPBasicBlock *VPBB) const
Returns the incoming value for VPBB. VPBB must be an incoming block.
void removeIncomingValueFor(VPBlockBase *IncomingBlock) const
Removes the incoming value for IncomingBlock, which must be a predecessor.
detail::zippy< llvm::detail::zip_first, VPUser::const_operand_range, const_incoming_blocks_range > incoming_values_and_blocks() const
Returns an iterator range over pairs of incoming values and corresponding incoming blocks.
Definition VPlan.h:1668
VPValue * getIncomingValue(unsigned Idx) const
Returns the incoming VPValue with index Idx.
Definition VPlan.h:1627
void printPhiOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the recipe.
void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const
Sets the incoming value for VPBB to V.
void execute(VPTransformState &State) override
Generates phi nodes for live-outs (from a replicate region) as needed to retain SSA form.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
bool mayReadFromMemory() const
Returns true if the recipe may read from memory.
bool mayHaveSideEffects() const
Returns true if the recipe may have side-effects.
virtual void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Each concrete VPRecipe prints itself, without printing common information, like debug info or metadat...
VPRegionBlock * getRegion()
Definition VPlan.h:4831
void dump() const
Dump the recipe to stderr (for debugging).
Definition VPlan.cpp:115
bool isPhi() const
Returns true for PHI-like recipes.
bool mayWriteToMemory() const
Returns true if the recipe may write to memory.
VPRecipeTy getVPRecipeID() const
Definition VPlan.h:521
virtual InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
VPBasicBlock * getParent()
Definition VPlan.h:475
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
bool isSafeToSpeculativelyExecute() const
Return true if we can safely execute this recipe unconditionally even if it is masked originally.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
VPRecipeBase(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:465
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this recipe, taking into account if the cost computation should be skipped and the...
void print(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const
Print the recipe, delegating to printRecipe().
void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
void moveAfter(VPRecipeBase *MovePos)
Unlink this recipe from its current VPBasicBlock and insert it into the VPBasicBlock that MovePos liv...
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:350
friend class VPValue
Definition VPlanValue.h:329
void execute(VPTransformState &State) override
Generate the reduction in the loop.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3393
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2921
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2940
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
bool isConditional() const
Return true if the in-loop reduction is conditional.
Definition VPlan.h:3332
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of VPReductionRecipe.
VPValue * getVecOp() const
The VPValue of the vector value to be reduced.
Definition VPlan.h:3345
VPValue * getCondOp() const
The VPValue of the condition for the block.
Definition VPlan.h:3347
RecurKind getRecurrenceKind() const
Return the recurrence kind for the in-loop reduction.
Definition VPlan.h:3328
bool isPartialReduction() const
Returns true if the reduction outputs a vector with a scaled down VF.
Definition VPlan.h:3334
VPValue * getChainOp() const
The VPValue of the scalar Chain being accumulated.
Definition VPlan.h:3343
bool isInLoop() const
Returns true if the reduction is in-loop.
Definition VPlan.h:3338
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the reduction in the loop.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4657
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4733
const VPBranchOnMaskRecipe * getEntryBranchOnMask() const
Return the VPBranchOnMaskRecipe from the entry block of this replicating region.
Definition VPlan.cpp:718
void execute(VPTransformState &State) override
Generate replicas of the desired Ingredient.
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3474
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPReplicateRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
static Type * computeScalarType(const Instruction *I, ArrayRef< VPValue * > Operands)
Compute the scalar result type for a VPReplicateRecipe wrapping I with Operands (excluding any predic...
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
unsigned getOpcode() const
Definition VPlan.h:3512
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPScalarIVStepsRecipe.
bool doesGeneratePerAllLanes() const
Returns true if this recipe produces scalar values for all VF lanes.
VPValue * getStepValue() const
Definition VPlan.h:4305
VPValue * getStartIndex() const
Return the StartIndex, or null if known to be zero, valid only after unrolling.
Definition VPlan.h:4313
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the scalarized versions of the phi node as needed by their users.
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
LLVM_DUMP_METHOD void dump() const
Print this VPSingleDefRecipe to dbgs() (for debugging).
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:613
This class can be used to assign names to VPValues.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:213
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:397
void printOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the operands to O.
Definition VPlan.cpp:1509
operand_range operands()
Definition VPlanValue.h:473
unsigned getNumOperands() const
Definition VPlanValue.h:437
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:438
bool operands_empty() const
Definition VPlanValue.h:477
VPValue * getLastOperand() const
Returns the last operand.
Definition VPlanValue.h:444
void addOperand(VPValue *Operand)
Definition VPlanValue.h:423
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
Definition VPlan.cpp:1461
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
void printAsOperand(raw_ostream &OS, VPSlotTracker &Tracker) const
Definition VPlan.cpp:1505
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
void setUnderlyingValue(Value *Val)
Definition VPlanValue.h:205
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
Definition VPlanValue.h:179
VPValue * getVFValue() const
Definition VPlan.h:2309
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Type * getSourceElementType() const
Definition VPlan.h:2306
int64_t getStride() const
Definition VPlan.h:2307
void materializeOffset(unsigned Part=0)
Adds the offset operand to the recipe.
VPValue * getStride() const
Definition VPlan.h:2383
Type * getSourceElementType() const
Definition VPlan.h:2398
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPValue * getVFxPart() const
Definition VPlan.h:2385
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
operand_range args()
Definition VPlan.h:2161
Function * getCalledScalarFunction() const
Definition VPlan.h:2157
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCallRecipe.
void execute(VPTransformState &State) override
Produce a widened version of the call instruction.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Instruction::CastOps getOpcode() const
Definition VPlan.h:1932
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Produce widened copies of the cast.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCastRecipe.
void execute(VPTransformState &State) override
Generate the gep nodes.
Type * getSourceElementType() const
Definition VPlan.h:2263
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2568
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2585
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenIntOrFpInductionRecipe.
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2674
bool isCanonical() const
Returns true if the induction is canonical, i.e.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
CallInst * createVectorCall(VPTransformState &State)
Helper function to produce the widened intrinsic call.
Intrinsic::ID getVectorIntrinsicID() const
Return the ID of the intrinsic.
Definition VPlan.h:2047
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
StringRef getIntrinsicName() const
Return to name of the intrinsic as string.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
void execute(VPTransformState &State) override
Produce a widened version of the vector intrinsic.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector intrinsic.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
void execute(VPTransformState &State) override
Produce a widened version of the vector memory intrinsic.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector memory intrinsic.
bool IsMasked
Whether the memory access is masked.
Definition VPlan.h:3779
bool isConsecutive() const
Return whether the loaded-from / stored-to addresses are consecutive.
Definition VPlan.h:3804
Instruction & Ingredient
Definition VPlan.h:3770
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Return the cost of this VPWidenMemoryRecipe.
bool Consecutive
Whether the accessed addresses are consecutive.
Definition VPlan.h:3776
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3814
Align Alignment
Alignment information for this memory access.
Definition VPlan.h:3773
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3807
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPHIRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
bool onlyScalarsGenerated(bool IsScalable)
Returns true if only scalar values will be generated.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPointerInductionRecipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
void execute(VPTransformState &State) override
Produce a widened instruction using the opcode and operands of the recipe, processing State....
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getOpcode() const
Definition VPlan.h:1874
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4844
const DataLayout & getDataLayout() const
Definition VPlan.h:5058
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5164
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
void mutateType(Type *Ty)
Mutate the type of this Value to be of the specified type.
Definition Value.h:809
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
self_iterator getIterator()
Definition ilist_node.h:123
iterator erase(iterator where)
Definition ilist.h:204
pointer remove(iterator &IT)
Definition ilist.h:188
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
LLVM_ABI Intrinsic::ID getDeinterleaveIntrinsicID(unsigned Factor)
Returns the corresponding llvm.vector.deinterleaveN intrinsic for factor N.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI AttributeSet getFnAttributes(LLVMContext &C, ID id)
Return the function attributes for an intrinsic.
LLVM_ABI StringRef getBaseName(ID id)
Return the LLVM name for an intrinsic, without encoded types for overloading, such as "llvm....
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
bool match(Val *V, const Pattern &P)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::Or, true > m_c_LogicalOr(const LHS &L, const RHS &R)
Matches L || R with LHS and RHS in either order.
auto m_ZExtOrTrunc(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExplicitVectorLength, Op0_t > m_EVL(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
BranchProbability getExecutionProbability(BlockFrequency Freq)
Returns Freq as a BranchProbability, relative to the full mass.
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
bool isAddressSCEVForCost(const SCEV *Addr, ScalarEvolution &SE, const Loop *L)
Returns true if Addr is an address SCEV that can be passed to TTI::getAddressComputationCost,...
bool onlyFirstPartUsed(const VPValue *Def)
Returns true if only the first part of Def is used.
Intrinsic::ID getIntrinsicID(const Ty *R)
Return the intrinsic ID underlying a call.
Definition VPlanUtils.h:93
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
bool onlyScalarValuesUsed(const VPValue *Def)
Returns true if only scalar values of Def are used by all users.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
LLVM_ABI_FOR_TEST const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
LLVM_ABI Value * createSimpleReduction(IRBuilderBase &B, Value *Src, RecurKind RdxKind)
Create a reduction of the given vector.
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
InstructionCost Cost
@ Undef
Value of the register doesn't matter.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
VectorInstrContext
Represents a hint about the context in which a vector instruction or intrinsic is used.
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
auto map_to_vector(ContainerTy &&C, FuncTy &&F)
Map a range to a SmallVector with element types deduced from the mapping.
Value * getRuntimeVF(IRBuilderBase &B, Type *Ty, ElementCount VF)
Return the runtime value for VF.
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
void interleaveComma(const Container &c, StreamT &os, UnaryFunctor each_fn)
Definition STLExtras.h:2329
auto cast_or_null(const Y &Val)
Definition Casting.h:714
LLVM_ABI Value * concatenateVectors(IRBuilderBase &Builder, ArrayRef< Value * > Vecs)
Concatenate a list of vectors.
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI Value * createMinMaxOp(IRBuilderBase &Builder, RecurKind RK, Value *Left, Value *Right)
Returns a Min/Max operation corresponding to MinMaxRecurrenceKind.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
static Error getOffset(const SymbolRef &Sym, SectionRef Sec, uint64_t &Result)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI Constant * createBitMaskForGaps(IRBuilderBase &Builder, unsigned VF, const InterleaveGroup< Instruction > &Group)
Create a mask that filters the members of an interleave group where there are gaps.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
cl::opt< unsigned > ForceTargetInstructionCost("force-target-instruction-cost", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's expected cost for " "an instruction to a single constant value. Mostly " "useful for getting consistent testing."))
Definition VPlan.cpp:58
ElementCount getVectorizedTypeVF(Type *Ty)
Returns the number of vector elements for a vectorized type.
LLVM_ABI llvm::SmallVector< int, 16 > createReplicatedMask(unsigned ReplicationFactor, unsigned VF)
Create a mask with replicated elements.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool isPointerTy(const Type *T)
Definition SPIRVUtils.h:383
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
Type * toVectorizedTy(Type *Ty, ElementCount EC)
A helper for converting to vectorized types.
LLVM_ABI Type * computeScalarTypeForInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands)
Compute the scalar result type for an IR Opcode given Operands.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
LLVM_ABI bool isVectorIntrinsicWithStructReturnOverloadAtField(Intrinsic::ID ID, int RetIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic that returns a struct is overloaded at the struct elem...
@ Other
Any other memory.
Definition ModRef.h:68
static const MachineInstrBuilder & addOffset(const MachineInstrBuilder &MIB, int Offset)
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FMinimumNum
FP min with llvm.minimumnum semantics.
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ FMinimum
FP min with llvm.minimum semantics.
@ FMaxNum
FP max with llvm.maxnum semantics including NaNs.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ None
Not a recurrence.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMaximum
FP max with llvm.maximum semantics.
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ FMul
Product of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ FMinNum
FP min with llvm.minnum semantics including NaNs.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
@ FMaximumNum
FP max with llvm.maximumnum semantics.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
DWARFExpression::Operation Op
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
LLVM_ABI Value * createOrderedReduction(IRBuilderBase &B, RecurKind RdxKind, Value *Src, Value *Start)
Create an ordered reduction intrinsic using the given recurrence kind RdxKind.
ArrayRef< Type * > getContainedTypes(Type *const &Ty)
Returns the types contained in Ty.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isVectorIntrinsicWithOverloadTypeAtArg(Intrinsic::ID ID, int OpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic is overloaded on the type of the operand at index OpdI...
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Struct to hold various analysis needed for cost computations.
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
Definition VPlan.cpp:1923
static bool executesAtMostOnce(const VPlan &Plan, ElementCount VF)
Returns true if the vector loop body of Plan is known to execute at most once at VF,...
TargetTransformInfo::TargetCostKind CostKind
The frequency with which a recipe executes, relative to the entry of the loop region.
Definition VPlan.h:1172
void execute(VPTransformState &State) override
Generate the phi nodes.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this first-order recurrence phi recipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use cast/dyn_cast/isa and exec...
Definition VPlan.h:1797
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
PHINode & getIRPhi() const
Definition VPlan.h:1810
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void execute(VPTransformState &State) override
Generate the instruction.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1117
InstructionCost getCostForRecipeWithOpcode(unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const
Compute the cost for this recipe for VF, using Opcode and Ctx.
SmallDenseMap< const VPBasicBlock *, BasicBlock * > VPBB2IRBB
A mapping of each VPBasicBlock to the corresponding BasicBlock.
VPTransformState holds information passed down when "executing" a VPlan, needed for generating the ou...
struct llvm::VPTransformState::CFGState CFG
IRBuilderBase & Builder
Hold a reference to the IRBuilder used to generate output IR code.
Value * get(const VPValue *Def, bool NeedsSingleScalar=false)
Get the generated vector Value for a given VPValue Def if NeedsSingleScalar is false,...
Definition VPlan.cpp:282
ElementCount VF
The chosen Vectorization Factor of the loop being vectorized.
void execute(VPTransformState &State) override
Generate the wide load or gather.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenLoadEVLRecipe.
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3905
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate a wide load or gather.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
VPValue * getStoredValue() const
Return the address accessed by this recipe.
Definition VPlan.h:4007
void execute(VPTransformState &State) override
Generate the wide store or scatter.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenStoreEVLRecipe.
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:4010
void execute(VPTransformState &State) override
Generate a wide store or scatter.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
VPValue * getStoredValue() const
Return the value stored by this recipe.
Definition VPlan.h:3955