LLVM 24.0.0git
BasicTTIImpl.h
Go to the documentation of this file.
1//===- BasicTTIImpl.h -------------------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file provides a helper that implements much of the TTI interface in
11/// terms of the target-independent code generator and TargetLowering
12/// interfaces.
13//
14//===----------------------------------------------------------------------===//
15
16#ifndef LLVM_CODEGEN_BASICTTIIMPL_H
17#define LLVM_CODEGEN_BASICTTIIMPL_H
18
19#include "llvm/ADT/APInt.h"
20#include "llvm/ADT/BitVector.h"
21#include "llvm/ADT/STLExtras.h"
35#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
40#include "llvm/IR/InstrTypes.h"
41#include "llvm/IR/Instruction.h"
43#include "llvm/IR/Intrinsics.h"
44#include "llvm/IR/Operator.h"
45#include "llvm/IR/Type.h"
46#include "llvm/IR/Value.h"
55#include <algorithm>
56#include <cassert>
57#include <cstdint>
58#include <limits>
59#include <optional>
60#include <utility>
61
62namespace llvm {
63
64class Function;
65class GlobalValue;
66class LLVMContext;
67class ScalarEvolution;
68class SCEV;
69class TargetMachine;
70
72
73/// Base class which can be used to help build a TTI implementation.
74///
75/// This class provides as much implementation of the TTI interface as is
76/// possible using the target independent parts of the code generator.
77///
78/// In order to subclass it, your class must implement a getST() method to
79/// return the subtarget, and a getTLI() method to return the target lowering.
80/// We need these methods implemented in the derived class so that this class
81/// doesn't have to duplicate storage for them.
82template <typename T>
84private:
86 using TTI = TargetTransformInfo;
87
88 /// Helper function to access this as a T.
89 const T *thisT() const { return static_cast<const T *>(this); }
90
91 /// Estimate a cost of Broadcast as an extract and sequence of insert
92 /// operations.
94 getBroadcastShuffleOverhead(FixedVectorType *VTy,
97 // Broadcast cost is equal to the cost of extracting the zero'th element
98 // plus the cost of inserting it into every element of the result vector.
99 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
100 CostKind, 0, nullptr, nullptr);
101
102 for (int i = 0, e = VTy->getNumElements(); i < e; ++i) {
103 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, VTy,
104 CostKind, i, nullptr, nullptr);
105 }
106 return Cost;
107 }
108
109 /// Estimate a cost of shuffle as a sequence of extract and insert
110 /// operations.
112 getPermuteShuffleOverhead(FixedVectorType *VTy,
115 // Shuffle cost is equal to the cost of extracting element from its argument
116 // plus the cost of inserting them onto the result vector.
117
118 // e.g. <4 x float> has a mask of <0,5,2,7> i.e we need to extract from
119 // index 0 of first vector, index 1 of second vector,index 2 of first
120 // vector and finally index 3 of second vector and insert them at index
121 // <0,1,2,3> of result vector.
122 for (int i = 0, e = VTy->getNumElements(); i < e; ++i) {
123 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, VTy,
124 CostKind, i, nullptr, nullptr);
125 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
126 CostKind, i, nullptr, nullptr);
127 }
128 return Cost;
129 }
130
131 /// Estimate a cost of subvector extraction as a sequence of extract and
132 /// insert operations.
133 InstructionCost getExtractSubvectorOverhead(VectorType *VTy,
135 int Index,
136 FixedVectorType *SubVTy) const {
137 assert(VTy && SubVTy &&
138 "Can only extract subvectors from vectors");
139 int NumSubElts = SubVTy->getNumElements();
141 (Index + NumSubElts) <=
142 (int)cast<FixedVectorType>(VTy)->getNumElements()) &&
143 "SK_ExtractSubvector index out of range");
144
146 // Subvector extraction cost is equal to the cost of extracting element from
147 // the source type plus the cost of inserting them into the result vector
148 // type.
149 for (int i = 0; i != NumSubElts; ++i) {
150 Cost +=
151 thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
152 CostKind, i + Index, nullptr, nullptr);
153 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, SubVTy,
154 CostKind, i, nullptr, nullptr);
155 }
156 return Cost;
157 }
158
159 /// Estimate a cost of subvector insertion as a sequence of extract and
160 /// insert operations.
161 InstructionCost getInsertSubvectorOverhead(VectorType *VTy,
163 int Index,
164 FixedVectorType *SubVTy) const {
165 assert(VTy && SubVTy &&
166 "Can only insert subvectors into vectors");
167 int NumSubElts = SubVTy->getNumElements();
169 (Index + NumSubElts) <=
170 (int)cast<FixedVectorType>(VTy)->getNumElements()) &&
171 "SK_InsertSubvector index out of range");
172
174 // Subvector insertion cost is equal to the cost of extracting element from
175 // the source type plus the cost of inserting them into the result vector
176 // type.
177 for (int i = 0; i != NumSubElts; ++i) {
178 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, SubVTy,
179 CostKind, i, nullptr, nullptr);
180 Cost +=
181 thisT()->getVectorInstrCost(Instruction::InsertElement, VTy, CostKind,
182 i + Index, nullptr, nullptr);
183 }
184 return Cost;
185 }
186
187 /// Local query method delegates up to T which *must* implement this!
188 const TargetSubtargetInfo *getST() const {
189 return static_cast<const T *>(this)->getST();
190 }
191
192 /// Local query method delegates up to T which *must* implement this!
193 const TargetLoweringBase *getTLI() const {
194 return static_cast<const T *>(this)->getTLI();
195 }
196
197 static ISD::MemIndexedMode getISDIndexedMode(TTI::MemIndexedMode M) {
198 switch (M) {
200 return ISD::UNINDEXED;
201 case TTI::MIM_PreInc:
202 return ISD::PRE_INC;
203 case TTI::MIM_PreDec:
204 return ISD::PRE_DEC;
205 case TTI::MIM_PostInc:
206 return ISD::POST_INC;
207 case TTI::MIM_PostDec:
208 return ISD::POST_DEC;
209 }
210 llvm_unreachable("Unexpected MemIndexedMode");
211 }
212
213 InstructionCost getCommonMaskedMemoryOpCost(unsigned Opcode, Type *DataTy,
214 Align Alignment,
215 bool VariableMask,
216 bool IsGatherScatter,
218 unsigned AddressSpace = 0) const {
219 // We cannot scalarize scalable vectors, so return Invalid.
220 if (isa<ScalableVectorType>(DataTy))
222
223 auto *VT = cast<FixedVectorType>(DataTy);
224 unsigned VF = VT->getNumElements();
225
226 // Assume the target does not have support for gather/scatter operations
227 // and provide a rough estimate.
228 //
229 // First, compute the cost of the individual memory operations.
230 InstructionCost AddrExtractCost =
231 IsGatherScatter ? getScalarizationOverhead(
233 PointerType::get(VT->getContext(), 0), VF),
234 /*Insert=*/false, /*Extract=*/true, CostKind)
235 : 0;
236
237 // The cost of the scalar loads/stores.
238 InstructionCost MemoryOpCost =
239 VF * thisT()->getMemoryOpCost(Opcode, VT->getElementType(), Alignment,
241
242 // Next, compute the cost of packing the result in a vector.
243 InstructionCost PackingCost =
244 getScalarizationOverhead(VT, Opcode != Instruction::Store,
245 Opcode == Instruction::Store, CostKind);
246
247 InstructionCost ConditionalCost = 0;
248 if (VariableMask) {
249 // Compute the cost of conditionally executing the memory operations with
250 // variable masks. This includes extracting the individual conditions, a
251 // branches and PHIs to combine the results.
252 // NOTE: Estimating the cost of conditionally executing the memory
253 // operations accurately is quite difficult and the current solution
254 // provides a very rough estimate only.
255 ConditionalCost =
258 /*Insert=*/false, /*Extract=*/true, CostKind) +
259 VF * (thisT()->getCFInstrCost(Instruction::CondBr, CostKind) +
260 thisT()->getCFInstrCost(Instruction::PHI, CostKind));
261 }
262
263 return AddrExtractCost + MemoryOpCost + PackingCost + ConditionalCost;
264 }
265
266 /// Checks if the provided mask \p is a splat mask, i.e. it contains only -1
267 /// or same non -1 index value and this index value contained at least twice.
268 /// So, mask <0, -1,-1, -1> is not considered splat (it is just identity),
269 /// same for <-1, 0, -1, -1> (just a slide), while <2, -1, 2, -1> is a splat
270 /// with \p Index=2.
271 static bool isSplatMask(ArrayRef<int> Mask, unsigned NumSrcElts, int &Index) {
272 // Check that the broadcast index meets at least twice.
273 bool IsCompared = false;
274 if (int SplatIdx = PoisonMaskElem;
275 all_of(enumerate(Mask), [&](const auto &P) {
276 if (P.value() == PoisonMaskElem)
277 return P.index() != Mask.size() - 1 || IsCompared;
278 if (static_cast<unsigned>(P.value()) >= NumSrcElts * 2)
279 return false;
280 if (SplatIdx == PoisonMaskElem) {
281 SplatIdx = P.value();
282 return P.index() != Mask.size() - 1;
283 }
284 IsCompared = true;
285 return SplatIdx == P.value();
286 })) {
287 Index = SplatIdx;
288 return true;
289 }
290 return false;
291 }
292
293 /// Several intrinsics that return structs (including llvm.sincos[pi] and
294 /// llvm.modf) can be lowered to a vector library call (for certain VFs). The
295 /// vector library functions correspond to the scalar calls (e.g. sincos or
296 /// modf), which unlike the intrinsic return values via output pointers. This
297 /// helper checks if a vector call exists for the given intrinsic, and returns
298 /// the cost, which includes the cost of the mask (if required), and the loads
299 /// for values returned via output pointers. \p LC is the scalar libcall and
300 /// \p CallRetElementIndex (optional) is the struct element which is mapped to
301 /// the call return value. If std::nullopt is returned, then no vector library
302 /// call is available, so the intrinsic should be assigned the default cost
303 /// (e.g. scalarization).
304 std::optional<InstructionCost> getMultipleResultIntrinsicVectorLibCallCost(
306 std::optional<unsigned> CallRetElementIndex = {}) const {
307 Type *RetTy = ICA.getReturnType();
308 // Vector variants of the intrinsic can be mapped to a vector library call.
309 if (!isa<StructType>(RetTy) ||
311 return std::nullopt;
312
313 Type *Ty = getContainedTypes(RetTy).front();
314 EVT VT = getTLI()->getValueType(DL, Ty);
315
316 RTLIB::Libcall LC = RTLIB::UNKNOWN_LIBCALL;
317
318 switch (ICA.getID()) {
319 case Intrinsic::modf:
320 LC = RTLIB::getMODF(VT);
321 break;
322 case Intrinsic::sincospi:
323 LC = RTLIB::getSINCOSPI(VT);
324 break;
325 case Intrinsic::sincos:
326 LC = RTLIB::getSINCOS(VT);
327 break;
328 default:
329 return std::nullopt;
330 }
331
332 // Find associated libcall.
333 RTLIB::LibcallImpl LibcallImpl = getTLI()->getLibcallImpl(LC);
334 if (LibcallImpl == RTLIB::Unsupported)
335 return std::nullopt;
336
337 LLVMContext &Ctx = RetTy->getContext();
338
339 // Cost the call + mask.
340 auto Cost =
341 thisT()->getCallInstrCost(nullptr, RetTy, ICA.getArgTypes(), CostKind);
342
345 auto VecTy = VectorType::get(IntegerType::getInt1Ty(Ctx), VF);
346 Cost += thisT()->getShuffleCost(TargetTransformInfo::SK_Broadcast, VecTy,
347 VecTy, {}, CostKind, 0, nullptr, {});
348 }
349
350 // Lowering to a library call (with output pointers) may require us to emit
351 // reloads for the results.
352 for (auto [Idx, VectorTy] : enumerate(getContainedTypes(RetTy))) {
353 if (Idx == CallRetElementIndex)
354 continue;
355 Cost += thisT()->getMemoryOpCost(
356 Instruction::Load, VectorTy,
357 thisT()->getDataLayout().getABITypeAlign(VectorTy), 0, CostKind);
358 }
359 return Cost;
360 }
361
362 /// Filter out constant and duplicated entries in \p Ops and return a vector
363 /// containing the types from \p Tys corresponding to the remaining operands.
365 filterConstantAndDuplicatedOperands(ArrayRef<const Value *> Ops,
366 ArrayRef<Type *> Tys) {
367 SmallPtrSet<const Value *, 4> UniqueOperands;
368 SmallVector<Type *, 4> FilteredTys;
369 for (const auto &[Op, Ty] : zip_equal(Ops, Tys)) {
370 if (isa<Constant>(Op) || !UniqueOperands.insert(Op).second)
371 continue;
372 FilteredTys.push_back(Ty);
373 }
374 return FilteredTys;
375 }
376
377protected:
378 explicit BasicTTIImplBase(const TargetMachine *TM, const DataLayout &DL)
379 : BaseT(DL) {}
380 ~BasicTTIImplBase() override = default;
381
384
385public:
386 /// \name Scalar TTI Implementations
387 /// @{
389 unsigned AddressSpace, Align Alignment,
390 unsigned *Fast) const override {
391 EVT E = EVT::getIntegerVT(Context, BitWidth);
392 return getTLI()->allowsMisalignedMemoryAccesses(
394 }
395
396 bool areInlineCompatible(const Function *Caller,
397 const Function *Callee) const override {
398 const TargetMachine &TM = getTLI()->getTargetMachine();
399
400 const TargetSubtargetInfo *CallerSTI = TM.getSubtargetImpl(*Caller);
401 const TargetSubtargetInfo *CalleeSTI = TM.getSubtargetImpl(*Callee);
402 FeatureBitset InlineIgnoreFeatures = CallerSTI->getInlineIgnoreFeatures();
403 FeatureBitset InlineInverseFeatures = CallerSTI->getInlineInverseFeatures();
404 FeatureBitset InlineMustMatchFeatures =
405 CallerSTI->getInlineMustMatchFeatures();
406
407 FeatureBitset CallerBits =
408 (CallerSTI->getFeatureBits() ^ InlineInverseFeatures) &
409 ~InlineIgnoreFeatures;
410 FeatureBitset CalleeBits =
411 (CalleeSTI->getFeatureBits() ^ InlineInverseFeatures) &
412 ~InlineIgnoreFeatures;
413
414 if ((CallerBits & InlineMustMatchFeatures) !=
415 (CalleeBits & InlineMustMatchFeatures))
416 return false;
417
418 // Inline a callee if its target-features are a subset of the callers
419 // target-features.
420 return (CallerBits & CalleeBits) == CalleeBits;
421 }
422
423 bool hasBranchDivergence(const Function *F = nullptr) const override {
424 return false;
425 }
426
427 bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override {
428 return false;
429 }
430
431 bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const override {
432 return true;
433 }
434
435 unsigned getFlatAddressSpace() const override {
436 // Return an invalid address space.
437 return -1;
438 }
439
441 Intrinsic::ID IID) const override {
442 return false;
443 }
444
445 bool isNoopAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override {
446 return getTLI()->getTargetMachine().isNoopAddrSpaceCast(FromAS, ToAS);
447 }
448
449 unsigned getAssumedAddrSpace(const Value *V) const override {
450 return getTLI()->getTargetMachine().getAssumedAddrSpace(V);
451 }
452
453 bool isSingleThreaded() const override {
454 return getTLI()->getTargetMachine().Options.ThreadModel ==
456 }
457
458 std::pair<const Value *, unsigned>
459 getPredicatedAddrSpace(const Value *V) const override {
460 return getTLI()->getTargetMachine().getPredicatedAddrSpace(V);
461 }
462
464 Value *NewV) const override {
465 return nullptr;
466 }
467
468 bool isLegalAddImmediate(int64_t imm) const override {
469 return getTLI()->isLegalAddImmediate(imm);
470 }
471
472 bool isLegalAddScalableImmediate(int64_t Imm) const override {
473 return getTLI()->isLegalAddScalableImmediate(Imm);
474 }
475
476 bool isLegalICmpImmediate(int64_t imm) const override {
477 return getTLI()->isLegalICmpImmediate(imm);
478 }
479
480 bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset,
481 bool HasBaseReg, int64_t Scale, unsigned AddrSpace,
482 Instruction *I = nullptr,
483 int64_t ScalableOffset = 0) const override {
485 AM.BaseGV = BaseGV;
486 AM.BaseOffs = BaseOffset;
487 AM.HasBaseReg = HasBaseReg;
488 AM.Scale = Scale;
489 AM.ScalableOffset = ScalableOffset;
490 return getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace, I);
491 }
492
493 int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset) {
494 return getTLI()->getPreferredLargeGEPBaseOffset(MinOffset, MaxOffset);
495 }
496
497 unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy,
498 Align Alignment,
499 unsigned AddrSpace) const override {
500 auto &&IsSupportedByTarget = [this, ScalarMemTy, ScalarValTy, Alignment,
501 AddrSpace](unsigned VF) {
502 auto *SrcTy = FixedVectorType::get(ScalarMemTy, VF / 2);
503 EVT VT = getTLI()->getValueType(DL, SrcTy);
504 if (getTLI()->isOperationLegal(ISD::STORE, VT) ||
505 getTLI()->isOperationCustom(ISD::STORE, VT))
506 return true;
507
508 EVT ValVT =
509 getTLI()->getValueType(DL, FixedVectorType::get(ScalarValTy, VF / 2));
510 EVT LegalizedVT =
511 getTLI()->getTypeToTransformTo(ScalarMemTy->getContext(), VT);
512 return getTLI()->isTruncStoreLegal(LegalizedVT, ValVT, Alignment,
513 AddrSpace);
514 };
515 while (VF > 2 && IsSupportedByTarget(VF))
516 VF /= 2;
517 return VF;
518 }
519
520 bool isIndexedLoadLegal(TTI::MemIndexedMode M, Type *Ty) const override {
521 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
522 return getTLI()->isIndexedLoadLegal(getISDIndexedMode(M), VT);
523 }
524
525 bool isIndexedStoreLegal(TTI::MemIndexedMode M, Type *Ty) const override {
526 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
527 return getTLI()->isIndexedStoreLegal(getISDIndexedMode(M), VT);
528 }
529
531 const TTI::LSRCost &C2) const override {
533 }
534
538
542
546
548 StackOffset BaseOffset, bool HasBaseReg,
549 int64_t Scale,
550 unsigned AddrSpace) const override {
552 AM.BaseGV = BaseGV;
553 AM.BaseOffs = BaseOffset.getFixed();
554 AM.HasBaseReg = HasBaseReg;
555 AM.Scale = Scale;
556 AM.ScalableOffset = BaseOffset.getScalable();
557 if (getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace))
558 return 0;
560 }
561
562 bool isTruncateFree(Type *Ty1, Type *Ty2) const override {
563 return getTLI()->isTruncateFree(Ty1, Ty2);
564 }
565
566 bool isProfitableToHoist(Instruction *I) const override {
567 return getTLI()->isProfitableToHoist(I);
568 }
569
570 bool useAA() const override { return getST()->useAA(); }
571
572 bool isTypeLegal(Type *Ty) const override {
573 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
574 return getTLI()->isTypeLegal(VT);
575 }
576
577 unsigned getRegUsageForType(Type *Ty) const override {
578 EVT ETy = getTLI()->getValueType(DL, Ty);
579 return getTLI()->getNumRegisters(Ty->getContext(), ETy);
580 }
581
582 InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr,
583 ArrayRef<const Value *> Operands, Type *AccessType,
584 TTI::TargetCostKind CostKind) const override {
585 return BaseT::getGEPCost(PointeeType, Ptr, Operands, AccessType, CostKind);
586 }
587
589 const SwitchInst &SI, unsigned &JumpTableSize, ProfileSummaryInfo *PSI,
590 BlockFrequencyInfo *BFI) const override {
591 /// Try to find the estimated number of clusters. Note that the number of
592 /// clusters identified in this function could be different from the actual
593 /// numbers found in lowering. This function ignore switches that are
594 /// lowered with a mix of jump table / bit test / BTree. This function was
595 /// initially intended to be used when estimating the cost of switch in
596 /// inline cost heuristic, but it's a generic cost model to be used in other
597 /// places (e.g., in loop unrolling).
598 unsigned N = SI.getNumCases();
599 const TargetLoweringBase *TLI = getTLI();
600 const DataLayout &DL = this->getDataLayout();
601
602 JumpTableSize = 0;
603 bool IsJTAllowed = TLI->areJTsAllowed(SI.getParent()->getParent());
604
605 // Early exit if both a jump table and bit test are not allowed.
606 if (N < 1 || (!IsJTAllowed && DL.getIndexSizeInBits(0u) < N))
607 return N;
608
609 APInt MaxCaseVal = SI.case_begin()->getCaseValue()->getValue();
610 APInt MinCaseVal = MaxCaseVal;
611 for (auto CI : SI.cases()) {
612 const APInt &CaseVal = CI.getCaseValue()->getValue();
613 if (CaseVal.sgt(MaxCaseVal))
614 MaxCaseVal = CaseVal;
615 if (CaseVal.slt(MinCaseVal))
616 MinCaseVal = CaseVal;
617 }
618
619 // Check if suitable for a bit test
620 if (N <= DL.getIndexSizeInBits(0u)) {
622 for (auto I : SI.cases()) {
623 const BasicBlock *BB = I.getCaseSuccessor();
624 ++DestMap[BB];
625 }
626
627 if (TLI->isSuitableForBitTests(DestMap, MinCaseVal, MaxCaseVal, DL))
628 return 1;
629 }
630
631 // Check if suitable for a jump table.
632 if (IsJTAllowed) {
633 if (N < 2 || N < TLI->getMinimumJumpTableEntries())
634 return N;
636 (MaxCaseVal - MinCaseVal)
637 .getLimitedValue(std::numeric_limits<uint64_t>::max() - 1) + 1;
638 // Check whether a range of clusters is dense enough for a jump table
639 if (TLI->isSuitableForJumpTable(&SI, N, Range, PSI, BFI)) {
640 JumpTableSize = Range;
641 return 1;
642 }
643 }
644 return N;
645 }
646
647 bool shouldBuildLookupTables() const override {
648 const TargetLoweringBase *TLI = getTLI();
649 return TLI->isOperationLegalOrCustom(ISD::BR_JT, MVT::Other) ||
650 TLI->isOperationLegalOrCustom(ISD::BRIND, MVT::Other);
651 }
652
653 bool shouldBuildRelLookupTables() const override {
654 const TargetMachine &TM = getTLI()->getTargetMachine();
655 // If non-PIC mode, do not generate a relative lookup table.
656 if (!TM.isPositionIndependent())
657 return false;
658
659 /// Relative lookup table entries consist of 32-bit offsets.
660 /// Do not generate relative lookup tables for large code models
661 /// in 64-bit achitectures where 32-bit offsets might not be enough.
662 if (TM.getCodeModel() == CodeModel::Medium ||
664 return false;
665
666 const Triple &TargetTriple = TM.getTargetTriple();
667 if (!TargetTriple.isArch64Bit())
668 return false;
669
670 // TODO: Triggers issues on aarch64 on darwin, so temporarily disable it
671 // there.
672 if (TargetTriple.getArch() == Triple::aarch64 && TargetTriple.isOSDarwin())
673 return false;
674
675 return true;
676 }
677
678 bool haveFastSqrt(Type *Ty) const override {
679 const TargetLoweringBase *TLI = getTLI();
680 EVT VT = TLI->getValueType(DL, Ty);
681 return TLI->isTypeLegal(VT) &&
683 }
684
685 bool haveFastClmul(IntegerType *Ty) const override {
686 // FIXME: clmul should really be Promote for any bitwidth under the largest
687 // legal bitwidth for clmul. Using IndexTy instead of Ty is a hack to get
688 // around that shortcoming.
689 const DataLayout &DL = thisT()->DL;
690 IntegerType *IndexTy =
691 DL.getIndexType(Ty->getContext(), DL.getAllocaAddrSpace());
692 if (Ty->getBitWidth() > IndexTy->getBitWidth())
693 return false;
694
695 const TargetLoweringBase *TLI = getTLI();
696 EVT VT = TLI->getValueType(DL, IndexTy);
697 return TLI->isOperationLegalOrCustom(ISD::CLMUL, VT);
698 }
699
700 bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const override { return true; }
701
702 InstructionCost getFPOpCost(Type *Ty) const override {
703 // Check whether FADD is available, as a proxy for floating-point in
704 // general.
705 const TargetLoweringBase *TLI = getTLI();
706 EVT VT = TLI->getValueType(DL, Ty);
710 }
711
713 const Function &Fn) const override {
714 switch (Inst.getOpcode()) {
715 default:
716 break;
717 case Instruction::SDiv:
718 case Instruction::SRem:
719 case Instruction::UDiv:
720 case Instruction::URem: {
721 if (!isa<ConstantInt>(Inst.getOperand(1)))
722 return false;
723 EVT VT = getTLI()->getValueType(DL, Inst.getType());
724 return !getTLI()->isIntDivCheap(VT, Fn.getAttributes());
725 }
726 };
727
728 return false;
729 }
730
731 unsigned getInliningThresholdMultiplier() const override { return 1; }
732 unsigned adjustInliningThreshold(const CallBase *CB) const override {
733 return 0;
734 }
735 unsigned getCallerAllocaCost(const CallBase *CB,
736 const AllocaInst *AI) const override {
737 return 0;
738 }
739
740 int getInlinerVectorBonusPercent() const override { return 150; }
741
744 OptimizationRemarkEmitter *ORE) const override {
745 // This unrolling functionality is target independent, but to provide some
746 // motivation for its intended use, for x86:
747
748 // According to the Intel 64 and IA-32 Architectures Optimization Reference
749 // Manual, Intel Core models and later have a loop stream detector (and
750 // associated uop queue) that can benefit from partial unrolling.
751 // The relevant requirements are:
752 // - The loop must have no more than 4 (8 for Nehalem and later) branches
753 // taken, and none of them may be calls.
754 // - The loop can have no more than 18 (28 for Nehalem and later) uops.
755
756 // According to the Software Optimization Guide for AMD Family 15h
757 // Processors, models 30h-4fh (Steamroller and later) have a loop predictor
758 // and loop buffer which can benefit from partial unrolling.
759 // The relevant requirements are:
760 // - The loop must have fewer than 16 branches
761 // - The loop must have less than 40 uops in all executed loop branches
762
763 // The number of taken branches in a loop is hard to estimate here, and
764 // benchmarking has revealed that it is better not to be conservative when
765 // estimating the branch count. As a result, we'll ignore the branch limits
766 // until someone finds a case where it matters in practice.
767
768 unsigned MaxOps;
769 const TargetSubtargetInfo *ST = getST();
770 if (PartialUnrollingThreshold.getNumOccurrences() > 0)
772 else if (ST->getSchedModel().LoopMicroOpBufferSize > 0)
773 MaxOps = ST->getSchedModel().LoopMicroOpBufferSize;
774 else
775 return;
776
777 // Scan the loop: don't unroll loops with calls.
778 for (BasicBlock *BB : L->blocks()) {
779 for (Instruction &I : *BB) {
780 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
781 if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
782 if (!thisT()->isLoweredToCall(F))
783 continue;
784 }
785
786 if (ORE) {
787 ORE->emit([&]() {
788 return OptimizationRemark("TTI", "DontUnroll", L->getStartLoc(),
789 L->getHeader())
790 << "advising against unrolling the loop because it "
791 "contains a "
792 << ore::NV("Call", &I);
793 });
794 }
795 return;
796 }
797 }
798 }
799
800 // Enable runtime and partial unrolling up to the specified size.
801 // Enable using trip count upper bound to unroll loops.
802 UP.Partial = UP.Runtime = UP.UpperBound = true;
803 UP.PartialThreshold = MaxOps;
804
805 // Avoid unrolling when optimizing for size.
806 UP.OptSizeThreshold = 0;
808
809 // Set number of instructions optimized when "back edge"
810 // becomes "fall through" to default value of 2.
811 UP.BEInsns = 2;
812 }
813
815 TTI::PeelingPreferences &PP) const override {
816 PP.PeelCount = 0;
817 PP.AllowPeeling = true;
818 PP.AllowLoopNestsPeeling = false;
819 PP.PeelProfiledIterations = true;
820 }
821
824 HardwareLoopInfo &HWLoopInfo) const override {
825 return BaseT::isHardwareLoopProfitable(L, SE, AC, LibInfo, HWLoopInfo);
826 }
827
828 unsigned getEpilogueVectorizationMinVF() const override {
830 }
831
835
839
840 std::optional<Instruction *>
843 }
844
845 std::optional<Value *>
847 APInt DemandedMask, KnownBits &Known,
848 bool &KnownBitsComputed) const override {
849 return BaseT::simplifyDemandedUseBitsIntrinsic(IC, II, DemandedMask, Known,
850 KnownBitsComputed);
851 }
852
854 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts,
855 APInt &UndefElts2, APInt &UndefElts3,
856 std::function<void(Instruction *, unsigned, APInt, APInt &)>
857 SimplifyAndSetOp) const override {
859 IC, II, DemandedElts, UndefElts, UndefElts2, UndefElts3,
860 SimplifyAndSetOp);
861 }
862
864 return getST()->getMispredictionPenalty();
865 }
866
867 std::optional<unsigned>
869 return std::optional<unsigned>(
870 getST()->getCacheSize(static_cast<unsigned>(Level)));
871 }
872
873 std::optional<unsigned>
875 std::optional<unsigned> TargetResult =
876 getST()->getCacheAssociativity(static_cast<unsigned>(Level));
877
878 if (TargetResult)
879 return TargetResult;
880
881 return BaseT::getCacheAssociativity(Level);
882 }
883
884 unsigned getCacheLineSize() const override {
885 return getST()->getCacheLineSize();
886 }
887
888 unsigned getPrefetchDistance() const override {
889 return getST()->getPrefetchDistance();
890 }
891
892 unsigned getMinPrefetchStride(unsigned NumMemAccesses,
893 unsigned NumStridedMemAccesses,
894 unsigned NumPrefetches,
895 bool HasCall) const override {
896 return getST()->getMinPrefetchStride(NumMemAccesses, NumStridedMemAccesses,
897 NumPrefetches, HasCall);
898 }
899
900 unsigned getMaxPrefetchIterationsAhead() const override {
901 return getST()->getMaxPrefetchIterationsAhead();
902 }
903
904 bool enableWritePrefetching() const override {
905 return getST()->enableWritePrefetching();
906 }
907
908 bool shouldPrefetchAddressSpace(unsigned AS) const override {
909 return getST()->shouldPrefetchAddressSpace(AS);
910 }
911
912 /// @}
913
914 /// \name Vector TTI Implementations
915 /// @{
916
921
922 std::optional<unsigned> getMaxVScale() const override { return std::nullopt; }
923 std::optional<unsigned> getVScaleForTuning() const override {
924 return std::nullopt;
925 }
926
927 /// Estimate the overhead of scalarizing an instruction. Insert and Extract
928 /// are set if the demanded result elements need to be inserted and/or
929 /// extracted from vectors.
931 getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts,
932 bool Insert, bool Extract,
934 bool ForPoisonSrc = true, ArrayRef<Value *> VL = {},
936 TTI::VectorInstrContext::None) const override {
937 /// FIXME: a bitfield is not a reasonable abstraction for talking about
938 /// which elements are needed from a scalable vector
939 if (isa<ScalableVectorType>(InTy))
941 auto *Ty = cast<FixedVectorType>(InTy);
942
943 assert(DemandedElts.getBitWidth() == Ty->getNumElements() &&
944 (VL.empty() || VL.size() == Ty->getNumElements()) &&
945 "Vector size mismatch");
946
948
949 for (int i = 0, e = Ty->getNumElements(); i < e; ++i) {
950 if (!DemandedElts[i])
951 continue;
952 if (Insert) {
953 Value *InsertedVal = VL.empty() ? nullptr : VL[i];
954 Cost +=
955 thisT()->getVectorInstrCost(Instruction::InsertElement, Ty,
956 CostKind, i, nullptr, InsertedVal, VIC);
957 }
958 if (Extract)
959 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
960 CostKind, i, nullptr, nullptr, VIC);
961 }
962
963 return Cost;
964 }
965
966 bool
968 unsigned ScalarOpdIdx) const override {
969 return false;
970 }
971
973 int OpdIdx) const override {
974 return OpdIdx == -1;
975 }
976
977 bool
979 int RetIdx) const override {
980 return RetIdx == 0;
981 }
982
983 /// Helper wrapper for the DemandedElts variant of getScalarizationOverhead.
985 VectorType *InTy, bool Insert, bool Extract, TTI::TargetCostKind CostKind,
986 bool ForPoisonSrc = true, ArrayRef<Value *> VL = {},
988 if (isa<ScalableVectorType>(InTy))
990 auto *Ty = cast<FixedVectorType>(InTy);
991
992 APInt DemandedElts = APInt::getAllOnes(Ty->getNumElements());
993 // Use CRTP to allow target overrides
994 return thisT()->getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
995 CostKind, ForPoisonSrc, VL, VIC);
996 }
997
998 /// Estimate the overhead of scalarizing an instruction's
999 /// operands. The (potentially vector) types to use for each of
1000 /// argument are passes via Tys.
1004 TTI::VectorInstrContext::None) const override {
1006 for (Type *Ty : Tys) {
1007 // Disregard things like metadata arguments.
1008 if (!Ty->isIntOrIntVectorTy() && !Ty->isFPOrFPVectorTy() &&
1009 !Ty->isPtrOrPtrVectorTy())
1010 continue;
1011
1012 if (auto *VecTy = dyn_cast<VectorType>(Ty))
1013 Cost += getScalarizationOverhead(VecTy, /*Insert*/ false,
1014 /*Extract*/ true, CostKind,
1015 /*ForPoisonSrc=*/true, {}, VIC);
1016 }
1017
1018 return Cost;
1019 }
1020
1021 /// Estimate the overhead of scalarizing the inputs and outputs of an
1022 /// instruction, with return type RetTy and arguments Args of type Tys. If
1023 /// Args are unknown (empty), then the cost associated with one argument is
1024 /// added as a heuristic.
1027 ArrayRef<Type *> Tys,
1030 RetTy, /*Insert*/ true, /*Extract*/ false, CostKind);
1031 if (!Args.empty())
1033 filterConstantAndDuplicatedOperands(Args, Tys), CostKind);
1034 else
1035 // When no information on arguments is provided, we add the cost
1036 // associated with one argument as a heuristic.
1037 Cost += getScalarizationOverhead(RetTy, /*Insert*/ false,
1038 /*Extract*/ true, CostKind);
1039
1040 return Cost;
1041 }
1042
1043 /// Estimate the cost of type-legalization and the legalized type.
1044 std::pair<InstructionCost, MVT> getTypeLegalizationCost(Type *Ty) const {
1045 LLVMContext &C = Ty->getContext();
1046 EVT MTy = getTLI()->getValueType(DL, Ty);
1047
1049 // We keep legalizing the type until we find a legal kind. We assume that
1050 // the only operation that costs anything is the split. After splitting
1051 // we need to handle two types.
1052 while (true) {
1053 TargetLoweringBase::LegalizeKind LK = getTLI()->getTypeConversion(C, MTy);
1054
1056 // Ensure we return a sensible simple VT here, since many callers of
1057 // this function require it.
1058 MVT VT = MTy.isSimple() ? MTy.getSimpleVT() : MVT::i64;
1059 return std::make_pair(InstructionCost::getInvalid(), VT);
1060 }
1061
1062 if (LK.first == TargetLoweringBase::TypeLegal)
1063 return std::make_pair(Cost, MTy.getSimpleVT());
1064
1065 if (LK.first == TargetLoweringBase::TypeSplitVector ||
1067 Cost *= 2;
1068
1069 // Do not loop with f128 type.
1070 if (MTy == LK.second)
1071 return std::make_pair(Cost, MTy.getSimpleVT());
1072
1073 // Keep legalizing the type.
1074 MTy = LK.second;
1075 }
1076 }
1077
1079 bool HasUnorderedReductions) const override {
1080 return 1;
1081 }
1082
1084 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
1087 ArrayRef<const Value *> Args = {},
1088 const Instruction *CxtI = nullptr) const override {
1089 // Check if any of the operands are vector operands.
1090 const TargetLoweringBase *TLI = getTLI();
1091 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1092 assert(ISD && "Invalid opcode");
1093
1094 // TODO: Handle more cost kinds.
1096 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind,
1097 Opd1Info, Opd2Info,
1098 Args, CxtI);
1099
1100 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1101
1102 bool IsFloat = Ty->isFPOrFPVectorTy();
1103 // Assume that floating point arithmetic operations cost twice as much as
1104 // integer operations.
1105 InstructionCost OpCost = (IsFloat ? 2 : 1);
1106
1107 if (TLI->isOperationLegalOrPromote(ISD, LT.second)) {
1108 // The operation is legal. Assume it costs 1.
1109 // TODO: Once we have extract/insert subvector cost we need to use them.
1110 return LT.first * OpCost;
1111 }
1112
1113 if (!TLI->isOperationExpand(ISD, LT.second)) {
1114 // If the operation is custom lowered, then assume that the code is twice
1115 // as expensive.
1116 return LT.first * 2 * OpCost;
1117 }
1118
1119 // An 'Expand' of URem and SRem is special because it may default
1120 // to expanding the operation into a sequence of sub-operations
1121 // i.e. X % Y -> X-(X/Y)*Y.
1122 if (ISD == ISD::UREM || ISD == ISD::SREM) {
1123 bool IsSigned = ISD == ISD::SREM;
1124 if (TLI->isOperationLegalOrCustom(IsSigned ? ISD::SDIVREM : ISD::UDIVREM,
1125 LT.second) ||
1126 TLI->isOperationLegalOrCustom(IsSigned ? ISD::SDIV : ISD::UDIV,
1127 LT.second)) {
1128 unsigned DivOpc = IsSigned ? Instruction::SDiv : Instruction::UDiv;
1129 InstructionCost DivCost = thisT()->getArithmeticInstrCost(
1130 DivOpc, Ty, CostKind, Opd1Info, Opd2Info);
1131 InstructionCost MulCost =
1132 thisT()->getArithmeticInstrCost(Instruction::Mul, Ty, CostKind);
1133 InstructionCost SubCost =
1134 thisT()->getArithmeticInstrCost(Instruction::Sub, Ty, CostKind);
1135 return DivCost + MulCost + SubCost;
1136 }
1137 }
1138
1139 // We cannot scalarize scalable vectors, so return Invalid.
1142
1143 // Else, assume that we need to scalarize this op.
1144 // TODO: If one of the types get legalized by splitting, handle this
1145 // similarly to what getCastInstrCost() does.
1146 if (auto *VTy = dyn_cast<FixedVectorType>(Ty)) {
1147 InstructionCost Cost = thisT()->getArithmeticInstrCost(
1148 Opcode, VTy->getScalarType(), CostKind, Opd1Info, Opd2Info,
1149 Args, CxtI);
1150 // Return the cost of multiple scalar invocation plus the cost of
1151 // inserting and extracting the values.
1152 SmallVector<Type *> Tys(Args.size(), Ty);
1153 return getScalarizationOverhead(VTy, Args, Tys, CostKind) +
1154 VTy->getNumElements() * Cost;
1155 }
1156
1157 // We don't know anything about this scalar instruction.
1158 return OpCost;
1159 }
1160
1162 ArrayRef<int> Mask,
1163 VectorType *SrcTy, int &Index,
1164 VectorType *&SubTy) const {
1165 if (Mask.empty())
1166 return Kind;
1167 int NumDstElts = Mask.size();
1168 int NumSrcElts = SrcTy->getElementCount().getKnownMinValue();
1169 switch (Kind) {
1171 if (ShuffleVectorInst::isReverseMask(Mask, NumSrcElts))
1172 return TTI::SK_Reverse;
1173 if (ShuffleVectorInst::isZeroEltSplatMask(Mask, NumSrcElts))
1174 return TTI::SK_Broadcast;
1175 if (isSplatMask(Mask, NumSrcElts, Index))
1176 return TTI::SK_Broadcast;
1177 if (ShuffleVectorInst::isExtractSubvectorMask(Mask, NumSrcElts, Index) &&
1178 (Index + NumDstElts) <= NumSrcElts) {
1179 SubTy = FixedVectorType::get(SrcTy->getElementType(), NumDstElts);
1181 }
1182 break;
1183 }
1184 case TTI::SK_PermuteTwoSrc: {
1185 if (all_of(Mask, [NumSrcElts](int M) { return M < NumSrcElts; }))
1187 Index, SubTy);
1188 int NumSubElts;
1189 if (NumDstElts > 2 && ShuffleVectorInst::isInsertSubvectorMask(
1190 Mask, NumSrcElts, NumSubElts, Index)) {
1191 if (Index + NumSubElts > NumSrcElts)
1192 return Kind;
1193 SubTy = FixedVectorType::get(SrcTy->getElementType(), NumSubElts);
1195 }
1196 if (ShuffleVectorInst::isSelectMask(Mask, NumSrcElts))
1197 return TTI::SK_Select;
1198 if (ShuffleVectorInst::isTransposeMask(Mask, NumSrcElts))
1199 return TTI::SK_Transpose;
1200 if (ShuffleVectorInst::isSpliceMask(Mask, NumSrcElts, Index))
1201 return TTI::SK_Splice;
1202 break;
1203 }
1204 case TTI::SK_Select:
1205 case TTI::SK_Reverse:
1206 case TTI::SK_Broadcast:
1207 case TTI::SK_Transpose:
1210 case TTI::SK_Splice:
1211 break;
1212 }
1213 return Kind;
1214 }
1215
1219 VectorType *SubTp, ArrayRef<const Value *> Args = {},
1220 const Instruction *CxtI = nullptr) const override {
1221 switch (improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp)) {
1222 case TTI::SK_Broadcast:
1223 if (auto *FVT = dyn_cast<FixedVectorType>(SrcTy))
1224 return getBroadcastShuffleOverhead(FVT, CostKind);
1226 case TTI::SK_Select:
1227 case TTI::SK_Splice:
1228 case TTI::SK_Reverse:
1229 case TTI::SK_Transpose:
1232 if (auto *FVT = dyn_cast<FixedVectorType>(SrcTy))
1233 return getPermuteShuffleOverhead(FVT, CostKind);
1236 return getExtractSubvectorOverhead(SrcTy, CostKind, Index,
1237 cast<FixedVectorType>(SubTp));
1239 return getInsertSubvectorOverhead(DstTy, CostKind, Index,
1240 cast<FixedVectorType>(SubTp));
1241 }
1242 llvm_unreachable("Unknown TTI::ShuffleKind");
1243 }
1244
1246 getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src,
1248 const Instruction *I = nullptr) const override {
1249 if (BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I) == 0)
1250 return 0;
1251
1252 const TargetLoweringBase *TLI = getTLI();
1253 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1254 assert(ISD && "Invalid opcode");
1255 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
1256 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
1257
1258 TypeSize SrcSize = SrcLT.second.getSizeInBits();
1259 TypeSize DstSize = DstLT.second.getSizeInBits();
1260 bool IntOrPtrSrc = Src->isIntegerTy() || Src->isPointerTy();
1261 bool IntOrPtrDst = Dst->isIntegerTy() || Dst->isPointerTy();
1262
1263 switch (Opcode) {
1264 default:
1265 break;
1266 case Instruction::Trunc:
1267 // Check for NOOP conversions.
1268 if (TLI->isTruncateFree(SrcLT.second, DstLT.second))
1269 return 0;
1270 [[fallthrough]];
1271 case Instruction::BitCast:
1272 // Bitcast between types that are legalized to the same type are free and
1273 // assume int to/from ptr of the same size is also free.
1274 if (SrcLT.first == DstLT.first && IntOrPtrSrc == IntOrPtrDst &&
1275 SrcSize == DstSize)
1276 return 0;
1277 break;
1278 case Instruction::FPExt:
1279 if (I && getTLI()->isExtFree(I))
1280 return 0;
1281 break;
1282 case Instruction::ZExt:
1283 if (TLI->isZExtFree(SrcLT.second, DstLT.second))
1284 return 0;
1285 [[fallthrough]];
1286 case Instruction::SExt:
1287 if (I && getTLI()->isExtFree(I))
1288 return 0;
1289
1290 // If this is a zext/sext of a load, return 0 if the corresponding
1291 // extending load exists on target and the result type is legal.
1292 if (CCH == TTI::CastContextHint::Normal) {
1293 EVT ExtVT = EVT::getEVT(Dst);
1294 EVT LoadVT = EVT::getEVT(Src);
1295 unsigned LType =
1296 Opcode == Instruction::ZExt ? ISD::ZEXTLOAD : ISD::SEXTLOAD;
1297 if (I) {
1298 if (auto *LI = dyn_cast<LoadInst>(I->getOperand(0))) {
1299 if (DstLT.first == SrcLT.first &&
1300 TLI->isLoadLegal(ExtVT, LoadVT, LI->getAlign(),
1301 LI->getPointerAddressSpace(), LType, false))
1302 return 0;
1303 } else if (auto *II = dyn_cast<IntrinsicInst>(I->getOperand(0))) {
1304 switch (II->getIntrinsicID()) {
1305 case Intrinsic::masked_load: {
1306 Type *PtrType = II->getArgOperand(0)->getType();
1307 assert(PtrType->isPointerTy());
1308
1309 if (DstLT.first == SrcLT.first &&
1310 TLI->isLoadLegal(
1311 ExtVT, LoadVT, II->getParamAlign(0).valueOrOne(),
1312 PtrType->getPointerAddressSpace(), LType, false))
1313 return 0;
1314
1315 break;
1316 }
1317 default:
1318 break;
1319 }
1320 }
1321 }
1322 }
1323 break;
1324 case Instruction::AddrSpaceCast:
1325 if (TLI->isFreeAddrSpaceCast(Src->getPointerAddressSpace(),
1326 Dst->getPointerAddressSpace()))
1327 return 0;
1328 break;
1329 }
1330
1331 auto *SrcVTy = dyn_cast<VectorType>(Src);
1332 auto *DstVTy = dyn_cast<VectorType>(Dst);
1333
1334 // If the cast is marked as legal (or promote) then assume low cost.
1335 if (SrcLT.first == DstLT.first &&
1336 TLI->isOperationLegalOrPromote(ISD, DstLT.second))
1337 return SrcLT.first;
1338
1339 // Handle scalar conversions.
1340 if (!SrcVTy && !DstVTy) {
1341 // Just check the op cost. If the operation is legal then assume it costs
1342 // 1.
1343 if (!TLI->isOperationExpand(ISD, DstLT.second))
1344 return 1;
1345
1346 // Assume that illegal scalar instruction are expensive.
1347 return 4;
1348 }
1349
1350 // Check vector-to-vector casts.
1351 if (DstVTy && SrcVTy) {
1352 // If the cast is between same-sized registers, then the check is simple.
1353 if (SrcLT.first == DstLT.first && SrcSize == DstSize) {
1354
1355 // Assume that Zext is done using AND.
1356 if (Opcode == Instruction::ZExt)
1357 return SrcLT.first;
1358
1359 // Assume that sext is done using SHL and SRA.
1360 if (Opcode == Instruction::SExt)
1361 return SrcLT.first * 2;
1362
1363 // Just check the op cost. If the operation is legal then assume it
1364 // costs
1365 // 1 and multiply by the type-legalization overhead.
1366 if (!TLI->isOperationExpand(ISD, DstLT.second))
1367 return SrcLT.first * 1;
1368 }
1369
1370 // If we are legalizing by splitting, query the concrete TTI for the cost
1371 // of casting the original vector twice. We also need to factor in the
1372 // cost of the split itself. Count that as 1, to be consistent with
1373 // getTypeLegalizationCost().
1374 bool SplitSrc =
1375 TLI->getTypeAction(Src->getContext(), TLI->getValueType(DL, Src)) ==
1377 bool SplitDst =
1378 TLI->getTypeAction(Dst->getContext(), TLI->getValueType(DL, Dst)) ==
1380 if ((SplitSrc || SplitDst) && SrcVTy->getElementCount().isKnownEven() &&
1381 DstVTy->getElementCount().isKnownEven()) {
1382 Type *SplitDstTy = VectorType::getHalfElementsVectorType(DstVTy);
1383 Type *SplitSrcTy = VectorType::getHalfElementsVectorType(SrcVTy);
1384 const T *TTI = thisT();
1385 // If both types need to be split then the split is free.
1386 InstructionCost SplitCost =
1387 (!SplitSrc || !SplitDst) ? TTI->getVectorSplitCost() : 0;
1388 return SplitCost +
1389 (2 * TTI->getCastInstrCost(Opcode, SplitDstTy, SplitSrcTy, CCH,
1390 CostKind, I));
1391 }
1392
1393 // Scalarization cost is Invalid, can't assume any num elements.
1394 if (isa<ScalableVectorType>(DstVTy))
1396
1397 // In other cases where the source or destination are illegal, assume
1398 // the operation will get scalarized.
1399 unsigned Num = cast<FixedVectorType>(DstVTy)->getNumElements();
1400 InstructionCost Cost = thisT()->getCastInstrCost(
1401 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH, CostKind, I);
1402
1403 // Return the cost of multiple scalar invocation plus the cost of
1404 // inserting and extracting the values.
1405 return getScalarizationOverhead(DstVTy, /*Insert*/ true, /*Extract*/ true,
1406 CostKind) +
1407 Num * Cost;
1408 }
1409
1410 // We already handled vector-to-vector and scalar-to-scalar conversions.
1411 // This
1412 // is where we handle bitcast between vectors and scalars. We need to assume
1413 // that the conversion is scalarized in one way or another.
1414 if (Opcode == Instruction::BitCast) {
1415 // Illegal bitcasts are done by storing and loading from a stack slot.
1416 return (SrcVTy ? getScalarizationOverhead(SrcVTy, /*Insert*/ false,
1417 /*Extract*/ true, CostKind)
1418 : 0) +
1419 (DstVTy ? getScalarizationOverhead(DstVTy, /*Insert*/ true,
1420 /*Extract*/ false, CostKind)
1421 : 0);
1422 }
1423
1424 llvm_unreachable("Unhandled cast");
1425 }
1426
1428 getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy,
1429 unsigned Index,
1430 TTI::TargetCostKind CostKind) const override {
1431 return thisT()->getVectorInstrCost(Instruction::ExtractElement, VecTy,
1432 CostKind, Index, nullptr, nullptr) +
1433 thisT()->getCastInstrCost(Opcode, Dst, VecTy->getElementType(),
1435 }
1436
1439 const Instruction *I = nullptr) const override {
1440 return BaseT::getCFInstrCost(Opcode, CostKind, I);
1441 }
1442
1444 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
1448 const Instruction *I = nullptr) const override {
1449 const TargetLoweringBase *TLI = getTLI();
1450 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1451 assert(ISD && "Invalid opcode");
1452
1453 if (getTLI()->getValueType(DL, ValTy, true) == MVT::Other)
1454 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
1455 Op1Info, Op2Info, I);
1456
1457 // Selects on vectors are actually vector selects.
1458 if (ISD == ISD::SELECT) {
1459 assert(CondTy && "CondTy must exist");
1460 if (CondTy->isVectorTy())
1461 ISD = ISD::VSELECT;
1462 }
1463 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
1464
1465 if (!(ValTy->isVectorTy() && !LT.second.isVector()) &&
1466 !TLI->isOperationExpand(ISD, LT.second)) {
1467 // The operation is legal. Assume it costs 1. Multiply
1468 // by the type-legalization overhead.
1469 return LT.first * 1;
1470 }
1471
1472 // Otherwise, assume that the cast is scalarized.
1473 // TODO: If one of the types get legalized by splitting, handle this
1474 // similarly to what getCastInstrCost() does.
1475 if (auto *ValVTy = dyn_cast<VectorType>(ValTy)) {
1476 if (isa<ScalableVectorType>(ValTy))
1478
1479 unsigned Num = cast<FixedVectorType>(ValVTy)->getNumElements();
1480 InstructionCost Cost = thisT()->getCmpSelInstrCost(
1481 Opcode, ValVTy->getScalarType(), CondTy->getScalarType(), VecPred,
1482 CostKind, Op1Info, Op2Info, I);
1483
1484 // Return the cost of multiple scalar invocation plus the cost of
1485 // inserting and extracting the values.
1486 return getScalarizationOverhead(ValVTy, /*Insert*/ true,
1487 /*Extract*/ false, CostKind) +
1488 Num * Cost;
1489 }
1490
1491 // Unknown scalar opcode.
1492 return 1;
1493 }
1494
1497 unsigned Index, const Value *Op0, const Value *Op1,
1499 TTI::VectorInstrContext::None) const override {
1500 return getRegUsageForType(Val->getScalarType());
1501 }
1502
1503 /// \param ScalarUserAndIdx encodes the information about extracts from a
1504 /// vector with 'Scalar' being the value being extracted,'User' being the user
1505 /// of the extract(nullptr if user is not known before vectorization) and
1506 /// 'Idx' being the extract lane.
1508 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
1509 Value *Scalar,
1510 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
1512 TTI::VectorInstrContext::None) const override {
1513 return getVectorInstrCost(Opcode, Val, CostKind, Index, nullptr, nullptr,
1514 VIC);
1515 }
1516
1519 TTI::TargetCostKind CostKind, unsigned Index,
1521 TTI::VectorInstrContext::None) const override {
1522 Value *Op0 = nullptr;
1523 Value *Op1 = nullptr;
1524 if (auto *IE = dyn_cast<InsertElementInst>(&I)) {
1525 Op0 = IE->getOperand(0);
1526 Op1 = IE->getOperand(1);
1527 }
1528 // If VIC is None, compute it from the instruction
1531 return thisT()->getVectorInstrCost(I.getOpcode(), Val, CostKind, Index, Op0,
1532 Op1, VIC);
1533 }
1534
1538 unsigned Index) const override {
1539 unsigned NewIndex = -1;
1540 if (auto *FVTy = dyn_cast<FixedVectorType>(Val)) {
1541 assert(Index < FVTy->getNumElements() &&
1542 "Unexpected index from end of vector");
1543 NewIndex = FVTy->getNumElements() - 1 - Index;
1544 }
1545 return thisT()->getVectorInstrCost(Opcode, Val, CostKind, NewIndex, nullptr,
1546 nullptr);
1547 }
1548
1550 getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF,
1551 const APInt &DemandedDstElts,
1552 TTI::TargetCostKind CostKind) const override {
1553 assert(DemandedDstElts.getBitWidth() == (unsigned)VF * ReplicationFactor &&
1554 "Unexpected size of DemandedDstElts.");
1555
1557
1558 auto *SrcVT = FixedVectorType::get(EltTy, VF);
1559 auto *ReplicatedVT = FixedVectorType::get(EltTy, VF * ReplicationFactor);
1560
1561 // The Mask shuffling cost is extract all the elements of the Mask
1562 // and insert each of them Factor times into the wide vector:
1563 //
1564 // E.g. an interleaved group with factor 3:
1565 // %mask = icmp ult <8 x i32> %vec1, %vec2
1566 // %interleaved.mask = shufflevector <8 x i1> %mask, <8 x i1> undef,
1567 // <24 x i32> <0,0,0,1,1,1,2,2,2,3,3,3,4,4,4,5,5,5,6,6,6,7,7,7>
1568 // The cost is estimated as extract all mask elements from the <8xi1> mask
1569 // vector and insert them factor times into the <24xi1> shuffled mask
1570 // vector.
1571 APInt DemandedSrcElts = APIntOps::ScaleBitMask(DemandedDstElts, VF);
1572 Cost += thisT()->getScalarizationOverhead(SrcVT, DemandedSrcElts,
1573 /*Insert*/ false,
1574 /*Extract*/ true, CostKind);
1575 Cost += thisT()->getScalarizationOverhead(ReplicatedVT, DemandedDstElts,
1576 /*Insert*/ true,
1577 /*Extract*/ false, CostKind);
1578
1579 return Cost;
1580 }
1581
1583 unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
1586 const Instruction *I = nullptr) const override {
1587 assert(!Src->isVoidTy() && "Invalid type");
1588 // Assume types, such as structs, are expensive.
1589 if (getTLI()->getValueType(DL, Src, true) == MVT::Other)
1590 return 4;
1591 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
1592
1593 // FIXME: Arbitrary cost
1594 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
1595 return 4;
1596
1597 // Assuming that all loads of legal types cost 1.
1598 InstructionCost Cost = LT.first;
1600 return Cost;
1601
1602 const DataLayout &DL = this->getDataLayout();
1603 if (Src->isVectorTy() &&
1604 // In practice it's not currently possible to have a change in lane
1605 // length for extending loads or truncating stores so both types should
1606 // have the same scalable property.
1607 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
1608 LT.second.getSizeInBits())) {
1609 // This is a vector load that legalizes to a larger type than the vector
1610 // itself. Unless the corresponding extending load or truncating store is
1611 // legal, then this will scalarize.
1613 EVT MemVT = getTLI()->getValueType(DL, Src);
1614 if (Opcode == Instruction::Store)
1615 LA = getTLI()->getTruncStoreAction(LT.second, MemVT, Alignment,
1616 AddressSpace);
1617 else
1618 LA = getTLI()->getLoadAction(LT.second, MemVT, Alignment, AddressSpace,
1619 ISD::EXTLOAD, false);
1620
1621 if (LA != TargetLowering::Legal && LA != TargetLowering::Custom) {
1622 // This is a vector load/store for some illegal type that is scalarized.
1623 // We must account for the cost of building or decomposing the vector.
1625 cast<VectorType>(Src), Opcode != Instruction::Store,
1626 Opcode == Instruction::Store, CostKind);
1627 }
1628 }
1629
1630 return Cost;
1631 }
1632
1634 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1635 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1636 bool UseMaskForCond = false, bool UseMaskForGaps = false) const override {
1637
1638 // We cannot scalarize scalable vectors, so return Invalid.
1639 if (isa<ScalableVectorType>(VecTy))
1641
1642 auto *VT = cast<FixedVectorType>(VecTy);
1643
1644 unsigned NumElts = VT->getNumElements();
1645 assert(Factor > 1 && NumElts % Factor == 0 && "Invalid interleave factor");
1646
1647 unsigned NumSubElts = NumElts / Factor;
1648 auto *SubVT = FixedVectorType::get(VT->getElementType(), NumSubElts);
1649
1650 // Firstly, the cost of load/store operation.
1652 if (UseMaskForCond || UseMaskForGaps) {
1653 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
1654 : Intrinsic::masked_store;
1655 Cost = thisT()->getMemIntrinsicInstrCost(
1656 MemIntrinsicCostAttributes(IID, VecTy, Alignment, AddressSpace),
1657 CostKind);
1658 } else
1659 Cost = thisT()->getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace,
1660 CostKind);
1661
1662 // Legalize the vector type, and get the legalized and unlegalized type
1663 // sizes.
1664 MVT VecTyLT = getTypeLegalizationCost(VecTy).second;
1665 unsigned VecTySize = thisT()->getDataLayout().getTypeStoreSize(VecTy);
1666 unsigned VecTyLTSize = VecTyLT.getStoreSize();
1667
1668 // Scale the cost of the memory operation by the fraction of legalized
1669 // instructions that will actually be used. We shouldn't account for the
1670 // cost of dead instructions since they will be removed.
1671 //
1672 // E.g., An interleaved load of factor 8:
1673 // %vec = load <16 x i64>, <16 x i64>* %ptr
1674 // %v0 = shufflevector %vec, undef, <0, 8>
1675 //
1676 // If <16 x i64> is legalized to 8 v2i64 loads, only 2 of the loads will be
1677 // used (those corresponding to elements [0:1] and [8:9] of the unlegalized
1678 // type). The other loads are unused.
1679 //
1680 // TODO: Note that legalization can turn masked loads/stores into unmasked
1681 // (legalized) loads/stores. This can be reflected in the cost.
1682 if (Cost.isValid() && VecTySize > VecTyLTSize) {
1683 // The number of loads of a legal type it will take to represent a load
1684 // of the unlegalized vector type.
1685 unsigned NumLegalInsts = divideCeil(VecTySize, VecTyLTSize);
1686
1687 // The number of elements of the unlegalized type that correspond to a
1688 // single legal instruction.
1689 unsigned NumEltsPerLegalInst = divideCeil(NumElts, NumLegalInsts);
1690
1691 // Determine which legal instructions will be used.
1692 BitVector UsedInsts(NumLegalInsts, false);
1693 for (unsigned Index : Indices)
1694 for (unsigned Elt = 0; Elt < NumSubElts; ++Elt)
1695 UsedInsts.set((Index + Elt * Factor) / NumEltsPerLegalInst);
1696
1697 // Scale the cost of the load by the fraction of legal instructions that
1698 // will be used.
1699 Cost = divideCeil(UsedInsts.count() * Cost.getValue(), NumLegalInsts);
1700 }
1701
1702 // Then plus the cost of interleave operation.
1703 assert(Indices.size() <= Factor &&
1704 "Interleaved memory op has too many members");
1705
1706 const APInt DemandedAllSubElts = APInt::getAllOnes(NumSubElts);
1707 const APInt DemandedAllResultElts = APInt::getAllOnes(NumElts);
1708
1709 APInt DemandedLoadStoreElts = APInt::getZero(NumElts);
1710 for (unsigned Index : Indices) {
1711 assert(Index < Factor && "Invalid index for interleaved memory op");
1712 for (unsigned Elm = 0; Elm < NumSubElts; Elm++)
1713 DemandedLoadStoreElts.setBit(Index + Elm * Factor);
1714 }
1715
1716 if (Opcode == Instruction::Load) {
1717 // The interleave cost is similar to extract sub vectors' elements
1718 // from the wide vector, and insert them into sub vectors.
1719 //
1720 // E.g. An interleaved load of factor 2 (with one member of index 0):
1721 // %vec = load <8 x i32>, <8 x i32>* %ptr
1722 // %v0 = shuffle %vec, undef, <0, 2, 4, 6> ; Index 0
1723 // The cost is estimated as extract elements at 0, 2, 4, 6 from the
1724 // <8 x i32> vector and insert them into a <4 x i32> vector.
1725 InstructionCost InsSubCost = thisT()->getScalarizationOverhead(
1726 SubVT, DemandedAllSubElts,
1727 /*Insert*/ true, /*Extract*/ false, CostKind);
1728 Cost += Indices.size() * InsSubCost;
1729 Cost += thisT()->getScalarizationOverhead(VT, DemandedLoadStoreElts,
1730 /*Insert*/ false,
1731 /*Extract*/ true, CostKind);
1732 } else {
1733 // The interleave cost is extract elements from sub vectors, and
1734 // insert them into the wide vector.
1735 //
1736 // E.g. An interleaved store of factor 3 with 2 members at indices 0,1:
1737 // (using VF=4):
1738 // %v0_v1 = shuffle %v0, %v1, <0,4,undef,1,5,undef,2,6,undef,3,7,undef>
1739 // %gaps.mask = <true, true, false, true, true, false,
1740 // true, true, false, true, true, false>
1741 // call llvm.masked.store <12 x i32> %v0_v1, <12 x i32>* %ptr,
1742 // i32 Align, <12 x i1> %gaps.mask
1743 // The cost is estimated as extract all elements (of actual members,
1744 // excluding gaps) from both <4 x i32> vectors and insert into the <12 x
1745 // i32> vector.
1746 InstructionCost ExtSubCost = thisT()->getScalarizationOverhead(
1747 SubVT, DemandedAllSubElts,
1748 /*Insert*/ false, /*Extract*/ true, CostKind);
1749 Cost += ExtSubCost * Indices.size();
1750 Cost += thisT()->getScalarizationOverhead(VT, DemandedLoadStoreElts,
1751 /*Insert*/ true,
1752 /*Extract*/ false, CostKind);
1753 }
1754
1755 if (!UseMaskForCond)
1756 return Cost;
1757
1758 Type *I8Type = Type::getInt8Ty(VT->getContext());
1759
1760 Cost += thisT()->getReplicationShuffleCost(
1761 I8Type, Factor, NumSubElts,
1762 UseMaskForGaps ? DemandedLoadStoreElts : DemandedAllResultElts,
1763 CostKind);
1764
1765 // The Gaps mask is invariant and created outside the loop, therefore the
1766 // cost of creating it is not accounted for here. However if we have both
1767 // a MaskForGaps and some other mask that guards the execution of the
1768 // memory access, we need to account for the cost of And-ing the two masks
1769 // inside the loop.
1770 if (UseMaskForGaps) {
1771 auto *MaskVT = FixedVectorType::get(I8Type, NumElts);
1772 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::And, MaskVT,
1773 CostKind);
1774 }
1775
1776 return Cost;
1777 }
1778
1779 /// Get intrinsic cost based on arguments.
1782 TTI::TargetCostKind CostKind) const override {
1783 // Check for generically free intrinsics.
1785 return 0;
1786
1787 // Assume that target intrinsics are cheap.
1788 Intrinsic::ID IID = ICA.getID();
1791
1792 // VP Intrinsics should have the same cost as their non-vp counterpart.
1793 // TODO: Adjust the cost to make the vp intrinsic cheaper than its non-vp
1794 // counterpart when the vector length argument is smaller than the maximum
1795 // vector length.
1796 // TODO: Support other kinds of VPIntrinsics
1797 if (VPIntrinsic::isVPIntrinsic(ICA.getID())) {
1798 std::optional<unsigned> FOp =
1800 if (FOp) {
1801 if (ICA.getID() == Intrinsic::vp_load) {
1802 Align Alignment;
1803 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1804 Alignment = VPI->getPointerAlignment().valueOrOne();
1805 unsigned AS = 0;
1806 if (ICA.getArgTypes().size() > 1)
1807 if (auto *PtrTy = dyn_cast<PointerType>(ICA.getArgTypes()[0]))
1808 AS = PtrTy->getAddressSpace();
1809 return thisT()->getMemoryOpCost(*FOp, ICA.getReturnType(), Alignment,
1810 AS, CostKind);
1811 }
1812 if (ICA.getID() == Intrinsic::vp_store) {
1813 Align Alignment;
1814 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1815 Alignment = VPI->getPointerAlignment().valueOrOne();
1816 unsigned AS = 0;
1817 if (ICA.getArgTypes().size() >= 2)
1818 if (auto *PtrTy = dyn_cast<PointerType>(ICA.getArgTypes()[1]))
1819 AS = PtrTy->getAddressSpace();
1820 return thisT()->getMemoryOpCost(*FOp, ICA.getArgTypes()[0], Alignment,
1821 AS, CostKind);
1822 }
1824 ICA.getID() == Intrinsic::vp_fneg) {
1825 return thisT()->getArithmeticInstrCost(*FOp, ICA.getReturnType(),
1826 CostKind);
1827 }
1828 if (VPCastIntrinsic::isVPCast(ICA.getID())) {
1829 return thisT()->getCastInstrCost(
1830 *FOp, ICA.getReturnType(), ICA.getArgTypes()[0],
1832 }
1833 if (VPCmpIntrinsic::isVPCmp(ICA.getID())) {
1834 // We can only handle vp_cmp intrinsics with underlying instructions.
1835 if (ICA.getInst()) {
1836 assert(FOp);
1837 auto *UI = cast<VPCmpIntrinsic>(ICA.getInst());
1838 return thisT()->getCmpSelInstrCost(*FOp, ICA.getArgTypes()[0],
1839 ICA.getReturnType(),
1840 UI->getPredicate(), CostKind);
1841 }
1842 }
1843 }
1844 if (ICA.getID() == Intrinsic::vp_load_ff) {
1845 Type *RetTy = ICA.getReturnType();
1846 Type *DataTy = cast<StructType>(RetTy)->getElementType(0);
1847 Align Alignment;
1848 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1849 Alignment = VPI->getPointerAlignment().valueOrOne();
1850 return thisT()->getMemIntrinsicInstrCost(
1851 MemIntrinsicCostAttributes(ICA.getID(), DataTy, Alignment),
1852 CostKind);
1853 }
1854 if (ICA.getID() == Intrinsic::vp_scatter) {
1855 if (ICA.isTypeBasedOnly()) {
1856 IntrinsicCostAttributes MaskedScatter(
1859 ICA.getFlags());
1860 return getTypeBasedIntrinsicInstrCost(MaskedScatter, CostKind);
1861 }
1862 Align Alignment;
1863 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1864 Alignment = VPI->getPointerAlignment().valueOrOne();
1865 bool VarMask = isa<Constant>(ICA.getArgs()[2]);
1866 return thisT()->getMemIntrinsicInstrCost(
1867 MemIntrinsicCostAttributes(Intrinsic::vp_scatter,
1868 ICA.getArgTypes()[0], ICA.getArgs()[1],
1869 VarMask, Alignment, nullptr),
1870 CostKind);
1871 }
1872 if (ICA.getID() == Intrinsic::vp_gather) {
1873 if (ICA.isTypeBasedOnly()) {
1874 IntrinsicCostAttributes MaskedGather(
1877 ICA.getFlags());
1878 return getTypeBasedIntrinsicInstrCost(MaskedGather, CostKind);
1879 }
1880 Align Alignment;
1881 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1882 Alignment = VPI->getPointerAlignment().valueOrOne();
1883 bool VarMask = isa<Constant>(ICA.getArgs()[1]);
1884 return thisT()->getMemIntrinsicInstrCost(
1885 MemIntrinsicCostAttributes(Intrinsic::vp_gather,
1886 ICA.getReturnType(), ICA.getArgs()[0],
1887 VarMask, Alignment, nullptr),
1888 CostKind);
1889 }
1890
1891 if (ICA.getID() == Intrinsic::vp_select ||
1892 ICA.getID() == Intrinsic::vp_merge) {
1893 TTI::OperandValueInfo OpInfoX, OpInfoY;
1894 if (!ICA.isTypeBasedOnly()) {
1895 OpInfoX = TTI::getOperandInfo(ICA.getArgs()[0]);
1896 OpInfoY = TTI::getOperandInfo(ICA.getArgs()[1]);
1897 }
1898 return getCmpSelInstrCost(
1899 Instruction::Select, ICA.getReturnType(), ICA.getArgTypes()[0],
1900 CmpInst::BAD_ICMP_PREDICATE, CostKind, OpInfoX, OpInfoY);
1901 }
1902
1903 std::optional<Intrinsic::ID> FID =
1905
1906 // Not functionally equivalent but close enough for cost modelling.
1907 if (ICA.getID() == Intrinsic::experimental_vp_reverse)
1908 FID = Intrinsic::vector_reverse;
1909
1910 if (FID) {
1911 // Non-vp version will have same arg types except mask and vector
1912 // length.
1913 assert(ICA.getArgTypes().size() >= 2 &&
1914 "Expected VPIntrinsic to have Mask and Vector Length args and "
1915 "types");
1916
1917 ArrayRef<const Value *> NewArgs = ArrayRef(ICA.getArgs());
1918 if (!ICA.isTypeBasedOnly())
1919 NewArgs = NewArgs.drop_back(2);
1921
1922 // VPReduction intrinsics have a start value argument that their non-vp
1923 // counterparts do not have, except for the fadd and fmul non-vp
1924 // counterpart.
1926 *FID != Intrinsic::vector_reduce_fadd &&
1927 *FID != Intrinsic::vector_reduce_fmul) {
1928 if (!ICA.isTypeBasedOnly())
1929 NewArgs = NewArgs.drop_front();
1930 NewTys = NewTys.drop_front();
1931 }
1932
1933 IntrinsicCostAttributes NewICA(*FID, ICA.getReturnType(), NewArgs,
1934 NewTys, ICA.getFlags());
1935 return thisT()->getIntrinsicInstrCost(NewICA, CostKind);
1936 }
1937 }
1938
1939 if (ICA.isTypeBasedOnly())
1941
1942 Type *RetTy = ICA.getReturnType();
1943
1944 ElementCount RetVF = isVectorizedTy(RetTy) ? getVectorizedTypeVF(RetTy)
1946
1947 const IntrinsicInst *I = ICA.getInst();
1948 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
1949 FastMathFlags FMF = ICA.getFlags();
1950 switch (IID) {
1951 default:
1952 break;
1953
1954 case Intrinsic::powi:
1955 if (auto *RHSC = dyn_cast<ConstantInt>(Args[1])) {
1956 bool ShouldOptForSize = I->getParent()->getParent()->hasOptSize();
1957 if (getTLI()->isBeneficialToExpandPowI(RHSC->getSExtValue(),
1958 ShouldOptForSize)) {
1959 // The cost is modeled on the expansion performed by ExpandPowI in
1960 // SelectionDAGBuilder.
1961 APInt Exponent = RHSC->getValue().abs();
1962 unsigned ActiveBits = Exponent.getActiveBits();
1963 unsigned PopCount = Exponent.popcount();
1964 InstructionCost Cost = (ActiveBits + PopCount - 2) *
1965 thisT()->getArithmeticInstrCost(
1966 Instruction::FMul, RetTy, CostKind);
1967 if (RHSC->isNegative())
1968 Cost += thisT()->getArithmeticInstrCost(Instruction::FDiv, RetTy,
1969 CostKind);
1970 return Cost;
1971 }
1972 }
1973 break;
1974 case Intrinsic::cttz:
1975 // FIXME: If necessary, this should go in target-specific overrides.
1976 if (RetVF.isScalar() && getTLI()->isCheapToSpeculateCttz(RetTy))
1978 break;
1979
1980 case Intrinsic::ctlz:
1981 // FIXME: If necessary, this should go in target-specific overrides.
1982 if (RetVF.isScalar() && getTLI()->isCheapToSpeculateCtlz(RetTy))
1984 break;
1985
1986 case Intrinsic::memcpy:
1987 return thisT()->getMemcpyCost(ICA.getInst());
1988
1989 case Intrinsic::masked_scatter: {
1990 const Value *Mask = Args[2];
1991 bool VarMask = !isa<Constant>(Mask);
1992 Align Alignment = I->getParamAlign(1).valueOrOne();
1993 return thisT()->getMemIntrinsicInstrCost(
1994 MemIntrinsicCostAttributes(Intrinsic::masked_scatter,
1995 ICA.getArgTypes()[0], Args[1], VarMask,
1996 Alignment, I),
1997 CostKind);
1998 }
1999 case Intrinsic::masked_gather: {
2000 const Value *Mask = Args[1];
2001 bool VarMask = !isa<Constant>(Mask);
2002 Align Alignment = I->getParamAlign(0).valueOrOne();
2003 return thisT()->getMemIntrinsicInstrCost(
2004 MemIntrinsicCostAttributes(Intrinsic::masked_gather, RetTy, Args[0],
2005 VarMask, Alignment, I),
2006 CostKind);
2007 }
2008 case Intrinsic::masked_compressstore: {
2009 const Value *Data = Args[0];
2010 const Value *Mask = Args[2];
2011 Align Alignment = I->getParamAlign(1).valueOrOne();
2012 return thisT()->getMemIntrinsicInstrCost(
2013 MemIntrinsicCostAttributes(IID, Data->getType(), !isa<Constant>(Mask),
2014 Alignment, I),
2015 CostKind);
2016 }
2017 case Intrinsic::masked_expandload: {
2018 const Value *Mask = Args[1];
2019 Align Alignment = I->getParamAlign(0).valueOrOne();
2020 return thisT()->getMemIntrinsicInstrCost(
2021 MemIntrinsicCostAttributes(IID, RetTy, !isa<Constant>(Mask),
2022 Alignment, I),
2023 CostKind);
2024 }
2025 case Intrinsic::experimental_vp_strided_store: {
2026 const Value *Data = Args[0];
2027 const Value *Ptr = Args[1];
2028 const Value *Mask = Args[3];
2029 const Value *EVL = Args[4];
2030 bool VarMask = !isa<Constant>(Mask) || !isa<Constant>(EVL);
2031 Type *EltTy = cast<VectorType>(Data->getType())->getElementType();
2032 Align Alignment =
2033 I->getParamAlign(1).value_or(thisT()->DL.getABITypeAlign(EltTy));
2034 return thisT()->getMemIntrinsicInstrCost(
2035 MemIntrinsicCostAttributes(IID, Data->getType(), Ptr, VarMask,
2036 Alignment, I),
2037 CostKind);
2038 }
2039 case Intrinsic::experimental_vp_strided_load: {
2040 const Value *Ptr = Args[0];
2041 const Value *Mask = Args[2];
2042 const Value *EVL = Args[3];
2043 bool VarMask = !isa<Constant>(Mask) || !isa<Constant>(EVL);
2044 Type *EltTy = cast<VectorType>(RetTy)->getElementType();
2045 Align Alignment =
2046 I->getParamAlign(0).value_or(thisT()->DL.getABITypeAlign(EltTy));
2047 return thisT()->getMemIntrinsicInstrCost(
2048 MemIntrinsicCostAttributes(IID, RetTy, Ptr, VarMask, Alignment, I),
2049 CostKind);
2050 }
2051 case Intrinsic::stepvector: {
2052 if (isa<ScalableVectorType>(RetTy))
2054 // The cost of materialising a constant integer vector.
2056 }
2057 case Intrinsic::vector_extract: {
2058 // FIXME: Handle case where a scalable vector is extracted from a scalable
2059 // vector
2060 if (isa<ScalableVectorType>(RetTy))
2062 unsigned Index = cast<ConstantInt>(Args[1])->getZExtValue();
2063 return thisT()->getShuffleCost(TTI::SK_ExtractSubvector,
2064 cast<VectorType>(RetTy),
2065 cast<VectorType>(Args[0]->getType()), {},
2066 CostKind, Index, cast<VectorType>(RetTy));
2067 }
2068 case Intrinsic::vector_insert: {
2069 // FIXME: Handle case where a scalable vector is inserted into a scalable
2070 // vector
2071 if (isa<ScalableVectorType>(Args[1]->getType()))
2073 unsigned Index = cast<ConstantInt>(Args[2])->getZExtValue();
2074 return thisT()->getShuffleCost(
2076 cast<VectorType>(Args[0]->getType()), {}, CostKind, Index,
2077 cast<VectorType>(Args[1]->getType()));
2078 }
2079 case Intrinsic::vector_splice_left:
2080 case Intrinsic::vector_splice_right: {
2081 auto *COffset = dyn_cast<ConstantInt>(Args[2]);
2082 if (!COffset)
2083 break;
2084 unsigned Index = COffset->getZExtValue();
2085 return thisT()->getShuffleCost(
2087 cast<VectorType>(Args[0]->getType()), {}, CostKind,
2088 IID == Intrinsic::vector_splice_left ? Index : -Index,
2089 cast<VectorType>(RetTy));
2090 }
2091 case Intrinsic::vector_reduce_add:
2092 case Intrinsic::vector_reduce_mul:
2093 case Intrinsic::vector_reduce_and:
2094 case Intrinsic::vector_reduce_or:
2095 case Intrinsic::vector_reduce_xor:
2096 case Intrinsic::vector_reduce_smax:
2097 case Intrinsic::vector_reduce_smin:
2098 case Intrinsic::vector_reduce_fmax:
2099 case Intrinsic::vector_reduce_fmin:
2100 case Intrinsic::vector_reduce_fmaximum:
2101 case Intrinsic::vector_reduce_fminimum:
2102 case Intrinsic::vector_reduce_umax:
2103 case Intrinsic::vector_reduce_umin: {
2104 IntrinsicCostAttributes Attrs(IID, RetTy, Args[0]->getType(), FMF, I, 1);
2106 }
2107 case Intrinsic::vector_reduce_fadd:
2108 case Intrinsic::vector_reduce_fmul: {
2110 IID, RetTy, {Args[0]->getType(), Args[1]->getType()}, FMF, I, 1);
2112 }
2113 case Intrinsic::fshl:
2114 case Intrinsic::fshr: {
2115 const Value *X = Args[0];
2116 const Value *Y = Args[1];
2117 const Value *Z = Args[2];
2120 const TTI::OperandValueInfo OpInfoZ = TTI::getOperandInfo(Z);
2121
2122 // fshl: (X << (Z % BW)) | (Y >> (BW - (Z % BW)))
2123 // fshr: (X << (BW - (Z % BW))) | (Y >> (Z % BW))
2125 Cost +=
2126 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
2127 Cost += thisT()->getArithmeticInstrCost(
2128 BinaryOperator::Shl, RetTy, CostKind, OpInfoX,
2129 {OpInfoZ.Kind, TTI::OP_None});
2130 Cost += thisT()->getArithmeticInstrCost(
2131 BinaryOperator::LShr, RetTy, CostKind, OpInfoY,
2132 {OpInfoZ.Kind, TTI::OP_None});
2133
2134 if (!OpInfoZ.isConstant()) {
2135 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy,
2136 CostKind);
2137 // Non-constant shift amounts requires a modulo. If the typesize is a
2138 // power-2 then this will be converted to an and, otherwise it will use
2139 // a urem.
2140 Cost += thisT()->getArithmeticInstrCost(
2141 isPowerOf2_32(RetTy->getScalarSizeInBits()) ? BinaryOperator::And
2142 : BinaryOperator::URem,
2143 RetTy, CostKind, OpInfoZ,
2144 {TTI::OK_UniformConstantValue, TTI::OP_None});
2145 // For non-rotates (X != Y) we must add shift-by-zero handling costs.
2146 if (X != Y) {
2147 Type *CondTy = RetTy->getWithNewBitWidth(1);
2148 Cost += thisT()->getCmpSelInstrCost(
2149 BinaryOperator::ICmp, RetTy, CondTy, CmpInst::ICMP_EQ, CostKind);
2150 Cost +=
2151 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2153 }
2154 }
2155 return Cost;
2156 }
2157 case Intrinsic::experimental_cttz_elts: {
2158 EVT ArgType = getTLI()->getValueType(DL, ICA.getArgTypes()[0], true);
2159
2160 // If we're not expanding the intrinsic then we assume this is cheap
2161 // to implement.
2162 if (!getTLI()->shouldExpandCttzElements(ArgType))
2163 return getTypeLegalizationCost(RetTy).first;
2164
2165 // TODO: The costs below reflect the expansion code in
2166 // SelectionDAGBuilder, but we may want to sacrifice some accuracy in
2167 // favour of compile time.
2168
2169 // Find the smallest "sensible" element type to use for the expansion.
2170 bool ZeroIsPoison = !cast<ConstantInt>(Args[1])->isZero();
2171 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
2172 if (isa<ScalableVectorType>(ICA.getArgTypes()[0]) && I && I->getCaller())
2173 VScaleRange = getVScaleRange(I->getCaller(), 64);
2174
2175 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
2176 getTLI()->getValueType(DL, RetTy), ArgType.getVectorElementCount(),
2177 ZeroIsPoison, &VScaleRange);
2178 Type *NewEltTy = IntegerType::getIntNTy(RetTy->getContext(), EltWidth);
2179
2180 // Create the new vector type & get the vector length
2181 Type *NewVecTy = VectorType::get(
2182 NewEltTy, cast<VectorType>(Args[0]->getType())->getElementCount());
2183
2184 IntrinsicCostAttributes StepVecAttrs(Intrinsic::stepvector, NewVecTy, {},
2185 FMF);
2187 thisT()->getIntrinsicInstrCost(StepVecAttrs, CostKind);
2188
2189 Cost +=
2190 thisT()->getArithmeticInstrCost(Instruction::Sub, NewVecTy, CostKind);
2191 Cost += thisT()->getCastInstrCost(Instruction::SExt, NewVecTy,
2192 Args[0]->getType(),
2194 Cost +=
2195 thisT()->getArithmeticInstrCost(Instruction::And, NewVecTy, CostKind);
2196
2197 IntrinsicCostAttributes ReducAttrs(Intrinsic::vector_reduce_umax,
2198 NewEltTy, NewVecTy, FMF, I, 1);
2199 Cost += thisT()->getTypeBasedIntrinsicInstrCost(ReducAttrs, CostKind);
2200 Cost +=
2201 thisT()->getArithmeticInstrCost(Instruction::Sub, NewEltTy, CostKind);
2202
2203 return Cost;
2204 }
2205 case Intrinsic::get_active_lane_mask:
2206 case Intrinsic::experimental_vector_match:
2207 case Intrinsic::experimental_vector_histogram_add:
2208 case Intrinsic::experimental_vector_histogram_uadd_sat:
2209 case Intrinsic::experimental_vector_histogram_umax:
2210 case Intrinsic::experimental_vector_histogram_umin:
2211 case Intrinsic::masked_udiv:
2212 case Intrinsic::masked_sdiv:
2213 case Intrinsic::masked_urem:
2214 case Intrinsic::masked_srem:
2215 return thisT()->getTypeBasedIntrinsicInstrCost(ICA, CostKind);
2216 case Intrinsic::modf:
2217 case Intrinsic::sincos:
2218 case Intrinsic::sincospi: {
2219 std::optional<unsigned> CallRetElementIndex;
2220 // The first element of the modf result is returned by value in the
2221 // libcall.
2222 if (ICA.getID() == Intrinsic::modf)
2223 CallRetElementIndex = 0;
2224
2225 if (auto Cost = getMultipleResultIntrinsicVectorLibCallCost(
2226 ICA, CostKind, CallRetElementIndex))
2227 return *Cost;
2228 // Otherwise, fallback to default scalarization cost.
2229 break;
2230 }
2231 case Intrinsic::loop_dependence_war_mask:
2232 case Intrinsic::loop_dependence_raw_mask: {
2233 // Compute the cost of the expanded version of these intrinsics:
2234 //
2235 // The possible expansions are...
2236 //
2237 // loop_dependence_war_mask:
2238 // diff = (addrB - addrA) / eltSize
2239 // cmp = icmp sle diff, 0
2240 // upper_bound = select cmp, -1, diff
2241 // mask = get_active_lane_mask 0, upper_bound
2242 //
2243 // loop_dependence_raw_mask:
2244 // diff = (abs(addrB - addrA)) / eltSize
2245 // cmp = icmp eq diff, 0
2246 // upper_bound = select cmp, -1, diff
2247 // mask = get_active_lane_mask 0, upper_bound
2248 //
2249 Type *AddrTy = ICA.getArgTypes()[0];
2250 bool IsReadAfterWrite = IID == Intrinsic::loop_dependence_raw_mask;
2251
2253 thisT()->getArithmeticInstrCost(Instruction::Sub, AddrTy, CostKind);
2254 if (IsReadAfterWrite) {
2255 IntrinsicCostAttributes AbsAttrs(Intrinsic::abs, AddrTy, {AddrTy}, {});
2256 Cost += thisT()->getIntrinsicInstrCost(AbsAttrs, CostKind);
2257 }
2258
2259 TTI::OperandValueInfo EltSizeOpInfo =
2260 TTI::getOperandInfo(ICA.getArgs()[2]);
2261 Cost += thisT()->getArithmeticInstrCost(Instruction::SDiv, AddrTy,
2262 CostKind, {}, EltSizeOpInfo);
2263
2264 Type *CondTy = IntegerType::getInt1Ty(RetTy->getContext());
2265 CmpInst::Predicate Pred =
2266 IsReadAfterWrite ? CmpInst::ICMP_EQ : CmpInst::ICMP_SLE;
2267 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CondTy, AddrTy,
2268 Pred, CostKind);
2269 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, AddrTy,
2270 CondTy, Pred, CostKind);
2271
2272 IntrinsicCostAttributes Attrs(Intrinsic::get_active_lane_mask, RetTy,
2273 {AddrTy, AddrTy}, FMF);
2274 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2275 return Cost;
2276 }
2277 }
2278
2279 // Assume that we need to scalarize this intrinsic.)
2280 // Compute the scalarization overhead based on Args for a vector
2281 // intrinsic.
2282 InstructionCost ScalarizationCost = InstructionCost::getInvalid();
2283 if (RetVF.isVector() && !RetVF.isScalable()) {
2284 ScalarizationCost = 0;
2285 if (!RetTy->isVoidTy()) {
2286 for (Type *VectorTy : getContainedTypes(RetTy)) {
2287 ScalarizationCost += getScalarizationOverhead(
2288 cast<VectorType>(VectorTy),
2289 /*Insert=*/true, /*Extract=*/false, CostKind);
2290 }
2291 }
2292 ScalarizationCost += getOperandsScalarizationOverhead(
2293 filterConstantAndDuplicatedOperands(Args, ICA.getArgTypes()),
2294 CostKind);
2295 }
2296
2297 IntrinsicCostAttributes Attrs(IID, RetTy, ICA.getArgTypes(), FMF, I,
2298 ScalarizationCost);
2299 return thisT()->getTypeBasedIntrinsicInstrCost(Attrs, CostKind);
2300 }
2301
2302 /// Get intrinsic cost based on argument types.
2303 /// If ScalarizationCostPassed is std::numeric_limits<unsigned>::max(), the
2304 /// cost of scalarizing the arguments and the return value will be computed
2305 /// based on types.
2309 Intrinsic::ID IID = ICA.getID();
2310 Type *RetTy = ICA.getReturnType();
2311 const SmallVectorImpl<Type *> &Tys = ICA.getArgTypes();
2312 FastMathFlags FMF = ICA.getFlags();
2313 InstructionCost ScalarizationCostPassed = ICA.getScalarizationCost();
2314 bool SkipScalarizationCost = ICA.skipScalarizationCost();
2315
2316 VectorType *VecOpTy = nullptr;
2317 if (!Tys.empty()) {
2318 // The vector reduction operand is operand 0 except for fadd/fmul.
2319 // Their operand 0 is a scalar start value, so the vector op is operand 1.
2320 unsigned VecTyIndex = 0;
2321 if (IID == Intrinsic::vector_reduce_fadd ||
2322 IID == Intrinsic::vector_reduce_fmul)
2323 VecTyIndex = 1;
2324 assert(Tys.size() > VecTyIndex && "Unexpected IntrinsicCostAttributes");
2325 VecOpTy = dyn_cast<VectorType>(Tys[VecTyIndex]);
2326 }
2327
2328 // Library call cost - other than size, make it expensive.
2329 unsigned SingleCallCost = CostKind == TTI::TCK_CodeSize ? 1 : 10;
2330 unsigned ISD = 0;
2331 switch (IID) {
2332 default: {
2333 // Scalable vectors cannot be scalarized, so return Invalid.
2334 if (isa<ScalableVectorType>(RetTy) || any_of(Tys, [](const Type *Ty) {
2335 return isa<ScalableVectorType>(Ty);
2336 }))
2338
2339 // Assume that we need to scalarize this intrinsic.
2340 InstructionCost ScalarizationCost =
2341 SkipScalarizationCost ? ScalarizationCostPassed : 0;
2342 unsigned ScalarCalls = 1;
2343 Type *ScalarRetTy = RetTy;
2344 if (auto *RetVTy = dyn_cast<VectorType>(RetTy)) {
2345 if (!SkipScalarizationCost)
2346 ScalarizationCost = getScalarizationOverhead(
2347 RetVTy, /*Insert*/ true, /*Extract*/ false, CostKind);
2348 ScalarCalls = std::max(ScalarCalls,
2349 cast<FixedVectorType>(RetVTy)->getNumElements());
2350 ScalarRetTy = RetTy->getScalarType();
2351 }
2352 SmallVector<Type *, 4> ScalarTys;
2353 for (Type *Ty : Tys) {
2354 if (auto *VTy = dyn_cast<VectorType>(Ty)) {
2355 if (!SkipScalarizationCost)
2356 ScalarizationCost += getScalarizationOverhead(
2357 VTy, /*Insert*/ false, /*Extract*/ true, CostKind);
2358 ScalarCalls = std::max(ScalarCalls,
2359 cast<FixedVectorType>(VTy)->getNumElements());
2360 Ty = Ty->getScalarType();
2361 }
2362 ScalarTys.push_back(Ty);
2363 }
2364 if (ScalarCalls == 1)
2365 return 1; // Return cost of a scalar intrinsic. Assume it to be cheap.
2366
2367 IntrinsicCostAttributes ScalarAttrs(IID, ScalarRetTy, ScalarTys, FMF);
2368 InstructionCost ScalarCost =
2369 thisT()->getIntrinsicInstrCost(ScalarAttrs, CostKind);
2370
2371 return ScalarCalls * ScalarCost + ScalarizationCost;
2372 }
2373 // Look for intrinsics that can be lowered directly or turned into a scalar
2374 // intrinsic call.
2375 case Intrinsic::sqrt:
2376 ISD = ISD::FSQRT;
2377 break;
2378 case Intrinsic::sin:
2379 ISD = ISD::FSIN;
2380 break;
2381 case Intrinsic::cos:
2382 ISD = ISD::FCOS;
2383 break;
2384 case Intrinsic::sincos:
2385 ISD = ISD::FSINCOS;
2386 break;
2387 case Intrinsic::sincospi:
2389 break;
2390 case Intrinsic::modf:
2391 ISD = ISD::FMODF;
2392 break;
2393 case Intrinsic::tan:
2394 ISD = ISD::FTAN;
2395 break;
2396 case Intrinsic::asin:
2397 ISD = ISD::FASIN;
2398 break;
2399 case Intrinsic::acos:
2400 ISD = ISD::FACOS;
2401 break;
2402 case Intrinsic::atan:
2403 ISD = ISD::FATAN;
2404 break;
2405 case Intrinsic::atan2:
2406 ISD = ISD::FATAN2;
2407 break;
2408 case Intrinsic::sinh:
2409 ISD = ISD::FSINH;
2410 break;
2411 case Intrinsic::cosh:
2412 ISD = ISD::FCOSH;
2413 break;
2414 case Intrinsic::tanh:
2415 ISD = ISD::FTANH;
2416 break;
2417 case Intrinsic::exp:
2418 ISD = ISD::FEXP;
2419 break;
2420 case Intrinsic::exp2:
2421 ISD = ISD::FEXP2;
2422 break;
2423 case Intrinsic::exp10:
2424 ISD = ISD::FEXP10;
2425 break;
2426 case Intrinsic::log:
2427 ISD = ISD::FLOG;
2428 break;
2429 case Intrinsic::log10:
2430 ISD = ISD::FLOG10;
2431 break;
2432 case Intrinsic::log2:
2433 ISD = ISD::FLOG2;
2434 break;
2435 case Intrinsic::ldexp:
2436 ISD = ISD::FLDEXP;
2437 break;
2438 case Intrinsic::fabs:
2439 ISD = ISD::FABS;
2440 break;
2441 case Intrinsic::canonicalize:
2443 break;
2444 case Intrinsic::minnum:
2445 ISD = ISD::FMINNUM;
2446 break;
2447 case Intrinsic::maxnum:
2448 ISD = ISD::FMAXNUM;
2449 break;
2450 case Intrinsic::minimum:
2452 break;
2453 case Intrinsic::maximum:
2455 break;
2456 case Intrinsic::minimumnum:
2458 break;
2459 case Intrinsic::maximumnum:
2461 break;
2462 case Intrinsic::copysign:
2464 break;
2465 case Intrinsic::floor:
2466 ISD = ISD::FFLOOR;
2467 break;
2468 case Intrinsic::ceil:
2469 ISD = ISD::FCEIL;
2470 break;
2471 case Intrinsic::trunc:
2472 ISD = ISD::FTRUNC;
2473 break;
2474 case Intrinsic::nearbyint:
2476 break;
2477 case Intrinsic::rint:
2478 ISD = ISD::FRINT;
2479 break;
2480 case Intrinsic::lrint:
2481 ISD = ISD::LRINT;
2482 break;
2483 case Intrinsic::llrint:
2484 ISD = ISD::LLRINT;
2485 break;
2486 case Intrinsic::round:
2487 ISD = ISD::FROUND;
2488 break;
2489 case Intrinsic::roundeven:
2491 break;
2492 case Intrinsic::lround:
2493 ISD = ISD::LROUND;
2494 break;
2495 case Intrinsic::llround:
2496 ISD = ISD::LLROUND;
2497 break;
2498 case Intrinsic::pow:
2499 ISD = ISD::FPOW;
2500 break;
2501 case Intrinsic::fma:
2502 ISD = ISD::FMA;
2503 break;
2504 case Intrinsic::fmuladd:
2505 ISD = ISD::FMA;
2506 break;
2507 case Intrinsic::experimental_constrained_fmuladd:
2509 break;
2510 // FIXME: We should return 0 whenever getIntrinsicCost == TCC_Free.
2511 case Intrinsic::lifetime_start:
2512 case Intrinsic::lifetime_end:
2513 case Intrinsic::sideeffect:
2514 case Intrinsic::pseudoprobe:
2515 case Intrinsic::arithmetic_fence:
2516 return 0;
2517 case Intrinsic::masked_store: {
2518 Type *Ty = Tys[0];
2519 Align TyAlign = thisT()->DL.getABITypeAlign(Ty);
2520 return thisT()->getMemIntrinsicInstrCost(
2521 MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
2522 }
2523 case Intrinsic::masked_load: {
2524 Type *Ty = RetTy;
2525 Align TyAlign = thisT()->DL.getABITypeAlign(Ty);
2526 return thisT()->getMemIntrinsicInstrCost(
2527 MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
2528 }
2529 case Intrinsic::experimental_vp_strided_store: {
2530 auto *Ty = cast<VectorType>(ICA.getArgTypes()[0]);
2531 Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
2532 return thisT()->getMemIntrinsicInstrCost(
2533 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr,
2534 /*VariableMask=*/true, Alignment,
2535 ICA.getInst()),
2536 CostKind);
2537 }
2538 case Intrinsic::experimental_vp_strided_load: {
2539 auto *Ty = cast<VectorType>(ICA.getReturnType());
2540 Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
2541 return thisT()->getMemIntrinsicInstrCost(
2542 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr,
2543 /*VariableMask=*/true, Alignment,
2544 ICA.getInst()),
2545 CostKind);
2546 }
2547 case Intrinsic::vector_reduce_add:
2548 case Intrinsic::vector_reduce_mul:
2549 case Intrinsic::vector_reduce_and:
2550 case Intrinsic::vector_reduce_or:
2551 case Intrinsic::vector_reduce_xor:
2552 return thisT()->getArithmeticReductionCost(
2553 getArithmeticReductionInstruction(IID), VecOpTy, std::nullopt,
2554 CostKind);
2555 case Intrinsic::vector_reduce_fadd:
2556 case Intrinsic::vector_reduce_fmul:
2557 return thisT()->getArithmeticReductionCost(
2558 getArithmeticReductionInstruction(IID), VecOpTy, FMF, CostKind);
2559 case Intrinsic::vector_reduce_smax:
2560 case Intrinsic::vector_reduce_smin:
2561 case Intrinsic::vector_reduce_umax:
2562 case Intrinsic::vector_reduce_umin:
2563 case Intrinsic::vector_reduce_fmax:
2564 case Intrinsic::vector_reduce_fmin:
2565 case Intrinsic::vector_reduce_fmaximum:
2566 case Intrinsic::vector_reduce_fminimum:
2567 return thisT()->getMinMaxReductionCost(getMinMaxReductionIntrinsicOp(IID),
2568 VecOpTy, ICA.getFlags(), CostKind);
2569 case Intrinsic::experimental_vector_match: {
2570 auto *SearchTy = cast<VectorType>(ICA.getArgTypes()[0]);
2571 auto *NeedleTy = cast<FixedVectorType>(ICA.getArgTypes()[1]);
2572 unsigned SearchSize = NeedleTy->getNumElements();
2573
2574 // If we're not expanding the intrinsic then we assume this is cheap to
2575 // implement.
2576 EVT SearchVT = getTLI()->getValueType(DL, SearchTy);
2577 if (!getTLI()->shouldExpandVectorMatch(SearchVT, SearchSize))
2578 return getTypeLegalizationCost(RetTy).first;
2579
2580 // Approximate the cost based on the expansion code in
2581 // SelectionDAGBuilder.
2583 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, NeedleTy,
2584 CostKind, 1, nullptr, nullptr);
2585 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, SearchTy,
2586 CostKind, 0, nullptr, nullptr);
2587 Cost += thisT()->getShuffleCost(TTI::SK_Broadcast, SearchTy, SearchTy, {},
2588 CostKind, 0, nullptr);
2589 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, SearchTy, RetTy,
2591 Cost +=
2592 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
2593 Cost *= SearchSize;
2594 Cost +=
2595 thisT()->getArithmeticInstrCost(BinaryOperator::And, RetTy, CostKind);
2596 return Cost;
2597 }
2598 case Intrinsic::vector_reverse:
2599 return thisT()->getShuffleCost(TTI::SK_Reverse, cast<VectorType>(RetTy),
2600 cast<VectorType>(ICA.getArgTypes()[0]), {},
2601 CostKind, 0, cast<VectorType>(RetTy));
2602 case Intrinsic::experimental_vector_histogram_add:
2603 case Intrinsic::experimental_vector_histogram_uadd_sat:
2604 case Intrinsic::experimental_vector_histogram_umax:
2605 case Intrinsic::experimental_vector_histogram_umin: {
2607 Type *EltTy = ICA.getArgTypes()[1];
2608
2609 // Targets with scalable vectors must handle this on their own.
2610 if (!PtrsTy)
2612
2613 Align Alignment = thisT()->DL.getABITypeAlign(EltTy);
2615 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, PtrsTy,
2616 CostKind, 1, nullptr, nullptr);
2617 Cost += thisT()->getMemoryOpCost(Instruction::Load, EltTy, Alignment, 0,
2618 CostKind);
2619 switch (IID) {
2620 default:
2621 llvm_unreachable("Unhandled histogram update operation.");
2622 case Intrinsic::experimental_vector_histogram_add:
2623 Cost +=
2624 thisT()->getArithmeticInstrCost(Instruction::Add, EltTy, CostKind);
2625 break;
2626 case Intrinsic::experimental_vector_histogram_uadd_sat: {
2627 IntrinsicCostAttributes UAddSat(Intrinsic::uadd_sat, EltTy, {EltTy});
2628 Cost += thisT()->getIntrinsicInstrCost(UAddSat, CostKind);
2629 break;
2630 }
2631 case Intrinsic::experimental_vector_histogram_umax: {
2632 IntrinsicCostAttributes UMax(Intrinsic::umax, EltTy, {EltTy});
2633 Cost += thisT()->getIntrinsicInstrCost(UMax, CostKind);
2634 break;
2635 }
2636 case Intrinsic::experimental_vector_histogram_umin: {
2637 IntrinsicCostAttributes UMin(Intrinsic::umin, EltTy, {EltTy});
2638 Cost += thisT()->getIntrinsicInstrCost(UMin, CostKind);
2639 break;
2640 }
2641 }
2642 Cost += thisT()->getMemoryOpCost(Instruction::Store, EltTy, Alignment, 0,
2643 CostKind);
2644 Cost *= PtrsTy->getNumElements();
2645 return Cost;
2646 }
2647 case Intrinsic::get_active_lane_mask: {
2648 Type *ArgTy = ICA.getArgTypes()[0];
2649 EVT ResVT = getTLI()->getValueType(DL, RetTy, true);
2650 EVT ArgVT = getTLI()->getValueType(DL, ArgTy, true);
2651
2652 // If we're not expanding the intrinsic then we assume this is cheap
2653 // to implement.
2654 if (!getTLI()->shouldExpandGetActiveLaneMask(ResVT, ArgVT))
2655 return getTypeLegalizationCost(RetTy).first;
2656
2657 // Create the expanded types that will be used to calculate the uadd_sat
2658 // operation.
2659 Type *ExpRetTy =
2660 VectorType::get(ArgTy, cast<VectorType>(RetTy)->getElementCount());
2661 IntrinsicCostAttributes Attrs(Intrinsic::uadd_sat, ExpRetTy, {}, FMF);
2663 thisT()->getTypeBasedIntrinsicInstrCost(Attrs, CostKind);
2664 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, ExpRetTy, RetTy,
2666 return Cost;
2667 }
2668 case Intrinsic::experimental_memset_pattern:
2669 // This cost is set to match the cost of the memset_pattern16 libcall.
2670 // It should likely be re-evaluated after migration to this intrinsic
2671 // is complete.
2672 return TTI::TCC_Basic * 4;
2673 case Intrinsic::abs:
2674 ISD = ISD::ABS;
2675 break;
2676 case Intrinsic::fshl:
2677 ISD = ISD::FSHL;
2678 break;
2679 case Intrinsic::fshr:
2680 ISD = ISD::FSHR;
2681 break;
2682 case Intrinsic::smax:
2683 ISD = ISD::SMAX;
2684 break;
2685 case Intrinsic::smin:
2686 ISD = ISD::SMIN;
2687 break;
2688 case Intrinsic::umax:
2689 ISD = ISD::UMAX;
2690 break;
2691 case Intrinsic::umin:
2692 ISD = ISD::UMIN;
2693 break;
2694 case Intrinsic::sadd_sat:
2695 ISD = ISD::SADDSAT;
2696 break;
2697 case Intrinsic::ssub_sat:
2698 ISD = ISD::SSUBSAT;
2699 break;
2700 case Intrinsic::uadd_sat:
2701 ISD = ISD::UADDSAT;
2702 break;
2703 case Intrinsic::usub_sat:
2704 ISD = ISD::USUBSAT;
2705 break;
2706 case Intrinsic::smul_fix:
2707 ISD = ISD::SMULFIX;
2708 break;
2709 case Intrinsic::umul_fix:
2710 ISD = ISD::UMULFIX;
2711 break;
2712 case Intrinsic::sadd_with_overflow:
2713 ISD = ISD::SADDO;
2714 break;
2715 case Intrinsic::ssub_with_overflow:
2716 ISD = ISD::SSUBO;
2717 break;
2718 case Intrinsic::uadd_with_overflow:
2719 ISD = ISD::UADDO;
2720 break;
2721 case Intrinsic::usub_with_overflow:
2722 ISD = ISD::USUBO;
2723 break;
2724 case Intrinsic::smul_with_overflow:
2725 ISD = ISD::SMULO;
2726 break;
2727 case Intrinsic::umul_with_overflow:
2728 ISD = ISD::UMULO;
2729 break;
2730 case Intrinsic::fptosi_sat:
2731 case Intrinsic::fptoui_sat: {
2732 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Tys[0]);
2733 std::pair<InstructionCost, MVT> RetLT = getTypeLegalizationCost(RetTy);
2734
2735 // For cast instructions, types are different between source and
2736 // destination. Also need to check if the source type can be legalize.
2737 if (!SrcLT.first.isValid() || !RetLT.first.isValid())
2739 ISD = IID == Intrinsic::fptosi_sat ? ISD::FP_TO_SINT_SAT
2741 break;
2742 }
2743 case Intrinsic::ctpop:
2744 ISD = ISD::CTPOP;
2745 // In case of legalization use TCC_Expensive. This is cheaper than a
2746 // library call but still not a cheap instruction.
2747 SingleCallCost = TargetTransformInfo::TCC_Expensive;
2748 break;
2749 case Intrinsic::ctlz:
2750 ISD = ISD::CTLZ;
2751 break;
2752 case Intrinsic::cttz:
2753 ISD = ISD::CTTZ;
2754 break;
2755 case Intrinsic::bswap:
2756 ISD = ISD::BSWAP;
2757 break;
2758 case Intrinsic::bitreverse:
2760 break;
2761 case Intrinsic::ucmp:
2762 ISD = ISD::UCMP;
2763 break;
2764 case Intrinsic::scmp:
2765 ISD = ISD::SCMP;
2766 break;
2767 case Intrinsic::clmul:
2768 ISD = ISD::CLMUL;
2769 break;
2770 case Intrinsic::masked_udiv:
2771 case Intrinsic::masked_sdiv:
2772 case Intrinsic::masked_urem:
2773 case Intrinsic::masked_srem: {
2774 unsigned UnmaskedOpc;
2775 switch (IID) {
2776 case Intrinsic::masked_udiv:
2778 UnmaskedOpc = Instruction::UDiv;
2779 break;
2780 case Intrinsic::masked_sdiv:
2782 UnmaskedOpc = Instruction::SDiv;
2783 break;
2784 case Intrinsic::masked_urem:
2786 UnmaskedOpc = Instruction::URem;
2787 break;
2788 case Intrinsic::masked_srem:
2790 UnmaskedOpc = Instruction::SRem;
2791 break;
2792 default:
2793 llvm_unreachable("Unexpected intrinsic ID");
2794 }
2796 thisT()->getArithmeticInstrCost(UnmaskedOpc, RetTy, CostKind);
2797
2798 // Expansion generates a (select %mask, %rhs, 1) for the divisor.
2799 MVT LT = getTypeLegalizationCost(RetTy).second;
2800 if (!getTLI()->isOperationLegalOrCustom(ISD, LT)) {
2801 Type *CondTy = cast<VectorType>(RetTy)->getWithNewType(
2803 Cost += thisT()->getCmpSelInstrCost(
2804 BinaryOperator::Select, RetTy, CondTy, CmpInst::BAD_ICMP_PREDICATE,
2806 }
2807
2808 return Cost;
2809 }
2810 }
2811
2812 auto *ST = dyn_cast<StructType>(RetTy);
2813 Type *LegalizeTy = ST ? ST->getContainedType(0) : RetTy;
2814 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(LegalizeTy);
2815
2816 const TargetLoweringBase *TLI = getTLI();
2817
2818 if (TLI->isOperationLegalOrPromote(ISD, LT.second)) {
2819 if (IID == Intrinsic::fabs && LT.second.isFloatingPoint() &&
2820 TLI->isFAbsFree(LT.second)) {
2821 return 0;
2822 }
2823
2824 // The operation is legal. Assume it costs 1.
2825 // If the type is split to multiple registers, assume that there is some
2826 // overhead to this.
2827 // TODO: Once we have extract/insert subvector cost we need to use them.
2828 if (LT.first > 1)
2829 return (LT.first * 2);
2830 else
2831 return (LT.first * 1);
2832 } else if (TLI->isOperationCustom(ISD, LT.second)) {
2833 // If the operation is custom lowered then assume
2834 // that the code is twice as expensive.
2835 return (LT.first * 2);
2836 }
2837
2838 switch (IID) {
2839 case Intrinsic::fmuladd: {
2840 // If we can't lower fmuladd into an FMA estimate the cost as a floating
2841 // point mul followed by an add.
2842
2843 return thisT()->getArithmeticInstrCost(BinaryOperator::FMul, RetTy,
2844 CostKind) +
2845 thisT()->getArithmeticInstrCost(BinaryOperator::FAdd, RetTy,
2846 CostKind);
2847 }
2848 case Intrinsic::experimental_constrained_fmuladd: {
2849 IntrinsicCostAttributes FMulAttrs(
2850 Intrinsic::experimental_constrained_fmul, RetTy, Tys);
2851 IntrinsicCostAttributes FAddAttrs(
2852 Intrinsic::experimental_constrained_fadd, RetTy, Tys);
2853 return thisT()->getIntrinsicInstrCost(FMulAttrs, CostKind) +
2854 thisT()->getIntrinsicInstrCost(FAddAttrs, CostKind);
2855 }
2856 case Intrinsic::smin:
2857 case Intrinsic::smax:
2858 case Intrinsic::umin:
2859 case Intrinsic::umax: {
2860 // minmax(X,Y) = select(icmp(X,Y),X,Y)
2861 Type *CondTy = RetTy->getWithNewBitWidth(1);
2862 bool IsUnsigned = IID == Intrinsic::umax || IID == Intrinsic::umin;
2863 CmpInst::Predicate Pred =
2864 IsUnsigned ? CmpInst::ICMP_UGT : CmpInst::ICMP_SGT;
2866 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
2867 Pred, CostKind);
2868 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2869 Pred, CostKind);
2870 return Cost;
2871 }
2872 case Intrinsic::sadd_with_overflow:
2873 case Intrinsic::ssub_with_overflow: {
2874 Type *SumTy = RetTy->getContainedType(0);
2875 Type *OverflowTy = RetTy->getContainedType(1);
2876 unsigned Opcode = IID == Intrinsic::sadd_with_overflow
2877 ? BinaryOperator::Add
2878 : BinaryOperator::Sub;
2879
2880 // Add:
2881 // Overflow -> (Result < LHS) ^ (RHS < 0)
2882 // Sub:
2883 // Overflow -> (Result < LHS) ^ (RHS > 0)
2885 Cost += thisT()->getArithmeticInstrCost(Opcode, SumTy, CostKind);
2886 Cost +=
2887 2 * thisT()->getCmpSelInstrCost(Instruction::ICmp, SumTy, OverflowTy,
2889 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Xor, OverflowTy,
2890 CostKind);
2891 return Cost;
2892 }
2893 case Intrinsic::uadd_with_overflow:
2894 case Intrinsic::usub_with_overflow: {
2895 Type *SumTy = RetTy->getContainedType(0);
2896 Type *OverflowTy = RetTy->getContainedType(1);
2897 unsigned Opcode = IID == Intrinsic::uadd_with_overflow
2898 ? BinaryOperator::Add
2899 : BinaryOperator::Sub;
2900 CmpInst::Predicate Pred = IID == Intrinsic::uadd_with_overflow
2903
2905 Cost += thisT()->getArithmeticInstrCost(Opcode, SumTy, CostKind);
2906 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, SumTy,
2907 OverflowTy, Pred, CostKind);
2908 return Cost;
2909 }
2910 case Intrinsic::smul_with_overflow:
2911 case Intrinsic::umul_with_overflow: {
2912 Type *MulTy = RetTy->getContainedType(0);
2913 Type *OverflowTy = RetTy->getContainedType(1);
2914 unsigned ExtSize = MulTy->getScalarSizeInBits() * 2;
2915 Type *ExtTy = MulTy->getWithNewBitWidth(ExtSize);
2916 bool IsSigned = IID == Intrinsic::smul_with_overflow;
2917
2918 unsigned ExtOp = IsSigned ? Instruction::SExt : Instruction::ZExt;
2920
2922 Cost += 2 * thisT()->getCastInstrCost(ExtOp, ExtTy, MulTy, CCH, CostKind);
2923 Cost +=
2924 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
2925 Cost += 2 * thisT()->getCastInstrCost(Instruction::Trunc, MulTy, ExtTy,
2926 CCH, CostKind);
2927 Cost += thisT()->getArithmeticInstrCost(
2928 Instruction::LShr, ExtTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
2930
2931 if (IsSigned)
2932 Cost += thisT()->getArithmeticInstrCost(
2933 Instruction::AShr, MulTy, CostKind,
2936
2937 Cost += thisT()->getCmpSelInstrCost(
2938 BinaryOperator::ICmp, MulTy, OverflowTy, CmpInst::ICMP_NE, CostKind);
2939 return Cost;
2940 }
2941 case Intrinsic::sadd_sat:
2942 case Intrinsic::ssub_sat: {
2943 // Assume a default expansion.
2944 Type *CondTy = RetTy->getWithNewBitWidth(1);
2945
2946 Type *OpTy = StructType::create({RetTy, CondTy});
2947 Intrinsic::ID OverflowOp = IID == Intrinsic::sadd_sat
2948 ? Intrinsic::sadd_with_overflow
2949 : Intrinsic::ssub_with_overflow;
2951
2952 // SatMax -> Overflow && SumDiff < 0
2953 // SatMin -> Overflow && SumDiff >= 0
2955 IntrinsicCostAttributes Attrs(OverflowOp, OpTy, {RetTy, RetTy}, FMF,
2956 nullptr, ScalarizationCostPassed);
2957 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2958 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
2959 Pred, CostKind);
2960 Cost += 2 * thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy,
2961 CondTy, Pred, CostKind);
2962 return Cost;
2963 }
2964 case Intrinsic::uadd_sat:
2965 case Intrinsic::usub_sat: {
2966 Type *CondTy = RetTy->getWithNewBitWidth(1);
2967
2968 Type *OpTy = StructType::create({RetTy, CondTy});
2969 Intrinsic::ID OverflowOp = IID == Intrinsic::uadd_sat
2970 ? Intrinsic::uadd_with_overflow
2971 : Intrinsic::usub_with_overflow;
2972
2974 IntrinsicCostAttributes Attrs(OverflowOp, OpTy, {RetTy, RetTy}, FMF,
2975 nullptr, ScalarizationCostPassed);
2976 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2977 Cost +=
2978 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2980 return Cost;
2981 }
2982 case Intrinsic::smul_fix:
2983 case Intrinsic::umul_fix: {
2984 unsigned ExtSize = RetTy->getScalarSizeInBits() * 2;
2985 Type *ExtTy = RetTy->getWithNewBitWidth(ExtSize);
2986
2987 unsigned ExtOp =
2988 IID == Intrinsic::smul_fix ? Instruction::SExt : Instruction::ZExt;
2990
2992 Cost += 2 * thisT()->getCastInstrCost(ExtOp, ExtTy, RetTy, CCH, CostKind);
2993 Cost +=
2994 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
2995 Cost += 2 * thisT()->getCastInstrCost(Instruction::Trunc, RetTy, ExtTy,
2996 CCH, CostKind);
2997 Cost += thisT()->getArithmeticInstrCost(
2998 Instruction::LShr, RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3000 Cost += thisT()->getArithmeticInstrCost(
3001 Instruction::Shl, RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3003 Cost += thisT()->getArithmeticInstrCost(Instruction::Or, RetTy, CostKind);
3004 return Cost;
3005 }
3006 case Intrinsic::abs: {
3007 // abs(X) = select(icmp(X,0),X,sub(0,X))
3008 Type *CondTy = RetTy->getWithNewBitWidth(1);
3011 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
3012 Pred, CostKind);
3013 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
3014 Pred, CostKind);
3015 // TODO: Should we add an OperandValueProperties::OP_Zero property?
3016 Cost += thisT()->getArithmeticInstrCost(
3017 BinaryOperator::Sub, RetTy, CostKind,
3019 return Cost;
3020 }
3021 case Intrinsic::fshl:
3022 case Intrinsic::fshr: {
3023 // fshl: (X << (Z % BW)) | (Y >> (BW - (Z % BW)))
3024 // fshr: (X << (BW - (Z % BW))) | (Y >> (Z % BW))
3025 Type *CondTy = RetTy->getWithNewBitWidth(1);
3027 Cost +=
3028 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
3029 Cost +=
3030 thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy, CostKind);
3031 Cost +=
3032 thisT()->getArithmeticInstrCost(BinaryOperator::Shl, RetTy, CostKind);
3033 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::LShr, RetTy,
3034 CostKind);
3035 // Non-constant shift amounts requires a modulo. If the typesize is a
3036 // power-2 then this will be converted to an and, otherwise it will use a
3037 // urem.
3038 Cost += thisT()->getArithmeticInstrCost(
3039 isPowerOf2_32(RetTy->getScalarSizeInBits()) ? BinaryOperator::And
3040 : BinaryOperator::URem,
3041 RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3042 {TTI::OK_UniformConstantValue, TTI::OP_None});
3043 // Shift-by-zero handling.
3044 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
3046 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
3048 return Cost;
3049 }
3050 case Intrinsic::fptosi_sat:
3051 case Intrinsic::fptoui_sat: {
3052 if (Tys.empty())
3053 break;
3054 Type *FromTy = Tys[0];
3055 bool IsSigned = IID == Intrinsic::fptosi_sat;
3056
3058 IntrinsicCostAttributes Attrs1(Intrinsic::minnum, FromTy,
3059 {FromTy, FromTy});
3060 Cost += thisT()->getIntrinsicInstrCost(Attrs1, CostKind);
3061 IntrinsicCostAttributes Attrs2(Intrinsic::maxnum, FromTy,
3062 {FromTy, FromTy});
3063 Cost += thisT()->getIntrinsicInstrCost(Attrs2, CostKind);
3064 Cost += thisT()->getCastInstrCost(
3065 IsSigned ? Instruction::FPToSI : Instruction::FPToUI, RetTy, FromTy,
3067 if (IsSigned) {
3068 Type *CondTy = RetTy->getWithNewBitWidth(1);
3069 Cost += thisT()->getCmpSelInstrCost(
3070 BinaryOperator::FCmp, FromTy, CondTy, CmpInst::FCMP_UNO, CostKind);
3071 Cost += thisT()->getCmpSelInstrCost(
3072 BinaryOperator::Select, RetTy, CondTy, CmpInst::FCMP_UNO, CostKind);
3073 }
3074 return Cost;
3075 }
3076 case Intrinsic::ucmp:
3077 case Intrinsic::scmp: {
3078 Type *CmpTy = Tys[0];
3079 Type *CondTy = RetTy->getWithNewBitWidth(1);
3081 thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CmpTy, CondTy,
3083 CostKind) +
3084 thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CmpTy, CondTy,
3086 CostKind);
3087
3088 EVT VT = TLI->getValueType(DL, CmpTy, true);
3090 // x < y ? -1 : (x > y ? 1 : 0)
3091 Cost += 2 * thisT()->getCmpSelInstrCost(
3092 BinaryOperator::Select, RetTy, CondTy,
3094 } else {
3095 // zext(x > y) - zext(x < y)
3096 Cost +=
3097 2 * thisT()->getCastInstrCost(CastInst::ZExt, RetTy, CondTy,
3099 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy,
3100 CostKind);
3101 }
3102 return Cost;
3103 }
3104 case Intrinsic::maximumnum:
3105 case Intrinsic::minimumnum: {
3106 // On platform that support FMAXNUM_IEEE/FMINNUM_IEEE, we expand
3107 // maximumnum/minimumnum to
3108 // ARG0 = fcanonicalize ARG0, ARG0 // to quiet ARG0
3109 // ARG1 = fcanonicalize ARG1, ARG1 // to quiet ARG1
3110 // RESULT = MAXNUM_IEEE ARG0, ARG1 // or MINNUM_IEEE
3111 // FIXME: In LangRef, we claimed FMAXNUM has the same behaviour of
3112 // FMAXNUM_IEEE, while the backend hasn't migrated the code yet.
3113 // Finally, we will remove FMAXNUM_IEEE and FMINNUM_IEEE.
3114 int IeeeISD =
3115 IID == Intrinsic::maximumnum ? ISD::FMAXNUM_IEEE : ISD::FMINNUM_IEEE;
3116 if (TLI->isOperationLegal(IeeeISD, LT.second)) {
3117 IntrinsicCostAttributes FCanonicalizeAttrs(Intrinsic::canonicalize,
3118 RetTy, Tys[0]);
3119 InstructionCost FCanonicalizeCost =
3120 thisT()->getIntrinsicInstrCost(FCanonicalizeAttrs, CostKind);
3121 return LT.first + FCanonicalizeCost * 2;
3122 }
3123 break;
3124 }
3125 case Intrinsic::clmul: {
3126 // This cost model should match the expansion in
3127 // TargetLowering::expandCLMUL.
3128 unsigned BW = RetTy->getScalarSizeInBits();
3129 InstructionCost AndCost =
3130 thisT()->getArithmeticInstrCost(Instruction::And, RetTy, CostKind);
3131 InstructionCost OrCost =
3132 thisT()->getArithmeticInstrCost(Instruction::Or, RetTy, CostKind);
3133 InstructionCost XorCost =
3134 thisT()->getArithmeticInstrCost(Instruction::Xor, RetTy, CostKind);
3135 InstructionCost MulCost =
3136 thisT()->getArithmeticInstrCost(Instruction::Mul, RetTy, CostKind);
3137
3138 // When the multiplication with holes approach is used, that emits 16
3139 // MULs, 8 + 4 ANDs, 12 XORs and 3 ORs.
3140 if (BW >= 32 && BW <= 64 &&
3142 TLI->getValueType(DL, RetTy))) {
3143 return 16 * MulCost + 12 * AndCost + 12 * XorCost + 3 * OrCost;
3144 }
3145
3146 InstructionCost PerBitCostMul = AndCost + MulCost + XorCost;
3147 InstructionCost PerBitCostBittest =
3148 AndCost +
3149 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, RetTy,
3151 thisT()->getCmpSelInstrCost(Instruction::ICmp, RetTy, RetTy,
3153 InstructionCost PerBitCost = std::min(PerBitCostMul, PerBitCostBittest);
3154 return BW * PerBitCost;
3155 }
3156 default:
3157 break;
3158 }
3159
3160 // Else, assume that we need to scalarize this intrinsic. For math builtins
3161 // this will emit a costly libcall, adding call overhead and spills. Make it
3162 // very expensive.
3163 if (isVectorizedTy(RetTy)) {
3164 ArrayRef<Type *> RetVTys = getContainedTypes(RetTy);
3165
3166 // Scalable vectors cannot be scalarized, so return Invalid.
3167 if (any_of(concat<Type *const>(RetVTys, Tys),
3168 [](Type *Ty) { return isa<ScalableVectorType>(Ty); }))
3170
3171 InstructionCost ScalarizationCost = ScalarizationCostPassed;
3172 if (!SkipScalarizationCost) {
3173 ScalarizationCost = 0;
3174 for (Type *RetVTy : RetVTys) {
3175 ScalarizationCost += getScalarizationOverhead(
3176 cast<VectorType>(RetVTy), /*Insert=*/true,
3177 /*Extract=*/false, CostKind);
3178 }
3179 }
3180
3181 unsigned ScalarCalls = getVectorizedTypeVF(RetTy).getFixedValue();
3182 SmallVector<Type *, 4> ScalarTys;
3183 for (Type *Ty : Tys) {
3184 if (Ty->isVectorTy())
3185 Ty = Ty->getScalarType();
3186 ScalarTys.push_back(Ty);
3187 }
3188 IntrinsicCostAttributes Attrs(IID, toScalarizedTy(RetTy), ScalarTys, FMF);
3189 InstructionCost ScalarCost =
3190 thisT()->getIntrinsicInstrCost(Attrs, CostKind);
3191 for (Type *Ty : Tys) {
3192 if (auto *VTy = dyn_cast<VectorType>(Ty)) {
3193 if (!ICA.skipScalarizationCost())
3194 ScalarizationCost += getScalarizationOverhead(
3195 VTy, /*Insert*/ false, /*Extract*/ true, CostKind);
3196 ScalarCalls = std::max(ScalarCalls,
3197 cast<FixedVectorType>(VTy)->getNumElements());
3198 }
3199 }
3200 return ScalarCalls * ScalarCost + ScalarizationCost;
3201 }
3202
3203 // This is going to be turned into a library call, make it expensive.
3204 return SingleCallCost;
3205 }
3206
3207 /// Get memory intrinsic cost based on arguments.
3210 TTI::TargetCostKind CostKind) const override {
3211 unsigned Id = MICA.getID();
3212 Type *DataTy = MICA.getDataType();
3213 bool VariableMask = MICA.getVariableMask();
3214 Align Alignment = MICA.getAlignment();
3215
3216 switch (Id) {
3217 case Intrinsic::experimental_vp_strided_load:
3218 case Intrinsic::experimental_vp_strided_store: {
3219 unsigned Opcode = Id == Intrinsic::experimental_vp_strided_load
3220 ? Instruction::Load
3221 : Instruction::Store;
3222 // For a target without strided memory operations (or for an illegal
3223 // operation type on one which does), assume we lower to a gather/scatter
3224 // operation. (Which may in turn be scalarized.)
3225 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3226 VariableMask, true, CostKind);
3227 }
3228 case Intrinsic::masked_scatter:
3229 case Intrinsic::masked_gather:
3230 case Intrinsic::vp_scatter:
3231 case Intrinsic::vp_gather: {
3232 unsigned Opcode = (MICA.getID() == Intrinsic::masked_gather ||
3233 MICA.getID() == Intrinsic::vp_gather)
3234 ? Instruction::Load
3235 : Instruction::Store;
3236
3237 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3238 VariableMask, true, CostKind);
3239 }
3240 case Intrinsic::vp_load:
3241 case Intrinsic::vp_store:
3243 case Intrinsic::masked_load:
3244 case Intrinsic::masked_store: {
3245 unsigned Opcode =
3246 Id == Intrinsic::masked_load ? Instruction::Load : Instruction::Store;
3247 // TODO: Pass on AddressSpace when we have test coverage.
3248 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment, true, false,
3249 CostKind);
3250 }
3251 case Intrinsic::masked_compressstore:
3252 case Intrinsic::masked_expandload: {
3253 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
3254 ? Instruction::Load
3255 : Instruction::Store;
3256 // Treat expand load/compress store as gather/scatter operation.
3257 // TODO: implement more precise cost estimation for these intrinsics.
3258 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3259 VariableMask,
3260 /*IsGatherScatter*/ true, CostKind);
3261 }
3262 case Intrinsic::vp_load_ff:
3264 default:
3265 llvm_unreachable("unexpected intrinsic");
3266 }
3267 }
3268
3269 /// Compute a cost of the given call instruction.
3270 ///
3271 /// Compute the cost of calling function F with return type RetTy and
3272 /// argument types Tys. F might be nullptr, in this case the cost of an
3273 /// arbitrary call with the specified signature will be returned.
3274 /// This is used, for instance, when we estimate call of a vector
3275 /// counterpart of the given function.
3276 /// \param F Called function, might be nullptr.
3277 /// \param RetTy Return value types.
3278 /// \param Tys Argument types.
3279 /// \returns The cost of Call instruction.
3282 TTI::TargetCostKind CostKind) const override {
3283 return 10;
3284 }
3285
3286 unsigned getNumberOfParts(Type *Tp) const override {
3287 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
3288 if (!LT.first.isValid())
3289 return 0;
3290 // Try to find actual number of parts for non-power-of-2 elements as
3291 // ceil(num-of-elements/num-of-subtype-elements).
3292 if (auto *FTp = dyn_cast<FixedVectorType>(Tp);
3293 Tp && LT.second.isFixedLengthVector() &&
3294 !has_single_bit(FTp->getNumElements())) {
3295 if (auto *SubTp = dyn_cast_if_present<FixedVectorType>(
3296 EVT(LT.second).getTypeForEVT(Tp->getContext()));
3297 SubTp && SubTp->getElementType() == FTp->getElementType())
3298 return divideCeil(FTp->getNumElements(), SubTp->getNumElements());
3299 }
3300 return LT.first.getValue();
3301 }
3302
3305 TTI::TargetCostKind) const override {
3306 return 0;
3307 }
3308
3309 /// Try to calculate arithmetic and shuffle op costs for reduction intrinsics.
3310 /// We're assuming that reduction operation are performing the following way:
3311 ///
3312 /// %val1 = shufflevector<n x t> %val, <n x t> %undef,
3313 /// <n x i32> <i32 n/2, i32 n/2 + 1, ..., i32 n, i32 undef, ..., i32 undef>
3314 /// \----------------v-------------/ \----------v------------/
3315 /// n/2 elements n/2 elements
3316 /// %red1 = op <n x t> %val, <n x t> val1
3317 /// After this operation we have a vector %red1 where only the first n/2
3318 /// elements are meaningful, the second n/2 elements are undefined and can be
3319 /// dropped. All other operations are actually working with the vector of
3320 /// length n/2, not n, though the real vector length is still n.
3321 /// %val2 = shufflevector<n x t> %red1, <n x t> %undef,
3322 /// <n x i32> <i32 n/4, i32 n/4 + 1, ..., i32 n/2, i32 undef, ..., i32 undef>
3323 /// \----------------v-------------/ \----------v------------/
3324 /// n/4 elements 3*n/4 elements
3325 /// %red2 = op <n x t> %red1, <n x t> val2 - working with the vector of
3326 /// length n/2, the resulting vector has length n/4 etc.
3327 ///
3328 /// The cost model should take into account that the actual length of the
3329 /// vector is reduced on each iteration.
3332 // Targets must implement a default value for the scalable case, since
3333 // we don't know how many lanes the vector has.
3336
3337 Type *ScalarTy = Ty->getElementType();
3338 unsigned NumVecElts = cast<FixedVectorType>(Ty)->getNumElements();
3339 if ((Opcode == Instruction::Or || Opcode == Instruction::And) &&
3340 ScalarTy == IntegerType::getInt1Ty(Ty->getContext()) &&
3341 NumVecElts >= 2) {
3342 // Or reduction for i1 is represented as:
3343 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
3344 // %res = cmp ne iReduxWidth %val, 0
3345 // And reduction for i1 is represented as:
3346 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
3347 // %res = cmp eq iReduxWidth %val, 11111
3348 Type *ValTy = IntegerType::get(Ty->getContext(), NumVecElts);
3349 return thisT()->getCastInstrCost(Instruction::BitCast, ValTy, Ty,
3351 thisT()->getCmpSelInstrCost(Instruction::ICmp, ValTy,
3354 }
3355 unsigned NumReduxLevels = Log2_32(NumVecElts);
3356 InstructionCost ArithCost = 0;
3357 InstructionCost ShuffleCost = 0;
3358 std::pair<InstructionCost, MVT> LT = thisT()->getTypeLegalizationCost(Ty);
3359 unsigned LongVectorCount = 0;
3360 unsigned MVTLen =
3361 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
3362 while (NumVecElts > MVTLen) {
3363 NumVecElts /= 2;
3364 VectorType *SubTy = FixedVectorType::get(ScalarTy, NumVecElts);
3365 ShuffleCost += thisT()->getShuffleCost(
3366 TTI::SK_ExtractSubvector, SubTy, Ty, {}, CostKind, NumVecElts, SubTy);
3367 ArithCost += thisT()->getArithmeticInstrCost(Opcode, SubTy, CostKind);
3368 Ty = SubTy;
3369 ++LongVectorCount;
3370 }
3371
3372 NumReduxLevels -= LongVectorCount;
3373
3374 // The minimal length of the vector is limited by the real length of vector
3375 // operations performed on the current platform. That's why several final
3376 // reduction operations are performed on the vectors with the same
3377 // architecture-dependent length.
3378
3379 // By default reductions need one shuffle per reduction level.
3380 ShuffleCost +=
3381 NumReduxLevels * thisT()->getShuffleCost(TTI::SK_PermuteSingleSrc, Ty,
3382 Ty, {}, CostKind, 0, Ty);
3383 ArithCost +=
3384 NumReduxLevels * thisT()->getArithmeticInstrCost(Opcode, Ty, CostKind);
3385 return ShuffleCost + ArithCost +
3386 thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
3387 CostKind, 0, nullptr, nullptr);
3388 }
3389
3390 /// Try to calculate the cost of performing strict (in-order) reductions,
3391 /// which involves doing a sequence of floating point additions in lane
3392 /// order, starting with an initial value. For example, consider a scalar
3393 /// initial value 'InitVal' of type float and a vector of type <4 x float>:
3394 ///
3395 /// Vector = <float %v0, float %v1, float %v2, float %v3>
3396 ///
3397 /// %add1 = %InitVal + %v0
3398 /// %add2 = %add1 + %v1
3399 /// %add3 = %add2 + %v2
3400 /// %add4 = %add3 + %v3
3401 ///
3402 /// As a simple estimate we can say the cost of such a reduction is 4 times
3403 /// the cost of a scalar FP addition. We can only estimate the costs for
3404 /// fixed-width vectors here because for scalable vectors we do not know the
3405 /// runtime number of operations.
3408 // Targets must implement a default value for the scalable case, since
3409 // we don't know how many lanes the vector has.
3412
3413 auto *VTy = cast<FixedVectorType>(Ty);
3415 VTy, /*Insert=*/false, /*Extract=*/true, CostKind);
3416 InstructionCost ArithCost = thisT()->getArithmeticInstrCost(
3417 Opcode, VTy->getElementType(), CostKind);
3418 ArithCost *= VTy->getNumElements();
3419
3420 return ExtractCost + ArithCost;
3421 }
3422
3425 std::optional<FastMathFlags> FMF,
3426 TTI::TargetCostKind CostKind) const override {
3427 assert(Ty && "Unknown reduction vector type");
3429 return getOrderedReductionCost(Opcode, Ty, CostKind);
3430 return getTreeReductionCost(Opcode, Ty, CostKind);
3431 }
3432
3433 /// Try to calculate op costs for min/max reduction operations.
3434 /// \param CondTy Conditional type for the Select instruction.
3437 TTI::TargetCostKind CostKind) const override {
3438 // Targets must implement a default value for the scalable case, since
3439 // we don't know how many lanes the vector has.
3442
3443 Type *ScalarTy = Ty->getElementType();
3444 unsigned NumVecElts = cast<FixedVectorType>(Ty)->getNumElements();
3445 unsigned NumReduxLevels = Log2_32(NumVecElts);
3446 InstructionCost MinMaxCost = 0;
3447 InstructionCost ShuffleCost = 0;
3448 std::pair<InstructionCost, MVT> LT = thisT()->getTypeLegalizationCost(Ty);
3449 unsigned LongVectorCount = 0;
3450 unsigned MVTLen =
3451 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
3452 while (NumVecElts > MVTLen) {
3453 NumVecElts /= 2;
3454 auto *SubTy = FixedVectorType::get(ScalarTy, NumVecElts);
3455
3456 ShuffleCost += thisT()->getShuffleCost(
3457 TTI::SK_ExtractSubvector, SubTy, Ty, {}, CostKind, NumVecElts, SubTy);
3458
3459 IntrinsicCostAttributes Attrs(IID, SubTy, {SubTy, SubTy}, FMF);
3460 MinMaxCost += getIntrinsicInstrCost(Attrs, CostKind);
3461 Ty = SubTy;
3462 ++LongVectorCount;
3463 }
3464
3465 NumReduxLevels -= LongVectorCount;
3466
3467 // The minimal length of the vector is limited by the real length of vector
3468 // operations performed on the current platform. That's why several final
3469 // reduction opertions are perfomed on the vectors with the same
3470 // architecture-dependent length.
3471 ShuffleCost +=
3472 NumReduxLevels * thisT()->getShuffleCost(TTI::SK_PermuteSingleSrc, Ty,
3473 Ty, {}, CostKind, 0, Ty);
3474 IntrinsicCostAttributes Attrs(IID, Ty, {Ty, Ty}, FMF);
3475 MinMaxCost += NumReduxLevels * getIntrinsicInstrCost(Attrs, CostKind);
3476 // The last min/max should be in vector registers and we counted it above.
3477 // So just need a single extractelement.
3478 return ShuffleCost + MinMaxCost +
3479 thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
3480 CostKind, 0, nullptr, nullptr);
3481 }
3482
3484 getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy,
3485 VectorType *Ty, std::optional<FastMathFlags> FMF,
3486 TTI::TargetCostKind CostKind) const override {
3487 if (auto *FTy = dyn_cast<FixedVectorType>(Ty);
3488 FTy && IsUnsigned && Opcode == Instruction::Add &&
3489 FTy->getElementType() == IntegerType::getInt1Ty(Ty->getContext())) {
3490 // Represent vector_reduce_add(ZExt(<n x i1>)) as
3491 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
3492 auto *IntTy =
3493 IntegerType::get(ResTy->getContext(), FTy->getNumElements());
3494 IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy},
3495 FMF ? *FMF : FastMathFlags());
3496 return thisT()->getCastInstrCost(Instruction::BitCast, IntTy, FTy,
3498 thisT()->getIntrinsicInstrCost(ICA, CostKind);
3499 }
3500 // Without any native support, this is equivalent to the cost of
3501 // vecreduce.opcode(ext(Ty A)).
3502 VectorType *ExtTy = VectorType::get(ResTy, Ty);
3503 InstructionCost RedCost =
3504 thisT()->getArithmeticReductionCost(Opcode, ExtTy, FMF, CostKind);
3505 InstructionCost ExtCost = thisT()->getCastInstrCost(
3506 IsUnsigned ? Instruction::ZExt : Instruction::SExt, ExtTy, Ty,
3508
3509 return RedCost + ExtCost;
3510 }
3511
3513 getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy,
3514 VectorType *Ty,
3515 TTI::TargetCostKind CostKind) const override {
3516 // Without any native support, this is equivalent to the cost of
3517 // vecreduce.add(mul(ext(Ty A), ext(Ty B))) or
3518 // vecreduce.add(mul(A, B)).
3519 assert((RedOpcode == Instruction::Add || RedOpcode == Instruction::Sub) &&
3520 "The reduction opcode is expected to be Add or Sub.");
3521 VectorType *ExtTy = VectorType::get(ResTy, Ty);
3522 InstructionCost RedCost = thisT()->getArithmeticReductionCost(
3523 RedOpcode, ExtTy, std::nullopt, CostKind);
3524 InstructionCost ExtCost = thisT()->getCastInstrCost(
3525 IsUnsigned ? Instruction::ZExt : Instruction::SExt, ExtTy, Ty,
3527
3528 InstructionCost MulCost =
3529 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
3530
3531 return RedCost + MulCost + 2 * ExtCost;
3532 }
3533
3535 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
3537 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
3539 std::optional<FastMathFlags> FMF) const override {
3540 unsigned EltSizeAcc = AccumType->getScalarSizeInBits();
3541 unsigned EltSizeInA = InputTypeA->getScalarSizeInBits();
3542 unsigned Ratio = EltSizeAcc / EltSizeInA;
3543 if (VF.getKnownMinValue() <= Ratio || VF.getKnownMinValue() % Ratio != 0 ||
3544 EltSizeAcc % EltSizeInA != 0 || (BinOp && InputTypeA != InputTypeB))
3546
3547 Type *InputVectorType = VectorType::get(InputTypeA, VF);
3548 Type *ExtInputVectorType = VectorType::get(AccumType, VF);
3549 Type *AccumVectorType =
3550 VectorType::get(AccumType, VF.divideCoefficientBy(Ratio));
3551
3552 InstructionCost ExtendCostA = 0;
3554 ExtendCostA = getCastInstrCost(
3556 ExtInputVectorType, InputVectorType, TTI::CastContextHint::None,
3557 CostKind);
3558
3559 // TODO: add cost of extracting subvectors from the source vector that
3560 // is to be partially reduced.
3561 InstructionCost ReductionOpCost =
3562 Ratio * getArithmeticInstrCost(Opcode, AccumVectorType, CostKind);
3563
3564 if (!BinOp)
3565 return ExtendCostA + ReductionOpCost;
3566
3567 InstructionCost ExtendCostB = 0;
3569 ExtendCostB = getCastInstrCost(
3571 ExtInputVectorType, InputVectorType, TTI::CastContextHint::None,
3572 CostKind);
3573 return ExtendCostA + ExtendCostB + ReductionOpCost +
3574 getArithmeticInstrCost(*BinOp, ExtInputVectorType, CostKind);
3575 }
3576
3578
3579 /// @}
3580};
3581
3582/// Concrete BasicTTIImpl that can be used if no further customization
3583/// is needed.
3584class BasicTTIImpl : public BasicTTIImplBase<BasicTTIImpl> {
3585 using BaseT = BasicTTIImplBase<BasicTTIImpl>;
3586
3587 friend class BasicTTIImplBase<BasicTTIImpl>;
3588
3589 const TargetSubtargetInfo *ST;
3590 const TargetLoweringBase *TLI;
3591
3592 const TargetSubtargetInfo *getST() const { return ST; }
3593 const TargetLoweringBase *getTLI() const { return TLI; }
3594
3595public:
3596 LLVM_ABI explicit BasicTTIImpl(const TargetMachine *TM, const Function &F);
3597};
3598
3599} // end namespace llvm
3600
3601#endif // LLVM_CODEGEN_BASICTTIIMPL_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
This file implements the BitVector class.
#define LLVM_ABI
Definition Compiler.h:215
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
#define T
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
static Type * getValueType(Value *V, bool LookThroughCmp=false)
Returns the "element type" of the given value/instruction V.
This file contains some templates that are useful if you are working with the STL at all.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This file describes how to lower LLVM code to machine code.
This file provides helpers for the implementation of a TargetTransformInfo-conforming class.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:235
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1355
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1210
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1513
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1139
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:201
an instruction to allocate memory on the stack
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
ArrayRef< T > drop_back(size_t N=1) const
Drop the last N elements of the array.
Definition ArrayRef.h:200
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
InstructionCost getFPOpCost(Type *Ty) const override
bool preferToKeepConstantsAttached(const Instruction &Inst, const Function &Fn) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool isIndexedLoadLegal(TTI::MemIndexedMode M, Type *Ty) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, Type *AccessType, TTI::TargetCostKind CostKind) const override
unsigned getCallerAllocaCost(const CallBase *CB, const AllocaInst *AI) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool shouldBuildLookupTables() const override
bool isNoopAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override
bool isProfitableToHoist(Instruction *I) const override
unsigned getNumberOfParts(Type *Tp) const override
unsigned getMinPrefetchStride(unsigned NumMemAccesses, unsigned NumStridedMemAccesses, unsigned NumPrefetches, bool HasCall) const override
bool useAA() const override
unsigned getPrefetchDistance() const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
InstructionCost getOperandsScalarizationOverhead(ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction's operands.
bool isLegalAddScalableImmediate(int64_t Imm) const override
bool haveFastClmul(IntegerType *Ty) const override
unsigned getAssumedAddrSpace(const Value *V) const override
std::optional< Value * > simplifyDemandedUseBitsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed) const override
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
bool isIndexedStoreLegal(TTI::MemIndexedMode M, Type *Ty) const override
bool haveFastSqrt(Type *Ty) const override
bool collectFlatAddressOperands(SmallVectorImpl< int > &OpIndexes, Intrinsic::ID IID) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
unsigned getEstimatedNumberOfCaseClusters(const SwitchInst &SI, unsigned &JumpTableSize, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) const override
unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy, Align Alignment, unsigned AddrSpace) const override
Value * rewriteIntrinsicWithAddressSpace(IntrinsicInst *II, Value *OldV, Value *NewV) const override
unsigned adjustInliningThreshold(const CallBase *CB) const override
unsigned getInliningThresholdMultiplier() const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, Value *Scalar, ArrayRef< std::tuple< Value *, User *, int > > ScalarUserAndIdx, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset)
bool shouldBuildRelLookupTables() const override
bool isTargetIntrinsicWithStructReturnOverloadAtField(Intrinsic::ID ID, int RetIdx) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getVectorInstrCost(const Instruction &I, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
InstructionCost getVectorSplitCost() const
bool isTruncateFree(Type *Ty1, Type *Ty2) const override
std::optional< unsigned > getMaxVScale() const override
unsigned getFlatAddressSpace() const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
Compute a cost of the given call instruction.
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
InstructionCost getTreeReductionCost(unsigned Opcode, VectorType *Ty, TTI::TargetCostKind CostKind) const
Try to calculate arithmetic and shuffle op costs for reduction intrinsics.
~BasicTTIImplBase() override=default
std::pair< const Value *, unsigned > getPredicatedAddrSpace(const Value *V) const override
unsigned getMaxPrefetchIterationsAhead() const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getTypeBasedIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const
Get intrinsic cost based on argument types.
bool hasBranchDivergence(const Function *F=nullptr) const override
InstructionCost getOrderedReductionCost(unsigned Opcode, VectorType *Ty, TTI::TargetCostKind CostKind) const
Try to calculate the cost of performing strict (in-order) reductions, which involves doing a sequence...
std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const override
bool shouldPrefetchAddressSpace(unsigned AS) const override
bool allowsMisalignedMemoryAccesses(LLVMContext &Context, unsigned BitWidth, unsigned AddressSpace, Align Alignment, unsigned *Fast) const override
unsigned getCacheLineSize() const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldDropLSRSolutionIfLessProfitable() const override
int getInlinerVectorBonusPercent() const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
Estimate the cost of type-legalization and the legalized type.
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF, const APInt &DemandedDstElts, TTI::TargetCostKind CostKind) const override
bool isSingleThreaded() const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isProfitableLSRChainElement(Instruction *I) const override
bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override
bool isTargetIntrinsicWithOverloadTypeAtArg(Intrinsic::ID ID, int OpdIdx) const override
bool isTargetIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const override
InstructionCost getScalarizationOverhead(VectorType *RetTy, ArrayRef< const Value * > Args, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const
Estimate the overhead of scalarizing the inputs and outputs of an instruction, with return type RetTy...
TailFoldingStyle getPreferredTailFoldingStyle() const override
std::optional< unsigned > getCacheSize(TargetTransformInfo::CacheLevel Level) const override
bool isLegalICmpImmediate(int64_t imm) const override
bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
BasicTTIImplBase(const TargetMachine *TM, const DataLayout &DL)
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
bool enableWritePrefetching() const override
bool isLSRCostLess(const TTI::LSRCost &C1, const TTI::LSRCost &C2) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
Helper wrapper for the DemandedElts variant of getScalarizationOverhead.
InstructionCost getBranchMispredictPenalty() const override
bool isNumRegsMajorCostOfLSR() const override
LLVM_ABI BasicTTIImpl(const TargetMachine *TM, const Function &F)
size_type count() const
Returns the number of bits which are set.
Definition BitVector.h:181
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static CmpInst::Predicate getGTPredicate(Intrinsic::ID ID)
static CmpInst::Predicate getLTPredicate(Intrinsic::ID ID)
This class represents a range of values.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:324
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:309
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:320
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Container class for subtarget features.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:328
The core instruction combiner logic.
static InstructionCost getInvalid(CostType Val=0)
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
InstructionCost getScalarizationCost() const
const IntrinsicInst * getInst() const
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
const FeatureBitset & getFeatureBits() const
Machine Value Type.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Information for memory intrinsic cost model.
The optimization diagnostic interface.
LLVM_ABI void emit(DiagnosticInfoOptimizationBase &OptDiag)
Output the remark via the diagnostic handler and to the optimization record file.
Diagnostic information for applied optimization remarks.
static LLVM_ABI PointerType * get(Type *ElementType, unsigned AddressSpace)
This constructs a pointer to an object of the specified type in a numbered address space.
Analysis providing profile information.
This class represents an analyzed expression in the program.
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isSpliceMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is a splice mask, concatenating the two inputs together and then ext...
static LLVM_ABI bool isSelectMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from its source vectors without lane crossings.
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isReverseMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask swaps the order of elements from exactly one source vector.
static LLVM_ABI bool isTransposeMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask is a transpose mask.
static LLVM_ABI bool isInsertSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &NumSubElts, int &Index)
Return true if this shuffle mask is an insert subvector mask.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
static StackOffset getScalable(int64_t Scalable)
Definition TypeSize.h:40
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
Definition Type.cpp:683
Multiway switch.
Provides information about what library functions are available for the current target.
This base class for TargetLowering contains the SelectionDAG-independent parts that can be used from ...
bool isOperationExpand(unsigned Op, EVT VT) const
Return true if the specified operation is illegal on this target or unlikely to be made legal with cu...
int InstructionOpcodeToISD(unsigned Opcode) const
Get the ISD node that corresponds to the Instruction class opcode.
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
LegalizeAction
This enum indicates whether operations are valid for a target, and if not, what action should be used...
virtual bool preferSelectsOverBooleanArithmetic(EVT VT) const
Should we prefer selects to doing arithmetic on boolean types.
virtual bool isZExtFree(Type *FromTy, Type *ToTy) const
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
virtual bool isSuitableForJumpTable(const SwitchInst *SI, uint64_t NumCases, uint64_t Range, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) const
Return true if lowering to a jump table is suitable for a set of case clusters which may contain NumC...
virtual bool areJTsAllowed(const Function *Fn) const
Return true if lowering to a jump table is allowed.
bool isOperationLegalOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal using promotion.
LegalizeAction getTruncStoreAction(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace) const
Return how this store with truncation should be treated: either it is legal, needs to be promoted to ...
bool isOperationCustom(unsigned Op, EVT VT) const
Return true if the operation uses custom lowering, regardless of whether the type is legal or not.
bool isSuitableForBitTests(const DenseMap< const BasicBlock *, unsigned int > &DestCmps, const APInt &Low, const APInt &High, const DataLayout &DL) const
Return true if lowering to a bit test is suitable for a set of case clusters which contains NumDests ...
virtual bool isTruncateFree(Type *FromTy, Type *ToTy) const
Return true if it's free to truncate a value of type FromTy to type ToTy.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
virtual bool isFreeAddrSpaceCast(unsigned SrcAS, unsigned DestAS) const
Returns true if a cast from SrcAS to DestAS is "cheap", such that e.g.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
LegalizeAction getLoadAction(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace, unsigned ExtType, bool Atomic) const
Return how this load with extension should be treated: either it is legal, needs to be promoted to a ...
LegalizeTypeAction getTypeAction(LLVMContext &Context, EVT VT) const
Return how we should legalize values of this type, either it is already legal (return 'Legal') or we ...
bool isLoadLegal(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace, unsigned ExtType, bool Atomic) const
Return true if the specified load with extension is legal on this target.
virtual bool isFAbsFree(EVT VT) const
Return true if an fabs operation is free to the point where it is never worthwhile to replace it with...
bool isOperationLegalOrCustomOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
Primary interface to the complete machine description for the target machine.
bool isPositionIndependent() const
const Triple & getTargetTriple() const
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
CodeModel::Model getCodeModel() const
Returns the code model.
TargetSubtargetInfo - Generic base class for all target subtargets.
virtual const FeatureBitset & getInlineMustMatchFeatures() const =0
Target features where all mismatches prevent inlining.
virtual const FeatureBitset & getInlineInverseFeatures() const =0
Target features where the callee may have an additional feature, instead of the caller.
virtual const FeatureBitset & getInlineIgnoreFeatures() const =0
Target features to ignore for inline compatibility check.
virtual bool isProfitableLSRChainElement(Instruction *I) const
virtual TailFoldingStyle getPreferredTailFoldingStyle() const
virtual const DataLayout & getDataLayout() const
virtual std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const
virtual std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const
virtual bool shouldDropLSRSolutionIfLessProfitable() const
virtual bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const
virtual std::optional< Value * > simplifyDemandedUseBitsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed) const
virtual bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const
virtual std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const
virtual unsigned getEpilogueVectorizationMinVF() const
virtual InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual bool isLoweredToCall(const Function *F) const
virtual InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI=nullptr) const
virtual InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const
virtual bool isLSRCostLess(const TTI::LSRCost &C1, const TTI::LSRCost &C2) const
virtual InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I) const
virtual InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const
virtual InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, const Instruction *I) const
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, Type *AccessType, TTI::TargetCostKind CostKind) const override
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Basic
The cost of a typical 'add' instruction.
static LLVM_ABI Instruction::CastOps getOpcodeForPartialReductionExtendKind(PartialReductionExtendKind Kind)
Get the cast opcode for an extension kind.
MemIndexedMode
The type of load/store indexing.
static LLVM_ABI VectorInstrContext getVectorInstrContextHint(const Instruction *I)
Calculates a VectorInstrContext from I.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
@ Normal
The cast is used with a normal load/store.
CacheLevel
The possible cache levels.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
ArchType getArch() const
Get the parsed architecture type of this triple.
Definition Triple.h:512
LLVM_ABI bool isArch64Bit() const
Test whether the architecture is 64-bit.
Definition Triple.cpp:1865
bool isOSDarwin() const
Is this a "Darwin" OS (macOS, iOS, tvOS, watchOS, DriverKit, XROS, or bridgeOS).
Definition Triple.h:721
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:343
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:307
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:306
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:227
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:397
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
Value * getOperand(unsigned i) const
Definition User.h:207
static LLVM_ABI bool isVPBinOp(Intrinsic::ID ID)
static LLVM_ABI bool isVPCast(Intrinsic::ID ID)
static LLVM_ABI bool isVPCmp(Intrinsic::ID ID)
static LLVM_ABI std::optional< unsigned > getFunctionalOpcodeForVP(Intrinsic::ID ID)
static LLVM_ABI std::optional< Intrinsic::ID > getFunctionalIntrinsicIDForVP(Intrinsic::ID ID)
static LLVM_ABI bool isVPIntrinsic(Intrinsic::ID)
static LLVM_ABI bool isVPReduction(Intrinsic::ID ID)
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
Base class of all SIMD vector types.
static VectorType * getHalfElementsVectorType(VectorType *VTy)
This static method returns a VectorType with half as many elements as the input type and the same ele...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
LLVM_ABI APInt ScaleBitMask(const APInt &A, unsigned NewBitWidth, bool MatchAllBits=false)
Splat/Merge neighboring bits to widen/narrow the bitmask represented by.
Definition APInt.cpp:3040
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:24
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:789
@ SMULFIX
RESULT = [US]MULFIX(LHS, RHS, SCALE) - Perform fixed point multiplication on 2 integers with the same...
Definition ISDOpcodes.h:394
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ FMODF
FMODF - Decomposes the operand into integral and fractional parts, each having the same type and sign...
@ FATAN2
FATAN2 - atan2, inspired by libm.
@ FSINCOSPI
FSINCOSPI - Compute both the sine and cosine times pi more accurately than FSINCOS(pi*x),...
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:749
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ CLMUL
Carry-less multiplication operations.
Definition ISDOpcodes.h:780
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:352
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:543
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:374
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:806
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:348
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
Definition ISDOpcodes.h:356
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:729
@ MASKED_UDIV
Masked vector arithmetic that returns poison on disabled lanes.
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:815
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ SCMP
[US]CMP - 3-way comparison of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:955
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:365
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
MemIndexedMode
MemIndexedMode enum - This enum defines the load / store indexed addressing modes.
LLVM_ABI bool isTargetIntrinsic(ID IID)
isTargetIntrinsic - Returns true if IID is an intrinsic specific to a certain target.
LLVM_ABI Libcall getSINCOSPI(EVT RetVT)
getSINCOSPI - Return the SINCOSPI_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getMODF(EVT VT)
getMODF - Return the MODF_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getSINCOS(EVT RetVT)
getSINCOS - Return the SINCOS_* value for the given types, or UNKNOWN_LIBCALL if there is none.
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:840
InstructionCost Cost
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
Type * toScalarizedTy(Type *Ty)
A helper for converting vectorized types to scalarized (non-vector) types.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_ABI unsigned getArithmeticReductionInstruction(Intrinsic::ID RdxID)
Returns the arithmetic instruction opcode used when expanding a reduction.
bool isVectorizedTy(Type *Ty)
Returns true if Ty is a vector type or a struct of vector types where all vector types share the same...
detail::concat_range< ValueT, RangeTs... > concat(RangeTs &&...Ranges)
Returns a concatenated range across two or more ranges.
Definition STLExtras.h:1151
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
ElementCount getVectorizedTypeVF(Type *Ty)
Returns the number of vector elements for a vectorized type.
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:395
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
ArrayRef< Type * > getContainedTypes(Type *const &Ty)
Returns the types contained in Ty.
LLVM_ABI cl::opt< unsigned > PartialUnrollingThreshold
LLVM_ABI bool isVectorizedStructTy(StructType *StructTy)
Returns true if StructTy is an unpacked literal struct where all elements are vectors of matching ele...
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
Attributes of a target dependent hardware loop.
static LLVM_ABI bool hasVectorMaskArgument(RTLIB::LibcallImpl Impl)
Returns true if the function has a vector mask argument, which is assumed to be the last argument.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
bool AllowPeeling
Allow peeling off loop iterations.
bool AllowLoopNestsPeeling
Allow peeling off loop iterations for loop nests.
bool PeelProfiledIterations
Allow peeling basing on profile.
unsigned PeelCount
A forced peeling factor (the number of bodied of the original loop that should be peeled off before t...
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
unsigned PartialThreshold
The cost threshold for the unrolled loop, like Threshold, but used for partial/runtime unrolling (set...
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).