LLVM 24.0.0git
SLPCostAnalysis.cpp
Go to the documentation of this file.
1//===- SLPCostAnalysis.cpp - SLP Vectorizer free cost helpers -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "SLPCostAnalysis.h"
10#include "SLPTypeUtils.h"
11#include "SLPUtils.h"
12
13#include "llvm/ADT/APInt.h"
14#include "llvm/ADT/STLExtras.h"
15#include "llvm/ADT/Sequence.h"
18#include "llvm/IR/Constants.h"
19#include "llvm/IR/DataLayout.h"
23#include "llvm/IR/Intrinsics.h"
24#include "llvm/IR/Operator.h"
26#include "llvm/IR/Type.h"
27#include "llvm/IR/Value.h"
30
31#include <cassert>
32#include <utility>
33
34using namespace llvm;
35using namespace llvm::PatternMatch;
36
37namespace llvm::slpvectorizer {
38
42 ArrayRef<int> Mask, int Index, VectorType *SubTp,
45 VectorType *DstTy = Tp;
46 if (!Mask.empty())
47 DstTy = FixedVectorType::get(Tp->getScalarType(), Mask.size());
48
49 if (Kind != TTI::SK_PermuteTwoSrc)
50 return TTI.getShuffleCost(Kind, DstTy, Tp, CostKind, Mask, Index, SubTp,
51 Args, /*CtxI=*/nullptr, VIC);
52 int NumSrcElts = Tp->getElementCount().getKnownMinValue();
53 int NumSubElts;
54 if (Mask.size() > 2 && ShuffleVectorInst::isInsertSubvectorMask(
55 Mask, NumSrcElts, NumSubElts, Index)) {
56 if (Index + NumSubElts > NumSrcElts &&
57 Index + NumSrcElts <= static_cast<int>(Mask.size()))
58 return TTI.getShuffleCost(TTI::SK_InsertSubvector, DstTy, Tp, CostKind,
59 Mask, Index, Tp);
60 }
61 return TTI.getShuffleCost(Kind, DstTy, Tp, CostKind, Mask, Index, SubTp, Args,
62 /*CtxI=*/nullptr, VIC);
63}
64
65std::pair<InstructionCost, InstructionCost>
67 Value *BasePtr, unsigned Opcode, const TTI::TargetCostKind CostKind,
68 Type *ScalarTy, VectorType *VecTy) {
69 InstructionCost ScalarCost = 0;
70 InstructionCost VecCost = 0;
71 // Here we differentiate two cases: (1) when Ptrs represent a regular
72 // vectorization tree node (as they are pointer arguments of scattered
73 // loads) or (2) when Ptrs are the arguments of loads or stores being
74 // vectorized as plane wide unit-stride load/store since all the
75 // loads/stores are known to be from/to adjacent locations.
76 if (Opcode == Instruction::Load || Opcode == Instruction::Store) {
77 // Case 2: estimate costs for pointer related costs when vectorizing to
78 // a wide load/store.
79 // Scalar cost is estimated as a set of pointers with known relationship
80 // between them.
81 // For vector code we will use BasePtr as argument for the wide load/store
82 // but we also need to account all the instructions which are going to
83 // stay in vectorized code due to uses outside of these scalar
84 // loads/stores.
85 ScalarCost = TTI.getPointersChainCost(
86 Ptrs, BasePtr, TTI::PointersChainInfo::getUnitStride(), ScalarTy,
87 CostKind);
88
89 SmallVector<const Value *> PtrsRetainedInVecCode;
90 for (Value *V : Ptrs) {
91 if (V == BasePtr) {
92 PtrsRetainedInVecCode.push_back(V);
93 continue;
94 }
95 auto *Ptr = dyn_cast<GetElementPtrInst>(V);
96 // For simplicity assume Ptr to stay in vectorized code if it's not a
97 // GEP instruction. We don't care since it's cost considered free.
98 // TODO: We should check for any uses outside of vectorizable tree
99 // rather than just single use.
100 if (!Ptr || !Ptr->hasOneUse())
101 PtrsRetainedInVecCode.push_back(V);
102 }
103
104 if (PtrsRetainedInVecCode.size() == Ptrs.size()) {
105 // If all pointers stay in vectorized code then we don't have
106 // any savings on that.
107 return std::make_pair(TTI::TCC_Free, TTI::TCC_Free);
108 }
109 VecCost = TTI.getPointersChainCost(PtrsRetainedInVecCode, BasePtr,
110 TTI::PointersChainInfo::getKnownStride(),
111 VecTy, CostKind);
112 } else {
113 // Case 1: Ptrs are the arguments of loads that we are going to transform
114 // into masked gather load intrinsic.
115 // All the scalar GEPs will be removed as a result of vectorization.
116 // For any external uses of some lanes extract element instructions will
117 // be generated (which cost is estimated separately).
118 TTI::PointersChainInfo PtrsInfo =
119 all_of(Ptrs,
120 [](const Value *V) {
121 auto *Ptr = dyn_cast<GetElementPtrInst>(V);
122 return Ptr && !Ptr->hasAllConstantIndices();
123 })
124 ? TTI::PointersChainInfo::getUnknownStride()
125 : TTI::PointersChainInfo::getKnownStride();
126
127 // The GEPs of the masked gather loads are accessed with the loaded type
128 // and form a chain only if the lanes share the base.
129 Type *AccessTy = ScalarTy;
130 if (all_of(Ptrs, [](const Value *V) {
131 auto *Ptr = dyn_cast<GetElementPtrInst>(V);
132 return Ptr && Ptr->hasOneUse() && isa<LoadInst>(Ptr->user_back());
133 })) {
134 PtrsInfo.IsSameBaseAddress = all_equal(map_range(Ptrs, [](Value *V) {
135 return cast<GetElementPtrInst>(V)->getPointerOperand();
136 }));
137 AccessTy = Ptrs.front()->user_back()->getType();
138 }
139 ScalarCost =
140 TTI.getPointersChainCost(Ptrs, BasePtr, PtrsInfo, AccessTy, CostKind);
141 auto *BaseGEP = dyn_cast<GEPOperator>(BasePtr);
142 if (!BaseGEP) {
143 auto *It = find_if(Ptrs, IsaPred<GEPOperator>);
144 if (It != Ptrs.end())
145 BaseGEP = cast<GEPOperator>(*It);
146 }
147 if (BaseGEP) {
148 SmallVector<const Value *> Indices(BaseGEP->indices());
149 VecCost = TTI.getGEPCost(BaseGEP->getSourceElementType(),
150 BaseGEP->getPointerOperand(), Indices, CostKind,
151 VecTy);
152 }
153 }
154
155 return std::make_pair(ScalarCost, VecCost);
156}
157
159 Align Alignment, unsigned AddressSpace,
161 Type *CmpTy = CmpInst::makeCmpResultType(VecTy);
162 return 2 * TTI.getMemIntrinsicInstrCost(
163 MemIntrinsicCostAttributes(Intrinsic::masked_load, VecTy,
164 Alignment, AddressSpace),
165 CostKind) +
166 TTI.getArithmeticInstrCost(Instruction::Xor, CmpTy, CostKind) +
167 TTI.getCmpSelInstrCost(Instruction::Select, VecTy, CmpTy,
169}
170
172 Type *SrcTy, Type *DstTy,
173 const DataLayout &DL,
176 bool ToPtr = cast<VectorType>(DstTy)->getElementType()->isPointerTy();
177 if (ToPtr == cast<VectorType>(SrcTy)->getElementType()->isPointerTy())
178 return TTI.getCastInstrCost(Instruction::BitCast, DstTy, SrcTy, CCH,
179 CostKind);
180 // The ptr/int conversion keeps the vector shape, the bitcast transforms the
181 // resulting integer vector.
182 if (ToPtr) {
183 Type *IntVecTy = DL.getIntPtrType(DstTy);
184 return TTI.getCastInstrCost(Instruction::IntToPtr, DstTy, IntVecTy, CCH,
185 CostKind) +
186 TTI.getCastInstrCost(Instruction::BitCast, IntVecTy, SrcTy, CCH,
187 CostKind);
188 }
189 Type *IntVecTy = DL.getIntPtrType(SrcTy);
190 return TTI.getCastInstrCost(Instruction::PtrToInt, IntVecTy, SrcTy, CCH,
191 CostKind) +
192 TTI.getCastInstrCost(Instruction::BitCast, DstTy, IntVecTy, CCH,
193 CostKind);
194}
195
197 unsigned Opcode, Type *ScalarTy,
198 unsigned NumElts,
200 FixedVectorType **PaddedTy) {
201 FixedVectorType *PaddedVecTy =
202 getMaskedDivRemType(TTI, Opcode, ScalarTy, NumElts, ReVec);
203 if (!PaddedVecTy)
205 // One mask bit per element of the padded vector, not per padded lane.
206 auto *MaskTy =
208 PaddedVecTy->getNumElements());
209 InstructionCost DirectCost = TTI.getArithmeticInstrCost(
210 Opcode, getWidenedType(ScalarTy, NumElts), CostKind);
211 IntrinsicCostAttributes ICA(getMaskedDivRemIntrinsic(Opcode), PaddedVecTy,
212 {PaddedVecTy, PaddedVecTy, MaskTy});
213 InstructionCost MaskedCost = TTI.getIntrinsicInstrCost(ICA, CostKind);
214 if (!MaskedCost.isValid() || MaskedCost >= DirectCost)
216 if (PaddedTy)
217 *PaddedTy = PaddedVecTy;
218 return MaskedCost;
219}
220
223 Type *ScalarTy, VectorType *Ty,
224 const APInt &DemandedElts, bool Insert, bool Extract,
225 const TTI::TargetCostKind CostKind, bool ForPoisonSrc,
228 "ScalableVectorType is not supported.");
229 assert(getNumElements(ScalarTy) * DemandedElts.getBitWidth() ==
230 getNumElements(Ty) &&
231 "Incorrect usage.");
232 if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
233 assert(ReVec && "Only supported by REVEC.");
234 // If ScalarTy is FixedVectorType, we should use CreateInsertVector instead
235 // of CreateInsertElement.
236 unsigned ScalarTyNumElements = VecTy->getNumElements();
238 for (unsigned I : seq(DemandedElts.getBitWidth())) {
239 if (!DemandedElts[I])
240 continue;
241 if (Insert)
243 I * ScalarTyNumElements, VecTy);
244 if (Extract)
246 I * ScalarTyNumElements, VecTy);
247 }
248 return Cost;
249 }
250 return TTI.getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
251 CostKind, ForPoisonSrc, VL, VIC);
252}
253
255 const TargetTransformInfo &TTI, bool ReVec, Type *ScalarTy, unsigned Opcode,
256 Type *Val, const TTI::TargetCostKind CostKind, unsigned Index,
257 Value *Scalar, ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
259 if (Opcode == Instruction::ExtractElement) {
260 if (auto *VecTy = dyn_cast<FixedVectorType>(ScalarTy)) {
261 assert(ReVec && "Only supported by REVEC.");
262 assert(isa<VectorType>(Val) && "Val must be a vector type.");
264 cast<VectorType>(Val), CostKind, {},
265 Index * VecTy->getNumElements(), VecTy);
266 }
267 }
268 return TTI.getVectorInstrCost(Opcode, Val, CostKind, Index, Scalar,
269 ScalarUserAndIdx, VIC);
270}
271
273 bool ReVec, unsigned Opcode, Type *Dst,
274 VectorType *VecTy, unsigned Index,
276 if (isVectorizedTy(Dst)) {
277 assert(ReVec && "Only supported by REVEC.");
278 auto *SubTp = cast<FixedVectorType>(
281 Index * getNumElements(Dst), SubTp) +
282 TTI.getCastInstrCost(Opcode, Dst, SubTp, TTI::CastContextHint::None,
283 CostKind);
284 }
285 return TTI.getExtractWithExtendCost(Opcode, Dst, VecTy, Index, CostKind);
286}
287
288/// Returns the cast context hint for the trunc of the booleanized reduction
289/// result, which inherits the uses of the reduction root \p Root.
302
304 RecurKind RdxKind,
305 FixedVectorType *VecTy,
306 const Value *Root, FastMathFlags FMF,
308 Type *I1Ty = Type::getInt1Ty(VecTy->getContext());
309 return TTI.getArithmeticReductionCost(
310 RecurrenceDescriptor::getOpcode(RdxKind), VecTy, FMF, CostKind) +
311 TTI.getCastInstrCost(Instruction::Trunc, I1Ty, VecTy->getScalarType(),
313}
314
316 RecurKind RdxKind,
317 FixedVectorType *VecTy,
318 const Value *Root,
319 ArrayRef<Instruction *> ChainInsts,
321 // The new instructions are costed in the context of the replaced cast chain
322 // instructions.
323 auto TruncIt =
324 find_if(ChainInsts, [](Instruction *I) { return isa<TruncInst>(I); });
325 const Instruction *TruncI = TruncIt == ChainInsts.end() ? nullptr : *TruncIt;
326 auto CmpIt =
327 find_if(ChainInsts, [](Instruction *I) { return isa<ICmpInst>(I); });
328 const Instruction *CmpI = CmpIt == ChainInsts.end() ? nullptr : *CmpIt;
329 unsigned VF = VecTy->getNumElements();
330 auto *I1VecTy =
332 Type *IntTy = IntegerType::get(VecTy->getContext(), VF);
333 Constant *CmpRHS = RdxKind == RecurKind::And
335 : Constant::getNullValue(IntTy);
336 return TTI.getCastInstrCost(Instruction::Trunc, I1VecTy, VecTy,
337 TTI.getCastContextHint(TruncI), CostKind,
338 TruncI) +
339 TTI.getCastInstrCost(Instruction::BitCast, IntTy, I1VecTy,
340 TTI.getCastContextHint(TruncI), CostKind) +
341 TTI.getCmpSelInstrCost(
342 Instruction::ICmp, IntTy, CmpInst::makeCmpResultType(IntTy),
344 CostKind, TTI.getOperandInfo(Root), TTI.getOperandInfo(CmpRHS),
345 CmpI);
346}
347
348static InstructionCost
352 assert((Kind == RecurKind::And || Kind == RecurKind::Or) &&
353 VectorTy->getElementType()->isIntegerTy(1) &&
354 "Expected and/or reduction of i1");
355 auto *IntTy =
356 IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
357 CmpInst::Predicate Pred =
359 // The compare is against the all-ones (and) or zero (or) constant.
360 return TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
361 CostKind) +
362 TTI.getCmpSelInstrCost(Instruction::ICmp, IntTy,
363 CmpInst::makeCmpResultType(IntTy), Pred,
364 CostKind, /*Op1Info=*/{},
366}
367
368std::pair<InstructionCost, bool>
370 FixedVectorType *VectorTy, Type *ScalarTy,
372 unsigned RdxOpcode = RecurrenceDescriptor::getOpcode(Kind);
373 if (Kind == RecurKind::And || Kind == RecurKind::Or) {
374 InstructionCost RdxCost = TTI.getArithmeticReductionCost(
375 RdxOpcode, VectorTy, std::nullopt, CostKind);
376 InstructionCost BitcastCost =
377 getBoolLogicRdxBitcastCost(Kind, TTI, VectorTy, Ctx, CostKind);
378 return {std::min(RdxCost, BitcastCost), BitcastCost < RdxCost};
379 }
380 assert(Kind == RecurKind::Add && !ScalarTy->isIntegerTy(1) &&
381 "Expected add reduction of zexted i1 values");
382 // The bitcast+ctpop form is estimated as the cheaper of the extended
383 // reduction cost, which models it for the zexted i1 add reduction, and the
384 // explicitly priced components, including the cast of the ctpop result to
385 // the destination type.
386 auto *IntTy =
387 IntegerType::get(VectorTy->getContext(), getNumElements(VectorTy));
388 InstructionCost ExplicitCost =
389 TTI.getCastInstrCost(Instruction::BitCast, IntTy, VectorTy, Ctx,
390 CostKind) +
391 TTI.getIntrinsicInstrCost(
392 IntrinsicCostAttributes(Intrinsic::ctpop, IntTy, {IntTy}), CostKind);
393 if (IntTy != ScalarTy)
394 ExplicitCost += TTI.getCastInstrCost(IntTy->getBitWidth() <
395 ScalarTy->getIntegerBitWidth()
396 ? Instruction::ZExt
397 : Instruction::Trunc,
398 ScalarTy, IntTy, Ctx, CostKind);
399 InstructionCost CtpopCost = std::min(
400 TTI.getExtendedReductionCost(RdxOpcode, /*IsUnsigned=*/true, ScalarTy,
401 VectorTy, std::nullopt, CostKind),
402 ExplicitCost);
403 // The plain form is the zext to the wide vector type plus the reduction.
404 auto *ExtTy = VectorType::get(ScalarTy, VectorTy);
405 InstructionCost ExtRdxCost =
406 TTI.getCastInstrCost(Instruction::ZExt, ExtTy, VectorTy, Ctx, CostKind) +
407 TTI.getArithmeticReductionCost(RdxOpcode, ExtTy, std::nullopt, CostKind);
408 return {std::min(ExtRdxCost, CtpopCost), CtpopCost <= ExtRdxCost};
409}
410
412 FixedVectorType *SrcTy, Type *ResultTy,
413 const BitPackInfo &Info, unsigned ZExtSrcWidth,
416 const TargetLibraryInfo *TLI,
417 const Instruction *CtxI, unsigned &ShiftWidth) {
418 unsigned BitWidth = SrcTy->getScalarSizeInBits();
419 unsigned NumElts = SrcTy->getNumElements();
420 uint64_t MaxAmt = *max_element(Info.LShrAmts);
421 // The shift amounts form a constant vector.
422 TTI::OperandValueInfo ShiftAmtInfo = {
425 all_of(Info.LShrAmts,
426 [](uint64_t A) { return A == 0 || isPowerOf2_64(A); })
428 : TTI::OP_None};
429 // After the shift the field content of each lane sits in the low bits of
430 // the lane, so the packing is a single byte shuffle of the shifted lanes.
431 // Pick the cheapest shift width: the narrowest type still holding the field
432 // content is not always the cheapest (e.g. missing narrow variable shifts).
433 Type *Int8Ty = Type::getInt8Ty(SrcTy->getContext());
434 assert(BitWidth % 8 == 0 &&
435 "The byte-multiple field width divides the result bit width.");
436 unsigned OutBytes = BitWidth / 8;
437 auto *PackTy = FixedVectorType::get(Int8Ty, OutBytes);
438 unsigned MinShiftWidth = 8;
439 while (MinShiftWidth < MaxAmt + Info.FieldWidth)
440 MinShiftWidth *= 2;
442 ShiftWidth = 0;
443 for (unsigned W2 = MinShiftWidth; W2 <= BitWidth; W2 *= 2) {
444 auto *ShiftTy = FixedVectorType::get(
445 IntegerType::get(SrcTy->getContext(), W2), NumElts);
446 unsigned BytesPerLane = W2 / 8;
447 unsigned InBytes = NumElts * BytesPerLane;
448 SmallVector<int> Mask =
449 getBitPackMask(Info, OutBytes, NumElts, BytesPerLane);
450 InstructionCost C = TTI.getCastInstrCost(Instruction::BitCast, ResultTy,
451 PackTy, CCH, CostKind);
452 // A plain byte reversal of the shifted lanes is a bswap, no shuffle.
453 if (ShuffleVectorInst::isReverseMask(Mask, InBytes)) {
454 IntrinsicCostAttributes CostAttrs(Intrinsic::bswap, ResultTy, {ResultTy});
455 C += TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
456 } else if (!ShuffleVectorInst::isIdentityMask(Mask, InBytes)) {
457 C += TTI.getShuffleCost(
458 is_contained(Info.LaneOfField, BitPackInfo::NoLane)
461 PackTy, FixedVectorType::get(Int8Ty, InBytes), CostKind, Mask,
462 /*Index=*/0, /*SubTp=*/nullptr, /*Args=*/{}, CtxI);
463 }
464 if (W2 != BitWidth && W2 != ZExtSrcWidth)
465 C += TTI.getCastInstrCost(Instruction::Trunc, ShiftTy, SrcTy, CCH,
466 CostKind);
467 if (Info.needsShift())
468 C += TTI.getArithmeticInstrCost(Instruction::LShr, ShiftTy, CostKind,
469 /*Opd1Info=*/{}, ShiftAmtInfo,
470 /*Args=*/{}, CtxI, TLI);
471 if (C.isValid() && (!NewCost.isValid() || C < NewCost)) {
472 NewCost = C;
473 ShiftWidth = W2;
474 }
475 }
476 return NewCost;
477}
478
480 bool NeedMask, Type *NarrowScalarTy,
481 Type *WideTy, unsigned VF,
482 ArrayRef<int> PermMask, const Value *Root,
484 auto *NarrowVecTy = cast<VectorType>(getWidenedType(NarrowScalarTy, VF));
485 Type *CmpTy = CmpInst::makeCmpResultType(NarrowVecTy);
486 auto *MaskTy = IntegerType::get(WideTy->getContext(), VF);
487 // The result cast inherits the uses of the reduction root.
489 const auto *CtxI = cast<Instruction>(Root);
491 if (NeedMask)
492 Cost += TTI.getArithmeticInstrCost(
493 Instruction::And, NarrowVecTy, CostKind,
496 if (!ShuffleVectorInst::isIdentityMask(PermMask, VF))
498 PermMask);
499 if (!NarrowScalarTy->isIntegerTy(1))
500 Cost += TTI.getCmpSelInstrCost(
501 Instruction::ICmp, NarrowVecTy, CmpTy, CmpInst::ICMP_NE, CostKind,
504 // Only the final cast inherits the uses of the reduction root.
505 Cost += TTI.getCastInstrCost(
506 Instruction::BitCast, MaskTy, CmpTy,
507 MaskTy == WideTy ? CCH : TTI::CastContextHint::None, CostKind);
508 if (MaskTy != WideTy)
509 Cost +=
510 TTI.getCastInstrCost(Instruction::ZExt, WideTy, MaskTy, CCH, CostKind);
511 return Cost;
512}
513
516 const SmallDenseMap<Value *, NarrowedLeafInfo> &NarrowedLeafShifts,
517 VectorType *NarrowVecTy, VectorType *WideVecTy, const Instruction *CtxI,
520 if (any_of(NarrowedLeafShifts,
521 [](const auto &P) { return P.second.Shift != 0; }))
522 Cost += TTI.getArithmeticInstrCost(
523 Instruction::Shl, WideVecTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
524 {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CtxI);
525 if (any_of(NarrowedLeafShifts,
526 [](const auto &P) { return !P.second.Mask.isAllOnes(); }))
527 Cost += TTI.getArithmeticInstrCost(
528 Instruction::And, NarrowVecTy, CostKind,
529 {TTI::OK_AnyValue, TTI::OP_None},
530 {TTI::OK_NonUniformConstantValue, TTI::OP_None}, {}, CtxI);
531 return Cost;
532}
533} // namespace llvm::slpvectorizer
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
#define I(x, y, z)
Definition MD5.cpp:57
#define P(N)
This file contains some templates that are useful if you are working with the STL at all.
Provides some synthesis utilities to produce sequences of values.
This file defines the SmallVector class.
Class for arbitrary precision integers.
Definition APInt.h:78
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
iterator end() const
Definition ArrayRef.h:130
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_NE
not equal
Definition InstrTypes.h:762
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
static InstructionCost getInvalid(CostType Val=0)
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
Information for memory intrinsic cost model.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isReverseMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask swaps the order of elements from exactly one source vector.
static LLVM_ABI bool isInsertSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &NumSubElts, int &Index)
Return true if this shuffle mask is an insert subvector mask.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Provides information about what library functions are available for the current target.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
TargetCostKind
The kind of cost model.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ Masked
The cast is used with a masked load/store.
@ Normal
The cast is used with a normal load/store.
@ GatherScatter
The cast is used with a gather/scatter.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
LLVM Value Representation.
Definition Value.h:75
user_iterator user_begin()
Definition Value.h:404
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
bool match(Val *V, const Pattern &P)
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
A private "module" namespace for types and utilities used by this pass.
std::pair< InstructionCost, bool > getI1ReductionCost(RecurKind Kind, const TargetTransformInfo &TTI, FixedVectorType *VectorTy, Type *ScalarTy, TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind)
i1 reductions can be emitted as the plain target reduction or in the bitcast-based form (bitcast to a...
InstructionCost getShuffleCost(const TargetTransformInfo &TTI, TTI::ShuffleKind Kind, VectorType *Tp, const TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args, TTI::VectorInstrContext VIC)
Returns the cost of the shuffle instructions with the given Kind, vector type Tp and optional Mask.
InstructionCost getNarrowedLeafOpsCost(const TargetTransformInfo &TTI, const SmallDenseMap< Value *, NarrowedLeafInfo > &NarrowedLeafShifts, VectorType *NarrowVecTy, VectorType *WideVecTy, const Instruction *CtxI, const TTI::TargetCostKind CostKind)
Returns the cost of the per-lane operations on the narrowed leaves NarrowedLeafShifts: the shl in the...
InstructionCost getBoolReduxBitcastCmpCost(const TargetTransformInfo &TTI, RecurKind RdxKind, FixedVectorType *VecTy, const Value *Root, ArrayRef< Instruction * > ChainInsts, const TTI::TargetCostKind CostKind)
Returns the cost of the booleanized logical and/or reduction of a vector of type VecTy with the i1 ro...
std::pair< InstructionCost, InstructionCost > getGEPCosts(const TargetTransformInfo &TTI, ArrayRef< Value * > Ptrs, Value *BasePtr, unsigned Opcode, const TTI::TargetCostKind CostKind, Type *ScalarTy, VectorType *VecTy)
Calculate the scalar and the vector costs from vectorizing set of GEPs.
Intrinsic::ID getMaskedDivRemIntrinsic(unsigned Opcode)
SmallVector< int > getBitPackMask(const BitPackInfo &Info, unsigned NumBytes, unsigned NumElts, unsigned BytesPerLane)
Returns the byte shuffle mask packing the per-lane fields of the shifted lanes (BytesPerLane bytes ea...
unsigned getNumElements(Type *Ty)
Definition SLPUtils.cpp:88
Type * getWidenedType(Type *ScalarTy, unsigned VF)
static TTI::CastContextHint getBoolReduxResultCCH(const Value *Root)
Returns the cast context hint for the trunc of the booleanized reduction result, which inherits the u...
InstructionCost getBlendedLoadCost(const TargetTransformInfo &TTI, Type *VecTy, Align Alignment, unsigned AddressSpace, const TTI::TargetCostKind CostKind)
Returns the cost of a BlendedLoadVectorize node loading VecTy: two masked loads (one per candidate ba...
FixedVectorType * getMaskedDivRemType(const TargetTransformInfo &TTI, unsigned Opcode, Type *ScalarTy, unsigned NumElts, bool ReVec)
For a non-power-of-2 NumElts-wide integer div/rem Opcode, returns the padded full-register vector typ...
InstructionCost getBoolBitmaskCost(const TargetTransformInfo &TTI, bool NeedMask, Type *NarrowScalarTy, Type *WideTy, unsigned VF, ArrayRef< int > PermMask, const Value *Root, const TTI::TargetCostKind CostKind)
Returns the cost of the boolean bitmask reduction of a vector of boolean leaves of type NarrowScalarT...
InstructionCost getBoolReduxWideRdxCost(const TargetTransformInfo &TTI, RecurKind RdxKind, FixedVectorType *VecTy, const Value *Root, FastMathFlags FMF, const TTI::TargetCostKind CostKind)
Returns the cost of the booleanized logical and/or reduction of a vector of type VecTy with the i1 ro...
InstructionCost getWidenedStridedCastCost(const TargetTransformInfo &TTI, Type *SrcTy, Type *DstTy, const DataLayout &DL, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind)
Returns the cost of the cast between the widened strided access type and the entry vector type.
InstructionCost getScalarizationOverhead(const TargetTransformInfo &TTI, bool ReVec, Type *ScalarTy, VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, const TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef< Value * > VL, TTI::VectorInstrContext VIC)
This is similar to TargetTransformInfo::getScalarizationOverhead, but if ScalarTy is a FixedVectorTyp...
InstructionCost getExtractWithExtendCost(const TargetTransformInfo &TTI, bool ReVec, unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, const TTI::TargetCostKind CostKind)
This is similar to TargetTransformInfo::getExtractWithExtendCost, but if Dst is a FixedVectorType,...
static InstructionCost getBoolLogicRdxBitcastCost(RecurKind Kind, const TargetTransformInfo &TTI, FixedVectorType *VectorTy, TTI::CastContextHint Ctx, TTI::TargetCostKind CostKind)
InstructionCost getVectorInstrCost(const TargetTransformInfo &TTI, bool ReVec, Type *ScalarTy, unsigned Opcode, Type *Val, const TTI::TargetCostKind CostKind, unsigned Index, Value *Scalar, ArrayRef< std::tuple< Value *, User *, int > > ScalarUserAndIdx, TTI::VectorInstrContext VIC)
This is similar to TargetTransformInfo::getVectorInstrCost, but if ScalarTy is a FixedVectorType,...
InstructionCost getMaskedDivRemCost(const TargetTransformInfo &TTI, bool ReVec, unsigned Opcode, Type *ScalarTy, unsigned NumElts, const TTI::TargetCostKind CostKind, FixedVectorType **PaddedTy)
For a non-power-of-2 NumElts-wide integer div/rem Opcode, checks if padding to a full register and us...
InstructionCost getBitPackCost(const TargetTransformInfo &TTI, FixedVectorType *SrcTy, Type *ResultTy, const BitPackInfo &Info, unsigned ZExtSrcWidth, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const TargetLibraryInfo *TLI, const Instruction *CtxI, unsigned &ShiftWidth)
Returns the cost of the bitfield packing of SrcTy into ResultTy, picking the cheapest shift width.
This is an optimization pass for GlobalISel generic memory operations.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
InstructionCost Cost
Type * toScalarizedTy(Type *Ty)
A helper for converting vectorized types to scalarized (non-vector) types.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isVectorizedTy(Type *Ty)
Returns true if Ty is a vector type or a struct of vector types where all vector types share the same...
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
bool isPointerTy(const Type *T)
Definition SPIRVUtils.h:383
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
TargetTransformInfo TTI
RecurKind
These are the kinds of recurrences that we support.
@ Or
Bitwise or logical OR of integers.
@ And
Bitwise or logical AND of integers.
@ Add
Sum of integers.
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2104
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2182
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Describe known properties for a set of pointers.
unsigned IsSameBaseAddress
All the GEPs in a set have same base address.
Description of a bitfield packing of vector lanes into a scalar value: every lane contributes a disjo...
Definition SLPUtils.h:473
static constexpr unsigned NoLane
Definition SLPUtils.h:474