LLVM 24.0.0git
RISCVTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- RISCVTargetTransformInfo.cpp - RISC-V specific TTI ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "RISCVVectorUtils.h"
12#include "llvm/ADT/STLExtras.h"
19#include "llvm/IR/IntrinsicsRISCV.h"
22#include <cmath>
23#include <limits>
24#include <optional>
25using namespace llvm;
26using namespace llvm::PatternMatch;
27
28#define DEBUG_TYPE "riscvtti"
29
31 "riscv-v-register-bit-width-lmul",
33 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
34 "by autovectorized code. Fractional LMULs are not supported."),
36
38 "riscv-v-slp-max-vf",
40 "Overrides result used for getMaximumVF query which is used "
41 "exclusively by SLP vectorizer."),
43
45 RVVMinTripCount("riscv-v-min-trip-count",
46 cl::desc("Set the lower bound of a trip count to decide on "
47 "vectorization while tail-folding."),
49
50static cl::opt<bool> EnableOrLikeSelectOpt("riscv-or-like-select",
51 cl::init(true), cl::Hidden);
52
54RISCVTTIImpl::getRISCVInstructionCost(ArrayRef<unsigned> OpCodes, MVT VT,
56 // Check if the type is valid for all CostKind
57 if (!VT.isVector())
59 size_t NumInstr = OpCodes.size();
61 return NumInstr;
62 InstructionCost LMULCost = TLI->getLMULCost(VT);
64 return LMULCost * NumInstr;
65 InstructionCost Cost = 0;
66 for (auto Op : OpCodes) {
67 switch (Op) {
68 case RISCV::VRGATHER_VI:
69 Cost += TLI->getVRGatherVICost(VT);
70 break;
71 case RISCV::VRGATHER_VV:
72 Cost += TLI->getVRGatherVVCost(VT);
73 break;
74 case RISCV::VSLIDEUP_VI:
75 case RISCV::VSLIDEDOWN_VI:
76 Cost += TLI->getVSlideVICost(VT);
77 break;
78 case RISCV::VSLIDEUP_VX:
79 case RISCV::VSLIDEDOWN_VX:
80 Cost += TLI->getVSlideVXCost(VT);
81 break;
82 case RISCV::VREDMAX_VS:
83 case RISCV::VREDMIN_VS:
84 case RISCV::VREDMAXU_VS:
85 case RISCV::VREDMINU_VS:
86 case RISCV::VREDSUM_VS:
87 case RISCV::VREDAND_VS:
88 case RISCV::VREDOR_VS:
89 case RISCV::VREDXOR_VS:
90 case RISCV::VFREDMAX_VS:
91 case RISCV::VFREDMIN_VS:
92 case RISCV::VFREDUSUM_VS: {
93 unsigned VL = VT.getVectorMinNumElements();
94 if (!VT.isFixedLengthVector())
95 VL *= *getVScaleForTuning();
96 Cost += Log2_32_Ceil(VL);
97 break;
98 }
99 case RISCV::VFREDOSUM_VS: {
100 unsigned VL = VT.getVectorMinNumElements();
101 if (!VT.isFixedLengthVector())
102 VL *= *getVScaleForTuning();
103 Cost += VL;
104 break;
105 }
106 case RISCV::VMV_X_S:
107 case RISCV::VFMV_F_S:
108 // Domain crossings from vector -> scalar are usually more expensive.
109 Cost += 2;
110 break;
111 case RISCV::VMV_S_X:
112 case RISCV::VFMV_S_F:
113 case RISCV::VMOR_MM:
114 case RISCV::VMXOR_MM:
115 case RISCV::VMAND_MM:
116 case RISCV::VMANDN_MM:
117 case RISCV::VMNAND_MM:
118 case RISCV::VCPOP_M:
119 case RISCV::VFIRST_M:
120 Cost += 1;
121 break;
122 case RISCV::VDIV_VV:
123 case RISCV::VREM_VV:
124 Cost += LMULCost * TTI::TCC_Expensive;
125 break;
126 default:
127 Cost += LMULCost;
128 }
129 }
130 return Cost;
131}
132
134 const RISCVSubtarget *ST,
135 const APInt &Imm, Type *Ty,
137 bool FreeZeroes) {
138 assert(Ty->isIntegerTy() &&
139 "getIntImmCost can only estimate cost of materialising integers");
140
141 // We have a Zero register, so 0 is always free.
142 if (Imm == 0)
143 return TTI::TCC_Free;
144
145 // Otherwise, we check how many instructions it will take to materialise.
146 return RISCVMatInt::getIntMatCost(Imm, DL.getTypeSizeInBits(Ty), *ST,
147 /*CompressionCost=*/false, FreeZeroes);
148}
149
153 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind, false);
154}
155
156// Look for patterns of shift followed by AND that can be turned into a pair of
157// shifts. We won't need to materialize an immediate for the AND so these can
158// be considered free.
159static bool canUseShiftPair(Instruction *Inst, const APInt &Imm) {
160 uint64_t Mask = Imm.getZExtValue();
161 auto *BO = dyn_cast<BinaryOperator>(Inst->getOperand(0));
162 if (!BO || !BO->hasOneUse())
163 return false;
164
165 if (BO->getOpcode() != Instruction::Shl)
166 return false;
167
168 if (!isa<ConstantInt>(BO->getOperand(1)))
169 return false;
170
171 unsigned ShAmt = cast<ConstantInt>(BO->getOperand(1))->getZExtValue();
172 // (and (shl x, c2), c1) will be matched to (srli (slli x, c2+c3), c3) if c1
173 // is a mask shifted by c2 bits with c3 leading zeros.
174 if (isShiftedMask_64(Mask)) {
175 unsigned Trailing = llvm::countr_zero(Mask);
176 if (ShAmt == Trailing)
177 return true;
178 }
179
180 return false;
181}
182
183// If this is i64 AND is part of (X & -(1 << C1) & 0xffffffff) == C2 << C1),
184// DAGCombiner can convert this to (sraiw X, C1) == sext(C2) for RV64. On RV32,
185// the type will be split so only the lower 32 bits need to be compared using
186// (srai/srli X, C) == C2.
187static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm) {
188 if (!Inst->hasOneUse())
189 return false;
190
191 // Look for equality comparison.
192 auto *Cmp = dyn_cast<ICmpInst>(*Inst->user_begin());
193 if (!Cmp || !Cmp->isEquality())
194 return false;
195
196 // Right hand side of comparison should be a constant.
197 auto *C = dyn_cast<ConstantInt>(Cmp->getOperand(1));
198 if (!C)
199 return false;
200
201 uint64_t Mask = Imm.getZExtValue();
202
203 // Mask should be of the form -(1 << C) in the lower 32 bits.
204 if (!isUInt<32>(Mask) || !isPowerOf2_32(-uint32_t(Mask)))
205 return false;
206
207 // Comparison constant should be a subset of Mask.
208 uint64_t CmpC = C->getZExtValue();
209 if ((CmpC & Mask) != CmpC)
210 return false;
211
212 // We'll need to sign extend the comparison constant and shift it right. Make
213 // sure the new constant can use addi/xori+seqz/snez.
214 unsigned ShiftBits = llvm::countr_zero(Mask);
215 int64_t NewCmpC = SignExtend64<32>(CmpC) >> ShiftBits;
216 return NewCmpC >= -2048 && NewCmpC <= 2048;
217}
218
220 const APInt &Imm, Type *Ty,
222 Instruction *Inst) const {
223 assert(Ty->isIntegerTy() &&
224 "getIntImmCost can only estimate cost of materialising integers");
225
226 // We have a Zero register, so 0 is always free.
227 if (Imm == 0)
228 return TTI::TCC_Free;
229
230 // Some instructions in RISC-V can take a 12-bit immediate. Some of these are
231 // commutative, in others the immediate comes from a specific argument index.
232 bool Takes12BitImm = false;
233 unsigned ImmArgIdx = ~0U;
234
235 switch (Opcode) {
236 case Instruction::GetElementPtr:
237 // Never hoist any arguments to a GetElementPtr. CodeGenPrepare will
238 // split up large offsets in GEP into better parts than ConstantHoisting
239 // can.
240 return TTI::TCC_Free;
241 case Instruction::Store: {
242 // Use the materialization cost regardless of if it's the address or the
243 // value that is constant, except for if the store is misaligned and
244 // misaligned accesses are not legal (experience shows constant hoisting
245 // can sometimes be harmful in such cases).
246 if (Idx == 1 || !Inst)
247 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
248 /*FreeZeroes=*/true);
249
250 StoreInst *ST = cast<StoreInst>(Inst);
251 if (!getTLI()->allowsMemoryAccessForAlignment(
252 Ty->getContext(), DL, getTLI()->getValueType(DL, Ty),
253 ST->getPointerAddressSpace(), ST->getAlign()))
254 return TTI::TCC_Free;
255
256 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
257 /*FreeZeroes=*/true);
258 }
259 case Instruction::Load:
260 // If the address is a constant, use the materialization cost.
261 return getIntImmCost(Imm, Ty, CostKind);
262 case Instruction::And:
263 // zext.h
264 if (Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
265 return TTI::TCC_Free;
266 // zext.w
267 if (Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
268 return TTI::TCC_Free;
269 // bclri
270 if (ST->hasStdExtZbs() && (~Imm).isPowerOf2())
271 return TTI::TCC_Free;
272 if (Inst && Idx == 1 && Imm.getBitWidth() <= ST->getXLen() &&
273 canUseShiftPair(Inst, Imm))
274 return TTI::TCC_Free;
275 if (Inst && Idx == 1 && Imm.getBitWidth() == 64 &&
276 canUseShiftCmp(Inst, Imm))
277 return TTI::TCC_Free;
278 Takes12BitImm = true;
279 break;
280 case Instruction::Add:
281 Takes12BitImm = true;
282 break;
283 case Instruction::Or:
284 case Instruction::Xor:
285 // bseti/binvi
286 if (ST->hasStdExtZbs() && Imm.isPowerOf2())
287 return TTI::TCC_Free;
288 Takes12BitImm = true;
289 break;
290 case Instruction::Mul:
291 // Power of 2 is a shift. Negated power of 2 is a shift and a negate.
292 if (Imm.isPowerOf2() || Imm.isNegatedPowerOf2())
293 return TTI::TCC_Free;
294 // One more or less than a power of 2 can use SLLI+ADD/SUB.
295 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2())
296 return TTI::TCC_Free;
297 // FIXME: There is no MULI instruction.
298 Takes12BitImm = true;
299 break;
300 case Instruction::Sub:
301 case Instruction::Shl:
302 case Instruction::LShr:
303 case Instruction::AShr:
304 Takes12BitImm = true;
305 ImmArgIdx = 1;
306 break;
307 default:
308 break;
309 }
310
311 if (Takes12BitImm) {
312 // Check immediate is the correct argument...
313 if (Instruction::isCommutative(Opcode) || Idx == ImmArgIdx) {
314 // ... and fits into the 12-bit immediate.
315 if (Imm.getSignificantBits() <= 64 &&
316 getTLI()->isLegalAddImmediate(Imm.getSExtValue())) {
317 return TTI::TCC_Free;
318 }
319 }
320
321 // Otherwise, use the full materialisation cost.
322 return getIntImmCost(Imm, Ty, CostKind);
323 }
324
325 // By default, prevent hoisting.
326 return TTI::TCC_Free;
327}
328
331 const APInt &Imm, Type *Ty,
333 // Prevent hoisting in unknown cases.
334 return TTI::TCC_Free;
335}
336
338 return ST->hasVInstructions();
339}
340
342RISCVTTIImpl::getPopcntSupport(unsigned TyWidth) const {
343 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
344 return ST->hasCPOPLike() ? TTI::PSK_FastHardware : TTI::PSK_Software;
345}
346
348 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
350 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
351 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
352 if (Opcode == Instruction::FAdd)
354
355 // zve32x is broken for partial_reduce_umla, but let's make sure we
356 // don't generate them.
357 // vdot4a* reduces four i8 products into an i32 result; an i64 accumulator is
358 // additionally supported by widening the i32 partial sums to i64 (see
359 // lowerPARTIAL_REDUCE_MLA). VF is the number of i8 input elements, so the
360 // reduction factor is AccumBits / 8 (4 for i32, 8 for i64).
361 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
362 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
363 InputTypeA != InputTypeB || !InputTypeA->isIntegerTy(8) ||
364 (!AccumType->isIntegerTy(32) && !AccumType->isIntegerTy(64)))
366
367 unsigned Ratio = AccumType->getScalarSizeInBits() / 8;
368 if (!VF.isKnownMultipleOf(Ratio))
370
371 // Cost of the vdot4a* itself, which operates on the i32 intermediate type
372 // holding VF/4 elements.
373 Type *DotTp = VectorType::get(Type::getInt32Ty(AccumType->getContext()),
374 VF.divideCoefficientBy(4));
375 std::pair<InstructionCost, MVT> DotLT = getTypeLegalizationCost(DotTp);
376 // Note: Asuming all vdot4a* variants are equal cost
378 DotLT.first *
379 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second, CostKind);
380
381 // Account for reducing the i32 partial sums down to the i64 accumulator's
382 // element count and accumulating into it (see lowerPARTIAL_REDUCE_MLA), which
383 // has two shapes depending on the accumulator's LMUL.
384 if (AccumType->isIntegerTy(64)) {
385 LLVMContext &Ctx = AccumType->getContext();
386 Type *I32Ty = Type::getInt32Ty(Ctx);
387 ElementCount AccVF = VF.divideCoefficientBy(Ratio);
388 std::pair<InstructionCost, MVT> AccLT =
389 getTypeLegalizationCost(VectorType::get(AccumType, AccVF));
390
391 // When the i32 subvectors of a single-vector scalable accumulator are a
392 // fractional LMUL, extracting the high subvector would need a vslidedown,
393 // so instead the i32 dot result is widened to i64 first (vsext.vf2 /
394 // vzext.vf2) and then reduced and accumulated with register-aligned i64
395 // vadd.vv.
396 bool WidenFirst = false;
397 if (VF.isScalable() && AccLT.second.isScalableVector()) {
398 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
399 WidenFirst =
401 .second;
402 }
403
404 if (WidenFirst) {
405 // The widened i64 dot result has VF/4 elements, i.e. twice the
406 // accumulator's element count, so the reduction plus the accumulate are
407 // two i64 vadd.vv.
408 std::pair<InstructionCost, MVT> WideLT = getTypeLegalizationCost(
409 VectorType::get(AccumType, VF.divideCoefficientBy(4)));
410 Cost +=
411 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
412 WideLT.second, CostKind) +
413 2 * AccLT.first *
414 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second, CostKind);
415 } else {
416 // Otherwise the scale-4 i32 sums are halved with a single i32 vadd.vv,
417 // then widened and added into the i64 result with a vwadd.wv.
418 std::pair<InstructionCost, MVT> RedLT =
420 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
421 RedLT.second, CostKind) +
422 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
423 AccLT.second, CostKind);
424 // Fixed-length vectors extract the high i32 subvector with a vslidedown.
425 if (VF.isFixed())
426 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
427 DotLT.second, CostKind);
428 }
429 }
430
431 return Cost;
432}
433
435 // Currently, the ExpandReductions pass can't expand scalable-vector
436 // reductions, but we still request expansion as RVV doesn't support certain
437 // reductions and the SelectionDAG can't legalize them either.
438 switch (II->getIntrinsicID()) {
439 default:
440 return false;
441 // These reductions have no equivalent in RVV
442 case Intrinsic::vector_reduce_mul:
443 case Intrinsic::vector_reduce_fmul:
444 return true;
445 }
446}
447
448std::optional<unsigned> RISCVTTIImpl::getVScaleForTuning() const {
449 if (ST->hasVInstructions())
450 if (unsigned MinVLen = ST->getRealMinVLen();
451 MinVLen >= RISCV::RVVBitsPerBlock)
452 return MinVLen / RISCV::RVVBitsPerBlock;
454}
455
458 unsigned LMUL =
459 llvm::bit_floor(std::clamp<unsigned>(RVVRegisterWidthLMUL, 1, 8));
460 switch (K) {
462 return TypeSize::getFixed(ST->getXLen());
464 return TypeSize::getFixed(
465 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
468 (ST->hasVInstructions() &&
469 ST->getRealMinVLen() >= RISCV::RVVBitsPerBlock)
471 : 0);
472 }
473
474 llvm_unreachable("Unsupported register kind");
475}
476
477InstructionCost RISCVTTIImpl::getStaticDataAddrGenerationCost(
478 const TTI::TargetCostKind CostKind) const {
479 switch (CostKind) {
482 // Always 2 instructions
483 return 2;
484 case TTI::TCK_Latency:
486 // Depending on the memory model the address generation will
487 // require AUIPC + ADDI (medany) or LUI + ADDI (medlow). Don't
488 // have a way of getting this information here, so conservatively
489 // require both.
490 // In practice, these are generally implemented together.
491 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
492 }
493 llvm_unreachable("Unsupported cost kind");
494}
495
497RISCVTTIImpl::getConstantPoolLoadCost(Type *Ty,
499 // Add a cost of address generation + the cost of the load. The address
500 // is expected to be a PC relative offset to a constant pool entry
501 // using auipc/addi.
503 Cost = getStaticDataAddrGenerationCost(CostKind) +
504 getMemoryOpCost(Instruction::Load, Ty, DL.getABITypeAlign(Ty),
505 /*AddressSpace=*/0, CostKind);
506 // Estimate the amount of 4 byte instructions that could fit
507 // instead of the constant pool, ignoring any extra padding.
509 Cost += ((InstructionCost)DL.getTypeAllocSize(Ty)) / 4;
510 return Cost;
511}
512
513static bool isRepeatedConcatMask(ArrayRef<int> Mask, int &SubVectorSize) {
514 unsigned Size = Mask.size();
515 if (!isPowerOf2_32(Size))
516 return false;
517 for (unsigned I = 0; I != Size; ++I) {
518 if (static_cast<unsigned>(Mask[I]) == I)
519 continue;
520 if (Mask[I] != 0)
521 return false;
522 if (Size % I != 0)
523 return false;
524 for (unsigned J = I + 1; J != Size; ++J)
525 // Check the pattern is repeated.
526 if (static_cast<unsigned>(Mask[J]) != J % I)
527 return false;
528 SubVectorSize = I;
529 return true;
530 }
531 // That means Mask is <0, 1, 2, 3>. This is not a concatenation.
532 return false;
533}
534
536 LLVMContext &C) {
537 assert((DataVT.getScalarSizeInBits() != 8 ||
538 DataVT.getVectorNumElements() <= 256) && "unhandled case in lowering");
539 MVT IndexVT = DataVT.changeTypeToInteger();
540 if (IndexVT.getScalarType().bitsGT(ST.getXLenVT()))
541 IndexVT = IndexVT.changeVectorElementType(MVT::i16);
542 return cast<VectorType>(EVT(IndexVT).getTypeForEVT(C));
543}
544
545/// Attempt to approximate the cost of a shuffle which will require splitting
546/// during legalization. Note that processShuffleMasks is not an exact proxy
547/// for the algorithm used in LegalizeVectorTypes, but hopefully it's a
548/// reasonably close upperbound.
550 MVT LegalVT, VectorType *Tp,
551 ArrayRef<int> Mask,
553 assert(LegalVT.isFixedLengthVector() && !Mask.empty() &&
554 "Expected fixed vector type and non-empty mask");
555 unsigned LegalNumElts = LegalVT.getVectorNumElements();
556 // Number of destination vectors after legalization:
557 unsigned NumOfDests = divideCeil(Mask.size(), LegalNumElts);
558 // We are going to permute multiple sources and the result will be in
559 // multiple destinations. Providing an accurate cost only for splits where
560 // the element type remains the same.
561 if (NumOfDests <= 1 ||
563 Tp->getElementType()->getPrimitiveSizeInBits() ||
564 LegalNumElts >= Tp->getElementCount().getFixedValue())
566
567 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
568 unsigned LegalVTSize = LegalVT.getStoreSize();
569 // Number of source vectors after legalization:
570 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
571
572 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(), LegalNumElts);
573
574 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
575 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
576 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
577 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
578 assert(NormalizedVF >= Mask.size() &&
579 "Normalized mask expected to be not shorter than original mask.");
580 copy(Mask, NormalizedMask.begin());
581 InstructionCost Cost = 0;
582 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
584 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
585 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
586 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
587 return;
588 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
589 .second)
590 return;
591 Cost += TTI.getShuffleCost(
593 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
594 SingleOpTy, CostKind, RegMask, 0, nullptr);
595 },
596 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
597 Cost += TTI.getShuffleCost(
599 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
600 SingleOpTy, CostKind, RegMask, 0, nullptr);
601 });
602 return Cost;
603}
604
605/// Try to perform better estimation of the permutation.
606/// 1. Split the source/destination vectors into real registers.
607/// 2. Do the mask analysis to identify which real registers are
608/// permuted. If more than 1 source registers are used for the
609/// destination register building, the cost for this destination register
610/// is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
611/// source register is used, build mask and calculate the cost as a cost
612/// of PermuteSingleSrc.
613/// Also, for the single register permute we try to identify if the
614/// destination register is just a copy of the source register or the
615/// copy of the previous destination register (the cost is
616/// TTI::TCC_Basic). If the source register is just reused, the cost for
617/// this operation is 0.
618static InstructionCost
620 std::optional<unsigned> VLen, VectorType *Tp,
622 assert(LegalVT.isFixedLengthVector());
623 if (!VLen || Mask.empty())
625 MVT ElemVT = LegalVT.getVectorElementType();
626 unsigned ElemsPerVReg = *VLen / ElemVT.getFixedSizeInBits();
627 LegalVT = TTI.getTypeLegalizationCost(
628 FixedVectorType::get(Tp->getElementType(), ElemsPerVReg))
629 .second;
630 // Number of destination vectors after legalization:
631 InstructionCost NumOfDests =
632 divideCeil(Mask.size(), LegalVT.getVectorNumElements());
633 if (NumOfDests <= 1 ||
635 Tp->getElementType()->getPrimitiveSizeInBits() ||
636 LegalVT.getVectorNumElements() >= Tp->getElementCount().getFixedValue())
638
639 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
640 unsigned LegalVTSize = LegalVT.getStoreSize();
641 // Number of source vectors after legalization:
642 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
643
644 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(),
645 LegalVT.getVectorNumElements());
646
647 unsigned E = NumOfDests.getValue();
648 unsigned NormalizedVF =
649 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
650 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
651 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
652 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
653 assert(NormalizedVF >= Mask.size() &&
654 "Normalized mask expected to be not shorter than original mask.");
655 copy(Mask, NormalizedMask.begin());
656 InstructionCost Cost = 0;
657 int NumShuffles = 0;
658 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
660 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
661 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
662 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
663 return;
664 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
665 .second)
666 return;
667 ++NumShuffles;
668 Cost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, SingleOpTy,
669 SingleOpTy, CostKind, RegMask, 0, nullptr);
670 },
671 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
672 Cost += TTI.getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
673 SingleOpTy, CostKind, RegMask, 0, nullptr);
674 NumShuffles += 2;
675 });
676 // Note: check that we do not emit too many shuffles here to prevent code
677 // size explosion.
678 // TODO: investigate, if it can be improved by extra analysis of the masks
679 // to check if the code is more profitable.
680 if ((NumOfDestRegs > 2 && NumShuffles <= static_cast<int>(NumOfDestRegs)) ||
681 (NumOfDestRegs <= 2 && NumShuffles < 4))
682 return Cost;
684}
685
686InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
687 ArrayRef<int> Mask,
689 // Avoid missing masks and length changing shuffles
690 if (Mask.size() <= 2 || Mask.size() != Tp->getNumElements())
692
693 int NumElts = Tp->getNumElements();
694 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
695 // Avoid scalarization cases
696 if (!LT.second.isFixedLengthVector())
698
699 // Requires moving elements between parts, which requires additional
700 // unmodeled instructions.
701 if (LT.first != 1)
703
704 auto GetSlideOpcode = [&](int SlideAmt) {
705 assert(SlideAmt != 0);
706 bool IsVI = isUInt<5>(std::abs(SlideAmt));
707 if (SlideAmt < 0)
708 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
709 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
710 };
711
712 std::array<std::pair<int, int>, 2> SrcInfo;
713 if (!isMaskedSlidePair(Mask, NumElts, SrcInfo))
715
716 if (SrcInfo[1].second == 0)
717 std::swap(SrcInfo[0], SrcInfo[1]);
718
719 if (ST->hasStdExtZvzip() && LT.second.getScalarSizeInBits() != 1) {
720 unsigned Factor;
721 if (isPairEven(SrcInfo, Mask, Factor) && Factor == 1)
722 return getRISCVInstructionCost(RISCV::VPAIRE_VV, LT.second, CostKind);
723 if (isPairOdd(SrcInfo, Mask, Factor) && Factor == 1)
724 return getRISCVInstructionCost(RISCV::VPAIRO_VV, LT.second, CostKind);
725 }
726
727 InstructionCost FirstSlideCost = 0;
728 if (SrcInfo[0].second != 0) {
729 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
730 FirstSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
731 }
732
733 if (SrcInfo[1].first == -1)
734 return FirstSlideCost;
735
736 InstructionCost SecondSlideCost = 0;
737 if (SrcInfo[1].second != 0) {
738 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
739 SecondSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
740 } else {
741 SecondSlideCost =
742 getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second, CostKind);
743 }
744
745 auto EC = Tp->getElementCount();
746 VectorType *MaskTy =
748 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
749 return FirstSlideCost + SecondSlideCost + MaskCost;
750}
751
752std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
753 assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
754 if (!InterleavedVT.getVectorElementCount().isKnownEven())
755 return std::nullopt;
756
757 unsigned EltBits = InterleavedVT.getScalarSizeInBits();
758 unsigned MinSize = InterleavedVT.getSizeInBits().getKnownMinValue();
759 unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
760 // Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
761 if (EltBits * 16 >
762 LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
763 return std::nullopt;
764 return InterleavedVT;
765}
766
767std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(MVT InterleavedVT) const {
768 assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
769 if (!InterleavedVT.getVectorElementCount().isKnownEven())
770 return std::nullopt;
771
772 MVT DeinterleavedVT = InterleavedVT.getHalfNumVectorElementsVT();
773 if (RISCVTargetLowering::getLMUL(DeinterleavedVT) == RISCVVType::LMUL_8)
774 return std::nullopt;
775 return InterleavedVT;
776}
777
779 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
781 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
782 TTI::VectorInstrContext VIC) const {
783 assert((Mask.empty() || DstTy->isScalableTy() ||
784 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
785 "Expected the Mask to match the return size if given");
786 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
787 "Expected the same scalar types");
788
789 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
790 if (VIC == TTI::VectorInstrContext::SplatOpFolded &&
791 ST->sinkSplatOperands() && Kind == TTI::SK_Broadcast)
792 return TTI::TCC_Free;
793
794 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
795 // For now, skip all fixed vector cost analysis when P extension is available
796 // to avoid crashes in getMinRVVVectorSizeInBits()
797 if (ST->hasStdExtP() && isa<FixedVectorType>(SrcTy))
798 return 1;
799
800 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
801
802 // First, handle cases where having a fixed length vector enables us to
803 // give a more accurate cost than falling back to generic scalable codegen.
804 // TODO: Each of these cases hints at a modeling gap around scalable vectors.
805 if (auto *FVTp = dyn_cast<FixedVectorType>(SrcTy);
806 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
808 *this, LT.second, ST->getRealVLen(),
809 Kind == TTI::SK_InsertSubvector ? DstTy : SrcTy, Mask, CostKind);
810 if (VRegSplittingCost.isValid())
811 return VRegSplittingCost;
812 switch (Kind) {
813 default:
814 break;
816 if (Mask.size() >= 2) {
817 MVT EltTp = LT.second.getVectorElementType();
818 // If the size of the element is < ELEN then shuffles of interleaves and
819 // deinterleaves of 2 vectors can be lowered into the following
820 // sequences
821 if (EltTp.getScalarSizeInBits() < ST->getELen()) {
822 // Example sequence:
823 // vsetivli zero, 4, e8, mf4, ta, ma (ignored)
824 // vwaddu.vv v10, v8, v9
825 // li a0, -1 (ignored)
826 // vwmaccu.vx v10, a0, v9
827 if (ShuffleVectorInst::isInterleaveMask(Mask, 2, Mask.size()))
828 return 2 * LT.first * TLI->getLMULCost(LT.second);
829
830 if (Mask[0] == 0 || Mask[0] == 1) {
831 auto DeinterleaveMask = createStrideMask(Mask[0], 2, Mask.size());
832 // Example sequence:
833 // vnsrl.wi v10, v8, 0
834 if (equal(DeinterleaveMask, Mask))
835 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
836 LT.second, CostKind);
837 }
838 }
839 int SubVectorSize;
840 if (LT.second.getScalarSizeInBits() != 1 &&
841 isRepeatedConcatMask(Mask, SubVectorSize)) {
843 unsigned NumSlides = Log2_32(Mask.size() / SubVectorSize);
844 // The cost of extraction from a subvector is 0 if the index is 0.
845 for (unsigned I = 0; I != NumSlides; ++I) {
846 unsigned InsertIndex = SubVectorSize * (1 << I);
847 FixedVectorType *SubTp =
848 FixedVectorType::get(SrcTy->getElementType(), InsertIndex);
849 FixedVectorType *DestTp =
851 std::pair<InstructionCost, MVT> DestLT =
853 // Add the cost of whole vector register move because the
854 // destination vector register group for vslideup cannot overlap the
855 // source.
856 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
858 CostKind, {}, InsertIndex, SubTp);
859 }
860 return Cost;
861 }
862 }
863
864 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
865 SlideCost.isValid())
866 return SlideCost;
867
868 // vrgather + cost of generating the mask constant.
869 // We model this for an unknown mask with a single vrgather.
870 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
871 LT.second.getVectorNumElements() <= 256)) {
872 VectorType *IdxTy =
873 getVRGatherIndexType(LT.second, *ST, SrcTy->getContext());
874 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
875 return IndexCost +
876 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second, CostKind);
877 }
878 break;
879 }
882
883 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
884 SlideCost.isValid())
885 return SlideCost;
886
887 // 2 x (vrgather + cost of generating the mask constant) + cost of mask
888 // register for the second vrgather. We model this for an unknown
889 // (shuffle) mask.
890 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
891 LT.second.getVectorNumElements() <= 256)) {
892 auto &C = SrcTy->getContext();
893 auto EC = SrcTy->getElementCount();
894 VectorType *IdxTy = getVRGatherIndexType(LT.second, *ST, C);
896 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
897 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
898 return 2 * IndexCost +
899 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
900 LT.second, CostKind) +
901 MaskCost;
902 }
903 break;
904 }
905 }
906
907 auto shouldSplit = [](TTI::ShuffleKind Kind) {
908 switch (Kind) {
909 default:
910 return false;
914 return true;
915 }
916 };
917
918 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
919 shouldSplit(Kind)) {
920 InstructionCost SplitCost =
921 costShuffleViaSplitting(*this, LT.second, FVTp, Mask, CostKind);
922 if (SplitCost.isValid())
923 return SplitCost;
924 }
925 }
926
927 // Handle scalable vectors (and fixed vectors legalized to scalable vectors).
928 switch (Kind) {
929 default:
930 // Fallthrough to generic handling.
931 // TODO: Most of these cases will return getInvalid in generic code, and
932 // must be implemented here.
933 break;
935 // Extract at zero is always a subregister extract
936 if (Index == 0)
937 return TTI::TCC_Free;
938
939 // If we're extracting a subvector of at most m1 size at a sub-register
940 // boundary - which unfortunately we need exact vlen to identify - this is
941 // a subregister extract at worst and thus won't require a vslidedown.
942 // TODO: Extend for aligned m2, m4 subvector extracts
943 // TODO: Extend for misalgined (but contained) extracts
944 // TODO: Extend for scalable subvector types
945 if (std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
946 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
947 if (std::optional<unsigned> VLen = ST->getRealVLen();
948 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
949 SubLT.second.getSizeInBits() <= *VLen)
950 return TTI::TCC_Free;
951 }
952
953 // Example sequence:
954 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
955 // vslidedown.vi v8, v9, 2
956 return LT.first *
957 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second, CostKind);
959 // Example sequence:
960 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
961 // vslideup.vi v8, v9, 2
962 LT = getTypeLegalizationCost(DstTy);
963 return LT.first *
964 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second, CostKind);
965 case TTI::SK_Select: {
966 // Example sequence:
967 // li a0, 90
968 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
969 // vmv.s.x v0, a0
970 // vmerge.vvm v8, v9, v8, v0
971 // We use 2 for the cost of the mask materialization as this is the true
972 // cost for small masks and most shuffles are small. At worst, this cost
973 // should be a very small constant for the constant pool load. As such,
974 // we may bias towards large selects slightly more than truly warranted.
975 return LT.first *
976 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
977 LT.second, CostKind));
978 }
979 case TTI::SK_Broadcast: {
980 // Check for broadcast loads, which are synthesized by optimized zero-stride
981 // loads (this is checked in RISCVTTIImpl::isLegalBroadcastLoad).
982 bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
983 if (IsLoad && LT.second.isVector() &&
984 isLegalBroadcastLoad(SrcTy->getElementType(),
985 LT.second.getVectorElementCount()))
986 return 0;
987
988 bool HasScalar = (Args.size() > 0) && (Operator::getOpcode(Args[0]) ==
989 Instruction::InsertElement);
990 if (LT.second.getScalarSizeInBits() == 1) {
991 if (HasScalar) {
992 // Example sequence:
993 // andi a0, a0, 1
994 // vsetivli zero, 2, e8, mf8, ta, ma (ignored)
995 // vmv.v.x v8, a0
996 // vmsne.vi v0, v8, 0
997 return LT.first *
998 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
999 LT.second, CostKind));
1000 }
1001 // Example sequence:
1002 // vsetivli zero, 2, e8, mf8, ta, mu (ignored)
1003 // vmv.v.i v8, 0
1004 // vmerge.vim v8, v8, 1, v0
1005 // vmv.x.s a0, v8
1006 // andi a0, a0, 1
1007 // vmv.v.x v8, a0
1008 // vmsne.vi v0, v8, 0
1009
1010 return LT.first *
1011 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
1012 RISCV::VMV_X_S, RISCV::VMV_V_X,
1013 RISCV::VMSNE_VI},
1014 LT.second, CostKind));
1015 }
1016
1017 if (HasScalar) {
1018 // Example sequence:
1019 // vmv.v.x v8, a0
1020 return LT.first *
1021 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second, CostKind);
1022 }
1023
1024 // Example sequence:
1025 // vrgather.vi v9, v8, 0
1026 return LT.first *
1027 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second, CostKind);
1028 }
1029 case TTI::SK_Splice: {
1030 // vslidedown+vslideup.
1031 // TODO: Multiplying by LT.first implies this legalizes into multiple copies
1032 // of similar code, but I think we expand through memory.
1033 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
1034 if (Index >= 0 && Index < 32)
1035 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
1036 else if (Index < 0 && Index > -32)
1037 Opcodes[1] = RISCV::VSLIDEUP_VI;
1038 return LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1039 }
1040 case TTI::SK_Reverse: {
1041
1042 if (!LT.second.isVector())
1044
1045 // TODO: Cases to improve here:
1046 // * Illegal vector types
1047 // * i64 on RV32
1048 if (SrcTy->getElementType()->isIntegerTy(1)) {
1049 VectorType *WideTy =
1050 VectorType::get(IntegerType::get(SrcTy->getContext(), 8),
1051 cast<VectorType>(SrcTy)->getElementCount());
1052 return getCastInstrCost(Instruction::ZExt, WideTy, SrcTy,
1054 getShuffleCost(TTI::SK_Reverse, WideTy, WideTy, CostKind, {}, 0,
1055 nullptr) +
1056 getCastInstrCost(Instruction::Trunc, SrcTy, WideTy,
1058 }
1059
1060 MVT ContainerVT = LT.second;
1061 if (LT.second.isFixedLengthVector())
1062 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1063 MVT M1VT = RISCVTargetLowering::getM1VT(ContainerVT);
1064 if (ContainerVT.bitsLE(M1VT)) {
1065 // Example sequence:
1066 // csrr a0, vlenb
1067 // srli a0, a0, 3
1068 // addi a0, a0, -1
1069 // vsetvli a1, zero, e8, mf8, ta, mu (ignored)
1070 // vid.v v9
1071 // vrsub.vx v10, v9, a0
1072 // vrgather.vv v9, v8, v10
1073 InstructionCost LenCost = 3;
1074 if (LT.second.isFixedLengthVector())
1075 // vrsub.vi has a 5 bit immediate field, otherwise an li suffices
1076 LenCost = isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1077 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1078 if (LT.second.isFixedLengthVector() &&
1079 isInt<5>(LT.second.getVectorNumElements() - 1))
1080 Opcodes[1] = RISCV::VRSUB_VI;
1081 InstructionCost GatherCost =
1082 getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1083 return LT.first * (LenCost + GatherCost);
1084 }
1085
1086 // At high LMUL, we split into a series of M1 reverses (see
1087 // lowerVECTOR_REVERSE) and then do a single slide at the end to eliminate
1088 // the resulting gap at the bottom (for fixed vectors only). The important
1089 // bit is that the cost scales linearly, not quadratically with LMUL.
1090 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1091 InstructionCost FixedCost =
1092 getRISCVInstructionCost(M1Opcodes, M1VT, CostKind) + 3;
1093 unsigned Ratio =
1094 ContainerVT.getVectorMinNumElements() / M1VT.getVectorMinNumElements();
1095 InstructionCost GatherCost =
1096 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT, CostKind) * Ratio;
1097 InstructionCost SlideCost = !LT.second.isFixedLengthVector() ? 0 :
1098 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second, CostKind);
1099 return FixedCost + LT.first * (GatherCost + SlideCost);
1100 }
1101 }
1102 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1103 SubTp);
1104}
1105
1106static unsigned isM1OrSmaller(MVT VT) {
1108 return (LMUL == RISCVVType::VLMUL::LMUL_F8 ||
1112}
1113
1115 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
1116 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
1117 TTI::VectorInstrContext VIC) const {
1120
1121 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1122 // For now, skip all fixed vector cost analysis when P extension is available
1123 // to avoid crashes in getMinRVVVectorSizeInBits()
1124 if (ST->hasStdExtP() && isa<FixedVectorType>(Ty)) {
1125 return 1; // Treat as single instruction cost for now
1126 }
1127
1128 // A build_vector (which is m1 sized or smaller) can be done in no
1129 // worse than one vslide1down.vx per element in the type. We could
1130 // in theory do an explode_vector in the inverse manner, but our
1131 // lowering today does not have a first class node for this pattern.
1133 Ty, DemandedElts, Insert, Extract, CostKind);
1134 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1135 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1136 if (Ty->getScalarSizeInBits() == 1) {
1137 auto *WideVecTy = cast<VectorType>(Ty->getWithNewBitWidth(8));
1138 // Note: Implicit scalar anyextend is assumed to be free since the i1
1139 // must be stored in a GPR.
1140 return getScalarizationOverhead(WideVecTy, DemandedElts, Insert, Extract,
1141 CostKind) +
1142 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
1144 }
1145
1146 assert(LT.second.isFixedLengthVector());
1147 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1148 if (isM1OrSmaller(ContainerVT)) {
1149 InstructionCost BV =
1150 cast<FixedVectorType>(Ty)->getNumElements() *
1151 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second, CostKind);
1152 if (BV < Cost)
1153 Cost = BV;
1154 }
1155 }
1156 return Cost;
1157}
1158
1162 Type *DataTy = MICA.getDataType();
1163 Align Alignment = MICA.getAlignment();
1164 switch (MICA.getID()) {
1165 case Intrinsic::vp_load_ff: {
1166 EVT DataTypeVT = TLI->getValueType(DL, DataTy);
1167 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1169
1170 unsigned AS = MICA.getAddressSpace();
1171 return getMemoryOpCost(Instruction::Load, DataTy, Alignment, AS, CostKind,
1172 {TTI::OK_AnyValue, TTI::OP_None}, nullptr);
1173 }
1174 case Intrinsic::experimental_vp_strided_load:
1175 case Intrinsic::experimental_vp_strided_store:
1176 return getStridedMemoryOpCost(MICA, CostKind);
1177 case Intrinsic::masked_compressstore:
1178 case Intrinsic::masked_expandload:
1180 case Intrinsic::vp_scatter:
1181 case Intrinsic::vp_gather:
1182 case Intrinsic::masked_scatter:
1183 case Intrinsic::masked_gather:
1184 return getGatherScatterOpCost(MICA, CostKind);
1185 case Intrinsic::vp_load:
1186 case Intrinsic::vp_store:
1187 case Intrinsic::masked_load:
1188 case Intrinsic::masked_store:
1189 return getMaskedMemoryOpCost(MICA, CostKind);
1190 }
1192}
1193
1197 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
1198 : Instruction::Store;
1199 Type *Src = MICA.getDataType();
1200 Align Alignment = MICA.getAlignment();
1201 unsigned AddressSpace = MICA.getAddressSpace();
1202
1203 if (!isLegalMaskedLoadStore(Src, Alignment) ||
1206
1207 // Splitting involves additional evl arithmetic and vl toggles.
1208 InstructionCost SplitCost = 0;
1209 if (MICA.getID() == Intrinsic::vp_load ||
1210 MICA.getID() == Intrinsic::vp_store) {
1211 auto LT = getTypeLegalizationCost(Src);
1212 if (LT.first > 1)
1213 SplitCost += LT.first * TTI::TCC_Expensive;
1214 }
1215
1216 return getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind);
1217}
1218
1220 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1221 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1222 bool UseMaskForCond, bool UseMaskForGaps) const {
1223
1224 // The interleaved memory access pass will lower (de)interleave ops combined
1225 // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
1226 // only support masking per-iteration (i.e. condition), not per-segment (i.e.
1227 // gap).
1228 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1229 auto *VTy = cast<VectorType>(VecTy);
1230 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
1231 // Need to make sure type has't been scalarized
1232 if (LT.second.isVector()) {
1234 return LT.first * TTI::TCC_Basic;
1235
1236 auto *SubVecTy =
1237 VectorType::get(VTy->getElementType(),
1238 VTy->getElementCount().divideCoefficientBy(Factor));
1239 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1240 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1241 AddressSpace, DL)) {
1242
1243 // Some processors optimize segment loads/stores as N * DLEN sized
1244 // load ops + Factor * LMUL shuffle ops.
1245 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1246 unsigned VecSizeInBits =
1247 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1248 unsigned VLENForTuning =
1250 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1251 InstructionCost Cost = divideCeil(VecSizeInBits, DLENForTuning);
1252 MVT SubVecVT = getTLI()->getValueType(DL, SubVecTy).getSimpleVT();
1253 Cost += Factor * TLI->getLMULCost(SubVecVT);
1254 return Cost;
1255 }
1256
1257 // Otherwise, the cost is proportional to the number of elements (VL *
1258 // Factor ops).
1259 unsigned NumLoads = getEstimatedVLFor(VTy);
1260 return NumLoads * TTI::TCC_Basic;
1261 }
1262 }
1263 }
1264
1265 // TODO: Return the cost of interleaved accesses for scalable vector when
1266 // unable to convert to segment accesses instructions.
1267 if (isa<ScalableVectorType>(VecTy))
1269
1270 auto *FVTy = cast<FixedVectorType>(VecTy);
1271 // When gaps are only at the tail, for interleaved load, we can emit a wide
1272 // masked load and shufflevectors. For interleaved store, we can emit
1273 // shufflevectors and a wide masked store. The interleaved memory access pass
1274 // will lower them into vlsseg/vssseg intrinsics.
1275 if (UseMaskForGaps) {
1276 assert(llvm::is_sorted(Indices) && "Indices must be sorted");
1277 assert(llvm::adjacent_find(Indices) == Indices.end() &&
1278 "Indices should not contain duplicate elements");
1279 unsigned NumOfFields = Indices.size();
1280 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
1281 if (IsTailGapOnly &&
1282 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1283 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
1284 if (LT.second.isVector() &&
1285 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1286 auto *SubVecTy = VectorType::get(
1287 FVTy->getElementType(),
1288 FVTy->getElementCount().divideCoefficientBy(Factor));
1289 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1290 AddressSpace, DL)) {
1291 // The cost is proportional to the total number of element accesses.
1292 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1293 return NumAccesses * TTI::TCC_Basic;
1294 }
1295 }
1296 }
1297 }
1298
1299 InstructionCost MemCost =
1300 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
1301 unsigned VF = FVTy->getNumElements() / Factor;
1302
1303 // An interleaved load will look like this for Factor=3:
1304 // %wide.vec = load <12 x i32>, ptr %3, align 4
1305 // %strided.vec = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1306 // %strided.vec1 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1307 // %strided.vec2 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1308 if (Opcode == Instruction::Load) {
1309 InstructionCost Cost = MemCost;
1310 for (unsigned Index : Indices) {
1311 FixedVectorType *VecTy =
1312 FixedVectorType::get(FVTy->getElementType(), VF * Factor);
1313 auto Mask = createStrideMask(Index, Factor, VF);
1314 Mask.resize(VF * Factor, -1);
1315 InstructionCost ShuffleCost =
1317 CostKind, Mask, 0, nullptr, {});
1318 Cost += ShuffleCost;
1319 }
1320 return Cost;
1321 }
1322
1323 // TODO: Model for NF > 2
1324 // We'll need to enhance getShuffleCost to model shuffles that are just
1325 // inserts and extracts into subvectors, since they won't have the full cost
1326 // of a vrgather.
1327 // An interleaved store for 3 vectors of 4 lanes will look like
1328 // %11 = shufflevector <4 x i32> %4, <4 x i32> %6, <8 x i32> <0...7>
1329 // %12 = shufflevector <4 x i32> %9, <4 x i32> poison, <8 x i32> <0...3>
1330 // %13 = shufflevector <8 x i32> %11, <8 x i32> %12, <12 x i32> <0...11>
1331 // %interleaved.vec = shufflevector %13, poison, <12 x i32> <interleave mask>
1332 // store <12 x i32> %interleaved.vec, ptr %10, align 4
1333 if (Factor != 2)
1334 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1335 Alignment, AddressSpace, CostKind,
1336 UseMaskForCond, UseMaskForGaps);
1337
1338 assert(Opcode == Instruction::Store && "Opcode must be a store");
1339 // For an interleaving store of 2 vectors, we perform one large interleaving
1340 // shuffle that goes into the wide store
1341 auto Mask = createInterleaveMask(VF, Factor);
1342 InstructionCost ShuffleCost =
1344 CostKind, Mask, 0, nullptr, {});
1345 return MemCost + ShuffleCost;
1346}
1347
1351
1352 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
1353 MICA.getID() == Intrinsic::vp_gather;
1354 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1355 Type *DataTy = MICA.getDataType();
1356 Type *PtrTy = DataTy->getWithNewType(
1357 DL.getAddressType(DataTy->getContext(), MICA.getAddressSpace()));
1358 Align Alignment = MICA.getAlignment();
1361
1362 if ((Opcode == Instruction::Load &&
1363 !isLegalMaskedGather(DataTy, Align(Alignment))) ||
1364 (Opcode == Instruction::Store &&
1365 !isLegalMaskedScatter(DataTy, Align(Alignment))))
1367
1368 // Splitting vp intrinsics involves additional evl arithmetic and vl toggles.
1369 InstructionCost SplitCost = 0;
1370 if (MICA.getID() == Intrinsic::vp_gather ||
1371 MICA.getID() == Intrinsic::vp_scatter) {
1372 auto DataLT = getTypeLegalizationCost(DataTy);
1373 auto PtrLT = getTypeLegalizationCost(PtrTy);
1374 if (DataLT.first > 1)
1375 SplitCost += DataLT.first * TTI::TCC_Expensive;
1376 if (PtrLT.first > 1)
1377 SplitCost += PtrLT.first * TTI::TCC_Expensive;
1378 }
1379
1380 // Cost is proportional to the number of memory operations implied. For
1381 // scalable vectors, we use an estimate on that number since we don't
1382 // know exactly what VL will be.
1383 auto &VTy = *cast<VectorType>(DataTy);
1384 unsigned NumLoads = getEstimatedVLFor(&VTy);
1385 return SplitCost + NumLoads * TTI::TCC_Basic;
1386}
1387
1389 const MemIntrinsicCostAttributes &MICA,
1391 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
1392 ? Instruction::Load
1393 : Instruction::Store;
1394 Type *DataTy = MICA.getDataType();
1395 bool VariableMask = MICA.getVariableMask();
1396 Align Alignment = MICA.getAlignment();
1397 bool IsLegal = (Opcode == Instruction::Store &&
1398 isLegalMaskedCompressStore(DataTy, Alignment)) ||
1399 (Opcode == Instruction::Load &&
1400 isLegalMaskedExpandLoad(DataTy, Alignment));
1401 if (!IsLegal || CostKind != TTI::TCK_RecipThroughput)
1403 // Example compressstore sequence:
1404 // vsetivli zero, 8, e32, m2, ta, ma (ignored)
1405 // vcompress.vm v10, v8, v0
1406 // vcpop.m a1, v0
1407 // vsetvli zero, a1, e32, m2, ta, ma
1408 // vse32.v v10, (a0)
1409 // Example expandload sequence:
1410 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
1411 // vcpop.m a1, v0
1412 // vsetvli zero, a1, e32, m2, ta, ma
1413 // vle32.v v10, (a0)
1414 // vsetivli zero, 8, e32, m2, ta, ma
1415 // viota.m v12, v0
1416 // vrgather.vv v8, v10, v12, v0.t
1417 auto MemOpCost =
1418 getMemoryOpCost(Opcode, DataTy, Alignment, /*AddressSpace*/ 0, CostKind);
1419 auto LT = getTypeLegalizationCost(DataTy);
1420 SmallVector<unsigned, 4> Opcodes{RISCV::VSETVLI};
1421 if (VariableMask)
1422 Opcodes.push_back(RISCV::VCPOP_M);
1423 if (Opcode == Instruction::Store)
1424 Opcodes.append({RISCV::VCOMPRESS_VM});
1425 else
1426 Opcodes.append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1427 return MemOpCost +
1428 LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1429}
1430
1434 Type *DataTy = MICA.getDataType();
1435 Align Alignment = MICA.getAlignment();
1436
1437 if (!isLegalStridedLoadStore(DataTy, Alignment))
1439
1441 return TTI::TCC_Basic;
1442
1443 // Splitting vp intrinsics involves additional evl arithmetic and vl toggles.
1444 InstructionCost SplitCost = 0;
1445 auto LT = getTypeLegalizationCost(DataTy);
1446 if (LT.first > 1)
1447 SplitCost += LT.first * TTI::TCC_Expensive;
1448
1449 // Cost is proportional to the number of memory operations implied. For
1450 // scalable vectors, we use an estimate on that number since we don't
1451 // know exactly what VL will be.
1452 auto &VTy = *cast<VectorType>(DataTy);
1453 unsigned NumLoads = getEstimatedVLFor(&VTy);
1454 // Performant implementations of the vector extension will coalesce
1455 // elements if they fall on the same cache line
1456 uint64_t CacheLineBytes = ST->getCacheLineSize();
1457 if (!CacheLineBytes) // If no value, use default value of 64
1458 CacheLineBytes = 64;
1459 if (const ConstantInt *StrideCI =
1461 int64_t Stride = StrideCI->getSExtValue();
1462 // Bail early to avoid UB with std:abs() call
1463 if (Stride != std::numeric_limits<int64_t>::min() && Stride != 0) {
1464 uint64_t AbsStride = (uint64_t)std::abs(Stride);
1465 if (AbsStride < CacheLineBytes) {
1466 uint64_t MaxCombines = ST->getMaxVectorCoalesceElts();
1467 if ((CacheLineBytes / AbsStride) >= MaxCombines)
1468 NumLoads = divideCeil(NumLoads, MaxCombines);
1469 else
1470 // If we were to calculate CacheLineBytes / AbsStride first, would
1471 // lose accuracy
1472 NumLoads = divideCeil((NumLoads * AbsStride), CacheLineBytes);
1473 }
1474 }
1475 }
1476 return SplitCost + NumLoads * TTI::TCC_Basic;
1477}
1478
1481 // FIXME: This is a property of the default vector convention, not
1482 // all possible calling conventions. Fixing that will require
1483 // some TTI API and SLP rework.
1486 for (auto *Ty : Tys) {
1487 if (!Ty->isVectorTy())
1488 continue;
1489 Align A = DL.getPrefTypeAlign(Ty);
1490 Cost += getMemoryOpCost(Instruction::Store, Ty, A, 0, CostKind) +
1491 getMemoryOpCost(Instruction::Load, Ty, A, 0, CostKind);
1492 }
1493 return Cost;
1494}
1495
1496// Currently, these represent both throughput and codesize costs
1497// for the respective intrinsics. The costs in this table are simply
1498// instruction counts with the following adjustments made:
1499// * One vsetvli is considered free.
1501 {Intrinsic::floor, MVT::f32, 9},
1502 {Intrinsic::floor, MVT::f64, 9},
1503 {Intrinsic::ceil, MVT::f32, 9},
1504 {Intrinsic::ceil, MVT::f64, 9},
1505 {Intrinsic::trunc, MVT::f32, 7},
1506 {Intrinsic::trunc, MVT::f64, 7},
1507 {Intrinsic::round, MVT::f32, 9},
1508 {Intrinsic::round, MVT::f64, 9},
1509 {Intrinsic::roundeven, MVT::f32, 9},
1510 {Intrinsic::roundeven, MVT::f64, 9},
1511 {Intrinsic::rint, MVT::f32, 7},
1512 {Intrinsic::rint, MVT::f64, 7},
1513 {Intrinsic::nearbyint, MVT::f32, 9},
1514 {Intrinsic::nearbyint, MVT::f64, 9},
1515 {Intrinsic::bswap, MVT::i16, 3},
1516 {Intrinsic::bswap, MVT::i32, 12},
1517 {Intrinsic::bswap, MVT::i64, 31},
1518 {Intrinsic::bitreverse, MVT::i8, 17},
1519 {Intrinsic::bitreverse, MVT::i16, 24},
1520 {Intrinsic::bitreverse, MVT::i32, 33},
1521 {Intrinsic::bitreverse, MVT::i64, 52},
1522 {Intrinsic::ctpop, MVT::i8, 12},
1523 {Intrinsic::ctpop, MVT::i16, 19},
1524 {Intrinsic::ctpop, MVT::i32, 20},
1525 {Intrinsic::ctpop, MVT::i64, 21},
1526 {Intrinsic::ctlz, MVT::i8, 19},
1527 {Intrinsic::ctlz, MVT::i16, 28},
1528 {Intrinsic::ctlz, MVT::i32, 31},
1529 {Intrinsic::ctlz, MVT::i64, 35},
1530 {Intrinsic::cttz, MVT::i8, 16},
1531 {Intrinsic::cttz, MVT::i16, 23},
1532 {Intrinsic::cttz, MVT::i32, 24},
1533 {Intrinsic::cttz, MVT::i64, 25},
1534};
1535
1539 auto *RetTy = ICA.getReturnType();
1540 switch (ICA.getID()) {
1541 case Intrinsic::lrint:
1542 case Intrinsic::llrint:
1543 case Intrinsic::lround:
1544 case Intrinsic::llround: {
1545 auto LT = getTypeLegalizationCost(RetTy);
1546 Type *SrcTy = ICA.getArgTypes().front();
1547 auto SrcLT = getTypeLegalizationCost(SrcTy);
1548 if (ST->hasVInstructions() && LT.second.isVector()) {
1550 unsigned SrcEltSz = DL.getTypeSizeInBits(SrcTy->getScalarType());
1551 unsigned DstEltSz = DL.getTypeSizeInBits(RetTy->getScalarType());
1552 if (LT.second.getVectorElementType() == MVT::bf16) {
1553 if (!ST->hasVInstructionsBF16Minimal())
1555 if (DstEltSz == 32)
1556 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1557 else
1558 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1559 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1560 !ST->hasVInstructionsF16()) {
1561 if (!ST->hasVInstructionsF16Minimal())
1563 if (DstEltSz == 32)
1564 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1565 else
1566 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1567
1568 } else if (SrcEltSz > DstEltSz) {
1569 Ops = {RISCV::VFNCVT_X_F_W};
1570 } else if (SrcEltSz < DstEltSz) {
1571 Ops = {RISCV::VFWCVT_X_F_V};
1572 } else {
1573 Ops = {RISCV::VFCVT_X_F_V};
1574 }
1575
1576 // We need to use the source LMUL in the case of a narrowing op, and the
1577 // destination LMUL otherwise.
1578 if (SrcEltSz > DstEltSz)
1579 return SrcLT.first *
1580 getRISCVInstructionCost(Ops, SrcLT.second, CostKind);
1581 return LT.first * getRISCVInstructionCost(Ops, LT.second, CostKind);
1582 }
1583 break;
1584 }
1585 case Intrinsic::ceil:
1586 case Intrinsic::floor:
1587 case Intrinsic::trunc:
1588 case Intrinsic::rint:
1589 case Intrinsic::round:
1590 case Intrinsic::roundeven: {
1591 // These all use the same code.
1592 auto LT = getTypeLegalizationCost(RetTy);
1593 if (!LT.second.isVector() && TLI->isOperationCustom(ISD::FCEIL, LT.second))
1594 return LT.first * 8;
1595 break;
1596 }
1597 case Intrinsic::umin:
1598 case Intrinsic::umax:
1599 case Intrinsic::smin:
1600 case Intrinsic::smax: {
1601 auto LT = getTypeLegalizationCost(RetTy);
1602 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1603 return LT.first;
1604
1605 if (ST->hasVInstructions() && LT.second.isVector()) {
1606 unsigned Op;
1607 switch (ICA.getID()) {
1608 case Intrinsic::umin:
1609 Op = RISCV::VMINU_VV;
1610 break;
1611 case Intrinsic::umax:
1612 Op = RISCV::VMAXU_VV;
1613 break;
1614 case Intrinsic::smin:
1615 Op = RISCV::VMIN_VV;
1616 break;
1617 case Intrinsic::smax:
1618 Op = RISCV::VMAX_VV;
1619 break;
1620 }
1621 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1622 }
1623 break;
1624 }
1625 case Intrinsic::sadd_sat:
1626 case Intrinsic::ssub_sat:
1627 case Intrinsic::uadd_sat:
1628 case Intrinsic::usub_sat: {
1629 auto LT = getTypeLegalizationCost(RetTy);
1630 if (ST->hasVInstructions() && LT.second.isVector()) {
1631 unsigned Op;
1632 switch (ICA.getID()) {
1633 case Intrinsic::sadd_sat:
1634 Op = RISCV::VSADD_VV;
1635 break;
1636 case Intrinsic::ssub_sat:
1637 Op = RISCV::VSSUB_VV;
1638 break;
1639 case Intrinsic::uadd_sat:
1640 Op = RISCV::VSADDU_VV;
1641 break;
1642 case Intrinsic::usub_sat:
1643 Op = RISCV::VSSUBU_VV;
1644 break;
1645 }
1646 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1647 }
1648 break;
1649 }
1650 case Intrinsic::fma:
1651 case Intrinsic::fmuladd: {
1652 // TODO: handle promotion with f16/bf16 with zvfhmin/zvfbfmin
1653 auto LT = getTypeLegalizationCost(RetTy);
1654 if (ST->hasVInstructions() && LT.second.isVector())
1655 return LT.first *
1656 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second, CostKind);
1657 break;
1658 }
1659 case Intrinsic::fabs: {
1660 auto LT = getTypeLegalizationCost(RetTy);
1661 if (ST->hasVInstructions() && LT.second.isVector()) {
1662 // lui a0, 8
1663 // addi a0, a0, -1
1664 // vsetvli a1, zero, e16, m1, ta, ma
1665 // vand.vx v8, v8, a0
1666 // f16 with zvfhmin and bf16 with zvfhbmin
1667 if (LT.second.getVectorElementType() == MVT::bf16 ||
1668 (LT.second.getVectorElementType() == MVT::f16 &&
1669 !ST->hasVInstructionsF16()))
1670 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1671 CostKind) +
1672 2;
1673 else
1674 return LT.first *
1675 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second, CostKind);
1676 }
1677 break;
1678 }
1679 case Intrinsic::sqrt: {
1680 auto LT = getTypeLegalizationCost(RetTy);
1681 if (ST->hasVInstructions() && LT.second.isVector()) {
1684 MVT ConvType = LT.second;
1685 MVT FsqrtType = LT.second;
1686 // f16 with zvfhmin and bf16 with zvfbfmin and the type of nxv32[b]f16
1687 // will be spilt.
1688 if (LT.second.getVectorElementType() == MVT::bf16) {
1689 if (LT.second == MVT::nxv32bf16) {
1690 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1691 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1692 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1693 ConvType = MVT::nxv16f16;
1694 FsqrtType = MVT::nxv16f32;
1695 } else {
1696 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1697 FsqrtOp = {RISCV::VFSQRT_V};
1698 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1699 }
1700 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1701 !ST->hasVInstructionsF16()) {
1702 if (LT.second == MVT::nxv32f16) {
1703 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1704 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1705 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1706 ConvType = MVT::nxv16f16;
1707 FsqrtType = MVT::nxv16f32;
1708 } else {
1709 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1710 FsqrtOp = {RISCV::VFSQRT_V};
1711 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1712 }
1713 } else {
1714 FsqrtOp = {RISCV::VFSQRT_V};
1715 }
1716
1717 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType, CostKind) +
1718 getRISCVInstructionCost(ConvOp, ConvType, CostKind));
1719 }
1720 break;
1721 }
1722 case Intrinsic::cttz:
1723 case Intrinsic::ctlz:
1724 case Intrinsic::ctpop: {
1725 auto LT = getTypeLegalizationCost(RetTy);
1726 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1727 unsigned Op;
1728 switch (ICA.getID()) {
1729 case Intrinsic::cttz:
1730 Op = RISCV::VCTZ_V;
1731 break;
1732 case Intrinsic::ctlz:
1733 Op = RISCV::VCLZ_V;
1734 break;
1735 case Intrinsic::ctpop:
1736 Op = RISCV::VCPOP_V;
1737 break;
1738 }
1739 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1740 }
1741 break;
1742 }
1743 case Intrinsic::abs: {
1744 auto LT = getTypeLegalizationCost(RetTy);
1745 if (ST->hasVInstructions() && LT.second.isVector()) {
1746 // vabs.v v10, v8 (alias for vabd.vx v10, v8, zero)
1747 if (ST->hasStdExtZvabd())
1748 return LT.first *
1749 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second, CostKind);
1750
1751 // vrsub.vi v10, v8, 0
1752 // vmax.vv v8, v8, v10
1753 return LT.first *
1754 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1755 LT.second, CostKind);
1756 }
1757 break;
1758 }
1759 case Intrinsic::fshl:
1760 case Intrinsic::fshr: {
1761 if (ICA.getArgs().empty())
1762 break;
1763
1764 // Funnel-shifts are ROTL/ROTR when the first and second operand are equal.
1765 // When Zbb/Zbkb is enabled we can use a single ROL(W)/ROR(I)(W)
1766 // instruction.
1767 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1768 ICA.getArgs()[0] == ICA.getArgs()[1] &&
1769 (RetTy->getIntegerBitWidth() == 32 ||
1770 RetTy->getIntegerBitWidth() == 64) &&
1771 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1772 return 1;
1773 }
1774 break;
1775 }
1776 case Intrinsic::clmul: {
1777 auto LT = getTypeLegalizationCost(RetTy);
1778 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbkc()) {
1779 // TODO: Once custom lowering in this case for RV32 is added, this guard
1780 // should be removed and the cost model should be updated.
1781 if (!ST->is64Bit() || LT.second != MVT::i64)
1782 break;
1783 // vmv.s.x v8, a0
1784 // vclmul.vx v8, v8, a1
1785 // vmv.x.s a0, v8
1786 MVT VecVT = MVT::getScalableVectorVT(LT.second, 1);
1787 return LT.first * getRISCVInstructionCost(
1788 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1789 VecVT, CostKind);
1790 }
1791 break;
1792 }
1793 case Intrinsic::masked_udiv:
1794 return getArithmeticInstrCost(Instruction::UDiv, ICA.getReturnType(),
1795 CostKind);
1796 case Intrinsic::masked_sdiv:
1797 return getArithmeticInstrCost(Instruction::SDiv, ICA.getReturnType(),
1798 CostKind);
1799 case Intrinsic::masked_urem:
1800 return getArithmeticInstrCost(Instruction::URem, ICA.getReturnType(),
1801 CostKind);
1802 case Intrinsic::masked_srem:
1803 return getArithmeticInstrCost(Instruction::SRem, ICA.getReturnType(),
1804 CostKind);
1805 case Intrinsic::get_active_lane_mask: {
1806 if (ST->hasVInstructions()) {
1807 Type *ExpRetTy = VectorType::get(
1808 ICA.getArgTypes()[0], cast<VectorType>(RetTy)->getElementCount());
1809 auto LT = getTypeLegalizationCost(ExpRetTy);
1810
1811 // vid.v v8 // considered hoisted
1812 // vsaddu.vx v8, v8, a0
1813 // vmsltu.vx v0, v8, a1
1814 return LT.first *
1815 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1816 LT.second, CostKind);
1817 }
1818 break;
1819 }
1820 // TODO: add more intrinsic
1821 case Intrinsic::stepvector: {
1822 auto LT = getTypeLegalizationCost(RetTy);
1823 // Legalisation of illegal types involves an `index' instruction plus
1824 // (LT.first - 1) vector adds.
1825 if (ST->hasVInstructions())
1826 return getRISCVInstructionCost(RISCV::VID_V, LT.second, CostKind) +
1827 (LT.first - 1) *
1828 getRISCVInstructionCost(RISCV::VADD_VX, LT.second, CostKind);
1829 return 1 + (LT.first - 1);
1830 }
1831 case Intrinsic::vector_splice_left:
1832 case Intrinsic::vector_splice_right: {
1833 auto LT = getTypeLegalizationCost(RetTy);
1834 // Constant offsets fall through to getShuffleCost.
1835 if (!ICA.isTypeBasedOnly() && isa<ConstantInt>(ICA.getArgs()[2]))
1836 break;
1837 if (ST->hasVInstructions() && LT.second.isVector()) {
1838 return LT.first *
1839 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1840 LT.second, CostKind);
1841 }
1842 break;
1843 }
1844 case Intrinsic::experimental_cttz_elts: {
1845 if (!ST->hasVInstructions())
1846 break;
1848 Type *ArgTy = ICA.getArgTypes()[0];
1849 auto LT = getTypeLegalizationCost(ArgTy);
1850 if (!LT.second.isVector())
1851 break;
1852
1853 // If the element type is not i1, do a comparison with all-zeros.
1854 if (LT.second.getVectorElementType() != MVT::i1)
1855 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second, CostKind);
1856
1857 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind);
1858
1859 // If zero_is_poison is false, then we will generate additional
1860 // cmp + select instructions to convert -1 to EVL.
1861 Type *BoolTy = Type::getInt1Ty(RetTy->getContext());
1862 if (ICA.getArgs().size() > 1 &&
1863 cast<ConstantInt>(ICA.getArgs()[1])->isZero())
1864 Cost += getCmpSelInstrCost(Instruction::ICmp, BoolTy, RetTy,
1866 getCmpSelInstrCost(Instruction::Select, RetTy, BoolTy,
1868
1869 return LT.first * Cost;
1870 }
1871 case Intrinsic::experimental_vp_splice: {
1872 // To support type-based query from vectorizer, set the index to 0.
1873 // Note that index only change the cost from vslide.vx to vslide.vi and in
1874 // current implementations they have same costs.
1876 cast<VectorType>(ICA.getArgTypes()[0]), CostKind, {},
1878 }
1879 case Intrinsic::vp_merge: {
1880 // If an operand is a binary op and the type is legal, RISCVVectorPeephole
1881 // will likely fold the resulting vmerge.vvm away.
1883 getTypeLegalizationCost(RetTy).first == 1)
1884 return TTI::TCC_Free;
1885 break;
1886 }
1887 case Intrinsic::fptoui_sat:
1888 case Intrinsic::fptosi_sat: {
1890 bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
1891 Type *SrcTy = ICA.getArgTypes()[0];
1892
1893 auto SrcLT = getTypeLegalizationCost(SrcTy);
1894 auto DstLT = getTypeLegalizationCost(RetTy);
1895 if (!SrcTy->isVectorTy())
1896 break;
1897
1898 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1900
1901 Cost +=
1902 getCastInstrCost(IsSigned ? Instruction::FPToSI : Instruction::FPToUI,
1903 RetTy, SrcTy, TTI::CastContextHint::None, CostKind);
1904
1905 // Handle NaN.
1906 // vmfne v0, v8, v8 # If v8[i] is NaN set v0[i] to 1.
1907 // vmerge.vim v8, v8, 0, v0 # Convert NaN to 0.
1908 Type *CondTy = RetTy->getWithNewBitWidth(1);
1909 Cost += getCmpSelInstrCost(BinaryOperator::FCmp, SrcTy, CondTy,
1911 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
1913 return Cost;
1914 }
1915 case Intrinsic::experimental_vector_extract_last_active: {
1916 auto *ValTy = cast<VectorType>(ICA.getArgTypes()[0]);
1917 auto *MaskTy = cast<VectorType>(ICA.getArgTypes()[1]);
1918
1919 auto ValLT = getTypeLegalizationCost(ValTy);
1920 auto MaskLT = getTypeLegalizationCost(MaskTy);
1921
1922 // TODO: Return cheaper cost when the entire lane is inactive.
1923 // The expected asm sequence is:
1924 // vcpop.m a0, v0
1925 // beqz a0, exit # Return passthru when the entire lane is inactive.
1926 // vid v10, v0.t
1927 // vredmaxu.vs v10, v10, v10
1928 // vmv.x.s a0, v10
1929 // zext.b a0, a0
1930 // vslidedown.vx v8, v8, a0
1931 // vmv.x.s a0, v8
1932 // exit:
1933 // ...
1934
1935 // Find a suitable type for a stepvector.
1936 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
1937 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1938 TLI->getVectorIdxTy(getDataLayout()), MaskTy->getElementCount(),
1939 /*ZeroIsPoison=*/true, &VScaleRange);
1940 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1941 Type *StepTy = Type::getIntNTy(MaskTy->getContext(), EltWidth);
1942 auto *StepVecTy = VectorType::get(StepTy, ValTy->getElementCount());
1943 auto StepLT = getTypeLegalizationCost(StepVecTy);
1944
1945 // Currently expandVectorFindLastActive cannot handle step vector split.
1946 // So return invalid when the type needs split.
1947 // FIXME: Remove this if expandVectorFindLastActive supports split vector.
1948 if (StepLT.first > 1)
1950
1952 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1953
1954 Cost += MaskLT.first *
1955 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second, CostKind);
1956 Cost += getCFInstrCost(Instruction::CondBr, CostKind, nullptr);
1957 Cost += StepLT.first *
1958 getRISCVInstructionCost(Opcodes, StepLT.second, CostKind);
1959 Cost += getCastInstrCost(Instruction::ZExt,
1960 Type::getInt64Ty(ValTy->getContext()), StepTy,
1962 Cost += ValLT.first *
1963 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1964 ValLT.second, CostKind);
1965 return Cost;
1966 }
1967 case Intrinsic::vector_interleave2:
1968 case Intrinsic::vector_deinterleave2: {
1969 if (!ST->hasStdExtZvzip())
1970 break;
1971
1972 bool IsInterleave = ICA.getID() == Intrinsic::vector_interleave2;
1973 Type *InterleavedTy = IsInterleave ? RetTy : ICA.getArgTypes().front();
1974 // ISel does not select vzip.vv if either interleave2 input is undef.
1975 if (IsInterleave && !ICA.isTypeBasedOnly() &&
1976 any_of(ICA.getArgs(),
1977 [](const Value *Arg) { return isa<UndefValue>(Arg); }))
1978 break;
1979 if (InterleavedTy->getScalarSizeInBits() == 1)
1980 break;
1981
1982 if (auto *FVT = dyn_cast<FixedVectorType>(InterleavedTy)) {
1983 auto *HalfFVT = FixedVectorType::getHalfElementsVectorType(FVT);
1984 unsigned HalfVF = HalfFVT->getNumElements();
1985 if (IsInterleave)
1986 return getShuffleCost(TTI::SK_PermuteTwoSrc, FVT, HalfFVT, CostKind,
1987 createInterleaveMask(HalfVF, 2), 0, nullptr);
1989 for (unsigned Start = 0; Start != 2; ++Start)
1991 createStrideMask(Start, 2, HalfVF), 0, nullptr);
1992 return Cost;
1993 }
1994
1995 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(InterleavedTy);
1996 if (!LT.second.isScalableVector())
1997 break;
1998 if (IsInterleave) {
1999 if (std::optional<MVT> CostVT = getZvzipVZIPCostVT(LT.second))
2000 return LT.first *
2001 getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT, CostKind);
2002 } else if (std::optional<MVT> CostVT = getZvzipVUNZIPCostVT(LT.second)) {
2003 return LT.first *
2004 getRISCVInstructionCost({RISCV::VUNZIPE_V, RISCV::VUNZIPO_V},
2005 *CostVT, CostKind);
2006 }
2007 break;
2008 }
2009 }
2010
2011 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
2012 if (auto LT = getTypeLegalizationCost(RetTy);
2013 LT.second.isVector()) {
2014 MVT EltTy = LT.second.getVectorElementType();
2015 if (const auto *Entry = CostTableLookup(VectorIntrinsicCostTable,
2016 ICA.getID(), EltTy))
2017 return LT.first * Entry->Cost;
2018 }
2019 }
2020
2022}
2023
2026 const SCEV *Ptr,
2028 // Address computations for vector indexed load/store likely require an offset
2029 // and/or scaling.
2030 if (ST->hasVInstructions() && PtrTy->isVectorTy())
2031 return getArithmeticInstrCost(Instruction::Add, PtrTy, CostKind);
2032
2033 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
2034}
2035
2037 Type *Src,
2040 const Instruction *I) const {
2041 bool IsVectorType = isa<VectorType>(Dst) && isa<VectorType>(Src);
2042 if (!IsVectorType)
2043 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2044
2045 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2046 // For now, skip all fixed vector cost analysis when P extension is available
2047 // to avoid crashes in getMinRVVVectorSizeInBits()
2048 if (ST->hasStdExtP() &&
2050 return 1; // Treat as single instruction cost for now
2051 }
2052
2053 // FIXME: Need to compute legalizing cost for illegal types. The current
2054 // code handles only legal types and those which can be trivially
2055 // promoted to legal.
2056 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
2057 Dst->getScalarSizeInBits() > ST->getELen())
2058 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2059
2060 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2061 assert(ISD && "Invalid opcode");
2062 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
2063 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
2064
2065 // Handle i1 source and dest cases *before* calling logic in BasicTTI.
2066 // The shared implementation doesn't model vector widening during legalization
2067 // and instead assumes scalarization. In order to scalarize an <N x i1>
2068 // vector, we need to extend/trunc to/from i8. If we don't special case
2069 // this, we can get an infinite recursion cycle.
2070 switch (ISD) {
2071 default:
2072 break;
2073 case ISD::SIGN_EXTEND:
2074 case ISD::ZERO_EXTEND:
2075 if (Src->getScalarSizeInBits() == 1) {
2076 // We do not use vsext/vzext to extend from mask vector.
2077 // Instead we use the following instructions to extend from mask vector:
2078 // vmv.v.i v8, 0
2079 // vmerge.vim v8, v8, -1, v0 (repeated per split)
2080 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second, CostKind) +
2081 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
2082 DstLT.second, CostKind) +
2083 DstLT.first - 1;
2084 }
2085 break;
2086 case ISD::TRUNCATE:
2087 if (Dst->getScalarSizeInBits() == 1) {
2088 // We do not use several vncvt to truncate to mask vector. So we could
2089 // not use PowDiff to calculate it.
2090 // Instead we use the following instructions to truncate to mask vector:
2091 // vand.vi v8, v8, 1
2092 // vmsne.vi v0, v8, 0
2093 return SrcLT.first *
2094 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
2095 SrcLT.second, CostKind) +
2096 SrcLT.first - 1;
2097 }
2098 break;
2099 };
2100
2101 // Our actual lowering for the case where a wider legal type is available
2102 // uses promotion to the wider type. This is reflected in the result of
2103 // getTypeLegalizationCost, but BasicTTI assumes the widened cases are
2104 // scalarized if the legalized Src and Dst are not equal sized.
2105 const DataLayout &DL = this->getDataLayout();
2106 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
2107 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2108 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Src),
2109 SrcLT.second.getSizeInBits()) ||
2110 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Dst),
2111 DstLT.second.getSizeInBits()) ||
2112 SrcLT.first > 1 || DstLT.first > 1)
2113 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2114
2115 // The split cost is handled by the base getCastInstrCost
2116 assert((SrcLT.first == 1) && (DstLT.first == 1) && "Illegal type");
2117
2118 int PowDiff = (int)Log2_32(DstLT.second.getScalarSizeInBits()) -
2119 (int)Log2_32(SrcLT.second.getScalarSizeInBits());
2120 switch (ISD) {
2121 case ISD::SIGN_EXTEND:
2122 case ISD::ZERO_EXTEND: {
2123 if ((PowDiff < 1) || (PowDiff > 3))
2124 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2125 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2126 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2127 unsigned Op =
2128 (ISD == ISD::SIGN_EXTEND) ? SExtOp[PowDiff - 1] : ZExtOp[PowDiff - 1];
2129 return getRISCVInstructionCost(Op, DstLT.second, CostKind);
2130 }
2131 case ISD::TRUNCATE:
2132 case ISD::FP_EXTEND:
2133 case ISD::FP_ROUND: {
2134 // Counts of narrow/widen instructions.
2135 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2136 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2137
2138 unsigned Op = (ISD == ISD::TRUNCATE) ? RISCV::VNSRL_WI
2139 : (ISD == ISD::FP_EXTEND) ? RISCV::VFWCVT_F_F_V
2140 : RISCV::VFNCVT_F_F_W;
2142 for (; SrcEltSize != DstEltSize;) {
2143 MVT ElementMVT = (ISD == ISD::TRUNCATE)
2144 ? MVT::getIntegerVT(DstEltSize)
2145 : MVT::getFloatingPointVT(DstEltSize);
2146 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2147 DstEltSize =
2148 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2149 Cost += getRISCVInstructionCost(Op, DstMVT, CostKind);
2150 }
2151 return Cost;
2152 }
2153 case ISD::FP_TO_SINT:
2154 case ISD::FP_TO_UINT: {
2155 unsigned IsSigned = ISD == ISD::FP_TO_SINT;
2156 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2157 unsigned FWCVT =
2158 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2159 unsigned FNCVT =
2160 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2161 unsigned SrcEltSize = Src->getScalarSizeInBits();
2162 unsigned DstEltSize = Dst->getScalarSizeInBits();
2164 if ((SrcEltSize == 16) &&
2165 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2166 // If the target only supports zvfhmin or it is fp16-to-i64 conversion
2167 // pre-widening to f32 and then convert f32 to integer
2168 VectorType *VecF32Ty =
2169 VectorType::get(Type::getFloatTy(Dst->getContext()),
2170 cast<VectorType>(Dst)->getElementCount());
2171 std::pair<InstructionCost, MVT> VecF32LT =
2172 getTypeLegalizationCost(VecF32Ty);
2173 Cost +=
2174 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2175 VecF32LT.second, CostKind);
2176 Cost += getCastInstrCost(Opcode, Dst, VecF32Ty, CCH, CostKind, I);
2177 return Cost;
2178 }
2179 if (DstEltSize == SrcEltSize)
2180 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2181 else if (DstEltSize > SrcEltSize)
2182 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2183 else { // (SrcEltSize > DstEltSize)
2184 // First do a narrowing conversion to an integer half the size, then
2185 // truncate if needed.
2186 MVT ElementVT = MVT::getIntegerVT(SrcEltSize / 2);
2187 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2188 Cost += getRISCVInstructionCost(FNCVT, VecVT, CostKind);
2189 if ((SrcEltSize / 2) > DstEltSize) {
2190 Type *VecTy = EVT(VecVT).getTypeForEVT(Dst->getContext());
2191 Cost +=
2192 getCastInstrCost(Instruction::Trunc, Dst, VecTy, CCH, CostKind, I);
2193 }
2194 }
2195 return Cost;
2196 }
2197 case ISD::SINT_TO_FP:
2198 case ISD::UINT_TO_FP: {
2199 unsigned IsSigned = ISD == ISD::SINT_TO_FP;
2200 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2201 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2202 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2203 unsigned SrcEltSize = Src->getScalarSizeInBits();
2204 unsigned DstEltSize = Dst->getScalarSizeInBits();
2205
2207 if ((DstEltSize == 16) &&
2208 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2209 // If the target only supports zvfhmin or it is i64-to-fp16 conversion
2210 // it is converted to f32 and then converted to f16
2211 VectorType *VecF32Ty =
2212 VectorType::get(Type::getFloatTy(Dst->getContext()),
2213 cast<VectorType>(Dst)->getElementCount());
2214 std::pair<InstructionCost, MVT> VecF32LT =
2215 getTypeLegalizationCost(VecF32Ty);
2216 Cost += getCastInstrCost(Opcode, VecF32Ty, Src, CCH, CostKind, I);
2217 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2218 DstLT.second, CostKind);
2219 return Cost;
2220 }
2221
2222 if (DstEltSize == SrcEltSize)
2223 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2224 else if (DstEltSize > SrcEltSize) {
2225 if ((DstEltSize / 2) > SrcEltSize) {
2226 VectorType *VecTy =
2227 VectorType::get(IntegerType::get(Dst->getContext(), DstEltSize / 2),
2228 cast<VectorType>(Dst)->getElementCount());
2229 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2230 Cost += getCastInstrCost(Op, VecTy, Src, CCH, CostKind, I);
2231 }
2232 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2233 } else
2234 Cost += getRISCVInstructionCost(FNCVT, DstLT.second, CostKind);
2235 return Cost;
2236 }
2237 }
2238 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2239}
2240
2241unsigned RISCVTTIImpl::getEstimatedVLFor(VectorType *Ty) const {
2242 if (isa<ScalableVectorType>(Ty)) {
2243 const unsigned EltSize = DL.getTypeSizeInBits(Ty->getElementType());
2244 const unsigned MinSize = DL.getTypeSizeInBits(Ty).getKnownMinValue();
2245 const unsigned VectorBits = *getVScaleForTuning() * RISCV::RVVBitsPerBlock;
2246 return RISCVTargetLowering::computeVLMAX(VectorBits, EltSize, MinSize);
2247 }
2248 return cast<FixedVectorType>(Ty)->getNumElements();
2249}
2250
2253 FastMathFlags FMF,
2255 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2256 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2257
2258 // Skip if scalar size of Ty is bigger than ELEN.
2259 if (Ty->getScalarSizeInBits() > ST->getELen())
2260 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2261
2262 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2263 if (Ty->getElementType()->isIntegerTy(1)) {
2264 // SelectionDAGBuilder does following transforms:
2265 // vector_reduce_{smin,umax}(<n x i1>) --> vector_reduce_or(<n x i1>)
2266 // vector_reduce_{smax,umin}(<n x i1>) --> vector_reduce_and(<n x i1>)
2267 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2268 return getArithmeticReductionCost(Instruction::Or, Ty, FMF, CostKind);
2269 else
2270 return getArithmeticReductionCost(Instruction::And, Ty, FMF, CostKind);
2271 }
2272
2273 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2275 InstructionCost ExtraCost = 0;
2276 switch (IID) {
2277 case Intrinsic::maximum:
2278 if (FMF.noNaNs()) {
2279 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2280 } else {
2281 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2282 RISCV::VFMV_F_S};
2283 // Cost of Canonical Nan + branch
2284 // lui a0, 523264
2285 // fmv.w.x fa0, a0
2286 Type *DstTy = Ty->getScalarType();
2287 const unsigned EltTyBits = DstTy->getScalarSizeInBits();
2288 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2289 ExtraCost = 1 +
2290 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2292 getCFInstrCost(Instruction::CondBr, CostKind);
2293 }
2294 break;
2295
2296 case Intrinsic::minimum:
2297 if (FMF.noNaNs()) {
2298 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2299 } else {
2300 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2301 RISCV::VFMV_F_S};
2302 // Cost of Canonical Nan + branch
2303 // lui a0, 523264
2304 // fmv.w.x fa0, a0
2305 Type *DstTy = Ty->getScalarType();
2306 const unsigned EltTyBits = DL.getTypeSizeInBits(DstTy);
2307 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2308 ExtraCost = 1 +
2309 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2311 getCFInstrCost(Instruction::CondBr, CostKind);
2312 }
2313 break;
2314 }
2315 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2316 }
2317
2318 // IR Reduction is composed by one rvv reduction instruction and vmv
2319 unsigned SplitOp;
2321 switch (IID) {
2322 default:
2323 llvm_unreachable("Unsupported intrinsic");
2324 case Intrinsic::smax:
2325 SplitOp = RISCV::VMAX_VV;
2326 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2327 break;
2328 case Intrinsic::smin:
2329 SplitOp = RISCV::VMIN_VV;
2330 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2331 break;
2332 case Intrinsic::umax:
2333 SplitOp = RISCV::VMAXU_VV;
2334 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2335 break;
2336 case Intrinsic::umin:
2337 SplitOp = RISCV::VMINU_VV;
2338 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2339 break;
2340 case Intrinsic::maxnum:
2341 SplitOp = RISCV::VFMAX_VV;
2342 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2343 break;
2344 case Intrinsic::minnum:
2345 SplitOp = RISCV::VFMIN_VV;
2346 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2347 break;
2348 }
2349 // Add a cost for data larger than LMUL8
2350 InstructionCost SplitCost =
2351 (LT.first > 1) ? (LT.first - 1) *
2352 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2353 : 0;
2354 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2355}
2356
2359 std::optional<FastMathFlags> FMF,
2361 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2362 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2363
2364 // Skip if scalar size of Ty is bigger than ELEN.
2365 if (Ty->getScalarSizeInBits() > ST->getELen())
2366 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2367
2368 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2369 assert(ISD && "Invalid opcode");
2370
2371 if (ISD != ISD::ADD && ISD != ISD::OR && ISD != ISD::XOR && ISD != ISD::AND &&
2372 ISD != ISD::FADD)
2373 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2374
2375 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2376 Type *ElementTy = Ty->getElementType();
2377 if (ElementTy->isIntegerTy(1)) {
2378 // Example sequences:
2379 // vfirst.m a0, v0
2380 // seqz a0, a0
2381 if (LT.second == MVT::v1i1)
2382 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind) +
2383 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2385
2386 if (ISD == ISD::AND) {
2387 // Example sequences:
2388 // vmand.mm v8, v9, v8 ; needed every time type is split
2389 // vmnot.m v8, v0 ; alias for vmnand
2390 // vcpop.m a0, v8
2391 // seqz a0, a0
2392
2393 // See the discussion: https://github.com/llvm/llvm-project/pull/119160
2394 // For LMUL <= 8, there is no splitting,
2395 // the sequences are vmnot, vcpop and seqz.
2396 // When LMUL > 8 and split = 1,
2397 // the sequences are vmnand, vcpop and seqz.
2398 // When LMUL > 8 and split > 1,
2399 // the sequences are (LT.first-2) * vmand, vmnand, vcpop and seqz.
2400 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2401 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second, CostKind) +
2402 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second, CostKind) +
2403 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2404 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2406 } else if (ISD == ISD::XOR || ISD == ISD::ADD) {
2407 // Example sequences:
2408 // vsetvli a0, zero, e8, mf8, ta, ma
2409 // vmxor.mm v8, v0, v8 ; needed every time type is split
2410 // vcpop.m a0, v8
2411 // andi a0, a0, 1
2412 return (LT.first - 1) *
2413 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind) +
2414 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) + 1;
2415 } else {
2416 assert(ISD == ISD::OR);
2417 // Example sequences:
2418 // vsetvli a0, zero, e8, mf8, ta, ma
2419 // vmor.mm v8, v9, v8 ; needed every time type is split
2420 // vcpop.m a0, v0
2421 // snez a0, a0
2422 return (LT.first - 1) *
2423 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second, CostKind) +
2424 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2425 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2427 }
2428 }
2429
2430 // IR Reduction of or/and is composed by one vmv and one rvv reduction
2431 // instruction, and others is composed by two vmv and one rvv reduction
2432 // instruction
2433 unsigned SplitOp;
2435 switch (ISD) {
2436 case ISD::ADD:
2437 SplitOp = RISCV::VADD_VV;
2438 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2439 break;
2440 case ISD::OR:
2441 SplitOp = RISCV::VOR_VV;
2442 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2443 break;
2444 case ISD::XOR:
2445 SplitOp = RISCV::VXOR_VV;
2446 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2447 break;
2448 case ISD::AND:
2449 SplitOp = RISCV::VAND_VV;
2450 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2451 break;
2452 case ISD::FADD:
2453 // We can't promote f16/bf16 fadd reductions.
2454 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2455 LT.second.getScalarType() == MVT::bf16)
2456 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2458 Opcodes.push_back(RISCV::VFMV_S_F);
2459 for (unsigned i = 0; i < LT.first.getValue(); i++)
2460 Opcodes.push_back(RISCV::VFREDOSUM_VS);
2461 Opcodes.push_back(RISCV::VFMV_F_S);
2462 return getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2463 }
2464 SplitOp = RISCV::VFADD_VV;
2465 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2466 break;
2467 }
2468 // Add a cost for data larger than LMUL8
2469 InstructionCost SplitCost =
2470 (LT.first > 1) ? (LT.first - 1) *
2471 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2472 : 0;
2473 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2474}
2475
2477 unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy,
2478 std::optional<FastMathFlags> FMF, TTI::TargetCostKind CostKind) const {
2479 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2480 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2481 FMF, CostKind);
2482
2483 // Skip if scalar size of ResTy is bigger than ELEN.
2484 if (ResTy->getScalarSizeInBits() > ST->getELen())
2485 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2486 FMF, CostKind);
2487
2488 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2489 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2490 FMF, CostKind);
2491
2492 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2493
2494 if (IsUnsigned && Opcode == Instruction::Add &&
2495 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2496 // Represent vector_reduce_add(ZExt(<n x i1>)) as
2497 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
2498 return LT.first *
2499 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind);
2500 }
2501
2502 if (ResTy->getScalarSizeInBits() != 2 * LT.second.getScalarSizeInBits())
2503 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2504 FMF, CostKind);
2505
2506 return (LT.first - 1) +
2507 getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
2508}
2509
2513 assert(OpInfo.isConstant() && "non constant operand?");
2514 if (!isa<VectorType>(Ty))
2515 // FIXME: We need to account for immediate materialization here, but doing
2516 // a decent job requires more knowledge about the immediate than we
2517 // currently have here.
2518 return 0;
2519
2520 if (OpInfo.isUniform())
2521 // vmv.v.i, vmv.v.x, or vfmv.v.f
2522 // We ignore the cost of the scalar constant materialization to be consistent
2523 // with how we treat scalar constants themselves just above.
2524 return 1;
2525
2526 return getConstantPoolLoadCost(Ty, CostKind);
2527}
2528
2530 Align Alignment,
2531 unsigned AddressSpace,
2533 TTI::OperandValueInfo OpInfo,
2534 const Instruction *I) const {
2535 EVT VT = TLI->getValueType(DL, Src, true);
2536 // Type legalization can't handle structs, and load latency isn't handled here
2537 if (VT == MVT::Other ||
2538 (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency))
2539 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2540 CostKind, OpInfo, I);
2541
2543 if (Opcode == Instruction::Store && OpInfo.isConstant())
2544 Cost += getStoreImmCost(Src, OpInfo, CostKind);
2545
2546 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
2547
2548 InstructionCost BaseCost = [&]() {
2549 InstructionCost Cost = LT.first;
2551 return Cost;
2552
2553 // Our actual lowering for the case where a wider legal type is available
2554 // uses the a VL predicated load on the wider type. This is reflected in
2555 // the result of getTypeLegalizationCost, but BasicTTI assumes the
2556 // widened cases are scalarized.
2557 const DataLayout &DL = this->getDataLayout();
2558 if (Src->isVectorTy() && LT.second.isVector() &&
2559 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
2560 LT.second.getSizeInBits()))
2561 return Cost;
2562
2563 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2564 CostKind, OpInfo, I);
2565 }();
2566
2567 // Assume memory ops cost scale with the number of vector registers
2568 // possible accessed by the instruction. Note that BasicTTI already
2569 // handles the LT.first term for us.
2570 if (ST->hasVInstructions() && LT.second.isVector() &&
2572 BaseCost *= TLI->getLMULCost(LT.second);
2573 return Cost + BaseCost;
2574}
2575
2577 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
2579 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
2581 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2582 Op1Info, Op2Info, I);
2583
2584 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2585 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2586 Op1Info, Op2Info, I);
2587
2588 // Skip if scalar size of ValTy is bigger than ELEN.
2589 if (ValTy->isVectorTy() && ValTy->getScalarSizeInBits() > ST->getELen())
2590 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2591 Op1Info, Op2Info, I);
2592
2593 auto GetConstantMatCost =
2594 [&](TTI::OperandValueInfo OpInfo) -> InstructionCost {
2595 if (OpInfo.isUniform())
2596 // We return 0 we currently ignore the cost of materializing scalar
2597 // constants in GPRs.
2598 return 0;
2599
2600 return getConstantPoolLoadCost(ValTy, CostKind);
2601 };
2602
2603 InstructionCost ConstantMatCost;
2604 if (Op1Info.isConstant())
2605 ConstantMatCost += GetConstantMatCost(Op1Info);
2606 if (Op2Info.isConstant())
2607 ConstantMatCost += GetConstantMatCost(Op2Info);
2608
2609 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2610 if (Opcode == Instruction::Select && LT.second.isVector()) {
2611 if (CondTy->isVectorTy()) {
2612 if (ValTy->getScalarSizeInBits() == 1) {
2613 // vmandn.mm v8, v8, v9
2614 // vmand.mm v9, v0, v9
2615 // vmor.mm v0, v9, v8
2616 return ConstantMatCost +
2617 LT.first *
2618 getRISCVInstructionCost(
2619 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2620 LT.second, CostKind);
2621 }
2622 // vselect and max/min are supported natively.
2623 return ConstantMatCost +
2624 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2625 CostKind);
2626 }
2627
2628 if (ValTy->getScalarSizeInBits() == 1) {
2629 // vmv.v.x v9, a0
2630 // vmsne.vi v9, v9, 0
2631 // vmandn.mm v8, v8, v9
2632 // vmand.mm v9, v0, v9
2633 // vmor.mm v0, v9, v8
2634 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2635 return ConstantMatCost +
2636 LT.first *
2637 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2638 InterimVT, CostKind) +
2639 LT.first * getRISCVInstructionCost(
2640 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2641 LT.second, CostKind);
2642 }
2643
2644 // vmv.v.x v10, a0
2645 // vmsne.vi v0, v10, 0
2646 // vmerge.vvm v8, v9, v8, v0
2647 return ConstantMatCost +
2648 LT.first * getRISCVInstructionCost(
2649 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2650 LT.second, CostKind);
2651 }
2652
2653 if ((Opcode == Instruction::ICmp) && ValTy->isVectorTy() &&
2654 CmpInst::isIntPredicate(VecPred)) {
2655 // Use VMSLT_VV to represent VMSEQ, VMSNE, VMSLTU, VMSLEU, VMSLT, VMSLE
2656 // provided they incur the same cost across all implementations
2657 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2658 LT.second,
2659 CostKind);
2660 }
2661
2662 if ((Opcode == Instruction::FCmp) && ValTy->isVectorTy() &&
2663 CmpInst::isFPPredicate(VecPred)) {
2664
2665 // Use VMXOR_MM and VMXNOR_MM to generate all true/false mask
2666 if ((VecPred == CmpInst::FCMP_FALSE) || (VecPred == CmpInst::FCMP_TRUE))
2667 return ConstantMatCost +
2668 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind);
2669
2670 // If we do not support the input floating point vector type, use the base
2671 // one which will calculate as:
2672 // ScalarizeCost + Num * Cost for fixed vector,
2673 // InvalidCost for scalable vector.
2674 if ((ValTy->getScalarSizeInBits() == 16 && !ST->hasVInstructionsF16()) ||
2675 (ValTy->getScalarSizeInBits() == 32 && !ST->hasVInstructionsF32()) ||
2676 (ValTy->getScalarSizeInBits() == 64 && !ST->hasVInstructionsF64()))
2677 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2678 Op1Info, Op2Info, I);
2679
2680 // Assuming vector fp compare and mask instructions are all the same cost
2681 // until a need arises to differentiate them.
2682 switch (VecPred) {
2683 case CmpInst::FCMP_ONE: // vmflt.vv + vmflt.vv + vmor.mm
2684 case CmpInst::FCMP_ORD: // vmfeq.vv + vmfeq.vv + vmand.mm
2685 case CmpInst::FCMP_UNO: // vmfne.vv + vmfne.vv + vmor.mm
2686 case CmpInst::FCMP_UEQ: // vmflt.vv + vmflt.vv + vmnor.mm
2687 return ConstantMatCost +
2688 LT.first * getRISCVInstructionCost(
2689 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2690 LT.second, CostKind);
2691
2692 case CmpInst::FCMP_UGT: // vmfle.vv + vmnot.m
2693 case CmpInst::FCMP_UGE: // vmflt.vv + vmnot.m
2694 case CmpInst::FCMP_ULT: // vmfle.vv + vmnot.m
2695 case CmpInst::FCMP_ULE: // vmflt.vv + vmnot.m
2696 return ConstantMatCost +
2697 LT.first *
2698 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2699 LT.second, CostKind);
2700
2701 case CmpInst::FCMP_OEQ: // vmfeq.vv
2702 case CmpInst::FCMP_OGT: // vmflt.vv
2703 case CmpInst::FCMP_OGE: // vmfle.vv
2704 case CmpInst::FCMP_OLT: // vmflt.vv
2705 case CmpInst::FCMP_OLE: // vmfle.vv
2706 case CmpInst::FCMP_UNE: // vmfne.vv
2707 return ConstantMatCost +
2708 LT.first *
2709 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second, CostKind);
2710 default:
2711 break;
2712 }
2713 }
2714
2715 // With ShortForwardBranchOpt or ConditionalMoveFusion, scalar icmp + select
2716 // instructions will lower to SELECT_CC and lower to PseudoCCMOVGPR which will
2717 // generate a conditional branch + mv. The cost of scalar (icmp + select) will
2718 // be (0 + select instr cost).
2719 if (ST->hasConditionalMoveFusion() && I && isa<ICmpInst>(I) &&
2720 ValTy->isIntegerTy() && !I->user_empty()) {
2721 if (all_of(I->users(), [&](const User *U) {
2722 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2723 U->getType()->isIntegerTy() &&
2724 !isa<ConstantData>(U->getOperand(1)) &&
2725 !isa<ConstantData>(U->getOperand(2));
2726 }))
2727 return 0;
2728 }
2729
2730 // TODO: Add cost for scalar type.
2731
2732 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2733 Op1Info, Op2Info, I);
2734}
2735
2738 const Instruction *I) const {
2740 return Opcode == Instruction::PHI ? 0 : 1;
2741 // Branches are assumed to be predicted.
2742 return 0;
2743}
2744
2746 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
2747 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
2748 assert(Val->isVectorTy() && "This must be a vector type");
2749
2750 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2751 // For now, skip all fixed vector cost analysis when P extension is available
2752 // to avoid crashes in getMinRVVVectorSizeInBits()
2753 if (ST->hasStdExtP() && isa<FixedVectorType>(Val)) {
2754 return 1; // Treat as single instruction cost for now
2755 }
2756
2757 if (Opcode != Instruction::ExtractElement &&
2758 Opcode != Instruction::InsertElement)
2759 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
2760 VIC);
2761
2762 // Scalar splat operand can be folded for vector ops that support splatting
2763 // the scalar operand, so the explicit insertelement is free in this context.
2764 if (Opcode == Instruction::InsertElement &&
2765 VIC == TTI::VectorInstrContext::SplatOpFolded &&
2766 ST->sinkSplatOperands() && Index == 0)
2767 return TTI::TCC_Free;
2768
2769 // Legalize the type.
2770 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
2771
2772 // This type is legalized to a scalar type.
2773 if (!LT.second.isVector()) {
2774 auto *FixedVecTy = cast<FixedVectorType>(Val);
2775 // If Index is a known constant, cost is zero.
2776 if (Index != -1U)
2777 return 0;
2778 // Extract/InsertElement with non-constant index is very costly when
2779 // scalarized; estimate cost of loads/stores sequence via the stack:
2780 // ExtractElement cost: store vector to stack, load scalar;
2781 // InsertElement cost: store vector to stack, store scalar, load vector.
2782 Type *ElemTy = FixedVecTy->getElementType();
2783 auto NumElems = FixedVecTy->getNumElements();
2784 auto Align = DL.getPrefTypeAlign(ElemTy);
2785 InstructionCost LoadCost =
2786 getMemoryOpCost(Instruction::Load, ElemTy, Align, 0, CostKind);
2787 InstructionCost StoreCost =
2788 getMemoryOpCost(Instruction::Store, ElemTy, Align, 0, CostKind);
2789 return Opcode == Instruction::ExtractElement
2790 ? StoreCost * NumElems + LoadCost
2791 : (StoreCost + LoadCost) * NumElems + StoreCost;
2792 }
2793
2794 // For unsupported scalable vector.
2795 if (LT.second.isScalableVector() && !LT.first.isValid())
2796 return LT.first;
2797
2798 // Mask vector extract/insert is expanded via e8.
2799 if (Val->getScalarSizeInBits() == 1) {
2800 VectorType *WideTy =
2802 cast<VectorType>(Val)->getElementCount());
2803 if (Opcode == Instruction::ExtractElement) {
2804 InstructionCost ExtendCost
2805 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2807 InstructionCost ExtractCost
2808 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2809 return ExtendCost + ExtractCost;
2810 }
2811 InstructionCost ExtendCost
2812 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2814 InstructionCost InsertCost
2815 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2816 InstructionCost TruncCost
2817 = getCastInstrCost(Instruction::Trunc, Val, WideTy,
2819 return ExtendCost + InsertCost + TruncCost;
2820 }
2821
2822
2823 // In RVV, we could use vslidedown + vmv.x.s to extract element from vector
2824 // and vslideup + vmv.s.x to insert element to vector.
2825 unsigned MoveOpc;
2826 if (LT.second.isFloatingPoint())
2827 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2828 : RISCV::VFMV_F_S;
2829 else
2830 MoveOpc =
2831 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2832 InstructionCost BaseCost =
2833 getRISCVInstructionCost(MoveOpc, LT.second, CostKind);
2834 // When insertelement we should add the index with 1 as the input of vslideup.
2835 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2836
2837 if (Index != -1U) {
2838 // The type may be split. For fixed-width vectors we can normalize the
2839 // index to the new type.
2840 if (LT.second.isFixedLengthVector()) {
2841 unsigned Width = LT.second.getVectorNumElements();
2842 Index = Index % Width;
2843 }
2844
2845 // If exact VLEN is known, we will insert/extract into the appropriate
2846 // subvector with no additional subvector insert/extract cost.
2847 if (auto VLEN = ST->getRealVLen()) {
2848 unsigned EltSize = LT.second.getScalarSizeInBits();
2849 unsigned M1Max = *VLEN / EltSize;
2850 Index = Index % M1Max;
2851 }
2852
2853 if (Index == 0)
2854 // We can extract/insert the first element without vslidedown/vslideup.
2855 SlideCost = 0;
2856 else if (Opcode == Instruction::InsertElement)
2857 SlideCost = 1; // With a constant index, we do not need to use addi.
2858 }
2859
2860 // When the vector needs to split into multiple register groups and the index
2861 // exceeds single vector register group, we need to insert/extract the element
2862 // via stack.
2863 if (LT.first > 1 &&
2864 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2865 LT.second.isScalableVector()))) {
2866 Type *ScalarType = Val->getScalarType();
2867 Align VecAlign = DL.getPrefTypeAlign(Val);
2868 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
2869 // Extra addi for unknown index.
2870 InstructionCost IdxCost = Index == -1U ? 1 : 0;
2871
2872 // Store all split vectors into stack and load the target element.
2873 if (Opcode == Instruction::ExtractElement)
2874 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2875 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
2876 CostKind) +
2877 IdxCost;
2878
2879 // Store all split vectors into stack and store the target element and load
2880 // vectors back.
2881 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2882 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind) +
2883 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
2884 CostKind) +
2885 IdxCost;
2886 }
2887
2888 // Extract i64 in the target that has XLEN=32 need more instruction.
2889 if (Val->getScalarType()->isIntegerTy() &&
2890 ST->getXLen() < Val->getScalarSizeInBits()) {
2891 // For extractelement, we need the following instructions:
2892 // vsetivli zero, 1, e64, m1, ta, mu (not count)
2893 // vslidedown.vx v8, v8, a0
2894 // vmv.x.s a0, v8
2895 // li a1, 32
2896 // vsrl.vx v8, v8, a1
2897 // vmv.x.s a1, v8
2898
2899 // For insertelement, we need the following instructions:
2900 // vsetivli zero, 2, e32, m4, ta, ma (don't count)
2901 // vslide1down.vx v12, v8, a0
2902 // vslide1down.vx v12, v12, a1
2903 // addi a0, a2, 1
2904 // vsetvli zero, a0, e64, m4, tu, ma (don't count)
2905 // vslideup.vx v8, v12, a2
2906
2907 // TODO: should we count these special vsetvlis?
2908 BaseCost =
2909 Opcode == Instruction::InsertElement
2910 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2911 RISCV::VSLIDE1DOWN_VX,
2912 RISCV::VSLIDEUP_VX},
2913 LT.second, CostKind)
2914 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2915 RISCV::VSRL_VX, RISCV::VMV_X_S},
2916 LT.second, CostKind);
2917 }
2918 return BaseCost + SlideCost;
2919}
2920
2924 unsigned Index) const {
2925 if (isa<FixedVectorType>(Val))
2927 Index);
2928
2929 // TODO: This code replicates what LoopVectorize.cpp used to do when asking
2930 // for the cost of extracting the last lane of a scalable vector. It probably
2931 // needs a more accurate cost.
2932 ElementCount EC = cast<VectorType>(Val)->getElementCount();
2933 assert(Index < EC.getKnownMinValue() && "Unexpected reverse index");
2934 return getVectorInstrCost(Opcode, Val, CostKind,
2935 EC.getKnownMinValue() - 1 - Index, nullptr,
2936 nullptr);
2937}
2938
2939/// Check to see if this instruction is expected to be combined to a simpler
2940/// operation during/before lowering. If so return the cost of the combined
2941/// operation rather than provided one. For instance, `udiv i16 %X, 2` is likely
2942/// to be combined to `lshr i16 %X, 1`, so return the cost of a `lshr` rather
2943/// than the cost of a `udiv`
2944std::optional<InstructionCost>
2946 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2948 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
2949 // Vector unsigned division/remainder will be simplified to shifts/masks.
2950 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2951 Opd2Info.isConstant() && Opd2Info.isPowerOf2()) {
2952 if (Opcode == Instruction::UDiv)
2953 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind, Opd1Info,
2954 Opd2Info.getNoProps());
2955 // UREM
2956 return getArithmeticInstrCost(Instruction::And, Ty, CostKind, Opd1Info,
2957 Opd2Info.getNoProps());
2958 }
2959 return std::nullopt;
2960}
2961
2963 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2965 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
2966
2967 // TODO: Handle more cost kinds.
2969 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2970 Args, CtxI);
2971
2972 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2973 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2974 Args, CtxI);
2975
2976 // Skip if scalar size of Ty is bigger than ELEN.
2977 if (isa<VectorType>(Ty) && Ty->getScalarSizeInBits() > ST->getELen())
2978 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2979 Args, CtxI);
2980
2981 if (std::optional<InstructionCost> CombinedCost =
2983 Op2Info, Args, CtxI))
2984 return *CombinedCost;
2985
2986 // Legalize the type.
2987 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2988 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2989
2990 // TODO: Handle scalar type.
2991 if (!LT.second.isVector()) {
2992 static const CostTblEntry DivTbl[]{
2993 {ISD::UDIV, MVT::i32, TTI::TCC_Expensive},
2994 {ISD::UDIV, MVT::i64, TTI::TCC_Expensive},
2995 {ISD::SDIV, MVT::i32, TTI::TCC_Expensive},
2996 {ISD::SDIV, MVT::i64, TTI::TCC_Expensive},
2997 {ISD::UREM, MVT::i32, TTI::TCC_Expensive},
2998 {ISD::UREM, MVT::i64, TTI::TCC_Expensive},
2999 {ISD::SREM, MVT::i32, TTI::TCC_Expensive},
3000 {ISD::SREM, MVT::i64, TTI::TCC_Expensive}};
3001 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
3002 if (const auto *Entry = CostTableLookup(DivTbl, ISDOpcode, LT.second))
3003 return Entry->Cost * LT.first;
3004
3005 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
3006 Args, CtxI);
3007 }
3008
3009 // f16 with zvfhmin and bf16 will be promoted to f32.
3010 // FIXME: nxv32[b]f16 will be custom lowered and split.
3011 InstructionCost CastCost = 0;
3012 if ((LT.second.getVectorElementType() == MVT::f16 ||
3013 LT.second.getVectorElementType() == MVT::bf16) &&
3014 TLI->getOperationAction(ISDOpcode, LT.second) ==
3016 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
3017 Type *PromotedTy = EVT(PromotedVT).getTypeForEVT(Ty->getContext());
3018 Type *LegalTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
3019 // Add cost of extending arguments
3020 CastCost += LT.first * Args.size() *
3021 getCastInstrCost(Instruction::FPExt, PromotedTy, LegalTy,
3023 // Add cost of truncating result
3024 CastCost +=
3025 LT.first * getCastInstrCost(Instruction::FPTrunc, LegalTy, PromotedTy,
3027 // Compute cost of op in promoted type
3028 LT.second = PromotedVT;
3029 }
3030
3031 auto getConstantMatCost =
3032 [&](unsigned Operand, TTI::OperandValueInfo OpInfo) -> InstructionCost {
3033 if (OpInfo.isUniform() && canSplatOperand(Opcode, Operand))
3034 // Two sub-cases:
3035 // * Has a 5 bit immediate operand which can be splatted.
3036 // * Has a larger immediate which must be materialized in scalar register
3037 // We return 0 for both as we currently ignore the cost of materializing
3038 // scalar constants in GPRs.
3039 return 0;
3040
3041 return getConstantPoolLoadCost(Ty, CostKind);
3042 };
3043
3044 // Add the cost of materializing any constant vectors required.
3045 InstructionCost ConstantMatCost = 0;
3046 if (Op1Info.isConstant())
3047 ConstantMatCost += getConstantMatCost(0, Op1Info);
3048 if (Op2Info.isConstant())
3049 ConstantMatCost += getConstantMatCost(1, Op2Info);
3050
3051 unsigned Op;
3052 switch (ISDOpcode) {
3053 case ISD::ADD:
3054 case ISD::SUB:
3055 Op = RISCV::VADD_VV;
3056 break;
3057 case ISD::SHL:
3058 case ISD::SRL:
3059 case ISD::SRA:
3060 Op = RISCV::VSLL_VV;
3061 break;
3062 case ISD::AND:
3063 case ISD::OR:
3064 case ISD::XOR:
3065 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
3066 break;
3067 case ISD::MUL:
3068 case ISD::MULHS:
3069 case ISD::MULHU:
3070 Op = RISCV::VMUL_VV;
3071 break;
3072 case ISD::SDIV:
3073 case ISD::UDIV:
3074 Op = RISCV::VDIV_VV;
3075 break;
3076 case ISD::SREM:
3077 case ISD::UREM:
3078 Op = RISCV::VREM_VV;
3079 break;
3080 case ISD::FADD:
3081 case ISD::FSUB:
3082 Op = RISCV::VFADD_VV;
3083 break;
3084 case ISD::FMUL:
3085 Op = RISCV::VFMUL_VV;
3086 break;
3087 case ISD::FDIV:
3088 Op = RISCV::VFDIV_VV;
3089 break;
3090 case ISD::FNEG:
3091 Op = RISCV::VFSGNJN_VV;
3092 break;
3093 default:
3094 // Assuming all other instructions have the same cost until a need arises to
3095 // differentiate them.
3096 return CastCost + ConstantMatCost +
3097 BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
3098 Args, CtxI);
3099 }
3100
3101 InstructionCost InstrCost = getRISCVInstructionCost(Op, LT.second, CostKind);
3102 // We use BasicTTIImpl to calculate scalar costs, which assumes floating point
3103 // ops are twice as expensive as integer ops. Do the same for vectors so
3104 // scalar floating point ops aren't cheaper than their vector equivalents.
3105 if (Ty->isFPOrFPVectorTy())
3106 InstrCost *= 2;
3107 return CastCost + ConstantMatCost + LT.first * InstrCost;
3108}
3109
3110// TODO: Deduplicate from TargetTransformInfoImplCRTPBase.
3112 ArrayRef<const Value *> Ptrs, const Value *Base,
3113 const TTI::PointersChainInfo &Info, Type *AccessTy,
3114 const TTI::TargetCostKind CostKind) const {
3116 // In the basic model we take into account GEP instructions only
3117 // (although here can come alloca instruction, a value, constants and/or
3118 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
3119 // pointer). Typically, if Base is a not a GEP-instruction and all the
3120 // pointers are relative to the same base address, all the rest are
3121 // either GEP instructions, PHIs, bitcasts or constants. When we have same
3122 // base, we just calculate cost of each non-Base GEP as an ADD operation if
3123 // any their index is a non-const.
3124 // If no known dependencies between the pointers cost is calculated as a sum
3125 // of costs of GEP instructions.
3126 for (auto [I, V] : enumerate(Ptrs)) {
3127 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
3128 if (!GEP)
3129 continue;
3130 if (Info.isSameBase() && V != Base) {
3131 if (GEP->hasAllConstantIndices())
3132 continue;
3133 // If the chain is unit-stride and BaseReg + stride*i is a legal
3134 // addressing mode, then presume the base GEP is sitting around in a
3135 // register somewhere and check if we can fold the offset relative to
3136 // it.
3137 unsigned Stride = DL.getTypeStoreSize(AccessTy);
3138 if (Info.isUnitStride() &&
3139 isLegalAddressingMode(AccessTy,
3140 /* BaseGV */ nullptr,
3141 /* BaseOffset */ Stride * I,
3142 /* HasBaseReg */ true,
3143 /* Scale */ 0,
3144 GEP->getType()->getPointerAddressSpace()))
3145 continue;
3146 Cost += getArithmeticInstrCost(Instruction::Add, GEP->getType(), CostKind,
3147 {TTI::OK_AnyValue, TTI::OP_None},
3148 {TTI::OK_AnyValue, TTI::OP_None}, {});
3149 } else {
3150 SmallVector<const Value *> Indices(GEP->indices());
3151 Cost += getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
3152 Indices, CostKind, AccessTy);
3153 }
3154 }
3155 return Cost;
3156}
3157
3160 OptimizationRemarkEmitter *ORE) const {
3161 // TODO: More tuning on benchmarks and metrics with changes as needed
3162 // would apply to all settings below to enable performance.
3163
3164
3165 if (ST->enableDefaultUnroll())
3166 return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP, ORE);
3167
3168 // Enable Upper bound unrolling universally, not dependent upon the conditions
3169 // below.
3170 UP.UpperBound = true;
3171
3172 // Disable loop unrolling for Oz and Os.
3173 UP.OptSizeThreshold = 0;
3175 if (L->getHeader()->getParent()->hasOptSize())
3176 return;
3177
3178 SmallVector<BasicBlock *, 4> ExitingBlocks;
3179 L->getExitingBlocks(ExitingBlocks);
3180 LLVM_DEBUG(dbgs() << "Loop has:\n"
3181 << "Blocks: " << L->getNumBlocks() << "\n"
3182 << "Exit blocks: " << ExitingBlocks.size() << "\n");
3183
3184 // Only allow another exit other than the latch. This acts as an early exit
3185 // as it mirrors the profitability calculation of the runtime unroller.
3186 if (ExitingBlocks.size() > 2)
3187 return;
3188
3189 // Limit the CFG of the loop body for targets with a branch predictor.
3190 // Allowing 4 blocks permits if-then-else diamonds in the body.
3191 if (L->getNumBlocks() > 4)
3192 return;
3193
3194 // Scan the loop: don't unroll loops with calls as this could prevent
3195 // inlining. Don't unroll auto-vectorized loops either, though do allow
3196 // unrolling of the scalar remainder.
3197 bool IsVectorized = getBooleanLoopAttribute(L, "llvm.loop.isvectorized");
3199 for (auto *BB : L->getBlocks()) {
3200 for (auto &I : *BB) {
3201 // Both auto-vectorized loops and the scalar remainder have the
3202 // isvectorized attribute, so differentiate between them by the presence
3203 // of vector instructions.
3204 if (IsVectorized && (I.getType()->isVectorTy() ||
3205 llvm::any_of(I.operand_values(), [](Value *V) {
3206 return V->getType()->isVectorTy();
3207 })))
3208 return;
3209
3210 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
3211 const Function *F = cast<CallBase>(I).getCalledFunction();
3212 if (!F || isLoweredToCall(F))
3213 return;
3214 }
3215
3216 SmallVector<const Value *> Operands(I.operand_values());
3219 }
3220 }
3221
3222 LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");
3223
3224 UP.Partial = true;
3225 UP.Runtime = true;
3226 UP.UnrollRemainder = true;
3227 UP.UnrollAndJam = true;
3228
3229 // Force unrolling small loops can be very useful because of the branch
3230 // taken cost of the backedge.
3231 if (Cost < 12)
3232 UP.Force = true;
3233}
3234
3239
3241 MemIntrinsicInfo &Info) const {
3242 const DataLayout &DL = getDataLayout();
3243 Intrinsic::ID IID = Inst->getIntrinsicID();
3244 LLVMContext &C = Inst->getContext();
3245 bool HasMask = false;
3246
3247 auto getSegNum = [](const IntrinsicInst *II, unsigned PtrOperandNo,
3248 bool IsWrite) -> int64_t {
3249 if (auto *TarExtTy =
3250 dyn_cast<TargetExtType>(II->getArgOperand(0)->getType()))
3251 return TarExtTy->getIntParameter(0);
3252
3253 return 1;
3254 };
3255
3256 switch (IID) {
3257 case Intrinsic::riscv_vle_mask:
3258 case Intrinsic::riscv_vse_mask:
3259 case Intrinsic::riscv_vlseg2_mask:
3260 case Intrinsic::riscv_vlseg3_mask:
3261 case Intrinsic::riscv_vlseg4_mask:
3262 case Intrinsic::riscv_vlseg5_mask:
3263 case Intrinsic::riscv_vlseg6_mask:
3264 case Intrinsic::riscv_vlseg7_mask:
3265 case Intrinsic::riscv_vlseg8_mask:
3266 case Intrinsic::riscv_vsseg2_mask:
3267 case Intrinsic::riscv_vsseg3_mask:
3268 case Intrinsic::riscv_vsseg4_mask:
3269 case Intrinsic::riscv_vsseg5_mask:
3270 case Intrinsic::riscv_vsseg6_mask:
3271 case Intrinsic::riscv_vsseg7_mask:
3272 case Intrinsic::riscv_vsseg8_mask:
3273 HasMask = true;
3274 [[fallthrough]];
3275 case Intrinsic::riscv_vle:
3276 case Intrinsic::riscv_vse:
3277 case Intrinsic::riscv_vlseg2:
3278 case Intrinsic::riscv_vlseg3:
3279 case Intrinsic::riscv_vlseg4:
3280 case Intrinsic::riscv_vlseg5:
3281 case Intrinsic::riscv_vlseg6:
3282 case Intrinsic::riscv_vlseg7:
3283 case Intrinsic::riscv_vlseg8:
3284 case Intrinsic::riscv_vsseg2:
3285 case Intrinsic::riscv_vsseg3:
3286 case Intrinsic::riscv_vsseg4:
3287 case Intrinsic::riscv_vsseg5:
3288 case Intrinsic::riscv_vsseg6:
3289 case Intrinsic::riscv_vsseg7:
3290 case Intrinsic::riscv_vsseg8: {
3291 // Intrinsic interface:
3292 // riscv_vle(merge, ptr, vl)
3293 // riscv_vle_mask(merge, ptr, mask, vl, policy)
3294 // riscv_vse(val, ptr, vl)
3295 // riscv_vse_mask(val, ptr, mask, vl, policy)
3296 // riscv_vlseg#(merge, ptr, vl, sew)
3297 // riscv_vlseg#_mask(merge, ptr, mask, vl, policy, sew)
3298 // riscv_vsseg#(val, ptr, vl, sew)
3299 // riscv_vsseg#_mask(val, ptr, mask, vl, sew)
3300 bool IsWrite = Inst->getType()->isVoidTy();
3301 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3302 // The results of segment loads are TargetExtType.
3303 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3304 unsigned SEW =
3305 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3306 ->getZExtValue();
3307 Ty = TarExtTy->getTypeParameter(0U);
3309 IntegerType::get(C, SEW),
3310 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3311 }
3312 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3313 unsigned VLIndex = RVVIInfo->VLOperand;
3314 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3315 MaybeAlign Alignment =
3316 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3317 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3318 Value *Mask = ConstantInt::getTrue(MaskType);
3319 if (HasMask)
3320 Mask = Inst->getArgOperand(VLIndex - 1);
3321 Value *EVL = Inst->getArgOperand(VLIndex);
3322 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3323 // RVV uses contiguous elements as a segment.
3324 if (SegNum > 1) {
3325 unsigned ElemSize = Ty->getScalarSizeInBits();
3326 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3327 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3328 }
3329 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3330 Alignment, Mask, EVL);
3331 return true;
3332 }
3333 case Intrinsic::riscv_vlse_mask:
3334 case Intrinsic::riscv_vsse_mask:
3335 case Intrinsic::riscv_vlsseg2_mask:
3336 case Intrinsic::riscv_vlsseg3_mask:
3337 case Intrinsic::riscv_vlsseg4_mask:
3338 case Intrinsic::riscv_vlsseg5_mask:
3339 case Intrinsic::riscv_vlsseg6_mask:
3340 case Intrinsic::riscv_vlsseg7_mask:
3341 case Intrinsic::riscv_vlsseg8_mask:
3342 case Intrinsic::riscv_vssseg2_mask:
3343 case Intrinsic::riscv_vssseg3_mask:
3344 case Intrinsic::riscv_vssseg4_mask:
3345 case Intrinsic::riscv_vssseg5_mask:
3346 case Intrinsic::riscv_vssseg6_mask:
3347 case Intrinsic::riscv_vssseg7_mask:
3348 case Intrinsic::riscv_vssseg8_mask:
3349 HasMask = true;
3350 [[fallthrough]];
3351 case Intrinsic::riscv_vlse:
3352 case Intrinsic::riscv_vsse:
3353 case Intrinsic::riscv_vlsseg2:
3354 case Intrinsic::riscv_vlsseg3:
3355 case Intrinsic::riscv_vlsseg4:
3356 case Intrinsic::riscv_vlsseg5:
3357 case Intrinsic::riscv_vlsseg6:
3358 case Intrinsic::riscv_vlsseg7:
3359 case Intrinsic::riscv_vlsseg8:
3360 case Intrinsic::riscv_vssseg2:
3361 case Intrinsic::riscv_vssseg3:
3362 case Intrinsic::riscv_vssseg4:
3363 case Intrinsic::riscv_vssseg5:
3364 case Intrinsic::riscv_vssseg6:
3365 case Intrinsic::riscv_vssseg7:
3366 case Intrinsic::riscv_vssseg8: {
3367 // Intrinsic interface:
3368 // riscv_vlse(merge, ptr, stride, vl)
3369 // riscv_vlse_mask(merge, ptr, stride, mask, vl, policy)
3370 // riscv_vsse(val, ptr, stride, vl)
3371 // riscv_vsse_mask(val, ptr, stride, mask, vl, policy)
3372 // riscv_vlsseg#(merge, ptr, offset, vl, sew)
3373 // riscv_vlsseg#_mask(merge, ptr, offset, mask, vl, policy, sew)
3374 // riscv_vssseg#(val, ptr, offset, vl, sew)
3375 // riscv_vssseg#_mask(val, ptr, offset, mask, vl, sew)
3376 bool IsWrite = Inst->getType()->isVoidTy();
3377 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3378 // The results of segment loads are TargetExtType.
3379 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3380 unsigned SEW =
3381 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3382 ->getZExtValue();
3383 Ty = TarExtTy->getTypeParameter(0U);
3385 IntegerType::get(C, SEW),
3386 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3387 }
3388 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3389 unsigned VLIndex = RVVIInfo->VLOperand;
3390 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3391 MaybeAlign Alignment =
3392 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3393
3394 Value *Stride = Inst->getArgOperand(PtrOperandNo + 1);
3395 // Use the pointer alignment as the element alignment if the stride is a
3396 // multiple of the pointer alignment. Otherwise, the element alignment
3397 // should be the greatest common divisor of pointer alignment and stride.
3398 // For simplicity, just consider unalignment for elements.
3399 unsigned PointerAlign = Alignment.valueOrOne().value();
3400 if (!isa<ConstantInt>(Stride) ||
3401 cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
3402 Alignment = Align(1);
3403
3404 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3405 Value *Mask = ConstantInt::getTrue(MaskType);
3406 if (HasMask)
3407 Mask = Inst->getArgOperand(VLIndex - 1);
3408 Value *EVL = Inst->getArgOperand(VLIndex);
3409 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3410 // RVV uses contiguous elements as a segment.
3411 if (SegNum > 1) {
3412 unsigned ElemSize = Ty->getScalarSizeInBits();
3413 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3414 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3415 }
3416 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3417 Alignment, Mask, EVL, Stride);
3418 return true;
3419 }
3420 case Intrinsic::riscv_vloxei_mask:
3421 case Intrinsic::riscv_vluxei_mask:
3422 case Intrinsic::riscv_vsoxei_mask:
3423 case Intrinsic::riscv_vsuxei_mask:
3424 case Intrinsic::riscv_vloxseg2_mask:
3425 case Intrinsic::riscv_vloxseg3_mask:
3426 case Intrinsic::riscv_vloxseg4_mask:
3427 case Intrinsic::riscv_vloxseg5_mask:
3428 case Intrinsic::riscv_vloxseg6_mask:
3429 case Intrinsic::riscv_vloxseg7_mask:
3430 case Intrinsic::riscv_vloxseg8_mask:
3431 case Intrinsic::riscv_vluxseg2_mask:
3432 case Intrinsic::riscv_vluxseg3_mask:
3433 case Intrinsic::riscv_vluxseg4_mask:
3434 case Intrinsic::riscv_vluxseg5_mask:
3435 case Intrinsic::riscv_vluxseg6_mask:
3436 case Intrinsic::riscv_vluxseg7_mask:
3437 case Intrinsic::riscv_vluxseg8_mask:
3438 case Intrinsic::riscv_vsoxseg2_mask:
3439 case Intrinsic::riscv_vsoxseg3_mask:
3440 case Intrinsic::riscv_vsoxseg4_mask:
3441 case Intrinsic::riscv_vsoxseg5_mask:
3442 case Intrinsic::riscv_vsoxseg6_mask:
3443 case Intrinsic::riscv_vsoxseg7_mask:
3444 case Intrinsic::riscv_vsoxseg8_mask:
3445 case Intrinsic::riscv_vsuxseg2_mask:
3446 case Intrinsic::riscv_vsuxseg3_mask:
3447 case Intrinsic::riscv_vsuxseg4_mask:
3448 case Intrinsic::riscv_vsuxseg5_mask:
3449 case Intrinsic::riscv_vsuxseg6_mask:
3450 case Intrinsic::riscv_vsuxseg7_mask:
3451 case Intrinsic::riscv_vsuxseg8_mask:
3452 HasMask = true;
3453 [[fallthrough]];
3454 case Intrinsic::riscv_vloxei:
3455 case Intrinsic::riscv_vluxei:
3456 case Intrinsic::riscv_vsoxei:
3457 case Intrinsic::riscv_vsuxei:
3458 case Intrinsic::riscv_vloxseg2:
3459 case Intrinsic::riscv_vloxseg3:
3460 case Intrinsic::riscv_vloxseg4:
3461 case Intrinsic::riscv_vloxseg5:
3462 case Intrinsic::riscv_vloxseg6:
3463 case Intrinsic::riscv_vloxseg7:
3464 case Intrinsic::riscv_vloxseg8:
3465 case Intrinsic::riscv_vluxseg2:
3466 case Intrinsic::riscv_vluxseg3:
3467 case Intrinsic::riscv_vluxseg4:
3468 case Intrinsic::riscv_vluxseg5:
3469 case Intrinsic::riscv_vluxseg6:
3470 case Intrinsic::riscv_vluxseg7:
3471 case Intrinsic::riscv_vluxseg8:
3472 case Intrinsic::riscv_vsoxseg2:
3473 case Intrinsic::riscv_vsoxseg3:
3474 case Intrinsic::riscv_vsoxseg4:
3475 case Intrinsic::riscv_vsoxseg5:
3476 case Intrinsic::riscv_vsoxseg6:
3477 case Intrinsic::riscv_vsoxseg7:
3478 case Intrinsic::riscv_vsoxseg8:
3479 case Intrinsic::riscv_vsuxseg2:
3480 case Intrinsic::riscv_vsuxseg3:
3481 case Intrinsic::riscv_vsuxseg4:
3482 case Intrinsic::riscv_vsuxseg5:
3483 case Intrinsic::riscv_vsuxseg6:
3484 case Intrinsic::riscv_vsuxseg7:
3485 case Intrinsic::riscv_vsuxseg8: {
3486 // Intrinsic interface (only listed ordered version):
3487 // riscv_vloxei(merge, ptr, index, vl)
3488 // riscv_vloxei_mask(merge, ptr, index, mask, vl, policy)
3489 // riscv_vsoxei(val, ptr, index, vl)
3490 // riscv_vsoxei_mask(val, ptr, index, mask, vl, policy)
3491 // riscv_vloxseg#(merge, ptr, index, vl, sew)
3492 // riscv_vloxseg#_mask(merge, ptr, index, mask, vl, policy, sew)
3493 // riscv_vsoxseg#(val, ptr, index, vl, sew)
3494 // riscv_vsoxseg#_mask(val, ptr, index, mask, vl, sew)
3495 bool IsWrite = Inst->getType()->isVoidTy();
3496 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3497 // The results of segment loads are TargetExtType.
3498 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3499 unsigned SEW =
3500 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3501 ->getZExtValue();
3502 Ty = TarExtTy->getTypeParameter(0U);
3504 IntegerType::get(C, SEW),
3505 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3506 }
3507 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3508 unsigned VLIndex = RVVIInfo->VLOperand;
3509 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3510 Value *Mask;
3511 if (HasMask) {
3512 Mask = Inst->getArgOperand(VLIndex - 1);
3513 } else {
3514 // Mask cannot be nullptr here: vector GEP produces <vscale x N x ptr>,
3515 // and casting that to scalar i64 triggers a vector/scalar mismatch
3516 // assertion in CreatePointerCast. Use an all-true mask so ASan lowers it
3517 // via extractelement instead.
3518 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3519 Mask = ConstantInt::getTrue(MaskType);
3520 }
3521 Value *EVL = Inst->getArgOperand(VLIndex);
3522 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3523 // RVV uses contiguous elements as a segment.
3524 if (SegNum > 1) {
3525 unsigned ElemSize = Ty->getScalarSizeInBits();
3526 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3527 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3528 }
3529 Value *OffsetOp = Inst->getArgOperand(PtrOperandNo + 1);
3530 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3531 Align(1), Mask, EVL,
3532 /* Stride */ nullptr, OffsetOp);
3533 return true;
3534 }
3535 }
3536 return false;
3537}
3538
3540 if (Ty->isVectorTy()) {
3541 // f16 with only zvfhmin and bf16 will be promoted to f32
3542 Type *EltTy = cast<VectorType>(Ty)->getElementType();
3543 if ((EltTy->isHalfTy() && !ST->hasVInstructionsF16()) ||
3544 EltTy->isBFloatTy())
3545 Ty = VectorType::get(Type::getFloatTy(Ty->getContext()),
3546 cast<VectorType>(Ty));
3547
3548 TypeSize Size = DL.getTypeSizeInBits(Ty);
3549 if (Size.isScalable() && ST->hasVInstructions())
3550 return divideCeil(Size.getKnownMinValue(), RISCV::RVVBitsPerBlock);
3551
3552 if (ST->useRVVForFixedLengthVectors())
3553 return divideCeil(Size, ST->getRealMinVLen());
3554 }
3555
3556 return BaseT::getRegUsageForType(Ty);
3557}
3558
3559unsigned RISCVTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
3560 if (SLPMaxVF.getNumOccurrences())
3561 return SLPMaxVF;
3562
3563 // Return how many elements can fit in getRegisterBitwidth. This is the
3564 // same routine as used in LoopVectorizer. We should probably be
3565 // accounting for whether we actually have instructions with the right
3566 // lane type, but we don't have enough information to do that without
3567 // some additional plumbing which hasn't been justified yet.
3568 TypeSize RegWidth =
3570 // If no vector registers, or absurd element widths, disable
3571 // vectorization by returning 1.
3572 return std::max<unsigned>(1U, RegWidth.getFixedValue() / ElemWidth);
3573}
3574
3578
3580 return ST->enableUnalignedVectorMem();
3581}
3582
3585 ScalarEvolution *SE) const {
3586 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3587 return TTI::AMK_PostIndexed;
3588
3590}
3591
3593 const TargetTransformInfo::LSRCost &C2) const {
3594 // RISC-V specific here are "instruction number 1st priority".
3595 // If we need to emit adds inside the loop to add up base registers, then
3596 // we need at least one extra temporary register.
3597 unsigned C1NumRegs = C1.NumRegs + (C1.NumBaseAdds != 0);
3598 unsigned C2NumRegs = C2.NumRegs + (C2.NumBaseAdds != 0);
3599 return std::tie(C1.Insns, C1NumRegs, C1.AddRecCost,
3600 C1.NumIVMuls, C1.NumBaseAdds,
3601 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
3602 std::tie(C2.Insns, C2NumRegs, C2.AddRecCost,
3603 C2.NumIVMuls, C2.NumBaseAdds,
3604 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
3605}
3606
3608 Align Alignment) const {
3609 auto *VTy = dyn_cast<VectorType>(DataTy);
3610 if (!VTy)
3611 return false;
3612
3613 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3614 return false;
3615
3616 // FIXME: If it is an i8 vector and the element count exceeds 256, we should
3617 // scalarize these types with LMUL >= maximum fixed-length LMUL.
3618 if (VTy->getElementType()->isIntegerTy(8)) {
3619 uint64_t MaxEltCount = VTy->getElementCount().getKnownMinValue();
3620 if (VTy->isScalableTy())
3621 MaxEltCount *= ST->getRealMaxVLen() / RISCV::RVVBitsPerBlock;
3622 // We can't yet split any widened indices type.
3623 if (MaxEltCount > 256)
3625 VTy->getWithNewType(Type::getInt16Ty(VTy->getContext())))
3626 .first == 1;
3627 }
3628 return true;
3629}
3630
3632 Align Alignment) const {
3633 return isLegalMaskedLoadStore(DataTy, Alignment);
3634}
3635
3637 ElementCount NumElements) const {
3638 // Optimized zero-stride loads can be treated as broadcasts.
3639 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3640 return false;
3641
3642 return TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, ElementTy));
3643}
3644
3645/// See if \p I should be considered for address type promotion. We check if \p
3646/// I is a sext with right type and used in memory accesses. If it used in a
3647/// "complex" getelementptr, we allow it to be promoted without finding other
3648/// sext instructions that sign extended the same initial value. A getelementptr
3649/// is considered as "complex" if it has more than 2 operands.
3651 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
3652 bool Considerable = false;
3653 AllowPromotionWithoutCommonHeader = false;
3654 if (!isa<SExtInst>(&I))
3655 return false;
3656 Type *ConsideredSExtType =
3657 Type::getInt64Ty(I.getParent()->getParent()->getContext());
3658 if (I.getType() != ConsideredSExtType)
3659 return false;
3660 // See if the sext is the one with the right type and used in at least one
3661 // GetElementPtrInst.
3662 for (const User *U : I.users()) {
3663 if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
3664 Considerable = true;
3665 // A getelementptr is considered as "complex" if it has more than 2
3666 // operands. We will promote a SExt used in such complex GEP as we
3667 // expect some computation to be merged if they are done on 64 bits.
3668 if (GEPInst->getNumOperands() > 2) {
3669 AllowPromotionWithoutCommonHeader = true;
3670 break;
3671 }
3672 }
3673 }
3674 return Considerable;
3675}
3676
3677bool RISCVTTIImpl::canSplatOperand(unsigned Opcode, int Operand) const {
3678 switch (Opcode) {
3679 case Instruction::Add:
3680 case Instruction::Sub:
3681 case Instruction::Mul:
3682 case Instruction::And:
3683 case Instruction::Or:
3684 case Instruction::Xor:
3685 case Instruction::FAdd:
3686 case Instruction::FSub:
3687 case Instruction::FMul:
3688 case Instruction::FDiv:
3689 case Instruction::ICmp:
3690 case Instruction::FCmp:
3691 return true;
3692 case Instruction::Shl:
3693 case Instruction::LShr:
3694 case Instruction::AShr:
3695 case Instruction::UDiv:
3696 case Instruction::SDiv:
3697 case Instruction::URem:
3698 case Instruction::SRem:
3699 case Instruction::Select:
3700 return Operand == 1;
3701 default:
3702 return false;
3703 }
3704}
3705
3707 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3708 return false;
3709
3710 if (canSplatOperand(I->getOpcode(), Operand))
3711 return true;
3712
3713 auto *II = dyn_cast<IntrinsicInst>(I);
3714 if (!II)
3715 return false;
3716
3717 switch (II->getIntrinsicID()) {
3718 case Intrinsic::fma:
3719 case Intrinsic::fmuladd:
3720 return Operand == 0 || Operand == 1;
3721 case Intrinsic::vp_udiv:
3722 case Intrinsic::vp_sdiv:
3723 case Intrinsic::vp_urem:
3724 case Intrinsic::vp_srem:
3725 case Intrinsic::ssub_sat:
3726 case Intrinsic::usub_sat:
3727 return Operand == 1;
3728 // These intrinsics are commutative.
3729 case Intrinsic::smin:
3730 case Intrinsic::umin:
3731 case Intrinsic::smax:
3732 case Intrinsic::umax:
3733 case Intrinsic::sadd_sat:
3734 case Intrinsic::uadd_sat:
3735 return Operand == 0 || Operand == 1;
3736 default:
3737 return false;
3738 }
3739}
3740
3742 ArrayRef<int> Mask, ArrayRef<Value *> Scalars,
3744 GatherUseOps) const {
3745 if (Scalars.empty() || !ST->hasVInstructions() || !ST->sinkSplatOperands() ||
3746 !ShuffleVectorInst::isZeroEltSplatMask(Mask, Mask.size()))
3748
3749 const auto *SplatIt = find_if_not(Scalars, IsaPred<UndefValue>);
3750 if (SplatIt == Scalars.end() || (*SplatIt)->getType()->isIntegerTy(1) ||
3751 isa<VectorType>((*SplatIt)->getType()) ||
3752 isa<ExtractElementInst>(*SplatIt))
3754
3756 if (!GatherUseOps(UserOps) || UserOps.empty())
3758
3759 if (all_of(UserOps,
3760 [this](const TargetTransformInfo::BuildVectorUseOp &UserOp) {
3761 return canSplatOperand(UserOp.Opcode, UserOp.OperandIndex);
3762 }))
3764
3766}
3767
3768/// Check if sinking \p I's operands to I's basic block is profitable, because
3769/// the operands can be folded into a target instruction, e.g.
3770/// splats of scalars can fold into vector instructions.
3773 using namespace llvm::PatternMatch;
3774
3775 if (I->isBitwiseLogicOp()) {
3776 if (!I->getType()->isVectorTy()) {
3777 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3778 for (auto &Op : I->operands()) {
3779 // (and/or/xor X, (not Y)) -> (andn/orn/xnor X, Y)
3780 if (match(Op.get(), m_Not(m_Value()))) {
3781 Ops.push_back(&Op);
3782 return true;
3783 }
3784 }
3785 }
3786 } else if (I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3787 for (auto &Op : I->operands()) {
3788 // (and X, (not Y)) -> (vandn.vv X, Y)
3789 if (match(Op.get(), m_Not(m_Value()))) {
3790 Ops.push_back(&Op);
3791 return true;
3792 }
3793 // (and X, (splat (not Y))) -> (vandn.vx X, Y)
3795 m_ZeroInt()),
3796 m_Value(), m_ZeroMask()))) {
3797 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
3798 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
3799 Ops.push_back(&Not);
3800 Ops.push_back(&InsertElt);
3801 Ops.push_back(&Op);
3802 return true;
3803 }
3804 }
3805 }
3806 }
3807
3808 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3809 return false;
3810
3811 // Don't sink splat operands if the target prefers it. Some targets requires
3812 // S2V transfer buffers and we can run out of them copying the same value
3813 // repeatedly.
3814 // FIXME: It could still be worth doing if it would improve vector register
3815 // pressure and prevent a vector spill.
3816 if (!ST->sinkSplatOperands())
3817 return false;
3818
3819 for (auto OpIdx : enumerate(I->operands())) {
3820 if (!canSplatOperand(I, OpIdx.index()))
3821 continue;
3822
3823 Instruction *Op = dyn_cast<Instruction>(OpIdx.value().get());
3824 // Make sure we are not already sinking this operand
3825 if (!Op || any_of(Ops, [&](Use *U) { return U->get() == Op; }))
3826 continue;
3827
3828 // We are looking for a splat that can be sunk.
3830 m_Value(), m_ZeroMask())))
3831 continue;
3832
3833 // Don't sink i1 splats.
3834 if (cast<VectorType>(Op->getType())->getElementType()->isIntegerTy(1))
3835 continue;
3836
3837 // All uses of the shuffle should be sunk to avoid duplicating it across gpr
3838 // and vector registers
3839 for (Use &U : Op->uses()) {
3840 Instruction *Insn = cast<Instruction>(U.getUser());
3841 if (!canSplatOperand(Insn, U.getOperandNo()))
3842 return false;
3843 }
3844
3845 // Sink any fpexts since they might be used in a widening fp pattern.
3846 Use *InsertEltUse = &Op->getOperandUse(0);
3847 auto *InsertElt = cast<InsertElementInst>(InsertEltUse);
3848 if (isa<FPExtInst>(InsertElt->getOperand(1)))
3849 Ops.push_back(&InsertElt->getOperandUse(1));
3850 Ops.push_back(InsertEltUse);
3851 Ops.push_back(&OpIdx.value());
3852 }
3853 return true;
3854}
3855
3857RISCVTTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
3859
3860 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3861 return Options;
3862
3863 Options.AllowOverlappingLoads = true;
3864 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3865 Options.NumLoadsPerBlock = IsZeroCmp ? Options.MaxNumLoads : 1;
3866 if (ST->is64Bit()) {
3867 Options.LoadSizes = {8, 4, 2, 1};
3868 Options.AllowedTailExpansions = {3, 5, 6};
3869 } else {
3870 Options.LoadSizes = {4, 2, 1};
3871 Options.AllowedTailExpansions = {3};
3872 }
3873
3874 if (IsZeroCmp && ST->hasVInstructions()) {
3875 unsigned VLenB = ST->getRealMinVLen() / 8;
3876 // The minimum size should be `XLen / 8 + 1`, and the maxinum size should be
3877 // `VLenB * MaxLMUL` so that it fits in a single register group.
3878 unsigned MinSize = ST->getXLen() / 8 + 1;
3879 unsigned MaxSize = VLenB * 8;
3880 for (unsigned Size = MinSize; Size <= MaxSize; Size++)
3881 Options.LoadSizes.insert(Options.LoadSizes.begin(), Size);
3882 }
3883 return Options;
3884}
3885
3887 const Instruction *I) const {
3889 // For the binary operators (e.g. or) we need to be more careful than
3890 // selects, here we only transform them if they are already at a natural
3891 // break point in the code - the end of a block with an unconditional
3892 // terminator.
3893 if (I->getOpcode() == Instruction::Or &&
3894 isa<UncondBrInst>(I->getNextNode()))
3895 return true;
3896
3897 if (I->getOpcode() == Instruction::Add ||
3898 I->getOpcode() == Instruction::Sub)
3899 return true;
3900 }
3902}
3903
3905 const Function *Caller, const Attribute &Attr) const {
3906 // "interrupt" controls the prolog/epilog of interrupt handlers (and includes
3907 // restrictions on their signatures). We can outline from the bodies of these
3908 // handlers, but when we do we need to make sure we don't mark the outlined
3909 // function as an interrupt handler too.
3910 if (Attr.isStringAttribute() && Attr.getKindAsString() == "interrupt")
3911 return false;
3912
3914}
3915
3916std::optional<Instruction *>
3918 // Attach a range return attribute describing the result of vsetvli/vsetvlimax
3919 // so generic value analyses can reason about it. The verifier guarantees an
3920 // XLen result and constant VSEW/VLMUL encoding a valid vtype, so no defensive
3921 // validation is needed here.
3922 if (is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3923 II.getIntrinsicID())) {
3924 // These intrinsics require the V extension; without it the VLEN queries
3925 // below would assert. Such IR would fail isel anyway, so just bail out.
3926 if (!ST->hasVInstructions())
3927 return {};
3928
3929 bool HasAVL = II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3930 unsigned Offset = HasAVL ? 1 : 0;
3931 unsigned BitWidth = II.getType()->getIntegerBitWidth();
3932 ConstantRange VLenRange(APInt(BitWidth, ST->getRealMinVLen()),
3933 APInt(BitWidth, ST->getRealMaxVLen()) + 1);
3934
3935 uint64_t VSEW = cast<ConstantInt>(II.getArgOperand(Offset))->getZExtValue();
3936 auto VLMUL = static_cast<RISCVVType::VLMUL>(
3937 cast<ConstantInt>(II.getArgOperand(Offset + 1))->getZExtValue());
3938 unsigned SEW = RISCVVType::decodeVSEW(VSEW);
3939 unsigned Ratio = RISCVVType::getSEWLMULRatio(SEW, VLMUL);
3940
3941 // VLMAX = VLEN / (SEW / LMUL), clamped to >= 1 for any usable vtype.
3942 ConstantRange VLMAXRange =
3943 VLenRange.udiv(ConstantRange(APInt(BitWidth, Ratio)))
3945
3946 // vsetvlimax returns exactly VLMAX; vsetvli returns vl with
3947 // 0 <= vl <= min(AVL, VLMAX). vl == AVL only when AVL <= the smallest
3948 // possible VLMAX; otherwise vl can shrink below VLMAX (to 0 at runtime), so
3949 // only the VLMAX upper bound is sound.
3950 ConstantRange VLRange = VLMAXRange;
3951 if (HasAVL) {
3952 // vl ≤ VLMAX
3953 VLRange =
3955
3956 Value *AVL = II.getArgOperand(0);
3958 AVL, /*ForSigned=*/false,
3960
3961 // vl = AVL if AVL ≤ VLMAX
3962 if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
3963 return IC.replaceInstUsesWith(II, AVL);
3964
3965 // vl ≤ AVL
3966 VLRange = VLRange.umin(AVLRange.getUnsignedMax());
3967
3968 // vl > 0 if AVL > 0
3970 VLRange = VLRange.umax(APInt(BitWidth, 1));
3971
3972 // vl = VLMAX if AVL ≥ (2 * VLMAX)
3973 ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
3974 if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
3975 VLRange = VLRange.intersectWith(VLMAXRange);
3976
3977 // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
3978 if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
3979 VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
3980 APInt(BitWidth, 2),
3982 }
3983
3984 ConstantRange OldRange =
3985 II.getRange().value_or(ConstantRange::getFull(BitWidth));
3986 ConstantRange NewRange = VLRange.intersectWith(OldRange);
3987 if (NewRange != OldRange) {
3988 II.addRangeRetAttr(NewRange);
3989 return &II;
3990 }
3991 return {};
3992 }
3993
3994 // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
3995 // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
3996 // creating redundant masks.
3997 const DataLayout &DL = IC.getDataLayout();
3998 if (II.user_empty())
3999 return {};
4000 auto *TargetVecTy = dyn_cast<ScalableVectorType>(II.user_back()->getType());
4001 if (!TargetVecTy)
4002 return {};
4003 const APInt *Scalar;
4004 uint64_t VL;
4006 m_Poison(), m_APInt(Scalar), m_ConstantInt(VL))) ||
4007 !all_of(II.users(), [TargetVecTy](User *U) {
4008 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
4009 }))
4010 return {};
4011 auto *SourceVecTy = cast<ScalableVectorType>(II.getType());
4012 unsigned TargetEltBW = DL.getTypeSizeInBits(TargetVecTy->getElementType());
4013 unsigned SourceEltBW = DL.getTypeSizeInBits(SourceVecTy->getElementType());
4014 if (TargetEltBW % SourceEltBW)
4015 return {};
4016 unsigned TargetScale = TargetEltBW / SourceEltBW;
4017 if (VL % TargetScale || TargetScale == 1)
4018 return {};
4019 Type *VLTy = II.getOperand(2)->getType();
4020 ElementCount SourceEC = SourceVecTy->getElementCount();
4021 unsigned NewEltBW = SourceEltBW * TargetScale;
4022 if (!SourceEC.isKnownMultipleOf(TargetScale) ||
4023 !DL.fitsInLegalInteger(NewEltBW))
4024 return {};
4025 auto *NewEltTy = IntegerType::get(II.getContext(), NewEltBW);
4026 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, NewEltTy)))
4027 return {};
4028 ElementCount NewEC = SourceEC.divideCoefficientBy(TargetScale);
4029 Type *RetTy = VectorType::get(NewEltTy, NewEC);
4030 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
4031 "Lossless bitcast between types expected");
4032 APInt NewScalar = APInt::getSplat(NewEltBW, *Scalar);
4033 return IC.replaceInstUsesWith(
4034 II,
4037 RetTy, Intrinsic::riscv_vmv_v_x,
4038 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
4039 ConstantInt::get(VLTy, VL / TargetScale)}),
4040 SourceVecTy));
4041}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
static InstructionCost costShuffleViaVRegSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, std::optional< unsigned > VLen, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Try to perform better estimation of the permutation.
static InstructionCost costShuffleViaSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Attempt to approximate the cost of a shuffle which will require splitting during legalization.
static bool isRepeatedConcatMask(ArrayRef< int > Mask, int &SubVectorSize)
static cl::opt< bool > EnableOrLikeSelectOpt("riscv-or-like-select", cl::init(true), cl::Hidden)
static unsigned isM1OrSmaller(MVT VT)
static cl::opt< unsigned > SLPMaxVF("riscv-v-slp-max-vf", cl::desc("Overrides result used for getMaximumVF query which is used " "exclusively by SLP vectorizer."), cl::Hidden)
static cl::opt< unsigned > RVVRegisterWidthLMUL("riscv-v-register-bit-width-lmul", cl::desc("The LMUL to use for getRegisterBitWidth queries. Affects LMUL used " "by autovectorized code. Fractional LMULs are not supported."), cl::init(2), cl::Hidden)
static cl::opt< unsigned > RVVMinTripCount("riscv-v-min-trip-count", cl::desc("Set the lower bound of a trip count to decide on " "vectorization while tail-folding."), cl::init(5), cl::Hidden)
static InstructionCost getIntImmCostImpl(const DataLayout &DL, const RISCVSubtarget *ST, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, bool FreeZeroes)
static VectorType * getVRGatherIndexType(MVT DataVT, const RISCVSubtarget &ST, LLVMContext &C)
static const CostTblEntry VectorIntrinsicCostTable[]
static bool canUseShiftPair(Instruction *Inst, const APInt &Imm)
static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm)
This file defines a TargetTransformInfoImplBase conforming object specific to the RISC-V target machi...
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static bool isIntPredicate(Predicate P)
Definition InstrTypes.h:839
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noNaNs() const
Definition FMF.h:65
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
static FixedVectorType * getHalfElementsVectorType(FixedVectorType *VTy)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2252
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
ElementCount getVectorElementCount() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
TargetTransformInfo::VectorInstrContext getBuildVectorContextHint(ArrayRef< int > Mask, ArrayRef< Value * > Scalars, function_ref< bool(SmallVectorImpl< TargetTransformInfo::BuildVectorUseOp > &)> GatherUseOps) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CtxI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:865
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual const DataLayout & getDataLayout() const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual bool isLoweredToCall(const Function *F) const
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
Definition TypeSize.h:176
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
An efficient, type-erasing, non-owning reference to a callable.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2801
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:26
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
@ SplatOpFolded
All of the value's users support splatting the value.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
Definition STLExtras.h:1834
bool isPairEven(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-even idiom.
bool isPairOdd(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-odd idiom.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1793
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1986
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1901
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2162
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const
Stores information about the uses of a build vector.
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).