LLVM 24.0.0git
RISCVTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- RISCVTargetTransformInfo.cpp - RISC-V specific TTI ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "llvm/ADT/STLExtras.h"
18#include "llvm/IR/IntrinsicsRISCV.h"
21#include <cmath>
22#include <optional>
23using namespace llvm;
24using namespace llvm::PatternMatch;
25
26#define DEBUG_TYPE "riscvtti"
27
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
34
36 "riscv-v-slp-max-vf",
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
41
43 RVVMinTripCount("riscv-v-min-trip-count",
44 cl::desc("Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
47
48static cl::opt<bool> EnableOrLikeSelectOpt("enable-riscv-or-like-select",
49 cl::init(true), cl::Hidden);
50
52RISCVTTIImpl::getRISCVInstructionCost(ArrayRef<unsigned> OpCodes, MVT VT,
54 // Check if the type is valid for all CostKind
55 if (!VT.isVector())
57 size_t NumInstr = OpCodes.size();
59 return NumInstr;
60 InstructionCost LMULCost = TLI->getLMULCost(VT);
62 return LMULCost * NumInstr;
63 InstructionCost Cost = 0;
64 for (auto Op : OpCodes) {
65 switch (Op) {
66 case RISCV::VRGATHER_VI:
67 Cost += TLI->getVRGatherVICost(VT);
68 break;
69 case RISCV::VRGATHER_VV:
70 Cost += TLI->getVRGatherVVCost(VT);
71 break;
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
74 Cost += TLI->getVSlideVICost(VT);
75 break;
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
78 Cost += TLI->getVSlideVXCost(VT);
79 break;
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
91 unsigned VL = VT.getVectorMinNumElements();
92 if (!VT.isFixedLengthVector())
93 VL *= *getVScaleForTuning();
94 Cost += Log2_32_Ceil(VL);
95 break;
96 }
97 case RISCV::VFREDOSUM_VS: {
98 unsigned VL = VT.getVectorMinNumElements();
99 if (!VT.isFixedLengthVector())
100 VL *= *getVScaleForTuning();
101 Cost += VL;
102 break;
103 }
104 case RISCV::VMV_X_S:
105 case RISCV::VFMV_F_S:
106 // Domain crossings from vector -> scalar are usually more expensive.
107 Cost += 2;
108 break;
109 case RISCV::VMV_S_X:
110 case RISCV::VFMV_S_F:
111 case RISCV::VMOR_MM:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
116 case RISCV::VCPOP_M:
117 case RISCV::VFIRST_M:
118 Cost += 1;
119 break;
120 case RISCV::VDIV_VV:
121 case RISCV::VREM_VV:
122 Cost += LMULCost * TTI::TCC_Expensive;
123 break;
124 default:
125 Cost += LMULCost;
126 }
127 }
128 return Cost;
129}
130
132 const RISCVSubtarget *ST,
133 const APInt &Imm, Type *Ty,
135 bool FreeZeroes) {
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
138
139 // We have a Zero register, so 0 is always free.
140 if (Imm == 0)
141 return TTI::TCC_Free;
142
143 // Otherwise, we check how many instructions it will take to materialise.
144 return RISCVMatInt::getIntMatCost(Imm, DL.getTypeSizeInBits(Ty), *ST,
145 /*CompressionCost=*/false, FreeZeroes);
146}
147
151 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind, false);
152}
153
154// Look for patterns of shift followed by AND that can be turned into a pair of
155// shifts. We won't need to materialize an immediate for the AND so these can
156// be considered free.
157static bool canUseShiftPair(Instruction *Inst, const APInt &Imm) {
158 uint64_t Mask = Imm.getZExtValue();
159 auto *BO = dyn_cast<BinaryOperator>(Inst->getOperand(0));
160 if (!BO || !BO->hasOneUse())
161 return false;
162
163 if (BO->getOpcode() != Instruction::Shl)
164 return false;
165
166 if (!isa<ConstantInt>(BO->getOperand(1)))
167 return false;
168
169 unsigned ShAmt = cast<ConstantInt>(BO->getOperand(1))->getZExtValue();
170 // (and (shl x, c2), c1) will be matched to (srli (slli x, c2+c3), c3) if c1
171 // is a mask shifted by c2 bits with c3 leading zeros.
172 if (isShiftedMask_64(Mask)) {
173 unsigned Trailing = llvm::countr_zero(Mask);
174 if (ShAmt == Trailing)
175 return true;
176 }
177
178 return false;
179}
180
181// If this is i64 AND is part of (X & -(1 << C1) & 0xffffffff) == C2 << C1),
182// DAGCombiner can convert this to (sraiw X, C1) == sext(C2) for RV64. On RV32,
183// the type will be split so only the lower 32 bits need to be compared using
184// (srai/srli X, C) == C2.
185static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm) {
186 if (!Inst->hasOneUse())
187 return false;
188
189 // Look for equality comparison.
190 auto *Cmp = dyn_cast<ICmpInst>(*Inst->user_begin());
191 if (!Cmp || !Cmp->isEquality())
192 return false;
193
194 // Right hand side of comparison should be a constant.
195 auto *C = dyn_cast<ConstantInt>(Cmp->getOperand(1));
196 if (!C)
197 return false;
198
199 uint64_t Mask = Imm.getZExtValue();
200
201 // Mask should be of the form -(1 << C) in the lower 32 bits.
202 if (!isUInt<32>(Mask) || !isPowerOf2_32(-uint32_t(Mask)))
203 return false;
204
205 // Comparison constant should be a subset of Mask.
206 uint64_t CmpC = C->getZExtValue();
207 if ((CmpC & Mask) != CmpC)
208 return false;
209
210 // We'll need to sign extend the comparison constant and shift it right. Make
211 // sure the new constant can use addi/xori+seqz/snez.
212 unsigned ShiftBits = llvm::countr_zero(Mask);
213 int64_t NewCmpC = SignExtend64<32>(CmpC) >> ShiftBits;
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
215}
216
218 const APInt &Imm, Type *Ty,
220 Instruction *Inst) const {
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
223
224 // We have a Zero register, so 0 is always free.
225 if (Imm == 0)
226 return TTI::TCC_Free;
227
228 // Some instructions in RISC-V can take a 12-bit immediate. Some of these are
229 // commutative, in others the immediate comes from a specific argument index.
230 bool Takes12BitImm = false;
231 unsigned ImmArgIdx = ~0U;
232
233 switch (Opcode) {
234 case Instruction::GetElementPtr:
235 // Never hoist any arguments to a GetElementPtr. CodeGenPrepare will
236 // split up large offsets in GEP into better parts than ConstantHoisting
237 // can.
238 return TTI::TCC_Free;
239 case Instruction::Store: {
240 // Use the materialization cost regardless of if it's the address or the
241 // value that is constant, except for if the store is misaligned and
242 // misaligned accesses are not legal (experience shows constant hoisting
243 // can sometimes be harmful in such cases).
244 if (Idx == 1 || !Inst)
245 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
246 /*FreeZeroes=*/true);
247
248 StoreInst *ST = cast<StoreInst>(Inst);
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(), DL, getTLI()->getValueType(DL, Ty),
251 ST->getPointerAddressSpace(), ST->getAlign()))
252 return TTI::TCC_Free;
253
254 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
255 /*FreeZeroes=*/true);
256 }
257 case Instruction::Load:
258 // If the address is a constant, use the materialization cost.
259 return getIntImmCost(Imm, Ty, CostKind);
260 case Instruction::And:
261 // zext.h
262 if (Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
263 return TTI::TCC_Free;
264 // zext.w
265 if (Imm == UINT64_C(0xffffffff) &&
266 ((ST->hasStdExtZba() && ST->isRV64()) || ST->isRV32()))
267 return TTI::TCC_Free;
268 // bclri
269 if (ST->hasStdExtZbs() && (~Imm).isPowerOf2())
270 return TTI::TCC_Free;
271 if (Inst && Idx == 1 && Imm.getBitWidth() <= ST->getXLen() &&
272 canUseShiftPair(Inst, Imm))
273 return TTI::TCC_Free;
274 if (Inst && Idx == 1 && Imm.getBitWidth() == 64 &&
275 canUseShiftCmp(Inst, Imm))
276 return TTI::TCC_Free;
277 Takes12BitImm = true;
278 break;
279 case Instruction::Add:
280 Takes12BitImm = true;
281 break;
282 case Instruction::Or:
283 case Instruction::Xor:
284 // bseti/binvi
285 if (ST->hasStdExtZbs() && Imm.isPowerOf2())
286 return TTI::TCC_Free;
287 Takes12BitImm = true;
288 break;
289 case Instruction::Mul:
290 // Power of 2 is a shift. Negated power of 2 is a shift and a negate.
291 if (Imm.isPowerOf2() || Imm.isNegatedPowerOf2())
292 return TTI::TCC_Free;
293 // One more or less than a power of 2 can use SLLI+ADD/SUB.
294 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2())
295 return TTI::TCC_Free;
296 // FIXME: There is no MULI instruction.
297 Takes12BitImm = true;
298 break;
299 case Instruction::Sub:
300 case Instruction::Shl:
301 case Instruction::LShr:
302 case Instruction::AShr:
303 Takes12BitImm = true;
304 ImmArgIdx = 1;
305 break;
306 default:
307 break;
308 }
309
310 if (Takes12BitImm) {
311 // Check immediate is the correct argument...
312 if (Instruction::isCommutative(Opcode) || Idx == ImmArgIdx) {
313 // ... and fits into the 12-bit immediate.
314 if (Imm.getSignificantBits() <= 64 &&
315 getTLI()->isLegalAddImmediate(Imm.getSExtValue())) {
316 return TTI::TCC_Free;
317 }
318 }
319
320 // Otherwise, use the full materialisation cost.
321 return getIntImmCost(Imm, Ty, CostKind);
322 }
323
324 // By default, prevent hoisting.
325 return TTI::TCC_Free;
326}
327
330 const APInt &Imm, Type *Ty,
332 // Prevent hoisting in unknown cases.
333 return TTI::TCC_Free;
334}
335
337 return ST->hasVInstructions();
338}
339
341RISCVTTIImpl::getPopcntSupport(unsigned TyWidth) const {
342 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
343 return ST->hasCPOPLike() ? TTI::PSK_FastHardware : TTI::PSK_Software;
344}
345
347 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
349 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
350 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
351 if (Opcode == Instruction::FAdd)
353
354 // zve32x is broken for partial_reduce_umla, but let's make sure we
355 // don't generate them.
356 // vdot4a* reduces four i8 products into an i32 result; an i64 accumulator is
357 // additionally supported by widening the i32 partial sums to i64 (see
358 // lowerPARTIAL_REDUCE_MLA). VF is the number of i8 input elements, so the
359 // reduction factor is AccumBits / 8 (4 for i32, 8 for i64).
360 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
361 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
362 InputTypeA != InputTypeB || !InputTypeA->isIntegerTy(8) ||
363 (!AccumType->isIntegerTy(32) && !AccumType->isIntegerTy(64)))
365
366 unsigned Ratio = AccumType->getScalarSizeInBits() / 8;
367 if (!VF.isKnownMultipleOf(Ratio))
369
370 // Cost of the vdot4a* itself, which operates on the i32 intermediate type
371 // holding VF/4 elements.
372 Type *DotTp = VectorType::get(Type::getInt32Ty(AccumType->getContext()),
373 VF.divideCoefficientBy(4));
374 std::pair<InstructionCost, MVT> DotLT = getTypeLegalizationCost(DotTp);
375 // Note: Asuming all vdot4a* variants are equal cost
377 DotLT.first *
378 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second, CostKind);
379
380 // Account for reducing the i32 partial sums down to the i64 accumulator's
381 // element count and accumulating into it (see lowerPARTIAL_REDUCE_MLA), which
382 // has two shapes depending on the accumulator's LMUL.
383 if (AccumType->isIntegerTy(64)) {
384 LLVMContext &Ctx = AccumType->getContext();
385 Type *I32Ty = Type::getInt32Ty(Ctx);
386 ElementCount AccVF = VF.divideCoefficientBy(Ratio);
387 std::pair<InstructionCost, MVT> AccLT =
388 getTypeLegalizationCost(VectorType::get(AccumType, AccVF));
389
390 // When the i32 subvectors of a single-vector scalable accumulator are a
391 // fractional LMUL, extracting the high subvector would need a vslidedown,
392 // so instead the i32 dot result is widened to i64 first (vsext.vf2 /
393 // vzext.vf2) and then reduced and accumulated with register-aligned i64
394 // vadd.vv.
395 bool WidenFirst = false;
396 if (VF.isScalable() && AccLT.second.isScalableVector()) {
397 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
398 WidenFirst =
400 .second;
401 }
402
403 if (WidenFirst) {
404 // The widened i64 dot result has VF/4 elements, i.e. twice the
405 // accumulator's element count, so the reduction plus the accumulate are
406 // two i64 vadd.vv.
407 std::pair<InstructionCost, MVT> WideLT = getTypeLegalizationCost(
408 VectorType::get(AccumType, VF.divideCoefficientBy(4)));
409 Cost +=
410 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
411 WideLT.second, CostKind) +
412 2 * AccLT.first *
413 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second, CostKind);
414 } else {
415 // Otherwise the scale-4 i32 sums are halved with a single i32 vadd.vv,
416 // then widened and added into the i64 result with a vwadd.wv.
417 std::pair<InstructionCost, MVT> RedLT =
419 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
420 RedLT.second, CostKind) +
421 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
422 AccLT.second, CostKind);
423 // Fixed-length vectors extract the high i32 subvector with a vslidedown.
424 if (VF.isFixed())
425 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
426 DotLT.second, CostKind);
427 }
428 }
429
430 return Cost;
431}
432
434 // Currently, the ExpandReductions pass can't expand scalable-vector
435 // reductions, but we still request expansion as RVV doesn't support certain
436 // reductions and the SelectionDAG can't legalize them either.
437 switch (II->getIntrinsicID()) {
438 default:
439 return false;
440 // These reductions have no equivalent in RVV
441 case Intrinsic::vector_reduce_mul:
442 case Intrinsic::vector_reduce_fmul:
443 return true;
444 }
445}
446
447std::optional<unsigned> RISCVTTIImpl::getVScaleForTuning() const {
448 if (ST->hasVInstructions())
449 if (unsigned MinVLen = ST->getRealMinVLen();
450 MinVLen >= RISCV::RVVBitsPerBlock)
451 return MinVLen / RISCV::RVVBitsPerBlock;
453}
454
457 unsigned LMUL =
458 llvm::bit_floor(std::clamp<unsigned>(RVVRegisterWidthLMUL, 1, 8));
459 switch (K) {
461 return TypeSize::getFixed(ST->getXLen());
463 return TypeSize::getFixed(
464 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
467 (ST->hasVInstructions() &&
468 ST->getRealMinVLen() >= RISCV::RVVBitsPerBlock)
470 : 0);
471 }
472
473 llvm_unreachable("Unsupported register kind");
474}
475
476InstructionCost RISCVTTIImpl::getStaticDataAddrGenerationCost(
477 const TTI::TargetCostKind CostKind) const {
478 switch (CostKind) {
481 // Always 2 instructions
482 return 2;
483 case TTI::TCK_Latency:
485 // Depending on the memory model the address generation will
486 // require AUIPC + ADDI (medany) or LUI + ADDI (medlow). Don't
487 // have a way of getting this information here, so conservatively
488 // require both.
489 // In practice, these are generally implemented together.
490 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
491 }
492 llvm_unreachable("Unsupported cost kind");
493}
494
496RISCVTTIImpl::getConstantPoolLoadCost(Type *Ty,
498 // Add a cost of address generation + the cost of the load. The address
499 // is expected to be a PC relative offset to a constant pool entry
500 // using auipc/addi.
501 return getStaticDataAddrGenerationCost(CostKind) +
502 getMemoryOpCost(Instruction::Load, Ty, DL.getABITypeAlign(Ty),
503 /*AddressSpace=*/0, CostKind);
504}
505
506static bool isRepeatedConcatMask(ArrayRef<int> Mask, int &SubVectorSize) {
507 unsigned Size = Mask.size();
508 if (!isPowerOf2_32(Size))
509 return false;
510 for (unsigned I = 0; I != Size; ++I) {
511 if (static_cast<unsigned>(Mask[I]) == I)
512 continue;
513 if (Mask[I] != 0)
514 return false;
515 if (Size % I != 0)
516 return false;
517 for (unsigned J = I + 1; J != Size; ++J)
518 // Check the pattern is repeated.
519 if (static_cast<unsigned>(Mask[J]) != J % I)
520 return false;
521 SubVectorSize = I;
522 return true;
523 }
524 // That means Mask is <0, 1, 2, 3>. This is not a concatenation.
525 return false;
526}
527
529 LLVMContext &C) {
530 assert((DataVT.getScalarSizeInBits() != 8 ||
531 DataVT.getVectorNumElements() <= 256) && "unhandled case in lowering");
532 MVT IndexVT = DataVT.changeTypeToInteger();
533 if (IndexVT.getScalarType().bitsGT(ST.getXLenVT()))
534 IndexVT = IndexVT.changeVectorElementType(MVT::i16);
535 return cast<VectorType>(EVT(IndexVT).getTypeForEVT(C));
536}
537
538/// Attempt to approximate the cost of a shuffle which will require splitting
539/// during legalization. Note that processShuffleMasks is not an exact proxy
540/// for the algorithm used in LegalizeVectorTypes, but hopefully it's a
541/// reasonably close upperbound.
543 MVT LegalVT, VectorType *Tp,
544 ArrayRef<int> Mask,
546 assert(LegalVT.isFixedLengthVector() && !Mask.empty() &&
547 "Expected fixed vector type and non-empty mask");
548 unsigned LegalNumElts = LegalVT.getVectorNumElements();
549 // Number of destination vectors after legalization:
550 unsigned NumOfDests = divideCeil(Mask.size(), LegalNumElts);
551 // We are going to permute multiple sources and the result will be in
552 // multiple destinations. Providing an accurate cost only for splits where
553 // the element type remains the same.
554 if (NumOfDests <= 1 ||
556 Tp->getElementType()->getPrimitiveSizeInBits() ||
557 LegalNumElts >= Tp->getElementCount().getFixedValue())
559
560 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
561 unsigned LegalVTSize = LegalVT.getStoreSize();
562 // Number of source vectors after legalization:
563 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
564
565 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(), LegalNumElts);
566
567 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
568 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
569 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
570 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
571 assert(NormalizedVF >= Mask.size() &&
572 "Normalized mask expected to be not shorter than original mask.");
573 copy(Mask, NormalizedMask.begin());
574 InstructionCost Cost = 0;
575 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
577 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
578 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
579 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
580 return;
581 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
582 .second)
583 return;
584 Cost += TTI.getShuffleCost(
586 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
587 SingleOpTy, CostKind, RegMask, 0, nullptr);
588 },
589 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
590 Cost += TTI.getShuffleCost(
592 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
593 SingleOpTy, CostKind, RegMask, 0, nullptr);
594 });
595 return Cost;
596}
597
598/// Try to perform better estimation of the permutation.
599/// 1. Split the source/destination vectors into real registers.
600/// 2. Do the mask analysis to identify which real registers are
601/// permuted. If more than 1 source registers are used for the
602/// destination register building, the cost for this destination register
603/// is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
604/// source register is used, build mask and calculate the cost as a cost
605/// of PermuteSingleSrc.
606/// Also, for the single register permute we try to identify if the
607/// destination register is just a copy of the source register or the
608/// copy of the previous destination register (the cost is
609/// TTI::TCC_Basic). If the source register is just reused, the cost for
610/// this operation is 0.
611static InstructionCost
613 std::optional<unsigned> VLen, VectorType *Tp,
615 assert(LegalVT.isFixedLengthVector());
616 if (!VLen || Mask.empty())
618 MVT ElemVT = LegalVT.getVectorElementType();
619 unsigned ElemsPerVReg = *VLen / ElemVT.getFixedSizeInBits();
620 LegalVT = TTI.getTypeLegalizationCost(
621 FixedVectorType::get(Tp->getElementType(), ElemsPerVReg))
622 .second;
623 // Number of destination vectors after legalization:
624 InstructionCost NumOfDests =
625 divideCeil(Mask.size(), LegalVT.getVectorNumElements());
626 if (NumOfDests <= 1 ||
628 Tp->getElementType()->getPrimitiveSizeInBits() ||
629 LegalVT.getVectorNumElements() >= Tp->getElementCount().getFixedValue())
631
632 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
633 unsigned LegalVTSize = LegalVT.getStoreSize();
634 // Number of source vectors after legalization:
635 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
636
637 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(),
638 LegalVT.getVectorNumElements());
639
640 unsigned E = NumOfDests.getValue();
641 unsigned NormalizedVF =
642 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
643 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
644 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
645 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
646 assert(NormalizedVF >= Mask.size() &&
647 "Normalized mask expected to be not shorter than original mask.");
648 copy(Mask, NormalizedMask.begin());
649 InstructionCost Cost = 0;
650 int NumShuffles = 0;
651 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
653 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
654 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
655 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
656 return;
657 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
658 .second)
659 return;
660 ++NumShuffles;
661 Cost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, SingleOpTy,
662 SingleOpTy, CostKind, RegMask, 0, nullptr);
663 },
664 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
665 Cost += TTI.getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
666 SingleOpTy, CostKind, RegMask, 0, nullptr);
667 NumShuffles += 2;
668 });
669 // Note: check that we do not emit too many shuffles here to prevent code
670 // size explosion.
671 // TODO: investigate, if it can be improved by extra analysis of the masks
672 // to check if the code is more profitable.
673 if ((NumOfDestRegs > 2 && NumShuffles <= static_cast<int>(NumOfDestRegs)) ||
674 (NumOfDestRegs <= 2 && NumShuffles < 4))
675 return Cost;
677}
678
679InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
680 ArrayRef<int> Mask,
682 // Avoid missing masks and length changing shuffles
683 if (Mask.size() <= 2 || Mask.size() != Tp->getNumElements())
685
686 int NumElts = Tp->getNumElements();
687 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
688 // Avoid scalarization cases
689 if (!LT.second.isFixedLengthVector())
691
692 // Requires moving elements between parts, which requires additional
693 // unmodeled instructions.
694 if (LT.first != 1)
696
697 auto GetSlideOpcode = [&](int SlideAmt) {
698 assert(SlideAmt != 0);
699 bool IsVI = isUInt<5>(std::abs(SlideAmt));
700 if (SlideAmt < 0)
701 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
702 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
703 };
704
705 std::array<std::pair<int, int>, 2> SrcInfo;
706 if (!isMaskedSlidePair(Mask, NumElts, SrcInfo))
708
709 if (SrcInfo[1].second == 0)
710 std::swap(SrcInfo[0], SrcInfo[1]);
711
712 InstructionCost FirstSlideCost = 0;
713 if (SrcInfo[0].second != 0) {
714 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
715 FirstSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
716 }
717
718 if (SrcInfo[1].first == -1)
719 return FirstSlideCost;
720
721 InstructionCost SecondSlideCost = 0;
722 if (SrcInfo[1].second != 0) {
723 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
724 SecondSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
725 } else {
726 SecondSlideCost =
727 getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second, CostKind);
728 }
729
730 auto EC = Tp->getElementCount();
731 VectorType *MaskTy =
733 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
734 return FirstSlideCost + SecondSlideCost + MaskCost;
735}
736
740 ArrayRef<int> Mask, int Index, VectorType *SubTp,
742 const Instruction *CxtI) const {
743 assert((Mask.empty() || DstTy->isScalableTy() ||
744 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
745 "Expected the Mask to match the return size if given");
746 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
747 "Expected the same scalar types");
748
749 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
750
751 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
752 // For now, skip all fixed vector cost analysis when P extension is available
753 // to avoid crashes in getMinRVVVectorSizeInBits()
754 if (ST->hasStdExtP() && isa<FixedVectorType>(SrcTy))
755 return 1;
756
757 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
758
759 // First, handle cases where having a fixed length vector enables us to
760 // give a more accurate cost than falling back to generic scalable codegen.
761 // TODO: Each of these cases hints at a modeling gap around scalable vectors.
762 if (auto *FVTp = dyn_cast<FixedVectorType>(SrcTy);
763 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
765 *this, LT.second, ST->getRealVLen(),
766 Kind == TTI::SK_InsertSubvector ? DstTy : SrcTy, Mask, CostKind);
767 if (VRegSplittingCost.isValid())
768 return VRegSplittingCost;
769 switch (Kind) {
770 default:
771 break;
773 if (Mask.size() >= 2) {
774 MVT EltTp = LT.second.getVectorElementType();
775 // If the size of the element is < ELEN then shuffles of interleaves and
776 // deinterleaves of 2 vectors can be lowered into the following
777 // sequences
778 if (EltTp.getScalarSizeInBits() < ST->getELen()) {
779 // Example sequence:
780 // vsetivli zero, 4, e8, mf4, ta, ma (ignored)
781 // vwaddu.vv v10, v8, v9
782 // li a0, -1 (ignored)
783 // vwmaccu.vx v10, a0, v9
784 if (ShuffleVectorInst::isInterleaveMask(Mask, 2, Mask.size()))
785 return 2 * LT.first * TLI->getLMULCost(LT.second);
786
787 if (Mask[0] == 0 || Mask[0] == 1) {
788 auto DeinterleaveMask = createStrideMask(Mask[0], 2, Mask.size());
789 // Example sequence:
790 // vnsrl.wi v10, v8, 0
791 if (equal(DeinterleaveMask, Mask))
792 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
793 LT.second, CostKind);
794 }
795 }
796 int SubVectorSize;
797 if (LT.second.getScalarSizeInBits() != 1 &&
798 isRepeatedConcatMask(Mask, SubVectorSize)) {
800 unsigned NumSlides = Log2_32(Mask.size() / SubVectorSize);
801 // The cost of extraction from a subvector is 0 if the index is 0.
802 for (unsigned I = 0; I != NumSlides; ++I) {
803 unsigned InsertIndex = SubVectorSize * (1 << I);
804 FixedVectorType *SubTp =
805 FixedVectorType::get(SrcTy->getElementType(), InsertIndex);
806 FixedVectorType *DestTp =
808 std::pair<InstructionCost, MVT> DestLT =
810 // Add the cost of whole vector register move because the
811 // destination vector register group for vslideup cannot overlap the
812 // source.
813 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
815 CostKind, {}, InsertIndex, SubTp);
816 }
817 return Cost;
818 }
819 }
820
821 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
822 SlideCost.isValid())
823 return SlideCost;
824
825 // vrgather + cost of generating the mask constant.
826 // We model this for an unknown mask with a single vrgather.
827 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
828 LT.second.getVectorNumElements() <= 256)) {
829 VectorType *IdxTy =
830 getVRGatherIndexType(LT.second, *ST, SrcTy->getContext());
831 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
832 return IndexCost +
833 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second, CostKind);
834 }
835 break;
836 }
839
840 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
841 SlideCost.isValid())
842 return SlideCost;
843
844 // 2 x (vrgather + cost of generating the mask constant) + cost of mask
845 // register for the second vrgather. We model this for an unknown
846 // (shuffle) mask.
847 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
848 LT.second.getVectorNumElements() <= 256)) {
849 auto &C = SrcTy->getContext();
850 auto EC = SrcTy->getElementCount();
851 VectorType *IdxTy = getVRGatherIndexType(LT.second, *ST, C);
853 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
854 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
855 return 2 * IndexCost +
856 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
857 LT.second, CostKind) +
858 MaskCost;
859 }
860 break;
861 }
862 }
863
864 auto shouldSplit = [](TTI::ShuffleKind Kind) {
865 switch (Kind) {
866 default:
867 return false;
871 return true;
872 }
873 };
874
875 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
876 shouldSplit(Kind)) {
877 InstructionCost SplitCost =
878 costShuffleViaSplitting(*this, LT.second, FVTp, Mask, CostKind);
879 if (SplitCost.isValid())
880 return SplitCost;
881 }
882 }
883
884 // Handle scalable vectors (and fixed vectors legalized to scalable vectors).
885 switch (Kind) {
886 default:
887 // Fallthrough to generic handling.
888 // TODO: Most of these cases will return getInvalid in generic code, and
889 // must be implemented here.
890 break;
892 // Extract at zero is always a subregister extract
893 if (Index == 0)
894 return TTI::TCC_Free;
895
896 // If we're extracting a subvector of at most m1 size at a sub-register
897 // boundary - which unfortunately we need exact vlen to identify - this is
898 // a subregister extract at worst and thus won't require a vslidedown.
899 // TODO: Extend for aligned m2, m4 subvector extracts
900 // TODO: Extend for misalgined (but contained) extracts
901 // TODO: Extend for scalable subvector types
902 if (std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
903 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
904 if (std::optional<unsigned> VLen = ST->getRealVLen();
905 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
906 SubLT.second.getSizeInBits() <= *VLen)
907 return TTI::TCC_Free;
908 }
909
910 // Example sequence:
911 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
912 // vslidedown.vi v8, v9, 2
913 return LT.first *
914 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second, CostKind);
916 // Example sequence:
917 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
918 // vslideup.vi v8, v9, 2
919 LT = getTypeLegalizationCost(DstTy);
920 return LT.first *
921 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second, CostKind);
922 case TTI::SK_Select: {
923 // Example sequence:
924 // li a0, 90
925 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
926 // vmv.s.x v0, a0
927 // vmerge.vvm v8, v9, v8, v0
928 // We use 2 for the cost of the mask materialization as this is the true
929 // cost for small masks and most shuffles are small. At worst, this cost
930 // should be a very small constant for the constant pool load. As such,
931 // we may bias towards large selects slightly more than truly warranted.
932 return LT.first *
933 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
934 LT.second, CostKind));
935 }
936 case TTI::SK_Broadcast: {
937 // Check for broadcast loads, which are synthesized by optimized zero-stride
938 // loads (this is checked in RISCVTTIImpl::isLegalBroadcastLoad).
939 bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
940 if (IsLoad && LT.second.isVector() &&
941 isLegalBroadcastLoad(SrcTy->getElementType(),
942 LT.second.getVectorElementCount()))
943 return 0;
944
945 bool HasScalar = (Args.size() > 0) && (Operator::getOpcode(Args[0]) ==
946 Instruction::InsertElement);
947 if (LT.second.getScalarSizeInBits() == 1) {
948 if (HasScalar) {
949 // Example sequence:
950 // andi a0, a0, 1
951 // vsetivli zero, 2, e8, mf8, ta, ma (ignored)
952 // vmv.v.x v8, a0
953 // vmsne.vi v0, v8, 0
954 return LT.first *
955 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
956 LT.second, CostKind));
957 }
958 // Example sequence:
959 // vsetivli zero, 2, e8, mf8, ta, mu (ignored)
960 // vmv.v.i v8, 0
961 // vmerge.vim v8, v8, 1, v0
962 // vmv.x.s a0, v8
963 // andi a0, a0, 1
964 // vmv.v.x v8, a0
965 // vmsne.vi v0, v8, 0
966
967 return LT.first *
968 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
969 RISCV::VMV_X_S, RISCV::VMV_V_X,
970 RISCV::VMSNE_VI},
971 LT.second, CostKind));
972 }
973
974 if (HasScalar) {
975 // Example sequence:
976 // vmv.v.x v8, a0
977 return LT.first *
978 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second, CostKind);
979 }
980
981 // Example sequence:
982 // vrgather.vi v9, v8, 0
983 return LT.first *
984 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second, CostKind);
985 }
986 case TTI::SK_Splice: {
987 // vslidedown+vslideup.
988 // TODO: Multiplying by LT.first implies this legalizes into multiple copies
989 // of similar code, but I think we expand through memory.
990 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
991 if (Index >= 0 && Index < 32)
992 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
993 else if (Index < 0 && Index > -32)
994 Opcodes[1] = RISCV::VSLIDEUP_VI;
995 return LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
996 }
997 case TTI::SK_Reverse: {
998
999 if (!LT.second.isVector())
1001
1002 // TODO: Cases to improve here:
1003 // * Illegal vector types
1004 // * i64 on RV32
1005 if (SrcTy->getElementType()->isIntegerTy(1)) {
1006 VectorType *WideTy =
1007 VectorType::get(IntegerType::get(SrcTy->getContext(), 8),
1008 cast<VectorType>(SrcTy)->getElementCount());
1009 return getCastInstrCost(Instruction::ZExt, WideTy, SrcTy,
1011 getShuffleCost(TTI::SK_Reverse, WideTy, WideTy, CostKind, {}, 0,
1012 nullptr) +
1013 getCastInstrCost(Instruction::Trunc, SrcTy, WideTy,
1015 }
1016
1017 MVT ContainerVT = LT.second;
1018 if (LT.second.isFixedLengthVector())
1019 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1020 MVT M1VT = RISCVTargetLowering::getM1VT(ContainerVT);
1021 if (ContainerVT.bitsLE(M1VT)) {
1022 // Example sequence:
1023 // csrr a0, vlenb
1024 // srli a0, a0, 3
1025 // addi a0, a0, -1
1026 // vsetvli a1, zero, e8, mf8, ta, mu (ignored)
1027 // vid.v v9
1028 // vrsub.vx v10, v9, a0
1029 // vrgather.vv v9, v8, v10
1030 InstructionCost LenCost = 3;
1031 if (LT.second.isFixedLengthVector())
1032 // vrsub.vi has a 5 bit immediate field, otherwise an li suffices
1033 LenCost = isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1034 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1035 if (LT.second.isFixedLengthVector() &&
1036 isInt<5>(LT.second.getVectorNumElements() - 1))
1037 Opcodes[1] = RISCV::VRSUB_VI;
1038 InstructionCost GatherCost =
1039 getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1040 return LT.first * (LenCost + GatherCost);
1041 }
1042
1043 // At high LMUL, we split into a series of M1 reverses (see
1044 // lowerVECTOR_REVERSE) and then do a single slide at the end to eliminate
1045 // the resulting gap at the bottom (for fixed vectors only). The important
1046 // bit is that the cost scales linearly, not quadratically with LMUL.
1047 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1048 InstructionCost FixedCost =
1049 getRISCVInstructionCost(M1Opcodes, M1VT, CostKind) + 3;
1050 unsigned Ratio =
1051 ContainerVT.getVectorMinNumElements() / M1VT.getVectorMinNumElements();
1052 InstructionCost GatherCost =
1053 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT, CostKind) * Ratio;
1054 InstructionCost SlideCost = !LT.second.isFixedLengthVector() ? 0 :
1055 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second, CostKind);
1056 return FixedCost + LT.first * (GatherCost + SlideCost);
1057 }
1058 }
1059 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1060 SubTp);
1061}
1062
1063static unsigned isM1OrSmaller(MVT VT) {
1065 return (LMUL == RISCVVType::VLMUL::LMUL_F8 ||
1069}
1070
1072 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
1073 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
1074 TTI::VectorInstrContext VIC) const {
1077
1078 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1079 // For now, skip all fixed vector cost analysis when P extension is available
1080 // to avoid crashes in getMinRVVVectorSizeInBits()
1081 if (ST->hasStdExtP() && isa<FixedVectorType>(Ty)) {
1082 return 1; // Treat as single instruction cost for now
1083 }
1084
1085 // A build_vector (which is m1 sized or smaller) can be done in no
1086 // worse than one vslide1down.vx per element in the type. We could
1087 // in theory do an explode_vector in the inverse manner, but our
1088 // lowering today does not have a first class node for this pattern.
1090 Ty, DemandedElts, Insert, Extract, CostKind);
1091 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1092 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1093 if (Ty->getScalarSizeInBits() == 1) {
1094 auto *WideVecTy = cast<VectorType>(Ty->getWithNewBitWidth(8));
1095 // Note: Implicit scalar anyextend is assumed to be free since the i1
1096 // must be stored in a GPR.
1097 return getScalarizationOverhead(WideVecTy, DemandedElts, Insert, Extract,
1098 CostKind) +
1099 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
1101 }
1102
1103 assert(LT.second.isFixedLengthVector());
1104 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1105 if (isM1OrSmaller(ContainerVT)) {
1106 InstructionCost BV =
1107 cast<FixedVectorType>(Ty)->getNumElements() *
1108 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second, CostKind);
1109 if (BV < Cost)
1110 Cost = BV;
1111 }
1112 }
1113 return Cost;
1114}
1115
1119 Type *DataTy = MICA.getDataType();
1120 Align Alignment = MICA.getAlignment();
1121 switch (MICA.getID()) {
1122 case Intrinsic::vp_load_ff: {
1123 EVT DataTypeVT = TLI->getValueType(DL, DataTy);
1124 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1126
1127 unsigned AS = MICA.getAddressSpace();
1128 return getMemoryOpCost(Instruction::Load, DataTy, Alignment, AS, CostKind,
1129 {TTI::OK_AnyValue, TTI::OP_None}, nullptr);
1130 }
1131 case Intrinsic::experimental_vp_strided_load:
1132 case Intrinsic::experimental_vp_strided_store:
1133 return getStridedMemoryOpCost(MICA, CostKind);
1134 case Intrinsic::masked_compressstore:
1135 case Intrinsic::masked_expandload:
1137 case Intrinsic::vp_scatter:
1138 case Intrinsic::vp_gather:
1139 case Intrinsic::masked_scatter:
1140 case Intrinsic::masked_gather:
1141 return getGatherScatterOpCost(MICA, CostKind);
1142 case Intrinsic::vp_load:
1143 case Intrinsic::vp_store:
1144 case Intrinsic::masked_load:
1145 case Intrinsic::masked_store:
1146 return getMaskedMemoryOpCost(MICA, CostKind);
1147 }
1149}
1150
1154 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
1155 : Instruction::Store;
1156 Type *Src = MICA.getDataType();
1157 Align Alignment = MICA.getAlignment();
1158 unsigned AddressSpace = MICA.getAddressSpace();
1159
1160 if (!isLegalMaskedLoadStore(Src, Alignment) ||
1163
1164 return getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind);
1165}
1166
1168 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1169 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1170 bool UseMaskForCond, bool UseMaskForGaps) const {
1171
1172 // The interleaved memory access pass will lower (de)interleave ops combined
1173 // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
1174 // only support masking per-iteration (i.e. condition), not per-segment (i.e.
1175 // gap).
1176 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1177 auto *VTy = cast<VectorType>(VecTy);
1178 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
1179 // Need to make sure type has't been scalarized
1180 if (LT.second.isVector()) {
1182 return LT.first * TTI::TCC_Basic;
1183
1184 auto *SubVecTy =
1185 VectorType::get(VTy->getElementType(),
1186 VTy->getElementCount().divideCoefficientBy(Factor));
1187 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1188 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1189 AddressSpace, DL)) {
1190
1191 // Some processors optimize segment loads/stores as N * DLEN sized
1192 // load ops + Factor * LMUL shuffle ops.
1193 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1194 unsigned VecSizeInBits =
1195 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1196 unsigned VLENForTuning =
1198 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1199 InstructionCost Cost = divideCeil(VecSizeInBits, DLENForTuning);
1200 MVT SubVecVT = getTLI()->getValueType(DL, SubVecTy).getSimpleVT();
1201 Cost += Factor * TLI->getLMULCost(SubVecVT);
1202 return Cost;
1203 }
1204
1205 // Otherwise, the cost is proportional to the number of elements (VL *
1206 // Factor ops).
1207 unsigned NumLoads = getEstimatedVLFor(VTy);
1208 return NumLoads * TTI::TCC_Basic;
1209 }
1210 }
1211 }
1212
1213 // TODO: Return the cost of interleaved accesses for scalable vector when
1214 // unable to convert to segment accesses instructions.
1215 if (isa<ScalableVectorType>(VecTy))
1217
1218 auto *FVTy = cast<FixedVectorType>(VecTy);
1219 // When gaps are only at the tail, for interleaved load, we can emit a wide
1220 // masked load and shufflevectors. For interleaved store, we can emit
1221 // shufflevectors and a wide masked store. The interleaved memory access pass
1222 // will lower them into vlsseg/vssseg intrinsics.
1223 if (UseMaskForGaps) {
1224 assert(llvm::is_sorted(Indices) && "Indices must be sorted");
1225 assert(llvm::adjacent_find(Indices) == Indices.end() &&
1226 "Indices should not contain duplicate elements");
1227 unsigned NumOfFields = Indices.size();
1228 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
1229 if (IsTailGapOnly &&
1230 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1231 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
1232 if (LT.second.isVector() &&
1233 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1234 auto *SubVecTy = VectorType::get(
1235 FVTy->getElementType(),
1236 FVTy->getElementCount().divideCoefficientBy(Factor));
1237 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1238 AddressSpace, DL)) {
1239 // The cost is proportional to the total number of element accesses.
1240 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1241 return NumAccesses * TTI::TCC_Basic;
1242 }
1243 }
1244 }
1245 }
1246
1247 InstructionCost MemCost =
1248 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
1249 unsigned VF = FVTy->getNumElements() / Factor;
1250
1251 // An interleaved load will look like this for Factor=3:
1252 // %wide.vec = load <12 x i32>, ptr %3, align 4
1253 // %strided.vec = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1254 // %strided.vec1 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1255 // %strided.vec2 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1256 if (Opcode == Instruction::Load) {
1257 InstructionCost Cost = MemCost;
1258 for (unsigned Index : Indices) {
1259 FixedVectorType *VecTy =
1260 FixedVectorType::get(FVTy->getElementType(), VF * Factor);
1261 auto Mask = createStrideMask(Index, Factor, VF);
1262 Mask.resize(VF * Factor, -1);
1263 InstructionCost ShuffleCost =
1265 CostKind, Mask, 0, nullptr, {});
1266 Cost += ShuffleCost;
1267 }
1268 return Cost;
1269 }
1270
1271 // TODO: Model for NF > 2
1272 // We'll need to enhance getShuffleCost to model shuffles that are just
1273 // inserts and extracts into subvectors, since they won't have the full cost
1274 // of a vrgather.
1275 // An interleaved store for 3 vectors of 4 lanes will look like
1276 // %11 = shufflevector <4 x i32> %4, <4 x i32> %6, <8 x i32> <0...7>
1277 // %12 = shufflevector <4 x i32> %9, <4 x i32> poison, <8 x i32> <0...3>
1278 // %13 = shufflevector <8 x i32> %11, <8 x i32> %12, <12 x i32> <0...11>
1279 // %interleaved.vec = shufflevector %13, poison, <12 x i32> <interleave mask>
1280 // store <12 x i32> %interleaved.vec, ptr %10, align 4
1281 if (Factor != 2)
1282 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1283 Alignment, AddressSpace, CostKind,
1284 UseMaskForCond, UseMaskForGaps);
1285
1286 assert(Opcode == Instruction::Store && "Opcode must be a store");
1287 // For an interleaving store of 2 vectors, we perform one large interleaving
1288 // shuffle that goes into the wide store
1289 auto Mask = createInterleaveMask(VF, Factor);
1290 InstructionCost ShuffleCost =
1292 CostKind, Mask, 0, nullptr, {});
1293 return MemCost + ShuffleCost;
1294}
1295
1299
1300 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
1301 MICA.getID() == Intrinsic::vp_gather;
1302 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1303 Type *DataTy = MICA.getDataType();
1304 Align Alignment = MICA.getAlignment();
1307
1308 if ((Opcode == Instruction::Load &&
1309 !isLegalMaskedGather(DataTy, Align(Alignment))) ||
1310 (Opcode == Instruction::Store &&
1311 !isLegalMaskedScatter(DataTy, Align(Alignment))))
1313
1314 // Cost is proportional to the number of memory operations implied. For
1315 // scalable vectors, we use an estimate on that number since we don't
1316 // know exactly what VL will be.
1317 auto &VTy = *cast<VectorType>(DataTy);
1318 unsigned NumLoads = getEstimatedVLFor(&VTy);
1319 return NumLoads * TTI::TCC_Basic;
1320}
1321
1323 const MemIntrinsicCostAttributes &MICA,
1325 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
1326 ? Instruction::Load
1327 : Instruction::Store;
1328 Type *DataTy = MICA.getDataType();
1329 bool VariableMask = MICA.getVariableMask();
1330 Align Alignment = MICA.getAlignment();
1331 bool IsLegal = (Opcode == Instruction::Store &&
1332 isLegalMaskedCompressStore(DataTy, Alignment)) ||
1333 (Opcode == Instruction::Load &&
1334 isLegalMaskedExpandLoad(DataTy, Alignment));
1335 if (!IsLegal || CostKind != TTI::TCK_RecipThroughput)
1337 // Example compressstore sequence:
1338 // vsetivli zero, 8, e32, m2, ta, ma (ignored)
1339 // vcompress.vm v10, v8, v0
1340 // vcpop.m a1, v0
1341 // vsetvli zero, a1, e32, m2, ta, ma
1342 // vse32.v v10, (a0)
1343 // Example expandload sequence:
1344 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
1345 // vcpop.m a1, v0
1346 // vsetvli zero, a1, e32, m2, ta, ma
1347 // vle32.v v10, (a0)
1348 // vsetivli zero, 8, e32, m2, ta, ma
1349 // viota.m v12, v0
1350 // vrgather.vv v8, v10, v12, v0.t
1351 auto MemOpCost =
1352 getMemoryOpCost(Opcode, DataTy, Alignment, /*AddressSpace*/ 0, CostKind);
1353 auto LT = getTypeLegalizationCost(DataTy);
1354 SmallVector<unsigned, 4> Opcodes{RISCV::VSETVLI};
1355 if (VariableMask)
1356 Opcodes.push_back(RISCV::VCPOP_M);
1357 if (Opcode == Instruction::Store)
1358 Opcodes.append({RISCV::VCOMPRESS_VM});
1359 else
1360 Opcodes.append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1361 return MemOpCost +
1362 LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1363}
1364
1368 Type *DataTy = MICA.getDataType();
1369 Align Alignment = MICA.getAlignment();
1370
1371 if (!isLegalStridedLoadStore(DataTy, Alignment))
1373
1375 return TTI::TCC_Basic;
1376
1377 // Cost is proportional to the number of memory operations implied. For
1378 // scalable vectors, we use an estimate on that number since we don't
1379 // know exactly what VL will be.
1380 auto &VTy = *cast<VectorType>(DataTy);
1381 unsigned NumLoads = getEstimatedVLFor(&VTy);
1382 return NumLoads * TTI::TCC_Basic;
1383}
1384
1387 // FIXME: This is a property of the default vector convention, not
1388 // all possible calling conventions. Fixing that will require
1389 // some TTI API and SLP rework.
1392 for (auto *Ty : Tys) {
1393 if (!Ty->isVectorTy())
1394 continue;
1395 Align A = DL.getPrefTypeAlign(Ty);
1396 Cost += getMemoryOpCost(Instruction::Store, Ty, A, 0, CostKind) +
1397 getMemoryOpCost(Instruction::Load, Ty, A, 0, CostKind);
1398 }
1399 return Cost;
1400}
1401
1402// Currently, these represent both throughput and codesize costs
1403// for the respective intrinsics. The costs in this table are simply
1404// instruction counts with the following adjustments made:
1405// * One vsetvli is considered free.
1407 {Intrinsic::floor, MVT::f32, 9},
1408 {Intrinsic::floor, MVT::f64, 9},
1409 {Intrinsic::ceil, MVT::f32, 9},
1410 {Intrinsic::ceil, MVT::f64, 9},
1411 {Intrinsic::trunc, MVT::f32, 7},
1412 {Intrinsic::trunc, MVT::f64, 7},
1413 {Intrinsic::round, MVT::f32, 9},
1414 {Intrinsic::round, MVT::f64, 9},
1415 {Intrinsic::roundeven, MVT::f32, 9},
1416 {Intrinsic::roundeven, MVT::f64, 9},
1417 {Intrinsic::rint, MVT::f32, 7},
1418 {Intrinsic::rint, MVT::f64, 7},
1419 {Intrinsic::nearbyint, MVT::f32, 9},
1420 {Intrinsic::nearbyint, MVT::f64, 9},
1421 {Intrinsic::bswap, MVT::i16, 3},
1422 {Intrinsic::bswap, MVT::i32, 12},
1423 {Intrinsic::bswap, MVT::i64, 31},
1424 {Intrinsic::bitreverse, MVT::i8, 17},
1425 {Intrinsic::bitreverse, MVT::i16, 24},
1426 {Intrinsic::bitreverse, MVT::i32, 33},
1427 {Intrinsic::bitreverse, MVT::i64, 52},
1428 {Intrinsic::ctpop, MVT::i8, 12},
1429 {Intrinsic::ctpop, MVT::i16, 19},
1430 {Intrinsic::ctpop, MVT::i32, 20},
1431 {Intrinsic::ctpop, MVT::i64, 21},
1432 {Intrinsic::ctlz, MVT::i8, 19},
1433 {Intrinsic::ctlz, MVT::i16, 28},
1434 {Intrinsic::ctlz, MVT::i32, 31},
1435 {Intrinsic::ctlz, MVT::i64, 35},
1436 {Intrinsic::cttz, MVT::i8, 16},
1437 {Intrinsic::cttz, MVT::i16, 23},
1438 {Intrinsic::cttz, MVT::i32, 24},
1439 {Intrinsic::cttz, MVT::i64, 25},
1440};
1441
1445 auto *RetTy = ICA.getReturnType();
1446 switch (ICA.getID()) {
1447 case Intrinsic::lrint:
1448 case Intrinsic::llrint:
1449 case Intrinsic::lround:
1450 case Intrinsic::llround: {
1451 auto LT = getTypeLegalizationCost(RetTy);
1452 Type *SrcTy = ICA.getArgTypes().front();
1453 auto SrcLT = getTypeLegalizationCost(SrcTy);
1454 if (ST->hasVInstructions() && LT.second.isVector()) {
1456 unsigned SrcEltSz = DL.getTypeSizeInBits(SrcTy->getScalarType());
1457 unsigned DstEltSz = DL.getTypeSizeInBits(RetTy->getScalarType());
1458 if (LT.second.getVectorElementType() == MVT::bf16) {
1459 if (!ST->hasVInstructionsBF16Minimal())
1461 if (DstEltSz == 32)
1462 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1463 else
1464 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1465 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1466 !ST->hasVInstructionsF16()) {
1467 if (!ST->hasVInstructionsF16Minimal())
1469 if (DstEltSz == 32)
1470 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1471 else
1472 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1473
1474 } else if (SrcEltSz > DstEltSz) {
1475 Ops = {RISCV::VFNCVT_X_F_W};
1476 } else if (SrcEltSz < DstEltSz) {
1477 Ops = {RISCV::VFWCVT_X_F_V};
1478 } else {
1479 Ops = {RISCV::VFCVT_X_F_V};
1480 }
1481
1482 // We need to use the source LMUL in the case of a narrowing op, and the
1483 // destination LMUL otherwise.
1484 if (SrcEltSz > DstEltSz)
1485 return SrcLT.first *
1486 getRISCVInstructionCost(Ops, SrcLT.second, CostKind);
1487 return LT.first * getRISCVInstructionCost(Ops, LT.second, CostKind);
1488 }
1489 break;
1490 }
1491 case Intrinsic::ceil:
1492 case Intrinsic::floor:
1493 case Intrinsic::trunc:
1494 case Intrinsic::rint:
1495 case Intrinsic::round:
1496 case Intrinsic::roundeven: {
1497 // These all use the same code.
1498 auto LT = getTypeLegalizationCost(RetTy);
1499 if (!LT.second.isVector() && TLI->isOperationCustom(ISD::FCEIL, LT.second))
1500 return LT.first * 8;
1501 break;
1502 }
1503 case Intrinsic::umin:
1504 case Intrinsic::umax:
1505 case Intrinsic::smin:
1506 case Intrinsic::smax: {
1507 auto LT = getTypeLegalizationCost(RetTy);
1508 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1509 return LT.first;
1510
1511 if (ST->hasVInstructions() && LT.second.isVector()) {
1512 unsigned Op;
1513 switch (ICA.getID()) {
1514 case Intrinsic::umin:
1515 Op = RISCV::VMINU_VV;
1516 break;
1517 case Intrinsic::umax:
1518 Op = RISCV::VMAXU_VV;
1519 break;
1520 case Intrinsic::smin:
1521 Op = RISCV::VMIN_VV;
1522 break;
1523 case Intrinsic::smax:
1524 Op = RISCV::VMAX_VV;
1525 break;
1526 }
1527 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1528 }
1529 break;
1530 }
1531 case Intrinsic::sadd_sat:
1532 case Intrinsic::ssub_sat:
1533 case Intrinsic::uadd_sat:
1534 case Intrinsic::usub_sat: {
1535 auto LT = getTypeLegalizationCost(RetTy);
1536 if (ST->hasVInstructions() && LT.second.isVector()) {
1537 unsigned Op;
1538 switch (ICA.getID()) {
1539 case Intrinsic::sadd_sat:
1540 Op = RISCV::VSADD_VV;
1541 break;
1542 case Intrinsic::ssub_sat:
1543 Op = RISCV::VSSUB_VV;
1544 break;
1545 case Intrinsic::uadd_sat:
1546 Op = RISCV::VSADDU_VV;
1547 break;
1548 case Intrinsic::usub_sat:
1549 Op = RISCV::VSSUBU_VV;
1550 break;
1551 }
1552 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1553 }
1554 break;
1555 }
1556 case Intrinsic::fma:
1557 case Intrinsic::fmuladd: {
1558 // TODO: handle promotion with f16/bf16 with zvfhmin/zvfbfmin
1559 auto LT = getTypeLegalizationCost(RetTy);
1560 if (ST->hasVInstructions() && LT.second.isVector())
1561 return LT.first *
1562 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second, CostKind);
1563 break;
1564 }
1565 case Intrinsic::fabs: {
1566 auto LT = getTypeLegalizationCost(RetTy);
1567 if (ST->hasVInstructions() && LT.second.isVector()) {
1568 // lui a0, 8
1569 // addi a0, a0, -1
1570 // vsetvli a1, zero, e16, m1, ta, ma
1571 // vand.vx v8, v8, a0
1572 // f16 with zvfhmin and bf16 with zvfhbmin
1573 if (LT.second.getVectorElementType() == MVT::bf16 ||
1574 (LT.second.getVectorElementType() == MVT::f16 &&
1575 !ST->hasVInstructionsF16()))
1576 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1577 CostKind) +
1578 2;
1579 else
1580 return LT.first *
1581 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second, CostKind);
1582 }
1583 break;
1584 }
1585 case Intrinsic::sqrt: {
1586 auto LT = getTypeLegalizationCost(RetTy);
1587 if (ST->hasVInstructions() && LT.second.isVector()) {
1590 MVT ConvType = LT.second;
1591 MVT FsqrtType = LT.second;
1592 // f16 with zvfhmin and bf16 with zvfbfmin and the type of nxv32[b]f16
1593 // will be spilt.
1594 if (LT.second.getVectorElementType() == MVT::bf16) {
1595 if (LT.second == MVT::nxv32bf16) {
1596 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1597 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1598 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1599 ConvType = MVT::nxv16f16;
1600 FsqrtType = MVT::nxv16f32;
1601 } else {
1602 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1603 FsqrtOp = {RISCV::VFSQRT_V};
1604 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1605 }
1606 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1607 !ST->hasVInstructionsF16()) {
1608 if (LT.second == MVT::nxv32f16) {
1609 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1610 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1611 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1612 ConvType = MVT::nxv16f16;
1613 FsqrtType = MVT::nxv16f32;
1614 } else {
1615 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1616 FsqrtOp = {RISCV::VFSQRT_V};
1617 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1618 }
1619 } else {
1620 FsqrtOp = {RISCV::VFSQRT_V};
1621 }
1622
1623 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType, CostKind) +
1624 getRISCVInstructionCost(ConvOp, ConvType, CostKind));
1625 }
1626 break;
1627 }
1628 case Intrinsic::cttz:
1629 case Intrinsic::ctlz:
1630 case Intrinsic::ctpop: {
1631 auto LT = getTypeLegalizationCost(RetTy);
1632 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1633 unsigned Op;
1634 switch (ICA.getID()) {
1635 case Intrinsic::cttz:
1636 Op = RISCV::VCTZ_V;
1637 break;
1638 case Intrinsic::ctlz:
1639 Op = RISCV::VCLZ_V;
1640 break;
1641 case Intrinsic::ctpop:
1642 Op = RISCV::VCPOP_V;
1643 break;
1644 }
1645 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1646 }
1647 break;
1648 }
1649 case Intrinsic::abs: {
1650 auto LT = getTypeLegalizationCost(RetTy);
1651 if (ST->hasVInstructions() && LT.second.isVector()) {
1652 // vabs.v v10, v8 (alias for vabd.vx v10, v8, zero)
1653 if (ST->hasStdExtZvabd())
1654 return LT.first *
1655 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second, CostKind);
1656
1657 // vrsub.vi v10, v8, 0
1658 // vmax.vv v8, v8, v10
1659 return LT.first *
1660 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1661 LT.second, CostKind);
1662 }
1663 break;
1664 }
1665 case Intrinsic::fshl:
1666 case Intrinsic::fshr: {
1667 if (ICA.getArgs().empty())
1668 break;
1669
1670 // Funnel-shifts are ROTL/ROTR when the first and second operand are equal.
1671 // When Zbb/Zbkb is enabled we can use a single ROL(W)/ROR(I)(W)
1672 // instruction.
1673 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1674 ICA.getArgs()[0] == ICA.getArgs()[1] &&
1675 (RetTy->getIntegerBitWidth() == 32 ||
1676 RetTy->getIntegerBitWidth() == 64) &&
1677 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1678 return 1;
1679 }
1680 break;
1681 }
1682 case Intrinsic::clmul: {
1683 auto LT = getTypeLegalizationCost(RetTy);
1684 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbc() &&
1685 !ST->hasStdExtZbkc()) {
1686 // TODO: Once custom lowering in this case for RV32 is added, this guard
1687 // should be removed and the cost model should be updated.
1688 if (!ST->is64Bit() || LT.second != MVT::i64)
1689 break;
1690 // vmv.s.x v8, a0
1691 // vclmul.vx v8, v8, a1
1692 // vmv.x.s a0, v8
1693 MVT VecVT = MVT::getScalableVectorVT(LT.second, 1);
1694 return LT.first * getRISCVInstructionCost(
1695 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1696 VecVT, CostKind);
1697 }
1698 break;
1699 }
1700 case Intrinsic::masked_udiv:
1701 return getArithmeticInstrCost(Instruction::UDiv, ICA.getReturnType(),
1702 CostKind);
1703 case Intrinsic::masked_sdiv:
1704 return getArithmeticInstrCost(Instruction::SDiv, ICA.getReturnType(),
1705 CostKind);
1706 case Intrinsic::masked_urem:
1707 return getArithmeticInstrCost(Instruction::URem, ICA.getReturnType(),
1708 CostKind);
1709 case Intrinsic::masked_srem:
1710 return getArithmeticInstrCost(Instruction::SRem, ICA.getReturnType(),
1711 CostKind);
1712 case Intrinsic::get_active_lane_mask: {
1713 if (ST->hasVInstructions()) {
1714 Type *ExpRetTy = VectorType::get(
1715 ICA.getArgTypes()[0], cast<VectorType>(RetTy)->getElementCount());
1716 auto LT = getTypeLegalizationCost(ExpRetTy);
1717
1718 // vid.v v8 // considered hoisted
1719 // vsaddu.vx v8, v8, a0
1720 // vmsltu.vx v0, v8, a1
1721 return LT.first *
1722 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1723 LT.second, CostKind);
1724 }
1725 break;
1726 }
1727 // TODO: add more intrinsic
1728 case Intrinsic::stepvector: {
1729 auto LT = getTypeLegalizationCost(RetTy);
1730 // Legalisation of illegal types involves an `index' instruction plus
1731 // (LT.first - 1) vector adds.
1732 if (ST->hasVInstructions())
1733 return getRISCVInstructionCost(RISCV::VID_V, LT.second, CostKind) +
1734 (LT.first - 1) *
1735 getRISCVInstructionCost(RISCV::VADD_VX, LT.second, CostKind);
1736 return 1 + (LT.first - 1);
1737 }
1738 case Intrinsic::vector_splice_left:
1739 case Intrinsic::vector_splice_right: {
1740 auto LT = getTypeLegalizationCost(RetTy);
1741 // Constant offsets fall through to getShuffleCost.
1742 if (!ICA.isTypeBasedOnly() && isa<ConstantInt>(ICA.getArgs()[2]))
1743 break;
1744 if (ST->hasVInstructions() && LT.second.isVector()) {
1745 return LT.first *
1746 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1747 LT.second, CostKind);
1748 }
1749 break;
1750 }
1751 case Intrinsic::experimental_cttz_elts: {
1752 if (!ST->hasVInstructions())
1753 break;
1755 Type *ArgTy = ICA.getArgTypes()[0];
1756 auto LT = getTypeLegalizationCost(ArgTy);
1757
1758 // If the element type is not i1, do a comparison with all-zeros.
1759 if (LT.second.getVectorElementType() != MVT::i1)
1760 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second, CostKind);
1761
1762 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind);
1763
1764 // If zero_is_poison is false, then we will generate additional
1765 // cmp + select instructions to convert -1 to EVL.
1766 Type *BoolTy = Type::getInt1Ty(RetTy->getContext());
1767 if (ICA.getArgs().size() > 1 &&
1768 cast<ConstantInt>(ICA.getArgs()[1])->isZero())
1769 Cost += getCmpSelInstrCost(Instruction::ICmp, BoolTy, RetTy,
1771 getCmpSelInstrCost(Instruction::Select, RetTy, BoolTy,
1773
1774 return LT.first * Cost;
1775 }
1776 case Intrinsic::experimental_vp_splice: {
1777 // To support type-based query from vectorizer, set the index to 0.
1778 // Note that index only change the cost from vslide.vx to vslide.vi and in
1779 // current implementations they have same costs.
1781 cast<VectorType>(ICA.getArgTypes()[0]), CostKind, {},
1783 }
1784 case Intrinsic::vp_merge: {
1785 // If an operand is a binary op and the type is legal, RISCVVectorPeephole
1786 // will likely fold the resulting vmerge.vvm away.
1788 getTypeLegalizationCost(RetTy).first == 1)
1789 return TTI::TCC_Free;
1790 break;
1791 }
1792 case Intrinsic::fptoui_sat:
1793 case Intrinsic::fptosi_sat: {
1795 bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
1796 Type *SrcTy = ICA.getArgTypes()[0];
1797
1798 auto SrcLT = getTypeLegalizationCost(SrcTy);
1799 auto DstLT = getTypeLegalizationCost(RetTy);
1800 if (!SrcTy->isVectorTy())
1801 break;
1802
1803 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1805
1806 Cost +=
1807 getCastInstrCost(IsSigned ? Instruction::FPToSI : Instruction::FPToUI,
1808 RetTy, SrcTy, TTI::CastContextHint::None, CostKind);
1809
1810 // Handle NaN.
1811 // vmfne v0, v8, v8 # If v8[i] is NaN set v0[i] to 1.
1812 // vmerge.vim v8, v8, 0, v0 # Convert NaN to 0.
1813 Type *CondTy = RetTy->getWithNewBitWidth(1);
1814 Cost += getCmpSelInstrCost(BinaryOperator::FCmp, SrcTy, CondTy,
1816 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
1818 return Cost;
1819 }
1820 case Intrinsic::experimental_vector_extract_last_active: {
1821 auto *ValTy = cast<VectorType>(ICA.getArgTypes()[0]);
1822 auto *MaskTy = cast<VectorType>(ICA.getArgTypes()[1]);
1823
1824 auto ValLT = getTypeLegalizationCost(ValTy);
1825 auto MaskLT = getTypeLegalizationCost(MaskTy);
1826
1827 // TODO: Return cheaper cost when the entire lane is inactive.
1828 // The expected asm sequence is:
1829 // vcpop.m a0, v0
1830 // beqz a0, exit # Return passthru when the entire lane is inactive.
1831 // vid v10, v0.t
1832 // vredmaxu.vs v10, v10, v10
1833 // vmv.x.s a0, v10
1834 // zext.b a0, a0
1835 // vslidedown.vx v8, v8, a0
1836 // vmv.x.s a0, v8
1837 // exit:
1838 // ...
1839
1840 // Find a suitable type for a stepvector.
1841 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
1842 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1843 TLI->getVectorIdxTy(getDataLayout()), MaskTy->getElementCount(),
1844 /*ZeroIsPoison=*/true, &VScaleRange);
1845 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1846 Type *StepTy = Type::getIntNTy(MaskTy->getContext(), EltWidth);
1847 auto *StepVecTy = VectorType::get(StepTy, ValTy->getElementCount());
1848 auto StepLT = getTypeLegalizationCost(StepVecTy);
1849
1850 // Currently expandVectorFindLastActive cannot handle step vector split.
1851 // So return invalid when the type needs split.
1852 // FIXME: Remove this if expandVectorFindLastActive supports split vector.
1853 if (StepLT.first > 1)
1855
1857 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1858
1859 Cost += MaskLT.first *
1860 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second, CostKind);
1861 Cost += getCFInstrCost(Instruction::CondBr, CostKind, nullptr);
1862 Cost += StepLT.first *
1863 getRISCVInstructionCost(Opcodes, StepLT.second, CostKind);
1864 Cost += getCastInstrCost(Instruction::ZExt,
1865 Type::getInt64Ty(ValTy->getContext()), StepTy,
1867 Cost += ValLT.first *
1868 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1869 ValLT.second, CostKind);
1870 return Cost;
1871 }
1872 }
1873
1874 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1875 if (auto LT = getTypeLegalizationCost(RetTy);
1876 LT.second.isVector()) {
1877 MVT EltTy = LT.second.getVectorElementType();
1878 if (const auto *Entry = CostTableLookup(VectorIntrinsicCostTable,
1879 ICA.getID(), EltTy))
1880 return LT.first * Entry->Cost;
1881 }
1882 }
1883
1885}
1886
1889 const SCEV *Ptr,
1891 // Address computations for vector indexed load/store likely require an offset
1892 // and/or scaling.
1893 if (ST->hasVInstructions() && PtrTy->isVectorTy())
1894 return getArithmeticInstrCost(Instruction::Add, PtrTy, CostKind);
1895
1896 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
1897}
1898
1900 Type *Src,
1903 const Instruction *I) const {
1904 bool IsVectorType = isa<VectorType>(Dst) && isa<VectorType>(Src);
1905 if (!IsVectorType)
1906 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1907
1908 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1909 // For now, skip all fixed vector cost analysis when P extension is available
1910 // to avoid crashes in getMinRVVVectorSizeInBits()
1911 if (ST->hasStdExtP() &&
1913 return 1; // Treat as single instruction cost for now
1914 }
1915
1916 // FIXME: Need to compute legalizing cost for illegal types. The current
1917 // code handles only legal types and those which can be trivially
1918 // promoted to legal.
1919 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1920 Dst->getScalarSizeInBits() > ST->getELen())
1921 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1922
1923 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1924 assert(ISD && "Invalid opcode");
1925 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
1926 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
1927
1928 // Handle i1 source and dest cases *before* calling logic in BasicTTI.
1929 // The shared implementation doesn't model vector widening during legalization
1930 // and instead assumes scalarization. In order to scalarize an <N x i1>
1931 // vector, we need to extend/trunc to/from i8. If we don't special case
1932 // this, we can get an infinite recursion cycle.
1933 switch (ISD) {
1934 default:
1935 break;
1936 case ISD::SIGN_EXTEND:
1937 case ISD::ZERO_EXTEND:
1938 if (Src->getScalarSizeInBits() == 1) {
1939 // We do not use vsext/vzext to extend from mask vector.
1940 // Instead we use the following instructions to extend from mask vector:
1941 // vmv.v.i v8, 0
1942 // vmerge.vim v8, v8, -1, v0 (repeated per split)
1943 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second, CostKind) +
1944 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1945 DstLT.second, CostKind) +
1946 DstLT.first - 1;
1947 }
1948 break;
1949 case ISD::TRUNCATE:
1950 if (Dst->getScalarSizeInBits() == 1) {
1951 // We do not use several vncvt to truncate to mask vector. So we could
1952 // not use PowDiff to calculate it.
1953 // Instead we use the following instructions to truncate to mask vector:
1954 // vand.vi v8, v8, 1
1955 // vmsne.vi v0, v8, 0
1956 return SrcLT.first *
1957 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
1958 SrcLT.second, CostKind) +
1959 SrcLT.first - 1;
1960 }
1961 break;
1962 };
1963
1964 // Our actual lowering for the case where a wider legal type is available
1965 // uses promotion to the wider type. This is reflected in the result of
1966 // getTypeLegalizationCost, but BasicTTI assumes the widened cases are
1967 // scalarized if the legalized Src and Dst are not equal sized.
1968 const DataLayout &DL = this->getDataLayout();
1969 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
1970 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
1971 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Src),
1972 SrcLT.second.getSizeInBits()) ||
1973 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Dst),
1974 DstLT.second.getSizeInBits()) ||
1975 SrcLT.first > 1 || DstLT.first > 1)
1976 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1977
1978 // The split cost is handled by the base getCastInstrCost
1979 assert((SrcLT.first == 1) && (DstLT.first == 1) && "Illegal type");
1980
1981 int PowDiff = (int)Log2_32(DstLT.second.getScalarSizeInBits()) -
1982 (int)Log2_32(SrcLT.second.getScalarSizeInBits());
1983 switch (ISD) {
1984 case ISD::SIGN_EXTEND:
1985 case ISD::ZERO_EXTEND: {
1986 if ((PowDiff < 1) || (PowDiff > 3))
1987 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1988 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
1989 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
1990 unsigned Op =
1991 (ISD == ISD::SIGN_EXTEND) ? SExtOp[PowDiff - 1] : ZExtOp[PowDiff - 1];
1992 return getRISCVInstructionCost(Op, DstLT.second, CostKind);
1993 }
1994 case ISD::TRUNCATE:
1995 case ISD::FP_EXTEND:
1996 case ISD::FP_ROUND: {
1997 // Counts of narrow/widen instructions.
1998 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
1999 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2000
2001 unsigned Op = (ISD == ISD::TRUNCATE) ? RISCV::VNSRL_WI
2002 : (ISD == ISD::FP_EXTEND) ? RISCV::VFWCVT_F_F_V
2003 : RISCV::VFNCVT_F_F_W;
2005 for (; SrcEltSize != DstEltSize;) {
2006 MVT ElementMVT = (ISD == ISD::TRUNCATE)
2007 ? MVT::getIntegerVT(DstEltSize)
2008 : MVT::getFloatingPointVT(DstEltSize);
2009 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2010 DstEltSize =
2011 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2012 Cost += getRISCVInstructionCost(Op, DstMVT, CostKind);
2013 }
2014 return Cost;
2015 }
2016 case ISD::FP_TO_SINT:
2017 case ISD::FP_TO_UINT: {
2018 unsigned IsSigned = ISD == ISD::FP_TO_SINT;
2019 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2020 unsigned FWCVT =
2021 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2022 unsigned FNCVT =
2023 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2024 unsigned SrcEltSize = Src->getScalarSizeInBits();
2025 unsigned DstEltSize = Dst->getScalarSizeInBits();
2027 if ((SrcEltSize == 16) &&
2028 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2029 // If the target only supports zvfhmin or it is fp16-to-i64 conversion
2030 // pre-widening to f32 and then convert f32 to integer
2031 VectorType *VecF32Ty =
2032 VectorType::get(Type::getFloatTy(Dst->getContext()),
2033 cast<VectorType>(Dst)->getElementCount());
2034 std::pair<InstructionCost, MVT> VecF32LT =
2035 getTypeLegalizationCost(VecF32Ty);
2036 Cost +=
2037 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2038 VecF32LT.second, CostKind);
2039 Cost += getCastInstrCost(Opcode, Dst, VecF32Ty, CCH, CostKind, I);
2040 return Cost;
2041 }
2042 if (DstEltSize == SrcEltSize)
2043 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2044 else if (DstEltSize > SrcEltSize)
2045 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2046 else { // (SrcEltSize > DstEltSize)
2047 // First do a narrowing conversion to an integer half the size, then
2048 // truncate if needed.
2049 MVT ElementVT = MVT::getIntegerVT(SrcEltSize / 2);
2050 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2051 Cost += getRISCVInstructionCost(FNCVT, VecVT, CostKind);
2052 if ((SrcEltSize / 2) > DstEltSize) {
2053 Type *VecTy = EVT(VecVT).getTypeForEVT(Dst->getContext());
2054 Cost +=
2055 getCastInstrCost(Instruction::Trunc, Dst, VecTy, CCH, CostKind, I);
2056 }
2057 }
2058 return Cost;
2059 }
2060 case ISD::SINT_TO_FP:
2061 case ISD::UINT_TO_FP: {
2062 unsigned IsSigned = ISD == ISD::SINT_TO_FP;
2063 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2064 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2065 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2066 unsigned SrcEltSize = Src->getScalarSizeInBits();
2067 unsigned DstEltSize = Dst->getScalarSizeInBits();
2068
2070 if ((DstEltSize == 16) &&
2071 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2072 // If the target only supports zvfhmin or it is i64-to-fp16 conversion
2073 // it is converted to f32 and then converted to f16
2074 VectorType *VecF32Ty =
2075 VectorType::get(Type::getFloatTy(Dst->getContext()),
2076 cast<VectorType>(Dst)->getElementCount());
2077 std::pair<InstructionCost, MVT> VecF32LT =
2078 getTypeLegalizationCost(VecF32Ty);
2079 Cost += getCastInstrCost(Opcode, VecF32Ty, Src, CCH, CostKind, I);
2080 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2081 DstLT.second, CostKind);
2082 return Cost;
2083 }
2084
2085 if (DstEltSize == SrcEltSize)
2086 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2087 else if (DstEltSize > SrcEltSize) {
2088 if ((DstEltSize / 2) > SrcEltSize) {
2089 VectorType *VecTy =
2090 VectorType::get(IntegerType::get(Dst->getContext(), DstEltSize / 2),
2091 cast<VectorType>(Dst)->getElementCount());
2092 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2093 Cost += getCastInstrCost(Op, VecTy, Src, CCH, CostKind, I);
2094 }
2095 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2096 } else
2097 Cost += getRISCVInstructionCost(FNCVT, DstLT.second, CostKind);
2098 return Cost;
2099 }
2100 }
2101 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2102}
2103
2104unsigned RISCVTTIImpl::getEstimatedVLFor(VectorType *Ty) const {
2105 if (isa<ScalableVectorType>(Ty)) {
2106 const unsigned EltSize = DL.getTypeSizeInBits(Ty->getElementType());
2107 const unsigned MinSize = DL.getTypeSizeInBits(Ty).getKnownMinValue();
2108 const unsigned VectorBits = *getVScaleForTuning() * RISCV::RVVBitsPerBlock;
2109 return RISCVTargetLowering::computeVLMAX(VectorBits, EltSize, MinSize);
2110 }
2111 return cast<FixedVectorType>(Ty)->getNumElements();
2112}
2113
2116 FastMathFlags FMF,
2118 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2119 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2120
2121 // Skip if scalar size of Ty is bigger than ELEN.
2122 if (Ty->getScalarSizeInBits() > ST->getELen())
2123 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2124
2125 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2126 if (Ty->getElementType()->isIntegerTy(1)) {
2127 // SelectionDAGBuilder does following transforms:
2128 // vector_reduce_{smin,umax}(<n x i1>) --> vector_reduce_or(<n x i1>)
2129 // vector_reduce_{smax,umin}(<n x i1>) --> vector_reduce_and(<n x i1>)
2130 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2131 return getArithmeticReductionCost(Instruction::Or, Ty, FMF, CostKind);
2132 else
2133 return getArithmeticReductionCost(Instruction::And, Ty, FMF, CostKind);
2134 }
2135
2136 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2138 InstructionCost ExtraCost = 0;
2139 switch (IID) {
2140 case Intrinsic::maximum:
2141 if (FMF.noNaNs()) {
2142 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2143 } else {
2144 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2145 RISCV::VFMV_F_S};
2146 // Cost of Canonical Nan + branch
2147 // lui a0, 523264
2148 // fmv.w.x fa0, a0
2149 Type *DstTy = Ty->getScalarType();
2150 const unsigned EltTyBits = DstTy->getScalarSizeInBits();
2151 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2152 ExtraCost = 1 +
2153 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2155 getCFInstrCost(Instruction::CondBr, CostKind);
2156 }
2157 break;
2158
2159 case Intrinsic::minimum:
2160 if (FMF.noNaNs()) {
2161 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2162 } else {
2163 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2164 RISCV::VFMV_F_S};
2165 // Cost of Canonical Nan + branch
2166 // lui a0, 523264
2167 // fmv.w.x fa0, a0
2168 Type *DstTy = Ty->getScalarType();
2169 const unsigned EltTyBits = DL.getTypeSizeInBits(DstTy);
2170 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2171 ExtraCost = 1 +
2172 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2174 getCFInstrCost(Instruction::CondBr, CostKind);
2175 }
2176 break;
2177 }
2178 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2179 }
2180
2181 // IR Reduction is composed by one rvv reduction instruction and vmv
2182 unsigned SplitOp;
2184 switch (IID) {
2185 default:
2186 llvm_unreachable("Unsupported intrinsic");
2187 case Intrinsic::smax:
2188 SplitOp = RISCV::VMAX_VV;
2189 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2190 break;
2191 case Intrinsic::smin:
2192 SplitOp = RISCV::VMIN_VV;
2193 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2194 break;
2195 case Intrinsic::umax:
2196 SplitOp = RISCV::VMAXU_VV;
2197 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2198 break;
2199 case Intrinsic::umin:
2200 SplitOp = RISCV::VMINU_VV;
2201 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2202 break;
2203 case Intrinsic::maxnum:
2204 SplitOp = RISCV::VFMAX_VV;
2205 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2206 break;
2207 case Intrinsic::minnum:
2208 SplitOp = RISCV::VFMIN_VV;
2209 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2210 break;
2211 }
2212 // Add a cost for data larger than LMUL8
2213 InstructionCost SplitCost =
2214 (LT.first > 1) ? (LT.first - 1) *
2215 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2216 : 0;
2217 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2218}
2219
2222 std::optional<FastMathFlags> FMF,
2224 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2225 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2226
2227 // Skip if scalar size of Ty is bigger than ELEN.
2228 if (Ty->getScalarSizeInBits() > ST->getELen())
2229 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2230
2231 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2232 assert(ISD && "Invalid opcode");
2233
2234 if (ISD != ISD::ADD && ISD != ISD::OR && ISD != ISD::XOR && ISD != ISD::AND &&
2235 ISD != ISD::FADD)
2236 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2237
2238 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2239 Type *ElementTy = Ty->getElementType();
2240 if (ElementTy->isIntegerTy(1)) {
2241 // Example sequences:
2242 // vfirst.m a0, v0
2243 // seqz a0, a0
2244 if (LT.second == MVT::v1i1)
2245 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind) +
2246 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2248
2249 if (ISD == ISD::AND) {
2250 // Example sequences:
2251 // vmand.mm v8, v9, v8 ; needed every time type is split
2252 // vmnot.m v8, v0 ; alias for vmnand
2253 // vcpop.m a0, v8
2254 // seqz a0, a0
2255
2256 // See the discussion: https://github.com/llvm/llvm-project/pull/119160
2257 // For LMUL <= 8, there is no splitting,
2258 // the sequences are vmnot, vcpop and seqz.
2259 // When LMUL > 8 and split = 1,
2260 // the sequences are vmnand, vcpop and seqz.
2261 // When LMUL > 8 and split > 1,
2262 // the sequences are (LT.first-2) * vmand, vmnand, vcpop and seqz.
2263 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2264 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second, CostKind) +
2265 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second, CostKind) +
2266 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2267 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2269 } else if (ISD == ISD::XOR || ISD == ISD::ADD) {
2270 // Example sequences:
2271 // vsetvli a0, zero, e8, mf8, ta, ma
2272 // vmxor.mm v8, v0, v8 ; needed every time type is split
2273 // vcpop.m a0, v8
2274 // andi a0, a0, 1
2275 return (LT.first - 1) *
2276 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind) +
2277 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) + 1;
2278 } else {
2279 assert(ISD == ISD::OR);
2280 // Example sequences:
2281 // vsetvli a0, zero, e8, mf8, ta, ma
2282 // vmor.mm v8, v9, v8 ; needed every time type is split
2283 // vcpop.m a0, v0
2284 // snez a0, a0
2285 return (LT.first - 1) *
2286 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second, CostKind) +
2287 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2288 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2290 }
2291 }
2292
2293 // IR Reduction of or/and is composed by one vmv and one rvv reduction
2294 // instruction, and others is composed by two vmv and one rvv reduction
2295 // instruction
2296 unsigned SplitOp;
2298 switch (ISD) {
2299 case ISD::ADD:
2300 SplitOp = RISCV::VADD_VV;
2301 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2302 break;
2303 case ISD::OR:
2304 SplitOp = RISCV::VOR_VV;
2305 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2306 break;
2307 case ISD::XOR:
2308 SplitOp = RISCV::VXOR_VV;
2309 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2310 break;
2311 case ISD::AND:
2312 SplitOp = RISCV::VAND_VV;
2313 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2314 break;
2315 case ISD::FADD:
2316 // We can't promote f16/bf16 fadd reductions.
2317 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2318 LT.second.getScalarType() == MVT::bf16)
2319 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2321 Opcodes.push_back(RISCV::VFMV_S_F);
2322 for (unsigned i = 0; i < LT.first.getValue(); i++)
2323 Opcodes.push_back(RISCV::VFREDOSUM_VS);
2324 Opcodes.push_back(RISCV::VFMV_F_S);
2325 return getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2326 }
2327 SplitOp = RISCV::VFADD_VV;
2328 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2329 break;
2330 }
2331 // Add a cost for data larger than LMUL8
2332 InstructionCost SplitCost =
2333 (LT.first > 1) ? (LT.first - 1) *
2334 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2335 : 0;
2336 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2337}
2338
2340 unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy,
2341 std::optional<FastMathFlags> FMF, TTI::TargetCostKind CostKind) const {
2342 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2343 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2344 FMF, CostKind);
2345
2346 // Skip if scalar size of ResTy is bigger than ELEN.
2347 if (ResTy->getScalarSizeInBits() > ST->getELen())
2348 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2349 FMF, CostKind);
2350
2351 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2352 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2353 FMF, CostKind);
2354
2355 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2356
2357 if (IsUnsigned && Opcode == Instruction::Add &&
2358 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2359 // Represent vector_reduce_add(ZExt(<n x i1>)) as
2360 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
2361 return LT.first *
2362 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind);
2363 }
2364
2365 if (ResTy->getScalarSizeInBits() != 2 * LT.second.getScalarSizeInBits())
2366 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2367 FMF, CostKind);
2368
2369 return (LT.first - 1) +
2370 getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
2371}
2372
2376 assert(OpInfo.isConstant() && "non constant operand?");
2377 if (!isa<VectorType>(Ty))
2378 // FIXME: We need to account for immediate materialization here, but doing
2379 // a decent job requires more knowledge about the immediate than we
2380 // currently have here.
2381 return 0;
2382
2383 if (OpInfo.isUniform())
2384 // vmv.v.i, vmv.v.x, or vfmv.v.f
2385 // We ignore the cost of the scalar constant materialization to be consistent
2386 // with how we treat scalar constants themselves just above.
2387 return 1;
2388
2389 return getConstantPoolLoadCost(Ty, CostKind);
2390}
2391
2393 Align Alignment,
2394 unsigned AddressSpace,
2396 TTI::OperandValueInfo OpInfo,
2397 const Instruction *I) const {
2398 EVT VT = TLI->getValueType(DL, Src, true);
2399 // Type legalization can't handle structs, and load latency isn't handled here
2400 if (VT == MVT::Other ||
2401 (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency))
2402 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2403 CostKind, OpInfo, I);
2404
2406 if (Opcode == Instruction::Store && OpInfo.isConstant())
2407 Cost += getStoreImmCost(Src, OpInfo, CostKind);
2408
2409 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
2410
2411 InstructionCost BaseCost = [&]() {
2412 InstructionCost Cost = LT.first;
2414 return Cost;
2415
2416 // Our actual lowering for the case where a wider legal type is available
2417 // uses the a VL predicated load on the wider type. This is reflected in
2418 // the result of getTypeLegalizationCost, but BasicTTI assumes the
2419 // widened cases are scalarized.
2420 const DataLayout &DL = this->getDataLayout();
2421 if (Src->isVectorTy() && LT.second.isVector() &&
2422 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
2423 LT.second.getSizeInBits()))
2424 return Cost;
2425
2426 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2427 CostKind, OpInfo, I);
2428 }();
2429
2430 // Assume memory ops cost scale with the number of vector registers
2431 // possible accessed by the instruction. Note that BasicTTI already
2432 // handles the LT.first term for us.
2433 if (ST->hasVInstructions() && LT.second.isVector() &&
2435 BaseCost *= TLI->getLMULCost(LT.second);
2436 return Cost + BaseCost;
2437}
2438
2440 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
2442 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
2444 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2445 Op1Info, Op2Info, I);
2446
2447 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2448 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2449 Op1Info, Op2Info, I);
2450
2451 // Skip if scalar size of ValTy is bigger than ELEN.
2452 if (ValTy->isVectorTy() && ValTy->getScalarSizeInBits() > ST->getELen())
2453 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2454 Op1Info, Op2Info, I);
2455
2456 auto GetConstantMatCost =
2457 [&](TTI::OperandValueInfo OpInfo) -> InstructionCost {
2458 if (OpInfo.isUniform())
2459 // We return 0 we currently ignore the cost of materializing scalar
2460 // constants in GPRs.
2461 return 0;
2462
2463 return getConstantPoolLoadCost(ValTy, CostKind);
2464 };
2465
2466 InstructionCost ConstantMatCost;
2467 if (Op1Info.isConstant())
2468 ConstantMatCost += GetConstantMatCost(Op1Info);
2469 if (Op2Info.isConstant())
2470 ConstantMatCost += GetConstantMatCost(Op2Info);
2471
2472 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2473 if (Opcode == Instruction::Select && LT.second.isVector()) {
2474 if (CondTy->isVectorTy()) {
2475 if (ValTy->getScalarSizeInBits() == 1) {
2476 // vmandn.mm v8, v8, v9
2477 // vmand.mm v9, v0, v9
2478 // vmor.mm v0, v9, v8
2479 return ConstantMatCost +
2480 LT.first *
2481 getRISCVInstructionCost(
2482 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2483 LT.second, CostKind);
2484 }
2485 // vselect and max/min are supported natively.
2486 return ConstantMatCost +
2487 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2488 CostKind);
2489 }
2490
2491 if (ValTy->getScalarSizeInBits() == 1) {
2492 // vmv.v.x v9, a0
2493 // vmsne.vi v9, v9, 0
2494 // vmandn.mm v8, v8, v9
2495 // vmand.mm v9, v0, v9
2496 // vmor.mm v0, v9, v8
2497 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2498 return ConstantMatCost +
2499 LT.first *
2500 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2501 InterimVT, CostKind) +
2502 LT.first * getRISCVInstructionCost(
2503 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2504 LT.second, CostKind);
2505 }
2506
2507 // vmv.v.x v10, a0
2508 // vmsne.vi v0, v10, 0
2509 // vmerge.vvm v8, v9, v8, v0
2510 return ConstantMatCost +
2511 LT.first * getRISCVInstructionCost(
2512 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2513 LT.second, CostKind);
2514 }
2515
2516 if ((Opcode == Instruction::ICmp) && ValTy->isVectorTy() &&
2517 CmpInst::isIntPredicate(VecPred)) {
2518 // Use VMSLT_VV to represent VMSEQ, VMSNE, VMSLTU, VMSLEU, VMSLT, VMSLE
2519 // provided they incur the same cost across all implementations
2520 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2521 LT.second,
2522 CostKind);
2523 }
2524
2525 if ((Opcode == Instruction::FCmp) && ValTy->isVectorTy() &&
2526 CmpInst::isFPPredicate(VecPred)) {
2527
2528 // Use VMXOR_MM and VMXNOR_MM to generate all true/false mask
2529 if ((VecPred == CmpInst::FCMP_FALSE) || (VecPred == CmpInst::FCMP_TRUE))
2530 return ConstantMatCost +
2531 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind);
2532
2533 // If we do not support the input floating point vector type, use the base
2534 // one which will calculate as:
2535 // ScalarizeCost + Num * Cost for fixed vector,
2536 // InvalidCost for scalable vector.
2537 if ((ValTy->getScalarSizeInBits() == 16 && !ST->hasVInstructionsF16()) ||
2538 (ValTy->getScalarSizeInBits() == 32 && !ST->hasVInstructionsF32()) ||
2539 (ValTy->getScalarSizeInBits() == 64 && !ST->hasVInstructionsF64()))
2540 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2541 Op1Info, Op2Info, I);
2542
2543 // Assuming vector fp compare and mask instructions are all the same cost
2544 // until a need arises to differentiate them.
2545 switch (VecPred) {
2546 case CmpInst::FCMP_ONE: // vmflt.vv + vmflt.vv + vmor.mm
2547 case CmpInst::FCMP_ORD: // vmfeq.vv + vmfeq.vv + vmand.mm
2548 case CmpInst::FCMP_UNO: // vmfne.vv + vmfne.vv + vmor.mm
2549 case CmpInst::FCMP_UEQ: // vmflt.vv + vmflt.vv + vmnor.mm
2550 return ConstantMatCost +
2551 LT.first * getRISCVInstructionCost(
2552 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2553 LT.second, CostKind);
2554
2555 case CmpInst::FCMP_UGT: // vmfle.vv + vmnot.m
2556 case CmpInst::FCMP_UGE: // vmflt.vv + vmnot.m
2557 case CmpInst::FCMP_ULT: // vmfle.vv + vmnot.m
2558 case CmpInst::FCMP_ULE: // vmflt.vv + vmnot.m
2559 return ConstantMatCost +
2560 LT.first *
2561 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2562 LT.second, CostKind);
2563
2564 case CmpInst::FCMP_OEQ: // vmfeq.vv
2565 case CmpInst::FCMP_OGT: // vmflt.vv
2566 case CmpInst::FCMP_OGE: // vmfle.vv
2567 case CmpInst::FCMP_OLT: // vmflt.vv
2568 case CmpInst::FCMP_OLE: // vmfle.vv
2569 case CmpInst::FCMP_UNE: // vmfne.vv
2570 return ConstantMatCost +
2571 LT.first *
2572 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second, CostKind);
2573 default:
2574 break;
2575 }
2576 }
2577
2578 // With ShortForwardBranchOpt or ConditionalMoveFusion, scalar icmp + select
2579 // instructions will lower to SELECT_CC and lower to PseudoCCMOVGPR which will
2580 // generate a conditional branch + mv. The cost of scalar (icmp + select) will
2581 // be (0 + select instr cost).
2582 if (ST->hasConditionalMoveFusion() && I && isa<ICmpInst>(I) &&
2583 ValTy->isIntegerTy() && !I->user_empty()) {
2584 if (all_of(I->users(), [&](const User *U) {
2585 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2586 U->getType()->isIntegerTy() &&
2587 !isa<ConstantData>(U->getOperand(1)) &&
2588 !isa<ConstantData>(U->getOperand(2));
2589 }))
2590 return 0;
2591 }
2592
2593 // TODO: Add cost for scalar type.
2594
2595 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2596 Op1Info, Op2Info, I);
2597}
2598
2601 const Instruction *I) const {
2603 return Opcode == Instruction::PHI ? 0 : 1;
2604 // Branches are assumed to be predicted.
2605 return 0;
2606}
2607
2609 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
2610 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
2611 assert(Val->isVectorTy() && "This must be a vector type");
2612
2613 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2614 // For now, skip all fixed vector cost analysis when P extension is available
2615 // to avoid crashes in getMinRVVVectorSizeInBits()
2616 if (ST->hasStdExtP() && isa<FixedVectorType>(Val)) {
2617 return 1; // Treat as single instruction cost for now
2618 }
2619
2620 if (Opcode != Instruction::ExtractElement &&
2621 Opcode != Instruction::InsertElement)
2622 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
2623 VIC);
2624
2625 // Legalize the type.
2626 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
2627
2628 // This type is legalized to a scalar type.
2629 if (!LT.second.isVector()) {
2630 auto *FixedVecTy = cast<FixedVectorType>(Val);
2631 // If Index is a known constant, cost is zero.
2632 if (Index != -1U)
2633 return 0;
2634 // Extract/InsertElement with non-constant index is very costly when
2635 // scalarized; estimate cost of loads/stores sequence via the stack:
2636 // ExtractElement cost: store vector to stack, load scalar;
2637 // InsertElement cost: store vector to stack, store scalar, load vector.
2638 Type *ElemTy = FixedVecTy->getElementType();
2639 auto NumElems = FixedVecTy->getNumElements();
2640 auto Align = DL.getPrefTypeAlign(ElemTy);
2641 InstructionCost LoadCost =
2642 getMemoryOpCost(Instruction::Load, ElemTy, Align, 0, CostKind);
2643 InstructionCost StoreCost =
2644 getMemoryOpCost(Instruction::Store, ElemTy, Align, 0, CostKind);
2645 return Opcode == Instruction::ExtractElement
2646 ? StoreCost * NumElems + LoadCost
2647 : (StoreCost + LoadCost) * NumElems + StoreCost;
2648 }
2649
2650 // For unsupported scalable vector.
2651 if (LT.second.isScalableVector() && !LT.first.isValid())
2652 return LT.first;
2653
2654 // Mask vector extract/insert is expanded via e8.
2655 if (Val->getScalarSizeInBits() == 1) {
2656 VectorType *WideTy =
2658 cast<VectorType>(Val)->getElementCount());
2659 if (Opcode == Instruction::ExtractElement) {
2660 InstructionCost ExtendCost
2661 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2663 InstructionCost ExtractCost
2664 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2665 return ExtendCost + ExtractCost;
2666 }
2667 InstructionCost ExtendCost
2668 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2670 InstructionCost InsertCost
2671 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2672 InstructionCost TruncCost
2673 = getCastInstrCost(Instruction::Trunc, Val, WideTy,
2675 return ExtendCost + InsertCost + TruncCost;
2676 }
2677
2678
2679 // In RVV, we could use vslidedown + vmv.x.s to extract element from vector
2680 // and vslideup + vmv.s.x to insert element to vector.
2681 unsigned MoveOpc;
2682 if (LT.second.isFloatingPoint())
2683 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2684 : RISCV::VFMV_F_S;
2685 else
2686 MoveOpc =
2687 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2688 InstructionCost BaseCost =
2689 getRISCVInstructionCost(MoveOpc, LT.second, CostKind);
2690 // When insertelement we should add the index with 1 as the input of vslideup.
2691 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2692
2693 if (Index != -1U) {
2694 // The type may be split. For fixed-width vectors we can normalize the
2695 // index to the new type.
2696 if (LT.second.isFixedLengthVector()) {
2697 unsigned Width = LT.second.getVectorNumElements();
2698 Index = Index % Width;
2699 }
2700
2701 // If exact VLEN is known, we will insert/extract into the appropriate
2702 // subvector with no additional subvector insert/extract cost.
2703 if (auto VLEN = ST->getRealVLen()) {
2704 unsigned EltSize = LT.second.getScalarSizeInBits();
2705 unsigned M1Max = *VLEN / EltSize;
2706 Index = Index % M1Max;
2707 }
2708
2709 if (Index == 0)
2710 // We can extract/insert the first element without vslidedown/vslideup.
2711 SlideCost = 0;
2712 else if (Opcode == Instruction::InsertElement)
2713 SlideCost = 1; // With a constant index, we do not need to use addi.
2714 }
2715
2716 // When the vector needs to split into multiple register groups and the index
2717 // exceeds single vector register group, we need to insert/extract the element
2718 // via stack.
2719 if (LT.first > 1 &&
2720 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2721 LT.second.isScalableVector()))) {
2722 Type *ScalarType = Val->getScalarType();
2723 Align VecAlign = DL.getPrefTypeAlign(Val);
2724 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
2725 // Extra addi for unknown index.
2726 InstructionCost IdxCost = Index == -1U ? 1 : 0;
2727
2728 // Store all split vectors into stack and load the target element.
2729 if (Opcode == Instruction::ExtractElement)
2730 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2731 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
2732 CostKind) +
2733 IdxCost;
2734
2735 // Store all split vectors into stack and store the target element and load
2736 // vectors back.
2737 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2738 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind) +
2739 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
2740 CostKind) +
2741 IdxCost;
2742 }
2743
2744 // Extract i64 in the target that has XLEN=32 need more instruction.
2745 if (Val->getScalarType()->isIntegerTy() &&
2746 ST->getXLen() < Val->getScalarSizeInBits()) {
2747 // For extractelement, we need the following instructions:
2748 // vsetivli zero, 1, e64, m1, ta, mu (not count)
2749 // vslidedown.vx v8, v8, a0
2750 // vmv.x.s a0, v8
2751 // li a1, 32
2752 // vsrl.vx v8, v8, a1
2753 // vmv.x.s a1, v8
2754
2755 // For insertelement, we need the following instructions:
2756 // vsetivli zero, 2, e32, m4, ta, ma (don't count)
2757 // vslide1down.vx v12, v8, a0
2758 // vslide1down.vx v12, v12, a1
2759 // addi a0, a2, 1
2760 // vsetvli zero, a0, e64, m4, tu, ma (don't count)
2761 // vslideup.vx v8, v12, a2
2762
2763 // TODO: should we count these special vsetvlis?
2764 BaseCost =
2765 Opcode == Instruction::InsertElement
2766 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2767 RISCV::VSLIDE1DOWN_VX,
2768 RISCV::VSLIDEUP_VX},
2769 LT.second, CostKind)
2770 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2771 RISCV::VSRL_VX, RISCV::VMV_X_S},
2772 LT.second, CostKind);
2773 }
2774 return BaseCost + SlideCost;
2775}
2776
2780 unsigned Index) const {
2781 if (isa<FixedVectorType>(Val))
2783 Index);
2784
2785 // TODO: This code replicates what LoopVectorize.cpp used to do when asking
2786 // for the cost of extracting the last lane of a scalable vector. It probably
2787 // needs a more accurate cost.
2788 ElementCount EC = cast<VectorType>(Val)->getElementCount();
2789 assert(Index < EC.getKnownMinValue() && "Unexpected reverse index");
2790 return getVectorInstrCost(Opcode, Val, CostKind,
2791 EC.getKnownMinValue() - 1 - Index, nullptr,
2792 nullptr);
2793}
2794
2795/// Check to see if this instruction is expected to be combined to a simpler
2796/// operation during/before lowering. If so return the cost of the combined
2797/// operation rather than provided one. For instance, `udiv i16 %X, 2` is likely
2798/// to be combined to `lshr i16 %X, 1`, so return the cost of a `lshr` rather
2799/// than the cost of a `udiv`
2800std::optional<InstructionCost>
2802 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2804 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2805 // Vector unsigned division/remainder will be simplified to shifts/masks.
2806 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2807 Opd2Info.isConstant() && Opd2Info.isPowerOf2()) {
2808 if (Opcode == Instruction::UDiv)
2809 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind, Opd1Info,
2810 Opd2Info.getNoProps());
2811 // UREM
2812 return getArithmeticInstrCost(Instruction::And, Ty, CostKind, Opd1Info,
2813 Opd2Info.getNoProps());
2814 }
2815 return std::nullopt;
2816}
2817
2819 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2821 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2822
2823 // TODO: Handle more cost kinds.
2825 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2826 Args, CxtI);
2827
2828 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2829 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2830 Args, CxtI);
2831
2832 // Skip if scalar size of Ty is bigger than ELEN.
2833 if (isa<VectorType>(Ty) && Ty->getScalarSizeInBits() > ST->getELen())
2834 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2835 Args, CxtI);
2836
2837 if (std::optional<InstructionCost> CombinedCost =
2839 Op2Info, Args, CxtI))
2840 return *CombinedCost;
2841
2842 // Legalize the type.
2843 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2844 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2845
2846 // TODO: Handle scalar type.
2847 if (!LT.second.isVector()) {
2848 static const CostTblEntry DivTbl[]{
2849 {ISD::UDIV, MVT::i32, TTI::TCC_Expensive},
2850 {ISD::UDIV, MVT::i64, TTI::TCC_Expensive},
2851 {ISD::SDIV, MVT::i32, TTI::TCC_Expensive},
2852 {ISD::SDIV, MVT::i64, TTI::TCC_Expensive},
2853 {ISD::UREM, MVT::i32, TTI::TCC_Expensive},
2854 {ISD::UREM, MVT::i64, TTI::TCC_Expensive},
2855 {ISD::SREM, MVT::i32, TTI::TCC_Expensive},
2856 {ISD::SREM, MVT::i64, TTI::TCC_Expensive}};
2857 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2858 if (const auto *Entry = CostTableLookup(DivTbl, ISDOpcode, LT.second))
2859 return Entry->Cost * LT.first;
2860
2861 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2862 Args, CxtI);
2863 }
2864
2865 // f16 with zvfhmin and bf16 will be promoted to f32.
2866 // FIXME: nxv32[b]f16 will be custom lowered and split.
2867 InstructionCost CastCost = 0;
2868 if ((LT.second.getVectorElementType() == MVT::f16 ||
2869 LT.second.getVectorElementType() == MVT::bf16) &&
2870 TLI->getOperationAction(ISDOpcode, LT.second) ==
2872 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2873 Type *PromotedTy = EVT(PromotedVT).getTypeForEVT(Ty->getContext());
2874 Type *LegalTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
2875 // Add cost of extending arguments
2876 CastCost += LT.first * Args.size() *
2877 getCastInstrCost(Instruction::FPExt, PromotedTy, LegalTy,
2879 // Add cost of truncating result
2880 CastCost +=
2881 LT.first * getCastInstrCost(Instruction::FPTrunc, LegalTy, PromotedTy,
2883 // Compute cost of op in promoted type
2884 LT.second = PromotedVT;
2885 }
2886
2887 auto getConstantMatCost =
2888 [&](unsigned Operand, TTI::OperandValueInfo OpInfo) -> InstructionCost {
2889 if (OpInfo.isUniform() && canSplatOperand(Opcode, Operand))
2890 // Two sub-cases:
2891 // * Has a 5 bit immediate operand which can be splatted.
2892 // * Has a larger immediate which must be materialized in scalar register
2893 // We return 0 for both as we currently ignore the cost of materializing
2894 // scalar constants in GPRs.
2895 return 0;
2896
2897 return getConstantPoolLoadCost(Ty, CostKind);
2898 };
2899
2900 // Add the cost of materializing any constant vectors required.
2901 InstructionCost ConstantMatCost = 0;
2902 if (Op1Info.isConstant())
2903 ConstantMatCost += getConstantMatCost(0, Op1Info);
2904 if (Op2Info.isConstant())
2905 ConstantMatCost += getConstantMatCost(1, Op2Info);
2906
2907 unsigned Op;
2908 switch (ISDOpcode) {
2909 case ISD::ADD:
2910 case ISD::SUB:
2911 Op = RISCV::VADD_VV;
2912 break;
2913 case ISD::SHL:
2914 case ISD::SRL:
2915 case ISD::SRA:
2916 Op = RISCV::VSLL_VV;
2917 break;
2918 case ISD::AND:
2919 case ISD::OR:
2920 case ISD::XOR:
2921 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2922 break;
2923 case ISD::MUL:
2924 case ISD::MULHS:
2925 case ISD::MULHU:
2926 Op = RISCV::VMUL_VV;
2927 break;
2928 case ISD::SDIV:
2929 case ISD::UDIV:
2930 Op = RISCV::VDIV_VV;
2931 break;
2932 case ISD::SREM:
2933 case ISD::UREM:
2934 Op = RISCV::VREM_VV;
2935 break;
2936 case ISD::FADD:
2937 case ISD::FSUB:
2938 Op = RISCV::VFADD_VV;
2939 break;
2940 case ISD::FMUL:
2941 Op = RISCV::VFMUL_VV;
2942 break;
2943 case ISD::FDIV:
2944 Op = RISCV::VFDIV_VV;
2945 break;
2946 case ISD::FNEG:
2947 Op = RISCV::VFSGNJN_VV;
2948 break;
2949 default:
2950 // Assuming all other instructions have the same cost until a need arises to
2951 // differentiate them.
2952 return CastCost + ConstantMatCost +
2953 BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2954 Args, CxtI);
2955 }
2956
2957 InstructionCost InstrCost = getRISCVInstructionCost(Op, LT.second, CostKind);
2958 // We use BasicTTIImpl to calculate scalar costs, which assumes floating point
2959 // ops are twice as expensive as integer ops. Do the same for vectors so
2960 // scalar floating point ops aren't cheaper than their vector equivalents.
2961 if (Ty->isFPOrFPVectorTy())
2962 InstrCost *= 2;
2963 return CastCost + ConstantMatCost + LT.first * InstrCost;
2964}
2965
2966// TODO: Deduplicate from TargetTransformInfoImplCRTPBase.
2968 ArrayRef<const Value *> Ptrs, const Value *Base,
2969 const TTI::PointersChainInfo &Info, Type *AccessTy,
2970 const TTI::TargetCostKind CostKind) const {
2972 // In the basic model we take into account GEP instructions only
2973 // (although here can come alloca instruction, a value, constants and/or
2974 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
2975 // pointer). Typically, if Base is a not a GEP-instruction and all the
2976 // pointers are relative to the same base address, all the rest are
2977 // either GEP instructions, PHIs, bitcasts or constants. When we have same
2978 // base, we just calculate cost of each non-Base GEP as an ADD operation if
2979 // any their index is a non-const.
2980 // If no known dependencies between the pointers cost is calculated as a sum
2981 // of costs of GEP instructions.
2982 for (auto [I, V] : enumerate(Ptrs)) {
2983 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
2984 if (!GEP)
2985 continue;
2986 if (Info.isSameBase() && V != Base) {
2987 if (GEP->hasAllConstantIndices())
2988 continue;
2989 // If the chain is unit-stride and BaseReg + stride*i is a legal
2990 // addressing mode, then presume the base GEP is sitting around in a
2991 // register somewhere and check if we can fold the offset relative to
2992 // it.
2993 unsigned Stride = DL.getTypeStoreSize(AccessTy);
2994 if (Info.isUnitStride() &&
2995 isLegalAddressingMode(AccessTy,
2996 /* BaseGV */ nullptr,
2997 /* BaseOffset */ Stride * I,
2998 /* HasBaseReg */ true,
2999 /* Scale */ 0,
3000 GEP->getType()->getPointerAddressSpace()))
3001 continue;
3002 Cost += getArithmeticInstrCost(Instruction::Add, GEP->getType(), CostKind,
3003 {TTI::OK_AnyValue, TTI::OP_None},
3004 {TTI::OK_AnyValue, TTI::OP_None}, {});
3005 } else {
3006 SmallVector<const Value *> Indices(GEP->indices());
3007 Cost += getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
3008 Indices, CostKind, AccessTy);
3009 }
3010 }
3011 return Cost;
3012}
3013
3016 OptimizationRemarkEmitter *ORE) const {
3017 // TODO: More tuning on benchmarks and metrics with changes as needed
3018 // would apply to all settings below to enable performance.
3019
3020
3021 if (ST->enableDefaultUnroll())
3022 return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP, ORE);
3023
3024 // Enable Upper bound unrolling universally, not dependent upon the conditions
3025 // below.
3026 UP.UpperBound = true;
3027
3028 // Disable loop unrolling for Oz and Os.
3029 UP.OptSizeThreshold = 0;
3031 if (L->getHeader()->getParent()->hasOptSize())
3032 return;
3033
3034 SmallVector<BasicBlock *, 4> ExitingBlocks;
3035 L->getExitingBlocks(ExitingBlocks);
3036 LLVM_DEBUG(dbgs() << "Loop has:\n"
3037 << "Blocks: " << L->getNumBlocks() << "\n"
3038 << "Exit blocks: " << ExitingBlocks.size() << "\n");
3039
3040 // Only allow another exit other than the latch. This acts as an early exit
3041 // as it mirrors the profitability calculation of the runtime unroller.
3042 if (ExitingBlocks.size() > 2)
3043 return;
3044
3045 // Limit the CFG of the loop body for targets with a branch predictor.
3046 // Allowing 4 blocks permits if-then-else diamonds in the body.
3047 if (L->getNumBlocks() > 4)
3048 return;
3049
3050 // Scan the loop: don't unroll loops with calls as this could prevent
3051 // inlining. Don't unroll auto-vectorized loops either, though do allow
3052 // unrolling of the scalar remainder.
3053 bool IsVectorized = getBooleanLoopAttribute(L, "llvm.loop.isvectorized");
3055 for (auto *BB : L->getBlocks()) {
3056 for (auto &I : *BB) {
3057 // Both auto-vectorized loops and the scalar remainder have the
3058 // isvectorized attribute, so differentiate between them by the presence
3059 // of vector instructions.
3060 if (IsVectorized && (I.getType()->isVectorTy() ||
3061 llvm::any_of(I.operand_values(), [](Value *V) {
3062 return V->getType()->isVectorTy();
3063 })))
3064 return;
3065
3066 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
3067 if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
3068 if (!isLoweredToCall(F))
3069 continue;
3070 }
3071 return;
3072 }
3073
3074 SmallVector<const Value *> Operands(I.operand_values());
3077 }
3078 }
3079
3080 LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");
3081
3082 UP.Partial = true;
3083 UP.Runtime = true;
3084 UP.UnrollRemainder = true;
3085 UP.UnrollAndJam = true;
3086
3087 // Force unrolling small loops can be very useful because of the branch
3088 // taken cost of the backedge.
3089 if (Cost < 12)
3090 UP.Force = true;
3091}
3092
3097
3099 MemIntrinsicInfo &Info) const {
3100 const DataLayout &DL = getDataLayout();
3101 Intrinsic::ID IID = Inst->getIntrinsicID();
3102 LLVMContext &C = Inst->getContext();
3103 bool HasMask = false;
3104
3105 auto getSegNum = [](const IntrinsicInst *II, unsigned PtrOperandNo,
3106 bool IsWrite) -> int64_t {
3107 if (auto *TarExtTy =
3108 dyn_cast<TargetExtType>(II->getArgOperand(0)->getType()))
3109 return TarExtTy->getIntParameter(0);
3110
3111 return 1;
3112 };
3113
3114 switch (IID) {
3115 case Intrinsic::riscv_vle_mask:
3116 case Intrinsic::riscv_vse_mask:
3117 case Intrinsic::riscv_vlseg2_mask:
3118 case Intrinsic::riscv_vlseg3_mask:
3119 case Intrinsic::riscv_vlseg4_mask:
3120 case Intrinsic::riscv_vlseg5_mask:
3121 case Intrinsic::riscv_vlseg6_mask:
3122 case Intrinsic::riscv_vlseg7_mask:
3123 case Intrinsic::riscv_vlseg8_mask:
3124 case Intrinsic::riscv_vsseg2_mask:
3125 case Intrinsic::riscv_vsseg3_mask:
3126 case Intrinsic::riscv_vsseg4_mask:
3127 case Intrinsic::riscv_vsseg5_mask:
3128 case Intrinsic::riscv_vsseg6_mask:
3129 case Intrinsic::riscv_vsseg7_mask:
3130 case Intrinsic::riscv_vsseg8_mask:
3131 HasMask = true;
3132 [[fallthrough]];
3133 case Intrinsic::riscv_vle:
3134 case Intrinsic::riscv_vse:
3135 case Intrinsic::riscv_vlseg2:
3136 case Intrinsic::riscv_vlseg3:
3137 case Intrinsic::riscv_vlseg4:
3138 case Intrinsic::riscv_vlseg5:
3139 case Intrinsic::riscv_vlseg6:
3140 case Intrinsic::riscv_vlseg7:
3141 case Intrinsic::riscv_vlseg8:
3142 case Intrinsic::riscv_vsseg2:
3143 case Intrinsic::riscv_vsseg3:
3144 case Intrinsic::riscv_vsseg4:
3145 case Intrinsic::riscv_vsseg5:
3146 case Intrinsic::riscv_vsseg6:
3147 case Intrinsic::riscv_vsseg7:
3148 case Intrinsic::riscv_vsseg8: {
3149 // Intrinsic interface:
3150 // riscv_vle(merge, ptr, vl)
3151 // riscv_vle_mask(merge, ptr, mask, vl, policy)
3152 // riscv_vse(val, ptr, vl)
3153 // riscv_vse_mask(val, ptr, mask, vl, policy)
3154 // riscv_vlseg#(merge, ptr, vl, sew)
3155 // riscv_vlseg#_mask(merge, ptr, mask, vl, policy, sew)
3156 // riscv_vsseg#(val, ptr, vl, sew)
3157 // riscv_vsseg#_mask(val, ptr, mask, vl, sew)
3158 bool IsWrite = Inst->getType()->isVoidTy();
3159 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3160 // The results of segment loads are TargetExtType.
3161 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3162 unsigned SEW =
3163 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3164 ->getZExtValue();
3165 Ty = TarExtTy->getTypeParameter(0U);
3167 IntegerType::get(C, SEW),
3168 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3169 }
3170 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3171 unsigned VLIndex = RVVIInfo->VLOperand;
3172 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3173 MaybeAlign Alignment =
3174 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3175 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3176 Value *Mask = ConstantInt::getTrue(MaskType);
3177 if (HasMask)
3178 Mask = Inst->getArgOperand(VLIndex - 1);
3179 Value *EVL = Inst->getArgOperand(VLIndex);
3180 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3181 // RVV uses contiguous elements as a segment.
3182 if (SegNum > 1) {
3183 unsigned ElemSize = Ty->getScalarSizeInBits();
3184 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3185 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3186 }
3187 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3188 Alignment, Mask, EVL);
3189 return true;
3190 }
3191 case Intrinsic::riscv_vlse_mask:
3192 case Intrinsic::riscv_vsse_mask:
3193 case Intrinsic::riscv_vlsseg2_mask:
3194 case Intrinsic::riscv_vlsseg3_mask:
3195 case Intrinsic::riscv_vlsseg4_mask:
3196 case Intrinsic::riscv_vlsseg5_mask:
3197 case Intrinsic::riscv_vlsseg6_mask:
3198 case Intrinsic::riscv_vlsseg7_mask:
3199 case Intrinsic::riscv_vlsseg8_mask:
3200 case Intrinsic::riscv_vssseg2_mask:
3201 case Intrinsic::riscv_vssseg3_mask:
3202 case Intrinsic::riscv_vssseg4_mask:
3203 case Intrinsic::riscv_vssseg5_mask:
3204 case Intrinsic::riscv_vssseg6_mask:
3205 case Intrinsic::riscv_vssseg7_mask:
3206 case Intrinsic::riscv_vssseg8_mask:
3207 HasMask = true;
3208 [[fallthrough]];
3209 case Intrinsic::riscv_vlse:
3210 case Intrinsic::riscv_vsse:
3211 case Intrinsic::riscv_vlsseg2:
3212 case Intrinsic::riscv_vlsseg3:
3213 case Intrinsic::riscv_vlsseg4:
3214 case Intrinsic::riscv_vlsseg5:
3215 case Intrinsic::riscv_vlsseg6:
3216 case Intrinsic::riscv_vlsseg7:
3217 case Intrinsic::riscv_vlsseg8:
3218 case Intrinsic::riscv_vssseg2:
3219 case Intrinsic::riscv_vssseg3:
3220 case Intrinsic::riscv_vssseg4:
3221 case Intrinsic::riscv_vssseg5:
3222 case Intrinsic::riscv_vssseg6:
3223 case Intrinsic::riscv_vssseg7:
3224 case Intrinsic::riscv_vssseg8: {
3225 // Intrinsic interface:
3226 // riscv_vlse(merge, ptr, stride, vl)
3227 // riscv_vlse_mask(merge, ptr, stride, mask, vl, policy)
3228 // riscv_vsse(val, ptr, stride, vl)
3229 // riscv_vsse_mask(val, ptr, stride, mask, vl, policy)
3230 // riscv_vlsseg#(merge, ptr, offset, vl, sew)
3231 // riscv_vlsseg#_mask(merge, ptr, offset, mask, vl, policy, sew)
3232 // riscv_vssseg#(val, ptr, offset, vl, sew)
3233 // riscv_vssseg#_mask(val, ptr, offset, mask, vl, sew)
3234 bool IsWrite = Inst->getType()->isVoidTy();
3235 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3236 // The results of segment loads are TargetExtType.
3237 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3238 unsigned SEW =
3239 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3240 ->getZExtValue();
3241 Ty = TarExtTy->getTypeParameter(0U);
3243 IntegerType::get(C, SEW),
3244 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3245 }
3246 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3247 unsigned VLIndex = RVVIInfo->VLOperand;
3248 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3249 MaybeAlign Alignment =
3250 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3251
3252 Value *Stride = Inst->getArgOperand(PtrOperandNo + 1);
3253 // Use the pointer alignment as the element alignment if the stride is a
3254 // multiple of the pointer alignment. Otherwise, the element alignment
3255 // should be the greatest common divisor of pointer alignment and stride.
3256 // For simplicity, just consider unalignment for elements.
3257 unsigned PointerAlign = Alignment.valueOrOne().value();
3258 if (!isa<ConstantInt>(Stride) ||
3259 cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
3260 Alignment = Align(1);
3261
3262 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3263 Value *Mask = ConstantInt::getTrue(MaskType);
3264 if (HasMask)
3265 Mask = Inst->getArgOperand(VLIndex - 1);
3266 Value *EVL = Inst->getArgOperand(VLIndex);
3267 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3268 // RVV uses contiguous elements as a segment.
3269 if (SegNum > 1) {
3270 unsigned ElemSize = Ty->getScalarSizeInBits();
3271 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3272 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3273 }
3274 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3275 Alignment, Mask, EVL, Stride);
3276 return true;
3277 }
3278 case Intrinsic::riscv_vloxei_mask:
3279 case Intrinsic::riscv_vluxei_mask:
3280 case Intrinsic::riscv_vsoxei_mask:
3281 case Intrinsic::riscv_vsuxei_mask:
3282 case Intrinsic::riscv_vloxseg2_mask:
3283 case Intrinsic::riscv_vloxseg3_mask:
3284 case Intrinsic::riscv_vloxseg4_mask:
3285 case Intrinsic::riscv_vloxseg5_mask:
3286 case Intrinsic::riscv_vloxseg6_mask:
3287 case Intrinsic::riscv_vloxseg7_mask:
3288 case Intrinsic::riscv_vloxseg8_mask:
3289 case Intrinsic::riscv_vluxseg2_mask:
3290 case Intrinsic::riscv_vluxseg3_mask:
3291 case Intrinsic::riscv_vluxseg4_mask:
3292 case Intrinsic::riscv_vluxseg5_mask:
3293 case Intrinsic::riscv_vluxseg6_mask:
3294 case Intrinsic::riscv_vluxseg7_mask:
3295 case Intrinsic::riscv_vluxseg8_mask:
3296 case Intrinsic::riscv_vsoxseg2_mask:
3297 case Intrinsic::riscv_vsoxseg3_mask:
3298 case Intrinsic::riscv_vsoxseg4_mask:
3299 case Intrinsic::riscv_vsoxseg5_mask:
3300 case Intrinsic::riscv_vsoxseg6_mask:
3301 case Intrinsic::riscv_vsoxseg7_mask:
3302 case Intrinsic::riscv_vsoxseg8_mask:
3303 case Intrinsic::riscv_vsuxseg2_mask:
3304 case Intrinsic::riscv_vsuxseg3_mask:
3305 case Intrinsic::riscv_vsuxseg4_mask:
3306 case Intrinsic::riscv_vsuxseg5_mask:
3307 case Intrinsic::riscv_vsuxseg6_mask:
3308 case Intrinsic::riscv_vsuxseg7_mask:
3309 case Intrinsic::riscv_vsuxseg8_mask:
3310 HasMask = true;
3311 [[fallthrough]];
3312 case Intrinsic::riscv_vloxei:
3313 case Intrinsic::riscv_vluxei:
3314 case Intrinsic::riscv_vsoxei:
3315 case Intrinsic::riscv_vsuxei:
3316 case Intrinsic::riscv_vloxseg2:
3317 case Intrinsic::riscv_vloxseg3:
3318 case Intrinsic::riscv_vloxseg4:
3319 case Intrinsic::riscv_vloxseg5:
3320 case Intrinsic::riscv_vloxseg6:
3321 case Intrinsic::riscv_vloxseg7:
3322 case Intrinsic::riscv_vloxseg8:
3323 case Intrinsic::riscv_vluxseg2:
3324 case Intrinsic::riscv_vluxseg3:
3325 case Intrinsic::riscv_vluxseg4:
3326 case Intrinsic::riscv_vluxseg5:
3327 case Intrinsic::riscv_vluxseg6:
3328 case Intrinsic::riscv_vluxseg7:
3329 case Intrinsic::riscv_vluxseg8:
3330 case Intrinsic::riscv_vsoxseg2:
3331 case Intrinsic::riscv_vsoxseg3:
3332 case Intrinsic::riscv_vsoxseg4:
3333 case Intrinsic::riscv_vsoxseg5:
3334 case Intrinsic::riscv_vsoxseg6:
3335 case Intrinsic::riscv_vsoxseg7:
3336 case Intrinsic::riscv_vsoxseg8:
3337 case Intrinsic::riscv_vsuxseg2:
3338 case Intrinsic::riscv_vsuxseg3:
3339 case Intrinsic::riscv_vsuxseg4:
3340 case Intrinsic::riscv_vsuxseg5:
3341 case Intrinsic::riscv_vsuxseg6:
3342 case Intrinsic::riscv_vsuxseg7:
3343 case Intrinsic::riscv_vsuxseg8: {
3344 // Intrinsic interface (only listed ordered version):
3345 // riscv_vloxei(merge, ptr, index, vl)
3346 // riscv_vloxei_mask(merge, ptr, index, mask, vl, policy)
3347 // riscv_vsoxei(val, ptr, index, vl)
3348 // riscv_vsoxei_mask(val, ptr, index, mask, vl, policy)
3349 // riscv_vloxseg#(merge, ptr, index, vl, sew)
3350 // riscv_vloxseg#_mask(merge, ptr, index, mask, vl, policy, sew)
3351 // riscv_vsoxseg#(val, ptr, index, vl, sew)
3352 // riscv_vsoxseg#_mask(val, ptr, index, mask, vl, sew)
3353 bool IsWrite = Inst->getType()->isVoidTy();
3354 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3355 // The results of segment loads are TargetExtType.
3356 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3357 unsigned SEW =
3358 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3359 ->getZExtValue();
3360 Ty = TarExtTy->getTypeParameter(0U);
3362 IntegerType::get(C, SEW),
3363 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3364 }
3365 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3366 unsigned VLIndex = RVVIInfo->VLOperand;
3367 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3368 Value *Mask;
3369 if (HasMask) {
3370 Mask = Inst->getArgOperand(VLIndex - 1);
3371 } else {
3372 // Mask cannot be nullptr here: vector GEP produces <vscale x N x ptr>,
3373 // and casting that to scalar i64 triggers a vector/scalar mismatch
3374 // assertion in CreatePointerCast. Use an all-true mask so ASan lowers it
3375 // via extractelement instead.
3376 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3377 Mask = ConstantInt::getTrue(MaskType);
3378 }
3379 Value *EVL = Inst->getArgOperand(VLIndex);
3380 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3381 // RVV uses contiguous elements as a segment.
3382 if (SegNum > 1) {
3383 unsigned ElemSize = Ty->getScalarSizeInBits();
3384 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3385 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3386 }
3387 Value *OffsetOp = Inst->getArgOperand(PtrOperandNo + 1);
3388 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3389 Align(1), Mask, EVL,
3390 /* Stride */ nullptr, OffsetOp);
3391 return true;
3392 }
3393 }
3394 return false;
3395}
3396
3398 if (Ty->isVectorTy()) {
3399 // f16 with only zvfhmin and bf16 will be promoted to f32
3400 Type *EltTy = cast<VectorType>(Ty)->getElementType();
3401 if ((EltTy->isHalfTy() && !ST->hasVInstructionsF16()) ||
3402 EltTy->isBFloatTy())
3403 Ty = VectorType::get(Type::getFloatTy(Ty->getContext()),
3404 cast<VectorType>(Ty));
3405
3406 TypeSize Size = DL.getTypeSizeInBits(Ty);
3407 if (Size.isScalable() && ST->hasVInstructions())
3408 return divideCeil(Size.getKnownMinValue(), RISCV::RVVBitsPerBlock);
3409
3410 if (ST->useRVVForFixedLengthVectors())
3411 return divideCeil(Size, ST->getRealMinVLen());
3412 }
3413
3414 return BaseT::getRegUsageForType(Ty);
3415}
3416
3417unsigned RISCVTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
3418 if (SLPMaxVF.getNumOccurrences())
3419 return SLPMaxVF;
3420
3421 // Return how many elements can fit in getRegisterBitwidth. This is the
3422 // same routine as used in LoopVectorizer. We should probably be
3423 // accounting for whether we actually have instructions with the right
3424 // lane type, but we don't have enough information to do that without
3425 // some additional plumbing which hasn't been justified yet.
3426 TypeSize RegWidth =
3428 // If no vector registers, or absurd element widths, disable
3429 // vectorization by returning 1.
3430 return std::max<unsigned>(1U, RegWidth.getFixedValue() / ElemWidth);
3431}
3432
3436
3438 return ST->enableUnalignedVectorMem();
3439}
3440
3443 ScalarEvolution *SE) const {
3444 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3445 return TTI::AMK_PostIndexed;
3446
3448}
3449
3451 const TargetTransformInfo::LSRCost &C2) const {
3452 // RISC-V specific here are "instruction number 1st priority".
3453 // If we need to emit adds inside the loop to add up base registers, then
3454 // we need at least one extra temporary register.
3455 unsigned C1NumRegs = C1.NumRegs + (C1.NumBaseAdds != 0);
3456 unsigned C2NumRegs = C2.NumRegs + (C2.NumBaseAdds != 0);
3457 return std::tie(C1.Insns, C1NumRegs, C1.AddRecCost,
3458 C1.NumIVMuls, C1.NumBaseAdds,
3459 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
3460 std::tie(C2.Insns, C2NumRegs, C2.AddRecCost,
3461 C2.NumIVMuls, C2.NumBaseAdds,
3462 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
3463}
3464
3466 Align Alignment) const {
3467 auto *VTy = dyn_cast<VectorType>(DataTy);
3468 if (!VTy || VTy->isScalableTy())
3469 return false;
3470
3471 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3472 return false;
3473
3474 // FIXME: If it is an i8 vector and the element count exceeds 256, we should
3475 // scalarize these types with LMUL >= maximum fixed-length LMUL.
3476 if (VTy->getElementType()->isIntegerTy(8))
3477 if (VTy->getElementCount().getFixedValue() > 256)
3478 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3479 ST->getMaxLMULForFixedLengthVectors();
3480 return true;
3481}
3482
3484 Align Alignment) const {
3485 auto *VTy = dyn_cast<VectorType>(DataTy);
3486 if (!VTy || VTy->isScalableTy())
3487 return false;
3488
3489 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3490 return false;
3491 return true;
3492}
3493
3495 ElementCount NumElements) const {
3496 // Optimized zero-stride loads can be treated as broadcasts.
3497 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3498 return false;
3499
3500 return TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, ElementTy));
3501}
3502
3503/// See if \p I should be considered for address type promotion. We check if \p
3504/// I is a sext with right type and used in memory accesses. If it used in a
3505/// "complex" getelementptr, we allow it to be promoted without finding other
3506/// sext instructions that sign extended the same initial value. A getelementptr
3507/// is considered as "complex" if it has more than 2 operands.
3509 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
3510 bool Considerable = false;
3511 AllowPromotionWithoutCommonHeader = false;
3512 if (!isa<SExtInst>(&I))
3513 return false;
3514 Type *ConsideredSExtType =
3515 Type::getInt64Ty(I.getParent()->getParent()->getContext());
3516 if (I.getType() != ConsideredSExtType)
3517 return false;
3518 // See if the sext is the one with the right type and used in at least one
3519 // GetElementPtrInst.
3520 for (const User *U : I.users()) {
3521 if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
3522 Considerable = true;
3523 // A getelementptr is considered as "complex" if it has more than 2
3524 // operands. We will promote a SExt used in such complex GEP as we
3525 // expect some computation to be merged if they are done on 64 bits.
3526 if (GEPInst->getNumOperands() > 2) {
3527 AllowPromotionWithoutCommonHeader = true;
3528 break;
3529 }
3530 }
3531 }
3532 return Considerable;
3533}
3534
3535bool RISCVTTIImpl::canSplatOperand(unsigned Opcode, int Operand) const {
3536 switch (Opcode) {
3537 case Instruction::Add:
3538 case Instruction::Sub:
3539 case Instruction::Mul:
3540 case Instruction::And:
3541 case Instruction::Or:
3542 case Instruction::Xor:
3543 case Instruction::FAdd:
3544 case Instruction::FSub:
3545 case Instruction::FMul:
3546 case Instruction::FDiv:
3547 case Instruction::ICmp:
3548 case Instruction::FCmp:
3549 return true;
3550 case Instruction::Shl:
3551 case Instruction::LShr:
3552 case Instruction::AShr:
3553 case Instruction::UDiv:
3554 case Instruction::SDiv:
3555 case Instruction::URem:
3556 case Instruction::SRem:
3557 case Instruction::Select:
3558 return Operand == 1;
3559 default:
3560 return false;
3561 }
3562}
3563
3565 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3566 return false;
3567
3568 if (canSplatOperand(I->getOpcode(), Operand))
3569 return true;
3570
3571 auto *II = dyn_cast<IntrinsicInst>(I);
3572 if (!II)
3573 return false;
3574
3575 switch (II->getIntrinsicID()) {
3576 case Intrinsic::fma:
3577 case Intrinsic::fmuladd:
3578 return Operand == 0 || Operand == 1;
3579 case Intrinsic::vp_udiv:
3580 case Intrinsic::vp_sdiv:
3581 case Intrinsic::vp_urem:
3582 case Intrinsic::vp_srem:
3583 case Intrinsic::ssub_sat:
3584 case Intrinsic::usub_sat:
3585 return Operand == 1;
3586 // These intrinsics are commutative.
3587 case Intrinsic::smin:
3588 case Intrinsic::umin:
3589 case Intrinsic::smax:
3590 case Intrinsic::umax:
3591 case Intrinsic::sadd_sat:
3592 case Intrinsic::uadd_sat:
3593 return Operand == 0 || Operand == 1;
3594 default:
3595 return false;
3596 }
3597}
3598
3599/// Check if sinking \p I's operands to I's basic block is profitable, because
3600/// the operands can be folded into a target instruction, e.g.
3601/// splats of scalars can fold into vector instructions.
3604 using namespace llvm::PatternMatch;
3605
3606 if (I->isBitwiseLogicOp()) {
3607 if (!I->getType()->isVectorTy()) {
3608 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3609 for (auto &Op : I->operands()) {
3610 // (and/or/xor X, (not Y)) -> (andn/orn/xnor X, Y)
3611 if (match(Op.get(), m_Not(m_Value()))) {
3612 Ops.push_back(&Op);
3613 return true;
3614 }
3615 }
3616 }
3617 } else if (I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3618 for (auto &Op : I->operands()) {
3619 // (and X, (not Y)) -> (vandn.vv X, Y)
3620 if (match(Op.get(), m_Not(m_Value()))) {
3621 Ops.push_back(&Op);
3622 return true;
3623 }
3624 // (and X, (splat (not Y))) -> (vandn.vx X, Y)
3626 m_ZeroInt()),
3627 m_Value(), m_ZeroMask()))) {
3628 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
3629 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
3630 Ops.push_back(&Not);
3631 Ops.push_back(&InsertElt);
3632 Ops.push_back(&Op);
3633 return true;
3634 }
3635 }
3636 }
3637 }
3638
3639 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3640 return false;
3641
3642 // Don't sink splat operands if the target prefers it. Some targets requires
3643 // S2V transfer buffers and we can run out of them copying the same value
3644 // repeatedly.
3645 // FIXME: It could still be worth doing if it would improve vector register
3646 // pressure and prevent a vector spill.
3647 if (!ST->sinkSplatOperands())
3648 return false;
3649
3650 for (auto OpIdx : enumerate(I->operands())) {
3651 if (!canSplatOperand(I, OpIdx.index()))
3652 continue;
3653
3654 Instruction *Op = dyn_cast<Instruction>(OpIdx.value().get());
3655 // Make sure we are not already sinking this operand
3656 if (!Op || any_of(Ops, [&](Use *U) { return U->get() == Op; }))
3657 continue;
3658
3659 // We are looking for a splat that can be sunk.
3661 m_Value(), m_ZeroMask())))
3662 continue;
3663
3664 // Don't sink i1 splats.
3665 if (cast<VectorType>(Op->getType())->getElementType()->isIntegerTy(1))
3666 continue;
3667
3668 // All uses of the shuffle should be sunk to avoid duplicating it across gpr
3669 // and vector registers
3670 for (Use &U : Op->uses()) {
3671 Instruction *Insn = cast<Instruction>(U.getUser());
3672 if (!canSplatOperand(Insn, U.getOperandNo()))
3673 return false;
3674 }
3675
3676 // Sink any fpexts since they might be used in a widening fp pattern.
3677 Use *InsertEltUse = &Op->getOperandUse(0);
3678 auto *InsertElt = cast<InsertElementInst>(InsertEltUse);
3679 if (isa<FPExtInst>(InsertElt->getOperand(1)))
3680 Ops.push_back(&InsertElt->getOperandUse(1));
3681 Ops.push_back(InsertEltUse);
3682 Ops.push_back(&OpIdx.value());
3683 }
3684 return true;
3685}
3686
3688RISCVTTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
3690
3691 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3692 return Options;
3693
3694 Options.AllowOverlappingLoads = true;
3695 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3696 Options.NumLoadsPerBlock = IsZeroCmp ? Options.MaxNumLoads : 1;
3697 if (ST->is64Bit()) {
3698 Options.LoadSizes = {8, 4, 2, 1};
3699 Options.AllowedTailExpansions = {3, 5, 6};
3700 } else {
3701 Options.LoadSizes = {4, 2, 1};
3702 Options.AllowedTailExpansions = {3};
3703 }
3704
3705 if (IsZeroCmp && ST->hasVInstructions()) {
3706 unsigned VLenB = ST->getRealMinVLen() / 8;
3707 // The minimum size should be `XLen / 8 + 1`, and the maxinum size should be
3708 // `VLenB * MaxLMUL` so that it fits in a single register group.
3709 unsigned MinSize = ST->getXLen() / 8 + 1;
3710 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3711 for (unsigned Size = MinSize; Size <= MaxSize; Size++)
3712 Options.LoadSizes.insert(Options.LoadSizes.begin(), Size);
3713 }
3714 return Options;
3715}
3716
3718 const Instruction *I) const {
3720 // For the binary operators (e.g. or) we need to be more careful than
3721 // selects, here we only transform them if they are already at a natural
3722 // break point in the code - the end of a block with an unconditional
3723 // terminator.
3724 if (I->getOpcode() == Instruction::Or &&
3725 isa<UncondBrInst>(I->getNextNode()))
3726 return true;
3727
3728 if (I->getOpcode() == Instruction::Add ||
3729 I->getOpcode() == Instruction::Sub)
3730 return true;
3731 }
3733}
3734
3736 const Function *Caller, const Attribute &Attr) const {
3737 // "interrupt" controls the prolog/epilog of interrupt handlers (and includes
3738 // restrictions on their signatures). We can outline from the bodies of these
3739 // handlers, but when we do we need to make sure we don't mark the outlined
3740 // function as an interrupt handler too.
3741 if (Attr.isStringAttribute() && Attr.getKindAsString() == "interrupt")
3742 return false;
3743
3745}
3746
3747std::optional<Instruction *>
3749 // Attach a range return attribute describing the result of vsetvli/vsetvlimax
3750 // so generic value analyses can reason about it. The verifier guarantees an
3751 // XLen result and constant VSEW/VLMUL encoding a valid vtype, so no defensive
3752 // validation is needed here.
3753 if (is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3754 II.getIntrinsicID())) {
3755 // These intrinsics require the V extension; without it the VLEN queries
3756 // below would assert. Such IR would fail isel anyway, so just bail out.
3757 if (!ST->hasVInstructions())
3758 return {};
3759
3760 bool HasAVL = II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3761 unsigned Offset = HasAVL ? 1 : 0;
3762 unsigned BitWidth = II.getType()->getIntegerBitWidth();
3763 ConstantRange VLenRange(APInt(BitWidth, ST->getRealMinVLen()),
3764 APInt(BitWidth, ST->getRealMaxVLen()) + 1);
3765
3766 uint64_t VSEW = cast<ConstantInt>(II.getArgOperand(Offset))->getZExtValue();
3767 auto VLMUL = static_cast<RISCVVType::VLMUL>(
3768 cast<ConstantInt>(II.getArgOperand(Offset + 1))->getZExtValue());
3769 unsigned SEW = RISCVVType::decodeVSEW(VSEW);
3770 unsigned Ratio = RISCVVType::getSEWLMULRatio(SEW, VLMUL);
3771
3772 // VLMAX = VLEN / (SEW / LMUL), clamped to >= 1 for any usable vtype.
3773 ConstantRange VLMAXRange =
3774 VLenRange.udiv(ConstantRange(APInt(BitWidth, Ratio)))
3776
3777 // vsetvlimax returns exactly VLMAX; vsetvli returns vl with
3778 // 0 <= vl <= min(AVL, VLMAX). vl == AVL only when AVL <= the smallest
3779 // possible VLMAX; otherwise vl can shrink below VLMAX (to 0 at runtime), so
3780 // only the VLMAX upper bound is sound.
3781 ConstantRange VLRange = VLMAXRange;
3782 if (HasAVL) {
3783 // vl ≤ VLMAX
3784 VLRange =
3786
3787 Value *AVL = II.getArgOperand(0);
3789 AVL, /*ForSigned=*/false,
3791
3792 // vl = AVL if AVL ≤ VLMAX
3793 if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
3794 return IC.replaceInstUsesWith(II, AVL);
3795
3796 // vl ≤ AVL
3797 VLRange = VLRange.umin(AVLRange.getUnsignedMax());
3798
3799 // vl > 0 if AVL > 0
3801 VLRange = VLRange.umax(APInt(BitWidth, 1));
3802
3803 // vl = VLMAX if AVL ≥ (2 * VLMAX)
3804 ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
3805 if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
3806 VLRange = VLRange.intersectWith(VLMAXRange);
3807
3808 // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
3809 if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
3810 VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
3811 APInt(BitWidth, 2),
3813 }
3814
3815 ConstantRange OldRange =
3816 II.getRange().value_or(ConstantRange::getFull(BitWidth));
3817 ConstantRange NewRange = VLRange.intersectWith(OldRange);
3818 if (NewRange != OldRange) {
3819 II.addRangeRetAttr(NewRange);
3820 return &II;
3821 }
3822 return {};
3823 }
3824
3825 // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
3826 // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
3827 // creating redundant masks.
3828 const DataLayout &DL = IC.getDataLayout();
3829 if (II.user_empty())
3830 return {};
3831 auto *TargetVecTy = dyn_cast<ScalableVectorType>(II.user_back()->getType());
3832 if (!TargetVecTy)
3833 return {};
3834 const APInt *Scalar;
3835 uint64_t VL;
3837 m_Poison(), m_APInt(Scalar), m_ConstantInt(VL))) ||
3838 !all_of(II.users(), [TargetVecTy](User *U) {
3839 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3840 }))
3841 return {};
3842 auto *SourceVecTy = cast<ScalableVectorType>(II.getType());
3843 unsigned TargetEltBW = DL.getTypeSizeInBits(TargetVecTy->getElementType());
3844 unsigned SourceEltBW = DL.getTypeSizeInBits(SourceVecTy->getElementType());
3845 if (TargetEltBW % SourceEltBW)
3846 return {};
3847 unsigned TargetScale = TargetEltBW / SourceEltBW;
3848 if (VL % TargetScale || TargetScale == 1)
3849 return {};
3850 Type *VLTy = II.getOperand(2)->getType();
3851 ElementCount SourceEC = SourceVecTy->getElementCount();
3852 unsigned NewEltBW = SourceEltBW * TargetScale;
3853 if (!SourceEC.isKnownMultipleOf(TargetScale) ||
3854 !DL.fitsInLegalInteger(NewEltBW))
3855 return {};
3856 auto *NewEltTy = IntegerType::get(II.getContext(), NewEltBW);
3857 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, NewEltTy)))
3858 return {};
3859 ElementCount NewEC = SourceEC.divideCoefficientBy(TargetScale);
3860 Type *RetTy = VectorType::get(NewEltTy, NewEC);
3861 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3862 "Lossless bitcast between types expected");
3863 APInt NewScalar = APInt::getSplat(NewEltBW, *Scalar);
3864 return IC.replaceInstUsesWith(
3865 II,
3868 RetTy, Intrinsic::riscv_vmv_v_x,
3869 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3870 ConstantInt::get(VLTy, VL / TargetScale)}),
3871 SourceVecTy));
3872}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableOrLikeSelectOpt("enable-aarch64-or-like-select", cl::init(true), cl::Hidden)
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
static InstructionCost costShuffleViaVRegSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, std::optional< unsigned > VLen, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Try to perform better estimation of the permutation.
static InstructionCost costShuffleViaSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Attempt to approximate the cost of a shuffle which will require splitting during legalization.
static bool isRepeatedConcatMask(ArrayRef< int > Mask, int &SubVectorSize)
static unsigned isM1OrSmaller(MVT VT)
static cl::opt< bool > EnableOrLikeSelectOpt("enable-riscv-or-like-select", cl::init(true), cl::Hidden)
static cl::opt< unsigned > SLPMaxVF("riscv-v-slp-max-vf", cl::desc("Overrides result used for getMaximumVF query which is used " "exclusively by SLP vectorizer."), cl::Hidden)
static cl::opt< unsigned > RVVRegisterWidthLMUL("riscv-v-register-bit-width-lmul", cl::desc("The LMUL to use for getRegisterBitWidth queries. Affects LMUL used " "by autovectorized code. Fractional LMULs are not supported."), cl::init(2), cl::Hidden)
static cl::opt< unsigned > RVVMinTripCount("riscv-v-min-trip-count", cl::desc("Set the lower bound of a trip count to decide on " "vectorization while tail-folding."), cl::init(5), cl::Hidden)
static InstructionCost getIntImmCostImpl(const DataLayout &DL, const RISCVSubtarget *ST, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, bool FreeZeroes)
static VectorType * getVRGatherIndexType(MVT DataVT, const RISCVSubtarget &ST, LLVMContext &C)
static const CostTblEntry VectorIntrinsicCostTable[]
static bool canUseShiftPair(Instruction *Inst, const APInt &Imm)
static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm)
This file defines a TargetTransformInfoImplBase conforming object specific to the RISC-V target machi...
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:647
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static bool isIntPredicate(Predicate P)
Definition InstrTypes.h:839
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noNaNs() const
Definition FMF.h:65
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2253
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:889
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual const DataLayout & getDataLayout() const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual bool isLoweredToCall(const Function *F) const
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:343
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:346
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:310
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
LLVM_ABI bool isScalableTy(SmallPtrSetImpl< const Type * > &Visited) const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:306
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:286
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
user_iterator user_begin()
Definition Value.h:402
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:439
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2799
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:24
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ BinaryOp
One of the operands is a binary op.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
Definition STLExtras.h:1818
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1970
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1885
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2146
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).