LLVM 24.0.0git
X86TargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- X86TargetTransformInfo.cpp - X86 specific TTI pass ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements a TargetTransformInfo analysis pass specific to the
10/// X86 target machine. It uses the target's detailed information to provide
11/// more precise answers to certain TTI queries, while letting the target
12/// independent and default TTI implementations handle the rest.
13///
14//===----------------------------------------------------------------------===//
15/// About Cost Model numbers used below it's necessary to say the following:
16/// the numbers correspond to some "generic" X86 CPU instead of usage of a
17/// specific CPU model. Usually the numbers correspond to the CPU where the
18/// feature first appeared. For example, if we do Subtarget.hasSSE42() in
19/// the lookups below the cost is based on Nehalem as that was the first CPU
20/// to support that feature level and thus has most likely the worst case cost,
21/// although we may discard an outlying worst cost from one CPU (e.g. Atom).
22///
23/// Some examples of other technologies/CPUs:
24/// SSE 3 - Pentium4 / Athlon64
25/// SSE 4.1 - Penryn
26/// SSE 4.2 - Nehalem / Silvermont
27/// AVX - Sandy Bridge / Jaguar / Bulldozer
28/// AVX2 - Haswell / Ryzen
29/// AVX-512 - Xeon Phi / Skylake
30///
31/// And some examples of instruction target dependent costs (latency)
32/// divss sqrtss rsqrtss
33/// AMD K7 11-16 19 3
34/// Piledriver 9-24 13-15 5
35/// Jaguar 14 16 2
36/// Pentium II,III 18 30 2
37/// Nehalem 7-14 7-18 3
38/// Haswell 10-13 11 5
39///
40/// Interpreting the 4 TargetCostKind types:
41/// TCK_RecipThroughput and TCK_Latency should try to match the worst case
42/// values reported by the CPU scheduler models (and llvm-mca).
43/// TCK_CodeSize should match the instruction count (e.g. divss = 1), NOT the
44/// actual encoding size of the instruction.
45/// TCK_SizeAndLatency should match the worst case micro-op counts reported by
46/// by the CPU scheduler models (and llvm-mca), to ensure that they are
47/// compatible with the MicroOpBufferSize and LoopMicroOpBufferSize values which are
48/// often used as the cost thresholds where TCK_SizeAndLatency is requested.
49//===----------------------------------------------------------------------===//
50
60#include <optional>
61
62using namespace llvm;
63
64#define DEBUG_TYPE "x86tti"
65
66//===----------------------------------------------------------------------===//
67//
68// X86 cost model.
69//
70//===----------------------------------------------------------------------===//
71
72// Helper struct to store/access costs for each cost kind.
73// TODO: Move this to allow other targets to use it?
75 unsigned RecipThroughputCost = ~0U;
76 unsigned LatencyCost = ~0U;
77 unsigned CodeSizeCost = ~0U;
78 unsigned SizeAndLatencyCost = ~0U;
79
80 std::optional<unsigned>
82 unsigned Cost = ~0U;
83 switch (Kind) {
86 break;
89 break;
92 break;
95 break;
96 }
97 if (Cost == ~0U)
98 return std::nullopt;
99 return Cost;
100 }
101};
104
106X86TTIImpl::getPopcntSupport(unsigned TyWidth) const {
107 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
108 // TODO: Currently the __builtin_popcount() implementation using SSE3
109 // instructions is inefficient. Once the problem is fixed, we should
110 // call ST->hasSSE3() instead of ST->hasPOPCNT().
111 return ST->hasPOPCNT() ? TTI::PSK_FastHardware : TTI::PSK_Software;
112}
113
114std::optional<unsigned> X86TTIImpl::getCacheSize(
116 switch (Level) {
118 // - Penryn
119 // - Nehalem
120 // - Westmere
121 // - Sandy Bridge
122 // - Ivy Bridge
123 // - Haswell
124 // - Broadwell
125 // - Skylake
126 // - Kabylake
127 return 32 * 1024; // 32 KiB
129 // - Penryn
130 // - Nehalem
131 // - Westmere
132 // - Sandy Bridge
133 // - Ivy Bridge
134 // - Haswell
135 // - Broadwell
136 // - Skylake
137 // - Kabylake
138 return 256 * 1024; // 256 KiB
139 }
140
141 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
142}
143
144std::optional<unsigned> X86TTIImpl::getCacheAssociativity(
146 // - Penryn
147 // - Nehalem
148 // - Westmere
149 // - Sandy Bridge
150 // - Ivy Bridge
151 // - Haswell
152 // - Broadwell
153 // - Skylake
154 // - Kabylake
155 switch (Level) {
157 [[fallthrough]];
159 return 8;
160 }
161
162 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
163}
164
166
168 return Vector ? VectorClass
169 : Ty && Ty->isFloatingPointTy() ? ScalarFPClass
170 : GPRClass;
171}
172
173unsigned X86TTIImpl::getNumberOfRegisters(unsigned ClassID) const {
174 if (ClassID == VectorClass && !ST->hasSSE1())
175 return 0;
176
177 if (!ST->is64Bit())
178 return 8;
179
180 if ((ClassID == GPRClass && ST->hasEGPR()) ||
181 (ClassID != GPRClass && ST->hasAVX512()))
182 return 32;
183
184 return 16;
185}
186
188 if (!ST->hasCF())
189 return false;
190 if (!Ty)
191 return true;
192 // Conditional faulting is supported by CFCMOV, which only accepts
193 // 16/32/64-bit operands.
194 // TODO: Support f32/f64 with VMOVSS/VMOVSD with zero mask when it's
195 // profitable.
196 auto *VTy = dyn_cast<FixedVectorType>(Ty);
197 if (!Ty->isIntegerTy() && (!VTy || VTy->getNumElements() != 1))
198 return false;
199 auto *ScalarTy = Ty->getScalarType();
200 switch (cast<IntegerType>(ScalarTy)->getBitWidth()) {
201 default:
202 return false;
203 case 16:
204 case 32:
205 case 64:
206 return true;
207 }
208}
209
212 unsigned PreferVectorWidth = ST->getPreferVectorWidth();
213 switch (K) {
215 return TypeSize::getFixed(ST->is64Bit() ? 64 : 32);
217 if (ST->hasAVX512() && PreferVectorWidth >= 512)
218 return TypeSize::getFixed(512);
219 if (ST->hasAVX() && PreferVectorWidth >= 256)
220 return TypeSize::getFixed(256);
221 if (ST->hasSSE1() && PreferVectorWidth >= 128)
222 return TypeSize::getFixed(128);
223 return TypeSize::getFixed(0);
225 return TypeSize::getScalable(0);
226 }
227
228 llvm_unreachable("Unsupported register kind");
229}
230
235
237 bool HasUnorderedReductions) const {
238 // If the loop will not be vectorized, don't interleave the loop.
239 // Let regular unroll to unroll the loop, which saves the overflow
240 // check and memory check cost.
241 if (VF.isScalar())
242 return 1;
243
244 if (ST->isAtom())
245 return 1;
246
247 // Sandybridge and Haswell have multiple execution ports and pipelined
248 // vector units.
249 if (ST->hasAVX())
250 return 4;
251
252 return 2;
253}
254
256 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
258 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
259
260 // vXi8 multiplications are always promoted to vXi16.
261 // Sub-128-bit types can be extended/packed more efficiently.
262 if (Opcode == Instruction::Mul && Ty->isVectorTy() &&
263 Ty->getPrimitiveSizeInBits() <= 64 && Ty->getScalarSizeInBits() == 8) {
264 Type *WideVecTy =
266 return getCastInstrCost(Instruction::ZExt, WideVecTy, Ty,
268 CostKind) +
269 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
271 CostKind) +
272 getArithmeticInstrCost(Opcode, WideVecTy, CostKind, Op1Info, Op2Info);
273 }
274
275 // Legalize the type.
276 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
277
278 int ISD = TLI->InstructionOpcodeToISD(Opcode);
279 assert(ISD && "Invalid opcode");
280
281 if (ISD == ISD::MUL && Args.size() == 2 && LT.second.isVector() &&
282 (LT.second.getScalarType() == MVT::i32 ||
283 LT.second.getScalarType() == MVT::i64)) {
284 // Check if the operands can be represented as a smaller datatype.
285 bool Op1Signed = false, Op2Signed = false;
286 unsigned Op1MinSize = BaseT::minRequiredElementSize(Args[0], Op1Signed);
287 unsigned Op2MinSize = BaseT::minRequiredElementSize(Args[1], Op2Signed);
288 unsigned OpMinSize = std::max(Op1MinSize, Op2MinSize);
289 bool SignedMode = Op1Signed || Op2Signed;
290
291 // If both vXi32 are representable as i15 and at least one is constant,
292 // zero-extended, or sign-extended from vXi16 (or less pre-SSE41) then we
293 // can treat this as PMADDWD which has the same costs as a vXi16 multiply.
294 if (OpMinSize <= 15 && !ST->isPMADDWDSlow() &&
295 LT.second.getScalarType() == MVT::i32) {
296 bool Op1Constant =
297 isa<ConstantDataVector>(Args[0]) || isa<ConstantVector>(Args[0]);
298 bool Op2Constant =
299 isa<ConstantDataVector>(Args[1]) || isa<ConstantVector>(Args[1]);
300 bool Op1Sext = isa<SExtInst>(Args[0]) &&
301 (Op1MinSize == 15 || (Op1MinSize < 15 && !ST->hasSSE41()));
302 bool Op2Sext = isa<SExtInst>(Args[1]) &&
303 (Op2MinSize == 15 || (Op2MinSize < 15 && !ST->hasSSE41()));
304
305 bool IsZeroExtended = !Op1Signed || !Op2Signed;
306 bool IsConstant = Op1Constant || Op2Constant;
307 bool IsSext = Op1Sext || Op2Sext;
308 if (IsConstant || IsZeroExtended || IsSext)
309 LT.second =
310 MVT::getVectorVT(MVT::i16, 2 * LT.second.getVectorNumElements());
311 }
312
313 // Check if the vXi32 operands can be shrunk into a smaller datatype.
314 // This should match the codegen from reduceVMULWidth.
315 // TODO: Make this generic (!ST->SSE41 || ST->isPMULLDSlow()).
316 if (ST->useSLMArithCosts() && LT.second == MVT::v4i32) {
317 if (OpMinSize <= 7)
318 return LT.first * 3; // pmullw/sext
319 if (!SignedMode && OpMinSize <= 8)
320 return LT.first * 3; // pmullw/zext
321 if (OpMinSize <= 15)
322 return LT.first * 5; // pmullw/pmulhw/pshuf
323 if (!SignedMode && OpMinSize <= 16)
324 return LT.first * 5; // pmullw/pmulhw/pshuf
325 }
326
327 // If both vXi64 are representable as (unsigned) i32, then we can perform
328 // the multiple with a single PMULUDQ instruction.
329 // TODO: Add (SSE41+) PMULDQ handling for signed extensions.
330 if (!SignedMode && OpMinSize <= 32 && LT.second.getScalarType() == MVT::i64)
331 ISD = X86ISD::PMULUDQ;
332 }
333
334 // Vector multiply by pow2 will be simplified to shifts.
335 // Vector multiply by -pow2 will be simplified to shifts/negates.
336 if (ISD == ISD::MUL && Op2Info.isConstant() &&
337 (Op2Info.isPowerOf2() || Op2Info.isNegatedPowerOf2())) {
339 getArithmeticInstrCost(Instruction::Shl, Ty, CostKind,
340 Op1Info.getNoProps(), Op2Info.getNoProps());
341 if (Op2Info.isNegatedPowerOf2())
342 Cost += getArithmeticInstrCost(Instruction::Sub, Ty, CostKind);
343 return Cost;
344 }
345
346 // On X86, vector signed division by constants power-of-two are
347 // normally expanded to the sequence SRA + SRL + ADD + SRA.
348 // The OperandValue properties may not be the same as that of the previous
349 // operation; conservatively assume OP_None.
350 if ((ISD == ISD::SDIV || ISD == ISD::SREM) &&
351 Op2Info.isConstant() && Op2Info.isPowerOf2()) {
353 2 * getArithmeticInstrCost(Instruction::AShr, Ty, CostKind,
354 Op1Info.getNoProps(), Op2Info.getNoProps());
355 Cost += getArithmeticInstrCost(Instruction::LShr, Ty, CostKind,
356 Op1Info.getNoProps(), Op2Info.getNoProps());
357 Cost += getArithmeticInstrCost(Instruction::Add, Ty, CostKind,
358 Op1Info.getNoProps(), Op2Info.getNoProps());
359
360 if (ISD == ISD::SREM) {
361 // For SREM: (X % C) is the equivalent of (X - (X/C)*C)
362 Cost += getArithmeticInstrCost(Instruction::Mul, Ty, CostKind, Op1Info.getNoProps(),
363 Op2Info.getNoProps());
364 Cost += getArithmeticInstrCost(Instruction::Sub, Ty, CostKind, Op1Info.getNoProps(),
365 Op2Info.getNoProps());
366 }
367
368 return Cost;
369 }
370
371 // Vector unsigned division/remainder will be simplified to shifts/masks.
372 if ((ISD == ISD::UDIV || ISD == ISD::UREM) &&
373 Op2Info.isConstant() && Op2Info.isPowerOf2()) {
374 if (ISD == ISD::UDIV)
375 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind,
376 Op1Info.getNoProps(), Op2Info.getNoProps());
377 // UREM
378 return getArithmeticInstrCost(Instruction::And, Ty, CostKind,
379 Op1Info.getNoProps(), Op2Info.getNoProps());
380 }
381
382 // A scalar integer divide/remainder by a constant is not a hardware divide;
383 // it lowers to a magic-number multiply-high plus a few fixup ops. Cost it as
384 // that sequence rather than the generic single-instruction divide, so the
385 // vectorizers do not compare against an artificially cheap scalar lane. The
386 // power-of-two cases are handled above; negated powers of two are left to the
387 // generic handling.
388 if (!Ty->isVectorTy() && Op2Info.isConstant() && !Op2Info.isNegatedPowerOf2() &&
389 (ISD == ISD::UDIV || ISD == ISD::SDIV || ISD == ISD::UREM ||
390 ISD == ISD::SREM)) {
391 unsigned Cost = ISD == ISD::UREM || ISD == ISD::SREM ? 6 : 5;
393 Cost += 2;
394 return LT.first * Cost;
395 }
396
397 static const CostKindTblEntry GFNIUniformConstCostTable[] = {
398 { ISD::SHL, MVT::v16i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
399 { ISD::SRL, MVT::v16i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
400 { ISD::SRA, MVT::v16i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
401 { ISD::SHL, MVT::v32i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
402 { ISD::SRL, MVT::v32i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
403 { ISD::SRA, MVT::v32i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
404 { ISD::SHL, MVT::v64i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
405 { ISD::SRL, MVT::v64i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
406 { ISD::SRA, MVT::v64i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
407 };
408
409 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasGFNI())
410 if (const auto *Entry =
411 CostTableLookup(GFNIUniformConstCostTable, ISD, LT.second))
412 if (auto KindCost = Entry->Cost[CostKind])
413 return LT.first * *KindCost;
414
415 static const CostKindTblEntry AVX512BWUniformConstCostTable[] = {
416 { ISD::SHL, MVT::v16i8, { 1, 7, 2, 3 } }, // psllw + pand.
417 { ISD::SRL, MVT::v16i8, { 1, 7, 2, 3 } }, // psrlw + pand.
418 { ISD::SRA, MVT::v16i8, { 1, 8, 4, 5 } }, // psrlw, pand, pxor, psubb.
419 { ISD::SHL, MVT::v32i8, { 1, 8, 2, 3 } }, // psllw + pand.
420 { ISD::SRL, MVT::v32i8, { 1, 8, 2, 3 } }, // psrlw + pand.
421 { ISD::SRA, MVT::v32i8, { 1, 9, 4, 5 } }, // psrlw, pand, pxor, psubb.
422 { ISD::SHL, MVT::v64i8, { 1, 8, 2, 3 } }, // psllw + pand.
423 { ISD::SRL, MVT::v64i8, { 1, 8, 2, 3 } }, // psrlw + pand.
424 { ISD::SRA, MVT::v64i8, { 1, 9, 4, 6 } }, // psrlw, pand, pxor, psubb.
425
426 { ISD::SHL, MVT::v16i16, { 1, 1, 1, 1 } }, // psllw
427 { ISD::SRL, MVT::v16i16, { 1, 1, 1, 1 } }, // psrlw
428 { ISD::SRA, MVT::v16i16, { 1, 1, 1, 1 } }, // psrlw
429 { ISD::SHL, MVT::v32i16, { 1, 1, 1, 1 } }, // psllw
430 { ISD::SRL, MVT::v32i16, { 1, 1, 1, 1 } }, // psrlw
431 { ISD::SRA, MVT::v32i16, { 1, 1, 1, 1 } }, // psrlw
432 };
433
434 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasBWI())
435 if (const auto *Entry =
436 CostTableLookup(AVX512BWUniformConstCostTable, ISD, LT.second))
437 if (auto KindCost = Entry->Cost[CostKind])
438 return LT.first * *KindCost;
439
440 static const CostKindTblEntry AVX512DQUniformConstCostTable[] = {
441 { ISD::SDIV, MVT::v4i64, { 15 } }, // vpmullq-based MULHS sequence
442 { ISD::SREM, MVT::v4i64, { 17 } }, // vpmullq-based MULHS+mul+sub sequence
443 { ISD::SDIV, MVT::v8i64, { 15 } }, // vpmullq-based MULHS sequence
444 { ISD::SREM, MVT::v8i64, { 17 } }, // vpmullq-based MULHS+mul+sub sequence
445 // The remainder's multiply-back is a single vpmullq with DQ, just like the
446 // pmulld the vXi32 entries above rely on. Without DQ it is another
447 // vpmuludq schoolbook, so the AVX512/AVX2 tables charge more.
448 { ISD::UREM, MVT::v4i64, { 17 } }, // MULHU + vpmullq + sub sequence
449 { ISD::UREM, MVT::v8i64, { 17 } }, // MULHU + vpmullq + sub sequence
450 };
451
452 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasDQI())
453 if (const auto *Entry =
454 CostTableLookup(AVX512DQUniformConstCostTable, ISD, LT.second))
455 if (auto KindCost = Entry->Cost[CostKind])
456 return LT.first * *KindCost;
457
458 static const CostKindTblEntry AVX512UniformConstCostTable[] = {
459 { ISD::SHL, MVT::v64i8, { 2, 12, 5, 6 } }, // psllw + pand.
460 { ISD::SRL, MVT::v64i8, { 2, 12, 5, 6 } }, // psrlw + pand.
461 { ISD::SRA, MVT::v64i8, { 3, 10, 12, 12 } }, // psrlw, pand, pxor, psubb.
462
463 { ISD::SHL, MVT::v16i16, { 2, 7, 4, 4 } }, // psllw + split.
464 { ISD::SRL, MVT::v16i16, { 2, 7, 4, 4 } }, // psrlw + split.
465 { ISD::SRA, MVT::v16i16, { 2, 7, 4, 4 } }, // psraw + split.
466
467 { ISD::SHL, MVT::v8i32, { 1, 1, 1, 1 } }, // pslld
468 { ISD::SRL, MVT::v8i32, { 1, 1, 1, 1 } }, // psrld
469 { ISD::SRA, MVT::v8i32, { 1, 1, 1, 1 } }, // psrad
470 { ISD::SHL, MVT::v16i32, { 1, 1, 1, 1 } }, // pslld
471 { ISD::SRL, MVT::v16i32, { 1, 1, 1, 1 } }, // psrld
472 { ISD::SRA, MVT::v16i32, { 1, 1, 1, 1 } }, // psrad
473
474 { ISD::SRA, MVT::v2i64, { 1, 1, 1, 1 } }, // psraq
475 { ISD::SHL, MVT::v4i64, { 1, 1, 1, 1 } }, // psllq
476 { ISD::SRL, MVT::v4i64, { 1, 1, 1, 1 } }, // psrlq
477 { ISD::SRA, MVT::v4i64, { 1, 1, 1, 1 } }, // psraq
478 { ISD::SHL, MVT::v8i64, { 1, 1, 1, 1 } }, // psllq
479 { ISD::SRL, MVT::v8i64, { 1, 1, 1, 1 } }, // psrlq
480 { ISD::SRA, MVT::v8i64, { 1, 1, 1, 1 } }, // psraq
481
482 { ISD::SDIV, MVT::v16i32, { 6 } }, // pmuludq sequence
483 { ISD::SREM, MVT::v16i32, { 8 } }, // pmuludq+mul+sub sequence
484 { ISD::UDIV, MVT::v16i32, { 5 } }, // pmuludq sequence
485 { ISD::UREM, MVT::v16i32, { 7 } }, // pmuludq+mul+sub sequence
486
487 { ISD::UDIV, MVT::v8i64, { 15 } }, // pmuludq-based MULHU sequence
488 { ISD::UREM, MVT::v8i64, { 21 } }, // pmuludq-based MULHU+mul+sub sequence
489 };
490
491 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX512())
492 if (const auto *Entry =
493 CostTableLookup(AVX512UniformConstCostTable, ISD, LT.second))
494 if (auto KindCost = Entry->Cost[CostKind])
495 return LT.first * *KindCost;
496
497 static const CostKindTblEntry AVX2UniformConstCostTable[] = {
498 { ISD::SHL, MVT::v16i8, { 1, 8, 2, 3 } }, // psllw + pand.
499 { ISD::SRL, MVT::v16i8, { 1, 8, 2, 3 } }, // psrlw + pand.
500 { ISD::SRA, MVT::v16i8, { 2, 10, 5, 6 } }, // psrlw, pand, pxor, psubb.
501 { ISD::SHL, MVT::v32i8, { 2, 8, 2, 4 } }, // psllw + pand.
502 { ISD::SRL, MVT::v32i8, { 2, 8, 2, 4 } }, // psrlw + pand.
503 { ISD::SRA, MVT::v32i8, { 3, 10, 5, 9 } }, // psrlw, pand, pxor, psubb.
504
505 { ISD::SHL, MVT::v8i16, { 1, 1, 1, 1 } }, // psllw
506 { ISD::SRL, MVT::v8i16, { 1, 1, 1, 1 } }, // psrlw
507 { ISD::SRA, MVT::v8i16, { 1, 1, 1, 1 } }, // psraw
508 { ISD::SHL, MVT::v16i16,{ 2, 2, 1, 2 } }, // psllw
509 { ISD::SRL, MVT::v16i16,{ 2, 2, 1, 2 } }, // psrlw
510 { ISD::SRA, MVT::v16i16,{ 2, 2, 1, 2 } }, // psraw
511
512 { ISD::SHL, MVT::v4i32, { 1, 1, 1, 1 } }, // pslld
513 { ISD::SRL, MVT::v4i32, { 1, 1, 1, 1 } }, // psrld
514 { ISD::SRA, MVT::v4i32, { 1, 1, 1, 1 } }, // psrad
515 { ISD::SHL, MVT::v8i32, { 2, 2, 1, 2 } }, // pslld
516 { ISD::SRL, MVT::v8i32, { 2, 2, 1, 2 } }, // psrld
517 { ISD::SRA, MVT::v8i32, { 2, 2, 1, 2 } }, // psrad
518
519 { ISD::SHL, MVT::v2i64, { 1, 1, 1, 1 } }, // psllq
520 { ISD::SRL, MVT::v2i64, { 1, 1, 1, 1 } }, // psrlq
521 { ISD::SRA, MVT::v2i64, { 2, 3, 3, 3 } }, // psrad + shuffle.
522 { ISD::SHL, MVT::v4i64, { 2, 2, 1, 2 } }, // psllq
523 { ISD::SRL, MVT::v4i64, { 2, 2, 1, 2 } }, // psrlq
524 { ISD::SRA, MVT::v4i64, { 4, 4, 3, 6 } }, // psrad + shuffle + split.
525
526 { ISD::SDIV, MVT::v8i32, { 6 } }, // pmuludq sequence
527 { ISD::SREM, MVT::v8i32, { 8 } }, // pmuludq+mul+sub sequence
528 { ISD::UDIV, MVT::v8i32, { 5 } }, // pmuludq sequence
529 { ISD::UREM, MVT::v8i32, { 7 } }, // pmuludq+mul+sub sequence
530
531 { ISD::UDIV, MVT::v4i64, { 15 } }, // pmuludq-based MULHU sequence
532 { ISD::UREM, MVT::v4i64, { 21 } }, // pmuludq-based MULHU+mul+sub sequence
533 };
534
535 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX2())
536 if (const auto *Entry =
537 CostTableLookup(AVX2UniformConstCostTable, ISD, LT.second))
538 if (auto KindCost = Entry->Cost[CostKind])
539 return LT.first * *KindCost;
540
541 static const CostKindTblEntry AVXUniformConstCostTable[] = {
542 { ISD::SHL, MVT::v16i8, { 2, 7, 2, 3 } }, // psllw + pand.
543 { ISD::SRL, MVT::v16i8, { 2, 7, 2, 3 } }, // psrlw + pand.
544 { ISD::SRA, MVT::v16i8, { 3, 9, 5, 6 } }, // psrlw, pand, pxor, psubb.
545 { ISD::SHL, MVT::v32i8, { 4, 7, 7, 8 } }, // 2*(psllw + pand) + split.
546 { ISD::SRL, MVT::v32i8, { 4, 7, 7, 8 } }, // 2*(psrlw + pand) + split.
547 { ISD::SRA, MVT::v32i8, { 7, 7, 12, 13 } }, // 2*(psrlw, pand, pxor, psubb) + split.
548
549 { ISD::SHL, MVT::v8i16, { 1, 2, 1, 1 } }, // psllw.
550 { ISD::SRL, MVT::v8i16, { 1, 2, 1, 1 } }, // psrlw.
551 { ISD::SRA, MVT::v8i16, { 1, 2, 1, 1 } }, // psraw.
552 { ISD::SHL, MVT::v16i16,{ 3, 6, 4, 5 } }, // psllw + split.
553 { ISD::SRL, MVT::v16i16,{ 3, 6, 4, 5 } }, // psrlw + split.
554 { ISD::SRA, MVT::v16i16,{ 3, 6, 4, 5 } }, // psraw + split.
555
556 { ISD::SHL, MVT::v4i32, { 1, 2, 1, 1 } }, // pslld.
557 { ISD::SRL, MVT::v4i32, { 1, 2, 1, 1 } }, // psrld.
558 { ISD::SRA, MVT::v4i32, { 1, 2, 1, 1 } }, // psrad.
559 { ISD::SHL, MVT::v8i32, { 3, 6, 4, 5 } }, // pslld + split.
560 { ISD::SRL, MVT::v8i32, { 3, 6, 4, 5 } }, // psrld + split.
561 { ISD::SRA, MVT::v8i32, { 3, 6, 4, 5 } }, // psrad + split.
562
563 { ISD::SHL, MVT::v2i64, { 1, 2, 1, 1 } }, // psllq.
564 { ISD::SRL, MVT::v2i64, { 1, 2, 1, 1 } }, // psrlq.
565 { ISD::SRA, MVT::v2i64, { 2, 3, 3, 3 } }, // psrad + shuffle.
566 { ISD::SHL, MVT::v4i64, { 3, 6, 4, 5 } }, // 2 x psllq + split.
567 { ISD::SRL, MVT::v4i64, { 3, 6, 4, 5 } }, // 2 x psllq + split.
568 { ISD::SRA, MVT::v4i64, { 5, 7, 8, 9 } }, // 2 x psrad + shuffle + split.
569
570 { ISD::SDIV, MVT::v8i32, { 14 } }, // 2*pmuludq sequence + split.
571 { ISD::SREM, MVT::v8i32, { 18 } }, // 2*pmuludq+mul+sub sequence + split.
572 { ISD::UDIV, MVT::v8i32, { 12 } }, // 2*pmuludq sequence + split.
573 { ISD::UREM, MVT::v8i32, { 16 } }, // 2*pmuludq+mul+sub sequence + split.
574 };
575
576 // XOP has faster vXi8 shifts.
577 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasAVX() &&
578 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
579 if (const auto *Entry =
580 CostTableLookup(AVXUniformConstCostTable, ISD, LT.second))
581 if (auto KindCost = Entry->Cost[CostKind])
582 return LT.first * *KindCost;
583
584 static const CostKindTblEntry SSE2UniformConstCostTable[] = {
585 { ISD::SHL, MVT::v16i8, { 1, 7, 2, 3 } }, // psllw + pand.
586 { ISD::SRL, MVT::v16i8, { 1, 7, 2, 3 } }, // psrlw + pand.
587 { ISD::SRA, MVT::v16i8, { 3, 9, 5, 6 } }, // psrlw, pand, pxor, psubb.
588
589 { ISD::SHL, MVT::v8i16, { 1, 1, 1, 1 } }, // psllw.
590 { ISD::SRL, MVT::v8i16, { 1, 1, 1, 1 } }, // psrlw.
591 { ISD::SRA, MVT::v8i16, { 1, 1, 1, 1 } }, // psraw.
592
593 { ISD::SHL, MVT::v4i32, { 1, 1, 1, 1 } }, // pslld
594 { ISD::SRL, MVT::v4i32, { 1, 1, 1, 1 } }, // psrld.
595 { ISD::SRA, MVT::v4i32, { 1, 1, 1, 1 } }, // psrad.
596
597 { ISD::SHL, MVT::v2i64, { 1, 1, 1, 1 } }, // psllq.
598 { ISD::SRL, MVT::v2i64, { 1, 1, 1, 1 } }, // psrlq.
599 { ISD::SRA, MVT::v2i64, { 3, 5, 6, 6 } }, // 2 x psrad + shuffle.
600
601 { ISD::SDIV, MVT::v4i32, { 6 } }, // pmuludq sequence
602 { ISD::SREM, MVT::v4i32, { 8 } }, // pmuludq+mul+sub sequence
603 { ISD::UDIV, MVT::v4i32, { 5 } }, // pmuludq sequence
604 { ISD::UREM, MVT::v4i32, { 7 } }, // pmuludq+mul+sub sequence
605 };
606
607 // XOP has faster vXi8 shifts.
608 if (Op2Info.isUniform() && Op2Info.isConstant() && ST->hasSSE2() &&
609 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
610 if (const auto *Entry =
611 CostTableLookup(SSE2UniformConstCostTable, ISD, LT.second))
612 if (auto KindCost = Entry->Cost[CostKind])
613 return LT.first * *KindCost;
614
615 static const CostKindTblEntry AVX512BWConstCostTable[] = {
616 { ISD::SDIV, MVT::v64i8, { 14 } }, // 2*ext+2*pmulhw sequence
617 { ISD::SREM, MVT::v64i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
618 { ISD::UDIV, MVT::v64i8, { 14 } }, // 2*ext+2*pmulhw sequence
619 { ISD::UREM, MVT::v64i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
620
621 { ISD::SDIV, MVT::v32i16, { 6 } }, // vpmulhw sequence
622 { ISD::SREM, MVT::v32i16, { 8 } }, // vpmulhw+mul+sub sequence
623 { ISD::UDIV, MVT::v32i16, { 6 } }, // vpmulhuw sequence
624 { ISD::UREM, MVT::v32i16, { 8 } }, // vpmulhuw+mul+sub sequence
625 };
626
627 if (Op2Info.isConstant() && ST->hasBWI())
628 if (const auto *Entry =
629 CostTableLookup(AVX512BWConstCostTable, ISD, LT.second))
630 if (auto KindCost = Entry->Cost[CostKind])
631 return LT.first * *KindCost;
632
633 static const CostKindTblEntry AVX512DQConstCostTable[] = {
634 { ISD::SDIV, MVT::v4i64, { 19 } }, // vpmullq-based MULHS sequence
635 { ISD::SREM, MVT::v4i64, { 21 } }, // vpmullq-based MULHS+mul+sub sequence
636 { ISD::SDIV, MVT::v8i64, { 19 } }, // vpmullq-based MULHS sequence
637 { ISD::SREM, MVT::v8i64, { 21 } }, // vpmullq-based MULHS+mul+sub sequence
638 // The remainder's multiply-back is a single vpmullq with DQ, whereas the
639 // AVX512/AVX2 tables have to charge for another vpmuludq schoolbook.
640 { ISD::UREM, MVT::v4i64, { 24 } }, // MULHU + vpmullq + sub sequence
641 { ISD::UREM, MVT::v8i64, { 24 } }, // MULHU + vpmullq + sub sequence
642 };
643
644 if (Op2Info.isConstant() && ST->hasDQI())
645 if (const auto *Entry =
646 CostTableLookup(AVX512DQConstCostTable, ISD, LT.second))
647 if (auto KindCost = Entry->Cost[CostKind])
648 return LT.first * *KindCost;
649
650 static const CostKindTblEntry AVX512ConstCostTable[] = {
651 { ISD::SDIV, MVT::v64i8, { 28 } }, // 4*ext+4*pmulhw sequence
652 { ISD::SREM, MVT::v64i8, { 32 } }, // 4*ext+4*pmulhw+mul+sub sequence
653 { ISD::UDIV, MVT::v64i8, { 28 } }, // 4*ext+4*pmulhw sequence
654 { ISD::UREM, MVT::v64i8, { 32 } }, // 4*ext+4*pmulhw+mul+sub sequence
655
656 { ISD::SDIV, MVT::v32i16, { 12 } }, // 2*vpmulhw sequence
657 { ISD::SREM, MVT::v32i16, { 16 } }, // 2*vpmulhw+mul+sub sequence
658 { ISD::UDIV, MVT::v32i16, { 12 } }, // 2*vpmulhuw sequence
659 { ISD::UREM, MVT::v32i16, { 16 } }, // 2*vpmulhuw+mul+sub sequence
660
661 { ISD::SDIV, MVT::v16i32, { 15 } }, // vpmuldq sequence
662 { ISD::SREM, MVT::v16i32, { 17 } }, // vpmuldq+mul+sub sequence
663 { ISD::UDIV, MVT::v16i32, { 15 } }, // vpmuludq sequence
664 { ISD::UREM, MVT::v16i32, { 17 } }, // vpmuludq+mul+sub sequence
665
666 { ISD::UDIV, MVT::v8i64, { 22 } }, // vpmuludq-based MULHU sequence
667 { ISD::UREM, MVT::v8i64, { 28 } }, // vpmuludq-based MULHU+mul+sub sequence
668 };
669
670 if (Op2Info.isConstant() && ST->hasAVX512())
671 if (const auto *Entry =
672 CostTableLookup(AVX512ConstCostTable, ISD, LT.second))
673 if (auto KindCost = Entry->Cost[CostKind])
674 return LT.first * *KindCost;
675
676 static const CostKindTblEntry AVX2ConstCostTable[] = {
677 { ISD::SDIV, MVT::v32i8, { 14 } }, // 2*ext+2*pmulhw sequence
678 { ISD::SREM, MVT::v32i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
679 { ISD::UDIV, MVT::v32i8, { 14 } }, // 2*ext+2*pmulhw sequence
680 { ISD::UREM, MVT::v32i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
681
682 { ISD::SDIV, MVT::v16i16, { 6 } }, // vpmulhw sequence
683 { ISD::SREM, MVT::v16i16, { 8 } }, // vpmulhw+mul+sub sequence
684 { ISD::UDIV, MVT::v16i16, { 6 } }, // vpmulhuw sequence
685 { ISD::UREM, MVT::v16i16, { 8 } }, // vpmulhuw+mul+sub sequence
686
687 { ISD::SDIV, MVT::v8i32, { 15 } }, // vpmuldq sequence
688 { ISD::SREM, MVT::v8i32, { 19 } }, // vpmuldq+mul+sub sequence
689 { ISD::UDIV, MVT::v8i32, { 15 } }, // vpmuludq sequence
690 { ISD::UREM, MVT::v8i32, { 19 } }, // vpmuludq+mul+sub sequence
691
692 { ISD::UDIV, MVT::v4i64, { 22 } }, // vpmuludq-based MULHU sequence
693 { ISD::UREM, MVT::v4i64, { 28 } }, // vpmuludq-based MULHU+mul+sub sequence
694 };
695
696 if (Op2Info.isConstant() && ST->hasAVX2())
697 if (const auto *Entry = CostTableLookup(AVX2ConstCostTable, ISD, LT.second))
698 if (auto KindCost = Entry->Cost[CostKind])
699 return LT.first * *KindCost;
700
701 static const CostKindTblEntry AVXConstCostTable[] = {
702 { ISD::SDIV, MVT::v32i8, { 30 } }, // 4*ext+4*pmulhw sequence + split.
703 { ISD::SREM, MVT::v32i8, { 34 } }, // 4*ext+4*pmulhw+mul+sub sequence + split.
704 { ISD::UDIV, MVT::v32i8, { 30 } }, // 4*ext+4*pmulhw sequence + split.
705 { ISD::UREM, MVT::v32i8, { 34 } }, // 4*ext+4*pmulhw+mul+sub sequence + split.
706
707 { ISD::SDIV, MVT::v16i16, { 14 } }, // 2*pmulhw sequence + split.
708 { ISD::SREM, MVT::v16i16, { 18 } }, // 2*pmulhw+mul+sub sequence + split.
709 { ISD::UDIV, MVT::v16i16, { 14 } }, // 2*pmulhuw sequence + split.
710 { ISD::UREM, MVT::v16i16, { 18 } }, // 2*pmulhuw+mul+sub sequence + split.
711
712 { ISD::SDIV, MVT::v8i32, { 32 } }, // vpmuludq sequence
713 { ISD::SREM, MVT::v8i32, { 38 } }, // vpmuludq+mul+sub sequence
714 { ISD::UDIV, MVT::v8i32, { 32 } }, // 2*pmuludq sequence + split.
715 { ISD::UREM, MVT::v8i32, { 42 } }, // 2*pmuludq+mul+sub sequence + split.
716 };
717
718 if (Op2Info.isConstant() && ST->hasAVX())
719 if (const auto *Entry = CostTableLookup(AVXConstCostTable, ISD, LT.second))
720 if (auto KindCost = Entry->Cost[CostKind])
721 return LT.first * *KindCost;
722
723 static const CostKindTblEntry SSE41ConstCostTable[] = {
724 { ISD::SDIV, MVT::v4i32, { 15 } }, // vpmuludq sequence
725 { ISD::SREM, MVT::v4i32, { 20 } }, // vpmuludq+mul+sub sequence
726 };
727
728 if (Op2Info.isConstant() && ST->hasSSE41())
729 if (const auto *Entry =
730 CostTableLookup(SSE41ConstCostTable, ISD, LT.second))
731 if (auto KindCost = Entry->Cost[CostKind])
732 return LT.first * *KindCost;
733
734 static const CostKindTblEntry SSE2ConstCostTable[] = {
735 { ISD::SDIV, MVT::v16i8, { 14 } }, // 2*ext+2*pmulhw sequence
736 { ISD::SREM, MVT::v16i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
737 { ISD::UDIV, MVT::v16i8, { 14 } }, // 2*ext+2*pmulhw sequence
738 { ISD::UREM, MVT::v16i8, { 16 } }, // 2*ext+2*pmulhw+mul+sub sequence
739
740 { ISD::SDIV, MVT::v8i16, { 6 } }, // pmulhw sequence
741 { ISD::SREM, MVT::v8i16, { 8 } }, // pmulhw+mul+sub sequence
742 { ISD::UDIV, MVT::v8i16, { 6 } }, // pmulhuw sequence
743 { ISD::UREM, MVT::v8i16, { 8 } }, // pmulhuw+mul+sub sequence
744
745 { ISD::SDIV, MVT::v4i32, { 19 } }, // pmuludq sequence
746 { ISD::SREM, MVT::v4i32, { 24 } }, // pmuludq+mul+sub sequence
747 { ISD::UDIV, MVT::v4i32, { 15 } }, // pmuludq sequence
748 { ISD::UREM, MVT::v4i32, { 20 } }, // pmuludq+mul+sub sequence
749 };
750
751 if (Op2Info.isConstant() && ST->hasSSE2())
752 if (const auto *Entry = CostTableLookup(SSE2ConstCostTable, ISD, LT.second))
753 if (auto KindCost = Entry->Cost[CostKind])
754 return LT.first * *KindCost;
755
756 // rem matches div because the divider returns the remainder for free.
757 static const CostKindTblEntry ScalarVarDivCostTable[] = {
758 { ISD::SDIV, MVT::i8, { 15, 20, 2, 4 } },
759 { ISD::UDIV, MVT::i8, { 15, 20, 2, 4 } },
760 { ISD::SREM, MVT::i8, { 15, 20, 2, 4 } },
761 { ISD::UREM, MVT::i8, { 15, 20, 2, 4 } },
762 { ISD::SDIV, MVT::i16, { 17, 20, 2, 4 } },
763 { ISD::UDIV, MVT::i16, { 17, 20, 2, 4 } },
764 { ISD::SREM, MVT::i16, { 17, 20, 2, 4 } },
765 { ISD::UREM, MVT::i16, { 17, 20, 2, 4 } },
766 { ISD::SDIV, MVT::i32, { 25, 22, 2, 4 } },
767 { ISD::UDIV, MVT::i32, { 25, 22, 2, 4 } },
768 { ISD::SREM, MVT::i32, { 25, 22, 2, 4 } },
769 { ISD::UREM, MVT::i32, { 25, 22, 2, 4 } },
770 { ISD::SDIV, MVT::i64, { 41, 24, 2, 4 } },
771 { ISD::UDIV, MVT::i64, { 41, 24, 2, 4 } },
772 { ISD::SREM, MVT::i64, { 41, 24, 2, 4 } },
773 { ISD::UREM, MVT::i64, { 41, 24, 2, 4 } },
774 };
775
776 if (!LT.second.isVector() && !Op2Info.isConstant())
777 if (const auto *Entry =
778 CostTableLookup(ScalarVarDivCostTable, ISD, LT.second))
779 if (auto KindCost = Entry->Cost[CostKind])
780 return LT.first * *KindCost;
781
782 // Variable divisors lower through a float divide. strictfp needs SAE
783 // rounding which is 512-bit only.
784 bool IsStrictFP =
785 CtxI && CtxI->getFunction()->hasFnAttribute(Attribute::StrictFP);
786 bool IsDivRem = ISD == ISD::UDIV || ISD == ISD::SDIV || ISD == ISD::UREM ||
787 ISD == ISD::SREM;
788 bool VarDivToFP = IsDivRem && !Op2Info.isConstant() &&
789 (!IsStrictFP || ST->useAVX512Regs());
790
791 // i64 needs the qq converts, which are AVX512DQ only. Two tables because the
792 // lowering picks by operand value and not by type.
793 static const CostKindTblEntry AVX512DQExactVarDivCostTable[] = {
794 { ISD::UDIV, MVT::v2i64, { 5 } }, // cvt+divpd sequence
795 { ISD::SDIV, MVT::v2i64, { 5 } },
796 { ISD::UREM, MVT::v2i64, { 5 } },
797 { ISD::SREM, MVT::v2i64, { 5 } },
798 { ISD::UDIV, MVT::v4i64, { 8 } },
799 { ISD::SDIV, MVT::v4i64, { 8 } },
800 { ISD::UREM, MVT::v4i64, { 8 } },
801 { ISD::SREM, MVT::v4i64, { 8 } },
802 { ISD::UDIV, MVT::v8i64, { 16 } },
803 { ISD::SDIV, MVT::v8i64, { 16 } },
804 { ISD::UREM, MVT::v8i64, { 16 } },
805 { ISD::SREM, MVT::v8i64, { 16 } },
806 };
807
808 static const CostKindTblEntry AVX512DQVarDivCostTable[] = {
809 { ISD::UDIV, MVT::v2i64, { 16 } },
810 { ISD::SDIV, MVT::v2i64, { 16 } },
811 { ISD::UREM, MVT::v2i64, { 16 } },
812 { ISD::SREM, MVT::v2i64, { 16 } },
813 { ISD::UDIV, MVT::v4i64, { 16 } },
814 { ISD::SDIV, MVT::v4i64, { 16 } },
815 { ISD::UREM, MVT::v4i64, { 16 } },
816 { ISD::SREM, MVT::v4i64, { 16 } },
817 { ISD::UDIV, MVT::v8i64, { 16 } },
818 { ISD::SDIV, MVT::v8i64, { 18 } },
819 { ISD::UREM, MVT::v8i64, { 16 } },
820 { ISD::SREM, MVT::v8i64, { 18 } },
821 };
822
823 // The DAG combine picks between the two sequences with these same two
824 // queries, so the cost cannot disagree with what codegen emits.
825 bool IsSignedDiv = ISD == ISD::SDIV || ISD == ISD::SREM;
826 auto OperandsFit = [&](unsigned Mantissa) {
827 if (Args.size() != 2 || !CtxI)
828 return false;
829 unsigned EltBits = LT.second.getScalarSizeInBits();
830 const DataLayout &DL = CtxI->getDataLayout();
831 auto Fits = [&](const Value *V) {
832 if (IsSignedDiv)
833 return ComputeNumSignBits(V, DL, /*AC=*/nullptr, CtxI) + Mantissa >
834 EltBits;
835 return computeKnownBits(V, DL, /*AC=*/nullptr, CtxI)
836 .countMaxActiveBits() <= Mantissa;
837 };
838 return Fits(Args[0]) && Fits(Args[1]);
839 };
840
841 // An i32 divide goes through f32 when both operands fit in 24 bits, and
842 // through f64 at twice the vector width when they do not.
843 bool ExactI32 =
844 VarDivToFP && LT.second.getScalarType() == MVT::i32 &&
846
847 if (VarDivToFP && ST->hasDQI() && ST->useAVX512Regs() &&
848 LT.second.getScalarType() == MVT::i64) {
849 bool ExactFPDiv =
851
852 // Only the reciprocal chain multiplies, so only it is vpmullq gated.
853 bool SlowMultiply =
854 !ExactFPDiv && LT.second == MVT::v2i64 && ST->isPMULLQSlow();
855
856 if (!SlowMultiply) {
858 ExactFPDiv ? AVX512DQExactVarDivCostTable : AVX512DQVarDivCostTable;
859 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
860 if (auto KindCost = Entry->Cost[CostKind])
861 return LT.first * *KindCost;
862 }
863 }
864
865 static const CostKindTblEntry AVX512BWVarDivCostTable[] = {
866 { ISD::UDIV, MVT::v16i8, { 10 } }, // unpack+cvt+divps sequence
867 { ISD::SDIV, MVT::v16i8, { 10 } },
868 { ISD::UREM, MVT::v16i8, { 10 } },
869 { ISD::SREM, MVT::v16i8, { 10 } },
870 { ISD::UDIV, MVT::v32i8, { 20 } },
871 { ISD::SDIV, MVT::v32i8, { 20 } },
872 { ISD::UREM, MVT::v32i8, { 20 } },
873 { ISD::SREM, MVT::v32i8, { 20 } },
874 { ISD::UDIV, MVT::v64i8, { 40 } },
875 { ISD::SDIV, MVT::v64i8, { 40 } },
876 { ISD::UREM, MVT::v64i8, { 40 } },
877 { ISD::SREM, MVT::v64i8, { 40 } },
878 { ISD::UDIV, MVT::v8i16, { 5 } },
879 { ISD::SDIV, MVT::v8i16, { 5 } },
880 { ISD::UREM, MVT::v8i16, { 5 } },
881 { ISD::SREM, MVT::v8i16, { 5 } },
882 { ISD::UDIV, MVT::v16i16, { 10 } },
883 { ISD::SDIV, MVT::v16i16, { 10 } },
884 { ISD::UREM, MVT::v16i16, { 10 } },
885 { ISD::SREM, MVT::v16i16, { 10 } },
886 { ISD::UDIV, MVT::v32i16, { 20 } },
887 { ISD::SDIV, MVT::v32i16, { 20 } },
888 { ISD::UREM, MVT::v32i16, { 20 } },
889 { ISD::SREM, MVT::v32i16, { 20 } },
890 { ISD::UDIV, MVT::v4i32, { 8 } }, // cvt+divpd sequence
891 { ISD::SDIV, MVT::v4i32, { 8 } },
892 { ISD::UREM, MVT::v4i32, { 8 } },
893 { ISD::SREM, MVT::v4i32, { 8 } },
894 { ISD::UDIV, MVT::v8i32, { 16 } },
895 { ISD::SDIV, MVT::v8i32, { 16 } },
896 { ISD::UREM, MVT::v8i32, { 16 } },
897 { ISD::SREM, MVT::v8i32, { 16 } },
898 { ISD::UDIV, MVT::v16i32, { 32 } },
899 { ISD::SDIV, MVT::v16i32, { 32 } },
900 { ISD::UREM, MVT::v16i32, { 32 } },
901 { ISD::SREM, MVT::v16i32, { 32 } },
902 };
903
904 static const CostKindTblEntry AVX512BWExactVarDivCostTable[] = {
905 { ISD::UDIV, MVT::v4i32, { 3 } }, // cvt+divps sequence
906 { ISD::SDIV, MVT::v4i32, { 3 } },
907 { ISD::UREM, MVT::v4i32, { 3 } },
908 { ISD::SREM, MVT::v4i32, { 3 } },
909 { ISD::UDIV, MVT::v8i32, { 5 } },
910 { ISD::SDIV, MVT::v8i32, { 5 } },
911 { ISD::UREM, MVT::v8i32, { 5 } },
912 { ISD::SREM, MVT::v8i32, { 5 } },
913 { ISD::UDIV, MVT::v16i32, { 10 } },
914 { ISD::SDIV, MVT::v16i32, { 10 } },
915 { ISD::UREM, MVT::v16i32, { 10 } },
916 { ISD::SREM, MVT::v16i32, { 10 } },
917 };
918
919 if (VarDivToFP && ST->hasBWI()) {
920 ArrayRef<CostKindTblEntry> Tbl = AVX512BWVarDivCostTable;
921 if (ExactI32)
922 Tbl = AVX512BWExactVarDivCostTable;
923 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
924 if (auto KindCost = Entry->Cost[CostKind])
925 return LT.first * *KindCost;
926 }
927
928 static const CostKindTblEntry AVX512VarDivCostTable[] = {
929 { ISD::UDIV, MVT::v16i8, { 14 } }, // unpack+cvt+divps sequence
930 { ISD::SDIV, MVT::v16i8, { 14 } },
931 { ISD::UREM, MVT::v16i8, { 14 } },
932 { ISD::SREM, MVT::v16i8, { 14 } },
933 { ISD::UDIV, MVT::v32i8, { 28 } },
934 { ISD::SDIV, MVT::v32i8, { 28 } },
935 { ISD::UREM, MVT::v32i8, { 28 } },
936 { ISD::SREM, MVT::v32i8, { 28 } },
937 { ISD::UDIV, MVT::v64i8, { 56 } },
938 { ISD::SDIV, MVT::v64i8, { 56 } },
939 { ISD::UREM, MVT::v64i8, { 56 } },
940 { ISD::SREM, MVT::v64i8, { 56 } },
941 { ISD::UDIV, MVT::v8i16, { 14 } },
942 { ISD::SDIV, MVT::v8i16, { 14 } },
943 { ISD::UREM, MVT::v8i16, { 14 } },
944 { ISD::SREM, MVT::v8i16, { 14 } },
945 { ISD::UDIV, MVT::v16i16, { 14 } },
946 { ISD::SDIV, MVT::v16i16, { 14 } },
947 { ISD::UREM, MVT::v16i16, { 14 } },
948 { ISD::SREM, MVT::v16i16, { 14 } },
949 { ISD::UDIV, MVT::v32i16, { 28 } },
950 { ISD::SDIV, MVT::v32i16, { 28 } },
951 { ISD::UREM, MVT::v32i16, { 28 } },
952 { ISD::SREM, MVT::v32i16, { 28 } },
953 { ISD::UDIV, MVT::v4i32, { 28 } }, // cvt+divpd sequence
954 { ISD::SDIV, MVT::v4i32, { 28 } },
955 { ISD::UREM, MVT::v4i32, { 28 } },
956 { ISD::SREM, MVT::v4i32, { 28 } },
957 { ISD::UDIV, MVT::v8i32, { 28 } },
958 { ISD::SDIV, MVT::v8i32, { 28 } },
959 { ISD::UREM, MVT::v8i32, { 28 } },
960 { ISD::SREM, MVT::v8i32, { 28 } },
961 { ISD::UDIV, MVT::v16i32, { 56 } },
962 { ISD::SDIV, MVT::v16i32, { 56 } },
963 { ISD::UREM, MVT::v16i32, { 56 } },
964 { ISD::SREM, MVT::v16i32, { 56 } },
965 };
966
967 static const CostKindTblEntry AVX512ExactVarDivCostTable[] = {
968 { ISD::UDIV, MVT::v4i32, { 7 } }, // cvt+divps sequence
969 { ISD::SDIV, MVT::v4i32, { 7 } },
970 { ISD::UREM, MVT::v4i32, { 7 } },
971 { ISD::SREM, MVT::v4i32, { 7 } },
972 { ISD::UDIV, MVT::v8i32, { 14 } },
973 { ISD::SDIV, MVT::v8i32, { 14 } },
974 { ISD::UREM, MVT::v8i32, { 14 } },
975 { ISD::SREM, MVT::v8i32, { 14 } },
976 { ISD::UDIV, MVT::v16i32, { 14 } },
977 { ISD::SDIV, MVT::v16i32, { 14 } },
978 { ISD::UREM, MVT::v16i32, { 14 } },
979 { ISD::SREM, MVT::v16i32, { 14 } },
980 };
981
982 if (VarDivToFP && ST->hasAVX512()) {
983 ArrayRef<CostKindTblEntry> Tbl = AVX512VarDivCostTable;
984 if (ExactI32)
985 Tbl = AVX512ExactVarDivCostTable;
986 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
987 if (auto KindCost = Entry->Cost[CostKind])
988 return LT.first * *KindCost;
989 }
990
991 static const CostKindTblEntry AVX2VarDivCostTable[] = {
992 { ISD::UDIV, MVT::v16i8, { 28 } }, // unpack+cvt+divps sequence
993 { ISD::SDIV, MVT::v16i8, { 28 } },
994 { ISD::UREM, MVT::v16i8, { 28 } },
995 { ISD::SREM, MVT::v16i8, { 28 } },
996 { ISD::UDIV, MVT::v32i8, { 56 } },
997 { ISD::SDIV, MVT::v32i8, { 56 } },
998 { ISD::UREM, MVT::v32i8, { 56 } },
999 { ISD::SREM, MVT::v32i8, { 56 } },
1000 { ISD::UDIV, MVT::v8i16, { 14 } },
1001 { ISD::SDIV, MVT::v8i16, { 14 } },
1002 { ISD::UREM, MVT::v8i16, { 14 } },
1003 { ISD::SREM, MVT::v8i16, { 14 } },
1004 { ISD::UDIV, MVT::v16i16, { 28 } },
1005 { ISD::SDIV, MVT::v16i16, { 28 } },
1006 { ISD::UREM, MVT::v16i16, { 28 } },
1007 { ISD::SREM, MVT::v16i16, { 28 } },
1008 { ISD::UDIV, MVT::v4i32, { 28 } }, // cvt+divpd sequence
1009 { ISD::SDIV, MVT::v4i32, { 28 } },
1010 { ISD::UREM, MVT::v4i32, { 28 } },
1011 { ISD::SREM, MVT::v4i32, { 28 } },
1012 { ISD::UDIV, MVT::v8i32, { 56 } },
1013 { ISD::SDIV, MVT::v8i32, { 56 } },
1014 { ISD::UREM, MVT::v8i32, { 56 } },
1015 { ISD::SREM, MVT::v8i32, { 56 } },
1016 };
1017
1018 static const CostKindTblEntry AVX2ExactVarDivCostTable[] = {
1019 { ISD::UDIV, MVT::v4i32, { 9 } }, // cvt+divps sequence
1020 { ISD::SDIV, MVT::v4i32, { 8 } },
1021 { ISD::UREM, MVT::v4i32, { 9 } },
1022 { ISD::SREM, MVT::v4i32, { 8 } },
1023 { ISD::UDIV, MVT::v8i32, { 14 } },
1024 { ISD::SDIV, MVT::v8i32, { 14 } },
1025 { ISD::UREM, MVT::v8i32, { 14 } },
1026 { ISD::SREM, MVT::v8i32, { 14 } },
1027 };
1028
1029 if (VarDivToFP && ST->hasAVX2()) {
1030 ArrayRef<CostKindTblEntry> Tbl = AVX2VarDivCostTable;
1031 if (ExactI32)
1032 Tbl = AVX2ExactVarDivCostTable;
1033 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
1034 if (auto KindCost = Entry->Cost[CostKind])
1035 return LT.first * *KindCost;
1036 }
1037
1038 // No unsigned i32 entries below AVX2, where the u32 to f64 converts are
1039 // emulated and the fold stays off.
1040 static const CostKindTblEntry AVX1VarDivCostTable[] = {
1041 { ISD::UDIV, MVT::v16i8, { 56 } }, // unpack+cvt+divps sequence
1042 { ISD::SDIV, MVT::v16i8, { 56 } },
1043 { ISD::UREM, MVT::v16i8, { 56 } },
1044 { ISD::SREM, MVT::v16i8, { 56 } },
1045 { ISD::UDIV, MVT::v32i8, { 112 } },
1046 { ISD::SDIV, MVT::v32i8, { 112 } },
1047 { ISD::UREM, MVT::v32i8, { 112 } },
1048 { ISD::SREM, MVT::v32i8, { 112 } },
1049 { ISD::UDIV, MVT::v8i16, { 28 } },
1050 { ISD::SDIV, MVT::v8i16, { 28 } },
1051 { ISD::UREM, MVT::v8i16, { 28 } },
1052 { ISD::SREM, MVT::v8i16, { 28 } },
1053 { ISD::UDIV, MVT::v16i16, { 56 } },
1054 { ISD::SDIV, MVT::v16i16, { 56 } },
1055 { ISD::UREM, MVT::v16i16, { 56 } },
1056 { ISD::SREM, MVT::v16i16, { 56 } },
1057 { ISD::SDIV, MVT::v4i32, { 44 } }, // cvt+divpd sequence
1058 { ISD::SREM, MVT::v4i32, { 44 } },
1059 { ISD::SDIV, MVT::v8i32, { 88 } },
1060 { ISD::SREM, MVT::v8i32, { 88 } },
1061 };
1062
1063 static const CostKindTblEntry AVX1ExactVarDivCostTable[] = {
1064 { ISD::SDIV, MVT::v4i32, { 14 } }, // cvt+divps sequence
1065 { ISD::SREM, MVT::v4i32, { 14 } },
1066 { ISD::SDIV, MVT::v8i32, { 28 } },
1067 { ISD::SREM, MVT::v8i32, { 28 } },
1068 };
1069
1070 if (VarDivToFP && ST->hasAVX()) {
1071 ArrayRef<CostKindTblEntry> Tbl = AVX1VarDivCostTable;
1072 if (ExactI32)
1073 Tbl = AVX1ExactVarDivCostTable;
1074 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
1075 if (auto KindCost = Entry->Cost[CostKind])
1076 return LT.first * *KindCost;
1077 }
1078
1079 static const CostKindTblEntry SSE2VarDivCostTable[] = {
1080 { ISD::UDIV, MVT::v16i8, { 56 } }, // unpack+cvt+divps sequence
1081 { ISD::SDIV, MVT::v16i8, { 56 } },
1082 { ISD::UREM, MVT::v16i8, { 56 } },
1083 { ISD::SREM, MVT::v16i8, { 56 } },
1084 { ISD::UDIV, MVT::v8i16, { 28 } },
1085 { ISD::SDIV, MVT::v8i16, { 28 } },
1086 { ISD::UREM, MVT::v8i16, { 28 } },
1087 { ISD::SREM, MVT::v8i16, { 28 } },
1088 { ISD::SDIV, MVT::v4i32, { 44 } }, // cvt+divpd sequence
1089 { ISD::SREM, MVT::v4i32, { 44 } },
1090 };
1091
1092 static const CostKindTblEntry SSE2ExactVarDivCostTable[] = {
1093 { ISD::SDIV, MVT::v4i32, { 14 } }, // cvt+divps sequence
1094 { ISD::SREM, MVT::v4i32, { 14 } },
1095 };
1096
1097 if (VarDivToFP && ST->hasSSE2()) {
1098 ArrayRef<CostKindTblEntry> Tbl = SSE2VarDivCostTable;
1099 if (ExactI32)
1100 Tbl = SSE2ExactVarDivCostTable;
1101 if (const auto *Entry = CostTableLookup(Tbl, ISD, LT.second))
1102 if (auto KindCost = Entry->Cost[CostKind])
1103 return LT.first * *KindCost;
1104 }
1105
1106 static const CostKindTblEntry AVX512BWUniformCostTable[] = {
1107 { ISD::SHL, MVT::v16i8, { 3, 5, 5, 7 } }, // psllw + pand.
1108 { ISD::SRL, MVT::v16i8, { 3,10, 5, 8 } }, // psrlw + pand.
1109 { ISD::SRA, MVT::v16i8, { 4,12, 8,12 } }, // psrlw, pand, pxor, psubb.
1110 { ISD::SHL, MVT::v32i8, { 4, 7, 6, 8 } }, // psllw + pand.
1111 { ISD::SRL, MVT::v32i8, { 4, 8, 7, 9 } }, // psrlw + pand.
1112 { ISD::SRA, MVT::v32i8, { 5,10,10,13 } }, // psrlw, pand, pxor, psubb.
1113 { ISD::SHL, MVT::v64i8, { 4, 7, 6, 8 } }, // psllw + pand.
1114 { ISD::SRL, MVT::v64i8, { 4, 8, 7,10 } }, // psrlw + pand.
1115 { ISD::SRA, MVT::v64i8, { 5,10,10,15 } }, // psrlw, pand, pxor, psubb.
1116
1117 { ISD::SHL, MVT::v32i16, { 2, 4, 2, 3 } }, // psllw
1118 { ISD::SRL, MVT::v32i16, { 2, 4, 2, 3 } }, // psrlw
1119 { ISD::SRA, MVT::v32i16, { 2, 4, 2, 3 } }, // psrqw
1120 };
1121
1122 if (ST->hasBWI() && Op2Info.isUniform())
1123 if (const auto *Entry =
1124 CostTableLookup(AVX512BWUniformCostTable, ISD, LT.second))
1125 if (auto KindCost = Entry->Cost[CostKind])
1126 return LT.first * *KindCost;
1127
1128 static const CostKindTblEntry AVX512UniformCostTable[] = {
1129 { ISD::SHL, MVT::v32i16, { 5,10, 5, 7 } }, // psllw + split.
1130 { ISD::SRL, MVT::v32i16, { 5,10, 5, 7 } }, // psrlw + split.
1131 { ISD::SRA, MVT::v32i16, { 5,10, 5, 7 } }, // psraw + split.
1132
1133 { ISD::SHL, MVT::v16i32, { 2, 4, 2, 3 } }, // pslld
1134 { ISD::SRL, MVT::v16i32, { 2, 4, 2, 3 } }, // psrld
1135 { ISD::SRA, MVT::v16i32, { 2, 4, 2, 3 } }, // psrad
1136
1137 { ISD::SRA, MVT::v2i64, { 1, 2, 1, 2 } }, // psraq
1138 { ISD::SHL, MVT::v4i64, { 1, 4, 1, 2 } }, // psllq
1139 { ISD::SRL, MVT::v4i64, { 1, 4, 1, 2 } }, // psrlq
1140 { ISD::SRA, MVT::v4i64, { 1, 4, 1, 2 } }, // psraq
1141 { ISD::SHL, MVT::v8i64, { 1, 4, 1, 2 } }, // psllq
1142 { ISD::SRL, MVT::v8i64, { 1, 4, 1, 2 } }, // psrlq
1143 { ISD::SRA, MVT::v8i64, { 1, 4, 1, 2 } }, // psraq
1144 };
1145
1146 if (ST->hasAVX512() && Op2Info.isUniform())
1147 if (const auto *Entry =
1148 CostTableLookup(AVX512UniformCostTable, ISD, LT.second))
1149 if (auto KindCost = Entry->Cost[CostKind])
1150 return LT.first * *KindCost;
1151
1152 static const CostKindTblEntry AVX2UniformCostTable[] = {
1153 // Uniform splats are cheaper for the following instructions.
1154 { ISD::SHL, MVT::v16i8, { 3, 5, 5, 7 } }, // psllw + pand.
1155 { ISD::SRL, MVT::v16i8, { 3, 9, 5, 8 } }, // psrlw + pand.
1156 { ISD::SRA, MVT::v16i8, { 4, 5, 9,13 } }, // psrlw, pand, pxor, psubb.
1157 { ISD::SHL, MVT::v32i8, { 4, 7, 6, 8 } }, // psllw + pand.
1158 { ISD::SRL, MVT::v32i8, { 4, 8, 7, 9 } }, // psrlw + pand.
1159 { ISD::SRA, MVT::v32i8, { 6, 9,11,16 } }, // psrlw, pand, pxor, psubb.
1160
1161 { ISD::SHL, MVT::v8i16, { 1, 2, 1, 2 } }, // psllw.
1162 { ISD::SRL, MVT::v8i16, { 1, 2, 1, 2 } }, // psrlw.
1163 { ISD::SRA, MVT::v8i16, { 1, 2, 1, 2 } }, // psraw.
1164 { ISD::SHL, MVT::v16i16, { 2, 4, 2, 3 } }, // psllw.
1165 { ISD::SRL, MVT::v16i16, { 2, 4, 2, 3 } }, // psrlw.
1166 { ISD::SRA, MVT::v16i16, { 2, 4, 2, 3 } }, // psraw.
1167
1168 { ISD::SHL, MVT::v4i32, { 1, 2, 1, 2 } }, // pslld
1169 { ISD::SRL, MVT::v4i32, { 1, 2, 1, 2 } }, // psrld
1170 { ISD::SRA, MVT::v4i32, { 1, 2, 1, 2 } }, // psrad
1171 { ISD::SHL, MVT::v8i32, { 2, 4, 2, 3 } }, // pslld
1172 { ISD::SRL, MVT::v8i32, { 2, 4, 2, 3 } }, // psrld
1173 { ISD::SRA, MVT::v8i32, { 2, 4, 2, 3 } }, // psrad
1174
1175 { ISD::SHL, MVT::v2i64, { 1, 2, 1, 2 } }, // psllq
1176 { ISD::SRL, MVT::v2i64, { 1, 2, 1, 2 } }, // psrlq
1177 { ISD::SRA, MVT::v2i64, { 2, 4, 5, 7 } }, // 2 x psrad + shuffle.
1178 { ISD::SHL, MVT::v4i64, { 2, 4, 1, 2 } }, // psllq
1179 { ISD::SRL, MVT::v4i64, { 2, 4, 1, 2 } }, // psrlq
1180 { ISD::SRA, MVT::v4i64, { 4, 6, 5, 9 } }, // 2 x psrad + shuffle.
1181 };
1182
1183 if (ST->hasAVX2() && Op2Info.isUniform())
1184 if (const auto *Entry =
1185 CostTableLookup(AVX2UniformCostTable, ISD, LT.second))
1186 if (auto KindCost = Entry->Cost[CostKind])
1187 return LT.first * *KindCost;
1188
1189 static const CostKindTblEntry AVXUniformCostTable[] = {
1190 { ISD::SHL, MVT::v16i8, { 4, 4, 6, 8 } }, // psllw + pand.
1191 { ISD::SRL, MVT::v16i8, { 4, 8, 5, 8 } }, // psrlw + pand.
1192 { ISD::SRA, MVT::v16i8, { 6, 6, 9,13 } }, // psrlw, pand, pxor, psubb.
1193 { ISD::SHL, MVT::v32i8, { 7, 8,11,14 } }, // psllw + pand + split.
1194 { ISD::SRL, MVT::v32i8, { 7, 9,10,14 } }, // psrlw + pand + split.
1195 { ISD::SRA, MVT::v32i8, { 10,11,16,21 } }, // psrlw, pand, pxor, psubb + split.
1196
1197 { ISD::SHL, MVT::v8i16, { 1, 3, 1, 2 } }, // psllw.
1198 { ISD::SRL, MVT::v8i16, { 1, 3, 1, 2 } }, // psrlw.
1199 { ISD::SRA, MVT::v8i16, { 1, 3, 1, 2 } }, // psraw.
1200 { ISD::SHL, MVT::v16i16, { 3, 7, 5, 7 } }, // psllw + split.
1201 { ISD::SRL, MVT::v16i16, { 3, 7, 5, 7 } }, // psrlw + split.
1202 { ISD::SRA, MVT::v16i16, { 3, 7, 5, 7 } }, // psraw + split.
1203
1204 { ISD::SHL, MVT::v4i32, { 1, 3, 1, 2 } }, // pslld.
1205 { ISD::SRL, MVT::v4i32, { 1, 3, 1, 2 } }, // psrld.
1206 { ISD::SRA, MVT::v4i32, { 1, 3, 1, 2 } }, // psrad.
1207 { ISD::SHL, MVT::v8i32, { 3, 7, 5, 7 } }, // pslld + split.
1208 { ISD::SRL, MVT::v8i32, { 3, 7, 5, 7 } }, // psrld + split.
1209 { ISD::SRA, MVT::v8i32, { 3, 7, 5, 7 } }, // psrad + split.
1210
1211 { ISD::SHL, MVT::v2i64, { 1, 3, 1, 2 } }, // psllq.
1212 { ISD::SRL, MVT::v2i64, { 1, 3, 1, 2 } }, // psrlq.
1213 { ISD::SRA, MVT::v2i64, { 3, 4, 5, 7 } }, // 2 x psrad + shuffle.
1214 { ISD::SHL, MVT::v4i64, { 3, 7, 4, 6 } }, // psllq + split.
1215 { ISD::SRL, MVT::v4i64, { 3, 7, 4, 6 } }, // psrlq + split.
1216 { ISD::SRA, MVT::v4i64, { 6, 7,10,13 } }, // 2 x (2 x psrad + shuffle) + split.
1217 };
1218
1219 // XOP has faster vXi8 shifts.
1220 if (ST->hasAVX() && Op2Info.isUniform() &&
1221 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
1222 if (const auto *Entry =
1223 CostTableLookup(AVXUniformCostTable, ISD, LT.second))
1224 if (auto KindCost = Entry->Cost[CostKind])
1225 return LT.first * *KindCost;
1226
1227 static const CostKindTblEntry SSE2UniformCostTable[] = {
1228 // Uniform splats are cheaper for the following instructions.
1229 { ISD::SHL, MVT::v16i8, { 9, 10, 6, 9 } }, // psllw + pand.
1230 { ISD::SRL, MVT::v16i8, { 9, 13, 5, 9 } }, // psrlw + pand.
1231 { ISD::SRA, MVT::v16i8, { 11, 15, 9,13 } }, // pcmpgtb sequence.
1232
1233 { ISD::SHL, MVT::v8i16, { 2, 2, 1, 2 } }, // psllw.
1234 { ISD::SRL, MVT::v8i16, { 2, 2, 1, 2 } }, // psrlw.
1235 { ISD::SRA, MVT::v8i16, { 2, 2, 1, 2 } }, // psraw.
1236
1237 { ISD::SHL, MVT::v4i32, { 2, 2, 1, 2 } }, // pslld
1238 { ISD::SRL, MVT::v4i32, { 2, 2, 1, 2 } }, // psrld.
1239 { ISD::SRA, MVT::v4i32, { 2, 2, 1, 2 } }, // psrad.
1240
1241 { ISD::SHL, MVT::v2i64, { 2, 2, 1, 2 } }, // psllq.
1242 { ISD::SRL, MVT::v2i64, { 2, 2, 1, 2 } }, // psrlq.
1243 { ISD::SRA, MVT::v2i64, { 5, 9, 5, 7 } }, // 2*psrlq + xor + sub.
1244 };
1245
1246 if (ST->hasSSE2() && Op2Info.isUniform() &&
1247 (!ST->hasXOP() || LT.second.getScalarSizeInBits() != 8))
1248 if (const auto *Entry =
1249 CostTableLookup(SSE2UniformCostTable, ISD, LT.second))
1250 if (auto KindCost = Entry->Cost[CostKind])
1251 return LT.first * *KindCost;
1252
1253 static const CostKindTblEntry AVX512DQCostTable[] = {
1254 { ISD::MUL, MVT::v2i64, { 2, 15, 1, 3 } }, // pmullq
1255 { ISD::MUL, MVT::v4i64, { 2, 15, 1, 3 } }, // pmullq
1256 { ISD::MUL, MVT::v8i64, { 3, 15, 1, 3 } } // pmullq
1257 };
1258
1259 // Look for AVX512DQ lowering tricks for custom cases.
1260 if (ST->hasDQI())
1261 if (const auto *Entry = CostTableLookup(AVX512DQCostTable, ISD, LT.second))
1262 if (auto KindCost = Entry->Cost[CostKind])
1263 return LT.first * *KindCost;
1264
1265 static const CostKindTblEntry AVX512BWCostTable[] = {
1266 { ISD::SHL, MVT::v16i8, { 4, 8, 4, 5 } }, // extend/vpsllvw/pack sequence.
1267 { ISD::SRL, MVT::v16i8, { 4, 8, 4, 5 } }, // extend/vpsrlvw/pack sequence.
1268 { ISD::SRA, MVT::v16i8, { 4, 8, 4, 5 } }, // extend/vpsravw/pack sequence.
1269 { ISD::SHL, MVT::v32i8, { 4, 23,11,16 } }, // extend/vpsllvw/pack sequence.
1270 { ISD::SRL, MVT::v32i8, { 4, 30,12,18 } }, // extend/vpsrlvw/pack sequence.
1271 { ISD::SRA, MVT::v32i8, { 6, 13,24,30 } }, // extend/vpsravw/pack sequence.
1272 { ISD::SHL, MVT::v64i8, { 6, 19,13,15 } }, // extend/vpsllvw/pack sequence.
1273 { ISD::SRL, MVT::v64i8, { 7, 27,15,18 } }, // extend/vpsrlvw/pack sequence.
1274 { ISD::SRA, MVT::v64i8, { 15, 15,30,30 } }, // extend/vpsravw/pack sequence.
1275
1276 { ISD::SHL, MVT::v8i16, { 1, 1, 1, 1 } }, // vpsllvw
1277 { ISD::SRL, MVT::v8i16, { 1, 1, 1, 1 } }, // vpsrlvw
1278 { ISD::SRA, MVT::v8i16, { 1, 1, 1, 1 } }, // vpsravw
1279 { ISD::SHL, MVT::v16i16, { 1, 1, 1, 1 } }, // vpsllvw
1280 { ISD::SRL, MVT::v16i16, { 1, 1, 1, 1 } }, // vpsrlvw
1281 { ISD::SRA, MVT::v16i16, { 1, 1, 1, 1 } }, // vpsravw
1282 { ISD::SHL, MVT::v32i16, { 1, 1, 1, 1 } }, // vpsllvw
1283 { ISD::SRL, MVT::v32i16, { 1, 1, 1, 1 } }, // vpsrlvw
1284 { ISD::SRA, MVT::v32i16, { 1, 1, 1, 1 } }, // vpsravw
1285
1286 { ISD::ADD, MVT::v64i8, { 1, 1, 1, 1 } }, // paddb
1287 { ISD::ADD, MVT::v32i16, { 1, 1, 1, 1 } }, // paddw
1288
1289 { ISD::ADD, MVT::v32i8, { 1, 1, 1, 1 } }, // paddb
1290 { ISD::ADD, MVT::v16i16, { 1, 1, 1, 1 } }, // paddw
1291 { ISD::ADD, MVT::v8i32, { 1, 1, 1, 1 } }, // paddd
1292 { ISD::ADD, MVT::v4i64, { 1, 1, 1, 1 } }, // paddq
1293
1294 { ISD::SUB, MVT::v64i8, { 1, 1, 1, 1 } }, // psubb
1295 { ISD::SUB, MVT::v32i16, { 1, 1, 1, 1 } }, // psubw
1296
1297 { ISD::MUL, MVT::v16i8, { 4, 12, 4, 5 } }, // extend/pmullw/trunc
1298 { ISD::MUL, MVT::v32i8, { 3, 10, 7,10 } }, // pmaddubsw
1299 { ISD::MUL, MVT::v64i8, { 3, 11, 7,10 } }, // pmaddubsw
1300 { ISD::MUL, MVT::v32i16, { 1, 5, 1, 1 } }, // pmullw
1301
1302 { ISD::SUB, MVT::v32i8, { 1, 1, 1, 1 } }, // psubb
1303 { ISD::SUB, MVT::v16i16, { 1, 1, 1, 1 } }, // psubw
1304 { ISD::SUB, MVT::v8i32, { 1, 1, 1, 1 } }, // psubd
1305 { ISD::SUB, MVT::v4i64, { 1, 1, 1, 1 } }, // psubq
1306 };
1307
1308 // Look for AVX512BW lowering tricks for custom cases.
1309 if (ST->hasBWI())
1310 if (const auto *Entry = CostTableLookup(AVX512BWCostTable, ISD, LT.second))
1311 if (auto KindCost = Entry->Cost[CostKind])
1312 return LT.first * *KindCost;
1313
1314 static const CostKindTblEntry AVX512CostTable[] = {
1315 { ISD::SHL, MVT::v64i8, { 15, 19,27,33 } }, // vpblendv+split sequence.
1316 { ISD::SRL, MVT::v64i8, { 15, 19,30,36 } }, // vpblendv+split sequence.
1317 { ISD::SRA, MVT::v64i8, { 37, 37,51,63 } }, // vpblendv+split sequence.
1318
1319 { ISD::SHL, MVT::v32i16, { 11, 16,11,15 } }, // 2*extend/vpsrlvd/pack sequence.
1320 { ISD::SRL, MVT::v32i16, { 11, 16,11,15 } }, // 2*extend/vpsrlvd/pack sequence.
1321 { ISD::SRA, MVT::v32i16, { 11, 16,11,15 } }, // 2*extend/vpsravd/pack sequence.
1322
1323 { ISD::SHL, MVT::v4i32, { 1, 1, 1, 1 } },
1324 { ISD::SRL, MVT::v4i32, { 1, 1, 1, 1 } },
1325 { ISD::SRA, MVT::v4i32, { 1, 1, 1, 1 } },
1326 { ISD::SHL, MVT::v8i32, { 1, 1, 1, 1 } },
1327 { ISD::SRL, MVT::v8i32, { 1, 1, 1, 1 } },
1328 { ISD::SRA, MVT::v8i32, { 1, 1, 1, 1 } },
1329 { ISD::SHL, MVT::v16i32, { 1, 1, 1, 1 } },
1330 { ISD::SRL, MVT::v16i32, { 1, 1, 1, 1 } },
1331 { ISD::SRA, MVT::v16i32, { 1, 1, 1, 1 } },
1332
1333 { ISD::SHL, MVT::v2i64, { 1, 1, 1, 1 } },
1334 { ISD::SRL, MVT::v2i64, { 1, 1, 1, 1 } },
1335 { ISD::SRA, MVT::v2i64, { 1, 1, 1, 1 } },
1336 { ISD::SHL, MVT::v4i64, { 1, 1, 1, 1 } },
1337 { ISD::SRL, MVT::v4i64, { 1, 1, 1, 1 } },
1338 { ISD::SRA, MVT::v4i64, { 1, 1, 1, 1 } },
1339 { ISD::SHL, MVT::v8i64, { 1, 1, 1, 1 } },
1340 { ISD::SRL, MVT::v8i64, { 1, 1, 1, 1 } },
1341 { ISD::SRA, MVT::v8i64, { 1, 1, 1, 1 } },
1342
1343 { ISD::ADD, MVT::v64i8, { 3, 7, 5, 5 } }, // 2*paddb + split
1344 { ISD::ADD, MVT::v32i16, { 3, 7, 5, 5 } }, // 2*paddw + split
1345
1346 { ISD::SUB, MVT::v64i8, { 3, 7, 5, 5 } }, // 2*psubb + split
1347 { ISD::SUB, MVT::v32i16, { 3, 7, 5, 5 } }, // 2*psubw + split
1348
1349 { ISD::AND, MVT::v32i8, { 1, 1, 1, 1 } },
1350 { ISD::AND, MVT::v16i16, { 1, 1, 1, 1 } },
1351 { ISD::AND, MVT::v8i32, { 1, 1, 1, 1 } },
1352 { ISD::AND, MVT::v4i64, { 1, 1, 1, 1 } },
1353
1354 { ISD::OR, MVT::v32i8, { 1, 1, 1, 1 } },
1355 { ISD::OR, MVT::v16i16, { 1, 1, 1, 1 } },
1356 { ISD::OR, MVT::v8i32, { 1, 1, 1, 1 } },
1357 { ISD::OR, MVT::v4i64, { 1, 1, 1, 1 } },
1358
1359 { ISD::XOR, MVT::v32i8, { 1, 1, 1, 1 } },
1360 { ISD::XOR, MVT::v16i16, { 1, 1, 1, 1 } },
1361 { ISD::XOR, MVT::v8i32, { 1, 1, 1, 1 } },
1362 { ISD::XOR, MVT::v4i64, { 1, 1, 1, 1 } },
1363
1364 { ISD::MUL, MVT::v16i32, { 1, 10, 1, 2 } }, // pmulld (Skylake from agner.org)
1365 { ISD::MUL, MVT::v8i32, { 1, 10, 1, 2 } }, // pmulld (Skylake from agner.org)
1366 { ISD::MUL, MVT::v4i32, { 1, 10, 1, 2 } }, // pmulld (Skylake from agner.org)
1367 { ISD::MUL, MVT::v8i64, { 6, 9, 8, 8 } }, // 3*pmuludq/3*shift/2*add
1368 { ISD::MUL, MVT::i64, { 1 } }, // Skylake from http://www.agner.org/
1369
1370 { X86ISD::PMULUDQ, MVT::v8i64, { 1, 5, 1, 1 } },
1371
1372 { ISD::FNEG, MVT::v8f64, { 1, 1, 1, 2 } }, // Skylake from http://www.agner.org/
1373 { ISD::FADD, MVT::v8f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1374 { ISD::FADD, MVT::v4f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1375 { ISD::FSUB, MVT::v8f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1376 { ISD::FSUB, MVT::v4f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1377 { ISD::FMUL, MVT::v8f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1378 { ISD::FMUL, MVT::v4f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1379 { ISD::FMUL, MVT::v2f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1380 { ISD::FMUL, MVT::f64, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1381
1382 { ISD::FDIV, MVT::f64, { 4, 14, 1, 1 } }, // Skylake from http://www.agner.org/
1383 { ISD::FDIV, MVT::v2f64, { 4, 14, 1, 1 } }, // Skylake from http://www.agner.org/
1384 { ISD::FDIV, MVT::v4f64, { 8, 14, 1, 1 } }, // Skylake from http://www.agner.org/
1385 { ISD::FDIV, MVT::v8f64, { 16, 23, 1, 3 } }, // Skylake from http://www.agner.org/
1386
1387 { ISD::FNEG, MVT::v16f32, { 1, 1, 1, 2 } }, // Skylake from http://www.agner.org/
1388 { ISD::FADD, MVT::v16f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1389 { ISD::FADD, MVT::v8f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1390 { ISD::FSUB, MVT::v16f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1391 { ISD::FSUB, MVT::v8f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1392 { ISD::FMUL, MVT::v16f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1393 { ISD::FMUL, MVT::v8f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1394 { ISD::FMUL, MVT::v4f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1395 { ISD::FMUL, MVT::f32, { 1, 4, 1, 1 } }, // Skylake from http://www.agner.org/
1396
1397 { ISD::FDIV, MVT::f32, { 3, 11, 1, 1 } }, // Skylake from http://www.agner.org/
1398 { ISD::FDIV, MVT::v4f32, { 3, 11, 1, 1 } }, // Skylake from http://www.agner.org/
1399 { ISD::FDIV, MVT::v8f32, { 5, 11, 1, 1 } }, // Skylake from http://www.agner.org/
1400 { ISD::FDIV, MVT::v16f32, { 10, 18, 1, 3 } }, // Skylake from http://www.agner.org/
1401 };
1402
1403 if (ST->hasAVX512())
1404 if (const auto *Entry = CostTableLookup(AVX512CostTable, ISD, LT.second))
1405 if (auto KindCost = Entry->Cost[CostKind])
1406 return LT.first * *KindCost;
1407
1408 static const CostKindTblEntry AVX2ShiftCostTable[] = {
1409 // Shifts on vXi64/vXi32 on AVX2 is legal even though we declare to
1410 // customize them to detect the cases where shift amount is a scalar one.
1411 { ISD::SHL, MVT::v4i32, { 2, 3, 1, 3 } }, // vpsllvd (Haswell from agner.org)
1412 { ISD::SRL, MVT::v4i32, { 2, 3, 1, 3 } }, // vpsrlvd (Haswell from agner.org)
1413 { ISD::SRA, MVT::v4i32, { 2, 3, 1, 3 } }, // vpsravd (Haswell from agner.org)
1414 { ISD::SHL, MVT::v8i32, { 4, 4, 1, 3 } }, // vpsllvd (Haswell from agner.org)
1415 { ISD::SRL, MVT::v8i32, { 4, 4, 1, 3 } }, // vpsrlvd (Haswell from agner.org)
1416 { ISD::SRA, MVT::v8i32, { 4, 4, 1, 3 } }, // vpsravd (Haswell from agner.org)
1417 { ISD::SHL, MVT::v2i64, { 2, 3, 1, 1 } }, // vpsllvq (Haswell from agner.org)
1418 { ISD::SRL, MVT::v2i64, { 2, 3, 1, 1 } }, // vpsrlvq (Haswell from agner.org)
1419 { ISD::SHL, MVT::v4i64, { 4, 4, 1, 2 } }, // vpsllvq (Haswell from agner.org)
1420 { ISD::SRL, MVT::v4i64, { 4, 4, 1, 2 } }, // vpsrlvq (Haswell from agner.org)
1421 };
1422
1423 if (ST->hasAVX512()) {
1424 if (ISD == ISD::SHL && LT.second == MVT::v32i16 && Op2Info.isConstant())
1425 // On AVX512, a packed v32i16 shift left by a constant build_vector
1426 // is lowered into a vector multiply (vpmullw).
1427 return getArithmeticInstrCost(Instruction::Mul, Ty, CostKind,
1428 Op1Info.getNoProps(), Op2Info.getNoProps());
1429 }
1430
1431 // Look for AVX2 lowering tricks (XOP is always better at v4i32 shifts).
1432 if (ST->hasAVX2() && !(ST->hasXOP() && LT.second == MVT::v4i32)) {
1433 if (ISD == ISD::SHL && LT.second == MVT::v16i16 &&
1434 Op2Info.isConstant())
1435 // On AVX2, a packed v16i16 shift left by a constant build_vector
1436 // is lowered into a vector multiply (vpmullw).
1437 return getArithmeticInstrCost(Instruction::Mul, Ty, CostKind,
1438 Op1Info.getNoProps(), Op2Info.getNoProps());
1439
1440 if (const auto *Entry = CostTableLookup(AVX2ShiftCostTable, ISD, LT.second))
1441 if (auto KindCost = Entry->Cost[CostKind])
1442 return LT.first * *KindCost;
1443 }
1444
1445 static const CostKindTblEntry XOPShiftCostTable[] = {
1446 // 128bit shifts take 1cy, but right shifts require negation beforehand.
1447 { ISD::SHL, MVT::v16i8, { 1, 3, 1, 1 } },
1448 { ISD::SRL, MVT::v16i8, { 2, 3, 1, 1 } },
1449 { ISD::SRA, MVT::v16i8, { 2, 3, 1, 1 } },
1450 { ISD::SHL, MVT::v8i16, { 1, 3, 1, 1 } },
1451 { ISD::SRL, MVT::v8i16, { 2, 3, 1, 1 } },
1452 { ISD::SRA, MVT::v8i16, { 2, 3, 1, 1 } },
1453 { ISD::SHL, MVT::v4i32, { 1, 3, 1, 1 } },
1454 { ISD::SRL, MVT::v4i32, { 2, 3, 1, 1 } },
1455 { ISD::SRA, MVT::v4i32, { 2, 3, 1, 1 } },
1456 { ISD::SHL, MVT::v2i64, { 1, 3, 1, 1 } },
1457 { ISD::SRL, MVT::v2i64, { 2, 3, 1, 1 } },
1458 { ISD::SRA, MVT::v2i64, { 2, 3, 1, 1 } },
1459 // 256bit shifts require splitting if AVX2 didn't catch them above.
1460 { ISD::SHL, MVT::v32i8, { 4, 7, 5, 6 } },
1461 { ISD::SRL, MVT::v32i8, { 6, 7, 5, 6 } },
1462 { ISD::SRA, MVT::v32i8, { 6, 7, 5, 6 } },
1463 { ISD::SHL, MVT::v16i16, { 4, 7, 5, 6 } },
1464 { ISD::SRL, MVT::v16i16, { 6, 7, 5, 6 } },
1465 { ISD::SRA, MVT::v16i16, { 6, 7, 5, 6 } },
1466 { ISD::SHL, MVT::v8i32, { 4, 7, 5, 6 } },
1467 { ISD::SRL, MVT::v8i32, { 6, 7, 5, 6 } },
1468 { ISD::SRA, MVT::v8i32, { 6, 7, 5, 6 } },
1469 { ISD::SHL, MVT::v4i64, { 4, 7, 5, 6 } },
1470 { ISD::SRL, MVT::v4i64, { 6, 7, 5, 6 } },
1471 { ISD::SRA, MVT::v4i64, { 6, 7, 5, 6 } },
1472 };
1473
1474 // Look for XOP lowering tricks.
1475 if (ST->hasXOP()) {
1476 // If the right shift is constant then we'll fold the negation so
1477 // it's as cheap as a left shift.
1478 int ShiftISD = ISD;
1479 if ((ShiftISD == ISD::SRL || ShiftISD == ISD::SRA) && Op2Info.isConstant())
1480 ShiftISD = ISD::SHL;
1481 if (const auto *Entry =
1482 CostTableLookup(XOPShiftCostTable, ShiftISD, LT.second))
1483 if (auto KindCost = Entry->Cost[CostKind])
1484 return LT.first * *KindCost;
1485 }
1486
1487 if (ISD == ISD::SHL && !Op2Info.isUniform() && Op2Info.isConstant()) {
1488 MVT VT = LT.second;
1489 // Vector shift left by non uniform constant can be lowered
1490 // into vector multiply.
1491 if (((VT == MVT::v8i16 || VT == MVT::v4i32) && ST->hasSSE2()) ||
1492 ((VT == MVT::v16i16 || VT == MVT::v8i32) && ST->hasAVX()))
1493 ISD = ISD::MUL;
1494 }
1495
1496 static const CostKindTblEntry GLMCostTable[] = {
1497 { ISD::FDIV, MVT::f32, { 18, 19, 1, 1 } }, // divss
1498 { ISD::FDIV, MVT::v4f32, { 35, 36, 1, 1 } }, // divps
1499 { ISD::FDIV, MVT::f64, { 33, 34, 1, 1 } }, // divsd
1500 { ISD::FDIV, MVT::v2f64, { 65, 66, 1, 1 } }, // divpd
1501 };
1502
1503 if (ST->useGLMDivSqrtCosts())
1504 if (const auto *Entry = CostTableLookup(GLMCostTable, ISD, LT.second))
1505 if (auto KindCost = Entry->Cost[CostKind])
1506 return LT.first * *KindCost;
1507
1508 static const CostKindTblEntry SLMCostTable[] = {
1509 { ISD::MUL, MVT::v4i32, { 11, 11, 1, 7 } }, // pmulld
1510 { ISD::MUL, MVT::v8i16, { 2, 5, 1, 1 } }, // pmullw
1511 { ISD::FMUL, MVT::f64, { 2, 5, 1, 1 } }, // mulsd
1512 { ISD::FMUL, MVT::f32, { 1, 4, 1, 1 } }, // mulss
1513 { ISD::FMUL, MVT::v2f64, { 4, 7, 1, 1 } }, // mulpd
1514 { ISD::FMUL, MVT::v4f32, { 2, 5, 1, 1 } }, // mulps
1515 { ISD::FDIV, MVT::f32, { 17, 19, 1, 1 } }, // divss
1516 { ISD::FDIV, MVT::v4f32, { 39, 39, 1, 6 } }, // divps
1517 { ISD::FDIV, MVT::f64, { 32, 34, 1, 1 } }, // divsd
1518 { ISD::FDIV, MVT::v2f64, { 69, 69, 1, 6 } }, // divpd
1519 { ISD::FADD, MVT::v2f64, { 2, 4, 1, 1 } }, // addpd
1520 { ISD::FSUB, MVT::v2f64, { 2, 4, 1, 1 } }, // subpd
1521 // v2i64/v4i64 mul is custom lowered as a series of long:
1522 // multiplies(3), shifts(3) and adds(2)
1523 // slm muldq version throughput is 2 and addq throughput 4
1524 // thus: 3X2 (muldq throughput) + 3X1 (shift throughput) +
1525 // 3X4 (addq throughput) = 17
1526 { ISD::MUL, MVT::v2i64, { 17, 22, 9, 9 } },
1527 // slm addq\subq throughput is 4
1528 { ISD::ADD, MVT::v2i64, { 4, 2, 1, 2 } },
1529 { ISD::SUB, MVT::v2i64, { 4, 2, 1, 2 } },
1530 };
1531
1532 if (ST->useSLMArithCosts())
1533 if (const auto *Entry = CostTableLookup(SLMCostTable, ISD, LT.second))
1534 if (auto KindCost = Entry->Cost[CostKind])
1535 return LT.first * *KindCost;
1536
1537 static const CostKindTblEntry AVX2CostTable[] = {
1538 { ISD::SHL, MVT::v16i8, { 6, 21,11,16 } }, // vpblendvb sequence.
1539 { ISD::SHL, MVT::v32i8, { 6, 23,11,22 } }, // vpblendvb sequence.
1540 { ISD::SHL, MVT::v8i16, { 5, 18, 5,10 } }, // extend/vpsrlvd/pack sequence.
1541 { ISD::SHL, MVT::v16i16, { 8, 10,10,14 } }, // extend/vpsrlvd/pack sequence.
1542
1543 { ISD::SRL, MVT::v16i8, { 6, 27,12,18 } }, // vpblendvb sequence.
1544 { ISD::SRL, MVT::v32i8, { 8, 30,12,24 } }, // vpblendvb sequence.
1545 { ISD::SRL, MVT::v8i16, { 5, 11, 5,10 } }, // extend/vpsrlvd/pack sequence.
1546 { ISD::SRL, MVT::v16i16, { 8, 10,10,14 } }, // extend/vpsrlvd/pack sequence.
1547
1548 { ISD::SRA, MVT::v16i8, { 17, 17,24,30 } }, // vpblendvb sequence.
1549 { ISD::SRA, MVT::v32i8, { 18, 20,24,43 } }, // vpblendvb sequence.
1550 { ISD::SRA, MVT::v8i16, { 5, 11, 5,10 } }, // extend/vpsravd/pack sequence.
1551 { ISD::SRA, MVT::v16i16, { 8, 10,10,14 } }, // extend/vpsravd/pack sequence.
1552 { ISD::SRA, MVT::v2i64, { 4, 5, 5, 5 } }, // srl/xor/sub sequence.
1553 { ISD::SRA, MVT::v4i64, { 8, 8, 5, 9 } }, // srl/xor/sub sequence.
1554
1555 { ISD::SUB, MVT::v32i8, { 1, 1, 1, 2 } }, // psubb
1556 { ISD::ADD, MVT::v32i8, { 1, 1, 1, 2 } }, // paddb
1557 { ISD::SUB, MVT::v16i16, { 1, 1, 1, 2 } }, // psubw
1558 { ISD::ADD, MVT::v16i16, { 1, 1, 1, 2 } }, // paddw
1559 { ISD::SUB, MVT::v8i32, { 1, 1, 1, 2 } }, // psubd
1560 { ISD::ADD, MVT::v8i32, { 1, 1, 1, 2 } }, // paddd
1561 { ISD::SUB, MVT::v4i64, { 1, 1, 1, 2 } }, // psubq
1562 { ISD::ADD, MVT::v4i64, { 1, 1, 1, 2 } }, // paddq
1563
1564 { ISD::MUL, MVT::v16i8, { 5, 18, 6,12 } }, // extend/pmullw/pack
1565 { ISD::MUL, MVT::v32i8, { 4, 8, 8,16 } }, // pmaddubsw
1566 { ISD::MUL, MVT::v16i16, { 2, 5, 1, 2 } }, // pmullw
1567 { ISD::MUL, MVT::v8i32, { 4, 10, 1, 2 } }, // pmulld
1568 { ISD::MUL, MVT::v4i32, { 2, 10, 1, 2 } }, // pmulld
1569 { ISD::MUL, MVT::v4i64, { 6, 10, 8,13 } }, // 3*pmuludq/3*shift/2*add
1570 { ISD::MUL, MVT::v2i64, { 6, 10, 8, 8 } }, // 3*pmuludq/3*shift/2*add
1571
1572 { X86ISD::PMULUDQ, MVT::v4i64, { 1, 5, 1, 1 } },
1573
1574 { ISD::FNEG, MVT::v4f64, { 1, 1, 1, 2 } }, // vxorpd
1575 { ISD::FNEG, MVT::v8f32, { 1, 1, 1, 2 } }, // vxorps
1576
1577 { ISD::FADD, MVT::f64, { 1, 4, 1, 1 } }, // vaddsd
1578 { ISD::FADD, MVT::f32, { 1, 4, 1, 1 } }, // vaddss
1579 { ISD::FADD, MVT::v2f64, { 1, 4, 1, 1 } }, // vaddpd
1580 { ISD::FADD, MVT::v4f32, { 1, 4, 1, 1 } }, // vaddps
1581 { ISD::FADD, MVT::v4f64, { 1, 4, 1, 2 } }, // vaddpd
1582 { ISD::FADD, MVT::v8f32, { 1, 4, 1, 2 } }, // vaddps
1583
1584 { ISD::FSUB, MVT::f64, { 1, 4, 1, 1 } }, // vsubsd
1585 { ISD::FSUB, MVT::f32, { 1, 4, 1, 1 } }, // vsubss
1586 { ISD::FSUB, MVT::v2f64, { 1, 4, 1, 1 } }, // vsubpd
1587 { ISD::FSUB, MVT::v4f32, { 1, 4, 1, 1 } }, // vsubps
1588 { ISD::FSUB, MVT::v4f64, { 1, 4, 1, 2 } }, // vsubpd
1589 { ISD::FSUB, MVT::v8f32, { 1, 4, 1, 2 } }, // vsubps
1590
1591 { ISD::FMUL, MVT::f64, { 1, 5, 1, 1 } }, // vmulsd
1592 { ISD::FMUL, MVT::f32, { 1, 5, 1, 1 } }, // vmulss
1593 { ISD::FMUL, MVT::v2f64, { 1, 5, 1, 1 } }, // vmulpd
1594 { ISD::FMUL, MVT::v4f32, { 1, 5, 1, 1 } }, // vmulps
1595 { ISD::FMUL, MVT::v4f64, { 1, 5, 1, 2 } }, // vmulpd
1596 { ISD::FMUL, MVT::v8f32, { 1, 5, 1, 2 } }, // vmulps
1597
1598 { ISD::FDIV, MVT::f32, { 7, 13, 1, 1 } }, // vdivss
1599 { ISD::FDIV, MVT::v4f32, { 7, 13, 1, 1 } }, // vdivps
1600 { ISD::FDIV, MVT::v8f32, { 14, 21, 1, 3 } }, // vdivps
1601 { ISD::FDIV, MVT::f64, { 14, 20, 1, 1 } }, // vdivsd
1602 { ISD::FDIV, MVT::v2f64, { 14, 20, 1, 1 } }, // vdivpd
1603 { ISD::FDIV, MVT::v4f64, { 28, 35, 1, 3 } }, // vdivpd
1604 };
1605
1606 // Look for AVX2 lowering tricks for custom cases.
1607 if (ST->hasAVX2())
1608 if (const auto *Entry = CostTableLookup(AVX2CostTable, ISD, LT.second))
1609 if (auto KindCost = Entry->Cost[CostKind])
1610 return LT.first * *KindCost;
1611
1612 static const CostKindTblEntry AVX1CostTable[] = {
1613 // We don't have to scalarize unsupported ops. We can issue two half-sized
1614 // operations and we only need to extract the upper YMM half.
1615 // Two ops + 1 extract + 1 insert = 4.
1616 { ISD::MUL, MVT::v32i8, { 10, 11, 18, 19 } }, // pmaddubsw + split
1617 { ISD::MUL, MVT::v16i8, { 5, 6, 8, 12 } }, // 2*pmaddubsw/3*and/psllw/or
1618 { ISD::MUL, MVT::v16i16, { 4, 8, 5, 6 } }, // pmullw + split
1619 { ISD::MUL, MVT::v8i32, { 5, 8, 5, 10 } }, // pmulld + split
1620 { ISD::MUL, MVT::v4i32, { 2, 5, 1, 3 } }, // pmulld
1621 { ISD::MUL, MVT::v4i64, { 12, 15, 19, 20 } },
1622
1623 { X86ISD::PMULUDQ, MVT::v4i64, { 3, 5, 5, 6 } }, // pmuludq + split
1624
1625 { ISD::AND, MVT::v32i8, { 1, 1, 1, 2 } }, // vandps
1626 { ISD::AND, MVT::v16i16, { 1, 1, 1, 2 } }, // vandps
1627 { ISD::AND, MVT::v8i32, { 1, 1, 1, 2 } }, // vandps
1628 { ISD::AND, MVT::v4i64, { 1, 1, 1, 2 } }, // vandps
1629
1630 { ISD::OR, MVT::v32i8, { 1, 1, 1, 2 } }, // vorps
1631 { ISD::OR, MVT::v16i16, { 1, 1, 1, 2 } }, // vorps
1632 { ISD::OR, MVT::v8i32, { 1, 1, 1, 2 } }, // vorps
1633 { ISD::OR, MVT::v4i64, { 1, 1, 1, 2 } }, // vorps
1634
1635 { ISD::XOR, MVT::v32i8, { 1, 1, 1, 2 } }, // vxorps
1636 { ISD::XOR, MVT::v16i16, { 1, 1, 1, 2 } }, // vxorps
1637 { ISD::XOR, MVT::v8i32, { 1, 1, 1, 2 } }, // vxorps
1638 { ISD::XOR, MVT::v4i64, { 1, 1, 1, 2 } }, // vxorps
1639
1640 { ISD::SUB, MVT::v32i8, { 4, 2, 5, 6 } }, // psubb + split
1641 { ISD::ADD, MVT::v32i8, { 4, 2, 5, 6 } }, // paddb + split
1642 { ISD::SUB, MVT::v16i16, { 4, 2, 5, 6 } }, // psubw + split
1643 { ISD::ADD, MVT::v16i16, { 4, 2, 5, 6 } }, // paddw + split
1644 { ISD::SUB, MVT::v8i32, { 4, 2, 5, 6 } }, // psubd + split
1645 { ISD::ADD, MVT::v8i32, { 4, 2, 5, 6 } }, // paddd + split
1646 { ISD::SUB, MVT::v4i64, { 4, 2, 5, 6 } }, // psubq + split
1647 { ISD::ADD, MVT::v4i64, { 4, 2, 5, 6 } }, // paddq + split
1648 { ISD::SUB, MVT::v2i64, { 1, 1, 1, 1 } }, // psubq
1649 { ISD::ADD, MVT::v2i64, { 1, 1, 1, 1 } }, // paddq
1650
1651 { ISD::SHL, MVT::v16i8, { 10, 21,11,17 } }, // pblendvb sequence.
1652 { ISD::SHL, MVT::v32i8, { 22, 22,27,40 } }, // pblendvb sequence + split.
1653 { ISD::SHL, MVT::v8i16, { 6, 9,11,11 } }, // pblendvb sequence.
1654 { ISD::SHL, MVT::v16i16, { 13, 16,24,25 } }, // pblendvb sequence + split.
1655 { ISD::SHL, MVT::v4i32, { 3, 11, 4, 6 } }, // pslld/paddd/cvttps2dq/pmulld
1656 { ISD::SHL, MVT::v8i32, { 9, 11,12,17 } }, // pslld/paddd/cvttps2dq/pmulld + split
1657 { ISD::SHL, MVT::v2i64, { 2, 4, 4, 6 } }, // Shift each lane + blend.
1658 { ISD::SHL, MVT::v4i64, { 6, 7,11,15 } }, // Shift each lane + blend + split.
1659
1660 { ISD::SRL, MVT::v16i8, { 11, 27,12,18 } }, // pblendvb sequence.
1661 { ISD::SRL, MVT::v32i8, { 23, 23,30,43 } }, // pblendvb sequence + split.
1662 { ISD::SRL, MVT::v8i16, { 13, 16,14,22 } }, // pblendvb sequence.
1663 { ISD::SRL, MVT::v16i16, { 28, 30,31,48 } }, // pblendvb sequence + split.
1664 { ISD::SRL, MVT::v4i32, { 6, 7,12,16 } }, // Shift each lane + blend.
1665 { ISD::SRL, MVT::v8i32, { 14, 14,26,34 } }, // Shift each lane + blend + split.
1666 { ISD::SRL, MVT::v2i64, { 2, 4, 4, 6 } }, // Shift each lane + blend.
1667 { ISD::SRL, MVT::v4i64, { 6, 7,11,15 } }, // Shift each lane + blend + split.
1668
1669 { ISD::SRA, MVT::v16i8, { 21, 22,24,36 } }, // pblendvb sequence.
1670 { ISD::SRA, MVT::v32i8, { 44, 45,51,76 } }, // pblendvb sequence + split.
1671 { ISD::SRA, MVT::v8i16, { 13, 16,14,22 } }, // pblendvb sequence.
1672 { ISD::SRA, MVT::v16i16, { 28, 30,31,48 } }, // pblendvb sequence + split.
1673 { ISD::SRA, MVT::v4i32, { 6, 7,12,16 } }, // Shift each lane + blend.
1674 { ISD::SRA, MVT::v8i32, { 14, 14,26,34 } }, // Shift each lane + blend + split.
1675 { ISD::SRA, MVT::v2i64, { 5, 6,10,14 } }, // Shift each lane + blend.
1676 { ISD::SRA, MVT::v4i64, { 12, 12,22,30 } }, // Shift each lane + blend + split.
1677
1678 { ISD::FNEG, MVT::v4f64, { 2, 2, 1, 2 } }, // BTVER2 from http://www.agner.org/
1679 { ISD::FNEG, MVT::v8f32, { 2, 2, 1, 2 } }, // BTVER2 from http://www.agner.org/
1680
1681 { ISD::FADD, MVT::f64, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1682 { ISD::FADD, MVT::f32, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1683 { ISD::FADD, MVT::v2f64, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1684 { ISD::FADD, MVT::v4f32, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1685 { ISD::FADD, MVT::v4f64, { 2, 5, 1, 2 } }, // BDVER2 from http://www.agner.org/
1686 { ISD::FADD, MVT::v8f32, { 2, 5, 1, 2 } }, // BDVER2 from http://www.agner.org/
1687
1688 { ISD::FSUB, MVT::f64, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1689 { ISD::FSUB, MVT::f32, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1690 { ISD::FSUB, MVT::v2f64, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1691 { ISD::FSUB, MVT::v4f32, { 1, 5, 1, 1 } }, // BDVER2 from http://www.agner.org/
1692 { ISD::FSUB, MVT::v4f64, { 2, 5, 1, 2 } }, // BDVER2 from http://www.agner.org/
1693 { ISD::FSUB, MVT::v8f32, { 2, 5, 1, 2 } }, // BDVER2 from http://www.agner.org/
1694
1695 { ISD::FMUL, MVT::f64, { 2, 5, 1, 1 } }, // BTVER2 from http://www.agner.org/
1696 { ISD::FMUL, MVT::f32, { 1, 5, 1, 1 } }, // BTVER2 from http://www.agner.org/
1697 { ISD::FMUL, MVT::v2f64, { 2, 5, 1, 1 } }, // BTVER2 from http://www.agner.org/
1698 { ISD::FMUL, MVT::v4f32, { 1, 5, 1, 1 } }, // BTVER2 from http://www.agner.org/
1699 { ISD::FMUL, MVT::v4f64, { 4, 5, 1, 2 } }, // BTVER2 from http://www.agner.org/
1700 { ISD::FMUL, MVT::v8f32, { 2, 5, 1, 2 } }, // BTVER2 from http://www.agner.org/
1701
1702 { ISD::FDIV, MVT::f32, { 14, 14, 1, 1 } }, // SNB from http://www.agner.org/
1703 { ISD::FDIV, MVT::v4f32, { 14, 14, 1, 1 } }, // SNB from http://www.agner.org/
1704 { ISD::FDIV, MVT::v8f32, { 28, 29, 1, 3 } }, // SNB from http://www.agner.org/
1705 { ISD::FDIV, MVT::f64, { 22, 22, 1, 1 } }, // SNB from http://www.agner.org/
1706 { ISD::FDIV, MVT::v2f64, { 22, 22, 1, 1 } }, // SNB from http://www.agner.org/
1707 { ISD::FDIV, MVT::v4f64, { 44, 45, 1, 3 } }, // SNB from http://www.agner.org/
1708 };
1709
1710 if (ST->hasAVX())
1711 if (const auto *Entry = CostTableLookup(AVX1CostTable, ISD, LT.second))
1712 if (auto KindCost = Entry->Cost[CostKind])
1713 return LT.first * *KindCost;
1714
1715 static const CostKindTblEntry SSE42CostTable[] = {
1716 { ISD::FADD, MVT::f64, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1717 { ISD::FADD, MVT::f32, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1718 { ISD::FADD, MVT::v2f64, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1719 { ISD::FADD, MVT::v4f32, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1720
1721 { ISD::FSUB, MVT::f64, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1722 { ISD::FSUB, MVT::f32 , { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1723 { ISD::FSUB, MVT::v2f64, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1724 { ISD::FSUB, MVT::v4f32, { 1, 3, 1, 1 } }, // Nehalem from http://www.agner.org/
1725
1726 { ISD::FMUL, MVT::f64, { 1, 5, 1, 1 } }, // Nehalem from http://www.agner.org/
1727 { ISD::FMUL, MVT::f32, { 1, 5, 1, 1 } }, // Nehalem from http://www.agner.org/
1728 { ISD::FMUL, MVT::v2f64, { 1, 5, 1, 1 } }, // Nehalem from http://www.agner.org/
1729 { ISD::FMUL, MVT::v4f32, { 1, 5, 1, 1 } }, // Nehalem from http://www.agner.org/
1730
1731 { ISD::FDIV, MVT::f32, { 14, 14, 1, 1 } }, // Nehalem from http://www.agner.org/
1732 { ISD::FDIV, MVT::v4f32, { 14, 14, 1, 1 } }, // Nehalem from http://www.agner.org/
1733 { ISD::FDIV, MVT::f64, { 22, 22, 1, 1 } }, // Nehalem from http://www.agner.org/
1734 { ISD::FDIV, MVT::v2f64, { 22, 22, 1, 1 } }, // Nehalem from http://www.agner.org/
1735
1736 { ISD::MUL, MVT::v2i64, { 6, 10,10,10 } } // 3*pmuludq/3*shift/2*add
1737 };
1738
1739 if (ST->hasSSE42())
1740 if (const auto *Entry = CostTableLookup(SSE42CostTable, ISD, LT.second))
1741 if (auto KindCost = Entry->Cost[CostKind])
1742 return LT.first * *KindCost;
1743
1744 static const CostKindTblEntry SSE41CostTable[] = {
1745 { ISD::SHL, MVT::v16i8, { 15, 24,17,22 } }, // pblendvb sequence.
1746 { ISD::SHL, MVT::v8i16, { 11, 14,11,11 } }, // pblendvb sequence.
1747 { ISD::SHL, MVT::v4i32, { 14, 20, 4,10 } }, // pslld/paddd/cvttps2dq/pmulld
1748
1749 { ISD::SRL, MVT::v16i8, { 16, 27,18,24 } }, // pblendvb sequence.
1750 { ISD::SRL, MVT::v8i16, { 22, 26,23,27 } }, // pblendvb sequence.
1751 { ISD::SRL, MVT::v4i32, { 16, 17,15,19 } }, // Shift each lane + blend.
1752 { ISD::SRL, MVT::v2i64, { 4, 6, 5, 7 } }, // splat+shuffle sequence.
1753
1754 { ISD::SRA, MVT::v16i8, { 38, 41,30,36 } }, // pblendvb sequence.
1755 { ISD::SRA, MVT::v8i16, { 22, 26,23,27 } }, // pblendvb sequence.
1756 { ISD::SRA, MVT::v4i32, { 16, 17,15,19 } }, // Shift each lane + blend.
1757 { ISD::SRA, MVT::v2i64, { 8, 17, 5, 7 } }, // splat+shuffle sequence.
1758
1759 { ISD::MUL, MVT::v4i32, { 2, 11, 1, 1 } } // pmulld (Nehalem from agner.org)
1760 };
1761
1762 if (ST->hasSSE41())
1763 if (const auto *Entry = CostTableLookup(SSE41CostTable, ISD, LT.second))
1764 if (auto KindCost = Entry->Cost[CostKind])
1765 return LT.first * *KindCost;
1766
1767 static const CostKindTblEntry SSSE3CostTable[] = {
1768 { ISD::MUL, MVT::v16i8, { 5, 18,10,12 } }, // 2*pmaddubsw/3*and/psllw/or
1769 };
1770
1771 if (ST->hasSSSE3())
1772 if (const auto *Entry = CostTableLookup(SSSE3CostTable, ISD, LT.second))
1773 if (auto KindCost = Entry->Cost[CostKind])
1774 return LT.first * *KindCost;
1775
1776 static const CostKindTblEntry SSE2CostTable[] = {
1777 // We don't correctly identify costs of casts because they are marked as
1778 // custom.
1779 { ISD::SHL, MVT::v16i8, { 13, 21,26,28 } }, // cmpgtb sequence.
1780 { ISD::SHL, MVT::v8i16, { 24, 27,16,20 } }, // cmpgtw sequence.
1781 { ISD::SHL, MVT::v4i32, { 17, 19,10,12 } }, // pslld/paddd/cvttps2dq/pmuludq.
1782 { ISD::SHL, MVT::v2i64, { 4, 6, 5, 7 } }, // splat+shuffle sequence.
1783
1784 { ISD::SRL, MVT::v16i8, { 14, 28,27,30 } }, // cmpgtb sequence.
1785 { ISD::SRL, MVT::v8i16, { 16, 19,31,31 } }, // cmpgtw sequence.
1786 { ISD::SRL, MVT::v4i32, { 12, 12,15,19 } }, // Shift each lane + blend.
1787 { ISD::SRL, MVT::v2i64, { 4, 6, 5, 7 } }, // splat+shuffle sequence.
1788
1789 { ISD::SRA, MVT::v16i8, { 27, 30,54,54 } }, // unpacked cmpgtb sequence.
1790 { ISD::SRA, MVT::v8i16, { 16, 19,31,31 } }, // cmpgtw sequence.
1791 { ISD::SRA, MVT::v4i32, { 12, 12,15,19 } }, // Shift each lane + blend.
1792 { ISD::SRA, MVT::v2i64, { 8, 11,12,16 } }, // srl/xor/sub splat+shuffle sequence.
1793
1794 { ISD::AND, MVT::v16i8, { 1, 1, 1, 1 } }, // pand
1795 { ISD::AND, MVT::v8i16, { 1, 1, 1, 1 } }, // pand
1796 { ISD::AND, MVT::v4i32, { 1, 1, 1, 1 } }, // pand
1797 { ISD::AND, MVT::v2i64, { 1, 1, 1, 1 } }, // pand
1798
1799 { ISD::OR, MVT::v16i8, { 1, 1, 1, 1 } }, // por
1800 { ISD::OR, MVT::v8i16, { 1, 1, 1, 1 } }, // por
1801 { ISD::OR, MVT::v4i32, { 1, 1, 1, 1 } }, // por
1802 { ISD::OR, MVT::v2i64, { 1, 1, 1, 1 } }, // por
1803
1804 { ISD::XOR, MVT::v16i8, { 1, 1, 1, 1 } }, // pxor
1805 { ISD::XOR, MVT::v8i16, { 1, 1, 1, 1 } }, // pxor
1806 { ISD::XOR, MVT::v4i32, { 1, 1, 1, 1 } }, // pxor
1807 { ISD::XOR, MVT::v2i64, { 1, 1, 1, 1 } }, // pxor
1808
1809 { ISD::ADD, MVT::v2i64, { 1, 2, 1, 2 } }, // paddq
1810 { ISD::SUB, MVT::v2i64, { 1, 2, 1, 2 } }, // psubq
1811
1812 { ISD::MUL, MVT::v16i8, { 6, 18,12,12 } }, // 2*unpack/2*pmullw/2*and/pack
1813 { ISD::MUL, MVT::v8i16, { 1, 5, 1, 1 } }, // pmullw
1814 { ISD::MUL, MVT::v4i32, { 6, 8, 7, 7 } }, // 3*pmuludq/4*shuffle
1815 { ISD::MUL, MVT::v2i64, { 7, 10,10,10 } }, // 3*pmuludq/3*shift/2*add
1816
1817 { X86ISD::PMULUDQ, MVT::v2i64, { 1, 5, 1, 1 } },
1818
1819 { ISD::FDIV, MVT::f32, { 23, 23, 1, 1 } }, // Pentium IV from http://www.agner.org/
1820 { ISD::FDIV, MVT::v4f32, { 39, 39, 1, 1 } }, // Pentium IV from http://www.agner.org/
1821 { ISD::FDIV, MVT::f64, { 38, 38, 1, 1 } }, // Pentium IV from http://www.agner.org/
1822 { ISD::FDIV, MVT::v2f64, { 69, 69, 1, 1 } }, // Pentium IV from http://www.agner.org/
1823
1824 { ISD::FNEG, MVT::f32, { 1, 1, 1, 1 } }, // Pentium IV from http://www.agner.org/
1825 { ISD::FNEG, MVT::f64, { 1, 1, 1, 1 } }, // Pentium IV from http://www.agner.org/
1826 { ISD::FNEG, MVT::v4f32, { 1, 1, 1, 1 } }, // Pentium IV from http://www.agner.org/
1827 { ISD::FNEG, MVT::v2f64, { 1, 1, 1, 1 } }, // Pentium IV from http://www.agner.org/
1828
1829 { ISD::FADD, MVT::f32, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1830 { ISD::FADD, MVT::f64, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1831 { ISD::FADD, MVT::v2f64, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1832
1833 { ISD::FSUB, MVT::f32, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1834 { ISD::FSUB, MVT::f64, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1835 { ISD::FSUB, MVT::v2f64, { 2, 3, 1, 1 } }, // Pentium IV from http://www.agner.org/
1836
1837 { ISD::FMUL, MVT::f64, { 2, 5, 1, 1 } }, // Pentium IV from http://www.agner.org/
1838 { ISD::FMUL, MVT::v2f64, { 2, 5, 1, 1 } }, // Pentium IV from http://www.agner.org/
1839 };
1840
1841 if (ST->hasSSE2())
1842 if (const auto *Entry = CostTableLookup(SSE2CostTable, ISD, LT.second))
1843 if (auto KindCost = Entry->Cost[CostKind])
1844 return LT.first * *KindCost;
1845
1846 static const CostKindTblEntry SSE1CostTable[] = {
1847 { ISD::FDIV, MVT::f32, { 17, 18, 1, 1 } }, // Pentium III from http://www.agner.org/
1848 { ISD::FDIV, MVT::v4f32, { 34, 48, 1, 1 } }, // Pentium III from http://www.agner.org/
1849
1850 { ISD::FNEG, MVT::f32, { 2, 2, 1, 2 } }, // Pentium III from http://www.agner.org/
1851 { ISD::FNEG, MVT::v4f32, { 2, 2, 1, 2 } }, // Pentium III from http://www.agner.org/
1852
1853 { ISD::FADD, MVT::f32, { 1, 3, 1, 1 } }, // Pentium III from http://www.agner.org/
1854 { ISD::FADD, MVT::v4f32, { 2, 3, 1, 1 } }, // Pentium III from http://www.agner.org/
1855
1856 { ISD::FSUB, MVT::f32, { 1, 3, 1, 1 } }, // Pentium III from http://www.agner.org/
1857 { ISD::FSUB, MVT::v4f32, { 2, 3, 1, 1 } }, // Pentium III from http://www.agner.org/
1858
1859 { ISD::FMUL, MVT::f32, { 2, 5, 1, 1 } }, // Pentium III from http://www.agner.org/
1860 { ISD::FMUL, MVT::v4f32, { 2, 5, 1, 1 } }, // Pentium III from http://www.agner.org/
1861 };
1862
1863 if (ST->hasSSE1())
1864 if (const auto *Entry = CostTableLookup(SSE1CostTable, ISD, LT.second))
1865 if (auto KindCost = Entry->Cost[CostKind])
1866 return LT.first * *KindCost;
1867
1868 static const CostKindTblEntry X64CostTbl[] = { // 64-bit targets
1869 { ISD::ADD, MVT::i64, { 1 } }, // Core (Merom) from http://www.agner.org/
1870 { ISD::SUB, MVT::i64, { 1 } }, // Core (Merom) from http://www.agner.org/
1871 { ISD::MUL, MVT::i64, { 2, 6, 1, 2 } },
1872 };
1873
1874 if (ST->is64Bit())
1875 if (const auto *Entry = CostTableLookup(X64CostTbl, ISD, LT.second))
1876 if (auto KindCost = Entry->Cost[CostKind])
1877 return LT.first * *KindCost;
1878
1879 static const CostKindTblEntry X86CostTbl[] = { // 32 or 64-bit targets
1880 { ISD::ADD, MVT::i8, { 1 } }, // Pentium III from http://www.agner.org/
1881 { ISD::ADD, MVT::i16, { 1 } }, // Pentium III from http://www.agner.org/
1882 { ISD::ADD, MVT::i32, { 1 } }, // Pentium III from http://www.agner.org/
1883
1884 { ISD::SUB, MVT::i8, { 1 } }, // Pentium III from http://www.agner.org/
1885 { ISD::SUB, MVT::i16, { 1 } }, // Pentium III from http://www.agner.org/
1886 { ISD::SUB, MVT::i32, { 1 } }, // Pentium III from http://www.agner.org/
1887
1888 { ISD::MUL, MVT::i8, { 3, 4, 1, 1 } },
1889 { ISD::MUL, MVT::i16, { 2, 4, 1, 1 } },
1890 { ISD::MUL, MVT::i32, { 1, 4, 1, 1 } },
1891
1892 { ISD::FNEG, MVT::f64, { 2, 2, 1, 3 } }, // (x87)
1893 { ISD::FADD, MVT::f64, { 2, 3, 1, 1 } }, // (x87)
1894 { ISD::FSUB, MVT::f64, { 2, 3, 1, 1 } }, // (x87)
1895 { ISD::FMUL, MVT::f64, { 2, 5, 1, 1 } }, // (x87)
1896 { ISD::FDIV, MVT::f64, { 38, 38, 1, 1 } }, // (x87)
1897 };
1898
1899 if (const auto *Entry = CostTableLookup(X86CostTbl, ISD, LT.second))
1900 if (auto KindCost = Entry->Cost[CostKind])
1901 return LT.first * *KindCost;
1902
1903 // It is not a good idea to vectorize division. We have to scalarize it and
1904 // in the process we will often end up having to spilling regular
1905 // registers. The overhead of division is going to dominate most kernels
1906 // anyways so try hard to prevent vectorization of division - it is
1907 // generally a bad idea. Assume somewhat arbitrarily that we have to be able
1908 // to hide "20 cycles" for each lane.
1909 if (CostKind == TTI::TCK_RecipThroughput && LT.second.isVector() &&
1910 (ISD == ISD::SDIV || ISD == ISD::SREM || ISD == ISD::UDIV ||
1911 ISD == ISD::UREM)) {
1912 InstructionCost ScalarCost =
1913 getArithmeticInstrCost(Opcode, Ty->getScalarType(), CostKind,
1914 Op1Info.getNoProps(), Op2Info.getNoProps());
1915 return 20 * LT.first * LT.second.getVectorNumElements() * ScalarCost;
1916 }
1917
1918 // Handle some basic single instruction code size cases.
1919 if (CostKind == TTI::TCK_CodeSize) {
1920 switch (ISD) {
1921 case ISD::FADD:
1922 case ISD::FSUB:
1923 case ISD::FMUL:
1924 case ISD::FDIV:
1925 case ISD::FNEG:
1926 case ISD::AND:
1927 case ISD::OR:
1928 case ISD::XOR:
1929 return LT.first;
1930 break;
1931 }
1932 }
1933
1934 // Fallback to the default implementation.
1935 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
1936 Args, CtxI);
1937}
1938
1941 unsigned Opcode1, const SmallBitVector &OpcodeMask,
1943 ArrayRef<const Value *> Scalars) const {
1944 if (isLegalAltInstr(VecTy, Opcode0, Opcode1, OpcodeMask, Scalars))
1945 return TTI::TCC_Basic;
1947}
1948
1950 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
1952 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
1953 TTI::VectorInstrContext VIC) const {
1954 assert((Mask.empty() || DstTy->isScalableTy() ||
1955 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
1956 "Expected the Mask to match the return size if given");
1957 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
1958 "Expected the same scalar types");
1959
1960 // 64-bit packed float vectors (v2f32) are widened to type v4f32.
1961 // 64-bit packed integer vectors (v2i32) are widened to type v4i32.
1962 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
1963
1964 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
1965
1966 // If all args are constant than this will be constant folded away.
1967 if (!Args.empty() &&
1968 all_of(Args, [](const Value *Arg) { return isa<Constant>(Arg); }))
1969 return TTI::TCC_Free;
1970
1971 // Recognize a basic concat_vector shuffle.
1972 if (Kind == TTI::SK_PermuteTwoSrc &&
1973 Mask.size() == (2 * SrcTy->getElementCount().getKnownMinValue()) &&
1974 ShuffleVectorInst::isIdentityMask(Mask, Mask.size()))
1978 CostKind, Mask, Mask.size() / 2, SrcTy);
1979
1980 // Treat Transpose as 2-op shuffles - there's no difference in lowering.
1981 if (Kind == TTI::SK_Transpose)
1982 if (LT.second != MVT::v4f64 && LT.second != MVT::v4i64)
1983 Kind = TTI::SK_PermuteTwoSrc;
1984
1985 if (Kind == TTI::SK_Broadcast) {
1986 // For Broadcasts we are splatting the first element from the first input
1987 // register, so only need to reference that input and all the output
1988 // registers are the same.
1989 LT.first = 1;
1990
1991 // If we're broadcasting a load then AVX/AVX2 can do this for free.
1992 // If many-used-load whose every use is one of a small set of operations
1993 // that SLP can rewrite into a single vector lane, codegen can fold it into
1994 // the free broadcast.
1995 using namespace PatternMatch;
1996 auto IsBroadcastLoadFoldUser = [&](const User *U) {
1997 if (isa<InsertElementInst>(U) && U->getOperand(1) == Args[0])
1998 return true;
1999 if (U->getType()->isVectorTy())
2000 return false;
2001 // Terminators (return/branch/switch/indirectbr/resume/invoke EH)
2002 // and phis carry the value across control flow.
2003 if (const auto *I = dyn_cast<Instruction>(U))
2004 if (I->isTerminator() ||
2006 return false;
2007 // Only pure calls can be folded.
2008 if (const auto *CB = dyn_cast<CallBase>(U))
2009 return CB->doesNotAccessMemory() && !CB->mayHaveSideEffects();
2010 return true;
2011 };
2012 auto IsFoldableSLPBroadcastLoad = [&]() {
2013 if (!match(Args[0], m_Load(m_Value())))
2014 return false;
2015 auto *FVT = dyn_cast<FixedVectorType>(DstTy);
2016 if (!FVT)
2017 return false;
2018 // getNumUses() counts each Use, matching the per-lane broadcast
2019 // accounting (a use like `op %x, %x` consumes two broadcast lanes).
2020 if (Args[0]->getNumUses() != FVT->getNumElements())
2021 return false;
2022 return all_of(Args[0]->users(), IsBroadcastLoadFoldUser);
2023 };
2024 if (!Args.empty() &&
2025 (match(Args[0], m_OneUse(m_Load(m_Value()))) ||
2026 IsFoldableSLPBroadcastLoad()) &&
2027 (ST->hasAVX2() ||
2028 (ST->hasAVX() && LT.second.getScalarSizeInBits() >= 32)))
2029 return TTI::TCC_Free;
2030 }
2031
2032 // Attempt to detect a cheaper inlane shuffle, avoiding 128-bit subvector
2033 // permutation.
2034 // Attempt to detect a shuffle mask with a single defined element.
2035 bool IsInLaneShuffle = false;
2036 bool IsSingleElementMask = false;
2037 if (SrcTy->getPrimitiveSizeInBits() > 0 &&
2038 (SrcTy->getPrimitiveSizeInBits() % 128) == 0 &&
2039 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
2040 Mask.size() == SrcTy->getElementCount().getKnownMinValue()) {
2041 unsigned NumLanes = SrcTy->getPrimitiveSizeInBits() / 128;
2042 unsigned NumEltsPerLane = Mask.size() / NumLanes;
2043 if ((Mask.size() % NumLanes) == 0) {
2044 IsInLaneShuffle = all_of(enumerate(Mask), [&](const auto &P) {
2045 return P.value() == PoisonMaskElem ||
2046 ((P.value() % Mask.size()) / NumEltsPerLane) ==
2047 (P.index() / NumEltsPerLane);
2048 });
2049 IsSingleElementMask =
2050 (Mask.size() - 1) == static_cast<unsigned>(count_if(Mask, [](int M) {
2051 return M == PoisonMaskElem;
2052 }));
2053 }
2054 }
2055
2056 // Treat <X x bfloat> shuffles as <X x half>.
2057 if (LT.second.isVectorOf(MVT::bf16))
2058 LT.second = LT.second.changeVectorElementType(MVT::f16);
2059
2060 // Subvector extractions are free if they start at the beginning of a
2061 // vector and cheap if the subvectors are aligned.
2062 if (Kind == TTI::SK_ExtractSubvector && LT.second.isVector()) {
2063 int NumElts = LT.second.getVectorNumElements();
2064 if ((Index % NumElts) == 0)
2065 return TTI::TCC_Free;
2066 std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
2067 if (SubLT.second.isVector()) {
2068 int NumSubElts = SubLT.second.getVectorNumElements();
2069 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
2070 return SubLT.first;
2071 // Handle some cases for widening legalization. For now we only handle
2072 // cases where the original subvector was naturally aligned and evenly
2073 // fit in its legalized subvector type.
2074 // FIXME: Remove some of the alignment restrictions.
2075 // FIXME: We can use permq for 64-bit or larger extracts from 256-bit
2076 // vectors.
2077 int OrigSubElts = cast<FixedVectorType>(SubTp)->getNumElements();
2078 if (NumSubElts > OrigSubElts && (Index % OrigSubElts) == 0 &&
2079 (NumSubElts % OrigSubElts) == 0 &&
2080 LT.second.getVectorElementType() ==
2081 SubLT.second.getVectorElementType() &&
2082 LT.second.getVectorElementType().getSizeInBits() ==
2083 SrcTy->getElementType()->getPrimitiveSizeInBits()) {
2084 assert(NumElts >= NumSubElts && NumElts > OrigSubElts &&
2085 "Unexpected number of elements!");
2086 auto *VecTy = FixedVectorType::get(SrcTy->getElementType(),
2087 LT.second.getVectorNumElements());
2088 auto *SubTy = FixedVectorType::get(SrcTy->getElementType(),
2089 SubLT.second.getVectorNumElements());
2090 int ExtractIndex = alignDown((Index % NumElts), NumSubElts);
2091 InstructionCost ExtractCost =
2093 ExtractIndex, SubTy);
2094
2095 // If the original size is 32-bits or more, we can use pshufd. Otherwise
2096 // if we have SSSE3 we can use pshufb.
2097 if (SubTp->getPrimitiveSizeInBits() >= 32 || ST->hasSSSE3())
2098 return ExtractCost + 1; // pshufd or pshufb
2099
2100 assert(SubTp->getPrimitiveSizeInBits() == 16 &&
2101 "Unexpected vector size");
2102
2103 return ExtractCost + 2; // worst case pshufhw + pshufd
2104 }
2105 }
2106 // If the extract subvector is not optimal, treat it as single op shuffle.
2108 }
2109
2110 // Subvector insertions are cheap if the subvectors are aligned.
2111 // Note that in general, the insertion starting at the beginning of a vector
2112 // isn't free, because we need to preserve the rest of the wide vector,
2113 // but if the destination vector legalizes to the same width as the subvector
2114 // then the insertion will simplify to a (free) register copy.
2115 if (Kind == TTI::SK_InsertSubvector && LT.second.isVector()) {
2116 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(DstTy);
2117 int NumElts = DstLT.second.getVectorNumElements();
2118 std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
2119 if (SubLT.second.isVector()) {
2120 int NumSubElts = SubLT.second.getVectorNumElements();
2121 bool MatchingTypes =
2122 NumElts == NumSubElts &&
2123 (SubTp->getElementCount().getKnownMinValue() % NumSubElts) == 0;
2124 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
2125 return MatchingTypes ? TTI::TCC_Free : SubLT.first;
2126 }
2127
2128 // Attempt to match MOVSS (Idx == 0) or INSERTPS pattern. This will have
2129 // been matched by improveShuffleKindFromMask as a SK_InsertSubvector of
2130 // v1f32 (legalised to f32) into a v4f32.
2131 if (LT.first == 1 && LT.second == MVT::v4f32 && SubLT.first == 1 &&
2132 SubLT.second == MVT::f32 && (Index == 0 || ST->hasSSE41()))
2133 return 1;
2134
2135 // If the insertion is the lowest subvector then it will be blended
2136 // otherwise treat it like a 2-op shuffle.
2137 Kind =
2138 (Index == 0 && LT.first == 1) ? TTI::SK_Select : TTI::SK_PermuteTwoSrc;
2139 }
2140
2141 // Handle some common (illegal) sub-vector types as they are often very cheap
2142 // to shuffle even on targets without PSHUFB.
2143 EVT VT = TLI->getValueType(DL, SrcTy);
2144 if (VT.isSimple() && VT.isVector() && VT.getSizeInBits() < 128 &&
2145 !ST->hasSSSE3()) {
2146 static const CostKindTblEntry SSE2SubVectorShuffleTbl[] = {
2147 {TTI::SK_Broadcast, MVT::v4i16, {1,1,1,1}}, // pshuflw
2148 {TTI::SK_Broadcast, MVT::v2i16, {1,1,1,1}}, // pshuflw
2149 {TTI::SK_Broadcast, MVT::v8i8, {2,2,2,2}}, // punpck/pshuflw
2150 {TTI::SK_Broadcast, MVT::v4i8, {2,2,2,2}}, // punpck/pshuflw
2151 {TTI::SK_Broadcast, MVT::v2i8, {1,1,1,1}}, // punpck
2152
2153 {TTI::SK_Reverse, MVT::v4i16, {1,1,1,1}}, // pshuflw
2154 {TTI::SK_Reverse, MVT::v2i16, {1,1,1,1}}, // pshuflw
2155 {TTI::SK_Reverse, MVT::v4i8, {3,3,3,3}}, // punpck/pshuflw/packus
2156 {TTI::SK_Reverse, MVT::v2i8, {1,1,1,1}}, // punpck
2157
2158 {TTI::SK_Splice, MVT::v4i16, {2,2,2,2}}, // punpck+psrldq
2159 {TTI::SK_Splice, MVT::v2i16, {2,2,2,2}}, // punpck+psrldq
2160 {TTI::SK_Splice, MVT::v4i8, {2,2,2,2}}, // punpck+psrldq
2161 {TTI::SK_Splice, MVT::v2i8, {2,2,2,2}}, // punpck+psrldq
2162
2163 {TTI::SK_PermuteTwoSrc, MVT::v4i16, {2,2,2,2}}, // punpck/pshuflw
2164 {TTI::SK_PermuteTwoSrc, MVT::v2i16, {2,2,2,2}}, // punpck/pshuflw
2165 {TTI::SK_PermuteTwoSrc, MVT::v8i8, {7,7,7,7}}, // punpck/pshuflw
2166 {TTI::SK_PermuteTwoSrc, MVT::v4i8, {4,4,4,4}}, // punpck/pshuflw
2167 {TTI::SK_PermuteTwoSrc, MVT::v2i8, {2,2,2,2}}, // punpck
2168
2169 {TTI::SK_PermuteSingleSrc, MVT::v4i16, {1,1,1,1}}, // pshuflw
2170 {TTI::SK_PermuteSingleSrc, MVT::v2i16, {1,1,1,1}}, // pshuflw
2171 {TTI::SK_PermuteSingleSrc, MVT::v8i8, {5,5,5,5}}, // punpck/pshuflw
2172 {TTI::SK_PermuteSingleSrc, MVT::v4i8, {3,3,3,3}}, // punpck/pshuflw
2173 {TTI::SK_PermuteSingleSrc, MVT::v2i8, {1,1,1,1}}, // punpck
2174 };
2175
2176 if (ST->hasSSE2())
2177 if (const auto *Entry =
2178 CostTableLookup(SSE2SubVectorShuffleTbl, Kind, VT.getSimpleVT()))
2179 if (auto KindCost = Entry->Cost[CostKind])
2180 return LT.first * *KindCost;
2181 }
2182
2183 // We are going to permute multiple sources and the result will be in multiple
2184 // destinations. Providing an accurate cost only for splits where the element
2185 // type remains the same.
2186 if (LT.first != 1) {
2187 MVT LegalVT = LT.second;
2188 if (LegalVT.isVector() &&
2189 LegalVT.getVectorElementType().getSizeInBits() ==
2190 SrcTy->getElementType()->getPrimitiveSizeInBits() &&
2191 LegalVT.getVectorNumElements() <
2192 cast<FixedVectorType>(SrcTy)->getNumElements()) {
2193 unsigned VecTySize = DL.getTypeStoreSize(SrcTy);
2194 unsigned LegalVTSize = LegalVT.getStoreSize();
2195 // Number of source vectors after legalization:
2196 unsigned NumOfSrcs = (VecTySize + LegalVTSize - 1) / LegalVTSize;
2197 // Number of destination vectors after legalization:
2198 InstructionCost NumOfDests = LT.first;
2199
2200 auto *SingleOpTy = FixedVectorType::get(SrcTy->getElementType(),
2201 LegalVT.getVectorNumElements());
2202
2203 if (!Mask.empty() && NumOfDests.isValid()) {
2204 // Try to perform better estimation of the permutation.
2205 // 1. Split the source/destination vectors into real registers.
2206 // 2. Do the mask analysis to identify which real registers are
2207 // permuted. If more than 1 source registers are used for the
2208 // destination register building, the cost for this destination register
2209 // is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
2210 // source register is used, build mask and calculate the cost as a cost
2211 // of PermuteSingleSrc.
2212 // Also, for the single register permute we try to identify if the
2213 // destination register is just a copy of the source register or the
2214 // copy of the previous destination register (the cost is
2215 // TTI::TCC_Basic). If the source register is just reused, the cost for
2216 // this operation is TTI::TCC_Free.
2217 NumOfDests =
2219 FixedVectorType::get(SrcTy->getElementType(), Mask.size()))
2220 .first;
2221 unsigned E = NumOfDests.getValue();
2222 unsigned NormalizedVF =
2223 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
2224 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
2225 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
2226 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
2227 copy(Mask, NormalizedMask.begin());
2228 unsigned PrevSrcReg = 0;
2229 ArrayRef<int> PrevRegMask;
2232 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
2233 [this, SingleOpTy, CostKind, &PrevSrcReg, &PrevRegMask,
2234 &Cost](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
2235 if (!ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size())) {
2236 // Check if the previous register can be just copied to the next
2237 // one.
2238 if (PrevRegMask.empty() || PrevSrcReg != SrcReg ||
2239 PrevRegMask != RegMask)
2240 Cost +=
2242 SingleOpTy, CostKind, RegMask, 0, nullptr);
2243 else
2244 // Just a copy of previous destination register.
2246 return;
2247 }
2248 if (SrcReg != DestReg &&
2249 any_of(RegMask, not_equal_to(PoisonMaskElem))) {
2250 // Just a copy of the source register.
2252 }
2253 PrevSrcReg = SrcReg;
2254 PrevRegMask = RegMask;
2255 },
2256 [this, SingleOpTy, CostKind,
2257 &Cost](ArrayRef<int> RegMask, unsigned /*Unused*/,
2258 unsigned /*Unused*/, bool /*Unused*/) {
2260 SingleOpTy, CostKind, RegMask, 0, nullptr);
2261 });
2262 return Cost;
2263 }
2264
2265 InstructionCost NumOfShuffles = (NumOfSrcs - 1) * NumOfDests;
2266 return NumOfShuffles * getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
2267 SingleOpTy, CostKind, {}, 0,
2268 nullptr);
2269 }
2270
2271 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
2272 SubTp);
2273 }
2274
2275 // If we're just moving a single element around (probably as an alternative to
2276 // extracting it), we can assume this is cheap.
2277 if (LT.first == 1 && IsInLaneShuffle && IsSingleElementMask)
2278 return TTI::TCC_Basic;
2279
2280 static const CostKindTblEntry AVX512VBMIShuffleTbl[] = {
2281 { TTI::SK_Reverse, MVT::v64i8, { 1, 1, 1, 1 } }, // vpermb
2282 { TTI::SK_Reverse, MVT::v32i8, { 1, 1, 1, 1 } }, // vpermb
2283 { TTI::SK_PermuteSingleSrc, MVT::v64i8, { 1, 1, 1, 1 } }, // vpermb
2284 { TTI::SK_PermuteSingleSrc, MVT::v32i8, { 1, 1, 1, 1 } }, // vpermb
2285 { TTI::SK_PermuteTwoSrc, MVT::v64i8, { 2, 2, 2, 2 } }, // vpermt2b
2286 { TTI::SK_PermuteTwoSrc, MVT::v32i8, { 2, 2, 2, 2 } }, // vpermt2b
2287 { TTI::SK_PermuteTwoSrc, MVT::v16i8, { 2, 2, 2, 2 } } // vpermt2b
2288 };
2289
2290 if (ST->hasVBMI())
2291 if (const auto *Entry =
2292 CostTableLookup(AVX512VBMIShuffleTbl, Kind, LT.second))
2293 if (auto KindCost = Entry->Cost[CostKind])
2294 return LT.first * *KindCost;
2295
2296 static const CostKindTblEntry AVX512BWShuffleTbl[] = {
2297 { TTI::SK_Broadcast, MVT::v32i16, { 1, 3, 1, 1 } }, // vpbroadcastw
2298 { TTI::SK_Broadcast, MVT::v32f16, { 1, 3, 1, 1 } }, // vpbroadcastw
2299 { TTI::SK_Broadcast, MVT::v64i8, { 1, 3, 1, 1 } }, // vpbroadcastb
2300
2301 { TTI::SK_Reverse, MVT::v32i16, { 2, 6, 2, 4 } }, // vpermw
2302 { TTI::SK_Reverse, MVT::v32f16, { 2, 6, 2, 4 } }, // vpermw
2303 { TTI::SK_Reverse, MVT::v16i16, { 2, 2, 2, 2 } }, // vpermw
2304 { TTI::SK_Reverse, MVT::v16f16, { 2, 2, 2, 2 } }, // vpermw
2305 { TTI::SK_Reverse, MVT::v64i8, { 2, 9, 2, 3 } }, // pshufb + vshufi64x2
2306
2307 { TTI::SK_PermuteSingleSrc, MVT::v32i16, { 2, 2, 2, 2 } }, // vpermw
2308 { TTI::SK_PermuteSingleSrc, MVT::v32f16, { 2, 2, 2, 2 } }, // vpermw
2309 { TTI::SK_PermuteSingleSrc, MVT::v16i16, { 2, 2, 2, 2 } }, // vpermw
2310 { TTI::SK_PermuteSingleSrc, MVT::v16f16, { 2, 2, 2, 2 } }, // vpermw
2311 { TTI::SK_PermuteSingleSrc, MVT::v64i8, { 8, 8, 8, 8 } }, // extend to v32i16
2312
2313 { TTI::SK_PermuteTwoSrc, MVT::v32i16,{ 2, 2, 2, 2 } }, // vpermt2w
2314 { TTI::SK_PermuteTwoSrc, MVT::v32f16,{ 2, 2, 2, 2 } }, // vpermt2w
2315 { TTI::SK_PermuteTwoSrc, MVT::v16i16,{ 2, 2, 2, 2 } }, // vpermt2w
2316 { TTI::SK_PermuteTwoSrc, MVT::v8i16, { 2, 2, 2, 2 } }, // vpermt2w
2317 { TTI::SK_PermuteTwoSrc, MVT::v64i8, { 19, 19, 19, 19 } }, // 6 * v32i8 + 1
2318
2319 { TTI::SK_Select, MVT::v32i16, { 1, 1, 1, 1 } }, // vblendmw
2320 { TTI::SK_Select, MVT::v64i8, { 1, 1, 1, 1 } }, // vblendmb
2321
2322 { TTI::SK_Splice, MVT::v32i16, { 2, 2, 2, 2 } }, // vshufi64x2 + palignr
2323 { TTI::SK_Splice, MVT::v32f16, { 2, 2, 2, 2 } }, // vshufi64x2 + palignr
2324 { TTI::SK_Splice, MVT::v64i8, { 2, 2, 2, 2 } }, // vshufi64x2 + palignr
2325 };
2326
2327 if (ST->hasBWI())
2328 if (const auto *Entry =
2329 CostTableLookup(AVX512BWShuffleTbl, Kind, LT.second))
2330 if (auto KindCost = Entry->Cost[CostKind])
2331 return LT.first * *KindCost;
2332
2333 static const CostKindTblEntry AVX512InLaneShuffleTbl[] = {
2334 {TTI::SK_PermuteTwoSrc, MVT::v8f64, { 1, 3, 1, 1 } },
2335 {TTI::SK_PermuteTwoSrc, MVT::v16f32, { 1, 3, 1, 1 } },
2336 {TTI::SK_PermuteTwoSrc, MVT::v8i64, { 1, 3, 1, 1 } },
2337 {TTI::SK_PermuteTwoSrc, MVT::v16i32, { 1, 3, 1, 1 } },
2338 {TTI::SK_PermuteTwoSrc, MVT::v4f64, { 1, 3, 1, 1 } },
2339 {TTI::SK_PermuteTwoSrc, MVT::v8f32, { 1, 3, 1, 1 } },
2340 {TTI::SK_PermuteTwoSrc, MVT::v4i64, { 1, 3, 1, 1 } },
2341 {TTI::SK_PermuteTwoSrc, MVT::v8i32, { 1, 3, 1, 1 } },
2342 };
2343
2344 if (IsInLaneShuffle && ST->hasAVX512())
2345 if (const auto *Entry =
2346 CostTableLookup(AVX512InLaneShuffleTbl, Kind, LT.second))
2347 if (auto KindCost = Entry->Cost[CostKind])
2348 return LT.first * *KindCost;
2349
2350 static const CostKindTblEntry AVX512ShuffleTbl[] = {
2351 {TTI::SK_Broadcast, MVT::v8f64, { 1, 3, 1, 1 } }, // vbroadcastsd
2352 {TTI::SK_Broadcast, MVT::v4f64, { 1, 3, 1, 1 } }, // vbroadcastsd
2353 {TTI::SK_Broadcast, MVT::v16f32, { 1, 3, 1, 1 } }, // vbroadcastss
2354 {TTI::SK_Broadcast, MVT::v8f32, { 1, 3, 1, 1 } }, // vbroadcastss
2355 {TTI::SK_Broadcast, MVT::v8i64, { 1, 3, 1, 1 } }, // vpbroadcastq
2356 {TTI::SK_Broadcast, MVT::v4i64, { 1, 3, 1, 1 } }, // vpbroadcastq
2357 {TTI::SK_Broadcast, MVT::v16i32, { 1, 3, 1, 1 } }, // vpbroadcastd
2358 {TTI::SK_Broadcast, MVT::v8i32, { 1, 3, 1, 1 } }, // vpbroadcastd
2359 {TTI::SK_Broadcast, MVT::v32i16, { 1, 3, 1, 1 } }, // vpbroadcastw
2360 {TTI::SK_Broadcast, MVT::v16i16, { 1, 3, 1, 1 } }, // vpbroadcastw
2361 {TTI::SK_Broadcast, MVT::v32f16, { 1, 3, 1, 1 } }, // vpbroadcastw
2362 {TTI::SK_Broadcast, MVT::v16f16, { 1, 3, 1, 1 } }, // vpbroadcastw
2363 {TTI::SK_Broadcast, MVT::v64i8, { 1, 3, 1, 1 } }, // vpbroadcastb
2364 {TTI::SK_Broadcast, MVT::v32i8, { 1, 3, 1, 1 }}, // vpbroadcastb
2365
2366 {TTI::SK_Reverse, MVT::v8f64, { 1, 5, 2, 3 } }, // vpermpd
2367 {TTI::SK_Reverse, MVT::v16f32, { 1, 3, 2, 3 } }, // vpermps
2368 {TTI::SK_Reverse, MVT::v8i64, { 1, 5, 2, 3 } }, // vpermq
2369 {TTI::SK_Reverse, MVT::v16i32, { 1, 3, 2, 3 } }, // vpermd
2370 {TTI::SK_Reverse, MVT::v32i16, { 7, 7, 7, 7 } }, // per mca
2371 {TTI::SK_Reverse, MVT::v32f16, { 7, 7, 7, 7 } }, // per mca
2372 {TTI::SK_Reverse, MVT::v64i8, { 7, 7, 7, 7 } }, // per mca
2373
2374 {TTI::SK_Splice, MVT::v8f64, { 1, 1, 1, 1 } }, // vpalignd
2375 {TTI::SK_Splice, MVT::v4f64, { 1, 1, 1, 1 } }, // vpalignd
2376 {TTI::SK_Splice, MVT::v16f32, { 1, 1, 1, 1 } }, // vpalignd
2377 {TTI::SK_Splice, MVT::v8f32, { 1, 1, 1, 1 } }, // vpalignd
2378 {TTI::SK_Splice, MVT::v8i64, { 1, 1, 1, 1 } }, // vpalignd
2379 {TTI::SK_Splice, MVT::v4i64, { 1, 1, 1, 1 } }, // vpalignd
2380 {TTI::SK_Splice, MVT::v16i32, { 1, 1, 1, 1 } }, // vpalignd
2381 {TTI::SK_Splice, MVT::v8i32, { 1, 1, 1, 1 } }, // vpalignd
2382 {TTI::SK_Splice, MVT::v32i16, { 4, 4, 4, 4 } }, // split + palignr
2383 {TTI::SK_Splice, MVT::v32f16, { 4, 4, 4, 4 } }, // split + palignr
2384 {TTI::SK_Splice, MVT::v64i8, { 4, 4, 4, 4 } }, // split + palignr
2385
2386 {TTI::SK_PermuteSingleSrc, MVT::v8f64, { 1, 3, 1, 1 } }, // vpermpd
2387 {TTI::SK_PermuteSingleSrc, MVT::v4f64, { 1, 3, 1, 1 } }, // vpermpd
2388 {TTI::SK_PermuteSingleSrc, MVT::v2f64, { 1, 3, 1, 1 } }, // vpermpd
2389 {TTI::SK_PermuteSingleSrc, MVT::v16f32, { 1, 3, 1, 1 } }, // vpermps
2390 {TTI::SK_PermuteSingleSrc, MVT::v8f32, { 1, 3, 1, 1 } }, // vpermps
2391 {TTI::SK_PermuteSingleSrc, MVT::v4f32, { 1, 3, 1, 1 } }, // vpermps
2392 {TTI::SK_PermuteSingleSrc, MVT::v8i64, { 1, 3, 1, 1 } }, // vpermq
2393 {TTI::SK_PermuteSingleSrc, MVT::v4i64, { 1, 3, 1, 1 } }, // vpermq
2394 {TTI::SK_PermuteSingleSrc, MVT::v2i64, { 1, 3, 1, 1 } }, // vpermq
2395 {TTI::SK_PermuteSingleSrc, MVT::v16i32, { 1, 3, 1, 1 } }, // vpermd
2396 {TTI::SK_PermuteSingleSrc, MVT::v8i32, { 1, 3, 1, 1 } }, // vpermd
2397 {TTI::SK_PermuteSingleSrc, MVT::v4i32, { 1, 3, 1, 1 } }, // vpermd
2398 {TTI::SK_PermuteSingleSrc, MVT::v16i8, { 1, 3, 1, 1 } }, // pshufb
2399
2400 {TTI::SK_PermuteTwoSrc, MVT::v8f64, { 2, 3, 1, 1 } }, // vpermt2pd
2401 {TTI::SK_PermuteTwoSrc, MVT::v16f32, { 2, 3, 1, 1 } }, // vpermt2ps
2402 {TTI::SK_PermuteTwoSrc, MVT::v8i64, { 2, 3, 1, 1 } }, // vpermt2q
2403 {TTI::SK_PermuteTwoSrc, MVT::v16i32, { 2, 3, 1, 1 } }, // vpermt2d
2404 {TTI::SK_PermuteTwoSrc, MVT::v4f64, { 2, 3, 1, 1 } }, // vpermt2pd
2405 {TTI::SK_PermuteTwoSrc, MVT::v8f32, { 2, 3, 1, 1 } }, // vpermt2ps
2406 {TTI::SK_PermuteTwoSrc, MVT::v4i64, { 2, 3, 1, 1 } }, // vpermt2q
2407 {TTI::SK_PermuteTwoSrc, MVT::v8i32, { 2, 3, 1, 1 } }, // vpermt2d
2408 {TTI::SK_PermuteTwoSrc, MVT::v2f64, { 1, 3, 1, 1 } },
2409 {TTI::SK_PermuteTwoSrc, MVT::v4f32, { 1, 3, 1, 1 } },
2410 {TTI::SK_PermuteTwoSrc, MVT::v2i64, { 1, 3, 1, 1 } },
2411 {TTI::SK_PermuteTwoSrc, MVT::v4i32, { 1, 3, 1, 1 } },
2412
2413 // FIXME: This just applies the type legalization cost rules above
2414 // assuming these completely split.
2415 {TTI::SK_PermuteSingleSrc, MVT::v32i16, { 14, 14, 14, 14 } },
2416 {TTI::SK_PermuteSingleSrc, MVT::v32f16, { 14, 14, 14, 14 } },
2417 {TTI::SK_PermuteSingleSrc, MVT::v64i8, { 14, 14, 14, 14 } },
2418 {TTI::SK_PermuteTwoSrc, MVT::v32i16, { 42, 42, 42, 42 } },
2419 {TTI::SK_PermuteTwoSrc, MVT::v32f16, { 42, 42, 42, 42 } },
2420 {TTI::SK_PermuteTwoSrc, MVT::v64i8, { 42, 42, 42, 42 } },
2421
2422 {TTI::SK_Select, MVT::v32i16, { 1, 1, 1, 1 } }, // vpternlogq
2423 {TTI::SK_Select, MVT::v32f16, { 1, 1, 1, 1 } }, // vpternlogq
2424 {TTI::SK_Select, MVT::v64i8, { 1, 1, 1, 1 } }, // vpternlogq
2425 {TTI::SK_Select, MVT::v8f64, { 1, 1, 1, 1 } }, // vblendmpd
2426 {TTI::SK_Select, MVT::v16f32, { 1, 1, 1, 1 } }, // vblendmps
2427 {TTI::SK_Select, MVT::v8i64, { 1, 1, 1, 1 } }, // vblendmq
2428 {TTI::SK_Select, MVT::v16i32, { 1, 1, 1, 1 } }, // vblendmd
2429 };
2430
2431 if (ST->hasAVX512())
2432 if (const auto *Entry = CostTableLookup(AVX512ShuffleTbl, Kind, LT.second))
2433 if (auto KindCost = Entry->Cost[CostKind])
2434 return LT.first * *KindCost;
2435
2436 static const CostKindTblEntry AVX2InLaneShuffleTbl[] = {
2437 { TTI::SK_PermuteSingleSrc, MVT::v16i16, { 1, 1, 1, 1 } }, // vpshufb
2438 { TTI::SK_PermuteSingleSrc, MVT::v16f16, { 1, 1, 1, 1 } }, // vpshufb
2439 { TTI::SK_PermuteSingleSrc, MVT::v32i8, { 1, 1, 1, 1 } }, // vpshufb
2440
2441 { TTI::SK_Transpose, MVT::v4f64, { 1, 1, 1, 1 } }, // vshufpd/vunpck
2442 { TTI::SK_Transpose, MVT::v4i64, { 1, 1, 1, 1 } }, // vshufpd/vunpck
2443
2444 { TTI::SK_PermuteTwoSrc, MVT::v4f64, { 2, 2, 2, 2 } }, // 2*vshufpd + vblendpd
2445 { TTI::SK_PermuteTwoSrc, MVT::v8f32, { 2, 2, 2, 2 } }, // 2*vshufps + vblendps
2446 { TTI::SK_PermuteTwoSrc, MVT::v4i64, { 2, 2, 2, 2 } }, // 2*vpshufd + vpblendd
2447 { TTI::SK_PermuteTwoSrc, MVT::v8i32, { 2, 2, 2, 2 } }, // 2*vpshufd + vpblendd
2448 { TTI::SK_PermuteTwoSrc, MVT::v16i16, { 2, 2, 2, 2 } }, // 2*vpshufb + vpor
2449 { TTI::SK_PermuteTwoSrc, MVT::v16f16, { 2, 2, 2, 2 } }, // 2*vpshufb + vpor
2450 { TTI::SK_PermuteTwoSrc, MVT::v32i8, { 2, 2, 2, 2 } }, // 2*vpshufb + vpor
2451 };
2452
2453 if (IsInLaneShuffle && ST->hasAVX2())
2454 if (const auto *Entry =
2455 CostTableLookup(AVX2InLaneShuffleTbl, Kind, LT.second))
2456 if (auto KindCost = Entry->Cost[CostKind])
2457 return LT.first * *KindCost;
2458
2459 static const CostKindTblEntry AVX2ShuffleTbl[] = {
2460 { TTI::SK_Broadcast, MVT::v4f64, { 1, 3, 1, 2 } }, // vbroadcastpd
2461 { TTI::SK_Broadcast, MVT::v8f32, { 1, 3, 1, 2 } }, // vbroadcastps
2462 { TTI::SK_Broadcast, MVT::v4i64, { 1, 3, 1, 2 } }, // vpbroadcastq
2463 { TTI::SK_Broadcast, MVT::v8i32, { 1, 3, 1, 2 } }, // vpbroadcastd
2464 { TTI::SK_Broadcast, MVT::v16i16, { 1, 3, 1, 2 } }, // vpbroadcastw
2465 { TTI::SK_Broadcast, MVT::v8i16, { 1, 3, 1, 1 } }, // vpbroadcastw
2466 { TTI::SK_Broadcast, MVT::v16f16, { 1, 3, 1, 2 } }, // vpbroadcastw
2467 { TTI::SK_Broadcast, MVT::v8f16, { 1, 3, 1, 1 } }, // vpbroadcastw
2468 { TTI::SK_Broadcast, MVT::v32i8, { 1, 3, 1, 2 } }, // vpbroadcastb
2469 { TTI::SK_Broadcast, MVT::v16i8, { 1, 3, 1, 1 } }, // vpbroadcastb
2470
2471 { TTI::SK_Reverse, MVT::v4f64, { 1, 6, 1, 2 } }, // vpermpd
2472 { TTI::SK_Reverse, MVT::v8f32, { 2, 7, 2, 4 } }, // vpermps
2473 { TTI::SK_Reverse, MVT::v4i64, { 1, 6, 1, 2 } }, // vpermq
2474 { TTI::SK_Reverse, MVT::v8i32, { 2, 7, 2, 4 } }, // vpermd
2475 { TTI::SK_Reverse, MVT::v16i16, { 2, 9, 2, 4 } }, // vperm2i128 + pshufb
2476 { TTI::SK_Reverse, MVT::v16f16, { 2, 9, 2, 4 } }, // vperm2i128 + pshufb
2477 { TTI::SK_Reverse, MVT::v32i8, { 2, 9, 2, 4 } }, // vperm2i128 + pshufb
2478
2479 { TTI::SK_Select, MVT::v16i16, { 1, 1, 1, 1 } }, // vpblendvb
2480 { TTI::SK_Select, MVT::v16f16, { 1, 1, 1, 1 } }, // vpblendvb
2481 { TTI::SK_Select, MVT::v32i8, { 1, 1, 1, 1 } }, // vpblendvb
2482
2483 { TTI::SK_Splice, MVT::v8i32, { 2, 2, 2, 2 } }, // vperm2i128 + vpalignr
2484 { TTI::SK_Splice, MVT::v8f32, { 2, 2, 2, 2 } }, // vperm2i128 + vpalignr
2485 { TTI::SK_Splice, MVT::v16i16, { 2, 2, 2, 2 } }, // vperm2i128 + vpalignr
2486 { TTI::SK_Splice, MVT::v16f16, { 2, 2, 2, 2 } }, // vperm2i128 + vpalignr
2487 { TTI::SK_Splice, MVT::v32i8, { 2, 2, 2, 2 } }, // vperm2i128 + vpalignr
2488
2489 { TTI::SK_PermuteSingleSrc, MVT::v4f64, { 1, 1, 1, 1 } }, // vpermpd
2490 { TTI::SK_PermuteSingleSrc, MVT::v8f32, { 1, 1, 1, 1 } }, // vpermps
2491 { TTI::SK_PermuteSingleSrc, MVT::v4i64, { 1, 1, 1, 1 } }, // vpermq
2492 { TTI::SK_PermuteSingleSrc, MVT::v8i32, { 1, 1, 1, 1 } }, // vpermd
2493 { TTI::SK_PermuteSingleSrc, MVT::v16i16, { 4, 4, 4, 4 } },
2494 { TTI::SK_PermuteSingleSrc, MVT::v16f16, { 4, 4, 4, 4 } },
2495 { TTI::SK_PermuteSingleSrc, MVT::v32i8, { 4, 4, 4, 4 } },
2496
2497 { TTI::SK_PermuteTwoSrc, MVT::v4f64, { 3, 3, 3, 3 } }, // 2*vpermpd + vblendpd
2498 { TTI::SK_PermuteTwoSrc, MVT::v8f32, { 3, 3, 3, 3 } }, // 2*vpermps + vblendps
2499 { TTI::SK_PermuteTwoSrc, MVT::v4i64, { 3, 3, 3, 3 } }, // 2*vpermq + vpblendd
2500 { TTI::SK_PermuteTwoSrc, MVT::v8i32, { 3, 3, 3, 3 } }, // 2*vpermd + vpblendd
2501 { TTI::SK_PermuteTwoSrc, MVT::v16i16, { 7, 7, 7, 7 } },
2502 { TTI::SK_PermuteTwoSrc, MVT::v16f16, { 7, 7, 7, 7 } },
2503 { TTI::SK_PermuteTwoSrc, MVT::v32i8, { 7, 7, 7, 7 } },
2504 };
2505
2506 if (ST->hasAVX2())
2507 if (const auto *Entry = CostTableLookup(AVX2ShuffleTbl, Kind, LT.second))
2508 if (auto KindCost = Entry->Cost[CostKind])
2509 return LT.first * *KindCost;
2510
2511 static const CostKindTblEntry XOPShuffleTbl[] = {
2512 { TTI::SK_PermuteSingleSrc, MVT::v4f64, { 2, 2, 2, 2 } }, // vperm2f128 + vpermil2pd
2513 { TTI::SK_PermuteSingleSrc, MVT::v8f32, { 2, 2, 2, 2 } }, // vperm2f128 + vpermil2ps
2514 { TTI::SK_PermuteSingleSrc, MVT::v4i64, { 2, 2, 2, 2 } }, // vperm2f128 + vpermil2pd
2515 { TTI::SK_PermuteSingleSrc, MVT::v8i32, { 2, 2, 2, 2 } }, // vperm2f128 + vpermil2ps
2516 { TTI::SK_PermuteSingleSrc, MVT::v16i16,{ 4, 4, 4, 4 } }, // vextractf128 + 2*vpperm
2517 // + vinsertf128
2518 { TTI::SK_PermuteSingleSrc, MVT::v32i8, { 4, 4, 4, 4 } }, // vextractf128 + 2*vpperm
2519 // + vinsertf128
2520
2521 { TTI::SK_PermuteTwoSrc, MVT::v16i16, { 9, 9, 9, 9 } }, // 2*vextractf128 + 6*vpperm
2522 // + vinsertf128
2523
2524 { TTI::SK_PermuteTwoSrc, MVT::v8i16, { 1, 1, 1, 1 } }, // vpperm
2525 { TTI::SK_PermuteTwoSrc, MVT::v32i8, { 9, 9, 9, 9 } }, // 2*vextractf128 + 6*vpperm
2526 // + vinsertf128
2527 { TTI::SK_PermuteTwoSrc, MVT::v16i8, { 1, 1, 1, 1 } }, // vpperm
2528 };
2529
2530 if (ST->hasXOP())
2531 if (const auto *Entry = CostTableLookup(XOPShuffleTbl, Kind, LT.second))
2532 if (auto KindCost = Entry->Cost[CostKind])
2533 return LT.first * *KindCost;
2534
2535 static const CostKindTblEntry AVX1InLaneShuffleTbl[] = {
2536 { TTI::SK_PermuteSingleSrc, MVT::v4f64, { 1, 1, 1, 1 } }, // vpermilpd
2537 { TTI::SK_PermuteSingleSrc, MVT::v4i64, { 1, 1, 1, 1 } }, // vpermilpd
2538 { TTI::SK_PermuteSingleSrc, MVT::v8f32, { 1, 1, 1, 1 } }, // vpermilps
2539 { TTI::SK_PermuteSingleSrc, MVT::v8i32, { 1, 1, 1, 1 } }, // vpermilps
2540
2541 { TTI::SK_PermuteSingleSrc, MVT::v16i16, { 4, 4, 4, 4 } }, // vextractf128 + 2*pshufb
2542 // + vpor + vinsertf128
2543 { TTI::SK_PermuteSingleSrc, MVT::v16f16, { 4, 4, 4, 4 } }, // vextractf128 + 2*pshufb
2544 // + vpor + vinsertf128
2545 { TTI::SK_PermuteSingleSrc, MVT::v32i8, { 4, 4, 4, 4 } }, // vextractf128 + 2*pshufb
2546 // + vpor + vinsertf128
2547
2548 { TTI::SK_Transpose, MVT::v4f64, { 1, 1, 1, 1 } }, // vshufpd/vunpck
2549 { TTI::SK_Transpose, MVT::v4i64, { 1, 1, 1, 1 } }, // vshufpd/vunpck
2550
2551 { TTI::SK_PermuteTwoSrc, MVT::v4f64, { 2, 2, 2, 2 } }, // 2*vshufpd + vblendpd
2552 { TTI::SK_PermuteTwoSrc, MVT::v8f32, { 2, 2, 2, 2 } }, // 2*vshufps + vblendps
2553 { TTI::SK_PermuteTwoSrc, MVT::v4i64, { 2, 2, 2, 2 } }, // 2*vpermilpd + vblendpd
2554 { TTI::SK_PermuteTwoSrc, MVT::v8i32, { 2, 2, 2, 2 } }, // 2*vpermilps + vblendps
2555 { TTI::SK_PermuteTwoSrc, MVT::v16i16, { 9, 9, 9, 9 } }, // 2*vextractf128 + 4*pshufb
2556 // + 2*vpor + vinsertf128
2557 { TTI::SK_PermuteTwoSrc, MVT::v16f16, { 9, 9, 9, 9 } }, // 2*vextractf128 + 4*pshufb
2558 // + 2*vpor + vinsertf128
2559 { TTI::SK_PermuteTwoSrc, MVT::v32i8, { 9, 9, 9, 9 } }, // 2*vextractf128 + 4*pshufb
2560 // + 2*vpor + vinsertf128
2561 };
2562
2563 if (IsInLaneShuffle && ST->hasAVX())
2564 if (const auto *Entry =
2565 CostTableLookup(AVX1InLaneShuffleTbl, Kind, LT.second))
2566 if (auto KindCost = Entry->Cost[CostKind])
2567 return LT.first * *KindCost;
2568
2569 static const CostKindTblEntry AVX1ShuffleTbl[] = {
2570 {TTI::SK_Broadcast, MVT::v4f64, {2,3,2,3}}, // vperm2f128 + vpermilpd
2571 {TTI::SK_Broadcast, MVT::v8f32, {2,3,2,3}}, // vperm2f128 + vpermilps
2572 {TTI::SK_Broadcast, MVT::v4i64, {2,3,2,3}}, // vperm2f128 + vpermilpd
2573 {TTI::SK_Broadcast, MVT::v8i32, {2,3,2,3}}, // vperm2f128 + vpermilps
2574 {TTI::SK_Broadcast, MVT::v16i16, {2,3,3,4}}, // vpshuflw + vpshufd + vinsertf128
2575 {TTI::SK_Broadcast, MVT::v16f16, {2,3,3,4}}, // vpshuflw + vpshufd + vinsertf128
2576 {TTI::SK_Broadcast, MVT::v32i8, {3,4,3,6}}, // vpshufb + vinsertf128
2577
2578 {TTI::SK_Reverse, MVT::v4f64, {2,6,2,2}}, // vperm2f128 + vpermilpd
2579 {TTI::SK_Reverse, MVT::v8f32, {2,7,2,4}}, // vperm2f128 + vpermilps
2580 {TTI::SK_Reverse, MVT::v4i64, {2,6,2,2}}, // vperm2f128 + vpermilpd
2581 {TTI::SK_Reverse, MVT::v8i32, {2,7,2,4}}, // vperm2f128 + vpermilps
2582 {TTI::SK_Reverse, MVT::v16i16, {2,9,5,5}}, // vextractf128 + 2*pshufb
2583 // + vinsertf128
2584 {TTI::SK_Reverse, MVT::v16f16, {2,9,5,5}}, // vextractf128 + 2*pshufb
2585 // + vinsertf128
2586 {TTI::SK_Reverse, MVT::v32i8, {2,9,5,5}}, // vextractf128 + 2*pshufb
2587 // + vinsertf128
2588
2589 {TTI::SK_Select, MVT::v4i64, {1,1,1,1}}, // vblendpd
2590 {TTI::SK_Select, MVT::v4f64, {1,1,1,1}}, // vblendpd
2591 {TTI::SK_Select, MVT::v8i32, {1,1,1,1}}, // vblendps
2592 {TTI::SK_Select, MVT::v8f32, {1,1,1,1}}, // vblendps
2593 {TTI::SK_Select, MVT::v16i16, {3,3,3,3}}, // vpand + vpandn + vpor
2594 {TTI::SK_Select, MVT::v16f16, {3,3,3,3}}, // vpand + vpandn + vpor
2595 {TTI::SK_Select, MVT::v32i8, {3,3,3,3}}, // vpand + vpandn + vpor
2596
2597 {TTI::SK_Splice, MVT::v4i64, {2,2,2,2}}, // vperm2f128 + shufpd
2598 {TTI::SK_Splice, MVT::v4f64, {2,2,2,2}}, // vperm2f128 + shufpd
2599 {TTI::SK_Splice, MVT::v8i32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2600 {TTI::SK_Splice, MVT::v8f32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2601 {TTI::SK_Splice, MVT::v16i16, {5,5,5,5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2602 {TTI::SK_Splice, MVT::v16f16, {5,5,5,5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2603 {TTI::SK_Splice, MVT::v32i8, {5,5,5,5}}, // 2*vperm2f128 + 2*vpalignr + vinsertf128
2604
2605 {TTI::SK_PermuteSingleSrc, MVT::v4f64, {2,2,2,2}}, // vperm2f128 + vshufpd
2606 {TTI::SK_PermuteSingleSrc, MVT::v4i64, {2,2,2,2}}, // vperm2f128 + vshufpd
2607 {TTI::SK_PermuteSingleSrc, MVT::v8f32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2608 {TTI::SK_PermuteSingleSrc, MVT::v8i32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2609 {TTI::SK_PermuteSingleSrc, MVT::v16i16,{8,8,8,8}}, // vextractf128 + 4*pshufb
2610 // + 2*por + vinsertf128
2611 {TTI::SK_PermuteSingleSrc, MVT::v16f16,{8,8,8,8}}, // vextractf128 + 4*pshufb
2612 // + 2*por + vinsertf128
2613 {TTI::SK_PermuteSingleSrc, MVT::v32i8, {8,8,8,8}}, // vextractf128 + 4*pshufb
2614 // + 2*por + vinsertf128
2615
2616 {TTI::SK_PermuteTwoSrc, MVT::v4f64, {3,3,3,3}}, // 2*vperm2f128 + vshufpd
2617 {TTI::SK_PermuteTwoSrc, MVT::v4i64, {3,3,3,3}}, // 2*vperm2f128 + vshufpd
2618 {TTI::SK_PermuteTwoSrc, MVT::v8f32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2619 {TTI::SK_PermuteTwoSrc, MVT::v8i32, {4,4,4,4}}, // 2*vperm2f128 + 2*vshufps
2620 {TTI::SK_PermuteTwoSrc, MVT::v16i16,{15,15,15,15}}, // 2*vextractf128 + 8*pshufb
2621 // + 4*por + vinsertf128
2622 {TTI::SK_PermuteTwoSrc, MVT::v16f16,{15,15,15,15}}, // 2*vextractf128 + 8*pshufb
2623 // + 4*por + vinsertf128
2624 {TTI::SK_PermuteTwoSrc, MVT::v32i8, {15,15,15,15}}, // 2*vextractf128 + 8*pshufb
2625 // + 4*por + vinsertf128
2626 };
2627
2628 if (ST->hasAVX())
2629 if (const auto *Entry = CostTableLookup(AVX1ShuffleTbl, Kind, LT.second))
2630 if (auto KindCost = Entry->Cost[CostKind])
2631 return LT.first * *KindCost;
2632
2633 static const CostKindTblEntry SSE41ShuffleTbl[] = {
2634 {TTI::SK_Select, MVT::v2i64, {1,1,1,1}}, // pblendw
2635 {TTI::SK_Select, MVT::v2f64, {1,1,1,1}}, // movsd
2636 {TTI::SK_Select, MVT::v4i32, {1,1,1,1}}, // pblendw
2637 {TTI::SK_Select, MVT::v4f32, {1,1,1,1}}, // blendps
2638 {TTI::SK_Select, MVT::v8i16, {1,1,1,1}}, // pblendw
2639 {TTI::SK_Select, MVT::v8f16, {1,1,1,1}}, // pblendw
2640 {TTI::SK_Select, MVT::v16i8, {1,1,1,1}} // pblendvb
2641 };
2642
2643 if (ST->hasSSE41())
2644 if (const auto *Entry = CostTableLookup(SSE41ShuffleTbl, Kind, LT.second))
2645 if (auto KindCost = Entry->Cost[CostKind])
2646 return LT.first * *KindCost;
2647
2648 static const CostKindTblEntry SSSE3ShuffleTbl[] = {
2649 {TTI::SK_Broadcast, MVT::v8i16, {1, 3, 2, 2}}, // pshufb
2650 {TTI::SK_Broadcast, MVT::v8f16, {1, 3, 2, 2}}, // pshufb
2651 {TTI::SK_Broadcast, MVT::v16i8, {1, 3, 2, 2}}, // pshufb
2652
2653 {TTI::SK_Reverse, MVT::v8i16, {1, 2, 1, 2}}, // pshufb
2654 {TTI::SK_Reverse, MVT::v8f16, {1, 2, 1, 2}}, // pshufb
2655 {TTI::SK_Reverse, MVT::v16i8, {1, 2, 1, 2}}, // pshufb
2656
2657 {TTI::SK_Splice, MVT::v4i32, {1, 1, 1, 1}}, // palignr
2658 {TTI::SK_Splice, MVT::v4f32, {1, 1, 1, 1}}, // palignr
2659 {TTI::SK_Splice, MVT::v8i16, {1, 1, 1, 1}}, // palignr
2660 {TTI::SK_Splice, MVT::v8f16, {1, 1, 1, 1}}, // palignr
2661 {TTI::SK_Splice, MVT::v16i8, {1, 1, 1, 1}}, // palignr
2662
2663 {TTI::SK_PermuteSingleSrc, MVT::v8i16, {1, 1, 1, 1}}, // pshufb
2664 {TTI::SK_PermuteSingleSrc, MVT::v8f16, {1, 1, 1, 1}}, // pshufb
2665 {TTI::SK_PermuteSingleSrc, MVT::v16i8, {1, 1, 1, 1}}, // pshufb
2666
2667 {TTI::SK_PermuteTwoSrc, MVT::v8i16, {3, 3, 3, 3}}, // 2*pshufb + por
2668 {TTI::SK_PermuteTwoSrc, MVT::v8f16, {3, 3, 3, 3}}, // 2*pshufb + por
2669 {TTI::SK_PermuteTwoSrc, MVT::v16i8, {3, 3, 3, 3}}, // 2*pshufb + por
2670 };
2671
2672 if (ST->hasSSSE3())
2673 if (const auto *Entry = CostTableLookup(SSSE3ShuffleTbl, Kind, LT.second))
2674 if (auto KindCost = Entry->Cost[CostKind])
2675 return LT.first * *KindCost;
2676
2677 static const CostKindTblEntry SSE2ShuffleTbl[] = {
2678 {TTI::SK_Broadcast, MVT::v2f64, {1, 1, 1, 1}}, // shufpd
2679 {TTI::SK_Broadcast, MVT::v2i64, {1, 1, 1, 1}}, // pshufd
2680 {TTI::SK_Broadcast, MVT::v4i32, {1, 1, 1, 1}}, // pshufd
2681 {TTI::SK_Broadcast, MVT::v8i16, {1, 2, 2, 2}}, // pshuflw + pshufd
2682 {TTI::SK_Broadcast, MVT::v8f16, {1, 2, 2, 2}}, // pshuflw + pshufd
2683 {TTI::SK_Broadcast, MVT::v16i8, {2, 3, 3, 4}}, // unpck + pshuflw + pshufd
2684
2685 {TTI::SK_Reverse, MVT::v2f64, {1, 1, 1, 1}}, // shufpd
2686 {TTI::SK_Reverse, MVT::v2i64, {1, 1, 1, 1}}, // pshufd
2687 {TTI::SK_Reverse, MVT::v4i32, {1, 1, 1, 1}}, // pshufd
2688 {TTI::SK_Reverse, MVT::v8i16, {2, 3, 3, 3}}, // pshuflw + pshufhw + pshufd
2689 {TTI::SK_Reverse, MVT::v8f16, {2, 3, 3, 3}}, // pshuflw + pshufhw + pshufd
2690 {TTI::SK_Reverse, MVT::v16i8, {5, 6,11,11}}, // 2*pshuflw + 2*pshufhw
2691 // + 2*pshufd + 2*unpck + packus
2692
2693 {TTI::SK_Select, MVT::v2i64, {1, 1, 1, 1}}, // movsd
2694 {TTI::SK_Select, MVT::v2f64, {1, 1, 1, 1}}, // movsd
2695 {TTI::SK_Select, MVT::v4i32, {2, 2, 2, 2}}, // 2*shufps
2696 {TTI::SK_Select, MVT::v8i16, {2, 2, 3, 3}}, // pand + pandn + por
2697 {TTI::SK_Select, MVT::v8f16, {2, 2, 3, 3}}, // pand + pandn + por
2698 {TTI::SK_Select, MVT::v16i8, {2, 2, 3, 3}}, // pand + pandn + por
2699
2700 {TTI::SK_Splice, MVT::v2i64, {1, 1, 1, 1}}, // shufpd
2701 {TTI::SK_Splice, MVT::v2f64, {1, 1, 1, 1}}, // shufpd
2702 {TTI::SK_Splice, MVT::v4i32, {2, 2, 2, 2}}, // 2*{unpck,movsd,pshufd}
2703 {TTI::SK_Splice, MVT::v8i16, {3, 3, 3, 3}}, // psrldq + psrlldq + por
2704 {TTI::SK_Splice, MVT::v8f16, {3, 3, 3, 3}}, // psrldq + psrlldq + por
2705 {TTI::SK_Splice, MVT::v16i8, {3, 3, 3, 3}}, // psrldq + psrlldq + por
2706
2707 {TTI::SK_PermuteSingleSrc, MVT::v2f64, {1, 1, 1, 1}}, // shufpd
2708 {TTI::SK_PermuteSingleSrc, MVT::v2i64, {1, 1, 1, 1}}, // pshufd
2709 {TTI::SK_PermuteSingleSrc, MVT::v4i32, {1, 1, 1, 1}}, // pshufd
2710 {TTI::SK_PermuteSingleSrc, MVT::v8i16, {3, 5, 5, 5}}, // 2*pshuflw + 2*pshufhw
2711 // + pshufd/unpck
2712 {TTI::SK_PermuteSingleSrc, MVT::v8f16, {3, 5, 5, 5}}, // 2*pshuflw + 2*pshufhw
2713 // + pshufd/unpck
2714 {TTI::SK_PermuteSingleSrc, MVT::v16i8, {8, 10, 10, 10}}, // 2*pshuflw + 2*pshufhw
2715 // + 2*pshufd + 2*unpck + 2*packus
2716
2717 {TTI::SK_PermuteTwoSrc, MVT::v2f64, {1, 1, 1, 1}}, // shufpd
2718 {TTI::SK_PermuteTwoSrc, MVT::v2i64, {1, 1, 1, 1}}, // shufpd
2719 {TTI::SK_PermuteTwoSrc, MVT::v4i32, {2, 2, 2, 2}}, // 2*{unpck,movsd,pshufd}
2720 {TTI::SK_PermuteTwoSrc, MVT::v8i16, {6, 8, 8, 8}}, // blend+permute
2721 {TTI::SK_PermuteTwoSrc, MVT::v8f16, {6, 8, 8, 8}}, // blend+permute
2722 {TTI::SK_PermuteTwoSrc, MVT::v16i8, {11, 13, 13, 13}}, // blend+permute
2723 };
2724
2725 static const CostTblEntry SSE3BroadcastLoadTbl[] = {
2726 {TTI::SK_Broadcast, MVT::v2f64, 0}, // broadcast handled by movddup
2727 };
2728
2729 if (ST->hasSSE2()) {
2730 bool IsLoad =
2731 llvm::any_of(Args, [](const auto &V) { return isa<LoadInst>(V); });
2732 if (ST->hasSSE3() && IsLoad)
2733 if (const auto *Entry =
2734 CostTableLookup(SSE3BroadcastLoadTbl, Kind, LT.second)) {
2735 assert(isLegalBroadcastLoad(SrcTy->getElementType(),
2736 LT.second.getVectorElementCount()) &&
2737 "Table entry missing from isLegalBroadcastLoad()");
2738 return LT.first * Entry->Cost;
2739 }
2740
2741 if (const auto *Entry = CostTableLookup(SSE2ShuffleTbl, Kind, LT.second))
2742 if (auto KindCost = Entry->Cost[CostKind])
2743 return LT.first * *KindCost;
2744 }
2745
2746 static const CostKindTblEntry SSE1ShuffleTbl[] = {
2747 { TTI::SK_Broadcast, MVT::v4f32, {1,1,1,1} }, // shufps
2748 { TTI::SK_Reverse, MVT::v4f32, {1,1,1,1} }, // shufps
2749 { TTI::SK_Select, MVT::v4f32, {2,2,2,2} }, // 2*shufps
2750 { TTI::SK_Splice, MVT::v4f32, {2,2,2,2} }, // 2*shufps
2751 { TTI::SK_PermuteSingleSrc, MVT::v4f32, {1,1,1,1} }, // shufps
2752 { TTI::SK_PermuteTwoSrc, MVT::v4f32, {2,2,2,2} }, // 2*shufps
2753 };
2754
2755 if (ST->hasSSE1()) {
2756 if (LT.first == 1 && LT.second == MVT::v4f32 && Mask.size() == 4) {
2757 // SHUFPS: both pairs must come from the same source register.
2758 auto MatchSHUFPS = [](int X, int Y) {
2759 return X < 0 || Y < 0 || ((X & 4) == (Y & 4));
2760 };
2761 if (MatchSHUFPS(Mask[0], Mask[1]) && MatchSHUFPS(Mask[2], Mask[3]))
2762 return 1;
2763 }
2764 if (const auto *Entry = CostTableLookup(SSE1ShuffleTbl, Kind, LT.second))
2765 if (auto KindCost = Entry->Cost[CostKind])
2766 return LT.first * *KindCost;
2767 }
2768
2769 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
2770 SubTp);
2771}
2772
2774 Type *Src,
2777 const Instruction *I) const {
2778 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2779 assert(ISD && "Invalid opcode");
2780
2781 // A narrow (i8/i16) zero-extension used as a GEP *index* can be folded into
2782 // the addressing mode of the consuming memory op, but only if the source is
2783 // already materialised zero-extended in a full register. X86's SIB form
2784 // [base + index*scale + disp] reads the index at full width and does NOT
2785 // zero-extend a narrow index (unlike AArch64's uxtw-extended addressing), so
2786 // a "dirty" narrow source (e.g. an i16 add result used only as an index)
2787 // still needs a dedicated movzx and is not free. Price it as free only with
2788 // positive evidence that no movzx is required.
2789 if (ISD == ISD::ZERO_EXTEND && I && I->hasOneUse() && Src->isIntegerTy() &&
2790 Src->getScalarSizeInBits() < 32) {
2791 const Use &U = *I->use_begin();
2792 if (isa<GetElementPtrInst>(U.getUser()) &&
2793 U.getOperandNo() != GetElementPtrInst::getPointerOperandIndex()) {
2794 const Value *Op = I->getOperand(0);
2795 // Clean sources: an extending load, a zeroext argument, or a value whose
2796 // high bits are provably zero (e.g. from a shift/mask). These mirror the
2797 // proof-based reasoning the middle end uses elsewhere (ValueTracking and
2798 // InstCombine's canEvaluateZExtd); we intentionally do NOT treat a merely
2799 // multiply-used operand as clean, since that is a guess rather than
2800 // proof.
2801 if (isa<LoadInst>(Op))
2802 return TTI::TCC_Free;
2803 if (const auto *A = dyn_cast<Argument>(Op))
2804 if (A->hasAttribute(Attribute::ZExt))
2805 return TTI::TCC_Free;
2806 if (computeKnownBits(Op, I->getDataLayout(), /*AC=*/nullptr, I)
2807 .countMinLeadingZeros() > 0)
2808 return TTI::TCC_Free;
2809 }
2810 }
2811
2812 // The cost tables include both specific, custom (non-legal) src/dst type
2813 // conversions and generic, legalized types. We test for customs first, before
2814 // falling back to legalization.
2815 // FIXME: Need a better design of the cost table to handle non-simple types of
2816 // potential massive combinations (elem_num x src_type x dst_type).
2817 static const TypeConversionCostKindTblEntry AVX512BWConversionTbl[]{
2818 { ISD::SIGN_EXTEND, MVT::v32i16, MVT::v32i8, { 1, 1, 1, 1 } },
2819 { ISD::ZERO_EXTEND, MVT::v32i16, MVT::v32i8, { 1, 1, 1, 1 } },
2820
2821 // Mask sign extend has an instruction.
2822 { ISD::SIGN_EXTEND, MVT::v2i8, MVT::v2i1, { 1, 1, 1, 1 } },
2823 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v2i1, { 1, 1, 1, 1 } },
2824 { ISD::SIGN_EXTEND, MVT::v2i16, MVT::v2i1, { 1, 1, 1, 1 } },
2825 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v2i1, { 1, 1, 1, 1 } },
2826 { ISD::SIGN_EXTEND, MVT::v4i8, MVT::v4i1, { 1, 1, 1, 1 } },
2827 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v4i1, { 1, 1, 1, 1 } },
2828 { ISD::SIGN_EXTEND, MVT::v4i16, MVT::v4i1, { 1, 1, 1, 1 } },
2829 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v4i1, { 1, 1, 1, 1 } },
2830 { ISD::SIGN_EXTEND, MVT::v8i8, MVT::v8i1, { 1, 1, 1, 1 } },
2831 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v8i1, { 1, 1, 1, 1 } },
2832 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v8i1, { 1, 1, 1, 1 } },
2833 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v16i1, { 1, 1, 1, 1 } },
2834 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, { 1, 1, 1, 1 } },
2835 { ISD::SIGN_EXTEND, MVT::v32i8, MVT::v32i1, { 1, 1, 1, 1 } },
2836 { ISD::SIGN_EXTEND, MVT::v32i16, MVT::v32i1, { 1, 1, 1, 1 } },
2837 { ISD::SIGN_EXTEND, MVT::v64i8, MVT::v64i1, { 1, 1, 1, 1 } },
2838 { ISD::SIGN_EXTEND, MVT::v32i16, MVT::v64i1, { 1, 1, 1, 1 } },
2839
2840 // Mask zero extend is a sext + shift.
2841 { ISD::ZERO_EXTEND, MVT::v2i8, MVT::v2i1, { 2, 1, 1, 1 } },
2842 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v2i1, { 2, 1, 1, 1 } },
2843 { ISD::ZERO_EXTEND, MVT::v2i16, MVT::v2i1, { 2, 1, 1, 1 } },
2844 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v2i1, { 2, 1, 1, 1 } },
2845 { ISD::ZERO_EXTEND, MVT::v4i8, MVT::v4i1, { 2, 1, 1, 1 } },
2846 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v4i1, { 2, 1, 1, 1 } },
2847 { ISD::ZERO_EXTEND, MVT::v4i16, MVT::v4i1, { 2, 1, 1, 1 } },
2848 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v4i1, { 2, 1, 1, 1 } },
2849 { ISD::ZERO_EXTEND, MVT::v8i8, MVT::v8i1, { 2, 1, 1, 1 } },
2850 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v8i1, { 2, 1, 1, 1 } },
2851 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v8i1, { 2, 1, 1, 1 } },
2852 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v16i1, { 2, 1, 1, 1 } },
2853 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, { 2, 1, 1, 1 } },
2854 { ISD::ZERO_EXTEND, MVT::v32i8, MVT::v32i1, { 2, 1, 1, 1 } },
2855 { ISD::ZERO_EXTEND, MVT::v32i16, MVT::v32i1, { 2, 1, 1, 1 } },
2856 { ISD::ZERO_EXTEND, MVT::v64i8, MVT::v64i1, { 2, 1, 1, 1 } },
2857 { ISD::ZERO_EXTEND, MVT::v32i16, MVT::v64i1, { 2, 1, 1, 1 } },
2858
2859 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 2, 1, 1, 1 } },
2860 { ISD::TRUNCATE, MVT::v2i1, MVT::v16i8, { 2, 1, 1, 1 } },
2861 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 2, 1, 1, 1 } },
2862 { ISD::TRUNCATE, MVT::v2i1, MVT::v8i16, { 2, 1, 1, 1 } },
2863 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 2, 1, 1, 1 } },
2864 { ISD::TRUNCATE, MVT::v4i1, MVT::v16i8, { 2, 1, 1, 1 } },
2865 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i16, { 2, 1, 1, 1 } },
2866 { ISD::TRUNCATE, MVT::v4i1, MVT::v8i16, { 2, 1, 1, 1 } },
2867 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i8, { 2, 1, 1, 1 } },
2868 { ISD::TRUNCATE, MVT::v8i1, MVT::v16i8, { 2, 1, 1, 1 } },
2869 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i16, { 2, 1, 1, 1 } },
2870 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i8, { 2, 1, 1, 1 } },
2871 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i16, { 2, 1, 1, 1 } },
2872 { ISD::TRUNCATE, MVT::v32i1, MVT::v32i8, { 2, 1, 1, 1 } },
2873 { ISD::TRUNCATE, MVT::v32i1, MVT::v32i16, { 2, 1, 1, 1 } },
2874 { ISD::TRUNCATE, MVT::v64i1, MVT::v64i8, { 2, 1, 1, 1 } },
2875 { ISD::TRUNCATE, MVT::v64i1, MVT::v32i16, { 2, 1, 1, 1 } },
2876
2877 { ISD::TRUNCATE, MVT::v32i8, MVT::v32i16, { 2, 1, 1, 1 } },
2878 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i16, { 2, 1, 1, 1 } }, // widen to zmm
2879 { ISD::TRUNCATE, MVT::v2i8, MVT::v2i16, { 2, 1, 1, 1 } }, // vpmovwb
2880 { ISD::TRUNCATE, MVT::v4i8, MVT::v4i16, { 2, 1, 1, 1 } }, // vpmovwb
2881 { ISD::TRUNCATE, MVT::v8i8, MVT::v8i16, { 2, 1, 1, 1 } }, // vpmovwb
2882 };
2883
2884 static const TypeConversionCostKindTblEntry AVX512DQConversionTbl[] = {
2885 // Mask sign extend has an instruction.
2886 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v2i1, { 1, 1, 1, 1 } },
2887 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v2i1, { 1, 1, 1, 1 } },
2888 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v4i1, { 1, 1, 1, 1 } },
2889 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 1, 1, 1, 1 } },
2890 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 1, 1, 1, 1 } },
2891 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v16i1, { 1, 1, 1, 1 } },
2892 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i1, { 1, 1, 1, 1 } },
2893 { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i1, { 1, 1, 1, 1 } },
2894
2895 // Mask zero extend is a sext + shift.
2896 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v2i1, { 2, 1, 1, 1, } },
2897 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v2i1, { 2, 1, 1, 1, } },
2898 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v4i1, { 2, 1, 1, 1, } },
2899 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 2, 1, 1, 1, } },
2900 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 2, 1, 1, 1, } },
2901 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v16i1, { 2, 1, 1, 1, } },
2902 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i1, { 2, 1, 1, 1, } },
2903 { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i1, { 2, 1, 1, 1, } },
2904
2905 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i64, { 2, 1, 1, 1 } },
2906 { ISD::TRUNCATE, MVT::v2i1, MVT::v4i32, { 2, 1, 1, 1 } },
2907 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i32, { 2, 1, 1, 1 } },
2908 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i64, { 2, 1, 1, 1 } },
2909 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 2, 1, 1, 1 } },
2910 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i64, { 2, 1, 1, 1 } },
2911 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i32, { 2, 1, 1, 1 } },
2912 { ISD::TRUNCATE, MVT::v16i1, MVT::v8i64, { 2, 1, 1, 1 } },
2913
2914 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i64, { 1, 1, 1, 1 } },
2915 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i64, { 1, 1, 1, 1 } },
2916
2917 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i64, { 1, 1, 1, 1 } },
2918 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i64, { 1, 1, 1, 1 } },
2919
2920 { ISD::FP_TO_SINT, MVT::v8i64, MVT::v8f32, { 1, 1, 1, 1 } },
2921 { ISD::FP_TO_SINT, MVT::v8i64, MVT::v8f64, { 1, 1, 1, 1 } },
2922
2923 { ISD::FP_TO_UINT, MVT::v8i64, MVT::v8f32, { 1, 1, 1, 1 } },
2924 { ISD::FP_TO_UINT, MVT::v8i64, MVT::v8f64, { 1, 1, 1, 1 } },
2925 };
2926
2927 // TODO: For AVX512DQ + AVX512VL, we also have cheap casts for 128-bit and
2928 // 256-bit wide vectors.
2929
2930 static const TypeConversionCostKindTblEntry AVX512FConversionTbl[] = {
2931 { ISD::FP_EXTEND, MVT::v8f64, MVT::v8f32, { 1, 1, 1, 1 } },
2932 { ISD::FP_EXTEND, MVT::v8f64, MVT::v16f32, { 3, 1, 1, 1 } },
2933 { ISD::FP_EXTEND, MVT::v16f64, MVT::v16f32, { 4, 1, 1, 1 } }, // 2*vcvtps2pd+vextractf64x4
2934 { ISD::FP_EXTEND, MVT::v16f32, MVT::v16f16, { 1, 1, 1, 1 } }, // vcvtph2ps
2935 { ISD::FP_EXTEND, MVT::v8f64, MVT::v8f16, { 2, 1, 1, 1 } }, // vcvtph2ps+vcvtps2pd
2936 { ISD::FP_ROUND, MVT::v8f32, MVT::v8f64, { 1, 1, 1, 1 } },
2937 { ISD::FP_ROUND, MVT::v16f16, MVT::v16f32, { 1, 1, 1, 1 } }, // vcvtps2ph
2938
2939 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
2940 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
2941 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
2942 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
2943 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
2944 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
2945 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
2946 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i16, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
2947 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i32, { 2, 1, 1, 1 } }, // zmm vpslld+vptestmd
2948 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i32, { 2, 1, 1, 1 } }, // zmm vpslld+vptestmd
2949 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 2, 1, 1, 1 } }, // zmm vpslld+vptestmd
2950 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i32, { 2, 1, 1, 1 } }, // vpslld+vptestmd
2951 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i64, { 2, 1, 1, 1 } }, // zmm vpsllq+vptestmq
2952 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i64, { 2, 1, 1, 1 } }, // zmm vpsllq+vptestmq
2953 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i64, { 2, 1, 1, 1 } }, // vpsllq+vptestmq
2954 { ISD::TRUNCATE, MVT::v2i8, MVT::v2i32, { 2, 1, 1, 1 } }, // vpmovdb
2955 { ISD::TRUNCATE, MVT::v4i8, MVT::v4i32, { 2, 1, 1, 1 } }, // vpmovdb
2956 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i32, { 2, 1, 1, 1 } }, // vpmovdb
2957 { ISD::TRUNCATE, MVT::v32i8, MVT::v16i32, { 2, 1, 1, 1 } }, // vpmovdb
2958 { ISD::TRUNCATE, MVT::v64i8, MVT::v16i32, { 2, 1, 1, 1 } }, // vpmovdb
2959 { ISD::TRUNCATE, MVT::v16i16, MVT::v16i32, { 2, 1, 1, 1 } }, // vpmovdw
2960 { ISD::TRUNCATE, MVT::v32i16, MVT::v16i32, { 2, 1, 1, 1 } }, // vpmovdw
2961 { ISD::TRUNCATE, MVT::v2i8, MVT::v2i64, { 2, 1, 1, 1 } }, // vpmovqb
2962 { ISD::TRUNCATE, MVT::v2i16, MVT::v2i64, { 1, 1, 1, 1 } }, // vpshufb
2963 { ISD::TRUNCATE, MVT::v8i8, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqb
2964 { ISD::TRUNCATE, MVT::v16i8, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqb
2965 { ISD::TRUNCATE, MVT::v32i8, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqb
2966 { ISD::TRUNCATE, MVT::v64i8, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqb
2967 { ISD::TRUNCATE, MVT::v8i16, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqw
2968 { ISD::TRUNCATE, MVT::v16i16, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqw
2969 { ISD::TRUNCATE, MVT::v32i16, MVT::v8i64, { 2, 1, 1, 1 } }, // vpmovqw
2970 { ISD::TRUNCATE, MVT::v8i32, MVT::v8i64, { 1, 1, 1, 1 } }, // vpmovqd
2971 { ISD::TRUNCATE, MVT::v4i32, MVT::v4i64, { 1, 1, 1, 1 } }, // zmm vpmovqd
2972 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i64, { 5, 1, 1, 1 } },// 2*vpmovqd+concat+vpmovdb
2973
2974 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i16, { 3, 1, 1, 1 } }, // extend to v16i32
2975 { ISD::TRUNCATE, MVT::v32i8, MVT::v32i16, { 8, 1, 1, 1 } },
2976 { ISD::TRUNCATE, MVT::v64i8, MVT::v32i16, { 8, 1, 1, 1 } },
2977
2978 // Sign extend is zmm vpternlogd+vptruncdb.
2979 // Zero extend is zmm broadcast load+vptruncdw.
2980 { ISD::SIGN_EXTEND, MVT::v2i8, MVT::v2i1, { 3, 1, 1, 1 } },
2981 { ISD::ZERO_EXTEND, MVT::v2i8, MVT::v2i1, { 4, 1, 1, 1 } },
2982 { ISD::SIGN_EXTEND, MVT::v4i8, MVT::v4i1, { 3, 1, 1, 1 } },
2983 { ISD::ZERO_EXTEND, MVT::v4i8, MVT::v4i1, { 4, 1, 1, 1 } },
2984 { ISD::SIGN_EXTEND, MVT::v8i8, MVT::v8i1, { 3, 1, 1, 1 } },
2985 { ISD::ZERO_EXTEND, MVT::v8i8, MVT::v8i1, { 4, 1, 1, 1 } },
2986 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v16i1, { 3, 1, 1, 1 } },
2987 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v16i1, { 4, 1, 1, 1 } },
2988
2989 // Sign extend is zmm vpternlogd+vptruncdw.
2990 // Zero extend is zmm vpternlogd+vptruncdw+vpsrlw.
2991 { ISD::SIGN_EXTEND, MVT::v2i16, MVT::v2i1, { 3, 1, 1, 1 } },
2992 { ISD::ZERO_EXTEND, MVT::v2i16, MVT::v2i1, { 4, 1, 1, 1 } },
2993 { ISD::SIGN_EXTEND, MVT::v4i16, MVT::v4i1, { 3, 1, 1, 1 } },
2994 { ISD::ZERO_EXTEND, MVT::v4i16, MVT::v4i1, { 4, 1, 1, 1 } },
2995 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v8i1, { 3, 1, 1, 1 } },
2996 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v8i1, { 4, 1, 1, 1 } },
2997 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, { 3, 1, 1, 1 } },
2998 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, { 4, 1, 1, 1 } },
2999
3000 { ISD::SIGN_EXTEND, MVT::v2i32, MVT::v2i1, { 1, 1, 1, 1 } }, // zmm vpternlogd
3001 { ISD::ZERO_EXTEND, MVT::v2i32, MVT::v2i1, { 2, 1, 1, 1 } }, // zmm vpternlogd+psrld
3002 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v4i1, { 1, 1, 1, 1 } }, // zmm vpternlogd
3003 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v4i1, { 2, 1, 1, 1 } }, // zmm vpternlogd+psrld
3004 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 1, 1, 1, 1 } }, // zmm vpternlogd
3005 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 2, 1, 1, 1 } }, // zmm vpternlogd+psrld
3006 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v2i1, { 1, 1, 1, 1 } }, // zmm vpternlogq
3007 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v2i1, { 2, 1, 1, 1 } }, // zmm vpternlogq+psrlq
3008 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 1, 1, 1, 1 } }, // zmm vpternlogq
3009 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 2, 1, 1, 1 } }, // zmm vpternlogq+psrlq
3010
3011 { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i1, { 1, 1, 1, 1 } }, // vpternlogd
3012 { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i1, { 2, 1, 1, 1 } }, // vpternlogd+psrld
3013 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i1, { 1, 1, 1, 1 } }, // vpternlogq
3014 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i1, { 2, 1, 1, 1 } }, // vpternlogq+psrlq
3015
3016 { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i8, { 1, 1, 1, 1 } },
3017 { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i8, { 1, 1, 1, 1 } },
3018 { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i16, { 1, 1, 1, 1 } },
3019 { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i16, { 1, 1, 1, 1 } },
3020 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i8, { 1, 1, 1, 1 } },
3021 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i8, { 1, 1, 1, 1 } },
3022 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i16, { 1, 1, 1, 1 } },
3023 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i16, { 1, 1, 1, 1 } },
3024 { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i32, { 1, 1, 1, 1 } },
3025 { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i32, { 1, 1, 1, 1 } },
3026
3027 { ISD::SIGN_EXTEND, MVT::v32i16, MVT::v32i8, { 3, 1, 1, 1 } }, // FIXME: May not be right
3028 { ISD::ZERO_EXTEND, MVT::v32i16, MVT::v32i8, { 3, 1, 1, 1 } }, // FIXME: May not be right
3029
3030 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i1, { 4, 1, 1, 1 } },
3031 { ISD::SINT_TO_FP, MVT::v16f32, MVT::v16i1, { 3, 1, 1, 1 } },
3032 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v16i8, { 2, 1, 1, 1 } },
3033 { ISD::SINT_TO_FP, MVT::v16f32, MVT::v16i8, { 1, 1, 1, 1 } },
3034 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i16, { 2, 1, 1, 1 } },
3035 { ISD::SINT_TO_FP, MVT::v16f32, MVT::v16i16, { 1, 1, 1, 1 } },
3036 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i32, { 1, 1, 1, 1 } },
3037 { ISD::SINT_TO_FP, MVT::v16f32, MVT::v16i32, { 1, 1, 1, 1 } },
3038
3039 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i1, { 4, 1, 1, 1 } },
3040 { ISD::UINT_TO_FP, MVT::v16f32, MVT::v16i1, { 3, 1, 1, 1 } },
3041 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v16i8, { 2, 1, 1, 1 } },
3042 { ISD::UINT_TO_FP, MVT::v16f32, MVT::v16i8, { 1, 1, 1, 1 } },
3043 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i16, { 2, 1, 1, 1 } },
3044 { ISD::UINT_TO_FP, MVT::v16f32, MVT::v16i16, { 1, 1, 1, 1 } },
3045 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i32, { 1, 1, 1, 1 } },
3046 { ISD::UINT_TO_FP, MVT::v16f32, MVT::v16i32, { 1, 1, 1, 1 } },
3047 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i64, {26, 1, 1, 1 } },
3048 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i64, { 5, 1, 1, 1 } },
3049
3050 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v16f32, { 2, 1, 1, 1 } },
3051 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v16f64, { 7, 1, 1, 1 } },
3052 { ISD::FP_TO_SINT, MVT::v32i8, MVT::v32f64, {15, 1, 1, 1 } },
3053 { ISD::FP_TO_SINT, MVT::v64i8, MVT::v64f32, {11, 1, 1, 1 } },
3054 { ISD::FP_TO_SINT, MVT::v64i8, MVT::v64f64, {31, 1, 1, 1 } },
3055 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v8f64, { 3, 1, 1, 1 } },
3056 { ISD::FP_TO_SINT, MVT::v16i16, MVT::v16f64, { 7, 1, 1, 1 } },
3057 { ISD::FP_TO_SINT, MVT::v32i16, MVT::v32f32, { 5, 1, 1, 1 } },
3058 { ISD::FP_TO_SINT, MVT::v32i16, MVT::v32f64, {15, 1, 1, 1 } },
3059 { ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f64, { 1, 1, 1, 1 } },
3060 { ISD::FP_TO_SINT, MVT::v16i32, MVT::v16f64, { 3, 1, 1, 1 } },
3061
3062 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f64, { 1, 1, 1, 1 } },
3063 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v8f64, { 3, 1, 1, 1 } },
3064 { ISD::FP_TO_UINT, MVT::v8i8, MVT::v8f64, { 3, 1, 1, 1 } },
3065 { ISD::FP_TO_UINT, MVT::v16i32, MVT::v16f32, { 1, 1, 1, 1 } },
3066 { ISD::FP_TO_UINT, MVT::v16i16, MVT::v16f32, { 3, 1, 1, 1 } },
3067 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v16f32, { 3, 1, 1, 1 } },
3068 };
3069
3070 static const TypeConversionCostKindTblEntry AVX512BWVLConversionTbl[] {
3071 // Mask sign extend has an instruction.
3072 { ISD::SIGN_EXTEND, MVT::v2i8, MVT::v2i1, { 1, 1, 1, 1 } },
3073 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v2i1, { 1, 1, 1, 1 } },
3074 { ISD::SIGN_EXTEND, MVT::v2i16, MVT::v2i1, { 1, 1, 1, 1 } },
3075 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v2i1, { 1, 1, 1, 1 } },
3076 { ISD::SIGN_EXTEND, MVT::v4i16, MVT::v4i1, { 1, 1, 1, 1 } },
3077 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v4i1, { 1, 1, 1, 1 } },
3078 { ISD::SIGN_EXTEND, MVT::v4i8, MVT::v4i1, { 1, 1, 1, 1 } },
3079 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v4i1, { 1, 1, 1, 1 } },
3080 { ISD::SIGN_EXTEND, MVT::v8i8, MVT::v8i1, { 1, 1, 1, 1 } },
3081 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v8i1, { 1, 1, 1, 1 } },
3082 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v8i1, { 1, 1, 1, 1 } },
3083 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v16i1, { 1, 1, 1, 1 } },
3084 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, { 1, 1, 1, 1 } },
3085 { ISD::SIGN_EXTEND, MVT::v32i8, MVT::v32i1, { 1, 1, 1, 1 } },
3086 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v32i1, { 1, 1, 1, 1 } },
3087 { ISD::SIGN_EXTEND, MVT::v32i8, MVT::v64i1, { 1, 1, 1, 1 } },
3088 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v64i1, { 1, 1, 1, 1 } },
3089
3090 // Mask zero extend is a sext + shift.
3091 { ISD::ZERO_EXTEND, MVT::v2i8, MVT::v2i1, { 2, 1, 1, 1 } },
3092 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v2i1, { 2, 1, 1, 1 } },
3093 { ISD::ZERO_EXTEND, MVT::v2i16, MVT::v2i1, { 2, 1, 1, 1 } },
3094 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v2i1, { 2, 1, 1, 1 } },
3095 { ISD::ZERO_EXTEND, MVT::v4i8, MVT::v4i1, { 2, 1, 1, 1 } },
3096 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v4i1, { 2, 1, 1, 1 } },
3097 { ISD::ZERO_EXTEND, MVT::v4i16, MVT::v4i1, { 2, 1, 1, 1 } },
3098 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v4i1, { 2, 1, 1, 1 } },
3099 { ISD::ZERO_EXTEND, MVT::v8i8, MVT::v8i1, { 2, 1, 1, 1 } },
3100 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v8i1, { 2, 1, 1, 1 } },
3101 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v8i1, { 2, 1, 1, 1 } },
3102 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v16i1, { 2, 1, 1, 1 } },
3103 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, { 2, 1, 1, 1 } },
3104 { ISD::ZERO_EXTEND, MVT::v32i8, MVT::v32i1, { 2, 1, 1, 1 } },
3105 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v32i1, { 2, 1, 1, 1 } },
3106 { ISD::ZERO_EXTEND, MVT::v32i8, MVT::v64i1, { 2, 1, 1, 1 } },
3107 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v64i1, { 2, 1, 1, 1 } },
3108
3109 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 2, 1, 1, 1 } },
3110 { ISD::TRUNCATE, MVT::v2i1, MVT::v16i8, { 2, 1, 1, 1 } },
3111 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 2, 1, 1, 1 } },
3112 { ISD::TRUNCATE, MVT::v2i1, MVT::v8i16, { 2, 1, 1, 1 } },
3113 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 2, 1, 1, 1 } },
3114 { ISD::TRUNCATE, MVT::v4i1, MVT::v16i8, { 2, 1, 1, 1 } },
3115 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i16, { 2, 1, 1, 1 } },
3116 { ISD::TRUNCATE, MVT::v4i1, MVT::v8i16, { 2, 1, 1, 1 } },
3117 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i8, { 2, 1, 1, 1 } },
3118 { ISD::TRUNCATE, MVT::v8i1, MVT::v16i8, { 2, 1, 1, 1 } },
3119 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i16, { 2, 1, 1, 1 } },
3120 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i8, { 2, 1, 1, 1 } },
3121 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i16, { 2, 1, 1, 1 } },
3122 { ISD::TRUNCATE, MVT::v32i1, MVT::v32i8, { 2, 1, 1, 1 } },
3123 { ISD::TRUNCATE, MVT::v32i1, MVT::v16i16, { 2, 1, 1, 1 } },
3124 { ISD::TRUNCATE, MVT::v64i1, MVT::v32i8, { 2, 1, 1, 1 } },
3125 { ISD::TRUNCATE, MVT::v64i1, MVT::v16i16, { 2, 1, 1, 1 } },
3126
3127 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i16, { 2, 1, 1, 1 } },
3128 };
3129
3130 static const TypeConversionCostKindTblEntry AVX512DQVLConversionTbl[] = {
3131 // Mask sign extend has an instruction.
3132 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v2i1, { 1, 1, 1, 1 } },
3133 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v2i1, { 1, 1, 1, 1 } },
3134 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v4i1, { 1, 1, 1, 1 } },
3135 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v16i1, { 1, 1, 1, 1 } },
3136 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 1, 1, 1, 1 } },
3137 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v8i1, { 1, 1, 1, 1 } },
3138 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v16i1, { 1, 1, 1, 1 } },
3139 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 1, 1, 1, 1 } },
3140
3141 // Mask zero extend is a sext + shift.
3142 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v2i1, { 2, 1, 1, 1 } },
3143 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v2i1, { 2, 1, 1, 1 } },
3144 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v4i1, { 2, 1, 1, 1 } },
3145 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v16i1, { 2, 1, 1, 1 } },
3146 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 2, 1, 1, 1 } },
3147 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v8i1, { 2, 1, 1, 1 } },
3148 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v16i1, { 2, 1, 1, 1 } },
3149 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 2, 1, 1, 1 } },
3150
3151 { ISD::TRUNCATE, MVT::v16i1, MVT::v4i64, { 2, 1, 1, 1 } },
3152 { ISD::TRUNCATE, MVT::v16i1, MVT::v8i32, { 2, 1, 1, 1 } },
3153 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i64, { 2, 1, 1, 1 } },
3154 { ISD::TRUNCATE, MVT::v2i1, MVT::v4i32, { 2, 1, 1, 1 } },
3155 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i32, { 2, 1, 1, 1 } },
3156 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i64, { 2, 1, 1, 1 } },
3157 { ISD::TRUNCATE, MVT::v8i1, MVT::v4i64, { 2, 1, 1, 1 } },
3158 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 2, 1, 1, 1 } },
3159
3160 { ISD::SINT_TO_FP, MVT::v2f32, MVT::v2i64, { 1, 1, 1, 1 } },
3161 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i64, { 1, 1, 1, 1 } },
3162 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i64, { 1, 1, 1, 1 } },
3163 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i64, { 1, 1, 1, 1 } },
3164
3165 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i64, { 1, 1, 1, 1 } },
3166 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, { 1, 1, 1, 1 } },
3167 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i64, { 1, 1, 1, 1 } },
3168 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i64, { 1, 1, 1, 1 } },
3169
3170 { ISD::FP_TO_SINT, MVT::v2i64, MVT::v4f32, { 1, 1, 1, 1 } },
3171 { ISD::FP_TO_SINT, MVT::v4i64, MVT::v4f32, { 1, 1, 1, 1 } },
3172 { ISD::FP_TO_SINT, MVT::v2i64, MVT::v2f64, { 1, 1, 1, 1 } },
3173 { ISD::FP_TO_SINT, MVT::v4i64, MVT::v4f64, { 1, 1, 1, 1 } },
3174
3175 { ISD::FP_TO_UINT, MVT::v2i64, MVT::v4f32, { 1, 1, 1, 1 } },
3176 { ISD::FP_TO_UINT, MVT::v4i64, MVT::v4f32, { 1, 1, 1, 1 } },
3177 { ISD::FP_TO_UINT, MVT::v2i64, MVT::v2f64, { 1, 1, 1, 1 } },
3178 { ISD::FP_TO_UINT, MVT::v4i64, MVT::v4f64, { 1, 1, 1, 1 } },
3179 };
3180
3181 static const TypeConversionCostKindTblEntry AVX512VLConversionTbl[] = {
3182 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
3183 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
3184 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i8, { 3, 1, 1, 1 } }, // sext+vpslld+vptestmd
3185 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i8, { 8, 1, 1, 1 } }, // split+2*v8i8
3186 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
3187 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
3188 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i16, { 3, 1, 1, 1 } }, // sext+vpsllq+vptestmq
3189 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i16, { 8, 1, 1, 1 } }, // split+2*v8i16
3190 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i32, { 2, 1, 1, 1 } }, // vpslld+vptestmd
3191 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i32, { 2, 1, 1, 1 } }, // vpslld+vptestmd
3192 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 2, 1, 1, 1 } }, // vpslld+vptestmd
3193 { ISD::TRUNCATE, MVT::v16i1, MVT::v8i32, { 2, 1, 1, 1 } }, // vpslld+vptestmd
3194 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i64, { 2, 1, 1, 1 } }, // vpsllq+vptestmq
3195 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i64, { 2, 1, 1, 1 } }, // vpsllq+vptestmq
3196 { ISD::TRUNCATE, MVT::v4i32, MVT::v4i64, { 1, 1, 1, 1 } }, // vpmovqd
3197 { ISD::TRUNCATE, MVT::v4i8, MVT::v4i64, { 2, 1, 1, 1 } }, // vpmovqb
3198 { ISD::TRUNCATE, MVT::v4i16, MVT::v4i64, { 2, 1, 1, 1 } }, // vpmovqw
3199 { ISD::TRUNCATE, MVT::v8i8, MVT::v8i32, { 2, 1, 1, 1 } }, // vpmovwb
3200
3201 // sign extend is vpcmpeq+maskedmove+vpmovdw+vpacksswb
3202 // zero extend is vpcmpeq+maskedmove+vpmovdw+vpsrlw+vpackuswb
3203 { ISD::SIGN_EXTEND, MVT::v2i8, MVT::v2i1, { 5, 1, 1, 1 } },
3204 { ISD::ZERO_EXTEND, MVT::v2i8, MVT::v2i1, { 6, 1, 1, 1 } },
3205 { ISD::SIGN_EXTEND, MVT::v4i8, MVT::v4i1, { 5, 1, 1, 1 } },
3206 { ISD::ZERO_EXTEND, MVT::v4i8, MVT::v4i1, { 6, 1, 1, 1 } },
3207 { ISD::SIGN_EXTEND, MVT::v8i8, MVT::v8i1, { 5, 1, 1, 1 } },
3208 { ISD::ZERO_EXTEND, MVT::v8i8, MVT::v8i1, { 6, 1, 1, 1 } },
3209 { ISD::SIGN_EXTEND, MVT::v16i8, MVT::v16i1, {10, 1, 1, 1 } },
3210 { ISD::ZERO_EXTEND, MVT::v16i8, MVT::v16i1, {12, 1, 1, 1 } },
3211
3212 // sign extend is vpcmpeq+maskedmove+vpmovdw
3213 // zero extend is vpcmpeq+maskedmove+vpmovdw+vpsrlw
3214 { ISD::SIGN_EXTEND, MVT::v2i16, MVT::v2i1, { 4, 1, 1, 1 } },
3215 { ISD::ZERO_EXTEND, MVT::v2i16, MVT::v2i1, { 5, 1, 1, 1 } },
3216 { ISD::SIGN_EXTEND, MVT::v4i16, MVT::v4i1, { 4, 1, 1, 1 } },
3217 { ISD::ZERO_EXTEND, MVT::v4i16, MVT::v4i1, { 5, 1, 1, 1 } },
3218 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v8i1, { 4, 1, 1, 1 } },
3219 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v8i1, { 5, 1, 1, 1 } },
3220 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, {10, 1, 1, 1 } },
3221 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, {12, 1, 1, 1 } },
3222
3223 { ISD::SIGN_EXTEND, MVT::v2i32, MVT::v2i1, { 1, 1, 1, 1 } }, // vpternlogd
3224 { ISD::ZERO_EXTEND, MVT::v2i32, MVT::v2i1, { 2, 1, 1, 1 } }, // vpternlogd+psrld
3225 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v4i1, { 1, 1, 1, 1 } }, // vpternlogd
3226 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v4i1, { 2, 1, 1, 1 } }, // vpternlogd+psrld
3227 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 1, 1, 1, 1 } }, // vpternlogd
3228 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 2, 1, 1, 1 } }, // vpternlogd+psrld
3229 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v16i1, { 1, 1, 1, 1 } }, // vpternlogd
3230 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v16i1, { 2, 1, 1, 1 } }, // vpternlogd+psrld
3231
3232 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v2i1, { 1, 1, 1, 1 } }, // vpternlogq
3233 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v2i1, { 2, 1, 1, 1 } }, // vpternlogq+psrlq
3234 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 1, 1, 1, 1 } }, // vpternlogq
3235 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 2, 1, 1, 1 } }, // vpternlogq+psrlq
3236
3237 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v16i8, { 1, 1, 1, 1 } },
3238 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v16i8, { 1, 1, 1, 1 } },
3239 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v16i8, { 1, 1, 1, 1 } },
3240 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v16i8, { 1, 1, 1, 1 } },
3241 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i8, { 1, 1, 1, 1 } },
3242 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i8, { 1, 1, 1, 1 } },
3243 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v8i16, { 1, 1, 1, 1 } },
3244 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v8i16, { 1, 1, 1, 1 } },
3245 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i16, { 1, 1, 1, 1 } },
3246 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i16, { 1, 1, 1, 1 } },
3247 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i32, { 1, 1, 1, 1 } },
3248 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i32, { 1, 1, 1, 1 } },
3249
3250 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v16i8, { 1, 1, 1, 1 } },
3251 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v16i8, { 1, 1, 1, 1 } },
3252 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v8i16, { 1, 1, 1, 1 } },
3253 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i16, { 1, 1, 1, 1 } },
3254
3255 { ISD::UINT_TO_FP, MVT::f32, MVT::i64, { 1, 1, 1, 1 } },
3256 { ISD::UINT_TO_FP, MVT::f64, MVT::i64, { 1, 1, 1, 1 } },
3257 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v16i8, { 1, 1, 1, 1 } },
3258 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v16i8, { 1, 1, 1, 1 } },
3259 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v8i16, { 1, 1, 1, 1 } },
3260 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i16, { 1, 1, 1, 1 } },
3261 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, { 1, 1, 1, 1 } },
3262 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, { 1, 1, 1, 1 } },
3263 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i32, { 1, 1, 1, 1 } },
3264 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i32, { 1, 1, 1, 1 } },
3265 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i64, { 5, 1, 1, 1 } },
3266 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, { 5, 1, 1, 1 } },
3267 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i64, { 5, 1, 1, 1 } },
3268
3269 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v8f32, { 2, 1, 1, 1 } },
3270 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v16f32, { 2, 1, 1, 1 } },
3271 { ISD::FP_TO_SINT, MVT::v32i8, MVT::v32f32, { 5, 1, 1, 1 } },
3272
3273 { ISD::FP_TO_UINT, MVT::i64, MVT::f32, { 1, 1, 1, 1 } },
3274 { ISD::FP_TO_UINT, MVT::i64, MVT::f64, { 1, 1, 1, 1 } },
3275 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, { 1, 1, 1, 1 } },
3276 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v2f64, { 1, 1, 1, 1 } },
3277 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f64, { 1, 1, 1, 1 } },
3278 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f32, { 1, 1, 1, 1 } },
3279 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f64, { 1, 1, 1, 1 } },
3280 };
3281
3282 static const TypeConversionCostKindTblEntry AVX2ConversionTbl[] = {
3283 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 3, 1, 1, 1 } },
3284 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 3, 1, 1, 1 } },
3285 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 3, 1, 1, 1 } },
3286 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 3, 1, 1, 1 } },
3287 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, { 1, 1, 1, 1 } },
3288 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, { 1, 1, 1, 1 } },
3289
3290 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v16i8, { 2, 1, 1, 1 } },
3291 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v16i8, { 2, 1, 1, 1 } },
3292 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v16i8, { 2, 1, 1, 1 } },
3293 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v16i8, { 2, 1, 1, 1 } },
3294 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i8, { 2, 1, 1, 1 } },
3295 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i8, { 2, 1, 1, 1 } },
3296 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v8i16, { 2, 1, 1, 1 } },
3297 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v8i16, { 2, 1, 1, 1 } },
3298 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i16, { 2, 1, 1, 1 } },
3299 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i16, { 2, 1, 1, 1 } },
3300 { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i16, { 3, 1, 1, 1 } },
3301 { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i16, { 3, 1, 1, 1 } },
3302 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i32, { 2, 1, 1, 1 } },
3303 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i32, { 2, 1, 1, 1 } },
3304
3305 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 2, 1, 1, 1 } },
3306
3307 { ISD::TRUNCATE, MVT::v16i16, MVT::v16i32, { 4, 1, 1, 1 } },
3308 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i32, { 4, 1, 1, 1 } },
3309 { ISD::TRUNCATE, MVT::v16i8, MVT::v8i16, { 1, 1, 1, 1 } },
3310 { ISD::TRUNCATE, MVT::v16i8, MVT::v4i32, { 1, 1, 1, 1 } },
3311 { ISD::TRUNCATE, MVT::v16i8, MVT::v2i64, { 1, 1, 1, 1 } },
3312 { ISD::TRUNCATE, MVT::v16i8, MVT::v8i32, { 4, 1, 1, 1 } },
3313 { ISD::TRUNCATE, MVT::v16i8, MVT::v4i64, { 4, 1, 1, 1 } },
3314 { ISD::TRUNCATE, MVT::v8i16, MVT::v4i32, { 1, 1, 1, 1 } },
3315 { ISD::TRUNCATE, MVT::v8i16, MVT::v2i64, { 1, 1, 1, 1 } },
3316 { ISD::TRUNCATE, MVT::v8i16, MVT::v4i64, { 5, 1, 1, 1 } },
3317 { ISD::TRUNCATE, MVT::v4i32, MVT::v4i64, { 1, 1, 1, 1 } },
3318 { ISD::TRUNCATE, MVT::v8i16, MVT::v8i32, { 2, 1, 1, 1 } },
3319
3320 { ISD::FP_EXTEND, MVT::v8f64, MVT::v8f32, { 3, 1, 1, 1 } },
3321 { ISD::FP_ROUND, MVT::v8f32, MVT::v8f64, { 3, 1, 1, 1 } },
3322
3323 { ISD::FP_TO_SINT, MVT::v16i16, MVT::v8f32, { 1, 1, 1, 1 } },
3324 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f64, { 1, 1, 1, 1 } },
3325 { ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f32, { 1, 1, 1, 1 } },
3326 { ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f64, { 3, 1, 1, 1 } },
3327
3328 { ISD::FP_TO_UINT, MVT::i64, MVT::f32, { 3, 1, 1, 1 } },
3329 { ISD::FP_TO_UINT, MVT::i64, MVT::f64, { 3, 1, 1, 1 } },
3330 { ISD::FP_TO_UINT, MVT::v16i16, MVT::v8f32, { 1, 1, 1, 1 } },
3331 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, { 3, 1, 1, 1 } },
3332 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v2f64, { 4, 1, 1, 1 } },
3333 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f64, { 4, 1, 1, 1 } },
3334 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f32, { 3, 1, 1, 1 } },
3335 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v4f64, { 4, 1, 1, 1 } },
3336
3337 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v16i8, { 2, 1, 1, 1 } },
3338 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v16i8, { 2, 1, 1, 1 } },
3339 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v8i16, { 2, 1, 1, 1 } },
3340 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i16, { 2, 1, 1, 1 } },
3341 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i32, { 1, 1, 1, 1 } },
3342 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i32, { 1, 1, 1, 1 } },
3343 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i32, { 3, 1, 1, 1 } },
3344
3345 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v16i8, { 2, 1, 1, 1 } },
3346 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v16i8, { 2, 1, 1, 1 } },
3347 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v8i16, { 2, 1, 1, 1 } },
3348 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i16, { 2, 1, 1, 1 } },
3349 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, { 2, 1, 1, 1 } },
3350 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i32, { 1, 1, 1, 1 } },
3351 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, { 2, 1, 1, 1 } },
3352 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i32, { 2, 1, 1, 1 } },
3353 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i32, { 2, 1, 1, 1 } },
3354 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i32, { 4, 1, 1, 1 } },
3355 };
3356
3357 static const TypeConversionCostKindTblEntry AVXConversionTbl[] = {
3358 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i1, { 4, 1, 1, 1 } },
3359 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i1, { 4, 1, 1, 1 } },
3360 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i1, { 4, 1, 1, 1 } },
3361 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i1, { 4, 1, 1, 1 } },
3362 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i1, { 4, 1, 1, 1 } },
3363 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i1, { 4, 1, 1, 1 } },
3364
3365 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v16i8, { 3, 1, 1, 1 } },
3366 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v16i8, { 3, 1, 1, 1 } },
3367 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v16i8, { 3, 1, 1, 1 } },
3368 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v16i8, { 3, 1, 1, 1 } },
3369 { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i8, { 3, 1, 1, 1 } },
3370 { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i8, { 3, 1, 1, 1 } },
3371 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v8i16, { 3, 1, 1, 1 } },
3372 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v8i16, { 3, 1, 1, 1 } },
3373 { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i16, { 3, 1, 1, 1 } },
3374 { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i16, { 3, 1, 1, 1 } },
3375 { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i32, { 3, 1, 1, 1 } },
3376 { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i32, { 3, 1, 1, 1 } },
3377
3378 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i64, { 4, 1, 1, 1 } },
3379 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i32, { 5, 1, 1, 1 } },
3380 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i16, { 4, 1, 1, 1 } },
3381 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i64, { 9, 1, 1, 1 } },
3382 { ISD::TRUNCATE, MVT::v16i1, MVT::v16i64, {11, 1, 1, 1 } },
3383
3384 { ISD::TRUNCATE, MVT::v16i16, MVT::v16i32, { 6, 1, 1, 1 } },
3385 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i32, { 6, 1, 1, 1 } },
3386 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i16, { 2, 1, 1, 1 } }, // and+extract+packuswb
3387 { ISD::TRUNCATE, MVT::v16i8, MVT::v8i32, { 5, 1, 1, 1 } },
3388 { ISD::TRUNCATE, MVT::v8i16, MVT::v8i32, { 5, 1, 1, 1 } },
3389 { ISD::TRUNCATE, MVT::v16i8, MVT::v4i64, { 5, 1, 1, 1 } },
3390 { ISD::TRUNCATE, MVT::v8i16, MVT::v4i64, { 3, 1, 1, 1 } }, // and+extract+2*packusdw
3391 { ISD::TRUNCATE, MVT::v4i32, MVT::v4i64, { 2, 1, 1, 1 } },
3392
3393 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i1, { 3, 1, 1, 1 } },
3394 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i1, { 3, 1, 1, 1 } },
3395 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i1, { 8, 1, 1, 1 } },
3396 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v16i8, { 4, 1, 1, 1 } },
3397 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v16i8, { 2, 1, 1, 1 } },
3398 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i16, { 4, 1, 1, 1 } },
3399 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v8i16, { 2, 1, 1, 1 } },
3400 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i32, { 2, 1, 1, 1 } },
3401 { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i32, { 2, 1, 1, 1 } },
3402 { ISD::SINT_TO_FP, MVT::v8f64, MVT::v8i32, { 4, 1, 1, 1 } },
3403 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v2i64, { 5, 1, 1, 1 } },
3404 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i64, { 8, 1, 1, 1 } },
3405
3406 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i1, { 7, 1, 1, 1 } },
3407 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i1, { 7, 1, 1, 1 } },
3408 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i1, { 6, 1, 1, 1 } },
3409 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v16i8, { 4, 1, 1, 1 } },
3410 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v16i8, { 2, 1, 1, 1 } },
3411 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i16, { 4, 1, 1, 1 } },
3412 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v8i16, { 2, 1, 1, 1 } },
3413 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, { 4, 1, 1, 1 } },
3414 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i32, { 4, 1, 1, 1 } },
3415 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, { 5, 1, 1, 1 } },
3416 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i32, { 6, 1, 1, 1 } },
3417 { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i32, { 8, 1, 1, 1 } },
3418 { ISD::UINT_TO_FP, MVT::v8f64, MVT::v8i32, {10, 1, 1, 1 } },
3419 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i64, {11, 1, 1, 1 } },
3420 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i64, {18, 1, 1, 1 } },
3421 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, { 5, 1, 1, 1 } },
3422 { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i64, {10, 1, 1, 1 } },
3423
3424 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v8f32, { 2, 1, 1, 1 } },
3425 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v4f64, { 2, 1, 1, 1 } },
3426 { ISD::FP_TO_SINT, MVT::v32i8, MVT::v8f32, { 2, 1, 1, 1 } },
3427 { ISD::FP_TO_SINT, MVT::v32i8, MVT::v4f64, { 2, 1, 1, 1 } },
3428 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v8f32, { 2, 1, 1, 1 } },
3429 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v4f64, { 2, 1, 1, 1 } },
3430 { ISD::FP_TO_SINT, MVT::v16i16, MVT::v8f32, { 2, 1, 1, 1 } },
3431 { ISD::FP_TO_SINT, MVT::v16i16, MVT::v4f64, { 2, 1, 1, 1 } },
3432 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f64, { 2, 1, 1, 1 } },
3433 { ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f32, { 2, 1, 1, 1 } },
3434 { ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f64, { 5, 1, 1, 1 } },
3435
3436 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v8f32, { 2, 1, 1, 1 } },
3437 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v4f64, { 2, 1, 1, 1 } },
3438 { ISD::FP_TO_UINT, MVT::v32i8, MVT::v8f32, { 2, 1, 1, 1 } },
3439 { ISD::FP_TO_UINT, MVT::v32i8, MVT::v4f64, { 2, 1, 1, 1 } },
3440 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v8f32, { 2, 1, 1, 1 } },
3441 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v4f64, { 2, 1, 1, 1 } },
3442 { ISD::FP_TO_UINT, MVT::v16i16, MVT::v8f32, { 2, 1, 1, 1 } },
3443 { ISD::FP_TO_UINT, MVT::v16i16, MVT::v4f64, { 2, 1, 1, 1 } },
3444 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, { 3, 1, 1, 1 } },
3445 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v2f64, { 4, 1, 1, 1 } },
3446 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f64, { 6, 1, 1, 1 } },
3447 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f32, { 7, 1, 1, 1 } },
3448 { ISD::FP_TO_UINT, MVT::v8i32, MVT::v4f64, { 7, 1, 1, 1 } },
3449
3450 { ISD::FP_EXTEND, MVT::v4f64, MVT::v4f32, { 1, 1, 1, 1 } },
3451 { ISD::FP_ROUND, MVT::v4f32, MVT::v4f64, { 1, 1, 1, 1 } },
3452 };
3453
3454 static const TypeConversionCostKindTblEntry SSE41ConversionTbl[] = {
3455 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v16i8, { 1, 1, 1, 1 } },
3456 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v16i8, { 1, 1, 1, 1 } },
3457 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v16i8, { 1, 1, 1, 1 } },
3458 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v16i8, { 1, 1, 1, 1 } },
3459 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v16i8, { 1, 1, 1, 1 } },
3460 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v16i8, { 1, 1, 1, 1 } },
3461 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v8i16, { 1, 1, 1, 1 } },
3462 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v8i16, { 1, 1, 1, 1 } },
3463 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v8i16, { 1, 1, 1, 1 } },
3464 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v8i16, { 1, 1, 1, 1 } },
3465 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v4i32, { 1, 1, 1, 1 } },
3466 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v4i32, { 1, 1, 1, 1 } },
3467
3468 // These truncates end up widening elements.
3469 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 1, 1, 1, 1 } }, // PMOVXZBQ
3470 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 1, 1, 1, 1 } }, // PMOVXZWQ
3471 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 1, 1, 1, 1 } }, // PMOVXZBD
3472
3473 { ISD::TRUNCATE, MVT::v16i8, MVT::v4i32, { 2, 1, 1, 1 } },
3474 { ISD::TRUNCATE, MVT::v8i16, MVT::v4i32, { 2, 1, 1, 1 } },
3475 { ISD::TRUNCATE, MVT::v16i8, MVT::v2i64, { 2, 1, 1, 1 } },
3476
3477 { ISD::SINT_TO_FP, MVT::f32, MVT::i32, { 1, 1, 1, 1 } },
3478 { ISD::SINT_TO_FP, MVT::f64, MVT::i32, { 1, 1, 1, 1 } },
3479 { ISD::SINT_TO_FP, MVT::f32, MVT::i64, { 1, 1, 1, 1 } },
3480 { ISD::SINT_TO_FP, MVT::f64, MVT::i64, { 1, 1, 1, 1 } },
3481 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v16i8, { 1, 1, 1, 1 } },
3482 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v16i8, { 1, 1, 1, 1 } },
3483 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v8i16, { 1, 1, 1, 1 } },
3484 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v8i16, { 1, 1, 1, 1 } },
3485 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i32, { 1, 1, 1, 1 } },
3486 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v4i32, { 1, 1, 1, 1 } },
3487 { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i32, { 2, 1, 1, 1 } },
3488
3489 { ISD::UINT_TO_FP, MVT::f32, MVT::i32, { 1, 1, 1, 1 } },
3490 { ISD::UINT_TO_FP, MVT::f64, MVT::i32, { 1, 1, 1, 1 } },
3491 { ISD::UINT_TO_FP, MVT::f32, MVT::i64, { 4, 1, 1, 1 } },
3492 { ISD::UINT_TO_FP, MVT::f64, MVT::i64, { 4, 1, 1, 1 } },
3493 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v16i8, { 1, 1, 1, 1 } },
3494 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v16i8, { 1, 1, 1, 1 } },
3495 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v8i16, { 1, 1, 1, 1 } },
3496 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v8i16, { 1, 1, 1, 1 } },
3497 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, { 3, 1, 1, 1 } },
3498 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, { 3, 1, 1, 1 } },
3499 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v4i32, { 2, 1, 1, 1 } },
3500 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v2i64, {12, 1, 1, 1 } },
3501 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i64, {22, 1, 1, 1 } },
3502 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, { 4, 1, 1, 1 } },
3503
3504 { ISD::FP_TO_SINT, MVT::i32, MVT::f32, { 1, 1, 1, 1 } },
3505 { ISD::FP_TO_SINT, MVT::i64, MVT::f32, { 1, 1, 1, 1 } },
3506 { ISD::FP_TO_SINT, MVT::i32, MVT::f64, { 1, 1, 1, 1 } },
3507 { ISD::FP_TO_SINT, MVT::i64, MVT::f64, { 1, 1, 1, 1 } },
3508 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v4f32, { 2, 1, 1, 1 } },
3509 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v2f64, { 2, 1, 1, 1 } },
3510 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v4f32, { 1, 1, 1, 1 } },
3511 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v2f64, { 1, 1, 1, 1 } },
3512 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f32, { 1, 1, 1, 1 } },
3513 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v2f64, { 1, 1, 1, 1 } },
3514
3515 { ISD::FP_TO_UINT, MVT::i32, MVT::f32, { 1, 1, 1, 1 } },
3516 { ISD::FP_TO_UINT, MVT::i64, MVT::f32, { 4, 1, 1, 1 } },
3517 { ISD::FP_TO_UINT, MVT::i32, MVT::f64, { 1, 1, 1, 1 } },
3518 { ISD::FP_TO_UINT, MVT::i64, MVT::f64, { 4, 1, 1, 1 } },
3519 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v4f32, { 2, 1, 1, 1 } },
3520 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v2f64, { 2, 1, 1, 1 } },
3521 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v4f32, { 1, 1, 1, 1 } },
3522 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v2f64, { 1, 1, 1, 1 } },
3523 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, { 4, 1, 1, 1 } },
3524 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v2f64, { 4, 1, 1, 1 } },
3525 };
3526
3527 static const TypeConversionCostKindTblEntry SSE2ConversionTbl[] = {
3528 // These are somewhat magic numbers justified by comparing the
3529 // output of llvm-mca for our various supported scheduler models
3530 // and basing it off the worst case scenario.
3531 { ISD::SINT_TO_FP, MVT::f32, MVT::i32, { 3, 1, 1, 1 } },
3532 { ISD::SINT_TO_FP, MVT::f64, MVT::i32, { 3, 1, 1, 1 } },
3533 { ISD::SINT_TO_FP, MVT::f32, MVT::i64, { 3, 1, 1, 1 } },
3534 { ISD::SINT_TO_FP, MVT::f64, MVT::i64, { 3, 1, 1, 1 } },
3535 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v16i8, { 3, 1, 1, 1 } },
3536 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v16i8, { 4, 1, 1, 1 } },
3537 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v8i16, { 3, 1, 1, 1 } },
3538 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v8i16, { 4, 1, 1, 1 } },
3539 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i32, { 3, 1, 1, 1 } },
3540 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v4i32, { 4, 1, 1, 1 } },
3541 { ISD::SINT_TO_FP, MVT::v4f32, MVT::v2i64, { 8, 1, 1, 1 } },
3542 { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i64, { 8, 1, 1, 1 } },
3543
3544 { ISD::UINT_TO_FP, MVT::f32, MVT::i32, { 3, 1, 1, 1 } },
3545 { ISD::UINT_TO_FP, MVT::f64, MVT::i32, { 3, 1, 1, 1 } },
3546 { ISD::UINT_TO_FP, MVT::f32, MVT::i64, { 8, 1, 1, 1 } },
3547 { ISD::UINT_TO_FP, MVT::f64, MVT::i64, { 9, 1, 1, 1 } },
3548 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v16i8, { 4, 1, 1, 1 } },
3549 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v16i8, { 4, 1, 1, 1 } },
3550 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v8i16, { 4, 1, 1, 1 } },
3551 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v8i16, { 4, 1, 1, 1 } },
3552 { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, { 7, 1, 1, 1 } },
3553 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v4i32, { 7, 1, 1, 1 } },
3554 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, { 5, 1, 1, 1 } },
3555 { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, {15, 1, 1, 1 } },
3556 { ISD::UINT_TO_FP, MVT::v4f32, MVT::v2i64, {18, 1, 1, 1 } },
3557
3558 { ISD::FP_TO_SINT, MVT::i32, MVT::f32, { 4, 1, 1, 1 } },
3559 { ISD::FP_TO_SINT, MVT::i64, MVT::f32, { 4, 1, 1, 1 } },
3560 { ISD::FP_TO_SINT, MVT::i32, MVT::f64, { 4, 1, 1, 1 } },
3561 { ISD::FP_TO_SINT, MVT::i64, MVT::f64, { 4, 1, 1, 1 } },
3562 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v4f32, { 6, 1, 1, 1 } },
3563 { ISD::FP_TO_SINT, MVT::v16i8, MVT::v2f64, { 6, 1, 1, 1 } },
3564 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v4f32, { 5, 1, 1, 1 } },
3565 { ISD::FP_TO_SINT, MVT::v8i16, MVT::v2f64, { 5, 1, 1, 1 } },
3566 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f32, { 4, 1, 1, 1 } },
3567 { ISD::FP_TO_SINT, MVT::v4i32, MVT::v2f64, { 4, 1, 1, 1 } },
3568
3569 { ISD::FP_TO_UINT, MVT::i32, MVT::f32, { 4, 1, 1, 1 } },
3570 { ISD::FP_TO_UINT, MVT::i64, MVT::f32, { 4, 1, 1, 1 } },
3571 { ISD::FP_TO_UINT, MVT::i32, MVT::f64, { 4, 1, 1, 1 } },
3572 { ISD::FP_TO_UINT, MVT::i64, MVT::f64, {15, 1, 1, 1 } },
3573 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v4f32, { 6, 1, 1, 1 } },
3574 { ISD::FP_TO_UINT, MVT::v16i8, MVT::v2f64, { 6, 1, 1, 1 } },
3575 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v4f32, { 5, 1, 1, 1 } },
3576 { ISD::FP_TO_UINT, MVT::v8i16, MVT::v2f64, { 5, 1, 1, 1 } },
3577 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, { 8, 1, 1, 1 } },
3578 { ISD::FP_TO_UINT, MVT::v4i32, MVT::v2f64, { 8, 1, 1, 1 } },
3579
3580 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v16i8, { 4, 1, 1, 1 } },
3581 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v16i8, { 4, 1, 1, 1 } },
3582 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v16i8, { 2, 1, 1, 1 } },
3583 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v16i8, { 3, 1, 1, 1 } },
3584 { ISD::ZERO_EXTEND, MVT::v8i16, MVT::v16i8, { 1, 1, 1, 1 } },
3585 { ISD::SIGN_EXTEND, MVT::v8i16, MVT::v16i8, { 2, 1, 1, 1 } },
3586 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v8i16, { 2, 1, 1, 1 } },
3587 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v8i16, { 3, 1, 1, 1 } },
3588 { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v8i16, { 1, 1, 1, 1 } },
3589 { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v8i16, { 2, 1, 1, 1 } },
3590 { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v4i32, { 1, 1, 1, 1 } },
3591 { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v4i32, { 2, 1, 1, 1 } },
3592
3593 // These truncates are really widening elements.
3594 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i32, { 1, 1, 1, 1 } }, // PSHUFD
3595 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i16, { 2, 1, 1, 1 } }, // PUNPCKLWD+DQ
3596 { ISD::TRUNCATE, MVT::v2i1, MVT::v2i8, { 3, 1, 1, 1 } }, // PUNPCKLBW+WD+PSHUFD
3597 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i16, { 1, 1, 1, 1 } }, // PUNPCKLWD
3598 { ISD::TRUNCATE, MVT::v4i1, MVT::v4i8, { 2, 1, 1, 1 } }, // PUNPCKLBW+WD
3599 { ISD::TRUNCATE, MVT::v8i1, MVT::v8i8, { 1, 1, 1, 1 } }, // PUNPCKLBW
3600
3601 { ISD::TRUNCATE, MVT::v16i8, MVT::v8i16, { 2, 1, 1, 1 } }, // PAND+PACKUSWB
3602 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i16, { 3, 1, 1, 1 } },
3603 { ISD::TRUNCATE, MVT::v16i8, MVT::v4i32, { 3, 1, 1, 1 } }, // PAND+2*PACKUSWB
3604 { ISD::TRUNCATE, MVT::v16i8, MVT::v16i32, { 7, 1, 1, 1 } },
3605 { ISD::TRUNCATE, MVT::v2i16, MVT::v2i32, { 1, 1, 1, 1 } },
3606 { ISD::TRUNCATE, MVT::v8i16, MVT::v4i32, { 3, 1, 1, 1 } },
3607 { ISD::TRUNCATE, MVT::v8i16, MVT::v8i32, { 5, 1, 1, 1 } },
3608 { ISD::TRUNCATE, MVT::v16i16, MVT::v16i32, {10, 1, 1, 1 } },
3609 { ISD::TRUNCATE, MVT::v16i8, MVT::v2i64, { 4, 1, 1, 1 } }, // PAND+3*PACKUSWB
3610 { ISD::TRUNCATE, MVT::v8i16, MVT::v2i64, { 2, 1, 1, 1 } }, // PSHUFD+PSHUFLW
3611 { ISD::TRUNCATE, MVT::v4i32, MVT::v2i64, { 1, 1, 1, 1 } }, // PSHUFD
3612 };
3613
3614 static const TypeConversionCostKindTblEntry F16ConversionTbl[] = {
3615 { ISD::FP_ROUND, MVT::f16, MVT::f32, { 1, 1, 1, 1 } },
3616 { ISD::FP_ROUND, MVT::v8f16, MVT::v8f32, { 1, 1, 1, 1 } },
3617 { ISD::FP_ROUND, MVT::v4f16, MVT::v4f32, { 1, 1, 1, 1 } },
3618 { ISD::FP_EXTEND, MVT::f32, MVT::f16, { 1, 1, 1, 1 } },
3619 { ISD::FP_EXTEND, MVT::f64, MVT::f16, { 2, 1, 1, 1 } }, // vcvtph2ps+vcvtps2pd
3620 { ISD::FP_EXTEND, MVT::v8f32, MVT::v8f16, { 1, 1, 1, 1 } },
3621 { ISD::FP_EXTEND, MVT::v4f32, MVT::v4f16, { 1, 1, 1, 1 } },
3622 { ISD::FP_EXTEND, MVT::v4f64, MVT::v4f16, { 2, 1, 1, 1 } }, // vcvtph2ps+vcvtps2pd
3623 };
3624
3625 // Attempt to map directly to (simple) MVT types to let us match custom entries.
3626 EVT SrcTy = TLI->getValueType(DL, Src);
3627 EVT DstTy = TLI->getValueType(DL, Dst);
3628
3629 // If we're sign-extending a vector comparison result back to the comparison
3630 // width, this will be free without AVX512 (or for 8/16-bit types without
3631 // BWI).
3632 if (!ST->hasAVX512() || (!ST->hasBWI() && DstTy.getScalarSizeInBits() < 32)) {
3633 if (I && Opcode == Instruction::CastOps::SExt &&
3634 SrcTy.isFixedLengthVectorOf(MVT::i1)) {
3635 if (auto *CmpI = dyn_cast<CmpInst>(I->getOperand(0))) {
3636 Type *CmpTy = CmpI->getOperand(0)->getType();
3637 if (CmpTy->getScalarSizeInBits() == DstTy.getScalarSizeInBits())
3638 return TTI::TCC_Free;
3639 }
3640 }
3641 }
3642
3643 // The function getSimpleVT only handles simple value types.
3644 if (SrcTy.isSimple() && DstTy.isSimple()) {
3645 MVT SimpleSrcTy = SrcTy.getSimpleVT();
3646 MVT SimpleDstTy = DstTy.getSimpleVT();
3647
3648 if (ST->useAVX512Regs()) {
3649 if (ST->hasBWI())
3650 if (const auto *Entry = ConvertCostTableLookup(
3651 AVX512BWConversionTbl, ISD, SimpleDstTy, SimpleSrcTy))
3652 if (auto KindCost = Entry->Cost[CostKind])
3653 return *KindCost;
3654
3655 if (ST->hasDQI())
3656 if (const auto *Entry = ConvertCostTableLookup(
3657 AVX512DQConversionTbl, ISD, SimpleDstTy, SimpleSrcTy))
3658 if (auto KindCost = Entry->Cost[CostKind])
3659 return *KindCost;
3660
3661 if (ST->hasAVX512())
3662 if (const auto *Entry = ConvertCostTableLookup(
3663 AVX512FConversionTbl, ISD, SimpleDstTy, SimpleSrcTy))
3664 if (auto KindCost = Entry->Cost[CostKind])
3665 return *KindCost;
3666 }
3667
3668 if (ST->hasBWI())
3669 if (const auto *Entry = ConvertCostTableLookup(
3670 AVX512BWVLConversionTbl, ISD, SimpleDstTy, SimpleSrcTy))
3671 if (auto KindCost = Entry->Cost[CostKind])
3672 return *KindCost;
3673
3674 if (ST->hasDQI())
3675 if (const auto *Entry = ConvertCostTableLookup(
3676 AVX512DQVLConversionTbl, ISD, SimpleDstTy, SimpleSrcTy))
3677 if (auto KindCost = Entry->Cost[CostKind])
3678 return *KindCost;
3679
3680 if (ST->hasAVX512())
3681 if (const auto *Entry = ConvertCostTableLookup(AVX512VLConversionTbl, ISD,
3682 SimpleDstTy, SimpleSrcTy))
3683 if (auto KindCost = Entry->Cost[CostKind])
3684 return *KindCost;
3685
3686 if (ST->hasAVX2()) {
3687 if (const auto *Entry = ConvertCostTableLookup(AVX2ConversionTbl, ISD,
3688 SimpleDstTy, SimpleSrcTy))
3689 if (auto KindCost = Entry->Cost[CostKind])
3690 return *KindCost;
3691 }
3692
3693 if (ST->hasAVX()) {
3694 if (const auto *Entry = ConvertCostTableLookup(AVXConversionTbl, ISD,
3695 SimpleDstTy, SimpleSrcTy))
3696 if (auto KindCost = Entry->Cost[CostKind])
3697 return *KindCost;
3698 }
3699
3700 if (ST->hasF16C()) {
3701 if (const auto *Entry = ConvertCostTableLookup(F16ConversionTbl, ISD,
3702 SimpleDstTy, SimpleSrcTy))
3703 if (auto KindCost = Entry->Cost[CostKind])
3704 return *KindCost;
3705 }
3706
3707 if (ST->hasSSE41()) {
3708 if (const auto *Entry = ConvertCostTableLookup(SSE41ConversionTbl, ISD,
3709 SimpleDstTy, SimpleSrcTy))
3710 if (auto KindCost = Entry->Cost[CostKind])
3711 return *KindCost;
3712 }
3713
3714 if (ST->hasSSE2()) {
3715 if (const auto *Entry = ConvertCostTableLookup(SSE2ConversionTbl, ISD,
3716 SimpleDstTy, SimpleSrcTy))
3717 if (auto KindCost = Entry->Cost[CostKind])
3718 return *KindCost;
3719 }
3720
3721 if ((ISD == ISD::FP_ROUND && SimpleDstTy == MVT::f16) ||
3722 (ISD == ISD::FP_EXTEND && SimpleSrcTy == MVT::f16)) {
3723 // fp16 conversions not covered by any table entries require a libcall.
3724 // Return a large (arbitrary) number to model this.
3725 return InstructionCost(64);
3726 }
3727 }
3728
3729 // Fall back to legalized types.
3730 std::pair<InstructionCost, MVT> LTSrc = getTypeLegalizationCost(Src);
3731 std::pair<InstructionCost, MVT> LTDest = getTypeLegalizationCost(Dst);
3732
3733 // If we're truncating to the same legalized type - just assume its free.
3734 if (ISD == ISD::TRUNCATE && LTSrc.second == LTDest.second)
3735 return TTI::TCC_Free;
3736
3737 if (ST->useAVX512Regs()) {
3738 if (ST->hasBWI())
3739 if (const auto *Entry = ConvertCostTableLookup(
3740 AVX512BWConversionTbl, ISD, LTDest.second, LTSrc.second))
3741 if (auto KindCost = Entry->Cost[CostKind])
3742 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3743
3744 if (ST->hasDQI())
3745 if (const auto *Entry = ConvertCostTableLookup(
3746 AVX512DQConversionTbl, ISD, LTDest.second, LTSrc.second))
3747 if (auto KindCost = Entry->Cost[CostKind])
3748 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3749
3750 if (ST->hasAVX512())
3751 if (const auto *Entry = ConvertCostTableLookup(
3752 AVX512FConversionTbl, ISD, LTDest.second, LTSrc.second))
3753 if (auto KindCost = Entry->Cost[CostKind])
3754 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3755 }
3756
3757 if (ST->hasBWI())
3758 if (const auto *Entry = ConvertCostTableLookup(AVX512BWVLConversionTbl, ISD,
3759 LTDest.second, LTSrc.second))
3760 if (auto KindCost = Entry->Cost[CostKind])
3761 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3762
3763 if (ST->hasDQI())
3764 if (const auto *Entry = ConvertCostTableLookup(AVX512DQVLConversionTbl, ISD,
3765 LTDest.second, LTSrc.second))
3766 if (auto KindCost = Entry->Cost[CostKind])
3767 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3768
3769 if (ST->hasAVX512())
3770 if (const auto *Entry = ConvertCostTableLookup(AVX512VLConversionTbl, ISD,
3771 LTDest.second, LTSrc.second))
3772 if (auto KindCost = Entry->Cost[CostKind])
3773 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3774
3775 if (ST->hasAVX2())
3776 if (const auto *Entry = ConvertCostTableLookup(AVX2ConversionTbl, ISD,
3777 LTDest.second, LTSrc.second))
3778 if (auto KindCost = Entry->Cost[CostKind])
3779 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3780
3781 if (ST->hasAVX())
3782 if (const auto *Entry = ConvertCostTableLookup(AVXConversionTbl, ISD,
3783 LTDest.second, LTSrc.second))
3784 if (auto KindCost = Entry->Cost[CostKind])
3785 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3786
3787 if (ST->hasF16C()) {
3788 if (const auto *Entry = ConvertCostTableLookup(F16ConversionTbl, ISD,
3789 LTDest.second, LTSrc.second))
3790 if (auto KindCost = Entry->Cost[CostKind])
3791 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3792 }
3793
3794 if (ST->hasSSE41())
3795 if (const auto *Entry = ConvertCostTableLookup(SSE41ConversionTbl, ISD,
3796 LTDest.second, LTSrc.second))
3797 if (auto KindCost = Entry->Cost[CostKind])
3798 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3799
3800 if (ST->hasSSE2())
3801 if (const auto *Entry = ConvertCostTableLookup(SSE2ConversionTbl, ISD,
3802 LTDest.second, LTSrc.second))
3803 if (auto KindCost = Entry->Cost[CostKind])
3804 return std::max(LTSrc.first, LTDest.first) * *KindCost;
3805
3806 // Fallback, for i8/i16 sitofp/uitofp cases we need to extend to i32 for
3807 // sitofp.
3808 if ((ISD == ISD::SINT_TO_FP || ISD == ISD::UINT_TO_FP) &&
3809 1 < Src->getScalarSizeInBits() && Src->getScalarSizeInBits() < 32) {
3810 Type *ExtSrc = Src->getWithNewBitWidth(32);
3811 unsigned ExtOpc =
3812 (ISD == ISD::SINT_TO_FP) ? Instruction::SExt : Instruction::ZExt;
3813
3814 // For scalar loads the extend would be free.
3815 InstructionCost ExtCost = 0;
3816 if (!(Src->isIntegerTy() && I && isa<LoadInst>(I->getOperand(0))))
3817 ExtCost = getCastInstrCost(ExtOpc, ExtSrc, Src, CCH, CostKind);
3818
3819 return ExtCost + getCastInstrCost(Instruction::SIToFP, Dst, ExtSrc,
3821 }
3822
3823 // Fallback for fptosi/fptoui i8/i16 cases we need to truncate from fptosi
3824 // i32.
3825 if ((ISD == ISD::FP_TO_SINT || ISD == ISD::FP_TO_UINT) &&
3826 1 < Dst->getScalarSizeInBits() && Dst->getScalarSizeInBits() < 32) {
3827 Type *TruncDst = Dst->getWithNewBitWidth(32);
3828 return getCastInstrCost(Instruction::FPToSI, TruncDst, Src, CCH, CostKind) +
3829 getCastInstrCost(Instruction::Trunc, Dst, TruncDst,
3831 }
3832
3833 // TODO: Allow non-throughput costs that aren't binary.
3834 auto AdjustCost = [&CostKind](InstructionCost Cost,
3837 return Cost == 0 ? 0 : N;
3838 return Cost * N;
3839 };
3840 return AdjustCost(
3841 BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
3842}
3843
3844// Additive cost for a predicated op's "predicate fanout": a select, masked
3845// load/store or gather/scatter keeps a compact <N x i1> mask, but if its data
3846// legalizes into more parts than the mask, codegen builds a sub-mask per extra
3847// part (kshiftr on AVX-512, unpack/extend on SSE/AVX) the per-part tables miss.
3848// FanoutCostPerPart is a tie-break: it beats the vectorizer's widest-VF bias.
3850 Type *DataVTy, Type *MaskVTy) {
3851 constexpr unsigned FanoutCostPerPart = 4;
3852 unsigned DataParts = TTI.getNumberOfParts(DataVTy);
3853 unsigned MaskParts = TTI.getNumberOfParts(MaskVTy);
3854 if (DataParts > MaskParts && MaskParts > 0)
3855 return InstructionCost((DataParts - MaskParts) * FanoutCostPerPart);
3856 return InstructionCost(0);
3857}
3858
3860 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
3862 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
3863 // Early out if this type isn't scalar/vector integer/float.
3864 if (!(ValTy->isIntOrIntVectorTy() || ValTy->isFPOrFPVectorTy()))
3865 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
3866 Op1Info, Op2Info, I);
3867
3868 // Legalize the type.
3869 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
3870
3871 MVT MTy = LT.second;
3872
3873 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3874 assert(ISD && "Invalid opcode");
3875
3876 // Predicate fanout (see getPredicateFanoutCost), AVX-512 only: on SSE/AVX a
3877 // select blend is already priced per part, so charging it there shrinks VF.
3878 InstructionCost MaskExpCost = 0;
3879 if (Opcode == Instruction::Select && ST->hasAVX512() &&
3881 MaskExpCost = getPredicateFanoutCost(*this, ValTy, CondTy);
3882
3883 InstructionCost ExtraCost = 0;
3884 if (Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) {
3885 // Some vector comparison predicates cost extra instructions.
3886 // TODO: Adjust ExtraCost based on CostKind?
3887 // TODO: Should we invert this and assume worst case cmp costs
3888 // and reduce for particular predicates?
3889 if (MTy.isVector() &&
3890 !((ST->hasXOP() && (!ST->hasAVX2() || MTy.is128BitVector())) ||
3891 (ST->hasAVX512() && 32 <= MTy.getScalarSizeInBits()) ||
3892 ST->hasBWI())) {
3893 // Fallback to I if a specific predicate wasn't specified.
3894 CmpInst::Predicate Pred = VecPred;
3895 if (I && (Pred == CmpInst::BAD_ICMP_PREDICATE ||
3897 Pred = cast<CmpInst>(I)->getPredicate();
3898
3899 bool CmpWithConstant = false;
3900 if (auto *CmpInstr = dyn_cast_or_null<CmpInst>(I))
3901 CmpWithConstant = isa<Constant>(CmpInstr->getOperand(1));
3902
3903 switch (Pred) {
3905 // xor(cmpeq(x,y),-1)
3906 ExtraCost = CmpWithConstant ? 0 : 1;
3907 break;
3910 // xor(cmpgt(x,y),-1)
3911 ExtraCost = CmpWithConstant ? 0 : 1;
3912 break;
3915 // cmpgt(xor(x,signbit),xor(y,signbit))
3916 // xor(cmpeq(pmaxu(x,y),x),-1)
3917 ExtraCost = CmpWithConstant ? 1 : 2;
3918 break;
3921 if ((ST->hasSSE41() && MTy.getScalarSizeInBits() == 32) ||
3922 (ST->hasSSE2() && MTy.getScalarSizeInBits() < 32)) {
3923 // cmpeq(psubus(x,y),0)
3924 // cmpeq(pminu(x,y),x)
3925 ExtraCost = 1;
3926 } else {
3927 // xor(cmpgt(xor(x,signbit),xor(y,signbit)),-1)
3928 ExtraCost = CmpWithConstant ? 2 : 3;
3929 }
3930 break;
3933 // Without AVX we need to expand FCMP_ONE/FCMP_UEQ cases.
3934 // Use FCMP_UEQ expansion - FCMP_ONE should be the same.
3935 if (CondTy && !ST->hasAVX())
3936 return getCmpSelInstrCost(Opcode, ValTy, CondTy,
3938 Op1Info, Op2Info) +
3939 getCmpSelInstrCost(Opcode, ValTy, CondTy,
3941 Op1Info, Op2Info) +
3942 getArithmeticInstrCost(Instruction::Or, CondTy, CostKind);
3943
3944 break;
3947 // Assume worst case scenario and add the maximum extra cost.
3948 ExtraCost = 3;
3949 break;
3950 default:
3951 break;
3952 }
3953 }
3954 }
3955
3956 static const CostKindTblEntry SLMCostTbl[] = {
3957 // slm pcmpeq/pcmpgt throughput is 2
3958 { ISD::SETCC, MVT::v2i64, { 2, 5, 1, 2 } },
3959 // slm pblendvb/blendvpd/blendvps throughput is 4
3960 { ISD::SELECT, MVT::v2f64, { 4, 4, 1, 3 } }, // vblendvpd
3961 { ISD::SELECT, MVT::v4f32, { 4, 4, 1, 3 } }, // vblendvps
3962 { ISD::SELECT, MVT::v2i64, { 4, 4, 1, 3 } }, // pblendvb
3963 { ISD::SELECT, MVT::v8i32, { 4, 4, 1, 3 } }, // pblendvb
3964 { ISD::SELECT, MVT::v8i16, { 4, 4, 1, 3 } }, // pblendvb
3965 { ISD::SELECT, MVT::v16i8, { 4, 4, 1, 3 } }, // pblendvb
3966 };
3967
3968 static const CostKindTblEntry AVX512BWCostTbl[] = {
3969 { ISD::SETCC, MVT::v32i16, { 1, 1, 1, 1 } },
3970 { ISD::SETCC, MVT::v16i16, { 1, 1, 1, 1 } },
3971 { ISD::SETCC, MVT::v64i8, { 1, 1, 1, 1 } },
3972 { ISD::SETCC, MVT::v32i8, { 1, 1, 1, 1 } },
3973
3974 { ISD::SELECT, MVT::v32i16, { 1, 1, 1, 1 } },
3975 { ISD::SELECT, MVT::v64i8, { 1, 1, 1, 1 } },
3976 };
3977
3978 static const CostKindTblEntry AVX512CostTbl[] = {
3979 { ISD::SETCC, MVT::v8f64, { 1, 4, 1, 1 } },
3980 { ISD::SETCC, MVT::v4f64, { 1, 4, 1, 1 } },
3981 { ISD::SETCC, MVT::v16f32, { 1, 4, 1, 1 } },
3982 { ISD::SETCC, MVT::v8f32, { 1, 4, 1, 1 } },
3983
3984 { ISD::SETCC, MVT::v8i64, { 1, 1, 1, 1 } },
3985 { ISD::SETCC, MVT::v4i64, { 1, 1, 1, 1 } },
3986 { ISD::SETCC, MVT::v2i64, { 1, 1, 1, 1 } },
3987 { ISD::SETCC, MVT::v16i32, { 1, 1, 1, 1 } },
3988 { ISD::SETCC, MVT::v8i32, { 1, 1, 1, 1 } },
3989 { ISD::SETCC, MVT::v32i16, { 3, 7, 5, 5 } },
3990 { ISD::SETCC, MVT::v64i8, { 3, 7, 5, 5 } },
3991
3992 { ISD::SELECT, MVT::v8i64, { 1, 1, 1, 1 } },
3993 { ISD::SELECT, MVT::v4i64, { 1, 1, 1, 1 } },
3994 { ISD::SELECT, MVT::v2i64, { 1, 1, 1, 1 } },
3995 { ISD::SELECT, MVT::v16i32, { 1, 1, 1, 1 } },
3996 { ISD::SELECT, MVT::v8i32, { 1, 1, 1, 1 } },
3997 { ISD::SELECT, MVT::v4i32, { 1, 1, 1, 1 } },
3998 { ISD::SELECT, MVT::v8f64, { 1, 1, 1, 1 } },
3999 { ISD::SELECT, MVT::v4f64, { 1, 1, 1, 1 } },
4000 { ISD::SELECT, MVT::v2f64, { 1, 1, 1, 1 } },
4001 { ISD::SELECT, MVT::f64, { 1, 1, 1, 1 } },
4002 { ISD::SELECT, MVT::v16f32, { 1, 1, 1, 1 } },
4003 { ISD::SELECT, MVT::v8f32 , { 1, 1, 1, 1 } },
4004 { ISD::SELECT, MVT::v4f32, { 1, 1, 1, 1 } },
4005 { ISD::SELECT, MVT::f32 , { 1, 1, 1, 1 } },
4006
4007 { ISD::SELECT, MVT::v32i16, { 2, 2, 4, 4 } },
4008 { ISD::SELECT, MVT::v16i16, { 1, 1, 1, 1 } },
4009 { ISD::SELECT, MVT::v8i16, { 1, 1, 1, 1 } },
4010 { ISD::SELECT, MVT::v64i8, { 2, 2, 4, 4 } },
4011 { ISD::SELECT, MVT::v32i8, { 1, 1, 1, 1 } },
4012 { ISD::SELECT, MVT::v16i8, { 1, 1, 1, 1 } },
4013 };
4014
4015 static const CostKindTblEntry AVX2CostTbl[] = {
4016 { ISD::SETCC, MVT::v4f64, { 1, 4, 1, 2 } },
4017 { ISD::SETCC, MVT::v2f64, { 1, 4, 1, 1 } },
4018 { ISD::SETCC, MVT::f64, { 1, 4, 1, 1 } },
4019 { ISD::SETCC, MVT::v8f32, { 1, 4, 1, 2 } },
4020 { ISD::SETCC, MVT::v4f32, { 1, 4, 1, 1 } },
4021 { ISD::SETCC, MVT::f32, { 1, 4, 1, 1 } },
4022
4023 { ISD::SETCC, MVT::v4i64, { 1, 1, 1, 2 } },
4024 { ISD::SETCC, MVT::v8i32, { 1, 1, 1, 2 } },
4025 { ISD::SETCC, MVT::v16i16, { 1, 1, 1, 2 } },
4026 { ISD::SETCC, MVT::v32i8, { 1, 1, 1, 2 } },
4027
4028 { ISD::SELECT, MVT::v4f64, { 2, 2, 1, 2 } }, // vblendvpd
4029 { ISD::SELECT, MVT::v8f32, { 2, 2, 1, 2 } }, // vblendvps
4030 { ISD::SELECT, MVT::v4i64, { 2, 2, 1, 2 } }, // pblendvb
4031 { ISD::SELECT, MVT::v8i32, { 2, 2, 1, 2 } }, // pblendvb
4032 { ISD::SELECT, MVT::v16i16, { 2, 2, 1, 2 } }, // pblendvb
4033 { ISD::SELECT, MVT::v32i8, { 2, 2, 1, 2 } }, // pblendvb
4034 };
4035
4036 static const CostKindTblEntry XOPCostTbl[] = {
4037 { ISD::SETCC, MVT::v4i64, { 4, 2, 5, 6 } },
4038 { ISD::SETCC, MVT::v2i64, { 1, 1, 1, 1 } },
4039 };
4040
4041 static const CostKindTblEntry AVX1CostTbl[] = {
4042 { ISD::SETCC, MVT::v4f64, { 2, 3, 1, 2 } },
4043 { ISD::SETCC, MVT::v2f64, { 1, 3, 1, 1 } },
4044 { ISD::SETCC, MVT::f64, { 1, 3, 1, 1 } },
4045 { ISD::SETCC, MVT::v8f32, { 2, 3, 1, 2 } },
4046 { ISD::SETCC, MVT::v4f32, { 1, 3, 1, 1 } },
4047 { ISD::SETCC, MVT::f32, { 1, 3, 1, 1 } },
4048
4049 // AVX1 does not support 8-wide integer compare.
4050 { ISD::SETCC, MVT::v4i64, { 4, 2, 5, 6 } },
4051 { ISD::SETCC, MVT::v8i32, { 4, 2, 5, 6 } },
4052 { ISD::SETCC, MVT::v16i16, { 4, 2, 5, 6 } },
4053 { ISD::SETCC, MVT::v32i8, { 4, 2, 5, 6 } },
4054
4055 { ISD::SELECT, MVT::v4f64, { 3, 3, 1, 2 } }, // vblendvpd
4056 { ISD::SELECT, MVT::v8f32, { 3, 3, 1, 2 } }, // vblendvps
4057 { ISD::SELECT, MVT::v4i64, { 3, 3, 1, 2 } }, // vblendvpd
4058 { ISD::SELECT, MVT::v8i32, { 3, 3, 1, 2 } }, // vblendvps
4059 { ISD::SELECT, MVT::v16i16, { 3, 3, 3, 3 } }, // vandps + vandnps + vorps
4060 { ISD::SELECT, MVT::v32i8, { 3, 3, 3, 3 } }, // vandps + vandnps + vorps
4061 };
4062
4063 static const CostKindTblEntry SSE42CostTbl[] = {
4064 { ISD::SETCC, MVT::v2i64, { 1, 2, 1, 2 } },
4065 };
4066
4067 static const CostKindTblEntry SSE41CostTbl[] = {
4068 { ISD::SETCC, MVT::v2f64, { 1, 5, 1, 1 } },
4069 { ISD::SETCC, MVT::v4f32, { 1, 5, 1, 1 } },
4070
4071 { ISD::SELECT, MVT::v2f64, { 2, 2, 1, 2 } }, // blendvpd
4072 { ISD::SELECT, MVT::f64, { 2, 2, 1, 2 } }, // blendvpd
4073 { ISD::SELECT, MVT::v4f32, { 2, 2, 1, 2 } }, // blendvps
4074 { ISD::SELECT, MVT::f32 , { 2, 2, 1, 2 } }, // blendvps
4075 { ISD::SELECT, MVT::v2i64, { 2, 2, 1, 2 } }, // pblendvb
4076 { ISD::SELECT, MVT::v4i32, { 2, 2, 1, 2 } }, // pblendvb
4077 { ISD::SELECT, MVT::v8i16, { 2, 2, 1, 2 } }, // pblendvb
4078 { ISD::SELECT, MVT::v16i8, { 2, 2, 1, 2 } }, // pblendvb
4079 };
4080
4081 static const CostKindTblEntry SSE2CostTbl[] = {
4082 { ISD::SETCC, MVT::v2f64, { 2, 5, 1, 1 } },
4083 { ISD::SETCC, MVT::f64, { 1, 5, 1, 1 } },
4084
4085 { ISD::SETCC, MVT::v2i64, { 5, 4, 5, 5 } }, // pcmpeqd/pcmpgtd expansion
4086 { ISD::SETCC, MVT::v4i32, { 1, 1, 1, 1 } },
4087 { ISD::SETCC, MVT::v8i16, { 1, 1, 1, 1 } },
4088 { ISD::SETCC, MVT::v16i8, { 1, 1, 1, 1 } },
4089
4090 { ISD::SELECT, MVT::v2f64, { 2, 2, 3, 3 } }, // andpd + andnpd + orpd
4091 { ISD::SELECT, MVT::f64, { 2, 2, 3, 3 } }, // andpd + andnpd + orpd
4092 { ISD::SELECT, MVT::v2i64, { 2, 2, 3, 3 } }, // pand + pandn + por
4093 { ISD::SELECT, MVT::v4i32, { 2, 2, 3, 3 } }, // pand + pandn + por
4094 { ISD::SELECT, MVT::v8i16, { 2, 2, 3, 3 } }, // pand + pandn + por
4095 { ISD::SELECT, MVT::v16i8, { 2, 2, 3, 3 } }, // pand + pandn + por
4096 };
4097
4098 static const CostKindTblEntry SSE1CostTbl[] = {
4099 { ISD::SETCC, MVT::v4f32, { 2, 5, 1, 1 } },
4100 { ISD::SETCC, MVT::f32, { 1, 5, 1, 1 } },
4101
4102 { ISD::SELECT, MVT::v4f32, { 2, 2, 3, 3 } }, // andps + andnps + orps
4103 { ISD::SELECT, MVT::f32, { 2, 2, 3, 3 } }, // andps + andnps + orps
4104 };
4105
4106 if (ST->useSLMArithCosts())
4107 if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
4108 if (auto KindCost = Entry->Cost[CostKind])
4109 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4110
4111 if (ST->hasBWI())
4112 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
4113 if (auto KindCost = Entry->Cost[CostKind])
4114 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4115
4116 if (ST->hasAVX512())
4117 if (const auto *Entry = CostTableLookup(AVX512CostTbl, ISD, MTy))
4118 if (auto KindCost = Entry->Cost[CostKind])
4119 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4120
4121 if (ST->hasAVX2())
4122 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
4123 if (auto KindCost = Entry->Cost[CostKind])
4124 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4125
4126 if (ST->hasXOP())
4127 if (const auto *Entry = CostTableLookup(XOPCostTbl, ISD, MTy))
4128 if (auto KindCost = Entry->Cost[CostKind])
4129 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4130
4131 if (ST->hasAVX())
4132 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
4133 if (auto KindCost = Entry->Cost[CostKind])
4134 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4135
4136 if (ST->hasSSE42())
4137 if (const auto *Entry = CostTableLookup(SSE42CostTbl, ISD, MTy))
4138 if (auto KindCost = Entry->Cost[CostKind])
4139 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4140
4141 if (ST->hasSSE41())
4142 if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
4143 if (auto KindCost = Entry->Cost[CostKind])
4144 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4145
4146 if (ST->hasSSE2())
4147 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
4148 if (auto KindCost = Entry->Cost[CostKind])
4149 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4150
4151 if (ST->hasSSE1())
4152 if (const auto *Entry = CostTableLookup(SSE1CostTbl, ISD, MTy))
4153 if (auto KindCost = Entry->Cost[CostKind])
4154 return LT.first * (ExtraCost + *KindCost) + MaskExpCost;
4155
4156 // Assume a 3cy latency for fp select ops.
4157 if (CostKind == TTI::TCK_Latency && Opcode == Instruction::Select)
4158 if (ValTy->getScalarType()->isFloatingPointTy())
4159 return 3;
4160
4161 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
4162 Op1Info, Op2Info, I) +
4163 MaskExpCost;
4164}
4165
4167
4168/// Returns true if the square root is divided into with the reciprocal
4169/// allowed and the target replaces the pair with the rsqrt* based estimate.
4171 const X86TargetLowering &TLI,
4172 const DataLayout &DL) {
4173 using namespace PatternMatch;
4174 const IntrinsicInst *II = ICA.getInst();
4175 if (!II || !II->hasOneUse() ||
4176 !match(II->user_back(), m_FDiv(m_Value(), m_Specific(II))) ||
4177 !cast<FPMathOperator>(II->user_back())->hasAllowReciprocal())
4178 return false;
4179 EVT VT = TLI.getValueType(DL, ICA.getReturnType());
4180 return TLI.hasSqrtEstimate(VT, /*Reciprocal=*/true) &&
4181 TLI.getRecipEstimateSqrtEnabled(VT, *II->getFunction()) !=
4183}
4184
4188 // Costs should match the codegen from:
4189 // BITREVERSE: llvm\test\CodeGen\X86\vector-bitreverse.ll
4190 // BSWAP: llvm\test\CodeGen\X86\bswap-vector.ll
4191 // CTLZ: llvm\test\CodeGen\X86\vector-lzcnt-*.ll
4192 // CTPOP: llvm\test\CodeGen\X86\vector-popcnt-*.ll
4193 // CTTZ: llvm\test\CodeGen\X86\vector-tzcnt-*.ll
4194
4195 // TODO: Overflow intrinsics (*ADDO, *SUBO, *MULO) with vector types are not
4196 // specialized in these tables yet.
4197 static const CostKindTblEntry AVX512VBMI2CostTbl[] = {
4198 { ISD::FSHL, MVT::v8i64, { 1, 1, 1, 1 } },
4199 { ISD::FSHL, MVT::v4i64, { 1, 1, 1, 1 } },
4200 { ISD::FSHL, MVT::v2i64, { 1, 1, 1, 1 } },
4201 { ISD::FSHL, MVT::v16i32, { 1, 1, 1, 1 } },
4202 { ISD::FSHL, MVT::v8i32, { 1, 1, 1, 1 } },
4203 { ISD::FSHL, MVT::v4i32, { 1, 1, 1, 1 } },
4204 { ISD::FSHL, MVT::v32i16, { 1, 1, 1, 1 } },
4205 { ISD::FSHL, MVT::v16i16, { 1, 1, 1, 1 } },
4206 { ISD::FSHL, MVT::v8i16, { 1, 1, 1, 1 } },
4207 { ISD::ROTL, MVT::v32i16, { 1, 1, 1, 1 } },
4208 { ISD::ROTL, MVT::v16i16, { 1, 1, 1, 1 } },
4209 { ISD::ROTL, MVT::v8i16, { 1, 1, 1, 1 } },
4210 { ISD::ROTR, MVT::v32i16, { 1, 1, 1, 1 } },
4211 { ISD::ROTR, MVT::v16i16, { 1, 1, 1, 1 } },
4212 { ISD::ROTR, MVT::v8i16, { 1, 1, 1, 1 } },
4213 { X86ISD::VROTLI, MVT::v32i16, { 1, 1, 1, 1 } },
4214 { X86ISD::VROTLI, MVT::v16i16, { 1, 1, 1, 1 } },
4215 { X86ISD::VROTLI, MVT::v8i16, { 1, 1, 1, 1 } },
4216 };
4217 static const CostKindTblEntry AVX512BITALGCostTbl[] = {
4218 { ISD::CTPOP, MVT::v32i16, { 1, 1, 1, 1 } },
4219 { ISD::CTPOP, MVT::v64i8, { 1, 1, 1, 1 } },
4220 { ISD::CTPOP, MVT::v16i16, { 1, 1, 1, 1 } },
4221 { ISD::CTPOP, MVT::v32i8, { 1, 1, 1, 1 } },
4222 { ISD::CTPOP, MVT::v8i16, { 1, 1, 1, 1 } },
4223 { ISD::CTPOP, MVT::v16i8, { 1, 1, 1, 1 } },
4224 };
4225 static const CostKindTblEntry AVX512VPOPCNTDQCostTbl[] = {
4226 { ISD::CTPOP, MVT::v8i64, { 1, 1, 1, 1 } },
4227 { ISD::CTPOP, MVT::v16i32, { 1, 1, 1, 1 } },
4228 { ISD::CTPOP, MVT::v4i64, { 1, 1, 1, 1 } },
4229 { ISD::CTPOP, MVT::v8i32, { 1, 1, 1, 1 } },
4230 { ISD::CTPOP, MVT::v2i64, { 1, 1, 1, 1 } },
4231 { ISD::CTPOP, MVT::v4i32, { 1, 1, 1, 1 } },
4232 };
4233 static const CostKindTblEntry AVX512CDCostTbl[] = {
4234 { ISD::CTLZ, MVT::v8i64, { 1, 5, 1, 1 } },
4235 { ISD::CTLZ, MVT::v16i32, { 1, 5, 1, 1 } },
4236 { ISD::CTLZ, MVT::v32i16, { 18, 27, 23, 27 } },
4237 { ISD::CTLZ, MVT::v64i8, { 3, 16, 9, 11 } },
4238 { ISD::CTLZ, MVT::v4i64, { 1, 5, 1, 1 } },
4239 { ISD::CTLZ, MVT::v8i32, { 1, 5, 1, 1 } },
4240 { ISD::CTLZ, MVT::v16i16, { 8, 19, 11, 13 } },
4241 { ISD::CTLZ, MVT::v32i8, { 2, 11, 9, 10 } },
4242 { ISD::CTLZ, MVT::v2i64, { 1, 5, 1, 1 } },
4243 { ISD::CTLZ, MVT::v4i32, { 1, 5, 1, 1 } },
4244 { ISD::CTLZ, MVT::v8i16, { 3, 15, 4, 6 } },
4245 { ISD::CTLZ, MVT::v16i8, { 2, 10, 9, 10 } },
4246
4247 { ISD::CTTZ, MVT::v8i64, { 2, 8, 6, 7 } },
4248 { ISD::CTTZ, MVT::v16i32, { 2, 8, 6, 7 } },
4249 { ISD::CTTZ, MVT::v4i64, { 1, 8, 6, 6 } },
4250 { ISD::CTTZ, MVT::v8i32, { 1, 8, 6, 6 } },
4251 { ISD::CTTZ, MVT::v2i64, { 1, 8, 6, 6 } },
4252 { ISD::CTTZ, MVT::v4i32, { 1, 8, 6, 6 } },
4253 };
4254 static const CostKindTblEntry AVX512BWCostTbl[] = {
4255 { ISD::ABS, MVT::v32i16, { 1, 1, 1, 1 } },
4256 { ISD::ABS, MVT::v64i8, { 1, 1, 1, 1 } },
4257 { ISD::BITREVERSE, MVT::v2i64, { 3, 10, 10, 11 } },
4258 { ISD::BITREVERSE, MVT::v4i64, { 3, 11, 10, 11 } },
4259 { ISD::BITREVERSE, MVT::v8i64, { 3, 12, 10, 14 } },
4260 { ISD::BITREVERSE, MVT::v4i32, { 3, 10, 10, 11 } },
4261 { ISD::BITREVERSE, MVT::v8i32, { 3, 11, 10, 11 } },
4262 { ISD::BITREVERSE, MVT::v16i32, { 3, 12, 10, 14 } },
4263 { ISD::BITREVERSE, MVT::v8i16, { 3, 10, 10, 11 } },
4264 { ISD::BITREVERSE, MVT::v16i16, { 3, 11, 10, 11 } },
4265 { ISD::BITREVERSE, MVT::v32i16, { 3, 12, 10, 14 } },
4266 { ISD::BITREVERSE, MVT::v16i8, { 2, 5, 9, 9 } },
4267 { ISD::BITREVERSE, MVT::v32i8, { 2, 5, 9, 9 } },
4268 { ISD::BITREVERSE, MVT::v64i8, { 2, 5, 9, 12 } },
4269 { ISD::BSWAP, MVT::v2i64, { 1, 1, 1, 2 } },
4270 { ISD::BSWAP, MVT::v4i64, { 1, 1, 1, 2 } },
4271 { ISD::BSWAP, MVT::v8i64, { 1, 1, 1, 2 } },
4272 { ISD::BSWAP, MVT::v4i32, { 1, 1, 1, 2 } },
4273 { ISD::BSWAP, MVT::v8i32, { 1, 1, 1, 2 } },
4274 { ISD::BSWAP, MVT::v16i32, { 1, 1, 1, 2 } },
4275 { ISD::BSWAP, MVT::v8i16, { 1, 1, 1, 2 } },
4276 { ISD::BSWAP, MVT::v16i16, { 1, 1, 1, 2 } },
4277 { ISD::BSWAP, MVT::v32i16, { 1, 1, 1, 2 } },
4278 { ISD::CTLZ, MVT::v8i64, { 8, 22, 23, 23 } },
4279 { ISD::CTLZ, MVT::v16i32, { 8, 23, 25, 25 } },
4280 { ISD::CTLZ, MVT::v32i16, { 4, 15, 15, 16 } },
4281 { ISD::CTLZ, MVT::v64i8, { 3, 12, 10, 9 } },
4282 { ISD::CTPOP, MVT::v2i64, { 3, 7, 10, 10 } },
4283 { ISD::CTPOP, MVT::v4i64, { 3, 7, 10, 10 } },
4284 { ISD::CTPOP, MVT::v8i64, { 3, 8, 10, 12 } },
4285 { ISD::CTPOP, MVT::v4i32, { 7, 11, 14, 14 } },
4286 { ISD::CTPOP, MVT::v8i32, { 7, 11, 14, 14 } },
4287 { ISD::CTPOP, MVT::v16i32, { 7, 12, 14, 16 } },
4288 { ISD::CTPOP, MVT::v8i16, { 2, 7, 11, 11 } },
4289 { ISD::CTPOP, MVT::v16i16, { 2, 7, 11, 11 } },
4290 { ISD::CTPOP, MVT::v32i16, { 3, 7, 11, 13 } },
4291 { ISD::CTPOP, MVT::v16i8, { 2, 4, 8, 8 } },
4292 { ISD::CTPOP, MVT::v32i8, { 2, 4, 8, 8 } },
4293 { ISD::CTPOP, MVT::v64i8, { 2, 5, 8, 10 } },
4294 { ISD::CTTZ, MVT::v8i16, { 3, 9, 14, 14 } },
4295 { ISD::CTTZ, MVT::v16i16, { 3, 9, 14, 14 } },
4296 { ISD::CTTZ, MVT::v32i16, { 3, 10, 14, 16 } },
4297 { ISD::CTTZ, MVT::v16i8, { 2, 6, 11, 11 } },
4298 { ISD::CTTZ, MVT::v32i8, { 2, 6, 11, 11 } },
4299 { ISD::CTTZ, MVT::v64i8, { 3, 7, 11, 13 } },
4300 { ISD::MULHS, MVT::v32i16, { 1, 5, 1, 1 } },
4301 { ISD::MULHU, MVT::v32i16, { 1, 5, 1, 1 } },
4302 { ISD::ROTL, MVT::v32i16, { 2, 8, 6, 8 } },
4303 { ISD::ROTL, MVT::v16i16, { 2, 8, 6, 7 } },
4304 { ISD::ROTL, MVT::v8i16, { 2, 7, 6, 7 } },
4305 { ISD::ROTL, MVT::v64i8, { 5, 6, 11, 12 } },
4306 { ISD::ROTL, MVT::v32i8, { 5, 15, 7, 10 } },
4307 { ISD::ROTL, MVT::v16i8, { 5, 15, 7, 10 } },
4308 { ISD::ROTR, MVT::v32i16, { 2, 8, 6, 8 } },
4309 { ISD::ROTR, MVT::v16i16, { 2, 8, 6, 7 } },
4310 { ISD::ROTR, MVT::v8i16, { 2, 7, 6, 7 } },
4311 { ISD::ROTR, MVT::v64i8, { 5, 6, 12, 14 } },
4312 { ISD::ROTR, MVT::v32i8, { 5, 14, 6, 9 } },
4313 { ISD::ROTR, MVT::v16i8, { 5, 14, 6, 9 } },
4314 { X86ISD::VROTLI, MVT::v32i16, { 2, 5, 3, 3 } },
4315 { X86ISD::VROTLI, MVT::v16i16, { 1, 5, 3, 3 } },
4316 { X86ISD::VROTLI, MVT::v8i16, { 1, 5, 3, 3 } },
4317 { X86ISD::VROTLI, MVT::v64i8, { 2, 9, 3, 4 } },
4318 { X86ISD::VROTLI, MVT::v32i8, { 1, 9, 3, 4 } },
4319 { X86ISD::VROTLI, MVT::v16i8, { 1, 8, 3, 4 } },
4320 { ISD::SADDSAT, MVT::v32i16, { 1, 1, 1, 1 } },
4321 { ISD::SADDSAT, MVT::v64i8, { 1, 1, 1, 1 } },
4322 { ISD::SMAX, MVT::v32i16, { 1, 1, 1, 1 } },
4323 { ISD::SMAX, MVT::v64i8, { 1, 1, 1, 1 } },
4324 { ISD::SMIN, MVT::v32i16, { 1, 1, 1, 1 } },
4325 { ISD::SMIN, MVT::v64i8, { 1, 1, 1, 1 } },
4326 { ISD::SMULO, MVT::v32i16, { 3, 6, 4, 4 } },
4327 { ISD::SMULO, MVT::v64i8, { 8, 21, 17, 18 } },
4328 { ISD::UMULO, MVT::v32i16, { 2, 5, 3, 3 } },
4329 { ISD::UMULO, MVT::v64i8, { 8, 15, 15, 16 } },
4330 { ISD::SSUBSAT, MVT::v32i16, { 1, 1, 1, 1 } },
4331 { ISD::SSUBSAT, MVT::v64i8, { 1, 1, 1, 1 } },
4332 { ISD::UADDSAT, MVT::v32i16, { 1, 1, 1, 1 } },
4333 { ISD::UADDSAT, MVT::v64i8, { 1, 1, 1, 1 } },
4334 { ISD::UMAX, MVT::v32i16, { 1, 1, 1, 1 } },
4335 { ISD::UMAX, MVT::v64i8, { 1, 1, 1, 1 } },
4336 { ISD::UMIN, MVT::v32i16, { 1, 1, 1, 1 } },
4337 { ISD::UMIN, MVT::v64i8, { 1, 1, 1, 1 } },
4338 { ISD::USUBSAT, MVT::v32i16, { 1, 1, 1, 1 } },
4339 { ISD::USUBSAT, MVT::v64i8, { 1, 1, 1, 1 } },
4340 };
4341 static const CostKindTblEntry AVX512CostTbl[] = {
4342 { ISD::ABS, MVT::v8i64, { 1, 1, 1, 1 } },
4343 { ISD::ABS, MVT::v4i64, { 1, 1, 1, 1 } },
4344 { ISD::ABS, MVT::v2i64, { 1, 1, 1, 1 } },
4345 { ISD::ABS, MVT::v16i32, { 1, 1, 1, 1 } },
4346 { ISD::ABS, MVT::v8i32, { 1, 1, 1, 1 } },
4347 { ISD::ABS, MVT::v32i16, { 2, 7, 4, 4 } },
4348 { ISD::ABS, MVT::v16i16, { 1, 1, 1, 1 } },
4349 { ISD::ABS, MVT::v64i8, { 2, 7, 4, 4 } },
4350 { ISD::ABS, MVT::v32i8, { 1, 1, 1, 1 } },
4351 { ISD::BITREVERSE, MVT::v8i64, { 9, 13, 20, 20 } },
4352 { ISD::BITREVERSE, MVT::v16i32, { 9, 13, 20, 20 } },
4353 { ISD::BITREVERSE, MVT::v32i16, { 9, 13, 20, 20 } },
4354 { ISD::BITREVERSE, MVT::v64i8, { 6, 11, 17, 17 } },
4355 { ISD::BSWAP, MVT::v8i64, { 4, 7, 5, 5 } },
4356 { ISD::BSWAP, MVT::v16i32, { 4, 7, 5, 5 } },
4357 { ISD::BSWAP, MVT::v32i16, { 4, 7, 5, 5 } },
4358 { ISD::CTLZ, MVT::v8i64, { 10, 28, 32, 32 } },
4359 { ISD::CTLZ, MVT::v16i32, { 12, 30, 38, 38 } },
4360 { ISD::CTLZ, MVT::v32i16, { 8, 15, 29, 29 } },
4361 { ISD::CTLZ, MVT::v64i8, { 6, 11, 19, 19 } },
4362 { ISD::CTPOP, MVT::v8i64, { 16, 16, 19, 19 } },
4363 { ISD::CTPOP, MVT::v16i32, { 24, 19, 27, 27 } },
4364 { ISD::CTPOP, MVT::v32i16, { 18, 15, 22, 22 } },
4365 { ISD::CTPOP, MVT::v64i8, { 12, 11, 16, 16 } },
4366 { ISD::CTTZ, MVT::v8i64, { 2, 8, 6, 7 } },
4367 { ISD::CTTZ, MVT::v16i32, { 2, 8, 6, 7 } },
4368 { ISD::CTTZ, MVT::v32i16, { 7, 17, 27, 27 } },
4369 { ISD::CTTZ, MVT::v64i8, { 6, 13, 21, 21 } },
4370 { ISD::MULHS, MVT::v16i32, { 3, 10, 6, 7 } },
4371 { ISD::MULHS, MVT::v8i32, { 3, 9, 6, 6 } },
4372 { ISD::MULHS, MVT::v32i16, { 3, 7, 5, 5 } },
4373 { ISD::MULHS, MVT::v16i16, { 1, 5, 1, 1 } },
4374 { ISD::MULHU, MVT::v16i32, { 3, 10, 6, 7 } },
4375 { ISD::MULHU, MVT::v8i32, { 3, 9, 6, 6 } },
4376 { ISD::MULHU, MVT::v32i16, { 3, 7, 5, 5 } },
4377 { ISD::MULHU, MVT::v16i16, { 1, 5, 1, 1 } },
4378 { ISD::ROTL, MVT::v8i64, { 1, 1, 1, 1 } },
4379 { ISD::ROTL, MVT::v4i64, { 1, 1, 1, 1 } },
4380 { ISD::ROTL, MVT::v2i64, { 1, 1, 1, 1 } },
4381 { ISD::ROTL, MVT::v16i32, { 1, 1, 1, 1 } },
4382 { ISD::ROTL, MVT::v8i32, { 1, 1, 1, 1 } },
4383 { ISD::ROTL, MVT::v4i32, { 1, 1, 1, 1 } },
4384 { ISD::ROTR, MVT::v8i64, { 1, 1, 1, 1 } },
4385 { ISD::ROTR, MVT::v4i64, { 1, 1, 1, 1 } },
4386 { ISD::ROTR, MVT::v2i64, { 1, 1, 1, 1 } },
4387 { ISD::ROTR, MVT::v16i32, { 1, 1, 1, 1 } },
4388 { ISD::ROTR, MVT::v8i32, { 1, 1, 1, 1 } },
4389 { ISD::ROTR, MVT::v4i32, { 1, 1, 1, 1 } },
4390 { X86ISD::VROTLI, MVT::v8i64, { 1, 1, 1, 1 } },
4391 { X86ISD::VROTLI, MVT::v4i64, { 1, 1, 1, 1 } },
4392 { X86ISD::VROTLI, MVT::v2i64, { 1, 1, 1, 1 } },
4393 { X86ISD::VROTLI, MVT::v16i32, { 1, 1, 1, 1 } },
4394 { X86ISD::VROTLI, MVT::v8i32, { 1, 1, 1, 1 } },
4395 { X86ISD::VROTLI, MVT::v4i32, { 1, 1, 1, 1 } },
4396 { ISD::SADDSAT, MVT::v2i64, { 3, 3, 8, 9 } },
4397 { ISD::SADDSAT, MVT::v4i64, { 2, 2, 6, 7 } },
4398 { ISD::SADDSAT, MVT::v8i64, { 3, 3, 6, 7 } },
4399 { ISD::SADDSAT, MVT::v4i32, { 2, 2, 6, 7 } },
4400 { ISD::SADDSAT, MVT::v8i32, { 2, 2, 6, 7 } },
4401 { ISD::SADDSAT, MVT::v16i32, { 3, 3, 6, 7 } },
4402 { ISD::SADDSAT, MVT::v32i16, { 2, 2, 2, 2 } },
4403 { ISD::SADDSAT, MVT::v64i8, { 2, 2, 2, 2 } },
4404 { ISD::SMAX, MVT::v8i64, { 1, 3, 1, 1 } },
4405 { ISD::SMAX, MVT::v16i32, { 1, 1, 1, 1 } },
4406 { ISD::SMAX, MVT::v32i16, { 3, 7, 5, 5 } },
4407 { ISD::SMAX, MVT::v64i8, { 3, 7, 5, 5 } },
4408 { ISD::SMAX, MVT::v4i64, { 1, 3, 1, 1 } },
4409 { ISD::SMAX, MVT::v2i64, { 1, 3, 1, 1 } },
4410 { ISD::SMIN, MVT::v8i64, { 1, 3, 1, 1 } },
4411 { ISD::SMIN, MVT::v16i32, { 1, 1, 1, 1 } },
4412 { ISD::SMIN, MVT::v32i16, { 3, 7, 5, 5 } },
4413 { ISD::SMIN, MVT::v64i8, { 3, 7, 5, 5 } },
4414 { ISD::SMIN, MVT::v4i64, { 1, 3, 1, 1 } },
4415 { ISD::SMIN, MVT::v2i64, { 1, 3, 1, 1 } },
4416 { ISD::SMULO, MVT::v8i64, { 44, 44, 81, 93 } },
4417 { ISD::SMULO, MVT::v16i32, { 5, 12, 9, 11 } },
4418 { ISD::SMULO, MVT::v32i16, { 6, 12, 17, 17 } },
4419 { ISD::SMULO, MVT::v64i8, { 22, 28, 42, 42 } },
4420 { ISD::SSUBSAT, MVT::v2i64, { 2, 13, 9, 10 } },
4421 { ISD::SSUBSAT, MVT::v4i64, { 2, 15, 7, 8 } },
4422 { ISD::SSUBSAT, MVT::v8i64, { 2, 14, 7, 8 } },
4423 { ISD::SSUBSAT, MVT::v4i32, { 2, 14, 7, 8 } },
4424 { ISD::SSUBSAT, MVT::v8i32, { 2, 15, 7, 8 } },
4425 { ISD::SSUBSAT, MVT::v16i32, { 2, 14, 7, 8 } },
4426 { ISD::SSUBSAT, MVT::v32i16, { 2, 2, 2, 2 } },
4427 { ISD::SSUBSAT, MVT::v64i8, { 2, 2, 2, 2 } },
4428 { ISD::UMAX, MVT::v8i64, { 1, 3, 1, 1 } },
4429 { ISD::UMAX, MVT::v16i32, { 1, 1, 1, 1 } },
4430 { ISD::UMAX, MVT::v32i16, { 3, 7, 5, 5 } },
4431 { ISD::UMAX, MVT::v64i8, { 3, 7, 5, 5 } },
4432 { ISD::UMAX, MVT::v4i64, { 1, 3, 1, 1 } },
4433 { ISD::UMAX, MVT::v2i64, { 1, 3, 1, 1 } },
4434 { ISD::UMIN, MVT::v8i64, { 1, 3, 1, 1 } },
4435 { ISD::UMIN, MVT::v16i32, { 1, 1, 1, 1 } },
4436 { ISD::UMIN, MVT::v32i16, { 3, 7, 5, 5 } },
4437 { ISD::UMIN, MVT::v64i8, { 3, 7, 5, 5 } },
4438 { ISD::UMIN, MVT::v4i64, { 1, 3, 1, 1 } },
4439 { ISD::UMIN, MVT::v2i64, { 1, 3, 1, 1 } },
4440 { ISD::UMULO, MVT::v8i64, { 52, 52, 95, 104} },
4441 { ISD::UMULO, MVT::v16i32, { 5, 12, 8, 10 } },
4442 { ISD::UMULO, MVT::v32i16, { 5, 13, 16, 16 } },
4443 { ISD::UMULO, MVT::v64i8, { 18, 24, 30, 30 } },
4444 { ISD::UADDSAT, MVT::v2i64, { 1, 4, 4, 4 } },
4445 { ISD::UADDSAT, MVT::v4i64, { 1, 4, 4, 4 } },
4446 { ISD::UADDSAT, MVT::v8i64, { 1, 4, 4, 4 } },
4447 { ISD::UADDSAT, MVT::v4i32, { 1, 2, 4, 4 } },
4448 { ISD::UADDSAT, MVT::v8i32, { 1, 2, 4, 4 } },
4449 { ISD::UADDSAT, MVT::v16i32, { 2, 2, 4, 4 } },
4450 { ISD::UADDSAT, MVT::v32i16, { 2, 2, 2, 2 } },
4451 { ISD::UADDSAT, MVT::v64i8, { 2, 2, 2, 2 } },
4452 { ISD::USUBSAT, MVT::v2i64, { 1, 4, 2, 2 } },
4453 { ISD::USUBSAT, MVT::v4i64, { 1, 4, 2, 2 } },
4454 { ISD::USUBSAT, MVT::v8i64, { 1, 4, 2, 2 } },
4455 { ISD::USUBSAT, MVT::v8i32, { 1, 2, 2, 2 } },
4456 { ISD::USUBSAT, MVT::v16i32, { 1, 2, 2, 2 } },
4457 { ISD::USUBSAT, MVT::v32i16, { 2, 2, 2, 2 } },
4458 { ISD::USUBSAT, MVT::v64i8, { 2, 2, 2, 2 } },
4459 { ISD::FMAXNUM, MVT::f32, { 2, 2, 3, 3 } },
4460 { ISD::FMAXNUM, MVT::v4f32, { 1, 1, 3, 3 } },
4461 { ISD::FMAXNUM, MVT::v8f32, { 2, 2, 3, 3 } },
4462 { ISD::FMAXNUM, MVT::v16f32, { 4, 4, 3, 3 } },
4463 { ISD::FMAXNUM, MVT::f64, { 2, 2, 3, 3 } },
4464 { ISD::FMAXNUM, MVT::v2f64, { 1, 1, 3, 3 } },
4465 { ISD::FMAXNUM, MVT::v4f64, { 2, 2, 3, 3 } },
4466 { ISD::FMAXNUM, MVT::v8f64, { 3, 3, 3, 3 } },
4467 { ISD::FSQRT, MVT::f32, { 3, 12, 1, 1 } }, // Skylake from http://www.agner.org/
4468 { ISD::FSQRT, MVT::v4f32, { 3, 12, 1, 1 } }, // Skylake from http://www.agner.org/
4469 { ISD::FSQRT, MVT::v8f32, { 6, 12, 1, 1 } }, // Skylake from http://www.agner.org/
4470 { ISD::FSQRT, MVT::v16f32, { 12, 20, 1, 3 } }, // Skylake from http://www.agner.org/
4471 { ISD::FSQRT, MVT::f64, { 6, 18, 1, 1 } }, // Skylake from http://www.agner.org/
4472 { ISD::FSQRT, MVT::v2f64, { 6, 18, 1, 1 } }, // Skylake from http://www.agner.org/
4473 { ISD::FSQRT, MVT::v4f64, { 12, 18, 1, 1 } }, // Skylake from http://www.agner.org/
4474 { ISD::FSQRT, MVT::v8f64, { 24, 32, 1, 3 } }, // Skylake from http://www.agner.org/
4475 };
4476 static const CostKindTblEntry XOPCostTbl[] = {
4477 { ISD::BITREVERSE, MVT::v4i64, { 3, 6, 5, 6 } },
4478 { ISD::BITREVERSE, MVT::v8i32, { 3, 6, 5, 6 } },
4479 { ISD::BITREVERSE, MVT::v16i16, { 3, 6, 5, 6 } },
4480 { ISD::BITREVERSE, MVT::v32i8, { 3, 6, 5, 6 } },
4481 { ISD::BITREVERSE, MVT::v2i64, { 2, 7, 1, 1 } },
4482 { ISD::BITREVERSE, MVT::v4i32, { 2, 7, 1, 1 } },
4483 { ISD::BITREVERSE, MVT::v8i16, { 2, 7, 1, 1 } },
4484 { ISD::BITREVERSE, MVT::v16i8, { 2, 7, 1, 1 } },
4485 { ISD::BITREVERSE, MVT::i64, { 2, 2, 3, 4 } },
4486 { ISD::BITREVERSE, MVT::i32, { 2, 2, 3, 4 } },
4487 { ISD::BITREVERSE, MVT::i16, { 2, 2, 3, 4 } },
4488 { ISD::BITREVERSE, MVT::i8, { 2, 2, 3, 4 } },
4489 // XOP: ROTL = VPROT(X,Y), ROTR = VPROT(X,SUB(0,Y))
4490 { ISD::ROTL, MVT::v4i64, { 4, 7, 5, 6 } },
4491 { ISD::ROTL, MVT::v8i32, { 4, 7, 5, 6 } },
4492 { ISD::ROTL, MVT::v16i16, { 4, 7, 5, 6 } },
4493 { ISD::ROTL, MVT::v32i8, { 4, 7, 5, 6 } },
4494 { ISD::ROTL, MVT::v2i64, { 1, 3, 1, 1 } },
4495 { ISD::ROTL, MVT::v4i32, { 1, 3, 1, 1 } },
4496 { ISD::ROTL, MVT::v8i16, { 1, 3, 1, 1 } },
4497 { ISD::ROTL, MVT::v16i8, { 1, 3, 1, 1 } },
4498 { ISD::ROTR, MVT::v4i64, { 4, 7, 8, 9 } },
4499 { ISD::ROTR, MVT::v8i32, { 4, 7, 8, 9 } },
4500 { ISD::ROTR, MVT::v16i16, { 4, 7, 8, 9 } },
4501 { ISD::ROTR, MVT::v32i8, { 4, 7, 8, 9 } },
4502 { ISD::ROTR, MVT::v2i64, { 1, 3, 3, 3 } },
4503 { ISD::ROTR, MVT::v4i32, { 1, 3, 3, 3 } },
4504 { ISD::ROTR, MVT::v8i16, { 1, 3, 3, 3 } },
4505 { ISD::ROTR, MVT::v16i8, { 1, 3, 3, 3 } },
4506 { X86ISD::VROTLI, MVT::v4i64, { 4, 7, 5, 6 } },
4507 { X86ISD::VROTLI, MVT::v8i32, { 4, 7, 5, 6 } },
4508 { X86ISD::VROTLI, MVT::v16i16, { 4, 7, 5, 6 } },
4509 { X86ISD::VROTLI, MVT::v32i8, { 4, 7, 5, 6 } },
4510 { X86ISD::VROTLI, MVT::v2i64, { 1, 3, 1, 1 } },
4511 { X86ISD::VROTLI, MVT::v4i32, { 1, 3, 1, 1 } },
4512 { X86ISD::VROTLI, MVT::v8i16, { 1, 3, 1, 1 } },
4513 { X86ISD::VROTLI, MVT::v16i8, { 1, 3, 1, 1 } },
4514 };
4515 static const CostKindTblEntry AVX2CostTbl[] = {
4516 { ISD::ABS, MVT::v2i64, { 2, 4, 3, 5 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4517 { ISD::ABS, MVT::v4i64, { 2, 4, 3, 5 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4518 { ISD::ABS, MVT::v4i32, { 1, 1, 1, 1 } },
4519 { ISD::ABS, MVT::v8i32, { 1, 1, 1, 2 } },
4520 { ISD::ABS, MVT::v8i16, { 1, 1, 1, 1 } },
4521 { ISD::ABS, MVT::v16i16, { 1, 1, 1, 2 } },
4522 { ISD::ABS, MVT::v16i8, { 1, 1, 1, 1 } },
4523 { ISD::ABS, MVT::v32i8, { 1, 1, 1, 2 } },
4524 { ISD::BITREVERSE, MVT::v2i64, { 3, 11, 10, 11 } },
4525 { ISD::BITREVERSE, MVT::v4i64, { 5, 11, 10, 17 } },
4526 { ISD::BITREVERSE, MVT::v4i32, { 3, 11, 10, 11 } },
4527 { ISD::BITREVERSE, MVT::v8i32, { 5, 11, 10, 17 } },
4528 { ISD::BITREVERSE, MVT::v8i16, { 3, 11, 10, 11 } },
4529 { ISD::BITREVERSE, MVT::v16i16, { 5, 11, 10, 17 } },
4530 { ISD::BITREVERSE, MVT::v16i8, { 3, 6, 9, 9 } },
4531 { ISD::BITREVERSE, MVT::v32i8, { 4, 5, 9, 15 } },
4532 { ISD::BSWAP, MVT::v2i64, { 1, 2, 1, 2 } },
4533 { ISD::BSWAP, MVT::v4i64, { 1, 3, 1, 2 } },
4534 { ISD::BSWAP, MVT::v4i32, { 1, 2, 1, 2 } },
4535 { ISD::BSWAP, MVT::v8i32, { 1, 3, 1, 2 } },
4536 { ISD::BSWAP, MVT::v8i16, { 1, 2, 1, 2 } },
4537 { ISD::BSWAP, MVT::v16i16, { 1, 3, 1, 2 } },
4538 { ISD::CTLZ, MVT::v2i64, { 7, 18, 24, 25 } },
4539 { ISD::CTLZ, MVT::v4i64, { 14, 18, 24, 44 } },
4540 { ISD::CTLZ, MVT::v4i32, { 5, 16, 19, 20 } },
4541 { ISD::CTLZ, MVT::v8i32, { 10, 16, 19, 34 } },
4542 { ISD::CTLZ, MVT::v8i16, { 4, 13, 14, 15 } },
4543 { ISD::CTLZ, MVT::v16i16, { 6, 14, 14, 24 } },
4544 { ISD::CTLZ, MVT::v16i8, { 3, 12, 9, 10 } },
4545 { ISD::CTLZ, MVT::v32i8, { 4, 12, 9, 14 } },
4546 { ISD::CTPOP, MVT::v2i64, { 3, 9, 10, 10 } },
4547 { ISD::CTPOP, MVT::v4i64, { 4, 9, 10, 14 } },
4548 { ISD::CTPOP, MVT::v4i32, { 7, 12, 14, 14 } },
4549 { ISD::CTPOP, MVT::v8i32, { 7, 12, 14, 18 } },
4550 { ISD::CTPOP, MVT::v8i16, { 3, 7, 11, 11 } },
4551 { ISD::CTPOP, MVT::v16i16, { 6, 8, 11, 18 } },
4552 { ISD::CTPOP, MVT::v16i8, { 2, 5, 8, 8 } },
4553 { ISD::CTPOP, MVT::v32i8, { 3, 5, 8, 12 } },
4554 { ISD::CTTZ, MVT::v2i64, { 4, 11, 13, 13 } },
4555 { ISD::CTTZ, MVT::v4i64, { 5, 11, 13, 20 } },
4556 { ISD::CTTZ, MVT::v4i32, { 7, 14, 17, 17 } },
4557 { ISD::CTTZ, MVT::v8i32, { 7, 15, 17, 24 } },
4558 { ISD::CTTZ, MVT::v8i16, { 4, 9, 14, 14 } },
4559 { ISD::CTTZ, MVT::v16i16, { 6, 9, 14, 24 } },
4560 { ISD::CTTZ, MVT::v16i8, { 3, 7, 11, 11 } },
4561 { ISD::CTTZ, MVT::v32i8, { 5, 7, 11, 18 } },
4562 { ISD::MULHS, MVT::v8i32, { 4, 9, 6, 12 } },
4563 { ISD::MULHS, MVT::v16i16, { 2, 5, 1, 2 } },
4564 { ISD::MULHU, MVT::v8i32, { 4, 9, 6, 12 } },
4565 { ISD::MULHU, MVT::v16i16, { 2, 5, 1, 2 } },
4566 { ISD::SADDSAT, MVT::v2i64, { 4, 13, 8, 11 } },
4567 { ISD::SADDSAT, MVT::v4i64, { 3, 10, 8, 12 } },
4568 { ISD::SADDSAT, MVT::v4i32, { 2, 6, 7, 9 } },
4569 { ISD::SADDSAT, MVT::v8i32, { 4, 6, 7, 13 } },
4570 { ISD::SADDSAT, MVT::v16i16, { 1, 1, 1, 2 } },
4571 { ISD::SADDSAT, MVT::v32i8, { 1, 1, 1, 2 } },
4572 { ISD::SMAX, MVT::v2i64, { 2, 7, 2, 3 } },
4573 { ISD::SMAX, MVT::v4i64, { 2, 7, 2, 3 } },
4574 { ISD::SMAX, MVT::v8i32, { 1, 1, 1, 2 } },
4575 { ISD::SMAX, MVT::v16i16, { 1, 1, 1, 2 } },
4576 { ISD::SMAX, MVT::v32i8, { 1, 1, 1, 2 } },
4577 { ISD::SMIN, MVT::v2i64, { 2, 7, 2, 3 } },
4578 { ISD::SMIN, MVT::v4i64, { 2, 7, 2, 3 } },
4579 { ISD::SMIN, MVT::v8i32, { 1, 1, 1, 2 } },
4580 { ISD::SMIN, MVT::v16i16, { 1, 1, 1, 2 } },
4581 { ISD::SMIN, MVT::v32i8, { 1, 1, 1, 2 } },
4582 { ISD::SMULO, MVT::v4i64, { 20, 20, 33, 37 } },
4583 { ISD::SMULO, MVT::v2i64, { 8, 8, 13, 15 } },
4584 { ISD::SMULO, MVT::v8i32, { 8, 20, 13, 24 } },
4585 { ISD::SMULO, MVT::v4i32, { 5, 15, 11, 12 } },
4586 { ISD::SMULO, MVT::v16i16, { 4, 14, 8, 14 } },
4587 { ISD::SMULO, MVT::v8i16, { 3, 9, 6, 6 } },
4588 { ISD::SMULO, MVT::v32i8, { 9, 15, 18, 35 } },
4589 { ISD::SMULO, MVT::v16i8, { 6, 22, 14, 21 } },
4590 { ISD::SSUBSAT, MVT::v2i64, { 4, 13, 9, 13 } },
4591 { ISD::SSUBSAT, MVT::v4i64, { 4, 15, 9, 13 } },
4592 { ISD::SSUBSAT, MVT::v4i32, { 3, 14, 9, 11 } },
4593 { ISD::SSUBSAT, MVT::v8i32, { 4, 15, 9, 16 } },
4594 { ISD::SSUBSAT, MVT::v16i16, { 1, 1, 1, 2 } },
4595 { ISD::SSUBSAT, MVT::v32i8, { 1, 1, 1, 2 } },
4596 { ISD::UADDSAT, MVT::v2i64, { 2, 8, 6, 6 } },
4597 { ISD::UADDSAT, MVT::v4i64, { 3, 8, 6, 10 } },
4598 { ISD::UADDSAT, MVT::v8i32, { 2, 2, 4, 8 } },
4599 { ISD::UADDSAT, MVT::v16i16, { 1, 1, 1, 2 } },
4600 { ISD::UADDSAT, MVT::v32i8, { 1, 1, 1, 2 } },
4601 { ISD::UMAX, MVT::v2i64, { 2, 8, 5, 6 } },
4602 { ISD::UMAX, MVT::v4i64, { 2, 8, 5, 8 } },
4603 { ISD::UMAX, MVT::v8i32, { 1, 1, 1, 2 } },
4604 { ISD::UMAX, MVT::v16i16, { 1, 1, 1, 2 } },
4605 { ISD::UMAX, MVT::v32i8, { 1, 1, 1, 2 } },
4606 { ISD::UMIN, MVT::v2i64, { 2, 8, 5, 6 } },
4607 { ISD::UMIN, MVT::v4i64, { 2, 8, 5, 8 } },
4608 { ISD::UMIN, MVT::v8i32, { 1, 1, 1, 2 } },
4609 { ISD::UMIN, MVT::v16i16, { 1, 1, 1, 2 } },
4610 { ISD::UMIN, MVT::v32i8, { 1, 1, 1, 2 } },
4611 { ISD::UMULO, MVT::v4i64, { 24, 24, 39, 43 } },
4612 { ISD::UMULO, MVT::v2i64, { 10, 10, 15, 19 } },
4613 { ISD::UMULO, MVT::v8i32, { 8, 11, 13, 23 } },
4614 { ISD::UMULO, MVT::v4i32, { 5, 12, 11, 12 } },
4615 { ISD::UMULO, MVT::v16i16, { 4, 6, 8, 13 } },
4616 { ISD::UMULO, MVT::v8i16, { 2, 8, 6, 6 } },
4617 { ISD::UMULO, MVT::v32i8, { 9, 13, 17, 33 } },
4618 { ISD::UMULO, MVT::v16i8, { 6, 19, 13, 20 } },
4619 { ISD::USUBSAT, MVT::v2i64, { 2, 7, 6, 6 } },
4620 { ISD::USUBSAT, MVT::v4i64, { 3, 7, 6, 10 } },
4621 { ISD::USUBSAT, MVT::v8i32, { 2, 2, 2, 4 } },
4622 { ISD::USUBSAT, MVT::v16i16, { 1, 1, 1, 2 } },
4623 { ISD::USUBSAT, MVT::v32i8, { 1, 1, 1, 2 } },
4624 { ISD::FMAXNUM, MVT::f32, { 2, 7, 3, 5 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4625 { ISD::FMAXNUM, MVT::v4f32, { 2, 7, 3, 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4626 { ISD::FMAXNUM, MVT::v8f32, { 3, 7, 3, 6 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4627 { ISD::FMAXNUM, MVT::f64, { 2, 7, 3, 5 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4628 { ISD::FMAXNUM, MVT::v2f64, { 2, 7, 3, 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4629 { ISD::FMAXNUM, MVT::v4f64, { 3, 7, 3, 6 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4630 { ISD::FSQRT, MVT::f32, { 7, 15, 1, 1 } }, // vsqrtss
4631 { ISD::FSQRT, MVT::v4f32, { 7, 15, 1, 1 } }, // vsqrtps
4632 { ISD::FSQRT, MVT::v8f32, { 14, 21, 1, 3 } }, // vsqrtps
4633 { ISD::FSQRT, MVT::f64, { 14, 21, 1, 1 } }, // vsqrtsd
4634 { ISD::FSQRT, MVT::v2f64, { 14, 21, 1, 1 } }, // vsqrtpd
4635 { ISD::FSQRT, MVT::v4f64, { 28, 35, 1, 3 } }, // vsqrtpd
4636 };
4637 static const CostKindTblEntry AVX1CostTbl[] = {
4638 { ISD::ABS, MVT::v4i64, { 6, 8, 6, 12 } }, // VBLENDVPD(X,VPSUBQ(0,X),X)
4639 { ISD::ABS, MVT::v8i32, { 3, 6, 4, 5 } },
4640 { ISD::ABS, MVT::v16i16, { 3, 6, 4, 5 } },
4641 { ISD::ABS, MVT::v32i8, { 3, 6, 4, 5 } },
4642 { ISD::BITREVERSE, MVT::v4i64, { 17, 20, 20, 33 } }, // 2 x 128-bit Op + extract/insert
4643 { ISD::BITREVERSE, MVT::v2i64, { 8, 13, 10, 16 } },
4644 { ISD::BITREVERSE, MVT::v8i32, { 17, 20, 20, 33 } }, // 2 x 128-bit Op + extract/insert
4645 { ISD::BITREVERSE, MVT::v4i32, { 8, 13, 10, 16 } },
4646 { ISD::BITREVERSE, MVT::v16i16, { 17, 20, 20, 33 } }, // 2 x 128-bit Op + extract/insert
4647 { ISD::BITREVERSE, MVT::v8i16, { 8, 13, 10, 16 } },
4648 { ISD::BITREVERSE, MVT::v32i8, { 13, 15, 17, 26 } }, // 2 x 128-bit Op + extract/insert
4649 { ISD::BITREVERSE, MVT::v16i8, { 7, 7, 9, 13 } },
4650 { ISD::BSWAP, MVT::v4i64, { 5, 6, 5, 10 } },
4651 { ISD::BSWAP, MVT::v2i64, { 2, 2, 1, 3 } },
4652 { ISD::BSWAP, MVT::v8i32, { 5, 6, 5, 10 } },
4653 { ISD::BSWAP, MVT::v4i32, { 2, 2, 1, 3 } },
4654 { ISD::BSWAP, MVT::v16i16, { 5, 6, 5, 10 } },
4655 { ISD::BSWAP, MVT::v8i16, { 2, 2, 1, 3 } },
4656 { ISD::CTLZ, MVT::v4i64, { 29, 33, 49, 58 } }, // 2 x 128-bit Op + extract/insert
4657 { ISD::CTLZ, MVT::v2i64, { 14, 24, 24, 28 } },
4658 { ISD::CTLZ, MVT::v8i32, { 24, 28, 39, 48 } }, // 2 x 128-bit Op + extract/insert
4659 { ISD::CTLZ, MVT::v4i32, { 12, 20, 19, 23 } },
4660 { ISD::CTLZ, MVT::v16i16, { 19, 22, 29, 38 } }, // 2 x 128-bit Op + extract/insert
4661 { ISD::CTLZ, MVT::v8i16, { 9, 16, 14, 18 } },
4662 { ISD::CTLZ, MVT::v32i8, { 14, 15, 19, 28 } }, // 2 x 128-bit Op + extract/insert
4663 { ISD::CTLZ, MVT::v16i8, { 7, 12, 9, 13 } },
4664 { ISD::CTPOP, MVT::v4i64, { 14, 18, 19, 28 } }, // 2 x 128-bit Op + extract/insert
4665 { ISD::CTPOP, MVT::v2i64, { 7, 14, 10, 14 } },
4666 { ISD::CTPOP, MVT::v8i32, { 18, 24, 27, 36 } }, // 2 x 128-bit Op + extract/insert
4667 { ISD::CTPOP, MVT::v4i32, { 9, 20, 14, 18 } },
4668 { ISD::CTPOP, MVT::v16i16, { 16, 21, 22, 31 } }, // 2 x 128-bit Op + extract/insert
4669 { ISD::CTPOP, MVT::v8i16, { 8, 18, 11, 15 } },
4670 { ISD::CTPOP, MVT::v32i8, { 13, 15, 16, 25 } }, // 2 x 128-bit Op + extract/insert
4671 { ISD::CTPOP, MVT::v16i8, { 6, 12, 8, 12 } },
4672 { ISD::CTTZ, MVT::v4i64, { 17, 22, 24, 33 } }, // 2 x 128-bit Op + extract/insert
4673 { ISD::CTTZ, MVT::v2i64, { 9, 19, 13, 17 } },
4674 { ISD::CTTZ, MVT::v8i32, { 21, 27, 32, 41 } }, // 2 x 128-bit Op + extract/insert
4675 { ISD::CTTZ, MVT::v4i32, { 11, 24, 17, 21 } },
4676 { ISD::CTTZ, MVT::v16i16, { 18, 24, 27, 36 } }, // 2 x 128-bit Op + extract/insert
4677 { ISD::CTTZ, MVT::v8i16, { 9, 21, 14, 18 } },
4678 { ISD::CTTZ, MVT::v32i8, { 15, 18, 21, 30 } }, // 2 x 128-bit Op + extract/insert
4679 { ISD::CTTZ, MVT::v16i8, { 8, 16, 11, 15 } },
4680 { ISD::MULHS, MVT::v8i32, { 9, 11, 14, 18 } },
4681 { ISD::MULHS, MVT::v16i16, { 3, 7, 5, 6 } },
4682 { ISD::MULHU, MVT::v8i32, { 9, 11, 14, 18 } },
4683 { ISD::MULHU, MVT::v16i16, { 3, 7, 5, 6 } },
4684 { ISD::SADDSAT, MVT::v2i64, { 6, 13, 8, 11 } },
4685 { ISD::SADDSAT, MVT::v4i64, { 13, 20, 15, 25 } }, // 2 x 128-bit Op + extract/insert
4686 { ISD::SADDSAT, MVT::v8i32, { 12, 18, 14, 24 } }, // 2 x 128-bit Op + extract/insert
4687 { ISD::SADDSAT, MVT::v16i16, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4688 { ISD::SADDSAT, MVT::v32i8, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4689 { ISD::SMAX, MVT::v4i64, { 6, 9, 6, 12 } }, // 2 x 128-bit Op + extract/insert
4690 { ISD::SMAX, MVT::v2i64, { 3, 7, 2, 4 } },
4691 { ISD::SMAX, MVT::v8i32, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4692 { ISD::SMAX, MVT::v16i16, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4693 { ISD::SMAX, MVT::v32i8, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4694 { ISD::SMIN, MVT::v4i64, { 6, 9, 6, 12 } }, // 2 x 128-bit Op + extract/insert
4695 { ISD::SMIN, MVT::v2i64, { 3, 7, 2, 3 } },
4696 { ISD::SMIN, MVT::v8i32, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4697 { ISD::SMIN, MVT::v16i16, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4698 { ISD::SMIN, MVT::v32i8, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4699 { ISD::SMULO, MVT::v4i64, { 20, 20, 33, 37 } },
4700 { ISD::SMULO, MVT::v2i64, { 9, 9, 13, 17 } },
4701 { ISD::SMULO, MVT::v8i32, { 15, 20, 24, 29 } },
4702 { ISD::SMULO, MVT::v4i32, { 7, 15, 11, 13 } },
4703 { ISD::SMULO, MVT::v16i16, { 8, 14, 14, 15 } },
4704 { ISD::SMULO, MVT::v8i16, { 3, 9, 6, 6 } },
4705 { ISD::SMULO, MVT::v32i8, { 20, 20, 37, 39 } },
4706 { ISD::SMULO, MVT::v16i8, { 9, 22, 18, 21 } },
4707 { ISD::SSUBSAT, MVT::v2i64, { 7, 13, 9, 13 } },
4708 { ISD::SSUBSAT, MVT::v4i64, { 15, 21, 18, 29 } }, // 2 x 128-bit Op + extract/insert
4709 { ISD::SSUBSAT, MVT::v8i32, { 15, 19, 18, 29 } }, // 2 x 128-bit Op + extract/insert
4710 { ISD::SSUBSAT, MVT::v16i16, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4711 { ISD::SSUBSAT, MVT::v32i8, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4712 { ISD::UADDSAT, MVT::v2i64, { 3, 8, 6, 6 } },
4713 { ISD::UADDSAT, MVT::v4i64, { 8, 11, 14, 15 } }, // 2 x 128-bit Op + extract/insert
4714 { ISD::UADDSAT, MVT::v8i32, { 6, 6, 10, 11 } }, // 2 x 128-bit Op + extract/insert
4715 { ISD::UADDSAT, MVT::v16i16, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4716 { ISD::UADDSAT, MVT::v32i8, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4717 { ISD::UMAX, MVT::v4i64, { 9, 10, 11, 17 } }, // 2 x 128-bit Op + extract/insert
4718 { ISD::UMAX, MVT::v2i64, { 4, 8, 5, 7 } },
4719 { ISD::UMAX, MVT::v8i32, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4720 { ISD::UMAX, MVT::v16i16, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4721 { ISD::UMAX, MVT::v32i8, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4722 { ISD::UMIN, MVT::v4i64, { 9, 10, 11, 17 } }, // 2 x 128-bit Op + extract/insert
4723 { ISD::UMIN, MVT::v2i64, { 4, 8, 5, 7 } },
4724 { ISD::UMIN, MVT::v8i32, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4725 { ISD::UMIN, MVT::v16i16, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4726 { ISD::UMIN, MVT::v32i8, { 4, 6, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4727 { ISD::UMULO, MVT::v4i64, { 24, 26, 39, 45 } },
4728 { ISD::UMULO, MVT::v2i64, { 10, 12, 15, 20 } },
4729 { ISD::UMULO, MVT::v8i32, { 14, 15, 23, 28 } },
4730 { ISD::UMULO, MVT::v4i32, { 7, 12, 11, 13 } },
4731 { ISD::UMULO, MVT::v16i16, { 7, 11, 13, 14 } },
4732 { ISD::UMULO, MVT::v8i16, { 3, 8, 6, 6 } },
4733 { ISD::UMULO, MVT::v32i8, { 19, 19, 35, 37 } },
4734 { ISD::UMULO, MVT::v16i8, { 9, 19, 17, 20 } },
4735 { ISD::USUBSAT, MVT::v2i64, { 3, 7, 6, 6 } },
4736 { ISD::USUBSAT, MVT::v4i64, { 8, 10, 14, 15 } }, // 2 x 128-bit Op + extract/insert
4737 { ISD::USUBSAT, MVT::v8i32, { 4, 4, 7, 8 } }, // 2 x 128-bit Op + extract/insert
4738 { ISD::USUBSAT, MVT::v8i32, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4739 { ISD::USUBSAT, MVT::v16i16, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4740 { ISD::USUBSAT, MVT::v32i8, { 3, 3, 5, 6 } }, // 2 x 128-bit Op + extract/insert
4741 { ISD::FMAXNUM, MVT::f32, { 3, 6, 3, 5 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4742 { ISD::FMAXNUM, MVT::v4f32, { 3, 6, 3, 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4743 { ISD::FMAXNUM, MVT::v8f32, { 5, 7, 3, 10 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4744 { ISD::FMAXNUM, MVT::f64, { 3, 6, 3, 5 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4745 { ISD::FMAXNUM, MVT::v2f64, { 3, 6, 3, 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4746 { ISD::FMAXNUM, MVT::v4f64, { 5, 7, 3, 10 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4747 { ISD::FSQRT, MVT::f32, { 21, 21, 1, 1 } }, // vsqrtss
4748 { ISD::FSQRT, MVT::v4f32, { 21, 21, 1, 1 } }, // vsqrtps
4749 { ISD::FSQRT, MVT::v8f32, { 42, 42, 1, 3 } }, // vsqrtps
4750 { ISD::FSQRT, MVT::f64, { 27, 27, 1, 1 } }, // vsqrtsd
4751 { ISD::FSQRT, MVT::v2f64, { 27, 27, 1, 1 } }, // vsqrtpd
4752 { ISD::FSQRT, MVT::v4f64, { 54, 54, 1, 3 } }, // vsqrtpd
4753 };
4754 static const CostKindTblEntry GFNICostTbl[] = {
4755 { ISD::BITREVERSE, MVT::i8, { 3, 3, 3, 4 } }, // gf2p8affineqb
4756 { ISD::BITREVERSE, MVT::i16, { 3, 3, 4, 6 } }, // gf2p8affineqb
4757 { ISD::BITREVERSE, MVT::i32, { 3, 3, 4, 5 } }, // gf2p8affineqb
4758 { ISD::BITREVERSE, MVT::i64, { 3, 3, 4, 6 } }, // gf2p8affineqb
4759 { ISD::BITREVERSE, MVT::v16i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4760 { ISD::BITREVERSE, MVT::v32i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4761 { ISD::BITREVERSE, MVT::v64i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4762 { ISD::BITREVERSE, MVT::v8i16, { 1, 8, 2, 4 } }, // gf2p8affineqb
4763 { ISD::BITREVERSE, MVT::v16i16, { 1, 9, 2, 4 } }, // gf2p8affineqb
4764 { ISD::BITREVERSE, MVT::v32i16, { 1, 9, 2, 4 } }, // gf2p8affineqb
4765 { ISD::BITREVERSE, MVT::v4i32, { 1, 8, 2, 4 } }, // gf2p8affineqb
4766 { ISD::BITREVERSE, MVT::v8i32, { 1, 9, 2, 4 } }, // gf2p8affineqb
4767 { ISD::BITREVERSE, MVT::v16i32, { 1, 9, 2, 4 } }, // gf2p8affineqb
4768 { ISD::BITREVERSE, MVT::v2i64, { 1, 8, 2, 4 } }, // gf2p8affineqb
4769 { ISD::BITREVERSE, MVT::v4i64, { 1, 9, 2, 4 } }, // gf2p8affineqb
4770 { ISD::BITREVERSE, MVT::v8i64, { 1, 9, 2, 4 } }, // gf2p8affineqb
4771 { X86ISD::VROTLI, MVT::v16i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4772 { X86ISD::VROTLI, MVT::v32i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4773 { X86ISD::VROTLI, MVT::v64i8, { 1, 6, 1, 2 } }, // gf2p8affineqb
4774 };
4775 static const CostKindTblEntry GLMCostTbl[] = {
4776 { ISD::FSQRT, MVT::f32, { 19, 20, 1, 1 } }, // sqrtss
4777 { ISD::FSQRT, MVT::v4f32, { 37, 41, 1, 5 } }, // sqrtps
4778 { ISD::FSQRT, MVT::f64, { 34, 35, 1, 1 } }, // sqrtsd
4779 { ISD::FSQRT, MVT::v2f64, { 67, 71, 1, 5 } }, // sqrtpd
4780 };
4781 static const CostKindTblEntry SLMCostTbl[] = {
4782 { ISD::BSWAP, MVT::v2i64, { 5, 5, 1, 5 } },
4783 { ISD::BSWAP, MVT::v4i32, { 5, 5, 1, 5 } },
4784 { ISD::BSWAP, MVT::v8i16, { 5, 5, 1, 5 } },
4785 { ISD::FSQRT, MVT::f32, { 20, 20, 1, 1 } }, // sqrtss
4786 { ISD::FSQRT, MVT::v4f32, { 40, 41, 1, 5 } }, // sqrtps
4787 { ISD::FSQRT, MVT::f64, { 35, 35, 1, 1 } }, // sqrtsd
4788 { ISD::FSQRT, MVT::v2f64, { 70, 71, 1, 5 } }, // sqrtpd
4789 };
4790 static const CostKindTblEntry SSE42CostTbl[] = {
4791 { ISD::FMAXNUM, MVT::f32, { 5, 5, 7, 7 } }, // MAXSS + CMPUNORDSS + BLENDVPS
4792 { ISD::FMAXNUM, MVT::v4f32, { 4, 4, 4, 5 } }, // MAXPS + CMPUNORDPS + BLENDVPS
4793 { ISD::FMAXNUM, MVT::f64, { 5, 5, 7, 7 } }, // MAXSD + CMPUNORDSD + BLENDVPD
4794 { ISD::FMAXNUM, MVT::v2f64, { 4, 4, 4, 5 } }, // MAXPD + CMPUNORDPD + BLENDVPD
4795 { ISD::FSQRT, MVT::f32, { 18, 18, 1, 1 } }, // Nehalem from http://www.agner.org/
4796 { ISD::FSQRT, MVT::v4f32, { 18, 18, 1, 1 } }, // Nehalem from http://www.agner.org/
4797 };
4798 static const CostKindTblEntry SSE41CostTbl[] = {
4799 { ISD::ABS, MVT::v2i64, { 3, 4, 3, 5 } }, // BLENDVPD(X,PSUBQ(0,X),X)
4800 { ISD::MULHS, MVT::v4i32, { 3, 9, 6, 7 } },
4801 { ISD::MULHU, MVT::v4i32, { 3, 9, 6, 7 } },
4802 { ISD::SADDSAT, MVT::v2i64, { 10, 14, 17, 21 } },
4803 { ISD::SADDSAT, MVT::v4i32, { 5, 11, 8, 10 } },
4804 { ISD::SSUBSAT, MVT::v2i64, { 12, 19, 25, 29 } },
4805 { ISD::SSUBSAT, MVT::v4i32, { 6, 14, 10, 12 } },
4806 { ISD::SMAX, MVT::v2i64, { 3, 7, 2, 3 } },
4807 { ISD::SMAX, MVT::v4i32, { 1, 1, 1, 1 } },
4808 { ISD::SMAX, MVT::v16i8, { 1, 1, 1, 1 } },
4809 { ISD::SMIN, MVT::v2i64, { 3, 7, 2, 3 } },
4810 { ISD::SMIN, MVT::v4i32, { 1, 1, 1, 1 } },
4811 { ISD::SMIN, MVT::v16i8, { 1, 1, 1, 1 } },
4812 { ISD::SMULO, MVT::v2i64, { 9, 11, 13, 17 } },
4813 { ISD::SMULO, MVT::v4i32, { 20, 24, 13, 19 } },
4814 { ISD::SMULO, MVT::v8i16, { 5, 9, 8, 8 } },
4815 { ISD::SMULO, MVT::v16i8, { 13, 22, 24, 25 } },
4816 { ISD::UADDSAT, MVT::v2i64, { 6, 13, 14, 14 } },
4817 { ISD::UADDSAT, MVT::v4i32, { 2, 2, 4, 4 } },
4818 { ISD::USUBSAT, MVT::v2i64, { 6, 10, 14, 14 } },
4819 { ISD::USUBSAT, MVT::v4i32, { 1, 2, 2, 2 } },
4820 { ISD::UMAX, MVT::v2i64, { 2, 11, 6, 7 } },
4821 { ISD::UMAX, MVT::v4i32, { 1, 1, 1, 1 } },
4822 { ISD::UMAX, MVT::v8i16, { 1, 1, 1, 1 } },
4823 { ISD::UMIN, MVT::v2i64, { 2, 11, 6, 7 } },
4824 { ISD::UMIN, MVT::v4i32, { 1, 1, 1, 1 } },
4825 { ISD::UMIN, MVT::v8i16, { 1, 1, 1, 1 } },
4826 { ISD::UMULO, MVT::v2i64, { 14, 20, 15, 20 } },
4827 { ISD::UMULO, MVT::v4i32, { 19, 22, 12, 18 } },
4828 { ISD::UMULO, MVT::v8i16, { 4, 9, 7, 7 } },
4829 { ISD::UMULO, MVT::v16i8, { 13, 19, 18, 20 } },
4830 };
4831 static const CostKindTblEntry SSSE3CostTbl[] = {
4832 { ISD::ABS, MVT::v4i32, { 1, 2, 1, 1 } },
4833 { ISD::ABS, MVT::v8i16, { 1, 2, 1, 1 } },
4834 { ISD::ABS, MVT::v16i8, { 1, 2, 1, 1 } },
4835 { ISD::BITREVERSE, MVT::v2i64, { 16, 20, 11, 21 } },
4836 { ISD::BITREVERSE, MVT::v4i32, { 16, 20, 11, 21 } },
4837 { ISD::BITREVERSE, MVT::v8i16, { 16, 20, 11, 21 } },
4838 { ISD::BITREVERSE, MVT::v16i8, { 11, 12, 10, 16 } },
4839 { ISD::BSWAP, MVT::v2i64, { 2, 3, 1, 5 } },
4840 { ISD::BSWAP, MVT::v4i32, { 2, 3, 1, 5 } },
4841 { ISD::BSWAP, MVT::v8i16, { 2, 3, 1, 5 } },
4842 { ISD::CTLZ, MVT::v2i64, { 18, 28, 28, 35 } },
4843 { ISD::CTLZ, MVT::v4i32, { 15, 20, 22, 28 } },
4844 { ISD::CTLZ, MVT::v8i16, { 13, 17, 16, 22 } },
4845 { ISD::CTLZ, MVT::v16i8, { 11, 15, 10, 16 } },
4846 { ISD::CTPOP, MVT::v2i64, { 13, 19, 12, 18 } },
4847 { ISD::CTPOP, MVT::v4i32, { 18, 24, 16, 22 } },
4848 { ISD::CTPOP, MVT::v8i16, { 13, 18, 14, 20 } },
4849 { ISD::CTPOP, MVT::v16i8, { 11, 12, 10, 16 } },
4850 { ISD::CTTZ, MVT::v2i64, { 13, 25, 15, 22 } },
4851 { ISD::CTTZ, MVT::v4i32, { 18, 26, 19, 25 } },
4852 { ISD::CTTZ, MVT::v8i16, { 13, 20, 17, 23 } },
4853 { ISD::CTTZ, MVT::v16i8, { 11, 16, 13, 19 } }
4854 };
4855 static const CostKindTblEntry SSE2CostTbl[] = {
4856 { ISD::ABS, MVT::v2i64, { 3, 6, 5, 5 } },
4857 { ISD::ABS, MVT::v4i32, { 1, 4, 4, 4 } },
4858 { ISD::ABS, MVT::v8i16, { 1, 2, 3, 3 } },
4859 { ISD::ABS, MVT::v16i8, { 1, 2, 3, 3 } },
4860 { ISD::BITREVERSE, MVT::v2i64, { 16, 20, 32, 32 } },
4861 { ISD::BITREVERSE, MVT::v4i32, { 16, 20, 30, 30 } },
4862 { ISD::BITREVERSE, MVT::v8i16, { 16, 20, 25, 25 } },
4863 { ISD::BITREVERSE, MVT::v16i8, { 11, 12, 21, 21 } },
4864 { ISD::BSWAP, MVT::v2i64, { 5, 6, 11, 11 } },
4865 { ISD::BSWAP, MVT::v4i32, { 5, 5, 9, 9 } },
4866 { ISD::BSWAP, MVT::v8i16, { 5, 5, 4, 5 } },
4867 { ISD::CTLZ, MVT::v2i64, { 10, 45, 36, 38 } },
4868 { ISD::CTLZ, MVT::v4i32, { 10, 45, 38, 40 } },
4869 { ISD::CTLZ, MVT::v8i16, { 9, 38, 32, 34 } },
4870 { ISD::CTLZ, MVT::v16i8, { 8, 39, 29, 32 } },
4871 { ISD::CTPOP, MVT::v2i64, { 12, 26, 16, 18 } },
4872 { ISD::CTPOP, MVT::v4i32, { 15, 29, 21, 23 } },
4873 { ISD::CTPOP, MVT::v8i16, { 13, 25, 18, 20 } },
4874 { ISD::CTPOP, MVT::v16i8, { 10, 21, 14, 16 } },
4875 { ISD::CTTZ, MVT::v2i64, { 14, 28, 19, 21 } },
4876 { ISD::CTTZ, MVT::v4i32, { 18, 31, 24, 26 } },
4877 { ISD::CTTZ, MVT::v8i16, { 16, 27, 21, 23 } },
4878 { ISD::CTTZ, MVT::v16i8, { 13, 23, 17, 19 } },
4879 { ISD::MULHS, MVT::v4i32, { 5, 11, 15, 15 } },
4880 { ISD::MULHS, MVT::v8i16, { 1, 5, 1, 1 } },
4881 { ISD::MULHU, MVT::v4i32, { 3, 9, 7, 7 } },
4882 { ISD::MULHU, MVT::v8i16, { 1, 5, 1, 1 } },
4883 { ISD::SADDSAT, MVT::v2i64, { 12, 14, 24, 24 } },
4884 { ISD::SADDSAT, MVT::v4i32, { 6, 11, 11, 12 } },
4885 { ISD::SADDSAT, MVT::v8i16, { 1, 2, 1, 1 } },
4886 { ISD::SADDSAT, MVT::v16i8, { 1, 2, 1, 1 } },
4887 { ISD::SMAX, MVT::v2i64, { 4, 8, 15, 15 } },
4888 { ISD::SMAX, MVT::v4i32, { 2, 4, 5, 5 } },
4889 { ISD::SMAX, MVT::v8i16, { 1, 1, 1, 1 } },
4890 { ISD::SMAX, MVT::v16i8, { 2, 4, 5, 5 } },
4891 { ISD::SMIN, MVT::v2i64, { 4, 8, 15, 15 } },
4892 { ISD::SMIN, MVT::v4i32, { 2, 4, 5, 5 } },
4893 { ISD::SMIN, MVT::v8i16, { 1, 1, 1, 1 } },
4894 { ISD::SMIN, MVT::v16i8, { 2, 4, 5, 5 } },
4895 { ISD::SMULO, MVT::v2i64, { 30, 33, 13, 23 } },
4896 { ISD::SMULO, MVT::v4i32, { 20, 24, 23, 23 } },
4897 { ISD::SMULO, MVT::v8i16, { 5, 10, 8, 8 } },
4898 { ISD::SMULO, MVT::v16i8, { 13, 23, 24, 25 } },
4899 { ISD::SSUBSAT, MVT::v2i64, { 16, 19, 31, 31 } },
4900 { ISD::SSUBSAT, MVT::v4i32, { 6, 14, 12, 13 } },
4901 { ISD::SSUBSAT, MVT::v8i16, { 1, 2, 1, 1 } },
4902 { ISD::SSUBSAT, MVT::v16i8, { 1, 2, 1, 1 } },
4903 { ISD::UADDSAT, MVT::v2i64, { 7, 13, 14, 14 } },
4904 { ISD::UADDSAT, MVT::v4i32, { 4, 5, 7, 7 } },
4905 { ISD::UADDSAT, MVT::v8i16, { 1, 2, 1, 1 } },
4906 { ISD::UADDSAT, MVT::v16i8, { 1, 2, 1, 1 } },
4907 { ISD::UMAX, MVT::v2i64, { 4, 8, 15, 15 } },
4908 { ISD::UMAX, MVT::v4i32, { 2, 5, 8, 8 } },
4909 { ISD::UMAX, MVT::v8i16, { 1, 3, 3, 3 } },
4910 { ISD::UMAX, MVT::v16i8, { 1, 1, 1, 1 } },
4911 { ISD::UMIN, MVT::v2i64, { 4, 8, 15, 15 } },
4912 { ISD::UMIN, MVT::v4i32, { 2, 5, 8, 8 } },
4913 { ISD::UMIN, MVT::v8i16, { 1, 3, 3, 3 } },
4914 { ISD::UMIN, MVT::v16i8, { 1, 1, 1, 1 } },
4915 { ISD::UMULO, MVT::v2i64, { 30, 33, 15, 29 } },
4916 { ISD::UMULO, MVT::v4i32, { 19, 22, 14, 18 } },
4917 { ISD::UMULO, MVT::v8i16, { 4, 9, 7, 7 } },
4918 { ISD::UMULO, MVT::v16i8, { 13, 19, 20, 20 } },
4919 { ISD::USUBSAT, MVT::v2i64, { 7, 10, 14, 14 } },
4920 { ISD::USUBSAT, MVT::v4i32, { 4, 4, 7, 7 } },
4921 { ISD::USUBSAT, MVT::v8i16, { 1, 2, 1, 1 } },
4922 { ISD::USUBSAT, MVT::v16i8, { 1, 2, 1, 1 } },
4923 { ISD::FMAXNUM, MVT::f64, { 5, 5, 7, 7 } },
4924 { ISD::FMAXNUM, MVT::v2f64, { 4, 6, 6, 6 } },
4925 { ISD::FSQRT, MVT::f64, { 32, 32, 1, 1 } }, // Nehalem from http://www.agner.org/
4926 { ISD::FSQRT, MVT::v2f64, { 32, 32, 1, 1 } }, // Nehalem from http://www.agner.org/
4927 };
4928 static const CostKindTblEntry SSE1CostTbl[] = {
4929 { ISD::FMAXNUM, MVT::f32, { 5, 5, 7, 7 } },
4930 { ISD::FMAXNUM, MVT::v4f32, { 4, 6, 6, 6 } },
4931 { ISD::FSQRT, MVT::f32, { 28, 30, 1, 2 } }, // Pentium III from http://www.agner.org/
4932 { ISD::FSQRT, MVT::v4f32, { 56, 56, 1, 2 } }, // Pentium III from http://www.agner.org/
4933 };
4934 static const CostKindTblEntry BMI64CostTbl[] = { // 64-bit targets
4935 { ISD::CTTZ, MVT::i64, { 1, 1, 1, 1 } },
4936 };
4937 static const CostKindTblEntry BMI32CostTbl[] = { // 32 or 64-bit targets
4938 { ISD::CTTZ, MVT::i32, { 1, 1, 1, 1 } },
4939 { ISD::CTTZ, MVT::i16, { 2, 1, 1, 1 } },
4940 { ISD::CTTZ, MVT::i8, { 2, 1, 1, 1 } },
4941 };
4942 static const CostKindTblEntry LZCNT64CostTbl[] = { // 64-bit targets
4943 { ISD::CTLZ, MVT::i64, { 1, 1, 1, 1 } },
4944 };
4945 static const CostKindTblEntry LZCNT32CostTbl[] = { // 32 or 64-bit targets
4946 { ISD::CTLZ, MVT::i32, { 1, 1, 1, 1 } },
4947 { ISD::CTLZ, MVT::i16, { 2, 1, 1, 1 } },
4948 { ISD::CTLZ, MVT::i8, { 2, 1, 1, 1 } },
4949 };
4950 static const CostKindTblEntry POPCNT64CostTbl[] = { // 64-bit targets
4951 { ISD::CTPOP, MVT::i64, { 1, 1, 1, 1 } }, // popcnt
4952 };
4953 static const CostKindTblEntry POPCNT32CostTbl[] = { // 32 or 64-bit targets
4954 { ISD::CTPOP, MVT::i32, { 1, 1, 1, 1 } }, // popcnt
4955 { ISD::CTPOP, MVT::i16, { 1, 1, 2, 2 } }, // popcnt(zext())
4956 { ISD::CTPOP, MVT::i8, { 1, 1, 2, 2 } }, // popcnt(zext())
4957 };
4958 static const CostKindTblEntry PCLMULCostTbl[] = {
4959 { ISD::CLMUL, MVT::v2i64, { 3, 12, 4, 8 } }, // MOV+2xPCLMUL+unpack
4960 { ISD::CLMUL, MVT::v4i32, { 8, 18, 12, 16 } }, // MOV+4xPCLMUL+unpack
4961 { ISD::CLMUL, MVT::i64, { 3, 12, 4, 8 } }, // MOV+PCLMUL+MOV
4962 { ISD::CLMUL, MVT::i32, { 3, 12, 4, 8 } }, // MOV+PCLMUL+MOV
4963 { ISD::CLMUL, MVT::i16, { 3, 12, 4, 8 } }, // MOV+PCLMUL+MOV
4964 { ISD::CLMUL, MVT::i8, { 3, 12, 4, 8 } }, // MOV+PCLMUL+MOV
4965 };
4966 static const CostKindTblEntry X64CostTbl[] = { // 64-bit targets
4967 { ISD::ABS, MVT::i64, { 1, 2, 3, 3 } }, // SUB+CMOV
4968 { ISD::BITREVERSE, MVT::i64, { 10, 12, 20, 22 } },
4969 { ISD::BSWAP, MVT::i64, { 1, 2, 1, 2 } },
4970 { ISD::CTLZ, MVT::i64, { 1, 2, 3, 3 } }, // MOV+BSR+XOR
4971 { ISD::CTLZ, MVT::i32, { 1, 2, 3, 3 } }, // MOV+BSR+XOR
4972 { ISD::CTLZ, MVT::i16, { 2, 2, 3, 3 } }, // MOV+BSR+XOR
4973 { ISD::CTLZ, MVT::i8, { 2, 2, 4, 3 } }, // MOV+BSR+XOR
4974 { ISD::CTLZ_ZERO_POISON,MVT::i64,{ 1, 2, 2, 2 } }, // BSR+XOR
4975 { ISD::CTTZ, MVT::i64, { 1, 2, 2, 2 } }, // MOV+BSF
4976 { ISD::CTTZ, MVT::i32, { 1, 2, 2, 2 } }, // MOV+BSF
4977 { ISD::CTTZ, MVT::i16, { 2, 2, 2, 2 } }, // MOV+BSF
4978 { ISD::CTTZ, MVT::i8, { 2, 2, 2, 2 } }, // MOV+BSF
4979 { ISD::CTTZ_ZERO_POISON,MVT::i64,{ 1, 2, 1, 2 } }, // BSF
4980 { ISD::CTPOP, MVT::i64, { 10, 6, 19, 19 } },
4981 { ISD::ROTL, MVT::i64, { 2, 3, 1, 3 } },
4982 { ISD::ROTR, MVT::i64, { 2, 3, 1, 3 } },
4983 { X86ISD::VROTLI, MVT::i64, { 1, 1, 1, 1 } },
4984 { ISD::FSHL, MVT::i64, { 4, 4, 1, 4 } },
4985 { ISD::SADDSAT, MVT::i64, { 4, 4, 7, 10 } },
4986 { ISD::SSUBSAT, MVT::i64, { 4, 5, 8, 11 } },
4987 { ISD::UADDSAT, MVT::i64, { 2, 3, 4, 7 } },
4988 { ISD::USUBSAT, MVT::i64, { 2, 3, 4, 7 } },
4989 { ISD::SMAX, MVT::i64, { 1, 3, 2, 3 } },
4990 { ISD::SMIN, MVT::i64, { 1, 3, 2, 3 } },
4991 { ISD::UMAX, MVT::i64, { 1, 3, 2, 3 } },
4992 { ISD::UMIN, MVT::i64, { 1, 3, 2, 3 } },
4993 { ISD::SADDO, MVT::i64, { 2, 2, 4, 6 } },
4994 { ISD::UADDO, MVT::i64, { 2, 2, 4, 6 } },
4995 { ISD::SMULO, MVT::i64, { 4, 4, 4, 6 } },
4996 { ISD::UMULO, MVT::i64, { 8, 8, 4, 7 } },
4997 };
4998 static const CostKindTblEntry X86CostTbl[] = { // 32 or 64-bit targets
4999 { ISD::ABS, MVT::i32, { 1, 2, 3, 3 } }, // SUB+XOR+SRA or SUB+CMOV
5000 { ISD::ABS, MVT::i16, { 2, 2, 3, 3 } }, // SUB+XOR+SRA or SUB+CMOV
5001 { ISD::ABS, MVT::i8, { 2, 4, 4, 3 } }, // SUB+XOR+SRA
5002 { ISD::BITREVERSE, MVT::i32, { 9, 12, 17, 19 } },
5003 { ISD::BITREVERSE, MVT::i16, { 9, 12, 17, 19 } },
5004 { ISD::BITREVERSE, MVT::i8, { 7, 9, 13, 14 } },
5005 { ISD::BSWAP, MVT::i32, { 1, 1, 1, 1 } },
5006 { ISD::BSWAP, MVT::i16, { 1, 2, 1, 2 } }, // ROL
5007 { ISD::CTLZ, MVT::i32, { 2, 2, 4, 5 } }, // BSR+XOR or BSR+XOR+CMOV
5008 { ISD::CTLZ, MVT::i16, { 2, 2, 4, 5 } }, // BSR+XOR or BSR+XOR+CMOV
5009 { ISD::CTLZ, MVT::i8, { 2, 2, 5, 6 } }, // BSR+XOR or BSR+XOR+CMOV
5010 { ISD::CTLZ_ZERO_POISON,MVT::i32,{ 1, 2, 2, 2 } }, // BSR+XOR
5011 { ISD::CTLZ_ZERO_POISON,MVT::i16,{ 2, 2, 2, 2 } }, // BSR+XOR
5012 { ISD::CTLZ_ZERO_POISON,MVT::i8, { 2, 2, 3, 3 } }, // BSR+XOR
5013 { ISD::CTTZ, MVT::i32, { 2, 2, 3, 3 } }, // TEST+BSF+CMOV/BRANCH
5014 { ISD::CTTZ, MVT::i16, { 2, 2, 2, 3 } }, // TEST+BSF+CMOV/BRANCH
5015 { ISD::CTTZ, MVT::i8, { 2, 2, 2, 3 } }, // TEST+BSF+CMOV/BRANCH
5016 { ISD::CTTZ_ZERO_POISON,MVT::i32,{ 1, 2, 1, 2 } }, // BSF
5017 { ISD::CTTZ_ZERO_POISON,MVT::i16,{ 2, 2, 1, 2 } }, // BSF
5018 { ISD::CTTZ_ZERO_POISON,MVT::i8, { 2, 2, 1, 2 } }, // BSF
5019 { ISD::CTPOP, MVT::i32, { 8, 7, 15, 15 } },
5020 { ISD::CTPOP, MVT::i16, { 9, 8, 17, 17 } },
5021 { ISD::CTPOP, MVT::i8, { 7, 6, 6, 6 } },
5022 { ISD::ROTL, MVT::i32, { 2, 3, 1, 3 } },
5023 { ISD::ROTL, MVT::i16, { 2, 3, 1, 3 } },
5024 { ISD::ROTL, MVT::i8, { 2, 3, 1, 3 } },
5025 { ISD::ROTR, MVT::i32, { 2, 3, 1, 3 } },
5026 { ISD::ROTR, MVT::i16, { 2, 3, 1, 3 } },
5027 { ISD::ROTR, MVT::i8, { 2, 3, 1, 3 } },
5028 { X86ISD::VROTLI, MVT::i32, { 1, 1, 1, 1 } },
5029 { X86ISD::VROTLI, MVT::i16, { 1, 1, 1, 1 } },
5030 { X86ISD::VROTLI, MVT::i8, { 1, 1, 1, 1 } },
5031 { ISD::FSHL, MVT::i32, { 4, 4, 1, 4 } },
5032 { ISD::FSHL, MVT::i16, { 4, 4, 2, 5 } },
5033 { ISD::FSHL, MVT::i8, { 4, 4, 2, 5 } },
5034 { ISD::SADDSAT, MVT::i32, { 3, 4, 6, 9 } },
5035 { ISD::SADDSAT, MVT::i16, { 4, 4, 7, 10 } },
5036 { ISD::SADDSAT, MVT::i8, { 4, 5, 8, 11 } },
5037 { ISD::SSUBSAT, MVT::i32, { 4, 4, 7, 10 } },
5038 { ISD::SSUBSAT, MVT::i16, { 4, 4, 7, 10 } },
5039 { ISD::SSUBSAT, MVT::i8, { 4, 5, 8, 11 } },
5040 { ISD::UADDSAT, MVT::i32, { 2, 3, 4, 7 } },
5041 { ISD::UADDSAT, MVT::i16, { 2, 3, 4, 7 } },
5042 { ISD::UADDSAT, MVT::i8, { 3, 3, 5, 8 } },
5043 { ISD::USUBSAT, MVT::i32, { 2, 3, 4, 7 } },
5044 { ISD::USUBSAT, MVT::i16, { 2, 3, 4, 7 } },
5045 { ISD::USUBSAT, MVT::i8, { 3, 3, 5, 8 } },
5046 { ISD::SMAX, MVT::i32, { 1, 2, 2, 3 } },
5047 { ISD::SMAX, MVT::i16, { 1, 4, 2, 4 } },
5048 { ISD::SMAX, MVT::i8, { 1, 4, 2, 4 } },
5049 { ISD::SMIN, MVT::i32, { 1, 2, 2, 3 } },
5050 { ISD::SMIN, MVT::i16, { 1, 4, 2, 4 } },
5051 { ISD::SMIN, MVT::i8, { 1, 4, 2, 4 } },
5052 { ISD::UMAX, MVT::i32, { 1, 2, 2, 3 } },
5053 { ISD::UMAX, MVT::i16, { 1, 4, 2, 4 } },
5054 { ISD::UMAX, MVT::i8, { 1, 4, 2, 4 } },
5055 { ISD::UMIN, MVT::i32, { 1, 2, 2, 3 } },
5056 { ISD::UMIN, MVT::i16, { 1, 4, 2, 4 } },
5057 { ISD::UMIN, MVT::i8, { 1, 4, 2, 4 } },
5058 { ISD::SADDO, MVT::i32, { 2, 2, 4, 6 } },
5059 { ISD::SADDO, MVT::i16, { 2, 2, 4, 6 } },
5060 { ISD::SADDO, MVT::i8, { 2, 2, 4, 6 } },
5061 { ISD::UADDO, MVT::i32, { 2, 2, 4, 6 } },
5062 { ISD::UADDO, MVT::i16, { 2, 2, 4, 6 } },
5063 { ISD::UADDO, MVT::i8, { 2, 2, 4, 6 } },
5064 { ISD::SMULO, MVT::i32, { 2, 2, 4, 6 } },
5065 { ISD::SMULO, MVT::i16, { 5, 5, 4, 6 } },
5066 { ISD::SMULO, MVT::i8, { 6, 6, 4, 6 } },
5067 { ISD::UMULO, MVT::i32, { 6, 6, 4, 8 } },
5068 { ISD::UMULO, MVT::i16, { 6, 6, 4, 9 } },
5069 { ISD::UMULO, MVT::i8, { 6, 6, 4, 6 } },
5070 };
5071
5072 Type *RetTy = ICA.getReturnType();
5073 Type *OpTy = RetTy;
5074 Intrinsic::ID IID = ICA.getID();
5075 unsigned ISD = ISD::DELETED_NODE;
5076 switch (IID) {
5077 default:
5078 break;
5079 case Intrinsic::abs:
5080 ISD = ISD::ABS;
5081 break;
5082 case Intrinsic::bitreverse:
5084 break;
5085 case Intrinsic::bswap:
5086 ISD = ISD::BSWAP;
5087 break;
5088 case Intrinsic::ctlz:
5089 ISD = ISD::CTLZ;
5090 break;
5091 case Intrinsic::ctpop:
5092 ISD = ISD::CTPOP;
5093 break;
5094 case Intrinsic::cttz:
5095 ISD = ISD::CTTZ;
5096 break;
5097 case Intrinsic::fshl:
5098 ISD = ISD::FSHL;
5099 if (!ICA.isTypeBasedOnly()) {
5100 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5101 if (Args[0] == Args[1]) {
5102 ISD = ISD::ROTL;
5103 // Handle uniform constant rotation amounts.
5104 // TODO: Handle funnel-shift cases.
5105 const APInt *Amt;
5106 if (Args[2] &&
5108 ISD = X86ISD::VROTLI;
5109 }
5110 }
5111 break;
5112 case Intrinsic::fshr:
5113 // FSHR has same costs so don't duplicate.
5114 ISD = ISD::FSHL;
5115 if (!ICA.isTypeBasedOnly()) {
5116 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5117 if (Args[0] == Args[1]) {
5118 ISD = ISD::ROTR;
5119 // Handle uniform constant rotation amount.
5120 // TODO: Handle funnel-shift cases.
5121 const APInt *Amt;
5122 if (Args[2] &&
5124 ISD = X86ISD::VROTLI;
5125 }
5126 }
5127 break;
5128 case Intrinsic::lrint:
5129 case Intrinsic::llrint: {
5130 // X86 can use the CVTP2SI instructions to lower lrint/llrint calls, which
5131 // have the same costs as the CVTTP2SI (fptosi) instructions
5132 const SmallVectorImpl<Type *> &ArgTys = ICA.getArgTypes();
5133 return getCastInstrCost(Instruction::FPToSI, RetTy, ArgTys[0],
5135 }
5136 case Intrinsic::maxnum:
5137 case Intrinsic::minnum:
5138 // FMINNUM has same costs so don't duplicate.
5139 ISD = ISD::FMAXNUM;
5140 break;
5141 case Intrinsic::sadd_sat:
5142 ISD = ISD::SADDSAT;
5143 break;
5144 case Intrinsic::smax:
5145 ISD = ISD::SMAX;
5146 break;
5147 case Intrinsic::smin:
5148 ISD = ISD::SMIN;
5149 break;
5150 case Intrinsic::smulh:
5151 ISD = ISD::MULHS;
5152 break;
5153 case Intrinsic::ssub_sat:
5154 ISD = ISD::SSUBSAT;
5155 break;
5156 case Intrinsic::uadd_sat:
5157 ISD = ISD::UADDSAT;
5158 break;
5159 case Intrinsic::umax:
5160 ISD = ISD::UMAX;
5161 break;
5162 case Intrinsic::umin:
5163 ISD = ISD::UMIN;
5164 break;
5165 case Intrinsic::usub_sat:
5166 ISD = ISD::USUBSAT;
5167 break;
5168 case Intrinsic::umulh:
5169 ISD = ISD::MULHU;
5170 break;
5171 case Intrinsic::sqrt:
5172 // The estimate sequence is costed with the division it is folded into.
5175 isFoldedIntoRsqrtEstimate(ICA, *TLI, DL))
5176 return TTI::TCC_Free;
5177 ISD = ISD::FSQRT;
5178 break;
5179 case Intrinsic::sadd_with_overflow:
5180 case Intrinsic::ssub_with_overflow:
5181 // SSUBO has same costs so don't duplicate.
5182 ISD = ISD::SADDO;
5183 OpTy = RetTy->getContainedType(0);
5184 break;
5185 case Intrinsic::uadd_with_overflow:
5186 case Intrinsic::usub_with_overflow:
5187 // USUBO has same costs so don't duplicate.
5188 ISD = ISD::UADDO;
5189 OpTy = RetTy->getContainedType(0);
5190 break;
5191 case Intrinsic::smul_with_overflow:
5192 ISD = ISD::SMULO;
5193 OpTy = RetTy->getContainedType(0);
5194 break;
5195 case Intrinsic::umul_with_overflow:
5196 ISD = ISD::UMULO;
5197 OpTy = RetTy->getContainedType(0);
5198 break;
5199 case Intrinsic::clmul:
5200 ISD = ISD::CLMUL;
5201 break;
5202 }
5203
5204 if (ISD != ISD::DELETED_NODE) {
5205 auto adjustTableCost = [&](int ISD, unsigned Cost,
5206 std::pair<InstructionCost, MVT> LT,
5208 InstructionCost LegalizationCost = LT.first;
5209 MVT MTy = LT.second;
5210
5211 // If there are no NANs to deal with, then these are reduced to a
5212 // single MIN** or MAX** instruction instead of the MIN/CMP/SELECT that we
5213 // assume is used in the non-fast case.
5214 if (ISD == ISD::FMAXNUM || ISD == ISD::FMINNUM) {
5215 if (FMF.noNaNs())
5216 return LegalizationCost * 1;
5217 }
5218
5219 // For cases where some ops can be folded into a load/store, assume free.
5220 if (MTy.isScalarInteger()) {
5221 if (ISD == ISD::BSWAP && ST->hasMOVBE() && ST->hasFastMOVBE()) {
5222 if (const Instruction *II = ICA.getInst()) {
5223 if (II->hasOneUse() && isa<StoreInst>(II->user_back()))
5224 return TTI::TCC_Free;
5225 if (auto *LI = dyn_cast<LoadInst>(II->getOperand(0))) {
5226 if (LI->hasOneUse())
5227 return TTI::TCC_Free;
5228 }
5229 }
5230 }
5231 }
5232
5233 return LegalizationCost * (int)Cost;
5234 };
5235
5236 // Legalize the type.
5237 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(OpTy);
5238 MVT MTy = LT.second;
5239
5240 // Without BMI/LZCNT see if we're only looking for a *_ZERO_POISON cost.
5241 if (((ISD == ISD::CTTZ && !ST->hasBMI()) ||
5242 (ISD == ISD::CTLZ && !ST->hasLZCNT())) &&
5243 !MTy.isVector() && !ICA.isTypeBasedOnly()) {
5244 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
5245 if (auto *Cst = dyn_cast<ConstantInt>(Args[1]))
5246 if (Cst->isAllOnesValue())
5247 ISD =
5249 }
5250
5251 // FSQRT is a single instruction.
5253 return LT.first;
5254
5255 if (ST->useGLMDivSqrtCosts())
5256 if (const auto *Entry = CostTableLookup(GLMCostTbl, ISD, MTy))
5257 if (auto KindCost = Entry->Cost[CostKind])
5258 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5259
5260 if (ST->useSLMArithCosts())
5261 if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
5262 if (auto KindCost = Entry->Cost[CostKind])
5263 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5264
5265 if (ST->hasVBMI2())
5266 if (const auto *Entry = CostTableLookup(AVX512VBMI2CostTbl, ISD, MTy))
5267 if (auto KindCost = Entry->Cost[CostKind])
5268 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5269
5270 if (ST->hasBITALG())
5271 if (const auto *Entry = CostTableLookup(AVX512BITALGCostTbl, ISD, MTy))
5272 if (auto KindCost = Entry->Cost[CostKind])
5273 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5274
5275 if (ST->hasVPOPCNTDQ())
5276 if (const auto *Entry = CostTableLookup(AVX512VPOPCNTDQCostTbl, ISD, MTy))
5277 if (auto KindCost = Entry->Cost[CostKind])
5278 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5279
5280 if (ST->hasGFNI())
5281 if (const auto *Entry = CostTableLookup(GFNICostTbl, ISD, MTy))
5282 if (auto KindCost = Entry->Cost[CostKind])
5283 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5284
5285 if (ST->hasCDI())
5286 if (const auto *Entry = CostTableLookup(AVX512CDCostTbl, ISD, MTy))
5287 if (auto KindCost = Entry->Cost[CostKind])
5288 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5289
5290 if (ST->hasBWI())
5291 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
5292 if (auto KindCost = Entry->Cost[CostKind])
5293 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5294
5295 if (ST->hasAVX512())
5296 if (const auto *Entry = CostTableLookup(AVX512CostTbl, ISD, MTy))
5297 if (auto KindCost = Entry->Cost[CostKind])
5298 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5299
5300 if (ST->hasXOP())
5301 if (const auto *Entry = CostTableLookup(XOPCostTbl, ISD, MTy))
5302 if (auto KindCost = Entry->Cost[CostKind])
5303 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5304
5305 if (ST->hasAVX2())
5306 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
5307 if (auto KindCost = Entry->Cost[CostKind])
5308 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5309
5310 if (ST->hasAVX())
5311 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
5312 if (auto KindCost = Entry->Cost[CostKind])
5313 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5314
5315 if (ST->hasSSE42())
5316 if (const auto *Entry = CostTableLookup(SSE42CostTbl, ISD, MTy))
5317 if (auto KindCost = Entry->Cost[CostKind])
5318 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5319
5320 if (ST->hasSSE41())
5321 if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
5322 if (auto KindCost = Entry->Cost[CostKind])
5323 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5324
5325 if (ST->hasSSSE3())
5326 if (const auto *Entry = CostTableLookup(SSSE3CostTbl, ISD, MTy))
5327 if (auto KindCost = Entry->Cost[CostKind])
5328 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5329
5330 if (ST->hasSSE2())
5331 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
5332 if (auto KindCost = Entry->Cost[CostKind])
5333 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5334
5335 if (ST->hasSSE1())
5336 if (const auto *Entry = CostTableLookup(SSE1CostTbl, ISD, MTy))
5337 if (auto KindCost = Entry->Cost[CostKind])
5338 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5339
5340 if (ST->hasBMI()) {
5341 if (ST->is64Bit())
5342 if (const auto *Entry = CostTableLookup(BMI64CostTbl, ISD, MTy))
5343 if (auto KindCost = Entry->Cost[CostKind])
5344 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5345
5346 if (const auto *Entry = CostTableLookup(BMI32CostTbl, ISD, MTy))
5347 if (auto KindCost = Entry->Cost[CostKind])
5348 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5349 }
5350
5351 if (ST->hasLZCNT()) {
5352 if (ST->is64Bit())
5353 if (const auto *Entry = CostTableLookup(LZCNT64CostTbl, ISD, MTy))
5354 if (auto KindCost = Entry->Cost[CostKind])
5355 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5356
5357 if (const auto *Entry = CostTableLookup(LZCNT32CostTbl, ISD, MTy))
5358 if (auto KindCost = Entry->Cost[CostKind])
5359 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5360 }
5361
5362 if (ST->hasPOPCNT()) {
5363 if (ST->is64Bit())
5364 if (const auto *Entry = CostTableLookup(POPCNT64CostTbl, ISD, MTy))
5365 if (auto KindCost = Entry->Cost[CostKind])
5366 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5367
5368 if (const auto *Entry = CostTableLookup(POPCNT32CostTbl, ISD, MTy))
5369 if (auto KindCost = Entry->Cost[CostKind])
5370 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5371 }
5372
5373 // FIXME: PCLMUL w/ AVX/AVX512 and VPCLMULQDQ are not handled properly.
5374 if (ST->hasPCLMUL())
5375 if (const auto *Entry = CostTableLookup(PCLMULCostTbl, ISD, MTy))
5376 if (auto KindCost = Entry->Cost[CostKind])
5377 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5378
5379 if (ST->is64Bit())
5380 if (const auto *Entry = CostTableLookup(X64CostTbl, ISD, MTy))
5381 if (auto KindCost = Entry->Cost[CostKind])
5382 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5383
5384 if (const auto *Entry = CostTableLookup(X86CostTbl, ISD, MTy))
5385 if (auto KindCost = Entry->Cost[CostKind])
5386 return adjustTableCost(Entry->ISD, *KindCost, LT, ICA.getFlags());
5387
5388 // Without arg data, we need to compute the expanded costs of custom lowered
5389 // intrinsics to prevent use of the (very low) default costs.
5390 if (ICA.isTypeBasedOnly() &&
5391 (IID == Intrinsic::fshl || IID == Intrinsic::fshr)) {
5392 Type *CondTy = RetTy->getWithNewBitWidth(1);
5394 Cost += getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
5395 Cost += getArithmeticInstrCost(BinaryOperator::Sub, RetTy, CostKind);
5396 Cost += getArithmeticInstrCost(BinaryOperator::Shl, RetTy, CostKind);
5397 Cost += getArithmeticInstrCost(BinaryOperator::LShr, RetTy, CostKind);
5398 Cost += getArithmeticInstrCost(BinaryOperator::And, RetTy, CostKind);
5399 Cost += getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
5401 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
5403 return Cost;
5404 }
5405 }
5406
5408}
5409
5411 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
5412 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
5413 static const CostTblEntry SLMCostTbl[] = {
5414 { ISD::EXTRACT_VECTOR_ELT, MVT::i8, 4 },
5415 { ISD::EXTRACT_VECTOR_ELT, MVT::i16, 4 },
5416 { ISD::EXTRACT_VECTOR_ELT, MVT::i32, 4 },
5417 { ISD::EXTRACT_VECTOR_ELT, MVT::i64, 7 }
5418 };
5419
5420 assert(Val->isVectorTy() && "This must be a vector type");
5421 auto *VT = cast<VectorType>(Val);
5422 if (VT->isScalableTy())
5424
5425 Type *ScalarType = Val->getScalarType();
5426 InstructionCost RegisterFileMoveCost = 0;
5427
5428 // Non-immediate extraction/insertion can be handled as a sequence of
5429 // aliased loads+stores via the stack.
5430 if (Index == -1U && (Opcode == Instruction::ExtractElement ||
5431 Opcode == Instruction::InsertElement)) {
5432 // TODO: On some SSE41+ targets, we expand to cmp+splat+select patterns:
5433 // inselt N0, N1, N2 --> select (SplatN2 == {0,1,2...}) ? SplatN1 : N0.
5434
5435 // TODO: Move this to BasicTTIImpl.h? We'd need better gep + index handling.
5436 assert(isa<FixedVectorType>(Val) && "Fixed vector type expected");
5437 Align VecAlign = DL.getPrefTypeAlign(Val);
5438 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
5439
5440 // Extract - store vector to stack, load scalar.
5441 if (Opcode == Instruction::ExtractElement) {
5442 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
5443 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
5444 CostKind);
5445 }
5446 // Insert - store vector to stack, store scalar, load vector.
5447 if (Opcode == Instruction::InsertElement) {
5448 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
5449 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
5450 CostKind) +
5451 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind);
5452 }
5453 }
5454
5455 if (Index != -1U && (Opcode == Instruction::ExtractElement ||
5456 Opcode == Instruction::InsertElement)) {
5457 // Extraction of vXi1 elements are now efficiently handled by MOVMSK.
5458 if (Opcode == Instruction::ExtractElement &&
5459 ScalarType->getScalarSizeInBits() == 1 &&
5460 cast<FixedVectorType>(Val)->getNumElements() > 1)
5461 return 1;
5462
5463 // Legalize the type.
5464 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
5465
5466 // This type is legalized to a scalar type.
5467 if (!LT.second.isVector())
5468 return TTI::TCC_Free;
5469
5470 // The type may be split. Normalize the index to the new type.
5471 unsigned SizeInBits = LT.second.getSizeInBits();
5472 unsigned NumElts = LT.second.getVectorNumElements();
5473 unsigned SubNumElts = NumElts;
5474 Index = Index % NumElts;
5475
5476 // For >128-bit vectors, we need to extract higher 128-bit subvectors.
5477 // For inserts, we also need to insert the subvector back.
5478 if (SizeInBits > 128) {
5479 assert((SizeInBits % 128) == 0 && "Illegal vector");
5480 unsigned NumSubVecs = SizeInBits / 128;
5481 SubNumElts = NumElts / NumSubVecs;
5482 if (SubNumElts <= Index) {
5483 RegisterFileMoveCost += (Opcode == Instruction::InsertElement ? 2 : 1);
5484 Index %= SubNumElts;
5485 }
5486 }
5487
5488 MVT MScalarTy = LT.second.getScalarType();
5489 auto IsCheapPInsrPExtrInsertPS = [&]() {
5490 // Assume pinsr/pextr XMM <-> GPR is relatively cheap on all targets.
5491 // Inserting f32 into index0 is just movss.
5492 // Also, assume insertps is relatively cheap on all >= SSE41 targets.
5493 return (MScalarTy == MVT::i16 && ST->hasSSE2()) ||
5494 (MScalarTy.isInteger() && ST->hasSSE41()) ||
5495 (MScalarTy == MVT::f32 && ST->hasSSE1() && Index == 0 &&
5496 Opcode == Instruction::InsertElement) ||
5497 (MScalarTy == MVT::f32 && ST->hasSSE41() &&
5498 Opcode == Instruction::InsertElement);
5499 };
5500
5501 if (Index == 0) {
5502 // Floating point scalars are already located in index #0.
5503 // Many insertions to #0 can fold away for scalar fp-ops, so let's assume
5504 // true for all.
5505 if (ScalarType->isFloatingPointTy() &&
5506 (Opcode != Instruction::InsertElement || !Op0 ||
5507 isa<UndefValue>(Op0)))
5508 return RegisterFileMoveCost;
5509
5510 if (Opcode == Instruction::InsertElement &&
5512 // Consider the gather cost to be cheap.
5514 return RegisterFileMoveCost;
5515 if (!IsCheapPInsrPExtrInsertPS()) {
5516 // mov constant-to-GPR + movd/movq GPR -> XMM.
5517 if (isa_and_nonnull<Constant>(Op1) && Op1->getType()->isIntegerTy())
5518 return 2 + RegisterFileMoveCost;
5519 // Assume movd/movq GPR -> XMM is relatively cheap on all targets.
5520 return 1 + RegisterFileMoveCost;
5521 }
5522 }
5523
5524 // Assume movd/movq XMM -> GPR is relatively cheap on all targets.
5525 if (ScalarType->isIntegerTy() && Opcode == Instruction::ExtractElement)
5526 return 1 + RegisterFileMoveCost;
5527 }
5528
5529 int ISD = TLI->InstructionOpcodeToISD(Opcode);
5530 assert(ISD && "Unexpected vector opcode");
5531 if (ST->useSLMArithCosts())
5532 if (auto *Entry = CostTableLookup(SLMCostTbl, ISD, MScalarTy))
5533 return Entry->Cost + RegisterFileMoveCost;
5534
5535 // Consider cheap cases.
5536 if (IsCheapPInsrPExtrInsertPS())
5537 return 1 + RegisterFileMoveCost;
5538
5539 // For extractions we just need to shuffle the element to index 0, which
5540 // should be very cheap (assume cost = 1). For insertions we need to shuffle
5541 // the elements to its destination. In both cases we must handle the
5542 // subvector move(s).
5543 // If the vector type is already less than 128-bits then don't reduce it.
5544 // TODO: Under what circumstances should we shuffle using the full width?
5545 InstructionCost ShuffleCost = 1;
5546 if (Opcode == Instruction::InsertElement) {
5547 auto *SubTy = cast<VectorType>(Val);
5548 EVT VT = TLI->getValueType(DL, Val);
5549 if (VT.getScalarType() != MScalarTy || VT.getSizeInBits() >= 128)
5550 SubTy = FixedVectorType::get(ScalarType, SubNumElts);
5551 ShuffleCost = getShuffleCost(TTI::SK_PermuteTwoSrc, SubTy, SubTy,
5552 CostKind, {}, 0, SubTy);
5553 }
5554 int IntOrFpCost = ScalarType->isFloatingPointTy() ? 0 : 1;
5555 return ShuffleCost + IntOrFpCost + RegisterFileMoveCost;
5556 }
5557
5558 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
5559 VIC) +
5560 RegisterFileMoveCost;
5561}
5562
5564 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
5565 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
5566 TTI::VectorInstrContext VIC) const {
5567 assert(DemandedElts.getBitWidth() ==
5568 cast<FixedVectorType>(Ty)->getNumElements() &&
5569 "Vector size mismatch");
5570
5571 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
5572 MVT MScalarTy = LT.second.getScalarType();
5573 unsigned LegalVectorBitWidth = LT.second.getSizeInBits();
5575
5576 constexpr unsigned LaneBitWidth = 128;
5577 assert((LegalVectorBitWidth < LaneBitWidth ||
5578 (LegalVectorBitWidth % LaneBitWidth) == 0) &&
5579 "Illegal vector");
5580
5581 const int NumLegalVectors = LT.first.getValue();
5582 assert(NumLegalVectors >= 0 && "Negative cost!");
5583
5584 // For insertions, a ISD::BUILD_VECTOR style vector initialization can be much
5585 // cheaper than an accumulation of ISD::INSERT_VECTOR_ELT. SLPVectorizer has
5586 // a special heuristic regarding poison input which is passed here in
5587 // ForPoisonSrc.
5588 if (Insert && !ForPoisonSrc) {
5589 // This is nearly identical to BaseT::getScalarizationOverhead(), except
5590 // it is passing nullptr to getVectorInstrCost() for Op0 (instead of
5591 // Constant::getNullValue()), which makes the X86TTIImpl
5592 // getVectorInstrCost() return 0 instead of 1.
5593 for (unsigned I : seq(DemandedElts.getBitWidth())) {
5594 if (!DemandedElts[I])
5595 continue;
5596 Cost += getVectorInstrCost(Instruction::InsertElement, Ty, CostKind, I,
5598 VL.empty() ? nullptr : VL[I],
5600 }
5601 return Cost;
5602 }
5603
5604 if (Insert) {
5605 if ((MScalarTy == MVT::i16 && ST->hasSSE2()) ||
5606 (MScalarTy.isInteger() && ST->hasSSE41()) ||
5607 (MScalarTy == MVT::f32 && ST->hasSSE41())) {
5608 // For types we can insert directly, insertion into 128-bit sub vectors is
5609 // cheap, followed by a cheap chain of concatenations.
5610 if (LegalVectorBitWidth <= LaneBitWidth) {
5611 Cost += BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert,
5612 /*Extract*/ false, CostKind);
5613 } else {
5614 // In each 128-lane, if at least one index is demanded but not all
5615 // indices are demanded and this 128-lane is not the first 128-lane of
5616 // the legalized-vector, then this 128-lane needs a extracti128; If in
5617 // each 128-lane, there is at least one demanded index, this 128-lane
5618 // needs a inserti128.
5619
5620 // The following cases will help you build a better understanding:
5621 // Assume we insert several elements into a v8i32 vector in avx2,
5622 // Case#1: inserting into 1th index needs vpinsrd + inserti128.
5623 // Case#2: inserting into 5th index needs extracti128 + vpinsrd +
5624 // inserti128.
5625 // Case#3: inserting into 4,5,6,7 index needs 4*vpinsrd + inserti128.
5626 assert((LegalVectorBitWidth % LaneBitWidth) == 0 && "Illegal vector");
5627 unsigned NumLegalLanes = LegalVectorBitWidth / LaneBitWidth;
5628 unsigned NumLanesTotal = NumLegalLanes * NumLegalVectors;
5629 unsigned NumLegalElts =
5630 LT.second.getVectorNumElements() * NumLegalVectors;
5631 assert(NumLegalElts >= DemandedElts.getBitWidth() &&
5632 "Vector has been legalized to smaller element count");
5633 assert((NumLegalElts % NumLanesTotal) == 0 &&
5634 "Unexpected elts per lane");
5635 unsigned NumEltsPerLane = NumLegalElts / NumLanesTotal;
5636
5637 APInt WidenedDemandedElts = DemandedElts.zext(NumLegalElts);
5638 auto *LaneTy =
5639 FixedVectorType::get(Ty->getElementType(), NumEltsPerLane);
5640
5641 for (unsigned I = 0; I != NumLanesTotal; ++I) {
5642 APInt LaneEltMask = WidenedDemandedElts.extractBits(
5643 NumEltsPerLane, NumEltsPerLane * I);
5644 if (LaneEltMask.isZero())
5645 continue;
5646 // FIXME: we don't need to extract if all non-demanded elements
5647 // are legalization-inserted padding.
5648 if (!LaneEltMask.isAllOnes())
5650 {}, I * NumEltsPerLane, LaneTy);
5651 Cost += BaseT::getScalarizationOverhead(LaneTy, LaneEltMask, Insert,
5652 /*Extract*/ false, CostKind);
5653 }
5654
5655 APInt AffectedLanes =
5656 APIntOps::ScaleBitMask(WidenedDemandedElts, NumLanesTotal);
5657 APInt FullyAffectedLegalVectors = APIntOps::ScaleBitMask(
5658 AffectedLanes, NumLegalVectors, /*MatchAllBits=*/true);
5659 for (int LegalVec = 0; LegalVec != NumLegalVectors; ++LegalVec) {
5660 for (unsigned Lane = 0; Lane != NumLegalLanes; ++Lane) {
5661 unsigned I = NumLegalLanes * LegalVec + Lane;
5662 // No need to insert unaffected lane; or lane 0 of each legal vector
5663 // iff ALL lanes of that vector were affected and will be inserted.
5664 if (!AffectedLanes[I] ||
5665 (Lane == 0 && FullyAffectedLegalVectors[LegalVec]))
5666 continue;
5668 {}, I * NumEltsPerLane, LaneTy);
5669 }
5670 }
5671 }
5672 } else if (LT.second.isVector()) {
5673 // Without fast insertion, we need to use MOVD/MOVQ to pass each demanded
5674 // integer element as a SCALAR_TO_VECTOR, then we build the vector as a
5675 // series of UNPCK followed by CONCAT_VECTORS - all of these can be
5676 // considered cheap.
5677 if (Ty->isIntOrIntVectorTy())
5678 Cost += DemandedElts.popcount();
5679
5680 // Get the smaller of the legalized or original pow2-extended number of
5681 // vector elements, which represents the number of unpacks we'll end up
5682 // performing.
5683 unsigned NumElts = LT.second.getVectorNumElements();
5684 unsigned Pow2Elts =
5685 PowerOf2Ceil(cast<FixedVectorType>(Ty)->getNumElements());
5686 Cost += (std::min<unsigned>(NumElts, Pow2Elts) - 1) * LT.first;
5687 }
5688 }
5689
5690 if (Extract) {
5691 // vXi1 can be efficiently extracted with MOVMSK.
5692 // TODO: AVX512 predicate mask handling.
5693 // NOTE: This doesn't work well for roundtrip scalarization.
5694 if (!Insert && Ty->getScalarSizeInBits() == 1 && !ST->hasAVX512()) {
5695 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
5696 unsigned MaxElts = ST->hasAVX2() ? 32 : 16;
5697 unsigned MOVMSKCost = (NumElts + MaxElts - 1) / MaxElts;
5698 return MOVMSKCost;
5699 }
5700
5701 if (LT.second.isVector()) {
5702 unsigned NumLegalElts =
5703 LT.second.getVectorNumElements() * NumLegalVectors;
5704 assert(NumLegalElts >= DemandedElts.getBitWidth() &&
5705 "Vector has been legalized to smaller element count");
5706
5707 // If we're extracting elements from a 128-bit subvector lane,
5708 // we only need to extract each lane once, not for every element.
5709 if (LegalVectorBitWidth > LaneBitWidth) {
5710 unsigned NumLegalLanes = LegalVectorBitWidth / LaneBitWidth;
5711 unsigned NumLanesTotal = NumLegalLanes * NumLegalVectors;
5712 assert((NumLegalElts % NumLanesTotal) == 0 &&
5713 "Unexpected elts per lane");
5714 unsigned NumEltsPerLane = NumLegalElts / NumLanesTotal;
5715
5716 // Add cost for each demanded 128-bit subvector extraction.
5717 // Luckily this is a lot easier than for insertion.
5718 APInt WidenedDemandedElts = DemandedElts.zext(NumLegalElts);
5719 auto *LaneTy =
5720 FixedVectorType::get(Ty->getElementType(), NumEltsPerLane);
5721
5722 for (unsigned I = 0; I != NumLanesTotal; ++I) {
5723 APInt LaneEltMask = WidenedDemandedElts.extractBits(
5724 NumEltsPerLane, I * NumEltsPerLane);
5725 if (LaneEltMask.isZero())
5726 continue;
5728 I * NumEltsPerLane, LaneTy);
5730 LaneTy, LaneEltMask, /*Insert*/ false, Extract, CostKind);
5731 }
5732
5733 return Cost;
5734 }
5735 }
5736
5737 // Fallback to default extraction.
5738 Cost += BaseT::getScalarizationOverhead(Ty, DemandedElts, /*Insert*/ false,
5739 Extract, CostKind);
5740 }
5741
5742 return Cost;
5743}
5744
5746X86TTIImpl::getReplicationShuffleCost(Type *EltTy, int ReplicationFactor,
5747 int VF, const APInt &DemandedDstElts,
5749 const unsigned EltTyBits = DL.getTypeSizeInBits(EltTy);
5750 // We don't differentiate element types here, only element bit width.
5751 EltTy = IntegerType::getIntNTy(EltTy->getContext(), EltTyBits);
5752
5753 auto bailout = [&]() {
5754 return BaseT::getReplicationShuffleCost(EltTy, ReplicationFactor, VF,
5755 DemandedDstElts, CostKind);
5756 };
5757
5758 // For now, only deal with AVX512 cases.
5759 if (!ST->hasAVX512())
5760 return bailout();
5761
5762 // Do we have a native shuffle for this element type, or should we promote?
5763 unsigned PromEltTyBits = EltTyBits;
5764 switch (EltTyBits) {
5765 case 32:
5766 case 64:
5767 break; // AVX512F.
5768 case 16:
5769 if (!ST->hasBWI())
5770 PromEltTyBits = 32; // promote to i32, AVX512F.
5771 break; // AVX512BW
5772 case 8:
5773 if (!ST->hasVBMI())
5774 PromEltTyBits = 32; // promote to i32, AVX512F.
5775 break; // AVX512VBMI
5776 case 1:
5777 // There is no support for shuffling i1 elements. We *must* promote.
5778 if (ST->hasBWI()) {
5779 if (ST->hasVBMI())
5780 PromEltTyBits = 8; // promote to i8, AVX512VBMI.
5781 else
5782 PromEltTyBits = 16; // promote to i16, AVX512BW.
5783 break;
5784 }
5785 PromEltTyBits = 32; // promote to i32, AVX512F.
5786 break;
5787 default:
5788 return bailout();
5789 }
5790 auto *PromEltTy = IntegerType::getIntNTy(EltTy->getContext(), PromEltTyBits);
5791
5792 auto *SrcVecTy = FixedVectorType::get(EltTy, VF);
5793 auto *PromSrcVecTy = FixedVectorType::get(PromEltTy, VF);
5794
5795 int NumDstElements = VF * ReplicationFactor;
5796 auto *PromDstVecTy = FixedVectorType::get(PromEltTy, NumDstElements);
5797 auto *DstVecTy = FixedVectorType::get(EltTy, NumDstElements);
5798
5799 // Legalize the types.
5800 MVT LegalSrcVecTy = getTypeLegalizationCost(SrcVecTy).second;
5801 MVT LegalPromSrcVecTy = getTypeLegalizationCost(PromSrcVecTy).second;
5802 MVT LegalPromDstVecTy = getTypeLegalizationCost(PromDstVecTy).second;
5803 MVT LegalDstVecTy = getTypeLegalizationCost(DstVecTy).second;
5804 // They should have legalized into vector types.
5805 if (!LegalSrcVecTy.isVector() || !LegalPromSrcVecTy.isVector() ||
5806 !LegalPromDstVecTy.isVector() || !LegalDstVecTy.isVector())
5807 return bailout();
5808
5809 if (PromEltTyBits != EltTyBits) {
5810 // If we have to perform the shuffle with wider elt type than our data type,
5811 // then we will first need to anyext (we don't care about the new bits)
5812 // the source elements, and then truncate Dst elements.
5813 InstructionCost PromotionCost;
5814 PromotionCost += getCastInstrCost(
5815 Instruction::SExt, /*Dst=*/PromSrcVecTy, /*Src=*/SrcVecTy,
5817 PromotionCost +=
5818 getCastInstrCost(Instruction::Trunc, /*Dst=*/DstVecTy,
5819 /*Src=*/PromDstVecTy,
5821 return PromotionCost + getReplicationShuffleCost(PromEltTy,
5822 ReplicationFactor, VF,
5823 DemandedDstElts, CostKind);
5824 }
5825
5826 assert(LegalSrcVecTy.getScalarSizeInBits() == EltTyBits &&
5827 LegalSrcVecTy.getScalarType() == LegalDstVecTy.getScalarType() &&
5828 "We expect that the legalization doesn't affect the element width, "
5829 "doesn't coalesce/split elements.");
5830
5831 unsigned NumEltsPerDstVec = LegalDstVecTy.getVectorNumElements();
5832 unsigned NumDstVectors =
5833 divideCeil(DstVecTy->getNumElements(), NumEltsPerDstVec);
5834
5835 auto *SingleDstVecTy = FixedVectorType::get(EltTy, NumEltsPerDstVec);
5836
5837 // Not all the produced Dst elements may be demanded. In our case,
5838 // given that a single Dst vector is formed by a single shuffle,
5839 // if all elements that will form a single Dst vector aren't demanded,
5840 // then we won't need to do that shuffle, so adjust the cost accordingly.
5841 APInt DemandedDstVectors = APIntOps::ScaleBitMask(
5842 DemandedDstElts.zext(NumDstVectors * NumEltsPerDstVec), NumDstVectors);
5843 unsigned NumDstVectorsDemanded = DemandedDstVectors.popcount();
5844
5845 InstructionCost SingleShuffleCost =
5846 getShuffleCost(TTI::SK_PermuteSingleSrc, SingleDstVecTy, SingleDstVecTy,
5847 CostKind, /*Mask=*/{},
5848 /*Index=*/0, /*SubTp=*/nullptr);
5849 return NumDstVectorsDemanded * SingleShuffleCost;
5850}
5851
5853 Align Alignment,
5854 unsigned AddressSpace,
5856 TTI::OperandValueInfo OpInfo,
5857 const Instruction *I) const {
5858 // FIXME: Load latency isn't handled here
5859 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
5860 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5861 CostKind, OpInfo, I);
5862
5863 // TODO: Handle other cost kinds.
5865 if (auto *SI = dyn_cast_or_null<StoreInst>(I)) {
5866 // Store instruction with index and scale costs 2 Uops.
5867 // Check the preceding GEP to identify non-const indices.
5868 if (auto *GEP = dyn_cast<GetElementPtrInst>(SI->getPointerOperand())) {
5869 if (!all_of(GEP->indices(), [](Value *V) { return isa<Constant>(V); }))
5870 return TTI::TCC_Basic * 2;
5871 }
5872 }
5873 return TTI::TCC_Basic;
5874 }
5875
5876 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5877 "Invalid Opcode");
5878 // Type legalization can't handle structs
5879 if (TLI->getValueType(DL, Src, true) == MVT::Other)
5880 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5881 CostKind, OpInfo, I);
5882
5883 // Legalize the type.
5884 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
5885
5886 auto *VTy = dyn_cast<FixedVectorType>(Src);
5887
5889
5890 // Add a cost for constant load to vector.
5891 if (Opcode == Instruction::Store && OpInfo.isConstant())
5892 Cost += getMemoryOpCost(Instruction::Load, Src, DL.getABITypeAlign(Src),
5893 /*AddressSpace=*/0, CostKind, OpInfo);
5894
5895 // Handle the simple case of non-vectors.
5896 // NOTE: this assumes that legalization never creates vector from scalars!
5897 if (!VTy || !LT.second.isVector()) {
5898 // Each load/store unit costs 1.
5899 return (LT.second.isFloatingPoint() ? Cost : 0) + LT.first * 1;
5900 }
5901
5902 bool IsLoad = Opcode == Instruction::Load;
5903
5904 Type *EltTy = VTy->getElementType();
5905
5906 const int EltTyBits = DL.getTypeSizeInBits(EltTy);
5907
5908 // Source of truth: how many elements were there in the original IR vector?
5909 const unsigned SrcNumElt = VTy->getNumElements();
5910
5911 // How far have we gotten?
5912 int NumEltRemaining = SrcNumElt;
5913 // Note that we intentionally capture by-reference, NumEltRemaining changes.
5914 auto NumEltDone = [&]() { return SrcNumElt - NumEltRemaining; };
5915
5916 const int MaxLegalOpSizeBytes = divideCeil(LT.second.getSizeInBits(), 8);
5917
5918 // Note that even if we can store 64 bits of an XMM, we still operate on XMM.
5919 const unsigned XMMBits = 128;
5920 if (XMMBits % EltTyBits != 0)
5921 // Vector size must be a multiple of the element size. I.e. no padding.
5922 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5923 CostKind, OpInfo, I);
5924 const int NumEltPerXMM = XMMBits / EltTyBits;
5925
5926 auto *XMMVecTy = FixedVectorType::get(EltTy, NumEltPerXMM);
5927
5928 for (int CurrOpSizeBytes = MaxLegalOpSizeBytes, SubVecEltsLeft = 0;
5929 NumEltRemaining > 0; CurrOpSizeBytes /= 2) {
5930 // How many elements would a single op deal with at once?
5931 if ((8 * CurrOpSizeBytes) % EltTyBits != 0)
5932 // Vector size must be a multiple of the element size. I.e. no padding.
5933 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
5934 CostKind, OpInfo, I);
5935 int CurrNumEltPerOp = (8 * CurrOpSizeBytes) / EltTyBits;
5936
5937 assert(CurrOpSizeBytes > 0 && CurrNumEltPerOp > 0 && "How'd we get here?");
5938 assert((((NumEltRemaining * EltTyBits) < (2 * 8 * CurrOpSizeBytes)) ||
5939 (CurrOpSizeBytes == MaxLegalOpSizeBytes)) &&
5940 "Unless we haven't halved the op size yet, "
5941 "we have less than two op's sized units of work left.");
5942
5943 auto *CurrVecTy = CurrNumEltPerOp > NumEltPerXMM
5944 ? FixedVectorType::get(EltTy, CurrNumEltPerOp)
5945 : XMMVecTy;
5946
5947 assert(CurrVecTy->getNumElements() % CurrNumEltPerOp == 0 &&
5948 "After halving sizes, the vector elt count is no longer a multiple "
5949 "of number of elements per operation?");
5950 auto *CoalescedVecTy =
5951 CurrNumEltPerOp == 1
5952 ? CurrVecTy
5954 IntegerType::get(Src->getContext(),
5955 EltTyBits * CurrNumEltPerOp),
5956 CurrVecTy->getNumElements() / CurrNumEltPerOp);
5957 assert(DL.getTypeSizeInBits(CoalescedVecTy) ==
5958 DL.getTypeSizeInBits(CurrVecTy) &&
5959 "coalesciing elements doesn't change vector width.");
5960
5961 while (NumEltRemaining > 0) {
5962 assert(SubVecEltsLeft >= 0 && "Subreg element count overconsumtion?");
5963
5964 // Can we use this vector size, as per the remaining element count?
5965 // Iff the vector is naturally aligned, we can do a wide load regardless.
5966 if (NumEltRemaining < CurrNumEltPerOp &&
5967 (!IsLoad || Alignment < CurrOpSizeBytes) && CurrOpSizeBytes != 1)
5968 break; // Try smalled vector size.
5969
5970 // This isn't exactly right. We're using slow unaligned 32-byte accesses
5971 // as a proxy for a double-pumped AVX memory interface such as on
5972 // Sandybridge.
5973 // Sub-32-bit loads/stores will be slower either with PINSR*/PEXTR* or
5974 // will be scalarized.
5975 //
5976 // For a vector load, each non-0th 1/2/4-byte in-lane remainder chunk is
5977 // materialized by a *single* folded PINSR*(mem) that both loads and
5978 // inserts the lane (1B->PINSRB, 2B->PINSRW, 4B->PINSRD; the byte and
5979 // dword folds need SSE4.1, the word fold only SSE2).
5980 // When that fold is available the chunk is one instruction, so it must be
5981 // priced once here (as a plain load) and the separate lane-insert charge
5982 // below must be skipped - otherwise the folded insert is double-counted.
5983 // Stores (the symmetric PEXTR*(mem) fold) are left unchanged here.
5984 bool Is0thSubVec = (NumEltDone() % LT.second.getVectorNumElements()) == 0;
5985 bool FoldedInLaneInsert = IsLoad && !Is0thSubVec &&
5986 ((CurrOpSizeBytes == 1 && ST->hasSSE41()) ||
5987 (CurrOpSizeBytes == 2 && ST->hasSSE2()) ||
5988 (CurrOpSizeBytes == 4 && ST->hasSSE41()));
5989 if (CurrOpSizeBytes == 32 && ST->isUnalignedMem32Slow())
5990 Cost += 2;
5991 else if (CurrOpSizeBytes < 4 && !FoldedInLaneInsert)
5992 Cost += 2;
5993 else
5994 Cost += 1;
5995
5996 // If we're loading a uniform value, then we don't need to split the load,
5997 // loading just a single (widest) vector can be reused by all splits.
5998 if (IsLoad && OpInfo.isUniform())
5999 return Cost;
6000
6001 // If we have fully processed the previous reg, we need to replenish it.
6002 if (SubVecEltsLeft == 0) {
6003 SubVecEltsLeft += CurrVecTy->getNumElements();
6004 // And that's free only for the 0'th subvector of a legalized vector.
6005 if (!Is0thSubVec)
6006 Cost +=
6009 VTy, VTy, CostKind, {}, NumEltDone(), CurrVecTy);
6010 }
6011
6012 // While we can directly load/store ZMM, YMM, and 64-bit halves of XMM,
6013 // for smaller widths (32/16/8) we have to insert/extract them separately.
6014 // Again, it's free for the 0'th subreg (if op is 32/64 bit wide,
6015 // but let's pretend that it is also true for 16/8 bit wide ops...)
6016 if (CurrOpSizeBytes <= 32 / 8 && !Is0thSubVec && !FoldedInLaneInsert) {
6017 int NumEltDoneInCurrXMM = NumEltDone() % NumEltPerXMM;
6018 assert(NumEltDoneInCurrXMM % CurrNumEltPerOp == 0 && "");
6019 int CoalescedVecEltIdx = NumEltDoneInCurrXMM / CurrNumEltPerOp;
6020 APInt DemandedElts =
6021 APInt::getBitsSet(CoalescedVecTy->getNumElements(),
6022 CoalescedVecEltIdx, CoalescedVecEltIdx + 1);
6023 assert(DemandedElts.popcount() == 1 && "Inserting single value");
6024 Cost += getScalarizationOverhead(CoalescedVecTy, DemandedElts, IsLoad,
6025 !IsLoad, CostKind);
6026 }
6027
6028 SubVecEltsLeft -= CurrNumEltPerOp;
6029 NumEltRemaining -= CurrNumEltPerOp;
6030 Alignment = commonAlignment(Alignment, CurrOpSizeBytes);
6031 }
6032 }
6033
6034 assert(NumEltRemaining <= 0 && "Should have processed all the elements.");
6035
6036 return Cost;
6037}
6038
6042 switch (MICA.getID()) {
6043 case Intrinsic::masked_scatter:
6044 case Intrinsic::masked_gather:
6045 return getGatherScatterOpCost(MICA, CostKind);
6046 case Intrinsic::masked_load:
6047 case Intrinsic::masked_store:
6048 return getMaskedMemoryOpCost(MICA, CostKind);
6049 case Intrinsic::masked_compressstore:
6050 // Fallback to scalarization for slow compressstore targets (e.g. znver4).
6051 if (ST->isVecCompressStoreSlow())
6053 break;
6054 }
6055
6056 static const CostKindTblEntry AVX512VBMI2CostTable[] = {
6057 { Intrinsic::masked_expandload, MVT::v16i8, { 2, 7, 1, 3 } },
6058 { Intrinsic::masked_expandload, MVT::v32i8, { 2, 8, 1, 3 } },
6059 { Intrinsic::masked_expandload, MVT::v64i8, { 2, 9, 1, 3 } },
6060
6061 { Intrinsic::masked_expandload, MVT::v8i16, { 2, 7, 1, 3 } },
6062 { Intrinsic::masked_expandload, MVT::v16i16, { 2, 8, 1, 3 } },
6063 { Intrinsic::masked_expandload, MVT::v32i16, { 2, 9, 1, 3 } },
6064
6065 { Intrinsic::masked_compressstore, MVT::v16i8, { 3,10, 1, 7 } },
6066 { Intrinsic::masked_compressstore, MVT::v32i8, { 3,11, 1, 7 } },
6067 { Intrinsic::masked_compressstore, MVT::v64i8, { 3,12, 1, 8 } },
6068
6069 { Intrinsic::masked_compressstore, MVT::v8i16, { 3,10, 1, 7 } },
6070 { Intrinsic::masked_compressstore, MVT::v16i16, { 3,11, 1, 7 } },
6071 { Intrinsic::masked_compressstore, MVT::v32i16, { 3,12, 1, 8 } },
6072 };
6073
6074 static const CostKindTblEntry AVX512CostTable[] = {
6075 { Intrinsic::masked_expandload, MVT::v4i32, { 2, 7, 1, 3 } },
6076 { Intrinsic::masked_expandload, MVT::v4f32, { 2, 7, 1, 3 } },
6077 { Intrinsic::masked_expandload, MVT::v8i32, { 2, 8, 1, 3 } },
6078 { Intrinsic::masked_expandload, MVT::v8f32, { 2, 8, 1, 3 } },
6079 { Intrinsic::masked_expandload, MVT::v16i32, { 2, 9, 1, 3 } },
6080 { Intrinsic::masked_expandload, MVT::v16f32, { 2, 9, 1, 3 } },
6081
6082 { Intrinsic::masked_expandload, MVT::v2i64, { 2, 7, 1, 3 } },
6083 { Intrinsic::masked_expandload, MVT::v2f64, { 2, 7, 1, 3 } },
6084 { Intrinsic::masked_expandload, MVT::v4i64, { 2, 8, 1, 3 } },
6085 { Intrinsic::masked_expandload, MVT::v4f64, { 2, 8, 1, 3 } },
6086 { Intrinsic::masked_expandload, MVT::v8i64, { 2, 9, 1, 3 } },
6087 { Intrinsic::masked_expandload, MVT::v8f64, { 2, 9, 1, 3 } },
6088
6089 { Intrinsic::masked_compressstore, MVT::v4i32, { 3,10, 1, 7 } },
6090 { Intrinsic::masked_compressstore, MVT::v4f32, { 3,10, 1, 7 } },
6091 { Intrinsic::masked_compressstore, MVT::v8i32, { 3,11, 1, 7 } },
6092 { Intrinsic::masked_compressstore, MVT::v8f32, { 3,11, 1, 7 } },
6093 { Intrinsic::masked_compressstore, MVT::v16i32, { 3,12, 1, 8 } },
6094 { Intrinsic::masked_compressstore, MVT::v16f32, { 3,12, 1, 8 } },
6095
6096 { Intrinsic::masked_compressstore, MVT::v2i64, { 3,10, 1, 7 } },
6097 { Intrinsic::masked_compressstore, MVT::v2f64, { 3,10, 1, 7 } },
6098 { Intrinsic::masked_compressstore, MVT::v4i64, { 3,11, 1, 7 } },
6099 { Intrinsic::masked_compressstore, MVT::v4f64, { 3,11, 1, 7 } },
6100 { Intrinsic::masked_compressstore, MVT::v8i64, { 3,12, 1, 8 } },
6101 { Intrinsic::masked_compressstore, MVT::v8f64, { 3,12, 1, 8 } },
6102 };
6103
6104 std::pair<InstructionCost, MVT> LT =
6106
6107 if (ST->hasVBMI2())
6108 if (const auto *Entry =
6109 CostTableLookup(AVX512VBMI2CostTable, MICA.getID(), LT.second))
6110 if (auto KindCost = Entry->Cost[CostKind])
6111 return LT.first * *KindCost;
6112
6113 if (ST->hasAVX512())
6114 if (const auto *Entry =
6115 CostTableLookup(AVX512CostTable, MICA.getID(), LT.second))
6116 if (auto KindCost = Entry->Cost[CostKind])
6117 return LT.first * *KindCost;
6118
6120}
6121
6125 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
6126 : Instruction::Store;
6127 Type *SrcTy = MICA.getDataType();
6128 Align Alignment = MICA.getAlignment();
6129 unsigned AddressSpace = MICA.getAddressSpace();
6130
6131 bool IsLoad = (Instruction::Load == Opcode);
6132 bool IsStore = (Instruction::Store == Opcode);
6133
6134 auto *SrcVTy = dyn_cast<FixedVectorType>(SrcTy);
6135 if (!SrcVTy)
6136 // To calculate scalar take the regular cost, without mask
6137 return getMemoryOpCost(Opcode, SrcTy, Alignment, AddressSpace, CostKind);
6138
6139 unsigned NumElem = SrcVTy->getNumElements();
6140 auto *MaskTy =
6141 FixedVectorType::get(Type::getInt8Ty(SrcVTy->getContext()), NumElem);
6142 if ((IsLoad && !isLegalMaskedLoad(SrcVTy, Alignment, AddressSpace)) ||
6143 (IsStore && !isLegalMaskedStore(SrcVTy, Alignment, AddressSpace))) {
6144 // Scalarization
6145 APInt DemandedElts = APInt::getAllOnes(NumElem);
6147 MaskTy, DemandedElts, /*Insert*/ false, /*Extract*/ true, CostKind);
6148 InstructionCost ScalarCompareCost = getCmpSelInstrCost(
6149 Instruction::ICmp, Type::getInt8Ty(SrcVTy->getContext()), nullptr,
6151 InstructionCost BranchCost = getCFInstrCost(Instruction::CondBr, CostKind);
6152 InstructionCost MaskCmpCost = NumElem * (BranchCost + ScalarCompareCost);
6154 SrcVTy, DemandedElts, IsLoad, IsStore, CostKind);
6155 InstructionCost MemopCost =
6156 NumElem * BaseT::getMemoryOpCost(Opcode, SrcVTy->getScalarType(),
6157 Alignment, AddressSpace, CostKind);
6158 return MemopCost + ValueSplitCost + MaskSplitCost + MaskCmpCost;
6159 }
6160
6161 // Legalize the type.
6162 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcVTy);
6163 auto VT = TLI->getValueType(DL, SrcVTy);
6165 MVT Ty = LT.second;
6166 if (Ty == MVT::i16 || Ty == MVT::i32 || Ty == MVT::i64)
6167 // APX masked load/store for scalar is cheap.
6168 return Cost + LT.first;
6169
6170 if (VT.isSimple() && Ty != VT.getSimpleVT() &&
6171 LT.second.getVectorNumElements() == NumElem)
6172 // Promotion requires extend/truncate for data and a shuffle for mask.
6173 Cost += getShuffleCost(TTI::SK_PermuteTwoSrc, SrcVTy, SrcVTy, CostKind, {},
6174 0, nullptr) +
6175 getShuffleCost(TTI::SK_PermuteTwoSrc, MaskTy, MaskTy, CostKind, {},
6176 0, nullptr);
6177
6178 else if (LT.first * Ty.getVectorNumElements() > NumElem) {
6179 auto *NewMaskTy = FixedVectorType::get(MaskTy->getElementType(),
6180 (unsigned)LT.first.getValue() *
6181 Ty.getVectorNumElements());
6182 // Expanding requires fill mask with zeroes
6183 Cost += getShuffleCost(TTI::SK_InsertSubvector, NewMaskTy, NewMaskTy,
6184 CostKind, {}, 0, MaskTy);
6185 }
6186
6187 // Predicate fanout (see getPredicateFanoutCost): only a variable mask needs
6188 // a per-part sub-mask; a constant mask is folded by codegen and pays nothing.
6189 InstructionCost MaskExpCost = 0;
6190 if (MICA.getVariableMask()) {
6191 auto *MaskVecTy =
6192 FixedVectorType::get(Type::getInt1Ty(SrcVTy->getContext()), NumElem);
6193 MaskExpCost = getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
6194 }
6195
6196 // Pre-AVX512 - each maskmov load costs 2 + store costs ~8.
6197 if (!ST->hasAVX512())
6198 return Cost + LT.first * (IsLoad ? 2 : 8) + MaskExpCost;
6199
6200 // AVX-512 masked load/store is cheaper.
6201 return Cost + LT.first + MaskExpCost;
6202}
6203
6205 ArrayRef<const Value *> Ptrs, const Value *Base,
6206 const TTI::PointersChainInfo &Info, Type *AccessTy,
6207 const TTI::TargetCostKind CostKind) const {
6208 if (Info.isSameBase() && Info.isKnownStride()) {
6209 // If all the pointers have known stride all the differences are translated
6210 // into constants. X86 memory addressing allows encoding it into
6211 // displacement. So we just need to take the base GEP cost.
6212 if (const auto *BaseGEP = dyn_cast<GetElementPtrInst>(Base)) {
6213 SmallVector<const Value *> Indices(BaseGEP->indices());
6214 return getGEPCost(BaseGEP->getSourceElementType(),
6215 BaseGEP->getPointerOperand(), Indices, CostKind,
6216 nullptr);
6217 }
6218 return TTI::TCC_Free;
6219 }
6220 return BaseT::getPointersChainCost(Ptrs, Base, Info, AccessTy, CostKind);
6221}
6222
6225 const SCEV *Ptr,
6227 // Address computations in vectorized code with non-consecutive addresses will
6228 // likely result in more instructions compared to scalar code where the
6229 // computation can more often be merged into the index mode. The resulting
6230 // extra micro-ops can significantly decrease throughput.
6231 const unsigned NumVectorInstToHideOverhead = 10;
6232
6233 // Cost modeling of Strided Access Computation is hidden by the indexing
6234 // modes of X86 regardless of the stride value. We dont believe that there
6235 // is a difference between constant strided access in gerenal and constant
6236 // strided value which is less than or equal to 64.
6237 // Even in the case of (loop invariant) stride whose value is not known at
6238 // compile time, the address computation will not incur more than one extra
6239 // ADD instruction.
6240 if (PtrTy->isVectorTy() && SE) {
6242 return 1;
6243 if (!ST->hasAVX2()) {
6244 // TODO: AVX2 is the current cut-off because we don't have correct
6245 // interleaving costs for prior ISA's.
6246 if (!BaseT::isStridedAccess(Ptr))
6247 return NumVectorInstToHideOverhead;
6248 }
6249 }
6250
6251 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
6252}
6253
6255 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
6257 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
6258 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
6259 auto ExpandCost = [&]() {
6260 return BaseT::getPartialReductionCost(Opcode, InputTypeA, InputTypeB,
6261 AccumType, VF, OpAExtend, OpBExtend,
6262 BinOp, CostKind, FMF);
6263 };
6264
6265 // The dot product instructions multiply-accumulate i8 x i8 -> i32,
6266 // i16 x i16 -> i32, bf16 x bf16 -> f32 or f16 x f16 -> f32. Partial
6267 // reductions may also multiply inputs extended from different types, which
6268 // they can't handle.
6269 if (VF.isScalable() || !BinOp || OpAExtend == TTI::PR_None ||
6270 OpBExtend == TTI::PR_None || InputTypeA != InputTypeB)
6271 return ExpandCost();
6272
6273 unsigned Opc;
6274 if (Opcode == Instruction::Add && *BinOp == Instruction::Mul &&
6275 AccumType->isIntegerTy(32) &&
6276 (InputTypeA->isIntegerTy(8) || InputTypeA->isIntegerTy(16))) {
6277 if (OpAExtend != OpBExtend)
6279 else if (OpAExtend == TTI::PR_SignExtend)
6281 else
6283 } else if (Opcode == Instruction::FAdd && *BinOp == Instruction::FMul &&
6284 AccumType->isFloatTy() &&
6285 (InputTypeA->isBFloatTy() || InputTypeA->isHalfTy())) {
6286 // VDPBF16PS and VDPPHPS, like the expansion, reassociate the additions and
6287 // fuse the multiplications.
6288 if (!FMF || !FMF->allowReassoc() || !FMF->allowContract())
6291 } else {
6292 return ExpandCost();
6293 }
6294
6295 unsigned Ratio =
6296 AccumType->getScalarSizeInBits() / InputTypeA->getScalarSizeInBits();
6297 if (!VF.isKnownMultipleOf(Ratio))
6298 return ExpandCost();
6299
6300 // One dot product per legal accumulator vector. Accumulators narrower than
6301 // a legal vector are widened by expanding the partial reduction instead.
6302 auto *AccVecTy = VectorType::get(AccumType, VF.divideCoefficientBy(Ratio));
6303 auto *InputVecTy = VectorType::get(InputTypeA, VF);
6304 std::pair<InstructionCost, MVT> AccLT = getTypeLegalizationCost(AccVecTy);
6305 std::pair<InstructionCost, MVT> InputLT = getTypeLegalizationCost(InputVecTy);
6306 if (AccLT.second.getFixedSizeInBits() >
6307 AccVecTy->getPrimitiveSizeInBits().getFixedValue() ||
6308 !TLI->isPartialReduceMLALegalOrCustom(Opc, AccLT.second, InputLT.second))
6309 return ExpandCost();
6310
6311 return AccLT.first;
6312}
6313
6316 std::optional<FastMathFlags> FMF,
6319 return BaseT::getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
6320
6321 // We use llvm-mca across all supported CPUs to measure the logic cost stats.
6322 // We use the Intel Architecture Code Analyzer(IACA) to measure the throughput
6323 // and make it as the cost. TODO: Update old IACA numbers to llvm-mca.
6324
6325 static const CostKindTblEntry SLMCostTbl[] = {
6326 { ISD::FADD, MVT::v2f64, {3, 3, 3, 3} },
6327 { ISD::ADD, MVT::v2i64, {5, 5, 5, 5} },
6328 };
6329
6330 static const CostKindTblEntry SSE2CostTbl[] = {
6331 { ISD::FADD, MVT::v2f64, {2, 2, 2, 2} },
6332 { ISD::FADD, MVT::v2f32, {2, 2, 2, 2} },
6333 { ISD::FADD, MVT::v4f32, {4, 4, 4, 4} },
6334 { ISD::ADD, MVT::v2i64, {2, 2, 2, 2} }, // The data reported by the IACA tool is "1.6".
6335 { ISD::ADD, MVT::v2i32, {2, 2, 2, 2} }, // FIXME: chosen to be less than v4i32
6336 { ISD::ADD, MVT::v4i32, {3, 3, 3, 3} }, // The data reported by the IACA tool is "3.3".
6337 { ISD::ADD, MVT::v2i16, {2, 2, 2, 2} }, // The data reported by the IACA tool is "4.3".
6338 { ISD::ADD, MVT::v4i16, {3, 3, 3, 3} }, // The data reported by the IACA tool is "4.3".
6339 { ISD::ADD, MVT::v8i16, {4, 4, 4, 4} }, // The data reported by the IACA tool is "4.3".
6340 { ISD::ADD, MVT::v2i8, {2, 2, 2, 2} },
6341 { ISD::ADD, MVT::v4i8, {2, 2, 2, 2} },
6342 { ISD::ADD, MVT::v8i8, {2, 2, 2, 2} },
6343 { ISD::ADD, MVT::v16i8, {3, 3, 3, 3} },
6344
6345 { ISD::AND, MVT::v2i64, {2, 2, 3, 3} },
6346 { ISD::AND, MVT::v4i32, {3, 4, 5, 5} },
6347 { ISD::AND, MVT::v8i16, {4, 7, 8, 8} },
6348 { ISD::AND, MVT::v16i8, {6,10,11,11} },
6349 { ISD::OR, MVT::v2i64, {2, 2, 3, 3} },
6350 { ISD::OR, MVT::v4i32, {3, 4, 5, 5} },
6351 { ISD::OR, MVT::v8i16, {4, 7, 8, 8} },
6352 { ISD::OR, MVT::v16i8, {6,10,11,11} },
6353 { ISD::XOR, MVT::v2i64, {2, 2, 3, 3} },
6354 { ISD::XOR, MVT::v4i32, {3, 4, 5, 5} },
6355 { ISD::XOR, MVT::v8i16, {4, 7, 8, 8} },
6356 { ISD::XOR, MVT::v16i8, {6,10,11,11} },
6357 };
6358
6359 static const CostKindTblEntry AVX1CostTbl[] = {
6360 { ISD::FADD, MVT::v4f64, {3, 3, 3, 3} },
6361 { ISD::FADD, MVT::v4f32, {3, 3, 3, 3} },
6362 { ISD::FADD, MVT::v8f32, {4, 4, 4, 4} },
6363 { ISD::ADD, MVT::v2i64, {1, 1, 1, 1} }, // The data reported by the IACA tool is "1.5".
6364 { ISD::ADD, MVT::v4i64, {3, 3, 3, 3} },
6365 { ISD::ADD, MVT::v8i32, {5, 5, 5, 5} },
6366 { ISD::ADD, MVT::v16i16, {5, 5, 5, 5} },
6367 { ISD::ADD, MVT::v32i8, {4, 4, 4, 4} },
6368
6369 { ISD::AND, MVT::v4i64, {3, 7, 5, 5} },
6370 { ISD::AND, MVT::v8i32, {4, 9, 7, 7} },
6371 { ISD::AND, MVT::v16i16, {5,11, 9, 9} },
6372 { ISD::AND, MVT::v8i16, {4, 7, 7, 7} },
6373 { ISD::AND, MVT::v32i8, {6,13,11,11} },
6374 { ISD::AND, MVT::v16i8, {5,10, 9, 9} },
6375 { ISD::OR, MVT::v4i64, {3, 7, 5, 5} },
6376 { ISD::OR, MVT::v8i32, {4, 9, 7, 7} },
6377 { ISD::OR, MVT::v16i16, {5,11, 9, 9} },
6378 { ISD::OR, MVT::v8i16, {4, 7, 7, 7} },
6379 { ISD::OR, MVT::v32i8, {6,13,11,11} },
6380 { ISD::OR, MVT::v16i8, {5,10, 9, 9} },
6381 { ISD::XOR, MVT::v4i64, {3, 7, 5, 5} },
6382 { ISD::XOR, MVT::v8i32, {4, 9, 7, 7} },
6383 { ISD::XOR, MVT::v16i16, {5,11, 9, 9} },
6384 { ISD::XOR, MVT::v8i16, {4, 7, 7, 7} },
6385 { ISD::XOR, MVT::v32i8, {6,13,11,11} },
6386 { ISD::XOR, MVT::v16i8, {5,10, 9, 9} },
6387 };
6388
6389 static const CostKindTblEntry AVX2CostTbl[] = {
6390 { ISD::AND, MVT::v4i64, {2, 7, 5, 5} },
6391 { ISD::AND, MVT::v2i64, {1, 2, 3, 3} },
6392 { ISD::AND, MVT::v8i32, {3, 9, 7, 7} },
6393 { ISD::AND, MVT::v4i32, {2, 4, 5, 5} },
6394 { ISD::AND, MVT::v16i16, {3,11, 9, 9} },
6395 { ISD::AND, MVT::v8i16, {2, 6, 7, 7} },
6396 { ISD::AND, MVT::v32i8, {3,13,11,11} },
6397 { ISD::AND, MVT::v16i8, {3, 8, 9, 9} },
6398 { ISD::OR, MVT::v4i64, {2, 7, 5, 5} },
6399 { ISD::OR, MVT::v2i64, {1, 2, 3, 3} },
6400 { ISD::OR, MVT::v8i32, {3, 9, 7, 7} },
6401 { ISD::OR, MVT::v4i32, {2, 4, 5, 5} },
6402 { ISD::OR, MVT::v16i16, {3,11, 9, 9} },
6403 { ISD::OR, MVT::v8i16, {2, 6, 7, 7} },
6404 { ISD::OR, MVT::v32i8, {3,13,11,11} },
6405 { ISD::OR, MVT::v16i8, {3, 8, 9, 9} },
6406 { ISD::XOR, MVT::v4i64, {2, 7, 5, 5} },
6407 { ISD::XOR, MVT::v2i64, {1, 2, 3, 3} },
6408 { ISD::XOR, MVT::v8i32, {3, 9, 7, 7} },
6409 { ISD::XOR, MVT::v4i32, {2, 4, 5, 5} },
6410 { ISD::XOR, MVT::v16i16, {3,11, 9, 9} },
6411 { ISD::XOR, MVT::v8i16, {2, 6, 7, 7} },
6412 { ISD::XOR, MVT::v32i8, {3,13,11,11} },
6413 { ISD::XOR, MVT::v16i8, {3, 8, 9, 9} },
6414 };
6415
6416 static const CostKindTblEntry AVX512FCostTbl[] = {
6417 { ISD::FADD, MVT::v8f64, {4, 4, 4, 4} },
6418 { ISD::FADD, MVT::v16f32, {5, 5, 5, 5} },
6419 { ISD::ADD, MVT::v8i64, {4, 4, 4, 4} },
6420 { ISD::ADD, MVT::v16i32, {6, 6, 6, 6} },
6421
6422 { ISD::AND, MVT::v8i64, {3,10, 7, 7} },
6423 { ISD::AND, MVT::v16i32, {4,12, 9, 9} },
6424 { ISD::AND, MVT::v32i16, {4,14,11,11} },
6425 { ISD::AND, MVT::v64i8, {4,16,13,13} },
6426 { ISD::AND, MVT::v16i8, {2, 8, 9, 9} },
6427 { ISD::OR, MVT::v8i64, {3,10, 7, 7} },
6428 { ISD::OR, MVT::v16i32, {4,12, 9, 9} },
6429 { ISD::OR, MVT::v32i16, {4,14,11,11} },
6430 { ISD::OR, MVT::v64i8, {4,16,13,13} },
6431 { ISD::OR, MVT::v16i8, {2, 8, 9, 9} },
6432 { ISD::XOR, MVT::v8i64, {3,10, 7, 7} },
6433 { ISD::XOR, MVT::v16i32, {4,12, 9, 9} },
6434 { ISD::XOR, MVT::v32i16, {4,14,11,11} },
6435 { ISD::XOR, MVT::v64i8, {4,16,13,13} },
6436 { ISD::XOR, MVT::v16i8, {2, 8, 9, 9} },
6437 };
6438
6439 static const CostKindTblEntry AVX512BWCostTbl[] = {
6440 { ISD::ADD, MVT::v32i16, {7, 7, 7, 7} },
6441 { ISD::ADD, MVT::v64i8, {4, 4, 4, 4} },
6442 };
6443
6444 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6445 assert(ISD && "Invalid opcode");
6446
6447 // Before legalizing the type, give a chance to look up illegal narrow types
6448 // in the table.
6449 // FIXME: Is there a better way to do this?
6450 EVT VT = TLI->getValueType(DL, ValTy);
6451 if (VT.isSimple()) {
6452 MVT MTy = VT.getSimpleVT();
6453 if (ST->useSLMArithCosts())
6454 if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
6455 if (auto KindCost = Entry->Cost[CostKind])
6456 return *KindCost;
6457
6458 if (ST->hasBWI())
6459 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
6460 if (auto KindCost = Entry->Cost[CostKind])
6461 return *KindCost;
6462
6463 if (ST->hasAVX512())
6464 if (const auto *Entry = CostTableLookup(AVX512FCostTbl, ISD, MTy))
6465 if (auto KindCost = Entry->Cost[CostKind])
6466 return *KindCost;
6467
6468 if (ST->hasAVX2())
6469 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
6470 if (auto KindCost = Entry->Cost[CostKind])
6471 return *KindCost;
6472
6473 if (ST->hasAVX())
6474 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
6475 if (auto KindCost = Entry->Cost[CostKind])
6476 return *KindCost;
6477
6478 if (ST->hasSSE2())
6479 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
6480 if (auto KindCost = Entry->Cost[CostKind])
6481 return *KindCost;
6482 }
6483
6484 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
6485
6486 MVT MTy = LT.second;
6487
6488 auto *ValVTy = cast<FixedVectorType>(ValTy);
6489
6490 InstructionCost ArithmeticCost = 0;
6491 if (LT.first != 1 && MTy.isVector() &&
6492 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6493 // Type needs to be split. We need LT.first - 1 arithmetic ops.
6494 auto *SingleOpTy = FixedVectorType::get(ValVTy->getElementType(),
6495 MTy.getVectorNumElements());
6496 ArithmeticCost = getArithmeticInstrCost(Opcode, SingleOpTy, CostKind);
6497 ArithmeticCost *= LT.first - 1;
6498 }
6499
6500 // FIXME: These assume a naive kshift+binop lowering, which is probably
6501 // conservative in most cases.
6502 static const CostKindTblEntry AVX512BoolReduction[] = {
6503 { ISD::AND, MVT::v2i1, { 3, 3, 3, 3} },
6504 { ISD::AND, MVT::v4i1, { 5, 5, 5, 5} },
6505 { ISD::AND, MVT::v8i1, { 7, 7, 7, 7} },
6506 { ISD::AND, MVT::v16i1, { 9, 9, 9, 9} },
6507 { ISD::AND, MVT::v32i1, {11,11,11,11} },
6508 { ISD::AND, MVT::v64i1, {13,13,13,13} },
6509 { ISD::OR, MVT::v2i1, { 3, 3, 3, 3} },
6510 { ISD::OR, MVT::v4i1, { 5, 5, 5, 5} },
6511 { ISD::OR, MVT::v8i1, { 7, 7, 7, 7} },
6512 { ISD::OR, MVT::v16i1, { 9, 9, 9, 9} },
6513 { ISD::OR, MVT::v32i1, {11,11,11,11} },
6514 { ISD::OR, MVT::v64i1, {13,13,13,13} },
6515 };
6516
6517 static const CostKindTblEntry AVX2BoolReduction[] = {
6518 { ISD::AND, MVT::v16i16, { 2, 2, 2, 2} }, // vpmovmskb + cmp
6519 { ISD::AND, MVT::v32i8, { 2, 2, 2, 2} }, // vpmovmskb + cmp
6520 { ISD::OR, MVT::v16i16, { 2, 2, 2, 2} }, // vpmovmskb + cmp
6521 { ISD::OR, MVT::v32i8, { 2, 2, 2, 2} }, // vpmovmskb + cmp
6522 };
6523
6524 static const CostKindTblEntry AVX1BoolReduction[] = {
6525 { ISD::AND, MVT::v4i64, {2, 2, 2, 2} }, // vmovmskpd + cmp
6526 { ISD::AND, MVT::v8i32, {2, 2, 2, 2} }, // vmovmskps + cmp
6527 { ISD::AND, MVT::v16i16, {4, 4, 4, 4} }, // vextractf128 + vpand + vpmovmskb + cmp
6528 { ISD::AND, MVT::v32i8, {4, 4, 4, 4} }, // vextractf128 + vpand + vpmovmskb + cmp
6529 { ISD::OR, MVT::v4i64, {2, 2, 2, 2} }, // vmovmskpd + cmp
6530 { ISD::OR, MVT::v8i32, {2, 2, 2, 2} }, // vmovmskps + cmp
6531 { ISD::OR, MVT::v16i16, {4, 4, 4, 4} }, // vextractf128 + vpor + vpmovmskb + cmp
6532 { ISD::OR, MVT::v32i8, {4, 4, 4, 4} }, // vextractf128 + vpor + vpmovmskb + cmp
6533 };
6534
6535 static const CostKindTblEntry SSE2BoolReduction[] = {
6536 { ISD::AND, MVT::v2i64, {2, 2, 2, 2} }, // movmskpd + cmp
6537 { ISD::AND, MVT::v4i32, {2, 2, 2, 2} }, // movmskps + cmp
6538 { ISD::AND, MVT::v8i16, {2, 2, 2, 2} }, // pmovmskb + cmp
6539 { ISD::AND, MVT::v16i8, {2, 2, 2, 2} }, // pmovmskb + cmp
6540 { ISD::OR, MVT::v2i64, {2, 2, 2, 2} }, // movmskpd + cmp
6541 { ISD::OR, MVT::v4i32, {2, 2, 2, 2} }, // movmskps + cmp
6542 { ISD::OR, MVT::v8i16, {2, 2, 2, 2} }, // pmovmskb + cmp
6543 { ISD::OR, MVT::v16i8, {2, 2, 2, 2} }, // pmovmskb + cmp
6544 };
6545
6546 // Handle bool allof/anyof vXi1 patterns before we check legal types.
6547 if (ValVTy->getElementType()->isIntegerTy(1)) {
6548 if (ISD == ISD::ADD) {
6549 // vXi1 addition reduction will bitcast to scalar and perform a popcount.
6550 auto *IntTy = IntegerType::getIntNTy(ValVTy->getContext(),
6551 ValVTy->getNumElements());
6552 IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy});
6553 return getCastInstrCost(Instruction::BitCast, IntTy, ValVTy,
6555 CostKind) +
6557 }
6558
6559 if (ST->hasAVX512())
6560 if (const auto *Entry = CostTableLookup(AVX512BoolReduction, ISD, MTy))
6561 if (auto KindCost = Entry->Cost[CostKind])
6562 return ArithmeticCost + *KindCost;
6563 if (ST->hasAVX2())
6564 if (const auto *Entry = CostTableLookup(AVX2BoolReduction, ISD, MTy))
6565 if (auto KindCost = Entry->Cost[CostKind])
6566 return ArithmeticCost + *KindCost;
6567 if (ST->hasAVX())
6568 if (const auto *Entry = CostTableLookup(AVX1BoolReduction, ISD, MTy))
6569 if (auto KindCost = Entry->Cost[CostKind])
6570 return ArithmeticCost + *KindCost;
6571 if (ST->hasSSE2())
6572 if (const auto *Entry = CostTableLookup(SSE2BoolReduction, ISD, MTy))
6573 if (auto KindCost = Entry->Cost[CostKind])
6574 return ArithmeticCost + *KindCost;
6575
6576 return BaseT::getArithmeticReductionCost(Opcode, ValVTy, FMF, CostKind);
6577 }
6578
6579 // Special case: vXi8 mul reductions are performed as vXi16.
6580 if (ISD == ISD::MUL && MTy.getScalarType() == MVT::i8) {
6581 auto *WideSclTy = IntegerType::get(ValVTy->getContext(), 16);
6582 auto *WideVecTy = FixedVectorType::get(WideSclTy, ValVTy->getNumElements());
6583 return getCastInstrCost(Instruction::ZExt, WideVecTy, ValTy,
6585 CostKind) +
6586 getArithmeticReductionCost(Opcode, WideVecTy, FMF, CostKind);
6587 }
6588
6589 if (ST->useSLMArithCosts())
6590 if (const auto *Entry = CostTableLookup(SLMCostTbl, ISD, MTy))
6591 if (auto KindCost = Entry->Cost[CostKind])
6592 return ArithmeticCost + *KindCost;
6593
6594 if (ST->hasBWI())
6595 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
6596 if (auto KindCost = Entry->Cost[CostKind])
6597 return ArithmeticCost + *KindCost;
6598
6599 if (ST->hasAVX512())
6600 if (const auto *Entry = CostTableLookup(AVX512FCostTbl, ISD, MTy))
6601 if (auto KindCost = Entry->Cost[CostKind])
6602 return ArithmeticCost + *KindCost;
6603
6604 if (ST->hasAVX2())
6605 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
6606 if (auto KindCost = Entry->Cost[CostKind])
6607 return ArithmeticCost + *KindCost;
6608
6609 if (ST->hasAVX())
6610 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
6611 if (auto KindCost = Entry->Cost[CostKind])
6612 return ArithmeticCost + *KindCost;
6613
6614 if (ST->hasSSE2())
6615 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
6616 if (auto KindCost = Entry->Cost[CostKind])
6617 return ArithmeticCost + *KindCost;
6618
6619 unsigned NumVecElts = ValVTy->getNumElements();
6620 unsigned ScalarSize = ValVTy->getScalarSizeInBits();
6621
6622 // Special case power of 2 reductions where the scalar type isn't changed
6623 // by type legalization.
6624 if (!isPowerOf2_32(NumVecElts) || ScalarSize != MTy.getScalarSizeInBits())
6625 return BaseT::getArithmeticReductionCost(Opcode, ValVTy, FMF, CostKind);
6626
6627 InstructionCost ReductionCost = 0;
6628
6629 auto *Ty = ValVTy;
6630 if (LT.first != 1 && MTy.isVector() &&
6631 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6632 // Type needs to be split. We need LT.first - 1 arithmetic ops.
6633 Ty = FixedVectorType::get(ValVTy->getElementType(),
6634 MTy.getVectorNumElements());
6635 ReductionCost = getArithmeticInstrCost(Opcode, Ty, CostKind);
6636 ReductionCost *= LT.first - 1;
6637 NumVecElts = MTy.getVectorNumElements();
6638 }
6639
6640 // Now handle reduction with the legal type, taking into account size changes
6641 // at each level.
6642 while (NumVecElts > 1) {
6643 // Determine the size of the remaining vector we need to reduce.
6644 unsigned Size = NumVecElts * ScalarSize;
6645 NumVecElts /= 2;
6646 // If we're reducing from 256/512 bits, use an extract_subvector.
6647 if (Size > 128) {
6648 auto *SubTy = FixedVectorType::get(ValVTy->getElementType(), NumVecElts);
6649 ReductionCost += getShuffleCost(TTI::SK_ExtractSubvector, Ty, Ty,
6650 CostKind, {}, NumVecElts, SubTy);
6651 Ty = SubTy;
6652 } else if (Size == 128) {
6653 // Reducing from 128 bits is a permute of v2f64/v2i64.
6654 FixedVectorType *ShufTy;
6655 if (ValVTy->isFloatingPointTy())
6656 ShufTy =
6657 FixedVectorType::get(Type::getDoubleTy(ValVTy->getContext()), 2);
6658 else
6659 ShufTy =
6660 FixedVectorType::get(Type::getInt64Ty(ValVTy->getContext()), 2);
6661 ReductionCost += getShuffleCost(TTI::SK_PermuteSingleSrc, ShufTy, ShufTy,
6662 CostKind, {}, 0, nullptr);
6663 } else if (Size == 64) {
6664 // Reducing from 64 bits is a shuffle of v4f32/v4i32.
6665 FixedVectorType *ShufTy;
6666 if (ValVTy->isFloatingPointTy())
6667 ShufTy =
6668 FixedVectorType::get(Type::getFloatTy(ValVTy->getContext()), 4);
6669 else
6670 ShufTy =
6671 FixedVectorType::get(Type::getInt32Ty(ValVTy->getContext()), 4);
6672 ReductionCost += getShuffleCost(TTI::SK_PermuteSingleSrc, ShufTy, ShufTy,
6673 CostKind, {}, 0, nullptr);
6674 } else {
6675 // Reducing from smaller size is a shift by immediate.
6676 auto *ShiftTy = FixedVectorType::get(
6677 Type::getIntNTy(ValVTy->getContext(), Size), 128 / Size);
6678 ReductionCost += getArithmeticInstrCost(
6679 Instruction::LShr, ShiftTy, CostKind,
6682 }
6683
6684 // Add the arithmetic op for this level.
6685 ReductionCost += getArithmeticInstrCost(Opcode, Ty, CostKind);
6686 }
6687
6688 // Add the final extract element to the cost.
6689 return ReductionCost + getVectorInstrCost(Instruction::ExtractElement, Ty,
6690 CostKind, 0, nullptr, nullptr,
6692}
6693
6696 FastMathFlags FMF) const {
6697 IntrinsicCostAttributes ICA(IID, Ty, {Ty, Ty}, FMF);
6698 return getIntrinsicInstrCost(ICA, CostKind);
6699}
6700
6703 FastMathFlags FMF,
6705 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
6706
6707 MVT MTy = LT.second;
6708
6710 if (ValTy->isIntOrIntVectorTy()) {
6711 ISD = (IID == Intrinsic::umin || IID == Intrinsic::umax) ? ISD::UMIN
6712 : ISD::SMIN;
6713 } else {
6714 assert(ValTy->isFPOrFPVectorTy() &&
6715 "Expected float point or integer vector type.");
6716 ISD = (IID == Intrinsic::minnum || IID == Intrinsic::maxnum)
6717 ? ISD::FMINNUM
6718 : ISD::FMINIMUM;
6719 }
6720
6721 // We use llvm-mca across all supported CPUs to measure the cost stats.
6722 static const CostKindTblEntry SSE2CostTbl[] = {
6723 {ISD::SMIN, MVT::v2i64, {3, 4, 5, 6}},
6724 {ISD::UMIN, MVT::v2i64, {3, 4, 5, 6}},
6725 {ISD::SMIN, MVT::v2i32, {2, 2, 5, 6}},
6726 {ISD::UMIN, MVT::v2i32, {2, 2, 5, 6}},
6727 {ISD::SMIN, MVT::v4i32, {3, 7,11,12}},
6728 {ISD::UMIN, MVT::v4i32, {4, 7,14,15}},
6729 {ISD::SMIN, MVT::v2i16, {2, 3, 4, 4}},
6730 {ISD::UMIN, MVT::v2i16, {2, 3, 4, 6}},
6731 {ISD::SMIN, MVT::v4i16, {3, 5, 6, 6}},
6732 {ISD::UMIN, MVT::v4i16, {3, 5, 8, 10}},
6733 {ISD::SMIN, MVT::v8i16, {3, 8, 8, 8}},
6734 {ISD::UMIN, MVT::v8i16, {4, 8,12,14}},
6735 {ISD::SMIN, MVT::v2i8, {2, 3, 5, 6}},
6736 {ISD::UMIN, MVT::v2i8, {2, 3, 4, 4}},
6737 {ISD::SMIN, MVT::v4i8, {4, 6,12,13}},
6738 {ISD::UMIN, MVT::v4i8, {3, 6, 7, 7}},
6739 {ISD::SMIN, MVT::v8i8, {5, 9,18,19}},
6740 {ISD::UMIN, MVT::v8i8, {4, 8, 9, 9}},
6741 {ISD::SMIN, MVT::v16i8, {7,13,24,25}},
6742 {ISD::UMIN, MVT::v16i8, {3,10,11,11}},
6743 };
6744
6745 static const CostKindTblEntry SSE41CostTbl[] = {
6746 {ISD::SMIN, MVT::v2i64, {3, 4, 4, 6}},
6747 {ISD::UMIN, MVT::v2i64, {3, 4, 4, 6}},
6748 {ISD::SMIN, MVT::v2i32, {2, 2, 3, 3}},
6749 {ISD::UMIN, MVT::v2i32, {2, 2, 3, 3}},
6750 {ISD::SMIN, MVT::v4i32, {3, 4, 5, 5}},
6751 {ISD::UMIN, MVT::v4i32, {3, 4, 5, 5}},
6752 {ISD::UMIN, MVT::v2i16, {2, 3, 4, 4}},
6753 {ISD::SMIN, MVT::v4i16, {3, 5, 6, 6}},
6754 {ISD::UMIN, MVT::v4i16, {3, 5, 6, 6}},
6755 {ISD::SMIN, MVT::v8i16, {2, 8, 4, 5}},
6756 {ISD::UMIN, MVT::v8i16, {2, 5, 2, 2}},
6757 {ISD::SMIN, MVT::v2i8, {2, 3, 4, 4}},
6758 {ISD::SMIN, MVT::v4i8, {3, 6, 7, 7}},
6759 {ISD::SMIN, MVT::v8i8, {4, 8, 9, 9}},
6760 {ISD::SMIN, MVT::v16i8, {3,10, 7, 8}},
6761 {ISD::UMIN, MVT::v16i8, {3, 8, 5, 5}},
6762 };
6763
6764 static const CostKindTblEntry AVX1CostTbl[] = {
6765 {ISD::SMIN, MVT::v4i64, {5,11, 7,10}},
6766 {ISD::UMIN, MVT::v4i64, {6,12,10,13}},
6767 {ISD::SMIN, MVT::v8i32, {4, 9, 7, 7}},
6768 {ISD::UMIN, MVT::v8i32, {4, 9, 7, 7}},
6769 {ISD::SMIN, MVT::v16i16, {3,15, 6, 7}},
6770 {ISD::UMIN, MVT::v16i16, {2, 9, 4, 4}},
6771 {ISD::SMIN, MVT::v32i8, {4,17, 8, 9}},
6772 {ISD::UMIN, MVT::v32i8, {3,11, 6, 6}},
6773 };
6774
6775 static const CostKindTblEntry AVX2CostTbl[] = {
6776 {ISD::SMIN, MVT::v4i64, {4,11, 7,10}},
6777 {ISD::UMIN, MVT::v4i64, {4,12,10,13}},
6778 {ISD::SMIN, MVT::v2i32, {1, 2, 3, 3}},
6779 {ISD::UMIN, MVT::v2i32, {1, 2, 3, 3}},
6780 {ISD::UMIN, MVT::v4i32, {2, 4, 5, 5}},
6781 {ISD::SMIN, MVT::v4i32, {2, 4, 5, 5}},
6782 {ISD::SMIN, MVT::v8i32, {3, 9, 7, 7}},
6783 {ISD::UMIN, MVT::v8i32, {3, 9, 7, 7}},
6784 {ISD::SMIN, MVT::v4i16, {2, 4, 5, 5}},
6785 {ISD::UMIN, MVT::v4i16, {2, 4, 5, 5}},
6786 {ISD::SMIN, MVT::v16i16, {2,15, 6, 7}},
6787 {ISD::SMIN, MVT::v8i8, {3, 6, 7, 7}},
6788 {ISD::UMIN, MVT::v8i8, {3, 6, 7, 7}},
6789 {ISD::SMIN, MVT::v32i8, {3,17, 8, 9}},
6790 };
6791
6792 static const CostKindTblEntry AVX512FCostTbl[] = {
6793 {ISD::SMIN, MVT::v2i64, {2, 4, 3, 3}},
6794 {ISD::UMIN, MVT::v2i64, {2, 4, 3, 3}},
6795 {ISD::SMIN, MVT::v4i64, {3,10, 5, 5}},
6796 {ISD::UMIN, MVT::v4i64, {3,10, 5, 5}},
6797 {ISD::SMIN, MVT::v8i64, {5,16, 7, 7}},
6798 {ISD::UMIN, MVT::v8i64, {5,16, 7, 7}},
6799 {ISD::SMIN, MVT::v16i32, {4,12, 9, 9}},
6800 {ISD::UMIN, MVT::v16i32, {4,12, 9, 9}},
6801 };
6802
6803 static const CostKindTblEntry AVX512BWCostTbl[] = {
6804 {ISD::SMIN, MVT::v2i16, {1, 2, 3, 3}},
6805 {ISD::UMIN, MVT::v2i16, {1, 2, 3, 3}},
6806 {ISD::SMIN, MVT::v32i16, {2,19, 8, 9}},
6807 {ISD::UMIN, MVT::v32i16, {2,12, 6, 6}},
6808 {ISD::SMIN, MVT::v2i8, {1, 2, 3, 3}},
6809 {ISD::UMIN, MVT::v2i8, {1, 2, 3, 3}},
6810 {ISD::SMIN, MVT::v4i8, {2, 4, 5, 5}},
6811 {ISD::UMIN, MVT::v4i8, {2, 4, 5, 5}},
6812 {ISD::SMIN, MVT::v16i8, {2,10, 6, 7}},
6813 {ISD::UMIN, MVT::v16i8, {2, 6, 4, 4}},
6814 {ISD::SMIN, MVT::v32i8, {2,17, 8, 9}},
6815 {ISD::UMIN, MVT::v32i8, {2,10, 6, 6}},
6816 {ISD::SMIN, MVT::v64i8, {2,21,10,11}},
6817 {ISD::UMIN, MVT::v64i8, {2,14, 8, 8}},
6818 };
6819
6820 // Before legalizing the type, give a chance to look up illegal narrow types
6821 // in the table.
6822 // FIXME: Is there a better way to do this?
6823 EVT VT = TLI->getValueType(DL, ValTy);
6824 if (VT.isSimple()) {
6825 MVT MTy = VT.getSimpleVT();
6826 if (ST->hasBWI())
6827 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
6828 if (auto KindCost = Entry->Cost[CostKind])
6829 return *KindCost;
6830
6831 if (ST->hasAVX512())
6832 if (const auto *Entry = CostTableLookup(AVX512FCostTbl, ISD, MTy))
6833 if (auto KindCost = Entry->Cost[CostKind])
6834 return *KindCost;
6835
6836 if (ST->hasAVX2())
6837 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
6838 if (auto KindCost = Entry->Cost[CostKind])
6839 return *KindCost;
6840
6841 if (ST->hasAVX())
6842 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
6843 if (auto KindCost = Entry->Cost[CostKind])
6844 return *KindCost;
6845
6846 if (ST->hasSSE41())
6847 if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
6848 if (auto KindCost = Entry->Cost[CostKind])
6849 return *KindCost;
6850
6851 if (ST->hasSSE2())
6852 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
6853 if (auto KindCost = Entry->Cost[CostKind])
6854 return *KindCost;
6855 }
6856
6857 auto *ValVTy = cast<FixedVectorType>(ValTy);
6858 unsigned NumVecElts = ValVTy->getNumElements();
6859
6860 auto *Ty = ValVTy;
6861 InstructionCost MinMaxCost = 0;
6862 if (LT.first != 1 && MTy.isVector() &&
6863 MTy.getVectorNumElements() < ValVTy->getNumElements()) {
6864 // Type needs to be split. We need LT.first - 1 operations ops.
6865 Ty = FixedVectorType::get(ValVTy->getElementType(),
6866 MTy.getVectorNumElements());
6867 MinMaxCost = getMinMaxCost(IID, Ty, CostKind, FMF);
6868 MinMaxCost *= LT.first - 1;
6869 NumVecElts = MTy.getVectorNumElements();
6870 }
6871
6872 if (ST->hasBWI())
6873 if (const auto *Entry = CostTableLookup(AVX512BWCostTbl, ISD, MTy))
6874 if (auto KindCost = Entry->Cost[CostKind])
6875 return MinMaxCost + *KindCost;
6876
6877 if (ST->hasAVX512())
6878 if (const auto *Entry = CostTableLookup(AVX512FCostTbl, ISD, MTy))
6879 if (auto KindCost = Entry->Cost[CostKind])
6880 return MinMaxCost + *KindCost;
6881
6882 if (ST->hasAVX2())
6883 if (const auto *Entry = CostTableLookup(AVX2CostTbl, ISD, MTy))
6884 if (auto KindCost = Entry->Cost[CostKind])
6885 return MinMaxCost + *KindCost;
6886
6887 if (ST->hasAVX())
6888 if (const auto *Entry = CostTableLookup(AVX1CostTbl, ISD, MTy))
6889 if (auto KindCost = Entry->Cost[CostKind])
6890 return MinMaxCost + *KindCost;
6891
6892 if (ST->hasSSE41())
6893 if (const auto *Entry = CostTableLookup(SSE41CostTbl, ISD, MTy))
6894 if (auto KindCost = Entry->Cost[CostKind])
6895 return MinMaxCost + *KindCost;
6896
6897 if (ST->hasSSE2())
6898 if (const auto *Entry = CostTableLookup(SSE2CostTbl, ISD, MTy))
6899 if (auto KindCost = Entry->Cost[CostKind])
6900 return MinMaxCost + *KindCost;
6901
6902 unsigned ScalarSize = ValTy->getScalarSizeInBits();
6903
6904 // Special case power of 2 reductions where the scalar type isn't changed
6905 // by type legalization.
6906 if (!isPowerOf2_32(ValVTy->getNumElements()) ||
6907 ScalarSize != MTy.getScalarSizeInBits())
6908 return BaseT::getMinMaxReductionCost(IID, ValTy, FMF, CostKind);
6909
6910 // Now handle reduction with the legal type, taking into account size changes
6911 // at each level.
6912 while (NumVecElts > 1) {
6913 // Determine the size of the remaining vector we need to reduce.
6914 unsigned Size = NumVecElts * ScalarSize;
6915 NumVecElts /= 2;
6916 // If we're reducing from 256/512 bits, use an extract_subvector.
6917 if (Size > 128) {
6918 auto *SubTy = FixedVectorType::get(ValVTy->getElementType(), NumVecElts);
6919 MinMaxCost += getShuffleCost(TTI::SK_ExtractSubvector, Ty, Ty, CostKind,
6920 {}, NumVecElts, SubTy);
6921 Ty = SubTy;
6922 } else if (Size == 128) {
6923 // Reducing from 128 bits is a permute of v2f64/v2i64.
6924 VectorType *ShufTy;
6925 if (ValTy->isFloatingPointTy())
6926 ShufTy =
6928 else
6929 ShufTy = FixedVectorType::get(Type::getInt64Ty(ValTy->getContext()), 2);
6930 MinMaxCost += getShuffleCost(TTI::SK_PermuteSingleSrc, ShufTy, ShufTy,
6931 CostKind, {}, 0, nullptr);
6932 } else if (Size == 64) {
6933 // Reducing from 64 bits is a shuffle of v4f32/v4i32.
6934 FixedVectorType *ShufTy;
6935 if (ValTy->isFloatingPointTy())
6936 ShufTy = FixedVectorType::get(Type::getFloatTy(ValTy->getContext()), 4);
6937 else
6938 ShufTy = FixedVectorType::get(Type::getInt32Ty(ValTy->getContext()), 4);
6939 MinMaxCost += getShuffleCost(TTI::SK_PermuteSingleSrc, ShufTy, ShufTy,
6940 CostKind, {}, 0, nullptr);
6941 } else {
6942 // Reducing from smaller size is a shift by immediate.
6943 auto *ShiftTy = FixedVectorType::get(
6944 Type::getIntNTy(ValTy->getContext(), Size), 128 / Size);
6945 MinMaxCost += getArithmeticInstrCost(
6946 Instruction::LShr, ShiftTy, TTI::TCK_RecipThroughput,
6949 }
6950
6951 // Add the arithmetic op for this level.
6952 MinMaxCost += getMinMaxCost(IID, Ty, CostKind, FMF);
6953 }
6954
6955 // Add the final extract element to the cost.
6956 return MinMaxCost + getVectorInstrCost(Instruction::ExtractElement, Ty,
6957 CostKind, 0, nullptr, nullptr,
6959}
6960
6961/// Calculate the cost of materializing a 64-bit value. This helper
6962/// method might only calculate a fraction of a larger immediate. Therefore it
6963/// is valid to return a cost of ZERO.
6965 if (Val == 0)
6966 return TTI::TCC_Free;
6967
6968 if (isInt<32>(Val))
6969 return TTI::TCC_Basic;
6970
6971 return 2 * TTI::TCC_Basic;
6972}
6973
6976 assert(Ty->isIntegerTy());
6977
6978 unsigned BitSize = Ty->getPrimitiveSizeInBits();
6979 if (BitSize == 0)
6980 return ~0U;
6981
6982 // Never hoist constants larger than 128bit, because this might lead to
6983 // incorrect code generation or assertions in codegen.
6984 // Fixme: Create a cost model for types larger than i128 once the codegen
6985 // issues have been fixed.
6986 if (BitSize > 128)
6987 return TTI::TCC_Free;
6988
6989 if (Imm == 0)
6990 return TTI::TCC_Free;
6991
6992 // Sign-extend all constants to a multiple of 64-bit.
6993 APInt ImmVal = Imm;
6994 if (BitSize % 64 != 0)
6995 ImmVal = Imm.sext(alignTo(BitSize, 64));
6996
6997 // Split the constant into 64-bit chunks and calculate the cost for each
6998 // chunk.
7000 for (unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
7001 APInt Tmp = ImmVal.ashr(ShiftVal).sextOrTrunc(64);
7002 int64_t Val = Tmp.getSExtValue();
7003 Cost += getIntImmCost(Val);
7004 }
7005 // We need at least one instruction to materialize the constant.
7006 return std::max<InstructionCost>(1, Cost);
7007}
7008
7010 const APInt &Imm, Type *Ty,
7012 Instruction *Inst) const {
7013 assert(Ty->isIntegerTy());
7014
7015 unsigned BitSize = Ty->getPrimitiveSizeInBits();
7016 unsigned ImmBitWidth = Imm.getBitWidth();
7017
7018 // There is no cost model for constants with a bit size of 0. Return TCC_Free
7019 // here, so that constant hoisting will ignore this constant.
7020 if (BitSize == 0)
7021 return TTI::TCC_Free;
7022
7023 unsigned ImmIdx = ~0U;
7024 switch (Opcode) {
7025 default:
7026 return TTI::TCC_Free;
7027 case Instruction::GetElementPtr:
7028 // Always hoist the base address of a GetElementPtr. This prevents the
7029 // creation of new constants for every base constant that gets constant
7030 // folded with the offset.
7031 if (Idx == 0)
7032 return 2 * TTI::TCC_Basic;
7033 return TTI::TCC_Free;
7034 case Instruction::Store:
7035 ImmIdx = 0;
7036 break;
7037 case Instruction::ICmp:
7038 // This is an imperfect hack to prevent constant hoisting of
7039 // compares that might be trying to check if a 64-bit value fits in
7040 // 32-bits. The backend can optimize these cases using a right shift by 32.
7041 // There are other predicates and immediates the backend can use shifts for.
7042 if (Idx == 1 && ImmBitWidth == 64) {
7043 uint64_t ImmVal = Imm.getZExtValue();
7044 if (ImmVal == 0x100000000ULL || ImmVal == 0xffffffff)
7045 return TTI::TCC_Free;
7046
7047 if (auto *Cmp = dyn_cast_or_null<CmpInst>(Inst)) {
7048 if (Cmp->isEquality()) {
7049 KnownBits Known = computeKnownBits(Cmp->getOperand(0), DL);
7050 if (Known.countMinTrailingZeros() >= 32)
7051 return TTI::TCC_Free;
7052 }
7053 }
7054 }
7055 ImmIdx = 1;
7056 break;
7057 case Instruction::And:
7058 // We support 64-bit ANDs with immediates with 32-bits of leading zeroes
7059 // by using a 32-bit operation with implicit zero extension. Detect such
7060 // immediates here as the normal path expects bit 31 to be sign extended.
7061 if (Idx == 1 && ImmBitWidth == 64 && Imm.isIntN(32))
7062 return TTI::TCC_Free;
7063 // If we have BMI then we can use BEXTR/BZHI to mask out upper i64 bits.
7064 if (Idx == 1 && ImmBitWidth == 64 && ST->is64Bit() && ST->hasBMI() &&
7065 Imm.isMask())
7066 return X86TTIImpl::getIntImmCost(ST->hasBMI2() ? 255 : 65535);
7067 ImmIdx = 1;
7068 break;
7069 case Instruction::Add:
7070 case Instruction::Sub:
7071 // For add/sub, we can use the opposite instruction for INT32_MIN.
7072 if (Idx == 1 && ImmBitWidth == 64 && Imm.getZExtValue() == 0x80000000)
7073 return TTI::TCC_Free;
7074 ImmIdx = 1;
7075 break;
7076 case Instruction::UDiv:
7077 case Instruction::SDiv:
7078 case Instruction::URem:
7079 case Instruction::SRem:
7080 // Division by constant is typically expanded later into a different
7081 // instruction sequence. This completely changes the constants.
7082 // Report them as "free" to stop ConstantHoist from marking them as opaque.
7083 return TTI::TCC_Free;
7084 case Instruction::Mul:
7085 case Instruction::Or:
7086 case Instruction::Xor:
7087 ImmIdx = 1;
7088 break;
7089 // Always return TCC_Free for the shift value of a shift instruction.
7090 case Instruction::Shl:
7091 case Instruction::LShr:
7092 case Instruction::AShr:
7093 if (Idx == 1)
7094 return TTI::TCC_Free;
7095 break;
7096 case Instruction::Trunc:
7097 case Instruction::ZExt:
7098 case Instruction::SExt:
7099 case Instruction::IntToPtr:
7100 case Instruction::PtrToInt:
7101 case Instruction::BitCast:
7102 case Instruction::PHI:
7103 case Instruction::Call:
7104 case Instruction::Select:
7105 case Instruction::Ret:
7106 case Instruction::Load:
7107 break;
7108 }
7109
7110 if (Idx == ImmIdx) {
7111 uint64_t NumConstants = divideCeil(BitSize, 64);
7113 return (Cost <= NumConstants * TTI::TCC_Basic)
7114 ? static_cast<int>(TTI::TCC_Free)
7115 : Cost;
7116 }
7117
7119}
7120
7123 const APInt &Imm, Type *Ty,
7125 assert(Ty->isIntegerTy());
7126
7127 unsigned BitSize = Ty->getPrimitiveSizeInBits();
7128 // There is no cost model for constants with a bit size of 0. Return TCC_Free
7129 // here, so that constant hoisting will ignore this constant.
7130 if (BitSize == 0)
7131 return TTI::TCC_Free;
7132
7133 switch (IID) {
7134 default:
7135 return TTI::TCC_Free;
7136 case Intrinsic::sadd_with_overflow:
7137 case Intrinsic::uadd_with_overflow:
7138 case Intrinsic::ssub_with_overflow:
7139 case Intrinsic::usub_with_overflow:
7140 case Intrinsic::smul_with_overflow:
7141 case Intrinsic::umul_with_overflow:
7142 if ((Idx == 1) && Imm.getBitWidth() <= 64 && Imm.isSignedIntN(32))
7143 return TTI::TCC_Free;
7144 break;
7145 case Intrinsic::experimental_stackmap:
7146 if ((Idx < 2) || (Imm.getBitWidth() <= 64 && Imm.isSignedIntN(64)))
7147 return TTI::TCC_Free;
7148 break;
7149 case Intrinsic::experimental_patchpoint_void:
7150 case Intrinsic::experimental_patchpoint:
7151 if ((Idx < 4) || (Imm.getBitWidth() <= 64 && Imm.isSignedIntN(64)))
7152 return TTI::TCC_Free;
7153 break;
7154 }
7156}
7157
7160 const Instruction *I) const {
7162 return Opcode == Instruction::PHI ? TTI::TCC_Free : TTI::TCC_Basic;
7163 // Branches are assumed to be predicted.
7164 return TTI::TCC_Free;
7165}
7166
7167int X86TTIImpl::getGatherOverhead() const {
7168 // Some CPUs have more overhead for gather. The specified overhead is relative
7169 // to the Load operation. "2" is the number provided by Intel architects. This
7170 // parameter is used for cost estimation of Gather Op and comparison with
7171 // other alternatives.
7172 // TODO: Remove the explicit hasAVX512()?, That would mean we would only
7173 // enable gather with a -march.
7174 if (ST->hasAVX512() || (ST->hasAVX2() && ST->hasFastGather()))
7175 return 2;
7176
7177 return 1024;
7178}
7179
7180int X86TTIImpl::getScatterOverhead() const {
7181 if (ST->hasAVX512())
7182 return 2;
7183
7184 return 1024;
7185}
7186
7187// Return an average cost of Gather / Scatter instruction, maybe improved later.
7188InstructionCost X86TTIImpl::getGSVectorCost(unsigned Opcode,
7190 Type *SrcVTy, const Value *Ptr,
7191 Align Alignment,
7192 unsigned AddressSpace) const {
7193
7194 assert(isa<VectorType>(SrcVTy) && "Unexpected type in getGSVectorCost");
7195 unsigned VF = cast<FixedVectorType>(SrcVTy)->getNumElements();
7196
7197 // Try to reduce index size from 64 bit (default for GEP)
7198 // to 32. It is essential for VF 16. If the index can't be reduced to 32, the
7199 // operation will use 16 x 64 indices which do not fit in a zmm and needs
7200 // to split. Also check that the base pointer is the same for all lanes,
7201 // and that there's at most one variable index.
7202 auto getIndexSizeInBits = [](const Value *Ptr, const DataLayout &DL) {
7203 unsigned IndexSize = DL.getPointerSizeInBits();
7204 const GetElementPtrInst *GEP = dyn_cast_or_null<GetElementPtrInst>(Ptr);
7205 if (IndexSize < 64 || !GEP)
7206 return IndexSize;
7207
7208 unsigned NumOfVarIndices = 0;
7209 const Value *Ptrs = GEP->getPointerOperand();
7210 if (Ptrs->getType()->isVectorTy() && !getSplatValue(Ptrs))
7211 return IndexSize;
7212 for (unsigned I = 1, E = GEP->getNumOperands(); I != E; ++I) {
7213 if (isa<Constant>(GEP->getOperand(I)))
7214 continue;
7215 Type *IndxTy = GEP->getOperand(I)->getType();
7216 if (auto *IndexVTy = dyn_cast<VectorType>(IndxTy))
7217 IndxTy = IndexVTy->getElementType();
7218 if ((IndxTy->getPrimitiveSizeInBits() == 64 &&
7219 !isa<SExtInst>(GEP->getOperand(I))) ||
7220 ++NumOfVarIndices > 1)
7221 return IndexSize; // 64
7222 }
7223 return (unsigned)32;
7224 };
7225
7226 // Trying to reduce IndexSize to 32 bits for vector 16.
7227 // By default the IndexSize is equal to pointer size.
7228 unsigned IndexSize = (ST->hasAVX512() && VF >= 16)
7229 ? getIndexSizeInBits(Ptr, DL)
7230 : DL.getPointerSizeInBits();
7231
7232 auto *IndexVTy = FixedVectorType::get(
7233 IntegerType::get(SrcVTy->getContext(), IndexSize), VF);
7234 std::pair<InstructionCost, MVT> IdxsLT = getTypeLegalizationCost(IndexVTy);
7235 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(SrcVTy);
7236 InstructionCost::CostType SplitFactor =
7237 std::max(IdxsLT.first, SrcLT.first).getValue();
7238 if (SplitFactor > 1) {
7239 // Handle splitting of vector of pointers
7240 auto *SplitSrcTy =
7241 FixedVectorType::get(SrcVTy->getScalarType(), VF / SplitFactor);
7242 return SplitFactor * getGSVectorCost(Opcode, CostKind, SplitSrcTy, Ptr,
7243 Alignment, AddressSpace);
7244 }
7245
7246 // If we didn't split, this will be a single gather/scatter instruction.
7248 return 1;
7249
7250 // The gather / scatter cost is given by Intel architects. It is a rough
7251 // number since we are looking at one instruction in a time.
7252 const int GSOverhead = (Opcode == Instruction::Load) ? getGatherOverhead()
7253 : getScatterOverhead();
7254 return GSOverhead + VF * getMemoryOpCost(Opcode, SrcVTy->getScalarType(),
7255 Alignment, AddressSpace, CostKind);
7256}
7257
7258/// Calculate the cost of Gather / Scatter operation
7262 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
7263 MICA.getID() == Intrinsic::vp_gather;
7264 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
7265 Type *SrcVTy = MICA.getDataType();
7266 const Value *Ptr = MICA.getPointer();
7267 Align Alignment = MICA.getAlignment();
7268 if ((Opcode == Instruction::Load &&
7269 (!isLegalMaskedGather(SrcVTy, Align(Alignment)) ||
7271 Align(Alignment)))) ||
7272 (Opcode == Instruction::Store &&
7273 (!isLegalMaskedScatter(SrcVTy, Align(Alignment)) ||
7275 Align(Alignment)))))
7277
7278 assert(SrcVTy->isVectorTy() && "Unexpected data type for Gather/Scatter");
7279 unsigned AddressSpace = MICA.getAddressSpace();
7281 getGSVectorCost(Opcode, CostKind, SrcVTy, Ptr, Alignment, AddressSpace);
7282
7283 // Predicate fanout (see getPredicateFanoutCost); variable mask only.
7284 if (MICA.getVariableMask()) {
7285 auto *MaskVecTy =
7287 cast<FixedVectorType>(SrcVTy)->getNumElements());
7288 Cost += getPredicateFanoutCost(*this, SrcVTy, MaskVecTy);
7289 }
7290 return Cost;
7291}
7292
7294 const TargetTransformInfo::LSRCost &C2) const {
7295 // X86 specific here are "instruction number 1st priority".
7296 return std::tie(C1.Insns, C1.NumRegs, C1.AddRecCost, C1.NumIVMuls,
7297 C1.NumBaseAdds, C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
7298 std::tie(C2.Insns, C2.NumRegs, C2.AddRecCost, C2.NumIVMuls,
7299 C2.NumBaseAdds, C2.ScaleCost, C2.ImmCost, C2.SetupCost);
7300}
7301
7303 return ST->hasMacroFusion() || ST->hasBranchFusion();
7304}
7305
7306static bool isLegalMaskedLoadStore(Type *ScalarTy, const X86Subtarget *ST) {
7307 if (!ST->hasAVX())
7308 return false;
7309
7310 if (ScalarTy->isPointerTy())
7311 return true;
7312
7313 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7314 return true;
7315
7316 if (ScalarTy->isHalfTy() && ST->hasBWI())
7317 return true;
7318
7319 if (ScalarTy->isBFloatTy() && ST->hasBF16())
7320 return true;
7321
7322 if (!ScalarTy->isIntegerTy())
7323 return false;
7324
7325 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7326 return IntWidth == 32 || IntWidth == 64 ||
7327 ((IntWidth == 8 || IntWidth == 16) && ST->hasBWI());
7328}
7329
7331 unsigned AddressSpace,
7332 TTI::MaskKind MaskKind) const {
7333 Type *ScalarTy = DataTy->getScalarType();
7334
7335 // The backend can't handle a single element vector w/o CFCMOV.
7336 if (isa<VectorType>(DataTy) &&
7337 cast<FixedVectorType>(DataTy)->getNumElements() == 1)
7338 return ST->hasCF() &&
7339 hasConditionalLoadStoreForType(ScalarTy, /*IsStore=*/false);
7340
7341 return isLegalMaskedLoadStore(ScalarTy, ST);
7342}
7343
7345 unsigned AddressSpace,
7346 TTI::MaskKind MaskKind) const {
7347 Type *ScalarTy = DataTy->getScalarType();
7348
7349 // The backend can't handle a single element vector w/o CFCMOV.
7350 if (isa<VectorType>(DataTy) &&
7351 cast<FixedVectorType>(DataTy)->getNumElements() == 1)
7352 return ST->hasCF() &&
7353 hasConditionalLoadStoreForType(ScalarTy, /*IsStore=*/true);
7354
7355 return isLegalMaskedLoadStore(ScalarTy, ST);
7356}
7357
7358bool X86TTIImpl::isLegalNTLoad(Type *DataType, Align Alignment) const {
7359 unsigned DataSize = DL.getTypeStoreSize(DataType);
7360 // The only supported nontemporal loads are for aligned vectors of 16 or 32
7361 // bytes. Note that 32-byte nontemporal vector loads are supported by AVX2
7362 // (the equivalent stores only require AVX).
7363 if (Alignment >= DataSize && (DataSize == 16 || DataSize == 32))
7364 return DataSize == 16 ? ST->hasSSE1() : ST->hasAVX2();
7365
7366 return false;
7367}
7368
7369bool X86TTIImpl::isLegalNTStore(Type *DataType, Align Alignment) const {
7370 unsigned DataSize = DL.getTypeStoreSize(DataType);
7371
7372 // SSE4A supports nontemporal stores of float and double at arbitrary
7373 // alignment.
7374 if (ST->hasSSE4A() && (DataType->isFloatTy() || DataType->isDoubleTy()))
7375 return true;
7376
7377 // Besides the SSE4A subtarget exception above, only aligned stores are
7378 // available nontemporaly on any other subtarget. And only stores with a size
7379 // of 4..32 bytes (powers of 2, only) are permitted.
7380 if (Alignment < DataSize || DataSize < 4 || DataSize > 32 ||
7381 !isPowerOf2_32(DataSize))
7382 return false;
7383
7384 // 32-byte vector nontemporal stores are supported by AVX (the equivalent
7385 // loads require AVX2).
7386 if (DataSize == 32)
7387 return ST->hasAVX();
7388 if (DataSize == 16)
7389 return ST->hasSSE1();
7390 return true;
7391}
7392
7394 ElementCount NumElements) const {
7395 // movddup
7396 return ST->hasSSE3() && !NumElements.isScalable() &&
7397 NumElements.getFixedValue() == 2 &&
7398 ElementTy == Type::getDoubleTy(ElementTy->getContext());
7399}
7400
7401bool X86TTIImpl::isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const {
7402 if (!isa<VectorType>(DataTy))
7403 return false;
7404
7405 if (!ST->hasAVX512())
7406 return false;
7407
7408 // The backend can't handle a single element vector.
7409 if (cast<FixedVectorType>(DataTy)->getNumElements() == 1)
7410 return false;
7411
7412 Type *ScalarTy = cast<VectorType>(DataTy)->getElementType();
7413
7414 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7415 return true;
7416
7417 if (!ScalarTy->isIntegerTy())
7418 return false;
7419
7420 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7421 return IntWidth == 32 || IntWidth == 64 ||
7422 ((IntWidth == 8 || IntWidth == 16) && ST->hasVBMI2());
7423}
7424
7426 Align Alignment) const {
7427 return isLegalMaskedExpandLoad(DataTy, Alignment);
7428}
7429
7430bool X86TTIImpl::supportsGather() const {
7431 // Some CPUs have better gather performance than others.
7432 // TODO: Remove the explicit ST->hasAVX512()?, That would mean we would only
7433 // enable gather with a -march.
7434 return ST->hasAVX512() || (ST->hasFastGather() && ST->hasAVX2());
7435}
7436
7438 Align Alignment) const {
7439 // Gather / Scatter for vector 2 is not profitable on KNL / SKX
7440 // Vector-4 of gather/scatter instruction does not exist on KNL. We can extend
7441 // it to 8 elements, but zeroing upper bits of the mask vector will add more
7442 // instructions. Right now we give the scalar cost of vector-4 for KNL. TODO:
7443 // Check, maybe the gather/scatter instruction is better in the VariableMask
7444 // case.
7445 unsigned NumElts = cast<FixedVectorType>(VTy)->getNumElements();
7446 return NumElts == 1 ||
7447 (ST->hasAVX512() && (NumElts == 2 || (NumElts == 4 && !ST->hasVLX())));
7448}
7449
7451 Align Alignment) const {
7452 Type *ScalarTy = DataTy->getScalarType();
7453 if (ScalarTy->isPointerTy())
7454 return true;
7455
7456 if (ScalarTy->isFloatTy() || ScalarTy->isDoubleTy())
7457 return true;
7458
7459 if (!ScalarTy->isIntegerTy())
7460 return false;
7461
7462 unsigned IntWidth = ScalarTy->getIntegerBitWidth();
7463 return IntWidth == 32 || IntWidth == 64;
7464}
7465
7466bool X86TTIImpl::isLegalMaskedGather(Type *DataTy, Align Alignment) const {
7467 if (!supportsGather() || !ST->preferGather())
7468 return false;
7469 return isLegalMaskedGatherScatter(DataTy, Alignment);
7470}
7471
7472bool X86TTIImpl::isLegalAltInstr(VectorType *VecTy, unsigned Opcode0,
7473 unsigned Opcode1,
7474 const SmallBitVector &OpcodeMask,
7475 ArrayRef<const Value *> Scalars) const {
7476 // ADDSUBPS 4xf32 SSE3
7477 // VADDSUBPS 4xf32 AVX
7478 // VADDSUBPS 8xf32 AVX2
7479 // ADDSUBPD 2xf64 SSE3
7480 // VADDSUBPD 2xf64 AVX
7481 // VADDSUBPD 4xf64 AVX2
7482
7483 unsigned NumElements = cast<FixedVectorType>(VecTy)->getNumElements();
7484 assert(OpcodeMask.size() == NumElements && "Mask and VecTy are incompatible");
7485 if (!isPowerOf2_32(NumElements))
7486 return false;
7487 // Check the opcode pattern. We apply the mask on the opcode arguments and
7488 // then check if it is what we expect.
7489 auto IsLaneOrder = [&](unsigned EvenOpc, unsigned OddOpc) {
7490 return all_of(seq<unsigned>(0, NumElements), [&](unsigned Lane) {
7491 return (OpcodeMask.test(Lane) ? Opcode1 : Opcode0) ==
7492 (Lane % 2 == 0 ? EvenOpc : OddOpc);
7493 });
7494 };
7495 // We expect FSub for even lanes and FAdd for odd lanes.
7496 const bool IsAddSub = IsLaneOrder(Instruction::FSub, Instruction::FAdd);
7497 // Now check that the pattern is supported by the target ISA.
7498 Type *ElemTy = cast<VectorType>(VecTy)->getElementType();
7499 if (IsAddSub && ST->hasSSE3() &&
7500 ((ElemTy->isFloatTy() && NumElements % 4 == 0) ||
7501 (ElemTy->isDoubleTy() && NumElements % 2 == 0)))
7502 return true;
7503 // The multiplication is fused into every lane: fmaddsub (the even lanes
7504 // subtract and the odd lanes add the product) or fmsubadd (the other way
7505 // round). The subtracted product is the first operand of the subtraction.
7506 using namespace PatternMatch;
7507 auto IsFusedWithFMul = [](const Value *V) {
7508 const auto *I = dyn_cast<Instruction>(V);
7509 if (!I || !I->hasAllowContract())
7510 return false;
7511 auto IsFMul = [&](const Value *Op) {
7512 return match(Op,
7514 cast<Instruction>(Op)->getParent() == I->getParent();
7515 };
7516 return IsFMul(I->getOperand(0)) ||
7517 (I->getOpcode() == Instruction::FAdd && IsFMul(I->getOperand(1)));
7518 };
7519 return ST->hasFMA() && !Scalars.empty() &&
7520 (ElemTy->isFloatTy() || ElemTy->isDoubleTy()) &&
7521 (IsAddSub || IsLaneOrder(Instruction::FAdd, Instruction::FSub)) &&
7522 all_of(Scalars, IsFusedWithFMul);
7523}
7524
7525bool X86TTIImpl::isLegalMaskedScatter(Type *DataType, Align Alignment) const {
7526 // AVX2 doesn't support scatter
7527 if (!ST->hasAVX512() || !ST->preferScatter())
7528 return false;
7529 return isLegalMaskedGatherScatter(DataType, Alignment);
7530}
7531
7532bool X86TTIImpl::hasDivRemOp(Type *DataType, bool IsSigned) const {
7533 EVT VT = TLI->getValueType(DL, DataType);
7534 return TLI->isOperationLegal(IsSigned ? ISD::SDIVREM : ISD::UDIVREM, VT);
7535}
7536
7538 // FDIV is always expensive, even if it has a very low uop count.
7539 // TODO: Still necessary for recent CPUs with low latency/throughput fdiv?
7540 if (I->getOpcode() == Instruction::FDiv)
7541 return true;
7542
7544}
7545
7546bool X86TTIImpl::isFCmpOrdCheaperThanFCmpZero(Type *Ty) const { return false; }
7547
7549 const Function *Callee) const {
7550 const TargetMachine &TM = getTLI()->getTargetMachine();
7551
7552 // Work this as a subsetting of subtarget features.
7553 const X86Subtarget &CallerSubtarget = TM.getSubtarget<X86Subtarget>(*Caller);
7554 const X86Subtarget &CalleeSubtarget = TM.getSubtarget<X86Subtarget>(*Callee);
7555 const FeatureBitset &CallerBits = CallerSubtarget.getFeatureBits();
7556 const FeatureBitset &CalleeBits = CalleeSubtarget.getFeatureBits();
7557
7558 // Check whether callee features are a subset of caller features
7559 // (apart from the ignore list).
7560 const FeatureBitset &InlineIgnoreFeatures =
7561 CallerSubtarget.getInlineIgnoreFeatures();
7562 FeatureBitset RealCallerBits = CallerBits & ~InlineIgnoreFeatures;
7563 FeatureBitset RealCalleeBits = CalleeBits & ~InlineIgnoreFeatures;
7564 if ((RealCallerBits & RealCalleeBits) != RealCalleeBits)
7565 return false;
7566
7567 // If the features are not exactly the same (or there is a difference in
7568 // AVX512 register usage), we need to additionally check for calls
7569 // that may become ABI-incompatible as a result of inlining.
7570 if (RealCallerBits == RealCalleeBits &&
7571 CallerSubtarget.useAVX512Regs() == CalleeSubtarget.useAVX512Regs())
7572 return true;
7573
7574 for (const Instruction &I : instructions(Callee)) {
7575 if (const auto *CB = dyn_cast<CallBase>(&I)) {
7576 // Having more target features is fine for inline ASM and intrinsics.
7577 if (CB->isInlineAsm() || CB->getIntrinsicID() != Intrinsic::not_intrinsic)
7578 continue;
7579
7581 for (Value *Arg : CB->args())
7582 Types.push_back(Arg->getType());
7583 if (!CB->getType()->isVoidTy())
7584 Types.push_back(CB->getType());
7585
7586 // Simple types are always ABI compatible.
7587 auto IsSimpleTy = [](Type *Ty) {
7588 return !Ty->isVectorTy() && !Ty->isAggregateType();
7589 };
7590 if (all_of(Types, IsSimpleTy))
7591 continue;
7592
7593 // Do a precise compatibility check.
7594 if (!areTypesABICompatible(Caller, Callee, Types))
7595 return false;
7596 }
7597 }
7598 return true;
7599}
7600
7602 const Function *Callee,
7603 ArrayRef<Type *> Types) const {
7604 const TargetMachine &TM = getTLI()->getTargetMachine();
7605 const TargetLowering *CallerTLI =
7606 TM.getSubtargetImpl(*Caller)->getTargetLowering();
7607 const TargetLowering *CalleeTLI =
7608 TM.getSubtargetImpl(*Callee)->getTargetLowering();
7609
7610 LLVMContext &Ctx = Caller->getContext();
7611 const DataLayout &DL = Caller->getDataLayout();
7612 CallingConv::ID CC = Callee->getCallingConv();
7613 return all_of(Types, [&](Type *Ty) {
7614 SmallVector<EVT> VTs;
7615 ComputeValueVTs(*CallerTLI, DL, Ty, VTs);
7616 return all_of(VTs, [&](EVT VT) {
7617 return CallerTLI->getRegisterTypeForCallingConv(Ctx, CC, VT) ==
7618 CalleeTLI->getRegisterTypeForCallingConv(Ctx, CC, VT);
7619 });
7620 });
7621}
7622
7624X86TTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
7626 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
7627 Options.NumLoadsPerBlock = IsZeroCmp ? 2 : 1;
7628 // All GPR and vector loads can be unaligned.
7629 Options.AllowOverlappingLoads = true;
7630 if (IsZeroCmp) {
7631 // Only enable vector loads for equality comparison. Right now the vector
7632 // version is not as fast for three way compare (see #33329).
7633 const unsigned PreferredWidth = ST->getPreferVectorWidth();
7634 if (PreferredWidth >= 512 && ST->hasAVX512())
7635 Options.LoadSizes.push_back(64);
7636 if (PreferredWidth >= 256 && ST->hasAVX()) Options.LoadSizes.push_back(32);
7637 if (PreferredWidth >= 128 && ST->hasSSE2()) Options.LoadSizes.push_back(16);
7638 }
7639 if (ST->is64Bit()) {
7640 Options.LoadSizes.push_back(8);
7641 }
7642 Options.LoadSizes.push_back(4);
7643 Options.LoadSizes.push_back(2);
7644 Options.LoadSizes.push_back(1);
7645 return Options;
7646}
7647
7649 return supportsGather();
7650}
7651
7653 return false;
7654}
7655
7657 // TODO: We expect this to be beneficial regardless of arch,
7658 // but there are currently some unexplained performance artifacts on Atom.
7659 // As a temporary solution, disable on Atom.
7660 return !(ST->isAtom());
7661}
7662
7664 switch (II->getIntrinsicID()) {
7665 default:
7666 return true;
7667 case Intrinsic::vector_reduce_and:
7668 case Intrinsic::vector_reduce_or:
7669 case Intrinsic::vector_reduce_xor:
7670 case Intrinsic::vector_reduce_mul:
7671 case Intrinsic::vector_reduce_smax:
7672 case Intrinsic::vector_reduce_smin:
7673 case Intrinsic::vector_reduce_umax:
7674 case Intrinsic::vector_reduce_umin:
7675 return false;
7676 }
7677}
7678
7679// Get estimation for interleaved load/store operations and strided load.
7680// \p Indices contains indices for strided load.
7681// \p Factor - the factor of interleaving.
7682// AVX-512 provides 3-src shuffles that significantly reduces the cost.
7684 unsigned Opcode, FixedVectorType *VecTy, unsigned Factor,
7685 ArrayRef<unsigned> Indices, Align Alignment, unsigned AddressSpace,
7686 TTI::TargetCostKind CostKind, bool UseMaskForCond,
7687 bool UseMaskForGaps) const {
7688 // VecTy for interleave memop is <VF*Factor x Elt>.
7689 // So, for VF=4, Interleave Factor = 3, Element type = i32 we have
7690 // VecTy = <12 x i32>.
7691
7692 // Calculate the number of memory operations (NumOfMemOps), required
7693 // for load/store the VecTy.
7694 MVT LegalVT = getTypeLegalizationCost(VecTy).second;
7695 unsigned VecTySize = DL.getTypeStoreSize(VecTy);
7696 unsigned LegalVTSize = LegalVT.getStoreSize();
7697 unsigned NumOfMemOps = (VecTySize + LegalVTSize - 1) / LegalVTSize;
7698
7699 // Get the cost of one memory operation.
7700 auto *SingleMemOpTy = FixedVectorType::get(VecTy->getElementType(),
7701 LegalVT.getVectorNumElements());
7702 InstructionCost MemOpCost;
7703 bool UseMaskedMemOp = UseMaskForCond || UseMaskForGaps;
7704 if (UseMaskedMemOp) {
7705 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
7706 : Intrinsic::masked_store;
7707 MemOpCost = getMaskedMemoryOpCost(
7708 {IID, SingleMemOpTy, Alignment, AddressSpace}, CostKind);
7709 } else
7710 MemOpCost = getMemoryOpCost(Opcode, SingleMemOpTy, Alignment, AddressSpace,
7711 CostKind);
7712
7713 unsigned VF = VecTy->getNumElements() / Factor;
7714 MVT VT =
7715 MVT::getVectorVT(TLI->getSimpleValueType(DL, VecTy->getScalarType()), VF);
7716
7717 InstructionCost MaskCost;
7718 if (UseMaskedMemOp) {
7719 APInt DemandedLoadStoreElts = APInt::getZero(VecTy->getNumElements());
7720 for (unsigned Index : Indices) {
7721 assert(Index < Factor && "Invalid index for interleaved memory op");
7722 for (unsigned Elm = 0; Elm < VF; Elm++)
7723 DemandedLoadStoreElts.setBit(Index + Elm * Factor);
7724 }
7725
7726 Type *I1Type = Type::getInt1Ty(VecTy->getContext());
7727
7728 MaskCost = getReplicationShuffleCost(
7729 I1Type, Factor, VF,
7730 UseMaskForGaps ? DemandedLoadStoreElts
7732 CostKind);
7733
7734 // The Gaps mask is invariant and created outside the loop, therefore the
7735 // cost of creating it is not accounted for here. However if we have both
7736 // a MaskForGaps and some other mask that guards the execution of the
7737 // memory access, we need to account for the cost of And-ing the two masks
7738 // inside the loop.
7739 if (UseMaskForGaps) {
7740 auto *MaskVT = FixedVectorType::get(I1Type, VecTy->getNumElements());
7741 MaskCost += getArithmeticInstrCost(BinaryOperator::And, MaskVT, CostKind);
7742 }
7743 }
7744
7745 if (Opcode == Instruction::Load) {
7746 // The tables (AVX512InterleavedLoadTbl and AVX512InterleavedStoreTbl)
7747 // contain the cost of the optimized shuffle sequence that the
7748 // X86InterleavedAccess pass will generate.
7749 // The cost of loads and stores are computed separately from the table.
7750
7751 // X86InterleavedAccess support only the following interleaved-access group.
7752 static const CostTblEntry AVX512InterleavedLoadTbl[] = {
7753 {3, MVT::v16i8, 12}, //(load 48i8 and) deinterleave into 3 x 16i8
7754 {3, MVT::v32i8, 14}, //(load 96i8 and) deinterleave into 3 x 32i8
7755 {3, MVT::v64i8, 22}, //(load 96i8 and) deinterleave into 3 x 32i8
7756 };
7757
7758 if (const auto *Entry =
7759 CostTableLookup(AVX512InterleavedLoadTbl, Factor, VT))
7760 return MaskCost + NumOfMemOps * MemOpCost + Entry->Cost;
7761 //If an entry does not exist, fallback to the default implementation.
7762
7763 // Kind of shuffle depends on number of loaded values.
7764 // If we load the entire data in one register, we can use a 1-src shuffle.
7765 // Otherwise, we'll merge 2 sources in each operation.
7766 TTI::ShuffleKind ShuffleKind =
7767 (NumOfMemOps > 1) ? TTI::SK_PermuteTwoSrc : TTI::SK_PermuteSingleSrc;
7768
7769 InstructionCost ShuffleCost = getShuffleCost(
7770 ShuffleKind, SingleMemOpTy, SingleMemOpTy, CostKind, {}, 0, nullptr);
7771
7772 unsigned NumOfLoadsInInterleaveGrp =
7773 Indices.size() ? Indices.size() : Factor;
7774 auto *ResultTy = FixedVectorType::get(VecTy->getElementType(),
7775 VecTy->getNumElements() / Factor);
7776 InstructionCost NumOfResults =
7777 getTypeLegalizationCost(ResultTy).first * NumOfLoadsInInterleaveGrp;
7778
7779 // About a half of the loads may be folded in shuffles when we have only
7780 // one result. If we have more than one result, or the loads are masked,
7781 // we do not fold loads at all.
7782 unsigned NumOfUnfoldedLoads =
7783 UseMaskedMemOp || NumOfResults > 1 ? NumOfMemOps : NumOfMemOps / 2;
7784
7785 // Get a number of shuffle operations per result.
7786 unsigned NumOfShufflesPerResult =
7787 std::max((unsigned)1, (unsigned)(NumOfMemOps - 1));
7788
7789 // The SK_MergeTwoSrc shuffle clobbers one of src operands.
7790 // When we have more than one destination, we need additional instructions
7791 // to keep sources.
7792 InstructionCost NumOfMoves = 0;
7793 if (NumOfResults > 1 && ShuffleKind == TTI::SK_PermuteTwoSrc)
7794 NumOfMoves = NumOfResults * NumOfShufflesPerResult / 2;
7795
7796 InstructionCost Cost = NumOfResults * NumOfShufflesPerResult * ShuffleCost +
7797 MaskCost + NumOfUnfoldedLoads * MemOpCost +
7798 NumOfMoves;
7799
7800 return Cost;
7801 }
7802
7803 // Store.
7804 assert(Opcode == Instruction::Store &&
7805 "Expected Store Instruction at this point");
7806 // X86InterleavedAccess support only the following interleaved-access group.
7807 static const CostTblEntry AVX512InterleavedStoreTbl[] = {
7808 {3, MVT::v16i8, 12}, // interleave 3 x 16i8 into 48i8 (and store)
7809 {3, MVT::v32i8, 14}, // interleave 3 x 32i8 into 96i8 (and store)
7810 {3, MVT::v64i8, 26}, // interleave 3 x 64i8 into 96i8 (and store)
7811
7812 {4, MVT::v8i8, 10}, // interleave 4 x 8i8 into 32i8 (and store)
7813 {4, MVT::v16i8, 11}, // interleave 4 x 16i8 into 64i8 (and store)
7814 {4, MVT::v32i8, 14}, // interleave 4 x 32i8 into 128i8 (and store)
7815 {4, MVT::v64i8, 24} // interleave 4 x 32i8 into 256i8 (and store)
7816 };
7817
7818 if (const auto *Entry =
7819 CostTableLookup(AVX512InterleavedStoreTbl, Factor, VT))
7820 return MaskCost + NumOfMemOps * MemOpCost + Entry->Cost;
7821 //If an entry does not exist, fallback to the default implementation.
7822
7823 // There is no strided stores meanwhile. And store can't be folded in
7824 // shuffle.
7825 unsigned NumOfSources = Factor; // The number of values to be merged.
7826 InstructionCost ShuffleCost =
7827 getShuffleCost(TTI::SK_PermuteTwoSrc, SingleMemOpTy, SingleMemOpTy,
7828 CostKind, {}, 0, nullptr);
7829 unsigned NumOfShufflesPerStore = NumOfSources - 1;
7830
7831 // The SK_MergeTwoSrc shuffle clobbers one of src operands.
7832 // We need additional instructions to keep sources.
7833 unsigned NumOfMoves = NumOfMemOps * NumOfShufflesPerStore / 2;
7835 MaskCost +
7836 NumOfMemOps * (MemOpCost + NumOfShufflesPerStore * ShuffleCost) +
7837 NumOfMoves;
7838 return Cost;
7839}
7840
7842 unsigned Opcode, Type *BaseTy, unsigned Factor, ArrayRef<unsigned> Indices,
7843 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
7844 bool UseMaskForCond, bool UseMaskForGaps) const {
7845 auto *VecTy = cast<FixedVectorType>(BaseTy);
7846
7847 auto isSupportedOnAVX512 = [&](Type *VecTy) {
7848 Type *EltTy = cast<VectorType>(VecTy)->getElementType();
7849 if (EltTy->isFloatTy() || EltTy->isDoubleTy() || EltTy->isIntegerTy(64) ||
7850 EltTy->isIntegerTy(32) || EltTy->isPointerTy())
7851 return true;
7852 if (EltTy->isIntegerTy(16) || EltTy->isIntegerTy(8) || EltTy->isHalfTy())
7853 return ST->hasBWI();
7854 if (EltTy->isBFloatTy())
7855 return ST->hasBF16();
7856 return false;
7857 };
7858 if (ST->hasAVX512() && isSupportedOnAVX512(VecTy))
7860 Opcode, VecTy, Factor, Indices, Alignment,
7861 AddressSpace, CostKind, UseMaskForCond, UseMaskForGaps);
7862
7863 if (UseMaskForCond || UseMaskForGaps)
7864 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7865 Alignment, AddressSpace, CostKind,
7866 UseMaskForCond, UseMaskForGaps);
7867
7868 // Get estimation for interleaved load/store operations for SSE-AVX2.
7869 // As opposed to AVX-512, SSE-AVX2 do not have generic shuffles that allow
7870 // computing the cost using a generic formula as a function of generic
7871 // shuffles. We therefore use a lookup table instead, filled according to
7872 // the instruction sequences that codegen currently generates.
7873
7874 // VecTy for interleave memop is <VF*Factor x Elt>.
7875 // So, for VF=4, Interleave Factor = 3, Element type = i32 we have
7876 // VecTy = <12 x i32>.
7877 MVT LegalVT = getTypeLegalizationCost(VecTy).second;
7878
7879 // This function can be called with VecTy=<6xi128>, Factor=3, in which case
7880 // the VF=2, while v2i128 is an unsupported MVT vector type
7881 // (see MachineValueType.h::getVectorVT()).
7882 if (!LegalVT.isVector())
7883 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7884 Alignment, AddressSpace, CostKind);
7885
7886 unsigned VF = VecTy->getNumElements() / Factor;
7887 Type *ScalarTy = VecTy->getElementType();
7888 // Deduplicate entries, model floats/pointers as appropriately-sized integers.
7889 if (!ScalarTy->isIntegerTy())
7890 ScalarTy =
7891 Type::getIntNTy(ScalarTy->getContext(), DL.getTypeSizeInBits(ScalarTy));
7892
7893 // Get the cost of all the memory operations.
7894 // FIXME: discount dead loads.
7895 InstructionCost MemOpCosts =
7896 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
7897
7898 auto *VT = FixedVectorType::get(ScalarTy, VF);
7899 EVT ETy = TLI->getValueType(DL, VT);
7900 if (!ETy.isSimple())
7901 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
7902 Alignment, AddressSpace, CostKind);
7903
7904 // TODO: Complete for other data-types and strides.
7905 // Each combination of Stride, element bit width and VF results in a different
7906 // sequence; The cost tables are therefore accessed with:
7907 // Factor (stride) and VectorType=VFxiN.
7908 // The Cost accounts only for the shuffle sequence;
7909 // The cost of the loads/stores is accounted for separately.
7910 //
7911 static const CostTblEntry AVX2InterleavedLoadTbl[] = {
7912 {2, MVT::v2i8, 2}, // (load 4i8 and) deinterleave into 2 x 2i8
7913 {2, MVT::v4i8, 2}, // (load 8i8 and) deinterleave into 2 x 4i8
7914 {2, MVT::v8i8, 2}, // (load 16i8 and) deinterleave into 2 x 8i8
7915 {2, MVT::v16i8, 4}, // (load 32i8 and) deinterleave into 2 x 16i8
7916 {2, MVT::v32i8, 6}, // (load 64i8 and) deinterleave into 2 x 32i8
7917
7918 {2, MVT::v8i16, 6}, // (load 16i16 and) deinterleave into 2 x 8i16
7919 {2, MVT::v16i16, 9}, // (load 32i16 and) deinterleave into 2 x 16i16
7920 {2, MVT::v32i16, 18}, // (load 64i16 and) deinterleave into 2 x 32i16
7921
7922 {2, MVT::v8i32, 4}, // (load 16i32 and) deinterleave into 2 x 8i32
7923 {2, MVT::v16i32, 8}, // (load 32i32 and) deinterleave into 2 x 16i32
7924 {2, MVT::v32i32, 16}, // (load 64i32 and) deinterleave into 2 x 32i32
7925
7926 {2, MVT::v4i64, 4}, // (load 8i64 and) deinterleave into 2 x 4i64
7927 {2, MVT::v8i64, 8}, // (load 16i64 and) deinterleave into 2 x 8i64
7928 {2, MVT::v16i64, 16}, // (load 32i64 and) deinterleave into 2 x 16i64
7929 {2, MVT::v32i64, 32}, // (load 64i64 and) deinterleave into 2 x 32i64
7930
7931 {3, MVT::v2i8, 3}, // (load 6i8 and) deinterleave into 3 x 2i8
7932 {3, MVT::v4i8, 3}, // (load 12i8 and) deinterleave into 3 x 4i8
7933 {3, MVT::v8i8, 6}, // (load 24i8 and) deinterleave into 3 x 8i8
7934 {3, MVT::v16i8, 11}, // (load 48i8 and) deinterleave into 3 x 16i8
7935 {3, MVT::v32i8, 14}, // (load 96i8 and) deinterleave into 3 x 32i8
7936
7937 {3, MVT::v2i16, 5}, // (load 6i16 and) deinterleave into 3 x 2i16
7938 {3, MVT::v4i16, 7}, // (load 12i16 and) deinterleave into 3 x 4i16
7939 {3, MVT::v8i16, 9}, // (load 24i16 and) deinterleave into 3 x 8i16
7940 {3, MVT::v16i16, 28}, // (load 48i16 and) deinterleave into 3 x 16i16
7941 {3, MVT::v32i16, 56}, // (load 96i16 and) deinterleave into 3 x 32i16
7942
7943 {3, MVT::v2i32, 3}, // (load 6i32 and) deinterleave into 3 x 2i32
7944 {3, MVT::v4i32, 3}, // (load 12i32 and) deinterleave into 3 x 4i32
7945 {3, MVT::v8i32, 7}, // (load 24i32 and) deinterleave into 3 x 8i32
7946 {3, MVT::v16i32, 14}, // (load 48i32 and) deinterleave into 3 x 16i32
7947 {3, MVT::v32i32, 32}, // (load 96i32 and) deinterleave into 3 x 32i32
7948
7949 {3, MVT::v2i64, 1}, // (load 6i64 and) deinterleave into 3 x 2i64
7950 {3, MVT::v4i64, 5}, // (load 12i64 and) deinterleave into 3 x 4i64
7951 {3, MVT::v8i64, 10}, // (load 24i64 and) deinterleave into 3 x 8i64
7952 {3, MVT::v16i64, 20}, // (load 48i64 and) deinterleave into 3 x 16i64
7953
7954 {4, MVT::v2i8, 4}, // (load 8i8 and) deinterleave into 4 x 2i8
7955 {4, MVT::v4i8, 4}, // (load 16i8 and) deinterleave into 4 x 4i8
7956 {4, MVT::v8i8, 12}, // (load 32i8 and) deinterleave into 4 x 8i8
7957 {4, MVT::v16i8, 24}, // (load 64i8 and) deinterleave into 4 x 16i8
7958 {4, MVT::v32i8, 56}, // (load 128i8 and) deinterleave into 4 x 32i8
7959
7960 {4, MVT::v2i16, 6}, // (load 8i16 and) deinterleave into 4 x 2i16
7961 {4, MVT::v4i16, 17}, // (load 16i16 and) deinterleave into 4 x 4i16
7962 {4, MVT::v8i16, 33}, // (load 32i16 and) deinterleave into 4 x 8i16
7963 {4, MVT::v16i16, 75}, // (load 64i16 and) deinterleave into 4 x 16i16
7964 {4, MVT::v32i16, 150}, // (load 128i16 and) deinterleave into 4 x 32i16
7965
7966 {4, MVT::v2i32, 4}, // (load 8i32 and) deinterleave into 4 x 2i32
7967 {4, MVT::v4i32, 8}, // (load 16i32 and) deinterleave into 4 x 4i32
7968 {4, MVT::v8i32, 16}, // (load 32i32 and) deinterleave into 4 x 8i32
7969 {4, MVT::v16i32, 32}, // (load 64i32 and) deinterleave into 4 x 16i32
7970 {4, MVT::v32i32, 68}, // (load 128i32 and) deinterleave into 4 x 32i32
7971
7972 {4, MVT::v2i64, 6}, // (load 8i64 and) deinterleave into 4 x 2i64
7973 {4, MVT::v4i64, 8}, // (load 16i64 and) deinterleave into 4 x 4i64
7974 {4, MVT::v8i64, 20}, // (load 32i64 and) deinterleave into 4 x 8i64
7975 {4, MVT::v16i64, 40}, // (load 64i64 and) deinterleave into 4 x 16i64
7976
7977 {6, MVT::v2i8, 6}, // (load 12i8 and) deinterleave into 6 x 2i8
7978 {6, MVT::v4i8, 14}, // (load 24i8 and) deinterleave into 6 x 4i8
7979 {6, MVT::v8i8, 18}, // (load 48i8 and) deinterleave into 6 x 8i8
7980 {6, MVT::v16i8, 43}, // (load 96i8 and) deinterleave into 6 x 16i8
7981 {6, MVT::v32i8, 82}, // (load 192i8 and) deinterleave into 6 x 32i8
7982
7983 {6, MVT::v2i16, 13}, // (load 12i16 and) deinterleave into 6 x 2i16
7984 {6, MVT::v4i16, 9}, // (load 24i16 and) deinterleave into 6 x 4i16
7985 {6, MVT::v8i16, 39}, // (load 48i16 and) deinterleave into 6 x 8i16
7986 {6, MVT::v16i16, 106}, // (load 96i16 and) deinterleave into 6 x 16i16
7987 {6, MVT::v32i16, 212}, // (load 192i16 and) deinterleave into 6 x 32i16
7988
7989 {6, MVT::v2i32, 6}, // (load 12i32 and) deinterleave into 6 x 2i32
7990 {6, MVT::v4i32, 15}, // (load 24i32 and) deinterleave into 6 x 4i32
7991 {6, MVT::v8i32, 31}, // (load 48i32 and) deinterleave into 6 x 8i32
7992 {6, MVT::v16i32, 64}, // (load 96i32 and) deinterleave into 6 x 16i32
7993
7994 {6, MVT::v2i64, 6}, // (load 12i64 and) deinterleave into 6 x 2i64
7995 {6, MVT::v4i64, 18}, // (load 24i64 and) deinterleave into 6 x 4i64
7996 {6, MVT::v8i64, 36}, // (load 48i64 and) deinterleave into 6 x 8i64
7997
7998 {8, MVT::v8i32, 40} // (load 64i32 and) deinterleave into 8 x 8i32
7999 };
8000
8001 static const CostTblEntry SSSE3InterleavedLoadTbl[] = {
8002 {2, MVT::v4i16, 2}, // (load 8i16 and) deinterleave into 2 x 4i16
8003 };
8004
8005 static const CostTblEntry SSE2InterleavedLoadTbl[] = {
8006 {2, MVT::v2i16, 2}, // (load 4i16 and) deinterleave into 2 x 2i16
8007 {2, MVT::v4i16, 7}, // (load 8i16 and) deinterleave into 2 x 4i16
8008
8009 {2, MVT::v2i32, 2}, // (load 4i32 and) deinterleave into 2 x 2i32
8010 {2, MVT::v4i32, 2}, // (load 8i32 and) deinterleave into 2 x 4i32
8011
8012 {2, MVT::v2i64, 2}, // (load 4i64 and) deinterleave into 2 x 2i64
8013 };
8014
8015 static const CostTblEntry AVX2InterleavedStoreTbl[] = {
8016 {2, MVT::v16i8, 3}, // interleave 2 x 16i8 into 32i8 (and store)
8017 {2, MVT::v32i8, 4}, // interleave 2 x 32i8 into 64i8 (and store)
8018
8019 {2, MVT::v8i16, 3}, // interleave 2 x 8i16 into 16i16 (and store)
8020 {2, MVT::v16i16, 4}, // interleave 2 x 16i16 into 32i16 (and store)
8021 {2, MVT::v32i16, 8}, // interleave 2 x 32i16 into 64i16 (and store)
8022
8023 {2, MVT::v4i32, 2}, // interleave 2 x 4i32 into 8i32 (and store)
8024 {2, MVT::v8i32, 4}, // interleave 2 x 8i32 into 16i32 (and store)
8025 {2, MVT::v16i32, 8}, // interleave 2 x 16i32 into 32i32 (and store)
8026 {2, MVT::v32i32, 16}, // interleave 2 x 32i32 into 64i32 (and store)
8027
8028 {2, MVT::v2i64, 2}, // interleave 2 x 2i64 into 4i64 (and store)
8029 {2, MVT::v4i64, 4}, // interleave 2 x 4i64 into 8i64 (and store)
8030 {2, MVT::v8i64, 8}, // interleave 2 x 8i64 into 16i64 (and store)
8031 {2, MVT::v16i64, 16}, // interleave 2 x 16i64 into 32i64 (and store)
8032 {2, MVT::v32i64, 32}, // interleave 2 x 32i64 into 64i64 (and store)
8033
8034 {3, MVT::v2i8, 4}, // interleave 3 x 2i8 into 6i8 (and store)
8035 {3, MVT::v4i8, 4}, // interleave 3 x 4i8 into 12i8 (and store)
8036 {3, MVT::v8i8, 6}, // interleave 3 x 8i8 into 24i8 (and store)
8037 {3, MVT::v16i8, 11}, // interleave 3 x 16i8 into 48i8 (and store)
8038 {3, MVT::v32i8, 13}, // interleave 3 x 32i8 into 96i8 (and store)
8039
8040 {3, MVT::v2i16, 4}, // interleave 3 x 2i16 into 6i16 (and store)
8041 {3, MVT::v4i16, 6}, // interleave 3 x 4i16 into 12i16 (and store)
8042 {3, MVT::v8i16, 12}, // interleave 3 x 8i16 into 24i16 (and store)
8043 {3, MVT::v16i16, 27}, // interleave 3 x 16i16 into 48i16 (and store)
8044 {3, MVT::v32i16, 54}, // interleave 3 x 32i16 into 96i16 (and store)
8045
8046 {3, MVT::v2i32, 4}, // interleave 3 x 2i32 into 6i32 (and store)
8047 {3, MVT::v4i32, 5}, // interleave 3 x 4i32 into 12i32 (and store)
8048 {3, MVT::v8i32, 11}, // interleave 3 x 8i32 into 24i32 (and store)
8049 {3, MVT::v16i32, 22}, // interleave 3 x 16i32 into 48i32 (and store)
8050 {3, MVT::v32i32, 48}, // interleave 3 x 32i32 into 96i32 (and store)
8051
8052 {3, MVT::v2i64, 4}, // interleave 3 x 2i64 into 6i64 (and store)
8053 {3, MVT::v4i64, 6}, // interleave 3 x 4i64 into 12i64 (and store)
8054 {3, MVT::v8i64, 12}, // interleave 3 x 8i64 into 24i64 (and store)
8055 {3, MVT::v16i64, 24}, // interleave 3 x 16i64 into 48i64 (and store)
8056
8057 {4, MVT::v2i8, 4}, // interleave 4 x 2i8 into 8i8 (and store)
8058 {4, MVT::v4i8, 4}, // interleave 4 x 4i8 into 16i8 (and store)
8059 {4, MVT::v8i8, 4}, // interleave 4 x 8i8 into 32i8 (and store)
8060 {4, MVT::v16i8, 8}, // interleave 4 x 16i8 into 64i8 (and store)
8061 {4, MVT::v32i8, 12}, // interleave 4 x 32i8 into 128i8 (and store)
8062
8063 {4, MVT::v2i16, 2}, // interleave 4 x 2i16 into 8i16 (and store)
8064 {4, MVT::v4i16, 6}, // interleave 4 x 4i16 into 16i16 (and store)
8065 {4, MVT::v8i16, 10}, // interleave 4 x 8i16 into 32i16 (and store)
8066 {4, MVT::v16i16, 32}, // interleave 4 x 16i16 into 64i16 (and store)
8067 {4, MVT::v32i16, 64}, // interleave 4 x 32i16 into 128i16 (and store)
8068
8069 {4, MVT::v2i32, 5}, // interleave 4 x 2i32 into 8i32 (and store)
8070 {4, MVT::v4i32, 6}, // interleave 4 x 4i32 into 16i32 (and store)
8071 {4, MVT::v8i32, 16}, // interleave 4 x 8i32 into 32i32 (and store)
8072 {4, MVT::v16i32, 32}, // interleave 4 x 16i32 into 64i32 (and store)
8073 {4, MVT::v32i32, 64}, // interleave 4 x 32i32 into 128i32 (and store)
8074
8075 {4, MVT::v2i64, 6}, // interleave 4 x 2i64 into 8i64 (and store)
8076 {4, MVT::v4i64, 8}, // interleave 4 x 4i64 into 16i64 (and store)
8077 {4, MVT::v8i64, 20}, // interleave 4 x 8i64 into 32i64 (and store)
8078 {4, MVT::v16i64, 40}, // interleave 4 x 16i64 into 64i64 (and store)
8079
8080 {6, MVT::v2i8, 7}, // interleave 6 x 2i8 into 12i8 (and store)
8081 {6, MVT::v4i8, 9}, // interleave 6 x 4i8 into 24i8 (and store)
8082 {6, MVT::v8i8, 16}, // interleave 6 x 8i8 into 48i8 (and store)
8083 {6, MVT::v16i8, 27}, // interleave 6 x 16i8 into 96i8 (and store)
8084 {6, MVT::v32i8, 90}, // interleave 6 x 32i8 into 192i8 (and store)
8085
8086 {6, MVT::v2i16, 10}, // interleave 6 x 2i16 into 12i16 (and store)
8087 {6, MVT::v4i16, 15}, // interleave 6 x 4i16 into 24i16 (and store)
8088 {6, MVT::v8i16, 21}, // interleave 6 x 8i16 into 48i16 (and store)
8089 {6, MVT::v16i16, 58}, // interleave 6 x 16i16 into 96i16 (and store)
8090 {6, MVT::v32i16, 90}, // interleave 6 x 32i16 into 192i16 (and store)
8091
8092 {6, MVT::v2i32, 9}, // interleave 6 x 2i32 into 12i32 (and store)
8093 {6, MVT::v4i32, 12}, // interleave 6 x 4i32 into 24i32 (and store)
8094 {6, MVT::v8i32, 33}, // interleave 6 x 8i32 into 48i32 (and store)
8095 {6, MVT::v16i32, 66}, // interleave 6 x 16i32 into 96i32 (and store)
8096
8097 {6, MVT::v2i64, 8}, // interleave 6 x 2i64 into 12i64 (and store)
8098 {6, MVT::v4i64, 15}, // interleave 6 x 4i64 into 24i64 (and store)
8099 {6, MVT::v8i64, 30}, // interleave 6 x 8i64 into 48i64 (and store)
8100 };
8101
8102 static const CostTblEntry SSE2InterleavedStoreTbl[] = {
8103 {2, MVT::v2i8, 1}, // interleave 2 x 2i8 into 4i8 (and store)
8104 {2, MVT::v4i8, 1}, // interleave 2 x 4i8 into 8i8 (and store)
8105 {2, MVT::v8i8, 1}, // interleave 2 x 8i8 into 16i8 (and store)
8106
8107 {2, MVT::v2i16, 1}, // interleave 2 x 2i16 into 4i16 (and store)
8108 {2, MVT::v4i16, 1}, // interleave 2 x 4i16 into 8i16 (and store)
8109
8110 {2, MVT::v2i32, 1}, // interleave 2 x 2i32 into 4i32 (and store)
8111 };
8112
8113 if (Opcode == Instruction::Load) {
8114 auto GetDiscountedCost = [Factor, NumMembers = Indices.size(),
8115 MemOpCosts](const CostTblEntry *Entry) {
8116 // NOTE: this is just an approximation!
8117 // It can over/under -estimate the cost!
8118 return MemOpCosts + divideCeil(NumMembers * Entry->Cost, Factor);
8119 };
8120
8121 if (ST->hasAVX2())
8122 if (const auto *Entry = CostTableLookup(AVX2InterleavedLoadTbl, Factor,
8123 ETy.getSimpleVT()))
8124 return GetDiscountedCost(Entry);
8125
8126 if (ST->hasSSSE3())
8127 if (const auto *Entry = CostTableLookup(SSSE3InterleavedLoadTbl, Factor,
8128 ETy.getSimpleVT()))
8129 return GetDiscountedCost(Entry);
8130
8131 if (ST->hasSSE2())
8132 if (const auto *Entry = CostTableLookup(SSE2InterleavedLoadTbl, Factor,
8133 ETy.getSimpleVT()))
8134 return GetDiscountedCost(Entry);
8135 } else {
8136 assert(Opcode == Instruction::Store &&
8137 "Expected Store Instruction at this point");
8138 assert((!Indices.size() || Indices.size() == Factor) &&
8139 "Interleaved store only supports fully-interleaved groups.");
8140 if (ST->hasAVX2())
8141 if (const auto *Entry = CostTableLookup(AVX2InterleavedStoreTbl, Factor,
8142 ETy.getSimpleVT()))
8143 return MemOpCosts + Entry->Cost;
8144
8145 if (ST->hasSSE2())
8146 if (const auto *Entry = CostTableLookup(SSE2InterleavedStoreTbl, Factor,
8147 ETy.getSimpleVT()))
8148 return MemOpCosts + Entry->Cost;
8149 }
8150
8151 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
8152 Alignment, AddressSpace, CostKind,
8153 UseMaskForCond, UseMaskForGaps);
8154}
8155
8157 StackOffset BaseOffset,
8158 bool HasBaseReg, int64_t Scale,
8159 unsigned AddrSpace) const {
8160 // Scaling factors are not free at all.
8161 // An indexed folded instruction, i.e., inst (reg1, reg2, scale),
8162 // will take 2 allocations in the out of order engine instead of 1
8163 // for plain addressing mode, i.e. inst (reg1).
8164 // E.g.,
8165 // vaddps (%rsi,%rdx), %ymm0, %ymm1
8166 // Requires two allocations (one for the load, one for the computation)
8167 // whereas:
8168 // vaddps (%rsi), %ymm0, %ymm1
8169 // Requires just 1 allocation, i.e., freeing allocations for other operations
8170 // and having less micro operations to execute.
8171 //
8172 // For some X86 architectures, this is even worse because for instance for
8173 // stores, the complex addressing mode forces the instruction to use the
8174 // "load" ports instead of the dedicated "store" port.
8175 // E.g., on Haswell:
8176 // vmovaps %ymm1, (%r8, %rdi) can use port 2 or 3.
8177 // vmovaps %ymm1, (%r8) can use port 2, 3, or 7.
8179 AM.BaseGV = BaseGV;
8180 AM.BaseOffs = BaseOffset.getFixed();
8181 AM.HasBaseReg = HasBaseReg;
8182 AM.Scale = Scale;
8183 AM.ScalableOffset = BaseOffset.getScalable();
8184 if (getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace))
8185 // Scale represents reg2 * scale, thus account for 1
8186 // as soon as we use a second register.
8187 return AM.Scale != 0;
8189}
8190
8192 // TODO: Hook MispredictPenalty of SchedMachineModel into this.
8193 return 14;
8194}
8195
8197 unsigned Bits = Ty->getScalarSizeInBits();
8198
8199 // XOP has v16i8/v8i16/v4i32/v2i64 variable vector shifts.
8200 // Splitting for v32i8/v16i16 on XOP+AVX2 targets is still preferred.
8201 if (ST->hasXOP() && (Bits == 8 || Bits == 16 || Bits == 32 || Bits == 64))
8202 return false;
8203
8204 // AVX2 has vpsllv[dq] instructions (and other shifts) that make variable
8205 // shifts just as cheap as scalar ones.
8206 if (ST->hasAVX2() && (Bits == 32 || Bits == 64))
8207 return false;
8208
8209 // AVX512BW has shifts such as vpsllvw.
8210 if (ST->hasBWI() && Bits == 16)
8211 return false;
8212
8213 // Otherwise, it's significantly cheaper to shift by a scalar amount than by a
8214 // fully general vector.
8215 return true;
8216}
8217
8218unsigned X86TTIImpl::getStoreMinimumVF(unsigned VF, Type *ScalarMemTy,
8219 Type *ScalarValTy, Align Alignment,
8220 unsigned AddrSpace) const {
8221 if (ST->hasF16C() && ScalarMemTy->isHalfTy()) {
8222 return 4;
8223 }
8224 return BaseT::getStoreMinimumVF(VF, ScalarMemTy, ScalarValTy, Alignment,
8225 AddrSpace);
8226}
8227
8229 SmallVectorImpl<Use *> &Ops) const {
8230 using namespace llvm::PatternMatch;
8231
8232 if (I->getOpcode() == Instruction::And &&
8233 (ST->hasBMI() || (I->getType()->isVectorTy() && ST->hasSSE2()))) {
8234 for (auto &Op : I->operands()) {
8235 // (and X, (not Y)) -> (andn X, Y)
8236 if (match(Op.get(), m_Not(m_Value())) && !I->getType()->isIntegerTy(8)) {
8237 Ops.push_back(&Op);
8238 return true;
8239 }
8240 // (and X, (splat (not Y))) -> (andn X, (splat Y))
8241 if (match(Op.get(),
8243 m_Value(), m_ZeroMask()))) {
8244 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
8245 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
8246 Ops.push_back(&Not);
8247 Ops.push_back(&InsertElt);
8248 Ops.push_back(&Op);
8249 return true;
8250 }
8251 }
8252 }
8253
8254 FixedVectorType *VTy = dyn_cast<FixedVectorType>(I->getType());
8255 if (!VTy)
8256 return false;
8257
8258 if (I->getOpcode() == Instruction::Mul &&
8259 VTy->getElementType()->isIntegerTy(64)) {
8260 for (auto &Op : I->operands()) {
8261 // Make sure we are not already sinking this operand
8262 if (any_of(Ops, [&](Use *U) { return U->get() == Op; }))
8263 continue;
8264
8265 // Look for PMULDQ pattern where the input is a sext_inreg from vXi32 or
8266 // the PMULUDQ pattern where the input is a zext_inreg from vXi32.
8267 if (ST->hasSSE41() &&
8268 match(Op.get(), m_AShr(m_Shl(m_Value(), m_SpecificInt(32)),
8269 m_SpecificInt(32)))) {
8270 Ops.push_back(&cast<Instruction>(Op)->getOperandUse(0));
8271 Ops.push_back(&Op);
8272 } else if (ST->hasSSE2() &&
8273 match(Op.get(),
8274 m_And(m_Value(), m_SpecificInt(UINT64_C(0xffffffff))))) {
8275 Ops.push_back(&Op);
8276 }
8277 }
8278
8279 return !Ops.empty();
8280 }
8281
8282 // A uniform shift amount in a vector shift or funnel shift may be much
8283 // cheaper than a generic variable vector shift, so make that pattern visible
8284 // to SDAG by sinking the shuffle instruction next to the shift.
8285 int ShiftAmountOpNum = -1;
8286 if (I->isShift())
8287 ShiftAmountOpNum = 1;
8288 else if (auto *II = dyn_cast<IntrinsicInst>(I)) {
8289 if (II->getIntrinsicID() == Intrinsic::fshl ||
8290 II->getIntrinsicID() == Intrinsic::fshr)
8291 ShiftAmountOpNum = 2;
8292 }
8293
8294 if (ShiftAmountOpNum == -1)
8295 return false;
8296
8297 auto *Shuf = dyn_cast<ShuffleVectorInst>(I->getOperand(ShiftAmountOpNum));
8298 if (Shuf && getSplatIndex(Shuf->getShuffleMask()) >= 0 &&
8299 isVectorShiftByScalarCheap(I->getType())) {
8300 Ops.push_back(&I->getOperandUse(ShiftAmountOpNum));
8301 return true;
8302 }
8303
8304 return false;
8305}
8306
8308 bool HasEGPR = ST->hasEGPR();
8309 const TargetMachine &TM = getTLI()->getTargetMachine();
8310
8311 for (User *U : F.users()) {
8313 if (!CB || CB->getCalledOperand() != &F)
8314 continue;
8315 Function *CallerFunc = CB->getFunction();
8316 if (TM.getSubtarget<X86Subtarget>(*CallerFunc).hasEGPR() != HasEGPR)
8317 return false;
8318 }
8319
8320 return true;
8321}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Expand Atomic instructions
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
iv users
Definition IVUsers.cpp:48
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
#define P(N)
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
CostTblEntryT< CostKindCosts > CostKindTblEntry
static bool isFoldedIntoRsqrtEstimate(const IntrinsicCostAttributes &ICA, const X86TargetLowering &TLI, const DataLayout &DL)
Returns true if the square root is divided into with the reciprocal allowed and the target replaces t...
static InstructionCost getPredicateFanoutCost(const X86TTIImpl &TTI, Type *DataVTy, Type *MaskVTy)
static bool isLegalMaskedLoadStore(Type *ScalarTy, const X86Subtarget *ST)
TypeConversionCostTblEntryT< CostKindCosts > TypeConversionCostKindTblEntry
This file a TargetTransformInfoImplBase conforming object specific to the X86 target machine.
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static LLVM_ABI unsigned int semanticsPrecision(const fltSemantics &)
Definition APFloat.cpp:329
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1086
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:829
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
LLVM_ABI APInt extractBits(unsigned numBits, unsigned bitPosition) const
Return an APInt with the extracted bits [bitPosition,bitPosition+numBits).
Definition APInt.cpp:478
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy, Align Alignment, unsigned AddrSpace) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF, const APInt &DemandedDstElts, TTI::TargetCostKind CostKind) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Value * getCalledOperand() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Container class for subtarget features.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
static unsigned getPointerOperandIndex()
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Machine Value Type.
bool is128BitVector() const
Return true if this is a 128-bit vector type.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
bool isScalarInteger() const
Return true if this is an integer, not including vectors.
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
This class represents an analyzed expression in the program.
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
bool test(unsigned Idx) const
Returns true if bit Idx is set.
size_type size() const
Returns the number of bits in this bitvector.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
static StackOffset getScalable(int64_t Scalable)
Definition TypeSize.h:40
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
int getRecipEstimateSqrtEnabled(EVT VT, const Function &F) const
Return a ReciprocalEstimate enum value for a square root of the given type based on the function's at...
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
Primary interface to the complete machine description for the target machine.
const STC & getSubtarget(const Function &F) const
This method returns a pointer to the specified type of TargetSubtargetInfo.
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
virtual const TargetLowering * getTargetLowering() const
bool isStridedAccess(const SCEV *Ptr) const
unsigned minRequiredElementSize(const Value *Val, bool &isSigned) const
const SCEVConstant * getConstantStrideStep(ScalarEvolution *SE, const SCEV *Ptr) const
virtual bool isExpensiveToSpeculativelyExecute(const Instruction *I) const
virtual InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const
MaskKind
Some targets only support masked load/store with a constant mask.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
CacheLevel
The possible cache levels.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
LLVM_ABI unsigned getIntegerBitWidth() const
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Definition Type.h:155
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
Definition Type.h:158
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
static LLVM_ABI Type * getDoubleTy(LLVMContext &C)
Definition Type.cpp:277
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:222
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:392
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
Base class of all SIMD vector types.
static VectorType * getExtendedElementVectorType(VectorType *VTy)
This static method is like getInteger except that the element types are twice as wide as the elements...
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
static VectorType * getDoubleElementsVectorType(VectorType *VTy)
This static method returns a VectorType with twice as many elements as the input type and the same el...
Type * getElementType() const
bool useAVX512Regs() const
bool hasAVX512() const
bool hasAVX2() const
bool useFastCCForInternalCall(Function &F) const override
bool isLegalAltInstr(VectorType *VecTy, unsigned Opcode0, unsigned Opcode1, const SmallBitVector &OpcodeMask, ArrayRef< const Value * > Scalars) const override
InstructionCost getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF, const APInt &DemandedDstElts, TTI::TargetCostKind CostKind) const override
bool isLegalNTLoad(Type *DataType, Align Alignment) const override
std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
unsigned getRegisterClassForType(bool Vector, Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool isLegalNTStore(Type *DataType, Align Alignment) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getInterleavedMemoryOpCostAVX512(unsigned Opcode, FixedVectorType *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isVectorShiftByScalarCheap(Type *Ty) const override
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
bool shouldExpandReduction(const IntrinsicInst *II) const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
unsigned getAtomicMemIntrinsicMaxElementSize() const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool forceScalarizeMaskedGather(VectorType *VTy, Align Alignment) const override
InstructionCost getBranchMispredictPenalty() const override
bool isExpensiveToSpeculativelyExecute(const Instruction *I) const override
bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const override
InstructionCost getAltInstrCost(VectorType *VecTy, unsigned Opcode0, unsigned Opcode1, const SmallBitVector &OpcodeMask, TTI::TargetCostKind CostKind, ArrayRef< const Value * > Scalars) const override
bool isLegalMaskedStore(Type *DataType, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind=TTI::MaskKind::VariableOrConstantMask) const override
std::optional< unsigned > getCacheSize(TargetTransformInfo::CacheLevel Level) const override
bool isLegalMaskedGatherScatter(Type *DataType, Align Alignment) const
bool isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind=TTI::MaskKind::VariableOrConstantMask) const override
bool enableInterleavedAccessVectorization() const override
unsigned getLoadStoreVecRegBitWidth(unsigned AS) const override
unsigned getNumberOfRegisters(unsigned ClassID) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy, Align Alignment, unsigned AddrSpace) const override
bool hasDivRemOp(Type *DataType, bool IsSigned) const override
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool supportsEfficientVectorElementLoadStore() const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
InstructionCost getIntImmCost(int64_t) const
Calculate the cost of materializing a 64-bit value.
InstructionCost getMinMaxCost(Intrinsic::ID IID, Type *Ty, TTI::TargetCostKind CostKind, FastMathFlags FMF) const
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool canMacroFuseCmp() const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool prefersVectorizedAddressing() const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Type) const override
bool forceScalarizeMaskedScatter(VectorType *VTy, Align Alignment) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
Calculate the cost of Gather / Scatter operation.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool hasSqrtEstimate(EVT VT, bool Reciprocal) const
Return true if VT has the rsqrt* based estimate of the square root, or of its reciprocal if Reciproca...
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt ScaleBitMask(const APInt &A, unsigned NewBitWidth, bool MatchAllBits=false)
Splat/Merge neighboring bits to widen/narrow the bitmask represented by.
Definition APInt.cpp:3043
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:26
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ DELETED_NODE
DELETED_NODE - This is an illegal value that is used to catch errors.
Definition ISDOpcodes.h:47
@ PARTIAL_REDUCE_SMLA
PARTIAL_REDUCE_[U|S]MLA(Accumulator, Input1, Input2) The partial reduction nodes sign or zero extend ...
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ CLMUL
Carry-less multiplication operations.
Definition ISDOpcodes.h:788
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ PARTIAL_REDUCE_UMLA
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ PARTIAL_REDUCE_FMLA
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:351
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
Definition ISDOpcodes.h:359
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:805
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ PARTIAL_REDUCE_SUMLA
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::FDiv > m_FDiv(const LHS &L, const RHS &R)
AllowFmf_match< T, FastMathFlags::AllowContract > m_AllowContract(const T &SubPattern)
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
This is an optimization pass for GlobalISel generic memory operations.
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
Definition STLExtras.h:2196
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
@ Known
Known to have no common set bits.
LLVM_ABI void ComputeValueVTs(const TargetLowering &TLI, const DataLayout &DL, Type *Ty, SmallVectorImpl< EVT > &ValueVTs, SmallVectorImpl< EVT > *MemVTs=nullptr, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
ComputeValueVTs - Given an LLVM IR type, compute a sequence of EVTs that represent all the individual...
Definition Analysis.cpp:121
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI unsigned ComputeNumSignBits(const Value *Op, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Return the number of times the sign bit of the register is replicated into the other bits.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1901
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
Definition STLExtras.h:2035
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
Definition CostTable.h:67
LLVM_ABI int getSplatIndex(ArrayRef< int > Mask)
If all non-negative Mask elements are the same value, return that value.
#define N
std::optional< unsigned > operator[](TargetTransformInfo::TargetCostKind Kind) const
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Cost Table Entry.
Definition CostTable.h:26
Extended Value Type.
Definition ValueTypes.h:35
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Type Conversion Cost Table.
Definition CostTable.h:56