47#define DEBUG_TYPE "vector-combine"
53STATISTIC(NumVecLoad,
"Number of vector loads formed");
54STATISTIC(NumVecCmp,
"Number of vector compares formed");
55STATISTIC(NumVecBO,
"Number of vector binops formed");
56STATISTIC(NumVecCmpBO,
"Number of vector compare + binop formed");
57STATISTIC(NumShufOfBitcast,
"Number of shuffles moved after bitcast");
58STATISTIC(NumScalarOps,
"Number of scalar unary + binary ops formed");
59STATISTIC(NumScalarCmp,
"Number of scalar compares formed");
60STATISTIC(NumScalarIntrinsic,
"Number of scalar intrinsic calls formed");
64 cl::desc(
"Disable all vector combine transforms"));
68 cl::desc(
"Disable binop extract to shuffle transforms"));
72 cl::desc(
"Max number of instructions to scan for vector combining."));
74static const unsigned InvalidIndex = std::numeric_limits<unsigned>::max();
82 bool TryEarlyFoldsOnly)
85 SQ(*
DL, nullptr, &DT, &AC),
86 TryEarlyFoldsOnly(TryEarlyFoldsOnly) {}
93 const TargetTransformInfo &TTI;
94 const DominatorTree &DT;
98 const SimplifyQuery SQ;
102 bool TryEarlyFoldsOnly;
104 InstructionWorklist Worklist;
113 bool vectorizeLoadInsert(Instruction &
I);
114 bool widenSubvectorLoad(Instruction &
I);
115 ExtractElementInst *getShuffleExtract(ExtractElementInst *Ext0,
116 ExtractElementInst *Ext1,
117 unsigned PreferredExtractIndex)
const;
118 bool isExtractExtractCheap(ExtractElementInst *Ext0, ExtractElementInst *Ext1,
119 const Instruction &
I,
120 ExtractElementInst *&ConvertToShuffle,
121 unsigned PreferredExtractIndex);
124 bool foldExtractExtract(Instruction &
I);
125 bool foldInsExtFNeg(Instruction &
I);
126 bool foldInsExtBinop(Instruction &
I);
127 bool foldInsExtVectorToShuffle(Instruction &
I);
128 bool foldBitOpOfCastops(Instruction &
I);
129 bool foldBitOpOfCastConstant(Instruction &
I);
130 bool foldBitcastShuffle(Instruction &
I);
131 bool scalarizeOpOrCmp(Instruction &
I);
132 bool foldExtractedCmps(Instruction &
I);
133 bool foldSelectsFromBitcast(Instruction &
I);
134 bool foldBinopOfReductions(Instruction &
I);
135 bool foldSingleElementStore(Instruction &
I);
136 bool scalarizeLoad(Instruction &
I);
137 bool scalarizeLoadExtract(LoadInst *LI, VectorType *VecTy,
Value *Ptr);
138 bool scalarizeLoadBitcast(LoadInst *LI, VectorType *VecTy,
Value *Ptr);
139 bool scalarizeExtExtract(Instruction &
I);
140 bool foldConcatOfBoolMasks(Instruction &
I);
141 bool foldPermuteOfBinops(Instruction &
I);
142 bool foldShuffleOfBinops(Instruction &
I);
143 bool foldShuffleOfSelects(Instruction &
I);
144 bool foldShuffleOfCastops(Instruction &
I);
145 bool foldShuffleOfShuffles(Instruction &
I);
146 bool foldPermuteOfIntrinsic(Instruction &
I);
147 bool foldShufflesOfLengthChangingShuffles(Instruction &
I);
148 bool foldShuffleOfIntrinsics(Instruction &
I);
149 bool foldShuffleToIdentity(Instruction &
I);
150 bool foldShuffleFromReductions(Instruction &
I);
151 bool foldShuffleChainsToReduce(Instruction &
I);
152 bool foldCastFromReductions(Instruction &
I);
153 bool foldSignBitReductionCmp(Instruction &
I);
154 bool foldReductionZeroTest(Instruction &
I);
155 bool foldICmpEqZeroVectorReduce(Instruction &
I);
156 bool foldEquivalentReductionCmp(Instruction &
I);
157 bool foldReduceAddCmpZero(Instruction &
I);
158 bool foldSelectShuffle(Instruction &
I,
bool FromReduction =
false);
159 bool foldInterleaveIntrinsics(Instruction &
I);
160 bool foldDeinterleaveIntrinsics(Instruction &
I);
161 bool foldBitcastOfVPLoad(Instruction &
I);
162 bool foldBitOrderReverseAndSwap(Instruction &
I);
163 bool shrinkType(Instruction &
I);
164 bool shrinkLoadForShuffles(Instruction &
I);
165 bool shrinkPhiOfShuffles(Instruction &
I);
166 bool foldDeinterleaveInterleavePair(Instruction &
I);
168 void replaceValue(Instruction &Old,
Value &New,
bool Erase =
true) {
174 Worklist.pushUsersToWorkList(*NewI);
175 Worklist.pushValue(NewI);
192 SmallPtrSet<Value *, 4> Visited;
197 OpI,
nullptr,
nullptr, [&](
Value *V) {
202 NextInst = NextInst->getNextNode();
207 Worklist.pushUsersToWorkList(*OpI);
208 Worklist.pushValue(OpI);
226 return X->getType() ==
Y->getType() &&
235 Load->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag) ||
241 Type *ScalarTy =
Load->getType()->getScalarType();
243 unsigned MinVectorSize =
TTI.getMinVectorRegisterBitWidth();
244 if (!ScalarSize || !MinVectorSize || MinVectorSize % ScalarSize != 0 ||
251bool VectorCombine::vectorizeLoadInsert(
Instruction &
I) {
277 Value *SrcPtr =
Load->getPointerOperand()->stripPointerCasts();
280 unsigned MinVecNumElts = MinVectorSize / ScalarSize;
281 auto *MinVecTy = VectorType::get(ScalarTy, MinVecNumElts,
false);
282 unsigned OffsetEltIndex = 0;
290 unsigned OffsetBitWidth =
DL->getIndexTypeSizeInBits(SrcPtr->
getType());
291 APInt
Offset(OffsetBitWidth, 0);
301 uint64_t ScalarSizeInBytes = ScalarSize / 8;
302 if (
Offset.urem(ScalarSizeInBytes) != 0)
306 APInt OffsetEltIndexAP =
Offset.udiv(ScalarSizeInBytes);
307 if (OffsetEltIndexAP.
uge(MinVecNumElts))
325 unsigned AS =
Load->getPointerAddressSpace();
344 unsigned OutputNumElts = Ty->getNumElements();
346 assert(OffsetEltIndex < MinVecNumElts &&
"Address offset too big");
347 Mask[0] = OffsetEltIndex;
354 if (OldCost < NewCost || !NewCost.
isValid())
365 replaceValue(
I, *VecLd);
373bool VectorCombine::widenSubvectorLoad(Instruction &
I) {
376 if (!Shuf->isIdentityWithPadding())
382 unsigned OpIndex =
any_of(Shuf->getShuffleMask(), [&NumOpElts](
int M) {
383 return M >= (int)(NumOpElts);
403 unsigned AS =
Load->getPointerAddressSpace();
418 if (OldCost < NewCost || !NewCost.
isValid())
425 replaceValue(
I, *VecLd);
432ExtractElementInst *VectorCombine::getShuffleExtract(
433 ExtractElementInst *Ext0, ExtractElementInst *Ext1,
437 assert(Index0C && Index1C &&
"Expected constant extract indexes");
439 unsigned Index0 = Index0C->getZExtValue();
440 unsigned Index1 = Index1C->getZExtValue();
443 if (Index0 == Index1)
467 if (PreferredExtractIndex == Index0)
469 if (PreferredExtractIndex == Index1)
473 return Index0 > Index1 ? Ext0 : Ext1;
481bool VectorCombine::isExtractExtractCheap(ExtractElementInst *Ext0,
482 ExtractElementInst *Ext1,
483 const Instruction &
I,
484 ExtractElementInst *&ConvertToShuffle,
485 unsigned PreferredExtractIndex) {
488 assert(Ext0IndexC && Ext1IndexC &&
"Expected constant extract indexes");
490 unsigned Opcode =
I.getOpcode();
503 assert((Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) &&
504 "Expected a compare");
514 unsigned Ext0Index = Ext0IndexC->getZExtValue();
515 unsigned Ext1Index = Ext1IndexC->getZExtValue();
529 unsigned BestExtIndex = Extract0Cost > Extract1Cost ? Ext0Index : Ext1Index;
530 unsigned BestInsIndex = Extract0Cost > Extract1Cost ? Ext1Index : Ext0Index;
531 InstructionCost CheapExtractCost = std::min(Extract0Cost, Extract1Cost);
536 if (Ext0Src == Ext1Src && Ext0Index == Ext1Index) {
541 bool HasUseTax = Ext0 == Ext1 ? !Ext0->
hasNUses(2)
543 OldCost = CheapExtractCost + ScalarOpCost;
544 NewCost = VectorOpCost + CheapExtractCost + HasUseTax * CheapExtractCost;
548 OldCost = Extract0Cost + Extract1Cost + ScalarOpCost;
549 NewCost = VectorOpCost + CheapExtractCost +
554 ConvertToShuffle = getShuffleExtract(Ext0, Ext1, PreferredExtractIndex);
555 if (ConvertToShuffle) {
567 SmallVector<int> ShuffleMask(FixedVecTy->getNumElements(),
569 ShuffleMask[BestInsIndex] = BestExtIndex;
571 VecTy, VecTy, ShuffleMask,
CostKind, 0,
572 nullptr, {ConvertToShuffle});
575 VecTy, VecTy, {},
CostKind, 0,
nullptr,
580 LLVM_DEBUG(
dbgs() <<
"Found a binop of extractions: " <<
I <<
"\n OldCost: "
581 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
586 return OldCost < NewCost;
598 ShufMask[NewIndex] = OldIndex;
599 return Builder.CreateShuffleVector(Vec, ShufMask,
"shift");
651 V1,
"foldExtExtBinop");
656 VecBOInst->copyIRFlags(&
I);
662bool VectorCombine::foldExtractExtract(Instruction &
I) {
683 unsigned NumElts = FixedVecTy->getNumElements();
684 if (C0 >= NumElts || C1 >= NumElts)
700 ExtractElementInst *ExtractToChange;
701 if (isExtractExtractCheap(Ext0, Ext1,
I, ExtractToChange, InsertIndex))
707 if (ExtractToChange) {
708 unsigned CheapExtractIdx = ExtractToChange == Ext0 ? C1 : C0;
713 if (ExtractToChange == Ext0)
722 ? foldExtExtCmp(ExtOp0, ExtOp1, ExtIndex,
I)
723 : foldExtExtBinop(ExtOp0, ExtOp1, ExtIndex,
I);
726 replaceValue(
I, *NewExt);
732bool VectorCombine::foldInsExtFNeg(Instruction &
I) {
750 auto *DstVecScalarTy = DstVecTy->getScalarType();
752 if (!SrcVecTy || DstVecScalarTy != SrcVecTy->getScalarType())
757 unsigned NumDstElts = DstVecTy->getNumElements();
758 unsigned NumSrcElts = SrcVecTy->getNumElements();
759 if (ExtIdx > NumSrcElts || InsIdx >= NumDstElts || NumDstElts == 1)
765 SmallVector<int>
Mask(NumDstElts);
766 std::iota(
Mask.begin(),
Mask.end(), 0);
767 Mask[InsIdx] = (ExtIdx % NumDstElts) + NumDstElts;
783 bool NeedLenChg = SrcVecTy->getNumElements() != NumDstElts;
786 SmallVector<int> SrcMask;
789 SrcMask[ExtIdx % NumDstElts] = ExtIdx;
791 DstVecTy, SrcVecTy, SrcMask,
CostKind);
795 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
797 if (NewCost > OldCost)
800 Value *NewShuf, *LenChgShuf =
nullptr;
814 replaceValue(
I, *NewShuf);
820bool VectorCombine::foldInsExtBinop(Instruction &
I) {
821 BinaryOperator *VecBinOp, *SclBinOp;
853 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
855 if (NewCost > OldCost)
866 NewInst->copyIRFlags(VecBinOp);
867 NewInst->andIRFlags(SclBinOp);
872 replaceValue(
I, *NewBO);
878bool VectorCombine::foldBitOpOfCastops(Instruction &
I) {
881 if (!BinOp || !BinOp->isBitwiseLogicOp())
887 if (!LHSCast || !RHSCast) {
888 LLVM_DEBUG(
dbgs() <<
" One or both operands are not cast instructions\n");
894 if (CastOpcode != RHSCast->getOpcode())
898 switch (CastOpcode) {
899 case Instruction::BitCast:
900 case Instruction::Trunc:
901 case Instruction::SExt:
902 case Instruction::ZExt:
908 Value *LHSSrc = LHSCast->getOperand(0);
909 Value *RHSSrc = RHSCast->getOperand(0);
915 auto *SrcTy = LHSSrc->
getType();
916 auto *DstTy =
I.getType();
919 if (CastOpcode != Instruction::BitCast &&
924 if (!SrcTy->getScalarType()->isIntegerTy() ||
925 !DstTy->getScalarType()->isIntegerTy())
940 LHSCastCost + RHSCastCost;
951 if (!LHSCast->hasOneUse())
952 NewCost += LHSCastCost;
953 if (!RHSCast->hasOneUse())
954 NewCost += RHSCastCost;
957 <<
" NewCost=" << NewCost <<
"\n");
959 if (NewCost > OldCost)
964 BinOp->getName() +
".inner");
966 NewBinOp->copyIRFlags(BinOp);
980 replaceValue(
I, *Result);
989bool VectorCombine::foldBitOpOfCastConstant(Instruction &
I) {
1005 switch (CastOpcode) {
1006 case Instruction::BitCast:
1007 case Instruction::ZExt:
1008 case Instruction::SExt:
1009 case Instruction::Trunc:
1015 Value *LHSSrc = LHSCast->getOperand(0);
1017 auto *SrcTy = LHSSrc->
getType();
1018 auto *DstTy =
I.getType();
1021 if (CastOpcode != Instruction::BitCast &&
1026 if (!SrcTy->getScalarType()->isIntegerTy() ||
1027 !DstTy->getScalarType()->isIntegerTy())
1031 PreservedCastFlags RHSFlags;
1056 if (!LHSCast->hasOneUse())
1057 NewCost += LHSCastCost;
1059 LLVM_DEBUG(
dbgs() <<
"foldBitOpOfCastConstant: OldCost=" << OldCost
1060 <<
" NewCost=" << NewCost <<
"\n");
1062 if (NewCost > OldCost)
1067 LHSSrc, InvC,
I.getName() +
".inner");
1069 NewBinOp->copyIRFlags(&
I);
1089 replaceValue(
I, *Result);
1096bool VectorCombine::foldBitcastShuffle(Instruction &
I) {
1110 if (!DestTy || !SrcTy)
1113 unsigned DestEltSize = DestTy->getScalarSizeInBits();
1114 unsigned SrcEltSize = SrcTy->getScalarSizeInBits();
1115 if (SrcTy->getPrimitiveSizeInBits() % DestEltSize != 0)
1125 if (!(BCTy0 && BCTy0->getElementType() == DestTy->getElementType()) &&
1126 !(BCTy1 && BCTy1->getElementType() == DestTy->getElementType()))
1130 SmallVector<int, 16> NewMask;
1131 if (DestEltSize <= SrcEltSize) {
1134 if (SrcEltSize % DestEltSize != 0)
1136 unsigned ScaleFactor = SrcEltSize / DestEltSize;
1141 if (DestEltSize % SrcEltSize != 0)
1143 unsigned ScaleFactor = DestEltSize / SrcEltSize;
1150 unsigned NumSrcElts = SrcTy->getPrimitiveSizeInBits() / DestEltSize;
1151 auto *NewShuffleTy =
1153 auto *OldShuffleTy =
1155 unsigned NumOps = IsUnary ? 1 : 2;
1165 TargetTransformInfo::CastContextHint::None,
1170 TargetTransformInfo::CastContextHint::None,
1173 LLVM_DEBUG(
dbgs() <<
"Found a bitcasted shuffle: " <<
I <<
"\n OldCost: "
1174 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
1176 if (NewCost > OldCost || !NewCost.
isValid())
1184 replaceValue(
I, *Shuf);
1191bool VectorCombine::scalarizeOpOrCmp(Instruction &
I) {
1196 if (!UO && !BO && !CI && !
II)
1204 if (Arg->getType() !=
II->getType() &&
1214 for (User *U :
I.users())
1221 std::optional<uint64_t>
Index;
1223 auto Ops =
II ?
II->args() :
I.operands();
1232 if (OpTy->getElementCount().getKnownMinValue() <= InsIdx)
1238 else if (InsIdx != *Index)
1255 if (!
Index.has_value())
1259 Type *ScalarTy = VecTy->getScalarType();
1260 assert(VecTy->isVectorTy() &&
1263 "Unexpected types for insert element into binop or cmp");
1265 unsigned Opcode =
I.getOpcode();
1273 }
else if (UO || BO) {
1277 IntrinsicCostAttributes ScalarICA(
1278 II->getIntrinsicID(), ScalarTy,
1281 IntrinsicCostAttributes VectorICA(
1282 II->getIntrinsicID(), VecTy,
1289 Value *NewVecC =
nullptr;
1291 NewVecC =
simplifyCmpInst(CI->getPredicate(), VecCs[0], VecCs[1], SQ);
1294 simplifyUnOp(UO->getOpcode(), VecCs[0], UO->getFastMathFlags(), SQ);
1296 NewVecC =
simplifyBinOp(BO->getOpcode(), VecCs[0], VecCs[1], SQ);
1310 for (
auto [Idx,
Op, VecC, Scalar] :
enumerate(
Ops, VecCs, ScalarOps)) {
1312 II->getIntrinsicID(), Idx, &
TTI)))
1315 Instruction::InsertElement, VecTy,
CostKind, *Index, VecC, Scalar);
1316 OldCost += InsertCost;
1317 NewCost += !
Op->hasOneUse() * InsertCost;
1321 if (OldCost < NewCost || !NewCost.
isValid())
1331 ++NumScalarIntrinsic;
1334 for (
auto [OpIdx, Scalar, VecC] :
enumerate(ScalarOps, VecCs))
1341 Scalar = Builder.
CreateCmp(CI->getPredicate(), ScalarOps[0], ScalarOps[1]);
1347 Scalar->setName(
I.getName() +
".scalar");
1352 ScalarInst->copyIRFlags(&
I);
1355 replaceValue(
I, *Insert);
1362bool VectorCombine::foldExtractedCmps(Instruction &
I) {
1367 if (!BI || !
I.getType()->isIntegerTy(1))
1372 Value *
B0 =
I.getOperand(0), *
B1 =
I.getOperand(1);
1375 CmpPredicate
P0,
P1;
1394 ExtractElementInst *ConvertToShuf = getShuffleExtract(Ext0, Ext1,
CostKind);
1397 assert((ConvertToShuf == Ext0 || ConvertToShuf == Ext1) &&
1398 "Unknown ExtractElementInst");
1403 unsigned CmpOpcode =
1409 if (Index0 >= VecTy->getNumElements() || Index1 >= VecTy->getNumElements())
1421 Ext0Cost + Ext1Cost + CmpCost * 2 +
1427 int CheapIndex = ConvertToShuf == Ext0 ? Index1 : Index0;
1428 int ExpensiveIndex = ConvertToShuf == Ext0 ? Index0 : Index1;
1433 ShufMask[CheapIndex] = ExpensiveIndex;
1438 NewCost += Ext0->
hasOneUse() ? 0 : Ext0Cost;
1439 NewCost += Ext1->
hasOneUse() ? 0 : Ext1Cost;
1444 if (OldCost < NewCost || !NewCost.
isValid())
1454 Value *
LHS = ConvertToShuf == Ext0 ? Shuf : VCmp;
1455 Value *
RHS = ConvertToShuf == Ext0 ? VCmp : Shuf;
1458 replaceValue(
I, *NewExt);
1485bool VectorCombine::foldSelectsFromBitcast(Instruction &
I) {
1492 if (!SrcVecTy || !DstVecTy)
1502 if (SrcEltBits != 32 && SrcEltBits != 64)
1505 if (!DstEltTy->
isIntegerTy() || DstEltBits >= SrcEltBits)
1522 if (!ScalarSelCost.
isValid() || ScalarSelCost == 0)
1525 unsigned MinSelects = (VecSelCost.
getValue() / ScalarSelCost.
getValue()) + 1;
1528 if (!BC->hasNUsesOrMore(MinSelects))
1533 DenseMap<Value *, SmallVector<SelectInst *, 8>> CondToSelects;
1535 for (User *U : BC->users()) {
1540 for (User *ExtUser : Ext->users()) {
1544 Cond->getType()->isIntegerTy(1))
1549 if (CondToSelects.
empty())
1552 bool MadeChange =
false;
1553 Value *SrcVec = BC->getOperand(0);
1556 for (
auto [
Cond, Selects] : CondToSelects) {
1558 if (Selects.size() < MinSelects) {
1559 LLVM_DEBUG(
dbgs() <<
"VectorCombine: foldSelectsFromBitcast not "
1560 <<
"profitable (VecCost=" << VecSelCost
1561 <<
", ScalarCost=" << ScalarSelCost
1562 <<
", NumSelects=" << Selects.size() <<
")\n");
1567 auto InsertPt = std::next(BC->getIterator());
1571 InsertPt = std::next(CondInst->getIterator());
1579 for (SelectInst *Sel : Selects) {
1581 Value *Idx = Ext->getIndexOperand();
1585 replaceValue(*Sel, *NewExt);
1590 <<
" selects into vector select\n");
1604 unsigned ReductionOpc =
1610 CostBeforeReduction =
1611 TTI.getCastInstrCost(RedOp->getOpcode(), VecRedTy, ExtType,
1613 CostAfterReduction =
1614 TTI.getExtendedReductionCost(ReductionOpc, IsUnsigned,
II.getType(),
1618 if (RedOp &&
II.getIntrinsicID() == Intrinsic::vector_reduce_add &&
1624 (Op0->
getOpcode() == RedOp->getOpcode() || Op0 == Op1)) {
1631 TTI.getCastInstrCost(Op0->
getOpcode(), MulType, ExtType,
1634 TTI.getArithmeticInstrCost(Instruction::Mul, MulType,
CostKind);
1636 TTI.getCastInstrCost(RedOp->getOpcode(), VecRedTy, MulType,
1639 CostBeforeReduction = ExtCost * 2 + MulCost + Ext2Cost;
1640 CostAfterReduction =
TTI.getMulAccReductionCost(
1641 IsUnsigned, ReductionOpc,
II.getType(), ExtType,
CostKind);
1644 CostAfterReduction =
TTI.getArithmeticReductionCost(ReductionOpc, VecRedTy,
1648bool VectorCombine::foldBinopOfReductions(Instruction &
I) {
1651 if (BinOpOpc == Instruction::Sub)
1652 ReductionIID = Intrinsic::vector_reduce_add;
1656 if (ReductionIID == Intrinsic::vector_reduce_fadd ||
1657 ReductionIID == Intrinsic::vector_reduce_fmul)
1660 auto checkIntrinsicAndGetItsArgument = [](
Value *
V,
1665 if (
II->getIntrinsicID() == IID &&
II->hasOneUse())
1666 return II->getArgOperand(0);
1670 Value *V0 = checkIntrinsicAndGetItsArgument(
I.getOperand(0), ReductionIID);
1673 Value *
V1 = checkIntrinsicAndGetItsArgument(
I.getOperand(1), ReductionIID);
1678 if (
V1->getType() != VTy)
1682 unsigned ReductionOpc =
1695 CostOfRedOperand0 + CostOfRedOperand1 +
1698 if (NewCost >= OldCost || !NewCost.
isValid())
1702 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
1705 if (BinOpOpc == Instruction::Or)
1712 replaceValue(
I, *Rdx);
1721 unsigned NumScanned = 0;
1722 if (std::any_of(Begin, End, [&](
const Instruction &Instr) {
1736class ScalarizationResult {
1737 enum class StatusTy { Unsafe, Safe, SafeWithFreeze };
1742 ScalarizationResult(StatusTy Status,
Value *ToFreeze =
nullptr)
1743 : Status(Status), ToFreeze(ToFreeze) {}
1746 ScalarizationResult(
const ScalarizationResult &
Other) =
default;
1747 ~ScalarizationResult() {
1748 assert(!ToFreeze &&
"freeze() not called with ToFreeze being set");
1751 static ScalarizationResult unsafe() {
return {StatusTy::Unsafe}; }
1752 static ScalarizationResult safe() {
return {StatusTy::Safe}; }
1753 static ScalarizationResult safeWithFreeze(
Value *ToFreeze) {
1754 return {StatusTy::SafeWithFreeze, ToFreeze};
1758 bool isSafe()
const {
return Status == StatusTy::Safe; }
1760 bool isUnsafe()
const {
return Status == StatusTy::Unsafe; }
1763 bool isSafeWithFreeze()
const {
return Status == StatusTy::SafeWithFreeze; }
1768 Status = StatusTy::Unsafe;
1772 void freeze(IRBuilderBase &Builder, Instruction &UserI) {
1773 assert(isSafeWithFreeze() &&
1774 "should only be used when freezing is required");
1776 "UserI must be a user of ToFreeze");
1777 IRBuilder<>::InsertPointGuard Guard(Builder);
1782 if (
U.get() == ToFreeze)
1797 uint64_t NumElements = VecTy->getElementCount().getKnownMinValue();
1801 if (
C->getValue().ult(NumElements))
1802 return ScalarizationResult::safe();
1803 return ScalarizationResult::unsafe();
1808 return ScalarizationResult::unsafe();
1810 APInt Zero(IntWidth, 0);
1811 APInt MaxElts(IntWidth, NumElements);
1818 return ScalarizationResult::safe();
1819 return ScalarizationResult::unsafe();
1832 if (ValidIndices.
contains(IdxRange))
1833 return ScalarizationResult::safeWithFreeze(IdxBase);
1834 return ScalarizationResult::unsafe();
1846 C->getZExtValue() *
DL.getTypeStoreSize(ScalarType));
1858bool VectorCombine::foldSingleElementStore(Instruction &
I) {
1870 if (!
match(
SI->getValueOperand(),
1877 Value *SrcAddr =
Load->getPointerOperand()->stripPointerCasts();
1880 if (!
Load->isSimple() ||
Load->getParent() !=
SI->getParent() ||
1881 !
DL->typeSizeEqualsStoreSize(
Load->getType()->getScalarType()) ||
1882 SrcAddr !=
SI->getPointerOperand()->stripPointerCasts())
1888 auto ScalarizableIdx =
1890 if (ScalarizableIdx.isUnsafe())
1897 if (ScalarizableIdx.isSafeWithFreeze())
1900 SI->getValueOperand()->getType(),
SI->getPointerOperand(),
1901 {ConstantInt::get(Idx->getType(), 0), Idx});
1906 NSI->
setMetadata(LLVMContext::MD_invariant_group,
nullptr);
1908 std::max(
SI->getAlign(),
Load->getAlign()), NewElement->
getType(), Idx,
1911 replaceValue(
I, *NSI);
1921bool VectorCombine::scalarizeLoad(Instruction &
I) {
1931 if (!LI->isSimple() || !
DL->typeSizeEqualsStoreSize(VecTy->getScalarType()))
1934 bool AllExtracts =
true;
1935 bool AllBitcasts =
true;
1937 unsigned NumInstChecked = 0;
1942 for (User *U : LI->users()) {
1944 if (!UI || UI->getParent() != LI->getParent())
1949 if (UI->use_empty())
1953 AllExtracts =
false;
1955 AllBitcasts =
false;
1959 for (Instruction &
I :
1960 make_range(std::next(LI->getIterator()), UI->getIterator())) {
1967 LastCheckedInst = UI;
1972 return scalarizeLoadExtract(LI, VecTy, Ptr);
1974 return scalarizeLoadBitcast(LI, VecTy, Ptr);
1979bool VectorCombine::scalarizeLoadExtract(LoadInst *LI, VectorType *VecTy,
1984 DenseMap<ExtractElementInst *, ScalarizationResult> NeedFreeze;
1987 for (
auto &Pair : NeedFreeze)
1988 Pair.second.discard();
1996 for (User *U : LI->
users()) {
2001 if (ScalarIdx.isUnsafe())
2003 if (ScalarIdx.isSafeWithFreeze()) {
2004 NeedFreeze.try_emplace(UI, ScalarIdx);
2005 ScalarIdx.discard();
2011 Index ?
Index->getZExtValue() : -1);
2019 LLVM_DEBUG(
dbgs() <<
"Found all extractions of a vector load: " << *LI
2020 <<
"\n LoadExtractCost: " << OriginalCost
2021 <<
" vs ScalarizedCost: " << ScalarizedCost <<
"\n");
2023 if (ScalarizedCost >= OriginalCost)
2030 Type *ElemType = VecTy->getElementType();
2033 for (User *U : LI->
users()) {
2035 Value *Idx = EI->getIndexOperand();
2038 auto It = NeedFreeze.find(EI);
2039 if (It != NeedFreeze.end())
2046 Builder.
CreateLoad(ElemType,
GEP, EI->getName() +
".scalar"));
2048 Align ScalarOpAlignment =
2050 NewLoad->setAlignment(ScalarOpAlignment);
2053 size_t Offset = ConstIdx->getZExtValue() *
DL->getTypeStoreSize(ElemType);
2058 replaceValue(*EI, *NewLoad,
false);
2061 FailureGuard.release();
2066bool VectorCombine::scalarizeLoadBitcast(LoadInst *LI, VectorType *VecTy,
2072 Type *TargetScalarType =
nullptr;
2073 unsigned VecBitWidth =
DL->getTypeSizeInBits(VecTy);
2075 for (User *U : LI->
users()) {
2078 Type *DestTy = BC->getDestTy();
2082 unsigned DestBitWidth =
DL->getTypeSizeInBits(DestTy);
2083 if (DestBitWidth != VecBitWidth)
2087 if (!TargetScalarType)
2088 TargetScalarType = DestTy;
2089 else if (TargetScalarType != DestTy)
2097 if (!TargetScalarType)
2105 LLVM_DEBUG(
dbgs() <<
"Found vector load feeding only bitcasts: " << *LI
2106 <<
"\n OriginalCost: " << OriginalCost
2107 <<
" vs ScalarizedCost: " << ScalarizedCost <<
"\n");
2109 if (ScalarizedCost >= OriginalCost)
2120 ScalarLoad->copyMetadata(*LI);
2123 for (User *U : LI->
users()) {
2125 replaceValue(*BC, *ScalarLoad,
false);
2131bool VectorCombine::scalarizeExtExtract(Instruction &
I) {
2146 Type *ScalarDstTy = DstTy->getElementType();
2147 if (
DL->getTypeSizeInBits(SrcTy) !=
DL->getTypeSizeInBits(ScalarDstTy))
2153 unsigned ExtCnt = 0;
2154 bool ExtLane0 =
false;
2155 for (User *U : Ext->users()) {
2169 Instruction::And, ScalarDstTy,
CostKind,
2172 (ExtCnt - ExtLane0) *
2174 Instruction::LShr, ScalarDstTy,
CostKind,
2177 if (ScalarCost > VectorCost)
2180 Value *ScalarV = Ext->getOperand(0);
2187 SmallDenseSet<ConstantInt *, 8> ExtractedLanes;
2188 bool AllExtractsTriggerUB =
true;
2189 ExtractElementInst *LastExtract =
nullptr;
2191 for (User *U : Ext->users()) {
2194 AllExtractsTriggerUB =
false;
2198 if (!LastExtract || LastExtract->
comesBefore(Extract))
2199 LastExtract = Extract;
2201 if (ExtractedLanes.
size() != DstTy->getNumElements() ||
2202 !AllExtractsTriggerUB ||
2210 uint64_t SrcEltSizeInBits =
DL->getTypeSizeInBits(SrcTy->getElementType());
2211 uint64_t TotalBits =
DL->getTypeSizeInBits(SrcTy);
2214 Value *
Mask = ConstantInt::get(PackedTy, EltBitMask);
2215 for (User *U : Ext->users()) {
2221 ? (TotalBits - SrcEltSizeInBits - Idx * SrcEltSizeInBits)
2222 : (Idx * SrcEltSizeInBits);
2225 U->replaceAllUsesWith(
And);
2233bool VectorCombine::foldConcatOfBoolMasks(Instruction &
I) {
2234 Type *Ty =
I.getType();
2239 if (
DL->isBigEndian())
2266 if (ShAmtX > ShAmtY) {
2274 uint64_t ShAmtDiff = ShAmtY - ShAmtX;
2275 unsigned NumSHL = (ShAmtX > 0) + (ShAmtY > 0);
2280 MaskTy->getNumElements() != ShAmtDiff ||
2281 MaskTy->getNumElements() > (
BitWidth / 2))
2286 Type::getIntNTy(Ty->
getContext(), ConcatTy->getNumElements());
2287 auto *MaskIntTy = Type::getIntNTy(Ty->
getContext(), ShAmtDiff);
2290 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
2307 if (Ty != ConcatIntTy)
2313 LLVM_DEBUG(
dbgs() <<
"Found a concatenation of bitcasted bool masks: " <<
I
2314 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2317 if (NewCost > OldCost)
2327 if (Ty != ConcatIntTy) {
2337 replaceValue(
I, *Result);
2343bool VectorCombine::foldPermuteOfBinops(Instruction &
I) {
2344 BinaryOperator *BinOp;
2345 ArrayRef<int> OuterMask;
2353 Value *Op00, *Op01, *Op10, *Op11;
2354 ArrayRef<int> Mask0, Mask1;
2359 if (!Match0 && !Match1)
2372 if (!ShuffleDstTy || !BinOpTy || !Op0Ty || !Op1Ty)
2375 unsigned NumSrcElts = BinOpTy->getNumElements();
2380 any_of(OuterMask, [NumSrcElts](
int M) {
return M >= (int)NumSrcElts; }))
2384 SmallVector<int> NewMask0, NewMask1;
2385 for (
int M : OuterMask) {
2386 if (M < 0 || M >= (
int)NumSrcElts) {
2390 NewMask0.
push_back(Match0 ? Mask0[M] : M);
2391 NewMask1.
push_back(Match1 ? Mask1[M] : M);
2395 unsigned NumOpElts = Op0Ty->getNumElements();
2396 bool IsIdentity0 = ShuffleDstTy == Op0Ty &&
2397 all_of(NewMask0, [NumOpElts](
int M) {
return M < (int)NumOpElts; }) &&
2399 bool IsIdentity1 = ShuffleDstTy == Op1Ty &&
2400 all_of(NewMask1, [NumOpElts](
int M) {
return M < (int)NumOpElts; }) &&
2409 ShuffleDstTy, BinOpTy, OuterMask,
CostKind,
2410 0,
nullptr, {BinOp}, &
I);
2412 NewCost += BinOpCost;
2418 OldCost += Shuf0Cost;
2420 NewCost += Shuf0Cost;
2426 OldCost += Shuf1Cost;
2428 NewCost += Shuf1Cost;
2436 Op0Ty, NewMask0,
CostKind, 0,
nullptr, {Op00, Op01});
2440 Op1Ty, NewMask1,
CostKind, 0,
nullptr, {Op10, Op11});
2442 LLVM_DEBUG(
dbgs() <<
"Found a shuffle feeding a shuffled binop: " <<
I
2443 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2447 if (NewCost > OldCost)
2458 NewInst->copyIRFlags(BinOp);
2462 replaceValue(
I, *NewBO);
2468bool VectorCombine::foldShuffleOfBinops(Instruction &
I) {
2469 ArrayRef<int> OldMask;
2476 if (
LHS->getOpcode() !=
RHS->getOpcode())
2480 bool IsCommutative =
false;
2489 IsCommutative = BinaryOperator::isCommutative(BO->getOpcode());
2500 if (!ShuffleDstTy || !BinResTy || !BinOpTy ||
X->getType() !=
Z->getType())
2503 bool SameBinOp =
LHS ==
RHS;
2504 unsigned NumSrcElts = BinOpTy->getNumElements();
2507 if (IsCommutative &&
X != Z &&
Y != W && (
X == W ||
Y == Z))
2510 auto ConvertToUnary = [NumSrcElts](
int &
M) {
2511 if (M >= (
int)NumSrcElts)
2515 SmallVector<int> NewMask0(OldMask);
2524 SmallVector<int> NewMask1(OldMask);
2543 ShuffleDstTy, BinResTy, OldMask,
CostKind, 0,
2553 ArrayRef<int> InnerMask;
2555 m_Mask(InnerMask)))) &&
2558 [NumSrcElts](
int M) {
return M < (int)NumSrcElts; })) {
2570 bool ReducedInstCount =
false;
2571 ReducedInstCount |= MergeInner(
X, 0, NewMask0,
CostKind);
2572 ReducedInstCount |= MergeInner(
Y, 0, NewMask1,
CostKind);
2573 ReducedInstCount |= MergeInner(Z, NumSrcElts, NewMask0,
CostKind);
2574 ReducedInstCount |= MergeInner(W, NumSrcElts, NewMask1,
CostKind);
2575 bool SingleSrcBinOp = (
X ==
Y) && (Z == W) && (NewMask0 == NewMask1);
2587 I.getType()->getScalarType()->isIntegerTy(1) &&
2591 auto *ShuffleCmpTy =
2594 SK0, ShuffleCmpTy, BinOpTy, NewMask0,
CostKind, 0,
nullptr, {
X,
Z});
2595 if (!SingleSrcBinOp)
2605 PredLHS,
CostKind, Op0Info, Op1Info);
2615 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2622 if (ReducedInstCount ? (NewCost > OldCost) : (NewCost >= OldCost))
2631 : Builder.
CreateCmp(PredLHS, Shuf0, Shuf1);
2635 NewInst->copyIRFlags(
LHS);
2636 NewInst->andIRFlags(
RHS);
2641 replaceValue(
I, *NewBO);
2648bool VectorCombine::foldShuffleOfSelects(Instruction &
I) {
2650 Value *C1, *
T1, *F1, *C2, *T2, *F2;
2661 if (!C1VecTy || !C2VecTy || C1VecTy != C2VecTy)
2667 if (((SI0FOp ==
nullptr) != (SI1FOp ==
nullptr)) ||
2668 ((SI0FOp !=
nullptr) &&
2669 (SI0FOp->getFastMathFlags() != SI1FOp->getFastMathFlags())))
2675 auto SelOp = Instruction::Select;
2683 CostSel1 + CostSel2 +
2685 {
I.getOperand(0),
I.getOperand(1)}, &
I);
2689 Mask,
CostKind, 0,
nullptr, {C1, C2});
2699 if (!Sel1->hasOneUse())
2700 NewCost += CostSel1;
2701 if (!Sel2->hasOneUse())
2702 NewCost += CostSel2;
2705 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2707 if (NewCost > OldCost)
2716 NewSel = Builder.
CreateSelectFMF(ShuffleCmp, ShuffleTrue, ShuffleFalse,
2717 SI0FOp->getFastMathFlags());
2719 NewSel = Builder.
CreateSelect(ShuffleCmp, ShuffleTrue, ShuffleFalse);
2724 replaceValue(
I, *NewSel);
2730bool VectorCombine::foldShuffleOfCastops(Instruction &
I) {
2732 ArrayRef<int> OldMask;
2741 if (!C0 || (IsBinaryShuffle && !C1))
2748 if (!IsBinaryShuffle && Opcode == Instruction::BitCast)
2751 if (IsBinaryShuffle) {
2752 if (C0->getSrcTy() != C1->getSrcTy())
2755 if (Opcode != C1->getOpcode()) {
2757 Opcode = Instruction::SExt;
2766 if (!ShuffleDstTy || !CastDstTy || !CastSrcTy)
2769 unsigned NumSrcElts = CastSrcTy->getNumElements();
2770 unsigned NumDstElts = CastDstTy->getNumElements();
2771 assert((NumDstElts == NumSrcElts || Opcode == Instruction::BitCast) &&
2772 "Only bitcasts expected to alter src/dst element counts");
2776 if (NumDstElts != NumSrcElts && (NumSrcElts % NumDstElts) != 0 &&
2777 (NumDstElts % NumSrcElts) != 0)
2780 SmallVector<int, 16> NewMask;
2781 if (NumSrcElts >= NumDstElts) {
2784 assert(NumSrcElts % NumDstElts == 0 &&
"Unexpected shuffle mask");
2785 unsigned ScaleFactor = NumSrcElts / NumDstElts;
2790 assert(NumDstElts % NumSrcElts == 0 &&
"Unexpected shuffle mask");
2791 unsigned ScaleFactor = NumDstElts / NumSrcElts;
2796 auto *NewShuffleDstTy =
2805 if (IsBinaryShuffle)
2820 if (IsBinaryShuffle) {
2830 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2832 if (NewCost > OldCost)
2836 if (IsBinaryShuffle)
2846 NewInst->copyIRFlags(C0);
2847 if (IsBinaryShuffle)
2848 NewInst->andIRFlags(C1);
2852 replaceValue(
I, *Cast);
2862bool VectorCombine::foldShuffleOfShuffles(Instruction &
I) {
2863 ArrayRef<int> OuterMask;
2864 Value *OuterV0, *OuterV1;
2869 ArrayRef<int> InnerMask0, InnerMask1;
2870 Value *X0, *X1, *Y0, *Y1;
2875 if (!Match0 && !Match1)
2880 SmallVector<int, 16> PoisonMask1;
2885 InnerMask1 = PoisonMask1;
2889 X0 = Match0 ? X0 : OuterV0;
2890 Y0 = Match0 ? Y0 : OuterV0;
2891 X1 = Match1 ? X1 : OuterV1;
2892 Y1 = Match1 ? Y1 : OuterV1;
2896 if (!ShuffleDstTy || !ShuffleSrcTy || !ShuffleImmTy ||
2900 unsigned NumSrcElts = ShuffleSrcTy->getNumElements();
2901 unsigned NumImmElts = ShuffleImmTy->getNumElements();
2906 SmallVector<int, 16> NewMask(OuterMask);
2907 Value *NewX =
nullptr, *NewY =
nullptr;
2908 for (
int &M : NewMask) {
2909 Value *Src =
nullptr;
2910 if (0 <= M && M < (
int)NumImmElts) {
2914 Src =
M >= (int)NumSrcElts ? Y0 : X0;
2915 M =
M >= (int)NumSrcElts ? (M - NumSrcElts) :
M;
2917 }
else if (M >= (
int)NumImmElts) {
2922 Src =
M >= (int)NumSrcElts ? Y1 : X1;
2923 M =
M >= (int)NumSrcElts ? (M - NumSrcElts) :
M;
2927 assert(0 <= M && M < (
int)NumSrcElts &&
"Unexpected shuffle mask index");
2936 if (!NewX || NewX == Src) {
2940 if (!NewY || NewY == Src) {
2959 replaceValue(
I, *NewX);
2976 bool IsUnary =
all_of(NewMask, [&](
int M) {
return M < (int)NumSrcElts; });
2982 nullptr, {NewX, NewY});
2984 NewCost += InnerCost0;
2986 NewCost += InnerCost1;
2989 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
2991 if (NewCost > OldCost)
2995 replaceValue(
I, *Shuf);
3011bool VectorCombine::foldShufflesOfLengthChangingShuffles(Instruction &
I) {
3016 unsigned ChainLength = 0;
3017 SmallVector<int>
Mask;
3018 SmallVector<int> YMask;
3028 ArrayRef<int> OuterMask;
3029 Value *OuterV0, *OuterV1;
3030 if (ChainLength != 0 && !Trunk->
hasOneUse())
3033 m_Mask(OuterMask))))
3035 if (OuterV0->
getType() != TrunkType) {
3041 ArrayRef<int> InnerMask0, InnerMask1;
3047 bool Match0Leaf = Match0 && A0->
getType() !=
I.getType();
3048 bool Match1Leaf = Match1 && A1->
getType() !=
I.getType();
3049 if (Match0Leaf == Match1Leaf) {
3055 SmallVector<int> CommutedOuterMask;
3062 for (
int &M : CommutedOuterMask) {
3065 if (M < (
int)NumTrunkElts)
3070 OuterMask = CommutedOuterMask;
3089 int NumLeafElts = YType->getNumElements();
3090 SmallVector<int> LocalYMask(InnerMask1);
3091 for (
int &M : LocalYMask) {
3092 if (M >= NumLeafElts)
3102 Mask.assign(OuterMask);
3103 YMask.
assign(LocalYMask);
3104 OldCost = NewCost = LocalOldCost;
3111 SmallVector<int> NewYMask(YMask);
3113 for (
auto [CombinedM, LeafM] :
llvm::zip(NewYMask, LocalYMask)) {
3114 if (LeafM == -1 || CombinedM == LeafM)
3116 if (CombinedM == -1) {
3126 SmallVector<int> NewMask;
3127 NewMask.
reserve(NumTrunkElts);
3128 for (
int M : Mask) {
3129 if (M < 0 || M >=
static_cast<int>(NumTrunkElts))
3144 if (LocalNewCost >= NewCost && LocalOldCost < LocalNewCost - NewCost)
3148 if (ChainLength == 1) {
3149 dbgs() <<
"Found chain of shuffles fed by length-changing shuffles: "
3152 dbgs() <<
" next chain link: " << *Trunk <<
'\n'
3153 <<
" old cost: " << (OldCost + LocalOldCost)
3154 <<
" new cost: " << LocalNewCost <<
'\n';
3159 OldCost += LocalOldCost;
3160 NewCost = LocalNewCost;
3164 if (ChainLength <= 1)
3172 return M < 0 || M >=
static_cast<int>(NumTrunkElts);
3175 for (
int &M : Mask) {
3176 if (M >=
static_cast<int>(NumTrunkElts))
3177 M = YMask[
M - NumTrunkElts];
3181 replaceValue(
I, *Root);
3188 replaceValue(
I, *Root);
3194bool VectorCombine::foldShuffleOfIntrinsics(Instruction &
I) {
3196 ArrayRef<int> OldMask;
3206 if (IID != II1->getIntrinsicID())
3215 if (!ShuffleDstTy || !II0Ty)
3221 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3222 Value *Arg0 = II0->getArgOperand(
I);
3223 Value *Arg1 = II1->getArgOperand(
I);
3240 II0Ty, OldMask,
CostKind, 0,
nullptr, {II0, II1}, &
I);
3244 SmallDenseSet<std::pair<Value *, Value *>> SeenOperandPairs;
3245 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3247 NewArgsTy.
push_back(II0->getArgOperand(
I)->getType());
3251 ShuffleDstTy->getNumElements());
3253 std::pair<Value *, Value *> OperandPair =
3254 std::make_pair(II0->getArgOperand(
I), II1->getArgOperand(
I));
3255 if (!SeenOperandPairs.
insert(OperandPair).second) {
3261 CostKind, 0,
nullptr, {II0->getArgOperand(
I), II1->getArgOperand(
I)});
3264 IntrinsicCostAttributes NewAttr(IID, ShuffleDstTy, NewArgsTy);
3267 if (!II0->hasOneUse())
3269 if (II1 != II0 && !II1->hasOneUse())
3273 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
3276 if (NewCost > OldCost)
3280 SmallDenseMap<std::pair<Value *, Value *>,
Value *> ShuffleCache;
3281 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I)
3285 std::pair<Value *, Value *> OperandPair =
3286 std::make_pair(II0->getArgOperand(
I), II1->getArgOperand(
I));
3287 auto It = ShuffleCache.
find(OperandPair);
3288 if (It != ShuffleCache.
end()) {
3294 II1->getArgOperand(
I), OldMask);
3295 ShuffleCache[OperandPair] = Shuf;
3303 NewInst->copyIRFlags(II0);
3304 NewInst->andIRFlags(II1);
3307 replaceValue(
I, *NewIntrinsic);
3313bool VectorCombine::foldPermuteOfIntrinsic(Instruction &
I) {
3325 if (!ShuffleDstTy || !IntrinsicSrcTy)
3329 unsigned NumSrcElts = IntrinsicSrcTy->getNumElements();
3330 if (
any_of(Mask, [NumSrcElts](
int M) {
return M >= (int)NumSrcElts; }))
3343 IntrinsicSrcTy, Mask,
CostKind, 0,
nullptr, {V0}, &
I);
3347 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3349 NewArgsTy.
push_back(II0->getArgOperand(
I)->getType());
3353 ShuffleDstTy->getNumElements());
3356 ArgTy, VecTy, Mask,
CostKind, 0,
nullptr,
3357 {II0->getArgOperand(
I)});
3360 IntrinsicCostAttributes NewAttr(IID, ShuffleDstTy, NewArgsTy);
3365 if (!II0->hasOneUse())
3368 LLVM_DEBUG(
dbgs() <<
"Found a permute of intrinsic: " <<
I <<
"\n OldCost: "
3369 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
3371 if (NewCost > OldCost)
3376 for (
unsigned I = 0,
E = II0->arg_size();
I !=
E; ++
I) {
3389 NewInst->copyIRFlags(II0);
3391 replaceValue(
I, *NewIntrinsic);
3401 int M = SV->getMaskValue(Lane);
3404 if (
static_cast<unsigned>(M) < NumElts) {
3405 V = SV->getOperand(0);
3408 V = SV->getOperand(1);
3419 auto [U, Lane] = IL;
3432 unsigned NumElts = Ty->getNumElements();
3433 if (Item.
size() == NumElts || NumElts == 1 || Item.
size() % NumElts != 0)
3439 std::iota(ConcatMask.
begin(), ConcatMask.
end(), 0);
3445 unsigned NumSlices = Item.
size() / NumElts;
3450 for (
unsigned Slice = 0; Slice < NumSlices; ++Slice) {
3451 Value *SliceV = Item[Slice * NumElts].first;
3452 if (!SliceV || SliceV->
getType() != Ty)
3454 for (
unsigned Elt = 0; Elt < NumElts; ++Elt) {
3455 auto [V, Lane] = Item[Slice * NumElts + Elt];
3456 if (Lane !=
static_cast<int>(Elt) || SliceV != V)
3465 const DenseSet<std::pair<Value *, Use *>> &IdentityLeafs,
3466 const DenseSet<std::pair<Value *, Use *>> &SplatLeafs,
3467 const DenseSet<std::pair<Value *, Use *>> &ConcatLeafs,
3470 auto [FrontV, FrontLane] = Item.
front();
3472 if (IdentityLeafs.contains(std::make_pair(FrontV, From))) {
3475 if (SplatLeafs.contains(std::make_pair(FrontV, From))) {
3477 return Builder.CreateShuffleVector(FrontV, Mask);
3479 if (ConcatLeafs.contains(std::make_pair(FrontV, From))) {
3483 for (
unsigned S = 0; S <
Values.size(); ++S)
3484 Values[S] = Item[S * NumElts].first;
3486 while (
Values.size() > 1) {
3489 std::iota(Mask.begin(), Mask.end(), 0);
3491 for (
unsigned S = 0; S < NewValues.
size(); ++S)
3493 Builder.CreateShuffleVector(
Values[S * 2],
Values[S * 2 + 1], Mask);
3507 if (BCDstTy && BCSrcTy &&
3508 BCDstTy->getElementCount() != BCSrcTy->getElementCount()) {
3509 unsigned DstElts = BCDstTy->getNumElements();
3510 unsigned SrcElts = BCSrcTy->getNumElements();
3512 if (DstElts > SrcElts) {
3514 unsigned R = DstElts / SrcElts;
3515 if (Item.
size() % R != 0)
3517 for (
unsigned Idx = 0,
E = Item.
size(); Idx <
E; Idx += R) {
3518 auto [V, Lane] = Item[Idx];
3528 unsigned R = SrcElts / DstElts;
3529 for (
auto [V, Lane] : Item) {
3535 for (
unsigned J = 0; J < R; ++J)
3540 IdentityLeafs, SplatLeafs, ConcatLeafs,
3541 Builder, WorkList,
TTI);
3543 return Builder.CreateBitCast(
3548 unsigned NumOps =
I->getNumOperands() - (
II ? 1 : 0);
3550 for (
unsigned Idx = 0; Idx <
NumOps; Idx++) {
3553 Ops[Idx] =
II->getOperand(Idx);
3558 IdentityLeafs, SplatLeafs, ConcatLeafs, Builder, WorkList,
TTI);
3568 for (
const auto &Lane : Item)
3581 auto *
Value = Builder.CreateCmp(CI->getPredicate(),
Ops[0],
Ops[1]);
3591 auto *
Value = Builder.CreateCast(CI->getOpcode(),
Ops[0], DstTy);
3596 auto *
Value = Builder.CreateIntrinsic(DstTy,
II->getIntrinsicID(),
Ops);
3610bool VectorCombine::foldShuffleToIdentity(Instruction &
I) {
3612 if (!Ty ||
I.use_empty())
3616 for (
unsigned M = 0,
E = Ty->getNumElements(); M <
E; ++M)
3620 Candidates.
push_back(std::make_pair(Start, &*
I.use_begin()));
3621 DenseSet<std::pair<Value *, Use *>> IdentityLeafs, SplatLeafs, ConcatLeafs;
3622 unsigned NumVisited = 0;
3623 bool TraversedElCountChangingBitcast =
false;
3625 while (!Candidates.
empty()) {
3630 auto Item = ItemFrom.first;
3631 auto From = ItemFrom.second;
3632 auto [FrontV, FrontLane] = Item.front();
3639 if (FrontLane == 0 &&
3643 Value *FrontV = Item.front().first;
3645 E.value().second == (int)
E.index());
3647 IdentityLeafs.
insert(std::make_pair(FrontV, From));
3652 C &&
C->getSplatValue() &&
3654 Value *FrontV = Item.front().first;
3660 SplatLeafs.
insert(std::make_pair(FrontV, From));
3665 auto [FrontV, FrontLane] = Item.front();
3666 auto [
V, Lane] = IL;
3667 return !
V || (
V == FrontV && Lane == FrontLane);
3669 SplatLeafs.
insert(std::make_pair(FrontV, From));
3675 auto CheckLaneIsEquivalentToFirst = [Item](
InstLane IL) {
3676 Value *FrontV = Item.front().first;
3685 if (CI->getPredicate() !=
cast<CmpInst>(FrontV)->getPredicate())
3688 if (CI->getSrcTy()->getScalarType() !=
3693 SI->getOperand(0)->getType() !=
3700 II->getIntrinsicID() ==
3702 !
II->hasOperandBundles());
3709 BO && BO->isIntDivRem())
3716 }
else if (
isa<UnaryOperator, TruncInst, ZExtInst, SExtInst, FPToSIInst,
3717 FPToUIInst, SIToFPInst, UIToFPInst>(FrontV)) {
3724 if (BCDstTy && BCSrcTy) {
3725 ElementCount DstEC = BCDstTy->getElementCount();
3726 ElementCount SrcEC = BCSrcTy->getElementCount();
3727 if (DstEC == SrcEC) {
3730 &BitCast->getOperandUse(0));
3735 if (DstElts > SrcElts && DstElts % SrcElts == 0) {
3739 unsigned R = DstElts / SrcElts;
3741 bool Valid = Item.size() %
R == 0;
3742 for (
unsigned Idx = 0,
E = Item.size(); Valid && Idx <
E;
3744 auto [V0, L0] = Item[Idx];
3747 [](
InstLane IL) {
return IL.first !=
nullptr; })) {
3758 for (
unsigned J = 1; J <
R; ++J) {
3759 auto [VJ, LJ] = Item[Idx + J];
3760 if (!VJ || VJ != V0 || LJ != L0 + (
int)J) {
3771 TraversedElCountChangingBitcast =
true;
3772 Candidates.
emplace_back(NItem, &BitCast->getOperandUse(0));
3775 }
else if (SrcElts > DstElts && SrcElts % DstElts == 0) {
3778 unsigned R = SrcElts / DstElts;
3780 for (
auto [V, Lane] : Item) {
3786 for (
unsigned J = 0; J <
R; ++J)
3789 TraversedElCountChangingBitcast =
true;
3790 Candidates.
emplace_back(NItem, &BitCast->getOperandUse(0));
3796 &Sel->getOperandUse(0));
3798 &Sel->getOperandUse(1));
3800 &Sel->getOperandUse(2));
3804 !
II->hasOperandBundles()) {
3805 for (
unsigned Op = 0,
E =
II->getNumOperands() - 1;
Op <
E;
Op++) {
3809 Value *FrontV = Item.front().first;
3826 ConcatLeafs.
insert(std::make_pair(FrontV, From));
3833 if (NumVisited <= 1)
3839 if (NumVisited == 2 && TraversedElCountChangingBitcast)
3842 LLVM_DEBUG(
dbgs() <<
"Found a superfluous identity shuffle: " <<
I <<
"\n");
3849 ConcatLeafs, Builder, Worklist, &
TTI);
3850 replaceValue(
I, *V);
3857bool VectorCombine::foldShuffleFromReductions(Instruction &
I) {
3861 switch (
II->getIntrinsicID()) {
3862 case Intrinsic::vector_reduce_add:
3863 case Intrinsic::vector_reduce_mul:
3864 case Intrinsic::vector_reduce_and:
3865 case Intrinsic::vector_reduce_or:
3866 case Intrinsic::vector_reduce_xor:
3867 case Intrinsic::vector_reduce_smin:
3868 case Intrinsic::vector_reduce_smax:
3869 case Intrinsic::vector_reduce_umin:
3870 case Intrinsic::vector_reduce_umax:
3879 std::queue<Value *> Worklist;
3880 SmallPtrSet<Value *, 4> Visited;
3881 ShuffleVectorInst *Shuffle =
nullptr;
3885 while (!Worklist.empty()) {
3886 Value *CV = Worklist.front();
3898 if (CI->isBinaryOp()) {
3899 for (
auto *
Op : CI->operand_values())
3903 if (Shuffle && Shuffle != SV)
3920 for (
auto *V : Visited)
3921 for (
auto *U :
V->users())
3922 if (!Visited.contains(U) && U != &
I)
3925 FixedVectorType *VecType =
3929 FixedVectorType *ShuffleInputType =
3931 if (!ShuffleInputType)
3937 SmallVector<int> ConcatMask;
3939 sort(ConcatMask, [](
int X,
int Y) {
return (
unsigned)
X < (unsigned)
Y; });
3940 bool UsesSecondVec =
3941 any_of(ConcatMask, [&](
int M) {
return M >= (int)NumInputElts; });
3948 ShuffleInputType, ConcatMask,
CostKind);
3950 LLVM_DEBUG(
dbgs() <<
"Found a reduction feeding from a shuffle: " << *Shuffle
3952 LLVM_DEBUG(
dbgs() <<
" OldCost: " << OldCost <<
" vs NewCost: " << NewCost
3954 bool MadeChanges =
false;
3955 if (NewCost < OldCost) {
3959 LLVM_DEBUG(
dbgs() <<
"Created new shuffle: " << *NewShuffle <<
"\n");
3960 replaceValue(*Shuffle, *NewShuffle);
3966 MadeChanges |= foldSelectShuffle(*Shuffle,
true);
3987bool VectorCombine::foldShuffleChainsToReduce(Instruction &
I) {
3996 if (FVT->getNumElements() < 2)
3999 std::optional<Instruction::BinaryOps> CommonBinOp;
4000 std::optional<Intrinsic::ID> CommonCallOp;
4005 CommonBinOp = BO->getOpcode();
4007 CommonCallOp = MMI->getIntrinsicID();
4013 FastMathFlags CommonFMF;
4014 bool IsFloatReduction =
false;
4018 auto IsChainNode = [&](
Value *
V) {
4020 return CommonBinOp && BO->getOpcode() == *CommonBinOp;
4022 return CommonCallOp && MMI->getIntrinsicID() == *CommonCallOp;
4030 constexpr unsigned MaxChainNodes = 32;
4031 SmallSetVector<Value *, 16> Nodes;
4032 SmallSetVector<Value *, 4> Sources;
4033 unsigned NumVisited = 0;
4034 auto AddSource = [&](
Value *
V) {
4040 auto Walk = [&](
Value *
V,
auto &&Walk) ->
bool {
4043 if (++NumVisited > MaxChainNodes)
4045 if (!IsChainNode(V))
4046 return AddSource(V);
4051 if (!Walk(
U->getOperand(
I), Walk))
4060 return AddSource(V);
4062 if (!Walk(VecOpEE, Walk) || Nodes.
empty())
4069 for (
Value *V : Nodes) {
4075 if (!IsFloatReduction) {
4077 IsFloatReduction =
true;
4091 DenseMap<Value *, Demand> Demands;
4092 auto DemandOf = [&](
Value *
V) -> Demand & {
4094 Demand &
D = Demands[
V];
4095 if (
D.Lanes.getBitWidth() !=
N)
4099 DemandOf(VecOpEE).Lanes.setBit(0);
4101 Demand DV = Demands.
lookup(V);
4102 if (DV.Lanes.isZero())
4105 ArrayRef<int>
Mask = SVI->getShuffleMask();
4106 Demand &
DS = DemandOf(SVI->getOperand(0));
4107 for (
unsigned I = 0,
E =
Mask.size();
I !=
E; ++
I) {
4109 if (!DV.Lanes[
I] || Mask[
I] < 0 ||
4110 (
unsigned)Mask[
I] >=
DS.Lanes.getBitWidth())
4112 if (
DS.Lanes[Mask[
I]] || DV.Duplicates[
I])
4113 DS.Duplicates.setBit(Mask[
I]);
4114 DS.Lanes.setBit(Mask[
I]);
4118 for (
Value *
Op : {
U->getOperand(0),
U->getOperand(1)}) {
4119 Demand &DOp = DemandOf(
Op);
4121 DOp.Duplicates |= DV.Duplicates | (DOp.Lanes & DV.Lanes);
4122 DOp.Lanes |= DV.Lanes;
4129 auto CoversChain = [&](
Value *
V) {
4130 SmallVector<Value *, 8> Worklist(1, VecOpEE);
4131 SmallPtrSet<Value *, 8> Seen;
4133 while (!Worklist.empty()) {
4136 for (
unsigned I = 0;
I !=
NumOps; ++
I) {
4140 if (!Nodes.contains(
Op))
4142 Worklist.push_back(
Op);
4150 struct ReductionCut {
4154 std::optional<ReductionCut> Cut;
4155 for (
Value *S : Sources) {
4156 auto It = Demands.
find(S);
4157 if (It == Demands.
end() || It->second.Lanes.isZero())
4159 if (!IsIdempotent && !It->second.Duplicates.isZero()) {
4164 Cut = ReductionCut{S, It->second.Lanes};
4171 if (!IsIdempotent && !(Cut->Elts & It->second.Lanes).isZero()) {
4175 Cut->Elts |= It->second.Lanes;
4178 for (
Value *V : Nodes) {
4181 auto It = Demands.
find(V);
4182 if (It == Demands.
end() || !It->second.Lanes.isAllOnes())
4184 if (!IsIdempotent && !It->second.Duplicates.isZero())
4186 if (!CoversChain(V))
4188 Cut = ReductionCut{
V, It->second.Lanes};
4193 if (!Cut || Cut->Elts.popcount() < 2)
4203 for (
Value *V : Nodes)
4207 bool IsPartialReduction = !Cut->Elts.isAllOnes();
4208 FixedVectorType *ReduceVecTy =
4213 SmallVector<int> ExtractMask;
4215 if (IsPartialReduction) {
4216 for (
unsigned I = 0,
E = Cut->Elts.getBitWidth();
I !=
E; ++
I)
4218 ExtractMask.push_back(
I);
4219 unsigned SubIdx = 0, SubLen;
4220 auto SK = Cut->Elts.isShiftedMask(SubIdx, SubLen)
4224 SubIdx, ReduceVecTy);
4227 IntrinsicCostAttributes ICA(
4228 ReducedOp, ReduceVecTy->getElementType(),
4232 IsFloatReduction ? CommonFMF : FastMathFlags());
4235 LLVM_DEBUG(
dbgs() <<
"Found reduction shuffle chain: " <<
I <<
"\n OldCost : "
4236 << OrigCost <<
" vs NewCost: " << NewCost <<
"\n");
4241 if (VecOpEE->
hasOneUse() ? (NewCost > OrigCost) : (NewCost >= OrigCost))
4244 Value *ReduceInput = Cut->Src;
4245 if (IsPartialReduction)
4248 Value *ReducedResult;
4249 if (IsFloatReduction) {
4251 *CommonBinOp, ReduceVecTy->getElementType(),
false,
4254 {Identity, ReduceInput}, CommonFMF);
4259 replaceValue(
I, *ReducedResult);
4268bool VectorCombine::foldCastFromReductions(Instruction &
I) {
4273 bool TruncOnly =
false;
4276 case Intrinsic::vector_reduce_add:
4277 case Intrinsic::vector_reduce_mul:
4280 case Intrinsic::vector_reduce_and:
4281 case Intrinsic::vector_reduce_or:
4282 case Intrinsic::vector_reduce_xor:
4289 Value *ReductionSrc =
I.getOperand(0);
4301 Type *ResultTy =
I.getType();
4304 ReductionOpc, ReductionSrcTy, std::nullopt,
CostKind);
4314 if (OldCost <= NewCost || !NewCost.
isValid())
4318 II->getIntrinsicID(), {Src});
4320 replaceValue(
I, *NewCast);
4348bool VectorCombine::foldSignBitReductionCmp(Instruction &
I) {
4350 IntrinsicInst *ReduceOp;
4351 const APInt *CmpVal;
4358 case Intrinsic::vector_reduce_or:
4359 case Intrinsic::vector_reduce_umax:
4360 case Intrinsic::vector_reduce_and:
4361 case Intrinsic::vector_reduce_umin:
4362 case Intrinsic::vector_reduce_add:
4373 unsigned BitWidth = VecTy->getScalarSizeInBits();
4377 unsigned NumElts = VecTy->getNumElements();
4386 case Intrinsic::vector_reduce_or:
4387 case Intrinsic::vector_reduce_umax:
4388 TreeOpcode = Instruction::Or;
4390 case Intrinsic::vector_reduce_and:
4391 case Intrinsic::vector_reduce_umin:
4392 TreeOpcode = Instruction::And;
4394 case Intrinsic::vector_reduce_add:
4395 TreeOpcode = Instruction::Add;
4403 SmallVector<Value *, 8> Worklist;
4404 SmallVector<Value *, 8> Sources;
4406 std::optional<bool> IsAShr;
4407 constexpr unsigned MaxSources = 8;
4412 while (!Worklist.
empty() && Worklist.
size() <= MaxSources &&
4413 Sources.
size() <= MaxSources) {
4422 bool ThisIsAShr = Shr->getOpcode() == Instruction::AShr;
4424 IsAShr = ThisIsAShr;
4425 else if (*IsAShr != ThisIsAShr)
4451 if (Sources.
empty() || Sources.
size() > MaxSources ||
4452 Worklist.
size() > MaxSources || !IsAShr)
4455 unsigned NumSources = Sources.
size();
4459 if (OrigIID == Intrinsic::vector_reduce_add &&
4467 (OrigIID == Intrinsic::vector_reduce_add) ? NumSources * NumElts : 1;
4470 NegativeVal.negate();
4502 TestsNegative =
false;
4503 }
else if (*CmpVal == NegativeVal) {
4504 TestsNegative =
true;
4508 IsEq = Pred == ICmpInst::ICMP_EQ;
4509 }
else if (Pred == ICmpInst::ICMP_SLT && *CmpVal == RangeHigh) {
4511 TestsNegative = (RangeHigh == NegativeVal);
4512 }
else if (Pred == ICmpInst::ICMP_SGT && *CmpVal == RangeHigh - 1) {
4514 TestsNegative = (RangeHigh == NegativeVal);
4515 }
else if (Pred == ICmpInst::ICMP_SGT && *CmpVal == RangeLow) {
4517 TestsNegative = (RangeLow == NegativeVal);
4518 }
else if (Pred == ICmpInst::ICMP_SLT && *CmpVal == RangeLow + 1) {
4520 TestsNegative = (RangeLow == NegativeVal);
4563 enum CheckKind :
unsigned {
4570 auto RequiresOr = [](CheckKind
C) ->
bool {
return C & 0b100; };
4572 auto IsNegativeCheck = [](CheckKind
C) ->
bool {
return C & 0b010; };
4574 auto Invert = [](CheckKind
C) {
return CheckKind(
C ^ 0b011); };
4578 case Intrinsic::vector_reduce_or:
4579 case Intrinsic::vector_reduce_umax:
4580 Base = TestsNegative ? AnyNeg : AllNonNeg;
4582 case Intrinsic::vector_reduce_and:
4583 case Intrinsic::vector_reduce_umin:
4584 Base = TestsNegative ? AllNeg : AnyNonNeg;
4586 case Intrinsic::vector_reduce_add:
4587 Base = TestsNegative ? AllNeg : AllNonNeg;
4602 return ArithCost <= MinMaxCost ? std::make_pair(Arith, ArithCost)
4603 : std::make_pair(MinMax, MinMaxCost);
4607 auto [NewIID, NewCost] = RequiresOr(
Check)
4608 ? PickCheaper(Intrinsic::vector_reduce_or,
4609 Intrinsic::vector_reduce_umax)
4610 : PickCheaper(
Intrinsic::vector_reduce_and,
4614 if (NumSources > 1) {
4615 unsigned CombineOpc =
4616 RequiresOr(
Check) ? Instruction::Or : Instruction::And;
4621 LLVM_DEBUG(
dbgs() <<
"Found sign-bit reduction cmp: " <<
I <<
"\n OldCost: "
4622 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
4624 if (NewCost > OldCost)
4629 Type *ScalarTy = VecTy->getScalarType();
4632 if (NumSources == 1) {
4643 replaceValue(
I, *NewCmp);
4674bool VectorCombine::foldReductionZeroTest(Instruction &
I) {
4683 if (!
II || !
II->hasOneUse())
4686 auto ReduceID =
II->getIntrinsicID();
4687 if (ReduceID != Intrinsic::vector_reduce_or &&
4688 ReduceID != Intrinsic::vector_reduce_umax)
4691 Value *Vec =
II->getArgOperand(0);
4693 if (!VecTy || !VecTy->getElementType()->isIntegerTy())
4698 ? Intrinsic::vector_reduce_or
4713 LLVM_DEBUG(
dbgs() <<
"Found a reduction zero test: " <<
I <<
"\n OldCost: "
4714 << OldCost <<
" vs NewCost: " << NewCost <<
"\n");
4716 if (!OldCost.
isValid() || !NewCost.
isValid() || NewCost > OldCost)
4722 replaceValue(
I, *NewReduce);
4747bool VectorCombine::foldICmpEqZeroVectorReduce(Instruction &
I) {
4758 switch (
II->getIntrinsicID()) {
4759 case Intrinsic::vector_reduce_add:
4760 case Intrinsic::vector_reduce_or:
4761 case Intrinsic::vector_reduce_umin:
4762 case Intrinsic::vector_reduce_umax:
4763 case Intrinsic::vector_reduce_smin:
4764 case Intrinsic::vector_reduce_smax:
4770 Value *InnerOp =
II->getArgOperand(0);
4813 switch (
II->getIntrinsicID()) {
4814 case Intrinsic::vector_reduce_add: {
4819 unsigned NumElems = XTy->getNumElements();
4825 if (LeadingZerosX <= LostBits || LeadingZerosFX <= LostBits)
4833 case Intrinsic::vector_reduce_smin:
4834 case Intrinsic::vector_reduce_smax:
4844 LLVM_DEBUG(
dbgs() <<
"Found a reduction to 0 comparison with removable op: "
4860 case Intrinsic::vector_reduce_add:
4861 case Intrinsic::vector_reduce_or:
4867 case Intrinsic::vector_reduce_umin:
4868 case Intrinsic::vector_reduce_umax:
4869 case Intrinsic::vector_reduce_smin:
4870 case Intrinsic::vector_reduce_smax:
4882 NewReduceCost + (InnerOp->
hasOneUse() ? 0 : ExtCost);
4884 LLVM_DEBUG(
dbgs() <<
"Found a removable extension before reduction: "
4885 << *InnerOp <<
"\n OldCost: " << OldCost
4886 <<
" vs NewCost: " << NewCost <<
"\n");
4892 if (NewCost > OldCost)
4901 Builder.
CreateICmp(Pred, NewReduce, ConstantInt::getNullValue(Ty));
4902 replaceValue(
I, *NewCmp);
4933bool VectorCombine::foldEquivalentReductionCmp(Instruction &
I) {
4936 const APInt *CmpVal;
4941 if (!
II || !
II->hasOneUse())
4944 const auto IsValidOrUmaxCmp = [&]() {
4953 bool IsPositive = CmpVal->
isAllOnes() && Pred == ICmpInst::ICMP_SGT;
4955 bool IsNegative = (CmpVal->
isZero() || CmpVal->
isOne() || *CmpVal == 2) &&
4956 Pred == ICmpInst::ICMP_SLT;
4957 return IsEquality || IsPositive || IsNegative;
4960 const auto IsValidAndUminCmp = [&]() {
4965 const auto LeadingOnes = CmpVal->
countl_one();
4972 bool IsNegative = CmpVal->
isZero() && Pred == ICmpInst::ICMP_SLT;
4981 ((*CmpVal)[0] || (*CmpVal)[1]) && Pred == ICmpInst::ICMP_SGT;
4982 return IsEquality || IsNegative || IsPositive;
4990 switch (OriginalIID) {
4991 case Intrinsic::vector_reduce_or:
4992 if (!IsValidOrUmaxCmp())
4994 AlternativeIID = Intrinsic::vector_reduce_umax;
4996 case Intrinsic::vector_reduce_umax:
4997 if (!IsValidOrUmaxCmp())
4999 AlternativeIID = Intrinsic::vector_reduce_or;
5001 case Intrinsic::vector_reduce_and:
5002 if (!IsValidAndUminCmp())
5004 AlternativeIID = Intrinsic::vector_reduce_umin;
5006 case Intrinsic::vector_reduce_umin:
5007 if (!IsValidAndUminCmp())
5009 AlternativeIID = Intrinsic::vector_reduce_and;
5022 if (ReductionOpc != Instruction::ICmp)
5033 <<
"\n OrigCost: " << OrigCost
5034 <<
" vs AltCost: " << AltCost <<
"\n");
5036 if (AltCost >= OrigCost)
5040 Type *ScalarTy = VecTy->getScalarType();
5043 Builder.
CreateICmp(Pred, NewReduce, ConstantInt::get(ScalarTy, *CmpVal));
5045 replaceValue(
I, *NewCmp);
5059 unsigned Depth = 0) {
5060 constexpr unsigned MaxLocalDepth = 2;
5061 if (
Depth > MaxLocalDepth)
5064 auto NumSignBits = [&](
const Value *
X) {
5067 if (NumSignBits(V) == V->getType()->getScalarSizeInBits())
5072 return NumSignBits(
A) >= 2 && NumSignBits(
B) >= 2 &&
5083bool VectorCombine::foldReduceAddCmpZero(Instruction &
I) {
5093 if (!VecTy || VecTy->getNumElements() < 2)
5099 if (!IsNonNegative && !IsNonPositive)
5104 unsigned NumElts = VecTy->getNumElements();
5106 if (
Log2_32(NumElts) >= NumSignBits)
5109 ICmpInst::Predicate NewPred;
5111 case ICmpInst::ICMP_EQ:
5112 case ICmpInst::ICMP_ULE:
5113 case ICmpInst::ICMP_SLE:
5114 case ICmpInst::ICMP_SGE:
5115 NewPred = ICmpInst::ICMP_EQ;
5117 case ICmpInst::ICMP_NE:
5118 case ICmpInst::ICMP_UGT:
5119 case ICmpInst::ICMP_SGT:
5120 case ICmpInst::ICMP_SLT:
5121 NewPred = ICmpInst::ICMP_NE;
5131 if (!IsNonNegative &&
5132 (Pred == ICmpInst::ICMP_SGT || Pred == ICmpInst::ICMP_SLE))
5134 if (!IsNonPositive &&
5135 (Pred == ICmpInst::ICMP_SLT || Pred == ICmpInst::ICMP_SGE))
5137 if ((Pred == ICmpInst::ICMP_SGT || Pred == ICmpInst::ICMP_SLE ||
5138 Pred == ICmpInst::ICMP_SLT || Pred == ICmpInst::ICMP_SGE) &&
5139 Log2_32(NumElts) >= NumSignBits - 1)
5143 Instruction::Add, VecTy, std::nullopt,
CostKind);
5145 Instruction::Or, VecTy, std::nullopt,
CostKind);
5147 Intrinsic::umax, VecTy, FastMathFlags(),
CostKind);
5150 bool UseOr = OrCost.
isValid() && (!UmaxCost.
isValid() || OrCost <= UmaxCost);
5152 if (AltCost > OrigCost)
5158 Intrinsic::vector_reduce_umax, {VecTy}, {Vec});
5159 Worklist.pushValue(NewReduce);
5161 NewPred, NewReduce, ConstantInt::getNullValue(VecTy->getScalarType()));
5162 replaceValue(
I, *NewCmp);
5171 constexpr unsigned MaxVisited = 32;
5174 bool FoundReduction =
false;
5177 while (!WorkList.
empty()) {
5179 for (
User *U :
I->users()) {
5181 if (!UI || !Visited.
insert(UI).second)
5183 if (Visited.
size() > MaxVisited)
5189 switch (
II->getIntrinsicID()) {
5190 case Intrinsic::vector_reduce_add:
5191 case Intrinsic::vector_reduce_mul:
5192 case Intrinsic::vector_reduce_and:
5193 case Intrinsic::vector_reduce_or:
5194 case Intrinsic::vector_reduce_xor:
5195 case Intrinsic::vector_reduce_smin:
5196 case Intrinsic::vector_reduce_smax:
5197 case Intrinsic::vector_reduce_umin:
5198 case Intrinsic::vector_reduce_umax:
5199 FoundReduction =
true;
5212 return FoundReduction;
5225bool VectorCombine::foldSelectShuffle(Instruction &
I,
bool FromReduction) {
5230 if (!Op0 || !Op1 || Op0 == Op1 || !Op0->isBinaryOp() || !Op1->isBinaryOp() ||
5231 VT != Op0->getType())
5238 SmallPtrSet<Instruction *, 4> InputShuffles({SVI0A, SVI0B, SVI1A, SVI1B});
5240 if (!
I ||
I->getOperand(0)->getType() != VT)
5242 return any_of(
I->users(), [&](User *U) {
5243 return U != Op0 && U != Op1 &&
5244 !(isa<ShuffleVectorInst>(U) &&
5245 (InputShuffles.contains(cast<Instruction>(U)) ||
5246 isInstructionTriviallyDead(cast<Instruction>(U))));
5249 if (checkSVNonOpUses(SVI0A) || checkSVNonOpUses(SVI0B) ||
5250 checkSVNonOpUses(SVI1A) || checkSVNonOpUses(SVI1B))
5258 for (
auto *U :
I->users()) {
5260 if (!SV || SV->getType() != VT)
5262 if ((SV->getOperand(0) != Op0 && SV->getOperand(0) != Op1) ||
5263 (SV->getOperand(1) != Op0 && SV->getOperand(1) != Op1))
5270 if (!collectShuffles(Op0) || !collectShuffles(Op1))
5274 if (FromReduction && Shuffles.
size() > 1)
5279 if (!FromReduction) {
5280 for (
size_t Idx = 0,
E = Shuffles.
size(); Idx !=
E; ++Idx) {
5281 for (
auto *U : Shuffles[Idx]->
users()) {
5296 int MaxV1Elt = 0, MaxV2Elt = 0;
5297 unsigned NumElts = VT->getNumElements();
5298 for (ShuffleVectorInst *SVN : Shuffles) {
5299 SmallVector<int>
Mask;
5300 SVN->getShuffleMask(Mask);
5304 Value *SVOp0 = SVN->getOperand(0);
5305 Value *SVOp1 = SVN->getOperand(1);
5310 for (
int &Elem : Mask) {
5316 if (SVOp0 == Op1 && SVOp1 == Op0) {
5320 if (SVOp0 != Op0 || SVOp1 != Op1)
5326 SmallVector<int> ReconstructMask;
5327 for (
unsigned I = 0;
I <
Mask.size();
I++) {
5330 }
else if (Mask[
I] <
static_cast<int>(NumElts)) {
5331 MaxV1Elt = std::max(MaxV1Elt, Mask[
I]);
5332 auto It =
find_if(
V1, [&](
const std::pair<int, int> &
A) {
5333 return Mask[
I] ==
A.first;
5339 V1.emplace_back(Mask[
I],
V1.size());
5342 MaxV2Elt = std::max<int>(MaxV2Elt, Mask[
I] - NumElts);
5343 auto It =
find_if(V2, [&](
const std::pair<int, int> &
A) {
5344 return Mask[
I] -
static_cast<int>(NumElts) ==
A.first;
5358 sort(ReconstructMask);
5359 OrigReconstructMasks.
push_back(std::move(ReconstructMask));
5366 if (
V1.empty() || V2.
empty() ||
5367 (MaxV1Elt ==
static_cast<int>(
V1.size()) - 1 &&
5368 MaxV2Elt ==
static_cast<int>(V2.
size()) - 1))
5380 if (InputShuffles.contains(SSV))
5382 return SV->getMaskValue(M);
5390 std::pair<int, int>
Y) {
5391 int MXA = GetBaseMaskValue(
A,
X.first);
5392 int MYA = GetBaseMaskValue(
A,
Y.first);
5396 return SortBase(SVI0A,
A,
B);
5398 stable_sort(V2, [&](std::pair<int, int>
A, std::pair<int, int>
B) {
5399 return SortBase(SVI1A,
A,
B);
5404 for (
const auto &Mask : OrigReconstructMasks) {
5405 SmallVector<int> ReconstructMask;
5406 for (
int M : Mask) {
5408 auto It =
find_if(V, [M](
auto A) {
return A.second ==
M; });
5409 assert(It !=
V.end() &&
"Expected all entries in Mask");
5410 return std::distance(
V.begin(), It);
5414 else if (M <
static_cast<int>(NumElts)) {
5417 ReconstructMask.
push_back(NumElts + FindIndex(V2, M));
5420 ReconstructMasks.
push_back(std::move(ReconstructMask));
5425 SmallVector<int> V1A, V1B, V2A, V2B;
5426 for (
unsigned I = 0;
I <
V1.size();
I++) {
5430 for (
unsigned I = 0;
I < V2.
size();
I++) {
5431 V2A.
push_back(GetBaseMaskValue(SVI1A, V2[
I].first));
5432 V2B.
push_back(GetBaseMaskValue(SVI1B, V2[
I].first));
5434 while (V1A.
size() < NumElts) {
5438 while (V2A.
size() < NumElts) {
5450 VT, VT, SV->getShuffleMask(),
CostKind);
5457 unsigned ElementSize = VT->getElementType()->getPrimitiveSizeInBits();
5458 unsigned MaxVectorSize =
5460 unsigned MaxElementsInVector = MaxVectorSize / ElementSize;
5461 if (MaxElementsInVector == 0)
5470 std::set<SmallVector<int, 4>> UniqueShuffles;
5475 unsigned NumFullVectors =
Mask.size() / MaxElementsInVector;
5476 if (NumFullVectors < 2)
5477 return C + ShuffleCost;
5478 SmallVector<int, 4> SubShuffle(MaxElementsInVector);
5479 unsigned NumUniqueGroups = 0;
5480 unsigned NumGroups =
Mask.size() / MaxElementsInVector;
5483 for (
unsigned I = 0;
I < NumFullVectors; ++
I) {
5484 for (
unsigned J = 0; J < MaxElementsInVector; ++J)
5485 SubShuffle[J] = Mask[MaxElementsInVector *
I + J];
5486 if (UniqueShuffles.insert(SubShuffle).second)
5487 NumUniqueGroups += 1;
5489 return C + ShuffleCost * NumUniqueGroups / NumGroups;
5495 SmallVector<int, 16>
Mask;
5496 SV->getShuffleMask(Mask);
5497 return AddShuffleMaskAdjustedCost(
C, Mask);
5500 auto AllShufflesHaveSameOperands =
5501 [](SmallPtrSetImpl<Instruction *> &InputShuffles) {
5502 if (InputShuffles.size() < 2)
5504 ShuffleVectorInst *FirstSV =
5511 std::next(InputShuffles.begin()), InputShuffles.end(),
5512 [&](Instruction *
I) {
5513 ShuffleVectorInst *SV = dyn_cast<ShuffleVectorInst>(I);
5514 return SV && SV->getOperand(0) == In0 && SV->getOperand(1) == In1;
5523 CostBefore += std::accumulate(Shuffles.begin(), Shuffles.end(),
5525 if (AllShufflesHaveSameOperands(InputShuffles)) {
5526 UniqueShuffles.clear();
5527 CostBefore += std::accumulate(InputShuffles.begin(), InputShuffles.end(),
5530 CostBefore += std::accumulate(InputShuffles.begin(), InputShuffles.end(),
5536 FixedVectorType *Op0SmallVT =
5538 FixedVectorType *Op1SmallVT =
5543 UniqueShuffles.clear();
5544 CostAfter += std::accumulate(ReconstructMasks.begin(), ReconstructMasks.end(),
5546 std::set<SmallVector<int>> OutputShuffleMasks({V1A, V1B, V2A, V2B});
5548 std::accumulate(OutputShuffleMasks.begin(), OutputShuffleMasks.end(),
5551 LLVM_DEBUG(
dbgs() <<
"Found a binop select shuffle pattern: " <<
I <<
"\n");
5553 <<
" vs CostAfter: " << CostAfter <<
"\n");
5554 if (CostBefore < CostAfter ||
5565 if (InputShuffles.contains(SSV))
5567 return SV->getOperand(
Op);
5571 GetShuffleOperand(SVI0A, 1), V1A);
5574 GetShuffleOperand(SVI0B, 1), V1B);
5577 GetShuffleOperand(SVI1A, 1), V2A);
5580 GetShuffleOperand(SVI1B, 1), V2B);
5585 I->copyIRFlags(Op0,
true);
5590 I->copyIRFlags(Op1,
true);
5592 for (
int S = 0,
E = ReconstructMasks.size(); S !=
E; S++) {
5595 replaceValue(*Shuffles[S], *NSV,
false);
5598 Worklist.pushValue(NSV0A);
5599 Worklist.pushValue(NSV0B);
5600 Worklist.pushValue(NSV1A);
5601 Worklist.pushValue(NSV1B);
5611bool VectorCombine::shrinkType(Instruction &
I) {
5612 Value *ZExted, *OtherOperand;
5618 Value *ZExtOperand =
I.getOperand(
I.getOperand(0) == OtherOperand ? 1 : 0);
5622 unsigned BW = SmallTy->getElementType()->getPrimitiveSizeInBits();
5624 if (
I.getOpcode() == Instruction::LShr) {
5641 Instruction::ZExt, BigTy, SmallTy,
5642 TargetTransformInfo::CastContextHint::None,
CostKind);
5647 for (User *U : ZExtOperand->
users()) {
5654 ShrinkCost += ZExtCost;
5669 ShrinkCost += ZExtCost;
5676 Instruction::Trunc, SmallTy, BigTy,
5677 TargetTransformInfo::CastContextHint::None,
CostKind);
5682 if (ShrinkCost > CurrentCost)
5686 Value *Op0 = ZExted;
5689 if (
I.getOperand(0) == OtherOperand)
5696 replaceValue(
I, *NewZExtr);
5702bool VectorCombine::foldInsExtVectorToShuffle(Instruction &
I) {
5703 Value *DstVec, *SrcVec;
5714 if (!DstVecTy || !SrcVecTy ||
5720 if (InsIdx >= NumDstElts || ExtIdx >= NumSrcElts || NumDstElts == 1)
5727 bool NeedExpOrNarrow = NumSrcElts != NumDstElts;
5729 if (NeedDstSrcSwap) {
5731 Mask[InsIdx] = ExtIdx % NumDstElts;
5735 std::iota(
Mask.begin(),
Mask.end(), 0);
5736 Mask[InsIdx] = (ExtIdx % NumDstElts) + NumDstElts;
5749 SmallVector<int> ExtToVecMask;
5750 if (!NeedExpOrNarrow) {
5755 nullptr, {DstVec, SrcVec});
5761 ExtToVecMask[ExtIdx % NumDstElts] = ExtIdx;
5764 DstVecTy, SrcVecTy, ExtToVecMask,
CostKind);
5768 if (!Ext->hasOneUse())
5771 LLVM_DEBUG(
dbgs() <<
"Found a insert/extract shuffle-like pair: " <<
I
5772 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
5775 if (OldCost < NewCost)
5778 if (NeedExpOrNarrow) {
5779 if (!NeedDstSrcSwap)
5792 replaceValue(
I, *Shuf);
5816bool VectorCombine::foldDeinterleaveInterleavePair(Instruction &
I) {
5833 if (
U.getUser()->isDroppable())
5837 if (!Extract || Extract->getNumIndices() != 1)
5840 unsigned Index = *Extract->idx_begin();
5841 if (Index >= Factor || CurrentUses[Index])
5849 IntrinsicInst *Interleave =
nullptr;
5850 unsigned NumVisited = 0;
5854 return CB->arg_size();
5855 return Inst->getNumOperands();
5858 auto IsSupportedElementwise = [&](
Instruction *Inst) {
5864 if (
II->hasOperandBundles() ||
5867 }
else if (!
isa<BinaryOperator, UnaryOperator, CastInst, CmpInst,
5868 SelectInst, FreezeInst>(Inst)) {
5874 for (
unsigned Op = 0,
E = GetNumDataOperands(Inst);
Op !=
E; ++
Op) {
5877 OperandTy->getElementCount() != ResultTy->getElementCount())
5889 NumVisited += Factor;
5891 for (Use *&CurrentUse : CurrentUses) {
5892 Use *NextUse = CurrentUse->getUser()->getSingleUndroppableUse();
5898 CurrentUse = NextUse;
5903 II &&
II->getIntrinsicID() == ExpectedInterleaveIID) {
5904 if (
II->hasOperandBundles())
5907 for (
unsigned Index = 0;
Index != Factor; ++
Index)
5908 if (CurrentUses[Index]->getUser() !=
II ||
5909 CurrentUses[Index]->getOperandNo() != Index)
5917 if (!IsSupportedElementwise(FirstInst))
5920 unsigned ChainOperand = CurrentUses.front()->getOperandNo();
5921 if (
any_of(CurrentUses, [&](Use *U) {
5923 return Inst != FirstInst && (
U->getOperandNo() != ChainOperand ||
5924 !FirstInst->isSameOperationAs(Inst));
5928 auto GetSplatOrScalar = [](
Value *
V) {
5935 for (
unsigned Op = 0,
E = GetNumDataOperands(FirstInst);
Op !=
E; ++
Op) {
5936 if (
Op == ChainOperand)
5939 Value *CommonValue = GetSplatOrScalar(FirstInst->getOperand(
Op));
5940 if (!CommonValue ||
any_of(CurrentUses, [&](Use *U) {
5942 return Inst != FirstInst &&
5956 ElementCount WideEC =
5959 auto CreateWideInstruction = [&](
Instruction *NarrowInst,
5962 assert(IsSupportedElementwise(NarrowInst) &&
5963 "Expected supported elementwise");
5967 return Builder.
CreateCast(Cast->getOpcode(), NewOperands[0],
5970 return Builder.
CreateCmp(
Cmp->getPredicate(), NewOperands[0],
5974 NewOperands[0], NewOperands[1], NewOperands[2],
"",
5986 for (
const ElementwiseStep &Step : Steps) {
5988 unsigned ChainOperand = Step.front()->getOperandNo();
5993 unsigned NumOperands = GetNumDataOperands(NarrowInst);
5994 SmallVector<Value *, 4> NewOperands;
5995 NewOperands.
reserve(NumOperands);
5997 for (
unsigned Op = 0;
Op != NumOperands; ++
Op) {
6000 if (
Op == ChainOperand)
6001 Operand = WideValue;
6007 auto *WideResultTy =
6010 CreateWideInstruction(NarrowInst, NewOperands, WideResultTy);
6019 WideValue = NewValue;
6023 replaceValue(*Interleave, *WideValue);
6031bool VectorCombine::foldInterleaveIntrinsics(Instruction &
I) {
6032 const APInt *SplatVal0, *SplatVal1;
6042 auto *ExtVTy = VectorType::getExtendedElementVectorType(VTy);
6043 unsigned Width = VTy->getElementType()->getIntegerBitWidth();
6052 LLVM_DEBUG(
dbgs() <<
"VC: The cost to cast from " << *ExtVTy <<
" to "
6053 << *
I.getType() <<
" is too high.\n");
6057 APInt NewSplatVal = SplatVal1->
zext(Width * 2);
6058 NewSplatVal <<= Width;
6059 NewSplatVal |= SplatVal0->
zext(Width * 2);
6061 ExtVTy->getElementCount(), ConstantInt::get(
F.getContext(), NewSplatVal));
6096bool VectorCombine::foldDeinterleaveIntrinsics(Instruction &
I) {
6097 if (foldDeinterleaveInterleavePair(
I))
6101 if (
DL->isBigEndian())
6104 using namespace PatternMatch;
6105 Value *DeinterleavedVal;
6116 unsigned HalfElementWidth = ElementWidth / 2;
6120 std::array<ExtractValueInst *, 2> OrigFields{};
6121 for (User *Usr :
I.users()) {
6124 if (!
E ||
E->getNumIndices() != 1)
6126 unsigned Idx = *
E->idx_begin();
6128 if (Idx >= 2 || OrigFields[Idx] || !
E->hasNUses(2))
6130 OrigFields[Idx] =
E;
6134 SmallVector<Instruction *, 2> MergeInsts;
6135 for (
auto *FieldUsr : OrigFields[0]->
users()) {
6143 auto MatchMerge = [&](void) ->
bool {
6146 return match(MergeInsts[0],
6150 match(MergeInsts[1],
6155 if (!MatchMerge()) {
6156 std::swap(MergeInsts[0], MergeInsts[1]);
6171 auto *NewFieldTy = VecTy->getWithNewBitWidth(HalfElementWidth);
6181 if (OldCost <= NewCost || !NewCost.
isValid()) {
6183 dbgs() <<
"VC: New deinterleave2 sequence cost (" << NewCost <<
")"
6184 <<
" is higher than that of the old one (" << OldCost <<
")\n");
6192 Intrinsic::vector_deinterleave2, {NewVecTy}, {NewVecCast});
6193 for (
auto [Idx, MergeInst] :
enumerate(MergeInsts)) {
6195 NewField = Builder.
CreateBitCast(NewField, MergeInst->getType());
6196 replaceValue(*MergeInst, *NewField);
6202bool VectorCombine::foldBitcastOfVPLoad(Instruction &
I) {
6203 const DataLayout &
DL =
I.getDataLayout();
6218 DL.getValueOrABITypeAlignment(
II->getPointerAlignment(), OrigVecTy);
6219 ElementCount OrigVecCnt = OrigVecTy->getElementCount();
6221 ElementCount NewVecCnt = NewVecTy->getElementCount();
6233 II->getMemoryPointerParam(),
false,
6239 {Intrinsic::vp_load, NewVecTy,
II->getMemoryPointerParam(),
false,
6243 <<
" NewCost=" << NewCost <<
"\n");
6244 if (NewCost > OldCost || !NewCost.
isValid())
6251 NewVecTy, Intrinsic::vp_load,
6252 {
II->getMemoryPointerParam(), NewMask, NewEVL});
6255 0, AttrBuilder(
II->getContext()).addAlignmentAttr(OrigAlign));
6256 replaceValue(*Cast, *NewVP);
6266bool VectorCombine::foldBitOrderReverseAndSwap(Instruction &
I) {
6270 Type *Ty =
X->getType();
6271 Type *VecTy =
I.getOperand(0)->getType();
6285 if (CanUseBswap || CanUseFshl) {
6296 IntrinsicCostAttributes ICABSwap(Intrinsic::bswap, Ty, {Ty});
6297 IntrinsicCostAttributes ICABFshl(Intrinsic::fshl, Ty, {
X,
X, HalfBW},
6299 IntrinsicCostAttributes ICABRev(Intrinsic::bitreverse, Ty, {Ty});
6304 if (!InnerCall->hasOneUse())
6307 else if (!InnerBitCast->hasOneUse())
6310 <<
"\n OldCost: " << OldCost
6311 <<
" vs NewCost: " << NewCost <<
"\n");
6312 if (NewCost.isValid() && NewCost < OldCost) {
6318 Worklist.pushValue(Swap);
6320 replaceValue(
I, *BRev);
6329 Type *Ty =
I.getType();
6331 TypeSize ElementSize =
DL->getTypeStoreSize(Ty);
6334 Type *NewVecTy = VectorType::get(I8Ty, NewVecCnt);
6347 IntrinsicCostAttributes ICANew(Intrinsic::bitreverse, NewVecTy, {NewVecTy});
6350 InstructionCost NewCost = CastToVecCost + NewIntrinsicCost + CastToOrigCost;
6351 if (!InnerII->hasOneUse())
6354 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
6356 if (!NewCost.
isValid() || NewCost >= OldCost)
6364 replaceValue(
I, *CastToOrig);
6374 unsigned RawNumElements = MaxIdx + 1u;
6377 if (!
TTI.isTypeLegal(ElemTy))
6378 return RawNumElements;
6380 TypeSize ElemSize =
DL.getTypeSizeInBits(ElemTy);
6382 return RawNumElements;
6387 return RawNumElements;
6392 if (ElemsPerReg == 0 || RawNumElements <= ElemsPerReg)
6393 return RawNumElements;
6395 return alignTo(RawNumElements, ElemsPerReg);
6399bool VectorCombine::shrinkLoadForShuffles(Instruction &
I) {
6401 if (!OldLoad || !OldLoad->isSimple())
6408 unsigned const OldNumElements = OldLoadTy->getNumElements();
6414 using IndexRange = std::pair<int, int>;
6415 auto GetIndexRangeInShuffles = [&]() -> std::optional<IndexRange> {
6416 IndexRange OutputRange = IndexRange(OldNumElements, -1);
6417 for (llvm::Use &Use :
I.uses()) {
6419 User *Shuffle =
Use.getUser();
6424 return std::nullopt;
6431 for (
int Index : Mask) {
6432 if (Index >= 0 && Index <
static_cast<int>(OldNumElements)) {
6433 OutputRange.first = std::min(Index, OutputRange.first);
6434 OutputRange.second = std::max(Index, OutputRange.second);
6439 if (OutputRange.second < OutputRange.first)
6440 return std::nullopt;
6446 if (std::optional<IndexRange> Indices = GetIndexRangeInShuffles()) {
6447 unsigned const NewNumElements =
6452 if (NewNumElements < OldNumElements) {
6457 Type *ElemTy = OldLoadTy->getElementType();
6459 Value *PtrOp = OldLoad->getPointerOperand();
6462 Instruction::Load, OldLoad->getType(), OldLoad->getAlign(),
6463 OldLoad->getPointerAddressSpace(),
CostKind);
6466 OldLoad->getPointerAddressSpace(),
CostKind);
6468 using UseEntry = std::pair<ShuffleVectorInst *, std::vector<int>>;
6470 unsigned const MaxIndex = NewNumElements * 2u;
6472 for (llvm::Use &Use :
I.uses()) {
6479 ArrayRef<int> OldMask = Shuffle->getShuffleMask();
6485 for (
int Index : OldMask) {
6486 if (Index >=
static_cast<int>(MaxIndex))
6500 dbgs() <<
"Found a load used only by shufflevector instructions: "
6501 <<
I <<
"\n OldCost: " << OldCost
6502 <<
" vs NewCost: " << NewCost <<
"\n");
6504 if (OldCost < NewCost || !NewCost.
isValid())
6510 NewLoad->copyMetadata(
I);
6513 for (UseEntry &Use : NewUses) {
6514 ShuffleVectorInst *Shuffle =
Use.first;
6515 std::vector<int> &NewMask =
Use.second;
6522 replaceValue(*Shuffle, *NewShuffle,
false);
6535bool VectorCombine::shrinkPhiOfShuffles(Instruction &
I) {
6537 if (!Phi ||
Phi->getNumIncomingValues() != 2u)
6541 ArrayRef<int> Mask0;
6542 ArrayRef<int> Mask1;
6555 auto const InputNumElements = InputVT->getNumElements();
6557 if (InputNumElements >= ResultVT->getNumElements())
6562 SmallVector<int, 16> NewMask;
6565 for (
auto [
M0,
M1] :
zip(Mask0, Mask1)) {
6566 if (
M0 >= 0 &&
M1 >= 0)
6568 else if (
M0 == -1 &&
M1 == -1)
6581 int MaskOffset = NewMask[0
u];
6582 unsigned Index = (InputNumElements + MaskOffset) % InputNumElements;
6585 for (
unsigned I = 0u;
I < InputNumElements; ++
I) {
6599 <<
"\n OldCost: " << OldCost <<
" vs NewCost: " << NewCost
6602 if (NewCost > OldCost)
6614 auto *NewPhi = Builder.
CreatePHI(NewShuf0->getType(), 2u);
6616 NewPhi->addIncoming(
Op,
Phi->getIncomingBlock(1u));
6622 replaceValue(*Phi, *NewShuf1);
6628bool VectorCombine::run() {
6642 auto Opcode =
I.getOpcode();
6650 if (IsFixedVectorType) {
6652 case Instruction::InsertElement:
6653 if (vectorizeLoadInsert(
I))
6656 case Instruction::ShuffleVector:
6657 if (widenSubvectorLoad(
I))
6668 if (scalarizeOpOrCmp(
I))
6670 if (scalarizeLoad(
I))
6672 if (scalarizeExtExtract(
I))
6674 if (foldInterleaveIntrinsics(
I))
6676 if (foldBitcastOfVPLoad(
I))
6680 if (foldDeinterleaveIntrinsics(
I))
6683 if (Opcode == Instruction::Store)
6684 if (foldSingleElementStore(
I))
6688 if (TryEarlyFoldsOnly)
6691 if (Opcode == Instruction::Call)
6692 if (foldBitOrderReverseAndSwap(
I))
6694 if (Opcode == Instruction::BitCast)
6695 if (foldBitOrderReverseAndSwap(
I))
6702 if (IsFixedVectorType) {
6704 case Instruction::InsertElement:
6705 if (foldInsExtFNeg(
I))
6707 if (foldInsExtBinop(
I))
6709 if (foldInsExtVectorToShuffle(
I))
6712 case Instruction::ShuffleVector:
6713 if (foldPermuteOfBinops(
I))
6715 if (foldShuffleOfBinops(
I))
6717 if (foldShuffleOfSelects(
I))
6719 if (foldShuffleOfCastops(
I))
6721 if (foldShuffleOfShuffles(
I))
6723 if (foldPermuteOfIntrinsic(
I))
6725 if (foldShufflesOfLengthChangingShuffles(
I))
6727 if (foldShuffleOfIntrinsics(
I))
6729 if (foldSelectShuffle(
I))
6731 if (foldShuffleToIdentity(
I))
6734 case Instruction::Load:
6735 if (shrinkLoadForShuffles(
I))
6738 case Instruction::BitCast:
6739 if (foldBitcastShuffle(
I))
6741 if (foldSelectsFromBitcast(
I))
6744 case Instruction::And:
6745 case Instruction::Or:
6746 case Instruction::Xor:
6747 if (foldBitOpOfCastops(
I))
6749 if (foldBitOpOfCastConstant(
I))
6752 case Instruction::PHI:
6753 if (shrinkPhiOfShuffles(
I))
6763 case Instruction::Call:
6764 if (foldShuffleFromReductions(
I))
6766 if (foldCastFromReductions(
I))
6769 case Instruction::ExtractElement:
6770 if (foldShuffleChainsToReduce(
I))
6773 case Instruction::ICmp:
6774 if (foldSignBitReductionCmp(
I))
6776 if (foldICmpEqZeroVectorReduce(
I))
6778 if (foldReductionZeroTest(
I))
6780 if (foldEquivalentReductionCmp(
I))
6782 if (foldReduceAddCmpZero(
I))
6785 case Instruction::FCmp:
6786 if (foldExtractExtract(
I))
6789 case Instruction::Or:
6790 if (foldConcatOfBoolMasks(
I))
6795 if (foldExtractExtract(
I))
6797 if (foldExtractedCmps(
I))
6799 if (foldBinopOfReductions(
I))
6808 bool MadeChange =
false;
6809 for (BasicBlock &BB :
F) {
6821 if (!
I->isDebugOrPseudoInst())
6822 MadeChange |= FoldInst(*
I);
6829 while (!Worklist.isEmpty()) {
6839 MadeChange |= FoldInst(*
I);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< unsigned > MaxInstrsToScan("aggressive-instcombine-max-scan-instrs", cl::init(64), cl::Hidden, cl::desc("Max number of instructions to scan for aggressive instcombine."))
This is the interface for LLVM's primary stateless and local alias analysis.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
This file defines the DenseMap class.
This is the interface for a simple mod/ref and alias analysis over globals.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static void eraseInstruction(Instruction &I, ICFLoopSafetyInfo &SafetyInfo, MemorySSAUpdater &MSSAU)
uint64_t IntrinsicInst * II
FunctionAnalysisManager FAM
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
This file defines the scope_exit class, which executes user-defined cleanup logic at scope exit.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
static bool isEquivBitcast(Value *X, Value *Y)
Helper to peek through bitcasts to the same value.
static bool isFreeConcat(ArrayRef< InstLane > Item, TTI::TargetCostKind CostKind, const TargetTransformInfo &TTI)
Detect concat of multiple values into a vector.
static void analyzeCostOfVecReduction(const IntrinsicInst &II, TTI::TargetCostKind CostKind, const TargetTransformInfo &TTI, InstructionCost &CostBeforeReduction, InstructionCost &CostAfterReduction)
static Value * generateNewInstTree(ArrayRef< InstLane > Item, Use *From, const DenseSet< std::pair< Value *, Use * > > &IdentityLeafs, const DenseSet< std::pair< Value *, Use * > > &SplatLeafs, const DenseSet< std::pair< Value *, Use * > > &ConcatLeafs, IRBuilderBase &Builder, InstructionWorklist &WorkList, const TargetTransformInfo *TTI)
static SmallVector< InstLane > generateInstLaneVectorFromOperand(ArrayRef< InstLane > Item, int Op)
static Value * createShiftShuffle(Value *Vec, unsigned OldIndex, unsigned NewIndex, IRBuilderBase &Builder)
Create a shuffle that translates (shifts) 1 element from the input vector to a new element location.
std::pair< Value *, int > InstLane
static bool isKnownNonPositive(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Used by foldReduceAddCmpZero to check if we can prove that a value is non-positive.
static Align computeAlignmentAfterScalarization(Align VectorAlignment, Type *ScalarType, Value *Idx, const DataLayout &DL)
The memory operation on a vector of ScalarType had alignment of VectorAlignment.
static bool feedsIntoVectorReduction(ShuffleVectorInst *SVI)
Returns true if this ShuffleVectorInst eventually feeds into a vector reduction intrinsic (e....
static cl::opt< bool > DisableVectorCombine("disable-vector-combine", cl::init(false), cl::Hidden, cl::desc("Disable all vector combine transforms"))
static bool canWidenLoad(LoadInst *Load, const TargetTransformInfo &TTI)
static const unsigned InvalidIndex
static Value * translateExtract(ExtractElementInst *ExtElt, unsigned NewIndex, IRBuilderBase &Builder)
Given an extract element instruction with constant index operand, shuffle the source vector (shift th...
static ScalarizationResult canScalarizeAccess(VectorType *VecTy, Value *Idx, const SimplifyQuery &SQ)
Check if it is legal to scalarize a memory access to VecTy at index Idx.
static cl::opt< unsigned > MaxInstrsToScan("vector-combine-max-scan-instrs", cl::init(30), cl::Hidden, cl::desc("Max number of instructions to scan for vector combining."))
static cl::opt< bool > DisableBinopExtractShuffle("disable-binop-extract-shuffle", cl::init(false), cl::Hidden, cl::desc("Disable binop extract to shuffle transforms"))
static unsigned getAlignedNumElements(unsigned MaxIdx, FixedVectorType *LoadTy, const TargetTransformInfo &TTI, const DataLayout &DL)
Given the maximum shuffle index and load vector type, compute the number of elements for the shrunk l...
static InstLane lookThroughShuffles(Value *V, int Lane)
static bool isMemModifiedBetween(BasicBlock::iterator Begin, BasicBlock::iterator End, const MemoryLocation &Loc, AAResults &AA)
static constexpr int Concat[]
A manager for alias analyses.
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
uint64_t getZExtValue() const
Get zero extended value.
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
unsigned getBitWidth() const
Return the number of bits in the APInt.
bool isNegative() const
Determine sign of this APInt.
unsigned countl_one() const
Count the number of leading one bits.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
bool isOne() const
Determine if this is a value of 1.
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
A function analysis which provides an AssumptionCache.
A cache of @llvm.assume calls within a function.
InstListType::iterator iterator
Instruction iterators...
BinaryOps getOpcode() const
Represents analyses that only rely on functions' control flow.
Value * getArgOperand(unsigned i) const
void addParamAttrs(unsigned ArgNo, const AttrBuilder &B)
Adds attributes to the indicated argument.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
bool isFPPredicate() const
static LLVM_ABI std::optional< CmpPredicate > getMatching(CmpPredicate A, CmpPredicate B)
Compares two CmpPredicates taking samesign into account and returns the canonicalized CmpPredicate if...
static LLVM_ABI Constant * getExtractElement(Constant *Vec, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI Constant * getBinOpIdentity(unsigned Opcode, Type *Ty, bool AllowRHSConstant=false, bool NSZ=false)
Return the identity constant for a binary opcode.
This is the shared class of boolean and integer constants.
const APInt & getValue() const
Return the constant as an APInt value reference.
This class represents a range of values.
LLVM_ABI ConstantRange urem(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned remainder operation of...
LLVM_ABI ConstantRange binaryAnd(const ConstantRange &Other) const
Return a new range representing the possible values resulting from a binary-and of a value in this ra...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
iterator find(const_arg_type_t< KeyT > Val)
Implements a dense probed hash-table based set.
Analysis pass which computes a DominatorTree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
LLVM_ABI bool isReachableFromEntry(const Use &U) const
Provide an overload for a Use.
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
static constexpr ElementCount get(ScalarTy MinVal, bool Scalable)
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
Common base class shared among various IRBuilders.
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
Value * CreateNUWMul(Value *LHS, Value *RHS, const Twine &Name="")
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
LLVM_ABI Value * CreateSelectFMF(Value *C, Value *True, Value *False, FMFSource FMFSource, const Twine &Name="", Instruction *MDFrom=nullptr)
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
ConstantInt * getTrue()
Get the constant value for i1 true.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateFreeze(Value *V, const Twine &Name="")
void SetCurrentDebugLocation(const DebugLoc &L)
Set location information used by debugging information.
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Value * CreateIsNotNeg(Value *Arg, const Twine &Name="")
Return a boolean value testing if Arg > -1.
Value * CreateInBoundsGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="")
Value * CreatePointerBitCastOrAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="")
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
LLVM_ABI Value * CreateOrReduce(Value *Src)
Create a vector int OR reduction intrinsic of the source vector.
ConstantInt * getInt32(uint32_t C)
Get a constant 32-bit value.
Value * CreateCmp(CmpInst::Predicate Pred, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
InstTy * Insert(InstTy *I, const Twine &Name="") const
Insert and return the specified instruction.
Value * CreateIsNeg(Value *Arg, const Twine &Name="")
Return a boolean value testing if Arg < 0.
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
LLVM_ABI Value * CreateNAryOp(unsigned Opc, ArrayRef< Value * > Ops, const Twine &Name="", MDNode *FPMathTag=nullptr)
Create either a UnaryOperator or BinaryOperator depending on Opc.
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Value * CreateAnd(Value *LHS, Value *RHS, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
Value * CreateTrunc(Value *V, Type *DestTy, const Twine &Name="", bool IsNUW=false, bool IsNSW=false)
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Value * CreateBinOp(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateFNegFMF(Value *V, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
IntegerType * getInt8Ty()
Fetch the type representing an 8-bit integer.
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
InstructionWorklist - This is the worklist management logic for InstCombine and other simplification ...
void push(Instruction *I)
Push the instruction onto the worklist stack.
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI void andIRFlags(const Value *V)
Logical 'and' of any supported wrapping, exact, and fast-math flags of V and this instruction.
LLVM_ABI void setNonNeg(bool b=true)
Set or clear the nneg flag on this instruction, which must be a zext instruction.
LLVM_ABI bool comesBefore(const Instruction *Other) const
Given an instruction Other in the same basic block as this instruction, return true if this instructi...
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
bool isIdempotent() const
Return true if the instruction is idempotent:
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI bool hasAllowReassoc() const LLVM_READONLY
Determine whether the allow-reassociation flag is set.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
void setAlignment(Align Align)
Type * getPointerOperandType() const
Align getAlign() const
Return the alignment of the access that is being performed.
Representation for a specific memory location.
static LLVM_ABI MemoryLocation get(const LoadInst *LI)
Return a location with information about the memory reference by the given instruction.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
const SDValue & getOperand(unsigned Num) const
bool contains(const_arg_type key) const
Check if the SetVector contains the given key.
bool empty() const
Determine if the SetVector is empty or not.
bool insert(const value_type &X)
Insert a new element into the SetVector.
This instruction constructs a fixed permutation of two input vectors.
int getMaskValue(unsigned Elt) const
Return the shuffle mask value of this instruction for the given element index.
VectorType * getType() const
Overload to return most specific vector type.
static LLVM_ABI void getShuffleMask(const Constant *Mask, SmallVectorImpl< int > &Result)
Convert the input shuffle mask operand to a vector of integers.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static void commuteShuffleMask(MutableArrayRef< int > Mask, unsigned InVecNumElts)
Change values in a shuffle permute mask assuming the two vector operands of length InVecNumElts have ...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
void setAlignment(Align Align)
Analysis pass providing the TargetTransformInfo.
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
bool isPointerTy() const
True if this is an instance of PointerType.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
const Value * stripAndAccumulateInBoundsConstantOffsets(const DataLayout &DL, APInt &Offset) const
This is a wrapper around stripAndAccumulateConstantOffsets with the in-bounds requirement set to fals...
LLVM_ABI bool hasOneUser() const
Return true if there is exactly one user of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
unsigned getValueID() const
Return an ID for the concrete type of this object.
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &)
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr bool isZero() const
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
const APInt & smin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be signed.
const APInt & smax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be signed.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Intrinsic::ID getInterleaveIntrinsicID(unsigned Factor)
Returns the corresponding llvm.vector.interleaveN intrinsic for factor N.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
DisjointOr_match< LHS, RHS > m_DisjointOr(const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_right_shift_op > m_Shr(const LHS &L, const RHS &R)
Matches logical shift operations.
CmpClass_match< LHS, RHS, ICmpInst, true > m_c_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
Matches an ICmp with a predicate over LHS and RHS in either order.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_non_zero_int > m_NonZeroInt()
Match a non-zero integer or a vector with all non-zero elements.
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
auto m_AnyIntrinsic()
Matches any intrinsic call and ignore it.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_bitwiselogic_op, true > m_c_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
match_combine_or< CastInst_match< OpTy, SExtInst >, NNegZExt_match< OpTy > > m_SExtLike(const OpTy &Op)
Match either "sext" or "zext nneg".
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_Deinterleave2(const Opnd &Op)
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ Valid
The data is already valid.
initializer< Ty > init(const Ty &Val)
DXILDebugInfoMap run(Module &M)
@ User
could "use" a pointer
NodeAddr< PhiNode * > Phi
NodeAddr< UseNode * > Use
friend class Instruction
Iterator for Instructions in a `BasicBlock.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
LLVM_ABI bool willNotFreeBetween(const Instruction *Assume, const Instruction *CtxI)
Returns true, if no instruction between Assume and CtxI may free (including through synchronization).
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
void stable_sort(R &&Range)
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
UnaryFunction for_each(R &&Range, UnaryFunction F)
Provide wrappers to std::for_each which take ranges instead of having to pass begin/end explicitly.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
LLVM_ABI bool RecursivelyDeleteTriviallyDeadInstructions(Value *V, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, std::function< void(Value *)> AboutToDeleteCallback=std::function< void(Value *)>())
If the specified value is a trivially dead instruction, delete it.
RelativeUniformCounterPtr Values
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI Value * simplifyUnOp(unsigned Opcode, Value *Op, const SimplifyQuery &Q)
Given operand for a UnaryOperator, fold the result or return null.
scope_exit(Callable) -> scope_exit< Callable >
@ Load
The value being inserted comes from a load (InsertElement only).
auto map_to_vector(ContainerTy &&C, FuncTy &&F)
Map a range to a SmallVector with element types deduced from the mapping.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI unsigned getArithmeticReductionInstruction(Intrinsic::ID RdxID)
Returns the arithmetic instruction opcode used when expanding a reduction.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
LLVM_ABI bool mustSuppressSpeculation(const LoadInst &LI)
Return true if speculation of the given load must be suppressed to avoid ordering or interfering with...
LLVM_ABI bool widenShuffleMaskElts(int Scale, ArrayRef< int > Mask, SmallVectorImpl< int > &ScaledMask)
Try to transform a shuffle mask by replacing elements with the scaled index for an equivalent mask of...
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Instruction * propagateMetadata(Instruction *I, ArrayRef< Value * > VL)
Specifically, let Kinds = [MD_tbaa, MD_alias_scope, MD_noalias, MD_fpmath, MD_nontemporal,...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
unsigned M1(unsigned Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isInstructionTriviallyDead(Instruction *I, const TargetLibraryInfo *TLI=nullptr)
Return true if the result produced by the instruction is not used, and the instruction will return.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
bool isModSet(const ModRefInfo MRI)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI bool programUndefinedIfPoison(const Instruction *Inst)
LLVM_ABI bool isSafeToLoadUnconditionally(Value *V, Align Alignment, const APInt &Size, const DataLayout &DL, Instruction *ScanFrom, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr)
Return true if we know that executing a load from this value cannot trap.
LLVM_ABI unsigned getDeinterleaveIntrinsicFactor(Intrinsic::ID ID)
Returns the corresponding factor of llvm.vector.deinterleaveN intrinsics.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ABI void propagateIRFlags(Value *I, ArrayRef< Value * > VL, Value *OpValue=nullptr, bool IncludeWrapFlags=true)
Get the intersection (logical and) of all of the potential IR flags of each scalar operation (VL) tha...
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr int PoisonMaskElem
IRBuilder(LLVMContext &, FolderTy, InserterTy, MDNode *, ArrayRef< OperandBundleDef >) -> IRBuilder< FolderTy, InserterTy >
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
LLVM_ABI void narrowShuffleMaskElts(int Scale, ArrayRef< int > Mask, SmallVectorImpl< int > &ScaledMask)
Replace each shuffle mask index with the scaled sequential indices for an equivalent mask of narrowed...
LLVM_ABI Intrinsic::ID getReductionForBinop(Instruction::BinaryOps Opc)
Returns the reduction intrinsic id corresponding to the binary operation.
@ And
Bitwise or logical AND of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI unsigned ComputeNumSignBits(const Value *Op, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Return the number of times the sign bit of the register is replicated into the other bits.
constexpr unsigned BitWidth
LLVM_ABI bool isGuaranteedToTransferExecutionToSuccessor(const Instruction *I)
Return true if this function can prove that the instruction I will always transfer execution to one o...
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI bool isGuaranteedNotToBePoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Returns true if V cannot be poison, but may be undef.
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicID(Intrinsic::ID IID)
Returns the llvm.vector.reduce min/max intrinsic that corresponds to the intrinsic op.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
LLVM_ABI AAMDNodes adjustForAccess(unsigned AccessSize)
Create a new AAMDNode for accessing AccessSize bytes of this AAMDNode.
This struct is a compact representation of a valid (non-zero power of two) alignment.
unsigned countMaxActiveBits() const
Returns the maximum number of bits needed to represent all possible unsigned values with these known ...
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
SimplifyQuery getWithInstruction(const Instruction *I) const