99#define DEBUG_TYPE "loop-idiom"
101STATISTIC(NumMemSet,
"Number of memset's formed from loop stores");
102STATISTIC(NumMemCpy,
"Number of memcpy's formed from loop load+stores");
103STATISTIC(NumMemMove,
"Number of memmove's formed from loop load+stores");
104STATISTIC(NumStrLen,
"Number of strlen's and wcslen's formed from loop loads");
106 NumShiftUntilBitTest,
107 "Number of uncountable loops recognized as 'shift until bitttest' idiom");
109 "Number of uncountable loops recognized as 'shift until zero' idiom");
115 cl::desc(
"Options to disable Loop Idiom Recognize Pass."),
122 cl::desc(
"Proceed with loop idiom recognize pass, but do "
123 "not convert loop(s) to memset."),
130 cl::desc(
"Proceed with loop idiom recognize pass, but do "
131 "not convert loop(s) to memcpy."),
138 cl::desc(
"Proceed with loop idiom recognize pass, but do "
139 "not convert loop(s) to strlen."),
146 cl::desc(
"Proceed with loop idiom recognize pass, "
147 "enable conversion of loop(s) to wcslen."),
154 cl::desc(
"Proceed with loop idiom recognize pass, "
155 "but do not optimize CRC loops."),
160 "use-lir-code-size-heurs",
161 cl::desc(
"Use loop idiom recognition code size heuristics when compiling "
166 "loop-idiom-force-memset-pattern-intrinsic",
167 cl::desc(
"Use memset.pattern intrinsic whenever possible"),
cl::init(
false),
171 "loop-idiom-force-crc-clmul",
172 cl::desc(
"Use the clmul-based CRC loop optimization whenever possible"),
181class LoopIdiomRecognize {
182 Loop *CurLoop =
nullptr;
191 bool ApplyCodeSizeHeuristics;
192 std::unique_ptr<MemorySSAUpdater> MSSAU;
201 :
AA(
AA), DT(DT), LI(LI), SE(SE), TLI(TLI),
TTI(
TTI),
DL(
DL), ORE(ORE) {
203 MSSAU = std::make_unique<MemorySSAUpdater>(MSSA);
206 bool runOnLoop(Loop *L);
209 using StoreList = SmallVector<StoreInst *, 8>;
210 using StoreListMap = MapVector<Value *, StoreList>;
212 StoreListMap StoreRefsForMemset;
213 StoreListMap StoreRefsForMemsetPattern;
214 StoreList StoreRefsForMemcpy;
216 bool HasMemsetPattern;
220 enum LegalStoreKind {
225 UnorderedAtomicMemcpy,
233 bool runOnCountableLoop();
234 bool runOnLoopBlock(BasicBlock *BB,
const SCEV *BECount,
235 SmallVectorImpl<BasicBlock *> &ExitBlocks);
237 void collectStores(BasicBlock *BB);
238 LegalStoreKind isLegalStore(StoreInst *SI);
239 enum class ForMemset {
No,
Yes };
240 bool processLoopStores(SmallVectorImpl<StoreInst *> &SL,
const SCEV *BECount,
243 template <
typename MemInst>
244 bool processLoopMemIntrinsic(
246 bool (LoopIdiomRecognize::*Processor)(MemInst *,
const SCEV *),
247 const SCEV *BECount);
248 bool processLoopMemCpy(MemCpyInst *MCI,
const SCEV *BECount);
249 bool processLoopMemSet(MemSetInst *MSI,
const SCEV *BECount);
251 bool processLoopStridedStore(
Value *DestPtr,
const SCEV *StoreSizeSCEV,
252 MaybeAlign StoreAlignment,
Value *StoredVal,
253 Instruction *TheStore,
254 SmallPtrSetImpl<Instruction *> &Stores,
255 const SCEVAddRecExpr *Ev,
const SCEV *BECount,
256 bool IsNegStride,
bool IsLoopMemset =
false);
257 bool processLoopStoreOfLoopLoad(StoreInst *SI,
const SCEV *BECount);
258 bool processLoopStoreOfLoopLoad(
Value *DestPtr,
Value *SourcePtr,
259 const SCEV *StoreSize, MaybeAlign StoreAlign,
260 MaybeAlign LoadAlign, Instruction *TheStore,
261 Instruction *TheLoad,
262 const SCEVAddRecExpr *StoreEv,
263 const SCEVAddRecExpr *LoadEv,
264 const SCEV *BECount);
265 bool avoidLIRForMultiBlockLoop(
bool IsMemset =
false,
266 bool IsLoopMemset =
false);
267 bool optimizeCRCLoop(
const PolynomialInfo &Info);
268 void optimizeCRCLoopUsingClmul(
const PolynomialInfo &Info);
269 void optimizeCRCLoopUsingTableLookup(
const PolynomialInfo &Info);
275 bool runOnNoncountableLoop();
277 bool recognizePopcount();
278 void transformLoopToPopcount(BasicBlock *PreCondBB, Instruction *CntInst,
279 PHINode *CntPhi,
Value *Var);
281 bool ZeroCheck,
size_t CanonicalSize);
283 Instruction *DefX, PHINode *CntPhi,
284 Instruction *CntInst);
285 bool recognizeAndInsertFFS();
286 bool recognizeShiftUntilLessThan();
287 void transformLoopToCountable(
Intrinsic::ID IntrinID, BasicBlock *PreCondBB,
288 Instruction *CntInst, PHINode *CntPhi,
289 Value *Var, Instruction *DefX,
291 bool IsCntPhiUsedOutsideLoop,
292 bool InsertSub =
false);
294 bool recognizeShiftUntilBitTest();
295 bool recognizeShiftUntilZero();
296 bool recognizeAndInsertStrLen();
308 const auto *
DL = &L.getHeader()->getDataLayout();
315 LoopIdiomRecognize LIR(&AR.
AA, &AR.
DT, &AR.
LI, &AR.
SE, &AR.
TLI, &AR.
TTI,
317 if (!LIR.runOnLoop(&L))
328 I->eraseFromParent();
337bool LoopIdiomRecognize::runOnLoop(
Loop *L) {
341 if (!
L->getLoopPreheader())
346 if (Name ==
"memset" || Name ==
"memcpy" || Name ==
"strlen" ||
351 ApplyCodeSizeHeuristics =
354 HasMemset = TLI->
has(LibFunc_memset);
360 HasMemsetPattern = TLI->
has(LibFunc_memset_pattern16);
361 HasMemcpy = TLI->
has(LibFunc_memcpy);
366 return runOnCountableLoop();
368 return runOnNoncountableLoop();
371bool LoopIdiomRecognize::runOnCountableLoop() {
374 "runOnCountableLoop() called on a loop without a predictable"
375 "backedge-taken count");
397 bool MadeChange =
false;
405 MadeChange |= runOnLoopBlock(BB, BECount, ExitBlocks);
411 MadeChange |= optimizeCRCLoop(*Res);
446 if (
DL->isBigEndian())
458 Type *CTy =
C->getType();
465LoopIdiomRecognize::LegalStoreKind
468 if (
SI->isVolatile())
469 return LegalStoreKind::None;
471 if (!
SI->isUnordered())
472 return LegalStoreKind::None;
475 if (
SI->getMetadata(LLVMContext::MD_nontemporal))
476 return LegalStoreKind::None;
478 Value *StoredVal =
SI->getValueOperand();
479 Value *StorePtr =
SI->getPointerOperand();
481 if (
DL->hasUnstableRepresentation(StoredVal->
getType()))
482 return LegalStoreKind::None;
491 bool MustPreserveExternalState =
DL->hasExternalState(StoredVal->
getType()) &&
500 return LegalStoreKind::None;
509 return LegalStoreKind::None;
520 bool UnorderedAtomic =
SI->isUnordered() && !
SI->isSimple();
524 if (!MustPreserveExternalState && !UnorderedAtomic && HasMemset &&
530 return LegalStoreKind::Memset;
532 if (!MustPreserveExternalState && !UnorderedAtomic &&
539 return LegalStoreKind::MemsetPattern;
546 unsigned StoreSize =
DL->getTypeStoreSize(
SI->getValueOperand()->getType());
548 if (StoreSize != StrideAP && StoreSize != -StrideAP)
549 return LegalStoreKind::None;
556 return LegalStoreKind::None;
559 return LegalStoreKind::None;
569 return LegalStoreKind::None;
572 UnorderedAtomic = UnorderedAtomic || LI->
isAtomic();
573 return UnorderedAtomic ? LegalStoreKind::UnorderedAtomicMemcpy
574 : LegalStoreKind::Memcpy;
577 return LegalStoreKind::None;
580void LoopIdiomRecognize::collectStores(
BasicBlock *BB) {
581 StoreRefsForMemset.clear();
582 StoreRefsForMemsetPattern.clear();
583 StoreRefsForMemcpy.clear();
590 switch (isLegalStore(
SI)) {
591 case LegalStoreKind::None:
594 case LegalStoreKind::Memset: {
597 StoreRefsForMemset[Ptr].push_back(
SI);
599 case LegalStoreKind::MemsetPattern: {
602 StoreRefsForMemsetPattern[Ptr].push_back(
SI);
604 case LegalStoreKind::Memcpy:
605 case LegalStoreKind::UnorderedAtomicMemcpy:
606 StoreRefsForMemcpy.push_back(
SI);
609 assert(
false &&
"unhandled return value");
618bool LoopIdiomRecognize::runOnLoopBlock(
628 bool MadeChange =
false;
635 for (
auto &SL : StoreRefsForMemset)
636 MadeChange |= processLoopStores(SL.second, BECount, ForMemset::Yes);
638 for (
auto &SL : StoreRefsForMemsetPattern)
639 MadeChange |= processLoopStores(SL.second, BECount, ForMemset::No);
642 for (
auto &
SI : StoreRefsForMemcpy)
643 MadeChange |= processLoopStoreOfLoopLoad(
SI, BECount);
645 MadeChange |= processLoopMemIntrinsic<MemCpyInst>(
646 BB, &LoopIdiomRecognize::processLoopMemCpy, BECount);
647 MadeChange |= processLoopMemIntrinsic<MemSetInst>(
648 BB, &LoopIdiomRecognize::processLoopMemSet, BECount);
655 const SCEV *BECount, ForMemset For) {
663 for (
unsigned i = 0, e = SL.
size(); i < e; ++i) {
664 assert(SL[i]->
isSimple() &&
"Expected only non-volatile stores.");
666 Value *FirstStoredVal = SL[i]->getValueOperand();
667 Value *FirstStorePtr = SL[i]->getPointerOperand();
671 unsigned FirstStoreSize =
DL->getTypeStoreSize(SL[i]->getValueOperand()->
getType());
674 if (FirstStride == FirstStoreSize || -FirstStride == FirstStoreSize) {
679 Value *FirstSplatValue =
nullptr;
680 Constant *FirstPatternValue =
nullptr;
682 if (For == ForMemset::Yes)
687 assert((FirstSplatValue || FirstPatternValue) &&
688 "Expected either splat value or pattern value.");
696 for (j = i + 1;
j <
e; ++
j)
698 for (j = i;
j > 0; --
j)
701 for (
auto &k : IndexQueue) {
702 assert(SL[k]->
isSimple() &&
"Expected only non-volatile stores.");
703 Value *SecondStorePtr = SL[
k]->getPointerOperand();
708 if (FirstStride != SecondStride)
711 Value *SecondStoredVal = SL[
k]->getValueOperand();
712 Value *SecondSplatValue =
nullptr;
713 Constant *SecondPatternValue =
nullptr;
715 if (For == ForMemset::Yes)
720 assert((SecondSplatValue || SecondPatternValue) &&
721 "Expected either splat value or pattern value.");
724 if (For == ForMemset::Yes) {
726 FirstSplatValue = SecondSplatValue;
727 if (FirstSplatValue != SecondSplatValue)
731 FirstPatternValue = SecondPatternValue;
732 if (FirstPatternValue != SecondPatternValue)
737 ConsecutiveChain[SL[i]] = SL[
k];
757 unsigned StoreSize = 0;
760 while (Tails.
count(
I) || Heads.count(
I)) {
761 if (TransformedStores.
count(
I))
765 StoreSize +=
DL->getTypeStoreSize(
I->getValueOperand()->getType());
767 I = ConsecutiveChain[
I];
777 if (StoreSize != Stride && StoreSize != -Stride)
780 bool IsNegStride = StoreSize == -Stride;
784 if (processLoopStridedStore(StorePtr, StoreSizeSCEV,
786 HeadStore, AdjacentStores, StoreEv, BECount,
798template <
typename MemInst>
799bool LoopIdiomRecognize::processLoopMemIntrinsic(
801 bool (LoopIdiomRecognize::*Processor)(MemInst *,
const SCEV *),
802 const SCEV *BECount) {
803 bool MadeChange =
false;
809 if (!(this->*Processor)(
MI, BECount))
823bool LoopIdiomRecognize::processLoopMemCpy(
MemCpyInst *MCI,
824 const SCEV *BECount) {
835 if (!Dest || !Source)
843 const APInt *StoreStrideValue, *LoadStrideValue;
854 if ((SizeInBytes >> 32) != 0)
862 if (SizeInBytes != *StoreStrideValue && SizeInBytes != -*StoreStrideValue) {
865 <<
ore::NV(
"Inst",
"memcpy") <<
" in "
867 <<
" function will not be hoisted: "
868 <<
ore::NV(
"Reason",
"memcpy size is not equal to stride");
873 int64_t StoreStrideInt = StoreStrideValue->
getSExtValue();
874 int64_t LoadStrideInt = LoadStrideValue->
getSExtValue();
876 if (StoreStrideInt != LoadStrideInt)
879 return processLoopStoreOfLoopLoad(
886bool LoopIdiomRecognize::processLoopMemSet(
MemSetInst *MSI,
887 const SCEV *BECount) {
902 const SCEV *PointerStrideSCEV;
911 bool IsNegStride =
false;
914 if (IsConstantSize) {
924 if (SizeInBytes != *Stride && SizeInBytes != -*Stride)
927 IsNegStride = SizeInBytes == -*Stride;
935 if (
Pointer->getType()->getPointerAddressSpace() != 0) {
951 LLVM_DEBUG(
dbgs() <<
" MemsetSizeSCEV: " << *MemsetSizeSCEV <<
"\n"
952 <<
" PositiveStrideSCEV: " << *PositiveStrideSCEV
955 if (PositiveStrideSCEV != MemsetSizeSCEV) {
958 const SCEV *FoldedPositiveStride =
960 const SCEV *FoldedMemsetSize =
964 <<
" FoldedMemsetSize: " << *FoldedMemsetSize <<
"\n"
965 <<
" FoldedPositiveStride: " << *FoldedPositiveStride
968 if (FoldedPositiveStride != FoldedMemsetSize) {
993 assert(SplatByte &&
"expected a bytewise splat value to match against");
995 if (!
SI || !
SI->isSimple() || !L->isLoopInvariant(
SI->getValueOperand()))
1007 const SCEV *BECount,
1010 Value *SplatByte =
nullptr,
1019 const APInt *BECst, *ConstSize;
1023 std::optional<uint64_t> SizeInt = ConstSize->
tryZExtValue();
1025 if (BEInt && SizeInt)
1037 bool TrySameByteValue = !AccessSize.
isPrecise() && SplatByte &&
DL;
1054 Type *IntPtr,
const SCEV *StoreSizeSCEV,
1057 if (!StoreSizeSCEV->
isOne()) {
1072 const SCEV *StoreSizeSCEV,
Loop *CurLoop,
1074 const SCEV *TripCountSCEV =
1083bool LoopIdiomRecognize::processLoopStridedStore(
1087 const SCEV *BECount,
bool IsNegStride,
bool IsLoopMemset) {
1099 Type *DestInt8PtrTy = Builder.getPtrTy(DestAS);
1110 if (!Expander.isSafeToExpand(Start))
1119 Expander.expandCodeFor(Start, DestInt8PtrTy, Preheader->
getTerminator());
1132 StoreSizeSCEV, *
AA, Stores, SplatValue,
DL))
1135 if (avoidLIRForMultiBlockLoop(
true, IsLoopMemset))
1146 std::optional<int64_t> BytesWritten;
1149 const SCEV *TripCountS =
1151 if (!Expander.isSafeToExpand(TripCountS))
1154 if (!ConstStoreSize)
1156 Value *TripCount = Expander.expandCodeFor(TripCountS, IntIdxTy,
1158 uint64_t PatternRepsPerTrip =
1159 (ConstStoreSize->
getValue()->getZExtValue() * 8) /
1160 DL->getTypeSizeInBits(PatternValue->
getType());
1165 PatternRepsPerTrip == 1
1167 : Builder.CreateMul(TripCount,
1169 PatternRepsPerTrip));
1175 const SCEV *NumBytesS =
1176 getNumBytes(BECount, IntIdxTy, StoreSizeSCEV, CurLoop,
DL, SE);
1180 if (!Expander.isSafeToExpand(NumBytesS))
1183 Expander.expandCodeFor(NumBytesS, IntIdxTy, Preheader->
getTerminator());
1185 BytesWritten = CI->getZExtValue();
1187 assert(MemsetArg &&
"MemsetArg should have been set");
1191 AATags = AATags.
merge(
Store->getAAMetadata());
1193 AATags = AATags.
extendTo(BytesWritten.value());
1199 NewCall = Builder.CreateMemSet(BasePtr, SplatValue, MemsetArg,
1206 NewCall = Builder.CreateIntrinsicWithoutFolding(
1207 Intrinsic::experimental_memset_pattern,
1208 {DestInt8PtrTy, PatternValue->
getType(), IntIdxTy},
1209 {
BasePtr, PatternValue, MemsetArg,
1222 MemoryAccess *NewMemAcc = MSSAU->createMemoryAccessInBB(
1228 <<
" from store to: " << *Ev <<
" at: " << *TheStore
1234 R <<
"Transformed loop-strided store in "
1236 <<
" function into a call to "
1239 if (!Stores.empty())
1241 for (
auto *
I : Stores) {
1242 R <<
ore::NV(
"FromBlock",
I->getParent()->getName())
1250 for (
auto *
I : Stores) {
1252 MSSAU->removeMemoryAccess(
I,
true);
1256 MSSAU->getMemorySSA()->verifyMemorySSA();
1258 ExpCleaner.markResultUsed();
1265bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
StoreInst *
SI,
1266 const SCEV *BECount) {
1267 assert(
SI->isUnordered() &&
"Expected only non-volatile non-ordered stores.");
1269 Value *StorePtr =
SI->getPointerOperand();
1271 unsigned StoreSize =
DL->getTypeStoreSize(
SI->getValueOperand()->getType());
1284 return processLoopStoreOfLoopLoad(StorePtr, LoadPtr, StoreSizeSCEV,
1286 StoreEv, LoadEv, BECount);
1290class MemmoveVerifier {
1292 explicit MemmoveVerifier(
const Value &LoadBasePtr,
const Value &StoreBasePtr,
1293 const DataLayout &
DL)
1295 LoadBasePtr.stripPointerCasts(), LoadOff,
DL)),
1297 StoreBasePtr.stripPointerCasts(), StoreOff,
DL)),
1298 IsSameObject(BP1 == BP2) {}
1300 bool loadAndStoreMayFormMemmove(
unsigned StoreSize,
bool IsNegStride,
1301 const Instruction &TheLoad,
1302 bool IsMemCpy)
const {
1306 if ((!IsNegStride && LoadOff <= StoreOff) ||
1307 (IsNegStride && LoadOff >= StoreOff))
1313 DL.getTypeSizeInBits(TheLoad.
getType()).getFixedValue() / 8;
1314 if (BP1 != BP2 || LoadSize != int64_t(StoreSize))
1316 if ((!IsNegStride && LoadOff < StoreOff + int64_t(StoreSize)) ||
1317 (IsNegStride && LoadOff + LoadSize > StoreOff))
1324 const DataLayout &
DL;
1325 int64_t LoadOff = 0;
1326 int64_t StoreOff = 0;
1331 const bool IsSameObject;
1335bool LoopIdiomRecognize::processLoopStoreOfLoopLoad(
1365 assert(ConstStoreSize &&
"store size is expected to be a constant");
1368 bool IsNegStride = StoreSize == -Stride;
1381 Value *StoreBasePtr = Expander.expandCodeFor(
1382 StrStart, Builder.getPtrTy(StrAS), Preheader->
getTerminator());
1394 IgnoredInsts.
insert(TheStore);
1397 const StringRef InstRemark = IsMemCpy ?
"memcpy" :
"load and store";
1399 bool LoopAccessStore =
1401 StoreSizeSCEV, *
AA, IgnoredInsts);
1402 if (LoopAccessStore) {
1408 IgnoredInsts.
insert(TheLoad);
1410 BECount, StoreSizeSCEV, *
AA, IgnoredInsts)) {
1414 <<
ore::NV(
"Inst", InstRemark) <<
" in "
1416 <<
" function will not be hoisted: "
1417 <<
ore::NV(
"Reason",
"The loop may access store location");
1421 IgnoredInsts.
erase(TheLoad);
1434 Value *LoadBasePtr = Expander.expandCodeFor(LdStart, Builder.getPtrTy(LdAS),
1439 MemmoveVerifier
Verifier(*LoadBasePtr, *StoreBasePtr, *
DL);
1440 if (IsMemCpy && !
Verifier.IsSameObject)
1441 IgnoredInsts.
erase(TheStore);
1443 StoreSizeSCEV, *
AA, IgnoredInsts)) {
1446 <<
ore::NV(
"Inst", InstRemark) <<
" in "
1448 <<
" function will not be hoisted: "
1449 <<
ore::NV(
"Reason",
"The loop may access load location");
1455 bool UseMemMove = IsMemCpy ?
Verifier.IsSameObject : LoopAccessStore;
1464 assert((StoreAlign && LoadAlign) &&
1465 "Expect unordered load/store to have align.");
1466 if (*StoreAlign < StoreSize || *LoadAlign < StoreSize)
1473 if (StoreSize >
TTI->getAtomicMemIntrinsicMaxElementSize())
1478 if (!
Verifier.loadAndStoreMayFormMemmove(StoreSize, IsNegStride, *TheLoad,
1482 if (avoidLIRForMultiBlockLoop())
1487 const SCEV *NumBytesS =
1488 getNumBytes(BECount, IntIdxTy, StoreSizeSCEV, CurLoop,
DL, SE);
1491 Expander.expandCodeFor(NumBytesS, IntIdxTy, Preheader->
getTerminator());
1495 AATags = AATags.
merge(StoreAATags);
1497 AATags = AATags.
extendTo(CI->getZExtValue());
1507 NewCall = Builder.CreateMemMove(StoreBasePtr, StoreAlign, LoadBasePtr,
1508 LoadAlign, NumBytes,
1512 Builder.CreateMemCpy(StoreBasePtr, StoreAlign, LoadBasePtr, LoadAlign,
1513 NumBytes,
false, AATags);
1518 NewCall = Builder.CreateElementUnorderedAtomicMemCpy(
1519 StoreBasePtr, *StoreAlign, LoadBasePtr, *LoadAlign, NumBytes, StoreSize,
1525 MemoryAccess *NewMemAcc = MSSAU->createMemoryAccessInBB(
1531 <<
" from load ptr=" << *LoadEv <<
" at: " << *TheLoad
1533 <<
" from store ptr=" << *StoreEv <<
" at: " << *TheStore
1539 <<
"Formed a call to "
1541 <<
"() intrinsic from " <<
ore::NV(
"Inst", InstRemark)
1552 MSSAU->removeMemoryAccess(TheStore,
true);
1555 MSSAU->getMemorySSA()->verifyMemorySSA();
1560 ExpCleaner.markResultUsed();
1567bool LoopIdiomRecognize::avoidLIRForMultiBlockLoop(
bool IsMemset,
1568 bool IsLoopMemset) {
1569 if (ApplyCodeSizeHeuristics && CurLoop->
getNumBlocks() > 1) {
1570 if (CurLoop->
isOutermost() && (!IsMemset || !IsLoopMemset)) {
1572 <<
" : LIR " << (IsMemset ?
"Memset" :
"Memcpy")
1573 <<
" avoided: multi-block top-level loop\n");
1581bool LoopIdiomRecognize::optimizeCRCLoop(
const PolynomialInfo &Info) {
1595 optimizeCRCLoopUsingClmul(Info);
1603 if (!ApplyCodeSizeHeuristics &&
Info.TripCount % 8 == 0) {
1604 optimizeCRCLoopUsingTableLookup(Info);
1612 unsigned ClmulMuBW =
Info.IsBigEndian ? 2 *
Info.TripCount :
Info.TripCount;
1613 unsigned ClmulGPBW =
1614 Info.LHS->getType()->getIntegerBitWidth() +
Info.TripCount;
1617 if (
TTI->haveFastClmul(WidestClmulTy)) {
1618 optimizeCRCLoopUsingClmul(Info);
1628void LoopIdiomRecognize::optimizeCRCLoopUsingClmul(
const PolynomialInfo &Info) {
1635 unsigned TC =
Info.TripCount;
1646 ConstantInt::get(Ctx, Mu.zextOrTrunc(ClmulMuTy->
getBitWidth()));
1647 Value *GenPolyConst =
1648 ConstantInt::get(Ctx, FullGenPoly.zext(ClmulGPTy->
getBitWidth()));
1655 bool SetupShiftNeeded =
Info.IsBigEndian && TC != CRCBW;
1671 Value *ClmulMuInput =
1672 Builder.CreateZExtOrTrunc(
Info.LHS, SetupTy,
"crc.cast");
1678 Data = Builder.CreateZExtOrTrunc(
Data, SetupTy,
"data.cast");
1680 ClmulMuInput = Builder.CreateXor(ClmulMuInput,
Data,
"xor.crc.data");
1684 if (SetupShiftNeeded) {
1687 ? Builder.CreateShl(ClmulMuInput, TC - CRCBW,
"crc.align.tc")
1688 : Builder.CreateLShr(ClmulMuInput, CRCBW - TC,
"crc.align.tc");
1693 if (SetupTy->getBitWidth() > TC) {
1696 ClmulMuInput = Builder.CreateAnd(ClmulMuInput, Mask,
"crc.tcbits");
1702 Builder.CreateZExtOrTrunc(ClmulMuInput, ClmulMuTy,
"tcbits.cast");
1703 Value *ClmulMu = Builder.CreateBinaryIntrinsic(
1704 Intrinsic::clmul, ClmulMuInput, MuConst, {},
"clmul.mu");
1707 Value *ClmulGPInput =
1708 Info.IsBigEndian ? Builder.CreateLShr(ClmulMu, TC,
"quot.lshr") : ClmulMu;
1713 Builder.CreateZExtOrTrunc(ClmulGPInput, ClmulGPTy,
"quot.cast");
1714 Value *ClmulGP = Builder.CreateBinaryIntrinsic(Intrinsic::clmul, ClmulGPInput,
1721 Value *CRCNext = Builder.CreateZExt(
Info.LHS, ClmulGPTy,
"crc.recast");
1722 if (
Info.IsBigEndian)
1723 CRCNext = Builder.CreateShl(CRCNext, TC,
"crc.shl");
1726 CRCNext = Builder.CreateXor(CRCNext, ClmulGP,
"xor.crc.mult");
1727 if (!
Info.IsBigEndian)
1728 CRCNext = Builder.CreateLShr(CRCNext, TC,
"crc.lshr");
1731 CRCNext = Builder.CreateTrunc(CRCNext, CRCTy,
"crc.next");
1734 Info.ComputedValue->replaceUsesOutsideBlock(CRCNext, CurLoop->
getLoopLatch());
1748 Ctx, BrInst->getSuccessor(0) == CurLoop->
getExitBlock()));
1753void LoopIdiomRecognize::optimizeCRCLoopUsingTableLookup(
1755 assert(
Info.TripCount % 8 == 0 &&
"A byte-multiple trip count is required");
1761 std::array<Constant *, 256> CRCConstants;
1763 CRCConstants.begin(),
1764 [CRCTy](
const APInt &
E) { return ConstantInt::get(CRCTy, E); });
1786 unsigned NewBTC = (
Info.TripCount / 8) - 1;
1793 Value *ExitLimit = ConstantInt::get(
IV->getType(), NewBTC);
1795 Value *NewExitCond =
1796 Builder.CreateICmp(ExitPred,
IV, ExitLimit,
"exit.cond");
1815 Type *OpTy =
Op->getType();
1819 return LoByte(Builder,
1820 CRCBW > 8 ? Builder.CreateLShr(
1821 Op, ConstantInt::get(OpTy, CRCBW - 8), Name)
1831 PHINode *CRCPhi = Builder.CreatePHI(CRCTy, 2,
"crc");
1835 Value *CRC = CRCPhi;
1839 Value *Indexer = CRC;
1847 Value *IVBits = Builder.CreateZExtOrTrunc(
1848 Builder.CreateShl(
IV, 3,
"iv.bits"), DataTy,
"iv.indexer");
1849 Value *DataIndexer =
1850 Info.IsBigEndian ? Builder.CreateShl(
Data, IVBits,
"data.indexer")
1851 : Builder.CreateLShr(
Data, IVBits,
"data.indexer");
1852 Indexer = Builder.CreateXor(
1854 Builder.CreateZExtOrTrunc(Indexer, DataTy,
"crc.indexer.cast"),
1855 "crc.data.indexer");
1858 Indexer =
Info.IsBigEndian ? HiIdx(Builder, Indexer,
"indexer.hi")
1859 : LoByte(Builder, Indexer,
"indexer.lo");
1862 Indexer = Builder.CreateZExt(
1867 Value *CRCTableGEP =
1868 Builder.CreateInBoundsGEP(CRCTy, GV, Indexer,
"tbl.ptradd");
1869 Value *CRCTableLd = Builder.CreateLoad(CRCTy, CRCTableGEP,
"tbl.ld");
1873 Value *CRCNext = CRCTableLd;
1876 ? Builder.CreateShl(CRC, 8,
"crc.be.shift")
1877 : Builder.CreateLShr(CRC, 8,
"crc.le.shift");
1878 CRCNext = Builder.CreateXor(CRCShift, CRCTableLd,
"crc.next");
1883 Info.ComputedValue->replaceUsesOutsideBlock(CRCNext,
1895bool LoopIdiomRecognize::runOnNoncountableLoop() {
1898 <<
"] Noncountable Loop %"
1901 return recognizePopcount() || recognizeAndInsertFFS() ||
1902 recognizeShiftUntilBitTest() || recognizeShiftUntilZero() ||
1903 recognizeShiftUntilLessThan() || recognizeAndInsertStrLen();
1913 bool JmpOnZero =
false) {
1919 if (!CmpZero || !CmpZero->isZero())
1930 return Cond->getOperand(0);
1937class StrlenVerifier {
1939 explicit StrlenVerifier(
const Loop *CurLoop, ScalarEvolution *SE,
1940 const TargetLibraryInfo *TLI)
1941 : CurLoop(CurLoop), SE(SE), TLI(TLI) {}
1943 bool isValidStrlenIdiom() {
1962 if (!LoopBody || LoopBody->
size() >= 15)
1983 const SCEV *LoadEv = SE->
getSCEV(IncPtr);
1996 if (OpWidth != StepSize * 8)
1998 if (OpWidth != 8 && OpWidth != 16 && OpWidth != 32)
2001 if (OpWidth != WcharSize * 8)
2005 for (Instruction &
I : *LoopBody)
2006 if (
I.mayHaveSideEffects())
2013 for (PHINode &PN : LoopExitBB->
phis()) {
2017 const SCEV *Ev = SE->
getSCEV(&PN);
2027 if (!AddRecEv || !AddRecEv->
isAffine())
2041 const Loop *CurLoop;
2042 ScalarEvolution *SE;
2043 const TargetLibraryInfo *TLI;
2046 ConstantInt *StepSizeCI;
2047 const SCEV *LoadBaseEv;
2112bool LoopIdiomRecognize::recognizeAndInsertStrLen() {
2116 StrlenVerifier
Verifier(CurLoop, SE, TLI);
2118 if (!
Verifier.isValidStrlenIdiom())
2125 assert(Preheader && LoopBody && LoopExitBB &&
2126 "Should be verified to be valid by StrlenVerifier");
2141 Builder.SetCurrentDebugLocation(CurLoop->
getStartLoc());
2143 Value *MaterialzedBase = Expander.expandCodeFor(
2145 Builder.GetInsertPoint());
2147 Value *StrLenFunc =
nullptr;
2149 StrLenFunc =
emitStrLen(MaterialzedBase, Builder, *
DL, TLI);
2151 StrLenFunc =
emitWcsLen(MaterialzedBase, Builder, *
DL, TLI);
2153 assert(StrLenFunc &&
"Failed to emit strlen function.");
2172 StrlenEv,
Base->getType())));
2174 Value *MaterializedPHI = Expander.expandCodeFor(NewEv, NewEv->
getType(),
2175 Builder.GetInsertPoint());
2190 "loop body must have a successor that is it self");
2192 ? Builder.getFalse()
2193 : Builder.getTrue();
2198 LLVM_DEBUG(
dbgs() <<
" Formed strlen idiom: " << *StrLenFunc <<
"\n");
2202 <<
"Transformed " << StrLenFunc->
getName() <<
" loop idiom";
2227 return Cond->getOperand(0);
2238 if (PhiX && PhiX->getParent() == LoopEntry &&
2239 (PhiX->getOperand(0) == DefX || PhiX->
getOperand(1) == DefX))
2306 if (DefX->
getOpcode() != Instruction::LShr)
2309 IntrinID = Intrinsic::ctlz;
2311 if (!Shft || !Shft->
isOne())
2325 if (Inst.
getOpcode() != Instruction::Add)
2377 Value *VarX1, *VarX0;
2380 DefX2 = CountInst =
nullptr;
2381 VarX1 = VarX0 =
nullptr;
2382 PhiX = CountPhi =
nullptr;
2395 if (!DefX2 || DefX2->
getOpcode() != Instruction::And)
2406 if (!SubOneOp || SubOneOp->
getOperand(0) != VarX1)
2412 (SubOneOp->
getOpcode() == Instruction::Add &&
2425 CountInst =
nullptr;
2428 if (Inst.
getOpcode() != Instruction::Add)
2432 if (!Inc || !Inc->
isOne())
2440 bool LiveOutLoop =
false;
2469 CntInst = CountInst;
2509 Value *VarX =
nullptr;
2523 if (!DefX || !DefX->
isShift())
2525 IntrinID = DefX->
getOpcode() == Instruction::Shl ? Intrinsic::cttz :
2528 if (!Shft || !Shft->
isOne())
2553 if (Inst.
getOpcode() != Instruction::Add)
2576bool LoopIdiomRecognize::isProfitableToInsertFFS(
Intrinsic::ID IntrinID,
2577 Value *InitX,
bool ZeroCheck,
2578 size_t CanonicalSize) {
2596bool LoopIdiomRecognize::insertFFSIfProfitable(
Intrinsic::ID IntrinID,
2600 bool IsCntPhiUsedOutsideLoop =
false;
2603 IsCntPhiUsedOutsideLoop =
true;
2606 bool IsCntInstUsedOutsideLoop =
false;
2609 IsCntInstUsedOutsideLoop =
true;
2614 if (IsCntInstUsedOutsideLoop && IsCntPhiUsedOutsideLoop)
2620 bool ZeroCheck =
false;
2629 if (!IsCntPhiUsedOutsideLoop) {
2648 size_t IdiomCanonicalSize = 6;
2649 if (!isProfitableToInsertFFS(IntrinID, InitX, ZeroCheck, IdiomCanonicalSize))
2652 transformLoopToCountable(IntrinID, PH, CntInst, CntPhi, InitX, DefX,
2654 IsCntPhiUsedOutsideLoop);
2661bool LoopIdiomRecognize::recognizeAndInsertFFS() {
2676 return insertFFSIfProfitable(IntrinID, InitX, DefX, CntPhi, CntInst);
2679bool LoopIdiomRecognize::recognizeShiftUntilLessThan() {
2690 APInt LoopThreshold;
2692 CntPhi, DefX, LoopThreshold))
2695 if (LoopThreshold == 2) {
2697 return insertFFSIfProfitable(IntrinID, InitX, DefX, CntPhi, CntInst);
2701 if (LoopThreshold != 4)
2719 APInt PreLoopThreshold;
2721 PreLoopThreshold != 2)
2724 bool ZeroCheck =
true;
2733 size_t IdiomCanonicalSize = 6;
2734 if (!isProfitableToInsertFFS(IntrinID, InitX, ZeroCheck, IdiomCanonicalSize))
2738 transformLoopToCountable(IntrinID, PH, CntInst, CntPhi, InitX, DefX,
2749bool LoopIdiomRecognize::recognizePopcount() {
2763 if (LoopBody->
size() >= 20) {
2791 transformLoopToPopcount(PreCondBB, CntInst, CntPhi, Val);
2845void LoopIdiomRecognize::transformLoopToCountable(
2848 bool ZeroCheck,
bool IsCntPhiUsedOutsideLoop,
bool InsertSub) {
2851 Builder.SetCurrentDebugLocation(
DL);
2860 if (IsCntPhiUsedOutsideLoop) {
2861 if (DefX->
getOpcode() == Instruction::AShr)
2862 InitXNext = Builder.CreateAShr(InitX, 1);
2863 else if (DefX->
getOpcode() == Instruction::LShr)
2864 InitXNext = Builder.CreateLShr(InitX, 1);
2865 else if (DefX->
getOpcode() == Instruction::Shl)
2866 InitXNext = Builder.CreateShl(InitX, 1);
2874 Count = Builder.CreateSub(
2877 Count = Builder.CreateSub(
Count, ConstantInt::get(CountTy, 1));
2879 if (IsCntPhiUsedOutsideLoop)
2880 Count = Builder.CreateAdd(
Count, ConstantInt::get(CountTy, 1));
2882 NewCount = Builder.CreateZExtOrTrunc(NewCount, CntInst->
getType());
2889 if (!InitConst || !InitConst->
isZero())
2890 NewCount = Builder.CreateAdd(NewCount, CntInitVal);
2894 NewCount = Builder.CreateSub(CntInitVal, NewCount);
2912 Builder.SetInsertPoint(LbCond);
2914 TcPhi, ConstantInt::get(CountTy, 1),
"tcdec",
false,
true));
2923 LbCond->
setOperand(1, ConstantInt::get(CountTy, 0));
2927 if (IsCntPhiUsedOutsideLoop)
2937void LoopIdiomRecognize::transformLoopToPopcount(
BasicBlock *PreCondBB,
2950 Value *PopCnt, *PopCntZext, *NewCount, *TripCnt;
2953 NewCount = PopCntZext =
2956 if (NewCount != PopCnt)
2965 if (!InitConst || !InitConst->
isZero()) {
2966 NewCount = Builder.CreateAdd(NewCount, CntInitVal);
2978 Value *Opnd0 = PopCntZext;
2979 Value *Opnd1 = ConstantInt::get(PopCntZext->
getType(), 0);
2984 Builder.CreateICmp(PreCond->
getPredicate(), Opnd0, Opnd1));
2985 PreCondBr->setCondition(NewPreCond);
3019 Builder.SetInsertPoint(LbCond);
3021 Builder.CreateSub(TcPhi, ConstantInt::get(Ty, 1),
3022 "tcdec",
false,
true));
3031 LbCond->
setOperand(1, ConstantInt::get(Ty, 0));
3052 template <
typename ITy>
bool match(ITy *V)
const {
3053 return L->isLoopInvariant(V) &&
SubPattern.match(V);
3058template <
typename Ty>
3089 " Performing shift-until-bittest idiom detection.\n");
3099 assert(LoopPreheaderBB &&
"There is always a loop preheader.");
3106 Value *CmpLHS, *CmpRHS;
3117 auto MatchVariableBitMask = [&]() {
3127 auto MatchDecomposableConstantBitMask = [&]() {
3129 CmpLHS, CmpRHS, Pred,
true,
3131 if (Res && Res->Mask.isPowerOf2()) {
3135 BitMask = ConstantInt::get(CurrX->
getType(), Res->Mask);
3136 BitPos = ConstantInt::get(CurrX->
getType(), Res->Mask.logBase2());
3142 if (!MatchVariableBitMask() && !MatchDecomposableConstantBitMask()) {
3149 if (!CurrXPN || CurrXPN->getParent() != LoopHeaderBB) {
3154 BaseX = CurrXPN->getIncomingValueForBlock(LoopPreheaderBB);
3159 "Expected BaseX to be available in the preheader!");
3170 "Should only get equality predicates here.");
3180 if (TrueBB != LoopHeaderBB) {
3239bool LoopIdiomRecognize::recognizeShiftUntilBitTest() {
3240 bool MadeChange =
false;
3242 Value *
X, *BitMask, *BitPos, *XCurr;
3247 " shift-until-bittest idiom detection failed.\n");
3257 assert(LoopPreheaderBB &&
"There is always a loop preheader.");
3260 assert(SuccessorBB &&
"There is only a single successor.");
3266 Type *Ty =
X->getType();
3280 " Intrinsic is too costly, not beneficial\n");
3283 if (
TTI->getArithmeticInstrCost(Instruction::Shl, Ty,
CostKind) >
3295 std::optional<BasicBlock::iterator> InsertPt = std::nullopt;
3297 InsertPt = BitPosI->getInsertionPointAfterDef();
3305 return U.getUser() != BitPosFrozen;
3307 BitPos = BitPosFrozen;
3313 BitPos->
getName() +
".lowbitmask");
3315 Builder.CreateOr(LowBitMask, BitMask, BitPos->
getName() +
".mask");
3316 Value *XMasked = Builder.CreateAnd(
X, Mask,
X->getName() +
".masked");
3317 Value *XMaskedNumLeadingZeros = Builder.CreateIntrinsic(
3318 IntrID, Ty, {XMasked, Builder.getTrue()},
3319 nullptr, XMasked->
getName() +
".numleadingzeros");
3320 Value *XMaskedNumActiveBits = Builder.CreateSub(
3322 XMasked->
getName() +
".numactivebits",
true,
3324 Value *XMaskedLeadingOnePos =
3326 XMasked->
getName() +
".leadingonepos",
false,
3329 Value *LoopBackedgeTakenCount = Builder.CreateSub(
3330 BitPos, XMaskedLeadingOnePos, CurLoop->
getName() +
".backedgetakencount",
3334 Value *LoopTripCount =
3335 Builder.CreateAdd(LoopBackedgeTakenCount, ConstantInt::get(Ty, 1),
3336 CurLoop->
getName() +
".tripcount",
true,
3343 Value *NewX = Builder.CreateShl(
X, LoopBackedgeTakenCount);
3346 I->copyIRFlags(XNext,
true);
3358 NewXNext = Builder.CreateShl(
X, LoopTripCount);
3363 NewXNext = Builder.CreateShl(NewX, ConstantInt::get(Ty, 1));
3368 I->copyIRFlags(XNext,
true);
3379 Builder.SetInsertPoint(LoopHeaderBB, LoopHeaderBB->
begin());
3380 auto *
IV = Builder.CreatePHI(Ty, 2, CurLoop->
getName() +
".iv");
3386 Builder.CreateAdd(
IV, ConstantInt::get(Ty, 1),
IV->getName() +
".next",
3387 true, Bitwidth != 2);
3390 auto *IVCheck = Builder.CreateICmpEQ(IVNext, LoopTripCount,
3391 CurLoop->
getName() +
".ivcheck");
3393 const bool HasBranchWeights =
3397 auto *BI = Builder.CreateCondBr(IVCheck, SuccessorBB, LoopHeaderBB);
3398 if (HasBranchWeights) {
3400 std::swap(BranchWeights[0], BranchWeights[1]);
3410 IV->addIncoming(ConstantInt::get(Ty, 0), LoopPreheaderBB);
3411 IV->addIncoming(IVNext, LoopHeaderBB);
3422 ++NumShiftUntilBitTest;
3458 const SCEV *&ExtraOffsetExpr,
3459 bool &InvertedCond) {
3461 " Performing shift-until-zero idiom detection.\n");
3474 assert(LoopPreheaderBB &&
"There is always a loop preheader.");
3485 !
match(ValShiftedIsZero,
3499 IntrinID = ValShifted->
getOpcode() == Instruction::Shl ? Intrinsic::cttz
3508 else if (
match(NBits,
3512 ExtraOffsetExpr = SE->
getSCEV(ExtraOffset);
3520 if (!IVPN || IVPN->getParent() != LoopHeaderBB) {
3525 Start = IVPN->getIncomingValueForBlock(LoopPreheaderBB);
3536 "Should only get equality predicates here.");
3547 if (FalseBB != LoopHeaderBB) {
3558 if (ValShifted->
getOpcode() == Instruction::AShr &&
3622bool LoopIdiomRecognize::recognizeShiftUntilZero() {
3623 bool MadeChange =
false;
3629 const SCEV *ExtraOffsetExpr;
3632 Start, Val, ExtraOffsetExpr, InvertedCond)) {
3634 " shift-until-zero idiom detection failed.\n");
3644 assert(LoopPreheaderBB &&
"There is always a loop preheader.");
3647 assert(SuccessorBB &&
"There is only a single successor.");
3650 Builder.SetCurrentDebugLocation(
IV->getDebugLoc());
3666 " Intrinsic is too costly, not beneficial\n");
3673 bool OffsetIsZero = ExtraOffsetExpr->
isZero();
3677 Value *ValNumLeadingZeros = Builder.CreateIntrinsic(
3678 IntrID, Ty, {Val, Builder.getFalse()},
3679 nullptr, Val->
getName() +
".numleadingzeros");
3680 Value *ValNumActiveBits = Builder.CreateSub(
3682 Val->
getName() +
".numactivebits",
true,
3686 Expander.setInsertPoint(&*Builder.GetInsertPoint());
3687 Value *ExtraOffset = Expander.expandCodeFor(ExtraOffsetExpr);
3689 Value *ValNumActiveBitsOffset = Builder.CreateAdd(
3690 ValNumActiveBits, ExtraOffset, ValNumActiveBits->
getName() +
".offset",
3691 OffsetIsZero,
true);
3692 Value *IVFinal = Builder.CreateIntrinsic(Intrinsic::smax, {Ty},
3693 {ValNumActiveBitsOffset,
Start},
3694 nullptr,
"iv.final");
3697 IVFinal, Start, CurLoop->
getName() +
".backedgetakencount",
3698 OffsetIsZero,
true));
3702 Value *LoopTripCount =
3703 Builder.CreateAdd(LoopBackedgeTakenCount, ConstantInt::get(Ty, 1),
3704 CurLoop->
getName() +
".tripcount",
true,
3710 IV->replaceUsesOutsideBlock(IVFinal, LoopHeaderBB);
3715 Builder.SetInsertPoint(LoopHeaderBB, LoopHeaderBB->
begin());
3716 auto *CIV = Builder.CreatePHI(Ty, 2, CurLoop->
getName() +
".iv");
3721 Builder.CreateAdd(CIV, ConstantInt::get(Ty, 1), CIV->getName() +
".next",
3722 true, Bitwidth != 2);
3725 auto *CIVCheck = Builder.CreateICmpEQ(CIVNext, LoopTripCount,
3726 CurLoop->
getName() +
".ivcheck");
3727 auto *NewIVCheck = CIVCheck;
3729 NewIVCheck = Builder.CreateNot(CIVCheck);
3730 NewIVCheck->takeName(ValShiftedIsZero);
3734 auto *IVDePHId = Builder.CreateAdd(CIV, Start,
"",
false,
3736 IVDePHId->takeName(
IV);
3741 const bool HasBranchWeights =
3745 auto *BI = Builder.CreateCondBr(CIVCheck, SuccessorBB, LoopHeaderBB);
3746 if (HasBranchWeights) {
3748 std::swap(BranchWeights[0], BranchWeights[1]);
3756 CIV->addIncoming(ConstantInt::get(Ty, 0), LoopPreheaderBB);
3757 CIV->addIncoming(CIVNext, LoopHeaderBB);
3765 IV->replaceAllUsesWith(IVDePHId);
3766 IV->eraseFromParent();
3775 ++NumShiftUntilZero;
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static const Function * getParent(const Value *V)
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
This file defines the DenseMap class.
ManagedStatic< HTTPClientCleanup > Cleanup
static bool mayLoopAccessLocation(Value *Ptr, ModRefInfo Access, Loop *L, const SCEV *BECount, unsigned StoreSize, AliasAnalysis &AA, SmallPtrSetImpl< Instruction * > &Ignored)
mayLoopAccessLocation - Return true if the specified loop might access the specified pointer location...
Module.h This file contains the declarations for the Module class.
This header defines various interfaces for pass management in LLVM.
This file defines an InstructionCost class that is used when calculating the cost of an instruction,...
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static PHINode * getRecurrenceVar(Value *VarX, Instruction *DefX, BasicBlock *LoopEntry)
static Value * createPopcntIntrinsic(IRBuilder<> &IRBuilder, Value *Val, const DebugLoc &DL)
static Value * matchShiftULTCondition(CondBrInst *BI, BasicBlock *LoopEntry, APInt &Threshold)
Check if the given conditional branch is based on an unsigned less-than comparison between a variable...
static bool detectShiftUntilLessThanIdiom(Loop *CurLoop, const DataLayout &DL, Intrinsic::ID &IntrinID, Value *&InitX, Instruction *&CntInst, PHINode *&CntPhi, Instruction *&DefX, APInt &Threshold)
Return true if the idiom is detected in the loop.
static Value * matchCondition(CondBrInst *BI, BasicBlock *LoopEntry, bool JmpOnZero=false)
Check if the given conditional branch is based on the comparison between a variable and zero,...
static bool detectShiftUntilBitTestIdiom(Loop *CurLoop, Value *&BaseX, Value *&BitMask, Value *&BitPos, Value *&CurrX, Instruction *&NextX)
Return true if the idiom is detected in the loop.
static bool detectPopcountIdiom(Loop *CurLoop, BasicBlock *PreCondBB, Instruction *&CntInst, PHINode *&CntPhi, Value *&Var)
Return true iff the idiom is detected in the loop.
static Constant * getMemSetPatternValue(Value *V, const DataLayout *DL)
getMemSetPatternValue - If a strided store of the specified value is safe to turn into a memset....
static const SCEV * getNumBytes(const SCEV *BECount, Type *IntPtr, const SCEV *StoreSizeSCEV, Loop *CurLoop, const DataLayout *DL, ScalarEvolution *SE)
Compute the number of bytes as a SCEV from the backedge taken count.
static bool detectShiftUntilZeroIdiom(Loop *CurLoop, const DataLayout &DL, Intrinsic::ID &IntrinID, Value *&InitX, Instruction *&CntInst, PHINode *&CntPhi, Instruction *&DefX)
Return true if the idiom is detected in the loop.
static Value * createFFSIntrinsic(IRBuilder<> &IRBuilder, Value *Val, const DebugLoc &DL, bool ZeroCheck, Intrinsic::ID IID)
static const SCEV * getStartForNegStride(const SCEV *Start, const SCEV *BECount, Type *IntPtr, const SCEV *StoreSizeSCEV, ScalarEvolution *SE)
static APInt getStoreStride(const SCEVAddRecExpr *StoreEv)
match_LoopInvariant< Ty > m_LoopInvariant(const Ty &M, const Loop *L)
Matches if the value is loop-invariant.
static bool isSameByteValueStore(Instruction &I, Value *SplatByte, Loop *L, const DataLayout &DL)
Return true if I is a (simple, loop-invariant-valued) store of the same bytewise value SplatByte.
static void deleteDeadInstruction(Instruction *I)
This file implements a map that provides insertion order iteration.
This file provides utility analysis objects describing memory locations.
This file exposes an interface to building/using memory SSA to walk memory instructions using a use/d...
Contains a collection of routines for determining if a given instruction is guaranteed to execute if ...
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
verify safepoint Safepoint IR Verifier
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static SymbolRef::Type getType(const Symbol *Sym)
static const uint32_t IV[8]
Class for arbitrary precision integers.
std::optional< uint64_t > tryZExtValue() const
Get zero extended value if possible.
uint64_t getZExtValue() const
Get zero extended value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
LLVM Basic Block Representation.
iterator begin()
Instruction iterator methods.
iterator_range< const_phi_iterator > phis() const
Returns a range that iterates over the phis in the basic block.
const Function * getParent() const
Return the enclosing method, or null if none.
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
InstListType::iterator iterator
Instruction iterators...
LLVM_ABI const_iterator getFirstNonPHIOrDbgOrAlloca() const
Returns an iterator to the first instruction in this block that is not a PHINode, a debug intrinsic,...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
LLVM_ABI const Module * getModule() const
Return the module owning the function this basic block belongs to, or nullptr if the function does no...
BinaryOps getOpcode() const
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
This class represents a function call, abstracting a target machine's calling convention.
void setPredicate(Predicate P)
Set the predicate for this instruction to the specified value.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ ICMP_SLE
signed less or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_ULT
unsigned less than
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Predicate getPredicate() const
Return the predicate for this instruction.
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
Conditional Branch instruction.
void setCondition(Value *V)
Value * getCondition() const
BasicBlock * getSuccessor(unsigned i) const
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
This is the shared class of boolean and integer constants.
bool isMinusOne() const
This function will return true iff every bit in this constant is set to true.
bool isOne() const
This is just a convenience method to make client code smaller for a common case.
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
This class represents a freeze function that returns random concrete value if an operand is either a ...
PointerType * getType() const
Global values are always pointers.
@ PrivateLinkage
Like Internal, but omit from symbol table.
static LLVM_ABI CRCTable genSarwateTable(const APInt &GenPoly, bool IsBigEndian)
Generate a lookup table of 256 entries by interleaving the generating polynomial.
static LLVM_ABI std::pair< APInt, APInt > genBarrettConstants(const PolynomialInfo &Info)
Auxilary entry point after analysis to generate constants for a GF(2) Barrett Reduction.
This instruction compares its operands according to the predicate given to the constructor.
bool isEquality() const
Return true if this predicate is either EQ or NE.
static bool isEquality(Predicate P)
Return true if this predicate is either EQ or NE.
Common base class shared among various IRBuilders.
ConstantInt * getInt1(bool V)
Get a constant value representing either true or false.
Value * CreateZExtOrTrunc(Value *V, Type *DestTy, const Twine &Name="")
Create a ZExt or Trunc from the integer value V to DestTy.
void SetCurrentDebugLocation(const DebugLoc &L)
Set location information used by debugging information.
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI bool hasNoUnsignedWrap() const LLVM_READONLY
Determine whether the no unsigned wrap flag is set.
LLVM_ABI bool hasNoSignedWrap() const LLVM_READONLY
Determine whether the no signed wrap flag is set.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isAtomic() const LLVM_READONLY
Return true if this instruction has an AtomicOrdering of unordered or higher.
LLVM_ABI void insertBefore(InstListType::iterator InsertPos)
Insert an unlinked instruction into a basic block immediately before the specified position.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI BasicBlock * getSuccessor(unsigned Idx) const LLVM_READONLY
Return the specified successor. This instruction must be a terminator.
LLVM_ABI AAMDNodes getAAMetadata() const
Returns the AA metadata for this instruction.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
This is an important class for using LLVM in a threaded context.
This class provides an interface for updating the loop pass manager based on mutations to the loop ne...
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
Align getAlign() const
Return the alignment of the access that is being performed.
static LocationSize precise(uint64_t Value)
static constexpr LocationSize afterPointer()
Any location after the base pointer (but still within the underlying object).
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
bool isOutermost() const
Return true if the loop does not have a parent (natural) loop.
BlockT * getLoopLatch() const
If there is a single latch block for this loop, return it.
unsigned getNumBlocks() const
Get the number of blocks in this loop in constant time.
unsigned getNumBackEdges() const
Calculate the number of back edges to the loop header.
BlockT * getHeader() const
BlockT * getExitBlock() const
If getExitBlocks would return exactly one block, return that block.
BlockT * getLoopPreheader() const
If there is a preheader for this loop, return it.
ArrayRef< BlockT * > getBlocks() const
Get a list of the basic blocks which make up this loop.
void getUniqueExitBlocks(SmallVectorImpl< BlockT * > &ExitBlocks) const
Return all unique successor blocks of this loop.
block_iterator block_begin() const
BlockT * getUniqueExitBlock() const
If getUniqueExitBlocks would return exactly one block, return that block.
LLVM_ABI PreservedAnalyses run(Loop &L, LoopAnalysisManager &AM, LoopStandardAnalysisResults &AR, LPMUpdater &U)
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
Represents a single loop in the control flow graph.
DebugLoc getStartLoc() const
Return the debug location of the start of this loop.
bool isLoopInvariant(const Value *V) const
Return true if the specified value is loop invariant.
ICmpInst * getLatchCmpInst() const
Get the latch condition instruction.
StringRef getName() const
PHINode * getCanonicalInductionVariable() const
Check to see if the loop has a canonical induction variable: an integer recurrence that starts at 0 a...
This class wraps the llvm.memcpy intrinsic.
Value * getLength() const
Value * getDest() const
This is just like getRawDest, but it strips off any cast instructions (including addrspacecast) that ...
MaybeAlign getDestAlign() const
bool isForceInlined() const
This class wraps the llvm.memset and llvm.memset.inline intrinsics.
MaybeAlign getSourceAlign() const
Value * getSource() const
This is just like getRawSource, but it strips off any cast instructions that feed it,...
Representation for a specific memory location.
An analysis that produces MemorySSA for a function.
Encapsulates MemorySSA, including all data associated with memory accesses.
A Module instance is used to store all the information related to an LLVM module.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
Value * getIncomingValueForBlock(const BasicBlock *BB) const
Value * getIncomingValue(unsigned i) const
Return incoming value number x.
int getBasicBlockIndex(const BasicBlock *BB) const
Return the first index of the specified basic block in the value list for this PHI.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
SCEVUse getStepRecurrence(ScalarEvolution &SE) const
Constructs and returns the recurrence indicating how much this expression steps by.
This class represents a constant integer value.
ConstantInt * getValue() const
const APInt & getAPInt() const
Helper to remove instructions inserted during SCEV expansion, unless they are marked as used.
This class uses information about analyze scalars to rewrite expressions in canonical form.
SCEVUse getOperand(unsigned i) const
This class represents an analyzed expression in the program.
LLVM_ABI bool isOne() const
Return true if the expression is a constant one.
static constexpr auto FlagNUW
LLVM_ABI bool isZero() const
Return true if the expression is a constant zero.
LLVM_ABI bool isNonConstantNegative() const
Return true if the specified scev is negated, but not a constant.
LLVM_ABI Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI bool isKnownNonNegative(const SCEV *S)
Test if the given expression is known to be non-negative.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
const SCEV * getZero(Type *Ty)
Return a SCEV for the constant 0 of a specific type.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI const SCEV * getTripCountFromExitCount(const SCEV *ExitCount)
A version of getTripCountFromExitCount below which always picks an evaluation type which can not resu...
LLVM_ABI void forgetLoop(const Loop *L)
This method should be called by the client when it has changed a loop in a way that may effect Scalar...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isSCEVable(Type *Ty) const
Test if values of the given type are analyzable within the SCEV framework.
LLVM_ABI bool hasLoopInvariantBackedgeTakenCount(const Loop *L)
Return true if the specified loop has an analyzable loop-invariant backedge-taken count.
LLVM_ABI const SCEV * getMulExpr(SmallVectorImpl< SCEVUse > &Ops, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Get a canonical multiply expression, or something simpler if possible.
LLVM_ABI const SCEV * getAddExpr(SmallVectorImpl< SCEVUse > &Ops, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Get a canonical add expression, or something simpler if possible.
LLVM_ABI const SCEV * applyLoopGuards(const SCEV *Expr, const Loop *L)
Try to apply information from loop guards for L to Expr.
LLVM_ABI const SCEV * getTruncateOrZeroExtend(const SCEV *V, Type *Ty, unsigned Depth=0)
Return a SCEV corresponding to a conversion of the input value to the specified type.
LLVM_ABI const SCEV * getTruncateOrSignExtend(const SCEV *V, Type *Ty, unsigned Depth=0)
Return a SCEV corresponding to a conversion of the input value to the specified type.
A vector that has set insertion semantics.
size_type count(const_arg_type key) const
Count the number of elements of a given key in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
Simple and conservative implementation of LoopSafetyInfo that can give false-positive answers to its ...
void computeLoopSafetyInfo(const Loop *CurLoop) override
Computes safety information for a loop checks loop body & header for the possibility of may throw exc...
bool anyBlockMayThrow() const override
Returns true iff any block of the loop for which this info is contains an instruction that may throw ...
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
bool erase(PtrType Ptr)
Remove pointer from the set.
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
void insert_range(Range &&R)
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Value * getValueOperand()
Value * getPointerOperand()
Represent a constant reference to a string, i.e.
Provides information about what library functions are available for the current target.
unsigned getWCharSize(const Module &M) const
Returns the size of the wchar_t type in bytes or 0 if the size is unknown.
bool has(LibFunc F) const
Tests whether a library function is available.
Triple - Helper class for working with autoconf configuration names.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI unsigned getIntegerBitWidth() const
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
unsigned getNumOperands() const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
iterator_range< user_iterator > users()
LLVM_ABI void replaceUsesOutsideBlock(Value *V, BasicBlock *BB)
replaceUsesOutsideBlock - Go through the uses list for this definition and make each use point to "V"...
LLVM_ABI bool replaceUsesWithIf(Value *New, llvm::function_ref< bool(Use &U)> ShouldReplace)
Go through the uses list for this definition and make each use point to "V" if the callback ShouldRep...
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Value handle that is nullable, but tries to track the Value.
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ C
The default llvm calling convention, compatible with C.
@ BasicBlock
Various leaf nodes.
OperandType
Operands are tagged with one of the values of this enum.
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
auto m_BasicBlock()
Match an arbitrary basic block value and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
cst_pred_ty< icmp_pred_with_threshold > m_SpecificInt_ICMP(ICmpInst::Predicate Predicate, const APInt &Threshold)
Match an integer or vector with every element comparing 'pred' (eg/ne/...) to Threshold.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
specificscev_ty m_scev_Specific(const SCEV *S)
Match if we have a specific specified SCEV.
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
DiagnosticInfoOptimizationBase::Argument NV
DiagnosticInfoOptimizationBase::setExtraArgs setExtraArgs
bool isSimple(Instruction *I)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
LLVM_ABI bool RecursivelyDeleteTriviallyDeadInstructions(Value *V, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, std::function< void(Value *)> AboutToDeleteCallback=std::function< void(Value *)>())
If the specified value is a trivially dead instruction, delete it.
static cl::opt< bool, true > EnableLIRPWcslen("disable-loop-idiom-wcslen", cl::desc("Proceed with loop idiom recognize pass, " "enable conversion of loop(s) to wcslen."), cl::location(DisableLIRP::Wcslen), cl::init(false), cl::ReallyHidden)
static cl::opt< bool, true > DisableLIRPMemcpy("disable-" DEBUG_TYPE "-memcpy", cl::desc("Proceed with loop idiom recognize pass, but do " "not convert loop(s) to memcpy."), cl::location(DisableLIRP::Memcpy), cl::init(false), cl::ReallyHidden)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
static cl::opt< bool, true > DisableLIRPStrlen("disable-loop-idiom-strlen", cl::desc("Proceed with loop idiom recognize pass, but do " "not convert loop(s) to strlen."), cl::location(DisableLIRP::Strlen), cl::init(false), cl::ReallyHidden)
@ Store
The extracted value is stored (ExtractElement only).
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
Value * GetPointerBaseWithConstantOffset(Value *Ptr, int64_t &Offset, const DataLayout &DL, bool AllowNonInbounds=true)
Analyze the specified pointer to see if it can be expressed as a base pointer plus a constant offset.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
static cl::opt< bool > ForceMemsetPatternIntrinsic("loop-idiom-force-memset-pattern-intrinsic", cl::desc("Use memset.pattern intrinsic whenever possible"), cl::init(false), cl::Hidden)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool isLibFuncEmittable(const Module *M, const TargetLibraryInfo *TLI, LibFunc TheLibFunc)
Check whether the library function is available on target and also that it in the current Module is a...
LLVM_ABI void setBranchWeights(Instruction &I, ArrayRef< uint32_t > Weights, bool IsExpected, bool ElideAllZero=false)
Create a new branch_weights metadata node and add or overwrite a prof metadata reference to instructi...
AnalysisManager< Loop, LoopStandardAnalysisResults & > LoopAnalysisManager
The loop analysis manager.
static cl::opt< bool > ForceCRCClmul("loop-idiom-force-crc-clmul", cl::desc("Use the clmul-based CRC loop optimization whenever possible"), cl::init(false), cl::Hidden)
auto dyn_cast_or_null(const Y &Val)
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
LLVM_ABI bool isMustProgress(const Loop *L)
Return true if this loop can be assumed to make progress.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool isModOrRefSet(const ModRefInfo MRI)
LLVM_ABI bool RecursivelyDeleteDeadPHINode(PHINode *PN, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, SmallPtrSetImpl< PHINode * > *KnownNonDeadPHIs=nullptr)
If the specified value is an effectively dead PHI node, due to being a def-use chain of single-use no...
LLVM_ABI Value * emitStrLen(Value *Ptr, IRBuilderBase &B, const DataLayout &DL, const TargetLibraryInfo *TLI)
Emit a call to the strlen function to the builder, for the specified pointer.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
ModRefInfo
Flags indicating whether a memory access modifies or references memory.
@ ModRef
The access may reference and may modify the value stored in memory.
@ Mod
The access may modify the value stored in memory.
static cl::opt< bool, true > DisableLIRPHashRecognize("disable-" DEBUG_TYPE "-hashrecognize", cl::desc("Proceed with loop idiom recognize pass, " "but do not optimize CRC loops."), cl::location(DisableLIRP::HashRecognize), cl::init(false), cl::ReallyHidden)
LLVM_ABI bool VerifyMemorySSA
Enables verification of MemorySSA.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
LLVM_ABI bool isConsecutiveAccess(Value *A, Value *B, const DataLayout &DL, ScalarEvolution &SE, bool CheckType=true)
Returns true if the memory operations A and B are consecutive.
DWARFExpression::Operation Op
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
LLVM_ABI Value * emitWcsLen(Value *Ptr, IRBuilderBase &B, const DataLayout &DL, const TargetLibraryInfo *TLI)
Emit a call to the wcslen function to the builder, for the specified pointer.
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI PreservedAnalyses getLoopPassPreservedAnalyses()
Returns the minimum set of Analyses that all loop passes must preserve.
LLVM_ABI Value * isBytewiseValue(Value *V, const DataLayout &DL)
If the specified value can be set by repeating the same byte in memory, return the i8 value that it i...
static cl::opt< bool > UseLIRCodeSizeHeurs("use-lir-code-size-heurs", cl::desc("Use loop idiom recognition code size heuristics when compiling " "with -Os/-Oz"), cl::init(true), cl::Hidden)
static cl::opt< bool, true > DisableLIRPMemset("disable-" DEBUG_TYPE "-memset", cl::desc("Proceed with loop idiom recognize pass, but do " "not convert loop(s) to memset."), cl::location(DisableLIRP::Memset), cl::init(false), cl::ReallyHidden)
static cl::opt< bool, true > DisableLIRPAll("disable-" DEBUG_TYPE "-all", cl::desc("Options to disable Loop Idiom Recognize Pass."), cl::location(DisableLIRP::All), cl::init(false), cl::ReallyHidden)
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI std::optional< DecomposedBitTest > decomposeBitTestICmp(Value *LHS, Value *RHS, CmpInst::Predicate Pred, bool LookThroughTrunc=true, bool AllowNonZeroC=false, bool DecomposeAnd=false)
Decompose an icmp into the form ((X & Mask) pred C) if possible.
SCEVUseT< const SCEV * > SCEVUse
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
LLVM_ABI AAMDNodes merge(const AAMDNodes &Other) const
Given two sets of AAMDNodes applying to potentially different locations, determine the best AAMDNodes...
AAMDNodes extendTo(ssize_t Len) const
Create a new AAMDNode that describes this AAMDNode after extending it to apply to a series of bytes o...
static LLVM_ABI bool Memcpy
When true, Memcpy is disabled.
static LLVM_ABI bool Wcslen
When true, Wcslen is disabled.
static LLVM_ABI bool Strlen
When true, Strlen is disabled.
static LLVM_ABI bool HashRecognize
When true, HashRecognize is disabled.
static LLVM_ABI bool Memset
When true, Memset is disabled.
static LLVM_ABI bool All
When true, the entire pass is disabled.
The adaptor from a function pass to a loop pass computes these analyses and makes them available to t...
TargetTransformInfo & TTI
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
The structure that is returned when a polynomial algorithm was recognized by the analysis.
Match loop-invariant value.
match_LoopInvariant(const SubPattern_t &SP, const Loop *L)