LLVM 24.0.0git
LoopStrengthReduce.cpp
Go to the documentation of this file.
1//===- LoopStrengthReduce.cpp - Strength Reduce IVs in Loops --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This transformation analyzes and transforms the induction variables (and
10// computations derived from them) into forms suitable for efficient execution
11// on the target.
12//
13// This pass performs a strength reduction on array references inside loops that
14// have as one or more of their components the loop induction variable, it
15// rewrites expressions to take advantage of scaled-index addressing modes
16// available on the target, and it performs a variety of other optimizations
17// related to loop induction variables.
18//
19// Terminology note: this code has a lot of handling for "post-increment" or
20// "post-inc" users. This is not talking about post-increment addressing modes;
21// it is instead talking about code like this:
22//
23// %i = phi [ 0, %entry ], [ %i.next, %latch ]
24// ...
25// %i.next = add %i, 1
26// %c = icmp eq %i.next, %n
27//
28// The SCEV for %i is {0,+,1}<%L>. The SCEV for %i.next is {1,+,1}<%L>, however
29// it's useful to think about these as the same register, with some uses using
30// the value of the register before the add and some using it after. In this
31// example, the icmp is a post-increment user, since it uses %i.next, which is
32// the value of the induction variable after the increment. The other common
33// case of post-increment users is users outside the loop.
34//
35// TODO: More sophistication in the way Formulae are generated and filtered.
36//
37// TODO: Handle multiple loops at a time.
38//
39// TODO: Should the addressing mode BaseGV be changed to a ConstantExpr instead
40// of a GlobalValue?
41//
42// TODO: When truncation is free, truncate ICmp users' operands to make it a
43// smaller encoding (on x86 at least).
44//
45// TODO: When a negated register is used by an add (such as in a list of
46// multiple base registers, or as the increment expression in an addrec),
47// we may not actually need both reg and (-1 * reg) in registers; the
48// negation can be implemented by using a sub instead of an add. The
49// lack of support for taking this into consideration when making
50// register pressure decisions is partly worked around by the "Special"
51// use kind.
52//
53//===----------------------------------------------------------------------===//
54
56#include "ScalarOptions.h"
57#include "llvm/ADT/APInt.h"
58#include "llvm/ADT/DenseMap.h"
59#include "llvm/ADT/DenseSet.h"
61#include "llvm/ADT/STLExtras.h"
62#include "llvm/ADT/SetVector.h"
65#include "llvm/ADT/SmallSet.h"
67#include "llvm/ADT/Statistic.h"
85#include "llvm/IR/BasicBlock.h"
86#include "llvm/IR/Constant.h"
87#include "llvm/IR/Constants.h"
90#include "llvm/IR/Dominators.h"
91#include "llvm/IR/GlobalValue.h"
92#include "llvm/IR/IRBuilder.h"
93#include "llvm/IR/InstrTypes.h"
94#include "llvm/IR/Instruction.h"
97#include "llvm/IR/Module.h"
98#include "llvm/IR/Operator.h"
99#include "llvm/IR/Type.h"
100#include "llvm/IR/Use.h"
101#include "llvm/IR/User.h"
102#include "llvm/IR/Value.h"
103#include "llvm/IR/ValueHandle.h"
105#include "llvm/Pass.h"
106#include "llvm/Support/Casting.h"
109#include "llvm/Support/Debug.h"
119#include <algorithm>
120#include <cassert>
121#include <cstddef>
122#include <cstdint>
123#include <iterator>
124#include <limits>
125#include <map>
126#include <numeric>
127#include <optional>
128#include <utility>
129
130using namespace llvm;
131using namespace SCEVPatternMatch;
132
133#define DEBUG_TYPE "loop-reduce"
134
135/// MaxIVUsers is an arbitrary threshold that provides an early opportunity for
136/// bail out. This threshold is far beyond the number of users that LSR can
137/// conceivably solve, so it should not affect generated code, but catches the
138/// worst cases before LSR burns too much compile time and stack space.
139static const unsigned MaxIVUsers = 200;
140
141/// Limit the size of expression that SCEV-based salvaging will attempt to
142/// translate into a DIExpression.
143/// Choose a maximum size such that debuginfo is not excessively increased and
144/// the salvaging is not too expensive for the compiler.
145static const unsigned MaxSCEVSalvageExpressionSize = 64;
146
147#ifndef NDEBUG
148// Stress test IV chain generation.
150 "stress-ivchain", cl::Hidden, cl::init(false),
151 cl::desc("Stress test LSR IV chains"));
152#else
153static bool StressIVChain = false;
154#endif
155
156namespace {
157
158struct MemAccessTy {
159 /// Used in situations where the accessed memory type is unknown.
160 static const unsigned UnknownAddressSpace =
161 std::numeric_limits<unsigned>::max();
162
163 Type *MemTy = nullptr;
164 unsigned AddrSpace = UnknownAddressSpace;
165
166 MemAccessTy() = default;
167 MemAccessTy(Type *Ty, unsigned AS) : MemTy(Ty), AddrSpace(AS) {}
168
169 bool operator==(MemAccessTy Other) const {
170 return MemTy == Other.MemTy && AddrSpace == Other.AddrSpace;
171 }
172
173 bool operator!=(MemAccessTy Other) const { return !(*this == Other); }
174
175 static MemAccessTy getUnknown(LLVMContext &Ctx,
176 unsigned AS = UnknownAddressSpace) {
177 return MemAccessTy(Type::getVoidTy(Ctx), AS);
178 }
179
180 Type *getType() { return MemTy; }
181};
182
183/// This class holds data which is used to order reuse candidates.
184class RegSortData {
185public:
186 /// This represents the set of LSRUse indices which reference
187 /// a particular register.
188 SmallBitVector UsedByIndices;
189
190 void print(raw_ostream &OS) const;
191 void dump() const;
192};
193
194// An offset from an address that is either scalable or fixed. Used for
195// per-target optimizations of addressing modes.
196class Immediate : public details::FixedOrScalableQuantity<Immediate, int64_t> {
197 constexpr Immediate(ScalarTy MinVal, bool Scalable)
198 : FixedOrScalableQuantity(MinVal, Scalable) {}
199
200 constexpr Immediate(const FixedOrScalableQuantity<Immediate, int64_t> &V)
201 : FixedOrScalableQuantity(V) {}
202
203public:
204 constexpr Immediate() = delete;
205
206 static constexpr Immediate getFixed(ScalarTy MinVal) {
207 return {MinVal, false};
208 }
209 static constexpr Immediate getScalable(ScalarTy MinVal) {
210 return {MinVal, true};
211 }
212 static constexpr Immediate get(ScalarTy MinVal, bool Scalable) {
213 return {MinVal, Scalable};
214 }
215 static constexpr Immediate getZero() { return {0, false}; }
216 static constexpr Immediate getFixedMin() {
217 return {std::numeric_limits<int64_t>::min(), false};
218 }
219 static constexpr Immediate getFixedMax() {
220 return {std::numeric_limits<int64_t>::max(), false};
221 }
222 static constexpr Immediate getScalableMin() {
223 return {std::numeric_limits<int64_t>::min(), true};
224 }
225 static constexpr Immediate getScalableMax() {
226 return {std::numeric_limits<int64_t>::max(), true};
227 }
228
229 constexpr bool isLessThanZero() const { return Quantity < 0; }
230
231 constexpr bool isGreaterThanZero() const { return Quantity > 0; }
232
233 constexpr bool isCompatibleImmediate(const Immediate &Imm) const {
234 return isZero() || Imm.isZero() || Imm.Scalable == Scalable;
235 }
236
237 constexpr bool isMin() const {
238 return Quantity == std::numeric_limits<ScalarTy>::min();
239 }
240
241 constexpr bool isMax() const {
242 return Quantity == std::numeric_limits<ScalarTy>::max();
243 }
244
245 // Arithmetic 'operators' that cast to unsigned types first.
246 constexpr Immediate addUnsigned(const Immediate &RHS) const {
247 assert(isCompatibleImmediate(RHS) && "Incompatible Immediates");
248 ScalarTy Value = (uint64_t)Quantity + RHS.getKnownMinValue();
249 return {Value, Scalable || RHS.isScalable()};
250 }
251
252 constexpr Immediate subUnsigned(const Immediate &RHS) const {
253 assert(isCompatibleImmediate(RHS) && "Incompatible Immediates");
254 ScalarTy Value = (uint64_t)Quantity - RHS.getKnownMinValue();
255 return {Value, Scalable || RHS.isScalable()};
256 }
257
258 // Scale the quantity by a constant without caring about runtime scalability.
259 constexpr Immediate mulUnsigned(const ScalarTy RHS) const {
260 ScalarTy Value = (uint64_t)Quantity * RHS;
261 return {Value, Scalable};
262 }
263
264 // Helpers for generating SCEVs with vscale terms where needed.
265 const SCEV *getSCEV(ScalarEvolution &SE, Type *Ty) const {
266 const SCEV *S = SE.getConstant(Ty, Quantity);
267 if (Scalable)
268 S = SE.getMulExpr(S, SE.getVScale(S->getType()));
269 return S;
270 }
271
272 const SCEV *getNegativeSCEV(ScalarEvolution &SE, Type *Ty) const {
273 const SCEV *NegS = SE.getConstant(Ty, -(uint64_t)Quantity);
274 if (Scalable)
275 NegS = SE.getMulExpr(NegS, SE.getVScale(NegS->getType()));
276 return NegS;
277 }
278
279 const SCEV *getUnknownSCEV(ScalarEvolution &SE, Type *Ty) const {
280 // TODO: Avoid implicit trunc?
281 // See https://github.com/llvm/llvm-project/issues/112510.
282 const SCEV *SU = SE.getUnknown(
283 ConstantInt::getSigned(Ty, Quantity, /*ImplicitTrunc=*/true));
284 if (Scalable)
285 SU = SE.getMulExpr(SU, SE.getVScale(SU->getType()));
286 return SU;
287 }
288};
289
290// This is needed for the Compare type of std::map when Immediate is used
291// as a key. We don't need it to be fully correct against any value of vscale,
292// just to make sure that vscale-related terms in the map are considered against
293// each other rather than being mixed up and potentially missing opportunities.
294struct KeyOrderTargetImmediate {
295 bool operator()(const Immediate &LHS, const Immediate &RHS) const {
296 if (LHS.isScalable() && !RHS.isScalable())
297 return false;
298 if (!LHS.isScalable() && RHS.isScalable())
299 return true;
300 return LHS.getKnownMinValue() < RHS.getKnownMinValue();
301 }
302};
303
304// This would be nicer if we could be generic instead of directly using size_t,
305// but there doesn't seem to be a type trait for is_orderable or
306// is_lessthan_comparable or similar.
307struct KeyOrderSizeTAndImmediate {
308 bool operator()(const std::pair<size_t, Immediate> &LHS,
309 const std::pair<size_t, Immediate> &RHS) const {
310 size_t LSize = LHS.first;
311 size_t RSize = RHS.first;
312 if (LSize != RSize)
313 return LSize < RSize;
314 return KeyOrderTargetImmediate()(LHS.second, RHS.second);
315 }
316};
317} // end anonymous namespace
318
319#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
320void RegSortData::print(raw_ostream &OS) const {
321 OS << "[NumUses=" << UsedByIndices.count() << ']';
322}
323
324LLVM_DUMP_METHOD void RegSortData::dump() const {
325 print(errs()); errs() << '\n';
326}
327#endif
328
329namespace {
330
331/// Map register candidates to information about how they are used.
332class RegUseTracker {
333 using RegUsesTy = DenseMap<const SCEV *, RegSortData>;
334
335 RegUsesTy RegUsesMap;
337
338public:
339 void countRegister(const SCEV *Reg, size_t LUIdx);
340 void dropRegister(const SCEV *Reg, size_t LUIdx);
341 void swapAndDropUse(size_t LUIdx, size_t LastLUIdx);
342
343 bool isRegUsedByUsesOtherThan(const SCEV *Reg, size_t LUIdx) const;
344
345 const SmallBitVector &getUsedByIndices(const SCEV *Reg) const;
346
347 void clear();
348
351
352 iterator begin() { return RegSequence.begin(); }
353 iterator end() { return RegSequence.end(); }
354 const_iterator begin() const { return RegSequence.begin(); }
355 const_iterator end() const { return RegSequence.end(); }
356};
357
358} // end anonymous namespace
359
360void
361RegUseTracker::countRegister(const SCEV *Reg, size_t LUIdx) {
362 std::pair<RegUsesTy::iterator, bool> Pair = RegUsesMap.try_emplace(Reg);
363 RegSortData &RSD = Pair.first->second;
364 if (Pair.second)
365 RegSequence.push_back(Reg);
366 RSD.UsedByIndices.resize(std::max(RSD.UsedByIndices.size(), LUIdx + 1));
367 RSD.UsedByIndices.set(LUIdx);
368}
369
370void
371RegUseTracker::dropRegister(const SCEV *Reg, size_t LUIdx) {
372 RegUsesTy::iterator It = RegUsesMap.find(Reg);
373 assert(It != RegUsesMap.end());
374 RegSortData &RSD = It->second;
375 assert(RSD.UsedByIndices.size() > LUIdx);
376 RSD.UsedByIndices.reset(LUIdx);
377}
378
379void
380RegUseTracker::swapAndDropUse(size_t LUIdx, size_t LastLUIdx) {
381 assert(LUIdx <= LastLUIdx);
382
383 // Update RegUses. The data structure is not optimized for this purpose;
384 // we must iterate through it and update each of the bit vectors.
385 for (auto &Pair : RegUsesMap) {
386 SmallBitVector &UsedByIndices = Pair.second.UsedByIndices;
387 if (LUIdx < UsedByIndices.size())
388 UsedByIndices[LUIdx] =
389 LastLUIdx < UsedByIndices.size() ? UsedByIndices[LastLUIdx] : false;
390 UsedByIndices.resize(std::min(UsedByIndices.size(), LastLUIdx));
391 }
392}
393
394bool
395RegUseTracker::isRegUsedByUsesOtherThan(const SCEV *Reg, size_t LUIdx) const {
396 RegUsesTy::const_iterator I = RegUsesMap.find(Reg);
397 if (I == RegUsesMap.end())
398 return false;
399 const SmallBitVector &UsedByIndices = I->second.UsedByIndices;
400 int i = UsedByIndices.find_first();
401 if (i == -1) return false;
402 if ((size_t)i != LUIdx) return true;
403 return UsedByIndices.find_next(i) != -1;
404}
405
406const SmallBitVector &RegUseTracker::getUsedByIndices(const SCEV *Reg) const {
407 RegUsesTy::const_iterator I = RegUsesMap.find(Reg);
408 assert(I != RegUsesMap.end() && "Unknown register!");
409 return I->second.UsedByIndices;
410}
411
412void RegUseTracker::clear() {
413 RegUsesMap.clear();
414 RegSequence.clear();
415}
416
417namespace {
418
419/// This class holds information that describes a formula for computing
420/// satisfying a use. It may include broken-out immediates and scaled registers.
421struct Formula {
422 /// Global base address used for complex addressing.
423 GlobalValue *BaseGV = nullptr;
424
425 /// Base offset for complex addressing.
426 Immediate BaseOffset = Immediate::getZero();
427
428 /// Whether any complex addressing has a base register.
429 bool HasBaseReg = false;
430
431 /// The scale of any complex addressing.
432 int64_t Scale = 0;
433
434 /// The list of "base" registers for this use. When this is non-empty. The
435 /// canonical representation of a formula is
436 /// 1. BaseRegs.size > 1 implies ScaledReg != NULL and
437 /// 2. ScaledReg != NULL implies Scale != 1 || !BaseRegs.empty().
438 /// 3. The reg containing recurrent expr related with currect loop in the
439 /// formula should be put in the ScaledReg.
440 /// #1 enforces that the scaled register is always used when at least two
441 /// registers are needed by the formula: e.g., reg1 + reg2 is reg1 + 1 * reg2.
442 /// #2 enforces that 1 * reg is reg.
443 /// #3 ensures invariant regs with respect to current loop can be combined
444 /// together in LSR codegen.
445 /// This invariant can be temporarily broken while building a formula.
446 /// However, every formula inserted into the LSRInstance must be in canonical
447 /// form.
449
450 /// The 'scaled' register for this use. This should be non-null when Scale is
451 /// not zero.
452 const SCEV *ScaledReg = nullptr;
453
454 /// An additional constant offset which added near the use. This requires a
455 /// temporary register, but the offset itself can live in an add immediate
456 /// field rather than a register.
457 Immediate UnfoldedOffset = Immediate::getZero();
458
459 Formula() = default;
460
461 void initialMatch(const SCEV *S, Loop *L, ScalarEvolution &SE);
462
463 bool isCanonical(const Loop &L) const;
464
465 void canonicalize(const Loop &L);
466
467 bool unscale();
468
469 bool hasZeroEnd() const;
470
471 bool countsDownToZero() const;
472
473 size_t getNumRegs() const;
474 Type *getType() const;
475
476 void deleteBaseReg(const SCEV *&S);
477
478 bool referencesReg(const SCEV *S) const;
479 bool hasRegsUsedByUsesOtherThan(size_t LUIdx,
480 const RegUseTracker &RegUses) const;
481
482 void print(raw_ostream &OS) const;
483 void dump() const;
484};
485
486} // end anonymous namespace
487
488/// Recursion helper for initialMatch.
489static void DoInitialMatch(const SCEV *S, Loop *L,
492 // Collect expressions which properly dominate the loop header.
493 if (SE.properlyDominates(S, L->getHeader())) {
494 Good.push_back(S);
495 return;
496 }
497
498 // Look at add operands.
499 if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
500 for (const SCEV *S : Add->operands())
501 DoInitialMatch(S, L, Good, Bad, SE);
502 return;
503 }
504
505 // Look at addrec operands.
506 const SCEV *Start, *Step;
507 const Loop *ARLoop;
508 if (match(S,
509 m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step), m_Loop(ARLoop))) &&
510 !Start->isZero()) {
511 DoInitialMatch(Start, L, Good, Bad, SE);
512 DoInitialMatch(SE.getAddRecExpr(SE.getConstant(S->getType(), 0), Step,
513 // FIXME: AR->getNoWrapFlags()
514 ARLoop, SCEV::FlagNone),
515 L, Good, Bad, SE);
516 return;
517 }
518
519 // Handle a multiplication by -1 (negation) if it didn't fold.
520 if (const SCEVMulExpr *Mul = dyn_cast<SCEVMulExpr>(S))
521 if (Mul->getOperand(0)->isAllOnesValue()) {
523 const SCEV *NewMul = SE.getMulExpr(Ops);
524
527 DoInitialMatch(NewMul, L, MyGood, MyBad, SE);
528 const SCEV *NegOne = SE.getSCEV(ConstantInt::getAllOnesValue(
529 SE.getEffectiveSCEVType(NewMul->getType())));
530 for (const SCEV *S : MyGood)
531 Good.push_back(SE.getMulExpr(NegOne, S));
532 for (const SCEV *S : MyBad)
533 Bad.push_back(SE.getMulExpr(NegOne, S));
534 return;
535 }
536
537 // Ok, we can't do anything interesting. Just stuff the whole thing into a
538 // register and hope for the best.
539 Bad.push_back(S);
540}
541
542/// Incorporate loop-variant parts of S into this Formula, attempting to keep
543/// all loop-invariant and loop-computable values in a single base register.
544void Formula::initialMatch(const SCEV *S, Loop *L, ScalarEvolution &SE) {
547 DoInitialMatch(S, L, Good, Bad, SE);
548 if (!Good.empty()) {
549 const SCEV *Sum = SE.getAddExpr(Good);
550 if (!Sum->isZero())
551 BaseRegs.push_back(Sum);
552 HasBaseReg = true;
553 }
554 if (!Bad.empty()) {
555 const SCEV *Sum = SE.getAddExpr(Bad);
556 if (!Sum->isZero())
557 BaseRegs.push_back(Sum);
558 HasBaseReg = true;
559 }
560 canonicalize(*L);
561}
562
563static bool containsAddRecDependentOnLoop(const SCEV *S, const Loop &L) {
564 return SCEVExprContains(S, [&L](const SCEV *S) {
565 return isa<SCEVAddRecExpr>(S) && (cast<SCEVAddRecExpr>(S)->getLoop() == &L);
566 });
567}
568
569/// Check whether or not this formula satisfies the canonical
570/// representation.
571/// \see Formula::BaseRegs.
572bool Formula::isCanonical(const Loop &L) const {
573 assert((Scale == 0 || ScaledReg) &&
574 "ScaledReg must be non-null if Scale is non-zero");
575
576 if (!ScaledReg)
577 return BaseRegs.size() <= 1;
578
579 if (Scale != 1)
580 return true;
581
582 if (Scale == 1 && BaseRegs.empty())
583 return false;
584
585 if (containsAddRecDependentOnLoop(ScaledReg, L))
586 return true;
587
588 // If ScaledReg is not a recurrent expr, or it is but its loop is not current
589 // loop, meanwhile BaseRegs contains a recurrent expr reg related with current
590 // loop, we want to swap the reg in BaseRegs with ScaledReg.
591 return none_of(BaseRegs, [&L](const SCEV *S) {
593 });
594}
595
596/// Helper method to morph a formula into its canonical representation.
597/// \see Formula::BaseRegs.
598/// Every formula having more than one base register, must use the ScaledReg
599/// field. Otherwise, we would have to do special cases everywhere in LSR
600/// to treat reg1 + reg2 + ... the same way as reg1 + 1*reg2 + ...
601/// On the other hand, 1*reg should be canonicalized into reg.
602void Formula::canonicalize(const Loop &L) {
603 if (isCanonical(L))
604 return;
605
606 if (BaseRegs.empty()) {
607 // No base reg? Use scale reg with scale = 1 as such.
608 assert(ScaledReg && "Expected 1*reg => reg");
609 assert(Scale == 1 && "Expected 1*reg => reg");
610 BaseRegs.push_back(ScaledReg);
611 Scale = 0;
612 ScaledReg = nullptr;
613 return;
614 }
615
616 // Keep the invariant sum in BaseRegs and one of the variant sum in ScaledReg.
617 if (!ScaledReg) {
618 ScaledReg = BaseRegs.pop_back_val();
619 Scale = 1;
620 }
621
622 // If ScaledReg is an invariant with respect to L, find the reg from
623 // BaseRegs containing the recurrent expr related with Loop L. Swap the
624 // reg with ScaledReg.
625 if (!containsAddRecDependentOnLoop(ScaledReg, L)) {
626 auto I = find_if(BaseRegs, [&L](const SCEV *S) {
628 });
629 if (I != BaseRegs.end())
630 std::swap(ScaledReg, *I);
631 }
632 assert(isCanonical(L) && "Failed to canonicalize?");
633}
634
635/// Get rid of the scale in the formula.
636/// In other words, this method morphes reg1 + 1*reg2 into reg1 + reg2.
637/// \return true if it was possible to get rid of the scale, false otherwise.
638/// \note After this operation the formula may not be in the canonical form.
639bool Formula::unscale() {
640 if (Scale != 1)
641 return false;
642 Scale = 0;
643 BaseRegs.push_back(ScaledReg);
644 ScaledReg = nullptr;
645 return true;
646}
647
648bool Formula::hasZeroEnd() const {
649 if (UnfoldedOffset || BaseOffset)
650 return false;
651 if (BaseRegs.size() != 1 || ScaledReg)
652 return false;
653 return true;
654}
655
656bool Formula::countsDownToZero() const {
657 if (!hasZeroEnd())
658 return false;
659 assert(BaseRegs.size() == 1 && "hasZeroEnd should mean one BaseReg");
660 const APInt *StepInt;
661 if (!match(BaseRegs[0], m_scev_AffineAddRec(m_SCEV(), m_scev_APInt(StepInt))))
662 return false;
663 return StepInt->isNegative();
664}
665
666/// Return the total number of register operands used by this formula. This does
667/// not include register uses implied by non-constant addrec strides.
668size_t Formula::getNumRegs() const {
669 return !!ScaledReg + BaseRegs.size();
670}
671
672/// Return the type of this formula, if it has one, or null otherwise. This type
673/// is meaningless except for the bit size.
674Type *Formula::getType() const {
675 return !BaseRegs.empty() ? BaseRegs.front()->getType() :
676 ScaledReg ? ScaledReg->getType() :
677 BaseGV ? BaseGV->getType() :
678 nullptr;
679}
680
681/// Delete the given base reg from the BaseRegs list.
682void Formula::deleteBaseReg(const SCEV *&S) {
683 if (&S != &BaseRegs.back())
684 std::swap(S, BaseRegs.back());
685 BaseRegs.pop_back();
686}
687
688/// Test if this formula references the given register.
689bool Formula::referencesReg(const SCEV *S) const {
690 return S == ScaledReg || is_contained(BaseRegs, S);
691}
692
693/// Test whether this formula uses registers which are used by uses other than
694/// the use with the given index.
695bool Formula::hasRegsUsedByUsesOtherThan(size_t LUIdx,
696 const RegUseTracker &RegUses) const {
697 if (ScaledReg)
698 if (RegUses.isRegUsedByUsesOtherThan(ScaledReg, LUIdx))
699 return true;
700 for (const SCEV *BaseReg : BaseRegs)
701 if (RegUses.isRegUsedByUsesOtherThan(BaseReg, LUIdx))
702 return true;
703 return false;
704}
705
706#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
707void Formula::print(raw_ostream &OS) const {
708 ListSeparator Plus(" + ");
709 if (BaseGV) {
710 OS << Plus;
711 BaseGV->printAsOperand(OS, /*PrintType=*/false);
712 }
713 if (BaseOffset.isNonZero())
714 OS << Plus << BaseOffset;
715
716 for (const SCEV *BaseReg : BaseRegs)
717 OS << Plus << "reg(" << *BaseReg << ')';
718
719 if (HasBaseReg && BaseRegs.empty())
720 OS << Plus << "**error: HasBaseReg**";
721 else if (!HasBaseReg && !BaseRegs.empty())
722 OS << Plus << "**error: !HasBaseReg**";
723
724 if (Scale != 0) {
725 OS << Plus << Scale << "*reg(";
726 if (ScaledReg)
727 OS << *ScaledReg;
728 else
729 OS << "<unknown>";
730 OS << ')';
731 }
732 if (UnfoldedOffset.isNonZero())
733 OS << Plus << "imm(" << UnfoldedOffset << ')';
734}
735
736LLVM_DUMP_METHOD void Formula::dump() const {
737 print(errs()); errs() << '\n';
738}
739#endif
740
741/// Return true if the given addrec can be sign-extended without changing its
742/// value.
744 Type *WideTy =
746 return isa<SCEVAddRecExpr>(SE.getSignExtendExpr(AR, WideTy));
747}
748
749/// Return true if the given add can be sign-extended without changing its
750/// value.
751static bool isAddSExtable(const SCEVAddExpr *A, ScalarEvolution &SE) {
752 Type *WideTy =
753 IntegerType::get(SE.getContext(), SE.getTypeSizeInBits(A->getType()) + 1);
754 return isa<SCEVAddExpr>(SE.getSignExtendExpr(A, WideTy));
755}
756
757/// Return true if the given mul can be sign-extended without changing its
758/// value.
759static bool isMulSExtable(const SCEVMulExpr *M, ScalarEvolution &SE) {
760 Type *WideTy =
762 SE.getTypeSizeInBits(M->getType()) * M->getNumOperands());
763 return isa<SCEVMulExpr>(SE.getSignExtendExpr(M, WideTy));
764}
765
766/// Return an expression for LHS /s RHS, if it can be determined and if the
767/// remainder is known to be zero, or null otherwise. If IgnoreSignificantBits
768/// is true, expressions like (X * Y) /s Y are simplified to X, ignoring that
769/// the multiplication may overflow, which is useful when the result will be
770/// used in a context where the most significant bits are ignored.
771static const SCEV *getExactSDiv(const SCEV *LHS, const SCEV *RHS,
772 ScalarEvolution &SE,
773 bool IgnoreSignificantBits = false) {
774 // Handle the trivial case, which works for any SCEV type.
775 if (LHS == RHS)
776 return SE.getConstant(LHS->getType(), 1);
777
778 // Handle a few RHS special cases.
780 if (RC) {
781 const APInt &RA = RC->getAPInt();
782 // Handle x /s -1 as x * -1, to give ScalarEvolution a chance to do
783 // some folding.
784 if (RA.isAllOnes()) {
785 if (LHS->getType()->isPointerTy())
786 return nullptr;
787 return SE.getMulExpr(LHS, RC);
788 }
789 // Handle x /s 1 as x.
790 if (RA == 1)
791 return LHS;
792 }
793
794 // Check for a division of a constant by a constant.
796 if (!RC)
797 return nullptr;
798 const APInt &LA = C->getAPInt();
799 const APInt &RA = RC->getAPInt();
800 if (LA.srem(RA) != 0)
801 return nullptr;
802 return SE.getConstant(LA.sdiv(RA));
803 }
804
805 // Distribute the sdiv over addrec operands, if the addrec doesn't overflow.
807 if ((IgnoreSignificantBits || isAddRecSExtable(AR, SE)) && AR->isAffine()) {
808 const SCEV *Step = getExactSDiv(AR->getStepRecurrence(SE), RHS, SE,
809 IgnoreSignificantBits);
810 if (!Step) return nullptr;
811 const SCEV *Start = getExactSDiv(AR->getStart(), RHS, SE,
812 IgnoreSignificantBits);
813 if (!Start) return nullptr;
814 // FlagNW is independent of the start value, step direction, and is
815 // preserved with smaller magnitude steps.
816 // FIXME: AR->getNoWrapFlags(SCEV::FlagNW)
817 return SE.getAddRecExpr(Start, Step, AR->getLoop(), SCEV::FlagNone);
818 }
819 return nullptr;
820 }
821
822 // Distribute the sdiv over add operands, if the add doesn't overflow.
824 if (IgnoreSignificantBits || isAddSExtable(Add, SE)) {
826 for (const SCEV *S : Add->operands()) {
827 const SCEV *Op = getExactSDiv(S, RHS, SE, IgnoreSignificantBits);
828 if (!Op) return nullptr;
829 Ops.push_back(Op);
830 }
831 return SE.getAddExpr(Ops);
832 }
833 return nullptr;
834 }
835
836 // Check for a multiply operand that we can pull RHS out of.
838 if (IgnoreSignificantBits || isMulSExtable(Mul, SE)) {
839 // Handle special case C1*X*Y /s C2*X*Y.
840 if (const SCEVMulExpr *MulRHS = dyn_cast<SCEVMulExpr>(RHS)) {
841 if (IgnoreSignificantBits || isMulSExtable(MulRHS, SE)) {
842 const SCEVConstant *LC = dyn_cast<SCEVConstant>(Mul->getOperand(0));
843 const SCEVConstant *RC =
844 dyn_cast<SCEVConstant>(MulRHS->getOperand(0));
845 if (LC && RC) {
847 SmallVector<const SCEV *, 4> ROps(drop_begin(MulRHS->operands()));
848 if (LOps == ROps)
849 return getExactSDiv(LC, RC, SE, IgnoreSignificantBits);
850 }
851 }
852 }
853
855 bool Found = false;
856 for (const SCEV *S : Mul->operands()) {
857 if (!Found)
858 if (const SCEV *Q = getExactSDiv(S, RHS, SE,
859 IgnoreSignificantBits)) {
860 S = Q;
861 Found = true;
862 }
863 Ops.push_back(S);
864 }
865 return Found ? SE.getMulExpr(Ops) : nullptr;
866 }
867 return nullptr;
868 }
869
870 // Otherwise we don't know.
871 return nullptr;
872}
873
874/// Extracts an immediate operand from \p Ops and replaces the operand with
875/// zero. If \p PreferScalable is true and \p Ops contains both a scalable and
876/// non-scalable offsets, the scalable offset will be extracted.
877static Immediate extractImmediateOperand(const ScalarOptions &Opts,
879 ScalarEvolution &SE,
880 bool PreferScalable) {
881 const APInt *C;
882 SCEVUse *Op = nullptr;
883 Immediate Result = Immediate::getZero();
884
885 // Ops are sorted by their SCEVType (the order of SCEVTypes enum). So, for an
886 // AddExpr the possible order of operands is:
887 // Constant < VScale < Truncate < ZeroExtend < SignExtend < MulExpr < ...
888
889 // This means fixed-size immediates will always appear on the LHS:
890 SCEVUse &S = Ops.front();
891 if (match(S, m_scev_APInt(C)) && !C->isZero() &&
892 C->getSignificantBits() <= 64) {
893 Op = &S;
894 Result = Immediate::getFixed(C->getSExtValue());
895 }
896
897 // But scalable immediates, which are MulExpr(Vscale, Constant), can appear
898 // later in the operand list:
899 if (Opts.lsr_enable_vscale_immediates &&
900 (Result.isZero() || PreferScalable)) {
901 for (SCEVUse &S : Ops) {
902 // We know anything past scMulExpr will not be a vscale immediate.
903 if (S->getSCEVType() > scMulExpr)
904 break;
906 Op = &S;
907 Result = Immediate::getScalable(C->getSExtValue());
908 break;
909 }
910 }
911 }
912
913 if (Result.isNonZero()) {
914 SCEVUse &S = *Op;
915 S = SE.getConstant(S->getType(), 0);
916 }
917
918 return Result;
919}
920
921/// If S involves the addition of a constant integer value, return that integer
922/// value, and mutate S to point to a new SCEV with that value excluded.
923static Immediate extractImmediate(const ScalarOptions &Opts, SCEVUse &S,
924 ScalarEvolution &SE,
925 bool PreferScalable = false) {
926 if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
927 SmallVector<SCEVUse, 8> NewOps(Add->operands());
928 Immediate Result =
929 extractImmediateOperand(Opts, NewOps, SE, PreferScalable);
930 if (Result.isZero())
931 Result = extractImmediate(Opts, NewOps.front(), SE, PreferScalable);
932 if (Result.isNonZero())
933 S = SE.getAddExpr(NewOps);
934 return Result;
935 } else if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(S)) {
936 SmallVector<SCEVUse, 8> NewOps(AR->operands());
937 Immediate Result =
938 extractImmediate(Opts, NewOps.front(), SE, PreferScalable);
939 if (Result.isNonZero())
940 S = SE.getAddRecExpr(NewOps, AR->getLoop(),
941 // FIXME: AR->getNoWrapFlags(SCEV::FlagNW)
943 return Result;
944 }
945 return extractImmediateOperand(Opts, {S}, SE, PreferScalable);
946}
947
948/// If S involves the addition of a GlobalValue address, return that symbol, and
949/// mutate S to point to a new SCEV with that value excluded.
951 if (const SCEVUnknown *U = dyn_cast<SCEVUnknown>(S)) {
952 if (GlobalValue *GV = dyn_cast<GlobalValue>(U->getValue())) {
953 S = SE.getConstant(GV->getType(), 0);
954 return GV;
955 }
956 } else if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
957 SmallVector<SCEVUse, 8> NewOps(Add->operands());
958 GlobalValue *Result = ExtractSymbol(NewOps.back(), SE);
959 if (Result)
960 S = SE.getAddExpr(NewOps);
961 return Result;
962 } else if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(S)) {
963 SmallVector<SCEVUse, 8> NewOps(AR->operands());
964 GlobalValue *Result = ExtractSymbol(NewOps.front(), SE);
965 if (Result)
966 S = SE.getAddRecExpr(NewOps, AR->getLoop(),
967 // FIXME: AR->getNoWrapFlags(SCEV::FlagNW)
969 return Result;
970 }
971 return nullptr;
972}
973
974/// Returns true if the specified instruction is using the specified value as an
975/// address.
977 Instruction *Inst, Value *OperandVal) {
978 bool isAddress = isa<LoadInst>(Inst);
979 if (StoreInst *SI = dyn_cast<StoreInst>(Inst)) {
980 if (SI->getPointerOperand() == OperandVal)
981 isAddress = true;
982 } else if (IntrinsicInst *II = dyn_cast<IntrinsicInst>(Inst)) {
983 // Addressing modes can also be folded into prefetches and a variety
984 // of intrinsics.
985 switch (II->getIntrinsicID()) {
986 case Intrinsic::memset:
987 case Intrinsic::prefetch:
988 case Intrinsic::masked_load:
989 if (II->getArgOperand(0) == OperandVal)
990 isAddress = true;
991 break;
992 case Intrinsic::masked_store:
993 if (II->getArgOperand(1) == OperandVal)
994 isAddress = true;
995 break;
996 case Intrinsic::memmove:
997 case Intrinsic::memcpy:
998 if (II->getArgOperand(0) == OperandVal ||
999 II->getArgOperand(1) == OperandVal)
1000 isAddress = true;
1001 break;
1002 default: {
1003 MemIntrinsicInfo IntrInfo;
1004 if (TTI.getTgtMemIntrinsic(II, IntrInfo)) {
1005 if (IntrInfo.PtrVal == OperandVal)
1006 isAddress = true;
1007 }
1008 }
1009 }
1010 } else if (AtomicRMWInst *RMW = dyn_cast<AtomicRMWInst>(Inst)) {
1011 if (RMW->getPointerOperand() == OperandVal)
1012 isAddress = true;
1013 } else if (AtomicCmpXchgInst *CmpX = dyn_cast<AtomicCmpXchgInst>(Inst)) {
1014 if (CmpX->getPointerOperand() == OperandVal)
1015 isAddress = true;
1016 }
1017 return isAddress;
1018}
1019
1020/// Return the type of the memory being accessed.
1021static MemAccessTy getAccessType(const TargetTransformInfo &TTI,
1022 Instruction *Inst, Value *OperandVal) {
1023 MemAccessTy AccessTy = MemAccessTy::getUnknown(Inst->getContext());
1024
1025 // First get the type of memory being accessed.
1026 if (Type *Ty = Inst->getAccessType())
1027 AccessTy.MemTy = Ty;
1028
1029 // Then get the pointer address space.
1030 if (const StoreInst *SI = dyn_cast<StoreInst>(Inst)) {
1031 AccessTy.AddrSpace = SI->getPointerAddressSpace();
1032 } else if (const LoadInst *LI = dyn_cast<LoadInst>(Inst)) {
1033 AccessTy.AddrSpace = LI->getPointerAddressSpace();
1034 } else if (const AtomicRMWInst *RMW = dyn_cast<AtomicRMWInst>(Inst)) {
1035 AccessTy.AddrSpace = RMW->getPointerAddressSpace();
1036 } else if (const AtomicCmpXchgInst *CmpX = dyn_cast<AtomicCmpXchgInst>(Inst)) {
1037 AccessTy.AddrSpace = CmpX->getPointerAddressSpace();
1038 } else if (IntrinsicInst *II = dyn_cast<IntrinsicInst>(Inst)) {
1039 switch (II->getIntrinsicID()) {
1040 case Intrinsic::prefetch:
1041 case Intrinsic::memset:
1042 AccessTy.AddrSpace = II->getArgOperand(0)->getType()->getPointerAddressSpace();
1043 AccessTy.MemTy = OperandVal->getType();
1044 break;
1045 case Intrinsic::memmove:
1046 case Intrinsic::memcpy:
1047 AccessTy.AddrSpace = OperandVal->getType()->getPointerAddressSpace();
1048 AccessTy.MemTy = OperandVal->getType();
1049 break;
1050 case Intrinsic::masked_load:
1051 AccessTy.AddrSpace =
1052 II->getArgOperand(0)->getType()->getPointerAddressSpace();
1053 break;
1054 case Intrinsic::masked_store:
1055 AccessTy.AddrSpace =
1056 II->getArgOperand(1)->getType()->getPointerAddressSpace();
1057 break;
1058 default: {
1059 MemIntrinsicInfo IntrInfo;
1060 if (TTI.getTgtMemIntrinsic(II, IntrInfo) && IntrInfo.PtrVal) {
1061 AccessTy.AddrSpace
1062 = IntrInfo.PtrVal->getType()->getPointerAddressSpace();
1063 }
1064
1065 break;
1066 }
1067 }
1068 }
1069
1070 return AccessTy;
1071}
1072
1073/// Return true if this AddRec is already a phi in its loop.
1074static bool isExistingPhi(const SCEVAddRecExpr *AR, ScalarEvolution &SE) {
1075 for (PHINode &PN : AR->getLoop()->getHeader()->phis()) {
1076 if (SE.isSCEVable(PN.getType()) &&
1077 (SE.getEffectiveSCEVType(PN.getType()) ==
1078 SE.getEffectiveSCEVType(AR->getType())) &&
1079 SE.getSCEV(&PN) == AR)
1080 return true;
1081 }
1082 return false;
1083}
1084
1085/// Check if expanding this expression is likely to incur significant cost. This
1086/// is tricky because SCEV doesn't track which expressions are actually computed
1087/// by the current IR.
1088///
1089/// We currently allow expansion of IV increments that involve adds,
1090/// multiplication by constants, and AddRecs from existing phis.
1091///
1092/// TODO: Allow UDivExpr if we can find an existing IV increment that is an
1093/// obvious multiple of the UDivExpr.
1094static bool isHighCostExpansion(const SCEV *S,
1096 ScalarEvolution &SE) {
1097 // Zero/One operand expressions
1098 switch (S->getSCEVType()) {
1099 case scUnknown:
1100 case scConstant:
1101 case scVScale:
1102 return false;
1103 case scTruncate:
1104 return isHighCostExpansion(cast<SCEVTruncateExpr>(S)->getOperand(),
1105 Processed, SE);
1106 case scZeroExtend:
1107 return isHighCostExpansion(cast<SCEVZeroExtendExpr>(S)->getOperand(),
1108 Processed, SE);
1109 case scSignExtend:
1110 return isHighCostExpansion(cast<SCEVSignExtendExpr>(S)->getOperand(),
1111 Processed, SE);
1112 default:
1113 break;
1114 }
1115
1116 if (!Processed.insert(S).second)
1117 return false;
1118
1119 if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
1120 for (const SCEV *S : Add->operands()) {
1121 if (isHighCostExpansion(S, Processed, SE))
1122 return true;
1123 }
1124 return false;
1125 }
1126
1127 const SCEV *Op0, *Op1;
1128 if (match(S, m_scev_Mul(m_SCEV(Op0), m_SCEV(Op1)))) {
1129 // Multiplication by a constant is ok
1130 if (isa<SCEVConstant>(Op0))
1131 return isHighCostExpansion(Op1, Processed, SE);
1132
1133 // If we have the value of one operand, check if an existing
1134 // multiplication already generates this expression.
1135 if (const auto *U = dyn_cast<SCEVUnknown>(Op1)) {
1136 Value *UVal = U->getValue();
1137 for (User *UR : UVal->users()) {
1138 // If U is a constant, it may be used by a ConstantExpr.
1140 if (UI && UI->getOpcode() == Instruction::Mul &&
1141 SE.isSCEVable(UI->getType())) {
1142 return SE.getSCEV(UI) == S;
1143 }
1144 }
1145 }
1146 }
1147
1148 if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(S)) {
1149 if (isExistingPhi(AR, SE))
1150 return false;
1151 }
1152
1153 // Fow now, consider any other type of expression (div/mul/min/max) high cost.
1154 return true;
1155}
1156
1157namespace {
1158
1159class LSRUse;
1160
1161} // end anonymous namespace
1162
1163/// Check if the addressing mode defined by \p F is completely
1164/// folded in \p LU at isel time.
1165/// This includes address-mode folding and special icmp tricks.
1166/// This function returns true if \p LU can accommodate what \p F
1167/// defines and up to 1 base + 1 scaled + offset.
1168/// In other words, if \p F has several base registers, this function may
1169/// still return true. Therefore, users still need to account for
1170/// additional base registers and/or unfolded offsets to derive an
1171/// accurate cost model.
1172static bool isAMCompletelyFolded(const TargetTransformInfo &TTI,
1173 const LSRUse &LU, const Formula &F);
1174
1175// Get the cost of the scaling factor used in F for LU.
1176static InstructionCost getScalingFactorCost(const TargetTransformInfo &TTI,
1177 const LSRUse &LU, const Formula &F,
1178 const Loop &L);
1179
1180namespace {
1181
1182/// This class is used to measure and compare candidate formulae.
1183class Cost {
1184 const ScalarOptions *Opts = nullptr;
1185 const Loop *L = nullptr;
1186 ScalarEvolution *SE = nullptr;
1187 const TargetTransformInfo *TTI = nullptr;
1188 TargetTransformInfo::LSRCost C;
1190
1191public:
1192 Cost() = delete;
1193 Cost(const ScalarOptions &Opts, const Loop *L, ScalarEvolution &SE,
1194 const TargetTransformInfo &TTI, TTI::AddressingModeKind AMK)
1195 : Opts(&Opts), L(L), SE(&SE), TTI(&TTI), AMK(AMK) {
1196 C.Insns = 0;
1197 C.NumRegs = 0;
1198 C.AddRecCost = 0;
1199 C.NumIVMuls = 0;
1200 C.NumBaseAdds = 0;
1201 C.ImmCost = 0;
1202 C.SetupCost = 0;
1203 C.ScaleCost = 0;
1204 }
1205
1206 bool isLess(const Cost &Other) const;
1207
1208 void Lose();
1209
1210#ifndef NDEBUG
1211 // Once any of the metrics loses, they must all remain losers.
1212 bool isValid() {
1213 return ((C.Insns | C.NumRegs | C.AddRecCost | C.NumIVMuls | C.NumBaseAdds
1214 | C.ImmCost | C.SetupCost | C.ScaleCost) != ~0u)
1215 || ((C.Insns & C.NumRegs & C.AddRecCost & C.NumIVMuls & C.NumBaseAdds
1216 & C.ImmCost & C.SetupCost & C.ScaleCost) == ~0u);
1217 }
1218#endif
1219
1220 bool isLoser() {
1221 assert(isValid() && "invalid cost");
1222 return C.NumRegs == ~0u;
1223 }
1224
1225 void RateFormula(const Formula &F, SmallPtrSetImpl<const SCEV *> &Regs,
1226 const DenseSet<const SCEV *> &VisitedRegs, const LSRUse &LU,
1227 bool HardwareLoopProfitable,
1228 SmallPtrSetImpl<const SCEV *> *LoserRegs = nullptr);
1229
1230 void print(raw_ostream &OS) const;
1231 void dump() const;
1232
1233private:
1234 void RateRegister(const Formula &F, const SCEV *Reg,
1235 SmallPtrSetImpl<const SCEV *> &Regs, const LSRUse &LU,
1236 bool HardwareLoopProfitable);
1237 void RatePrimaryRegister(const Formula &F, const SCEV *Reg,
1238 SmallPtrSetImpl<const SCEV *> &Regs,
1239 const LSRUse &LU, bool HardwareLoopProfitable,
1240 SmallPtrSetImpl<const SCEV *> *LoserRegs);
1241};
1242
1243/// An operand value in an instruction which is to be replaced with some
1244/// equivalent, possibly strength-reduced, replacement.
1245struct LSRFixup {
1246 /// The instruction which will be updated.
1247 Instruction *UserInst = nullptr;
1248
1249 /// The operand of the instruction which will be replaced. The operand may be
1250 /// used more than once; every instance will be replaced.
1251 Value *OperandValToReplace = nullptr;
1252
1253 /// If this user is to use the post-incremented value of an induction
1254 /// variable, this set is non-empty and holds the loops associated with the
1255 /// induction variable.
1256 PostIncLoopSet PostIncLoops;
1257
1258 /// A constant offset to be added to the LSRUse expression. This allows
1259 /// multiple fixups to share the same LSRUse with different offsets, for
1260 /// example in an unrolled loop.
1261 Immediate Offset = Immediate::getZero();
1262
1263 LSRFixup() = default;
1264
1265 bool isUseFullyOutsideLoop(const Loop *L) const;
1266
1267 void print(raw_ostream &OS) const;
1268 void dump() const;
1269};
1270
1271/// This class holds the state that LSR keeps for each use in IVUsers, as well
1272/// as uses invented by LSR itself. It includes information about what kinds of
1273/// things can be folded into the user, information about the user itself, and
1274/// information about how the use may be satisfied. TODO: Represent multiple
1275/// users of the same expression in common?
1276class LSRUse {
1277 DenseSet<SmallVector<const SCEV *, 4>> Uniquifier;
1278
1279public:
1280 /// An enum for a kind of use, indicating what types of scaled and immediate
1281 /// operands it might support.
1282 enum KindType {
1283 Basic, ///< A normal use, with no folding.
1284 Special, ///< A special case of basic, allowing -1 scales.
1285 Address, ///< An address use; folding according to TargetLowering
1286 ICmpZero ///< An equality icmp with both operands folded into one.
1287 // TODO: Add a generic icmp too?
1288 };
1289
1290 using SCEVUseKindPair = PointerIntPair<const SCEV *, 2, KindType>;
1291
1292 KindType Kind;
1293 MemAccessTy AccessTy;
1294
1295 /// The list of operands which are to be replaced.
1297
1298 /// Keep track of the min and max offsets of the fixups.
1299 Immediate MinOffset = Immediate::getFixedMax();
1300 Immediate MaxOffset = Immediate::getFixedMin();
1301
1302 /// This records whether all of the fixups using this LSRUse are outside of
1303 /// the loop, in which case some special-case heuristics may be used.
1304 bool AllFixupsOutsideLoop = true;
1305
1306 /// This records whether all of the fixups using this LSRUse are unconditional
1307 /// within the loop, meaning they will be executed on every path to the loop
1308 /// latch. This includes fixups before early exits.
1309 bool AllFixupsUnconditional = true;
1310
1311 /// RigidFormula is set to true to guarantee that this use will be associated
1312 /// with a single formula--the one that initially matched. Some SCEV
1313 /// expressions cannot be expanded. This allows LSR to consider the registers
1314 /// used by those expressions without the need to expand them later after
1315 /// changing the formula.
1316 bool RigidFormula = false;
1317
1318 /// A list of ways to build a value that can satisfy this user. After the
1319 /// list is populated, one of these is selected heuristically and used to
1320 /// formulate a replacement for OperandValToReplace in UserInst.
1321 SmallVector<Formula, 12> Formulae;
1322
1323 /// The set of register candidates used by all formulae in this LSRUse.
1324 SmallPtrSet<const SCEV *, 4> Regs;
1325
1326 LSRUse(KindType K, MemAccessTy AT) : Kind(K), AccessTy(AT) {}
1327
1328 LSRFixup &getNewFixup() {
1329 Fixups.push_back(LSRFixup());
1330 return Fixups.back();
1331 }
1332
1333 void pushFixup(LSRFixup &f) {
1334 Fixups.push_back(f);
1335 if (Immediate::isKnownGT(f.Offset, MaxOffset))
1336 MaxOffset = f.Offset;
1337 if (Immediate::isKnownLT(f.Offset, MinOffset))
1338 MinOffset = f.Offset;
1339 }
1340
1341 bool HasFormulaWithSameRegs(const Formula &F) const;
1342 float getNotSelectedProbability(const SCEV *Reg) const;
1343 bool InsertFormula(const Formula &F, const Loop &L);
1344 void DeleteFormula(Formula &F);
1345 void RecomputeRegs(size_t LUIdx, RegUseTracker &Reguses);
1346
1347 void print(raw_ostream &OS) const;
1348 void dump() const;
1349};
1350
1351} // end anonymous namespace
1352
1353static bool isAMCompletelyFolded(const TargetTransformInfo &TTI,
1354 LSRUse::KindType Kind, MemAccessTy AccessTy,
1355 GlobalValue *BaseGV, Immediate BaseOffset,
1356 bool HasBaseReg, int64_t Scale,
1357 Instruction *Fixup = nullptr);
1358
1359static unsigned getSetupCost(const SCEV *Reg, unsigned Depth,
1360 const TargetTransformInfo &TTI) {
1361 if (isa<SCEVUnknown>(Reg))
1362 return 1;
1363 if (const auto *C = dyn_cast<SCEVConstant>(Reg)) {
1364 if (TTI.getIntImmCost(C->getAPInt(), C->getType(),
1367 return 0;
1368 return 1;
1369 }
1370 if (Depth == 0)
1371 return 0;
1372 if (const auto *S = dyn_cast<SCEVAddRecExpr>(Reg))
1373 return getSetupCost(S->getStart(), Depth - 1, TTI);
1374 if (auto S = dyn_cast<SCEVIntegralCastExpr>(Reg))
1375 return getSetupCost(S->getOperand(), Depth - 1, TTI);
1376 if (auto S = dyn_cast<SCEVNAryExpr>(Reg))
1377 return std::accumulate(S->operands().begin(), S->operands().end(), 0,
1378 [&](unsigned i, const SCEV *Reg) {
1379 return i + getSetupCost(Reg, Depth - 1, TTI);
1380 });
1381 if (auto S = dyn_cast<SCEVUDivExpr>(Reg))
1382 return getSetupCost(S->getLHS(), Depth - 1, TTI) +
1383 getSetupCost(S->getRHS(), Depth - 1, TTI);
1384 return 0;
1385}
1386
1387/// Tally up interesting quantities from the given register.
1388void Cost::RateRegister(const Formula &F, const SCEV *Reg,
1389 SmallPtrSetImpl<const SCEV *> &Regs, const LSRUse &LU,
1390 bool HardwareLoopProfitable) {
1391 if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(Reg)) {
1392 // If this is an addrec for another loop, it should be an invariant
1393 // with respect to L since L is the innermost loop (at least
1394 // for now LSR only handles innermost loops).
1395 if (AR->getLoop() != L) {
1396 // If the AddRec exists, consider it's register free and leave it alone.
1397 if (isExistingPhi(AR, *SE) && !(AMK & TTI::AMK_PostIndexed))
1398 return;
1399
1400 // It is bad to allow LSR for current loop to add induction variables
1401 // for its sibling loops.
1402 if (!AR->getLoop()->contains(L)) {
1403 Lose();
1404 return;
1405 }
1406
1407 // Otherwise, it will be an invariant with respect to Loop L.
1408 ++C.NumRegs;
1409 return;
1410 }
1411
1412 unsigned LoopCost = 1;
1413 if (TTI->isIndexedLoadLegal(TTI->MIM_PostInc, AR->getType()) ||
1414 TTI->isIndexedStoreLegal(TTI->MIM_PostInc, AR->getType())) {
1415 const SCEV *Start;
1416 const APInt *Step;
1417 if (match(AR, m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(Step)))) {
1418 // If the step size matches the base offset, we could use pre-indexed
1419 // addressing.
1420 bool CanPreIndex = (AMK & TTI::AMK_PreIndexed) &&
1421 F.BaseOffset.isFixed() &&
1422 *Step == F.BaseOffset.getFixedValue();
1423 bool CanPostIndex = (AMK & TTI::AMK_PostIndexed) &&
1424 !isa<SCEVConstant>(Start) &&
1425 SE->isLoopInvariant(Start, L);
1426 // We can only pre or post index when the load/store is unconditional.
1427 if ((CanPreIndex || CanPostIndex) && LU.AllFixupsUnconditional)
1428 LoopCost = 0;
1429 }
1430 }
1431
1432 // If the loop counts down to zero and we'll be using a hardware loop then
1433 // the addrec will be combined into the hardware loop instruction.
1434 if (LU.Kind == LSRUse::ICmpZero && F.countsDownToZero() &&
1435 HardwareLoopProfitable)
1436 LoopCost = 0;
1437 C.AddRecCost += LoopCost;
1438
1439 // Add the step value register, if it needs one.
1440 // TODO: The non-affine case isn't precisely modeled here.
1441 const SCEV *StepReg = AR->getOperand(1);
1442 if (!AR->isAffine() || !isa<SCEVConstant>(StepReg)) {
1443 // If the step amount is a constant multiplied by vscale then it can form
1444 // the immediate value of an add and doesn't use a register, so long as
1445 // the immediate value is legal.
1446 auto IsVScaleStep = [](const SCEV *Reg, const TargetTransformInfo *TTI) {
1447 const APInt *X;
1449 return false;
1450 return TTI->isLegalAddScalableImmediate(X->getLimitedValue());
1451 };
1452 if (!Regs.count(StepReg) && !IsVScaleStep(StepReg, TTI)) {
1453 RateRegister(F, StepReg, Regs, LU, HardwareLoopProfitable);
1454 if (isLoser())
1455 return;
1456 }
1457 }
1458 }
1459 ++C.NumRegs;
1460
1461 // Rough heuristic; favor registers which don't require extra setup
1462 // instructions in the preheader.
1463 C.SetupCost += getSetupCost(Reg, Opts->lsr_setupcost_depth_limit, *TTI);
1464 // Ensure we don't, even with the recusion limit, produce invalid costs.
1465 C.SetupCost = std::min<unsigned>(C.SetupCost, 1 << 16);
1466
1467 C.NumIVMuls += isa<SCEVMulExpr>(Reg) &&
1469}
1470
1471/// Record this register in the set. If we haven't seen it before, rate
1472/// it. Optional LoserRegs provides a way to declare any formula that refers to
1473/// one of those regs an instant loser.
1474void Cost::RatePrimaryRegister(const Formula &F, const SCEV *Reg,
1475 SmallPtrSetImpl<const SCEV *> &Regs,
1476 const LSRUse &LU, bool HardwareLoopProfitable,
1477 SmallPtrSetImpl<const SCEV *> *LoserRegs) {
1478 if (LoserRegs && LoserRegs->count(Reg)) {
1479 Lose();
1480 return;
1481 }
1482 if (Regs.insert(Reg).second) {
1483 RateRegister(F, Reg, Regs, LU, HardwareLoopProfitable);
1484 if (LoserRegs && isLoser())
1485 LoserRegs->insert(Reg);
1486 }
1487}
1488
1489void Cost::RateFormula(const Formula &F, SmallPtrSetImpl<const SCEV *> &Regs,
1490 const DenseSet<const SCEV *> &VisitedRegs,
1491 const LSRUse &LU, bool HardwareLoopProfitable,
1492 SmallPtrSetImpl<const SCEV *> *LoserRegs) {
1493 if (isLoser())
1494 return;
1495 assert(F.isCanonical(*L) && "Cost is accurate only for canonical formula");
1496 // Tally up the registers.
1497 unsigned PrevAddRecCost = C.AddRecCost;
1498 unsigned PrevNumRegs = C.NumRegs;
1499 unsigned PrevNumBaseAdds = C.NumBaseAdds;
1500 if (const SCEV *ScaledReg = F.ScaledReg) {
1501 if (VisitedRegs.count(ScaledReg)) {
1502 Lose();
1503 return;
1504 }
1505 RatePrimaryRegister(F, ScaledReg, Regs, LU, HardwareLoopProfitable,
1506 LoserRegs);
1507 if (isLoser())
1508 return;
1509 }
1510 for (const SCEV *BaseReg : F.BaseRegs) {
1511 if (VisitedRegs.count(BaseReg)) {
1512 Lose();
1513 return;
1514 }
1515 RatePrimaryRegister(F, BaseReg, Regs, LU, HardwareLoopProfitable,
1516 LoserRegs);
1517 if (isLoser())
1518 return;
1519 }
1520
1521 // Determine how many (unfolded) adds we'll need inside the loop.
1522 size_t NumBaseParts = F.getNumRegs();
1523 if (NumBaseParts > 1)
1524 // Do not count the base and a possible second register if the target
1525 // allows to fold 2 registers.
1526 C.NumBaseAdds +=
1527 NumBaseParts - (1 + (F.Scale && isAMCompletelyFolded(*TTI, LU, F)));
1528 C.NumBaseAdds += (F.UnfoldedOffset.isNonZero());
1529
1530 // Accumulate non-free scaling amounts.
1531 C.ScaleCost += getScalingFactorCost(*TTI, LU, F, *L).getValue();
1532
1533 // Tally up the non-zero immediates.
1534 for (const LSRFixup &Fixup : LU.Fixups) {
1535 if (Fixup.Offset.isCompatibleImmediate(F.BaseOffset)) {
1536 Immediate Offset = Fixup.Offset.addUnsigned(F.BaseOffset);
1537 if (F.BaseGV)
1538 C.ImmCost += 64; // Handle symbolic values conservatively.
1539 // TODO: This should probably be the pointer size.
1540 else if (Offset.isNonZero())
1541 C.ImmCost +=
1542 APInt(64, Offset.getKnownMinValue(), true).getSignificantBits();
1543
1544 // Check with target if this offset with this instruction is
1545 // specifically not supported.
1546 if (LU.Kind == LSRUse::Address && Offset.isNonZero() &&
1547 !isAMCompletelyFolded(*TTI, LSRUse::Address, LU.AccessTy, F.BaseGV,
1548 Offset, F.HasBaseReg, F.Scale, Fixup.UserInst))
1549 C.NumBaseAdds++;
1550 } else {
1551 // Incompatible immediate type, increase cost to avoid using
1552 C.ImmCost += 2048;
1553 }
1554 }
1555
1556 // If we don't count instruction cost exit here.
1557 if (!valueOr(Opts->lsr_insns_cost, true)) {
1558 assert(isValid() && "invalid cost");
1559 return;
1560 }
1561
1562 // Treat every new register that exceeds TTI.getNumberOfRegisters() - 1 as
1563 // additional instruction (at least fill).
1564 // TODO: Need distinguish register class?
1565 unsigned TTIRegNum = TTI->getNumberOfRegisters(
1566 TTI->getRegisterClassForType(false, F.getType())) - 1;
1567 if (C.NumRegs > TTIRegNum) {
1568 // Cost already exceeded TTIRegNum, then only newly added register can add
1569 // new instructions.
1570 if (PrevNumRegs > TTIRegNum)
1571 C.Insns += (C.NumRegs - PrevNumRegs);
1572 else
1573 C.Insns += (C.NumRegs - TTIRegNum);
1574 }
1575
1576 // If ICmpZero formula ends with not 0, it could not be replaced by
1577 // just add or sub. We'll need to compare final result of AddRec.
1578 // That means we'll need an additional instruction. But if the target can
1579 // macro-fuse a compare with a branch, don't count this extra instruction.
1580 // For -10 + {0, +, 1}:
1581 // i = i + 1;
1582 // cmp i, 10
1583 //
1584 // For {-10, +, 1}:
1585 // i = i + 1;
1586 if (LU.Kind == LSRUse::ICmpZero && !F.hasZeroEnd() &&
1587 !TTI->canMacroFuseCmp())
1588 C.Insns++;
1589 // Each new AddRec adds 1 instruction to calculation.
1590 C.Insns += (C.AddRecCost - PrevAddRecCost);
1591
1592 // BaseAdds adds instructions for unfolded registers.
1593 if (LU.Kind != LSRUse::ICmpZero)
1594 C.Insns += C.NumBaseAdds - PrevNumBaseAdds;
1595 assert(isValid() && "invalid cost");
1596}
1597
1598/// Set this cost to a losing value.
1599void Cost::Lose() {
1600 C.Insns = std::numeric_limits<unsigned>::max();
1601 C.NumRegs = std::numeric_limits<unsigned>::max();
1602 C.AddRecCost = std::numeric_limits<unsigned>::max();
1603 C.NumIVMuls = std::numeric_limits<unsigned>::max();
1604 C.NumBaseAdds = std::numeric_limits<unsigned>::max();
1605 C.ImmCost = std::numeric_limits<unsigned>::max();
1606 C.SetupCost = std::numeric_limits<unsigned>::max();
1607 C.ScaleCost = std::numeric_limits<unsigned>::max();
1608}
1609
1610/// Choose the lower cost.
1611bool Cost::isLess(const Cost &Other) const {
1612 if (Opts->lsr_insns_cost == BoolOrDefault::True && C.Insns != Other.C.Insns)
1613 return C.Insns < Other.C.Insns;
1614 return TTI->isLSRCostLess(C, Other.C);
1615}
1616
1617#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1618void Cost::print(raw_ostream &OS) const {
1619 if (valueOr(Opts->lsr_insns_cost, true))
1620 OS << C.Insns << " instruction" << (C.Insns == 1 ? " " : "s ");
1621 OS << C.NumRegs << " reg" << (C.NumRegs == 1 ? "" : "s");
1622 if (C.AddRecCost != 0)
1623 OS << ", with addrec cost " << C.AddRecCost;
1624 if (C.NumIVMuls != 0)
1625 OS << ", plus " << C.NumIVMuls << " IV mul"
1626 << (C.NumIVMuls == 1 ? "" : "s");
1627 if (C.NumBaseAdds != 0)
1628 OS << ", plus " << C.NumBaseAdds << " base add"
1629 << (C.NumBaseAdds == 1 ? "" : "s");
1630 if (C.ScaleCost != 0)
1631 OS << ", plus " << C.ScaleCost << " scale cost";
1632 if (C.ImmCost != 0)
1633 OS << ", plus " << C.ImmCost << " imm cost";
1634 if (C.SetupCost != 0)
1635 OS << ", plus " << C.SetupCost << " setup cost";
1636}
1637
1638LLVM_DUMP_METHOD void Cost::dump() const {
1639 print(errs()); errs() << '\n';
1640}
1641#endif
1642
1643/// Test whether this fixup always uses its value outside of the given loop.
1644bool LSRFixup::isUseFullyOutsideLoop(const Loop *L) const {
1645 // PHI nodes use their value in their incoming blocks.
1646 if (const PHINode *PN = dyn_cast<PHINode>(UserInst)) {
1647 for (unsigned i = 0, e = PN->getNumIncomingValues(); i != e; ++i)
1648 if (PN->getIncomingValue(i) == OperandValToReplace &&
1649 L->contains(PN->getIncomingBlock(i)))
1650 return false;
1651 return true;
1652 }
1653
1654 return !L->contains(UserInst);
1655}
1656
1657#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1658void LSRFixup::print(raw_ostream &OS) const {
1659 OS << "UserInst=";
1660 // Store is common and interesting enough to be worth special-casing.
1661 if (StoreInst *Store = dyn_cast<StoreInst>(UserInst)) {
1662 OS << "store ";
1663 Store->getOperand(0)->printAsOperand(OS, /*PrintType=*/false);
1664 } else if (UserInst->getType()->isVoidTy())
1665 OS << UserInst->getOpcodeName();
1666 else
1667 UserInst->printAsOperand(OS, /*PrintType=*/false);
1668
1669 OS << ", OperandValToReplace=";
1670 OperandValToReplace->printAsOperand(OS, /*PrintType=*/false);
1671
1672 for (const Loop *PIL : PostIncLoops) {
1673 OS << ", PostIncLoop=";
1674 PIL->getHeader()->printAsOperand(OS, /*PrintType=*/false);
1675 }
1676
1677 if (Offset.isNonZero())
1678 OS << ", Offset=" << Offset;
1679}
1680
1681LLVM_DUMP_METHOD void LSRFixup::dump() const {
1682 print(errs()); errs() << '\n';
1683}
1684#endif
1685
1686/// Test whether this use as a formula which has the same registers as the given
1687/// formula.
1688bool LSRUse::HasFormulaWithSameRegs(const Formula &F) const {
1690 if (F.ScaledReg) Key.push_back(F.ScaledReg);
1691 // Unstable sort by host order ok, because this is only used for uniquifying.
1692 llvm::sort(Key);
1693 return Uniquifier.count(Key);
1694}
1695
1696/// The function returns a probability of selecting formula without Reg.
1697float LSRUse::getNotSelectedProbability(const SCEV *Reg) const {
1698 unsigned FNum = 0;
1699 for (const Formula &F : Formulae)
1700 if (F.referencesReg(Reg))
1701 FNum++;
1702 return ((float)(Formulae.size() - FNum)) / Formulae.size();
1703}
1704
1705/// If the given formula has not yet been inserted, add it to the list, and
1706/// return true. Return false otherwise. The formula must be in canonical form.
1707bool LSRUse::InsertFormula(const Formula &F, const Loop &L) {
1708 assert(F.isCanonical(L) && "Invalid canonical representation");
1709
1710 if (!Formulae.empty() && RigidFormula)
1711 return false;
1712
1714 if (F.ScaledReg) Key.push_back(F.ScaledReg);
1715 // Unstable sort by host order ok, because this is only used for uniquifying.
1716 llvm::sort(Key);
1717
1718 if (!Uniquifier.insert(Key).second)
1719 return false;
1720
1721 // Using a register to hold the value of 0 is not profitable.
1722 assert((!F.ScaledReg || !F.ScaledReg->isZero()) &&
1723 "Zero allocated in a scaled register!");
1724#ifndef NDEBUG
1725 for (const SCEV *BaseReg : F.BaseRegs)
1726 assert(!BaseReg->isZero() && "Zero allocated in a base register!");
1727#endif
1728
1729 // Add the formula to the list.
1730 Formulae.push_back(F);
1731
1732 // Record registers now being used by this use.
1733 Regs.insert_range(F.BaseRegs);
1734 if (F.ScaledReg)
1735 Regs.insert(F.ScaledReg);
1736
1737 return true;
1738}
1739
1740/// Remove the given formula from this use's list.
1741void LSRUse::DeleteFormula(Formula &F) {
1742 if (&F != &Formulae.back())
1743 std::swap(F, Formulae.back());
1744 Formulae.pop_back();
1745}
1746
1747/// Recompute the Regs field, and update RegUses.
1748void LSRUse::RecomputeRegs(size_t LUIdx, RegUseTracker &RegUses) {
1749 // Now that we've filtered out some formulae, recompute the Regs set.
1750 SmallPtrSet<const SCEV *, 4> OldRegs = std::move(Regs);
1751 Regs.clear();
1752 for (const Formula &F : Formulae) {
1753 if (F.ScaledReg) Regs.insert(F.ScaledReg);
1754 Regs.insert_range(F.BaseRegs);
1755 }
1756
1757 // Update the RegTracker.
1758 for (const SCEV *S : OldRegs)
1759 if (!Regs.count(S))
1760 RegUses.dropRegister(S, LUIdx);
1761}
1762
1763#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1764void LSRUse::print(raw_ostream &OS) const {
1765 OS << "LSR Use: Kind=";
1766 switch (Kind) {
1767 case Basic: OS << "Basic"; break;
1768 case Special: OS << "Special"; break;
1769 case ICmpZero: OS << "ICmpZero"; break;
1770 case Address:
1771 OS << "Address of ";
1772 if (AccessTy.MemTy->isPointerTy())
1773 OS << "pointer"; // the full pointer type could be really verbose
1774 else {
1775 OS << *AccessTy.MemTy;
1776 }
1777
1778 OS << " in addrspace(" << AccessTy.AddrSpace << ')';
1779 }
1780
1781 OS << ", Offsets={";
1782 bool NeedComma = false;
1783 for (const LSRFixup &Fixup : Fixups) {
1784 if (NeedComma) OS << ',';
1785 OS << Fixup.Offset;
1786 NeedComma = true;
1787 }
1788 OS << '}';
1789
1790 if (AllFixupsOutsideLoop)
1791 OS << ", all-fixups-outside-loop";
1792
1793 if (AllFixupsUnconditional)
1794 OS << ", all-fixups-unconditional";
1795}
1796
1797LLVM_DUMP_METHOD void LSRUse::dump() const {
1798 print(errs()); errs() << '\n';
1799}
1800#endif
1801
1803 LSRUse::KindType Kind, MemAccessTy AccessTy,
1804 GlobalValue *BaseGV, Immediate BaseOffset,
1805 bool HasBaseReg, int64_t Scale,
1806 Instruction *Fixup /* = nullptr */) {
1807 switch (Kind) {
1808 case LSRUse::Address: {
1809 int64_t FixedOffset =
1810 BaseOffset.isScalable() ? 0 : BaseOffset.getFixedValue();
1811 int64_t ScalableOffset =
1812 BaseOffset.isScalable() ? BaseOffset.getKnownMinValue() : 0;
1813 return TTI.isLegalAddressingMode(AccessTy.MemTy, BaseGV, FixedOffset,
1814 HasBaseReg, Scale, AccessTy.AddrSpace,
1815 Fixup, ScalableOffset);
1816 }
1817 case LSRUse::ICmpZero:
1818 // There's not even a target hook for querying whether it would be legal to
1819 // fold a GV into an ICmp.
1820 if (BaseGV)
1821 return false;
1822
1823 // ICmp only has two operands; don't allow more than two non-trivial parts.
1824 if (Scale != 0 && HasBaseReg && BaseOffset.isNonZero())
1825 return false;
1826
1827 // ICmp only supports no scale or a -1 scale, as we can "fold" a -1 scale by
1828 // putting the scaled register in the other operand of the icmp.
1829 if (Scale != 0 && Scale != -1)
1830 return false;
1831
1832 // If we have low-level target information, ask the target if it can fold an
1833 // integer immediate on an icmp.
1834 if (BaseOffset.isNonZero()) {
1835 // We don't have an interface to query whether the target supports
1836 // icmpzero against scalable quantities yet.
1837 if (BaseOffset.isScalable())
1838 return false;
1839
1840 // We have one of:
1841 // ICmpZero BaseReg + BaseOffset => ICmp BaseReg, -BaseOffset
1842 // ICmpZero -1*ScaleReg + BaseOffset => ICmp ScaleReg, BaseOffset
1843 // Offs is the ICmp immediate.
1844 if (Scale == 0)
1845 // The cast does the right thing with
1846 // std::numeric_limits<int64_t>::min().
1847 BaseOffset = BaseOffset.getFixed(-(uint64_t)BaseOffset.getFixedValue());
1848 return TTI.isLegalICmpImmediate(BaseOffset.getFixedValue());
1849 }
1850
1851 // ICmpZero BaseReg + -1*ScaleReg => ICmp BaseReg, ScaleReg
1852 return true;
1853
1854 case LSRUse::Basic:
1855 // Only handle single-register values.
1856 return !BaseGV && Scale == 0 && BaseOffset.isZero();
1857
1858 case LSRUse::Special:
1859 // Special case Basic to handle -1 scales.
1860 return !BaseGV && (Scale == 0 || Scale == -1) && BaseOffset.isZero();
1861 }
1862
1863 llvm_unreachable("Invalid LSRUse Kind!");
1864}
1865
1867 Immediate MinOffset, Immediate MaxOffset,
1868 LSRUse::KindType Kind, MemAccessTy AccessTy,
1869 GlobalValue *BaseGV, Immediate BaseOffset,
1870 bool HasBaseReg, int64_t Scale) {
1871 if (BaseOffset.isNonZero() &&
1872 (BaseOffset.isScalable() != MinOffset.isScalable() ||
1873 BaseOffset.isScalable() != MaxOffset.isScalable()))
1874 return false;
1875 // Check for overflow.
1876 int64_t Base = BaseOffset.getKnownMinValue();
1877 int64_t Min = MinOffset.getKnownMinValue();
1878 int64_t Max = MaxOffset.getKnownMinValue();
1879 if (((int64_t)((uint64_t)Base + Min) > Base) != (Min > 0))
1880 return false;
1881 MinOffset = Immediate::get((uint64_t)Base + Min, MinOffset.isScalable());
1882 if (((int64_t)((uint64_t)Base + Max) > Base) != (Max > 0))
1883 return false;
1884 MaxOffset = Immediate::get((uint64_t)Base + Max, MaxOffset.isScalable());
1885
1886 return isAMCompletelyFolded(TTI, Kind, AccessTy, BaseGV, MinOffset,
1887 HasBaseReg, Scale) &&
1888 isAMCompletelyFolded(TTI, Kind, AccessTy, BaseGV, MaxOffset,
1889 HasBaseReg, Scale);
1890}
1891
1893 Immediate MinOffset, Immediate MaxOffset,
1894 LSRUse::KindType Kind, MemAccessTy AccessTy,
1895 const Formula &F, const Loop &L) {
1896 // For the purpose of isAMCompletelyFolded either having a canonical formula
1897 // or a scale not equal to zero is correct.
1898 // Problems may arise from non canonical formulae having a scale == 0.
1899 // Strictly speaking it would best to just rely on canonical formulae.
1900 // However, when we generate the scaled formulae, we first check that the
1901 // scaling factor is profitable before computing the actual ScaledReg for
1902 // compile time sake.
1903 assert((F.isCanonical(L) || F.Scale != 0));
1904 return isAMCompletelyFolded(TTI, MinOffset, MaxOffset, Kind, AccessTy,
1905 F.BaseGV, F.BaseOffset, F.HasBaseReg, F.Scale);
1906}
1907
1908/// Test whether we know how to expand the current formula.
1909static bool isLegalUse(const TargetTransformInfo &TTI, Immediate MinOffset,
1910 Immediate MaxOffset, LSRUse::KindType Kind,
1911 MemAccessTy AccessTy, GlobalValue *BaseGV,
1912 Immediate BaseOffset, bool HasBaseReg, int64_t Scale) {
1913 // We know how to expand completely foldable formulae.
1914 return isAMCompletelyFolded(TTI, MinOffset, MaxOffset, Kind, AccessTy, BaseGV,
1915 BaseOffset, HasBaseReg, Scale) ||
1916 // Or formulae that use a base register produced by a sum of base
1917 // registers.
1918 (Scale == 1 &&
1919 isAMCompletelyFolded(TTI, MinOffset, MaxOffset, Kind, AccessTy,
1920 BaseGV, BaseOffset, true, 0));
1921}
1922
1923static bool isLegalUse(const TargetTransformInfo &TTI, Immediate MinOffset,
1924 Immediate MaxOffset, LSRUse::KindType Kind,
1925 MemAccessTy AccessTy, const Formula &F) {
1926 return isLegalUse(TTI, MinOffset, MaxOffset, Kind, AccessTy, F.BaseGV,
1927 F.BaseOffset, F.HasBaseReg, F.Scale);
1928}
1929
1931 Immediate Offset) {
1932 if (Offset.isScalable())
1933 return TTI.isLegalAddScalableImmediate(Offset.getKnownMinValue());
1934
1935 return TTI.isLegalAddImmediate(Offset.getFixedValue());
1936}
1937
1939 const LSRUse &LU, const Formula &F) {
1940 // Target may want to look at the user instructions.
1941 if (LU.Kind == LSRUse::Address && TTI.LSRWithInstrQueries()) {
1942 for (const LSRFixup &Fixup : LU.Fixups)
1943 if (!isAMCompletelyFolded(TTI, LSRUse::Address, LU.AccessTy, F.BaseGV,
1944 (F.BaseOffset + Fixup.Offset), F.HasBaseReg,
1945 F.Scale, Fixup.UserInst))
1946 return false;
1947 return true;
1948 }
1949
1950 return isAMCompletelyFolded(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind,
1951 LU.AccessTy, F.BaseGV, F.BaseOffset, F.HasBaseReg,
1952 F.Scale);
1953}
1954
1956 const LSRUse &LU, const Formula &F,
1957 const Loop &L) {
1958 if (!F.Scale)
1959 return 0;
1960
1961 // If the use is not completely folded in that instruction, we will have to
1962 // pay an extra cost only for scale != 1.
1963 if (!isAMCompletelyFolded(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind,
1964 LU.AccessTy, F, L))
1965 return F.Scale != 1;
1966
1967 switch (LU.Kind) {
1968 case LSRUse::Address: {
1969 // Check the scaling factor cost with both the min and max offsets.
1970 int64_t ScalableMin = 0, ScalableMax = 0, FixedMin = 0, FixedMax = 0;
1971 if (F.BaseOffset.isScalable()) {
1972 ScalableMin = (F.BaseOffset + LU.MinOffset).getKnownMinValue();
1973 ScalableMax = (F.BaseOffset + LU.MaxOffset).getKnownMinValue();
1974 } else {
1975 FixedMin = (F.BaseOffset + LU.MinOffset).getFixedValue();
1976 FixedMax = (F.BaseOffset + LU.MaxOffset).getFixedValue();
1977 }
1978 InstructionCost ScaleCostMinOffset = TTI.getScalingFactorCost(
1979 LU.AccessTy.MemTy, F.BaseGV, StackOffset::get(FixedMin, ScalableMin),
1980 F.HasBaseReg, F.Scale, LU.AccessTy.AddrSpace);
1981 InstructionCost ScaleCostMaxOffset = TTI.getScalingFactorCost(
1982 LU.AccessTy.MemTy, F.BaseGV, StackOffset::get(FixedMax, ScalableMax),
1983 F.HasBaseReg, F.Scale, LU.AccessTy.AddrSpace);
1984
1985 assert(ScaleCostMinOffset.isValid() && ScaleCostMaxOffset.isValid() &&
1986 "Legal addressing mode has an illegal cost!");
1987 return std::max(ScaleCostMinOffset, ScaleCostMaxOffset);
1988 }
1989 case LSRUse::ICmpZero:
1990 case LSRUse::Basic:
1991 case LSRUse::Special:
1992 // The use is completely folded, i.e., everything is folded into the
1993 // instruction.
1994 return 0;
1995 }
1996
1997 llvm_unreachable("Invalid LSRUse Kind!");
1998}
1999
2000static bool isAlwaysFoldable(const ScalarOptions &Opts,
2001 const TargetTransformInfo &TTI,
2002 LSRUse::KindType Kind, MemAccessTy AccessTy,
2003 GlobalValue *BaseGV, Immediate BaseOffset,
2004 bool HasBaseReg) {
2005 // Fast-path: zero is always foldable.
2006 if (BaseOffset.isZero() && !BaseGV)
2007 return true;
2008
2009 // Conservatively, create an address with an immediate and a
2010 // base and a scale.
2011 int64_t Scale = Kind == LSRUse::ICmpZero ? -1 : 1;
2012
2013 // Canonicalize a scale of 1 to a base register if the formula doesn't
2014 // already have a base register.
2015 if (!HasBaseReg && Scale == 1) {
2016 Scale = 0;
2017 HasBaseReg = true;
2018 }
2019
2020 // FIXME: Try with + without a scale? Maybe based on TTI?
2021 // I think basereg + scaledreg + immediateoffset isn't a good 'conservative'
2022 // default for many architectures, not just AArch64 SVE. More investigation
2023 // needed later to determine if this should be used more widely than just
2024 // on scalable types.
2025 if (HasBaseReg && BaseOffset.isNonZero() && Kind != LSRUse::ICmpZero &&
2026 AccessTy.MemTy && AccessTy.MemTy->isScalableTy() &&
2027 Opts.lsr_drop_scaled_reg_for_vscale)
2028 Scale = 0;
2029
2030 return isAMCompletelyFolded(TTI, Kind, AccessTy, BaseGV, BaseOffset,
2031 HasBaseReg, Scale);
2032}
2033
2034static bool isAlwaysFoldable(const ScalarOptions &Opts,
2035 const TargetTransformInfo &TTI,
2036 ScalarEvolution &SE, Immediate MinOffset,
2037 Immediate MaxOffset, LSRUse::KindType Kind,
2038 MemAccessTy AccessTy, const SCEV *S,
2039 bool HasBaseReg) {
2040 // Fast-path: zero is always foldable.
2041 if (S->isZero()) return true;
2042
2043 // Conservatively, create an address with an immediate and a
2044 // base and a scale.
2045 SCEVUse SCopy = S;
2046 Immediate BaseOffset = extractImmediate(Opts, SCopy, SE);
2047 GlobalValue *BaseGV = ExtractSymbol(SCopy, SE);
2048
2049 // If there's anything else involved, it's not foldable.
2050 if (!SCopy->isZero())
2051 return false;
2052
2053 // Fast-path: zero is always foldable.
2054 if (BaseOffset.isZero() && !BaseGV)
2055 return true;
2056
2057 if (BaseOffset.isScalable())
2058 return false;
2059
2060 // Conservatively, create an address with an immediate and a
2061 // base and a scale.
2062 int64_t Scale = Kind == LSRUse::ICmpZero ? -1 : 1;
2063
2064 return isAMCompletelyFolded(TTI, MinOffset, MaxOffset, Kind, AccessTy, BaseGV,
2065 BaseOffset, HasBaseReg, Scale);
2066}
2067
2068namespace {
2069
2070/// An individual increment in a Chain of IV increments. Relate an IV user to
2071/// an expression that computes the IV it uses from the IV used by the previous
2072/// link in the Chain.
2073///
2074/// For the head of a chain, IncExpr holds the absolute SCEV expression for the
2075/// original IVOperand. The head of the chain's IVOperand is only valid during
2076/// chain collection, before LSR replaces IV users. During chain generation,
2077/// IncExpr can be used to find the new IVOperand that computes the same
2078/// expression.
2079struct IVInc {
2080 Instruction *UserInst;
2081 Value* IVOperand;
2082 const SCEV *IncExpr;
2083
2084 IVInc(Instruction *U, Value *O, const SCEV *E)
2085 : UserInst(U), IVOperand(O), IncExpr(E) {}
2086};
2087
2088// The list of IV increments in program order. We typically add the head of a
2089// chain without finding subsequent links.
2090struct IVChain {
2092 const SCEV *ExprBase = nullptr;
2093
2094 IVChain() = default;
2095 IVChain(const IVInc &Head, const SCEV *Base)
2096 : Incs(1, Head), ExprBase(Base) {}
2097
2098 using const_iterator = SmallVectorImpl<IVInc>::const_iterator;
2099
2100 // Return the first increment in the chain.
2101 const_iterator begin() const {
2102 assert(!Incs.empty());
2103 return std::next(Incs.begin());
2104 }
2105 const_iterator end() const {
2106 return Incs.end();
2107 }
2108
2109 // Returns true if this chain contains any increments.
2110 bool hasIncs() const { return Incs.size() >= 2; }
2111
2112 // Add an IVInc to the end of this chain.
2113 void add(const IVInc &X) { Incs.push_back(X); }
2114
2115 // Returns the last UserInst in the chain.
2116 Instruction *tailUserInst() const { return Incs.back().UserInst; }
2117
2118 // Returns true if IncExpr can be profitably added to this chain.
2119 bool isProfitableIncrement(const SCEV *OperExpr,
2120 const SCEV *IncExpr,
2121 ScalarEvolution&);
2122};
2123
2124/// Helper for CollectChains to track multiple IV increment uses. Distinguish
2125/// between FarUsers that definitely cross IV increments and NearUsers that may
2126/// be used between IV increments.
2127struct ChainUsers {
2128 SmallPtrSet<Instruction*, 4> FarUsers;
2129 SmallPtrSet<Instruction*, 4> NearUsers;
2130};
2131
2132/// This class holds state for the main loop strength reduction logic.
2133class LSRInstance {
2134 const ScalarOptions &Opts;
2135 IVUsers &IU;
2136 ScalarEvolution &SE;
2137 DominatorTree &DT;
2138 LoopInfo &LI;
2139 AssumptionCache &AC;
2140 TargetLibraryInfo &TLI;
2141 const TargetTransformInfo &TTI;
2142 Loop *const L;
2143 MemorySSAUpdater *MSSAU;
2145 mutable SCEVExpander Rewriter;
2146 bool Changed = false;
2147 bool HardwareLoopProfitable = false;
2148 bool ShouldPreserveLCSSA = false;
2149
2150 /// This is the insert position that the current loop's induction variable
2151 /// increment should be placed. In simple loops, this is the latch block's
2152 /// terminator. But in more complicated cases, this is a position which will
2153 /// dominate all the in-loop post-increment users.
2154 Instruction *IVIncInsertPos = nullptr;
2155
2156 /// Interesting factors between use strides.
2157 ///
2158 /// We explicitly use a SetVector which contains a SmallSet, instead of the
2159 /// default, a SmallDenseSet, because we need to use the full range of
2160 /// int64_ts, and there's currently no good way of doing that with
2161 /// SmallDenseSet.
2162 SetVector<int64_t, SmallVector<int64_t, 8>, SmallSet<int64_t, 8>> Factors;
2163
2164 /// The cost of the current SCEV, the best solution by LSR will be dropped if
2165 /// the solution is not profitable.
2166 Cost BaselineCost;
2167
2168 /// Interesting use types, to facilitate truncation reuse.
2169 SmallSetVector<Type *, 4> Types;
2170
2171 /// The list of interesting uses.
2173
2174 /// Track which uses use which register candidates.
2175 RegUseTracker RegUses;
2176
2177 // Limit the number of chains to avoid quadratic behavior. We don't expect to
2178 // have more than a few IV increment chains in a loop. Missing a Chain falls
2179 // back to normal LSR behavior for those uses.
2180 static const unsigned MaxChains = 8;
2181
2182 /// IV users can form a chain of IV increments.
2184
2185 /// IV users that belong to profitable IVChains.
2186 SmallPtrSet<Use*, MaxChains> IVIncSet;
2187
2188 /// Induction variables that were generated and inserted by the SCEV Expander.
2189 SmallVector<llvm::WeakVH, 2> ScalarEvolutionIVs;
2190
2191 // Inserting instructions in the loop and using them as PHI's input could
2192 // break LCSSA in case if PHI's parent block is not a loop exit (i.e. the
2193 // corresponding incoming block is not loop exiting). So collect all such
2194 // instructions to form LCSSA for them later.
2195 SmallSetVector<Instruction *, 4> InsertedNonLCSSAInsts;
2196
2197 void OptimizeShadowIV();
2198 bool FindIVUserForCond(Instruction *Cond, IVStrideUse *&CondUse);
2199 Instruction *OptimizeMax(ICmpInst *Cond, IVStrideUse *&CondUse);
2200 void OptimizeLoopTermCond();
2201
2202 void ChainInstruction(Instruction *UserInst, Instruction *IVOper,
2203 SmallVectorImpl<ChainUsers> &ChainUsersVec);
2204 void FinalizeChain(IVChain &Chain);
2205 void CollectChains();
2206 void GenerateIVChain(const IVChain &Chain,
2207 SmallVectorImpl<WeakTrackingVH> &DeadInsts);
2208
2209 void CollectInterestingTypesAndFactors();
2210 void CollectFixupsAndInitialFormulae();
2211
2212 // Support for sharing of LSRUses between LSRFixups.
2213 using UseMapTy = DenseMap<LSRUse::SCEVUseKindPair, size_t>;
2214 UseMapTy UseMap;
2215
2216 bool reconcileNewOffset(LSRUse &LU, Immediate NewOffset, bool HasBaseReg,
2217 LSRUse::KindType Kind, MemAccessTy AccessTy);
2218
2219 std::pair<size_t, Immediate> getUse(const SCEV *&Expr, LSRUse::KindType Kind,
2220 MemAccessTy AccessTy);
2221
2222 void DeleteUse(LSRUse &LU, size_t LUIdx);
2223
2224 LSRUse *FindUseWithSimilarFormula(const Formula &F, const LSRUse &OrigLU);
2225
2226 void InsertInitialFormula(const SCEV *S, LSRUse &LU, size_t LUIdx);
2227 void InsertSupplementalFormula(const SCEV *S, LSRUse &LU, size_t LUIdx);
2228 void CountRegisters(const Formula &F, size_t LUIdx);
2229 bool InsertFormula(LSRUse &LU, unsigned LUIdx, const Formula &F);
2230 bool IsFixupExecutedEachIncrement(const LSRFixup &LF) const;
2231
2232 void CollectLoopInvariantFixupsAndFormulae();
2233
2234 void GenerateReassociations(LSRUse &LU, unsigned LUIdx, Formula Base,
2235 unsigned Depth = 0);
2236
2237 void GenerateReassociationsImpl(LSRUse &LU, unsigned LUIdx,
2238 const Formula &Base, unsigned Depth,
2239 size_t Idx, bool IsScaledReg = false);
2240 void GenerateCombinations(LSRUse &LU, unsigned LUIdx, Formula Base);
2241 void GenerateSymbolicOffsetsImpl(LSRUse &LU, unsigned LUIdx,
2242 const Formula &Base, size_t Idx,
2243 bool IsScaledReg = false);
2244 void GenerateSymbolicOffsets(LSRUse &LU, unsigned LUIdx, Formula Base);
2245 void GenerateConstantOffsetsImpl(LSRUse &LU, unsigned LUIdx,
2246 const Formula &Base,
2247 const SmallVectorImpl<Immediate> &Worklist,
2248 size_t Idx, bool IsScaledReg = false);
2249 void GenerateConstantOffsets(LSRUse &LU, unsigned LUIdx, Formula Base);
2250 void GenerateICmpZeroScales(LSRUse &LU, unsigned LUIdx, Formula Base);
2251 void GenerateScales(LSRUse &LU, unsigned LUIdx, Formula Base);
2252 void GenerateTruncates(LSRUse &LU, unsigned LUIdx, Formula Base);
2253 void GenerateCrossUseConstantOffsets();
2254 void GenerateAllReuseFormulae();
2255
2256 void FilterOutUndesirableDedicatedRegisters();
2257
2258 size_t EstimateSearchSpaceComplexity() const;
2259 void NarrowSearchSpaceByDetectingSupersets();
2260 void NarrowSearchSpaceByCollapsingUnrolledCode();
2261 void NarrowSearchSpaceByRefilteringUndesirableDedicatedRegisters();
2262 void NarrowSearchSpaceByFilterFormulaWithSameScaledReg();
2263 void NarrowSearchSpaceByFilterPostInc();
2264 void NarrowSearchSpaceByMergingUsesOutsideLoop();
2265 void NarrowSearchSpaceByDeletingCostlyFormulas();
2266 void NarrowSearchSpaceByPickingWinnerRegs();
2267 void NarrowSearchSpaceUsingHeuristics();
2268
2269 void SolveRecurse(SmallVectorImpl<const Formula *> &Solution,
2270 Cost &SolutionCost,
2271 SmallVectorImpl<const Formula *> &Workspace,
2272 const Cost &CurCost,
2273 const SmallPtrSet<const SCEV *, 16> &CurRegs,
2274 DenseSet<const SCEV *> &VisitedRegs) const;
2275 void Solve(SmallVectorImpl<const Formula *> &Solution) const;
2276
2278 HoistInsertPosition(BasicBlock::iterator IP,
2279 const SmallVectorImpl<Instruction *> &Inputs) const;
2280 BasicBlock::iterator AdjustInsertPositionForExpand(BasicBlock::iterator IP,
2281 const LSRFixup &LF,
2282 const LSRUse &LU) const;
2283
2284 Value *Expand(const LSRUse &LU, const LSRFixup &LF, const Formula &F,
2286 SmallVectorImpl<WeakTrackingVH> &DeadInsts) const;
2287 void RewriteForPHI(PHINode *PN, const LSRUse &LU, const LSRFixup &LF,
2288 const Formula &F,
2289 SmallVectorImpl<WeakTrackingVH> &DeadInsts);
2290 void Rewrite(const LSRUse &LU, const LSRFixup &LF, const Formula &F,
2291 SmallVectorImpl<WeakTrackingVH> &DeadInsts);
2292 void ImplementSolution(const SmallVectorImpl<const Formula *> &Solution);
2293
2294public:
2295 // TODO(boomanaiden154): The PreserveLCSSA flag is a hack to allow
2296 // experimentation with the NewPM which requires LCSSA preservation while
2297 // some of the details are worked out in LSR. Eventually it should be set
2298 // to true and removed.
2299 LSRInstance(const ScalarOptions &Opts, Loop *L, IVUsers &IU,
2300 ScalarEvolution &SE, DominatorTree &DT, LoopInfo &LI,
2301 const TargetTransformInfo &TTI, AssumptionCache &AC,
2302 TargetLibraryInfo &TLI, MemorySSAUpdater *MSSAU,
2303 bool PreserveLCSSA);
2304
2305 bool getChanged() const { return Changed; }
2306 const SmallVectorImpl<WeakVH> &getScalarEvolutionIVs() const {
2307 return ScalarEvolutionIVs;
2308 }
2309
2310 void print_factors_and_types(raw_ostream &OS) const;
2311 void print_fixups(raw_ostream &OS) const;
2312 void print_uses(raw_ostream &OS) const;
2313 void print(raw_ostream &OS) const;
2314 void dump() const;
2315};
2316
2317} // end anonymous namespace
2318
2319/// If IV is used in a int-to-float cast inside the loop then try to eliminate
2320/// the cast operation.
2321void LSRInstance::OptimizeShadowIV() {
2322 const SCEV *BackedgeTakenCount = SE.getBackedgeTakenCount(L);
2323 if (isa<SCEVCouldNotCompute>(BackedgeTakenCount))
2324 return;
2325
2326 for (IVUsers::const_iterator UI = IU.begin(), E = IU.end();
2327 UI != E; /* empty */) {
2328 IVUsers::const_iterator CandidateUI = UI;
2329 ++UI;
2330 Instruction *ShadowUse = CandidateUI->getUser();
2331 Type *DestTy = nullptr;
2332 bool IsSigned = false;
2333
2334 /* If shadow use is a int->float cast then insert a second IV
2335 to eliminate this cast.
2336
2337 for (unsigned i = 0; i < n; ++i)
2338 foo((double)i);
2339
2340 is transformed into
2341
2342 double d = 0.0;
2343 for (unsigned i = 0; i < n; ++i, ++d)
2344 foo(d);
2345 */
2346 if (UIToFPInst *UCast = dyn_cast<UIToFPInst>(CandidateUI->getUser())) {
2347 IsSigned = false;
2348 DestTy = UCast->getDestTy();
2349 }
2350 else if (SIToFPInst *SCast = dyn_cast<SIToFPInst>(CandidateUI->getUser())) {
2351 IsSigned = true;
2352 DestTy = SCast->getDestTy();
2353 }
2354 if (!DestTy) continue;
2355
2356 // If target does not support DestTy natively then do not apply
2357 // this transformation.
2358 if (!TTI.isTypeLegal(DestTy)) continue;
2359
2360 PHINode *PH = dyn_cast<PHINode>(ShadowUse->getOperand(0));
2361 if (!PH) continue;
2362 if (PH->getNumIncomingValues() != 2) continue;
2363
2364 // If the calculation in integers overflows, the result in FP type will
2365 // differ. So we only can do this transformation if we are guaranteed to not
2366 // deal with overflowing values
2367 const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(PH));
2368 if (!AR) continue;
2369 if (IsSigned && !AR->hasNoSignedWrap()) continue;
2370 if (!IsSigned && !AR->hasNoUnsignedWrap()) continue;
2371
2372 Type *SrcTy = PH->getType();
2373 int Mantissa = DestTy->getFPMantissaWidth();
2374 if (Mantissa == -1) continue;
2375 if ((int)SE.getTypeSizeInBits(SrcTy) > Mantissa)
2376 continue;
2377
2378 unsigned Entry, Latch;
2379 if (PH->getIncomingBlock(0) == L->getLoopPreheader()) {
2380 Entry = 0;
2381 Latch = 1;
2382 } else {
2383 Entry = 1;
2384 Latch = 0;
2385 }
2386
2387 ConstantInt *Init = dyn_cast<ConstantInt>(PH->getIncomingValue(Entry));
2388 if (!Init) continue;
2389 Constant *NewInit = ConstantFP::get(DestTy, IsSigned ?
2390 (double)Init->getSExtValue() :
2391 (double)Init->getZExtValue());
2392
2393 BinaryOperator *Incr =
2395 if (!Incr) continue;
2396 if (Incr->getOpcode() != Instruction::Add
2397 && Incr->getOpcode() != Instruction::Sub)
2398 continue;
2399
2400 /* Initialize new IV, double d = 0.0 in above example. */
2401 ConstantInt *C = nullptr;
2402 if (Incr->getOperand(0) == PH)
2404 else if (Incr->getOperand(1) == PH)
2406 else
2407 continue;
2408
2409 if (!C) continue;
2410
2411 // Ignore negative constants, as the code below doesn't handle them
2412 // correctly. TODO: Remove this restriction.
2413 if (!C->getValue().isStrictlyPositive())
2414 continue;
2415
2416 /* Add new PHINode. */
2417 PHINode *NewPH = PHINode::Create(DestTy, 2, "IV.S.", PH->getIterator());
2418 NewPH->setDebugLoc(PH->getDebugLoc());
2419
2420 /* create new increment. '++d' in above example. */
2421 Constant *CFP = ConstantFP::get(DestTy, C->getZExtValue());
2422 BinaryOperator *NewIncr = BinaryOperator::Create(
2423 Incr->getOpcode() == Instruction::Add ? Instruction::FAdd
2424 : Instruction::FSub,
2425 NewPH, CFP, "IV.S.next.", Incr->getIterator());
2426 NewIncr->setDebugLoc(Incr->getDebugLoc());
2427
2428 NewPH->addIncoming(NewInit, PH->getIncomingBlock(Entry));
2429 NewPH->addIncoming(NewIncr, PH->getIncomingBlock(Latch));
2430
2431 /* Remove cast operation */
2432 ShadowUse->replaceAllUsesWith(NewPH);
2433 ShadowUse->eraseFromParent();
2434 Changed = true;
2435 break;
2436 }
2437}
2438
2439/// If Cond has an operand that is an expression of an IV, set the IV user and
2440/// stride information and return true, otherwise return false.
2441bool LSRInstance::FindIVUserForCond(Instruction *Cond, IVStrideUse *&CondUse) {
2442 for (IVStrideUse &U : IU)
2443 if (U.getUser() == Cond) {
2444 // NOTE: we could handle setcc instructions with multiple uses here, but
2445 // InstCombine does it as well for simple uses, it's not clear that it
2446 // occurs enough in real life to handle.
2447 CondUse = &U;
2448 return true;
2449 }
2450 return false;
2451}
2452
2453/// Rewrite the loop's terminating condition if it uses a max computation.
2454///
2455/// This is a narrow solution to a specific, but acute, problem. For loops
2456/// like this:
2457///
2458/// i = 0;
2459/// do {
2460/// p[i] = 0.0;
2461/// } while (++i < n);
2462///
2463/// the trip count isn't just 'n', because 'n' might not be positive. And
2464/// unfortunately this can come up even for loops where the user didn't use
2465/// a C do-while loop. For example, seemingly well-behaved top-test loops
2466/// will commonly be lowered like this:
2467///
2468/// if (n > 0) {
2469/// i = 0;
2470/// do {
2471/// p[i] = 0.0;
2472/// } while (++i < n);
2473/// }
2474///
2475/// and then it's possible for subsequent optimization to obscure the if
2476/// test in such a way that indvars can't find it.
2477///
2478/// When indvars can't find the if test in loops like this, it creates a
2479/// max expression, which allows it to give the loop a canonical
2480/// induction variable:
2481///
2482/// i = 0;
2483/// max = n < 1 ? 1 : n;
2484/// do {
2485/// p[i] = 0.0;
2486/// } while (++i != max);
2487///
2488/// Canonical induction variables are necessary because the loop passes
2489/// are designed around them. The most obvious example of this is the
2490/// LoopInfo analysis, which doesn't remember trip count values. It
2491/// expects to be able to rediscover the trip count each time it is
2492/// needed, and it does this using a simple analysis that only succeeds if
2493/// the loop has a canonical induction variable.
2494///
2495/// However, when it comes time to generate code, the maximum operation
2496/// can be quite costly, especially if it's inside of an outer loop.
2497///
2498/// This function solves this problem by detecting this type of loop and
2499/// rewriting their conditions from ICMP_NE back to ICMP_SLT, and deleting
2500/// the instructions for the maximum computation.
2501Instruction *LSRInstance::OptimizeMax(ICmpInst *Cond, IVStrideUse *&CondUse) {
2502 // Check that the loop matches the pattern we're looking for.
2503 if (Cond->getPredicate() != CmpInst::ICMP_EQ &&
2504 Cond->getPredicate() != CmpInst::ICMP_NE)
2505 return Cond;
2506
2507 SelectInst *Sel = dyn_cast<SelectInst>(Cond->getOperand(1));
2508 if (!Sel || !Sel->hasOneUse()) return Cond;
2509
2510 const SCEV *BackedgeTakenCount = SE.getBackedgeTakenCount(L);
2511 if (isa<SCEVCouldNotCompute>(BackedgeTakenCount))
2512 return Cond;
2513 const SCEV *One = SE.getConstant(BackedgeTakenCount->getType(), 1);
2514
2515 // Add one to the backedge-taken count to get the trip count.
2516 const SCEV *IterationCount = SE.getAddExpr(One, BackedgeTakenCount);
2517 if (IterationCount != SE.getSCEV(Sel)) return Cond;
2518
2519 // Check for a max calculation that matches the pattern. There's no check
2520 // for ICMP_ULE here because the comparison would be with zero, which
2521 // isn't interesting.
2522 CmpInst::Predicate Pred = ICmpInst::BAD_ICMP_PREDICATE;
2523 const SCEVNAryExpr *Max = nullptr;
2524 if (const SCEVSMaxExpr *S = dyn_cast<SCEVSMaxExpr>(BackedgeTakenCount)) {
2525 Pred = ICmpInst::ICMP_SLE;
2526 Max = S;
2527 } else if (const SCEVSMaxExpr *S = dyn_cast<SCEVSMaxExpr>(IterationCount)) {
2528 Pred = ICmpInst::ICMP_SLT;
2529 Max = S;
2530 } else if (const SCEVUMaxExpr *U = dyn_cast<SCEVUMaxExpr>(IterationCount)) {
2531 Pred = ICmpInst::ICMP_ULT;
2532 Max = U;
2533 } else {
2534 // No match; bail.
2535 return Cond;
2536 }
2537
2538 // To handle a max with more than two operands, this optimization would
2539 // require additional checking and setup.
2540 if (Max->getNumOperands() != 2)
2541 return Cond;
2542
2543 const SCEV *MaxLHS = Max->getOperand(0);
2544 const SCEV *MaxRHS = Max->getOperand(1);
2545
2546 // ScalarEvolution canonicalizes constants to the left. For < and >, look
2547 // for a comparison with 1. For <= and >=, a comparison with zero.
2548 if (!MaxLHS ||
2549 (ICmpInst::isTrueWhenEqual(Pred) ? !MaxLHS->isZero() : (MaxLHS != One)))
2550 return Cond;
2551
2552 // Check the relevant induction variable for conformance to
2553 // the pattern.
2554 const SCEV *IV = SE.getSCEV(Cond->getOperand(0));
2555 if (!match(IV,
2557 return Cond;
2558
2559 assert(cast<SCEVAddRecExpr>(IV)->getLoop() == L &&
2560 "Loop condition operand is an addrec in a different loop!");
2561
2562 // Check the right operand of the select, and remember it, as it will
2563 // be used in the new comparison instruction.
2564 Value *NewRHS = nullptr;
2565 if (ICmpInst::isTrueWhenEqual(Pred)) {
2566 // Look for n+1, and grab n.
2567 if (AddOperator *BO = dyn_cast<AddOperator>(Sel->getOperand(1)))
2568 if (ConstantInt *BO1 = dyn_cast<ConstantInt>(BO->getOperand(1)))
2569 if (BO1->isOne() && SE.getSCEV(BO->getOperand(0)) == MaxRHS)
2570 NewRHS = BO->getOperand(0);
2571 if (AddOperator *BO = dyn_cast<AddOperator>(Sel->getOperand(2)))
2572 if (ConstantInt *BO1 = dyn_cast<ConstantInt>(BO->getOperand(1)))
2573 if (BO1->isOne() && SE.getSCEV(BO->getOperand(0)) == MaxRHS)
2574 NewRHS = BO->getOperand(0);
2575 if (!NewRHS)
2576 return Cond;
2577 } else if (SE.getSCEV(Sel->getOperand(1)) == MaxRHS)
2578 NewRHS = Sel->getOperand(1);
2579 else if (SE.getSCEV(Sel->getOperand(2)) == MaxRHS)
2580 NewRHS = Sel->getOperand(2);
2581 else if (const SCEVUnknown *SU = dyn_cast<SCEVUnknown>(MaxRHS))
2582 NewRHS = SU->getValue();
2583 else
2584 // Max doesn't match expected pattern.
2585 return Cond;
2586
2587 // Determine the new comparison opcode. It may be signed or unsigned,
2588 // and the original comparison may be either equality or inequality.
2589 if (Cond->getPredicate() == CmpInst::ICMP_EQ)
2590 Pred = CmpInst::getInversePredicate(Pred);
2591
2592 // Ok, everything looks ok to change the condition into an SLT or SGE and
2593 // delete the max calculation.
2594 ICmpInst *NewCond = new ICmpInst(Cond->getIterator(), Pred,
2595 Cond->getOperand(0), NewRHS, "scmp");
2596
2597 // Delete the max calculation instructions.
2598 NewCond->setDebugLoc(Cond->getDebugLoc());
2599 Cond->replaceAllUsesWith(NewCond);
2600 CondUse->setUser(NewCond);
2602 Cond->eraseFromParent();
2603 Sel->eraseFromParent();
2604 if (Cmp->use_empty()) {
2605 salvageDebugInfo(*Cmp);
2606 Cmp->eraseFromParent();
2607 }
2608 return NewCond;
2609}
2610
2611/// Change loop terminating condition to use the postinc iv when possible.
2612void
2613LSRInstance::OptimizeLoopTermCond() {
2614 SmallPtrSet<Instruction *, 4> PostIncs;
2615
2616 // We need a different set of heuristics for rotated and non-rotated loops.
2617 // If a loop is rotated then the latch is also the backedge, so inserting
2618 // post-inc expressions just before the latch is ideal. To reduce live ranges
2619 // it also makes sense to rewrite terminating conditions to use post-inc
2620 // expressions.
2621 //
2622 // If the loop is not rotated then the latch is not a backedge; the latch
2623 // check is done in the loop head. Adding post-inc expressions before the
2624 // latch will cause overlapping live-ranges of pre-inc and post-inc expressions
2625 // in the loop body. In this case we do *not* want to use post-inc expressions
2626 // in the latch check, and we want to insert post-inc expressions before
2627 // the backedge.
2628 BasicBlock *LatchBlock = L->getLoopLatch();
2629 SmallVector<BasicBlock*, 8> ExitingBlocks;
2630 L->getExitingBlocks(ExitingBlocks);
2631 if (!llvm::is_contained(ExitingBlocks, LatchBlock)) {
2632 // The backedge doesn't exit the loop; treat this as a head-tested loop.
2633 IVIncInsertPos = LatchBlock->getTerminator();
2634 return;
2635 }
2636
2637 // Otherwise treat this as a rotated loop.
2638 for (BasicBlock *ExitingBlock : ExitingBlocks) {
2639 // Get the terminating condition for the loop if possible. If we
2640 // can, we want to change it to use a post-incremented version of its
2641 // induction variable, to allow coalescing the live ranges for the IV into
2642 // one register value.
2643
2644 CondBrInst *TermBr = dyn_cast<CondBrInst>(ExitingBlock->getTerminator());
2645 if (!TermBr)
2646 continue;
2647
2649 // If the argument to TermBr is an extractelement, then the source of that
2650 // instruction is what's generated the condition.
2652 if (Extract)
2653 Cond = dyn_cast<Instruction>(Extract->getVectorOperand());
2654 // FIXME: We could do more here, like handling logical operations where one
2655 // side is a cmp that uses an induction variable.
2656 if (!Cond)
2657 continue;
2658
2659 // Search IVUsesByStride to find Cond's IVUse if there is one.
2660 IVStrideUse *CondUse = nullptr;
2661 if (!FindIVUserForCond(Cond, CondUse))
2662 continue;
2663
2664 // If the trip count is computed in terms of a max (due to ScalarEvolution
2665 // being unable to find a sufficient guard, for example), change the loop
2666 // comparison to use SLT or ULT instead of NE.
2667 // One consequence of doing this now is that it disrupts the count-down
2668 // optimization. That's not always a bad thing though, because in such
2669 // cases it may still be worthwhile to avoid a max.
2670 if (auto *Cmp = dyn_cast<ICmpInst>(Cond))
2671 Cond = OptimizeMax(Cmp, CondUse);
2672
2673 // If this exiting block dominates the latch block, it may also use
2674 // the post-inc value if it won't be shared with other uses.
2675 // Check for dominance.
2676 if (!DT.dominates(ExitingBlock, LatchBlock))
2677 continue;
2678
2679 // Conservatively avoid trying to use the post-inc value in non-latch
2680 // exits if there may be pre-inc users in intervening blocks.
2681 if (LatchBlock != ExitingBlock)
2682 for (const IVStrideUse &UI : IU)
2683 // Test if the use is reachable from the exiting block. This dominator
2684 // query is a conservative approximation of reachability.
2685 if (&UI != CondUse &&
2686 !DT.properlyDominates(UI.getUser()->getParent(), ExitingBlock)) {
2687 // Conservatively assume there may be reuse if the quotient of their
2688 // strides could be a legal scale.
2689 const SCEV *A = IU.getStride(*CondUse, L);
2690 const SCEV *B = IU.getStride(UI, L);
2691 if (!A || !B) continue;
2692 if (SE.getTypeSizeInBits(A->getType()) !=
2693 SE.getTypeSizeInBits(B->getType())) {
2694 if (SE.getTypeSizeInBits(A->getType()) >
2695 SE.getTypeSizeInBits(B->getType()))
2696 B = SE.getSignExtendExpr(B, A->getType());
2697 else
2698 A = SE.getSignExtendExpr(A, B->getType());
2699 }
2700 if (const SCEVConstant *D =
2702 const ConstantInt *C = D->getValue();
2703 // Stride of one or negative one can have reuse with non-addresses.
2704 if (C->isOne() || C->isMinusOne())
2705 goto decline_post_inc;
2706 // Avoid weird situations.
2707 if (C->getValue().getSignificantBits() >= 64 ||
2708 C->getValue().isMinSignedValue())
2709 goto decline_post_inc;
2710 // Check for possible scaled-address reuse.
2711 if (isAddressUse(TTI, UI.getUser(), UI.getOperandValToReplace())) {
2712 MemAccessTy AccessTy =
2713 getAccessType(TTI, UI.getUser(), UI.getOperandValToReplace());
2714 int64_t Scale = C->getSExtValue();
2715 if (TTI.isLegalAddressingMode(AccessTy.MemTy, /*BaseGV=*/nullptr,
2716 /*BaseOffset=*/0,
2717 /*HasBaseReg=*/true, Scale,
2718 AccessTy.AddrSpace))
2719 goto decline_post_inc;
2720 Scale = -Scale;
2721 if (TTI.isLegalAddressingMode(AccessTy.MemTy, /*BaseGV=*/nullptr,
2722 /*BaseOffset=*/0,
2723 /*HasBaseReg=*/true, Scale,
2724 AccessTy.AddrSpace))
2725 goto decline_post_inc;
2726 }
2727 }
2728 }
2729
2730 LLVM_DEBUG(dbgs() << " Change loop exiting icmp to use postinc iv: "
2731 << *Cond << '\n');
2732
2733 // It's possible for the setcc instruction to be anywhere in the loop, and
2734 // possible for it to have multiple users. If it is not immediately before
2735 // the exiting block branch, move it.
2736 if (isa_and_nonnull<CmpInst>(Cond) && Cond->getNextNode() != TermBr &&
2737 !Extract) {
2738 if (Cond->hasOneUse()) {
2739 Cond->moveBefore(TermBr->getIterator());
2740 } else {
2741 // Clone the terminating condition and insert into the loopend.
2742 Instruction *OldCond = Cond;
2743 Cond = Cond->clone();
2744 Cond->setName(L->getHeader()->getName() + ".termcond");
2745 Cond->insertInto(ExitingBlock, TermBr->getIterator());
2746
2747 // Clone the IVUse, as the old use still exists!
2748 CondUse = &IU.AddUser(Cond, CondUse->getOperandValToReplace());
2749 TermBr->replaceUsesOfWith(OldCond, Cond);
2750 }
2751 }
2752
2753 // If we get to here, we know that we can transform the setcc instruction to
2754 // use the post-incremented version of the IV, allowing us to coalesce the
2755 // live ranges for the IV correctly.
2756 CondUse->transformToPostInc(L);
2757 Changed = true;
2758
2759 PostIncs.insert(Cond);
2760 decline_post_inc:;
2761 }
2762
2763 // Determine an insertion point for the loop induction variable increment. It
2764 // must dominate all the post-inc comparisons we just set up, and it must
2765 // dominate the loop latch edge.
2766 IVIncInsertPos = L->getLoopLatch()->getTerminator();
2767 for (Instruction *Inst : PostIncs)
2768 IVIncInsertPos = DT.findNearestCommonDominator(IVIncInsertPos, Inst);
2769}
2770
2771/// Determine if the given use can accommodate a fixup at the given offset and
2772/// other details. If so, update the use and return true.
2773bool LSRInstance::reconcileNewOffset(LSRUse &LU, Immediate NewOffset,
2774 bool HasBaseReg, LSRUse::KindType Kind,
2775 MemAccessTy AccessTy) {
2776 Immediate NewMinOffset = LU.MinOffset;
2777 Immediate NewMaxOffset = LU.MaxOffset;
2778 MemAccessTy NewAccessTy = AccessTy;
2779
2780 // Check for a mismatched kind. It's tempting to collapse mismatched kinds to
2781 // something conservative, however this can pessimize in the case that one of
2782 // the uses will have all its uses outside the loop, for example.
2783 if (LU.Kind != Kind)
2784 return false;
2785
2786 // Check for a mismatched access type, and fall back conservatively as needed.
2787 // TODO: Be less conservative when the type is similar and can use the same
2788 // addressing modes.
2789 if (Kind == LSRUse::Address) {
2790 if (AccessTy.MemTy != LU.AccessTy.MemTy) {
2791 NewAccessTy = MemAccessTy::getUnknown(AccessTy.MemTy->getContext(),
2792 AccessTy.AddrSpace);
2793 }
2794 }
2795
2796 // Conservatively assume HasBaseReg is true for now.
2797 if (Immediate::isKnownLT(NewOffset, LU.MinOffset)) {
2798 if (!isAlwaysFoldable(Opts, TTI, Kind, NewAccessTy, /*BaseGV=*/nullptr,
2799 LU.MaxOffset - NewOffset, HasBaseReg))
2800 return false;
2801 NewMinOffset = NewOffset;
2802 } else if (Immediate::isKnownGT(NewOffset, LU.MaxOffset)) {
2803 if (!isAlwaysFoldable(Opts, TTI, Kind, NewAccessTy, /*BaseGV=*/nullptr,
2804 NewOffset - LU.MinOffset, HasBaseReg))
2805 return false;
2806 NewMaxOffset = NewOffset;
2807 }
2808
2809 // FIXME: We should be able to handle some level of scalable offset support
2810 // for 'void', but in order to get basic support up and running this is
2811 // being left out.
2812 if (NewAccessTy.MemTy && NewAccessTy.MemTy->isVoidTy() &&
2813 (NewMinOffset.isScalable() || NewMaxOffset.isScalable()))
2814 return false;
2815
2816 // Update the use.
2817 LU.MinOffset = NewMinOffset;
2818 LU.MaxOffset = NewMaxOffset;
2819 LU.AccessTy = NewAccessTy;
2820 return true;
2821}
2822
2823/// Return an LSRUse index and an offset value for a fixup which needs the given
2824/// expression, with the given kind and optional access type. Either reuse an
2825/// existing use or create a new one, as needed.
2826std::pair<size_t, Immediate> LSRInstance::getUse(const SCEV *&Expr,
2827 LSRUse::KindType Kind,
2828 MemAccessTy AccessTy) {
2829 const SCEV *Copy = Expr;
2830 SCEVUse ExprUse = Expr;
2831 Immediate Offset = extractImmediate(
2832 Opts, ExprUse, SE, AccessTy.MemTy && AccessTy.MemTy->isScalableTy());
2833 Expr = ExprUse;
2834
2835 // Basic uses can't accept any offset, for example.
2836 if (!isAlwaysFoldable(Opts, TTI, Kind, AccessTy, /*BaseGV=*/nullptr, Offset,
2837 /*HasBaseReg=*/true)) {
2838 Expr = Copy;
2839 Offset = Immediate::getFixed(0);
2840 }
2841
2842 std::pair<UseMapTy::iterator, bool> P =
2843 UseMap.try_emplace(LSRUse::SCEVUseKindPair(Expr, Kind));
2844 if (!P.second) {
2845 // A use already existed with this base.
2846 size_t LUIdx = P.first->second;
2847 LSRUse &LU = Uses[LUIdx];
2848 if (reconcileNewOffset(LU, Offset, /*HasBaseReg=*/true, Kind, AccessTy))
2849 // Reuse this use.
2850 return std::make_pair(LUIdx, Offset);
2851 }
2852
2853 // Create a new use.
2854 size_t LUIdx = Uses.size();
2855 P.first->second = LUIdx;
2856 Uses.push_back(LSRUse(Kind, AccessTy));
2857 LSRUse &LU = Uses[LUIdx];
2858
2859 LU.MinOffset = Offset;
2860 LU.MaxOffset = Offset;
2861 return std::make_pair(LUIdx, Offset);
2862}
2863
2864/// Delete the given use from the Uses list.
2865void LSRInstance::DeleteUse(LSRUse &LU, size_t LUIdx) {
2866 if (&LU != &Uses.back())
2867 std::swap(LU, Uses.back());
2868 Uses.pop_back();
2869
2870 // Update RegUses.
2871 RegUses.swapAndDropUse(LUIdx, Uses.size());
2872}
2873
2874/// Look for a use distinct from OrigLU which is has a formula that has the same
2875/// registers as the given formula.
2876LSRUse *
2877LSRInstance::FindUseWithSimilarFormula(const Formula &OrigF,
2878 const LSRUse &OrigLU) {
2879 // Search all uses for the formula. This could be more clever.
2880 for (LSRUse &LU : Uses) {
2881 // Check whether this use is close enough to OrigLU, to see whether it's
2882 // worthwhile looking through its formulae.
2883 // Ignore ICmpZero uses because they may contain formulae generated by
2884 // GenerateICmpZeroScales, in which case adding fixup offsets may
2885 // be invalid.
2886 if (&LU != &OrigLU && LU.Kind != LSRUse::ICmpZero &&
2887 LU.Kind == OrigLU.Kind && OrigLU.AccessTy == LU.AccessTy &&
2888 LU.HasFormulaWithSameRegs(OrigF)) {
2889 // Scan through this use's formulae.
2890 for (const Formula &F : LU.Formulae) {
2891 // Check to see if this formula has the same registers and symbols
2892 // as OrigF.
2893 if (F.BaseRegs == OrigF.BaseRegs &&
2894 F.ScaledReg == OrigF.ScaledReg &&
2895 F.BaseGV == OrigF.BaseGV &&
2896 F.Scale == OrigF.Scale &&
2897 F.UnfoldedOffset == OrigF.UnfoldedOffset) {
2898 if (F.BaseOffset.isZero())
2899 return &LU;
2900 // This is the formula where all the registers and symbols matched;
2901 // there aren't going to be any others. Since we declined it, we
2902 // can skip the rest of the formulae and proceed to the next LSRUse.
2903 break;
2904 }
2905 }
2906 }
2907 }
2908
2909 // Nothing looked good.
2910 return nullptr;
2911}
2912
2913void LSRInstance::CollectInterestingTypesAndFactors() {
2914 SmallSetVector<const SCEV *, 4> Strides;
2915
2916 // Collect interesting types and strides.
2918 for (const IVStrideUse &U : IU) {
2919 const SCEV *Expr = IU.getExpr(U);
2920 if (!Expr)
2921 continue;
2922
2923 // Collect interesting types.
2924 Types.insert(SE.getEffectiveSCEVType(Expr->getType()));
2925
2926 // Add strides for mentioned loops.
2927 Worklist.push_back(Expr);
2928 do {
2929 const SCEV *S = Worklist.pop_back_val();
2930 if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(S)) {
2931 if (AR->getLoop() == L)
2932 Strides.insert(AR->getStepRecurrence(SE));
2933 Worklist.push_back(AR->getStart());
2934 } else if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
2935 append_range(Worklist, Add->operands());
2936 }
2937 } while (!Worklist.empty());
2938 }
2939
2940 // Compute interesting factors from the set of interesting strides.
2941 for (SmallSetVector<const SCEV *, 4>::const_iterator
2942 I = Strides.begin(), E = Strides.end(); I != E; ++I)
2943 for (SmallSetVector<const SCEV *, 4>::const_iterator NewStrideIter =
2944 std::next(I); NewStrideIter != E; ++NewStrideIter) {
2945 const SCEV *OldStride = *I;
2946 const SCEV *NewStride = *NewStrideIter;
2947
2948 if (SE.getTypeSizeInBits(OldStride->getType()) !=
2949 SE.getTypeSizeInBits(NewStride->getType())) {
2950 if (SE.getTypeSizeInBits(OldStride->getType()) >
2951 SE.getTypeSizeInBits(NewStride->getType()))
2952 NewStride = SE.getSignExtendExpr(NewStride, OldStride->getType());
2953 else
2954 OldStride = SE.getSignExtendExpr(OldStride, NewStride->getType());
2955 }
2956 if (const SCEVConstant *Factor =
2957 dyn_cast_or_null<SCEVConstant>(getExactSDiv(NewStride, OldStride,
2958 SE, true))) {
2959 if (Factor->getAPInt().getSignificantBits() <= 64 && !Factor->isZero())
2960 Factors.insert(Factor->getAPInt().getSExtValue());
2961 } else if (const SCEVConstant *Factor =
2963 NewStride,
2964 SE, true))) {
2965 if (Factor->getAPInt().getSignificantBits() <= 64 && !Factor->isZero())
2966 Factors.insert(Factor->getAPInt().getSExtValue());
2967 }
2968 }
2969
2970 // If all uses use the same type, don't bother looking for truncation-based
2971 // reuse.
2972 if (Types.size() == 1)
2973 Types.clear();
2974
2975 LLVM_DEBUG(print_factors_and_types(dbgs()));
2976}
2977
2978/// Helper for CollectChains that finds an IV operand (computed by an AddRec in
2979/// this loop) within [OI,OE) or returns OE. If IVUsers mapped Instructions to
2980/// IVStrideUses, we could partially skip this.
2981static User::op_iterator
2983 Loop *L, ScalarEvolution &SE) {
2984 for(; OI != OE; ++OI) {
2985 if (Instruction *Oper = dyn_cast<Instruction>(*OI)) {
2986 if (!SE.isSCEVable(Oper->getType()))
2987 continue;
2988
2989 if (const SCEVAddRecExpr *AR =
2991 if (AR->getLoop() == L)
2992 break;
2993 }
2994 }
2995 }
2996 return OI;
2997}
2998
2999/// IVChain logic must consistently peek base TruncInst operands, so wrap it in
3000/// a convenient helper.
3002 if (TruncInst *Trunc = dyn_cast<TruncInst>(Oper))
3003 return Trunc->getOperand(0);
3004 return Oper;
3005}
3006
3007/// Return an approximation of this SCEV expression's "base", or NULL for any
3008/// constant. Returning the expression itself is conservative. Returning a
3009/// deeper subexpression is more precise and valid as long as it isn't less
3010/// complex than another subexpression. For expressions involving multiple
3011/// unscaled values, we need to return the pointer-type SCEVUnknown. This avoids
3012/// forming chains across objects, such as: PrevOper==a[i], IVOper==b[i],
3013/// IVInc==b-a.
3014///
3015/// Since SCEVUnknown is the rightmost type, and pointers are the rightmost
3016/// SCEVUnknown, we simply return the rightmost SCEV operand.
3017static const SCEV *getExprBase(const SCEV *S) {
3018 switch (S->getSCEVType()) {
3019 default: // including scUnknown.
3020 return S;
3021 case scConstant:
3022 case scVScale:
3023 return nullptr;
3024 case scTruncate:
3025 return getExprBase(cast<SCEVTruncateExpr>(S)->getOperand());
3026 case scZeroExtend:
3027 return getExprBase(cast<SCEVZeroExtendExpr>(S)->getOperand());
3028 case scSignExtend:
3029 return getExprBase(cast<SCEVSignExtendExpr>(S)->getOperand());
3030 case scAddExpr: {
3031 // Skip over scaled operands (scMulExpr) to follow add operands as long as
3032 // there's nothing more complex.
3033 // FIXME: not sure if we want to recognize negation.
3034 const SCEVAddExpr *Add = cast<SCEVAddExpr>(S);
3035 for (const SCEV *SubExpr : reverse(Add->operands())) {
3036 if (SubExpr->getSCEVType() == scAddExpr)
3037 return getExprBase(SubExpr);
3038
3039 if (SubExpr->getSCEVType() != scMulExpr)
3040 return SubExpr;
3041 }
3042 return S; // all operands are scaled, be conservative.
3043 }
3044 case scAddRecExpr:
3045 return getExprBase(cast<SCEVAddRecExpr>(S)->getStart());
3046 }
3047 llvm_unreachable("Unknown SCEV kind!");
3048}
3049
3050/// Return true if the chain increment is profitable to expand into a loop
3051/// invariant value, which may require its own register. A profitable chain
3052/// increment will be an offset relative to the same base. We allow such offsets
3053/// to potentially be used as chain increment as long as it's not obviously
3054/// expensive to expand using real instructions.
3055bool IVChain::isProfitableIncrement(const SCEV *OperExpr,
3056 const SCEV *IncExpr,
3057 ScalarEvolution &SE) {
3058 // Aggressively form chains when -stress-ivchain.
3059 if (StressIVChain)
3060 return true;
3061
3062 // Do not replace a constant offset from IV head with a nonconstant IV
3063 // increment.
3064 if (!isa<SCEVConstant>(IncExpr)) {
3065 const SCEV *HeadExpr = SE.getSCEV(getWideOperand(Incs[0].IVOperand));
3066 if (isa<SCEVConstant>(SE.getMinusSCEV(OperExpr, HeadExpr)))
3067 return false;
3068 }
3069
3070 SmallPtrSet<const SCEV*, 8> Processed;
3071 return !isHighCostExpansion(IncExpr, Processed, SE);
3072}
3073
3074/// Return true if the number of registers needed for the chain is estimated to
3075/// be less than the number required for the individual IV users. First prohibit
3076/// any IV users that keep the IV live across increments (the Users set should
3077/// be empty). Next count the number and type of increments in the chain.
3078///
3079/// Chaining IVs can lead to considerable code bloat if ISEL doesn't
3080/// effectively use postinc addressing modes. Only consider it profitable it the
3081/// increments can be computed in fewer registers when chained.
3082///
3083/// TODO: Consider IVInc free if it's already used in another chains.
3084static bool isProfitableChain(IVChain &Chain,
3086 ScalarEvolution &SE,
3087 const TargetTransformInfo &TTI) {
3088 if (StressIVChain)
3089 return true;
3090
3091 if (!Chain.hasIncs())
3092 return false;
3093
3094 if (!Users.empty()) {
3095 LLVM_DEBUG(dbgs() << "Chain: " << *Chain.Incs[0].UserInst << " users:\n";
3096 for (Instruction *Inst
3097 : Users) { dbgs() << " " << *Inst << "\n"; });
3098 return false;
3099 }
3100 assert(!Chain.Incs.empty() && "empty IV chains are not allowed");
3101
3102 // The chain itself may require a register, so initialize cost to 1.
3103 int cost = 1;
3104
3105 // A complete chain likely eliminates the need for keeping the original IV in
3106 // a register. LSR does not currently know how to form a complete chain unless
3107 // the header phi already exists.
3108 if (isa<PHINode>(Chain.tailUserInst())
3109 && SE.getSCEV(Chain.tailUserInst()) == Chain.Incs[0].IncExpr) {
3110 --cost;
3111 }
3112 const SCEV *LastIncExpr = nullptr;
3113 unsigned NumConstIncrements = 0;
3114 unsigned NumVarIncrements = 0;
3115 unsigned NumReusedIncrements = 0;
3116
3117 if (TTI.isProfitableLSRChainElement(Chain.Incs[0].UserInst))
3118 return true;
3119
3120 for (const IVInc &Inc : Chain) {
3121 if (TTI.isProfitableLSRChainElement(Inc.UserInst))
3122 return true;
3123 if (Inc.IncExpr->isZero())
3124 continue;
3125
3126 // Incrementing by zero or some constant is neutral. We assume constants can
3127 // be folded into an addressing mode or an add's immediate operand.
3128 if (isa<SCEVConstant>(Inc.IncExpr)) {
3129 ++NumConstIncrements;
3130 continue;
3131 }
3132
3133 if (Inc.IncExpr == LastIncExpr)
3134 ++NumReusedIncrements;
3135 else
3136 ++NumVarIncrements;
3137
3138 LastIncExpr = Inc.IncExpr;
3139 }
3140 // An IV chain with a single increment is handled by LSR's postinc
3141 // uses. However, a chain with multiple increments requires keeping the IV's
3142 // value live longer than it needs to be if chained.
3143 if (NumConstIncrements > 1)
3144 --cost;
3145
3146 // Materializing increment expressions in the preheader that didn't exist in
3147 // the original code may cost a register. For example, sign-extended array
3148 // indices can produce ridiculous increments like this:
3149 // IV + ((sext i32 (2 * %s) to i64) + (-1 * (sext i32 %s to i64)))
3150 cost += NumVarIncrements;
3151
3152 // Reusing variable increments likely saves a register to hold the multiple of
3153 // the stride.
3154 cost -= NumReusedIncrements;
3155
3156 LLVM_DEBUG(dbgs() << "Chain: " << *Chain.Incs[0].UserInst << " Cost: " << cost
3157 << "\n");
3158
3159 return cost < 0;
3160}
3161
3162/// Add this IV user to an existing chain or make it the head of a new chain.
3163void LSRInstance::ChainInstruction(Instruction *UserInst, Instruction *IVOper,
3164 SmallVectorImpl<ChainUsers> &ChainUsersVec) {
3165 // When IVs are used as types of varying widths, they are generally converted
3166 // to a wider type with some uses remaining narrow under a (free) trunc.
3167 Value *const NextIV = getWideOperand(IVOper);
3168 const SCEV *const OperExpr = SE.getSCEV(NextIV);
3169 const SCEV *const OperExprBase = getExprBase(OperExpr);
3170
3171 // Visit all existing chains. Check if its IVOper can be computed as a
3172 // profitable loop invariant increment from the last link in the Chain.
3173 unsigned ChainIdx = 0, NChains = IVChainVec.size();
3174 const SCEV *LastIncExpr = nullptr;
3175 for (; ChainIdx < NChains; ++ChainIdx) {
3176 IVChain &Chain = IVChainVec[ChainIdx];
3177
3178 // Prune the solution space aggressively by checking that both IV operands
3179 // are expressions that operate on the same unscaled SCEVUnknown. This
3180 // "base" will be canceled by the subsequent getMinusSCEV call. Checking
3181 // first avoids creating extra SCEV expressions.
3182 if (!StressIVChain && Chain.ExprBase != OperExprBase)
3183 continue;
3184
3185 Value *PrevIV = getWideOperand(Chain.Incs.back().IVOperand);
3186 if (PrevIV->getType() != NextIV->getType())
3187 continue;
3188
3189 // A phi node terminates a chain.
3190 if (isa<PHINode>(UserInst) && isa<PHINode>(Chain.tailUserInst()))
3191 continue;
3192
3193 // The increment must be loop-invariant so it can be kept in a register.
3194 const SCEV *PrevExpr = SE.getSCEV(PrevIV);
3195 const SCEV *IncExpr = SE.getMinusSCEV(OperExpr, PrevExpr);
3196 if (isa<SCEVCouldNotCompute>(IncExpr) || !SE.isLoopInvariant(IncExpr, L))
3197 continue;
3198
3199 if (Chain.isProfitableIncrement(OperExpr, IncExpr, SE)) {
3200 LastIncExpr = IncExpr;
3201 break;
3202 }
3203 }
3204 // If we haven't found a chain, create a new one, unless we hit the max. Don't
3205 // bother for phi nodes, because they must be last in the chain.
3206 if (ChainIdx == NChains) {
3207 if (isa<PHINode>(UserInst))
3208 return;
3209 if (NChains >= MaxChains && !StressIVChain) {
3210 LLVM_DEBUG(dbgs() << "IV Chain Limit\n");
3211 return;
3212 }
3213 LastIncExpr = OperExpr;
3214 // IVUsers may have skipped over sign/zero extensions. We don't currently
3215 // attempt to form chains involving extensions unless they can be hoisted
3216 // into this loop's AddRec.
3217 if (!isa<SCEVAddRecExpr>(LastIncExpr))
3218 return;
3219 ++NChains;
3220 IVChainVec.push_back(IVChain(IVInc(UserInst, IVOper, LastIncExpr),
3221 OperExprBase));
3222 ChainUsersVec.resize(NChains);
3223 LLVM_DEBUG(dbgs() << "IV Chain#" << ChainIdx << " Head: (" << *UserInst
3224 << ") IV=" << *LastIncExpr << "\n");
3225 } else {
3226 LLVM_DEBUG(dbgs() << "IV Chain#" << ChainIdx << " Inc: (" << *UserInst
3227 << ") IV+" << *LastIncExpr << "\n");
3228 // Add this IV user to the end of the chain.
3229 IVChainVec[ChainIdx].add(IVInc(UserInst, IVOper, LastIncExpr));
3230 }
3231 IVChain &Chain = IVChainVec[ChainIdx];
3232
3233 SmallPtrSet<Instruction*,4> &NearUsers = ChainUsersVec[ChainIdx].NearUsers;
3234 // This chain's NearUsers become FarUsers.
3235 if (!LastIncExpr->isZero()) {
3236 ChainUsersVec[ChainIdx].FarUsers.insert_range(NearUsers);
3237 NearUsers.clear();
3238 }
3239
3240 // All other uses of IVOperand become near uses of the chain.
3241 // We currently ignore intermediate values within SCEV expressions, assuming
3242 // they will eventually be used be the current chain, or can be computed
3243 // from one of the chain increments. To be more precise we could
3244 // transitively follow its user and only add leaf IV users to the set.
3245 for (User *U : IVOper->users()) {
3246 Instruction *OtherUse = dyn_cast<Instruction>(U);
3247 if (!OtherUse)
3248 continue;
3249 // Uses in the chain will no longer be uses if the chain is formed.
3250 // Include the head of the chain in this iteration (not Chain.begin()).
3251 IVChain::const_iterator IncIter = Chain.Incs.begin();
3252 IVChain::const_iterator IncEnd = Chain.Incs.end();
3253 for( ; IncIter != IncEnd; ++IncIter) {
3254 if (IncIter->UserInst == OtherUse)
3255 break;
3256 }
3257 if (IncIter != IncEnd)
3258 continue;
3259
3260 if (SE.isSCEVable(OtherUse->getType())
3261 && !isa<SCEVUnknown>(SE.getSCEV(OtherUse))
3262 && IU.isIVUserOrOperand(OtherUse)) {
3263 continue;
3264 }
3265 NearUsers.insert(OtherUse);
3266 }
3267
3268 // Since this user is part of the chain, it's no longer considered a use
3269 // of the chain.
3270 ChainUsersVec[ChainIdx].FarUsers.erase(UserInst);
3271}
3272
3273/// Populate the vector of Chains.
3274///
3275/// This decreases ILP at the architecture level. Targets with ample registers,
3276/// multiple memory ports, and no register renaming probably don't want
3277/// this. However, such targets should probably disable LSR altogether.
3278///
3279/// The job of LSR is to make a reasonable choice of induction variables across
3280/// the loop. Subsequent passes can easily "unchain" computation exposing more
3281/// ILP *within the loop* if the target wants it.
3282///
3283/// Finding the best IV chain is potentially a scheduling problem. Since LSR
3284/// will not reorder memory operations, it will recognize this as a chain, but
3285/// will generate redundant IV increments. Ideally this would be corrected later
3286/// by a smart scheduler:
3287/// = A[i]
3288/// = A[i+x]
3289/// A[i] =
3290/// A[i+x] =
3291///
3292/// TODO: Walk the entire domtree within this loop, not just the path to the
3293/// loop latch. This will discover chains on side paths, but requires
3294/// maintaining multiple copies of the Chains state.
3295void LSRInstance::CollectChains() {
3296 LLVM_DEBUG(dbgs() << "Collecting IV Chains.\n");
3297 SmallVector<ChainUsers, 8> ChainUsersVec;
3298
3299 SmallVector<BasicBlock *,8> LatchPath;
3300 BasicBlock *LoopHeader = L->getHeader();
3301 for (DomTreeNode *Rung = DT.getNode(L->getLoopLatch());
3302 Rung->getBlock() != LoopHeader; Rung = Rung->getIDom()) {
3303 LatchPath.push_back(Rung->getBlock());
3304 }
3305 LatchPath.push_back(LoopHeader);
3306
3307 // Walk the instruction stream from the loop header to the loop latch.
3308 for (BasicBlock *BB : reverse(LatchPath)) {
3309 for (Instruction &I : *BB) {
3310 // Skip instructions that weren't seen by IVUsers analysis.
3311 if (isa<PHINode>(I) || !IU.isIVUserOrOperand(&I))
3312 continue;
3313
3314 // Skip ephemeral values, as they don't produce real code.
3315 if (IU.isEphemeral(&I))
3316 continue;
3317
3318 // Ignore users that are part of a SCEV expression. This way we only
3319 // consider leaf IV Users. This effectively rediscovers a portion of
3320 // IVUsers analysis but in program order this time.
3321 if (SE.isSCEVable(I.getType()) && !isa<SCEVUnknown>(SE.getSCEV(&I)))
3322 continue;
3323
3324 // Remove this instruction from any NearUsers set it may be in.
3325 for (unsigned ChainIdx = 0, NChains = IVChainVec.size();
3326 ChainIdx < NChains; ++ChainIdx) {
3327 ChainUsersVec[ChainIdx].NearUsers.erase(&I);
3328 }
3329 // Search for operands that can be chained.
3330 SmallPtrSet<Instruction*, 4> UniqueOperands;
3331 User::op_iterator IVOpEnd = I.op_end();
3332 User::op_iterator IVOpIter = findIVOperand(I.op_begin(), IVOpEnd, L, SE);
3333 while (IVOpIter != IVOpEnd) {
3334 Instruction *IVOpInst = cast<Instruction>(*IVOpIter);
3335 if (UniqueOperands.insert(IVOpInst).second)
3336 ChainInstruction(&I, IVOpInst, ChainUsersVec);
3337 IVOpIter = findIVOperand(std::next(IVOpIter), IVOpEnd, L, SE);
3338 }
3339 } // Continue walking down the instructions.
3340 } // Continue walking down the domtree.
3341 // Visit phi backedges to determine if the chain can generate the IV postinc.
3342 for (PHINode &PN : L->getHeader()->phis()) {
3343 if (!SE.isSCEVable(PN.getType()))
3344 continue;
3345
3346 Instruction *IncV =
3347 dyn_cast<Instruction>(PN.getIncomingValueForBlock(L->getLoopLatch()));
3348 if (IncV)
3349 ChainInstruction(&PN, IncV, ChainUsersVec);
3350 }
3351 // Remove any unprofitable chains.
3352 unsigned ChainIdx = 0;
3353 for (unsigned UsersIdx = 0, NChains = IVChainVec.size();
3354 UsersIdx < NChains; ++UsersIdx) {
3355 if (!isProfitableChain(IVChainVec[UsersIdx],
3356 ChainUsersVec[UsersIdx].FarUsers, SE, TTI))
3357 continue;
3358 // Preserve the chain at UsesIdx.
3359 if (ChainIdx != UsersIdx)
3360 IVChainVec[ChainIdx] = IVChainVec[UsersIdx];
3361 FinalizeChain(IVChainVec[ChainIdx]);
3362 ++ChainIdx;
3363 }
3364 IVChainVec.resize(ChainIdx);
3365}
3366
3367void LSRInstance::FinalizeChain(IVChain &Chain) {
3368 assert(!Chain.Incs.empty() && "empty IV chains are not allowed");
3369 LLVM_DEBUG(dbgs() << "Final Chain: " << *Chain.Incs[0].UserInst << "\n");
3370
3371 for (const IVInc &Inc : Chain) {
3372 LLVM_DEBUG(dbgs() << " Inc: " << *Inc.UserInst << "\n");
3373 auto UseI = find(Inc.UserInst->operands(), Inc.IVOperand);
3374 assert(UseI != Inc.UserInst->op_end() && "cannot find IV operand");
3375 IVIncSet.insert(UseI);
3376 }
3377}
3378
3379/// Return true if the IVInc can be folded into an addressing mode.
3380static bool canFoldIVIncExpr(const ScalarOptions &Opts, const SCEV *IncExpr,
3381 Instruction *UserInst, Value *Operand,
3382 const TargetTransformInfo &TTI) {
3383 const SCEVConstant *IncConst = dyn_cast<SCEVConstant>(IncExpr);
3384 Immediate IncOffset = Immediate::getZero();
3385 if (IncConst) {
3386 if (IncConst && IncConst->getAPInt().getSignificantBits() > 64)
3387 return false;
3388 IncOffset = Immediate::getFixed(IncConst->getValue()->getSExtValue());
3389 } else {
3390 // Look for mul(vscale, constant), to detect a scalable offset.
3391 const APInt *C;
3392 if (!match(IncExpr, m_scev_Mul(m_scev_APInt(C), m_SCEVVScale())) ||
3393 C->getSignificantBits() > 64)
3394 return false;
3395 IncOffset = Immediate::getScalable(C->getSExtValue());
3396 }
3397
3398 if (!isAddressUse(TTI, UserInst, Operand))
3399 return false;
3400
3401 MemAccessTy AccessTy = getAccessType(TTI, UserInst, Operand);
3402 if (!isAlwaysFoldable(Opts, TTI, LSRUse::Address, AccessTy,
3403 /*BaseGV=*/nullptr, IncOffset, /*HasBaseReg=*/false))
3404 return false;
3405
3406 return true;
3407}
3408
3409/// Generate an add or subtract for each IVInc in a chain to materialize the IV
3410/// user's operand from the previous IV user's operand.
3411void LSRInstance::GenerateIVChain(const IVChain &Chain,
3412 SmallVectorImpl<WeakTrackingVH> &DeadInsts) {
3413 // Find the new IVOperand for the head of the chain. It may have been replaced
3414 // by LSR.
3415 const IVInc &Head = Chain.Incs[0];
3416 User::op_iterator IVOpEnd = Head.UserInst->op_end();
3417 // findIVOperand returns IVOpEnd if it can no longer find a valid IV user.
3418 User::op_iterator IVOpIter = findIVOperand(Head.UserInst->op_begin(),
3419 IVOpEnd, L, SE);
3420 Value *IVSrc = nullptr;
3421 while (IVOpIter != IVOpEnd) {
3422 IVSrc = getWideOperand(*IVOpIter);
3423
3424 // If this operand computes the expression that the chain needs, we may use
3425 // it. (Check this after setting IVSrc which is used below.)
3426 //
3427 // Note that if Head.IncExpr is wider than IVSrc, then this phi is too
3428 // narrow for the chain, so we can no longer use it. We do allow using a
3429 // wider phi, assuming the LSR checked for free truncation. In that case we
3430 // should already have a truncate on this operand such that
3431 // getSCEV(IVSrc) == IncExpr.
3432 if (SE.getSCEV(*IVOpIter) == Head.IncExpr
3433 || SE.getSCEV(IVSrc) == Head.IncExpr) {
3434 break;
3435 }
3436 IVOpIter = findIVOperand(std::next(IVOpIter), IVOpEnd, L, SE);
3437 }
3438 if (IVOpIter == IVOpEnd) {
3439 // Gracefully give up on this chain.
3440 LLVM_DEBUG(dbgs() << "Concealed chain head: " << *Head.UserInst << "\n");
3441 return;
3442 }
3443 assert(IVSrc && "Failed to find IV chain source");
3444
3445 LLVM_DEBUG(dbgs() << "Generate chain at: " << *IVSrc << "\n");
3446 Type *IVTy = IVSrc->getType();
3447 Type *IntTy = SE.getEffectiveSCEVType(IVTy);
3448 const SCEV *LeftOverExpr = nullptr;
3449 const SCEV *Accum = SE.getZero(IntTy);
3451 Bases.emplace_back(Accum, IVSrc);
3452
3453 for (const IVInc &Inc : Chain) {
3454 Instruction *InsertPt = Inc.UserInst;
3455 if (isa<PHINode>(InsertPt))
3456 InsertPt = L->getLoopLatch()->getTerminator();
3457
3458 // IVOper will replace the current IV User's operand. IVSrc is the IV
3459 // value currently held in a register.
3460 Value *IVOper = IVSrc;
3461 if (!Inc.IncExpr->isZero()) {
3462 // IncExpr was the result of subtraction of two narrow values, so must
3463 // be signed.
3464 const SCEV *IncExpr = SE.getNoopOrSignExtend(Inc.IncExpr, IntTy);
3465 Accum = SE.getAddExpr(Accum, IncExpr);
3466 LeftOverExpr = LeftOverExpr
3467 ? SE.getAddExpr(LeftOverExpr, IncExpr).getPointer()
3468 : IncExpr;
3469 }
3470
3471 // Look through each base to see if any can produce a nice addressing mode.
3472 bool FoundBase = false;
3473 for (auto [MapScev, MapIVOper] : reverse(Bases)) {
3474 const SCEV *Remainder = SE.getMinusSCEV(Accum, MapScev);
3475 if (canFoldIVIncExpr(Opts, Remainder, Inc.UserInst, Inc.IVOperand, TTI)) {
3476 if (!Remainder->isZero()) {
3477 Rewriter.clearPostInc();
3478 Value *IncV = Rewriter.expandCodeFor(Remainder, IntTy, InsertPt);
3479 const SCEV *IVOperExpr =
3480 SE.getAddExpr(SE.getUnknown(MapIVOper), SE.getUnknown(IncV));
3481 IVOper = Rewriter.expandCodeFor(IVOperExpr, IVTy, InsertPt);
3482 } else {
3483 IVOper = MapIVOper;
3484 }
3485
3486 FoundBase = true;
3487 break;
3488 }
3489 }
3490 if (!FoundBase && LeftOverExpr && !LeftOverExpr->isZero()) {
3491 // Expand the IV increment.
3492 Rewriter.clearPostInc();
3493 Value *IncV = Rewriter.expandCodeFor(LeftOverExpr, IntTy, InsertPt);
3494 const SCEV *IVOperExpr = SE.getAddExpr(SE.getUnknown(IVSrc),
3495 SE.getUnknown(IncV));
3496 IVOper = Rewriter.expandCodeFor(IVOperExpr, IVTy, InsertPt);
3497
3498 // If an IV increment can't be folded, use it as the next IV value.
3499 if (!canFoldIVIncExpr(Opts, LeftOverExpr, Inc.UserInst, Inc.IVOperand,
3500 TTI)) {
3501 assert(IVTy == IVOper->getType() && "inconsistent IV increment type");
3502 Bases.emplace_back(Accum, IVOper);
3503 IVSrc = IVOper;
3504 LeftOverExpr = nullptr;
3505 }
3506 }
3507 Type *OperTy = Inc.IVOperand->getType();
3508 if (IVTy != OperTy) {
3509 assert(SE.getTypeSizeInBits(IVTy) >= SE.getTypeSizeInBits(OperTy) &&
3510 "cannot extend a chained IV");
3511 IRBuilder<> Builder(InsertPt);
3512 IVOper = Builder.CreateTruncOrBitCast(IVOper, OperTy, "lsr.chain");
3513 }
3514 Inc.UserInst->replaceUsesOfWith(Inc.IVOperand, IVOper);
3515 if (auto *OperandIsInstr = dyn_cast<Instruction>(Inc.IVOperand))
3516 DeadInsts.emplace_back(OperandIsInstr);
3517 }
3518 // If LSR created a new, wider phi, we may also replace its postinc. We only
3519 // do this if we also found a wide value for the head of the chain.
3520 if (isa<PHINode>(Chain.tailUserInst())) {
3521 for (PHINode &Phi : L->getHeader()->phis()) {
3522 if (Phi.getType() != IVSrc->getType())
3523 continue;
3525 Phi.getIncomingValueForBlock(L->getLoopLatch()));
3526 if (!PostIncV || (SE.getSCEV(PostIncV) != SE.getSCEV(IVSrc)))
3527 continue;
3528 Value *IVOper = IVSrc;
3529 Type *PostIncTy = PostIncV->getType();
3530 if (IVTy != PostIncTy) {
3531 assert(PostIncTy->isPointerTy() && "mixing int/ptr IV types");
3532 IRBuilder<> Builder(L->getLoopLatch()->getTerminator());
3533 Builder.SetCurrentDebugLocation(PostIncV->getDebugLoc());
3534 IVOper = Builder.CreatePointerCast(IVSrc, PostIncTy, "lsr.chain");
3535 }
3536 Phi.replaceUsesOfWith(PostIncV, IVOper);
3537 DeadInsts.emplace_back(PostIncV);
3538 }
3539 }
3540}
3541
3542void LSRInstance::CollectFixupsAndInitialFormulae() {
3543 CondBrInst *ExitBranch = nullptr;
3544 bool SaveCmp = TTI.canSaveCmp(L, &ExitBranch, &SE, &LI, &DT, &AC, &TLI);
3545
3546 // For calculating baseline cost
3547 SmallPtrSet<const SCEV *, 16> Regs;
3548 DenseSet<const SCEV *> VisitedRegs;
3549 DenseSet<size_t> VisitedLSRUse;
3550
3551 for (const IVStrideUse &U : IU) {
3552 Instruction *UserInst = U.getUser();
3553 // Skip IV users that are part of profitable IV Chains.
3554 User::op_iterator UseI =
3555 find(UserInst->operands(), U.getOperandValToReplace());
3556 assert(UseI != UserInst->op_end() && "cannot find IV operand");
3557 if (IVIncSet.count(UseI)) {
3558 LLVM_DEBUG(dbgs() << "Use is in profitable chain: " << **UseI << '\n');
3559 continue;
3560 }
3561
3562 LSRUse::KindType Kind = LSRUse::Basic;
3563 MemAccessTy AccessTy;
3564 if (isAddressUse(TTI, UserInst, U.getOperandValToReplace())) {
3565 Kind = LSRUse::Address;
3566 AccessTy = getAccessType(TTI, UserInst, U.getOperandValToReplace());
3567 }
3568
3569 const SCEV *S = IU.getExpr(U);
3570 if (!S)
3571 continue;
3572 PostIncLoopSet TmpPostIncLoops = U.getPostIncLoops();
3573
3574 // Equality (== and !=) ICmps are special. We can rewrite (i == N) as
3575 // (N - i == 0), and this allows (N - i) to be the expression that we work
3576 // with rather than just N or i, so we can consider the register
3577 // requirements for both N and i at the same time. Limiting this code to
3578 // equality icmps is not a problem because all interesting loops use
3579 // equality icmps, thanks to IndVarSimplify.
3580 if (ICmpInst *CI = dyn_cast<ICmpInst>(UserInst)) {
3581 // If CI can be saved in some target, like replaced inside hardware loop
3582 // in PowerPC, no need to generate initial formulae for it.
3583 if (SaveCmp && CI == dyn_cast<ICmpInst>(ExitBranch->getCondition()))
3584 continue;
3585 if (CI->isEquality()) {
3586 // Swap the operands if needed to put the OperandValToReplace on the
3587 // left, for consistency.
3588 Value *NV = CI->getOperand(1);
3589 if (NV == U.getOperandValToReplace()) {
3590 CI->setOperand(1, CI->getOperand(0));
3591 CI->setOperand(0, NV);
3592 NV = CI->getOperand(1);
3593 Changed = true;
3594 }
3595
3596 // x == y --> x - y == 0
3597 const SCEV *N = SE.getSCEV(NV);
3598 if (SE.isLoopInvariant(N, L) && Rewriter.isSafeToExpand(N) &&
3599 (!NV->getType()->isPointerTy() ||
3600 SE.getPointerBase(N) == SE.getPointerBase(S))) {
3601 // S is normalized, so normalize N before folding it into S
3602 // to keep the result normalized.
3603 N = normalizeForPostIncUse(N, TmpPostIncLoops, SE);
3604 if (!N)
3605 continue;
3606 Kind = LSRUse::ICmpZero;
3607 S = SE.getMinusSCEV(N, S);
3608 } else if (L->isLoopInvariant(NV) &&
3609 (!isa<Instruction>(NV) ||
3610 DT.dominates(cast<Instruction>(NV), L->getHeader())) &&
3611 !NV->getType()->isPointerTy()) {
3612 // If we can't generally expand the expression (e.g. it contains
3613 // a divide), but it is already at a loop invariant point before the
3614 // loop, wrap it in an unknown (to prevent the expander from trying
3615 // to re-expand in a potentially unsafe way.) The restriction to
3616 // integer types is required because the unknown hides the base, and
3617 // SCEV can't compute the difference of two unknown pointers.
3618 N = SE.getUnknown(NV);
3619 N = normalizeForPostIncUse(N, TmpPostIncLoops, SE);
3620 if (!N)
3621 continue;
3622 Kind = LSRUse::ICmpZero;
3623 S = SE.getMinusSCEV(N, S);
3625 }
3626
3627 // -1 and the negations of all interesting strides (except the negation
3628 // of -1) are now also interesting.
3629 for (size_t i = 0, e = Factors.size(); i != e; ++i)
3630 if (Factors[i] != -1)
3631 Factors.insert(-(uint64_t)Factors[i]);
3632 Factors.insert(-1);
3633 }
3634 }
3635
3636 // Get or create an LSRUse.
3637 std::pair<size_t, Immediate> P = getUse(S, Kind, AccessTy);
3638 size_t LUIdx = P.first;
3639 Immediate Offset = P.second;
3640 LSRUse &LU = Uses[LUIdx];
3641
3642 // Record the fixup.
3643 LSRFixup &LF = LU.getNewFixup();
3644 LF.UserInst = UserInst;
3645 LF.OperandValToReplace = U.getOperandValToReplace();
3646 LF.PostIncLoops = TmpPostIncLoops;
3647 LF.Offset = Offset;
3648 LU.AllFixupsOutsideLoop &= LF.isUseFullyOutsideLoop(L);
3649 LU.AllFixupsUnconditional &= IsFixupExecutedEachIncrement(LF);
3650
3651 // Create SCEV as Formula for calculating baseline cost
3652 if (!VisitedLSRUse.count(LUIdx) && !LF.isUseFullyOutsideLoop(L)) {
3653 Formula F;
3654 F.initialMatch(S, L, SE);
3655 BaselineCost.RateFormula(F, Regs, VisitedRegs, LU,
3656 HardwareLoopProfitable);
3657 VisitedLSRUse.insert(LUIdx);
3658 }
3659
3660 // If this is the first use of this LSRUse, give it a formula.
3661 if (LU.Formulae.empty()) {
3662 InsertInitialFormula(S, LU, LUIdx);
3663 CountRegisters(LU.Formulae.back(), LUIdx);
3664 }
3665 }
3666
3667 LLVM_DEBUG(print_fixups(dbgs()));
3668}
3669
3670/// Insert a formula for the given expression into the given use, separating out
3671/// loop-variant portions from loop-invariant and loop-computable portions.
3672void LSRInstance::InsertInitialFormula(const SCEV *S, LSRUse &LU,
3673 size_t LUIdx) {
3674 // Mark uses whose expressions cannot be expanded.
3675 if (!Rewriter.isSafeToExpand(S))
3676 LU.RigidFormula = true;
3677
3678 Formula F;
3679 F.initialMatch(S, L, SE);
3680 bool Inserted = InsertFormula(LU, LUIdx, F);
3681 assert(Inserted && "Initial formula already exists!"); (void)Inserted;
3682}
3683
3684/// Insert a simple single-register formula for the given expression into the
3685/// given use.
3686void
3687LSRInstance::InsertSupplementalFormula(const SCEV *S,
3688 LSRUse &LU, size_t LUIdx) {
3689 Formula F;
3690 F.BaseRegs.push_back(S);
3691 F.HasBaseReg = true;
3692 bool Inserted = InsertFormula(LU, LUIdx, F);
3693 assert(Inserted && "Supplemental formula already exists!"); (void)Inserted;
3694}
3695
3696/// Note which registers are used by the given formula, updating RegUses.
3697void LSRInstance::CountRegisters(const Formula &F, size_t LUIdx) {
3698 if (F.ScaledReg)
3699 RegUses.countRegister(F.ScaledReg, LUIdx);
3700 for (const SCEV *BaseReg : F.BaseRegs)
3701 RegUses.countRegister(BaseReg, LUIdx);
3702}
3703
3704/// If the given formula has not yet been inserted, add it to the list, and
3705/// return true. Return false otherwise.
3706bool LSRInstance::InsertFormula(LSRUse &LU, unsigned LUIdx, const Formula &F) {
3707 // Do not insert formula that we will not be able to expand.
3708 assert(isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy, F) &&
3709 "Formula is illegal");
3710
3711 if (!LU.InsertFormula(F, *L))
3712 return false;
3713
3714 CountRegisters(F, LUIdx);
3715 return true;
3716}
3717
3718/// Test whether this fixup will be executed each time the corresponding IV
3719/// increment instruction is executed.
3720bool LSRInstance::IsFixupExecutedEachIncrement(const LSRFixup &LF) const {
3721 // If the fixup block dominates the IV increment block then there is no path
3722 // through the loop to the increment that doesn't pass through the fixup.
3723 return DT.dominates(LF.UserInst->getParent(), IVIncInsertPos->getParent());
3724}
3725
3726/// Check for other uses of loop-invariant values which we're tracking. These
3727/// other uses will pin these values in registers, making them less profitable
3728/// for elimination.
3729/// TODO: This currently misses non-constant addrec step registers.
3730/// TODO: Should this give more weight to users inside the loop?
3731void
3732LSRInstance::CollectLoopInvariantFixupsAndFormulae() {
3733 SmallVector<const SCEV *, 8> Worklist(RegUses.begin(), RegUses.end());
3734 SmallPtrSet<const SCEV *, 32> Visited;
3735
3736 // Don't collect outside uses if we are favoring postinc - the instructions in
3737 // the loop are more important than the ones outside of it.
3738 if (AMK == TTI::AMK_PostIndexed)
3739 return;
3740
3741 while (!Worklist.empty()) {
3742 const SCEV *S = Worklist.pop_back_val();
3743
3744 // Don't process the same SCEV twice
3745 if (!Visited.insert(S).second)
3746 continue;
3747
3748 if (const SCEVNAryExpr *N = dyn_cast<SCEVNAryExpr>(S))
3749 append_range(Worklist, N->operands());
3750 else if (const SCEVIntegralCastExpr *C = dyn_cast<SCEVIntegralCastExpr>(S))
3751 Worklist.push_back(C->getOperand());
3752 else if (const SCEVUDivExpr *D = dyn_cast<SCEVUDivExpr>(S)) {
3753 Worklist.push_back(D->getLHS());
3754 Worklist.push_back(D->getRHS());
3755 } else if (const SCEVUnknown *US = dyn_cast<SCEVUnknown>(S)) {
3756 const Value *V = US->getValue();
3757 if (const Instruction *Inst = dyn_cast<Instruction>(V)) {
3758 // Look for instructions defined outside the loop.
3759 if (L->contains(Inst)) continue;
3760 } else if (isa<Constant>(V))
3761 // Constants can be re-materialized.
3762 continue;
3763 for (const Use &U : V->uses()) {
3764 const Instruction *UserInst = dyn_cast<Instruction>(U.getUser());
3765 // Ignore non-instructions.
3766 if (!UserInst)
3767 continue;
3768 // Don't bother if the instruction is an EHPad.
3769 if (UserInst->isEHPad())
3770 continue;
3771 // Ignore instructions in other functions (as can happen with
3772 // Constants).
3773 if (UserInst->getParent()->getParent() != L->getHeader()->getParent())
3774 continue;
3775 // Ignore instructions not dominated by the loop.
3776 const BasicBlock *UseBB = !isa<PHINode>(UserInst) ?
3777 UserInst->getParent() :
3778 cast<PHINode>(UserInst)->getIncomingBlock(
3780 if (!DT.dominates(L->getHeader(), UseBB))
3781 continue;
3782 // Don't bother if the instruction is in a BB which ends in an EHPad.
3783 if (UseBB->getTerminator()->isEHPad())
3784 continue;
3785
3786 // Ignore cases in which the currently-examined value could come from
3787 // a basic block terminated with an EHPad. This checks all incoming
3788 // blocks of the phi node since it is possible that the same incoming
3789 // value comes from multiple basic blocks, only some of which may end
3790 // in an EHPad. If any of them do, a subsequent rewrite attempt by this
3791 // pass would try to insert instructions into an EHPad, hitting an
3792 // assertion.
3793 if (isa<PHINode>(UserInst)) {
3794 const auto *PhiNode = cast<PHINode>(UserInst);
3795 bool HasIncompatibleEHPTerminatedBlock = false;
3796 llvm::Value *ExpectedValue = U;
3797 for (unsigned int I = 0; I < PhiNode->getNumIncomingValues(); I++) {
3798 if (PhiNode->getIncomingValue(I) == ExpectedValue) {
3799 if (PhiNode->getIncomingBlock(I)->getTerminator()->isEHPad()) {
3800 HasIncompatibleEHPTerminatedBlock = true;
3801 break;
3802 }
3803 }
3804 }
3805 if (HasIncompatibleEHPTerminatedBlock) {
3806 continue;
3807 }
3808 }
3809
3810 // Don't bother rewriting PHIs in catchswitch blocks.
3811 if (isa<CatchSwitchInst>(UserInst->getParent()->getTerminator()))
3812 continue;
3813 // Ignore uses which are part of other SCEV expressions, to avoid
3814 // analyzing them multiple times.
3815 if (SE.isSCEVable(UserInst->getType())) {
3816 const SCEV *UserS = SE.getSCEV(const_cast<Instruction *>(UserInst));
3817 // If the user is a no-op, look through to its uses.
3818 if (!isa<SCEVUnknown>(UserS))
3819 continue;
3820 if (UserS == US) {
3821 Worklist.push_back(
3822 SE.getUnknown(const_cast<Instruction *>(UserInst)));
3823 continue;
3824 }
3825 }
3826 // Ignore icmp instructions which are already being analyzed.
3827 if (const ICmpInst *ICI = dyn_cast<ICmpInst>(UserInst)) {
3828 unsigned OtherIdx = !U.getOperandNo();
3829 Value *OtherOp = ICI->getOperand(OtherIdx);
3830 if (SE.hasComputableLoopEvolution(SE.getSCEV(OtherOp), L))
3831 continue;
3832 }
3833
3834 // Do not consider uses inside lifetime intrinsics. These are not
3835 // actually materialized.
3836 if (UserInst->isLifetimeStartOrEnd())
3837 continue;
3838
3839 std::pair<size_t, Immediate> P =
3840 getUse(S, LSRUse::Basic, MemAccessTy());
3841 size_t LUIdx = P.first;
3842 Immediate Offset = P.second;
3843 LSRUse &LU = Uses[LUIdx];
3844 LSRFixup &LF = LU.getNewFixup();
3845 LF.UserInst = const_cast<Instruction *>(UserInst);
3846 LF.OperandValToReplace = U;
3847 LF.Offset = Offset;
3848 LU.AllFixupsOutsideLoop &= LF.isUseFullyOutsideLoop(L);
3849 LU.AllFixupsUnconditional &= IsFixupExecutedEachIncrement(LF);
3850 InsertSupplementalFormula(US, LU, LUIdx);
3851 CountRegisters(LU.Formulae.back(), Uses.size() - 1);
3852 break;
3853 }
3854 }
3855 }
3856}
3857
3858/// Split S into subexpressions which can be pulled out into separate
3859/// registers. If C is non-null, multiply each subexpression by C.
3860///
3861/// Return remainder expression after factoring the subexpressions captured by
3862/// Ops. If Ops is complete, return NULL.
3863static const SCEV *CollectSubexprs(const SCEV *S, const SCEVConstant *C,
3865 const Loop *L,
3866 ScalarEvolution &SE,
3867 unsigned Depth = 0) {
3868 // Arbitrarily cap recursion to protect compile time.
3869 if (Depth >= 3)
3870 return S;
3871
3872 if (const SCEVAddExpr *Add = dyn_cast<SCEVAddExpr>(S)) {
3873 // Break out add operands.
3874 for (const SCEV *S : Add->operands()) {
3875 const SCEV *Remainder = CollectSubexprs(S, C, Ops, L, SE, Depth+1);
3876 if (Remainder)
3877 Ops.push_back(C ? SE.getMulExpr(C, Remainder).getPointer() : Remainder);
3878 }
3879 return nullptr;
3880 }
3881 const SCEV *Start, *Step;
3882 const SCEVConstant *Op0;
3883 const SCEV *Op1;
3884 if (match(S, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step)))) {
3885 // Split a non-zero base out of an addrec.
3886 if (Start->isZero())
3887 return S;
3888
3889 const SCEV *Remainder = CollectSubexprs(Start, C, Ops, L, SE, Depth + 1);
3890 // Split the non-zero AddRec unless it is part of a nested recurrence that
3891 // does not pertain to this loop.
3892 if (Remainder && (cast<SCEVAddRecExpr>(S)->getLoop() == L ||
3893 !isa<SCEVAddRecExpr>(Remainder))) {
3894 Ops.push_back(C ? SE.getMulExpr(C, Remainder).getPointer() : Remainder);
3895 Remainder = nullptr;
3896 }
3897 if (Remainder != Start) {
3898 if (!Remainder)
3899 Remainder = SE.getConstant(S->getType(), 0);
3900 return SE.getAddRecExpr(Remainder, Step,
3901 cast<SCEVAddRecExpr>(S)->getLoop(),
3902 // FIXME: AR->getNoWrapFlags(SCEV::FlagNW)
3904 }
3905 } else if (match(S, m_scev_Mul(m_SCEVConstant(Op0), m_SCEV(Op1)))) {
3906 // Break (C * (a + b + c)) into C*a + C*b + C*c.
3907 C = C ? cast<SCEVConstant>(SE.getMulExpr(C, Op0)) : Op0;
3908 const SCEV *Remainder = CollectSubexprs(Op1, C, Ops, L, SE, Depth + 1);
3909 if (Remainder)
3910 Ops.push_back(SE.getMulExpr(C, Remainder));
3911 return nullptr;
3912 }
3913 return S;
3914}
3915
3916/// Return true if the SCEV represents a value that may end up as a
3917/// post-increment operation.
3919 LSRUse &LU, const SCEV *S, const Loop *L,
3920 ScalarEvolution &SE) {
3921 if (LU.Kind != LSRUse::Address ||
3922 !LU.AccessTy.getType()->isIntOrIntVectorTy())
3923 return false;
3924 const SCEV *Start;
3925 if (!match(S, m_scev_AffineAddRec(m_SCEV(Start), m_SCEVConstant())))
3926 return false;
3927 // Check if a post-indexed load/store can be used.
3928 if (TTI.isIndexedLoadLegal(TTI.MIM_PostInc, S->getType()) ||
3929 TTI.isIndexedStoreLegal(TTI.MIM_PostInc, S->getType())) {
3930 if (!isa<SCEVConstant>(Start) && SE.isLoopInvariant(Start, L))
3931 return true;
3932 }
3933 return false;
3934}
3935
3936/// Helper function for LSRInstance::GenerateReassociations.
3937void LSRInstance::GenerateReassociationsImpl(LSRUse &LU, unsigned LUIdx,
3938 const Formula &Base,
3939 unsigned Depth, size_t Idx,
3940 bool IsScaledReg) {
3941 const SCEV *BaseReg = IsScaledReg ? Base.ScaledReg : Base.BaseRegs[Idx];
3942 // Don't generate reassociations for the base register of a value that
3943 // may generate a post-increment operator. The reason is that the
3944 // reassociations cause extra base+register formula to be created,
3945 // and possibly chosen, but the post-increment is more efficient.
3946 if (AMK == TTI::AMK_PostIndexed && mayUsePostIncMode(TTI, LU, BaseReg, L, SE))
3947 return;
3949 const SCEV *Remainder = CollectSubexprs(BaseReg, nullptr, AddOps, L, SE);
3950 if (Remainder)
3951 AddOps.push_back(Remainder);
3952
3953 if (AddOps.size() == 1)
3954 return;
3955
3957 JE = AddOps.end();
3958 J != JE; ++J) {
3959 // Loop-variant "unknown" values are uninteresting; we won't be able to
3960 // do anything meaningful with them.
3961 if (isa<SCEVUnknown>(*J) && !SE.isLoopInvariant(*J, L))
3962 continue;
3963
3964 // Don't pull a constant into a register if the constant could be folded
3965 // into an immediate field.
3966 if (isAlwaysFoldable(Opts, TTI, SE, LU.MinOffset, LU.MaxOffset, LU.Kind,
3967 LU.AccessTy, *J, Base.getNumRegs() > 1))
3968 continue;
3969
3970 // Collect all operands except *J.
3971 SmallVector<SCEVUse, 8> InnerAddOps(std::as_const(AddOps).begin(), J);
3972 InnerAddOps.append(std::next(J), std::as_const(AddOps).end());
3973
3974 // Don't leave just a constant behind in a register if the constant could
3975 // be folded into an immediate field.
3976 if (InnerAddOps.size() == 1 &&
3977 isAlwaysFoldable(Opts, TTI, SE, LU.MinOffset, LU.MaxOffset, LU.Kind,
3978 LU.AccessTy, InnerAddOps[0], Base.getNumRegs() > 1))
3979 continue;
3980
3981 const SCEV *InnerSum = SE.getAddExpr(InnerAddOps);
3982 if (InnerSum->isZero())
3983 continue;
3984 Formula F = Base;
3985
3986 if (F.UnfoldedOffset.isNonZero() && F.UnfoldedOffset.isScalable())
3987 continue;
3988
3989 // Add the remaining pieces of the add back into the new formula.
3990 const SCEVConstant *InnerSumSC = dyn_cast<SCEVConstant>(InnerSum);
3991 if (InnerSumSC && SE.getTypeSizeInBits(InnerSumSC->getType()) <= 64 &&
3992 TTI.isLegalAddImmediate((uint64_t)F.UnfoldedOffset.getFixedValue() +
3993 InnerSumSC->getValue()->getZExtValue())) {
3994 F.UnfoldedOffset =
3995 Immediate::getFixed((uint64_t)F.UnfoldedOffset.getFixedValue() +
3996 InnerSumSC->getValue()->getZExtValue());
3997 if (IsScaledReg) {
3998 F.ScaledReg = nullptr;
3999 F.Scale = 0;
4000 } else
4001 F.BaseRegs.erase(F.BaseRegs.begin() + Idx);
4002 } else if (IsScaledReg)
4003 F.ScaledReg = InnerSum;
4004 else
4005 F.BaseRegs[Idx] = InnerSum;
4006
4007 // Add J as its own register, or an unfolded immediate.
4008 const SCEVConstant *SC = dyn_cast<SCEVConstant>(*J);
4009 if (SC && SE.getTypeSizeInBits(SC->getType()) <= 64 &&
4010 TTI.isLegalAddImmediate((uint64_t)F.UnfoldedOffset.getFixedValue() +
4011 SC->getValue()->getZExtValue()))
4012 F.UnfoldedOffset =
4013 Immediate::getFixed((uint64_t)F.UnfoldedOffset.getFixedValue() +
4014 SC->getValue()->getZExtValue());
4015 else
4016 F.BaseRegs.push_back(*J);
4017 // We may have changed the number of register in base regs, adjust the
4018 // formula accordingly.
4019 F.canonicalize(*L);
4020
4021 if (InsertFormula(LU, LUIdx, F))
4022 // If that formula hadn't been seen before, recurse to find more like
4023 // it.
4024 // Add check on Log16(AddOps.size()) - same as Log2_32(AddOps.size()) >> 2)
4025 // Because just Depth is not enough to bound compile time.
4026 // This means that every time AddOps.size() is greater 16^x we will add
4027 // x to Depth.
4028 GenerateReassociations(LU, LUIdx, LU.Formulae.back(),
4029 Depth + 1 + (Log2_32(AddOps.size()) >> 2));
4030 }
4031}
4032
4033/// Split out subexpressions from adds and the bases of addrecs.
4034void LSRInstance::GenerateReassociations(LSRUse &LU, unsigned LUIdx,
4035 Formula Base, unsigned Depth) {
4036 assert(Base.isCanonical(*L) && "Input must be in the canonical form");
4037 // Arbitrarily cap recursion to protect compile time.
4038 if (Depth >= 3)
4039 return;
4040
4041 for (size_t i = 0, e = Base.BaseRegs.size(); i != e; ++i)
4042 GenerateReassociationsImpl(LU, LUIdx, Base, Depth, i);
4043
4044 if (Base.Scale == 1)
4045 GenerateReassociationsImpl(LU, LUIdx, Base, Depth,
4046 /* Idx */ -1, /* IsScaledReg */ true);
4047}
4048
4049/// Generate a formula consisting of all of the loop-dominating registers added
4050/// into a single register.
4051void LSRInstance::GenerateCombinations(LSRUse &LU, unsigned LUIdx,
4052 Formula Base) {
4053 // This method is only interesting on a plurality of registers.
4054 if (Base.BaseRegs.size() + (Base.Scale == 1) +
4055 (Base.UnfoldedOffset.isNonZero()) <=
4056 1)
4057 return;
4058
4059 // Flatten the representation, i.e., reg1 + 1*reg2 => reg1 + reg2, before
4060 // processing the formula.
4061 Base.unscale();
4063 Formula NewBase = Base;
4064 NewBase.BaseRegs.clear();
4065 Type *CombinedIntegerType = nullptr;
4066 for (const SCEV *BaseReg : Base.BaseRegs) {
4067 if (SE.properlyDominates(BaseReg, L->getHeader()) &&
4068 !SE.hasComputableLoopEvolution(BaseReg, L)) {
4069 if (!CombinedIntegerType)
4070 CombinedIntegerType = SE.getEffectiveSCEVType(BaseReg->getType());
4071 Ops.push_back(BaseReg);
4072 }
4073 else
4074 NewBase.BaseRegs.push_back(BaseReg);
4075 }
4076
4077 // If no register is relevant, we're done.
4078 if (Ops.size() == 0)
4079 return;
4080
4081 // Utility function for generating the required variants of the combined
4082 // registers.
4083 auto GenerateFormula = [&](const SCEV *Sum) {
4084 Formula F = NewBase;
4085
4086 // TODO: If Sum is zero, it probably means ScalarEvolution missed an
4087 // opportunity to fold something. For now, just ignore such cases
4088 // rather than proceed with zero in a register.
4089 if (Sum->isZero())
4090 return;
4091
4092 F.BaseRegs.push_back(Sum);
4093 F.canonicalize(*L);
4094 (void)InsertFormula(LU, LUIdx, F);
4095 };
4096
4097 // If we collected at least two registers, generate a formula combining them.
4098 if (Ops.size() > 1) {
4099 SmallVector<SCEVUse, 4> OpsCopy(Ops); // Don't let SE modify Ops.
4100 GenerateFormula(SE.getAddExpr(OpsCopy));
4101 }
4102
4103 // If we have an unfolded offset, generate a formula combining it with the
4104 // registers collected.
4105 if (NewBase.UnfoldedOffset.isNonZero() && NewBase.UnfoldedOffset.isFixed()) {
4106 assert(CombinedIntegerType && "Missing a type for the unfolded offset");
4107 Ops.push_back(SE.getConstant(CombinedIntegerType,
4108 NewBase.UnfoldedOffset.getFixedValue(), true));
4109 NewBase.UnfoldedOffset = Immediate::getFixed(0);
4110 GenerateFormula(SE.getAddExpr(Ops));
4111 }
4112}
4113
4114/// Helper function for LSRInstance::GenerateSymbolicOffsets.
4115void LSRInstance::GenerateSymbolicOffsetsImpl(LSRUse &LU, unsigned LUIdx,
4116 const Formula &Base, size_t Idx,
4117 bool IsScaledReg) {
4118 SCEVUse G = IsScaledReg ? Base.ScaledReg : Base.BaseRegs[Idx];
4119 GlobalValue *GV = ExtractSymbol(G, SE);
4120 if (G->isZero() || !GV)
4121 return;
4122 Formula F = Base;
4123 F.BaseGV = GV;
4124 if (!isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy, F))
4125 return;
4126 if (IsScaledReg)
4127 F.ScaledReg = G;
4128 else
4129 F.BaseRegs[Idx] = G;
4130 (void)InsertFormula(LU, LUIdx, F);
4131}
4132
4133/// Generate reuse formulae using symbolic offsets.
4134void LSRInstance::GenerateSymbolicOffsets(LSRUse &LU, unsigned LUIdx,
4135 Formula Base) {
4136 // We can't add a symbolic offset if the address already contains one.
4137 if (Base.BaseGV) return;
4138
4139 for (size_t i = 0, e = Base.BaseRegs.size(); i != e; ++i)
4140 GenerateSymbolicOffsetsImpl(LU, LUIdx, Base, i);
4141 if (Base.Scale == 1)
4142 GenerateSymbolicOffsetsImpl(LU, LUIdx, Base, /* Idx */ -1,
4143 /* IsScaledReg */ true);
4144}
4145
4146/// Helper function for LSRInstance::GenerateConstantOffsets.
4147void LSRInstance::GenerateConstantOffsetsImpl(
4148 LSRUse &LU, unsigned LUIdx, const Formula &Base,
4149 const SmallVectorImpl<Immediate> &Worklist, size_t Idx, bool IsScaledReg) {
4150
4151 auto GenerateOffset = [&](const SCEV *G, Immediate Offset) {
4152 Formula F = Base;
4153 if (!Base.BaseOffset.isCompatibleImmediate(Offset))
4154 return;
4155 F.BaseOffset = Base.BaseOffset.subUnsigned(Offset);
4156
4157 if (isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy, F)) {
4158 // Add the offset to the base register.
4159 const SCEV *NewOffset = Offset.getSCEV(SE, G->getType());
4160 const SCEV *NewG = SE.getAddExpr(NewOffset, G);
4161 // If it cancelled out, drop the base register, otherwise update it.
4162 if (NewG->isZero()) {
4163 if (IsScaledReg) {
4164 F.Scale = 0;
4165 F.ScaledReg = nullptr;
4166 } else
4167 F.deleteBaseReg(F.BaseRegs[Idx]);
4168 F.canonicalize(*L);
4169 } else if (IsScaledReg)
4170 F.ScaledReg = NewG;
4171 else
4172 F.BaseRegs[Idx] = NewG;
4173
4174 (void)InsertFormula(LU, LUIdx, F);
4175 }
4176 };
4177
4178 SCEVUse G = IsScaledReg ? Base.ScaledReg : Base.BaseRegs[Idx];
4179
4180 // With constant offsets and constant steps, we can generate pre-inc
4181 // accesses by having the offset equal the step. So, for access #0 with a
4182 // step of 8, we generate a G - 8 base which would require the first access
4183 // to be ((G - 8) + 8),+,8. The pre-indexed access then updates the pointer
4184 // for itself and hopefully becomes the base for other accesses. This means
4185 // means that a single pre-indexed access can be generated to become the new
4186 // base pointer for each iteration of the loop, resulting in no extra add/sub
4187 // instructions for pointer updating.
4188 if ((AMK & TTI::AMK_PreIndexed) && LU.Kind == LSRUse::Address) {
4189 const APInt *StepInt;
4190 if (match(G, m_scev_AffineAddRec(m_SCEV(), m_scev_APInt(StepInt)))) {
4191 int64_t Step = StepInt->isNegative() ? StepInt->getSExtValue()
4192 : StepInt->getZExtValue();
4193
4194 for (Immediate Offset : Worklist) {
4195 if (Offset.isFixed()) {
4196 Offset = Immediate::getFixed(Offset.getFixedValue() - Step);
4197 GenerateOffset(G, Offset);
4198 }
4199 }
4200 }
4201 }
4202 for (Immediate Offset : Worklist)
4203 GenerateOffset(G, Offset);
4204
4205 // TODO: It likely makes sense to extract the immediate corresponding to the
4206 // access type (i.e., set PreferScalable to AccessTy.MemTy &&
4207 // AccessTy.MemTy->isScalableTy()).
4208 Immediate Imm = extractImmediate(Opts, G, SE, /*PreferScalable=*/false);
4209 if (G->isZero() || Imm.isZero() ||
4210 !Base.BaseOffset.isCompatibleImmediate(Imm))
4211 return;
4212 Formula F = Base;
4213 F.BaseOffset = F.BaseOffset.addUnsigned(Imm);
4214 if (!isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy, F))
4215 return;
4216 if (IsScaledReg) {
4217 F.ScaledReg = G;
4218 } else {
4219 F.BaseRegs[Idx] = G;
4220 // We may generate non canonical Formula if G is a recurrent expr reg
4221 // related with current loop while F.ScaledReg is not.
4222 F.canonicalize(*L);
4223 }
4224 (void)InsertFormula(LU, LUIdx, F);
4225}
4226
4227/// GenerateConstantOffsets - Generate reuse formulae using symbolic offsets.
4228void LSRInstance::GenerateConstantOffsets(LSRUse &LU, unsigned LUIdx,
4229 Formula Base) {
4230 // TODO: For now, just add the min and max offset, because it usually isn't
4231 // worthwhile looking at everything inbetween.
4233 Worklist.push_back(LU.MinOffset);
4234 if (LU.MaxOffset != LU.MinOffset)
4235 Worklist.push_back(LU.MaxOffset);
4236
4237 for (size_t i = 0, e = Base.BaseRegs.size(); i != e; ++i)
4238 GenerateConstantOffsetsImpl(LU, LUIdx, Base, Worklist, i);
4239 if (Base.Scale == 1)
4240 GenerateConstantOffsetsImpl(LU, LUIdx, Base, Worklist, /* Idx */ -1,
4241 /* IsScaledReg */ true);
4242}
4243
4244/// For ICmpZero, check to see if we can scale up the comparison. For example, x
4245/// == y -> x*c == y*c.
4246void LSRInstance::GenerateICmpZeroScales(LSRUse &LU, unsigned LUIdx,
4247 Formula Base) {
4248 if (LU.Kind != LSRUse::ICmpZero) return;
4249
4250 // Determine the integer type for the base formula.
4251 Type *IntTy = Base.getType();
4252 if (!IntTy) return;
4253 if (SE.getTypeSizeInBits(IntTy) > 64) return;
4254
4255 // Don't do this if there is more than one offset.
4256 if (LU.MinOffset != LU.MaxOffset) return;
4257
4258 // Check if transformation is valid. It is illegal to multiply pointer.
4259 if (Base.ScaledReg && Base.ScaledReg->getType()->isPointerTy())
4260 return;
4261 for (const SCEV *BaseReg : Base.BaseRegs)
4262 if (BaseReg->getType()->isPointerTy())
4263 return;
4264 assert(!Base.BaseGV && "ICmpZero use is not legal!");
4265
4266 // Check each interesting stride.
4267 for (int64_t Factor : Factors) {
4268 // Check that Factor can be represented by IntTy
4269 if (!ConstantInt::isValueValidForType(IntTy, Factor))
4270 continue;
4271 // Check that the multiplication doesn't overflow.
4272 if (Base.BaseOffset.isMin() && Factor == -1)
4273 continue;
4274 // Not supporting scalable immediates.
4275 if (Base.BaseOffset.isNonZero() && Base.BaseOffset.isScalable())
4276 continue;
4277 Immediate NewBaseOffset = Base.BaseOffset.mulUnsigned(Factor);
4278 assert(Factor != 0 && "Zero factor not expected!");
4279 if (NewBaseOffset.getFixedValue() / Factor !=
4280 Base.BaseOffset.getFixedValue())
4281 continue;
4282 // If the offset will be truncated at this use, check that it is in bounds.
4283 if (!IntTy->isPointerTy() &&
4284 !ConstantInt::isValueValidForType(IntTy, NewBaseOffset.getFixedValue()))
4285 continue;
4286
4287 // Check that multiplying with the use offset doesn't overflow.
4288 Immediate Offset = LU.MinOffset;
4289 if (Offset.isMin() && Factor == -1)
4290 continue;
4291 Offset = Offset.mulUnsigned(Factor);
4292 if (Offset.getFixedValue() / Factor != LU.MinOffset.getFixedValue())
4293 continue;
4294 // If the offset will be truncated at this use, check that it is in bounds.
4295 if (!IntTy->isPointerTy() &&
4296 !ConstantInt::isValueValidForType(IntTy, Offset.getFixedValue()))
4297 continue;
4298
4299 Formula F = Base;
4300 F.BaseOffset = NewBaseOffset;
4301
4302 // Check that this scale is legal.
4303 if (!isLegalUse(TTI, Offset, Offset, LU.Kind, LU.AccessTy, F))
4304 continue;
4305
4306 // Compensate for the use having MinOffset built into it.
4307 F.BaseOffset = F.BaseOffset.addUnsigned(Offset).subUnsigned(LU.MinOffset);
4308
4309 const SCEV *FactorS = SE.getConstant(IntTy, Factor);
4310
4311 // Check that multiplying with each base register doesn't overflow.
4312 for (size_t i = 0, e = F.BaseRegs.size(); i != e; ++i) {
4313 F.BaseRegs[i] = SE.getMulExpr(F.BaseRegs[i], FactorS);
4314 if (getExactSDiv(F.BaseRegs[i], FactorS, SE) != Base.BaseRegs[i])
4315 goto next;
4316 }
4317
4318 // Check that multiplying with the scaled register doesn't overflow.
4319 if (F.ScaledReg) {
4320 F.ScaledReg = SE.getMulExpr(F.ScaledReg, FactorS);
4321 if (getExactSDiv(F.ScaledReg, FactorS, SE) != Base.ScaledReg)
4322 continue;
4323 }
4324
4325 // Check that multiplying with the unfolded offset doesn't overflow.
4326 if (F.UnfoldedOffset.isNonZero()) {
4327 if (F.UnfoldedOffset.isMin() && Factor == -1)
4328 continue;
4329 F.UnfoldedOffset = F.UnfoldedOffset.mulUnsigned(Factor);
4330 if (F.UnfoldedOffset.getFixedValue() / Factor !=
4331 Base.UnfoldedOffset.getFixedValue())
4332 continue;
4333 // If the offset will be truncated, check that it is in bounds.
4335 IntTy, F.UnfoldedOffset.getFixedValue()))
4336 continue;
4337 }
4338
4339 // If we make it here and it's legal, add it.
4340 (void)InsertFormula(LU, LUIdx, F);
4341 next:;
4342 }
4343}
4344
4345/// Generate stride factor reuse formulae by making use of scaled-offset address
4346/// modes, for example.
4347void LSRInstance::GenerateScales(LSRUse &LU, unsigned LUIdx, Formula Base) {
4348 // Determine the integer type for the base formula.
4349 Type *IntTy = Base.getType();
4350 if (!IntTy) return;
4351
4352 // If this Formula already has a scaled register, we can't add another one.
4353 // Try to unscale the formula to generate a better scale.
4354 if (Base.Scale != 0 && !Base.unscale())
4355 return;
4356
4357 assert(Base.Scale == 0 && "unscale did not did its job!");
4358
4359 // Check each interesting stride.
4360 for (int64_t Factor : Factors) {
4361 Base.Scale = Factor;
4362 Base.HasBaseReg = Base.BaseRegs.size() > 1;
4363 // Check whether this scale is going to be legal.
4364 if (!isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy,
4365 Base)) {
4366 // As a special-case, handle special out-of-loop Basic users specially.
4367 // TODO: Reconsider this special case.
4368 if (LU.Kind == LSRUse::Basic &&
4369 isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LSRUse::Special,
4370 LU.AccessTy, Base) &&
4371 LU.AllFixupsOutsideLoop)
4372 LU.Kind = LSRUse::Special;
4373 else
4374 continue;
4375 }
4376 // For an ICmpZero, negating a solitary base register won't lead to
4377 // new solutions.
4378 if (LU.Kind == LSRUse::ICmpZero && !Base.HasBaseReg &&
4379 Base.BaseOffset.isZero() && !Base.BaseGV)
4380 continue;
4381 // For each addrec base reg, if its loop is current loop, apply the scale.
4382 for (size_t i = 0, e = Base.BaseRegs.size(); i != e; ++i) {
4383 const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(Base.BaseRegs[i]);
4384 if (AR && (AR->getLoop() == L || LU.AllFixupsOutsideLoop)) {
4385 const SCEV *FactorS = SE.getConstant(IntTy, Factor);
4386 if (FactorS->isZero())
4387 continue;
4388 // Divide out the factor, ignoring high bits, since we'll be
4389 // scaling the value back up in the end.
4390 if (const SCEV *Quotient = getExactSDiv(AR, FactorS, SE, true))
4391 if (!Quotient->isZero()) {
4392 // TODO: This could be optimized to avoid all the copying.
4393 Formula F = Base;
4394 F.ScaledReg = Quotient;
4395 F.deleteBaseReg(F.BaseRegs[i]);
4396 // The canonical representation of 1*reg is reg, which is already in
4397 // Base. In that case, do not try to insert the formula, it will be
4398 // rejected anyway.
4399 if (F.Scale == 1 && (F.BaseRegs.empty() ||
4400 (AR->getLoop() != L && LU.AllFixupsOutsideLoop)))
4401 continue;
4402 // If AllFixupsOutsideLoop is true and F.Scale is 1, we may generate
4403 // non canonical Formula with ScaledReg's loop not being L.
4404 if (F.Scale == 1 && LU.AllFixupsOutsideLoop)
4405 F.canonicalize(*L);
4406 (void)InsertFormula(LU, LUIdx, F);
4407 }
4408 }
4409 }
4410 }
4411}
4412
4413/// Extend/Truncate \p Expr to \p ToTy considering post-inc uses in \p Loops.
4414/// For all PostIncLoopSets in \p Loops, first de-normalize \p Expr, then
4415/// perform the extension/truncate and normalize again, as the normalized form
4416/// can result in folds that are not valid in the post-inc use contexts. The
4417/// expressions for all PostIncLoopSets must match, otherwise return nullptr.
4418static const SCEV *
4420 const SCEV *Expr, Type *ToTy,
4421 ScalarEvolution &SE) {
4422 const SCEV *Result = nullptr;
4423 for (auto &L : Loops) {
4424 auto *DenormExpr = denormalizeForPostIncUse(Expr, L, SE);
4425 const SCEV *NewDenormExpr = SE.getAnyExtendExpr(DenormExpr, ToTy);
4426 const SCEV *New = normalizeForPostIncUse(NewDenormExpr, L, SE);
4427 if (!New || (Result && New != Result))
4428 return nullptr;
4429 Result = New;
4430 }
4431
4432 assert(Result && "failed to create expression");
4433 return Result;
4434}
4435
4436/// Generate reuse formulae from different IV types.
4437void LSRInstance::GenerateTruncates(LSRUse &LU, unsigned LUIdx, Formula Base) {
4438 // Don't bother truncating symbolic values.
4439 if (Base.BaseGV) return;
4440
4441 // Determine the integer type for the base formula.
4442 Type *DstTy = Base.getType();
4443 if (!DstTy) return;
4444 if (DstTy->isPointerTy())
4445 return;
4446
4447 // It is invalid to extend a pointer type so exit early if ScaledReg or
4448 // any of the BaseRegs are pointers.
4449 if (Base.ScaledReg && Base.ScaledReg->getType()->isPointerTy())
4450 return;
4451 if (any_of(Base.BaseRegs,
4452 [](const SCEV *S) { return S->getType()->isPointerTy(); }))
4453 return;
4454
4456 for (auto &LF : LU.Fixups)
4457 Loops.push_back(LF.PostIncLoops);
4458
4459 for (Type *SrcTy : Types) {
4460 if (SrcTy != DstTy && TTI.isTruncateFree(SrcTy, DstTy)) {
4461 Formula F = Base;
4462
4463 // Sometimes SCEV is able to prove zero during ext transform. It may
4464 // happen if SCEV did not do all possible transforms while creating the
4465 // initial node (maybe due to depth limitations), but it can do them while
4466 // taking ext.
4467 if (F.ScaledReg) {
4468 const SCEV *NewScaledReg =
4469 getAnyExtendConsideringPostIncUses(Loops, F.ScaledReg, SrcTy, SE);
4470 if (!NewScaledReg || NewScaledReg->isZero())
4471 continue;
4472 F.ScaledReg = NewScaledReg;
4473 }
4474 bool HasZeroBaseReg = false;
4475 for (const SCEV *&BaseReg : F.BaseRegs) {
4476 const SCEV *NewBaseReg =
4477 getAnyExtendConsideringPostIncUses(Loops, BaseReg, SrcTy, SE);
4478 if (!NewBaseReg || NewBaseReg->isZero()) {
4479 HasZeroBaseReg = true;
4480 break;
4481 }
4482 BaseReg = NewBaseReg;
4483 }
4484 if (HasZeroBaseReg)
4485 continue;
4486
4487 // TODO: This assumes we've done basic processing on all uses and
4488 // have an idea what the register usage is.
4489 if (!F.hasRegsUsedByUsesOtherThan(LUIdx, RegUses))
4490 continue;
4491
4492 F.canonicalize(*L);
4493 (void)InsertFormula(LU, LUIdx, F);
4494 }
4495 }
4496}
4497
4498namespace {
4499
4500/// Helper class for GenerateCrossUseConstantOffsets. It's used to defer
4501/// modifications so that the search phase doesn't have to worry about the data
4502/// structures moving underneath it.
4503struct WorkItem {
4504 size_t LUIdx;
4505 Immediate Imm;
4506 const SCEV *OrigReg;
4507
4508 WorkItem(size_t LI, Immediate I, const SCEV *R)
4509 : LUIdx(LI), Imm(I), OrigReg(R) {}
4510
4511 void print(raw_ostream &OS) const;
4512 void dump() const;
4513};
4514
4515} // end anonymous namespace
4516
4517#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4518void WorkItem::print(raw_ostream &OS) const {
4519 OS << "in formulae referencing " << *OrigReg << " in use " << LUIdx
4520 << " , add offset " << Imm;
4521}
4522
4523LLVM_DUMP_METHOD void WorkItem::dump() const {
4524 print(errs()); errs() << '\n';
4525}
4526#endif
4527
4528/// Look for registers which are a constant distance apart and try to form reuse
4529/// opportunities between them.
4530void LSRInstance::GenerateCrossUseConstantOffsets() {
4531 // Group the registers by their value without any added constant offset.
4532 using ImmMapTy = std::map<Immediate, const SCEV *, KeyOrderTargetImmediate>;
4533
4534 DenseMap<const SCEV *, ImmMapTy> Map;
4535 DenseMap<const SCEV *, SmallBitVector> UsedByIndicesMap;
4537 for (const SCEV *Use : RegUses) {
4538 SCEVUse Reg = Use; // Make a copy for extractImmediate to modify.
4539 // TODO: Extract both scalable and fixed immediates (if present)?
4540 Immediate Imm = extractImmediate(Opts, Reg, SE);
4541 auto Pair = Map.try_emplace(Reg);
4542 if (Pair.second)
4543 Sequence.push_back(Reg);
4544 Pair.first->second.insert(std::make_pair(Imm, Use));
4545 UsedByIndicesMap[Reg] |= RegUses.getUsedByIndices(Use);
4546 }
4547
4548 // Now examine each set of registers with the same base value. Build up
4549 // a list of work to do and do the work in a separate step so that we're
4550 // not adding formulae and register counts while we're searching.
4551 SmallVector<WorkItem, 32> WorkItems;
4552 SmallSet<std::pair<size_t, Immediate>, 32, KeyOrderSizeTAndImmediate>
4553 UniqueItems;
4554 for (const SCEV *Reg : Sequence) {
4555 const ImmMapTy &Imms = Map.find(Reg)->second;
4556
4557 // It's not worthwhile looking for reuse if there's only one offset.
4558 if (Imms.size() == 1)
4559 continue;
4560
4561 LLVM_DEBUG(dbgs() << "Generating cross-use offsets for " << *Reg << ':';
4562 for (const auto &Entry
4563 : Imms) dbgs()
4564 << ' ' << Entry.first;
4565 dbgs() << '\n');
4566
4567 // Examine each offset.
4568 for (ImmMapTy::const_iterator J = Imms.begin(), JE = Imms.end();
4569 J != JE; ++J) {
4570 const SCEV *OrigReg = J->second;
4571
4572 Immediate JImm = J->first;
4573 const SmallBitVector &UsedByIndices = RegUses.getUsedByIndices(OrigReg);
4574
4575 if (!isa<SCEVConstant>(OrigReg) &&
4576 UsedByIndicesMap[Reg].count() == 1) {
4577 LLVM_DEBUG(dbgs() << "Skipping cross-use reuse for " << *OrigReg
4578 << '\n');
4579 continue;
4580 }
4581
4582 // Conservatively examine offsets between this orig reg a few selected
4583 // other orig regs.
4584 Immediate First = Imms.begin()->first;
4585 Immediate Last = std::prev(Imms.end())->first;
4586 if (!First.isCompatibleImmediate(Last)) {
4587 LLVM_DEBUG(dbgs() << "Skipping cross-use reuse for " << *OrigReg
4588 << "\n");
4589 continue;
4590 }
4591 // Only scalable if both terms are scalable, or if one is scalable and
4592 // the other is 0.
4593 bool Scalable = First.isScalable() || Last.isScalable();
4594 int64_t FI = First.getKnownMinValue();
4595 int64_t LI = Last.getKnownMinValue();
4596 // Compute (First + Last) / 2 without overflow using the fact that
4597 // First + Last = 2 * (First + Last) + (First ^ Last).
4598 int64_t Avg = (FI & LI) + ((FI ^ LI) >> 1);
4599 // If the result is negative and FI is odd and LI even (or vice versa),
4600 // we rounded towards -inf. Add 1 in that case, to round towards 0.
4601 Avg = Avg + ((FI ^ LI) & ((uint64_t)Avg >> 63));
4602 ImmMapTy::const_iterator OtherImms[] = {
4603 Imms.begin(), std::prev(Imms.end()),
4604 Imms.lower_bound(Immediate::get(Avg, Scalable))};
4605 for (const auto &M : OtherImms) {
4606 if (M == J || M == JE) continue;
4607 if (!JImm.isCompatibleImmediate(M->first))
4608 continue;
4609
4610 // Compute the difference between the two.
4611 Immediate Imm = JImm.subUnsigned(M->first);
4612 for (unsigned LUIdx : UsedByIndices.set_bits())
4613 // Make a memo of this use, offset, and register tuple.
4614 if (UniqueItems.insert(std::make_pair(LUIdx, Imm)).second)
4615 WorkItems.push_back(WorkItem(LUIdx, Imm, OrigReg));
4616 }
4617 }
4618 }
4619
4620 Map.clear();
4621 Sequence.clear();
4622 UsedByIndicesMap.clear();
4623 UniqueItems.clear();
4624
4625 // Now iterate through the worklist and add new formulae.
4626 for (const WorkItem &WI : WorkItems) {
4627 size_t LUIdx = WI.LUIdx;
4628 LSRUse &LU = Uses[LUIdx];
4629 Immediate Imm = WI.Imm;
4630 const SCEV *OrigReg = WI.OrigReg;
4631
4632 Type *IntTy = SE.getEffectiveSCEVType(OrigReg->getType());
4633 const SCEV *NegImmS = Imm.getNegativeSCEV(SE, IntTy);
4634 unsigned BitWidth = SE.getTypeSizeInBits(IntTy);
4635
4636 // TODO: Use a more targeted data structure.
4637 for (size_t L = 0, LE = LU.Formulae.size(); L != LE; ++L) {
4638 Formula F = LU.Formulae[L];
4639 // FIXME: The code for the scaled and unscaled registers looks
4640 // very similar but slightly different. Investigate if they
4641 // could be merged. That way, we would not have to unscale the
4642 // Formula.
4643 F.unscale();
4644 // Use the immediate in the scaled register.
4645 if (F.ScaledReg == OrigReg) {
4646 if (!F.BaseOffset.isCompatibleImmediate(Imm))
4647 continue;
4648 Immediate Offset = F.BaseOffset.addUnsigned(Imm.mulUnsigned(F.Scale));
4649 // Don't create 50 + reg(-50).
4650 const SCEV *S = Offset.getNegativeSCEV(SE, IntTy);
4651 if (F.referencesReg(S))
4652 continue;
4653 Formula NewF = F;
4654 NewF.BaseOffset = Offset;
4655 if (!isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy,
4656 NewF))
4657 continue;
4658 NewF.ScaledReg = SE.getAddExpr(NegImmS, NewF.ScaledReg);
4659
4660 // If the new scale is a constant in a register, and adding the constant
4661 // value to the immediate would produce a value closer to zero than the
4662 // immediate itself, then the formula isn't worthwhile.
4663 if (const SCEVConstant *C = dyn_cast<SCEVConstant>(NewF.ScaledReg)) {
4664 // FIXME: Do we need to do something for scalable immediates here?
4665 // A scalable SCEV won't be constant, but we might still have
4666 // something in the offset? Bail out for now to be safe.
4667 if (NewF.BaseOffset.isNonZero() && NewF.BaseOffset.isScalable())
4668 continue;
4669 if (C->getValue()->isNegative() !=
4670 (NewF.BaseOffset.isLessThanZero()) &&
4671 (C->getAPInt().abs() * APInt(BitWidth, F.Scale))
4672 .ule(std::abs(NewF.BaseOffset.getFixedValue())))
4673 continue;
4674 }
4675
4676 // OK, looks good.
4677 NewF.canonicalize(*this->L);
4678 (void)InsertFormula(LU, LUIdx, NewF);
4679 } else {
4680 // Use the immediate in a base register.
4681 for (size_t N = 0, NE = F.BaseRegs.size(); N != NE; ++N) {
4682 const SCEV *BaseReg = F.BaseRegs[N];
4683 if (BaseReg != OrigReg)
4684 continue;
4685 Formula NewF = F;
4686 if (!NewF.BaseOffset.isCompatibleImmediate(Imm) ||
4687 !NewF.UnfoldedOffset.isCompatibleImmediate(Imm) ||
4688 !NewF.BaseOffset.isCompatibleImmediate(NewF.UnfoldedOffset))
4689 continue;
4690 NewF.BaseOffset = NewF.BaseOffset.addUnsigned(Imm);
4691 if (!isLegalUse(TTI, LU.MinOffset, LU.MaxOffset,
4692 LU.Kind, LU.AccessTy, NewF)) {
4693 if (AMK == TTI::AMK_PostIndexed &&
4694 mayUsePostIncMode(TTI, LU, OrigReg, this->L, SE))
4695 continue;
4696 Immediate NewUnfoldedOffset = NewF.UnfoldedOffset.addUnsigned(Imm);
4697 if (!isLegalAddImmediate(TTI, NewUnfoldedOffset))
4698 continue;
4699 NewF = F;
4700 NewF.UnfoldedOffset = NewUnfoldedOffset;
4701 }
4702 NewF.BaseRegs[N] = SE.getAddExpr(NegImmS, BaseReg);
4703
4704 // If the new formula has a constant in a register, and adding the
4705 // constant value to the immediate would produce a value closer to
4706 // zero than the immediate itself, then the formula isn't worthwhile.
4707 for (const SCEV *NewReg : NewF.BaseRegs)
4708 if (const SCEVConstant *C = dyn_cast<SCEVConstant>(NewReg)) {
4709 if (NewF.BaseOffset.isNonZero() && NewF.BaseOffset.isScalable())
4710 goto skip_formula;
4711 if ((C->getAPInt() + NewF.BaseOffset.getFixedValue())
4712 .abs()
4713 .slt(std::abs(NewF.BaseOffset.getFixedValue())) &&
4714 (C->getAPInt() + NewF.BaseOffset.getFixedValue())
4715 .countr_zero() >=
4717 NewF.BaseOffset.getFixedValue()))
4718 goto skip_formula;
4719 }
4720
4721 // Ok, looks good.
4722 NewF.canonicalize(*this->L);
4723 (void)InsertFormula(LU, LUIdx, NewF);
4724 break;
4725 skip_formula:;
4726 }
4727 }
4728 }
4729 }
4730}
4731
4732/// Generate formulae for each use.
4733void
4734LSRInstance::GenerateAllReuseFormulae() {
4735 // This is split into multiple loops so that hasRegsUsedByUsesOtherThan
4736 // queries are more precise.
4737 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4738 LSRUse &LU = Uses[LUIdx];
4739 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4740 GenerateReassociations(LU, LUIdx, LU.Formulae[i]);
4741 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4742 GenerateCombinations(LU, LUIdx, LU.Formulae[i]);
4743 }
4744 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4745 LSRUse &LU = Uses[LUIdx];
4746 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4747 GenerateSymbolicOffsets(LU, LUIdx, LU.Formulae[i]);
4748 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4749 GenerateConstantOffsets(LU, LUIdx, LU.Formulae[i]);
4750 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4751 GenerateICmpZeroScales(LU, LUIdx, LU.Formulae[i]);
4752 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4753 GenerateScales(LU, LUIdx, LU.Formulae[i]);
4754 }
4755 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4756 LSRUse &LU = Uses[LUIdx];
4757 for (size_t i = 0, f = LU.Formulae.size(); i != f; ++i)
4758 GenerateTruncates(LU, LUIdx, LU.Formulae[i]);
4759 }
4760
4761 GenerateCrossUseConstantOffsets();
4762
4763 LLVM_DEBUG(dbgs() << "\n"
4764 "After generating reuse formulae:\n";
4765 print_uses(dbgs()));
4766}
4767
4768/// If there are multiple formulae with the same set of registers used
4769/// by other uses, pick the best one and delete the others.
4770void LSRInstance::FilterOutUndesirableDedicatedRegisters() {
4771 DenseSet<const SCEV *> VisitedRegs;
4772 SmallPtrSet<const SCEV *, 16> Regs;
4773 SmallPtrSet<const SCEV *, 16> LoserRegs;
4774#ifndef NDEBUG
4775 bool ChangedFormulae = false;
4776#endif
4777
4778 // Collect the best formula for each unique set of shared registers. This
4779 // is reset for each use.
4780 using BestFormulaeTy = DenseMap<SmallVector<const SCEV *, 4>, size_t>;
4781
4782 BestFormulaeTy BestFormulae;
4783
4784 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4785 LSRUse &LU = Uses[LUIdx];
4786 LLVM_DEBUG(dbgs() << "Filtering for use "; LU.print(dbgs());
4787 dbgs() << '\n');
4788
4789 bool Any = false;
4790 for (size_t FIdx = 0, NumForms = LU.Formulae.size();
4791 FIdx != NumForms; ++FIdx) {
4792 Formula &F = LU.Formulae[FIdx];
4793
4794 // Some formulas are instant losers. For example, they may depend on
4795 // nonexistent AddRecs from other loops. These need to be filtered
4796 // immediately, otherwise heuristics could choose them over others leading
4797 // to an unsatisfactory solution. Passing LoserRegs into RateFormula here
4798 // avoids the need to recompute this information across formulae using the
4799 // same bad AddRec. Passing LoserRegs is also essential unless we remove
4800 // the corresponding bad register from the Regs set.
4801 Cost CostF(Opts, L, SE, TTI, AMK);
4802 Regs.clear();
4803 CostF.RateFormula(F, Regs, VisitedRegs, LU, HardwareLoopProfitable,
4804 &LoserRegs);
4805 if (CostF.isLoser()) {
4806 // During initial formula generation, undesirable formulae are generated
4807 // by uses within other loops that have some non-trivial address mode or
4808 // use the postinc form of the IV. LSR needs to provide these formulae
4809 // as the basis of rediscovering the desired formula that uses an AddRec
4810 // corresponding to the existing phi. Once all formulae have been
4811 // generated, these initial losers may be pruned.
4812 LLVM_DEBUG(dbgs() << " Filtering loser "; F.print(dbgs());
4813 dbgs() << "\n");
4814 }
4815 else {
4817 for (const SCEV *Reg : F.BaseRegs) {
4818 if (RegUses.isRegUsedByUsesOtherThan(Reg, LUIdx))
4819 Key.push_back(Reg);
4820 }
4821 if (F.ScaledReg &&
4822 RegUses.isRegUsedByUsesOtherThan(F.ScaledReg, LUIdx))
4823 Key.push_back(F.ScaledReg);
4824 // Unstable sort by host order ok, because this is only used for
4825 // uniquifying.
4826 llvm::sort(Key);
4827
4828 std::pair<BestFormulaeTy::const_iterator, bool> P =
4829 BestFormulae.insert(std::make_pair(Key, FIdx));
4830 if (P.second)
4831 continue;
4832
4833 Formula &Best = LU.Formulae[P.first->second];
4834
4835 Cost CostBest(Opts, L, SE, TTI, AMK);
4836 Regs.clear();
4837 CostBest.RateFormula(Best, Regs, VisitedRegs, LU,
4838 HardwareLoopProfitable);
4839 if (CostF.isLess(CostBest))
4840 std::swap(F, Best);
4841 LLVM_DEBUG(dbgs() << " Filtering out formula "; F.print(dbgs());
4842 dbgs() << "\n"
4843 " in favor of formula ";
4844 Best.print(dbgs()); dbgs() << '\n');
4845 }
4846#ifndef NDEBUG
4847 ChangedFormulae = true;
4848#endif
4849 LU.DeleteFormula(F);
4850 --FIdx;
4851 --NumForms;
4852 Any = true;
4853 }
4854
4855 // Now that we've filtered out some formulae, recompute the Regs set.
4856 if (Any)
4857 LU.RecomputeRegs(LUIdx, RegUses);
4858
4859 // Reset this to prepare for the next use.
4860 BestFormulae.clear();
4861 }
4862
4863 LLVM_DEBUG(if (ChangedFormulae) {
4864 dbgs() << "\n"
4865 "After filtering out undesirable candidates:\n";
4866 print_uses(dbgs());
4867 });
4868}
4869
4870/// Estimate the worst-case number of solutions the solver might have to
4871/// consider. It almost never considers this many solutions because it prune the
4872/// search space, but the pruning isn't always sufficient.
4873size_t LSRInstance::EstimateSearchSpaceComplexity() const {
4874 size_t Power = 1;
4875 for (const LSRUse &LU : Uses) {
4876 size_t FSize = LU.Formulae.size();
4877 if (FSize >= Opts.lsr_complexity_limit) {
4878 Power = Opts.lsr_complexity_limit;
4879 break;
4880 }
4881 Power *= FSize;
4882 if (Power >= Opts.lsr_complexity_limit)
4883 break;
4884 }
4885 return Power;
4886}
4887
4888/// When one formula uses a superset of the registers of another formula, it
4889/// won't help reduce register pressure (though it may not necessarily hurt
4890/// register pressure); remove it to simplify the system.
4891void LSRInstance::NarrowSearchSpaceByDetectingSupersets() {
4892 if (EstimateSearchSpaceComplexity() >= Opts.lsr_complexity_limit) {
4893 LLVM_DEBUG(dbgs() << "The search space is too complex.\n");
4894
4895 LLVM_DEBUG(dbgs() << "Narrowing the search space by eliminating formulae "
4896 "which use a superset of registers used by other "
4897 "formulae.\n");
4898
4899 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4900 LSRUse &LU = Uses[LUIdx];
4901 bool Any = false;
4902 for (size_t i = 0, e = LU.Formulae.size(); i != e; ++i) {
4903 Formula &F = LU.Formulae[i];
4904 if (F.BaseOffset.isNonZero() && F.BaseOffset.isScalable())
4905 continue;
4906 // Look for a formula with a constant or GV in a register. If the use
4907 // also has a formula with that same value in an immediate field,
4908 // delete the one that uses a register.
4910 I = F.BaseRegs.begin(), E = F.BaseRegs.end(); I != E; ++I) {
4911 if (const SCEVConstant *C = dyn_cast<SCEVConstant>(*I)) {
4912 Formula NewF = F;
4913 //FIXME: Formulas should store bitwidth to do wrapping properly.
4914 // See PR41034.
4915 NewF.BaseOffset =
4916 Immediate::getFixed(NewF.BaseOffset.getFixedValue() +
4917 (uint64_t)C->getValue()->getSExtValue());
4918 NewF.BaseRegs.erase(NewF.BaseRegs.begin() +
4919 (I - F.BaseRegs.begin()));
4920 if (LU.HasFormulaWithSameRegs(NewF)) {
4921 LLVM_DEBUG(dbgs() << " Deleting "; F.print(dbgs());
4922 dbgs() << '\n');
4923 LU.DeleteFormula(F);
4924 --i;
4925 --e;
4926 Any = true;
4927 break;
4928 }
4929 } else if (const SCEVUnknown *U = dyn_cast<SCEVUnknown>(*I)) {
4930 if (GlobalValue *GV = dyn_cast<GlobalValue>(U->getValue()))
4931 if (!F.BaseGV) {
4932 Formula NewF = F;
4933 NewF.BaseGV = GV;
4934 NewF.BaseRegs.erase(NewF.BaseRegs.begin() +
4935 (I - F.BaseRegs.begin()));
4936 if (LU.HasFormulaWithSameRegs(NewF)) {
4937 LLVM_DEBUG(dbgs() << " Deleting "; F.print(dbgs());
4938 dbgs() << '\n');
4939 LU.DeleteFormula(F);
4940 --i;
4941 --e;
4942 Any = true;
4943 break;
4944 }
4945 }
4946 }
4947 }
4948 }
4949 if (Any)
4950 LU.RecomputeRegs(LUIdx, RegUses);
4951 }
4952
4953 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
4954 }
4955}
4956
4957/// When there are many registers for expressions like A, A+1, A+2, etc.,
4958/// allocate a single register for them.
4959void LSRInstance::NarrowSearchSpaceByCollapsingUnrolledCode() {
4960 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
4961 return;
4962
4963 LLVM_DEBUG(
4964 dbgs() << "The search space is too complex.\n"
4965 "Narrowing the search space by assuming that uses separated "
4966 "by a constant offset will use the same registers.\n");
4967
4968 // This is especially useful for unrolled loops.
4969
4970 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
4971 LSRUse &LU = Uses[LUIdx];
4972 for (const Formula &F : LU.Formulae) {
4973 if (F.BaseOffset.isZero() || (F.Scale != 0 && F.Scale != 1))
4974 continue;
4975 assert((LU.Kind == LSRUse::Address || LU.Kind == LSRUse::ICmpZero) &&
4976 "Only address and cmp uses expected to have nonzero BaseOffset");
4977
4978 LSRUse *LUThatHas = FindUseWithSimilarFormula(F, LU);
4979 if (!LUThatHas)
4980 continue;
4981
4982 if (!reconcileNewOffset(*LUThatHas, F.BaseOffset, /*HasBaseReg=*/ false,
4983 LU.Kind, LU.AccessTy))
4984 continue;
4985
4986 LLVM_DEBUG(dbgs() << " Deleting use "; LU.print(dbgs()); dbgs() << '\n');
4987
4988 LUThatHas->AllFixupsOutsideLoop &= LU.AllFixupsOutsideLoop;
4989 LUThatHas->AllFixupsUnconditional &= LU.AllFixupsUnconditional;
4990
4991 // Transfer the fixups of LU to LUThatHas.
4992 for (LSRFixup &Fixup : LU.Fixups) {
4993 Fixup.Offset += F.BaseOffset;
4994 LUThatHas->pushFixup(Fixup);
4995 LLVM_DEBUG(dbgs() << "New fixup has offset " << Fixup.Offset << '\n');
4996 }
4997
4998#ifndef NDEBUG
4999 Type *FixupType = LUThatHas->Fixups[0].OperandValToReplace->getType();
5000 for (LSRFixup &Fixup : LUThatHas->Fixups)
5001 assert(Fixup.OperandValToReplace->getType() == FixupType &&
5002 "Expected all fixups to have the same type");
5003#endif
5004
5005 // Delete formulae from the new use which are no longer legal.
5006 bool Any = false;
5007 for (size_t i = 0, e = LUThatHas->Formulae.size(); i != e; ++i) {
5008 Formula &F = LUThatHas->Formulae[i];
5009 if (!isLegalUse(TTI, LUThatHas->MinOffset, LUThatHas->MaxOffset,
5010 LUThatHas->Kind, LUThatHas->AccessTy, F)) {
5011 LLVM_DEBUG(dbgs() << " Deleting "; F.print(dbgs()); dbgs() << '\n');
5012 LUThatHas->DeleteFormula(F);
5013 --i;
5014 --e;
5015 Any = true;
5016 }
5017 }
5018
5019 if (Any)
5020 LUThatHas->RecomputeRegs(LUThatHas - &Uses.front(), RegUses);
5021
5022 // Delete the old use.
5023 DeleteUse(LU, LUIdx);
5024 --LUIdx;
5025 --NumUses;
5026 break;
5027 }
5028 }
5029
5030 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5031}
5032
5033/// Call FilterOutUndesirableDedicatedRegisters again, if necessary, now that
5034/// we've done more filtering, as it may be able to find more formulae to
5035/// eliminate.
5036void LSRInstance::NarrowSearchSpaceByRefilteringUndesirableDedicatedRegisters(){
5037 if (EstimateSearchSpaceComplexity() >= Opts.lsr_complexity_limit) {
5038 LLVM_DEBUG(dbgs() << "The search space is too complex.\n");
5039
5040 LLVM_DEBUG(dbgs() << "Narrowing the search space by re-filtering out "
5041 "undesirable dedicated registers.\n");
5042
5043 FilterOutUndesirableDedicatedRegisters();
5044
5045 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5046 }
5047}
5048
5049/// If a LSRUse has multiple formulae with the same ScaledReg and Scale.
5050/// Pick the best one and delete the others.
5051/// This narrowing heuristic is to keep as many formulae with different
5052/// Scale and ScaledReg pair as possible while narrowing the search space.
5053/// The benefit is that it is more likely to find out a better solution
5054/// from a formulae set with more Scale and ScaledReg variations than
5055/// a formulae set with the same Scale and ScaledReg. The picking winner
5056/// reg heuristic will often keep the formulae with the same Scale and
5057/// ScaledReg and filter others, and we want to avoid that if possible.
5058void LSRInstance::NarrowSearchSpaceByFilterFormulaWithSameScaledReg() {
5059 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
5060 return;
5061
5062 LLVM_DEBUG(
5063 dbgs() << "The search space is too complex.\n"
5064 "Narrowing the search space by choosing the best Formula "
5065 "from the Formulae with the same Scale and ScaledReg.\n");
5066
5067 // Map the "Scale * ScaledReg" pair to the best formula of current LSRUse.
5068 using BestFormulaeTy = DenseMap<std::pair<const SCEV *, int64_t>, size_t>;
5069
5070 BestFormulaeTy BestFormulae;
5071#ifndef NDEBUG
5072 bool ChangedFormulae = false;
5073#endif
5074 DenseSet<const SCEV *> VisitedRegs;
5075 SmallPtrSet<const SCEV *, 16> Regs;
5076
5077 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
5078 LSRUse &LU = Uses[LUIdx];
5079 LLVM_DEBUG(dbgs() << "Filtering for use "; LU.print(dbgs());
5080 dbgs() << '\n');
5081
5082 // Return true if Formula FA is better than Formula FB.
5083 auto IsBetterThan = [&](Formula &FA, Formula &FB) {
5084 // First we will try to choose the Formula with fewer new registers.
5085 // For a register used by current Formula, the more the register is
5086 // shared among LSRUses, the less we increase the register number
5087 // counter of the formula.
5088 size_t FARegNum = 0;
5089 for (const SCEV *Reg : FA.BaseRegs) {
5090 const SmallBitVector &UsedByIndices = RegUses.getUsedByIndices(Reg);
5091 FARegNum += (NumUses - UsedByIndices.count() + 1);
5092 }
5093 size_t FBRegNum = 0;
5094 for (const SCEV *Reg : FB.BaseRegs) {
5095 const SmallBitVector &UsedByIndices = RegUses.getUsedByIndices(Reg);
5096 FBRegNum += (NumUses - UsedByIndices.count() + 1);
5097 }
5098 if (FARegNum != FBRegNum)
5099 return FARegNum < FBRegNum;
5100
5101 // If the new register numbers are the same, choose the Formula with
5102 // less Cost.
5103 Cost CostFA(Opts, L, SE, TTI, AMK);
5104 Cost CostFB(Opts, L, SE, TTI, AMK);
5105 Regs.clear();
5106 CostFA.RateFormula(FA, Regs, VisitedRegs, LU, HardwareLoopProfitable);
5107 Regs.clear();
5108 CostFB.RateFormula(FB, Regs, VisitedRegs, LU, HardwareLoopProfitable);
5109 return CostFA.isLess(CostFB);
5110 };
5111
5112 bool Any = false;
5113 for (size_t FIdx = 0, NumForms = LU.Formulae.size(); FIdx != NumForms;
5114 ++FIdx) {
5115 Formula &F = LU.Formulae[FIdx];
5116 if (!F.ScaledReg)
5117 continue;
5118 auto P = BestFormulae.insert({{F.ScaledReg, F.Scale}, FIdx});
5119 if (P.second)
5120 continue;
5121
5122 Formula &Best = LU.Formulae[P.first->second];
5123 if (IsBetterThan(F, Best))
5124 std::swap(F, Best);
5125 LLVM_DEBUG(dbgs() << " Filtering out formula "; F.print(dbgs());
5126 dbgs() << "\n"
5127 " in favor of formula ";
5128 Best.print(dbgs()); dbgs() << '\n');
5129#ifndef NDEBUG
5130 ChangedFormulae = true;
5131#endif
5132 LU.DeleteFormula(F);
5133 --FIdx;
5134 --NumForms;
5135 Any = true;
5136 }
5137 if (Any)
5138 LU.RecomputeRegs(LUIdx, RegUses);
5139
5140 // Reset this to prepare for the next use.
5141 BestFormulae.clear();
5142 }
5143
5144 LLVM_DEBUG(if (ChangedFormulae) {
5145 dbgs() << "\n"
5146 "After filtering out undesirable candidates:\n";
5147 print_uses(dbgs());
5148 });
5149}
5150
5151/// If we are over the complexity limit, filter out any post-inc prefering
5152/// variables to only post-inc values.
5153void LSRInstance::NarrowSearchSpaceByFilterPostInc() {
5154 if (AMK != TTI::AMK_PostIndexed)
5155 return;
5156 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
5157 return;
5158
5159 LLVM_DEBUG(dbgs() << "The search space is too complex.\n"
5160 "Narrowing the search space by choosing the lowest "
5161 "register Formula for PostInc Uses.\n");
5162
5163 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
5164 LSRUse &LU = Uses[LUIdx];
5165
5166 if (LU.Kind != LSRUse::Address)
5167 continue;
5168 if (!TTI.isIndexedLoadLegal(TTI.MIM_PostInc, LU.AccessTy.getType()) &&
5169 !TTI.isIndexedStoreLegal(TTI.MIM_PostInc, LU.AccessTy.getType()))
5170 continue;
5171
5172 size_t MinRegs = std::numeric_limits<size_t>::max();
5173 for (const Formula &F : LU.Formulae)
5174 MinRegs = std::min(F.getNumRegs(), MinRegs);
5175
5176 bool Any = false;
5177 for (size_t FIdx = 0, NumForms = LU.Formulae.size(); FIdx != NumForms;
5178 ++FIdx) {
5179 Formula &F = LU.Formulae[FIdx];
5180 if (F.getNumRegs() > MinRegs) {
5181 LLVM_DEBUG(dbgs() << " Filtering out formula "; F.print(dbgs());
5182 dbgs() << "\n");
5183 LU.DeleteFormula(F);
5184 --FIdx;
5185 --NumForms;
5186 Any = true;
5187 }
5188 }
5189 if (Any)
5190 LU.RecomputeRegs(LUIdx, RegUses);
5191
5192 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
5193 break;
5194 }
5195
5196 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5197}
5198
5199void LSRInstance::NarrowSearchSpaceByMergingUsesOutsideLoop() {
5200 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
5201 return;
5202
5203 LLVM_DEBUG(
5204 dbgs() << "The search space is too complex.\n"
5205 "Narrowing the search space by merging uses with fixups "
5206 "entirely outside the loop with uses inside the loop.\n");
5207
5208 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
5209 LSRUse &LU = Uses[LUIdx];
5210 // Don't merge ICmpZero uses outside the loop, as ICmpZero needs to be
5211 // handled specially when expanding.
5212 if (!LU.AllFixupsOutsideLoop || LU.Formulae.empty() ||
5213 LU.Kind == LSRUse::ICmpZero)
5214 continue;
5215
5216 LLVM_DEBUG(dbgs() << " Trying to eliminate use "; LU.print(dbgs());
5217 dbgs() << '\n');
5218
5219 // Find a compatible LSRUse inside the loop that we could merge LU with
5220 LSRUse *LUToMergeWith = nullptr;
5221 const Formula &ThisF = LU.Formulae[0];
5222 for (LSRUse &OtherLU : Uses) {
5223 // Only merge with uses inside the loop
5224 if (OtherLU.AllFixupsOutsideLoop)
5225 continue;
5226 // Can't merge with ICmpZero uses as they're handled specially when
5227 // expanding
5228 if (OtherLU.Kind == LSRUse::ICmpZero)
5229 continue;
5230 // Can't merge with uses without any formulae
5231 if (OtherLU.Formulae.empty())
5232 continue;
5233 // Can't merge if LU's offsets aren't legal for all of OtherLU's formulae
5234 if (any_of(OtherLU.Formulae, [&](const Formula &F) {
5235 return !isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, OtherLU.Kind,
5236 OtherLU.AccessTy, F);
5237 }))
5238 continue;
5239 // We can merge with uses that have the same initial formula. We allow
5240 // merging of uses with different Kind and AccessTy which means that the
5241 // cost may end up being inaccurate, but it's also what we would have
5242 // gotten if we'd ignored uses outside the loop entirely.
5243 const Formula &OtherF = OtherLU.Formulae[0];
5244 if (ThisF.BaseRegs == OtherF.BaseRegs &&
5245 ThisF.ScaledReg == OtherF.ScaledReg &&
5246 ThisF.BaseGV == OtherF.BaseGV && ThisF.Scale == OtherF.Scale &&
5247 ThisF.UnfoldedOffset == OtherF.UnfoldedOffset &&
5248 ThisF.BaseOffset == OtherF.BaseOffset) {
5249 LUToMergeWith = &OtherLU;
5250 break;
5251 }
5252 }
5253 if (!LUToMergeWith)
5254 continue;
5255
5256 LLVM_DEBUG(dbgs() << " Merging with "; LUToMergeWith->print(dbgs());
5257 dbgs() << '\n');
5258
5259 // Copy fixups
5260 for (LSRFixup &Fixup : LU.Fixups) {
5261 LUToMergeWith->pushFixup(Fixup);
5262 }
5263
5264 // Delete the old use.
5265 DeleteUse(LU, LUIdx);
5266 --LUIdx;
5267 --NumUses;
5268 }
5269
5270 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5271}
5272
5273/// The function delete formulas with high registers number expectation.
5274/// Assuming we don't know the value of each formula (already delete
5275/// all inefficient), generate probability of not selecting for each
5276/// register.
5277/// For example,
5278/// Use1:
5279/// reg(a) + reg({0,+,1})
5280/// reg(a) + reg({-1,+,1}) + 1
5281/// reg({a,+,1})
5282/// Use2:
5283/// reg(b) + reg({0,+,1})
5284/// reg(b) + reg({-1,+,1}) + 1
5285/// reg({b,+,1})
5286/// Use3:
5287/// reg(c) + reg(b) + reg({0,+,1})
5288/// reg(c) + reg({b,+,1})
5289///
5290/// Probability of not selecting
5291/// Use1 Use2 Use3
5292/// reg(a) (1/3) * 1 * 1
5293/// reg(b) 1 * (1/3) * (1/2)
5294/// reg({0,+,1}) (2/3) * (2/3) * (1/2)
5295/// reg({-1,+,1}) (2/3) * (2/3) * 1
5296/// reg({a,+,1}) (2/3) * 1 * 1
5297/// reg({b,+,1}) 1 * (2/3) * (2/3)
5298/// reg(c) 1 * 1 * 0
5299///
5300/// Now count registers number mathematical expectation for each formula:
5301/// Note that for each use we exclude probability if not selecting for the use.
5302/// For example for Use1 probability for reg(a) would be just 1 * 1 (excluding
5303/// probabilty 1/3 of not selecting for Use1).
5304/// Use1:
5305/// reg(a) + reg({0,+,1}) 1 + 1/3 -- to be deleted
5306/// reg(a) + reg({-1,+,1}) + 1 1 + 4/9 -- to be deleted
5307/// reg({a,+,1}) 1
5308/// Use2:
5309/// reg(b) + reg({0,+,1}) 1/2 + 1/3 -- to be deleted
5310/// reg(b) + reg({-1,+,1}) + 1 1/2 + 2/3 -- to be deleted
5311/// reg({b,+,1}) 2/3
5312/// Use3:
5313/// reg(c) + reg(b) + reg({0,+,1}) 1 + 1/3 + 4/9 -- to be deleted
5314/// reg(c) + reg({b,+,1}) 1 + 2/3
5315void LSRInstance::NarrowSearchSpaceByDeletingCostlyFormulas() {
5316 if (EstimateSearchSpaceComplexity() < Opts.lsr_complexity_limit)
5317 return;
5318 // Ok, we have too many of formulae on our hands to conveniently handle.
5319 // Use a rough heuristic to thin out the list.
5320
5321 // Set of Regs wich will be 100% used in final solution.
5322 // Used in each formula of a solution (in example above this is reg(c)).
5323 // We can skip them in calculations.
5324 SmallPtrSet<const SCEV *, 4> UniqRegs;
5325 LLVM_DEBUG(dbgs() << "The search space is too complex.\n");
5326
5327 // Map each register to probability of not selecting
5328 DenseMap <const SCEV *, float> RegNumMap;
5329 for (const SCEV *Reg : RegUses) {
5330 if (UniqRegs.count(Reg))
5331 continue;
5332 float PNotSel = 1;
5333 for (const LSRUse &LU : Uses) {
5334 if (!LU.Regs.count(Reg))
5335 continue;
5336 float P = LU.getNotSelectedProbability(Reg);
5337 if (P != 0.0)
5338 PNotSel *= P;
5339 else
5340 UniqRegs.insert(Reg);
5341 }
5342 RegNumMap.insert(std::make_pair(Reg, PNotSel));
5343 }
5344
5345 LLVM_DEBUG(
5346 dbgs() << "Narrowing the search space by deleting costly formulas\n");
5347
5348 // Delete formulas where registers number expectation is high.
5349 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
5350 LSRUse &LU = Uses[LUIdx];
5351 // If nothing to delete - continue.
5352 if (LU.Formulae.size() < 2)
5353 continue;
5354 // This is temporary solution to test performance. Float should be
5355 // replaced with round independent type (based on integers) to avoid
5356 // different results for different target builds.
5357 float FMinRegNum = LU.Formulae[0].getNumRegs();
5358 float FMinARegNum = LU.Formulae[0].getNumRegs();
5359 size_t MinIdx = 0;
5360 for (size_t i = 0, e = LU.Formulae.size(); i != e; ++i) {
5361 Formula &F = LU.Formulae[i];
5362 float FRegNum = 0;
5363 float FARegNum = 0;
5364 for (const SCEV *BaseReg : F.BaseRegs) {
5365 if (UniqRegs.count(BaseReg))
5366 continue;
5367 FRegNum += RegNumMap[BaseReg] / LU.getNotSelectedProbability(BaseReg);
5368 if (isa<SCEVAddRecExpr>(BaseReg))
5369 FARegNum +=
5370 RegNumMap[BaseReg] / LU.getNotSelectedProbability(BaseReg);
5371 }
5372 if (const SCEV *ScaledReg = F.ScaledReg) {
5373 if (!UniqRegs.count(ScaledReg)) {
5374 FRegNum +=
5375 RegNumMap[ScaledReg] / LU.getNotSelectedProbability(ScaledReg);
5376 if (isa<SCEVAddRecExpr>(ScaledReg))
5377 FARegNum +=
5378 RegNumMap[ScaledReg] / LU.getNotSelectedProbability(ScaledReg);
5379 }
5380 }
5381 if (FMinRegNum > FRegNum ||
5382 (FMinRegNum == FRegNum && FMinARegNum > FARegNum)) {
5383 FMinRegNum = FRegNum;
5384 FMinARegNum = FARegNum;
5385 MinIdx = i;
5386 }
5387 }
5388 LLVM_DEBUG(dbgs() << " The formula "; LU.Formulae[MinIdx].print(dbgs());
5389 dbgs() << " with min reg num " << FMinRegNum << '\n');
5390 if (MinIdx != 0)
5391 std::swap(LU.Formulae[MinIdx], LU.Formulae[0]);
5392 while (LU.Formulae.size() != 1) {
5393 LLVM_DEBUG(dbgs() << " Deleting "; LU.Formulae.back().print(dbgs());
5394 dbgs() << '\n');
5395 LU.Formulae.pop_back();
5396 }
5397 LU.RecomputeRegs(LUIdx, RegUses);
5398 assert(LU.Formulae.size() == 1 && "Should be exactly 1 min regs formula");
5399 Formula &F = LU.Formulae[0];
5400 LLVM_DEBUG(dbgs() << " Leaving only "; F.print(dbgs()); dbgs() << '\n');
5401 // When we choose the formula, the regs become unique.
5402 UniqRegs.insert_range(F.BaseRegs);
5403 if (F.ScaledReg)
5404 UniqRegs.insert(F.ScaledReg);
5405 }
5406 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5407}
5408
5409// Check if Best and Reg are SCEVs separated by a constant amount C, and if so
5410// would the addressing offset +C would be legal where the negative offset -C is
5411// not.
5413 ScalarEvolution &SE, const SCEV *Best,
5414 const SCEV *Reg,
5415 MemAccessTy AccessType) {
5416 if (Best->getType() != Reg->getType() ||
5418 cast<SCEVAddRecExpr>(Best)->getLoop() !=
5419 cast<SCEVAddRecExpr>(Reg)->getLoop()))
5420 return false;
5421 std::optional<APInt> Diff = SE.computeConstantDifference(Best, Reg);
5422 if (!Diff)
5423 return false;
5424
5425 return TTI.isLegalAddressingMode(
5426 AccessType.MemTy, /*BaseGV=*/nullptr,
5427 /*BaseOffset=*/Diff->getSExtValue(),
5428 /*HasBaseReg=*/true, /*Scale=*/0, AccessType.AddrSpace) &&
5429 !TTI.isLegalAddressingMode(
5430 AccessType.MemTy, /*BaseGV=*/nullptr,
5431 /*BaseOffset=*/-Diff->getSExtValue(),
5432 /*HasBaseReg=*/true, /*Scale=*/0, AccessType.AddrSpace);
5433}
5434
5435/// Pick a register which seems likely to be profitable, and then in any use
5436/// which has any reference to that register, delete all formulae which do not
5437/// reference that register.
5438void LSRInstance::NarrowSearchSpaceByPickingWinnerRegs() {
5439 // With all other options exhausted, loop until the system is simple
5440 // enough to handle.
5441 SmallPtrSet<const SCEV *, 4> Taken;
5442 while (EstimateSearchSpaceComplexity() >= Opts.lsr_complexity_limit) {
5443 // Ok, we have too many of formulae on our hands to conveniently handle.
5444 // Use a rough heuristic to thin out the list.
5445 LLVM_DEBUG(dbgs() << "The search space is too complex.\n");
5446
5447 // Pick the register which is used by the most LSRUses, which is likely
5448 // to be a good reuse register candidate.
5449 const SCEV *Best = nullptr;
5450 unsigned BestNum = 0;
5451 for (const SCEV *Reg : RegUses) {
5452 if (Taken.count(Reg))
5453 continue;
5454 if (!Best) {
5455 Best = Reg;
5456 BestNum = RegUses.getUsedByIndices(Reg).count();
5457 } else {
5458 unsigned Count = RegUses.getUsedByIndices(Reg).count();
5459 if (Count > BestNum) {
5460 Best = Reg;
5461 BestNum = Count;
5462 }
5463
5464 // If the scores are the same, but the Reg is simpler for the target
5465 // (for example {x,+,1} as opposed to {x+C,+,1}, where the target can
5466 // handle +C but not -C), opt for the simpler formula.
5467 if (Count == BestNum) {
5468 int LUIdx = RegUses.getUsedByIndices(Reg).find_first();
5469 if (LUIdx >= 0 && Uses[LUIdx].Kind == LSRUse::Address &&
5471 Uses[LUIdx].AccessTy)) {
5472 Best = Reg;
5473 BestNum = Count;
5474 }
5475 }
5476 }
5477 }
5478 assert(Best && "Failed to find best LSRUse candidate");
5479
5480 LLVM_DEBUG(dbgs() << "Narrowing the search space by assuming " << *Best
5481 << " will yield profitable reuse.\n");
5482 Taken.insert(Best);
5483
5484 // In any use with formulae which references this register, delete formulae
5485 // which don't reference it.
5486 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx) {
5487 LSRUse &LU = Uses[LUIdx];
5488 if (!LU.Regs.count(Best)) continue;
5489
5490 bool Any = false;
5491 for (size_t i = 0, e = LU.Formulae.size(); i != e; ++i) {
5492 Formula &F = LU.Formulae[i];
5493 if (!F.referencesReg(Best)) {
5494 LLVM_DEBUG(dbgs() << " Deleting "; F.print(dbgs()); dbgs() << '\n');
5495 LU.DeleteFormula(F);
5496 --e;
5497 --i;
5498 Any = true;
5499 assert(e != 0 && "Use has no formulae left! Is Regs inconsistent?");
5500 continue;
5501 }
5502 }
5503
5504 if (Any)
5505 LU.RecomputeRegs(LUIdx, RegUses);
5506 }
5507
5508 LLVM_DEBUG(dbgs() << "After pre-selection:\n"; print_uses(dbgs()));
5509 }
5510}
5511
5512/// If there are an extraordinary number of formulae to choose from, use some
5513/// rough heuristics to prune down the number of formulae. This keeps the main
5514/// solver from taking an extraordinary amount of time in some worst-case
5515/// scenarios.
5516void LSRInstance::NarrowSearchSpaceUsingHeuristics() {
5517 NarrowSearchSpaceByDetectingSupersets();
5518 NarrowSearchSpaceByCollapsingUnrolledCode();
5519 NarrowSearchSpaceByRefilteringUndesirableDedicatedRegisters();
5520 if (Opts.lsr_filter_same_scaled_reg)
5521 NarrowSearchSpaceByFilterFormulaWithSameScaledReg();
5522 NarrowSearchSpaceByFilterPostInc();
5523 NarrowSearchSpaceByMergingUsesOutsideLoop();
5524 if (Opts.lsr_exp_narrow)
5525 NarrowSearchSpaceByDeletingCostlyFormulas();
5526 else
5527 NarrowSearchSpaceByPickingWinnerRegs();
5528}
5529
5530/// This is the recursive solver.
5531void LSRInstance::SolveRecurse(SmallVectorImpl<const Formula *> &Solution,
5532 Cost &SolutionCost,
5533 SmallVectorImpl<const Formula *> &Workspace,
5534 const Cost &CurCost,
5535 const SmallPtrSet<const SCEV *, 16> &CurRegs,
5536 DenseSet<const SCEV *> &VisitedRegs) const {
5537 // Some ideas:
5538 // - prune more:
5539 // - use more aggressive filtering
5540 // - sort the formula so that the most profitable solutions are found first
5541 // - sort the uses too
5542 // - search faster:
5543 // - don't compute a cost, and then compare. compare while computing a cost
5544 // and bail early.
5545 // - track register sets with SmallBitVector
5546
5547 const LSRUse &LU = Uses[Workspace.size()];
5548
5549 // If this use references any register that's already a part of the
5550 // in-progress solution, consider it a requirement that a formula must
5551 // reference that register in order to be considered. This prunes out
5552 // unprofitable searching.
5553 SmallSetVector<const SCEV *, 4> ReqRegs;
5554 for (const SCEV *S : CurRegs)
5555 if (LU.Regs.count(S))
5556 ReqRegs.insert(S);
5557
5558 SmallPtrSet<const SCEV *, 16> NewRegs;
5559 Cost NewCost(Opts, L, SE, TTI, AMK);
5560 for (const Formula &F : LU.Formulae) {
5561 // Ignore formulae which may not be ideal in terms of register reuse of
5562 // ReqRegs. The formula should use all required registers before
5563 // introducing new ones.
5564 // This can sometimes (notably when trying to favour postinc) lead to
5565 // sub-optimial decisions. There it is best left to the cost modelling to
5566 // get correct.
5567 if (!(AMK & TTI::AMK_PostIndexed) || LU.Kind != LSRUse::Address) {
5568 int NumReqRegsToFind = std::min(F.getNumRegs(), ReqRegs.size());
5569 for (const SCEV *Reg : ReqRegs) {
5570 if ((F.ScaledReg && F.ScaledReg == Reg) ||
5571 is_contained(F.BaseRegs, Reg)) {
5572 --NumReqRegsToFind;
5573 if (NumReqRegsToFind == 0)
5574 break;
5575 }
5576 }
5577 if (NumReqRegsToFind != 0) {
5578 // If none of the formulae satisfied the required registers, then we could
5579 // clear ReqRegs and try again. Currently, we simply give up in this case.
5580 continue;
5581 }
5582 }
5583
5584 // Evaluate the cost of the current formula. If it's already worse than
5585 // the current best, prune the search at that point.
5586 NewCost = CurCost;
5587 NewRegs = CurRegs;
5588 NewCost.RateFormula(F, NewRegs, VisitedRegs, LU, HardwareLoopProfitable);
5589 if (NewCost.isLess(SolutionCost)) {
5590 Workspace.push_back(&F);
5591 if (Workspace.size() != Uses.size()) {
5592 SolveRecurse(Solution, SolutionCost, Workspace, NewCost,
5593 NewRegs, VisitedRegs);
5594 if (F.getNumRegs() == 1 && Workspace.size() == 1)
5595 VisitedRegs.insert(F.ScaledReg ? F.ScaledReg : F.BaseRegs[0]);
5596 } else {
5597 LLVM_DEBUG(dbgs() << "New best at "; NewCost.print(dbgs());
5598 dbgs() << ".\nRegs:\n";
5599 for (const SCEV *S : NewRegs) dbgs()
5600 << "- " << *S << "\n";
5601 dbgs() << '\n');
5602
5603 SolutionCost = NewCost;
5604 Solution = Workspace;
5605 }
5606 Workspace.pop_back();
5607 }
5608 }
5609}
5610
5611/// Choose one formula from each use. Return the results in the given Solution
5612/// vector.
5613void LSRInstance::Solve(SmallVectorImpl<const Formula *> &Solution) const {
5615 Cost SolutionCost(Opts, L, SE, TTI, AMK);
5616 SolutionCost.Lose();
5617 Cost CurCost(Opts, L, SE, TTI, AMK);
5618 SmallPtrSet<const SCEV *, 16> CurRegs;
5619 DenseSet<const SCEV *> VisitedRegs;
5620 Workspace.reserve(Uses.size());
5621
5622 // SolveRecurse does all the work.
5623 SolveRecurse(Solution, SolutionCost, Workspace, CurCost,
5624 CurRegs, VisitedRegs);
5625 if (Solution.empty()) {
5626 LLVM_DEBUG(dbgs() << "\nNo Satisfactory Solution\n");
5627 return;
5628 }
5629
5630 // Ok, we've now made all our decisions.
5631 LLVM_DEBUG(dbgs() << "\n"
5632 "The chosen solution requires ";
5633 SolutionCost.print(dbgs()); dbgs() << ":\n";
5634 for (size_t i = 0, e = Uses.size(); i != e; ++i) {
5635 dbgs() << " ";
5636 Uses[i].print(dbgs());
5637 dbgs() << "\n"
5638 " ";
5639 Solution[i]->print(dbgs());
5640 dbgs() << '\n';
5641 });
5642
5643 assert(Solution.size() == Uses.size() && "Malformed solution!");
5644
5645 const bool EnableDropUnprofitableSolution = valueOr(
5646 Opts.lsr_drop_solution, TTI.shouldDropLSRSolutionIfLessProfitable());
5647
5648 if (BaselineCost.isLess(SolutionCost)) {
5649 if (!EnableDropUnprofitableSolution)
5650 LLVM_DEBUG(
5651 dbgs() << "Baseline is more profitable than chosen solution, "
5652 "add option 'lsr-drop-solution' to drop LSR solution.\n");
5653 else {
5654 LLVM_DEBUG(dbgs() << "Baseline is more profitable than chosen "
5655 "solution, dropping LSR solution.\n";);
5656 Solution.clear();
5657 }
5658 }
5659}
5660
5661/// Helper for AdjustInsertPositionForExpand. Climb up the dominator tree far as
5662/// we can go while still being dominated by the input positions. This helps
5663/// canonicalize the insert position, which encourages sharing.
5665LSRInstance::HoistInsertPosition(BasicBlock::iterator IP,
5666 const SmallVectorImpl<Instruction *> &Inputs)
5667 const {
5668 Instruction *Tentative = &*IP;
5669 while (true) {
5670 bool AllDominate = true;
5671 Instruction *BetterPos = nullptr;
5672 // Don't bother attempting to insert before a catchswitch, their basic block
5673 // cannot have other non-PHI instructions.
5674 if (isa<CatchSwitchInst>(Tentative))
5675 return IP;
5676
5677 for (Instruction *Inst : Inputs) {
5678 if (Inst == Tentative || !DT.dominates(Inst, Tentative)) {
5679 AllDominate = false;
5680 break;
5681 }
5682 // Attempt to find an insert position in the middle of the block,
5683 // instead of at the end, so that it can be used for other expansions.
5684 if (Tentative->getParent() == Inst->getParent() &&
5685 (!BetterPos || !DT.dominates(Inst, BetterPos)))
5686 BetterPos = &*std::next(BasicBlock::iterator(Inst));
5687 }
5688 if (!AllDominate)
5689 break;
5690 if (BetterPos)
5691 IP = BetterPos->getIterator();
5692 else
5693 IP = Tentative->getIterator();
5694
5695 const Loop *IPLoop = LI.getLoopFor(IP->getParent());
5696 unsigned IPLoopDepth = IPLoop ? IPLoop->getLoopDepth() : 0;
5697
5698 BasicBlock *IDom;
5699 for (DomTreeNode *Rung = DT.getNode(IP->getParent()); ; ) {
5700 if (!Rung) return IP;
5701 Rung = Rung->getIDom();
5702 if (!Rung) return IP;
5703 IDom = Rung->getBlock();
5704
5705 // Don't climb into a loop though.
5706 const Loop *IDomLoop = LI.getLoopFor(IDom);
5707 unsigned IDomDepth = IDomLoop ? IDomLoop->getLoopDepth() : 0;
5708 if (IDomDepth <= IPLoopDepth &&
5709 (IDomDepth != IPLoopDepth || IDomLoop == IPLoop))
5710 break;
5711 }
5712
5713 Tentative = IDom->getTerminator();
5714 }
5715
5716 return IP;
5717}
5718
5719/// Determine an input position which will be dominated by the operands and
5720/// which will dominate the result.
5721BasicBlock::iterator LSRInstance::AdjustInsertPositionForExpand(
5722 BasicBlock::iterator LowestIP, const LSRFixup &LF, const LSRUse &LU) const {
5723 // Collect some instructions which must be dominated by the
5724 // expanding replacement. These must be dominated by any operands that
5725 // will be required in the expansion.
5726 SmallVector<Instruction *, 4> Inputs;
5727 if (Instruction *I = dyn_cast<Instruction>(LF.OperandValToReplace))
5728 Inputs.push_back(I);
5729 if (LU.Kind == LSRUse::ICmpZero)
5730 if (Instruction *I =
5731 dyn_cast<Instruction>(cast<ICmpInst>(LF.UserInst)->getOperand(1)))
5732 Inputs.push_back(I);
5733 if (LF.PostIncLoops.count(L)) {
5734 if (LF.isUseFullyOutsideLoop(L))
5735 Inputs.push_back(L->getLoopLatch()->getTerminator());
5736 else
5737 Inputs.push_back(IVIncInsertPos);
5738 }
5739 // The expansion must also be dominated by the increment positions of any
5740 // loops it for which it is using post-inc mode.
5741 for (const Loop *PIL : LF.PostIncLoops) {
5742 if (PIL == L) continue;
5743
5744 // Be dominated by the loop exit.
5745 SmallVector<BasicBlock *, 4> ExitingBlocks;
5746 PIL->getExitingBlocks(ExitingBlocks);
5747 if (!ExitingBlocks.empty()) {
5748 BasicBlock *BB = ExitingBlocks[0];
5749 for (unsigned i = 1, e = ExitingBlocks.size(); i != e; ++i)
5750 BB = DT.findNearestCommonDominator(BB, ExitingBlocks[i]);
5751 Inputs.push_back(BB->getTerminator());
5752 }
5753 }
5754
5755 assert(!isa<PHINode>(LowestIP) && !LowestIP->isEHPad() &&
5756 "Insertion point must be a normal instruction");
5757
5758 // Then, climb up the immediate dominator tree as far as we can go while
5759 // still being dominated by the input positions.
5760 BasicBlock::iterator IP = HoistInsertPosition(LowestIP, Inputs);
5761
5762 // Don't insert instructions before PHI nodes.
5763 while (isa<PHINode>(IP)) ++IP;
5764
5765 // Ignore landingpad instructions.
5766 while (IP->isEHPad()) ++IP;
5767
5768 // Set IP below instructions recently inserted by SCEVExpander. This keeps the
5769 // IP consistent across expansions and allows the previously inserted
5770 // instructions to be reused by subsequent expansion.
5771 while (Rewriter.isInsertedInstruction(&*IP) && IP != LowestIP)
5772 ++IP;
5773
5774 return IP;
5775}
5776
5777/// Emit instructions for the leading candidate expression for this LSRUse (this
5778/// is called "expanding").
5779Value *LSRInstance::Expand(const LSRUse &LU, const LSRFixup &LF,
5780 const Formula &F, BasicBlock::iterator IP,
5781 SmallVectorImpl<WeakTrackingVH> &DeadInsts) const {
5782 if (LU.RigidFormula)
5783 return LF.OperandValToReplace;
5784
5785 // Determine an input position which will be dominated by the operands and
5786 // which will dominate the result.
5787 IP = AdjustInsertPositionForExpand(IP, LF, LU);
5788 Rewriter.setInsertPoint(&*IP);
5789
5790 // Inform the Rewriter if we have a post-increment use, so that it can
5791 // perform an advantageous expansion.
5792 Rewriter.setPostInc(LF.PostIncLoops);
5793
5794 // This is the type that the user actually needs.
5795 Type *OpTy = LF.OperandValToReplace->getType();
5796 // This will be the type that we'll initially expand to.
5797 Type *Ty = F.getType();
5798 if (!Ty)
5799 // No type known; just expand directly to the ultimate type.
5800 Ty = OpTy;
5801 else if (SE.getEffectiveSCEVType(Ty) == SE.getEffectiveSCEVType(OpTy))
5802 // Expand directly to the ultimate type if it's the right size.
5803 Ty = OpTy;
5804 // This is the type to do integer arithmetic in.
5805 Type *IntTy = SE.getEffectiveSCEVType(Ty);
5806 // For ICmpZero with pointer-typed operands, keep the comparison in the
5807 // integer domain to avoid generating inttoptr casts. Use IntTy (the
5808 // formula's arithmetic width) so that both icmp operands match even when
5809 // the IV is wider than the pointer.
5810 if (LU.Kind == LSRUse::ICmpZero && OpTy->isPointerTy()) {
5811 OpTy = IntTy;
5812 Ty = IntTy;
5813 }
5814
5815 // Build up a list of operands to add together to form the full base.
5817
5818 // Expand the BaseRegs portion.
5819 for (const SCEV *Reg : F.BaseRegs) {
5820 assert(!Reg->isZero() && "Zero allocated in a base register!");
5821
5822 // If we're expanding for a post-inc user, make the post-inc adjustment.
5823 Reg = denormalizeForPostIncUse(Reg, LF.PostIncLoops, SE);
5824 Ops.push_back(SE.getUnknown(Rewriter.expandCodeFor(Reg, nullptr)));
5825 }
5826
5827 // Expand the ScaledReg portion.
5828 Value *ICmpScaledV = nullptr;
5829 if (F.Scale != 0) {
5830 const SCEV *ScaledS = F.ScaledReg;
5831
5832 // If we're expanding for a post-inc user, make the post-inc adjustment.
5833 PostIncLoopSet &Loops = const_cast<PostIncLoopSet &>(LF.PostIncLoops);
5834 ScaledS = denormalizeForPostIncUse(ScaledS, Loops, SE);
5835
5836 if (LU.Kind == LSRUse::ICmpZero) {
5837 // Expand ScaleReg as if it was part of the base regs.
5838 if (F.Scale == 1)
5839 Ops.push_back(
5840 SE.getUnknown(Rewriter.expandCodeFor(ScaledS, nullptr)));
5841 else {
5842 // An interesting way of "folding" with an icmp is to use a negated
5843 // scale, which we'll implement by inserting it into the other operand
5844 // of the icmp.
5845 assert(F.Scale == -1 &&
5846 "The only scale supported by ICmpZero uses is -1!");
5847 ICmpScaledV = Rewriter.expandCodeFor(ScaledS, nullptr);
5848 }
5849 } else {
5850 // Otherwise just expand the scaled register and an explicit scale,
5851 // which is expected to be matched as part of the address.
5852
5853 // Flush the operand list to suppress SCEVExpander hoisting address modes.
5854 // Unless the addressing mode will not be folded.
5855 if (!Ops.empty() && LU.Kind == LSRUse::Address &&
5856 isAMCompletelyFolded(TTI, LU, F)) {
5857 Value *FullV = Rewriter.expandCodeFor(SE.getAddExpr(Ops), nullptr);
5858 Ops.clear();
5859 Ops.push_back(SE.getUnknown(FullV));
5860 }
5861 ScaledS = SE.getUnknown(Rewriter.expandCodeFor(ScaledS, nullptr));
5862 if (F.Scale != 1)
5863 ScaledS =
5864 SE.getMulExpr(ScaledS, SE.getConstant(ScaledS->getType(), F.Scale));
5865 Ops.push_back(ScaledS);
5866 }
5867 }
5868
5869 // Expand the GV portion.
5870 if (F.BaseGV) {
5871 // Flush the operand list to suppress SCEVExpander hoisting.
5872 if (!Ops.empty()) {
5873 Value *FullV = Rewriter.expandCodeFor(SE.getAddExpr(Ops), IntTy);
5874 Ops.clear();
5875 Ops.push_back(SE.getUnknown(FullV));
5876 }
5877 Ops.push_back(SE.getUnknown(F.BaseGV));
5878 }
5879
5880 // Flush the operand list to suppress SCEVExpander hoisting of both folded and
5881 // unfolded offsets. LSR assumes they both live next to their uses.
5882 if (!Ops.empty()) {
5883 Value *FullV = Rewriter.expandCodeFor(SE.getAddExpr(Ops), Ty);
5884 Ops.clear();
5885 Ops.push_back(SE.getUnknown(FullV));
5886 }
5887
5888 // FIXME: Are we sure we won't get a mismatch here? Is there a way to bail
5889 // out at this point, or should we generate a SCEV adding together mixed
5890 // offsets?
5891 assert(F.BaseOffset.isCompatibleImmediate(LF.Offset) &&
5892 "Expanding mismatched offsets\n");
5893 // Expand the immediate portion.
5894 Immediate Offset = F.BaseOffset.addUnsigned(LF.Offset);
5895 if (Offset.isNonZero()) {
5896 if (LU.Kind == LSRUse::ICmpZero) {
5897 // The other interesting way of "folding" with an ICmpZero is to use a
5898 // negated immediate.
5899 if (!ICmpScaledV) {
5900 // TODO: Avoid implicit trunc?
5901 // See https://github.com/llvm/llvm-project/issues/112510.
5902 ICmpScaledV = ConstantInt::getSigned(
5903 IntTy, -(uint64_t)Offset.getFixedValue(), /*ImplicitTrunc=*/true);
5904 } else {
5905 Ops.push_back(SE.getUnknown(ICmpScaledV));
5906 ICmpScaledV = ConstantInt::getSigned(IntTy, Offset.getFixedValue(),
5907 /*ImplicitTrunc=*/true);
5908 }
5909 } else {
5910 // Just add the immediate values. These again are expected to be matched
5911 // as part of the address.
5912 Ops.push_back(Offset.getUnknownSCEV(SE, IntTy));
5913 }
5914 }
5915
5916 // Expand the unfolded offset portion.
5917 Immediate UnfoldedOffset = F.UnfoldedOffset;
5918 if (UnfoldedOffset.isNonZero()) {
5919 // Just add the immediate values.
5920 Ops.push_back(UnfoldedOffset.getUnknownSCEV(SE, IntTy));
5921 }
5922
5923 // Emit instructions summing all the operands.
5924 const SCEV *FullS =
5925 Ops.empty() ? SE.getConstant(IntTy, 0) : SE.getAddExpr(Ops).getPointer();
5926 Value *FullV = Rewriter.expandCodeFor(FullS, Ty);
5927
5928 // We're done expanding now, so reset the rewriter.
5929 Rewriter.clearPostInc();
5930
5931 // An ICmpZero Formula represents an ICmp which we're handling as a
5932 // comparison against zero. Now that we've expanded an expression for that
5933 // form, update the ICmp's other operand.
5934 if (LU.Kind == LSRUse::ICmpZero) {
5935 ICmpInst *CI = cast<ICmpInst>(LF.UserInst);
5936 if (auto *OperandIsInstr = dyn_cast<Instruction>(CI->getOperand(1)))
5937 DeadInsts.emplace_back(OperandIsInstr);
5938 assert(!F.BaseGV && "ICmp does not support folding a global value and "
5939 "a scale at the same time!");
5940 if (F.Scale == -1) {
5941 if (ICmpScaledV->getType() != OpTy) {
5943 CastInst::getCastOpcode(ICmpScaledV, false, OpTy, false),
5944 ICmpScaledV, OpTy, "tmp", CI->getIterator());
5945 ICmpScaledV = Cast;
5946 }
5947 CI->setOperand(1, ICmpScaledV);
5948 } else {
5949 // A scale of 1 means that the scale has been expanded as part of the
5950 // base regs.
5951 assert((F.Scale == 0 || F.Scale == 1) &&
5952 "ICmp does not support folding a global value and "
5953 "a scale at the same time!");
5954 // TODO: Avoid implicit trunc?
5955 // See https://github.com/llvm/llvm-project/issues/112510.
5957 -(uint64_t)Offset.getFixedValue(),
5958 /*ImplicitTrunc=*/true);
5959 if (C->getType() != OpTy) {
5961 CastInst::getCastOpcode(C, false, OpTy, false), C, OpTy,
5962 CI->getDataLayout());
5963 assert(C && "Cast of ConstantInt should have folded");
5964 }
5965
5966 CI->setOperand(1, C);
5967 }
5968 }
5969
5970 return FullV;
5971}
5972
5973/// Helper for Rewrite. PHI nodes are special because the use of their operands
5974/// effectively happens in their predecessor blocks, so the expression may need
5975/// to be expanded in multiple places.
5976void LSRInstance::RewriteForPHI(PHINode *PN, const LSRUse &LU,
5977 const LSRFixup &LF, const Formula &F,
5978 SmallVectorImpl<WeakTrackingVH> &DeadInsts) {
5979 DenseMap<BasicBlock *, Value *> Inserted;
5980
5981 for (unsigned i = 0, e = PN->getNumIncomingValues(); i != e; ++i)
5982 if (PN->getIncomingValue(i) == LF.OperandValToReplace) {
5983 bool needUpdateFixups = false;
5984 BasicBlock *BB = PN->getIncomingBlock(i);
5985
5986 // If this is a critical edge, split the edge so that we do not insert
5987 // the code on all predecessor/successor paths. We do this unless this
5988 // is the canonical backedge for this loop, which complicates post-inc
5989 // users.
5990 if (e != 1 && BB->getTerminator()->getNumSuccessors() > 1 &&
5993 BasicBlock *Parent = PN->getParent();
5994 Loop *PNLoop = LI.getLoopFor(Parent);
5995 if (!PNLoop || Parent != PNLoop->getHeader()) {
5996 // Split the critical edge.
5997 BasicBlock *NewBB = nullptr;
5998 if (!Parent->isLandingPad()) {
5999 CriticalEdgeSplittingOptions SplitOptions(&DT, &LI, MSSAU);
6000 SplitOptions =
6001 SplitOptions.setMergeIdenticalEdges().setKeepOneInputPHIs();
6002 if (ShouldPreserveLCSSA)
6003 SplitOptions = SplitOptions.setPreserveLCSSA();
6004 NewBB = SplitCriticalEdge(BB, Parent, SplitOptions);
6005 } else {
6007 DomTreeUpdater DTU(DT, DomTreeUpdater::UpdateStrategy::Eager);
6008 SplitLandingPadPredecessors(Parent, BB, "", "", NewBBs, &DTU, &LI);
6009 NewBB = NewBBs[0];
6010 }
6011 // If NewBB==NULL, then SplitCriticalEdge refused to split because all
6012 // phi predecessors are identical. The simple thing to do is skip
6013 // splitting in this case rather than complicate the API.
6014 if (NewBB) {
6015 // If PN is outside of the loop and BB is in the loop, we want to
6016 // move the block to be immediately before the PHI block, not
6017 // immediately after BB.
6018 if (L->contains(BB) && !L->contains(PN))
6019 NewBB->moveBefore(PN->getParent());
6020
6021 // Splitting the edge can reduce the number of PHI entries we have.
6022 e = PN->getNumIncomingValues();
6023 BB = NewBB;
6024 i = PN->getBasicBlockIndex(BB);
6025
6026 needUpdateFixups = true;
6027 }
6028 }
6029 }
6030
6031 std::pair<DenseMap<BasicBlock *, Value *>::iterator, bool> Pair =
6032 Inserted.try_emplace(BB);
6033 if (!Pair.second)
6034 PN->setIncomingValue(i, Pair.first->second);
6035 else {
6036 Value *FullV =
6037 Expand(LU, LF, F, BB->getTerminator()->getIterator(), DeadInsts);
6038
6039 // If this is reuse-by-noop-cast, insert the noop cast.
6040 Type *OpTy = LF.OperandValToReplace->getType();
6041 if (FullV->getType() != OpTy)
6042 FullV = CastInst::Create(
6043 CastInst::getCastOpcode(FullV, false, OpTy, false), FullV,
6044 LF.OperandValToReplace->getType(), "tmp",
6045 BB->getTerminator()->getIterator());
6046
6047 // If the incoming block for this value is not in the loop, it means the
6048 // current PHI is not in a loop exit, so we must create a LCSSA PHI for
6049 // the inserted value.
6050 if (auto *I = dyn_cast<Instruction>(FullV))
6051 if (L->contains(I) && !L->contains(BB))
6052 InsertedNonLCSSAInsts.insert(I);
6053
6054 PN->setIncomingValue(i, FullV);
6055 Pair.first->second = FullV;
6056 }
6057
6058 // If LSR splits critical edge and phi node has other pending
6059 // fixup operands, we need to update those pending fixups. Otherwise
6060 // formulae will not be implemented completely and some instructions
6061 // will not be eliminated.
6062 if (needUpdateFixups) {
6063 for (LSRUse &LU : Uses)
6064 for (LSRFixup &Fixup : LU.Fixups)
6065 // If fixup is supposed to rewrite some operand in the phi
6066 // that was just updated, it may be already moved to
6067 // another phi node. Such fixup requires update.
6068 if (Fixup.UserInst == PN) {
6069 // Check if the operand we try to replace still exists in the
6070 // original phi.
6071 bool foundInOriginalPHI = false;
6072 for (const auto &val : PN->incoming_values())
6073 if (val == Fixup.OperandValToReplace) {
6074 foundInOriginalPHI = true;
6075 break;
6076 }
6077
6078 // If fixup operand found in original PHI - nothing to do.
6079 if (foundInOriginalPHI)
6080 continue;
6081
6082 // Otherwise it might be moved to another PHI and requires update.
6083 // If fixup operand not found in any of the incoming blocks that
6084 // means we have already rewritten it - nothing to do.
6085 for (const auto &Block : PN->blocks())
6086 for (BasicBlock::iterator I = Block->begin(); isa<PHINode>(I);
6087 ++I) {
6088 PHINode *NewPN = cast<PHINode>(I);
6089 for (const auto &val : NewPN->incoming_values())
6090 if (val == Fixup.OperandValToReplace)
6091 Fixup.UserInst = NewPN;
6092 }
6093 }
6094 }
6095 }
6096}
6097
6098/// Emit instructions for the leading candidate expression for this LSRUse (this
6099/// is called "expanding"), and update the UserInst to reference the newly
6100/// expanded value.
6101void LSRInstance::Rewrite(const LSRUse &LU, const LSRFixup &LF,
6102 const Formula &F,
6103 SmallVectorImpl<WeakTrackingVH> &DeadInsts) {
6104 // First, find an insertion point that dominates UserInst. For PHI nodes,
6105 // find the nearest block which dominates all the relevant uses.
6106 if (PHINode *PN = dyn_cast<PHINode>(LF.UserInst)) {
6107 RewriteForPHI(PN, LU, LF, F, DeadInsts);
6108 } else {
6109 Value *FullV = Expand(LU, LF, F, LF.UserInst->getIterator(), DeadInsts);
6110
6111 // If this is reuse-by-noop-cast, insert the noop cast.
6112 // For ICmpZero with pointer operands, Expand() already set both operands
6113 // in integer domain, so no cast is needed here.
6114 Type *OpTy = LF.OperandValToReplace->getType();
6115 if (FullV->getType() != OpTy &&
6116 !(LU.Kind == LSRUse::ICmpZero && OpTy->isPointerTy())) {
6117 Instruction *Cast =
6118 CastInst::Create(CastInst::getCastOpcode(FullV, false, OpTy, false),
6119 FullV, OpTy, "tmp", LF.UserInst->getIterator());
6120 FullV = Cast;
6121 }
6122
6123 // Update the user. ICmpZero is handled specially here (for now) because
6124 // Expand may have updated one of the operands of the icmp already, and
6125 // its new value may happen to be equal to LF.OperandValToReplace, in
6126 // which case doing replaceUsesOfWith leads to replacing both operands
6127 // with the same value. TODO: Reorganize this.
6128 if (LU.Kind == LSRUse::ICmpZero)
6129 LF.UserInst->setOperand(0, FullV);
6130 else
6131 LF.UserInst->replaceUsesOfWith(LF.OperandValToReplace, FullV);
6132 }
6133
6134 if (auto *OperandIsInstr = dyn_cast<Instruction>(LF.OperandValToReplace))
6135 DeadInsts.emplace_back(OperandIsInstr);
6136}
6137
6138// Determine where to insert the transformed IV increment instruction for this
6139// fixup. By default this is the default insert position, but if this is a
6140// postincrement opportunity then we try to insert it in the same block as the
6141// fixup user instruction, as this is needed for a postincrement instruction to
6142// be generated.
6144 const LSRFixup &Fixup, const LSRUse &LU,
6145 Instruction *IVIncInsertPos,
6146 DominatorTree &DT) {
6147 // Only address uses can be postincremented
6148 if (LU.Kind != LSRUse::Address)
6149 return IVIncInsertPos;
6150
6151 // Don't try to postincrement if it's not legal
6152 Instruction *I = Fixup.UserInst;
6153 Type *Ty = I->getType();
6154 if (!(isa<LoadInst>(I) && TTI.isIndexedLoadLegal(TTI.MIM_PostInc, Ty)) &&
6155 !(isa<StoreInst>(I) && TTI.isIndexedStoreLegal(TTI.MIM_PostInc, Ty)))
6156 return IVIncInsertPos;
6157
6158 // It's only legal to hoist to the user block if it dominates the default
6159 // insert position.
6160 BasicBlock *HoistBlock = I->getParent();
6161 BasicBlock *IVIncBlock = IVIncInsertPos->getParent();
6162 if (!DT.dominates(I, IVIncBlock))
6163 return IVIncInsertPos;
6164
6165 return HoistBlock->getTerminator();
6166}
6167
6168/// Rewrite all the fixup locations with new values, following the chosen
6169/// solution.
6170void LSRInstance::ImplementSolution(
6171 const SmallVectorImpl<const Formula *> &Solution) {
6172 // Keep track of instructions we may have made dead, so that
6173 // we can remove them after we are done working.
6175
6176 // Mark phi nodes that terminate chains so the expander tries to reuse them.
6177 for (const IVChain &Chain : IVChainVec) {
6178 if (PHINode *PN = dyn_cast<PHINode>(Chain.tailUserInst()))
6179 Rewriter.setChainedPhi(PN);
6180 }
6181
6182 // Expand the new value definitions and update the users.
6183 for (size_t LUIdx = 0, NumUses = Uses.size(); LUIdx != NumUses; ++LUIdx)
6184 for (const LSRFixup &Fixup : Uses[LUIdx].Fixups) {
6185 Instruction *InsertPos =
6186 getFixupInsertPos(TTI, Fixup, Uses[LUIdx], IVIncInsertPos, DT);
6187 Rewriter.setIVIncInsertPos(L, InsertPos);
6188 Rewrite(Uses[LUIdx], Fixup, *Solution[LUIdx], DeadInsts);
6189 Changed = true;
6190 }
6191
6192 auto InsertedInsts = InsertedNonLCSSAInsts.takeVector();
6193 formLCSSAForInstructions(InsertedInsts, DT, LI, &SE);
6194
6195 for (const IVChain &Chain : IVChainVec) {
6196 GenerateIVChain(Chain, DeadInsts);
6197 Changed = true;
6198 }
6199
6200 for (const WeakVH &IV : Rewriter.getInsertedIVs())
6201 if (IV && dyn_cast<Instruction>(&*IV)->getParent())
6202 ScalarEvolutionIVs.push_back(IV);
6203
6204 // Clean up after ourselves. This must be done before deleting any
6205 // instructions.
6206 Rewriter.clear();
6207
6209 &TLI, MSSAU);
6210
6211 // In our cost analysis above, we assume that each addrec consumes exactly
6212 // one register, and arrange to have increments inserted just before the
6213 // latch to maximimize the chance this is true. However, if we reused
6214 // existing IVs, we now need to move the increments to match our
6215 // expectations. Otherwise, our cost modeling results in us having a
6216 // chosen a non-optimal result for the actual schedule. (And yes, this
6217 // scheduling decision does impact later codegen.)
6218 for (PHINode &PN : L->getHeader()->phis()) {
6219 BinaryOperator *BO = nullptr;
6220 Value *Start = nullptr, *Step = nullptr;
6221 if (!matchSimpleRecurrence(&PN, BO, Start, Step))
6222 continue;
6223
6224 switch (BO->getOpcode()) {
6225 case Instruction::Sub:
6226 if (BO->getOperand(0) != &PN)
6227 // sub is non-commutative - match handling elsewhere in LSR
6228 continue;
6229 break;
6230 case Instruction::Add:
6231 break;
6232 default:
6233 continue;
6234 };
6235
6236 if (!isa<Constant>(Step))
6237 // If not a constant step, might increase register pressure
6238 // (We assume constants have been canonicalized to RHS)
6239 continue;
6240
6241 if (BO->getParent() == IVIncInsertPos->getParent())
6242 // Only bother moving across blocks. Isel can handle block local case.
6243 continue;
6244
6245 // Can we legally schedule inc at the desired point?
6246 if (!llvm::all_of(BO->uses(),
6247 [&](Use &U) {return DT.dominates(IVIncInsertPos, U);}))
6248 continue;
6249 BO->moveBefore(IVIncInsertPos->getIterator());
6250 Changed = true;
6251 }
6252
6253
6254}
6255
6256LSRInstance::LSRInstance(const ScalarOptions &Opts, Loop *L, IVUsers &IU,
6257 ScalarEvolution &SE, DominatorTree &DT, LoopInfo &LI,
6258 const TargetTransformInfo &TTI, AssumptionCache &AC,
6259 TargetLibraryInfo &TLI, MemorySSAUpdater *MSSAU,
6260 bool PreserveLCSSA)
6261 : Opts(Opts), IU(IU), SE(SE), DT(DT), LI(LI), AC(AC), TLI(TLI), TTI(TTI),
6262 L(L), MSSAU(MSSAU), AMK(Opts.lsr_preferred_addressing_mode.value_or(
6263 TTI.getPreferredAddressingMode(L, &SE))),
6264 Rewriter(SE, "lsr", PreserveLCSSA), ShouldPreserveLCSSA(PreserveLCSSA),
6265 BaselineCost(Opts, L, SE, TTI, AMK) {
6266 // If LoopSimplify form is not available, stay out of trouble.
6267 if (!L->isLoopSimplifyForm())
6268 return;
6269
6270 // If there's no interesting work to be done, bail early.
6271 if (IU.empty()) return;
6272
6273 // If there's too much analysis to be done, bail early. We won't be able to
6274 // model the problem anyway.
6275 unsigned NumUsers = 0;
6276 for (const IVStrideUse &U : IU) {
6277 if (++NumUsers > MaxIVUsers) {
6278 (void)U;
6279 LLVM_DEBUG(dbgs() << "LSR skipping loop, too many IV Users in " << U
6280 << "\n");
6281 return;
6282 }
6283 // Bail out if we have a PHI on an EHPad that gets a value from a
6284 // CatchSwitchInst. Because the CatchSwitchInst cannot be split, there is
6285 // no good place to stick any instructions.
6286 if (auto *PN = dyn_cast<PHINode>(U.getUser())) {
6287 auto FirstNonPHI = PN->getParent()->getFirstNonPHIIt();
6288 if (isa<FuncletPadInst>(FirstNonPHI) ||
6289 isa<CatchSwitchInst>(FirstNonPHI))
6290 for (BasicBlock *PredBB : PN->blocks())
6291 if (isa<CatchSwitchInst>(PredBB->getFirstNonPHIIt()))
6292 return;
6293 }
6294 }
6295
6296 LLVM_DEBUG(dbgs() << "\nLSR on loop ";
6297 L->getHeader()->printAsOperand(dbgs(), /*PrintType=*/false);
6298 dbgs() << ":\n");
6299
6300 // Check if we expect this loop to use a hardware loop instruction, which will
6301 // be used when calculating the costs of formulas.
6302 HardwareLoopInfo HWLoopInfo(L);
6303 HardwareLoopProfitable =
6304 TTI.isHardwareLoopProfitable(L, SE, AC, &TLI, HWLoopInfo);
6305
6306 // Configure SCEVExpander already now, so the correct mode is used for
6307 // isSafeToExpand() checks.
6308#if LLVM_ENABLE_ABI_BREAKING_CHECKS
6309 Rewriter.setDebugType(DEBUG_TYPE);
6310#endif
6311 Rewriter.disableCanonicalMode();
6312 Rewriter.enableLSRMode();
6313
6314 // First, perform some low-level loop optimizations.
6315 OptimizeShadowIV();
6316 OptimizeLoopTermCond();
6317
6318 // If loop preparation eliminates all interesting IV users, bail.
6319 if (IU.empty()) return;
6320
6321 // Skip nested loops until we can model them better with formulae.
6322 if (!L->isInnermost()) {
6323 LLVM_DEBUG(dbgs() << "LSR skipping outer loop " << *L << "\n");
6324 return;
6325 }
6326
6327 // Start collecting data and preparing for the solver.
6328 // If number of registers is not the major cost, we cannot benefit from the
6329 // current profitable chain optimization which is based on number of
6330 // registers.
6331 // FIXME: add profitable chain optimization for other kinds major cost, for
6332 // example number of instructions.
6333 if (TTI.isNumRegsMajorCostOfLSR() || StressIVChain)
6334 CollectChains();
6335 CollectInterestingTypesAndFactors();
6336 CollectFixupsAndInitialFormulae();
6337 CollectLoopInvariantFixupsAndFormulae();
6338
6339 if (Uses.empty())
6340 return;
6341
6342 LLVM_DEBUG(dbgs() << "LSR found " << Uses.size() << " uses:\n";
6343 print_uses(dbgs()));
6344 LLVM_DEBUG(dbgs() << "The baseline solution requires ";
6345 BaselineCost.print(dbgs()); dbgs() << "\n");
6346
6347 // Now use the reuse data to generate a bunch of interesting ways
6348 // to formulate the values needed for the uses.
6349 GenerateAllReuseFormulae();
6350
6351 FilterOutUndesirableDedicatedRegisters();
6352 NarrowSearchSpaceUsingHeuristics();
6353
6355 Solve(Solution);
6356
6357 // Release memory that is no longer needed.
6358 Factors.clear();
6359 Types.clear();
6360 RegUses.clear();
6361
6362 if (Solution.empty())
6363 return;
6364
6365#ifndef NDEBUG
6366 // Formulae should be legal.
6367 for (const LSRUse &LU : Uses) {
6368 for (const Formula &F : LU.Formulae)
6369 assert(isLegalUse(TTI, LU.MinOffset, LU.MaxOffset, LU.Kind, LU.AccessTy,
6370 F) && "Illegal formula generated!");
6371 };
6372#endif
6373
6374 // Now that we've decided what we want, make it so.
6375 ImplementSolution(Solution);
6376}
6377
6378#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
6379void LSRInstance::print_factors_and_types(raw_ostream &OS) const {
6380 if (Factors.empty() && Types.empty()) return;
6381
6382 OS << "LSR has identified the following interesting factors and types: ";
6383 ListSeparator LS;
6384
6385 for (int64_t Factor : Factors)
6386 OS << LS << '*' << Factor;
6387
6388 for (Type *Ty : Types)
6389 OS << LS << '(' << *Ty << ')';
6390 OS << '\n';
6391}
6392
6393void LSRInstance::print_fixups(raw_ostream &OS) const {
6394 OS << "LSR is examining the following fixup sites:\n";
6395 for (const LSRUse &LU : Uses)
6396 for (const LSRFixup &LF : LU.Fixups) {
6397 dbgs() << " ";
6398 LF.print(OS);
6399 OS << '\n';
6400 }
6401}
6402
6403void LSRInstance::print_uses(raw_ostream &OS) const {
6404 OS << "LSR is examining the following uses:\n";
6405 for (const LSRUse &LU : Uses) {
6406 dbgs() << " ";
6407 LU.print(OS);
6408 OS << '\n';
6409 for (const Formula &F : LU.Formulae) {
6410 OS << " ";
6411 F.print(OS);
6412 OS << '\n';
6413 }
6414 }
6415}
6416
6417void LSRInstance::print(raw_ostream &OS) const {
6418 print_factors_and_types(OS);
6419 print_fixups(OS);
6420 print_uses(OS);
6421}
6422
6423LLVM_DUMP_METHOD void LSRInstance::dump() const {
6424 print(errs()); errs() << '\n';
6425}
6426#endif
6427
6428namespace {
6429
6430class LoopStrengthReduce : public LoopPass {
6431public:
6432 static char ID; // Pass ID, replacement for typeid
6433
6434 LoopStrengthReduce();
6435
6436private:
6437 bool runOnLoop(Loop *L, LPPassManager &LPM) override;
6438 void getAnalysisUsage(AnalysisUsage &AU) const override;
6439};
6440
6441} // end anonymous namespace
6442
6443LoopStrengthReduce::LoopStrengthReduce() : LoopPass(ID) {
6445}
6446
6447void LoopStrengthReduce::getAnalysisUsage(AnalysisUsage &AU) const {
6448 // We split critical edges, so we change the CFG. However, we do update
6449 // many analyses if they are around.
6451
6452 AU.addRequired<LoopInfoWrapperPass>();
6453 AU.addPreserved<LoopInfoWrapperPass>();
6455 AU.addRequired<DominatorTreeWrapperPass>();
6456 AU.addPreserved<DominatorTreeWrapperPass>();
6457 AU.addRequired<ScalarEvolutionWrapperPass>();
6458 AU.addPreserved<ScalarEvolutionWrapperPass>();
6459 AU.addRequired<AssumptionCacheTracker>();
6460 AU.addRequired<TargetLibraryInfoWrapperPass>();
6461 // Requiring LoopSimplify a second time here prevents IVUsers from running
6462 // twice, since LoopSimplify was invalidated by running ScalarEvolution.
6464 AU.addRequired<IVUsersWrapperPass>();
6465 AU.addPreserved<IVUsersWrapperPass>();
6466 AU.addRequired<TargetTransformInfoWrapperPass>();
6467 AU.addPreserved<MemorySSAWrapperPass>();
6468}
6469
6470namespace {
6471
6472/// Enables more convenient iteration over a DWARF expression vector.
6474ToDwarfOpIter(SmallVectorImpl<uint64_t> &Expr) {
6475 llvm::DIExpression::expr_op_iterator Begin =
6476 llvm::DIExpression::expr_op_iterator(Expr.begin());
6477 llvm::DIExpression::expr_op_iterator End =
6478 llvm::DIExpression::expr_op_iterator(Expr.end());
6479 return {Begin, End};
6480}
6481
6482struct SCEVDbgValueBuilder {
6483 SCEVDbgValueBuilder() = default;
6484 SCEVDbgValueBuilder(const SCEVDbgValueBuilder &Base) { clone(Base); }
6485
6486 void clone(const SCEVDbgValueBuilder &Base) {
6487 LocationOps = Base.LocationOps;
6488 Expr = Base.Expr;
6489 }
6490
6491 void clear() {
6492 LocationOps.clear();
6493 Expr.clear();
6494 }
6495
6496 /// The DIExpression as we translate the SCEV.
6498 /// The location ops of the DIExpression.
6499 SmallVector<Value *, 2> LocationOps;
6500
6501 void pushOperator(uint64_t Op) { Expr.push_back(Op); }
6502 void pushUInt(uint64_t Operand) { Expr.push_back(Operand); }
6503
6504 /// Add a DW_OP_LLVM_arg to the expression, followed by the index of the value
6505 /// in the set of values referenced by the expression.
6506 void pushLocation(llvm::Value *V) {
6508 auto *It = llvm::find(LocationOps, V);
6509 unsigned ArgIndex = 0;
6510 if (It != LocationOps.end()) {
6511 ArgIndex = std::distance(LocationOps.begin(), It);
6512 } else {
6513 ArgIndex = LocationOps.size();
6514 LocationOps.push_back(V);
6515 }
6516 Expr.push_back(ArgIndex);
6517 }
6518
6519 void pushValue(const SCEVUnknown *U) {
6520 llvm::Value *V = cast<SCEVUnknown>(U)->getValue();
6521 pushLocation(V);
6522 }
6523
6524 bool pushConst(const SCEVConstant *C) {
6525 if (C->getAPInt().getSignificantBits() > 64)
6526 return false;
6527 Expr.push_back(llvm::dwarf::DW_OP_consts);
6528 Expr.push_back(C->getAPInt().getSExtValue());
6529 return true;
6530 }
6531
6532 // Iterating the expression as DWARF ops is convenient when updating
6533 // DWARF_OP_LLVM_args.
6535 return ToDwarfOpIter(Expr);
6536 }
6537
6538 /// Several SCEV types are sequences of the same arithmetic operator applied
6539 /// to constants and values that may be extended or truncated.
6540 bool pushArithmeticExpr(const llvm::SCEVCommutativeExpr *CommExpr,
6541 uint64_t DwarfOp) {
6542 assert((isa<llvm::SCEVAddExpr>(CommExpr) || isa<SCEVMulExpr>(CommExpr)) &&
6543 "Expected arithmetic SCEV type");
6544 bool Success = true;
6545 unsigned EmitOperator = 0;
6546 for (const auto &Op : CommExpr->operands()) {
6547 Success &= pushSCEV(Op);
6548
6549 if (EmitOperator >= 1)
6550 pushOperator(DwarfOp);
6551 ++EmitOperator;
6552 }
6553 return Success;
6554 }
6555
6556 // TODO: Identify and omit noop casts.
6557 bool pushCast(const llvm::SCEVCastExpr *C, bool IsSigned) {
6558 const llvm::SCEV *Inner = C->getOperand(0);
6559 const llvm::Type *Type = C->getType();
6560 uint64_t ToWidth = Type->getIntegerBitWidth();
6561 bool Success = pushSCEV(Inner);
6562 uint64_t CastOps[] = {dwarf::DW_OP_LLVM_convert, ToWidth,
6563 IsSigned ? llvm::dwarf::DW_ATE_signed
6564 : llvm::dwarf::DW_ATE_unsigned};
6565 for (const auto &Op : CastOps)
6566 pushOperator(Op);
6567 return Success;
6568 }
6569
6570 // TODO: MinMax - although these haven't been encountered in the test suite.
6571 bool pushSCEV(const llvm::SCEV *S) {
6572 bool Success = true;
6573 if (const SCEVConstant *StartInt = dyn_cast<SCEVConstant>(S)) {
6574 Success &= pushConst(StartInt);
6575
6576 } else if (const SCEVUnknown *U = dyn_cast<SCEVUnknown>(S)) {
6577 if (!U->getValue())
6578 return false;
6579 pushLocation(U->getValue());
6580
6581 } else if (const SCEVMulExpr *MulRec = dyn_cast<SCEVMulExpr>(S)) {
6582 Success &= pushArithmeticExpr(MulRec, llvm::dwarf::DW_OP_mul);
6583
6584 } else if (const SCEVUDivExpr *UDiv = dyn_cast<SCEVUDivExpr>(S)) {
6585 Success &= pushSCEV(UDiv->getLHS());
6586 Success &= pushSCEV(UDiv->getRHS());
6587 pushOperator(llvm::dwarf::DW_OP_div);
6588
6589 } else if (const SCEVCastExpr *Cast = dyn_cast<SCEVCastExpr>(S)) {
6590 // Assert if a new and unknown SCEVCastEXpr type is encountered.
6593 "Unexpected cast type in SCEV.");
6594 Success &= pushCast(Cast, (isa<SCEVSignExtendExpr>(Cast)));
6595
6596 } else if (const SCEVAddExpr *AddExpr = dyn_cast<SCEVAddExpr>(S)) {
6597 Success &= pushArithmeticExpr(AddExpr, llvm::dwarf::DW_OP_plus);
6598
6599 } else if (isa<SCEVAddRecExpr>(S)) {
6600 // Nested SCEVAddRecExpr are generated by nested loops and are currently
6601 // unsupported.
6602 return false;
6603
6604 } else {
6605 return false;
6606 }
6607 return Success;
6608 }
6609
6610 /// Return true if the combination of arithmetic operator and underlying
6611 /// SCEV constant value is an identity function.
6612 bool isIdentityFunction(uint64_t Op, const SCEV *S) {
6613 if (const SCEVConstant *C = dyn_cast<SCEVConstant>(S)) {
6614 if (C->getAPInt().getSignificantBits() > 64)
6615 return false;
6616 int64_t I = C->getAPInt().getSExtValue();
6617 switch (Op) {
6618 case llvm::dwarf::DW_OP_plus:
6619 case llvm::dwarf::DW_OP_minus:
6620 return I == 0;
6621 case llvm::dwarf::DW_OP_mul:
6622 case llvm::dwarf::DW_OP_div:
6623 return I == 1;
6624 }
6625 }
6626 return false;
6627 }
6628
6629 /// Convert a SCEV of a value to a DIExpression that is pushed onto the
6630 /// builder's expression stack. The stack should already contain an
6631 /// expression for the iteration count, so that it can be multiplied by
6632 /// the stride and added to the start.
6633 /// Components of the expression are omitted if they are an identity function.
6634 /// Chain (non-affine) SCEVs are not supported.
6635 bool SCEVToValueExpr(const llvm::SCEVAddRecExpr &SAR, ScalarEvolution &SE) {
6636 assert(SAR.isAffine() && "Expected affine SCEV");
6637 const SCEV *Start = SAR.getStart();
6638 const SCEV *Stride = SAR.getStepRecurrence(SE);
6639
6640 // Skip pushing arithmetic noops.
6641 if (!isIdentityFunction(llvm::dwarf::DW_OP_mul, Stride)) {
6642 if (!pushSCEV(Stride))
6643 return false;
6644 pushOperator(llvm::dwarf::DW_OP_mul);
6645 }
6646 if (!isIdentityFunction(llvm::dwarf::DW_OP_plus, Start)) {
6647 if (!pushSCEV(Start))
6648 return false;
6649 pushOperator(llvm::dwarf::DW_OP_plus);
6650 }
6651 return true;
6652 }
6653
6654 /// Create an expression that is an offset from a value (usually the IV).
6655 void createOffsetExpr(int64_t Offset, Value *OffsetValue) {
6656 pushLocation(OffsetValue);
6658 LLVM_DEBUG(
6659 dbgs() << "scev-salvage: Generated IV offset expression. Offset: "
6660 << std::to_string(Offset) << "\n");
6661 }
6662
6663 /// Combine a translation of the SCEV and the IV to create an expression that
6664 /// recovers a location's value.
6665 /// returns true if an expression was created.
6666 bool createIterCountExpr(const SCEV *S,
6667 const SCEVDbgValueBuilder &IterationCount,
6668 ScalarEvolution &SE) {
6669 // SCEVs for SSA values are most frquently of the form
6670 // {start,+,stride}, but sometimes they are ({start,+,stride} + %a + ..).
6671 // This is because %a is a PHI node that is not the IV. However, these
6672 // SCEVs have not been observed to result in debuginfo-lossy optimisations,
6673 // so its not expected this point will be reached.
6674 if (!isa<SCEVAddRecExpr>(S))
6675 return false;
6676
6677 LLVM_DEBUG(dbgs() << "scev-salvage: Location to salvage SCEV: " << *S
6678 << '\n');
6679
6680 const auto *Rec = cast<SCEVAddRecExpr>(S);
6681 if (!Rec->isAffine())
6682 return false;
6683
6685 return false;
6686
6687 // Initialise a new builder with the iteration count expression. In
6688 // combination with the value's SCEV this enables recovery.
6689 clone(IterationCount);
6690 if (!SCEVToValueExpr(*Rec, SE))
6691 return false;
6692
6693 return true;
6694 }
6695
6696 /// Convert a SCEV of a value to a DIExpression that is pushed onto the
6697 /// builder's expression stack. The stack should already contain an
6698 /// expression for the iteration count, so that it can be multiplied by
6699 /// the stride and added to the start.
6700 /// Components of the expression are omitted if they are an identity function.
6701 bool SCEVToIterCountExpr(const llvm::SCEVAddRecExpr &SAR,
6702 ScalarEvolution &SE) {
6703 assert(SAR.isAffine() && "Expected affine SCEV");
6704 const SCEV *Start = SAR.getStart();
6705 const SCEV *Stride = SAR.getStepRecurrence(SE);
6706
6707 // Skip pushing arithmetic noops.
6708 if (!isIdentityFunction(llvm::dwarf::DW_OP_minus, Start)) {
6709 if (!pushSCEV(Start))
6710 return false;
6711 pushOperator(llvm::dwarf::DW_OP_minus);
6712 }
6713 if (!isIdentityFunction(llvm::dwarf::DW_OP_div, Stride)) {
6714 if (!pushSCEV(Stride))
6715 return false;
6716 pushOperator(llvm::dwarf::DW_OP_div);
6717 }
6718 return true;
6719 }
6720
6721 // Append the current expression and locations to a location list and an
6722 // expression list. Modify the DW_OP_LLVM_arg indexes to account for
6723 // the locations already present in the destination list.
6724 void appendToVectors(SmallVectorImpl<uint64_t> &DestExpr,
6725 SmallVectorImpl<Value *> &DestLocations) {
6726 assert(!DestLocations.empty() &&
6727 "Expected the locations vector to contain the IV");
6728 // The DWARF_OP_LLVM_arg arguments of the expression being appended must be
6729 // modified to account for the locations already in the destination vector.
6730 // All builders contain the IV as the first location op.
6731 assert(!LocationOps.empty() &&
6732 "Expected the location ops to contain the IV.");
6733 // DestIndexMap[n] contains the index in DestLocations for the nth
6734 // location in this SCEVDbgValueBuilder.
6735 SmallVector<uint64_t, 2> DestIndexMap;
6736 for (const auto &Op : LocationOps) {
6737 auto It = find(DestLocations, Op);
6738 if (It != DestLocations.end()) {
6739 // Location already exists in DestLocations, reuse existing ArgIndex.
6740 DestIndexMap.push_back(std::distance(DestLocations.begin(), It));
6741 continue;
6742 }
6743 // Location is not in DestLocations, add it.
6744 DestIndexMap.push_back(DestLocations.size());
6745 DestLocations.push_back(Op);
6746 }
6747
6748 for (const auto &Op : expr_ops()) {
6750 if (!Arg) {
6751 Op.appendToVector(DestExpr);
6752 continue;
6753 }
6754
6756 // `DW_OP_LLVM_arg n` represents the nth LocationOp in this SCEV,
6757 // DestIndexMap[n] contains its new index in DestLocations.
6758 uint64_t NewIndex = DestIndexMap[Arg.getIndex()];
6759 DestExpr.push_back(NewIndex);
6760 }
6761 }
6762};
6763
6764/// Holds all the required data to salvage a dbg.value using the pre-LSR SCEVs
6765/// and DIExpression.
6766struct DVIRecoveryRec {
6767 DVIRecoveryRec(DbgVariableRecord *DVR)
6768 : DbgRef(DVR), Expr(DVR->getExpression()), HadLocationArgList(false) {}
6769
6770 DbgVariableRecord *DbgRef;
6771 DIExpression *Expr;
6772 bool HadLocationArgList;
6773 SmallVector<WeakVH, 2> LocationOps;
6776
6777 void clear() {
6778 for (auto &RE : RecoveryExprs)
6779 RE.reset();
6780 RecoveryExprs.clear();
6781 }
6782
6783 ~DVIRecoveryRec() { clear(); }
6784};
6785} // namespace
6786
6787/// Returns the total number of DW_OP_llvm_arg operands in the expression.
6788/// This helps in determining if a DIArglist is necessary or can be omitted from
6789/// the dbg.value.
6791 auto expr_ops = ToDwarfOpIter(Expr);
6792 unsigned Count = 0;
6793 for (auto Op : expr_ops)
6794 if (Op.getOp() == dwarf::DW_OP_LLVM_arg)
6795 Count++;
6796 return Count;
6797}
6798
6799/// Overwrites DVI with the location and Ops as the DIExpression. This will
6800/// create an invalid expression if Ops has any dwarf::DW_OP_llvm_arg operands,
6801/// because a DIArglist is not created for the first argument of the dbg.value.
6802template <typename T>
6803static void updateDVIWithLocation(T &DbgVal, Value *Location,
6805 assert(numLLVMArgOps(Ops) == 0 && "Expected expression that does not "
6806 "contain any DW_OP_llvm_arg operands.");
6807 DbgVal.setRawLocation(ValueAsMetadata::get(Location));
6808 DbgVal.setExpression(DIExpression::get(DbgVal.getContext(), Ops));
6809}
6810
6811/// Overwrite DVI with locations placed into a DIArglist.
6812template <typename T>
6813static void updateDVIWithLocations(T &DbgVal,
6814 SmallVectorImpl<Value *> &Locations,
6816 assert(numLLVMArgOps(Ops) != 0 &&
6817 "Expected expression that references DIArglist locations using "
6818 "DW_OP_llvm_arg operands.");
6820 for (Value *V : Locations)
6821 MetadataLocs.push_back(ValueAsMetadata::get(V));
6822 auto ValArrayRef = llvm::ArrayRef<llvm::ValueAsMetadata *>(MetadataLocs);
6823 DbgVal.setRawLocation(llvm::DIArgList::get(DbgVal.getContext(), ValArrayRef));
6824 DbgVal.setExpression(DIExpression::get(DbgVal.getContext(), Ops));
6825}
6826
6827/// Write the new expression and new location ops for the dbg.value. If possible
6828/// reduce the szie of the dbg.value by omitting DIArglist. This
6829/// can be omitted if:
6830/// 1. There is only a single location, refenced by a single DW_OP_llvm_arg.
6831/// 2. The DW_OP_LLVM_arg is the first operand in the expression.
6832static void UpdateDbgValue(DVIRecoveryRec &DVIRec,
6833 SmallVectorImpl<Value *> &NewLocationOps,
6835 DbgVariableRecord *DbgVal = DVIRec.DbgRef;
6836 unsigned NumLLVMArgs = numLLVMArgOps(NewExpr);
6837 if (NumLLVMArgs == 0) {
6838 // Location assumed to be on the stack.
6839 updateDVIWithLocation(*DbgVal, NewLocationOps[0], NewExpr);
6840 } else if (NumLLVMArgs == 1 && NewExpr[0] == dwarf::DW_OP_LLVM_arg) {
6841 // There is only a single DW_OP_llvm_arg at the start of the expression,
6842 // so it can be omitted along with DIArglist.
6843 assert(NewExpr[1] == 0 &&
6844 "Lone LLVM_arg in a DIExpression should refer to location-op 0.");
6846 updateDVIWithLocation(*DbgVal, NewLocationOps[0], ShortenedOps);
6847 } else {
6848 // Multiple DW_OP_llvm_arg, so DIArgList is strictly necessary.
6849 updateDVIWithLocations(*DbgVal, NewLocationOps, NewExpr);
6850 }
6851
6852 // If the DIExpression was previously empty then add the stack terminator.
6853 // Non-empty expressions have only had elements inserted into them and so
6854 // the terminator should already be present e.g. stack_value or fragment.
6855 DIExpression *SalvageExpr = DbgVal->getExpression();
6856 if (!DVIRec.Expr->isComplex() && SalvageExpr->isComplex()) {
6857 SalvageExpr = DIExpression::append(SalvageExpr, {dwarf::DW_OP_stack_value});
6858 DbgVal->setExpression(SalvageExpr);
6859 }
6860}
6861
6862/// Cached location ops may be erased during LSR, in which case a poison is
6863/// required when restoring from the cache. The type of that location is no
6864/// longer available, so just use int8. The poison will be replaced by one or
6865/// more locations later when a SCEVDbgValueBuilder selects alternative
6866/// locations to use for the salvage.
6868 return (VH) ? VH : PoisonValue::get(llvm::Type::getInt8Ty(C));
6869}
6870
6871/// Restore the DVI's pre-LSR arguments. Substitute undef for any erased values.
6872static void restorePreTransformState(DVIRecoveryRec &DVIRec) {
6873 DbgVariableRecord *DbgVal = DVIRec.DbgRef;
6874 LLVM_DEBUG(dbgs() << "scev-salvage: restore dbg.value to pre-LSR state\n"
6875 << "scev-salvage: post-LSR: " << *DbgVal << '\n');
6876 assert(DVIRec.Expr && "Expected an expression");
6877 DbgVal->setExpression(DVIRec.Expr);
6878
6879 // Even a single location-op may be inside a DIArgList and referenced with
6880 // DW_OP_LLVM_arg, which is valid only with a DIArgList.
6881 if (!DVIRec.HadLocationArgList) {
6882 assert(DVIRec.LocationOps.size() == 1 &&
6883 "Unexpected number of location ops.");
6884 // LSR's unsuccessful salvage attempt may have added DIArgList, which in
6885 // this case was not present before, so force the location back to a
6886 // single uncontained Value.
6887 Value *CachedValue =
6888 getValueOrPoison(DVIRec.LocationOps[0], DbgVal->getContext());
6889 DbgVal->setRawLocation(ValueAsMetadata::get(CachedValue));
6890 } else {
6892 for (WeakVH VH : DVIRec.LocationOps) {
6893 Value *CachedValue = getValueOrPoison(VH, DbgVal->getContext());
6894 MetadataLocs.push_back(ValueAsMetadata::get(CachedValue));
6895 }
6896 auto ValArrayRef = llvm::ArrayRef<llvm::ValueAsMetadata *>(MetadataLocs);
6897 DbgVal->setRawLocation(
6898 llvm::DIArgList::get(DbgVal->getContext(), ValArrayRef));
6899 }
6900 LLVM_DEBUG(dbgs() << "scev-salvage: pre-LSR: " << *DbgVal << '\n');
6901}
6902
6904 llvm::PHINode *LSRInductionVar, DVIRecoveryRec &DVIRec,
6905 const SCEV *SCEVInductionVar,
6906 SCEVDbgValueBuilder IterCountExpr) {
6907
6908 if (!DVIRec.DbgRef->isKillLocation())
6909 return false;
6910
6911 // LSR may have caused several changes to the dbg.value in the failed salvage
6912 // attempt. So restore the DIExpression, the location ops and also the
6913 // location ops format, which is always DIArglist for multiple ops, but only
6914 // sometimes for a single op.
6916
6917 // LocationOpIndexMap[i] will store the post-LSR location index of
6918 // the non-optimised out location at pre-LSR index i.
6919 SmallVector<int64_t, 2> LocationOpIndexMap;
6920 LocationOpIndexMap.assign(DVIRec.LocationOps.size(), -1);
6921 SmallVector<Value *, 2> NewLocationOps;
6922 NewLocationOps.push_back(LSRInductionVar);
6923
6924 for (unsigned i = 0; i < DVIRec.LocationOps.size(); i++) {
6925 WeakVH VH = DVIRec.LocationOps[i];
6926 // Place the locations not optimised out in the list first, avoiding
6927 // inserts later. The map is used to update the DIExpression's
6928 // DW_OP_LLVM_arg arguments as the expression is updated.
6929 if (VH && !isa<UndefValue>(VH)) {
6930 NewLocationOps.push_back(VH);
6931 LocationOpIndexMap[i] = NewLocationOps.size() - 1;
6932 LLVM_DEBUG(dbgs() << "scev-salvage: Location index " << i
6933 << " now at index " << LocationOpIndexMap[i] << "\n");
6934 continue;
6935 }
6936
6937 // It's possible that a value referred to in the SCEV may have been
6938 // optimised out by LSR.
6939 if (SE.containsErasedValue(DVIRec.SCEVs[i]) ||
6940 SE.containsUndefs(DVIRec.SCEVs[i])) {
6941 LLVM_DEBUG(dbgs() << "scev-salvage: SCEV for location at index: " << i
6942 << " refers to a location that is now undef or erased. "
6943 "Salvage abandoned.\n");
6944 return false;
6945 }
6946
6947 LLVM_DEBUG(dbgs() << "scev-salvage: salvaging location at index " << i
6948 << " with SCEV: " << *DVIRec.SCEVs[i] << "\n");
6949
6950 DVIRec.RecoveryExprs[i] = std::make_unique<SCEVDbgValueBuilder>();
6951 SCEVDbgValueBuilder *SalvageExpr = DVIRec.RecoveryExprs[i].get();
6952
6953 // Create an offset-based salvage expression if possible, as it requires
6954 // less DWARF ops than an iteration count-based expression.
6955 if (std::optional<APInt> Offset =
6956 SE.computeConstantDifference(DVIRec.SCEVs[i], SCEVInductionVar)) {
6957 if (Offset->getSignificantBits() <= 64)
6958 SalvageExpr->createOffsetExpr(Offset->getSExtValue(), LSRInductionVar);
6959 else
6960 return false;
6961 } else if (!SalvageExpr->createIterCountExpr(DVIRec.SCEVs[i], IterCountExpr,
6962 SE))
6963 return false;
6964 }
6965
6966 // Merge the DbgValueBuilder generated expressions and the original
6967 // DIExpression, place the result into an new vector.
6969 if (DVIRec.Expr->getNumElements() == 0) {
6970 assert(DVIRec.RecoveryExprs.size() == 1 &&
6971 "Expected only a single recovery expression for an empty "
6972 "DIExpression.");
6973 assert(DVIRec.RecoveryExprs[0] &&
6974 "Expected a SCEVDbgSalvageBuilder for location 0");
6975 SCEVDbgValueBuilder *B = DVIRec.RecoveryExprs[0].get();
6976 B->appendToVectors(NewExpr, NewLocationOps);
6977 }
6978 for (const auto &Op : DVIRec.Expr->expr_ops()) {
6979 // Most Ops needn't be updated.
6981 if (!Arg) {
6982 Op.appendToVector(NewExpr);
6983 continue;
6984 }
6985
6986 uint64_t LocationArgIndex = Arg.getIndex();
6987 SCEVDbgValueBuilder *DbgBuilder =
6988 DVIRec.RecoveryExprs[LocationArgIndex].get();
6989 // The location doesn't have s SCEVDbgValueBuilder, so LSR did not
6990 // optimise it away. So just translate the argument to the updated
6991 // location index.
6992 if (!DbgBuilder) {
6993 NewExpr.push_back(dwarf::DW_OP_LLVM_arg);
6994 assert(LocationOpIndexMap[LocationArgIndex] != -1 &&
6995 "Expected a positive index for the location-op position.");
6996 NewExpr.push_back(LocationOpIndexMap[LocationArgIndex]);
6997 continue;
6998 }
6999 // The location has a recovery expression.
7000 DbgBuilder->appendToVectors(NewExpr, NewLocationOps);
7001 }
7002
7003 UpdateDbgValue(DVIRec, NewLocationOps, NewExpr);
7004 LLVM_DEBUG(dbgs() << "scev-salvage: Updated DVI: " << *DVIRec.DbgRef << "\n");
7005 return true;
7006}
7007
7008/// Obtain an expression for the iteration count, then attempt to salvage the
7009/// dbg.value intrinsics.
7011 llvm::Loop *L, ScalarEvolution &SE, llvm::PHINode *LSRInductionVar,
7012 SmallVector<std::unique_ptr<DVIRecoveryRec>, 2> &DVIToUpdate) {
7013 if (DVIToUpdate.empty())
7014 return;
7015
7016 const llvm::SCEV *SCEVInductionVar = SE.getSCEV(LSRInductionVar);
7017 assert(SCEVInductionVar &&
7018 "Anticipated a SCEV for the post-LSR induction variable");
7019
7020 if (const SCEVAddRecExpr *IVAddRec =
7021 dyn_cast<SCEVAddRecExpr>(SCEVInductionVar)) {
7022 if (!IVAddRec->isAffine())
7023 return;
7024
7025 // Prevent translation using excessive resources.
7026 if (IVAddRec->getExpressionSize() > MaxSCEVSalvageExpressionSize)
7027 return;
7028
7029 // The iteration count is required to recover location values.
7030 SCEVDbgValueBuilder IterCountExpr;
7031 IterCountExpr.pushLocation(LSRInductionVar);
7032 if (!IterCountExpr.SCEVToIterCountExpr(*IVAddRec, SE))
7033 return;
7034
7035 LLVM_DEBUG(dbgs() << "scev-salvage: IV SCEV: " << *SCEVInductionVar
7036 << '\n');
7037
7038 for (auto &DVIRec : DVIToUpdate) {
7039 SalvageDVI(L, SE, LSRInductionVar, *DVIRec, SCEVInductionVar,
7040 IterCountExpr);
7041 }
7042 }
7043}
7044
7045/// Identify and cache salvageable DVI locations and expressions along with the
7046/// corresponding SCEV(s). Also ensure that the DVI is not deleted between
7047/// cacheing and salvaging.
7049 Loop *L, ScalarEvolution &SE,
7050 SmallVector<std::unique_ptr<DVIRecoveryRec>, 2> &SalvageableDVISCEVs) {
7051 for (const auto &B : L->getBlocks()) {
7052 for (auto &I : *B) {
7053 for (DbgVariableRecord &DbgVal : filterDbgVars(I.getDbgRecordRange())) {
7054 if (!DbgVal.isDbgValue() && !DbgVal.isDbgAssign())
7055 continue;
7056
7057 // Ensure that if any location op is undef that the dbg.vlue is not
7058 // cached.
7059 if (DbgVal.isKillLocation())
7060 continue;
7061
7062 // Check that the location op SCEVs are suitable for translation to
7063 // DIExpression.
7064 const auto &HasTranslatableLocationOps =
7065 [&](const DbgVariableRecord &DbgValToTranslate) -> bool {
7066 for (const auto LocOp : DbgValToTranslate.location_ops()) {
7067 if (!LocOp)
7068 return false;
7069
7070 if (!SE.isSCEVable(LocOp->getType()))
7071 return false;
7072
7073 const SCEV *S = SE.getSCEV(LocOp);
7074 if (SE.containsUndefs(S))
7075 return false;
7076 }
7077 return true;
7078 };
7079
7080 if (!HasTranslatableLocationOps(DbgVal))
7081 continue;
7082
7083 std::unique_ptr<DVIRecoveryRec> NewRec =
7084 std::make_unique<DVIRecoveryRec>(&DbgVal);
7085 // Each location Op may need a SCEVDbgValueBuilder in order to recover
7086 // it. Pre-allocating a vector will enable quick lookups of the builder
7087 // later during the salvage.
7088 NewRec->RecoveryExprs.resize(DbgVal.getNumVariableLocationOps());
7089 for (const auto LocOp : DbgVal.location_ops()) {
7090 NewRec->SCEVs.push_back(SE.getSCEV(LocOp));
7091 NewRec->LocationOps.push_back(LocOp);
7092 NewRec->HadLocationArgList = DbgVal.hasArgList();
7093 }
7094 SalvageableDVISCEVs.push_back(std::move(NewRec));
7095 }
7096 }
7097 }
7098}
7099
7100/// Ideally pick the PHI IV inserted by ScalarEvolutionExpander. As a fallback
7101/// any PHi from the loop header is usable, but may have less chance of
7102/// surviving subsequent transforms.
7104 const LSRInstance &LSR) {
7105
7106 auto IsSuitableIV = [&](PHINode *P) {
7107 if (!SE.isSCEVable(P->getType()))
7108 return false;
7109 if (const SCEVAddRecExpr *Rec = dyn_cast<SCEVAddRecExpr>(SE.getSCEV(P)))
7110 return Rec->isAffine() && !SE.containsUndefs(SE.getSCEV(P));
7111 return false;
7112 };
7113
7114 // For now, just pick the first IV that was generated and inserted by
7115 // ScalarEvolution. Ideally pick an IV that is unlikely to be optimised away
7116 // by subsequent transforms.
7117 for (const WeakVH &IV : LSR.getScalarEvolutionIVs()) {
7118 if (!IV)
7119 continue;
7120
7121 // There should only be PHI node IVs.
7122 PHINode *P = cast<PHINode>(&*IV);
7123
7124 if (IsSuitableIV(P))
7125 return P;
7126 }
7127
7128 for (PHINode &P : L.getHeader()->phis()) {
7129 if (IsSuitableIV(&P))
7130 return &P;
7131 }
7132 return nullptr;
7133}
7134
7136 DominatorTree &DT, LoopInfo &LI,
7137 const TargetTransformInfo &TTI,
7139 MemorySSA *MSSA, bool PreserveLCSSA) {
7140 const ScalarOptions &Opts = ScalarOptions::Global;
7141
7142 // Debug preservation - before we start removing anything identify which DVI
7143 // meet the salvageable criteria and store their DIExpression and SCEVs.
7144 SmallVector<std::unique_ptr<DVIRecoveryRec>, 2> SalvageableDVIRecords;
7145 DbgGatherSalvagableDVI(L, SE, SalvageableDVIRecords);
7146
7147 bool Changed = false;
7148 std::unique_ptr<MemorySSAUpdater> MSSAU;
7149 if (MSSA)
7150 MSSAU = std::make_unique<MemorySSAUpdater>(MSSA);
7151
7152 // Run the main LSR transformation.
7153 const LSRInstance &Reducer = LSRInstance(Opts, L, IU, SE, DT, LI, TTI, AC,
7154 TLI, MSSAU.get(), PreserveLCSSA);
7155 Changed |= Reducer.getChanged();
7156
7157 // Remove any extra phis created by processing inner loops.
7158 Changed |= DeleteDeadPHIs(L->getHeader(), &TLI, MSSAU.get());
7159 if (Opts.enable_lsr_phielim && L->isLoopSimplifyForm()) {
7161 SCEVExpander Rewriter(SE, "lsr", false);
7162#if LLVM_ENABLE_ABI_BREAKING_CHECKS
7163 Rewriter.setDebugType(DEBUG_TYPE);
7164#endif
7165 unsigned numFolded = Rewriter.replaceCongruentIVs(L, &DT, DeadInsts, &TTI);
7166 Rewriter.clear();
7167 if (numFolded) {
7168 Changed = true;
7170 MSSAU.get());
7171 DeleteDeadPHIs(L->getHeader(), &TLI, MSSAU.get());
7172 }
7173 }
7174 // LSR may at times remove all uses of an induction variable from a loop.
7175 // The only remaining use is the PHI in the exit block.
7176 // When this is the case, if the exit value of the IV can be calculated using
7177 // SCEV, we can replace the exit block PHI with the final value of the IV and
7178 // skip the updates in each loop iteration.
7179 if (L->isRecursivelyLCSSAForm(DT, LI) && L->getExitBlock()) {
7181 SCEVExpander Rewriter(SE, "lsr", true);
7182 int Rewrites = rewriteLoopExitValues(L, &LI, &TLI, &SE, &TTI, Rewriter, &DT,
7183 UnusedIndVarInLoop, DeadInsts);
7184 Rewriter.clear();
7185 if (Rewrites) {
7186 Changed = true;
7188 MSSAU.get());
7189 DeleteDeadPHIs(L->getHeader(), &TLI, MSSAU.get());
7190 }
7191 }
7192
7193 if (SalvageableDVIRecords.empty())
7194 return Changed;
7195
7196 // Obtain relevant IVs and attempt to rewrite the salvageable DVIs with
7197 // expressions composed using the derived iteration count.
7198 // TODO: Allow for multiple IV references for nested AddRecSCEVs
7199 for (const auto &L : LI) {
7200 if (llvm::PHINode *IV = GetInductionVariable(*L, SE, Reducer))
7201 DbgRewriteSalvageableDVIs(L, SE, IV, SalvageableDVIRecords);
7202 else {
7203 LLVM_DEBUG(dbgs() << "scev-salvage: SCEV salvaging not possible. An IV "
7204 "could not be identified.\n");
7205 }
7206 }
7207
7208 for (auto &Rec : SalvageableDVIRecords)
7209 Rec->clear();
7210 SalvageableDVIRecords.clear();
7211 return Changed;
7212}
7213
7214bool LoopStrengthReduce::runOnLoop(Loop *L, LPPassManager & /*LPM*/) {
7215 if (skipLoop(L))
7216 return false;
7217
7218 auto &IU = getAnalysis<IVUsersWrapperPass>().getIU();
7219 auto &SE = getAnalysis<ScalarEvolutionWrapperPass>().getSE();
7220 auto &DT = getAnalysis<DominatorTreeWrapperPass>().getDomTree();
7221 auto &LI = getAnalysis<LoopInfoWrapperPass>().getLoopInfo();
7222 const auto &TTI = getAnalysis<TargetTransformInfoWrapperPass>().getTTI(
7223 *L->getHeader()->getParent());
7224 auto &AC = getAnalysis<AssumptionCacheTracker>().getAssumptionCache(
7225 *L->getHeader()->getParent());
7226 auto &TLI = getAnalysis<TargetLibraryInfoWrapperPass>().getTLI(
7227 *L->getHeader()->getParent());
7228 auto *MSSAAnalysis = getAnalysisIfAvailable<MemorySSAWrapperPass>();
7229 MemorySSA *MSSA = nullptr;
7230 if (MSSAAnalysis)
7231 MSSA = &MSSAAnalysis->getMSSA();
7232 return ReduceLoopStrength(L, IU, SE, DT, LI, TTI, AC, TLI, MSSA,
7233 /*PreserveLCSSA=*/false);
7234}
7235
7238 LPMUpdater &) {
7239 if (!ReduceLoopStrength(&L, AM.getResult<IVUsersAnalysis>(L, AR), AR.SE,
7240 AR.DT, AR.LI, AR.TTI, AR.AC, AR.TLI, AR.MSSA,
7241 /*PreserveLCSSA=*/true))
7242 return PreservedAnalyses::all();
7243
7244 auto PA = getLoopPassPreservedAnalyses();
7245 if (AR.MSSA)
7246 PA.preserve<MemorySSAAnalysis>();
7247 return PA;
7248}
7249
7250char LoopStrengthReduce::ID = 0;
7251
7252INITIALIZE_PASS_BEGIN(LoopStrengthReduce, "loop-reduce",
7253 "Loop Strength Reduction", false, false)
7259INITIALIZE_PASS_DEPENDENCY(LoopSimplify)
7260INITIALIZE_PASS_END(LoopStrengthReduce, "loop-reduce",
7261 "Loop Strength Reduction", false, false)
7262
7263Pass *llvm::createLoopStrengthReducePass() { return new LoopStrengthReduce(); }
#define Success
for(const MachineOperand &MO :llvm::drop_begin(OldMI.operands(), Desc.getNumOperands()))
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AArch64 Predicate As Counter Loop Rewrites
unsigned Imm
unsigned uint64_t
This file implements a class to represent arbitrary precision integral constant values and operations...
Function Alias Analysis false
static void print(raw_ostream &Out, object::Archive::Kind Kind, T Val)
static const Function * getParent(const Value *V)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:686
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static bool isCanonical(const MDString *S)
This file defines the DenseMap class.
This file defines the DenseSet and SmallDenseSet classes.
This file contains constants used for implementing Dwarf debug support.
early cse Early CSE w MemorySSA
#define DEBUG_TYPE
Hexagon Hardware Loops
Module.h This file contains the declarations for the Module class.
This defines the Use class.
iv Induction Variable Users
Definition IVUsers.cpp:48
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
Definition Lint.cpp:540
This header provides classes for managing per-loop analyses.
static bool SalvageDVI(llvm::Loop *L, ScalarEvolution &SE, llvm::PHINode *LSRInductionVar, DVIRecoveryRec &DVIRec, const SCEV *SCEVInductionVar, SCEVDbgValueBuilder IterCountExpr)
static Value * getWideOperand(Value *Oper)
IVChain logic must consistently peek base TruncInst operands, so wrap it in a convenient helper.
static bool isAddSExtable(const SCEVAddExpr *A, ScalarEvolution &SE)
Return true if the given add can be sign-extended without changing its value.
static bool mayUsePostIncMode(const TargetTransformInfo &TTI, LSRUse &LU, const SCEV *S, const Loop *L, ScalarEvolution &SE)
Return true if the SCEV represents a value that may end up as a post-increment operation.
static void restorePreTransformState(DVIRecoveryRec &DVIRec)
Restore the DVI's pre-LSR arguments. Substitute undef for any erased values.
static bool containsAddRecDependentOnLoop(const SCEV *S, const Loop &L)
static User::op_iterator findIVOperand(User::op_iterator OI, User::op_iterator OE, Loop *L, ScalarEvolution &SE)
Helper for CollectChains that finds an IV operand (computed by an AddRec in this loop) within [OI,...
static bool isLegalUse(const TargetTransformInfo &TTI, Immediate MinOffset, Immediate MaxOffset, LSRUse::KindType Kind, MemAccessTy AccessTy, GlobalValue *BaseGV, Immediate BaseOffset, bool HasBaseReg, int64_t Scale)
Test whether we know how to expand the current formula.
static void DbgGatherSalvagableDVI(Loop *L, ScalarEvolution &SE, SmallVector< std::unique_ptr< DVIRecoveryRec >, 2 > &SalvageableDVISCEVs)
Identify and cache salvageable DVI locations and expressions along with the corresponding SCEV(s).
static bool isMulSExtable(const SCEVMulExpr *M, ScalarEvolution &SE)
Return true if the given mul can be sign-extended without changing its value.
static const unsigned MaxSCEVSalvageExpressionSize
Limit the size of expression that SCEV-based salvaging will attempt to translate into a DIExpression.
static bool isExistingPhi(const SCEVAddRecExpr *AR, ScalarEvolution &SE)
Return true if this AddRec is already a phi in its loop.
static InstructionCost getScalingFactorCost(const TargetTransformInfo &TTI, const LSRUse &LU, const Formula &F, const Loop &L)
static cl::opt< bool > StressIVChain("stress-ivchain", cl::Hidden, cl::init(false), cl::desc("Stress test LSR IV chains"))
static bool isAddressUse(const TargetTransformInfo &TTI, Instruction *Inst, Value *OperandVal)
Returns true if the specified instruction is using the specified value as an address.
static void DoInitialMatch(const SCEV *S, Loop *L, SmallVectorImpl< SCEVUse > &Good, SmallVectorImpl< SCEVUse > &Bad, ScalarEvolution &SE)
Recursion helper for initialMatch.
static void updateDVIWithLocation(T &DbgVal, Value *Location, SmallVectorImpl< uint64_t > &Ops)
Overwrites DVI with the location and Ops as the DIExpression.
static bool ReduceLoopStrength(Loop *L, IVUsers &IU, ScalarEvolution &SE, DominatorTree &DT, LoopInfo &LI, const TargetTransformInfo &TTI, AssumptionCache &AC, TargetLibraryInfo &TLI, MemorySSA *MSSA, bool PreserveLCSSA)
static bool isLegalAddImmediate(const TargetTransformInfo &TTI, Immediate Offset)
static Instruction * getFixupInsertPos(const TargetTransformInfo &TTI, const LSRFixup &Fixup, const LSRUse &LU, Instruction *IVIncInsertPos, DominatorTree &DT)
static const SCEV * getExprBase(const SCEV *S)
Return an approximation of this SCEV expression's "base", or NULL for any constant.
static llvm::PHINode * GetInductionVariable(const Loop &L, ScalarEvolution &SE, const LSRInstance &LSR)
Ideally pick the PHI IV inserted by ScalarEvolutionExpander.
static bool IsSimplerBaseSCEVForTarget(const TargetTransformInfo &TTI, ScalarEvolution &SE, const SCEV *Best, const SCEV *Reg, MemAccessTy AccessType)
static const unsigned MaxIVUsers
MaxIVUsers is an arbitrary threshold that provides an early opportunity for bail out.
static Immediate extractImmediate(const ScalarOptions &Opts, SCEVUse &S, ScalarEvolution &SE, bool PreferScalable=false)
If S involves the addition of a constant integer value, return that integer value,...
static bool isHighCostExpansion(const SCEV *S, SmallPtrSetImpl< const SCEV * > &Processed, ScalarEvolution &SE)
Check if expanding this expression is likely to incur significant cost.
static Value * getValueOrPoison(WeakVH &VH, LLVMContext &C)
Cached location ops may be erased during LSR, in which case a poison is required when restoring from ...
static MemAccessTy getAccessType(const TargetTransformInfo &TTI, Instruction *Inst, Value *OperandVal)
Return the type of the memory being accessed.
static unsigned numLLVMArgOps(SmallVectorImpl< uint64_t > &Expr)
Returns the total number of DW_OP_llvm_arg operands in the expression.
static bool isAlwaysFoldable(const ScalarOptions &Opts, const TargetTransformInfo &TTI, LSRUse::KindType Kind, MemAccessTy AccessTy, GlobalValue *BaseGV, Immediate BaseOffset, bool HasBaseReg)
static void DbgRewriteSalvageableDVIs(llvm::Loop *L, ScalarEvolution &SE, llvm::PHINode *LSRInductionVar, SmallVector< std::unique_ptr< DVIRecoveryRec >, 2 > &DVIToUpdate)
Obtain an expression for the iteration count, then attempt to salvage the dbg.value intrinsics.
static void UpdateDbgValue(DVIRecoveryRec &DVIRec, SmallVectorImpl< Value * > &NewLocationOps, SmallVectorImpl< uint64_t > &NewExpr)
Write the new expression and new location ops for the dbg.value.
static bool isAddRecSExtable(const SCEVAddRecExpr *AR, ScalarEvolution &SE)
Return true if the given addrec can be sign-extended without changing its value.
static bool isAMCompletelyFolded(const TargetTransformInfo &TTI, const LSRUse &LU, const Formula &F)
Check if the addressing mode defined by F is completely folded in LU at isel time.
static Immediate extractImmediateOperand(const ScalarOptions &Opts, MutableArrayRef< SCEVUse > Ops, ScalarEvolution &SE, bool PreferScalable)
Extracts an immediate operand from Ops and replaces the operand with zero.
static void updateDVIWithLocations(T &DbgVal, SmallVectorImpl< Value * > &Locations, SmallVectorImpl< uint64_t > &Ops)
Overwrite DVI with locations placed into a DIArglist.
static bool canFoldIVIncExpr(const ScalarOptions &Opts, const SCEV *IncExpr, Instruction *UserInst, Value *Operand, const TargetTransformInfo &TTI)
Return true if the IVInc can be folded into an addressing mode.
static GlobalValue * ExtractSymbol(SCEVUse &S, ScalarEvolution &SE)
If S involves the addition of a GlobalValue address, return that symbol, and mutate S to point to a n...
static bool isProfitableChain(IVChain &Chain, SmallPtrSetImpl< Instruction * > &Users, ScalarEvolution &SE, const TargetTransformInfo &TTI)
Return true if the number of registers needed for the chain is estimated to be less than the number r...
static const SCEV * CollectSubexprs(const SCEV *S, const SCEVConstant *C, SmallVectorImpl< const SCEV * > &Ops, const Loop *L, ScalarEvolution &SE, unsigned Depth=0)
Split S into subexpressions which can be pulled out into separate registers.
static const SCEV * getExactSDiv(const SCEV *LHS, const SCEV *RHS, ScalarEvolution &SE, bool IgnoreSignificantBits=false)
Return an expression for LHS /s RHS, if it can be determined and if the remainder is known to be zero...
static const SCEV * getAnyExtendConsideringPostIncUses(ArrayRef< PostIncLoopSet > Loops, const SCEV *Expr, Type *ToTy, ScalarEvolution &SE)
Extend/Truncate Expr to ToTy considering post-inc uses in Loops.
static unsigned getSetupCost(const SCEV *Reg, unsigned Depth, const TargetTransformInfo &TTI)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
Register Reg
This file exposes an interface to building/using memory SSA to walk memory instructions using a use/d...
#define T
uint64_t IntrinsicInst * II
#define P(N)
PowerPC TLS Dynamic Call Fixup
if(PassOpts->AAPipeline)
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
This file defines the PointerIntPair class.
const SmallVectorImpl< MachineOperand > & Cond
Remove Loads Into Fake Uses
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
SI optimize exec mask operations pre RA
This file contains some templates that are useful if you are working with the STL at all.
This file implements a set that has insertion order iteration characteristics.
This file implements the SmallBitVector class.
This file defines the SmallPtrSet class.
This file defines the SmallSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
static const unsigned UnknownAddressSpace
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This pass exposes codegen information to IR-level passes.
virt reg Virtual Register Rewriter
Value * RHS
Value * LHS
BinaryOperator * Mul
static const uint32_t IV[8]
Definition blake3_impl.h:83
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
bool isNegative() const
Determine sign of this APInt.
Definition APInt.h:325
LLVM_ABI APInt sdiv(const APInt &RHS) const
Signed division function for APInt.
Definition APInt.cpp:1673
unsigned getSignificantBits() const
Get the minimum bit size for this signed APInt.
Definition APInt.h:1551
LLVM_ABI APInt srem(const APInt &RHS) const
Function for signed remainder operation.
Definition APInt.cpp:1774
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
LLVM_ABI AnalysisUsage & addRequiredID(const void *ID)
Definition Pass.cpp:292
AnalysisUsage & addPreservedID(const void *ID)
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
A cache of @llvm.assume calls within a function.
An instruction that atomically checks whether a specified value is in a memory location,...
an instruction that atomically reads a memory location, combines it with another value,...
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator_range< const_phi_iterator > phis() const
Returns a range that iterates over the phis in the basic block.
Definition BasicBlock.h:515
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
void moveBefore(BasicBlock *MovePos)
Unlink this basic block from its current function and insert it into the function that MovePos lives ...
Definition BasicBlock.h:373
LLVM_ABI bool isLandingPad() const
Return true if this basic block is a landing pad.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
BinaryOps getOpcode() const
Definition InstrTypes.h:409
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_NE
not equal
Definition InstrTypes.h:762
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
Value * getCondition() const
static LLVM_ABI bool isValueValidForType(Type *Ty, uint64_t V)
This static method returns true if the type Ty is big enough to represent the value V.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI DIArgList * get(LLVMContext &Context, ArrayRef< ValueAsMetadata * > Args)
DWARF expression.
iterator_range< expr_op_iterator > expr_ops() const
static LLVM_ABI DIExpression * append(const DIExpression *Expr, ArrayRef< uint64_t > Ops)
Append the opcodes Ops to DIExpr.
unsigned getNumElements() const
static LLVM_ABI void appendOffset(SmallVectorImpl< uint64_t > &Ops, int64_t Offset)
Append Ops with operations to apply the Offset.
LLVM_ABI bool isComplex() const
Return whether the location is computed on the expression stack, meaning it cannot be a simple regist...
LLVM_ABI LLVMContext & getContext()
Record of a variable value-assignment, aka a non instruction representation of the dbg....
LLVM_ABI bool isKillLocation() const
void setRawLocation(Metadata *NewLocation)
Use of this should generally be avoided; instead, replaceVariableLocationOp and addVariableLocationOp...
void setExpression(DIExpression *NewExpr)
DIExpression * getExpression() const
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:828
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:857
NodeT * getBlock() const
DomTreeNodeBase< NodeT > * getNode(const NodeT *BB) const
getNode - return the (Post)DominatorTree node for the specified basic block.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
Legacy analysis pass which computes a DominatorTree.
Definition Dominators.h:277
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
LLVM_ABI Instruction * findNearestCommonDominator(Instruction *I1, Instruction *I2) const
Find the nearest instruction I that dominates both I1 and I2, in the sense that a result produced bef...
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
PointerType * getType() const
Global values are always pointers.
IVStrideUse - Keep track of one use of a strided induction variable.
Definition IVUsers.h:36
void transformToPostInc(const Loop *L)
transformToPostInc - Transform the expression to post-inc form for the given loop.
Definition IVUsers.cpp:365
Value * getOperandValToReplace() const
getOperandValToReplace - Return the Value of the operand in the user instruction that this IVStrideUs...
Definition IVUsers.h:55
void setUser(Instruction *NewUser)
setUser - Assign a new user instruction for this use.
Definition IVUsers.h:49
Analysis pass that exposes the IVUsers for a loop.
Definition IVUsers.h:187
ilist< IVStrideUse >::const_iterator const_iterator
Definition IVUsers.h:143
iterator end()
Definition IVUsers.h:145
iterator begin()
Definition IVUsers.h:144
bool empty() const
Definition IVUsers.h:148
LLVM_ABI void print(raw_ostream &OS) const
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isLifetimeStartOrEnd() const LLVM_READONLY
Return true if the instruction is a llvm.lifetime.start or llvm.lifetime.end marker.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
bool isEHPad() const
Return true if the instruction is a variety of EH-block.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI Type * getAccessType() const LLVM_READONLY
Return the type this instruction accesses in memory, if any.
iterator_range< user_iterator > users()
const char * getOpcodeName() const
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
This class provides an interface for updating the loop pass manager based on mutations to the loop ne...
An instruction for reading from memory.
void getExitingBlocks(SmallVectorImpl< BlockT * > &ExitingBlocks) const
Return all blocks inside the loop that have successors outside of the loop.
BlockT * getHeader() const
unsigned getLoopDepth() const
Return the nesting level of this loop.
The legacy pass manager's analysis pass to compute loop information.
Definition LoopInfo.h:619
LLVM_ABI PreservedAnalyses run(Loop &L, LoopAnalysisManager &AM, LoopStandardAnalysisResults &AR, LPMUpdater &U)
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
An analysis that produces MemorySSA for a function.
Definition MemorySSA.h:922
Encapsulates MemorySSA, including all data associated with memory accesses.
Definition MemorySSA.h:702
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:294
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
iterator_range< const_block_iterator > blocks() const
op_range incoming_values()
void setIncomingValue(unsigned i, Value *V)
BasicBlock * getIncomingBlock(unsigned i) const
Return incoming basic block number i.
Value * getIncomingValue(unsigned i) const
Return incoming value number x.
static unsigned getIncomingValueNumForOperand(unsigned i)
int getBasicBlockIndex(const BasicBlock *BB) const
Return the first index of the specified basic block in the value list for this PHI.
unsigned getNumIncomingValues() const
Return the number of incoming edges.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
Pass interface - Implemented by all 'passes'.
Definition Pass.h:99
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
This node represents an addition of some number of SCEVs.
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
SCEVUse getStepRecurrence(ScalarEvolution &SE) const
Constructs and returns the recurrence indicating how much this expression steps by.
This class represents a constant integer value.
ConstantInt * getValue() const
const APInt & getAPInt() const
This class uses information about analyze scalars to rewrite expressions in canonical form.
This node represents multiplication of some number of SCEVs.
ArrayRef< SCEVUse > operands() const
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
unsigned short getExpressionSize() const
LLVM_ABI bool isZero() const
Return true if the expression is a constant zero.
LLVM_ABI ArrayRef< SCEVUse > operands() const
Return operands of this SCEV expression.
Type * getType() const
Return the LLVM type of this SCEV expression.
static constexpr auto FlagNone
SCEVTypes getSCEVType() const
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
const SCEV * getZero(Type *Ty)
Return a SCEV for the constant 0 of a specific type.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEVFlags Flags=SCEV::FlagNone, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI uint64_t getTypeSizeInBits(Type *Ty) const
Return the size in bits of the specified type, for which isSCEVable must return true.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI const SCEV * getNoopOrSignExtend(const SCEV *V, Type *Ty)
Return a SCEV corresponding to a conversion of the input value to the specified type.
LLVM_ABI SCEVUse getAddRecExpr(SCEVUse Start, SCEVUse Step, const Loop *L, SCEVFlagsPair Flags)
Get an add recurrence expression for the specified loop.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI SCEVUse getAddExpr(SmallVectorImpl< SCEVUse > &Ops, SCEVFlagsPair Flags={}, unsigned Depth=0)
Get a canonical add expression, or something simpler if possible.
LLVM_ABI bool isSCEVable(Type *Ty) const
Test if values of the given type are analyzable within the SCEV framework.
LLVM_ABI Type * getEffectiveSCEVType(Type *Ty) const
Return a type with the same bitwidth as the given type and which represents how SCEV will treat the g...
LLVM_ABI const SCEV * getAnyExtendExpr(SCEVUse Op, Type *Ty)
getAnyExtendExpr - Return a SCEV for the given operand extended with unspecified bits out to the give...
LLVM_ABI const SCEV * getSignExtendExpr(SCEVUse Op, Type *Ty, unsigned Depth=0)
LLVM_ABI bool containsUndefs(const SCEV *S) const
Return true if the SCEV expression contains an undef value.
LLVM_ABI const SCEV * getVScale(Type *Ty)
LLVM_ABI SCEVUse getMulExpr(SmallVectorImpl< SCEVUse > &Ops, SCEVFlagsPair Flags={}, unsigned Depth=0)
Get a canonical multiply expression, or something simpler if possible.
LLVM_ABI bool hasComputableLoopEvolution(const SCEV *S, const Loop *L)
Return true if the given SCEV changes value in a known way in the specified loop.
LLVM_ABI const SCEV * getPointerBase(const SCEV *V)
Transitively follow the chain of pointer-type operands until reaching a SCEV that does not have a sin...
LLVM_ABI const SCEV * getUnknown(Value *V)
LLVM_ABI std::optional< APInt > computeConstantDifference(const SCEV *LHS, const SCEV *RHS)
Compute LHS - RHS and returns the result as an APInt if it is a constant, and std::nullopt if it isn'...
LLVM_ABI bool properlyDominates(const SCEV *S, const BasicBlock *BB)
Return true if elements that makes up the given SCEV properly dominate the specified basic block.
LLVM_ABI bool containsErasedValue(const SCEV *S) const
Return true if the SCEV expression contains a Value that has been optimised out and is now a nullptr.
LLVMContext & getContext() const
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
iterator end()
Get an iterator to the end of the SetVector.
Definition SetVector.h:118
iterator begin()
Get an iterator to the beginning of the SetVector.
Definition SetVector.h:112
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
int find_first() const
Returns the index of the first set bit, -1 if none of the bits are set.
SmallBitVector & set()
iterator_range< const_set_bits_iterator > set_bits() const
int find_next(unsigned Prev) const
Returns the index of the next set bit following the "Prev" bit.
size_type size() const
Returns the number of bits in this bitvector.
void resize(unsigned N, bool t=false)
Grow or shrink the bitvector.
size_type count() const
Returns the number of bits which are set.
SmallBitVector & reset()
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
void insert_range(Range &&R)
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
iterator erase(const_iterator CI)
typename SuperClass::const_iterator const_iterator
typename SuperClass::iterator iterator
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
static StackOffset get(int64_t Fixed, int64_t Scalable)
Definition TypeSize.h:41
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Wrapper pass for TargetTransformInfo.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
LLVM_ABI bool shouldDropLSRSolutionIfLessProfitable() const
Return true if LSR should drop a found solution if it's calculated to be less profitable than the bas...
LLVM_ABI bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const
Return true if LSR cost of C1 is lower than C2.
LLVM_ABI bool isIndexedStoreLegal(enum MemIndexedMode Mode, Type *Ty) const
LLVM_ABI unsigned getRegisterClassForType(bool Vector, Type *Ty=nullptr) const
LLVM_ABI bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace=0, Instruction *I=nullptr, int64_t ScalableOffset=0) const
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
@ TCK_RecipThroughput
Reciprocal throughput.
LLVM_ABI bool isIndexedLoadLegal(enum MemIndexedMode Mode, Type *Ty) const
LLVM_ABI bool isTypeLegal(Type *Ty) const
Return true if this type is legal.
LLVM_ABI bool isLegalAddImmediate(int64_t Imm) const
Return true if the specified immediate is legal add immediate, that is the target has add instruction...
LLVM_ABI bool canSaveCmp(Loop *L, CondBrInst **BI, ScalarEvolution *SE, LoopInfo *LI, DominatorTree *DT, AssumptionCache *AC, TargetLibraryInfo *LibInfo) const
Return true if the target can save a compare for loop count, for example hardware loop saves a compar...
LLVM_ABI unsigned getNumberOfRegisters(unsigned ClassID) const
LLVM_ABI bool isLegalAddScalableImmediate(int64_t Imm) const
Return true if adding the specified scalable immediate is legal, that is the target has add instructi...
@ TCC_Free
Expected to fold away in lowering.
LLVM_ABI bool canMacroFuseCmp() const
Return true if the target can fuse a compare and branch.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
@ AMK_PreIndexed
Prefer pre-indexed addressing mode.
@ AMK_None
Don't prefer any addressing mode.
LLVM_ABI bool isTruncateFree(Type *Ty1, Type *Ty2) const
Return true if it's free to truncate a value of type Ty1 to type Ty2.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
LLVM_ABI int getFPMantissaWidth() const
Return the width of the mantissa of this type.
Definition Type.cpp:227
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
Use * op_iterator
Definition User.h:254
op_range operands()
Definition User.h:267
op_iterator op_begin()
Definition User.h:259
void setOperand(unsigned i, Value *Val)
Definition User.h:212
LLVM_ABI bool replaceUsesOfWith(Value *From, Value *To)
Replace uses of one Value with another.
Definition User.cpp:25
Value * getOperand(unsigned i) const
Definition User.h:207
op_iterator op_end()
Definition User.h:261
static LLVM_ABI ValueAsMetadata * get(Value *V)
Definition Metadata.cpp:514
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
LLVM_ABI void printAsOperand(raw_ostream &O, bool PrintType=true, const Module *M=nullptr) const
Print the name of this Value out to the specified raw_ostream.
iterator_range< use_iterator > uses()
Definition Value.h:382
A nullable Value handle that is nullable.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
Definition DenseSet.h:187
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
Changed
This provides a very simple, boring adaptor for a begin and end iterator into a range type.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ Entry
Definition COFF.h:862
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
match_bind< const SCEVMulExpr > m_scev_Mul(const SCEVMulExpr *&V)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
cst_pred_ty< is_specific_cst > m_scev_SpecificInt(uint64_t V)
Match an SCEV constant with a plain unsigned integer.
initializer< Ty > init(const Ty &Val)
@ DW_OP_LLVM_arg
Only used in LLVM metadata.
Definition Dwarf.h:149
@ DW_OP_LLVM_convert
Only used in LLVM metadata.
Definition Dwarf.h:145
constexpr double e
Sequence
A sequence of states that a pointer may go through in which an objc_retain and objc_release are actua...
Definition PtrState.h:41
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< PhiNode * > Phi
Definition RDFGraph.h:390
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
iterator end() const
Definition BasicBlock.h:89
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
LLVM_ABI iterator begin() const
BaseReg
Stack frame base register. Bit 0 of FREInfo.Info.
Definition SFrame.h:77
unsigned KindType
For isa, dyn_cast, etc operations on TelemetryInfo.
Definition Telemetry.h:83
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
@ Offset
Definition DWP.cpp:577
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
InstructionCost Cost
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI void salvageDebugInfo(const MachineRegisterInfo &MRI, MachineInstr &MI)
Assuming the instruction MI is going to be deleted, attempt to salvage debug users of MI by writing t...
Definition Utils.cpp:1676
@ Store
The extracted value is stored (ExtractElement only).
bool operator!=(uint64_t V1, const APInt &V2)
Definition APInt.h:2139
LLVM_ABI bool DeleteDeadPHIs(BasicBlock *BB, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, SmallPtrSetImpl< PHINode * > *KnownNonDeadPHIs=nullptr)
Examine each PHI in the given block and delete it if it is dead.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
LLVM_ABI char & LoopSimplifyID
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
DomTreeNodeBase< BasicBlock > DomTreeNode
Definition Dominators.h:65
AnalysisManager< Loop, LoopStandardAnalysisResults & > LoopAnalysisManager
The loop analysis manager.
LLVM_ABI bool matchSimpleRecurrence(const PHINode *P, BinaryOperator *&BO, Value *&Start, Value *&Step)
Attempt to match a simple first order recurrence cycle of the form: iv = phi Ty [Start,...
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI void initializeLoopStrengthReducePass(PassRegistry &)
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
LLVM_ABI const SCEV * denormalizeForPostIncUse(const SCEV *S, const PostIncLoopSet &Loops, ScalarEvolution &SE)
Denormalize S to be post-increment for all loops present in Loops.
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
LLVM_ABI Constant * ConstantFoldCastOperand(unsigned Opcode, Constant *C, Type *DestTy, const DataLayout &DL)
Attempt to constant fold a cast with the specified operand.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI void SplitLandingPadPredecessors(BasicBlock *OrigBB, ArrayRef< BasicBlock * > Preds, const char *Suffix, const char *Suffix2, SmallVectorImpl< BasicBlock * > &NewBBs, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, bool PreserveLCSSA=false)
This method transforms the landing pad, OrigBB, by introducing two new basic blocks into the function...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI const SCEV * normalizeForPostIncUse(const SCEV *S, const PostIncLoopSet &Loops, ScalarEvolution &SE, bool CheckInvertible=true)
Normalize S to be post-increment for all loops present in Loops.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
iterator_range(Container &&) -> iterator_range< llvm::detail::IterOfRange< Container > >
@ Other
Any other memory.
Definition ModRef.h:68
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
TargetTransformInfo TTI
@ Add
Sum of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2028
DWARFExpression::Operation Op
LLVM_ABI Pass * createLoopStrengthReducePass()
LLVM_ABI BasicBlock * SplitCriticalEdge(Instruction *TI, unsigned SuccNum, const CriticalEdgeSplittingOptions &Options=CriticalEdgeSplittingOptions(), const Twine &BBName="")
If this edge is a critical edge, insert a new node to split the critical edge.
LLVM_ABI bool RecursivelyDeleteTriviallyDeadInstructionsPermissive(SmallVectorImpl< WeakTrackingVH > &DeadInsts, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, std::function< void(Value *)> AboutToDeleteCallback=std::function< void(Value *)>())
Same functionality as RecursivelyDeleteTriviallyDeadInstructions, but allow instructions that are not...
Definition Local.cpp:541
constexpr unsigned BitWidth
LLVM_ABI bool formLCSSAForInstructions(SmallVectorImpl< Instruction * > &Worklist, const DominatorTree &DT, const LoopInfo &LI, ScalarEvolution *SE, SmallVectorImpl< PHINode * > *PHIsToRemove=nullptr, SmallVectorImpl< PHINode * > *InsertedPHIs=nullptr)
Ensures LCSSA form for every instruction from the Worklist in the scope of innermost containing loop.
Definition LCSSA.cpp:328
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI PreservedAnalyses getLoopPassPreservedAnalyses()
Returns the minimum set of Analyses that all loop passes must preserve.
SmallPtrSet< const Loop *, 2 > PostIncLoopSet
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
LLVM_ABI int rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI, ScalarEvolution *SE, const TargetTransformInfo *TTI, SCEVExpander &Rewriter, DominatorTree *DT, ReplaceExitVal ReplaceExitValue, SmallVector< WeakTrackingVH, 16 > &DeadInsts)
If the final value of any expressions that are recurrent in the loop can be computed,...
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr bool valueOr(BoolOrDefault X, bool Default)
@ UnusedIndVarInLoop
Definition LoopUtils.h:604
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
SCEVUseT< const SCEV * > SCEVUse
bool SCEVExprContains(const SCEV *Root, PredTy Pred)
Return true if any node in Root satisfies the predicate Pred.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Attributes of a target dependent hardware loop.
The adaptor from a function pass to a loop pass computes these analyses and makes them available to t...
Information about a load/store intrinsic defined by the target.
Value * PtrVal
This is the pointer that the intrinsic is loading from or storing to.
SCEVPtrT getPointer() const