26#include "llvm/IR/IntrinsicsHexagon.h"
36#define DEBUG_TYPE "hexagon-subtarget"
38#define GET_SUBTARGETINFO_CTOR
39#define GET_SUBTARGETINFO_TARGET_DESC
40#include "HexagonGenSubtargetInfo.inc"
50 cl::desc(
"Enable the scheduler to generate .cur"));
54 cl::desc(
"Disable Hexagon MI Scheduling"));
58 cl::desc(
"If present, forces/disables the use of long calls"));
62 cl::desc(
"Consider calls to be predicable"));
72 cl::desc(
"Enable checking for cache bank conflicts"));
80 TLInfo(TM, *this), InstrItins(getInstrItineraryForCPU(CPUString)) {
84 assert(InstrItins.Itineraries !=
nullptr &&
"InstrItins not initialized");
95 UseHVX128BOps =
false;
107 return F ==
"+hvx-qfloat" ||
F ==
"-hvx-qfloat";
112 if (
F.starts_with(
"+hvxv"))
118 if (
F.starts_with(
"+hvx") ||
F ==
"-hvx")
119 return F.take_front(4);
124 bool AddQFloat =
false;
130 }
else if (HvxVer ==
"+hvx") {
139 std::string FeatureString = Features.
getString();
147 static_assert(Hexagon::R27 - Hexagon::R16 == 11,
148 "Callee-saved R16-R27 are assumed to be consecutive");
149 SCSPReg = Hexagon::R18;
150 bool SCSRegSelected =
false;
151 for (
unsigned Reg = Hexagon::R16; Reg <= Hexagon::R27; ++Reg) {
152 if (!SCSPointerRegister[Reg])
156 "Only one shadow call stack pointer register may be selected");
158 SCSRegSelected =
true;
162 UseHVXFloatingPoint = UseHVXIEEEFPOps || UseHVXQFloatOps;
164 if (UseHVXQFloatOps && UseHVXIEEEFPOps && UseHVXFloatingPoint)
166 dbgs() <<
"Behavior is undefined for simultaneous qfloat and ieee hvx codegen...");
182 setFeatureBits(FeatureBits.
reset(Hexagon::FeatureDuplex));
194 static const struct {
195 const RTLIB::Libcall
Op;
196 const RTLIB::LibcallImpl Impl;
198 {RTLIB::SDIV_I32, RTLIB::impl___hexagon_divsi3},
199 {RTLIB::SDIV_I64, RTLIB::impl___hexagon_divdi3},
200 {RTLIB::UDIV_I32, RTLIB::impl___hexagon_udivsi3},
201 {RTLIB::UDIV_I64, RTLIB::impl___hexagon_udivdi3},
202 {RTLIB::SREM_I32, RTLIB::impl___hexagon_modsi3},
203 {RTLIB::SREM_I64, RTLIB::impl___hexagon_moddi3},
204 {RTLIB::UREM_I32, RTLIB::impl___hexagon_umodsi3},
205 {RTLIB::UREM_I64, RTLIB::impl___hexagon_umoddi3},
206 {RTLIB::ADD_F64, RTLIB::impl___hexagon_adddf3},
207 {RTLIB::SUB_F64, RTLIB::impl___hexagon_subdf3},
208 {RTLIB::MUL_F64, RTLIB::impl___hexagon_muldf3},
209 {RTLIB::DIV_F64, RTLIB::impl___hexagon_divdf3},
210 {RTLIB::DIV_F32, RTLIB::impl___hexagon_divsf3},
212 for (
const auto &LC : LibraryCalls)
213 Info.setLibcallImpl(LC.Op, LC.Impl);
220 Ty = Ty.getVectorElementType();
221 if (IncludeBool && Ty == MVT::i1)
233 if (!IncludeBool && ElemTy == MVT::i1)
240 if (IncludeBool && ElemTy == MVT::i1) {
243 for (
MVT T : ElemTypes)
244 if (NumElems *
T.getSizeInBits() == 8 * HwLen)
250 if (VecWidth != 8 * HwLen && VecWidth != 16 * HwLen)
266 if (!Ty.getVectorElementType().isSimple())
269 auto isHvxTy = [
this, IncludeBool](
MVT SimpleTy) {
279 unsigned VecLen =
PowerOf2Ceil(Ty.getVectorNumElements());
282 if (SimpleTy.
isValid() && isHvxTy(SimpleTy))
296 if (
D.getKind() ==
SDep::Output &&
D.getReg() == Hexagon::USR_OVF)
298 for (
auto &E : Erase)
311 bool IsLoadMI1 = MI1.
mayLoad();
312 if (!QII->isHVXVec(MI1) || !(IsStoreMI1 || IsLoadMI1))
318 if (!QII->isHVXVec(MI2))
324 for (
SDep &PI :
SI.getSUnit()->Preds) {
325 if (PI.getSUnit() != &SU || PI.getKind() !=
SDep::Order)
328 SI.getSUnit()->setDepthDirty();
342bool HexagonSubtarget::CallMutation::shouldTFRICallBind(
344 const SUnit &Inst2)
const {
356 SUnit* LastSequentialCall =
nullptr;
367 for (
unsigned su = 0, e = DAG->
SUnits.size(); su != e; ++su) {
369 if (DAG->
SUnits[su].getInstr()->isCall())
370 LastSequentialCall = &DAG->
SUnits[su];
372 else if (DAG->
SUnits[su].getInstr()->isCompare() && LastSequentialCall)
376 shouldTFRICallBind(HII, DAG->
SUnits[su], DAG->
SUnits[su+1]))
394 if (
MI->isCopy() &&
MI->getOperand(1).getReg().isPhysical()) {
396 VRegHoldingReg[
MI->getOperand(0).getReg()] =
MI->getOperand(1).getReg();
397 LastVRegUse.
erase(
MI->getOperand(1).getReg());
402 if (MO.isUse() && !
MI->isCopy() &&
403 VRegHoldingReg.
count(MO.getReg())) {
405 LastVRegUse[VRegHoldingReg[MO.getReg()]] = &DAG->
SUnits[su];
406 }
else if (MO.isDef() && MO.getReg().isPhysical()) {
409 if (
auto It = LastVRegUse.
find(*AI); It != LastVRegUse.
end()) {
410 if (It->second != &DAG->
SUnits[su])
414 LastVRegUse.
erase(It);
433 for (
unsigned i = 0, e = DAG->
SUnits.size(); i != e; ++i) {
443 if (BaseOp0 ==
nullptr || !BaseOp0->
isReg() || !Size0.
hasValue() ||
447 for (
unsigned j = i+1, m = std::min(i+32, e); j != m; ++j) {
456 if (BaseOp1 ==
nullptr || !BaseOp1->
isReg() || !Size0.
hasValue() ||
461 if (((Offset0 ^ Offset1) & 0x18) != 0)
485 if (!Src->isInstr() || !Dst->isInstr())
496 isBestZeroLatency(Src, Dst, QII, ExclSrc, ExclDst)) {
513 std::optional<unsigned> DLatency;
514 for (
const auto &DDep : Dst->Succs) {
517 for (
unsigned OpNum = 0; OpNum < DDst->
getNumOperands(); OpNum++) {
528 std::optional<unsigned>
Latency =
529 InstrInfo.getOperandLatency(&InstrItins, *SrcInst, 0, *DDst, UseIdx);
538 DLatency = std::nullopt;
549 isBestZeroLatency(Src, Dst, QII, ExclSrc, ExclDst)) {
555 Latency = updateLatency(*SrcInst, *DstInst, IsArtificial,
Latency);
560 std::vector<std::unique_ptr<ScheduleDAGMutation>> &Mutations)
const {
561 Mutations.push_back(std::make_unique<UsrOverflowMutation>());
562 Mutations.push_back(std::make_unique<HVXMemLatencyMutation>());
563 Mutations.push_back(std::make_unique<BankConflictMutation>());
567 std::vector<std::unique_ptr<ScheduleDAGMutation>> &Mutations)
const {
568 Mutations.push_back(std::make_unique<UsrOverflowMutation>());
569 Mutations.push_back(std::make_unique<HVXMemLatencyMutation>());
573void HexagonSubtarget::anchor() {}
585int HexagonSubtarget::updateLatency(
MachineInstr &SrcInst,
600void HexagonSubtarget::restoreLatency(
SUnit *Src,
SUnit *Dst)
const {
602 for (
auto &
I : Src->Succs) {
603 if (!
I.isAssignedRegDep() ||
I.getSUnit() != Dst)
607 for (
unsigned OpNum = 0; OpNum < SrcI->
getNumOperands(); OpNum++) {
609 bool IsSameOrSubReg =
false;
613 IsSameOrSubReg = (MOReg == DepR);
617 if (MO.
isDef() && IsSameOrSubReg)
621 assert(DefIdx >= 0 &&
"Def Reg not found in Src MI");
622 MachineInstr *DstI = Dst->getInstr();
624 for (
unsigned OpNum = 0; OpNum < DstI->
getNumOperands(); OpNum++) {
625 const MachineOperand &MO = DstI->
getOperand(OpNum);
627 std::optional<unsigned>
Latency = InstrInfo.getOperandLatency(
628 &InstrItins, *SrcI, DefIdx, *DstI, OpNum);
634 bool IsArtificial =
I.isArtificial();
642 auto F =
find(Dst->Preds,
T);
644 F->setLatency(
I.getLatency());
649void HexagonSubtarget::changeLatency(SUnit *Src, SUnit *Dst,
unsigned Lat)
651 for (
auto &
I : Src->Succs) {
652 if (!
I.isAssignedRegDep() ||
I.getSUnit() != Dst)
659 auto F =
find(Dst->Preds,
T);
668 if (
I.isAssignedRegDep() &&
I.getLatency() == 0 &&
669 !
I.getSUnit()->getInstr()->isPseudo())
678bool HexagonSubtarget::isBestZeroLatency(
679 SUnit *Src, SUnit *Dst,
const HexagonInstrInfo *
TII,
680 SmallPtrSet<SUnit *, 4> &ExclSrc, SmallPtrSet<SUnit *, 4> &ExclDst)
const {
681 MachineInstr &SrcInst = *Src->getInstr();
682 MachineInstr &DstInst = *Dst->getInstr();
685 if (Dst->isBoundaryNode())
702 SUnit *Best =
nullptr;
703 SUnit *DstBest =
nullptr;
705 if (SrcBest ==
nullptr || Src->NodeNum >= SrcBest->
NodeNum) {
708 if (DstBest ==
nullptr || Dst->NodeNum <= DstBest->
NodeNum)
716 if ((Src == SrcBest && Dst == DstBest ) ||
717 (SrcBest ==
nullptr && Dst == DstBest) ||
718 (Src == SrcBest && Dst ==
nullptr))
723 if (SrcBest !=
nullptr) {
725 changeLatency(SrcBest, Dst, 1);
727 restoreLatency(SrcBest, Dst);
729 if (DstBest !=
nullptr) {
731 changeLatency(Src, DstBest, 1);
733 restoreLatency(Src, DstBest);
738 if (SrcBest && DstBest)
741 changeLatency(SrcBest, DstBest, 0);
746 for (
auto &
I : DstBest->
Preds)
747 if (ExclSrc.
count(
I.getSUnit()) == 0 &&
748 isBestZeroLatency(
I.getSUnit(), DstBest,
TII, ExclSrc, ExclDst))
749 changeLatency(
I.getSUnit(), DstBest, 0);
750 }
else if (SrcBest) {
754 for (
auto &
I : SrcBest->
Succs)
755 if (ExclDst.
count(
I.getSUnit()) == 0 &&
756 isBestZeroLatency(SrcBest,
I.getSUnit(),
TII, ExclSrc, ExclDst))
757 changeLatency(SrcBest,
I.getSUnit(), 0);
783 static Scalar ScalarInts[] = {
784#define GET_SCALAR_INTRINSICS
786#undef GET_SCALAR_INTRINSICS
789 static Hvx HvxInts[] = {
790#define GET_HVX_INTRINSICS
792#undef GET_HVX_INTRINSICS
795 const auto CmpOpcode = [](
auto A,
auto B) {
return A.Opcode <
B.Opcode; };
796 [[maybe_unused]]
static bool SortedScalar =
798 [[maybe_unused]]
static bool SortedHvx =
801 auto [BS, ES] = std::make_pair(std::begin(ScalarInts), std::end(ScalarInts));
802 auto [BH, EH] = std::make_pair(std::begin(HvxInts), std::end(HvxInts));
804 auto FoundScalar = std::lower_bound(BS, ES, Scalar{
Opc, 0}, CmpOpcode);
805 if (FoundScalar != ES && FoundScalar->Opcode ==
Opc)
806 return FoundScalar->IntId;
808 auto FoundHvx = std::lower_bound(BH, EH, Hvx{
Opc, 0, 0}, CmpOpcode);
809 if (FoundHvx != EH && FoundHvx->Opcode ==
Opc) {
812 return FoundHvx->Int64Id;
814 return FoundHvx->Int128Id;
817 std::string
error =
"Invalid opcode (" + std::to_string(
Opc) +
")";
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
const HexagonInstrInfo * TII
static cl::opt< bool > DisableHexagonMISched("disable-hexagon-misched", cl::Hidden, cl::desc("Disable Hexagon MI Scheduling"))
static cl::opt< bool > EnableDotCurSched("enable-cur-sched", cl::Hidden, cl::init(true), cl::desc("Enable the scheduler to generate .cur"))
static cl::opt< bool > EnableCheckBankConflict("hexagon-check-bank-conflict", cl::Hidden, cl::init(true), cl::desc("Enable checking for cache bank conflicts"))
static cl::opt< bool > OverrideLongCalls("hexagon-long-calls", cl::Hidden, cl::desc("If present, forces/disables the use of long calls"))
static cl::opt< bool > SchedPredsCloser("sched-preds-closer", cl::Hidden, cl::init(true))
static cl::opt< bool > SchedRetvalOptimization("sched-retval-optimization", cl::Hidden, cl::init(true))
static cl::opt< bool > EnableTCLatencySched("enable-tc-latency-sched", cl::Hidden, cl::init(false))
static cl::opt< bool > EnableBSBSched("enable-bsb-sched", cl::Hidden, cl::init(true))
static SUnit * getZeroLatency(SUnit *N, SmallVector< SDep, 4 > &Deps)
If the SUnit has a zero latency edge, return the other SUnit.
static cl::opt< bool > EnablePredicatedCalls("hexagon-pred-calls", cl::Hidden, cl::desc("Consider calls to be predicable"))
Register const TargetRegisterInfo * TRI
This file defines the SmallVector class.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
iterator find(const_arg_type_t< KeyT > Val)
bool erase(const KeyT &Val)
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Container class for subtarget features.
constexpr FeatureBitset & reset(unsigned I)
unsigned getAddrMode(const MachineInstr &MI) const
bool canExecuteInBundle(const MachineInstr &First, const MachineInstr &Second) const
Can these instructions execute at the same time in a bundle.
bool isHVXVec(const MachineInstr &MI) const
bool isToBeScheduledASAP(const MachineInstr &MI1, const MachineInstr &MI2) const
MachineOperand * getBaseAndOffset(const MachineInstr &MI, int64_t &Offset, LocationSize &AccessSize) const
uint64_t getType(const MachineInstr &MI) const
Hexagon::ArchEnum HexagonArchVersion
void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep, const TargetSchedModel *SchedModel) const override
Perform target specific adjustments to the latency of a schedule dependency.
bool usePredicatedCalls() const
const HexagonInstrInfo * getInstrInfo() const override
const HexagonRegisterInfo * getRegisterInfo() const override
void getSMSMutations(std::vector< std::unique_ptr< ScheduleDAGMutation > > &Mutations) const override
HexagonSubtarget(const Triple &TT, StringRef CPU, StringRef FS, const TargetMachine &TM)
bool isHVXVectorType(EVT VecTy, bool IncludeBool=false) const
void getPostRAMutations(std::vector< std::unique_ptr< ScheduleDAGMutation > > &Mutations) const override
const HexagonTargetLowering * getTargetLowering() const override
bool UseBSBScheduling
True if the target should use Back-Skip-Back scheduling.
unsigned getL1PrefetchDistance() const
ArrayRef< MVT > getHVXElementTypes() const
bool useHVXFloatingPoint() const
bool enableSubRegLiveness() const override
unsigned getVectorLength() const
void initLibcallLoweringInfo(LibcallLoweringInfo &Info) const override
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
ParseSubtargetFeatures - Parses features string setting specified subtarget options.
bool useHVXV68Ops() const
unsigned getL1CacheLineSize() const
bool isTypeForHVX(Type *VecTy, bool IncludeBool=false) const
Intrinsic::ID getIntrinsicId(unsigned Opc) const
HexagonSubtarget & initializeSubtargetDependencies(StringRef CPU, StringRef FS)
bool enableMachineScheduler() const override
bool useBSBScheduling() const
bool isHVXElementType(MVT Ty, bool IncludeBool=false) const
bool useAA() const override
Enable use of alias analysis during code generation (during MI scheduling, DAGCombine,...
LegalizeTypeAction getPreferredVectorAction(MVT VT) const override
Return the preferred vector type legalization action.
Tracks which library functions to use for a particular subtarget.
static LocationSize precise(uint64_t Value)
TypeSize getValue() const
MCRegAliasIterator enumerates all registers aliasing Reg.
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isValid() const
Return true if this is a valid simple valuetype.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
unsigned getNumOperands() const
Retuns the total number of operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
bool isRegSequence() const
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
Register getReg() const
getReg - Returns the register number.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
@ Output
A register output-dependence (aka WAW).
@ Order
Any other ordering dependency.
void setLatency(unsigned Lat)
Sets the latency for this edge.
@ Barrier
An unknown scheduling barrier.
@ Artificial
Arbitrary strong DAG edge (no real dependence).
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
bool isArtificial() const
Tests if this is an Order dependence that is marked as "artificial", meaning it isn't necessary for c...
Scheduling unit. This is a node in the scheduling DAG.
bool isInstr() const
Returns true if this SUnit refers to a machine instruction as opposed to an SDNode.
unsigned NodeNum
Entry # of node in the node vector.
LLVM_ABI void setHeightDirty()
Sets a flag in this node to indicate that its stored Height value will require recomputation the next...
LLVM_ABI void removePred(const SDep &D)
Removes the specified edge as a pred of the current node if it exists.
SmallVector< SDep, 4 > Succs
All sunit successors.
SmallVector< SDep, 4 > Preds
All sunit predecessors.
MachineInstr * getInstr() const
Returns the representative MachineInstr for this SUnit.
A ScheduleDAG for scheduling lists of MachineInstr.
bool addEdge(SUnit *SuccSU, const SDep &PredDep)
Add a DAG edge to the given SU with the given predecessor dependence data.
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
const TargetInstrInfo * TII
Target instruction information.
std::vector< SUnit > SUnits
The scheduling units.
MachineFunction & MF
Machine function.
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
bool consumeInteger(unsigned Radix, T &Result)
Parse the current string as an integer of the specified radix.
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
Manages the enabling and disabling of subtarget specific features.
const std::vector< std::string > & getFeatures() const
Returns the vector of individual subtarget features.
LLVM_ABI std::string getString() const
Returns features as a string.
LLVM_ABI void AddFeature(StringRef String, bool Enable=true)
Adds Features.
Primary interface to the complete machine description for the target machine.
Provide an instruction scheduling machine model to CodeGen passes.
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
Triple - Helper class for working with autoconf configuration names.
The instances of the Type class are immutable: once they are created, they are never changed.
bool isVectorTy() const
True if this is an instance of VectorType.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
void addArchSubtarget(MCSubtargetInfo const *STI, StringRef FS)
FeatureBitset completeHVXFeatures(const FeatureBitset &FB)
std::optional< Hexagon::ArchEnum > getCpu(StringRef CPU)
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto reverse(ContainerTy &&C)
void sort(IteratorTy Start, IteratorTy End)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
DWARFExpression::Operation Op
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
cl::opt< bool > HexagonDisableDuplex
Implement std::hash so that hash_code can be used in STL containers.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
void apply(ScheduleDAGInstrs *DAG) override
void apply(ScheduleDAGInstrs *DAG) override
void apply(ScheduleDAGInstrs *DAG) override
void apply(ScheduleDAGInstrs *DAG) override