33#include "llvm/IR/IntrinsicsAMDGPU.h"
41#define DEBUG_TYPE "si-instr-info"
43#define GET_INSTRINFO_CTOR_DTOR
44#include "AMDGPUGenInstrInfo.inc"
47#define GET_ImageDimIntrinsicTable_IMPL
48#define GET_RsrcIntrinsics_IMPL
49#define GET_GFX1250BlockingCyclesTable_DECL
50#define GET_GFX1250BlockingCyclesTable_IMPL
57#include "AMDGPUGenSearchableTables.inc"
65 cl::desc(
"Restrict range of branch instructions (DEBUG)"));
68 "amdgpu-fix-16-bit-physreg-copies",
69 cl::desc(
"Fix copies between 32 and 16 bit registers by extending to 32 bit"),
85 unsigned N =
Node->getNumOperands();
86 while (
N &&
Node->getOperand(
N - 1).getValueType() == MVT::Glue)
98 int Op0Idx = AMDGPU::getNamedOperandIdx(Opc0,
OpName);
99 int Op1Idx = AMDGPU::getNamedOperandIdx(Opc1,
OpName);
101 if (Op0Idx == -1 && Op1Idx == -1)
105 if ((Op0Idx == -1 && Op1Idx != -1) ||
106 (Op1Idx == -1 && Op0Idx != -1))
127 return !
MI.memoperands_empty() &&
129 return MMO->isLoad() && MMO->isInvariant();
138static std::tuple<unsigned, unsigned, unsigned>
146 unsigned LoReloc, HiReloc;
176 return {BaseFlags, LoReloc, HiReloc};
194 if (!
MI.hasImplicitDef() &&
195 MI.getNumImplicitOperands() ==
MI.getDesc().implicit_uses().size() &&
196 !
MI.mayRaiseFPException())
205 if (!
MI.getNumOperands() || !
MI.getOperand(0).isReg())
220 if (
MI.isNotDuplicable() ||
MI.mayStore() ||
MI.mayRaiseFPException() ||
221 MI.hasUnmodeledSideEffects())
226 if (
MI.isInlineAsm())
230 if (
MI.mayLoad() && !
MI.isDereferenceableInvariantLoad())
245 if (Reg.isPhysical()) {
261 if (MO.isDef() && Reg != DefReg)
271 case AMDGPU::V_SUBREV_U16_e32:
272 case AMDGPU::V_SUBREV_U16_e64:
274 case AMDGPU::V_SUBREV_U32_e32:
275 case AMDGPU::V_SUBREV_U32_e64:
277 case AMDGPU::V_SUBREV_CO_U32_e32:
278 case AMDGPU::V_SUBREV_CO_U32_e64:
280 case AMDGPU::V_SUBBREV_U32_e32:
281 case AMDGPU::V_SUBBREV_U32_e64:
284 case AMDGPU::V_ASHRREV_I16_e32:
285 case AMDGPU::V_ASHRREV_I16_e64:
286 case AMDGPU::V_ASHRREV_I32_e32:
287 case AMDGPU::V_ASHRREV_I32_e64:
288 case AMDGPU::V_ASHRREV_I64_e64:
289 case AMDGPU::V_LSHLREV_B16_e32:
290 case AMDGPU::V_LSHLREV_B16_e64:
291 case AMDGPU::V_LSHLREV_B32_e32:
292 case AMDGPU::V_LSHLREV_B32_e64:
293 case AMDGPU::V_LSHLREV_B64_e64:
294 case AMDGPU::V_LSHRREV_B16_e32:
295 case AMDGPU::V_LSHRREV_B16_e64:
296 case AMDGPU::V_LSHRREV_B32_e32:
297 case AMDGPU::V_LSHRREV_B32_e64:
298 case AMDGPU::V_LSHRREV_B64_e64:
299 return !ST.hasGFX11Insts();
306bool SIInstrInfo::resultDependsOnExec(
const MachineInstr &
MI)
const {
310 if (
MI.isConvergent())
338 if (
MI.getOpcode() == AMDGPU::SI_IF_BREAK)
343 for (
auto Op :
MI.uses()) {
344 if (
Op.isReg() &&
Op.getReg().isVirtual() &&
358 while (FromCycle && !(ToCycle && CI->
contains(FromCycle, ToCycle))) {
378 int64_t &Offset1)
const {
386 if (!
get(Opc0).mayLoad() || !
get(Opc1).mayLoad())
390 if (!
get(Opc0).getNumDefs() || !
get(Opc1).getNumDefs())
406 int Offset0Idx = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
407 int Offset1Idx = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
408 if (Offset0Idx == -1 || Offset1Idx == -1)
415 Offset0Idx -=
get(Opc0).NumDefs;
416 Offset1Idx -=
get(Opc1).NumDefs;
446 if (!Load0Offset || !Load1Offset)
463 int OffIdx0 = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
464 int OffIdx1 = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
466 if (OffIdx0 == -1 || OffIdx1 == -1)
472 OffIdx0 -=
get(Opc0).NumDefs;
473 OffIdx1 -=
get(Opc1).NumDefs;
492 case AMDGPU::DS_READ2ST64_B32:
493 case AMDGPU::DS_READ2ST64_B64:
494 case AMDGPU::DS_WRITE2ST64_B32:
495 case AMDGPU::DS_WRITE2ST64_B64:
509 OffsetIsScalable =
false;
526 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
528 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
529 if (
Opc == AMDGPU::DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64)
542 unsigned Offset0 = Offset0Op->
getImm() & 0xff;
543 unsigned Offset1 = Offset1Op->
getImm() & 0xff;
544 if (Offset0 + 1 != Offset1)
555 int Data0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
556 EltSize = RI.getRegSizeInBits(*
getOpRegClass(LdSt, Data0Idx)) / 8;
563 Offset = EltSize * Offset0;
565 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
566 if (DataOpIdx == -1) {
567 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
569 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data1);
585 if (BaseOp && !BaseOp->
isFI())
593 if (SOffset->
isReg())
599 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
601 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
610 isMIMG(LdSt) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
611 int SRsrcIdx = AMDGPU::getNamedOperandIdx(
Opc, RsrcOpName);
613 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
614 if (VAddr0Idx >= 0) {
616 for (
int I = VAddr0Idx;
I < SRsrcIdx; ++
I)
623 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
638 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::sdst);
655 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
657 DataOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
674 if (BaseOps1.
front()->isIdenticalTo(*BaseOps2.
front()))
682 if (MO1->getAddrSpace() != MO2->getAddrSpace())
685 const auto *Base1 = MO1->getValue();
686 const auto *Base2 = MO2->getValue();
687 if (!Base1 || !Base2)
695 return Base1 == Base2;
699 int64_t Offset1,
bool OffsetIsScalable1,
701 int64_t Offset2,
bool OffsetIsScalable2,
702 unsigned ClusterSize,
703 unsigned NumBytes)
const {
716 }
else if (!BaseOps1.
empty() || !BaseOps2.
empty()) {
735 const unsigned LoadSize = NumBytes / ClusterSize;
736 const unsigned NumDWords = ((LoadSize + 3) / 4) * ClusterSize;
737 return NumDWords <= MaxMemoryClusterDWords;
751 int64_t Offset0, int64_t Offset1,
752 unsigned NumLoads)
const {
753 assert(Offset1 > Offset0 &&
754 "Second offset should be larger than first offset!");
759 return (NumLoads <= 16 && (Offset1 - Offset0) < 64);
766 const char *
Msg =
"illegal VGPR to SGPR copy") {
785 assert((
TII.getSubtarget().hasMAIInsts() &&
786 !
TII.getSubtarget().hasGFX90AInsts()) &&
787 "Expected GFX908 subtarget.");
790 AMDGPU::AGPR_32RegClass.
contains(SrcReg)) &&
791 "Source register of the copy should be either an SGPR or an AGPR.");
794 "Destination register of the copy should be an AGPR.");
803 for (
auto Def =
MI,
E =
MBB.begin(); Def !=
E; ) {
806 if (!Def->modifiesRegister(SrcReg, &RI))
809 if (Def->getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
810 Def->getOperand(0).getReg() != SrcReg)
817 bool SafeToPropagate =
true;
820 for (
auto I = Def;
I !=
MI && SafeToPropagate; ++
I)
821 if (
I->modifiesRegister(DefOp.
getReg(), &RI))
822 SafeToPropagate =
false;
824 if (!SafeToPropagate)
827 for (
auto I = Def;
I !=
MI; ++
I)
828 I->clearRegisterKills(DefOp.
getReg(), &RI);
836 if (ImpUseSuperReg) {
837 Builder.addReg(ImpUseSuperReg,
845 RS.enterBasicBlockEnd(
MBB);
846 RS.backward(std::next(
MI));
855 unsigned RegNo = (DestReg - AMDGPU::AGPR0) % 3;
858 assert(
MBB.getParent()->getRegInfo().isReserved(Tmp) &&
859 "VGPR used for an intermediate copy should have been reserved.");
864 Register Tmp2 = RS.scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
MI,
874 unsigned TmpCopyOp = AMDGPU::V_MOV_B32_e32;
875 if (AMDGPU::AGPR_32RegClass.
contains(SrcReg)) {
876 TmpCopyOp = AMDGPU::V_ACCVGPR_READ_B32_e64;
883 if (ImpUseSuperReg) {
884 UseBuilder.
addReg(ImpUseSuperReg,
901 for (
unsigned Idx = 0; Idx < BaseIndices.
size(); ++Idx) {
902 int16_t SubIdx = BaseIndices[Idx];
903 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
904 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
905 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
906 unsigned Opcode = AMDGPU::S_MOV_B32;
909 bool AlignedDest = ((DestSubReg - AMDGPU::SGPR0) % 2) == 0;
910 bool AlignedSrc = ((SrcSubReg - AMDGPU::SGPR0) % 2) == 0;
911 if (AlignedDest && AlignedSrc && (Idx + 1 < BaseIndices.
size())) {
915 DestSubReg = RI.getSubReg(DestReg, SubIdx);
916 SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
917 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
918 Opcode = AMDGPU::S_MOV_B64;
933 assert(FirstMI && LastMI);
938 LastMI->addRegisterKilled(SrcReg, &RI);
944 Register SrcReg,
bool KillSrc,
bool RenamableDest,
945 bool RenamableSrc)
const {
947 unsigned Size = RI.getRegSizeInBits(*RC);
949 unsigned SrcSize = RI.getRegSizeInBits(*SrcRC);
955 if (((
Size == 16) != (SrcSize == 16))) {
957 assert(ST.useRealTrue16Insts());
959 MCRegister SubReg = RI.getSubReg(RegToFix, AMDGPU::lo16);
962 if (DestReg == SrcReg) {
968 RC = RI.getPhysRegBaseClass(DestReg);
969 Size = RI.getRegSizeInBits(*RC);
970 SrcRC = RI.getPhysRegBaseClass(SrcReg);
971 SrcSize = RI.getRegSizeInBits(*SrcRC);
975 if (RC == &AMDGPU::VGPR_32RegClass) {
977 AMDGPU::SReg_32RegClass.
contains(SrcReg) ||
978 AMDGPU::AGPR_32RegClass.
contains(SrcReg));
979 unsigned Opc = AMDGPU::AGPR_32RegClass.contains(SrcReg) ?
980 AMDGPU::V_ACCVGPR_READ_B32_e64 : AMDGPU::V_MOV_B32_e32;
986 if (RC == &AMDGPU::SReg_32_XM0RegClass ||
987 RC == &AMDGPU::SReg_32RegClass) {
988 if (SrcReg == AMDGPU::SCC) {
995 if (!AMDGPU::SReg_32RegClass.
contains(SrcReg)) {
996 if (DestReg == AMDGPU::VCC_LO) {
1014 if (RC == &AMDGPU::SReg_64RegClass) {
1015 if (SrcReg == AMDGPU::SCC) {
1022 if (!AMDGPU::SReg_64_EncodableRegClass.
contains(SrcReg)) {
1023 if (DestReg == AMDGPU::VCC) {
1041 if (DestReg == AMDGPU::SCC) {
1044 if (AMDGPU::SReg_64RegClass.
contains(SrcReg)) {
1048 assert(ST.hasScalarCompareEq64());
1062 if (RC == &AMDGPU::AGPR_32RegClass) {
1063 if (AMDGPU::VGPR_32RegClass.
contains(SrcReg) ||
1064 (ST.hasGFX90AInsts() && AMDGPU::SReg_32RegClass.contains(SrcReg))) {
1070 if (AMDGPU::AGPR_32RegClass.
contains(SrcReg) && ST.hasGFX90AInsts()) {
1079 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1086 AMDGPU::SReg_LO16RegClass.
contains(SrcReg) ||
1087 AMDGPU::AGPR_LO16RegClass.
contains(SrcReg));
1089 bool IsSGPRDst = AMDGPU::SReg_LO16RegClass.contains(DestReg);
1090 bool IsSGPRSrc = AMDGPU::SReg_LO16RegClass.contains(SrcReg);
1091 bool IsAGPRDst = AMDGPU::AGPR_LO16RegClass.contains(DestReg);
1092 bool IsAGPRSrc = AMDGPU::AGPR_LO16RegClass.contains(SrcReg);
1095 MCRegister NewDestReg = RI.get32BitRegister(DestReg);
1096 MCRegister NewSrcReg = RI.get32BitRegister(SrcReg);
1109 if (IsAGPRDst || IsAGPRSrc) {
1110 if (!DstLow || !SrcLow) {
1112 "Cannot use hi16 subreg with an AGPR!");
1119 if (ST.useRealTrue16Insts()) {
1125 if (AMDGPU::VGPR_16_Lo128RegClass.
contains(DestReg) &&
1126 (IsSGPRSrc || AMDGPU::VGPR_16_Lo128RegClass.
contains(SrcReg))) {
1138 if (IsSGPRSrc && !ST.hasSDWAScalar()) {
1139 if (!DstLow || !SrcLow) {
1141 "Cannot use hi16 subreg on VI!");
1167 unsigned SrcOp = 1) {
1171 return DstOpRC && SrcOpRC && DstOpRC->
contains(Dst) &&
1175 if (RC == RI.getVGPR64Class() && (SrcRC == RC || RI.isSGPRClass(SrcRC))) {
1176 if (ST.hasVMovB64Inst() &&
1177 CanCopyWith(AMDGPU::V_MOV_B64_e32, DestReg, SrcReg)) {
1182 if (ST.hasPkMovB32() &&
1183 CanCopyWith(AMDGPU::V_PK_MOV_B32, DestReg, SrcReg, 2)) {
1199 const bool Forward = RI.getHWRegIndex(DestReg) <= RI.getHWRegIndex(SrcReg);
1200 if (RI.isSGPRClass(RC)) {
1201 if (!RI.isSGPRClass(SrcRC)) {
1205 const bool CanKillSuperReg = KillSrc && !RI.regsOverlap(SrcReg, DestReg);
1211 unsigned Opcode = AMDGPU::V_MOV_B32_e32;
1212 unsigned WideOpcode = AMDGPU::INSTRUCTION_LIST_END;
1213 if (RI.isAGPRClass(RC)) {
1214 if (ST.hasGFX90AInsts() && RI.isAGPRClass(SrcRC))
1215 Opcode = AMDGPU::V_ACCVGPR_MOV_B32;
1216 else if (RI.hasVGPRs(SrcRC) ||
1217 (ST.hasGFX90AInsts() && RI.isSGPRClass(SrcRC)))
1218 Opcode = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
1220 Opcode = AMDGPU::INSTRUCTION_LIST_END;
1221 }
else if (RI.hasVGPRs(RC) && RI.isAGPRClass(SrcRC)) {
1222 Opcode = AMDGPU::V_ACCVGPR_READ_B32_e64;
1223 }
else if (RI.isVGPRClass(RC)) {
1224 if (ST.hasVMovB64Inst())
1225 WideOpcode = AMDGPU::V_MOV_B64_e32;
1226 else if (ST.hasPkMovB32())
1227 WideOpcode = AMDGPU::V_PK_MOV_B32;
1231 if (WideOpcode != AMDGPU::INSTRUCTION_LIST_END) {
1233 unsigned SrcOp = WideOpcode == AMDGPU::V_PK_MOV_B32 ? 2 : 1;
1240 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1241 const bool CanKillSuperReg = KillSrc && !Overlap;
1248 std::unique_ptr<RegScavenger> RS;
1249 if (Opcode == AMDGPU::INSTRUCTION_LIST_END)
1250 RS = std::make_unique<RegScavenger>();
1254 for (
unsigned Idx{}; Idx < SubIndices.
size();) {
1255 unsigned NumRegs = 1;
1256 unsigned ThisOpcode = Opcode;
1258 Forward ? SubIndices[Idx] : SubIndices[SubIndices.
size() - Idx - 1];
1260 if (WideDstRC && WideSrcRC && Idx + 1 < SubIndices.
size()) {
1261 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
1265 unsigned WideSubIdx = RI.getSubRegFromChannel(Channel, 2);
1266 Register WideDst = RI.getSubReg(DestReg, WideSubIdx);
1267 Register WideSrc = RI.getSubReg(SrcReg, WideSubIdx);
1269 if (WideDst && WideSrc && WideDstRC->
contains(WideDst) &&
1270 WideSrcRC->contains(WideSrc)) {
1271 SubIdx = WideSubIdx;
1273 ThisOpcode = WideOpcode;
1277 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
1278 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
1279 assert(DestSubReg && SrcSubReg &&
"Failed to find subregs!");
1282 bool UseKill = CanKillSuperReg && Idx == SubIndices.
size();
1284 if (ThisOpcode == AMDGPU::INSTRUCTION_LIST_END) {
1287 *RS, Overlap, ImpUseSuper);
1288 }
else if (ThisOpcode == AMDGPU::V_PK_MOV_B32) {
1329 int64_t &ImmVal)
const {
1330 switch (
MI.getOpcode()) {
1331 case AMDGPU::V_MOV_B32_e32:
1332 case AMDGPU::S_MOV_B32:
1333 case AMDGPU::S_MOVK_I32:
1334 case AMDGPU::S_MOV_B64:
1335 case AMDGPU::V_MOV_B64_e32:
1336 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
1337 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
1338 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
1339 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
1340 case AMDGPU::V_MOV_B64_PSEUDO:
1341 case AMDGPU::V_MOV_B16_t16_e32: {
1344 ImmVal = Src0.getImm();
1345 return MI.getOperand(0).getReg() == Reg;
1350 case AMDGPU::V_MOV_B16_t16_e64: {
1352 if (Src0.isImm() && !
MI.getOperand(1).getImm()) {
1353 ImmVal = Src0.getImm();
1354 return MI.getOperand(0).getReg() == Reg;
1359 case AMDGPU::S_BREV_B32:
1360 case AMDGPU::V_BFREV_B32_e32:
1361 case AMDGPU::V_BFREV_B32_e64: {
1365 return MI.getOperand(0).getReg() == Reg;
1370 case AMDGPU::S_NOT_B32:
1371 case AMDGPU::V_NOT_B32_e32:
1372 case AMDGPU::V_NOT_B32_e64: {
1375 ImmVal =
static_cast<int64_t
>(~static_cast<int32_t>(Src0.getImm()));
1376 return MI.getOperand(0).getReg() == Reg;
1386std::optional<int64_t>
1396 if (!
Op.isReg() || !
Op.getReg().isVirtual())
1397 return std::nullopt;
1399 if (Def && Def->isMoveImmediate()) {
1401 if (ImmSrc.
isImm()) {
1408 return std::nullopt;
1411std::optional<int64_t>
1420 if (RI.isAGPRClass(DstRC))
1421 return AMDGPU::COPY;
1422 if (RI.getRegSizeInBits(*DstRC) == 16) {
1425 return RI.isSGPRClass(DstRC) ? AMDGPU::COPY : AMDGPU::V_MOV_B16_t16_e64;
1427 if (RI.getRegSizeInBits(*DstRC) == 32)
1428 return RI.isSGPRClass(DstRC) ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1429 if (RI.getRegSizeInBits(*DstRC) == 64 && RI.isSGPRClass(DstRC))
1430 return AMDGPU::S_MOV_B64;
1431 if (RI.getRegSizeInBits(*DstRC) == 64 && !RI.isSGPRClass(DstRC))
1432 return AMDGPU::V_MOV_B64_PSEUDO;
1433 return AMDGPU::COPY;
1438 bool IsIndirectSrc)
const {
1439 if (IsIndirectSrc) {
1441 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1);
1443 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2);
1445 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3);
1447 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4);
1449 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5);
1451 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6);
1453 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7);
1455 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8);
1457 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9);
1459 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10);
1461 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11);
1463 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12);
1465 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16);
1466 if (VecSize <= 1024)
1467 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32);
1473 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1);
1475 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2);
1477 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3);
1479 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4);
1481 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5);
1483 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6);
1485 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7);
1487 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8);
1489 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9);
1491 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10);
1493 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11);
1495 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12);
1497 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16);
1498 if (VecSize <= 1024)
1499 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32);
1506 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1508 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1510 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1512 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1514 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1516 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1518 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1520 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1522 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1524 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1526 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1528 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1530 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1531 if (VecSize <= 1024)
1532 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1539 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1541 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1543 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1545 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1547 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1549 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1551 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1553 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1555 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1557 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1559 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1561 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1563 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1564 if (VecSize <= 1024)
1565 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1572 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1;
1574 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2;
1576 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4;
1578 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8;
1579 if (VecSize <= 1024)
1580 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16;
1587 bool IsSGPR)
const {
1599 assert(EltSize == 32 &&
"invalid reg indexing elt size");
1606 return NeedsCFI ? AMDGPU::SI_SPILL_S32_CFI_SAVE : AMDGPU::SI_SPILL_S32_SAVE;
1608 return NeedsCFI ? AMDGPU::SI_SPILL_S64_CFI_SAVE : AMDGPU::SI_SPILL_S64_SAVE;
1610 return NeedsCFI ? AMDGPU::SI_SPILL_S96_CFI_SAVE : AMDGPU::SI_SPILL_S96_SAVE;
1612 return NeedsCFI ? AMDGPU::SI_SPILL_S128_CFI_SAVE
1613 : AMDGPU::SI_SPILL_S128_SAVE;
1615 return NeedsCFI ? AMDGPU::SI_SPILL_S160_CFI_SAVE
1616 : AMDGPU::SI_SPILL_S160_SAVE;
1618 return NeedsCFI ? AMDGPU::SI_SPILL_S192_CFI_SAVE
1619 : AMDGPU::SI_SPILL_S192_SAVE;
1621 return NeedsCFI ? AMDGPU::SI_SPILL_S224_CFI_SAVE
1622 : AMDGPU::SI_SPILL_S224_SAVE;
1624 return AMDGPU::SI_SPILL_S256_SAVE;
1626 return AMDGPU::SI_SPILL_S288_SAVE;
1628 return AMDGPU::SI_SPILL_S320_SAVE;
1630 return AMDGPU::SI_SPILL_S352_SAVE;
1632 return AMDGPU::SI_SPILL_S384_SAVE;
1634 return NeedsCFI ? AMDGPU::SI_SPILL_S512_CFI_SAVE
1635 : AMDGPU::SI_SPILL_S512_SAVE;
1637 return NeedsCFI ? AMDGPU::SI_SPILL_S1024_CFI_SAVE
1638 : AMDGPU::SI_SPILL_S1024_SAVE;
1647 return AMDGPU::SI_SPILL_V16_SAVE;
1649 return NeedsCFI ? AMDGPU::SI_SPILL_V32_CFI_SAVE : AMDGPU::SI_SPILL_V32_SAVE;
1651 return NeedsCFI ? AMDGPU::SI_SPILL_V64_CFI_SAVE : AMDGPU::SI_SPILL_V64_SAVE;
1653 return NeedsCFI ? AMDGPU::SI_SPILL_V96_CFI_SAVE : AMDGPU::SI_SPILL_V96_SAVE;
1655 return NeedsCFI ? AMDGPU::SI_SPILL_V128_CFI_SAVE
1656 : AMDGPU::SI_SPILL_V128_SAVE;
1658 return NeedsCFI ? AMDGPU::SI_SPILL_V160_CFI_SAVE
1659 : AMDGPU::SI_SPILL_V160_SAVE;
1661 return NeedsCFI ? AMDGPU::SI_SPILL_V192_CFI_SAVE
1662 : AMDGPU::SI_SPILL_V192_SAVE;
1664 return NeedsCFI ? AMDGPU::SI_SPILL_V224_CFI_SAVE
1665 : AMDGPU::SI_SPILL_V224_SAVE;
1667 return NeedsCFI ? AMDGPU::SI_SPILL_V256_CFI_SAVE
1668 : AMDGPU::SI_SPILL_V256_SAVE;
1670 return NeedsCFI ? AMDGPU::SI_SPILL_V288_CFI_SAVE
1671 : AMDGPU::SI_SPILL_V288_SAVE;
1673 return NeedsCFI ? AMDGPU::SI_SPILL_V320_CFI_SAVE
1674 : AMDGPU::SI_SPILL_V320_SAVE;
1676 return NeedsCFI ? AMDGPU::SI_SPILL_V352_CFI_SAVE
1677 : AMDGPU::SI_SPILL_V352_SAVE;
1679 return NeedsCFI ? AMDGPU::SI_SPILL_V384_CFI_SAVE
1680 : AMDGPU::SI_SPILL_V384_SAVE;
1682 return NeedsCFI ? AMDGPU::SI_SPILL_V512_CFI_SAVE
1683 : AMDGPU::SI_SPILL_V512_SAVE;
1685 return NeedsCFI ? AMDGPU::SI_SPILL_V1024_CFI_SAVE
1686 : AMDGPU::SI_SPILL_V1024_SAVE;
1695 return NeedsCFI ? AMDGPU::SI_SPILL_AV32_CFI_SAVE
1696 : AMDGPU::SI_SPILL_AV32_SAVE;
1698 return NeedsCFI ? AMDGPU::SI_SPILL_AV64_CFI_SAVE
1699 : AMDGPU::SI_SPILL_AV64_SAVE;
1701 return NeedsCFI ? AMDGPU::SI_SPILL_AV96_CFI_SAVE
1702 : AMDGPU::SI_SPILL_AV96_SAVE;
1704 return NeedsCFI ? AMDGPU::SI_SPILL_AV128_CFI_SAVE
1705 : AMDGPU::SI_SPILL_AV128_SAVE;
1707 return NeedsCFI ? AMDGPU::SI_SPILL_AV160_CFI_SAVE
1708 : AMDGPU::SI_SPILL_AV160_SAVE;
1710 return NeedsCFI ? AMDGPU::SI_SPILL_AV192_CFI_SAVE
1711 : AMDGPU::SI_SPILL_AV192_SAVE;
1713 return NeedsCFI ? AMDGPU::SI_SPILL_AV224_CFI_SAVE
1714 : AMDGPU::SI_SPILL_AV224_SAVE;
1716 return NeedsCFI ? AMDGPU::SI_SPILL_AV256_CFI_SAVE
1717 : AMDGPU::SI_SPILL_AV256_SAVE;
1719 return AMDGPU::SI_SPILL_AV288_SAVE;
1721 return AMDGPU::SI_SPILL_AV320_SAVE;
1723 return AMDGPU::SI_SPILL_AV352_SAVE;
1725 return AMDGPU::SI_SPILL_AV384_SAVE;
1727 return NeedsCFI ? AMDGPU::SI_SPILL_AV512_CFI_SAVE
1728 : AMDGPU::SI_SPILL_AV512_SAVE;
1730 return NeedsCFI ? AMDGPU::SI_SPILL_AV1024_CFI_SAVE
1731 : AMDGPU::SI_SPILL_AV1024_SAVE;
1738 bool IsVectorSuperClass) {
1743 if (IsVectorSuperClass)
1744 return AMDGPU::SI_SPILL_WWM_AV32_SAVE;
1746 return AMDGPU::SI_SPILL_WWM_V32_SAVE;
1752 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1759 if (ST.hasMAIInsts())
1765void SIInstrInfo::storeRegToStackSlotImpl(
1778 FrameInfo.getObjectAlign(FrameIndex));
1779 unsigned SpillSize = RI.getSpillSize(*RC);
1785 assert(SrcReg != AMDGPU::M0 &&
"m0 should not be spilled");
1786 assert(SrcReg != AMDGPU::EXEC_LO && SrcReg != AMDGPU::EXEC_HI &&
1787 SrcReg != AMDGPU::EXEC &&
"exec should not be spilled");
1796 if (SrcReg.
isVirtual() && SpillSize == 4) {
1810 SpillSize, *MFI, NeedsCFI);
1825 storeRegToStackSlotImpl(
MBB,
MI, SrcReg, isKill, FrameIndex, RC, VReg, Flags,
1834 storeRegToStackSlotImpl(
MBB,
MI, SrcReg, isKill, FrameIndex, RC,
Register(),
1841 return AMDGPU::SI_SPILL_S32_RESTORE;
1843 return AMDGPU::SI_SPILL_S64_RESTORE;
1845 return AMDGPU::SI_SPILL_S96_RESTORE;
1847 return AMDGPU::SI_SPILL_S128_RESTORE;
1849 return AMDGPU::SI_SPILL_S160_RESTORE;
1851 return AMDGPU::SI_SPILL_S192_RESTORE;
1853 return AMDGPU::SI_SPILL_S224_RESTORE;
1855 return AMDGPU::SI_SPILL_S256_RESTORE;
1857 return AMDGPU::SI_SPILL_S288_RESTORE;
1859 return AMDGPU::SI_SPILL_S320_RESTORE;
1861 return AMDGPU::SI_SPILL_S352_RESTORE;
1863 return AMDGPU::SI_SPILL_S384_RESTORE;
1865 return AMDGPU::SI_SPILL_S512_RESTORE;
1867 return AMDGPU::SI_SPILL_S1024_RESTORE;
1876 return AMDGPU::SI_SPILL_V16_RESTORE;
1878 return AMDGPU::SI_SPILL_V32_RESTORE;
1880 return AMDGPU::SI_SPILL_V64_RESTORE;
1882 return AMDGPU::SI_SPILL_V96_RESTORE;
1884 return AMDGPU::SI_SPILL_V128_RESTORE;
1886 return AMDGPU::SI_SPILL_V160_RESTORE;
1888 return AMDGPU::SI_SPILL_V192_RESTORE;
1890 return AMDGPU::SI_SPILL_V224_RESTORE;
1892 return AMDGPU::SI_SPILL_V256_RESTORE;
1894 return AMDGPU::SI_SPILL_V288_RESTORE;
1896 return AMDGPU::SI_SPILL_V320_RESTORE;
1898 return AMDGPU::SI_SPILL_V352_RESTORE;
1900 return AMDGPU::SI_SPILL_V384_RESTORE;
1902 return AMDGPU::SI_SPILL_V512_RESTORE;
1904 return AMDGPU::SI_SPILL_V1024_RESTORE;
1913 return AMDGPU::SI_SPILL_AV32_RESTORE;
1915 return AMDGPU::SI_SPILL_AV64_RESTORE;
1917 return AMDGPU::SI_SPILL_AV96_RESTORE;
1919 return AMDGPU::SI_SPILL_AV128_RESTORE;
1921 return AMDGPU::SI_SPILL_AV160_RESTORE;
1923 return AMDGPU::SI_SPILL_AV192_RESTORE;
1925 return AMDGPU::SI_SPILL_AV224_RESTORE;
1927 return AMDGPU::SI_SPILL_AV256_RESTORE;
1929 return AMDGPU::SI_SPILL_AV288_RESTORE;
1931 return AMDGPU::SI_SPILL_AV320_RESTORE;
1933 return AMDGPU::SI_SPILL_AV352_RESTORE;
1935 return AMDGPU::SI_SPILL_AV384_RESTORE;
1937 return AMDGPU::SI_SPILL_AV512_RESTORE;
1939 return AMDGPU::SI_SPILL_AV1024_RESTORE;
1946 bool IsVectorSuperClass) {
1951 if (IsVectorSuperClass)
1952 return AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
1954 return AMDGPU::SI_SPILL_WWM_V32_RESTORE;
1960 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1967 if (ST.hasMAIInsts())
1970 assert(!RI.isAGPRClass(RC));
1984 unsigned SpillSize = RI.getSpillSize(*RC);
1991 FrameInfo.getObjectAlign(FrameIndex));
1993 if (RI.isSGPRClass(RC)) {
1996 assert(DestReg != AMDGPU::M0 &&
"m0 should not be reloaded into");
1997 assert(DestReg != AMDGPU::EXEC_LO && DestReg != AMDGPU::EXEC_HI &&
1998 DestReg != AMDGPU::EXEC &&
"exec should not be spilled");
2003 if (DestReg.
isVirtual() && SpillSize == 4) {
2032 unsigned Quantity)
const {
2034 unsigned MaxSNopCount = 1u << ST.getSNopBits();
2035 while (Quantity > 0) {
2036 unsigned Arg = std::min(Quantity, MaxSNopCount);
2047 constexpr unsigned DoorbellIDMask = 0x3ff;
2048 constexpr unsigned ECQueueWaveAbort = 0x400;
2053 if (!
MBB.succ_empty() || std::next(
MI.getIterator()) !=
MBB.end()) {
2054 MBB.splitAt(
MI,
false);
2058 MBB.addSuccessor(TrapBB);
2068 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::TTMP2)
2072 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_AND_B32), DoorbellRegMasked)
2078 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_OR_B32), SetWaveAbortBit)
2079 .
addUse(DoorbellRegMasked)
2080 .
addImm(ECQueueWaveAbort)
2082 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2083 .
addUse(SetWaveAbortBit);
2086 BuildMI(*TrapBB, TrapBB->
end(),
DL,
get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2097 return MBB.getNextNode();
2101 switch (
MI.getOpcode()) {
2103 if (
MI.isMetaInstruction())
2108 return MI.getOperand(0).getImm() + 1;
2119 switch (
MI.getOpcode()) {
2121 case AMDGPU::S_MOV_B64_term:
2124 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2127 case AMDGPU::S_MOV_B32_term:
2130 MI.setDesc(
get(AMDGPU::S_MOV_B32));
2133 case AMDGPU::S_XOR_B64_term:
2136 MI.setDesc(
get(AMDGPU::S_XOR_B64));
2139 case AMDGPU::S_XOR_B32_term:
2142 MI.setDesc(
get(AMDGPU::S_XOR_B32));
2144 case AMDGPU::S_OR_B64_term:
2147 MI.setDesc(
get(AMDGPU::S_OR_B64));
2149 case AMDGPU::S_OR_B32_term:
2152 MI.setDesc(
get(AMDGPU::S_OR_B32));
2155 case AMDGPU::S_ANDN2_B64_term:
2158 MI.setDesc(
get(AMDGPU::S_ANDN2_B64));
2161 case AMDGPU::S_ANDN2_B32_term:
2164 MI.setDesc(
get(AMDGPU::S_ANDN2_B32));
2167 case AMDGPU::S_AND_B64_term:
2170 MI.setDesc(
get(AMDGPU::S_AND_B64));
2173 case AMDGPU::S_AND_B32_term:
2176 MI.setDesc(
get(AMDGPU::S_AND_B32));
2179 case AMDGPU::S_AND_SAVEEXEC_B64_term:
2182 MI.setDesc(
get(AMDGPU::S_AND_SAVEEXEC_B64));
2185 case AMDGPU::S_AND_SAVEEXEC_B32_term:
2188 MI.setDesc(
get(AMDGPU::S_AND_SAVEEXEC_B32));
2191 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
2192 MI.setDesc(
get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32));
2194 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
2195 MI.setDesc(
get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32));
2198 case AMDGPU::SI_SPILL_S32_TO_VGPR:
2199 MI.setDesc(
get(AMDGPU::V_WRITELANE_B32));
2202 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
2203 MI.setDesc(
get(AMDGPU::V_READLANE_B32));
2205 case AMDGPU::AV_MOV_B32_IMM_PSEUDO: {
2209 get(IsAGPR ? AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::V_MOV_B32_e32));
2212 case AMDGPU::AV_MOV_B64_IMM_PSEUDO: {
2215 int64_t
Imm =
MI.getOperand(1).getImm();
2217 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2218 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2223 MI.eraseFromParent();
2229 case AMDGPU::V_MOV_B64_PSEUDO: {
2231 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2232 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2240 if (ST.hasVMovB64Inst() && Mov64RC->
contains(Dst)) {
2241 MI.setDesc(Mov64Desc);
2245 (
SrcOp.isGlobal() && ST.has64BitLiterals()))
2248 if (
SrcOp.isGlobal()) {
2253 unsigned BaseFlags, LoReloc, HiReloc;
2254 std::tie(BaseFlags, LoReloc, HiReloc) =
2261 }
else if (
SrcOp.isImm()) {
2263 APInt Lo(32,
Imm.getLoBits(32).getZExtValue());
2264 APInt Hi(32,
Imm.getHiBits(32).getZExtValue());
2288 if (ST.hasPkMovB32() &&
2307 MI.eraseFromParent();
2310 case AMDGPU::V_MOV_B64_DPP_PSEUDO: {
2314 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2318 if (ST.has64BitLiterals()) {
2319 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2323 if (
SrcOp.isGlobal()) {
2325 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2326 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2329 unsigned BaseFlags, LoReloc, HiReloc;
2330 std::tie(BaseFlags, LoReloc, HiReloc) =
2337 MI.eraseFromParent();
2344 MI.setDesc(
get(AMDGPU::S_MOV_B64));
2349 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2350 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2352 APInt Lo(32,
Imm.getLoBits(32).getZExtValue());
2353 APInt Hi(32,
Imm.getHiBits(32).getZExtValue());
2358 MI.eraseFromParent();
2361 case AMDGPU::V_SET_INACTIVE_B32: {
2365 .
add(
MI.getOperand(3))
2366 .
add(
MI.getOperand(4))
2367 .
add(
MI.getOperand(1))
2368 .
add(
MI.getOperand(2))
2369 .
add(
MI.getOperand(5));
2370 MI.eraseFromParent();
2373 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2374 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2375 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2376 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2377 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2378 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2379 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2380 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2381 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2382 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2383 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2384 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2385 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2386 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2387 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2388 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2389 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2390 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2391 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2392 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2393 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2394 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2395 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2396 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2397 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2398 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2399 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2400 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2401 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1:
2402 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2:
2403 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4:
2404 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8:
2405 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16: {
2409 if (RI.hasVGPRs(EltRC)) {
2410 Opc = AMDGPU::V_MOVRELD_B32_e32;
2412 Opc = RI.getRegSizeInBits(*EltRC) == 64 ? AMDGPU::S_MOVRELD_B64
2413 : AMDGPU::S_MOVRELD_B32;
2418 bool IsUndef =
MI.getOperand(1).isUndef();
2419 unsigned SubReg =
MI.getOperand(3).getImm();
2420 assert(VecReg ==
MI.getOperand(1).getReg());
2425 .
add(
MI.getOperand(2))
2429 const int ImpDefIdx =
2431 const int ImpUseIdx = ImpDefIdx + 1;
2433 MI.eraseFromParent();
2436 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1:
2437 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2:
2438 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3:
2439 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4:
2440 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5:
2441 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6:
2442 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7:
2443 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8:
2444 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9:
2445 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10:
2446 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11:
2447 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12:
2448 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16:
2449 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32: {
2450 assert(ST.useVGPRIndexMode());
2452 bool IsUndef =
MI.getOperand(1).isUndef();
2461 const MCInstrDesc &OpDesc =
get(AMDGPU::V_MOV_B32_indirect_write);
2465 .
add(
MI.getOperand(2))
2469 const int ImpDefIdx =
2471 const int ImpUseIdx = ImpDefIdx + 1;
2478 MI.eraseFromParent();
2481 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1:
2482 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2:
2483 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3:
2484 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4:
2485 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5:
2486 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6:
2487 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7:
2488 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8:
2489 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9:
2490 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10:
2491 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11:
2492 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12:
2493 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16:
2494 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32: {
2495 assert(ST.useVGPRIndexMode());
2498 bool IsUndef =
MI.getOperand(1).isUndef();
2502 .
add(
MI.getOperand(2))
2515 MI.eraseFromParent();
2518 case AMDGPU::SI_PC_ADD_REL_OFFSET: {
2521 Register RegLo = RI.getSubReg(Reg, AMDGPU::sub0);
2522 Register RegHi = RI.getSubReg(Reg, AMDGPU::sub1);
2541 if (ST.hasGetPCZeroExtension()) {
2545 BuildMI(MF,
DL,
get(AMDGPU::S_SEXT_I32_I16), RegHi).addReg(RegHi));
2552 BuildMI(MF,
DL,
get(AMDGPU::S_ADD_U32), RegLo).addReg(RegLo).add(OpLo));
2562 MI.eraseFromParent();
2565 case AMDGPU::SI_PC_ADD_REL_OFFSET64: {
2575 Op.setOffset(
Op.getOffset() + 4);
2577 BuildMI(MF,
DL,
get(AMDGPU::S_ADD_U64), Reg).addReg(Reg).add(
Op));
2581 MI.eraseFromParent();
2584 case AMDGPU::ENTER_STRICT_WWM: {
2590 case AMDGPU::ENTER_STRICT_WQM: {
2597 MI.eraseFromParent();
2600 case AMDGPU::EXIT_STRICT_WWM:
2601 case AMDGPU::EXIT_STRICT_WQM: {
2607 case AMDGPU::SI_RETURN: {
2621 MI.eraseFromParent();
2625 case AMDGPU::S_MUL_U64_U32_PSEUDO:
2626 case AMDGPU::S_MUL_I64_I32_PSEUDO:
2627 MI.setDesc(
get(AMDGPU::S_MUL_U64));
2630 case AMDGPU::S_GETPC_B64_pseudo:
2631 MI.setDesc(
get(AMDGPU::S_GETPC_B64));
2632 if (ST.hasGetPCZeroExtension()) {
2634 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2643 case AMDGPU::V_MAX_BF16_PSEUDO_e64: {
2644 assert(ST.hasBF16PackedInsts());
2645 MI.setDesc(
get(AMDGPU::V_PK_MAX_NUM_BF16));
2656 case AMDGPU::GET_STACK_BASE:
2659 if (ST.getFrameLowering()->mayReserveScratchForCWSR(*
MBB.getParent())) {
2666 Register DestReg =
MI.getOperand(0).getReg();
2676 MI.getOperand(
MI.getNumExplicitOperands()).setIsDead(
false);
2677 MI.getOperand(
MI.getNumExplicitOperands()).setIsUse();
2678 MI.setDesc(
get(AMDGPU::S_CMOVK_I32));
2681 MI.setDesc(
get(AMDGPU::S_MOV_B32));
2684 MI.getNumExplicitOperands());
2702 case AMDGPU::S_MOV_B64:
2703 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2712 if (UsedLanes.
all())
2717 unsigned LoSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub0);
2718 unsigned HiSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub1);
2720 bool NeedLo = (UsedLanes & RI.getSubRegIndexLaneMask(LoSubReg)).any();
2721 bool NeedHi = (UsedLanes & RI.getSubRegIndexLaneMask(HiSubReg)).any();
2723 if (NeedLo && NeedHi)
2727 int32_t Imm32 = NeedLo ?
Lo_32(Imm64) :
Hi_32(Imm64);
2729 unsigned UseSubReg = NeedLo ? LoSubReg : HiSubReg;
2738 case AMDGPU::S_LOAD_DWORDX16_IMM:
2739 case AMDGPU::S_LOAD_DWORDX8_IMM: {
2752 for (
auto &CandMO :
I->operands()) {
2753 if (!CandMO.isReg() || CandMO.getReg() != RegToFind || CandMO.isDef())
2761 if (!UseMO || UseMO->
getSubReg() == AMDGPU::NoSubRegister)
2765 unsigned SubregSize = RI.getSubRegIdxSize(UseMO->
getSubReg());
2771 unsigned NewOpcode = -1;
2772 if (SubregSize == 256)
2773 NewOpcode = AMDGPU::S_LOAD_DWORDX8_IMM;
2774 else if (SubregSize == 128)
2775 NewOpcode = AMDGPU::S_LOAD_DWORDX4_IMM;
2785 UseMO->
setSubReg(AMDGPU::NoSubRegister);
2790 MI->getOperand(0).setReg(DestReg);
2791 MI->getOperand(0).setSubReg(AMDGPU::NoSubRegister);
2795 OffsetMO->
setImm(FinalOffset);
2801 MI->setMemRefs(*MF, NewMMOs);
2814std::pair<MachineInstr*, MachineInstr*>
2816 assert (
MI.getOpcode() == AMDGPU::V_MOV_B64_DPP_PSEUDO);
2818 if (ST.hasVMovB64Inst() && ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
2821 MI.setDesc(
get(AMDGPU::V_MOV_B64_dpp));
2822 return std::pair(&
MI,
nullptr);
2833 for (
auto Sub : { AMDGPU::sub0, AMDGPU::sub1 }) {
2835 if (Dst.isPhysical()) {
2836 MovDPP.addDef(RI.getSubReg(Dst,
Sub));
2843 for (
unsigned I = 1;
I <= 2; ++
I) {
2846 if (
SrcOp.isImm()) {
2848 Imm.ashrInPlace(Part * 32);
2849 MovDPP.addImm(
Imm.getLoBits(32).getZExtValue());
2853 if (Src.isPhysical())
2854 MovDPP.addReg(RI.getSubReg(Src,
Sub));
2861 MovDPP.addImm(MO.getImm());
2863 Split[Part] = MovDPP;
2867 if (Dst.isVirtual())
2874 MI.eraseFromParent();
2875 return std::pair(Split[0], Split[1]);
2878std::optional<DestSourcePair>
2880 if (
MI.getOpcode() == AMDGPU::WWM_COPY)
2883 return std::nullopt;
2887 AMDGPU::OpName Src0OpName,
2889 AMDGPU::OpName Src1OpName)
const {
2896 "All commutable instructions have both src0 and src1 modifiers");
2898 int Src0ModsVal = Src0Mods->
getImm();
2899 int Src1ModsVal = Src1Mods->
getImm();
2901 Src1Mods->
setImm(Src0ModsVal);
2902 Src0Mods->
setImm(Src1ModsVal);
2911 bool IsKill = RegOp.
isKill();
2913 bool IsUndef = RegOp.
isUndef();
2914 bool IsDebug = RegOp.
isDebug();
2916 if (NonRegOp.
isImm())
2918 else if (NonRegOp.
isFI())
2939 int64_t NonRegVal = NonRegOp1.
getImm();
2942 NonRegOp2.
setImm(NonRegVal);
2949 unsigned OpIdx1)
const {
2954 unsigned Opc =
MI.getOpcode();
2955 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
2965 if ((
int)OpIdx0 == Src0Idx && !MO0.
isReg() &&
2968 if ((
int)OpIdx1 == Src0Idx && !MO1.
isReg() &&
2973 if ((
int)OpIdx1 != Src0Idx && MO0.
isReg()) {
2979 if ((
int)OpIdx0 != Src0Idx && MO1.
isReg()) {
3001 unsigned Src1Idx)
const {
3002 assert(!NewMI &&
"this should never be used");
3007 unsigned Opc =
MI.getOpcode();
3009 if (CommutedOpcode == -1)
3012 if (Src0Idx > Src1Idx)
3015 assert(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0) ==
3016 static_cast<int>(Src0Idx) &&
3017 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1) ==
3018 static_cast<int>(Src1Idx) &&
3019 "inconsistency with findCommutedOpIndices");
3027 if (Src0.isReg() && Src1.isReg()) {
3031 }
else if (Src0.isReg() && !Src1.isReg()) {
3033 }
else if (!Src0.isReg() && Src1.isReg()) {
3035 }
else if (Src0.isImm() && Src1.isImm()) {
3044 Src1, AMDGPU::OpName::src1_modifiers);
3047 AMDGPU::OpName::src1_sel);
3059 unsigned &SrcOpIdx0,
3060 unsigned &SrcOpIdx1)
const {
3068 unsigned &SrcOpIdx0,
3069 unsigned &SrcOpIdx1)
const {
3070 if (!
Desc.isCommutable())
3073 unsigned Opc =
Desc.getOpcode();
3074 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
3078 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
3082 return fixCommutedOpIndices(SrcOpIdx0, SrcOpIdx1, Src0Idx, Src1Idx);
3086 int64_t BrOffset)
const {
3103 return MI.getOperand(0).getMBB();
3108 if (
MI.getOpcode() == AMDGPU::SI_IF ||
MI.getOpcode() == AMDGPU::SI_ELSE ||
3109 MI.getOpcode() == AMDGPU::SI_LOOP ||
3110 MI.getOpcode() == AMDGPU::SI_WATERFALL_LOOP)
3122 "new block should be inserted for expanding unconditional branch");
3125 "restore block should be inserted for restoring clobbered registers");
3133 if (ST.useAddPC64Inst()) {
3135 MCCtx.createTempSymbol(
"offset",
true);
3139 MCCtx.createTempSymbol(
"post_addpc",
true);
3140 AddPC->setPostInstrSymbol(*MF, PostAddPCLabel);
3144 Offset->setVariableValue(OffsetExpr);
3148 assert(RS &&
"RegScavenger required for long branching");
3156 const bool FlushSGPRWrites = (ST.isWave64() && ST.hasVALUMaskWriteHazard()) ||
3157 ST.hasVALUReadSGPRHazard();
3158 auto ApplyHazardWorkarounds = [
this, &
MBB, &
I, &
DL, FlushSGPRWrites]() {
3159 if (FlushSGPRWrites)
3167 ApplyHazardWorkarounds();
3170 MCCtx.createTempSymbol(
"post_getpc",
true);
3174 MCCtx.createTempSymbol(
"offset_lo",
true);
3176 MCCtx.createTempSymbol(
"offset_hi",
true);
3179 .
addReg(PCReg, {}, AMDGPU::sub0)
3183 .
addReg(PCReg, {}, AMDGPU::sub1)
3185 ApplyHazardWorkarounds();
3226 if (LongBranchReservedReg) {
3227 RS->enterBasicBlock(
MBB);
3228 Scav = LongBranchReservedReg;
3230 RS->enterBasicBlockEnd(
MBB);
3231 Scav = RS->scavengeRegisterBackwards(
3236 RS->setRegUsed(Scav);
3244 TRI->spillEmergencySGPR(GetPC, RestoreBB, AMDGPU::SGPR0_SGPR1, RS);
3261unsigned SIInstrInfo::getBranchOpcode(SIInstrInfo::BranchPredicate
Cond) {
3263 case SIInstrInfo::SCC_TRUE:
3264 return AMDGPU::S_CBRANCH_SCC1;
3265 case SIInstrInfo::SCC_FALSE:
3266 return AMDGPU::S_CBRANCH_SCC0;
3267 case SIInstrInfo::VCCNZ:
3268 return AMDGPU::S_CBRANCH_VCCNZ;
3269 case SIInstrInfo::VCCZ:
3270 return AMDGPU::S_CBRANCH_VCCZ;
3271 case SIInstrInfo::EXECNZ:
3272 return AMDGPU::S_CBRANCH_EXECNZ;
3273 case SIInstrInfo::EXECZ:
3274 return AMDGPU::S_CBRANCH_EXECZ;
3280SIInstrInfo::BranchPredicate SIInstrInfo::getBranchPredicate(
unsigned Opcode) {
3282 case AMDGPU::S_CBRANCH_SCC0:
3284 case AMDGPU::S_CBRANCH_SCC1:
3286 case AMDGPU::S_CBRANCH_VCCNZ:
3288 case AMDGPU::S_CBRANCH_VCCZ:
3290 case AMDGPU::S_CBRANCH_EXECNZ:
3292 case AMDGPU::S_CBRANCH_EXECZ:
3304 bool AllowModify)
const {
3305 if (
I->getOpcode() == AMDGPU::S_BRANCH) {
3307 TBB =
I->getOperand(0).getMBB();
3311 BranchPredicate Pred = getBranchPredicate(
I->getOpcode());
3312 if (Pred == INVALID_BR)
3317 Cond.push_back(
I->getOperand(1));
3321 if (
I ==
MBB.end()) {
3327 if (
I->getOpcode() == AMDGPU::S_BRANCH) {
3329 FBB =
I->getOperand(0).getMBB();
3339 bool AllowModify)
const {
3347 while (
I != E && !
I->isBranch() && !
I->isReturn()) {
3348 switch (
I->getOpcode()) {
3349 case AMDGPU::S_MOV_B64_term:
3350 case AMDGPU::S_XOR_B64_term:
3351 case AMDGPU::S_OR_B64_term:
3352 case AMDGPU::S_ANDN2_B64_term:
3353 case AMDGPU::S_AND_B64_term:
3354 case AMDGPU::S_AND_SAVEEXEC_B64_term:
3355 case AMDGPU::S_MOV_B32_term:
3356 case AMDGPU::S_XOR_B32_term:
3357 case AMDGPU::S_OR_B32_term:
3358 case AMDGPU::S_ANDN2_B32_term:
3359 case AMDGPU::S_AND_B32_term:
3360 case AMDGPU::S_AND_SAVEEXEC_B32_term:
3361 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
3362 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
3365 case AMDGPU::SI_ELSE:
3366 case AMDGPU::SI_KILL_I1_TERMINATOR:
3367 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
3384 int *BytesRemoved)
const {
3386 unsigned RemovedSize = 0;
3389 if (
MI.isBranch() ||
MI.isReturn()) {
3391 MI.eraseFromParent();
3397 *BytesRemoved = RemovedSize;
3414 int *BytesAdded)
const {
3415 if (!FBB &&
Cond.empty()) {
3419 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3426 = getBranchOpcode(
static_cast<BranchPredicate
>(
Cond[0].
getImm()));
3438 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3456 *BytesAdded = ST.hasOffset3fBug() ? 16 : 8;
3463 if (
Cond.size() != 2) {
3467 if (
Cond[0].isImm()) {
3488 bool shouldIgnoreForPipelining(
const MachineInstr *
MI)
const override {
3492 std::optional<bool> createTripCountGreaterCondition(
3493 int TC, MachineBasicBlock &
MBB,
3494 SmallVectorImpl<MachineOperand> &CondParam)
override {
3495 CondParam = this->
Cond;
3499 void adjustTripCount(
int TripCountAdjust)
override {}
3501 void setPreheader(MachineBasicBlock *NewPreheader)
override {}
3505std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
3514 if (
TBB == LoopBB && FBB == LoopBB)
3521 assert((
TBB == LoopBB || FBB == LoopBB) &&
3522 "The Loop must be a single-basic-block loop");
3525 BranchPredicate Pred =
static_cast<BranchPredicate
>(
Cond[0].getImm());
3526 if (Pred != SCC_TRUE && Pred != SCC_FALSE)
3531 if (
MI.isCall() ||
MI.isInlineAsm())
3546 if (CmpI == Instructions.end() || CmpI->isPHI())
3550 return std::make_unique<AMDGPUPipelinerLoopInfo>(
CmpInst,
Cond);
3556 Register FalseReg,
int &CondCycles,
3557 int &TrueCycles,
int &FalseCycles)
const {
3567 CondCycles = TrueCycles = FalseCycles = NumInsts;
3570 return RI.hasVGPRs(RC) && NumInsts <= 6;
3584 if (NumInsts % 2 == 0)
3587 CondCycles = TrueCycles = FalseCycles = NumInsts;
3588 return RI.isSGPRClass(RC);
3599 BranchPredicate Pred =
static_cast<BranchPredicate
>(
Cond[0].getImm());
3600 if (Pred == VCCZ || Pred == SCC_FALSE) {
3601 Pred =
static_cast<BranchPredicate
>(-Pred);
3607 unsigned DstSize = RI.getRegSizeInBits(*DstRC);
3609 if (DstSize == 32) {
3611 if (Pred == SCC_TRUE) {
3626 if (DstSize == 64 && Pred == SCC_TRUE) {
3636 static const int16_t Sub0_15[] = {
3637 AMDGPU::sub0, AMDGPU::sub1, AMDGPU::sub2, AMDGPU::sub3,
3638 AMDGPU::sub4, AMDGPU::sub5, AMDGPU::sub6, AMDGPU::sub7,
3639 AMDGPU::sub8, AMDGPU::sub9, AMDGPU::sub10, AMDGPU::sub11,
3640 AMDGPU::sub12, AMDGPU::sub13, AMDGPU::sub14, AMDGPU::sub15,
3643 static const int16_t Sub0_15_64[] = {
3644 AMDGPU::sub0_sub1, AMDGPU::sub2_sub3,
3645 AMDGPU::sub4_sub5, AMDGPU::sub6_sub7,
3646 AMDGPU::sub8_sub9, AMDGPU::sub10_sub11,
3647 AMDGPU::sub12_sub13, AMDGPU::sub14_sub15,
3650 unsigned SelOp = AMDGPU::V_CNDMASK_B32_e32;
3652 const int16_t *SubIndices = Sub0_15;
3653 int NElts = DstSize / 32;
3657 if (Pred == SCC_TRUE) {
3659 SelOp = AMDGPU::S_CSELECT_B32;
3660 EltRC = &AMDGPU::SGPR_32RegClass;
3662 SelOp = AMDGPU::S_CSELECT_B64;
3663 EltRC = &AMDGPU::SGPR_64RegClass;
3664 SubIndices = Sub0_15_64;
3670 MBB,
I,
DL,
get(AMDGPU::REG_SEQUENCE), DstReg);
3675 for (
int Idx = 0; Idx != NElts; ++Idx) {
3679 unsigned SubIdx = SubIndices[Idx];
3682 if (SelOp == AMDGPU::V_CNDMASK_B32_e32) {
3684 .
addReg(FalseReg, {}, SubIdx)
3685 .addReg(TrueReg, {}, SubIdx);
3688 .
addReg(TrueReg, {}, SubIdx)
3689 .addReg(FalseReg, {}, SubIdx);
3702 if (
MI.isBranch() ||
MI.isCall() ||
MI.isReturn() ||
MI.isIndirectBranch())
3705 switch (
MI.getOpcode()) {
3706 case AMDGPU::S_ENDPGM:
3707 case AMDGPU::S_ENDPGM_SAVED:
3708 case AMDGPU::S_TRAP:
3709 case AMDGPU::S_GETREG_B32:
3710 case AMDGPU::S_SETREG_B32:
3711 case AMDGPU::S_SETREG_B32_mode:
3712 case AMDGPU::S_SETREG_IMM32_B32:
3713 case AMDGPU::S_SETREG_IMM32_B32_mode:
3714 case AMDGPU::S_SENDMSG:
3715 case AMDGPU::S_SENDMSGHALT:
3716 case AMDGPU::S_SENDMSG_RTN_B32:
3717 case AMDGPU::S_SENDMSG_RTN_B64:
3718 case AMDGPU::S_BARRIER_WAIT:
3719 case AMDGPU::S_BARRIER_SIGNAL_M0:
3720 case AMDGPU::S_BARRIER_SIGNAL_IMM:
3721 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0:
3722 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM:
3730 switch (
MI.getOpcode()) {
3731 case AMDGPU::V_MOV_B16_t16_e32:
3732 case AMDGPU::V_MOV_B16_t16_e64:
3733 case AMDGPU::V_MOV_B32_e32:
3734 case AMDGPU::V_MOV_B32_e64:
3735 case AMDGPU::V_MOV_B64_PSEUDO:
3736 case AMDGPU::V_MOV_B64_e32:
3737 case AMDGPU::V_MOV_B64_e64:
3738 case AMDGPU::S_MOV_B32:
3739 case AMDGPU::S_MOV_B64:
3740 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3742 case AMDGPU::WWM_COPY:
3743 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3744 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3745 case AMDGPU::V_ACCVGPR_MOV_B32:
3746 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3747 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3755 switch (
MI.getOpcode()) {
3756 case AMDGPU::V_MOV_B16_t16_e32:
3757 case AMDGPU::V_MOV_B16_t16_e64:
3759 case AMDGPU::V_MOV_B32_e32:
3760 case AMDGPU::V_MOV_B32_e64:
3761 case AMDGPU::V_MOV_B64_PSEUDO:
3762 case AMDGPU::V_MOV_B64_e32:
3763 case AMDGPU::V_MOV_B64_e64:
3764 case AMDGPU::S_MOV_B32:
3765 case AMDGPU::S_MOV_B64:
3766 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3768 case AMDGPU::WWM_COPY:
3769 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3770 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3771 case AMDGPU::V_ACCVGPR_MOV_B32:
3772 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3773 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3781 AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src1_modifiers,
3782 AMDGPU::OpName::src2_modifiers, AMDGPU::OpName::clamp,
3783 AMDGPU::OpName::omod, AMDGPU::OpName::op_sel};
3786 unsigned Opc =
MI.getOpcode();
3788 int Idx = AMDGPU::getNamedOperandIdx(
Opc, Name);
3790 MI.removeOperand(Idx);
3796 MI.setDesc(NewDesc);
3802 unsigned NumOps =
Desc.getNumOperands() +
Desc.implicit_uses().size() +
3803 Desc.implicit_defs().size();
3805 for (
unsigned I =
MI.getNumOperands() - 1;
I >=
NumOps; --
I)
3806 MI.removeOperand(
I);
3810 unsigned SubRegIndex) {
3811 switch (SubRegIndex) {
3812 case AMDGPU::NoSubRegister:
3822 case AMDGPU::sub1_lo16:
3824 case AMDGPU::sub1_hi16:
3827 return std::nullopt;
3835 case AMDGPU::V_MAC_F16_e32:
3836 case AMDGPU::V_MAC_F16_e64:
3837 case AMDGPU::V_MAD_F16_e64:
3838 return AMDGPU::V_MADAK_F16;
3839 case AMDGPU::V_MAC_F32_e32:
3840 case AMDGPU::V_MAC_F32_e64:
3841 case AMDGPU::V_MAD_F32_e64:
3842 return AMDGPU::V_MADAK_F32;
3843 case AMDGPU::V_FMAC_F32_e32:
3844 case AMDGPU::V_FMAC_F32_e64:
3845 case AMDGPU::V_FMA_F32_e64:
3846 return AMDGPU::V_FMAAK_F32;
3847 case AMDGPU::V_FMAC_F16_e32:
3848 case AMDGPU::V_FMAC_F16_e64:
3849 case AMDGPU::V_FMAC_F16_t16_e64:
3850 case AMDGPU::V_FMAC_F16_fake16_e64:
3851 case AMDGPU::V_FMAC_F16_t16_e32:
3852 case AMDGPU::V_FMAC_F16_fake16_e32:
3853 case AMDGPU::V_FMA_F16_e64:
3854 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3855 ? AMDGPU::V_FMAAK_F16_t16
3856 : AMDGPU::V_FMAAK_F16_fake16
3857 : AMDGPU::V_FMAAK_F16;
3858 case AMDGPU::V_FMAC_F64_e32:
3859 case AMDGPU::V_FMAC_F64_e64:
3860 case AMDGPU::V_FMA_F64_e64:
3861 return AMDGPU::V_FMAAK_F64;
3869 case AMDGPU::V_MAC_F16_e32:
3870 case AMDGPU::V_MAC_F16_e64:
3871 case AMDGPU::V_MAD_F16_e64:
3872 return AMDGPU::V_MADMK_F16;
3873 case AMDGPU::V_MAC_F32_e32:
3874 case AMDGPU::V_MAC_F32_e64:
3875 case AMDGPU::V_MAD_F32_e64:
3876 return AMDGPU::V_MADMK_F32;
3877 case AMDGPU::V_FMAC_F32_e32:
3878 case AMDGPU::V_FMAC_F32_e64:
3879 case AMDGPU::V_FMA_F32_e64:
3880 return AMDGPU::V_FMAMK_F32;
3881 case AMDGPU::V_FMAC_F16_e32:
3882 case AMDGPU::V_FMAC_F16_e64:
3883 case AMDGPU::V_FMAC_F16_t16_e64:
3884 case AMDGPU::V_FMAC_F16_fake16_e64:
3885 case AMDGPU::V_FMAC_F16_t16_e32:
3886 case AMDGPU::V_FMAC_F16_fake16_e32:
3887 case AMDGPU::V_FMA_F16_e64:
3888 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3889 ? AMDGPU::V_FMAMK_F16_t16
3890 : AMDGPU::V_FMAMK_F16_fake16
3891 : AMDGPU::V_FMAMK_F16;
3892 case AMDGPU::V_FMAC_F64_e32:
3893 case AMDGPU::V_FMAC_F64_e64:
3894 case AMDGPU::V_FMA_F64_e64:
3895 return AMDGPU::V_FMAMK_F64;
3909 assert(!
DefMI.getOperand(0).getSubReg() &&
"Expected SSA form");
3912 if (
Opc == AMDGPU::COPY) {
3913 assert(!
UseMI.getOperand(0).getSubReg() &&
"Expected SSA form");
3920 if (HasMultipleUses) {
3923 unsigned ImmDefSize = RI.getRegSizeInBits(*MRI->
getRegClass(Reg));
3926 if (UseSubReg != AMDGPU::NoSubRegister && ImmDefSize == 64)
3934 if (ImmDefSize == 32 &&
3939 bool Is16Bit = UseSubReg != AMDGPU::NoSubRegister &&
3940 RI.getSubRegIdxSize(UseSubReg) == 16;
3943 if (RI.hasVGPRs(DstRC))
3946 if (DstReg.
isVirtual() && UseSubReg != AMDGPU::lo16)
3952 unsigned NewOpc = AMDGPU::INSTRUCTION_LIST_END;
3959 for (
unsigned MovOp :
3960 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
3961 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_ACCVGPR_WRITE_B32_e64}) {
3969 MovDstRC = RI.getMatchingSuperRegClass(MovDstRC, DstRC, AMDGPU::lo16);
3973 if (MovDstPhysReg) {
3977 RI.getMatchingSuperReg(MovDstPhysReg, AMDGPU::lo16, MovDstRC);
3984 if (MovDstPhysReg) {
3985 if (!MovDstRC->
contains(MovDstPhysReg))
4001 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType) &&
4009 if (NewOpc == AMDGPU::INSTRUCTION_LIST_END)
4013 UseMI.getOperand(0).setSubReg(AMDGPU::NoSubRegister);
4015 UseMI.getOperand(0).setReg(MovDstPhysReg);
4020 UseMI.setDesc(NewMCID);
4021 UseMI.getOperand(1).ChangeToImmediate(*SubRegImm);
4022 UseMI.addImplicitDefUseOperands(*MF);
4026 if (HasMultipleUses)
4029 if (
Opc == AMDGPU::V_MAD_F32_e64 ||
Opc == AMDGPU::V_MAC_F32_e64 ||
4030 Opc == AMDGPU::V_MAD_F16_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4031 Opc == AMDGPU::V_FMA_F32_e64 ||
Opc == AMDGPU::V_FMAC_F32_e64 ||
4032 Opc == AMDGPU::V_FMA_F16_e64 ||
Opc == AMDGPU::V_FMAC_F16_e64 ||
4033 Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4034 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
Opc == AMDGPU::V_FMA_F64_e64 ||
4035 Opc == AMDGPU::V_FMAC_F64_e64) {
4044 int Src0Idx = getNamedOperandIdx(
UseMI.getOpcode(), AMDGPU::OpName::src0);
4055 auto CopyRegOperandToNarrowerRC =
4058 if (!
MI.getOperand(OpNo).isReg())
4062 if (RI.getCommonSubClass(RC, NewRC) != NewRC)
4065 BuildMI(*
MI.getParent(),
MI.getIterator(),
MI.getDebugLoc(),
4066 get(AMDGPU::COPY), Tmp)
4068 MI.getOperand(OpNo).setReg(Tmp);
4069 MI.getOperand(OpNo).setIsKill();
4073 if ((Src0->isReg() && Src0->getReg() == Reg) ||
4074 (Src1->isReg() && Src1->getReg() == Reg)) {
4076 Src1->isReg() && Src1->getReg() == Reg ? Src0 : Src1;
4077 if (!RegSrc->
isReg())
4080 ST.getConstantBusLimit(
Opc) < 2)
4083 if (!Src2->isReg() || RI.isSGPRClass(MRI->
getRegClass(Src2->getReg())))
4095 if (Def && Def->isMoveImmediate() &&
4104 Imm, RegSrc == Src1 ? Src0->getSubReg() : Src1->getSubReg());
4110 unsigned SrcSubReg = RegSrc->
getSubReg();
4111 Src0->setReg(SrcReg);
4112 Src0->setSubReg(SrcSubReg);
4113 Src0->setIsKill(RegSrc->
isKill());
4115 if (
Opc == AMDGPU::V_MAC_F32_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4116 Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4117 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4118 Opc == AMDGPU::V_FMAC_F16_e64 ||
Opc == AMDGPU::V_FMAC_F64_e64)
4119 UseMI.untieRegOperand(
4120 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2));
4122 Src1->ChangeToImmediate(*SubRegImm);
4127 if (NewOpc == AMDGPU::V_FMAMK_F16_t16 ||
4128 NewOpc == AMDGPU::V_FMAMK_F16_fake16) {
4132 UseMI.getDebugLoc(),
get(AMDGPU::COPY),
4133 UseMI.getOperand(0).getReg())
4135 UseMI.getOperand(0).setReg(Tmp);
4136 CopyRegOperandToNarrowerRC(
UseMI, 1, NewRC);
4137 CopyRegOperandToNarrowerRC(
UseMI, 3, NewRC);
4142 DefMI.eraseFromParent();
4148 if (Src2->isReg() && Src2->getReg() == Reg) {
4149 if (ST.getConstantBusLimit(
Opc) < 2) {
4152 bool Src0Inlined =
false;
4153 if (Src0->isReg()) {
4158 if (Def && Def->isMoveImmediate() &&
4161 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4163 }
else if (ST.getConstantBusLimit(
Opc) <= 1 &&
4164 RI.isSGPRReg(*MRI, Src0->getReg())) {
4170 if (Src1->isReg() && !Src0Inlined) {
4173 if (Def && Def->isMoveImmediate() &&
4176 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4177 else if (RI.isSGPRReg(*MRI, Src1->getReg()))
4190 if (
Opc == AMDGPU::V_MAC_F32_e64 ||
Opc == AMDGPU::V_MAC_F16_e64 ||
4191 Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4192 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4193 Opc == AMDGPU::V_FMAC_F16_e64 ||
Opc == AMDGPU::V_FMAC_F64_e64)
4194 UseMI.untieRegOperand(
4195 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2));
4197 const std::optional<int64_t> SubRegImm =
4201 Src2->ChangeToImmediate(*SubRegImm);
4207 if (NewOpc == AMDGPU::V_FMAAK_F16_t16 ||
4208 NewOpc == AMDGPU::V_FMAAK_F16_fake16) {
4212 UseMI.getDebugLoc(),
get(AMDGPU::COPY),
4213 UseMI.getOperand(0).getReg())
4215 UseMI.getOperand(0).setReg(Tmp);
4216 CopyRegOperandToNarrowerRC(
UseMI, 1, NewRC);
4217 CopyRegOperandToNarrowerRC(
UseMI, 2, NewRC);
4226 AMDGPU::getNamedOperandIdx(
UseMI.getOpcode(), AMDGPU::OpName::src0);
4232 DefMI.eraseFromParent();
4244 if (BaseOps1.
size() != BaseOps2.
size())
4246 for (
size_t I = 0,
E = BaseOps1.
size();
I <
E; ++
I) {
4247 if (!BaseOps1[
I]->isIdenticalTo(*BaseOps2[
I]))
4255 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
4256 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
4257 LocationSize LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
4259 LowOffset + (int)LowWidth.
getValue() <= HighOffset;
4262bool SIInstrInfo::checkInstOffsetsDoNotOverlap(
const MachineInstr &MIa,
4265 int64_t Offset0, Offset1;
4268 bool Offset0IsScalable, Offset1IsScalable;
4282 LocationSize Width0 = MIa.
memoperands().front()->getSize();
4283 LocationSize Width1 = MIb.
memoperands().front()->getSize();
4290 "MIa must load from or modify a memory location");
4292 "MIb must load from or modify a memory location");
4314 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4321 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4331 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4345 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4356 case AMDGPU::V_MAC_F16_e32:
4357 case AMDGPU::V_MAC_F16_e64:
4358 return AMDGPU::V_MAD_F16_e64;
4359 case AMDGPU::V_MAC_F32_e32:
4360 case AMDGPU::V_MAC_F32_e64:
4361 return AMDGPU::V_MAD_F32_e64;
4362 case AMDGPU::V_MAC_LEGACY_F32_e32:
4363 case AMDGPU::V_MAC_LEGACY_F32_e64:
4364 return AMDGPU::V_MAD_LEGACY_F32_e64;
4365 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4366 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4367 return AMDGPU::V_FMA_LEGACY_F32_e64;
4368 case AMDGPU::V_FMAC_F16_e32:
4369 case AMDGPU::V_FMAC_F16_e64:
4370 case AMDGPU::V_FMAC_F16_t16_e64:
4371 case AMDGPU::V_FMAC_F16_fake16_e64:
4372 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
4373 ? AMDGPU::V_FMA_F16_gfx9_t16_e64
4374 : AMDGPU::V_FMA_F16_gfx9_fake16_e64
4375 : AMDGPU::V_FMA_F16_gfx9_e64;
4376 case AMDGPU::V_FMAC_F32_e32:
4377 case AMDGPU::V_FMAC_F32_e64:
4378 return AMDGPU::V_FMA_F32_e64;
4379 case AMDGPU::V_FMAC_F64_e32:
4380 case AMDGPU::V_FMAC_F64_e64:
4381 return AMDGPU::V_FMA_F64_e64;
4400 if (
MI.isBundle()) {
4403 if (
MI.getBundleSize() != 1)
4405 CandidateMI =
MI.getNextNode();
4409 MachineInstr *NewMI = convertToThreeAddressImpl(*CandidateMI, U);
4413 if (
MI.isBundle()) {
4418 MI.untieRegOperand(MO.getOperandNo());
4425 if (Def.isEarlyClobber() && Def.isReg() &&
4430 auto UpdateDefIndex = [&](
LiveRange &LR) {
4431 auto *S = LR.find(OldIndex);
4432 if (S != LR.end() && S->start == OldIndex) {
4433 assert(S->valno && S->valno->def == OldIndex);
4434 S->start = NewIndex;
4435 S->valno->def = NewIndex;
4439 for (
auto &SR : LI.subranges())
4445 if (U.RemoveMIUse) {
4448 Register DefReg = U.RemoveMIUse->getOperand(0).getReg();
4452 U.RemoveMIUse->setDesc(
get(AMDGPU::IMPLICIT_DEF));
4453 U.RemoveMIUse->getOperand(0).setIsDead(
true);
4454 for (
unsigned I = U.RemoveMIUse->getNumOperands() - 1;
I != 0; --
I)
4455 U.RemoveMIUse->removeOperand(
I);
4458 if (
MI.isBundle()) {
4462 if (MO.isReg() && MO.getReg() == DefReg) {
4463 assert(MO.getSubReg() == 0 &&
4464 "tied sub-registers in bundles currently not supported");
4465 MI.removeOperand(MO.getOperandNo());
4482 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4483 MIOp.setIsUndef(
true);
4484 MIOp.setReg(DummyReg);
4488 if (
MI.isBundle()) {
4492 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4493 MIOp.setIsUndef(
true);
4494 MIOp.setReg(DummyReg);
4507 return MI.isBundle() ? &
MI : NewMI;
4512 ThreeAddressUpdates &U)
const {
4514 unsigned Opc =
MI.getOpcode();
4518 if (NewMFMAOpc != -1) {
4521 for (
unsigned I = 0, E =
MI.getNumExplicitOperands();
I != E; ++
I)
4522 MIB.
add(
MI.getOperand(
I));
4530 for (
unsigned I = 0,
E =
MI.getNumExplicitOperands();
I !=
E; ++
I)
4535 assert(
Opc != AMDGPU::V_FMAC_F16_t16_e32 &&
4536 Opc != AMDGPU::V_FMAC_F16_fake16_e32 &&
4537 "V_FMAC_F16_t16/fake16_e32 is not supported and not expected to be "
4541 bool IsF64 =
Opc == AMDGPU::V_FMAC_F64_e32 ||
Opc == AMDGPU::V_FMAC_F64_e64;
4542 bool IsLegacy =
Opc == AMDGPU::V_MAC_LEGACY_F32_e32 ||
4543 Opc == AMDGPU::V_MAC_LEGACY_F32_e64 ||
4544 Opc == AMDGPU::V_FMAC_LEGACY_F32_e32 ||
4545 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64;
4546 bool Src0Literal =
false;
4551 case AMDGPU::V_MAC_F16_e64:
4552 case AMDGPU::V_FMAC_F16_e64:
4553 case AMDGPU::V_FMAC_F16_t16_e64:
4554 case AMDGPU::V_FMAC_F16_fake16_e64:
4555 case AMDGPU::V_MAC_F32_e64:
4556 case AMDGPU::V_MAC_LEGACY_F32_e64:
4557 case AMDGPU::V_FMAC_F32_e64:
4558 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4559 case AMDGPU::V_FMAC_F64_e64:
4561 case AMDGPU::V_MAC_F16_e32:
4562 case AMDGPU::V_FMAC_F16_e32:
4563 case AMDGPU::V_MAC_F32_e32:
4564 case AMDGPU::V_MAC_LEGACY_F32_e32:
4565 case AMDGPU::V_FMAC_F32_e32:
4566 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4567 case AMDGPU::V_FMAC_F64_e32: {
4568 int Src0Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
4569 AMDGPU::OpName::src0);
4570 const MachineOperand *
Src0 = &
MI.getOperand(Src0Idx);
4571 if (!
Src0->isReg() && !
Src0->isImm())
4581 MachineInstrBuilder MIB;
4584 const MachineOperand *Src0Mods =
4587 const MachineOperand *Src1Mods =
4590 const MachineOperand *Src2Mods =
4596 if (!Src0Mods && !Src1Mods && !Src2Mods && !Clamp && !Omod && !IsLegacy &&
4597 (!IsF64 || ST.hasFmaakFmamkF64Insts()) &&
4599 (ST.getConstantBusLimit(
Opc) > 1 || !
Src0->isReg() ||
4601 MachineInstr *
DefMI =
nullptr;
4603 std::optional<int64_t> ImmOpt;
4638 MI, AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::src0),
4654 if (Src0Literal && !ST.hasVOP3Literal())
4682 switch (
MI.getOpcode()) {
4683 case AMDGPU::S_SET_GPR_IDX_ON:
4684 case AMDGPU::S_SET_GPR_IDX_MODE:
4685 case AMDGPU::S_SET_GPR_IDX_OFF:
4703 if (
MI.isTerminator() ||
MI.isPosition())
4707 if (
MI.getOpcode() == TargetOpcode::INLINEASM_BR)
4710 if (
MI.getOpcode() == AMDGPU::SCHED_BARRIER &&
MI.getOperand(0).getImm() == 0)
4716 return MI.modifiesRegister(AMDGPU::EXEC, &RI) ||
4717 MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32 ||
4718 MI.getOpcode() == AMDGPU::S_SETREG_B32 ||
4719 MI.getOpcode() == AMDGPU::S_SETPRIO ||
4720 MI.getOpcode() == AMDGPU::S_SETPRIO_INC_WG ||
4725 return Opcode == AMDGPU::DS_ORDERED_COUNT ||
4726 Opcode == AMDGPU::DS_ADD_GS_REG_RTN ||
4727 Opcode == AMDGPU::DS_SUB_GS_REG_RTN ||
isGWS(Opcode);
4741 if (
MI.getMF()->getFunction().hasFnAttribute(
"amdgpu-no-flat-scratch-init"))
4746 if (
MI.memoperands_empty())
4751 unsigned AS = Memop->getAddrSpace();
4752 if (AS == AMDGPUAS::FLAT_ADDRESS) {
4753 const MDNode *MD = Memop->getAAInfo().NoAliasAddrSpace;
4754 return !MD || !AMDGPU::hasValueInRangeLikeMetadata(
4755 *MD, AMDGPUAS::PRIVATE_ADDRESS);
4770 if (
MI.memoperands_empty())
4779 unsigned AS = Memop->getAddrSpace();
4789 bool TgSplit)
const {
4802 if (
MI.memoperands_empty())
4807 unsigned AS = Memop->getAddrSpace();
4823 unsigned Opcode =
MI.getOpcode();
4838 if (Opcode == AMDGPU::S_SENDMSG || Opcode == AMDGPU::S_SENDMSGHALT ||
4839 isEXP(Opcode) || Opcode == AMDGPU::DS_ORDERED_COUNT ||
4840 Opcode == AMDGPU::S_TRAP || Opcode == AMDGPU::S_WAIT_EVENT ||
4841 Opcode == AMDGPU::S_SETHALT)
4844 if (
MI.isCall() ||
MI.isInlineAsm())
4850 if (ST.hasVPermPk16Hazard() &&
isVPermPk16(Opcode))
4866 if (Opcode == AMDGPU::V_READFIRSTLANE_B32 ||
4867 Opcode == AMDGPU::V_READLANE_B32 || Opcode == AMDGPU::V_WRITELANE_B32 ||
4868 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
4869 Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR)
4877 if (
MI.isMetaInstruction())
4881 if (
MI.isCopyLike()) {
4882 if (!RI.isSGPRReg(MRI,
MI.getOperand(0).getReg()))
4886 return MI.readsRegister(AMDGPU::EXEC, &RI);
4897 return !
isSALU(
MI) ||
MI.readsRegister(AMDGPU::EXEC, &RI);
4901 switch (
Imm.getBitWidth()) {
4907 ST.hasInv2PiInlineImm());
4910 ST.hasInv2PiInlineImm());
4912 return ST.has16BitInsts() &&
4914 ST.hasInv2PiInlineImm());
4921 APInt IntImm =
Imm.bitcastToAPInt();
4923 bool HasInv2Pi = ST.hasInv2PiInlineImm();
4931 return ST.has16BitInsts() &&
4934 return ST.has16BitInsts() &&
4944 switch (OperandType) {
4954 int32_t Trunc =
static_cast<int32_t
>(
Imm);
4998 int16_t Trunc =
static_cast<int16_t
>(
Imm);
4999 return ST.has16BitInsts() &&
5008 int16_t Trunc =
static_cast<int16_t
>(
Imm);
5009 return ST.has16BitInsts() &&
5061 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType))
5067 return ST.hasVOP3Literal();
5071 int64_t ImmVal)
const {
5073 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
5074 if (Src1Idx != -1 &&
isDPP(
Opc) && !ST.hasDPPSrc1SGPR() &&
5075 OpNo ==
static_cast<unsigned>(Src1Idx))
5080 if (
isMAI(InstDesc) && ST.hasMFMAInlineLiteralBug() &&
5081 OpNo == (
unsigned)AMDGPU::getNamedOperandIdx(InstDesc.
getOpcode(),
5082 AMDGPU::OpName::src2))
5085 if (ST.hasBF16InlineConstFromUpperFP32() &&
isVOP1(
Opc)) {
5092 return RI.opCanUseInlineConstant(OpInfo.OperandType);
5104 "unexpected imm-like operand kind");
5117 if (Opcode == AMDGPU::V_MUL_LEGACY_F32_e64 && ST.hasGFX90AInsts())
5137 return Op32 != -1 &&
TII.isVOPC(Op32);
5142 unsigned Depth)
const {
5143 assert(MRI.
isSSA() &&
"isMaskedByExec requires SSA form");
5152 constexpr unsigned MaxDepth = 6;
5153 if (
Depth >= MaxDepth || !Reg.isVirtual())
5159 if (!Def || Def->getParent() !=
MBB)
5167 auto Recurse = [&](
unsigned OpIdx) {
5173 unsigned Opc = Def->getOpcode();
5174 if (
Opc == AMDGPU::COPY && Recurse(1))
5176 if (
Opc == LMC.
AndOpc && (Recurse(1) || Recurse(2)))
5196 AMDGPU::OpName
OpName)
const {
5198 return Mods && Mods->
getImm();
5211 switch (
MI.getOpcode()) {
5212 default:
return false;
5214 case AMDGPU::V_ADDC_U32_e64:
5215 case AMDGPU::V_SUBB_U32_e64:
5216 case AMDGPU::V_SUBBREV_U32_e64: {
5219 if (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()))
5224 case AMDGPU::V_MAC_F16_e64:
5225 case AMDGPU::V_MAC_F32_e64:
5226 case AMDGPU::V_MAC_LEGACY_F32_e64:
5227 case AMDGPU::V_FMAC_F16_e64:
5228 case AMDGPU::V_FMAC_F16_t16_e64:
5229 case AMDGPU::V_FMAC_F16_fake16_e64:
5230 case AMDGPU::V_FMAC_F32_e64:
5231 case AMDGPU::V_FMAC_F64_e64:
5232 case AMDGPU::V_FMAC_LEGACY_F32_e64:
5233 if (!Src2->isReg() || !RI.isVGPR(MRI, Src2->getReg()) ||
5238 case AMDGPU::V_CNDMASK_B32_e64:
5244 if (Src1 && (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()) ||
5257 if (Src0 && Src0->isImm()) {
5260 get(Op32), AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src0),
5282 (
Use.getReg() == AMDGPU::VCC ||
Use.getReg() == AMDGPU::VCC_LO)) {
5291 unsigned Op32)
const {
5305 Inst32.
add(
MI.getOperand(
I));
5309 int Idx =
MI.getNumExplicitDefs();
5311 int OpTy =
MI.getDesc().operands()[Idx++].OperandType;
5316 if (AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src2) == -1) {
5336 if (OldSDst && OldSDst->
isDead()) {
5339 NewVCC->setIsDead();
5348 if (Reg == AMDGPU::SGPR_NULL || Reg == AMDGPU::SGPR_NULL64)
5356 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::M0;
5359 return AMDGPU::SReg_32RegClass.contains(Reg) ||
5360 AMDGPU::SReg_64RegClass.contains(Reg);
5388 switch (MO.getReg()) {
5390 case AMDGPU::VCC_LO:
5391 case AMDGPU::VCC_HI:
5393 case AMDGPU::FLAT_SCR:
5406 switch (
MI.getOpcode()) {
5407 case AMDGPU::V_READLANE_B32:
5408 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
5409 case AMDGPU::V_WRITELANE_B32:
5410 case AMDGPU::SI_SPILL_S32_TO_VGPR:
5417 if (
MI.isPreISelOpcode() ||
5418 SIInstrInfo::isGenericOpcode(
MI.getOpcode()) ||
5436 return SubReg.
getSubReg() != AMDGPU::NoSubRegister &&
5447 if (RI.isVectorRegister(MRI, SrcReg) && RI.isSGPRReg(MRI, DstReg)) {
5448 ErrInfo =
"illegal copy from vector register to SGPR";
5466 if (!MRI.
isSSA() &&
MI.isCopy())
5467 return verifyCopy(
MI, MRI, ErrInfo);
5469 if (SIInstrInfo::isGenericOpcode(Opcode))
5472 int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
5473 int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
5474 int Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2);
5476 if (Src0Idx == -1) {
5478 Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0X);
5479 Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1X);
5480 Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0Y);
5481 Src3Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1Y);
5486 if (!
Desc.isVariadic() &&
5487 Desc.getNumOperands() !=
MI.getNumExplicitOperands()) {
5488 ErrInfo =
"Instruction has wrong number of operands.";
5492 if (
MI.isInlineAsm()) {
5505 if (!Reg.isVirtual() && !RC->
contains(Reg)) {
5506 ErrInfo =
"inlineasm operand has incorrect register class.";
5514 if (
isImage(
MI) &&
MI.memoperands_empty() &&
MI.mayLoadOrStore()) {
5515 ErrInfo =
"missing memory operand from image instruction.";
5520 for (
int i = 0, e =
Desc.getNumOperands(); i != e; ++i) {
5523 ErrInfo =
"FPImm Machine Operands are not supported. ISel should bitcast "
5524 "all fp values to integers.";
5530 switch (OpInfo.OperandType) {
5532 if (
MI.getOperand(i).isImm() ||
MI.getOperand(i).isGlobal()) {
5533 ErrInfo =
"Illegal immediate value for operand.";
5565 ErrInfo =
"Illegal immediate value for operand.";
5574 if (ST.has64BitLiterals() &&
Desc.getSize() != 4 && MO.
isImm() &&
5577 OpInfo.OperandType ==
5579 ErrInfo =
"illegal 64-bit immediate value for operand.";
5586 ErrInfo =
"Expected inline constant for operand.";
5600 if (!
MI.getOperand(i).isImm() && !
MI.getOperand(i).isFI()) {
5601 ErrInfo =
"Expected immediate, but got non-immediate";
5610 if (OpInfo.isGenericType())
5618 if (!ST.hasSDWA()) {
5619 ErrInfo =
"SDWA is not supported on this target";
5623 for (
auto Op : {AMDGPU::OpName::src0_sel, AMDGPU::OpName::src1_sel,
5624 AMDGPU::OpName::dst_sel}) {
5630 ErrInfo =
"Invalid SDWA selection";
5635 int DstIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdst);
5637 for (
int OpIdx : {DstIdx, Src0Idx, Src1Idx, Src2Idx}) {
5642 if (!ST.hasSDWAScalar()) {
5644 if (!MO.
isReg() || !RI.hasVGPRs(RI.getRegClassForReg(MRI, MO.
getReg()))) {
5645 ErrInfo =
"Only VGPRs allowed as operands in SDWA instructions on VI";
5652 "Only reg allowed as operands in SDWA instructions on GFX9+";
5658 if (!ST.hasSDWAOmod()) {
5661 if (OMod !=
nullptr &&
5663 ErrInfo =
"OMod not allowed in SDWA instructions on VI";
5668 if (Opcode == AMDGPU::V_CVT_F32_FP8_sdwa ||
5669 Opcode == AMDGPU::V_CVT_F32_BF8_sdwa ||
5670 Opcode == AMDGPU::V_CVT_PK_F32_FP8_sdwa ||
5671 Opcode == AMDGPU::V_CVT_PK_F32_BF8_sdwa) {
5674 unsigned Mods = Src0ModsMO->
getImm();
5677 ErrInfo =
"sext, abs and neg are not allowed on this instruction";
5683 if (
isVOPC(BasicOpcode)) {
5684 if (!ST.hasSDWASdst() && DstIdx != -1) {
5687 if (!Dst.isReg() || Dst.getReg() != AMDGPU::VCC) {
5688 ErrInfo =
"Only VCC allowed as dst in SDWA instructions on VI";
5691 }
else if (!ST.hasSDWAOutModsVOPC()) {
5694 if (Clamp && (!Clamp->
isImm() || Clamp->
getImm() != 0)) {
5695 ErrInfo =
"Clamp not allowed in VOPC SDWA instructions on VI";
5701 if (OMod && (!OMod->
isImm() || OMod->
getImm() != 0)) {
5702 ErrInfo =
"OMod not allowed in VOPC SDWA instructions on VI";
5709 if (DstUnused && DstUnused->isImm() &&
5712 if (!Dst.isReg() || !Dst.isTied()) {
5713 ErrInfo =
"Dst register should have tied register";
5718 MI.getOperand(
MI.findTiedOperandIdx(DstIdx));
5721 "Dst register should be tied to implicit use of preserved register";
5725 ErrInfo =
"Dst register should use same physical register as preserved";
5731 if (
isDPP(
MI) && !ST.hasDPPSrc1SGPR() && Src1Idx != -1) {
5733 if (Src1MO.
isReg() && RI.isSGPRReg(MRI, Src1MO.
getReg())) {
5734 ErrInfo =
"DPP src1 cannot be SGPR on this subtarget";
5737 if (Src1MO.
isImm()) {
5738 ErrInfo =
"DPP src1 cannot be an immediate on this subtarget";
5744 if (
isImage(Opcode) && !
MI.mayStore()) {
5749 uint64_t DMaskImm = DMask->
getImm();
5756 if (D16 && D16->getImm() && !ST.hasUnpackedD16VMem())
5764 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdata);
5768 uint32_t DstSize = RI.getRegSizeInBits(*DstRC) / 32;
5769 if (RegCount > DstSize) {
5770 ErrInfo =
"Image instruction returns too many registers for dst "
5780 Desc.getOpcode() != AMDGPU::V_WRITELANE_B32) {
5781 unsigned ConstantBusCount = 0;
5782 bool UsesLiteral =
false;
5785 int ImmIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::imm);
5789 LiteralVal = &
MI.getOperand(ImmIdx);
5798 for (
int OpIdx : {Src0Idx, Src1Idx, Src2Idx, Src3Idx}) {
5809 }
else if (!MO.
isFI()) {
5816 ErrInfo =
"VOP2/VOP3 instruction uses more than one literal";
5826 if (
llvm::all_of(SGPRsUsed, [
this, SGPRUsed](
unsigned SGPR) {
5827 return !RI.regsOverlap(SGPRUsed, SGPR);
5836 if (ConstantBusCount > ST.getConstantBusLimit(Opcode) &&
5837 Opcode != AMDGPU::V_WRITELANE_B32) {
5838 ErrInfo =
"VOP* instruction violates constant bus restriction";
5842 if (
isVOP3(
MI) && UsesLiteral && !ST.hasVOP3Literal()) {
5843 ErrInfo =
"VOP3 instruction uses literal";
5850 if (
Desc.getOpcode() == AMDGPU::V_WRITELANE_B32) {
5851 unsigned SGPRCount = 0;
5854 for (
int OpIdx : {Src0Idx, Src1Idx}) {
5862 if (MO.
getReg() != SGPRUsed)
5867 if (SGPRCount > ST.getConstantBusLimit(Opcode)) {
5868 ErrInfo =
"WRITELANE instruction violates constant bus restriction";
5875 if (
Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F32_e64 ||
5876 Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F64_e64) {
5880 if (Src0.isReg() && Src1.isReg() && Src2.isReg()) {
5883 ErrInfo =
"v_div_scale_{f32|f64} require src0 = src1 or src2";
5893 ErrInfo =
"ABS not allowed in VOP3B instructions";
5905 !Src0.isIdenticalTo(Src1)) {
5906 ErrInfo =
"SOP2/SOPC instruction requires too many immediate constants";
5913 if (
Desc.isBranch()) {
5915 ErrInfo =
"invalid branch target for SOPK instruction";
5919 uint64_t
Imm =
Op->getImm();
5922 ErrInfo =
"invalid immediate for SOPK instruction";
5927 ErrInfo =
"invalid immediate for SOPK instruction";
5934 if (
Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e32 ||
5935 Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e64 ||
5936 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5937 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
5938 const bool IsDst =
Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5939 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
5941 const unsigned StaticNumOps =
5942 Desc.getNumOperands() +
Desc.implicit_uses().size();
5943 const unsigned NumImplicitOps = IsDst ? 2 : 1;
5949 if (
MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
5950 ErrInfo =
"missing implicit register operands";
5956 if (!Dst->isUse()) {
5957 ErrInfo =
"v_movreld_b32 vdst should be a use operand";
5962 if (!
MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
5963 UseOpIdx != StaticNumOps + 1) {
5964 ErrInfo =
"movrel implicit operands should be tied";
5971 =
MI.getOperand(StaticNumOps + NumImplicitOps - 1);
5973 !
isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
5974 ErrInfo =
"src0 should be subreg of implicit vector use";
5982 if (!
MI.hasRegisterImplicitUseOperand(AMDGPU::EXEC)) {
5983 ErrInfo =
"VALU instruction does not implicitly read exec mask";
5989 if (
MI.mayStore() &&
5994 if (Soff && Soff->
getReg() != AMDGPU::M0) {
5995 ErrInfo =
"scalar stores must use m0 as offset register";
6001 if (
isFLAT(
MI) && !ST.hasFlatInstOffsets()) {
6003 if (
Offset->getImm() != 0) {
6004 ErrInfo =
"subtarget does not support offsets in flat instructions";
6009 if (
isDS(
MI) && !ST.hasGDS()) {
6011 if (GDSOp && GDSOp->
getImm() != 0) {
6012 ErrInfo =
"GDS is not supported on this subtarget";
6020 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opcode,
6021 AMDGPU::OpName::vaddr0);
6022 AMDGPU::OpName RSrcOpName =
6023 isMIMG(
MI) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
6024 int RsrcIdx = AMDGPU::getNamedOperandIdx(Opcode, RSrcOpName);
6032 ErrInfo =
"dim is out of range";
6037 if (ST.hasR128A16()) {
6039 IsA16 = R128A16->
getImm() != 0;
6040 }
else if (ST.hasA16()) {
6042 IsA16 = A16->
getImm() != 0;
6045 bool IsNSA = RsrcIdx - VAddr0Idx > 1;
6047 unsigned AddrWords =
6050 unsigned VAddrWords;
6052 VAddrWords = RsrcIdx - VAddr0Idx;
6053 if (ST.hasPartialNSAEncoding() &&
6055 unsigned LastVAddrIdx = RsrcIdx - 1;
6056 VAddrWords +=
getOpSize(
MI, LastVAddrIdx) / 4 - 1;
6064 if (VAddrWords != AddrWords) {
6066 <<
" but got " << VAddrWords <<
"\n");
6067 ErrInfo =
"bad vaddr size";
6077 unsigned DC = DppCt->
getImm();
6078 if (DC == DppCtrl::DPP_UNUSED1 || DC == DppCtrl::DPP_UNUSED2 ||
6079 DC == DppCtrl::DPP_UNUSED3 || DC > DppCtrl::DPP_LAST ||
6080 (DC >= DppCtrl::DPP_UNUSED4_FIRST && DC <= DppCtrl::DPP_UNUSED4_LAST) ||
6081 (DC >= DppCtrl::DPP_UNUSED5_FIRST && DC <= DppCtrl::DPP_UNUSED5_LAST) ||
6082 (DC >= DppCtrl::DPP_UNUSED6_FIRST && DC <= DppCtrl::DPP_UNUSED6_LAST) ||
6083 (DC >= DppCtrl::DPP_UNUSED7_FIRST && DC <= DppCtrl::DPP_UNUSED7_LAST) ||
6084 (DC >= DppCtrl::DPP_UNUSED8_FIRST && DC <= DppCtrl::DPP_UNUSED8_LAST)) {
6085 ErrInfo =
"Invalid dpp_ctrl value";
6088 if (DC >= DppCtrl::WAVE_SHL1 && DC <= DppCtrl::WAVE_ROR1 &&
6089 !ST.hasDPPWavefrontShifts()) {
6090 ErrInfo =
"Invalid dpp_ctrl value: "
6091 "wavefront shifts are not supported on GFX10+";
6094 if (DC >= DppCtrl::BCAST15 && DC <= DppCtrl::BCAST31 &&
6095 !ST.hasDPPBroadcasts()) {
6096 ErrInfo =
"Invalid dpp_ctrl value: "
6097 "broadcasts are not supported on GFX10+";
6100 if (DC >= DppCtrl::ROW_SHARE_FIRST && DC <= DppCtrl::ROW_XMASK_LAST &&
6102 if (DC >= DppCtrl::ROW_NEWBCAST_FIRST &&
6103 DC <= DppCtrl::ROW_NEWBCAST_LAST &&
6104 !ST.hasGFX90AInsts()) {
6105 ErrInfo =
"Invalid dpp_ctrl value: "
6106 "row_newbroadcast/row_share is not supported before "
6110 if (DC > DppCtrl::ROW_NEWBCAST_LAST || !ST.hasGFX90AInsts()) {
6111 ErrInfo =
"Invalid dpp_ctrl value: "
6112 "row_share and row_xmask are not supported before GFX10";
6117 if (Opcode != AMDGPU::V_MOV_B64_DPP_PSEUDO &&
6119 ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
6121 ErrInfo =
"Invalid dpp_ctrl value: "
6122 "DP ALU dpp only support row_newbcast";
6129 AMDGPU::OpName DataName =
6130 isDS(Opcode) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata;
6136 if (!ST.hasGFX90AInsts()) {
6137 if ((Dst && RI.isAGPR(MRI, Dst->getReg())) ||
6138 (
Data && RI.isAGPR(MRI,
Data->getReg())) ||
6139 (Data2 && RI.isAGPR(MRI, Data2->
getReg()))) {
6140 ErrInfo =
"Invalid register class: "
6141 "agpr loads and stores not supported on this GPU";
6147 if (ST.needsAlignedVGPRs()) {
6148 const auto isAlignedReg = [&
MI, &MRI,
this](AMDGPU::OpName
OpName) ->
bool {
6153 if (Reg.isPhysical())
6154 return !(RI.getHWRegIndex(Reg) & 1);
6156 return RI.getRegSizeInBits(RC) > 32 && RI.isProperlyAlignedRC(RC) &&
6157 !(RI.getChannelFromSubReg(
Op->getSubReg()) & 1);
6161 if (!isAlignedReg(AMDGPU::OpName::vaddr)) {
6162 ErrInfo =
"Subtarget requires even aligned vector registers "
6163 "for vaddr operand of image instructions";
6169 if (Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts()) {
6171 if (Src->isReg() && RI.isSGPRReg(MRI, Src->getReg())) {
6172 ErrInfo =
"Invalid register class: "
6173 "v_accvgpr_write with an SGPR is not supported on this GPU";
6178 if (
Desc.getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS) {
6181 ErrInfo =
"pseudo expects only physical SGPRs";
6188 if (!ST.hasScaleOffset()) {
6189 ErrInfo =
"Subtarget does not support offset scaling";
6193 ErrInfo =
"Instruction does not support offset scaling";
6201 for (
unsigned I = 0;
I < 3; ++
I) {
6207 if (ST.hasFlatScratchHiInB64InstHazard() &&
isSALU(
MI) &&
6208 MI.readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI,
nullptr)) {
6210 if ((Dst && RI.getRegClassForReg(MRI, Dst->getReg()) ==
6211 &AMDGPU::SReg_64RegClass) ||
6212 Opcode == AMDGPU::S_BITCMP0_B64 || Opcode == AMDGPU::S_BITCMP1_B64) {
6213 ErrInfo =
"Instruction cannot read flat_scratch_base_hi";
6222 if (
MI.getOpcode() == AMDGPU::S_MOV_B32) {
6224 return MI.getOperand(1).isReg() || RI.isAGPR(MRI,
MI.getOperand(0).getReg())
6226 : AMDGPU::V_MOV_B32_e32;
6236 default:
return AMDGPU::INSTRUCTION_LIST_END;
6237 case AMDGPU::REG_SEQUENCE:
return AMDGPU::REG_SEQUENCE;
6238 case AMDGPU::COPY:
return AMDGPU::COPY;
6239 case AMDGPU::PHI:
return AMDGPU::PHI;
6240 case AMDGPU::INSERT_SUBREG:
return AMDGPU::INSERT_SUBREG;
6241 case AMDGPU::WQM:
return AMDGPU::WQM;
6242 case AMDGPU::SOFT_WQM:
return AMDGPU::SOFT_WQM;
6243 case AMDGPU::STRICT_WWM:
return AMDGPU::STRICT_WWM;
6244 case AMDGPU::STRICT_WQM:
return AMDGPU::STRICT_WQM;
6245 case AMDGPU::S_ADD_I32:
6246 return ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
6247 case AMDGPU::S_ADDC_U32:
6248 return AMDGPU::V_ADDC_U32_e32;
6249 case AMDGPU::S_SUB_I32:
6250 return ST.hasAddNoCarryInsts() ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_SUB_CO_U32_e32;
6253 case AMDGPU::S_ADD_U32:
6254 return AMDGPU::V_ADD_CO_U32_e32;
6255 case AMDGPU::S_SUB_U32:
6256 return AMDGPU::V_SUB_CO_U32_e32;
6257 case AMDGPU::S_ADD_U64_PSEUDO:
6258 return AMDGPU::V_ADD_U64_PSEUDO;
6259 case AMDGPU::S_SUB_U64_PSEUDO:
6260 return AMDGPU::V_SUB_U64_PSEUDO;
6261 case AMDGPU::S_SUBB_U32:
return AMDGPU::V_SUBB_U32_e32;
6262 case AMDGPU::S_MUL_I32:
return AMDGPU::V_MUL_LO_U32_e64;
6263 case AMDGPU::S_MUL_HI_U32:
return AMDGPU::V_MUL_HI_U32_e64;
6264 case AMDGPU::S_MUL_HI_I32:
return AMDGPU::V_MUL_HI_I32_e64;
6265 case AMDGPU::S_AND_B32:
return AMDGPU::V_AND_B32_e64;
6266 case AMDGPU::S_OR_B32:
return AMDGPU::V_OR_B32_e64;
6267 case AMDGPU::S_XOR_B32:
return AMDGPU::V_XOR_B32_e64;
6268 case AMDGPU::S_XNOR_B32:
6269 return ST.hasDLInsts() ? AMDGPU::V_XNOR_B32_e64 : AMDGPU::INSTRUCTION_LIST_END;
6270 case AMDGPU::S_MIN_I32:
return AMDGPU::V_MIN_I32_e64;
6271 case AMDGPU::S_MIN_U32:
return AMDGPU::V_MIN_U32_e64;
6272 case AMDGPU::S_MAX_I32:
return AMDGPU::V_MAX_I32_e64;
6273 case AMDGPU::S_MAX_U32:
return AMDGPU::V_MAX_U32_e64;
6274 case AMDGPU::S_ASHR_I32:
return AMDGPU::V_ASHR_I32_e32;
6275 case AMDGPU::S_ASHR_I64:
return AMDGPU::V_ASHR_I64_e64;
6276 case AMDGPU::S_LSHL_B32:
return AMDGPU::V_LSHL_B32_e32;
6277 case AMDGPU::S_LSHL_B64:
return AMDGPU::V_LSHL_B64_e64;
6278 case AMDGPU::S_LSHR_B32:
return AMDGPU::V_LSHR_B32_e32;
6279 case AMDGPU::S_LSHR_B64:
return AMDGPU::V_LSHR_B64_e64;
6280 case AMDGPU::S_SEXT_I32_I8:
return AMDGPU::V_BFE_I32_e64;
6281 case AMDGPU::S_SEXT_I32_I16:
return AMDGPU::V_BFE_I32_e64;
6282 case AMDGPU::S_BFE_U32:
return AMDGPU::V_BFE_U32_e64;
6283 case AMDGPU::S_BFE_I32:
return AMDGPU::V_BFE_I32_e64;
6284 case AMDGPU::S_BFM_B32:
return AMDGPU::V_BFM_B32_e64;
6285 case AMDGPU::S_BREV_B32:
return AMDGPU::V_BFREV_B32_e32;
6286 case AMDGPU::S_NOT_B32:
return AMDGPU::V_NOT_B32_e32;
6287 case AMDGPU::S_NOT_B64:
return AMDGPU::V_NOT_B32_e32;
6288 case AMDGPU::S_CMP_EQ_I32:
return AMDGPU::V_CMP_EQ_I32_e64;
6289 case AMDGPU::S_CMP_LG_I32:
return AMDGPU::V_CMP_NE_I32_e64;
6290 case AMDGPU::S_CMP_GT_I32:
return AMDGPU::V_CMP_GT_I32_e64;
6291 case AMDGPU::S_CMP_GE_I32:
return AMDGPU::V_CMP_GE_I32_e64;
6292 case AMDGPU::S_CMP_LT_I32:
return AMDGPU::V_CMP_LT_I32_e64;
6293 case AMDGPU::S_CMP_LE_I32:
return AMDGPU::V_CMP_LE_I32_e64;
6294 case AMDGPU::S_CMP_EQ_U32:
return AMDGPU::V_CMP_EQ_U32_e64;
6295 case AMDGPU::S_CMP_LG_U32:
return AMDGPU::V_CMP_NE_U32_e64;
6296 case AMDGPU::S_CMP_GT_U32:
return AMDGPU::V_CMP_GT_U32_e64;
6297 case AMDGPU::S_CMP_GE_U32:
return AMDGPU::V_CMP_GE_U32_e64;
6298 case AMDGPU::S_CMP_LT_U32:
return AMDGPU::V_CMP_LT_U32_e64;
6299 case AMDGPU::S_CMP_LE_U32:
return AMDGPU::V_CMP_LE_U32_e64;
6300 case AMDGPU::S_CMP_EQ_U64:
return AMDGPU::V_CMP_EQ_U64_e64;
6301 case AMDGPU::S_CMP_LG_U64:
return AMDGPU::V_CMP_NE_U64_e64;
6302 case AMDGPU::S_BCNT1_I32_B32:
return AMDGPU::V_BCNT_U32_B32_e64;
6303 case AMDGPU::S_FF1_I32_B32:
return AMDGPU::V_FFBL_B32_e32;
6304 case AMDGPU::S_FLBIT_I32_B32:
return AMDGPU::V_FFBH_U32_e32;
6305 case AMDGPU::S_FLBIT_I32:
return AMDGPU::V_FFBH_I32_e64;
6306 case AMDGPU::S_CBRANCH_SCC0:
return AMDGPU::S_CBRANCH_VCCZ;
6307 case AMDGPU::S_CBRANCH_SCC1:
return AMDGPU::S_CBRANCH_VCCNZ;
6308 case AMDGPU::S_CVT_F32_I32:
return AMDGPU::V_CVT_F32_I32_e64;
6309 case AMDGPU::S_CVT_F32_U32:
return AMDGPU::V_CVT_F32_U32_e64;
6310 case AMDGPU::S_CVT_I32_F32:
return AMDGPU::V_CVT_I32_F32_e64;
6311 case AMDGPU::S_CVT_U32_F32:
return AMDGPU::V_CVT_U32_F32_e64;
6312 case AMDGPU::S_CVT_F32_F16:
6313 case AMDGPU::S_CVT_HI_F32_F16:
6314 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F32_F16_t16_e64
6315 : AMDGPU::V_CVT_F32_F16_fake16_e64;
6316 case AMDGPU::S_CVT_F16_F32:
6317 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F16_F32_t16_e64
6318 : AMDGPU::V_CVT_F16_F32_fake16_e64;
6319 case AMDGPU::S_CEIL_F32:
return AMDGPU::V_CEIL_F32_e64;
6320 case AMDGPU::S_FLOOR_F32:
return AMDGPU::V_FLOOR_F32_e64;
6321 case AMDGPU::S_TRUNC_F32:
return AMDGPU::V_TRUNC_F32_e64;
6322 case AMDGPU::S_RNDNE_F32:
return AMDGPU::V_RNDNE_F32_e64;
6323 case AMDGPU::S_CEIL_F16:
6324 return ST.useRealTrue16Insts() ? AMDGPU::V_CEIL_F16_t16_e64
6325 : AMDGPU::V_CEIL_F16_fake16_e64;
6326 case AMDGPU::S_FLOOR_F16:
6327 return ST.useRealTrue16Insts() ? AMDGPU::V_FLOOR_F16_t16_e64
6328 : AMDGPU::V_FLOOR_F16_fake16_e64;
6329 case AMDGPU::S_TRUNC_F16:
6330 return ST.useRealTrue16Insts() ? AMDGPU::V_TRUNC_F16_t16_e64
6331 : AMDGPU::V_TRUNC_F16_fake16_e64;
6332 case AMDGPU::S_RNDNE_F16:
6333 return ST.useRealTrue16Insts() ? AMDGPU::V_RNDNE_F16_t16_e64
6334 : AMDGPU::V_RNDNE_F16_fake16_e64;
6335 case AMDGPU::S_ADD_F32:
return AMDGPU::V_ADD_F32_e64;
6336 case AMDGPU::S_SUB_F32:
return AMDGPU::V_SUB_F32_e64;
6337 case AMDGPU::S_MIN_F32:
return AMDGPU::V_MIN_F32_e64;
6338 case AMDGPU::S_MAX_F32:
return AMDGPU::V_MAX_F32_e64;
6339 case AMDGPU::S_MINIMUM_F32:
return AMDGPU::V_MINIMUM_F32_e64;
6340 case AMDGPU::S_MAXIMUM_F32:
return AMDGPU::V_MAXIMUM_F32_e64;
6341 case AMDGPU::S_MUL_F32:
return AMDGPU::V_MUL_F32_e64;
6342 case AMDGPU::S_ADD_F16:
6343 return ST.useRealTrue16Insts() ? AMDGPU::V_ADD_F16_t16_e64
6344 : AMDGPU::V_ADD_F16_fake16_e64;
6345 case AMDGPU::S_SUB_F16:
6346 return ST.useRealTrue16Insts() ? AMDGPU::V_SUB_F16_t16_e64
6347 : AMDGPU::V_SUB_F16_fake16_e64;
6348 case AMDGPU::S_MIN_F16:
6349 return ST.useRealTrue16Insts() ? AMDGPU::V_MIN_F16_t16_e64
6350 : AMDGPU::V_MIN_F16_fake16_e64;
6351 case AMDGPU::S_MAX_F16:
6352 return ST.useRealTrue16Insts() ? AMDGPU::V_MAX_F16_t16_e64
6353 : AMDGPU::V_MAX_F16_fake16_e64;
6354 case AMDGPU::S_MINIMUM_F16:
6355 return ST.useRealTrue16Insts() ? AMDGPU::V_MINIMUM_F16_t16_e64
6356 : AMDGPU::V_MINIMUM_F16_fake16_e64;
6357 case AMDGPU::S_MAXIMUM_F16:
6358 return ST.useRealTrue16Insts() ? AMDGPU::V_MAXIMUM_F16_t16_e64
6359 : AMDGPU::V_MAXIMUM_F16_fake16_e64;
6360 case AMDGPU::S_MUL_F16:
6361 return ST.useRealTrue16Insts() ? AMDGPU::V_MUL_F16_t16_e64
6362 : AMDGPU::V_MUL_F16_fake16_e64;
6363 case AMDGPU::S_CVT_PK_RTZ_F16_F32:
return AMDGPU::V_CVT_PKRTZ_F16_F32_e64;
6364 case AMDGPU::S_FMAC_F32:
return AMDGPU::V_FMAC_F32_e64;
6365 case AMDGPU::S_FMAC_F16:
6366 return ST.useRealTrue16Insts() ? AMDGPU::V_FMAC_F16_t16_e64
6367 : AMDGPU::V_FMAC_F16_fake16_e64;
6368 case AMDGPU::S_FMAMK_F32:
return AMDGPU::V_FMAMK_F32;
6369 case AMDGPU::S_FMAAK_F32:
return AMDGPU::V_FMAAK_F32;
6370 case AMDGPU::S_CMP_LT_F32:
return AMDGPU::V_CMP_LT_F32_e64;
6371 case AMDGPU::S_CMP_EQ_F32:
return AMDGPU::V_CMP_EQ_F32_e64;
6372 case AMDGPU::S_CMP_LE_F32:
return AMDGPU::V_CMP_LE_F32_e64;
6373 case AMDGPU::S_CMP_GT_F32:
return AMDGPU::V_CMP_GT_F32_e64;
6374 case AMDGPU::S_CMP_LG_F32:
return AMDGPU::V_CMP_LG_F32_e64;
6375 case AMDGPU::S_CMP_GE_F32:
return AMDGPU::V_CMP_GE_F32_e64;
6376 case AMDGPU::S_CMP_O_F32:
return AMDGPU::V_CMP_O_F32_e64;
6377 case AMDGPU::S_CMP_U_F32:
return AMDGPU::V_CMP_U_F32_e64;
6378 case AMDGPU::S_CMP_NGE_F32:
return AMDGPU::V_CMP_NGE_F32_e64;
6379 case AMDGPU::S_CMP_NLG_F32:
return AMDGPU::V_CMP_NLG_F32_e64;
6380 case AMDGPU::S_CMP_NGT_F32:
return AMDGPU::V_CMP_NGT_F32_e64;
6381 case AMDGPU::S_CMP_NLE_F32:
return AMDGPU::V_CMP_NLE_F32_e64;
6382 case AMDGPU::S_CMP_NEQ_F32:
return AMDGPU::V_CMP_NEQ_F32_e64;
6383 case AMDGPU::S_CMP_NLT_F32:
return AMDGPU::V_CMP_NLT_F32_e64;
6384 case AMDGPU::S_CMP_LT_F16:
6385 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LT_F16_t16_e64
6386 : AMDGPU::V_CMP_LT_F16_fake16_e64;
6387 case AMDGPU::S_CMP_EQ_F16:
6388 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_EQ_F16_t16_e64
6389 : AMDGPU::V_CMP_EQ_F16_fake16_e64;
6390 case AMDGPU::S_CMP_LE_F16:
6391 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LE_F16_t16_e64
6392 : AMDGPU::V_CMP_LE_F16_fake16_e64;
6393 case AMDGPU::S_CMP_GT_F16:
6394 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GT_F16_t16_e64
6395 : AMDGPU::V_CMP_GT_F16_fake16_e64;
6396 case AMDGPU::S_CMP_LG_F16:
6397 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LG_F16_t16_e64
6398 : AMDGPU::V_CMP_LG_F16_fake16_e64;
6399 case AMDGPU::S_CMP_GE_F16:
6400 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GE_F16_t16_e64
6401 : AMDGPU::V_CMP_GE_F16_fake16_e64;
6402 case AMDGPU::S_CMP_O_F16:
6403 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_O_F16_t16_e64
6404 : AMDGPU::V_CMP_O_F16_fake16_e64;
6405 case AMDGPU::S_CMP_U_F16:
6406 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_U_F16_t16_e64
6407 : AMDGPU::V_CMP_U_F16_fake16_e64;
6408 case AMDGPU::S_CMP_NGE_F16:
6409 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGE_F16_t16_e64
6410 : AMDGPU::V_CMP_NGE_F16_fake16_e64;
6411 case AMDGPU::S_CMP_NLG_F16:
6412 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLG_F16_t16_e64
6413 : AMDGPU::V_CMP_NLG_F16_fake16_e64;
6414 case AMDGPU::S_CMP_NGT_F16:
6415 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGT_F16_t16_e64
6416 : AMDGPU::V_CMP_NGT_F16_fake16_e64;
6417 case AMDGPU::S_CMP_NLE_F16:
6418 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLE_F16_t16_e64
6419 : AMDGPU::V_CMP_NLE_F16_fake16_e64;
6420 case AMDGPU::S_CMP_NEQ_F16:
6421 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NEQ_F16_t16_e64
6422 : AMDGPU::V_CMP_NEQ_F16_fake16_e64;
6423 case AMDGPU::S_CMP_NLT_F16:
6424 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLT_F16_t16_e64
6425 : AMDGPU::V_CMP_NLT_F16_fake16_e64;
6426 case AMDGPU::V_S_EXP_F32_e64:
return AMDGPU::V_EXP_F32_e64;
6427 case AMDGPU::V_S_EXP_F16_e64:
6428 return ST.useRealTrue16Insts() ? AMDGPU::V_EXP_F16_t16_e64
6429 : AMDGPU::V_EXP_F16_fake16_e64;
6430 case AMDGPU::V_S_LOG_F32_e64:
return AMDGPU::V_LOG_F32_e64;
6431 case AMDGPU::V_S_LOG_F16_e64:
6432 return ST.useRealTrue16Insts() ? AMDGPU::V_LOG_F16_t16_e64
6433 : AMDGPU::V_LOG_F16_fake16_e64;
6434 case AMDGPU::V_S_RCP_F32_e64:
return AMDGPU::V_RCP_F32_e64;
6435 case AMDGPU::V_S_RCP_F16_e64:
6436 return ST.useRealTrue16Insts() ? AMDGPU::V_RCP_F16_t16_e64
6437 : AMDGPU::V_RCP_F16_fake16_e64;
6438 case AMDGPU::V_S_RSQ_F32_e64:
return AMDGPU::V_RSQ_F32_e64;
6439 case AMDGPU::V_S_RSQ_F16_e64:
6440 return ST.useRealTrue16Insts() ? AMDGPU::V_RSQ_F16_t16_e64
6441 : AMDGPU::V_RSQ_F16_fake16_e64;
6442 case AMDGPU::V_S_SQRT_F32_e64:
return AMDGPU::V_SQRT_F32_e64;
6443 case AMDGPU::V_S_SQRT_F16_e64:
6444 return ST.useRealTrue16Insts() ? AMDGPU::V_SQRT_F16_t16_e64
6445 : AMDGPU::V_SQRT_F16_fake16_e64;
6448 "Unexpected scalar opcode without corresponding vector one!");
6497 "Not a whole wave func");
6500 if (
MI.getOpcode() == AMDGPU::SI_WHOLE_WAVE_FUNC_SETUP ||
6501 MI.getOpcode() == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
6508 unsigned OpNo)
const {
6510 if (
MI.isVariadic() || OpNo >=
Desc.getNumOperands() ||
6511 Desc.operands()[OpNo].RegClass == -1) {
6514 if (Reg.isVirtual()) {
6518 return RI.getPhysRegBaseClass(Reg);
6521 int16_t RegClass = getOpRegClassID(
Desc.operands()[OpNo]);
6522 return RegClass < 0 ? nullptr : RI.getRegClass(RegClass);
6527 constexpr AMDGPU::OpName OpNames[] = {
6528 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2};
6531 int SrcIdx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[
I]);
6532 if (
static_cast<unsigned>(SrcIdx) == OpIdx)
6544 unsigned RCID = getOpRegClassID(
get(
MI.getOpcode()).operands()[OpIdx]);
6546 unsigned Size = RI.getRegSizeInBits(*RC);
6547 unsigned Opcode = (
Size == 64) ? AMDGPU::V_MOV_B64_PSEUDO
6548 :
Size == 16 ? AMDGPU::V_MOV_B16_t16_e64
6549 : AMDGPU::V_MOV_B32_e32;
6551 Opcode = AMDGPU::COPY;
6552 else if (RI.isSGPRClass(RC))
6553 Opcode = (
Size == 64) ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
6578 .
addImm(AMDGPU::sub0_sub1)
6580 .
addImm(AMDGPU::sub2_sub3);
6581 }
else if (Opcode == AMDGPU::V_MOV_B16_t16_e64) {
6598 return RI.getSubReg(SuperReg.
getReg(), SubIdx);
6604 unsigned NewSubIdx = RI.composeSubRegIndices(SuperReg.
getSubReg(), SubIdx);
6615 if (SubIdx == AMDGPU::sub0)
6617 if (SubIdx == AMDGPU::sub1)
6629void SIInstrInfo::swapOperands(
MachineInstr &Inst)
const {
6645 if (Reg.isPhysical())
6652 RI.getLargestLegalSuperClass(RC, MRI.
getMF());
6655 return RI.getMatchingSuperRegClass(SuperRC, DRC, MO.
getSubReg()) !=
nullptr;
6658 return RI.getCommonSubClass(DRC, RC) !=
nullptr;
6665 unsigned Opc =
MI.getOpcode();
6668 if (MO.
isReg() && RI.isSGPRReg(MRI, MO.
getReg()) &&
6678 bool IsAGPR = RI.isAGPR(MRI, MO.
getReg());
6679 if (IsAGPR && !ST.hasMAIInsts())
6685 const int VDstIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
6686 const int DataIdx = AMDGPU::getNamedOperandIdx(
6687 Opc,
isDS(
Opc) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata);
6688 if ((
int)OpIdx == VDstIdx && DataIdx != -1 &&
6689 MI.getOperand(DataIdx).isReg() &&
6690 RI.isAGPR(MRI,
MI.getOperand(DataIdx).getReg()) != IsAGPR)
6692 if ((
int)OpIdx == DataIdx) {
6693 if (VDstIdx != -1 &&
6694 RI.isAGPR(MRI,
MI.getOperand(VDstIdx).getReg()) != IsAGPR)
6697 const int Data1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data1);
6698 if (Data1Idx != -1 &&
MI.getOperand(Data1Idx).isReg() &&
6699 RI.isAGPR(MRI,
MI.getOperand(Data1Idx).getReg()) != IsAGPR)
6704 if (
Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts() &&
6705 (
int)OpIdx == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0) &&
6706 RI.isSGPRReg(MRI, MO.
getReg()))
6709 if (ST.hasFlatScratchHiInB64InstHazard() &&
6716 if (
Opc == AMDGPU::S_BITCMP0_B64 ||
Opc == AMDGPU::S_BITCMP1_B64)
6719 if (!ST.hasDPPSrc1SGPR() &&
isDPP(
MI) && RI.isSGPRReg(MRI, MO.
getReg()) &&
6720 (
int)OpIdx == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1))
6729 constexpr unsigned NumOps = 3;
6730 constexpr AMDGPU::OpName OpNames[
NumOps * 2] = {
6731 AMDGPU::OpName::src0, AMDGPU::OpName::src1,
6732 AMDGPU::OpName::src2, AMDGPU::OpName::src0_modifiers,
6733 AMDGPU::OpName::src1_modifiers, AMDGPU::OpName::src2_modifiers};
6738 int SrcIdx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[SrcN]);
6741 MO = &
MI.getOperand(SrcIdx);
6744 if (!MO->
isReg() || !RI.isSGPRReg(MRI, MO->
getReg()))
6748 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OpNames[
NumOps + SrcN]);
6752 unsigned Mods =
MI.getOperand(ModsIdx).getImm();
6756 return !OpSel && !OpSelHi;
6765 int64_t RegClass = getOpRegClassID(OpInfo);
6767 RegClass != -1 ? RI.getRegClass(RegClass) :
nullptr;
6769 MO = &
MI.getOperand(OpIdx);
6773 if (
isVALU(
MI,
false) && !IsInlineConst &&
6777 int ConstantBusLimit = ST.getConstantBusLimit(
MI.getOpcode());
6778 int LiteralLimit = !
isVOP3(
MI) || ST.hasVOP3Literal() ? 1 : 0;
6782 if (!LiteralLimit--)
6792 for (
unsigned i = 0, e =
MI.getNumOperands(); i != e; ++i) {
6800 if (--ConstantBusLimit <= 0)
6812 if (!LiteralLimit--)
6814 if (--ConstantBusLimit <= 0)
6820 for (
unsigned i = 0, e =
MI.getNumOperands(); i != e; ++i) {
6824 if (!
Op.isReg() && !
Op.isFI() && !
Op.isRegMask() &&
6826 !
Op.isIdenticalTo(*MO))
6848 bool Is64BitOp = Is64BitFPOp ||
6856 (!ST.has64BitLiterals() || InstDesc.
getSize() != 4))
6865 if (!Is64BitFPOp && (int32_t)
Imm < 0 &&
6883 bool IsGFX950Only = ST.hasGFX950Insts();
6884 bool IsGFX940Only = ST.hasGFX940Insts();
6886 if (!IsGFX950Only && !IsGFX940Only)
6904 unsigned Opcode =
MI.getOpcode();
6906 case AMDGPU::V_CVT_PK_BF8_F32_e64:
6907 case AMDGPU::V_CVT_PK_FP8_F32_e64:
6908 case AMDGPU::V_MQSAD_PK_U16_U8_e64:
6909 case AMDGPU::V_MQSAD_U32_U8_e64:
6910 case AMDGPU::V_PK_ADD_F16:
6911 case AMDGPU::V_PK_ADD_F32:
6912 case AMDGPU::V_PK_ADD_I16:
6913 case AMDGPU::V_PK_ADD_U16:
6914 case AMDGPU::V_PK_ASHRREV_I16:
6915 case AMDGPU::V_PK_FMA_F16:
6916 case AMDGPU::V_PK_FMA_F32:
6917 case AMDGPU::V_PK_FMAC_F16_e32:
6918 case AMDGPU::V_PK_FMAC_F16_e64:
6919 case AMDGPU::V_PK_LSHLREV_B16:
6920 case AMDGPU::V_PK_LSHRREV_B16:
6921 case AMDGPU::V_PK_MAD_I16:
6922 case AMDGPU::V_PK_MAD_U16:
6923 case AMDGPU::V_PK_MAX_F16:
6924 case AMDGPU::V_PK_MAX_I16:
6925 case AMDGPU::V_PK_MAX_U16:
6926 case AMDGPU::V_PK_MIN_F16:
6927 case AMDGPU::V_PK_MIN_I16:
6928 case AMDGPU::V_PK_MIN_U16:
6929 case AMDGPU::V_PK_MOV_B32:
6930 case AMDGPU::V_PK_MUL_F16:
6931 case AMDGPU::V_PK_MUL_F32:
6932 case AMDGPU::V_PK_MUL_LO_U16:
6933 case AMDGPU::V_PK_SUB_I16:
6934 case AMDGPU::V_PK_SUB_U16:
6935 case AMDGPU::V_QSAD_PK_U16_U8_e64:
6944 unsigned Opc =
MI.getOpcode();
6947 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
6950 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
6956 if (HasImplicitSGPR && ST.getConstantBusLimit(
Opc) <= 1 && Src0.isReg() &&
6957 RI.isSGPRReg(MRI, Src0.getReg()))
6963 if (
Opc == AMDGPU::V_WRITELANE_B32) {
6965 if (Src0.isReg() && RI.isVGPR(MRI, Src0.getReg())) {
6969 Src0.ChangeToRegister(Reg,
false);
6971 if (Src1.isReg() && RI.isVGPR(MRI, Src1.getReg())) {
6976 Src1.ChangeToRegister(Reg,
false);
6982 if (
Opc == AMDGPU::V_FMAC_F32_e32 ||
Opc == AMDGPU::V_FMAC_F16_e32) {
6983 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
6984 if (!RI.isVGPR(MRI,
MI.getOperand(Src2Idx).getReg()))
6996 if (
Opc == AMDGPU::V_READLANE_B32 && Src1.isReg() &&
6997 RI.isVGPR(MRI, Src1.getReg())) {
7002 Src1.ChangeToRegister(Reg,
false);
7010 if (HasImplicitSGPR || !
MI.isCommutable()) {
7020 if ((!Src1.isImm() && !Src1.isReg()) ||
7027 if (CommutedOpc == -1) {
7032 MI.setDesc(
get(CommutedOpc));
7035 unsigned Src0SubReg = Src0.getSubReg();
7036 bool Src0Kill = Src0.isKill();
7039 Src0.ChangeToImmediate(Src1.getImm());
7040 else if (Src1.isReg()) {
7041 Src0.ChangeToRegister(Src1.getReg(),
false,
false, Src1.isKill());
7042 Src0.setSubReg(Src1.getSubReg());
7046 Src1.ChangeToRegister(Src0Reg,
false,
false, Src0Kill);
7047 Src1.setSubReg(Src0SubReg);
7055 unsigned Opc =
MI.getOpcode();
7058 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0),
7059 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1),
7060 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2)
7063 if (
Opc == AMDGPU::V_PERMLANE16_B32_e64 ||
7064 Opc == AMDGPU::V_PERMLANEX16_B32_e64 ||
7065 Opc == AMDGPU::V_PERMLANE_BCAST_B32_e64 ||
7066 Opc == AMDGPU::V_PERMLANE_UP_B32_e64 ||
7067 Opc == AMDGPU::V_PERMLANE_DOWN_B32_e64 ||
7068 Opc == AMDGPU::V_PERMLANE_XOR_B32_e64 ||
7069 Opc == AMDGPU::V_PERMLANE_IDX_GEN_B32_e64) {
7073 if (Src1.isReg() && !RI.isSGPRClass(MRI.
getRegClass(Src1.getReg()))) {
7077 Src1.ChangeToRegister(Reg,
false);
7079 if (VOP3Idx[2] != -1) {
7081 if (Src2.isReg() && !RI.isSGPRClass(MRI.
getRegClass(Src2.getReg()))) {
7085 Src2.ChangeToRegister(Reg,
false);
7091 int ConstantBusLimit = ST.getConstantBusLimit(
Opc);
7092 int LiteralLimit = ST.hasVOP3Literal() ? 1 : 0;
7094 Register SGPRReg = findUsedSGPR(
MI, VOP3Idx);
7096 SGPRsUsed.
insert(SGPRReg);
7100 for (
int Idx : VOP3Idx) {
7109 if (LiteralLimit > 0 && ConstantBusLimit > 0) {
7121 if (!RI.isSGPRClass(RI.getRegClassForReg(MRI, MO.
getReg())))
7128 if (ConstantBusLimit > 0) {
7140 if ((
Opc == AMDGPU::V_FMAC_F32_e64 ||
Opc == AMDGPU::V_FMAC_F16_e64) &&
7141 !RI.isVGPR(MRI,
MI.getOperand(VOP3Idx[2]).getReg()))
7147 for (
unsigned I = 0;
I < 3; ++
I) {
7160 SRC = RI.getCommonSubClass(SRC, DstRC);
7163 unsigned SubRegs = RI.getRegSizeInBits(*VRC) / 32;
7165 if (RI.hasAGPRs(VRC)) {
7166 VRC = RI.getEquivalentVGPRClass(VRC);
7169 get(TargetOpcode::COPY), NewSrcReg)
7176 get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
7182 for (
unsigned i = 0; i < SubRegs; ++i) {
7185 get(AMDGPU::V_READFIRSTLANE_B32), SGPR)
7186 .
addReg(SrcReg, {}, RI.getSubRegFromChannel(i));
7192 get(AMDGPU::REG_SEQUENCE), DstReg);
7193 for (
unsigned i = 0; i < SubRegs; ++i) {
7195 MIB.
addImm(RI.getSubRegFromChannel(i));
7208 if (SBase && !RI.isSGPRClass(MRI.
getRegClass(SBase->getReg()))) {
7210 SBase->setReg(SGPR);
7213 if (SOff && !RI.isSGPRReg(MRI, SOff->
getReg())) {
7221 int OldSAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
7222 if (OldSAddrIdx < 0)
7235 if (RI.isSGPRReg(MRI, SAddr.
getReg()))
7238 int NewVAddrIdx = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vaddr);
7239 if (NewVAddrIdx < 0)
7242 int OldVAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
7246 if (OldVAddrIdx >= 0) {
7260 if (OldVAddrIdx == NewVAddrIdx) {
7271 assert(OldSAddrIdx == NewVAddrIdx);
7273 if (OldVAddrIdx >= 0) {
7274 int NewVDstIn = AMDGPU::getNamedOperandIdx(NewOpc,
7275 AMDGPU::OpName::vdst_in);
7279 if (NewVDstIn != -1) {
7280 int OldVDstIn = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
7286 if (NewVDstIn != -1) {
7287 int NewVDst = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
7328 unsigned OpSubReg =
Op.getSubReg();
7331 RI.getRegClassForReg(MRI, OpReg), OpSubReg);
7338 auto Copy =
BuildMI(InsertMBB,
I,
DL,
get(AMDGPU::COPY), DstReg)
7339 .
addReg(OpReg, {}, OpSubReg);
7341 Op.setSubReg(AMDGPU::NoSubRegister);
7348 if (Def->isMoveImmediate() && DstRC != &AMDGPU::VReg_1RegClass)
7351 bool ImpDef = Def->isImplicitDef();
7352 while (!ImpDef && Def && Def->isCopy()) {
7353 if (Def->getOperand(1).getReg().isPhysical())
7356 ImpDef = Def && Def->isImplicitDef();
7358 if (!RI.isSGPRClass(DstRC) && !Copy->readsRegister(AMDGPU::EXEC, &RI) &&
7374 const auto *BoolXExecRC =
TRI->getWaveMaskRegClass();
7379 bool UseNewExecInstructions =
7388 if (UseNewExecInstructions) {
7423 for (
auto [Idx, ScalarOp] :
enumerate(ScalarOps)) {
7424 unsigned RegSize =
TRI->getRegSizeInBits(ScalarOp->getReg(), MRI);
7425 unsigned NumSubRegs =
RegSize / 32;
7426 Register VScalarOp = ScalarOp->getReg();
7429 TII.getRegClass(
TII.get(AMDGPU::V_READFIRSTLANE_B32), 1);
7431 if (NumSubRegs == 1) {
7434 TRI->getCommonSubClass(VScalarOpRC, RFLSrcRC);
7435 Common != VScalarOpRC) {
7442 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurReg)
7445 if (UseNewExecInstructions) {
7447 TII.get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term))
7450 if (
I == LoopBB.
end())
7455 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_CMP_EQ_U32_e64), NewCondReg)
7461 CondReg = NewCondReg;
7473 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7474 ScalarOp->setReg(CurReg);
7477 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(),
DL,
7478 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7480 ScalarOp->setReg(PhySGPRs[Idx]);
7482 ScalarOp->setIsKill();
7486 assert(NumSubRegs % 2 == 0 && NumSubRegs <= 32 &&
7487 "Unhandled register size");
7489 for (
unsigned Idx = 0;
Idx < NumSubRegs;
Idx += 2) {
7496 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegLo)
7497 .
addReg(VScalarOp, VScalarOpUndef,
TRI->getSubRegFromChannel(Idx));
7500 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegHi)
7501 .
addReg(VScalarOp, VScalarOpUndef,
7502 TRI->getSubRegFromChannel(Idx + 1));
7509 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::REG_SEQUENCE), CurReg)
7516 NumSubRegs <= 2 ? 0 :
TRI->getSubRegFromChannel(Idx, 2);
7518 if (UseNewExecInstructions) {
7520 TII.get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term))
7522 .
addReg(VScalarOp, VScalarOpUndef, SubReg);
7523 if (
I == LoopBB.
end())
7527 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::V_CMP_EQ_U64_e64), NewCondReg)
7529 .
addReg(VScalarOp, VScalarOpUndef, SubReg);
7533 CondReg = NewCondReg;
7545 const auto *SScalarOpRC =
7551 BuildMI(LoopBB,
I,
DL,
TII.get(AMDGPU::REG_SEQUENCE), SScalarOp);
7552 unsigned Channel = 0;
7553 for (
Register Piece : ReadlanePieces) {
7554 Merge.addReg(Piece).addImm(
TRI->getSubRegFromChannel(Channel++));
7558 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7559 ScalarOp->setReg(SScalarOp);
7561 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(),
DL,
7562 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7564 ScalarOp->setReg(PhySGPRs[Idx]);
7566 ScalarOp->setIsKill();
7573 if (!UseNewExecInstructions) {
7586 if (UseNewExecInstructions) {
7619 assert((PhySGPRs.empty() || PhySGPRs.size() == ScalarOps.
size()) &&
7620 "Physical SGPRs must be empty or match the number of scalar operands");
7626 if (!Begin.isValid())
7628 if (!End.isValid()) {
7634 const auto *BoolXExecRC =
TRI->getWaveMaskRegClass();
7643 std::numeric_limits<unsigned>::max()) !=
7661 for (
auto I = Begin;
I != AfterMI;
I++) {
7662 for (
auto &MO :
I->all_uses())
7698 for (
auto &Succ : RemainderBB->
successors()) {
7723static std::tuple<unsigned, unsigned>
7731 TII.buildExtractSubReg(
MI, MRI, Rsrc, &AMDGPU::VReg_128RegClass,
7732 AMDGPU::sub0_sub1, &AMDGPU::VReg_64RegClass);
7739 uint64_t RsrcDataFormat =
TII.getDefaultRsrcDataFormat();
7756 .
addImm(AMDGPU::sub0_sub1)
7762 return std::tuple(RsrcPtr, NewSRsrc);
7773 if (ST.useRealTrue16Insts())
7803 if (
MI.getOpcode() == AMDGPU::PHI) {
7805 assert(!RI.isSGPRClass(VRC));
7808 for (
unsigned I = 1, E =
MI.getNumOperands();
I != E;
I += 2) {
7810 if (!
Op.isReg() || !
Op.getReg().isVirtual())
7826 if (
MI.getOpcode() == AMDGPU::REG_SEQUENCE) {
7829 if (RI.hasVGPRs(DstRC)) {
7833 for (
unsigned I = 1, E =
MI.getNumOperands();
I != E;
I += 2) {
7835 if (!
Op.isReg() || !
Op.getReg().isVirtual())
7853 if (
MI.getOpcode() == AMDGPU::INSERT_SUBREG) {
7858 if (DstRC != Src0RC) {
7867 if (
MI.getOpcode() == AMDGPU::SI_INIT_M0) {
7869 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7875 if (
MI.getOpcode() == AMDGPU::S_BITREPLICATE_B64_B32 ||
7876 MI.getOpcode() == AMDGPU::S_QUADMASK_B32 ||
7877 MI.getOpcode() == AMDGPU::S_QUADMASK_B64 ||
7878 MI.getOpcode() == AMDGPU::S_WQM_B32 ||
7879 MI.getOpcode() == AMDGPU::S_WQM_B64 ||
7880 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U32 ||
7881 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U64) {
7883 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7896 ? AMDGPU::OpName::rsrc
7897 : AMDGPU::OpName::srsrc;
7902 AMDGPU::OpName SampOpName =
7903 isMIMG(
MI) ? AMDGPU::OpName::ssamp : AMDGPU::OpName::samp;
7912 if (
MI.getOpcode() == AMDGPU::SI_CALL_ISEL) {
7920 if (
MI.getOpcode() == AMDGPU::S_SLEEP_VAR) {
7924 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src0);
7928 Src0.ChangeToRegister(Reg,
false);
7934 if (
MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d2 ||
7935 MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d4 ||
7936 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d2 ||
7937 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d4) {
7939 if (Src.isReg() && RI.hasVectorRegisters(MRI.
getRegClass(Src.getReg())))
7946 bool isSoffsetLegal =
true;
7948 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::soffset);
7949 if (SoffsetIdx != -1) {
7953 isSoffsetLegal =
false;
7957 bool isRsrcLegal =
true;
7959 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::srsrc);
7960 if (RsrcIdx != -1) {
7962 if (Rsrc->
isReg() && !RI.isSGPRReg(MRI, Rsrc->
getReg()))
7963 isRsrcLegal =
false;
7967 if (isRsrcLegal && isSoffsetLegal)
7995 const auto *BoolXExecRC = RI.getWaveMaskRegClass();
7999 unsigned RsrcPtr, NewSRsrc;
8006 .
addReg(RsrcPtr, {}, AMDGPU::sub0)
8007 .addReg(VAddr->
getReg(), {}, AMDGPU::sub0)
8013 .
addReg(RsrcPtr, {}, AMDGPU::sub1)
8014 .addReg(VAddr->
getReg(), {}, AMDGPU::sub1)
8027 }
else if (!VAddr && ST.hasAddr64()) {
8031 "FIXME: Need to emit flat atomics here");
8033 unsigned RsrcPtr, NewSRsrc;
8059 MIB.
addImm(CPol->getImm());
8064 MIB.
addImm(TFE->getImm());
8084 MI.removeFromParent();
8089 .
addReg(RsrcPtr, {}, AMDGPU::sub0)
8090 .addImm(AMDGPU::sub0)
8091 .
addReg(RsrcPtr, {}, AMDGPU::sub1)
8092 .addImm(AMDGPU::sub1);
8095 if (!isSoffsetLegal) {
8106 if (!isSoffsetLegal) {
8115 if (InSet.insert(
MI).second)
8119 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::srsrc);
8120 if (RsrcIdx != -1) {
8121 DeferredList.insert(
MI);
8126 return DeferredList.contains(
MI);
8136 if (!ST.useRealTrue16Insts())
8139 unsigned Opcode =
MI.getOpcode();
8142 if (OpIdx >=
MI.getNumExplicitOperands() ||
8143 OpIdx >=
get(Opcode).getNumOperands() ||
8144 get(Opcode).operands()[OpIdx].RegClass == -1)
8148 if (!
Op.isReg() || !
Op.getReg().isVirtual() ||
Op.isDef())
8152 if (!RI.isVGPRClass(CurrRC))
8155 int16_t RCID = getOpRegClassID(
get(Opcode).operands()[OpIdx]);
8157 if (RI.getMatchingSuperRegClass(CurrRC, ExpectedRC, AMDGPU::lo16)) {
8159 if (
Op.getSubReg() == AMDGPU::NoSubRegister)
8160 Op.setSubReg(AMDGPU::lo16);
8165 RI.getSubRegisterClass(CurrRC,
Op.getSubReg());
8166 if (RI.getMatchingSuperRegClass(ExpectedRC, CurrSRC, AMDGPU::lo16)) {
8176 Op.setReg(NewDstReg);
8177 Op.setSubReg(AMDGPU::NoSubRegister);
8182 for (
unsigned OpIdx = 0; OpIdx <
MI.getNumExplicitOperands(); OpIdx++)
8190 assert(
MI->getOpcode() == AMDGPU::SI_CALL_ISEL &&
8191 "This only handle waterfall for SI_CALL_ISEL");
8198 while (Start->getOpcode() != AMDGPU::ADJCALLSTACKUP)
8201 while (End->getOpcode() != AMDGPU::ADJCALLSTACKDOWN)
8206 while (End !=
MBB.end() && End->isCopy() &&
8207 MI->definesRegister(End->getOperand(1).getReg(), &RI))
8217 while (!Worklist.
empty()) {
8223 moveToVALUImpl(Worklist, MDT, Inst, WaterFalls, V2SPhyCopiesToErase);
8229 moveToVALUImpl(Worklist, MDT, *Inst, WaterFalls, V2SPhyCopiesToErase);
8231 "Deferred MachineInstr are not supposed to re-populate worklist");
8234 for (
auto &Entry : WaterFalls) {
8235 if (Entry.first->getOpcode() == AMDGPU::SI_CALL_ISEL)
8237 Entry.second.SGPRs);
8240 for (std::pair<MachineInstr *, bool> Entry : V2SPhyCopiesToErase)
8242 Entry.first->eraseFromParent();
8250 if (SubRegIndices.
size() <= 1) {
8253 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8260 for (int16_t Indice : SubRegIndices) {
8263 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8270 get(AMDGPU::REG_SEQUENCE), DstReg);
8271 for (
unsigned i = 0; i < SubRegIndices.size(); ++i) {
8273 MIB.
addImm(RI.getSubRegFromChannel(i));
8283 if (DstReg == AMDGPU::M0) {
8296 if (
I->getOpcode() == AMDGPU::SI_CALL_ISEL) {
8298 for (
unsigned i = 0; i <
UseMI->getNumOperands(); ++i) {
8299 if (
UseMI->getOperand(i).isReg() &&
8300 UseMI->getOperand(i).getReg() == DstReg) {
8304 V2SCopyInfo.MOs.push_back(MO);
8305 V2SCopyInfo.SGPRs.push_back(DstReg);
8309 }
else if (
I->getOpcode() == AMDGPU::SI_RETURN_TO_EPILOG &&
8310 I->getOperand(0).isReg() &&
8311 I->getOperand(0).getReg() == DstReg) {
8314 }
else if (
I->readsRegister(DstReg, &RI)) {
8316 V2SPhyCopiesToErase[&Inst] =
false;
8318 if (
I->findRegisterDefOperand(DstReg, &RI))
8340 case AMDGPU::S_ADD_I32:
8341 case AMDGPU::S_SUB_I32: {
8345 std::tie(
Changed, CreatedBBTmp) = moveScalarAddSub(Worklist, Inst, MDT);
8353 case AMDGPU::S_MUL_U64:
8354 if (ST.useVMulU64Inst()) {
8355 NewOpcode = AMDGPU::V_MUL_U64_e64;
8359 splitScalarSMulU64(Worklist, Inst, MDT);
8363 case AMDGPU::S_MUL_U64_U32_PSEUDO:
8364 case AMDGPU::S_MUL_I64_I32_PSEUDO:
8367 splitScalarSMulPseudo(Worklist, Inst, MDT);
8371 case AMDGPU::S_AND_B64:
8372 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_AND_B32, MDT);
8376 case AMDGPU::S_OR_B64:
8377 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_OR_B32, MDT);
8381 case AMDGPU::S_XOR_B64:
8382 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XOR_B32, MDT);
8386 case AMDGPU::S_NAND_B64:
8387 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NAND_B32, MDT);
8391 case AMDGPU::S_NOR_B64:
8392 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NOR_B32, MDT);
8396 case AMDGPU::S_XNOR_B64:
8397 if (ST.hasDLInsts())
8398 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XNOR_B32, MDT);
8400 splitScalar64BitXnor(Worklist, Inst, MDT);
8404 case AMDGPU::S_ANDN2_B64:
8405 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ANDN2_B32, MDT);
8409 case AMDGPU::S_ORN2_B64:
8410 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ORN2_B32, MDT);
8414 case AMDGPU::S_BREV_B64:
8415 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_BREV_B32,
true);
8419 case AMDGPU::S_NOT_B64:
8420 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_NOT_B32);
8424 case AMDGPU::S_BCNT1_I32_B64:
8425 splitScalar64BitBCNT(Worklist, Inst);
8429 case AMDGPU::S_BFE_I64:
8430 splitScalar64BitBFE(Worklist, Inst);
8434 case AMDGPU::S_FLBIT_I32_B64:
8435 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBH_U32_e32);
8438 case AMDGPU::S_FF1_I32_B64:
8439 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBL_B32_e32);
8443 case AMDGPU::S_LSHL_B32:
8444 if (ST.hasOnlyRevVALUShifts()) {
8445 NewOpcode = AMDGPU::V_LSHLREV_B32_e64;
8449 case AMDGPU::S_ASHR_I32:
8450 if (ST.hasOnlyRevVALUShifts()) {
8451 NewOpcode = AMDGPU::V_ASHRREV_I32_e64;
8455 case AMDGPU::S_LSHR_B32:
8456 if (ST.hasOnlyRevVALUShifts()) {
8457 NewOpcode = AMDGPU::V_LSHRREV_B32_e64;
8461 case AMDGPU::S_LSHL_B64:
8462 if (ST.hasOnlyRevVALUShifts()) {
8464 ? AMDGPU::V_LSHLREV_B64_pseudo_e64
8465 : AMDGPU::V_LSHLREV_B64_e64;
8469 case AMDGPU::S_ASHR_I64:
8470 if (ST.hasOnlyRevVALUShifts()) {
8471 NewOpcode = AMDGPU::V_ASHRREV_I64_e64;
8475 case AMDGPU::S_LSHR_B64:
8476 if (ST.hasOnlyRevVALUShifts()) {
8477 NewOpcode = AMDGPU::V_LSHRREV_B64_e64;
8482 case AMDGPU::S_ABS_I32:
8483 lowerScalarAbs(Worklist, Inst);
8487 case AMDGPU::S_ABSDIFF_I32:
8488 lowerScalarAbsDiff(Worklist, Inst);
8492 case AMDGPU::S_CBRANCH_SCC0:
8493 case AMDGPU::S_CBRANCH_SCC1: {
8496 bool IsSCC = CondReg == AMDGPU::SCC;
8505 case AMDGPU::S_BFE_U64:
8506 case AMDGPU::S_BFM_B64:
8509 case AMDGPU::S_PACK_LL_B32_B16:
8510 case AMDGPU::S_PACK_LH_B32_B16:
8511 case AMDGPU::S_PACK_HL_B32_B16:
8512 case AMDGPU::S_PACK_HH_B32_B16:
8513 movePackToVALU(Worklist, MRI, Inst);
8517 case AMDGPU::S_XNOR_B32:
8518 lowerScalarXnor(Worklist, Inst);
8522 case AMDGPU::S_NAND_B32:
8523 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_AND_B32);
8527 case AMDGPU::S_NOR_B32:
8528 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_OR_B32);
8532 case AMDGPU::S_ANDN2_B32:
8533 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_AND_B32);
8537 case AMDGPU::S_ORN2_B32:
8538 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_OR_B32);
8546 case AMDGPU::S_ADD_CO_PSEUDO:
8547 case AMDGPU::S_SUB_CO_PSEUDO: {
8548 unsigned Opc = (Inst.
getOpcode() == AMDGPU::S_ADD_CO_PSEUDO)
8549 ? AMDGPU::V_ADDC_U32_e64
8550 : AMDGPU::V_SUBB_U32_e64;
8551 const auto *CarryRC = RI.getWaveMaskRegClass();
8573 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8577 case AMDGPU::S_UADDO_PSEUDO:
8578 case AMDGPU::S_USUBO_PSEUDO: {
8584 unsigned Opc = (Inst.
getOpcode() == AMDGPU::S_UADDO_PSEUDO)
8585 ? AMDGPU::V_ADD_CO_U32_e64
8586 : AMDGPU::V_SUB_CO_U32_e64;
8598 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8602 case AMDGPU::S_LSHL1_ADD_U32:
8603 case AMDGPU::S_LSHL2_ADD_U32:
8604 case AMDGPU::S_LSHL3_ADD_U32:
8605 case AMDGPU::S_LSHL4_ADD_U32: {
8609 unsigned ShiftAmt = (Opcode == AMDGPU::S_LSHL1_ADD_U32 ? 1
8610 : Opcode == AMDGPU::S_LSHL2_ADD_U32 ? 2
8611 : Opcode == AMDGPU::S_LSHL3_ADD_U32 ? 3
8625 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8629 case AMDGPU::S_CSELECT_B32:
8630 case AMDGPU::S_CSELECT_B64:
8631 lowerSelect(Worklist, Inst, MDT);
8634 case AMDGPU::S_CMP_EQ_I32:
8635 case AMDGPU::S_CMP_LG_I32:
8636 case AMDGPU::S_CMP_GT_I32:
8637 case AMDGPU::S_CMP_GE_I32:
8638 case AMDGPU::S_CMP_LT_I32:
8639 case AMDGPU::S_CMP_LE_I32:
8640 case AMDGPU::S_CMP_EQ_U32:
8641 case AMDGPU::S_CMP_LG_U32:
8642 case AMDGPU::S_CMP_GT_U32:
8643 case AMDGPU::S_CMP_GE_U32:
8644 case AMDGPU::S_CMP_LT_U32:
8645 case AMDGPU::S_CMP_LE_U32:
8646 case AMDGPU::S_CMP_EQ_U64:
8647 case AMDGPU::S_CMP_LG_U64:
8648 case AMDGPU::S_CMP_LT_F32:
8649 case AMDGPU::S_CMP_EQ_F32:
8650 case AMDGPU::S_CMP_LE_F32:
8651 case AMDGPU::S_CMP_GT_F32:
8652 case AMDGPU::S_CMP_LG_F32:
8653 case AMDGPU::S_CMP_GE_F32:
8654 case AMDGPU::S_CMP_O_F32:
8655 case AMDGPU::S_CMP_U_F32:
8656 case AMDGPU::S_CMP_NGE_F32:
8657 case AMDGPU::S_CMP_NLG_F32:
8658 case AMDGPU::S_CMP_NGT_F32:
8659 case AMDGPU::S_CMP_NLE_F32:
8660 case AMDGPU::S_CMP_NEQ_F32:
8661 case AMDGPU::S_CMP_NLT_F32: {
8666 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src0_modifiers) >=
8680 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8684 case AMDGPU::S_CMP_LT_F16:
8685 case AMDGPU::S_CMP_EQ_F16:
8686 case AMDGPU::S_CMP_LE_F16:
8687 case AMDGPU::S_CMP_GT_F16:
8688 case AMDGPU::S_CMP_LG_F16:
8689 case AMDGPU::S_CMP_GE_F16:
8690 case AMDGPU::S_CMP_O_F16:
8691 case AMDGPU::S_CMP_U_F16:
8692 case AMDGPU::S_CMP_NGE_F16:
8693 case AMDGPU::S_CMP_NLG_F16:
8694 case AMDGPU::S_CMP_NGT_F16:
8695 case AMDGPU::S_CMP_NLE_F16:
8696 case AMDGPU::S_CMP_NEQ_F16:
8697 case AMDGPU::S_CMP_NLT_F16: {
8719 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8723 case AMDGPU::S_CVT_HI_F32_F16: {
8726 if (ST.useRealTrue16Insts()) {
8731 .
addReg(TmpReg, {}, AMDGPU::hi16)
8747 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8751 case AMDGPU::S_MINIMUM_F32:
8752 case AMDGPU::S_MAXIMUM_F32: {
8764 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8768 case AMDGPU::S_MINIMUM_F16:
8769 case AMDGPU::S_MAXIMUM_F16: {
8771 ? &AMDGPU::VGPR_16RegClass
8772 : &AMDGPU::VGPR_32RegClass);
8783 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8787 case AMDGPU::V_S_EXP_F16_e64:
8788 case AMDGPU::V_S_LOG_F16_e64:
8789 case AMDGPU::V_S_RCP_F16_e64:
8790 case AMDGPU::V_S_RSQ_F16_e64:
8791 case AMDGPU::V_S_SQRT_F16_e64: {
8793 ? &AMDGPU::VGPR_16RegClass
8794 : &AMDGPU::VGPR_32RegClass);
8805 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8811 if (NewOpcode == AMDGPU::INSTRUCTION_LIST_END) {
8819 if (NewOpcode == Opcode) {
8826 V2SPhyCopiesToErase);
8834 RI.getCommonSubClass(NewDstRC, SrcRC)) {
8841 addUsersToMoveToVALUWorklist(DstReg, MRI, Worklist);
8847 RI.composeSubRegIndices(SrcSubReg, UseMO.getSubReg()));
8848 UseMO.setReg(NewDstReg);
8867 unsigned OpIdx =
UseMI.getOperandNo(&UseMO);
8880 if (ST.useRealTrue16Insts() && Inst.
isCopy() &&
8884 if (RI.getMatchingSuperRegClass(NewDstRC, SrcRegRC, AMDGPU::lo16)) {
8890 get(AMDGPU::REG_SEQUENCE), NewDstReg)
8897 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8899 }
else if (RI.getMatchingSuperRegClass(SrcRegRC, NewDstRC,
8904 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8912 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8922 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8923 AMDGPU::OpName::src0_modifiers) >= 0)
8927 NewInstr->addOperand(Src);
8930 if (Opcode == AMDGPU::S_SEXT_I32_I8 || Opcode == AMDGPU::S_SEXT_I32_I16) {
8933 unsigned Size = (Opcode == AMDGPU::S_SEXT_I32_I8) ? 8 : 16;
8935 NewInstr.addImm(
Size);
8936 }
else if (Opcode == AMDGPU::S_BCNT1_I32_B32) {
8940 }
else if (Opcode == AMDGPU::S_BFE_I32 || Opcode == AMDGPU::S_BFE_U32) {
8945 "Scalar BFE is only implemented for constant width and offset");
8953 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8954 AMDGPU::OpName::src1_modifiers) >= 0)
8956 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src1) >= 0)
8958 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8959 AMDGPU::OpName::src2_modifiers) >= 0)
8961 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src2) >= 0)
8963 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::clamp) >= 0)
8965 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::omod) >= 0)
8967 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::op_sel) >= 0)
8973 NewInstr->addOperand(
Op);
8979 bool DeadSCCDef =
false;
8981 if (
Op.getReg() == AMDGPU::SCC) {
8987 addSCCDefUsersToVALUWorklist(
Op, Inst, Worklist);
8991 addSCCDefsToVALUWorklist(NewInstr, Worklist);
8996 if (NewInstr->getOperand(0).isReg() && NewInstr->getOperand(0).isDef()) {
8997 Register DstReg = NewInstr->getOperand(0).getReg();
9011 NewInstr->findRegisterDefOperand(RI.getVCC(), &RI))
9012 VCCDef->setIsDead();
9018 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
9022std::pair<bool, MachineBasicBlock *>
9025 if (ST.hasAddNoCarryInsts()) {
9037 assert(
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_SUB_I32);
9039 unsigned NewOpc =
Opc == AMDGPU::S_ADD_I32 ?
9040 AMDGPU::V_ADD_U32_e64 : AMDGPU::V_SUB_U32_e64;
9051 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9052 return std::pair(
true, NewBB);
9055 return std::pair(
false,
nullptr);
9072 bool IsSCC = (CondReg == AMDGPU::SCC);
9078 if (!IsSCC &&
Src0.isImm() && (
Src0.getImm() == -1) &&
Src1.isImm() &&
9079 (
Src1.getImm() == 0)) {
9080 for (MachineOperand &UseMO :
9082 MachineInstr &
UseMI = *UseMO.getParent();
9083 switch (
UseMI.getOpcode()) {
9084 case AMDGPU::V_CNDMASK_B16_fake16_e32:
9085 case AMDGPU::V_CNDMASK_B16_fake16_e64:
9086 case AMDGPU::V_CNDMASK_B16_t16_e32:
9087 case AMDGPU::V_CNDMASK_B16_t16_e64:
9088 case AMDGPU::V_CNDMASK_B32_e32:
9089 case AMDGPU::V_CNDMASK_B32_e64:
9090 case AMDGPU::V_CNDMASK_B64_PSEUDO:
9091 if (UseMO.isImplicit() ||
9093 UseMO.setReg(CondReg);
9107 bool CopyFound =
false;
9108 for (MachineInstr &CandI :
9111 if (CandI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI,
false,
false) !=
9113 if (CandI.isCopy() && CandI.getOperand(0).getReg() == AMDGPU::SCC) {
9115 .
addReg(CandI.getOperand(1).getReg());
9127 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
9136 MachineInstr *NewInst;
9137 if (Inst.
getOpcode() == AMDGPU::S_CSELECT_B32) {
9138 NewInst =
BuildMI(
MBB, MII,
DL,
get(AMDGPU::V_CNDMASK_B32_e64), NewDestReg)
9153 addUsersToMoveToVALUWorklist(NewDestReg, MRI, Worklist);
9168 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9170 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9172 MachineInstrBuilder
Sub =
9182 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9199 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9201 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9203 MachineInstrBuilder Sub1 =
BuildMI(
MBB, MII,
DL,
get(SubOp), SubResultReg)
9207 MachineInstrBuilder Sub2 =
9220 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9234 if (ST.hasDLInsts()) {
9244 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9250 bool Src0IsSGPR =
Src0.isReg() &&
9252 bool Src1IsSGPR =
Src1.isReg() &&
9266 }
else if (Src1IsSGPR) {
9284 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9290 unsigned Opcode)
const {
9314 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9319 unsigned Opcode)
const {
9343 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9358 const MCInstrDesc &InstDesc =
get(Opcode);
9361 &AMDGPU::SGPR_32RegClass;
9364 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9367 AMDGPU::sub0, Src0SubRC);
9372 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9375 MachineInstr &LoHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub0).
add(SrcReg0Sub0);
9378 AMDGPU::sub1, Src0SubRC);
9381 MachineInstr &HiHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub1).
add(SrcReg0Sub1);
9395 Worklist.
insert(&LoHalf);
9396 Worklist.
insert(&HiHalf);
9402 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9426 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9427 if (RI.isSGPRClass(Src0SubRC))
9428 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9430 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9431 if (RI.isSGPRClass(Src1SubRC))
9432 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9436 MachineOperand Op0L =
9438 MachineOperand Op1L =
9440 MachineOperand Op0H =
9442 MachineOperand Op1H =
9461 MachineInstr *Op1L_Op0H =
9467 MachineInstr *Op1H_Op0L =
9473 MachineInstr *Carry =
9478 MachineInstr *LoHalf =
9488 MachineInstr *HiHalf =
9511 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9535 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9536 if (RI.isSGPRClass(Src0SubRC))
9537 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9539 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9540 if (RI.isSGPRClass(Src1SubRC))
9541 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9545 MachineOperand Op0L =
9547 MachineOperand Op1L =
9551 unsigned NewOpc =
Opc == AMDGPU::S_MUL_U64_U32_PSEUDO
9552 ? AMDGPU::V_MUL_HI_U32_e64
9553 : AMDGPU::V_MUL_HI_I32_e64;
9554 MachineInstr *HiHalf =
9557 MachineInstr *LoHalf =
9576 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9592 const MCInstrDesc &InstDesc =
get(Opcode);
9595 &AMDGPU::SGPR_32RegClass;
9598 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9601 &AMDGPU::SGPR_32RegClass;
9604 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9607 AMDGPU::sub0, Src0SubRC);
9609 AMDGPU::sub0, Src1SubRC);
9611 AMDGPU::sub1, Src0SubRC);
9613 AMDGPU::sub1, Src1SubRC);
9618 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9621 MachineInstr &LoHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub0)
9626 MachineInstr &HiHalf = *
BuildMI(
MBB, MII,
DL, InstDesc, DestSub1)
9639 Worklist.
insert(&LoHalf);
9640 Worklist.
insert(&HiHalf);
9643 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9663 MachineOperand* Op0;
9664 MachineOperand* Op1;
9666 if (
Src0.isReg() && RI.isSGPRReg(MRI,
Src0.getReg())) {
9699 const MCInstrDesc &InstDesc =
get(AMDGPU::V_BCNT_U32_B32_e64);
9702 &AMDGPU::SGPR_32RegClass;
9708 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9711 AMDGPU::sub0, SrcSubRC);
9713 AMDGPU::sub1, SrcSubRC);
9723 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9742 Offset == 0 &&
"Not implemented");
9765 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9775 .
addReg(Src.getReg(), {}, AMDGPU::sub0);
9778 .
addReg(Src.getReg(), {}, AMDGPU::sub0)
9784 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9803 const MCInstrDesc &InstDesc =
get(Opcode);
9805 bool IsCtlz = Opcode == AMDGPU::V_FFBH_U32_e32;
9808 Src.isReg() ? MRI.
getRegClass(Src.getReg()) : &AMDGPU::SGPR_32RegClass;
9810 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9812 MachineOperand SrcRegSub0 =
9814 MachineOperand SrcRegSub1 =
9828 .
addReg(IsCtlz ? MidReg1 : MidReg2);
9832 .
addReg(IsCtlz ? MidReg2 : MidReg1);
9836 addUsersToMoveToVALUWorklist(MidReg4, MRI, Worklist);
9839void SIInstrInfo::addUsersToMoveToVALUWorklist(
9843 MachineInstr &
UseMI = *MO.getParent();
9847 switch (
UseMI.getOpcode()) {
9850 case AMDGPU::SOFT_WQM:
9851 case AMDGPU::STRICT_WWM:
9852 case AMDGPU::STRICT_WQM:
9853 case AMDGPU::REG_SEQUENCE:
9855 case AMDGPU::INSERT_SUBREG:
9858 OpNo = MO.getOperandNo();
9865 if (!RI.hasVectorRegisters(OpRC))
9882 if (ST.useRealTrue16Insts()) {
9884 if (!
Src0.isReg() || !RI.isVGPR(MRI,
Src0.getReg())) {
9887 get(
Src0.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg0)
9890 SrcReg0 =
Src0.getReg();
9893 if (!
Src1.isReg() || !RI.isVGPR(MRI,
Src1.getReg())) {
9896 get(
Src1.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg1)
9899 SrcReg1 =
Src1.getReg();
9905 auto NewMI =
BuildMI(*
MBB, Inst,
DL,
get(AMDGPU::REG_SEQUENCE), ResultReg);
9907 case AMDGPU::S_PACK_LL_B32_B16:
9909 .addReg(SrcReg0, {},
9910 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9911 .addImm(AMDGPU::lo16)
9912 .addReg(SrcReg1, {},
9913 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9914 .addImm(AMDGPU::hi16);
9916 case AMDGPU::S_PACK_LH_B32_B16:
9918 .addReg(SrcReg0, {},
9919 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9920 .addImm(AMDGPU::lo16)
9921 .addReg(SrcReg1, {}, AMDGPU::hi16)
9922 .addImm(AMDGPU::hi16);
9924 case AMDGPU::S_PACK_HL_B32_B16:
9925 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9926 .addImm(AMDGPU::lo16)
9927 .addReg(SrcReg1, {},
9928 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9929 .addImm(AMDGPU::hi16);
9931 case AMDGPU::S_PACK_HH_B32_B16:
9932 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9933 .addImm(AMDGPU::lo16)
9934 .addReg(SrcReg1, {}, AMDGPU::hi16)
9935 .addImm(AMDGPU::hi16);
9943 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9948 case AMDGPU::S_PACK_LL_B32_B16: {
9967 case AMDGPU::S_PACK_LH_B32_B16: {
9977 case AMDGPU::S_PACK_HL_B32_B16: {
9988 case AMDGPU::S_PACK_HH_B32_B16: {
10008 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
10017 assert(
Op.isReg() &&
Op.getReg() == AMDGPU::SCC &&
Op.isDef() &&
10018 !
Op.isDead() &&
Op.getParent() == &SCCDefInst);
10019 SmallVector<MachineInstr *, 4> CopyToDelete;
10022 for (MachineInstr &
MI :
10026 int SCCIdx =
MI.findRegisterUseOperandIdx(AMDGPU::SCC, &RI,
false);
10027 if (SCCIdx != -1) {
10030 Register DestReg =
MI.getOperand(0).getReg();
10037 MI.getOperand(SCCIdx).setReg(NewCond);
10043 if (
MI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI,
false,
false) != -1)
10046 for (
auto &Copy : CopyToDelete)
10047 Copy->eraseFromParent();
10055void SIInstrInfo::addSCCDefsToVALUWorklist(
MachineInstr *SCCUseInst,
10061 for (MachineInstr &
MI :
10064 if (
MI.modifiesRegister(AMDGPU::VCC, &RI))
10066 if (
MI.definesRegister(AMDGPU::SCC, &RI)) {
10083 case AMDGPU::REG_SEQUENCE:
10084 case AMDGPU::INSERT_SUBREG:
10086 case AMDGPU::SOFT_WQM:
10087 case AMDGPU::STRICT_WWM:
10088 case AMDGPU::STRICT_WQM: {
10090 if (RI.isAGPRClass(SrcRC)) {
10091 if (RI.isAGPRClass(NewDstRC))
10096 case AMDGPU::REG_SEQUENCE:
10097 case AMDGPU::INSERT_SUBREG:
10098 NewDstRC = RI.getEquivalentAGPRClass(NewDstRC);
10101 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10107 if (!RI.isSGPRClass(NewDstRC) || NewDstRC == &AMDGPU::VReg_1RegClass)
10110 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10124 int OpIndices[3])
const {
10125 const MCInstrDesc &
Desc =
MI.getDesc();
10143 for (
unsigned i = 0; i < 3; ++i) {
10144 int Idx = OpIndices[i];
10148 const MachineOperand &MO =
MI.getOperand(Idx);
10155 RI.getRegClass(getOpRegClassID(
Desc.operands()[Idx]));
10156 bool IsRequiredSGPR = RI.isSGPRClass(OpRC);
10157 if (IsRequiredSGPR)
10163 if (RI.isSGPRClass(RegRC))
10164 UsedSGPRs[i] =
Reg;
10180 if (UsedSGPRs[0]) {
10181 if (UsedSGPRs[0] == UsedSGPRs[1] || UsedSGPRs[0] == UsedSGPRs[2])
10182 SGPRReg = UsedSGPRs[0];
10185 if (!SGPRReg && UsedSGPRs[1]) {
10186 if (UsedSGPRs[1] == UsedSGPRs[2])
10187 SGPRReg = UsedSGPRs[1];
10194 AMDGPU::OpName OperandName)
const {
10195 if (OperandName == AMDGPU::OpName::NUM_OPERAND_NAMES)
10198 int Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), OperandName);
10202 return &
MI.getOperand(Idx);
10216 if (ST.isAmdHsaOS()) {
10219 RsrcDataFormat |= (1ULL << 56);
10224 RsrcDataFormat |= (2ULL << 59);
10227 return RsrcDataFormat;
10237 uint64_t EltSizeValue =
Log2_32(ST.getMaxPrivateElementSize(
true)) - 1;
10242 uint64_t IndexStride = ST.isWave64() ? 3 : 2;
10249 Rsrc23 &=
~AMDGPU::RSRC_DATA_FORMAT;
10255 unsigned Opc =
MI.getOpcode();
10261 return get(
Opc).mayLoad() &&
10268 if (!Addr || !Addr->
isFI())
10277 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::vdata);
10279 return MI.getOperand(VDataIdx).getReg();
10289 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::data);
10291 return MI.getOperand(DataIdx).getReg();
10312 if (!
MI.mayStore())
10325 unsigned Opc =
MI.getOpcode();
10327 unsigned DescSize =
Desc.getSize();
10332 unsigned Size = DescSize;
10336 if (
MI.isBranch() && ST.hasOffset3fBug())
10347 bool HasLiteral =
false;
10348 unsigned LiteralSize = 4;
10349 for (
int I = 0, E =
MI.getNumExplicitOperands();
I != E; ++
I) {
10354 if (ST.has64BitLiterals()) {
10355 switch (OpInfo.OperandType) {
10380 return HasLiteral ? DescSize + LiteralSize : DescSize;
10385 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
10389 int RSrcIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::srsrc);
10390 return 8 + 4 * ((RSrcIdx - VAddr0Idx + 2) / 4);
10394 case TargetOpcode::BUNDLE:
10395 return getInstBundleSize(
MI);
10396 case TargetOpcode::INLINEASM:
10397 case TargetOpcode::INLINEASM_BR: {
10399 const char *AsmStr =
MI.getOperand(0).getSymbolName();
10403 if (
MI.isMetaInstruction())
10407 const auto *D16Info = AMDGPU::getT16D16Helper(
Opc);
10410 unsigned LoInstOpcode = D16Info->LoOp;
10412 DescSize =
Desc.getSize();
10416 if (
Opc == AMDGPU::V_FMA_MIX_F16_t16 ||
Opc == AMDGPU::V_FMA_MIX_BF16_t16) {
10419 DescSize =
Desc.getSize();
10428 if (
MI.isBranch() && ST.hasOffset3fBug())
10429 return InstSizeVerifyMode::NoVerify;
10430 return InstSizeVerifyMode::ExactSize;
10435 static const std::pair<int, const char *> TargetIndices[] = {
10475std::pair<unsigned, unsigned>
10482 static const std::pair<unsigned, const char *> TargetFlags[] = {
10500 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
10516 return AMDGPU::WWM_COPY;
10518 return AMDGPU::COPY;
10535 if (!IsLRSplitInst && Opcode != AMDGPU::IMPLICIT_DEF)
10539 if (RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg)))
10540 return IsLRSplitInst;
10553 bool IsNullOrVectorRegister =
true;
10557 IsNullOrVectorRegister = !RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg));
10560 return IsNullOrVectorRegister &&
10562 (!
MI.isTerminator() &&
MI.getOpcode() != AMDGPU::COPY &&
10563 MI.modifiesRegister(AMDGPU::EXEC, &RI)));
10571 if (ST.hasAddNoCarryInsts())
10587 if (ST.hasAddNoCarryInsts())
10591 Register UnusedCarry = !RS.isRegUsed(AMDGPU::VCC)
10593 : RS.scavengeRegisterBackwards(
10594 *RI.getBoolRC(),
I,
false,
10607 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
10608 case AMDGPU::SI_KILL_I1_TERMINATOR:
10617 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
10618 return get(AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR);
10619 case AMDGPU::SI_KILL_I1_PSEUDO:
10620 return get(AMDGPU::SI_KILL_I1_TERMINATOR);
10632 const unsigned OffsetBits =
10634 return (1 << OffsetBits) - 1;
10638 if (!ST.isWave32())
10641 if (
MI.isInlineAsm())
10644 if (
MI.getNumOperands() <
MI.getDesc().getNumOperands())
10647 for (
auto &
Op :
MI.implicit_operands()) {
10648 if (
Op.isReg() &&
Op.getReg() == AMDGPU::VCC)
10649 Op.setReg(AMDGPU::VCC_LO);
10658 int Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::sbase);
10662 const int16_t RCID = getOpRegClassID(
MI.getDesc().operands()[Idx]);
10663 return RI.getRegClass(RCID)->hasSubClassEq(&AMDGPU::SGPR_128RegClass);
10679 if (
Imm > MaxImm) {
10680 if (
Imm <= MaxImm + 64) {
10682 Overflow =
Imm - MaxImm;
10697 Overflow =
High - Alignment.value();
10701 if (Overflow > 0) {
10709 if (ST.hasRestrictedSOffset())
10714 SOffset = Overflow;
10752 if (!ST.hasFlatInstOffsets())
10756 if (ST.hasFlatSegmentOffsetBug() && FlatVariant == FlatAddrSpace::FLAT &&
10761 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10762 FlatVariant == FlatAddrSpace::FlatScratch &&
Offset < 0 &&
10773std::pair<int64_t, int64_t>
10776 int64_t RemainderOffset = COffsetVal;
10777 int64_t ImmField = 0;
10782 if (AllowNegative) {
10784 int64_t
D = 1LL << NumBits;
10785 RemainderOffset = (COffsetVal /
D) *
D;
10786 ImmField = COffsetVal - RemainderOffset;
10788 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10790 (ImmField % 4) != 0) {
10792 RemainderOffset += ImmField % 4;
10793 ImmField -= ImmField % 4;
10795 }
else if (COffsetVal >= 0) {
10797 RemainderOffset = COffsetVal - ImmField;
10801 assert(RemainderOffset + ImmField == COffsetVal);
10802 return {ImmField, RemainderOffset};
10807 if (ST.hasNegativeScratchOffsetBug() &&
10815 switch (ST.getGeneration()) {
10849 case AMDGPU::V_MOVRELS_B32_dpp_gfx10:
10850 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
10851 case AMDGPU::V_MOVRELD_B32_dpp_gfx10:
10852 case AMDGPU::V_MOVRELD_B32_sdwa_gfx10:
10853 case AMDGPU::V_MOVRELSD_B32_dpp_gfx10:
10854 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
10855 case AMDGPU::V_MOVRELSD_2_B32_dpp_gfx10:
10856 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
10863#define GENERATE_RENAMED_GFX9_CASES(OPCODE) \
10864 case OPCODE##_dpp: \
10865 case OPCODE##_e32: \
10866 case OPCODE##_e64: \
10867 case OPCODE##_e64_dpp: \
10868 case OPCODE##_sdwa:
10882 case AMDGPU::V_DIV_FIXUP_F16_gfx9_e64:
10883 case AMDGPU::V_DIV_FIXUP_F16_gfx9_fake16_e64:
10884 case AMDGPU::V_FMA_F16_gfx9_e64:
10885 case AMDGPU::V_FMA_F16_gfx9_fake16_e64:
10886 case AMDGPU::V_INTERP_P2_F16:
10887 case AMDGPU::V_MAD_F16_e64:
10888 case AMDGPU::V_MAD_U16_e64:
10889 case AMDGPU::V_MAD_I16_e64:
10898 "SIInsertWaitcnts should have promoted soft waitcnt instructions!");
10906 switch (ST.getGeneration()) {
10919 if (
isMAI(Opcode)) {
10933 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX11_7Insts())
10936 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX1250Insts())
10943 if (ST.hasGFX90AInsts()) {
10944 uint32_t NMCOp = AMDGPU::INSTRUCTION_LIST_END;
10945 if (ST.hasGFX940Insts())
10947 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10949 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10951 if (NMCOp != AMDGPU::INSTRUCTION_LIST_END)
10957 if (MCOp == AMDGPU::INSTRUCTION_LIST_END)
10976 for (
unsigned I = 0, E = (
MI.getNumOperands() - 1)/ 2;
I < E; ++
I)
10977 if (
MI.getOperand(1 + 2 *
I + 1).getImm() == SubReg) {
10978 auto &RegOp =
MI.getOperand(1 + 2 *
I);
10990 switch (
MI.getOpcode()) {
10992 case AMDGPU::REG_SEQUENCE:
10996 case AMDGPU::INSERT_SUBREG:
10997 if (RSR.
SubReg == (
unsigned)
MI.getOperand(3).getImm())
11014 if (!
P.Reg.isVirtual())
11019 while (
auto *
MI = DefInst) {
11021 switch (
MI->getOpcode()) {
11023 case AMDGPU::V_MOV_B32_e32: {
11024 auto &Op1 =
MI->getOperand(1);
11053 auto *DefBB =
DefMI.getParent();
11057 if (
UseMI.getParent() != DefBB)
11060 const int MaxInstScan = 20;
11064 auto E =
UseMI.getIterator();
11065 for (
auto I = std::next(
DefMI.getIterator());
I != E; ++
I) {
11066 if (
I->isDebugInstr())
11069 if (++NumInst > MaxInstScan)
11072 if (
I->modifiesRegister(AMDGPU::EXEC,
TRI))
11085 auto *DefBB =
DefMI.getParent();
11087 const int MaxUseScan = 10;
11091 auto &UseInst = *
Use.getParent();
11094 if (UseInst.getParent() != DefBB || UseInst.isPHI())
11097 if (++NumUse > MaxUseScan)
11104 const int MaxInstScan = 20;
11108 for (
auto I = std::next(
DefMI.getIterator()); ; ++
I) {
11111 if (
I->isDebugInstr())
11114 if (++NumInst > MaxInstScan)
11127 if (Reg == VReg && --NumUse == 0)
11129 }
else if (
TRI->regsOverlap(Reg, AMDGPU::EXEC))
11138 auto Cur =
MBB.begin();
11139 if (Cur !=
MBB.end())
11141 if (!Cur->isPHI() && Cur->readsRegister(Dst,
nullptr))
11144 }
while (Cur !=
MBB.end() && Cur != LastPHIIt);
11153 if (InsPt !=
MBB.end() &&
11154 (InsPt->getOpcode() == AMDGPU::SI_IF ||
11155 InsPt->getOpcode() == AMDGPU::SI_ELSE ||
11156 InsPt->getOpcode() == AMDGPU::SI_IF_BREAK) &&
11157 InsPt->definesRegister(Src,
nullptr)) {
11161 .
addReg(Src, {}, SrcSubReg)
11204 if (isFullCopyInstr(
MI)) {
11205 Register DstReg =
MI.getOperand(0).getReg();
11206 Register SrcReg =
MI.getOperand(1).getReg();
11228 unsigned *PredCost)
const {
11229 if (
MI.isBundle()) {
11232 unsigned Lat = 0,
Count = 0;
11233 for (++
I;
I != E &&
I->isBundledWithPred(); ++
I) {
11235 Lat = std::max(Lat, SchedModel.computeInstrLatency(&*
I));
11237 return Lat +
Count - 1;
11240 return SchedModel.computeInstrLatency(&
MI);
11244 if (!ST.hasGFX1250VALUBlockingCycles())
11251 if (
const auto *Entry = AMDGPU::getGFX1250BlockingCyclesInfo(
MI.getOpcode()))
11252 return Entry->GFX1250BlockingCycles;
11260 return *CallAddrOp;
11267 unsigned Opcode =
MI.getOpcode();
11269 auto HandleAddrSpaceCast = [
this, &MRI](
const MachineInstr &
MI) {
11275 unsigned SrcAS = SrcTy.getAddressSpace();
11278 ST.hasGloballyAddressableScratch()
11286 if (Opcode == TargetOpcode::G_ADDRSPACE_CAST)
11287 return HandleAddrSpaceCast(
MI);
11290 auto IID = GI->getIntrinsicID();
11297 case Intrinsic::amdgcn_if:
11298 case Intrinsic::amdgcn_else:
11312 if (Opcode == AMDGPU::G_LOAD || Opcode == AMDGPU::G_ZEXTLOAD ||
11313 Opcode == AMDGPU::G_SEXTLOAD) {
11314 if (
MI.memoperands_empty())
11318 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11319 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11327 if (SIInstrInfo::isGenericAtomicRMWOpcode(Opcode) ||
11328 Opcode == AMDGPU::G_ATOMIC_CMPXCHG ||
11329 Opcode == AMDGPU::G_ATOMIC_CMPXCHG_WITH_SUCCESS ||
11335 if (Opcode == TargetOpcode::G_DYN_STACKALLOC)
11338 if (Opcode == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
11346 Formatter = std::make_unique<AMDGPUMIRFormatter>(ST);
11347 return Formatter.get();
11355 unsigned opcode =
MI.getOpcode();
11356 if (opcode == AMDGPU::V_READLANE_B32 ||
11357 opcode == AMDGPU::V_READFIRSTLANE_B32 ||
11358 opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR)
11363 if (
MI.isInlineAsm()) {
11369 if (!RC || !RI.isSGPRClass(RC))
11374 if (isCopyInstr(
MI)) {
11378 RI.getPhysRegBaseClass(srcOp.
getReg());
11386 if (
MI.isPreISelOpcode())
11401 if (
MI.memoperands_empty())
11405 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11406 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11421 for (
unsigned I = 0, E =
MI.getNumOperands();
I != E; ++
I) {
11423 if (!
SrcOp.isReg())
11427 if (!Reg || !
SrcOp.readsReg())
11433 if (RegBank && RegBank->
getID() != AMDGPU::SGPRRegBankID)
11460 F,
"ds_ordered_count unsupported for this calling conv"));
11474 Register &SrcReg2, int64_t &CmpMask,
11475 int64_t &CmpValue)
const {
11476 if (!
MI.getOperand(0).isReg() ||
MI.getOperand(0).getSubReg())
11479 switch (
MI.getOpcode()) {
11482 case AMDGPU::S_CMP_EQ_U32:
11483 case AMDGPU::S_CMP_EQ_I32:
11484 case AMDGPU::S_CMP_LG_U32:
11485 case AMDGPU::S_CMP_LG_I32:
11486 case AMDGPU::S_CMP_LT_U32:
11487 case AMDGPU::S_CMP_LT_I32:
11488 case AMDGPU::S_CMP_GT_U32:
11489 case AMDGPU::S_CMP_GT_I32:
11490 case AMDGPU::S_CMP_LE_U32:
11491 case AMDGPU::S_CMP_LE_I32:
11492 case AMDGPU::S_CMP_GE_U32:
11493 case AMDGPU::S_CMP_GE_I32:
11494 case AMDGPU::S_CMP_EQ_U64:
11495 case AMDGPU::S_CMP_LG_U64:
11496 SrcReg =
MI.getOperand(0).getReg();
11497 if (
MI.getOperand(1).isReg()) {
11498 if (
MI.getOperand(1).getSubReg())
11500 SrcReg2 =
MI.getOperand(1).getReg();
11502 }
else if (
MI.getOperand(1).isImm()) {
11504 CmpValue =
MI.getOperand(1).getImm();
11510 case AMDGPU::S_CMPK_EQ_U32:
11511 case AMDGPU::S_CMPK_EQ_I32:
11512 case AMDGPU::S_CMPK_LG_U32:
11513 case AMDGPU::S_CMPK_LG_I32:
11514 case AMDGPU::S_CMPK_LT_U32:
11515 case AMDGPU::S_CMPK_LT_I32:
11516 case AMDGPU::S_CMPK_GT_U32:
11517 case AMDGPU::S_CMPK_GT_I32:
11518 case AMDGPU::S_CMPK_LE_U32:
11519 case AMDGPU::S_CMPK_LE_I32:
11520 case AMDGPU::S_CMPK_GE_U32:
11521 case AMDGPU::S_CMPK_GE_I32:
11522 SrcReg =
MI.getOperand(0).getReg();
11524 CmpValue =
MI.getOperand(1).getImm();
11534 if (S->isLiveIn(AMDGPU::SCC))
11543bool SIInstrInfo::invertSCCUse(
MachineInstr *SCCDef)
const {
11546 bool SCCIsDead =
false;
11549 constexpr unsigned ScanLimit = 12;
11550 unsigned Count = 0;
11553 if (++
Count > ScanLimit)
11555 if (
MI.readsRegister(AMDGPU::SCC, &RI)) {
11556 if (
MI.getOpcode() == AMDGPU::S_CSELECT_B32 ||
11557 MI.getOpcode() == AMDGPU::S_CSELECT_B64 ||
11558 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11559 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC1)
11564 if (
MI.definesRegister(AMDGPU::SCC, &RI)) {
11577 for (MachineInstr *
MI : InvertInstr) {
11578 if (
MI->getOpcode() == AMDGPU::S_CSELECT_B32 ||
11579 MI->getOpcode() == AMDGPU::S_CSELECT_B64) {
11581 }
else if (
MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11582 MI->getOpcode() == AMDGPU::S_CBRANCH_SCC1) {
11583 MI->setDesc(
get(
MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0
11584 ? AMDGPU::S_CBRANCH_SCC1
11585 : AMDGPU::S_CBRANCH_SCC0));
11598 bool NeedInversion)
const {
11599 MachineInstr *KillsSCC =
nullptr;
11604 if (
MI.modifiesRegister(AMDGPU::SCC, &RI))
11606 if (
MI.killsRegister(AMDGPU::SCC, &RI))
11609 if (NeedInversion && !invertSCCUse(SCCRedefine))
11611 if (MachineOperand *SccDef =
11613 SccDef->setIsDead(
false);
11622static std::optional<std::pair<int64_t, int64_t>>
11626 if (
Opc != AMDGPU::S_CSELECT_B32 &&
Opc != AMDGPU::S_CSELECT_B64)
11628 std::optional<int64_t>
A =
11632 std::optional<int64_t>
B =
11636 if (
Opc == AMDGPU::S_CSELECT_B32) {
11642 return std::pair(*
A, *
B);
11646 unsigned &NewDefOpc) {
11649 if (Def.getOpcode() != AMDGPU::S_ADD_I32 &&
11650 Def.getOpcode() != AMDGPU::S_ADD_U32)
11656 Def.getMF()->getSubtarget().getInstrInfo());
11658 auto Imm1 =
TII->getImmOrMaterializedImm(MRI, AddSrc1);
11659 auto Imm2 =
TII->getImmOrMaterializedImm(MRI, AddSrc2);
11660 if ((!Imm1 || *Imm1 != 1) && (!Imm2 || *Imm2 != 1))
11663 if (Def.getOpcode() == AMDGPU::S_ADD_I32) {
11665 Def.findRegisterDefOperand(AMDGPU::SCC,
nullptr);
11668 NewDefOpc = AMDGPU::S_ADD_U32;
11670 NeedInversion = !NeedInversion;
11675 Register SrcReg2, int64_t CmpMask,
11685 CmpValue = *ImmOpt;
11688 const auto optimizeCmpSelect = [&CmpInstr, SrcReg, CmpValue, MRI,
11689 this](
bool NeedInversion) ->
bool {
11694 unsigned NewDefOpc = Def->getOpcode();
11701 auto [
A,
B] = *Consts;
11702 int64_t
C = Def->getOpcode() == AMDGPU::S_CSELECT_B32 ?
Lo_32(CmpValue)
11705 NeedInversion = !NeedInversion;
11717 if (CmpValue != 0 ||
11723 if (!optimizeSCC(Def, &CmpInstr, NeedInversion))
11726 if (NewDefOpc != Def->getOpcode())
11727 Def->setDesc(
get(NewDefOpc));
11736 if (Def->getOpcode() == AMDGPU::S_OR_B32 &&
11743 if (Def1 && Def1->
getOpcode() == AMDGPU::COPY && Def2 &&
11752 auto [
A,
B] = *Consts;
11753 if (
A == 0 ||
B == 0)
11754 optimizeSCC(
Select, Def,
A == 0);
11763 const auto optimizeCmpAnd = [&CmpInstr, SrcReg, CmpValue, MRI,
11764 this](int64_t ExpectedValue,
unsigned SrcSize,
11765 bool IsReversible,
bool IsSigned) ->
bool {
11793 if (Def->getOpcode() != AMDGPU::S_AND_B32 &&
11794 Def->getOpcode() != AMDGPU::S_AND_B64)
11798 const auto isMask = [&Mask, SrcSize, MRI,
11810 SrcOp = &Def->getOperand(2);
11811 else if (isMask(&Def->getOperand(2)))
11812 SrcOp = &Def->getOperand(1);
11820 if (IsSigned && BitNo == SrcSize - 1)
11823 ExpectedValue <<= BitNo;
11825 bool IsReversedCC =
false;
11826 if (CmpValue != ExpectedValue) {
11829 IsReversedCC = CmpValue == (ExpectedValue ^ Mask);
11834 Register DefReg = Def->getOperand(0).getReg();
11835 if (IsReversedCC && !MRI->hasOneNonDBGUse(DefReg))
11838 if (!optimizeSCC(Def, &CmpInstr,
false))
11841 if (!MRI->use_nodbg_empty(DefReg)) {
11849 unsigned NewOpc = (SrcSize == 32) ? IsReversedCC ? AMDGPU::S_BITCMP0_B32
11850 : AMDGPU::S_BITCMP1_B32
11851 : IsReversedCC ? AMDGPU::S_BITCMP0_B64
11852 : AMDGPU::S_BITCMP1_B64;
11857 Def->eraseFromParent();
11865 case AMDGPU::S_CMP_EQ_U32:
11866 case AMDGPU::S_CMP_EQ_I32:
11867 case AMDGPU::S_CMPK_EQ_U32:
11868 case AMDGPU::S_CMPK_EQ_I32:
11869 return optimizeCmpAnd(1, 32,
true,
false) ||
11870 optimizeCmpSelect(
true);
11871 case AMDGPU::S_CMP_GE_U32:
11872 case AMDGPU::S_CMPK_GE_U32:
11873 return optimizeCmpAnd(1, 32,
false,
false);
11874 case AMDGPU::S_CMP_GE_I32:
11875 case AMDGPU::S_CMPK_GE_I32:
11876 return optimizeCmpAnd(1, 32,
false,
true);
11877 case AMDGPU::S_CMP_EQ_U64:
11878 return optimizeCmpAnd(1, 64,
true,
false) ||
11879 optimizeCmpSelect(
true);
11880 case AMDGPU::S_CMP_LG_U32:
11881 case AMDGPU::S_CMP_LG_I32:
11882 case AMDGPU::S_CMPK_LG_U32:
11883 case AMDGPU::S_CMPK_LG_I32:
11884 return optimizeCmpAnd(0, 32,
true,
false) ||
11885 optimizeCmpSelect(
false);
11886 case AMDGPU::S_CMP_GT_U32:
11887 case AMDGPU::S_CMPK_GT_U32:
11888 return optimizeCmpAnd(0, 32,
false,
false);
11889 case AMDGPU::S_CMP_GT_I32:
11890 case AMDGPU::S_CMPK_GT_I32:
11891 return optimizeCmpAnd(0, 32,
false,
true);
11892 case AMDGPU::S_CMP_LG_U64:
11893 return optimizeCmpAnd(0, 64,
true,
false) ||
11894 optimizeCmpSelect(
false);
11901 AMDGPU::OpName
OpName)
const {
11902 if (!ST.needsAlignedVGPRs())
11905 int OpNo = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
OpName);
11917 bool IsAGPR = RI.isAGPR(MRI, DataReg);
11919 IsAGPR ? &AMDGPU::AGPR_32RegClass : &AMDGPU::VGPR_32RegClass);
11923 : &AMDGPU::VReg_64_Align2RegClass);
11925 .
addReg(DataReg, {},
Op.getSubReg())
11930 Op.setSubReg(AMDGPU::sub0);
11935 if (!SchedModel.hasInstrSchedModel())
11941 unsigned RepeatRate = 0;
11943 PI = SchedModel.getWriteProcResBegin(SCDesc),
11944 PE = SchedModel.getWriteProcResEnd(SCDesc);
11946 RepeatRate = std::max(RepeatRate, (
unsigned)PI->ReleaseAtCycle);
11963 if (ST.hasGFX1250Insts())
11970 unsigned Opcode =
MI.getOpcode();
11976 Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
11977 Opcode == AMDGPU::V_ACCVGPR_READ_B32_e64)
11980 if (!ST.hasGFX940Insts())
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isUndef(const MachineInstr &MI)
TargetInstrInfo::RegSubRegPair RegSubRegPair
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file declares the machine register scavenger class.
static cl::opt< bool > Fix16BitCopies("amdgpu-fix-16-bit-physreg-copies", cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"), cl::init(true), cl::ReallyHidden)
static void expandSGPRCopy(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const TargetRegisterClass *RC, bool Forward)
static std::optional< std::pair< int64_t, int64_t > > getSelectConstants(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, const MachineInstr &Sel)
If Sel is an S_CSELECT* of two different constants A and B, return them, truncated to the width of th...
static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc)
static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize)
static bool compareMachineOp(const MachineOperand &Op0, const MachineOperand &Op1)
static bool isStride64(unsigned Opc)
static MachineBasicBlock * generateWaterFallLoop(const SIInstrInfo &TII, MachineInstr &MI, ArrayRef< MachineOperand * > ScalarOps, MachineDominatorTree *MDT, MachineBasicBlock::iterator Begin=nullptr, MachineBasicBlock::iterator End=nullptr, ArrayRef< Register > PhySGPRs={})
#define GENERATE_RENAMED_GFX9_CASES(OPCODE)
static std::tuple< unsigned, unsigned > extractRsrcPtr(const SIInstrInfo &TII, MachineInstr &MI, MachineOperand &Rsrc)
static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx)
static bool followSubRegDef(MachineInstr &MI, TargetInstrInfo::RegSubRegPair &RSR)
static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize)
static MachineInstr * swapImmOperands(MachineInstr &MI, MachineOperand &NonRegOp1, MachineOperand &NonRegOp2)
static void copyFlagsToImplicitVCC(MachineInstr &MI, const MachineOperand &Orig)
static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA, LocationSize WidthB, int OffsetB)
static void indirectCopyToAGPR(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, RegScavenger &RS, bool RegsOverlap, Register ImpUseSuperReg=Register())
Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908.
static unsigned getWWMRegSpillSaveOpcode(unsigned Size, bool IsVectorSuperClass)
static bool memOpsHaveSameBaseOperands(ArrayRef< const MachineOperand * > BaseOps1, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getWWMRegSpillRestoreOpcode(unsigned Size, bool IsVectorSuperClass)
static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion, unsigned &NewDefOpc)
static bool isSCCDeadOnExit(MachineBasicBlock *MBB)
static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize)
static unsigned subtargetEncodingFamily(const GCNSubtarget &ST)
static void preserveCondRegFlags(MachineOperand &CondReg, const MachineOperand &OrigCond)
static Register findImplicitSGPRRead(const MachineInstr &MI)
static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc)
static cl::opt< unsigned > BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16), cl::desc("Restrict range of branch instructions (DEBUG)"))
static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool memOpsHaveSameBasePtr(const MachineInstr &MI1, ArrayRef< const MachineOperand * > BaseOps1, const MachineInstr &MI2, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getSGPRSpillRestoreOpcode(unsigned Size)
static bool isRegOrFI(const MachineOperand &MO)
static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI)
Return true if MI is a VALU comparison, i.e.
static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static constexpr AMDGPU::OpName ModifierOpNames[]
static void reportIllegalCopy(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const char *Msg="illegal VGPR to SGPR copy")
static MachineInstr * swapRegAndNonRegOperand(MachineInstr &MI, MachineOperand &RegOp, MachineOperand &NonRegOp)
static bool shouldReadExec(const MachineInstr &MI)
static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc)
static bool isRenamedInGFX9(int Opcode)
static TargetInstrInfo::RegSubRegPair getRegOrUndef(const MachineOperand &RegOpnd)
static std::tuple< unsigned, unsigned, unsigned > splitGlobalAddressRelocFlags(const GCNSubtarget &ST, const MachineOperand &SrcOp)
static bool changesVGPRIndexingMode(const MachineInstr &MI)
static bool isSubRegOf(const SIRegisterInfo &TRI, const MachineOperand &SuperVec, const MachineOperand &SubReg)
static bool nodesHaveSameOperandValue(SDNode *N0, SDNode *N1, AMDGPU::OpName OpName)
Returns true if both nodes have the same value for the given operand Op, or if both nodes do not have...
static unsigned getNumOperandsNoGlue(SDNode *Node)
static bool canRemat(const MachineInstr &MI)
static unsigned getAVSpillRestoreOpcode(unsigned Size)
static void emitLoadScalarOpsFromVGPRLoop(const SIInstrInfo &TII, MachineRegisterInfo &MRI, MachineBasicBlock &PredBB, MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={})
static unsigned getVGPRSpillRestoreOpcode(unsigned Size)
Interface definition for SIInstrInfo.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
const unsigned AndN2WrExecOpc
static const LaneMaskConstants & get(const GCNSubtarget &ST)
const unsigned XorTermOpc
const unsigned MovTermOpc
const unsigned OrSaveExecOpc
const unsigned AndSaveExecOpc
static LLVM_ABI Semantics SemanticsToEnum(const llvm::fltSemantics &Sem)
Class for arbitrary precision integers.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
This class is the base class for the comparison instructions.
uint64_t getZExtValue() const
Opaque handle to a cycle within a GenericCycleInfo that wraps the cycle's preorder index.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Diagnostic information for unsupported feature in backend.
void changeImmediateDominator(DomTreeNodeBase< NodeT > *N, DomTreeNodeBase< NodeT > *NewIDom)
changeImmediateDominator - This method is used to update the dominator tree information when a node's...
DomTreeNodeBase< NodeT > * addNewBlock(NodeT *BB, NodeT *DomBB)
Add a new node to the dominator tree information.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void getExitingBlocks(CycleRef C, SmallVectorImpl< BlockT * > &TmpStorage) const
Return all blocks of C that have a successor outside of C.
CycleRef getParentCycle(CycleRef C) const
bool contains(CycleRef Outer, CycleRef Inner) const
Returns true iff Outer contains Inner. O(1). Non-strict.
CycleRef getCycle(const BlockT *Block) const
Find the innermost cycle containing Block.
Itinerary data supplied by a subtarget to be used by a target.
constexpr unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasInterval(Register Reg) const
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
This class represents the liveness of a register, stack slot, etc.
static LocationSize precise(uint64_t Value)
TypeSize getValue() const
static const MCBinaryExpr * createAnd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createAShr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
unsigned getSize() const
Return the number of bytes in the encoding of this instruction, or zero if the encoding size cannot b...
ArrayRef< MCPhysReg > implicit_uses() const
Return a list of registers that are potentially read by any instance of this machine instruction.
unsigned getOpcode() const
Return the opcode number for this descriptor.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
int16_t RegClass
This specifies the register class enumeration of the operand if the operand is a register.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Wrapper class representing physical registers. Should be passed by value.
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
LLVM_ABI void setVariableValue(const MCExpr *Value)
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
MIBundleBuilder & append(MachineInstr *MI)
Insert MI into MBB by appending it to the instructions in the bundle.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI MCSymbol * getSymbol() const
Return the MCSymbol for this basic block.
void push_back(MachineInstr *MI)
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
Instructions::const_iterator const_instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool isImmutableObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to an immutable object.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & copyImplicitOps(const MachineInstr &OtherMI) const
Copy all the implicit operands from OtherMI onto this one.
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
const MachineBasicBlock * getParent() const
LLVM_ABI void addImplicitDefUseOperands(MachineFunction &MF)
Add all implicit def and use operands to this instruction.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
bool modifiesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr modifies (fully define or partially define) the specified register.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
LLVM_ABI bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by mayLoad / mayStore,...
void untieRegOperand(unsigned OpIdx)
Break any tie involving OpIdx.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
LLVM_ABI void eraseFromBundle()
Unlink 'this' from its basic block and delete it.
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mop_range explicit_operands()
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
LLVM_ABI bool hasOrderedMemoryRef() const
Return true if this instruction may have an ordered or volatile memory reference, or if the informati...
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
ArrayRef< MachineMemOperand * > memoperands() const
Access to memory operands of the instruction.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
filtered_mop_range all_uses()
Returns an iterator range over all operands that are (explicit or implicit) register uses.
LLVM_ABI void setPostInstrSymbol(MachineFunction &MF, MCSymbol *Symbol)
Set a symbol that will be emitted just after the instruction itself.
LLVM_ABI void clearRegisterKills(Register Reg, const TargetRegisterInfo *RegInfo)
Clear all kill flags affecting Reg.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI int findRegisterDefOperandIdx(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false) const
Returns the operand index that is a def of the specified register or -1 if it is not found.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand * findRegisterDefOperand(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false)
Wrapper for findRegisterDefOperandIdx, it returns a pointer to the MachineOperand rather than an inde...
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
const GlobalValue * getGlobal() const
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
void setOffset(int64_t Offset)
unsigned getTargetFlags() const
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isTargetIndex() const
isTargetIndex - Tests if this is a MO_TargetIndex operand.
void setTargetFlags(unsigned F)
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_Register
Register operand.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
int64_t getOffset() const
Return the offset from the symbol in this operand.
bool isFPImm() const
isFPImm - Tests if this is a MO_FPImmediate operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI void moveOperands(MachineOperand *Dst, MachineOperand *Src, unsigned NumOps)
Move NumOps operands from Src to Dst, updating use-def lists as needed.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool reservedRegsFrozen() const
reservedRegsFrozen - Returns true after freezeReservedRegs() was called to ensure the set of reserved...
LLVM_ABI void clearVirtRegs()
clearVirtRegs - Remove all virtual registers (after physreg assignment).
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
const MachineFunction & getMF() const
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI bool isConstantPhysReg(MCRegister PhysReg) const
Returns true if PhysReg is unallocatable and constant throughout the function.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
iterator_range< use_iterator > use_operands(Register Reg) const
LLVM_ABI void removeRegOperandFromUseList(MachineOperand *MO)
Remove MO from its use-def list.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
LLVM_ABI void addRegOperandToUseList(MachineOperand *MO)
Add MO to the linked list of operands for its register.
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isLegalMUBUFImmOffset(unsigned Imm) const
bool isInlineConstant(const APInt &Imm) const
void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const
Fix operands in MI to satisfy constant bus requirements.
bool canAddToBBProlog(const MachineInstr &MI) const
static bool isDS(const MachineInstr &MI)
MachineBasicBlock * legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT=nullptr) const
Legalize all operands in this instruction.
bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0, int64_t &Offset1) const override
unsigned getLiveRangeSplitOpcode(Register Reg, const MachineFunction &MF) const override
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
static bool isNeverUniform(const MachineInstr &MI)
bool isXDLWMMA(const MachineInstr &MI) const
bool isBasicBlockPrologue(const MachineInstr &MI, Register Reg=Register()) const override
uint64_t getDefaultRsrcDataFormat() const
static bool isSOPP(const MachineInstr &MI)
bool mayAccessScratch(const MachineInstr &MI) const
bool isIGLP(unsigned Opcode) const
static bool isFLATScratch(const MachineInstr &MI)
bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Returns if Offset is legal for the subtarget as the offset to a FLAT encoded instruction with the giv...
const MCInstrDesc & getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize, bool IsSGPR) const
MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg) const
Return a partially built integer add instruction without carry.
bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0, int64_t Offset1, unsigned NumLoads) const override
bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset, Align Alignment=Align(4)) const
bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const
Replace the instructions opcode with the equivalent VALU opcode.
static bool isSMRD(const MachineInstr &MI)
unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const
GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, SlotIndexes *Indexes=nullptr) const
void storeRegToStackSlotCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC) const
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineOperand &MO, const MCOperandInfo &OpInfo) const
Returns true if this operand uses the constant bus.
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getFoldableCopySrcIdx(const MachineInstr &MI)
unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const
Return the size in bytes of the operand OpNo on the given.
void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const final
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
static bool isMTBUF(const MachineInstr &MI)
const MCInstrDesc & getIndirectGPRIDXPseudo(unsigned VecSize, bool IsIndirectSrc) const
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool setsSCCIfResultIsNonZero(const MachineInstr &MI)
const MIRFormatter * getMIRFormatter() const override
static bool isXcntDrain(const MachineInstr &MI)
True if MI implicitly drains XCNT.
void legalizeGenericOperand(MachineBasicBlock &InsertMBB, MachineBasicBlock::iterator I, const TargetRegisterClass *DstRC, MachineOperand &Op, MachineRegisterInfo &MRI, const DebugLoc &DL) const
MachineInstr * buildShrunkInst(MachineInstr &MI, unsigned NewOpcode) const
static bool isVOP2(const MachineInstr &MI)
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
static bool isSDWA(const MachineInstr &MI)
const MCInstrDesc & getKillTerminatorFromPseudo(unsigned Opcode) const
void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, unsigned Quantity) const override
static bool isGather4(const MachineInstr &MI)
MachineInstr * getWholeWaveFunctionSetup(MachineFunction &MF) const
static bool isDOT(const MachineInstr &MI)
std::unique_ptr< PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
InstSizeVerifyMode getInstSizeVerifyMode(const MachineInstr &MI) const override
MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const override
bool hasModifiers(unsigned Opcode) const
Return true if this instruction has any modifiers.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
static bool isSWMMAC(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
bool isHighLatencyDef(int Opc) const override
void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const
Legalize the OpIndex operand of this instruction by inserting a MOV.
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
static bool isVOPC(const MachineInstr &MI)
void removeModOperands(MachineInstr &MI) const
unsigned getRepeatRate(const MachineInstr &MI) const
Get the repeat rate for a VALU instruction from the scheduling model.
unsigned getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI) const
bool isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN, const MachineOperand *MO=nullptr) const
Check if MO would be a legal operand for a single-SGPR-read instruction.
bool isXDL(const MachineInstr &MI) const
Register isStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
static bool isVIMAGE(const MachineInstr &MI)
void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const
static bool isSOP2(const MachineInstr &MI)
static bool isGWS(const MachineInstr &MI)
bool hasRAWDependency(const MachineInstr &FirstMI, const MachineInstr &SecondMI) const
bool isLegalAV64PseudoImm(uint64_t Imm) const
Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
bool isNeverCoissue(MachineInstr &MI) const
static bool isBUF(const MachineInstr &MI)
bool isNonCommutableDPP(const MachineInstr &MI) const
void handleCopyToPhysHelper(SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst, MachineRegisterInfo &MRI, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const
bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx, unsigned toIdx) const
static bool isFLATGlobal(const MachineInstr &MI)
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
bool isGlobalMemoryObject(const MachineInstr *MI) const override
static bool isVSAMPLE(const MachineInstr &MI)
bool isBufferSMRD(const MachineInstr &MI) const
static bool isKillTerminator(unsigned Opcode)
bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const
If OpX is multicycle, anti-dependencies are not allowed.
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, bool IsSCCLive, SlotIndexes *Indexes=nullptr) const
bool hasVALU32BitEncoding(unsigned Opcode) const
Return true if this 64-bit VALU instruction has a 32-bit encoding.
unsigned getBlockingCycles(const MachineInstr &MI) const
unsigned getMovOpcode(const TargetRegisterClass *DstRC) const
Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
unsigned buildExtractSubReg(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const
Legalize operands in MI by either commuting it or inserting a copy of src1.
static bool isVPermPk16(unsigned Opcode)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const final
static bool isTRANS(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isSOPK(const MachineInstr &MI)
const TargetRegisterClass * getOpRegClass(const MachineInstr &MI, unsigned OpNo) const
Return the correct register class for OpNo.
MachineBasicBlock * insertSimulatedTrap(MachineRegisterInfo &MRI, MachineBasicBlock &MBB, MachineInstr &MI, const DebugLoc &DL) const
Build instructions that simulate the behavior of a s_trap 2 instructions for hardware (namely,...
static unsigned getNonSoftWaitcntOpcode(unsigned Opcode)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static bool isFoldableCopy(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
bool expandPostRAPseudo(MachineInstr &MI) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={}) const
Wrapper function for generating waterfall for instruction MI This function take into consideration of...
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isSegmentSpecificFLAT(const MachineInstr &MI)
bool isReMaterializableImpl(const MachineInstr &MI) const override
static bool isVOP3(const MCInstrDesc &Desc)
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool physRegUsesConstantBus(const MachineOperand &Reg) const
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const
static bool isDPP(const MachineInstr &MI)
bool analyzeBranchImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const
static bool isMFMA(const MachineInstr &MI)
bool isLowLatencyInstruction(const MachineInstr &MI) const
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is a instruction that moves/copies value from one register to ano...
void mutateAndCleanupImplicit(MachineInstr &MI, const MCInstrDesc &NewDesc) const
ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const
static bool isMAI(const MCInstrDesc &Desc)
static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode)
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
static bool usesLGKM_CNT(const MachineInstr &MI)
void legalizeOperandsVALUt16(MachineInstr &Inst, MachineRegisterInfo &MRI) const
Fix operands in Inst to fix 16bit SALU to VALU lowering.
bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo, const MachineOperand &MO) const
bool canShrink(const MachineInstr &MI, const MachineRegisterInfo &MRI) const
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool isAsmOnlyOpcode(int MCOp) const
Check if this instruction should only be used by assembler.
bool isAlwaysGDS(uint32_t Opcode) const
static bool isVGPRSpill(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
This is used by the post-RA scheduler (SchedulePostRAList.cpp).
bool verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const override
unsigned getInstrLatency(const InstrItineraryData *ItinData, const MachineInstr &MI, unsigned *PredCost=nullptr) const override
unsigned getVectorRegSpillSaveOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI, bool NeedsCFI) const
int64_t getNamedImmOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
Get required immediate operand.
ArrayRef< std::pair< int, const char * > > getSerializableTargetIndices() const override
bool regUsesConstantBus(const MachineOperand &Reg, const MachineRegisterInfo &MRI) const
static bool isMIMG(const MachineInstr &MI)
MachineOperand buildExtractSubRegOrImm(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
bool isLegalRegOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO (a register operand) is a legal register for the given operand description or operand ind...
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
unsigned getVALUOp(const MachineInstr &MI) const
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI, MachineRegisterInfo &MRI, const TargetRegisterClass *DstRC=nullptr) const
Copy a value from a VGPR (SrcReg) to SGPR.
bool hasDivergentBranch(const MachineBasicBlock *MBB) const
Return whether the block terminate with divergent branch.
std::pair< int64_t, int64_t > splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Split COffsetVal into {immediate offset field, remainder offset} values.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
void fixImplicitOperands(MachineInstr &MI) const
bool moveFlatAddrToVGPR(MachineInstr &Inst) const
Change SADDR form of a FLAT Inst to its VADDR form if saddr operand was moved to VGPR.
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool isMaskedByExec(Register Reg, const MachineInstr &Use, const MachineRegisterInfo &MRI, unsigned Depth=0) const
Return true if Reg is a lane mask that already has 0 in every bit corresponding to a lane that is ina...
void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const
bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0, AMDGPU::OpName Src0OpName, MachineOperand &Src1, AMDGPU::OpName Src1OpName) const
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const
This function is used to determine if an instruction can be safely executed under EXEC = 0 without ha...
bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg, int64_t &ImmVal) const override
static bool isAtomic(const MachineInstr &MI)
bool canInsertSelect(const MachineBasicBlock &MBB, ArrayRef< MachineOperand > Cond, Register DstReg, Register TrueReg, Register FalseReg, int &CondCycles, int &TrueCycles, int &FalseCycles) const override
bool isLiteralOperandLegal(const MCInstrDesc &InstDesc, const MCOperandInfo &OpInfo) const
static bool isWWMRegSpillOpcode(uint32_t Opcode)
static bool sopkIsZext(unsigned Opcode)
static bool isSGPRSpill(const MachineInstr &MI)
static bool isWMMA(const MachineInstr &MI)
ArrayRef< std::pair< MachineMemOperand::Flags, const char * > > getSerializableMachineMemOperandTargetFlags() const override
bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const
Returns true if the instruction could potentially depend on the value of exec.
void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
std::pair< MachineInstr *, MachineInstr * > expandMovDPP64(MachineInstr &MI) const
static bool isSOPC(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
bool isBarrier(unsigned Opcode) const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx0, unsigned OpIdx1) const override
bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const
int pseudoToMCOpcode(int Opcode) const
Return a target-specific opcode if Opcode is a pseudo instruction.
const MCInstrDesc & getMCOpcodeFromPseudo(unsigned Opcode) const
Return the descriptor of the target-specific machine instruction that corresponds to the specified ps...
static bool usesVM_CNT(const MachineInstr &MI)
MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const override
static bool isFixedSize(const MachineInstr &MI)
bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo, MachineCycleInfo *CI) const override
LLVM_READONLY int commuteOpcode(unsigned Opc) const
ValueUniformity getValueUniformity(const MachineInstr &MI) const final
uint64_t getScratchRsrcWords23() const
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
MachineInstr * convertToThreeAddress(MachineInstr &MI, LiveIntervals *LIS) const override
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx, const MachineOperand *MO=nullptr) const
Check if MO is a legal operand if it was the OpIdx Operand for MI.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const
Returns true if negative offsets are allowed for the given FlatVariant.
void moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
static bool isLDSDMA(const MachineInstr &MI)
static bool isVOP1(const MachineInstr &MI)
SIInstrInfo(const GCNSubtarget &ST)
std::optional< int64_t > getImmOrMaterializedImm(const MachineRegisterInfo &MRI, const MachineOperand &Op, MachineInstr **DefMI=nullptr) const
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
bool hasAnyModifiersSet(const MachineInstr &MI) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getLongBranchReservedReg() const
bool isWholeWaveFunction() const
Register getStackPtrOffsetReg() const
unsigned getMaxMemoryClusterDWords() const
void setHasSpilledVGPRs(bool Spill=true)
bool isWWMReg(Register Reg) const
bool checkFlag(Register Reg, uint8_t Flag) const
void setHasSpilledSGPRs(bool Spill=true)
unsigned getScratchReservedForDynamicVGPRs() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
unsigned getHWRegIndex(MCRegister Reg) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
unsigned getChannelFromSubReg(unsigned SubReg) const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
MachineFunction & MF
Machine function.
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
SlotIndex - An opaque wrapper around machine indexes.
SlotIndex getRegSlot(bool EC=false) const
Returns the register use/def slot in the current instruction for a normal or early-clobber def.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Object returned by analyzeLoopForPipelining.
virtual ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *, const ScheduleDAGMI *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual const MachineOperand & getCalleeOperand(const MachineInstr &MI) const
Returns the callee operand from the given MI.
virtual void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const
Re-issue the specified 'original' instruction at the specific location targeting a new destination re...
virtual MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const
This method commutes the operands of the given machine instruction MI.
virtual bool isGlobalMemoryObject(const MachineInstr *MI) const
Returns true if MI is an instruction we are unable to reason about (like a call or something with unm...
virtual bool expandPostRAPseudo(MachineInstr &MI) const
This function is called for all pseudo instructions that remain after register allocation.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
const MCWriteProcResEntry * ProcResIter
static constexpr TypeSize getFixed(ScalarTy ExactSize)
A Use represents the edge between a Value definition and its users.
std::pair< iterator, bool > insert(const ValueT &V)
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
const uint64_t RSRC_DATA_FORMAT
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool getWMMAIsXDL(unsigned Opc)
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_READONLY int32_t getCommuteRev(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteOrig(uint32_t Opcode)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool isGFX12Plus(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_READONLY int32_t getGlobalVaddrOp(uint32_t Opcode)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
bool getMAIIsGFX940XDL(unsigned Opc)
const uint64_t RSRC_ELEMENT_SIZE_SHIFT
bool isIntrinsicAlwaysUniform(unsigned IntrID)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getIfAddr64Inst(uint32_t Opcode)
Check if Opcode is an Addr64 opcode.
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
const uint64_t RSRC_TID_ENABLE
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
bool isGenericAtomic(unsigned Opc)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
LLVM_READONLY int32_t getAddr64Inst(uint32_t Opcode)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_IMM_NOINLINE_FP16
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
@ OPERAND_REG_IMM_V2FP16_SPLAT
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_IMM_V2INT32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_INLINE_C_AV64_PSEUDO
@ OPERAND_REG_INLINE_AC_FP64
@ OPERAND_REG_INLINE_C_FP16
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
LLVM_READONLY int32_t getBasicFromSDWAOp(uint32_t Opcode)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
const uint64_t RSRC_INDEX_STRIDE_SHIFT
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Fast
Attempts to make calls as fast as possible (e.g.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Not(const Pred &P) -> Not< Pred >
constexpr bool isSDWA(const T &...O)
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
constexpr uint64_t maxUIntN(uint64_t N)
Gets the maximum value for a N-bit unsigned integer.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI, unsigned SubReg)
Return the SubReg component from REG_SEQUENCE.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
constexpr bool has_single_bit(T Value) noexcept
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
MachineInstr * getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P, const MachineRegisterInfo &MRI)
Return the defining instruction for a given reg:subreg pair skipping copy like instructions and subre...
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
auto instructionsWithoutDebug(IterT It, IterT End, bool SkipPseudoOp=true)
Construct a range iterator which begins at It and moves forwards until End is reached,...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ABI VirtRegInfo AnalyzeVirtRegInBundle(MachineInstr &MI, Register Reg, SmallVectorImpl< std::pair< MachineInstr *, unsigned > > *Ops=nullptr)
AnalyzeVirtRegInBundle - Analyze how the current instruction or bundle uses a virtual register.
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
@ Xor
Bitwise or logical XOR of integers.
@ Sub
Subtraction of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
bool isTargetSpecificOpcode(unsigned Opcode)
Check whether the given Opcode is a target-specific opcode.
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned DefaultMemoryClusterDWordsLimit
constexpr unsigned BitWidth
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
constexpr T reverseBits(T Val)
Reverse the bits in Val.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
constexpr RegState getUndefRegState(bool B)
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
@ AlwaysUniform
The result value is always uniform.
@ NeverUniform
The result value can never be assumed to be uniform.
@ Default
The result value is uniform if and only if all operands are uniform.
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI)
Return false if EXEC is not changed between the def of VReg at DefMI and all its uses.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Helper struct for the implementation of 3-address conversion to communicate updates made to instructi...
MachineInstr * RemoveMIUse
Other instruction whose def is no longer used by the converted instruction.
uint8_t GFX1250BlockingCycles
static constexpr uint64_t encode(Fields... Values)
This struct is a compact representation of a valid (non-zero power of two) alignment.
constexpr bool all() const
Summarize the scheduling resources required for an instruction of a particular scheduling class.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
Utility to store machine instructions worklist.
MachineInstr * top() const
bool isDeferred(MachineInstr *MI)
SetVector< MachineInstr * > & getDeferredList()
void insert(MachineInstr *MI)
A pair composed of a register and a sub-register index.
VirtRegInfo - Information about a virtual register used by a set of operands.
bool Reads
Reads - One of the operands read the virtual register.
bool Writes
Writes - One of the operands writes the virtual register.