28#define DEBUG_TYPE "amdgpu-si-register-info"
32#define GET_REGINFO_TARGET_DESC
33#include "AMDGPUGenRegisterInfo.inc"
36 "amdgpu-spill-sgpr-to-vgpr",
37 cl::desc(
"Enable spilling SGPRs to VGPRs"),
42 "amdgpu-spill-cfi-saved-regs",
43 cl::desc(
"Enable spilling the registers required for CFI emission"),
48 cl::desc(
"Limit VGPRs to N registers by reserving the rest"));
52 cl::desc(
"Limit AGPRs to N registers by reserving the rest"));
56 cl::desc(
"Limit SGPRs to N registers by reserving the rest"));
58std::array<std::vector<int16_t>, 32> SIRegisterInfo::RegSplitParts;
59std::array<std::array<uint16_t, 32>, 9> SIRegisterInfo::SubRegFromChannelTable;
66 0, 1, 2, 3, 4, 5, 6, 7, 8, 0, 0, 0, 0, 0, 0, 0, 9};
69 const Twine &ErrMsg) {
142 MI->getOperand(0).isKill(),
Index,
RS) {}
157 MovOpc = AMDGPU::S_MOV_B32;
158 NotOpc = AMDGPU::S_NOT_B32;
161 MovOpc = AMDGPU::S_MOV_B64;
162 NotOpc = AMDGPU::S_NOT_B64;
167 SuperReg != AMDGPU::EXEC &&
"exec should never spill");
198 assert(
RS &&
"Cannot spill SGPR to memory without RegScavenger");
199 TmpVGPR =
RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
MI,
false,
227 IsWave32 ? AMDGPU::SGPR_32RegClass : AMDGPU::SGPR_64RegClass;
247 if (
RS->isRegUsed(AMDGPU::SCC))
249 "unhandled SGPR spill to memory");
259 I->getOperand(2).setIsDead();
294 I->getOperand(2).setIsDead();
323 if (
RS->isRegUsed(AMDGPU::SCC))
325 "unhandled SGPR spill to memory");
350 ST.getAMDGPUDwarfFlavour(),
355 assert(getSubRegIndexLaneMask(AMDGPU::sub0).getAsInteger() == 3 &&
356 getSubRegIndexLaneMask(AMDGPU::sub31).getAsInteger() == (3ULL << 62) &&
357 (getSubRegIndexLaneMask(AMDGPU::lo16) |
358 getSubRegIndexLaneMask(AMDGPU::hi16)).getAsInteger() ==
359 getSubRegIndexLaneMask(AMDGPU::sub0).getAsInteger() &&
360 "getNumCoveredRegs() will not work with generated subreg masks!");
362 RegPressureIgnoredUnits.resize(getNumRegUnits());
363 RegPressureIgnoredUnits.set(
365 for (
auto Reg : AMDGPU::VGPR_16RegClass) {
367 RegPressureIgnoredUnits.set(
368 static_cast<unsigned>(*regunits(Reg).begin()));
374 static auto InitializeRegSplitPartsOnce = [
this]() {
375 for (
unsigned Idx = 1, E = getNumSubRegIndices() - 1; Idx < E; ++Idx) {
376 unsigned Size = getSubRegIdxSize(Idx);
379 std::vector<int16_t> &Vec = RegSplitParts[
Size / 16 - 1];
380 unsigned Pos = getSubRegIdxOffset(Idx);
385 unsigned MaxNumParts = 1024 /
Size;
386 Vec.resize(MaxNumParts);
394 static auto InitializeSubRegFromChannelTableOnce = [
this]() {
395 for (
auto &Row : SubRegFromChannelTable)
396 Row.fill(AMDGPU::NoSubRegister);
397 for (
unsigned Idx = 1; Idx < getNumSubRegIndices(); ++Idx) {
398 unsigned Width = getSubRegIdxSize(Idx) / 32;
399 unsigned Offset = getSubRegIdxOffset(Idx) / 32;
404 unsigned TableIdx = Width - 1;
405 assert(TableIdx < SubRegFromChannelTable.size());
407 SubRegFromChannelTable[TableIdx][
Offset] = Idx;
411 llvm::call_once(InitializeRegSplitPartsFlag, InitializeRegSplitPartsOnce);
413 InitializeSubRegFromChannelTableOnce);
430 return ST.hasGFX90AInsts() ? CSR_AMDGPU_GFX90AInsts_SaveList
431 : CSR_AMDGPU_SaveList;
434 return ST.hasGFX90AInsts() ? CSR_AMDGPU_SI_Gfx_GFX90AInsts_SaveList
435 : CSR_AMDGPU_SI_Gfx_SaveList;
437 return CSR_AMDGPU_CS_ChainPreserve_SaveList;
440 static const MCPhysReg NoCalleeSavedReg = AMDGPU::NoRegister;
441 return &NoCalleeSavedReg;
457 return ST.hasGFX90AInsts() ? CSR_AMDGPU_GFX90AInsts_RegMask
458 : CSR_AMDGPU_RegMask;
461 return ST.hasGFX90AInsts() ? CSR_AMDGPU_SI_Gfx_GFX90AInsts_RegMask
462 : CSR_AMDGPU_SI_Gfx_RegMask;
467 return AMDGPU_AllVGPRs_RegMask;
474 return CSR_AMDGPU_NoRegs_RegMask;
478 return VGPR >= AMDGPU::VGPR0 && VGPR < AMDGPU::VGPR8;
489 if (RC == &AMDGPU::VGPR_32RegClass || RC == &AMDGPU::AGPR_32RegClass)
490 return &AMDGPU::AV_32RegClass;
491 if (RC == &AMDGPU::VReg_64RegClass || RC == &AMDGPU::AReg_64RegClass)
492 return &AMDGPU::AV_64RegClass;
493 if (RC == &AMDGPU::VReg_64_Align2RegClass ||
494 RC == &AMDGPU::AReg_64_Align2RegClass)
495 return &AMDGPU::AV_64_Align2RegClass;
496 if (RC == &AMDGPU::VReg_96RegClass || RC == &AMDGPU::AReg_96RegClass)
497 return &AMDGPU::AV_96RegClass;
498 if (RC == &AMDGPU::VReg_96_Align2RegClass ||
499 RC == &AMDGPU::AReg_96_Align2RegClass)
500 return &AMDGPU::AV_96_Align2RegClass;
501 if (RC == &AMDGPU::VReg_128RegClass || RC == &AMDGPU::AReg_128RegClass)
502 return &AMDGPU::AV_128RegClass;
503 if (RC == &AMDGPU::VReg_128_Align2RegClass ||
504 RC == &AMDGPU::AReg_128_Align2RegClass)
505 return &AMDGPU::AV_128_Align2RegClass;
506 if (RC == &AMDGPU::VReg_160RegClass || RC == &AMDGPU::AReg_160RegClass)
507 return &AMDGPU::AV_160RegClass;
508 if (RC == &AMDGPU::VReg_160_Align2RegClass ||
509 RC == &AMDGPU::AReg_160_Align2RegClass)
510 return &AMDGPU::AV_160_Align2RegClass;
511 if (RC == &AMDGPU::VReg_192RegClass || RC == &AMDGPU::AReg_192RegClass)
512 return &AMDGPU::AV_192RegClass;
513 if (RC == &AMDGPU::VReg_192_Align2RegClass ||
514 RC == &AMDGPU::AReg_192_Align2RegClass)
515 return &AMDGPU::AV_192_Align2RegClass;
516 if (RC == &AMDGPU::VReg_256RegClass || RC == &AMDGPU::AReg_256RegClass)
517 return &AMDGPU::AV_256RegClass;
518 if (RC == &AMDGPU::VReg_256_Align2RegClass ||
519 RC == &AMDGPU::AReg_256_Align2RegClass)
520 return &AMDGPU::AV_256_Align2RegClass;
521 if (RC == &AMDGPU::VReg_512RegClass || RC == &AMDGPU::AReg_512RegClass)
522 return &AMDGPU::AV_512RegClass;
523 if (RC == &AMDGPU::VReg_512_Align2RegClass ||
524 RC == &AMDGPU::AReg_512_Align2RegClass)
525 return &AMDGPU::AV_512_Align2RegClass;
526 if (RC == &AMDGPU::VReg_1024RegClass || RC == &AMDGPU::AReg_1024RegClass)
527 return &AMDGPU::AV_1024RegClass;
528 if (RC == &AMDGPU::VReg_1024_Align2RegClass ||
529 RC == &AMDGPU::AReg_1024_Align2RegClass)
530 return &AMDGPU::AV_1024_Align2RegClass;
560 return AMDGPU_AllVGPRs_RegMask;
564 return AMDGPU_AllAGPRs_RegMask;
568 return AMDGPU_AllVectorRegs_RegMask;
575 assert(NumRegIndex &&
"Not implemented");
576 assert(Channel < SubRegFromChannelTable[NumRegIndex - 1].
size());
577 return SubRegFromChannelTable[NumRegIndex - 1][Channel];
586 const unsigned Align,
589 MCRegister BaseReg(AMDGPU::SGPR_32RegClass.getRegister(BaseIdx));
590 return getMatchingSuperReg(BaseReg, AMDGPU::sub0, RC);
608 reserveRegisterTuples(
Reserved, AMDGPU::EXEC);
609 reserveRegisterTuples(
Reserved, AMDGPU::FLAT_SCR);
612 reserveRegisterTuples(
Reserved, AMDGPU::M0);
615 reserveRegisterTuples(
Reserved, AMDGPU::SRC_VCCZ);
616 reserveRegisterTuples(
Reserved, AMDGPU::SRC_EXECZ);
617 reserveRegisterTuples(
Reserved, AMDGPU::SRC_SCC);
620 reserveRegisterTuples(
Reserved, AMDGPU::SRC_SHARED_BASE);
621 reserveRegisterTuples(
Reserved, AMDGPU::SRC_SHARED_LIMIT);
622 reserveRegisterTuples(
Reserved, AMDGPU::SRC_PRIVATE_BASE);
623 reserveRegisterTuples(
Reserved, AMDGPU::SRC_PRIVATE_LIMIT);
624 reserveRegisterTuples(
Reserved, AMDGPU::SRC_FLAT_SCRATCH_BASE_LO);
625 reserveRegisterTuples(
Reserved, AMDGPU::SRC_FLAT_SCRATCH_BASE_HI);
628 reserveRegisterTuples(
Reserved, AMDGPU::ASYNCcnt);
629 reserveRegisterTuples(
Reserved, AMDGPU::TENSORcnt);
632 reserveRegisterTuples(
Reserved, AMDGPU::SRC_POPS_EXITING_WAVE_ID);
635 reserveRegisterTuples(
Reserved, AMDGPU::XNACK_MASK);
638 reserveRegisterTuples(
Reserved, AMDGPU::LDS_DIRECT);
641 reserveRegisterTuples(
Reserved, AMDGPU::TBA);
642 reserveRegisterTuples(
Reserved, AMDGPU::TMA);
643 reserveRegisterTuples(
Reserved, AMDGPU::TTMP0_TTMP1);
644 reserveRegisterTuples(
Reserved, AMDGPU::TTMP2_TTMP3);
645 reserveRegisterTuples(
Reserved, AMDGPU::TTMP4_TTMP5);
646 reserveRegisterTuples(
Reserved, AMDGPU::TTMP6_TTMP7);
647 reserveRegisterTuples(
Reserved, AMDGPU::TTMP8_TTMP9);
648 reserveRegisterTuples(
Reserved, AMDGPU::TTMP10_TTMP11);
649 reserveRegisterTuples(
Reserved, AMDGPU::TTMP12_TTMP13);
650 reserveRegisterTuples(
Reserved, AMDGPU::TTMP14_TTMP15);
653 reserveRegisterTuples(
Reserved, AMDGPU::SGPR_NULL64);
657 unsigned MaxNumSGPRs = ST.getMaxNumSGPRs(MF);
660 unsigned TotalNumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
663 unsigned NumRegs =
divideCeil(getRegSizeInBits(RC), 32);
666 if (Index + NumRegs > MaxNumSGPRs && Index < TotalNumSGPRs &&
667 Reg != AMDGPU::VCC_LO && Reg != AMDGPU::VCC_HI &&
675 if (ScratchRSrcReg != AMDGPU::NoRegister) {
679 reserveRegisterTuples(
Reserved, ScratchRSrcReg);
683 if (LongBranchReservedReg)
684 reserveRegisterTuples(
Reserved, LongBranchReservedReg);
691 reserveRegisterTuples(
Reserved, StackPtrReg);
692 assert(!isSubRegister(ScratchRSrcReg, StackPtrReg));
697 reserveRegisterTuples(
Reserved, FrameReg);
698 assert(!isSubRegister(ScratchRSrcReg, FrameReg));
703 reserveRegisterTuples(
Reserved, BasePtrReg);
704 assert(!isSubRegister(ScratchRSrcReg, BasePtrReg));
711 reserveRegisterTuples(
Reserved, ExecCopyReg);
715 auto [MaxNumVGPRs, MaxNumAGPRs] = ST.getMaxNumVectorRegs(MF.
getFunction());
725 unsigned NumRegs =
divideCeil(getRegSizeInBits(RC), 32);
728 if (Index + NumRegs > MaxNumVGPRs)
735 if (!ST.hasMAIInsts())
739 unsigned NumRegs =
divideCeil(getRegSizeInBits(RC), 32);
742 if (Index + NumRegs > MaxNumAGPRs)
750 if (ST.hasMAIInsts() && !ST.hasGFX90AInsts()) {
758 if (!PerLaneVGPRMask.
empty()) {
759 for (
unsigned RegI = AMDGPU::VGPR0, RegE = AMDGPU::VGPR0 + MaxNumVGPRs;
760 RegI < RegE; ++RegI) {
761 if (PerLaneVGPRMask.
test(RegI))
762 reserveRegisterTuples(
Reserved, RegI);
767 reserveRegisterTuples(
Reserved, Reg);
771 reserveRegisterTuples(
Reserved, Reg);
774 reserveRegisterTuples(
Reserved, Reg);
791 if (Info->isBottomOfStack())
799 if (Info->isEntryFunction()) {
832 int OffIdx = AMDGPU::getNamedOperandIdx(
MI->getOpcode(),
833 AMDGPU::OpName::offset);
834 return MI->getOperand(OffIdx).getImm();
839 switch (
MI->getOpcode()) {
840 case AMDGPU::V_ADD_U32_e32:
841 case AMDGPU::V_ADD_U32_e64:
842 case AMDGPU::V_ADD_CO_U32_e32: {
843 int OtherIdx = Idx == 1 ? 2 : 1;
847 case AMDGPU::V_ADD_CO_U32_e64: {
848 int OtherIdx = Idx == 2 ? 3 : 2;
859 assert((Idx == AMDGPU::getNamedOperandIdx(
MI->getOpcode(),
860 AMDGPU::OpName::vaddr) ||
861 (Idx == AMDGPU::getNamedOperandIdx(
MI->getOpcode(),
862 AMDGPU::OpName::saddr))) &&
863 "Should never see frame index on non-address operand");
875 return Src1.isImm() || (Src1.isReg() &&
TRI.isVGPR(
MI.getMF()->getRegInfo(),
880 return Src0.isImm() || (Src0.isReg() &&
TRI.isVGPR(
MI.getMF()->getRegInfo(),
889 switch (
MI->getOpcode()) {
890 case AMDGPU::V_ADD_U32_e32: {
893 if (ST.getConstantBusLimit(AMDGPU::V_ADD_U32_e32) < 2 &&
898 case AMDGPU::V_ADD_U32_e64:
908 return !ST.hasFlatScratchEnabled();
909 case AMDGPU::V_ADD_CO_U32_e32:
910 if (ST.getConstantBusLimit(AMDGPU::V_ADD_CO_U32_e32) < 2 &&
915 return MI->getOperand(3).isDead();
916 case AMDGPU::V_ADD_CO_U32_e64:
918 return MI->getOperand(1).isDead();
930 return !
TII->isLegalMUBUFImmOffset(FullOffset);
942 if (Ins !=
MBB->end())
943 DL = Ins->getDebugLoc();
949 ST.hasFlatScratchEnabled() ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
952 ST.hasFlatScratchEnabled() ? &AMDGPU::SReg_32_XEXEC_HIRegClass
953 : &AMDGPU::VGPR_32RegClass);
964 ? &AMDGPU::SReg_32_XM0RegClass
965 : &AMDGPU::VGPR_32RegClass);
972 if (ST.hasFlatScratchEnabled()) {
981 TII->getAddNoCarry(*
MBB, Ins,
DL, BaseReg)
993 switch (
MI.getOpcode()) {
994 case AMDGPU::V_ADD_U32_e32:
995 case AMDGPU::V_ADD_CO_U32_e32: {
1001 if (!ImmOp->
isImm()) {
1004 TII->legalizeOperandsVOP2(
MI.getMF()->getRegInfo(),
MI);
1009 if (TotalOffset == 0) {
1010 MI.setDesc(
TII->get(AMDGPU::COPY));
1011 for (
unsigned I =
MI.getNumOperands() - 1;
I != 1; --
I)
1012 MI.removeOperand(
I);
1014 MI.getOperand(1).ChangeToRegister(BaseReg,
false);
1018 ImmOp->
setImm(TotalOffset);
1033 MI.getOperand(2).ChangeToRegister(BaseRegVGPR,
false);
1035 MI.getOperand(2).ChangeToRegister(BaseReg,
false);
1039 case AMDGPU::V_ADD_U32_e64:
1040 case AMDGPU::V_ADD_CO_U32_e64: {
1041 int Src0Idx =
MI.getNumExplicitDefs();
1047 if (!ImmOp->
isImm()) {
1049 TII->legalizeOperandsVOP3(
MI.getMF()->getRegInfo(),
MI);
1054 if (TotalOffset == 0) {
1055 MI.setDesc(
TII->get(AMDGPU::COPY));
1057 for (
unsigned I =
MI.getNumOperands() - 1;
I != 1; --
I)
1058 MI.removeOperand(
I);
1060 MI.getOperand(1).ChangeToRegister(BaseReg,
false);
1063 ImmOp->
setImm(TotalOffset);
1072 bool IsFlat =
TII->isFLATScratch(
MI);
1076 bool SeenFI =
false;
1088 TII->getNamedOperand(
MI, IsFlat ? AMDGPU::OpName::saddr
1089 : AMDGPU::OpName::vaddr);
1094 assert(FIOp && FIOp->
isFI() &&
"frame index must be address operand");
1100 "offset should be legal");
1111 assert(
TII->isLegalMUBUFImmOffset(NewOffset) &&
"offset should be legal");
1121 switch (
MI->getOpcode()) {
1122 case AMDGPU::V_ADD_U32_e32:
1123 case AMDGPU::V_ADD_CO_U32_e32:
1125 case AMDGPU::V_ADD_U32_e64:
1126 case AMDGPU::V_ADD_CO_U32_e64:
1139 return TII->isLegalMUBUFImmOffset(NewOffset);
1147 return RC == &AMDGPU::SCC_CLASSRegClass ? &AMDGPU::SReg_32RegClass : RC;
1153 unsigned Op =
MI.getOpcode();
1155 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
1156 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
1157 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE:
1162 (
uint64_t)
TII->getNamedOperand(
MI, AMDGPU::OpName::mask)->getImm());
1163 case AMDGPU::SI_SPILL_S1024_SAVE:
1164 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
1165 case AMDGPU::SI_SPILL_S1024_RESTORE:
1166 case AMDGPU::SI_SPILL_V1024_SAVE:
1167 case AMDGPU::SI_SPILL_V1024_CFI_SAVE:
1168 case AMDGPU::SI_SPILL_V1024_RESTORE:
1169 case AMDGPU::SI_SPILL_A1024_SAVE:
1170 case AMDGPU::SI_SPILL_A1024_CFI_SAVE:
1171 case AMDGPU::SI_SPILL_A1024_RESTORE:
1172 case AMDGPU::SI_SPILL_AV1024_SAVE:
1173 case AMDGPU::SI_SPILL_AV1024_CFI_SAVE:
1174 case AMDGPU::SI_SPILL_AV1024_RESTORE:
1176 case AMDGPU::SI_SPILL_S512_SAVE:
1177 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
1178 case AMDGPU::SI_SPILL_S512_RESTORE:
1179 case AMDGPU::SI_SPILL_V512_SAVE:
1180 case AMDGPU::SI_SPILL_V512_CFI_SAVE:
1181 case AMDGPU::SI_SPILL_V512_RESTORE:
1182 case AMDGPU::SI_SPILL_A512_SAVE:
1183 case AMDGPU::SI_SPILL_A512_CFI_SAVE:
1184 case AMDGPU::SI_SPILL_A512_RESTORE:
1185 case AMDGPU::SI_SPILL_AV512_SAVE:
1186 case AMDGPU::SI_SPILL_AV512_CFI_SAVE:
1187 case AMDGPU::SI_SPILL_AV512_RESTORE:
1189 case AMDGPU::SI_SPILL_S384_SAVE:
1190 case AMDGPU::SI_SPILL_S384_RESTORE:
1191 case AMDGPU::SI_SPILL_V384_SAVE:
1192 case AMDGPU::SI_SPILL_V384_RESTORE:
1193 case AMDGPU::SI_SPILL_A384_SAVE:
1194 case AMDGPU::SI_SPILL_A384_RESTORE:
1195 case AMDGPU::SI_SPILL_AV384_SAVE:
1196 case AMDGPU::SI_SPILL_AV384_RESTORE:
1198 case AMDGPU::SI_SPILL_S352_SAVE:
1199 case AMDGPU::SI_SPILL_S352_RESTORE:
1200 case AMDGPU::SI_SPILL_V352_SAVE:
1201 case AMDGPU::SI_SPILL_V352_RESTORE:
1202 case AMDGPU::SI_SPILL_A352_SAVE:
1203 case AMDGPU::SI_SPILL_A352_RESTORE:
1204 case AMDGPU::SI_SPILL_AV352_SAVE:
1205 case AMDGPU::SI_SPILL_AV352_RESTORE:
1207 case AMDGPU::SI_SPILL_S320_SAVE:
1208 case AMDGPU::SI_SPILL_S320_RESTORE:
1209 case AMDGPU::SI_SPILL_V320_SAVE:
1210 case AMDGPU::SI_SPILL_V320_RESTORE:
1211 case AMDGPU::SI_SPILL_A320_SAVE:
1212 case AMDGPU::SI_SPILL_A320_RESTORE:
1213 case AMDGPU::SI_SPILL_AV320_SAVE:
1214 case AMDGPU::SI_SPILL_AV320_RESTORE:
1216 case AMDGPU::SI_SPILL_S288_SAVE:
1217 case AMDGPU::SI_SPILL_S288_RESTORE:
1218 case AMDGPU::SI_SPILL_V288_SAVE:
1219 case AMDGPU::SI_SPILL_V288_RESTORE:
1220 case AMDGPU::SI_SPILL_A288_SAVE:
1221 case AMDGPU::SI_SPILL_A288_RESTORE:
1222 case AMDGPU::SI_SPILL_AV288_SAVE:
1223 case AMDGPU::SI_SPILL_AV288_RESTORE:
1225 case AMDGPU::SI_SPILL_S256_SAVE:
1226 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
1227 case AMDGPU::SI_SPILL_S256_RESTORE:
1228 case AMDGPU::SI_SPILL_V256_SAVE:
1229 case AMDGPU::SI_SPILL_V256_CFI_SAVE:
1230 case AMDGPU::SI_SPILL_V256_RESTORE:
1231 case AMDGPU::SI_SPILL_A256_SAVE:
1232 case AMDGPU::SI_SPILL_A256_CFI_SAVE:
1233 case AMDGPU::SI_SPILL_A256_RESTORE:
1234 case AMDGPU::SI_SPILL_AV256_SAVE:
1235 case AMDGPU::SI_SPILL_AV256_CFI_SAVE:
1236 case AMDGPU::SI_SPILL_AV256_RESTORE:
1238 case AMDGPU::SI_SPILL_S224_SAVE:
1239 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
1240 case AMDGPU::SI_SPILL_S224_RESTORE:
1241 case AMDGPU::SI_SPILL_V224_SAVE:
1242 case AMDGPU::SI_SPILL_V224_CFI_SAVE:
1243 case AMDGPU::SI_SPILL_V224_RESTORE:
1244 case AMDGPU::SI_SPILL_A224_SAVE:
1245 case AMDGPU::SI_SPILL_A224_CFI_SAVE:
1246 case AMDGPU::SI_SPILL_A224_RESTORE:
1247 case AMDGPU::SI_SPILL_AV224_SAVE:
1248 case AMDGPU::SI_SPILL_AV224_CFI_SAVE:
1249 case AMDGPU::SI_SPILL_AV224_RESTORE:
1251 case AMDGPU::SI_SPILL_S192_SAVE:
1252 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
1253 case AMDGPU::SI_SPILL_S192_RESTORE:
1254 case AMDGPU::SI_SPILL_V192_SAVE:
1255 case AMDGPU::SI_SPILL_V192_CFI_SAVE:
1256 case AMDGPU::SI_SPILL_V192_RESTORE:
1257 case AMDGPU::SI_SPILL_A192_SAVE:
1258 case AMDGPU::SI_SPILL_A192_CFI_SAVE:
1259 case AMDGPU::SI_SPILL_A192_RESTORE:
1260 case AMDGPU::SI_SPILL_AV192_SAVE:
1261 case AMDGPU::SI_SPILL_AV192_CFI_SAVE:
1262 case AMDGPU::SI_SPILL_AV192_RESTORE:
1264 case AMDGPU::SI_SPILL_S160_SAVE:
1265 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
1266 case AMDGPU::SI_SPILL_S160_RESTORE:
1267 case AMDGPU::SI_SPILL_V160_SAVE:
1268 case AMDGPU::SI_SPILL_V160_CFI_SAVE:
1269 case AMDGPU::SI_SPILL_V160_RESTORE:
1270 case AMDGPU::SI_SPILL_A160_SAVE:
1271 case AMDGPU::SI_SPILL_A160_CFI_SAVE:
1272 case AMDGPU::SI_SPILL_A160_RESTORE:
1273 case AMDGPU::SI_SPILL_AV160_SAVE:
1274 case AMDGPU::SI_SPILL_AV160_CFI_SAVE:
1275 case AMDGPU::SI_SPILL_AV160_RESTORE:
1277 case AMDGPU::SI_SPILL_S128_SAVE:
1278 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
1279 case AMDGPU::SI_SPILL_S128_RESTORE:
1280 case AMDGPU::SI_SPILL_V128_SAVE:
1281 case AMDGPU::SI_SPILL_V128_CFI_SAVE:
1282 case AMDGPU::SI_SPILL_V128_RESTORE:
1283 case AMDGPU::SI_SPILL_A128_SAVE:
1284 case AMDGPU::SI_SPILL_A128_CFI_SAVE:
1285 case AMDGPU::SI_SPILL_A128_RESTORE:
1286 case AMDGPU::SI_SPILL_AV128_SAVE:
1287 case AMDGPU::SI_SPILL_AV128_CFI_SAVE:
1288 case AMDGPU::SI_SPILL_AV128_RESTORE:
1290 case AMDGPU::SI_SPILL_S96_SAVE:
1291 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
1292 case AMDGPU::SI_SPILL_S96_RESTORE:
1293 case AMDGPU::SI_SPILL_V96_SAVE:
1294 case AMDGPU::SI_SPILL_V96_CFI_SAVE:
1295 case AMDGPU::SI_SPILL_V96_RESTORE:
1296 case AMDGPU::SI_SPILL_A96_SAVE:
1297 case AMDGPU::SI_SPILL_A96_CFI_SAVE:
1298 case AMDGPU::SI_SPILL_A96_RESTORE:
1299 case AMDGPU::SI_SPILL_AV96_SAVE:
1300 case AMDGPU::SI_SPILL_AV96_CFI_SAVE:
1301 case AMDGPU::SI_SPILL_AV96_RESTORE:
1303 case AMDGPU::SI_SPILL_S64_SAVE:
1304 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
1305 case AMDGPU::SI_SPILL_S64_RESTORE:
1306 case AMDGPU::SI_SPILL_V64_SAVE:
1307 case AMDGPU::SI_SPILL_V64_CFI_SAVE:
1308 case AMDGPU::SI_SPILL_V64_RESTORE:
1309 case AMDGPU::SI_SPILL_A64_SAVE:
1310 case AMDGPU::SI_SPILL_A64_CFI_SAVE:
1311 case AMDGPU::SI_SPILL_A64_RESTORE:
1312 case AMDGPU::SI_SPILL_AV64_SAVE:
1313 case AMDGPU::SI_SPILL_AV64_CFI_SAVE:
1314 case AMDGPU::SI_SPILL_AV64_RESTORE:
1316 case AMDGPU::SI_SPILL_S32_SAVE:
1317 case AMDGPU::SI_SPILL_S32_CFI_SAVE:
1318 case AMDGPU::SI_SPILL_S32_RESTORE:
1319 case AMDGPU::SI_SPILL_V32_SAVE:
1320 case AMDGPU::SI_SPILL_V32_CFI_SAVE:
1321 case AMDGPU::SI_SPILL_V32_RESTORE:
1322 case AMDGPU::SI_SPILL_A32_SAVE:
1323 case AMDGPU::SI_SPILL_A32_CFI_SAVE:
1324 case AMDGPU::SI_SPILL_A32_RESTORE:
1325 case AMDGPU::SI_SPILL_AV32_SAVE:
1326 case AMDGPU::SI_SPILL_AV32_CFI_SAVE:
1327 case AMDGPU::SI_SPILL_AV32_RESTORE:
1328 case AMDGPU::SI_SPILL_WWM_V32_SAVE:
1329 case AMDGPU::SI_SPILL_WWM_V32_RESTORE:
1330 case AMDGPU::SI_SPILL_WWM_AV32_SAVE:
1331 case AMDGPU::SI_SPILL_WWM_AV32_RESTORE:
1332 case AMDGPU::SI_SPILL_V16_SAVE:
1333 case AMDGPU::SI_SPILL_V16_RESTORE:
1341 case AMDGPU::BUFFER_STORE_DWORD_OFFEN:
1342 return AMDGPU::BUFFER_STORE_DWORD_OFFSET;
1343 case AMDGPU::BUFFER_STORE_BYTE_OFFEN:
1344 return AMDGPU::BUFFER_STORE_BYTE_OFFSET;
1345 case AMDGPU::BUFFER_STORE_SHORT_OFFEN:
1346 return AMDGPU::BUFFER_STORE_SHORT_OFFSET;
1347 case AMDGPU::BUFFER_STORE_DWORDX2_OFFEN:
1348 return AMDGPU::BUFFER_STORE_DWORDX2_OFFSET;
1349 case AMDGPU::BUFFER_STORE_DWORDX3_OFFEN:
1350 return AMDGPU::BUFFER_STORE_DWORDX3_OFFSET;
1351 case AMDGPU::BUFFER_STORE_DWORDX4_OFFEN:
1352 return AMDGPU::BUFFER_STORE_DWORDX4_OFFSET;
1353 case AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFEN:
1354 return AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFSET;
1355 case AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFEN:
1356 return AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFSET;
1364 case AMDGPU::BUFFER_LOAD_DWORD_OFFEN:
1365 return AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
1366 case AMDGPU::BUFFER_LOAD_UBYTE_OFFEN:
1367 return AMDGPU::BUFFER_LOAD_UBYTE_OFFSET;
1368 case AMDGPU::BUFFER_LOAD_SBYTE_OFFEN:
1369 return AMDGPU::BUFFER_LOAD_SBYTE_OFFSET;
1370 case AMDGPU::BUFFER_LOAD_USHORT_OFFEN:
1371 return AMDGPU::BUFFER_LOAD_USHORT_OFFSET;
1372 case AMDGPU::BUFFER_LOAD_SSHORT_OFFEN:
1373 return AMDGPU::BUFFER_LOAD_SSHORT_OFFSET;
1374 case AMDGPU::BUFFER_LOAD_DWORDX2_OFFEN:
1375 return AMDGPU::BUFFER_LOAD_DWORDX2_OFFSET;
1376 case AMDGPU::BUFFER_LOAD_DWORDX3_OFFEN:
1377 return AMDGPU::BUFFER_LOAD_DWORDX3_OFFSET;
1378 case AMDGPU::BUFFER_LOAD_DWORDX4_OFFEN:
1379 return AMDGPU::BUFFER_LOAD_DWORDX4_OFFSET;
1380 case AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFEN:
1381 return AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFSET;
1382 case AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFEN:
1383 return AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFSET;
1384 case AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFEN:
1385 return AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFSET;
1386 case AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFEN:
1387 return AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFSET;
1388 case AMDGPU::BUFFER_LOAD_SHORT_D16_OFFEN:
1389 return AMDGPU::BUFFER_LOAD_SHORT_D16_OFFSET;
1390 case AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFEN:
1391 return AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFSET;
1399 case AMDGPU::BUFFER_STORE_DWORD_OFFSET:
1400 return AMDGPU::BUFFER_STORE_DWORD_OFFEN;
1401 case AMDGPU::BUFFER_STORE_BYTE_OFFSET:
1402 return AMDGPU::BUFFER_STORE_BYTE_OFFEN;
1403 case AMDGPU::BUFFER_STORE_SHORT_OFFSET:
1404 return AMDGPU::BUFFER_STORE_SHORT_OFFEN;
1405 case AMDGPU::BUFFER_STORE_DWORDX2_OFFSET:
1406 return AMDGPU::BUFFER_STORE_DWORDX2_OFFEN;
1407 case AMDGPU::BUFFER_STORE_DWORDX3_OFFSET:
1408 return AMDGPU::BUFFER_STORE_DWORDX3_OFFEN;
1409 case AMDGPU::BUFFER_STORE_DWORDX4_OFFSET:
1410 return AMDGPU::BUFFER_STORE_DWORDX4_OFFEN;
1411 case AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFSET:
1412 return AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFEN;
1413 case AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFSET:
1414 return AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFEN;
1422 case AMDGPU::BUFFER_LOAD_DWORD_OFFSET:
1423 return AMDGPU::BUFFER_LOAD_DWORD_OFFEN;
1424 case AMDGPU::BUFFER_LOAD_UBYTE_OFFSET:
1425 return AMDGPU::BUFFER_LOAD_UBYTE_OFFEN;
1426 case AMDGPU::BUFFER_LOAD_SBYTE_OFFSET:
1427 return AMDGPU::BUFFER_LOAD_SBYTE_OFFEN;
1428 case AMDGPU::BUFFER_LOAD_USHORT_OFFSET:
1429 return AMDGPU::BUFFER_LOAD_USHORT_OFFEN;
1430 case AMDGPU::BUFFER_LOAD_SSHORT_OFFSET:
1431 return AMDGPU::BUFFER_LOAD_SSHORT_OFFEN;
1432 case AMDGPU::BUFFER_LOAD_DWORDX2_OFFSET:
1433 return AMDGPU::BUFFER_LOAD_DWORDX2_OFFEN;
1434 case AMDGPU::BUFFER_LOAD_DWORDX3_OFFSET:
1435 return AMDGPU::BUFFER_LOAD_DWORDX3_OFFEN;
1436 case AMDGPU::BUFFER_LOAD_DWORDX4_OFFSET:
1437 return AMDGPU::BUFFER_LOAD_DWORDX4_OFFEN;
1438 case AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFSET:
1439 return AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFEN;
1440 case AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFSET:
1441 return AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFEN;
1442 case AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFSET:
1443 return AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFEN;
1444 case AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFSET:
1445 return AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFEN;
1446 case AMDGPU::BUFFER_LOAD_SHORT_D16_OFFSET:
1447 return AMDGPU::BUFFER_LOAD_SHORT_D16_OFFEN;
1448 case AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFSET:
1449 return AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFEN;
1458 unsigned ValueReg,
bool IsKill,
bool NeedsCFI) {
1466 if (
Reg == AMDGPU::NoRegister)
1469 bool IsStore =
MI->mayStore();
1473 unsigned Dst = IsStore ?
Reg : ValueReg;
1474 unsigned Src = IsStore ? ValueReg :
Reg;
1475 bool IsVGPR =
TRI->isVGPR(MRI,
Reg);
1477 if (IsVGPR ==
TRI->isVGPR(MRI, ValueReg)) {
1489 unsigned Opc = (IsStore ^ IsVGPR) ? AMDGPU::V_ACCVGPR_WRITE_B32_e64
1490 : AMDGPU::V_ACCVGPR_READ_B32_e64;
1510 bool IsStore =
MI->mayStore();
1512 unsigned Opc =
MI->getOpcode();
1513 int LoadStoreOp = IsStore ?
1515 if (LoadStoreOp == -1)
1526 .
add(*
TII->getNamedOperand(*
MI, AMDGPU::OpName::srsrc))
1527 .
add(*
TII->getNamedOperand(*
MI, AMDGPU::OpName::soffset))
1534 AMDGPU::OpName::vdata_in);
1536 NewMI.
add(*VDataIn);
1541 unsigned LoadStoreOp,
1543 bool IsStore =
TII->get(LoadStoreOp).mayStore();
1549 if (
TII->isBlockLoadStore(LoadStoreOp))
1554 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
1555 : AMDGPU::SCRATCH_LOAD_DWORD_SADDR;
1558 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX2_SADDR
1559 : AMDGPU::SCRATCH_LOAD_DWORDX2_SADDR;
1562 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX3_SADDR
1563 : AMDGPU::SCRATCH_LOAD_DWORDX3_SADDR;
1566 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX4_SADDR
1567 : AMDGPU::SCRATCH_LOAD_DWORDX4_SADDR;
1583 unsigned LoadStoreOp,
int Index,
Register ValueReg,
bool IsKill,
1586 assert((!RS || !LiveUnits) &&
"Only RS or LiveUnits can be set but not both");
1595 bool IsStore =
Desc->mayStore();
1596 bool IsFlat =
TII->isFLATScratch(LoadStoreOp);
1597 bool IsBlock =
TII->isBlockLoadStore(LoadStoreOp);
1599 bool CanClobberSCC =
false;
1600 bool Scavenged =
false;
1605 const bool IsAGPR = !ST.hasGFX90AInsts() &&
isAGPRClass(RC);
1616 bool IsRegMisaligned =
false;
1617 if (!IsBlock && !IsAGPR && RegWidth > 4 && IsFlat) {
1618 unsigned SpillOpcode =
1621 IsStore ? AMDGPU::getNamedOperandIdx(SpillOpcode, AMDGPU::OpName::vdata)
1624 TII->getRegClass(
TII->get(SpillOpcode), VDataIdx);
1625 if (!ExpectedRC->
contains(ValueReg)) {
1629 getMatchingSuperRegClass(RC, ExpectedRC, SubIdx);
1630 if (!MatchRC || !MatchRC->
contains(ValueReg))
1631 IsRegMisaligned =
true;
1635 if (IsRegMisaligned)
1640 unsigned EltSize = IsBlock ? RegWidth
1641 : (IsFlat && !IsAGPR) ? std::min(RegWidth, 16u)
1643 unsigned NumSubRegs = RegWidth / EltSize;
1644 unsigned Size = NumSubRegs * EltSize;
1645 unsigned RemSize = RegWidth -
Size;
1646 unsigned NumRemSubRegs = RemSize ? 1 : 0;
1648 if (IsRegMisaligned)
1651 int64_t MaterializedOffset =
Offset;
1656 int64_t MaxOffset =
Offset +
Size - (RemSize ? 0 : EltSize);
1657 int64_t ScratchOffsetRegDelta = 0;
1658 int64_t AdditionalCFIOffset = 0;
1660 if (IsFlat && EltSize > 4) {
1662 Desc = &
TII->get(LoadStoreOp);
1669 "unexpected VGPR spill offset");
1676 bool UseVGPROffset =
false;
1683 if (IsFlat && SGPRBase) {
1688 if (ST.getConstantBusLimit(AMDGPU::V_ADD_U32_e64) >= 2) {
1707 bool IsOffsetLegal =
1710 :
TII->isLegalMUBUFImmOffset(MaxOffset);
1711 if (!IsOffsetLegal || (IsFlat && !SOffset && !ST.hasFlatScratchSTMode())) {
1719 SOffset = RS->scavengeRegisterBackwards(AMDGPU::SGPR_32RegClass,
MI,
false, 0,
false);
1722 CanClobberSCC = !RS->isRegUsed(AMDGPU::SCC);
1723 }
else if (LiveUnits) {
1724 CanClobberSCC = LiveUnits->
available(AMDGPU::SCC);
1725 for (
MCRegister Reg : AMDGPU::SGPR_32RegClass) {
1733 if (ScratchOffsetReg != AMDGPU::NoRegister && !CanClobberSCC)
1737 UseVGPROffset =
true;
1740 TmpOffsetVGPR = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
MI,
false, 0);
1743 for (
MCRegister Reg : AMDGPU::VGPR_32RegClass) {
1745 TmpOffsetVGPR = Reg;
1752 }
else if (!SOffset && CanClobberSCC) {
1763 if (!ScratchOffsetReg)
1765 SOffset = ScratchOffsetReg;
1766 ScratchOffsetRegDelta =
Offset;
1771 AdditionalCFIOffset =
Offset;
1775 if (!IsFlat && !UseVGPROffset)
1776 Offset *= ST.getWavefrontSize();
1778 if (!UseVGPROffset && !SOffset)
1781 if (UseVGPROffset) {
1783 MaterializeVOffset(ScratchOffsetReg, TmpOffsetVGPR,
Offset);
1784 }
else if (ScratchOffsetReg == AMDGPU::NoRegister) {
1789 .
addReg(ScratchOffsetReg)
1791 Add->getOperand(3).setIsDead();
1797 if (IsFlat && SOffset == AMDGPU::NoRegister) {
1798 assert(AMDGPU::getNamedOperandIdx(LoadStoreOp, AMDGPU::OpName::vaddr) < 0
1799 &&
"Unexpected vaddr for flat scratch with a FI operand");
1801 if (UseVGPROffset) {
1804 assert(ST.hasFlatScratchSTMode());
1805 assert(!
TII->isBlockLoadStore(LoadStoreOp) &&
"Block ops don't have ST");
1809 Desc = &
TII->get(LoadStoreOp);
1814 unsigned OrigEltSize = EltSize;
1815 for (
unsigned i = 0, e = NumSubRegs + NumRemSubRegs, RegOffset = 0; i != e;
1816 ++i, RegOffset += EltSize) {
1817 if (IsRegMisaligned) {
1825 IsRegMisaligned =
false;
1826 EltSize = OrigEltSize;
1830 if (i == NumSubRegs) {
1834 Desc = &
TII->get(LoadStoreOp);
1836 if (!IsFlat && UseVGPROffset) {
1839 Desc = &
TII->get(NewLoadStoreOp);
1842 if (UseVGPROffset && TmpOffsetVGPR == TmpIntermediateVGPR) {
1849 MaterializeVOffset(ScratchOffsetReg, TmpOffsetVGPR, MaterializedOffset);
1852 unsigned NumRegs = EltSize / 4;
1860 const bool IsLastSubReg = i + 1 == e;
1861 const bool IsFirstSubReg = i == 0;
1870 bool NeedSuperRegDef = e > 1 && IsStore && IsFirstSubReg;
1871 bool NeedSuperRegImpOperand = e > 1;
1875 unsigned RemEltSize = EltSize;
1883 for (
int LaneS = (RegOffset + EltSize) / 4 - 1, Lane = LaneS,
1884 LaneE = RegOffset / 4;
1885 Lane >= LaneE; --Lane) {
1886 bool IsSubReg = e > 1 || EltSize > 4;
1892 if (!MIB.getInstr())
1894 if (NeedSuperRegDef || (IsSubReg && IsStore && Lane == LaneS && IsFirstSubReg)) {
1896 NeedSuperRegDef =
false;
1898 if ((IsSubReg || NeedSuperRegImpOperand) && (IsFirstSubReg || IsLastSubReg)) {
1899 NeedSuperRegImpOperand =
true;
1901 if (!IsLastSubReg || (Lane != LaneE))
1903 if (!IsFirstSubReg || (Lane != LaneS))
1913 if (RemEltSize != EltSize) {
1914 assert(IsFlat && EltSize > 4);
1916 unsigned NumRegs = RemEltSize / 4;
1917 SubReg =
Register(getSubReg(ValueReg,
1923 unsigned FinalReg = SubReg;
1928 if (!TmpIntermediateVGPR) {
1934 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64),
1935 TmpIntermediateVGPR)
1937 if (NeedSuperRegDef)
1939 if (NeedSuperRegImpOperand && (IsFirstSubReg || IsLastSubReg))
1943 SubReg = TmpIntermediateVGPR;
1944 }
else if (UseVGPROffset) {
1945 if (!TmpOffsetVGPR) {
1946 TmpOffsetVGPR = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
1948 RS->setRegUsed(TmpOffsetVGPR);
1953 if (LoadStoreOp == AMDGPU::SCRATCH_LOAD_USHORT_SADDR ||
1954 LoadStoreOp == AMDGPU::SCRATCH_LOAD_USHORT_ST) {
1958 RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
MI,
false, 0);
1974 if (UseVGPROffset) {
1983 if (SOffset == AMDGPU::NoRegister) {
1985 if (UseVGPROffset && ScratchOffsetReg) {
1986 MIB.addReg(ScratchOffsetReg);
1993 MIB.addReg(SOffset, SOffsetRegState);
2003 MIB.addMemOperand(NewMMO);
2005 if (FinalValueReg != ValueReg) {
2007 ValueReg = getSubReg(ValueReg, AMDGPU::lo16);
2013 ValueReg = FinalValueReg;
2016 if (IsStore && NeedsCFI) {
2017 if (
TII->isBlockLoadStore(LoadStoreOp)) {
2019 "expected whole register block to be treated as single element");
2024 (
Offset + RegOffset) * ST.getWavefrontSize() + AdditionalCFIOffset);
2028 if (!IsAGPR && NeedSuperRegDef)
2031 if (!IsStore && IsAGPR && TmpIntermediateVGPR != AMDGPU::NoRegister) {
2039 bool PartialReloadCopy = (RemEltSize != EltSize) && !IsStore;
2040 if (NeedSuperRegImpOperand &&
2041 (IsFirstSubReg || (IsLastSubReg && !IsSrcDstDef))) {
2043 if (PartialReloadCopy)
2068 if (!IsStore &&
MI !=
MBB.end() &&
MI->isReturn() &&
2069 MI->readsRegister(SubReg,
this)) {
2071 MIB->tieOperands(0, MIB->getNumOperands() - 1);
2079 if (!IsStore &&
TII->isBlockLoadStore(LoadStoreOp))
2083 if (ScratchOffsetRegDelta != 0) {
2087 .
addImm(-ScratchOffsetRegDelta);
2096 Register BaseVGPR = getSubReg(BlockReg, AMDGPU::sub0);
2097 for (
unsigned RegOffset = 1; RegOffset < 32; ++RegOffset)
2098 if (!(Mask & (1 << RegOffset)) &&
2099 isCalleeSavedPhysReg(BaseVGPR + RegOffset, *MF))
2110 Register BaseVGPR = getSubReg(BlockReg, AMDGPU::sub0);
2111 for (
unsigned RegOffset = 0; RegOffset < 32; ++RegOffset) {
2112 Register VGPR = BaseVGPR + RegOffset;
2113 if (Mask & (1 << RegOffset)) {
2114 assert(isCalleeSavedPhysReg(VGPR, *MF));
2115 ST.getFrameLowering()->buildCFIForVGPRToVMEMSpill(
2117 (
Offset + RegOffset) * ST.getWavefrontSize());
2118 }
else if (isCalleeSavedPhysReg(VGPR, *MF)) {
2123 BaseVGPR + RegOffset);
2130 bool IsKill)
const {
2140 Align Alignment = FrameInfo.getObjectAlign(Index);
2147 unsigned Opc = ST.hasFlatScratchEnabled()
2148 ? AMDGPU::SCRATCH_LOAD_DWORD_SADDR
2149 : AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
2153 unsigned Opc = ST.hasFlatScratchEnabled()
2154 ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
2155 : AMDGPU::BUFFER_STORE_DWORD_OFFSET;
2166 bool SpillToPhysVGPRLane,
bool NeedsCFI)
const {
2167 assert(!
MI->getOperand(0).isUndef() &&
2168 "undef spill should have been deleted earlier");
2175 bool SpillToVGPR = !VGPRSpills.
empty();
2176 if (OnlyToVGPR && !SpillToVGPR)
2191 "Num of SGPRs spilled should be less than or equal to num of "
2194 for (
unsigned i = 0, e = SB.
NumSubRegs; i < e; ++i) {
2201 bool IsFirstSubreg = i == 0;
2203 bool UseKill = SB.
IsKill && IsLastSubreg;
2209 SB.
TII.get(AMDGPU::SI_SPILL_S32_TO_VGPR), Spill.VGPR)
2219 AMDGPU::PC_REG, VGPRSpills);
2222 Spill.VGPR, Spill.Lane);
2242 if (SB.
NumSubRegs > 1 && (IsFirstSubreg || IsLastSubreg))
2262 for (
unsigned i =
Offset * PVD.PerVGPR,
2272 SB.
TII.get(AMDGPU::SI_SPILL_S32_TO_VGPR), SB.
TmpVGPR)
2273 .
addReg(SubReg, SubKillState)
2304 ST.getWavefrontSize();
2306 AMDGPU::PC_REG, CFIOffset);
2315 MI->eraseFromParent();
2327 bool SpillToPhysVGPRLane)
const {
2333 bool SpillToVGPR = !VGPRSpills.
empty();
2334 if (OnlyToVGPR && !SpillToVGPR)
2338 for (
unsigned i = 0, e = SB.
NumSubRegs; i < e; ++i) {
2346 SB.
TII.get(AMDGPU::SI_RESTORE_S32_FROM_VGPR), SubReg)
2369 for (
unsigned i =
Offset * PVD.PerVGPR,
2377 bool LastSubReg = (i + 1 == e);
2379 SB.
TII.get(AMDGPU::SI_RESTORE_S32_FROM_VGPR), SubReg)
2396 MI->eraseFromParent();
2416 for (
unsigned i =
Offset * PVD.PerVGPR,
2427 .
addReg(SubReg, SubKillState)
2445 MI = RestoreMBB.
end();
2451 for (
unsigned i =
Offset * PVD.PerVGPR,
2460 bool LastSubReg = (i + 1 == e);
2481 bool NeedsCFI =
false;
2482 switch (
MI->getOpcode()) {
2483 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
2484 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
2485 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
2486 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
2487 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
2488 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
2489 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
2490 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
2491 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
2492 case AMDGPU::SI_SPILL_S32_CFI_SAVE:
2495 case AMDGPU::SI_SPILL_S1024_SAVE:
2496 case AMDGPU::SI_SPILL_S512_SAVE:
2497 case AMDGPU::SI_SPILL_S384_SAVE:
2498 case AMDGPU::SI_SPILL_S352_SAVE:
2499 case AMDGPU::SI_SPILL_S320_SAVE:
2500 case AMDGPU::SI_SPILL_S288_SAVE:
2501 case AMDGPU::SI_SPILL_S256_SAVE:
2502 case AMDGPU::SI_SPILL_S224_SAVE:
2503 case AMDGPU::SI_SPILL_S192_SAVE:
2504 case AMDGPU::SI_SPILL_S160_SAVE:
2505 case AMDGPU::SI_SPILL_S128_SAVE:
2506 case AMDGPU::SI_SPILL_S96_SAVE:
2507 case AMDGPU::SI_SPILL_S64_SAVE:
2508 case AMDGPU::SI_SPILL_S32_SAVE:
2509 return spillSGPR(
MI, FI, RS, Indexes, LIS,
true, SpillToPhysVGPRLane,
2511 case AMDGPU::SI_SPILL_S1024_RESTORE:
2512 case AMDGPU::SI_SPILL_S512_RESTORE:
2513 case AMDGPU::SI_SPILL_S384_RESTORE:
2514 case AMDGPU::SI_SPILL_S352_RESTORE:
2515 case AMDGPU::SI_SPILL_S320_RESTORE:
2516 case AMDGPU::SI_SPILL_S288_RESTORE:
2517 case AMDGPU::SI_SPILL_S256_RESTORE:
2518 case AMDGPU::SI_SPILL_S224_RESTORE:
2519 case AMDGPU::SI_SPILL_S192_RESTORE:
2520 case AMDGPU::SI_SPILL_S160_RESTORE:
2521 case AMDGPU::SI_SPILL_S128_RESTORE:
2522 case AMDGPU::SI_SPILL_S96_RESTORE:
2523 case AMDGPU::SI_SPILL_S64_RESTORE:
2524 case AMDGPU::SI_SPILL_S32_RESTORE:
2525 return restoreSGPR(
MI, FI, RS, Indexes, LIS,
true, SpillToPhysVGPRLane);
2548 return (RS.isRegUsed(AMDGPU::SCC) &&
2549 !
MI.definesRegister(AMDGPU::SCC,
nullptr)) ||
2550 MI.readsRegister(AMDGPU::SCC,
nullptr);
2554 int SPAdj,
unsigned FIOperandNum,
2563 assert(SPAdj == 0 &&
"unhandled SP adjustment in call sequence?");
2566 "unreserved scratch RSRC register");
2569 int Index =
MI->getOperand(FIOperandNum).getIndex();
2575 bool NeedsCFI =
false;
2577 switch (
MI->getOpcode()) {
2579 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
2580 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
2581 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
2582 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
2583 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
2584 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
2585 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
2586 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
2587 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
2588 case AMDGPU::SI_SPILL_S32_CFI_SAVE: {
2592 case AMDGPU::SI_SPILL_S1024_SAVE:
2593 case AMDGPU::SI_SPILL_S512_SAVE:
2594 case AMDGPU::SI_SPILL_S384_SAVE:
2595 case AMDGPU::SI_SPILL_S352_SAVE:
2596 case AMDGPU::SI_SPILL_S320_SAVE:
2597 case AMDGPU::SI_SPILL_S288_SAVE:
2598 case AMDGPU::SI_SPILL_S256_SAVE:
2599 case AMDGPU::SI_SPILL_S224_SAVE:
2600 case AMDGPU::SI_SPILL_S192_SAVE:
2601 case AMDGPU::SI_SPILL_S160_SAVE:
2602 case AMDGPU::SI_SPILL_S128_SAVE:
2603 case AMDGPU::SI_SPILL_S96_SAVE:
2604 case AMDGPU::SI_SPILL_S64_SAVE:
2605 case AMDGPU::SI_SPILL_S32_SAVE: {
2612 case AMDGPU::SI_SPILL_S1024_RESTORE:
2613 case AMDGPU::SI_SPILL_S512_RESTORE:
2614 case AMDGPU::SI_SPILL_S384_RESTORE:
2615 case AMDGPU::SI_SPILL_S352_RESTORE:
2616 case AMDGPU::SI_SPILL_S320_RESTORE:
2617 case AMDGPU::SI_SPILL_S288_RESTORE:
2618 case AMDGPU::SI_SPILL_S256_RESTORE:
2619 case AMDGPU::SI_SPILL_S224_RESTORE:
2620 case AMDGPU::SI_SPILL_S192_RESTORE:
2621 case AMDGPU::SI_SPILL_S160_RESTORE:
2622 case AMDGPU::SI_SPILL_S128_RESTORE:
2623 case AMDGPU::SI_SPILL_S96_RESTORE:
2624 case AMDGPU::SI_SPILL_S64_RESTORE:
2625 case AMDGPU::SI_SPILL_S32_RESTORE: {
2627 FrameInfo.getStackID(Index) ==
2632 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
2633 case AMDGPU::SI_SPILL_V1024_CFI_SAVE:
2634 case AMDGPU::SI_SPILL_V512_CFI_SAVE:
2635 case AMDGPU::SI_SPILL_V256_CFI_SAVE:
2636 case AMDGPU::SI_SPILL_V224_CFI_SAVE:
2637 case AMDGPU::SI_SPILL_V192_CFI_SAVE:
2638 case AMDGPU::SI_SPILL_V160_CFI_SAVE:
2639 case AMDGPU::SI_SPILL_V128_CFI_SAVE:
2640 case AMDGPU::SI_SPILL_V96_CFI_SAVE:
2641 case AMDGPU::SI_SPILL_V64_CFI_SAVE:
2642 case AMDGPU::SI_SPILL_V32_CFI_SAVE:
2643 case AMDGPU::SI_SPILL_A1024_CFI_SAVE:
2644 case AMDGPU::SI_SPILL_A512_CFI_SAVE:
2645 case AMDGPU::SI_SPILL_A256_CFI_SAVE:
2646 case AMDGPU::SI_SPILL_A224_CFI_SAVE:
2647 case AMDGPU::SI_SPILL_A192_CFI_SAVE:
2648 case AMDGPU::SI_SPILL_A160_CFI_SAVE:
2649 case AMDGPU::SI_SPILL_A128_CFI_SAVE:
2650 case AMDGPU::SI_SPILL_A96_CFI_SAVE:
2651 case AMDGPU::SI_SPILL_A64_CFI_SAVE:
2652 case AMDGPU::SI_SPILL_A32_CFI_SAVE:
2653 case AMDGPU::SI_SPILL_AV1024_CFI_SAVE:
2654 case AMDGPU::SI_SPILL_AV512_CFI_SAVE:
2655 case AMDGPU::SI_SPILL_AV256_CFI_SAVE:
2656 case AMDGPU::SI_SPILL_AV224_CFI_SAVE:
2657 case AMDGPU::SI_SPILL_AV192_CFI_SAVE:
2658 case AMDGPU::SI_SPILL_AV160_CFI_SAVE:
2659 case AMDGPU::SI_SPILL_AV128_CFI_SAVE:
2660 case AMDGPU::SI_SPILL_AV96_CFI_SAVE:
2661 case AMDGPU::SI_SPILL_AV64_CFI_SAVE:
2662 case AMDGPU::SI_SPILL_AV32_CFI_SAVE:
2665 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
2666 case AMDGPU::SI_SPILL_V1024_SAVE:
2667 case AMDGPU::SI_SPILL_V512_SAVE:
2668 case AMDGPU::SI_SPILL_V384_SAVE:
2669 case AMDGPU::SI_SPILL_V352_SAVE:
2670 case AMDGPU::SI_SPILL_V320_SAVE:
2671 case AMDGPU::SI_SPILL_V288_SAVE:
2672 case AMDGPU::SI_SPILL_V256_SAVE:
2673 case AMDGPU::SI_SPILL_V224_SAVE:
2674 case AMDGPU::SI_SPILL_V192_SAVE:
2675 case AMDGPU::SI_SPILL_V160_SAVE:
2676 case AMDGPU::SI_SPILL_V128_SAVE:
2677 case AMDGPU::SI_SPILL_V96_SAVE:
2678 case AMDGPU::SI_SPILL_V64_SAVE:
2679 case AMDGPU::SI_SPILL_V32_SAVE:
2680 case AMDGPU::SI_SPILL_V16_SAVE:
2681 case AMDGPU::SI_SPILL_A1024_SAVE:
2682 case AMDGPU::SI_SPILL_A512_SAVE:
2683 case AMDGPU::SI_SPILL_A384_SAVE:
2684 case AMDGPU::SI_SPILL_A352_SAVE:
2685 case AMDGPU::SI_SPILL_A320_SAVE:
2686 case AMDGPU::SI_SPILL_A288_SAVE:
2687 case AMDGPU::SI_SPILL_A256_SAVE:
2688 case AMDGPU::SI_SPILL_A224_SAVE:
2689 case AMDGPU::SI_SPILL_A192_SAVE:
2690 case AMDGPU::SI_SPILL_A160_SAVE:
2691 case AMDGPU::SI_SPILL_A128_SAVE:
2692 case AMDGPU::SI_SPILL_A96_SAVE:
2693 case AMDGPU::SI_SPILL_A64_SAVE:
2694 case AMDGPU::SI_SPILL_A32_SAVE:
2695 case AMDGPU::SI_SPILL_AV1024_SAVE:
2696 case AMDGPU::SI_SPILL_AV512_SAVE:
2697 case AMDGPU::SI_SPILL_AV384_SAVE:
2698 case AMDGPU::SI_SPILL_AV352_SAVE:
2699 case AMDGPU::SI_SPILL_AV320_SAVE:
2700 case AMDGPU::SI_SPILL_AV288_SAVE:
2701 case AMDGPU::SI_SPILL_AV256_SAVE:
2702 case AMDGPU::SI_SPILL_AV224_SAVE:
2703 case AMDGPU::SI_SPILL_AV192_SAVE:
2704 case AMDGPU::SI_SPILL_AV160_SAVE:
2705 case AMDGPU::SI_SPILL_AV128_SAVE:
2706 case AMDGPU::SI_SPILL_AV96_SAVE:
2707 case AMDGPU::SI_SPILL_AV64_SAVE:
2708 case AMDGPU::SI_SPILL_AV32_SAVE:
2709 case AMDGPU::SI_SPILL_WWM_V32_SAVE:
2710 case AMDGPU::SI_SPILL_WWM_AV32_SAVE: {
2712 MI->getOpcode() != AMDGPU::SI_BLOCK_SPILL_V1024_SAVE &&
2713 "block spill does not currenty support spilling non-CSR registers");
2715 if (
MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE)
2719 .
add(*
TII->getNamedOperand(*
MI, AMDGPU::OpName::mask));
2722 AMDGPU::OpName::vdata);
2724 MI->eraseFromParent();
2728 assert(
TII->getNamedOperand(*
MI, AMDGPU::OpName::soffset)->getReg() ==
2732 if (
MI->getOpcode() == AMDGPU::SI_SPILL_V16_SAVE) {
2733 assert(ST.hasFlatScratchEnabled() &&
"Flat Scratch is not enabled!");
2734 Opc = AMDGPU::SCRATCH_STORE_SHORT_SADDR_t16;
2736 Opc =
MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE
2737 ? AMDGPU::SCRATCH_STORE_BLOCK_SADDR
2738 : ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
2739 : AMDGPU::BUFFER_STORE_DWORD_OFFSET;
2742 auto *
MBB =
MI->getParent();
2743 bool IsWWMRegSpill =
TII->isWWMRegSpillOpcode(
MI->getOpcode());
2744 if (IsWWMRegSpill) {
2746 RS->isRegUsed(AMDGPU::SCC));
2750 TII->getNamedOperand(*
MI, AMDGPU::OpName::offset)->getImm(),
2751 *
MI->memoperands_begin(), RS,
nullptr, NeedsCFI);
2756 MI->eraseFromParent();
2759 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE: {
2763 .
add(*
TII->getNamedOperand(*
MI, AMDGPU::OpName::mask));
2766 case AMDGPU::SI_SPILL_V16_RESTORE:
2767 case AMDGPU::SI_SPILL_V32_RESTORE:
2768 case AMDGPU::SI_SPILL_V64_RESTORE:
2769 case AMDGPU::SI_SPILL_V96_RESTORE:
2770 case AMDGPU::SI_SPILL_V128_RESTORE:
2771 case AMDGPU::SI_SPILL_V160_RESTORE:
2772 case AMDGPU::SI_SPILL_V192_RESTORE:
2773 case AMDGPU::SI_SPILL_V224_RESTORE:
2774 case AMDGPU::SI_SPILL_V256_RESTORE:
2775 case AMDGPU::SI_SPILL_V288_RESTORE:
2776 case AMDGPU::SI_SPILL_V320_RESTORE:
2777 case AMDGPU::SI_SPILL_V352_RESTORE:
2778 case AMDGPU::SI_SPILL_V384_RESTORE:
2779 case AMDGPU::SI_SPILL_V512_RESTORE:
2780 case AMDGPU::SI_SPILL_V1024_RESTORE:
2781 case AMDGPU::SI_SPILL_A32_RESTORE:
2782 case AMDGPU::SI_SPILL_A64_RESTORE:
2783 case AMDGPU::SI_SPILL_A96_RESTORE:
2784 case AMDGPU::SI_SPILL_A128_RESTORE:
2785 case AMDGPU::SI_SPILL_A160_RESTORE:
2786 case AMDGPU::SI_SPILL_A192_RESTORE:
2787 case AMDGPU::SI_SPILL_A224_RESTORE:
2788 case AMDGPU::SI_SPILL_A256_RESTORE:
2789 case AMDGPU::SI_SPILL_A288_RESTORE:
2790 case AMDGPU::SI_SPILL_A320_RESTORE:
2791 case AMDGPU::SI_SPILL_A352_RESTORE:
2792 case AMDGPU::SI_SPILL_A384_RESTORE:
2793 case AMDGPU::SI_SPILL_A512_RESTORE:
2794 case AMDGPU::SI_SPILL_A1024_RESTORE:
2795 case AMDGPU::SI_SPILL_AV32_RESTORE:
2796 case AMDGPU::SI_SPILL_AV64_RESTORE:
2797 case AMDGPU::SI_SPILL_AV96_RESTORE:
2798 case AMDGPU::SI_SPILL_AV128_RESTORE:
2799 case AMDGPU::SI_SPILL_AV160_RESTORE:
2800 case AMDGPU::SI_SPILL_AV192_RESTORE:
2801 case AMDGPU::SI_SPILL_AV224_RESTORE:
2802 case AMDGPU::SI_SPILL_AV256_RESTORE:
2803 case AMDGPU::SI_SPILL_AV288_RESTORE:
2804 case AMDGPU::SI_SPILL_AV320_RESTORE:
2805 case AMDGPU::SI_SPILL_AV352_RESTORE:
2806 case AMDGPU::SI_SPILL_AV384_RESTORE:
2807 case AMDGPU::SI_SPILL_AV512_RESTORE:
2808 case AMDGPU::SI_SPILL_AV1024_RESTORE:
2809 case AMDGPU::SI_SPILL_WWM_V32_RESTORE:
2810 case AMDGPU::SI_SPILL_WWM_AV32_RESTORE: {
2812 AMDGPU::OpName::vdata);
2813 assert(
TII->getNamedOperand(*
MI, AMDGPU::OpName::soffset)->getReg() ==
2817 if (
MI->getOpcode() == AMDGPU::SI_SPILL_V16_RESTORE) {
2818 assert(ST.hasFlatScratchEnabled() &&
"Flat Scratch is not enabled!");
2819 Opc = ST.d16PreservesUnusedBits()
2820 ? AMDGPU::SCRATCH_LOAD_SHORT_D16_SADDR_t16
2821 : AMDGPU::SCRATCH_LOAD_USHORT_SADDR;
2823 Opc =
MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE
2824 ? AMDGPU::SCRATCH_LOAD_BLOCK_SADDR
2825 : ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_LOAD_DWORD_SADDR
2826 : AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
2829 auto *
MBB =
MI->getParent();
2830 bool IsWWMRegSpill =
TII->isWWMRegSpillOpcode(
MI->getOpcode());
2831 if (IsWWMRegSpill) {
2833 RS->isRegUsed(AMDGPU::SCC));
2838 TII->getNamedOperand(*
MI, AMDGPU::OpName::offset)->getImm(),
2839 *
MI->memoperands_begin(), RS);
2844 MI->eraseFromParent();
2847 case AMDGPU::V_ADD_U32_e32:
2848 case AMDGPU::V_ADD_U32_e64:
2849 case AMDGPU::V_ADD_CO_U32_e32:
2850 case AMDGPU::V_ADD_CO_U32_e64: {
2852 unsigned NumDefs =
MI->getNumExplicitDefs();
2853 unsigned Src0Idx = NumDefs;
2855 bool HasClamp =
false;
2858 switch (
MI->getOpcode()) {
2859 case AMDGPU::V_ADD_U32_e32:
2861 case AMDGPU::V_ADD_U32_e64:
2862 HasClamp =
MI->getOperand(3).getImm();
2864 case AMDGPU::V_ADD_CO_U32_e32:
2865 VCCOp = &
MI->getOperand(3);
2867 case AMDGPU::V_ADD_CO_U32_e64:
2868 VCCOp = &
MI->getOperand(1);
2869 HasClamp =
MI->getOperand(4).getImm();
2874 bool DeadVCC = !VCCOp || VCCOp->
isDead();
2878 unsigned OtherOpIdx =
2879 FIOperandNum == Src0Idx ? FIOperandNum + 1 : Src0Idx;
2882 unsigned Src1Idx = Src0Idx + 1;
2883 Register MaterializedReg = FrameReg;
2886 int64_t
Offset = FrameInfo.getObjectOffset(Index);
2890 if ((!DeadVCC || HasClamp) &&
2897 if (OtherOp->
isImm()) {
2908 OtherOp->
setImm(TotalOffset);
2912 if (FrameReg && !ST.hasFlatScratchEnabled()) {
2920 ScavengedVGPR = RS->scavengeRegisterBackwards(
2921 AMDGPU::VGPR_32RegClass,
MI,
false, 0);
2927 .
addImm(ST.getWavefrontSizeLog2())
2929 MaterializedReg = ScavengedVGPR;
2932 if ((!OtherOp->
isImm() || OtherOp->
getImm() != 0) && MaterializedReg) {
2933 if (OtherOp->
isImm()) {
2935 FIOp->
setIsKill(MaterializedReg != FrameReg);
2937 if (ST.hasFlatScratchEnabled() &&
2938 !
TII->isOperandLegal(*
MI, Src1Idx, OtherOp)) {
2946 if (!ScavengedVGPR) {
2947 ScavengedVGPR = RS->scavengeRegisterBackwards(
2948 AMDGPU::VGPR_32RegClass,
MI,
false,
2952 assert(ScavengedVGPR != DstReg);
2958 MaterializedReg = ScavengedVGPR;
2967 AddI32.
add(
MI->getOperand(1));
2972 if (
isVGPRClass(getPhysRegBaseClass(MaterializedReg))) {
2975 AddI32.add(*OtherOp).addReg(MaterializedReg, MaterializedRegFlags);
2979 AddI32.addReg(MaterializedReg, MaterializedRegFlags).add(*OtherOp);
2982 if (
MI->getOpcode() == AMDGPU::V_ADD_CO_U32_e64 ||
2983 MI->getOpcode() == AMDGPU::V_ADD_U32_e64)
2986 if (
MI->getOpcode() == AMDGPU::V_ADD_CO_U32_e32)
2987 AddI32.setOperandDead(3);
2989 MaterializedReg = DstReg;
2996 }
else if (
Offset != 0) {
2997 assert(!MaterializedReg);
3001 if (DeadVCC && !HasClamp) {
3006 if (OtherOp->
isReg() && OtherOp->
getReg() == DstReg) {
3008 MI->eraseFromParent();
3013 MI->setDesc(
TII->get(AMDGPU::V_MOV_B32_e32));
3014 MI->removeOperand(FIOperandNum);
3016 unsigned NumOps =
MI->getNumOperands();
3017 for (
unsigned I =
NumOps - 2;
I >= NumDefs + 1; --
I)
3018 MI->removeOperand(
I);
3021 MI->removeOperand(1);
3033 if (!
TII->isOperandLegal(*
MI, Src1Idx) &&
TII->commuteInstruction(*
MI)) {
3041 for (
unsigned SrcIdx : {FIOperandNum, OtherOpIdx}) {
3042 if (!
TII->isOperandLegal(*
MI, SrcIdx)) {
3046 if (!ScavengedVGPR) {
3047 ScavengedVGPR = RS->scavengeRegisterBackwards(
3048 AMDGPU::VGPR_32RegClass,
MI,
false,
3052 assert(ScavengedVGPR != DstReg);
3058 Src.ChangeToRegister(ScavengedVGPR,
false);
3059 Src.setIsKill(
true);
3065 if (FIOp->
isImm() && FIOp->
getImm() == 0 && DeadVCC && !HasClamp) {
3066 if (OtherOp->
isReg() && OtherOp->
getReg() != DstReg) {
3070 MI->eraseFromParent();
3075 case AMDGPU::S_ADD_I32:
3076 case AMDGPU::S_ADD_U32: {
3078 unsigned OtherOpIdx = FIOperandNum == 1 ? 2 : 1;
3085 Register MaterializedReg = FrameReg;
3087 int64_t
Offset = FrameInfo.getObjectOffset(Index);
3090 bool DeadSCC =
MI->getOperand(3).isDead();
3101 if (FrameReg && !ST.hasFlatScratchEnabled()) {
3106 TmpReg = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3113 .
addImm(ST.getWavefrontSizeLog2())
3116 MaterializedReg = TmpReg;
3122 if (OtherOp.
isImm()) {
3126 if (MaterializedReg)
3130 }
else if (MaterializedReg) {
3134 if (!TmpReg && MaterializedReg == FrameReg) {
3135 TmpReg = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3149 MaterializedReg = DstReg;
3162 if (DeadSCC && OtherOp.
isImm() && OtherOp.
getImm() == 0) {
3164 MI->removeOperand(3);
3165 MI->removeOperand(OtherOpIdx);
3167 MI->setDesc(
TII->get(Src.isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32));
3168 }
else if (DeadSCC && FIOp->
isImm() && FIOp->
getImm() == 0) {
3170 MI->removeOperand(3);
3171 MI->removeOperand(FIOperandNum);
3173 MI->setDesc(
TII->get(Src.isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32));
3184 int64_t
Offset = FrameInfo.getObjectOffset(Index);
3185 if (ST.hasFlatScratchEnabled()) {
3186 if (
TII->isFLATScratch(*
MI)) {
3188 (int16_t)FIOperandNum ==
3189 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::saddr));
3196 TII->getNamedOperand(*
MI, AMDGPU::OpName::offset);
3207 unsigned Opc =
MI->getOpcode();
3211 }
else if (ST.hasFlatScratchSTMode()) {
3221 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
3222 bool TiedVDst = VDstIn != -1 &&
MI->getOperand(VDstIn).isReg() &&
3223 MI->getOperand(VDstIn).isTied();
3225 MI->untieRegOperand(VDstIn);
3228 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr));
3232 AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
3234 AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst_in);
3235 assert(NewVDst != -1 && NewVDstIn != -1 &&
"Must be tied!");
3236 MI->tieOperands(NewVDst, NewVDstIn);
3238 MI->setDesc(
TII->get(NewOpc));
3246 if (
TII->isOperandLegal(*
MI, FIOperandNum, FIOp))
3253 bool UseSGPR =
TII->isOperandLegal(*
MI, FIOperandNum, FIOp);
3255 if (!
Offset && FrameReg && UseSGPR) {
3261 UseSGPR ? &AMDGPU::SReg_32_XM0RegClass : &AMDGPU::VGPR_32RegClass;
3264 RS->scavengeRegisterBackwards(*RC,
MI,
false, 0, !UseSGPR);
3268 if ((!FrameReg || !
Offset) && TmpReg) {
3269 unsigned Opc = UseSGPR ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
3272 MIB.addReg(FrameReg);
3283 : RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3284 MI,
false, 0, !UseSGPR);
3290 if ((!TmpSReg && !FrameReg) || (!TmpReg && !UseSGPR)) {
3291 int SVfromSSOpcode =
3293 int SVfromSVSOpcode =
3295 int SVOpcode = SVfromSSOpcode != -1 ? SVfromSSOpcode : SVfromSVSOpcode;
3296 if (ST.hasFlatScratchSVSMode() && SVOpcode != -1) {
3303 "SV-form fallback cannot encode a frame register");
3309 int64_t FullOffset =
3311 TII->getNamedOperand(*
MI, AMDGPU::OpName::offset)->getImm();
3312 auto [ImmOffset, RemainderOffset] =
3318 TII->getNamedOperand(*
MI, AMDGPU::OpName::vaddr)) {
3320 TII->getNamedOperand(*
MI, AMDGPU::OpName::vdata);
3324 bool CanReuseVAddr = VAddr->isKill() &&
3325 !(VData && regsOverlap(Src, VData->
getReg()));
3327 : RS->scavengeRegisterBackwards(
3328 AMDGPU::VGPR_32RegClass,
MI,
3336 UsedVAddr = RS->scavengeRegisterBackwards(
3337 AMDGPU::VGPR_32RegClass,
MI,
false, 0,
true);
3339 .
addImm(RemainderOffset);
3342 .
add(
MI->getOperand(0))
3345 .
add(*
TII->getNamedOperand(*
MI, AMDGPU::OpName::cpol));
3346 MI->eraseFromParent();
3360 assert(!(
Offset & 0x1) &&
"Flat scratch offset must be aligned!");
3380 if (TmpSReg == FrameReg) {
3383 !
MI->registerDefIsDead(AMDGPU::SCC,
nullptr)) {
3407 bool IsMUBUF =
TII->isMUBUF(*
MI);
3416 bool SCCLiveAfterMI = RS->isRegUsed(AMDGPU::SCC);
3418 ? &AMDGPU::SReg_32RegClass
3419 : &AMDGPU::VGPR_32RegClass;
3420 bool IsCopy =
MI->getOpcode() == AMDGPU::V_MOV_B32_e32 ||
3421 MI->getOpcode() == AMDGPU::V_MOV_B32_e64 ||
3422 MI->getOpcode() == AMDGPU::S_MOV_B32;
3424 int64_t
Offset = FrameInfo.getObjectOffset(Index);
3429 bool CanUseFrameRegAsSGPRScratch = IsSALU && FrameReg &&
3430 !
MI->readsRegister(FrameReg,
this) &&
3431 !
MI->modifiesRegister(FrameReg,
this);
3434 bool CanUseFrameRegAsScratch = CanUseFrameRegAsSGPRScratch && !LiveSCC;
3436 bool RestoreFrameReg =
false;
3440 ResultReg =
MI->getOperand(0).getReg();
3442 ResultReg = RS->scavengeRegisterBackwards(*RC,
MI,
false, 0,
3445 if (CanUseFrameRegAsScratch) {
3448 ResultReg = FrameReg;
3449 RestoreFrameReg =
true;
3451 ResultReg = RS->scavengeRegisterBackwards(*RC,
MI,
false, 0);
3460 isWave32 ?
Add.getReg(1)
3461 :
Register(getSubReg(
Add.getReg(1), AMDGPU::sub0));
3464 return ConstOffsetReg;
3469 IsSALU && !LiveSCC ? AMDGPU::S_LSHR_B32 : AMDGPU::V_LSHRREV_B32_e64;
3471 if (IsSALU && LiveSCC) {
3472 TmpResultReg = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
3477 if (OpCode == AMDGPU::V_LSHRREV_B32_e64)
3480 Shift.addImm(ST.getWavefrontSizeLog2()).addReg(FrameReg);
3482 Shift.addReg(FrameReg).addImm(ST.getWavefrontSizeLog2());
3483 if (IsSALU && !LiveSCC)
3484 Shift.getInstr()->getOperand(3).setIsDead();
3485 if (IsSALU && LiveSCC) {
3489 NewDest = ResultReg;
3494 NewDest = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3498 if (CanUseFrameRegAsSGPRScratch) {
3500 RestoreFrameReg =
true;
3505 "unhandled SGPR spill to memory");
3506 NewDest = RS->scavengeRegisterBackwards(
3507 AMDGPU::SReg_32_XM0RegClass, Shift,
false, 0);
3513 ResultReg = NewDest;
3518 if ((MIB =
TII->getAddNoCarry(*
MBB,
MI,
DL, ResultReg, *RS)) !=
3525 .
addImm(ST.getWavefrontSizeLog2())
3528 const bool IsVOP2 = MIB->
getOpcode() == AMDGPU::V_ADD_U32_e32;
3540 "Need to reuse carry out register");
3548 if (!MIB || IsSALU) {
3555 Register TmpScaledReg = IsCopy && IsSALU
3557 : RS->scavengeRegisterBackwards(
3558 AMDGPU::SReg_32_XM0RegClass,
MI,
3564 ScaledReg = IsSALU ? ResultReg : FrameReg;
3570 .
addImm(ST.getWavefrontSizeLog2());
3575 TmpResultReg = RS->scavengeRegisterBackwards(
3576 AMDGPU::VGPR_32RegClass,
MI,
false, 0,
true);
3579 if ((
Add =
TII->getAddNoCarry(*
MBB,
MI,
DL, TmpResultReg, *RS))) {
3582 .
addImm(ST.getWavefrontSizeLog2())
3584 if (
Add->getOpcode() == AMDGPU::V_ADD_CO_U32_e64) {
3592 "offset is unsafe for v_mad_u32_u24");
3601 bool IsInlinableLiteral =
3603 if (!IsInlinableLiteral) {
3612 if (!IsInlinableLiteral) {
3618 Add.addImm(ST.getWavefrontSize()).addReg(FrameReg).addImm(0);
3621 .
addImm(ST.getWavefrontSizeLog2())
3627 NewDest = ResultReg;
3630 NewDest = RS->scavengeRegisterBackwards(
3631 AMDGPU::SReg_32_XM0RegClass, *
Add,
false, 0,
3634 if (CanUseFrameRegAsSGPRScratch) {
3636 RestoreFrameReg =
true;
3640 "unhandled SGPR spill to memory");
3641 NewDest = RS->scavengeRegisterBackwards(
3642 AMDGPU::SReg_32_XM0RegClass, *
Add,
false, 0);
3650 ResultReg = NewDest;
3658 if (!TmpScaledReg.
isValid()) {
3664 .
addImm(ST.getWavefrontSizeLog2());
3670 if (RestoreFrameReg) {
3679 .
addImm(ST.getWavefrontSize());
3682 int64_t ScaledOffset = -
Offset * ST.getWavefrontSize();
3683 if (!SCCLiveAfterMI) {
3703 MI->eraseFromParent();
3714 static_cast<int>(FIOperandNum) ==
3715 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::vaddr));
3717 auto &SOffset = *
TII->getNamedOperand(*
MI, AMDGPU::OpName::soffset);
3718 assert((SOffset.isImm() && SOffset.getImm() == 0));
3720 if (FrameReg != AMDGPU::NoRegister)
3721 SOffset.ChangeToRegister(FrameReg,
false);
3723 int64_t
Offset = FrameInfo.getObjectOffset(Index);
3725 TII->getNamedOperand(*
MI, AMDGPU::OpName::offset)->getImm();
3726 int64_t NewOffset = OldImm +
Offset;
3728 if (
TII->isLegalMUBUFImmOffset(NewOffset) &&
3730 MI->eraseFromParent();
3741 if (!
TII->isOperandLegal(*
MI, FIOperandNum, FIOp)) {
3743 TII->getRegClass(
MI->getDesc(), FIOperandNum);
3747 UseSGPR ? &AMDGPU::SReg_32_XM0RegClass : &AMDGPU::VGPR_32RegClass;
3748 Register TmpReg = RS->scavengeRegisterBackwards(*RC,
MI,
false, 0);
3750 TII->get(UseSGPR ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32),
3770 return &AMDGPU::VReg_64RegClass;
3772 return &AMDGPU::VReg_96RegClass;
3774 return &AMDGPU::VReg_128RegClass;
3776 return &AMDGPU::VReg_160RegClass;
3778 return &AMDGPU::VReg_192RegClass;
3780 return &AMDGPU::VReg_224RegClass;
3782 return &AMDGPU::VReg_256RegClass;
3784 return &AMDGPU::VReg_288RegClass;
3786 return &AMDGPU::VReg_320RegClass;
3788 return &AMDGPU::VReg_352RegClass;
3790 return &AMDGPU::VReg_384RegClass;
3792 return &AMDGPU::VReg_512RegClass;
3794 return &AMDGPU::VReg_1024RegClass;
3802 return &AMDGPU::VReg_64_Align2RegClass;
3804 return &AMDGPU::VReg_96_Align2RegClass;
3806 return &AMDGPU::VReg_128_Align2RegClass;
3808 return &AMDGPU::VReg_160_Align2RegClass;
3810 return &AMDGPU::VReg_192_Align2RegClass;
3812 return &AMDGPU::VReg_224_Align2RegClass;
3814 return &AMDGPU::VReg_256_Align2RegClass;
3816 return &AMDGPU::VReg_288_Align2RegClass;
3818 return &AMDGPU::VReg_320_Align2RegClass;
3820 return &AMDGPU::VReg_352_Align2RegClass;
3822 return &AMDGPU::VReg_384_Align2RegClass;
3824 return &AMDGPU::VReg_512_Align2RegClass;
3826 return &AMDGPU::VReg_1024_Align2RegClass;
3834 return &AMDGPU::VReg_1RegClass;
3836 return &AMDGPU::VGPR_16RegClass;
3838 return &AMDGPU::VGPR_32RegClass;
3846 return &AMDGPU::VGPR_32_Lo256RegClass;
3848 return &AMDGPU::VReg_64_Lo256_Align2RegClass;
3850 return &AMDGPU::VReg_96_Lo256_Align2RegClass;
3852 return &AMDGPU::VReg_128_Lo256_Align2RegClass;
3854 return &AMDGPU::VReg_160_Lo256_Align2RegClass;
3856 return &AMDGPU::VReg_192_Lo256_Align2RegClass;
3858 return &AMDGPU::VReg_224_Lo256_Align2RegClass;
3860 return &AMDGPU::VReg_256_Lo256_Align2RegClass;
3862 return &AMDGPU::VReg_288_Lo256_Align2RegClass;
3864 return &AMDGPU::VReg_320_Lo256_Align2RegClass;
3866 return &AMDGPU::VReg_352_Lo256_Align2RegClass;
3868 return &AMDGPU::VReg_384_Lo256_Align2RegClass;
3870 return &AMDGPU::VReg_512_Lo256_Align2RegClass;
3872 return &AMDGPU::VReg_1024_Lo256_Align2RegClass;
3880 return &AMDGPU::AReg_64RegClass;
3882 return &AMDGPU::AReg_96RegClass;
3884 return &AMDGPU::AReg_128RegClass;
3886 return &AMDGPU::AReg_160RegClass;
3888 return &AMDGPU::AReg_192RegClass;
3890 return &AMDGPU::AReg_224RegClass;
3892 return &AMDGPU::AReg_256RegClass;
3894 return &AMDGPU::AReg_288RegClass;
3896 return &AMDGPU::AReg_320RegClass;
3898 return &AMDGPU::AReg_352RegClass;
3900 return &AMDGPU::AReg_384RegClass;
3902 return &AMDGPU::AReg_512RegClass;
3904 return &AMDGPU::AReg_1024RegClass;
3912 return &AMDGPU::AReg_64_Align2RegClass;
3914 return &AMDGPU::AReg_96_Align2RegClass;
3916 return &AMDGPU::AReg_128_Align2RegClass;
3918 return &AMDGPU::AReg_160_Align2RegClass;
3920 return &AMDGPU::AReg_192_Align2RegClass;
3922 return &AMDGPU::AReg_224_Align2RegClass;
3924 return &AMDGPU::AReg_256_Align2RegClass;
3926 return &AMDGPU::AReg_288_Align2RegClass;
3928 return &AMDGPU::AReg_320_Align2RegClass;
3930 return &AMDGPU::AReg_352_Align2RegClass;
3932 return &AMDGPU::AReg_384_Align2RegClass;
3934 return &AMDGPU::AReg_512_Align2RegClass;
3936 return &AMDGPU::AReg_1024_Align2RegClass;
3944 return &AMDGPU::AGPR_LO16RegClass;
3946 return &AMDGPU::AGPR_32RegClass;
3954 return &AMDGPU::AV_64RegClass;
3956 return &AMDGPU::AV_96RegClass;
3958 return &AMDGPU::AV_128RegClass;
3960 return &AMDGPU::AV_160RegClass;
3962 return &AMDGPU::AV_192RegClass;
3964 return &AMDGPU::AV_224RegClass;
3966 return &AMDGPU::AV_256RegClass;
3968 return &AMDGPU::AV_288RegClass;
3970 return &AMDGPU::AV_320RegClass;
3972 return &AMDGPU::AV_352RegClass;
3974 return &AMDGPU::AV_384RegClass;
3976 return &AMDGPU::AV_512RegClass;
3978 return &AMDGPU::AV_1024RegClass;
3986 return &AMDGPU::AV_64_Align2RegClass;
3988 return &AMDGPU::AV_96_Align2RegClass;
3990 return &AMDGPU::AV_128_Align2RegClass;
3992 return &AMDGPU::AV_160_Align2RegClass;
3994 return &AMDGPU::AV_192_Align2RegClass;
3996 return &AMDGPU::AV_224_Align2RegClass;
3998 return &AMDGPU::AV_256_Align2RegClass;
4000 return &AMDGPU::AV_288_Align2RegClass;
4002 return &AMDGPU::AV_320_Align2RegClass;
4004 return &AMDGPU::AV_352_Align2RegClass;
4006 return &AMDGPU::AV_384_Align2RegClass;
4008 return &AMDGPU::AV_512_Align2RegClass;
4010 return &AMDGPU::AV_1024_Align2RegClass;
4018 return &AMDGPU::AV_32RegClass;
4019 return ST.needsAlignedVGPRs()
4038 return &AMDGPU::SReg_32RegClass;
4040 return &AMDGPU::SReg_64RegClass;
4042 return &AMDGPU::SGPR_96RegClass;
4044 return &AMDGPU::SGPR_128RegClass;
4046 return &AMDGPU::SGPR_160RegClass;
4048 return &AMDGPU::SGPR_192RegClass;
4050 return &AMDGPU::SGPR_224RegClass;
4052 return &AMDGPU::SGPR_256RegClass;
4054 return &AMDGPU::SGPR_288RegClass;
4056 return &AMDGPU::SGPR_320RegClass;
4058 return &AMDGPU::SGPR_352RegClass;
4060 return &AMDGPU::SGPR_384RegClass;
4062 return &AMDGPU::SGPR_512RegClass;
4064 return &AMDGPU::SGPR_1024RegClass;
4072 if (Reg.isVirtual())
4075 RC = getPhysRegBaseClass(Reg);
4081 unsigned Size = getRegSizeInBits(*SRC);
4083 switch (SRC->
getID()) {
4086 case AMDGPU::VS_16_Lo128RegClassID:
4087 return getAllocatableClass(&AMDGPU::VGPR_16_Lo128RegClass);
4088 case AMDGPU::VS_32_Lo128RegClassID:
4089 return getAllocatableClass(&AMDGPU::VGPR_32_Lo128RegClass);
4090 case AMDGPU::VS_32_Lo256RegClassID:
4091 case AMDGPU::VS_64_Lo256RegClassID:
4097 assert(VRC &&
"Invalid register class size");
4103 unsigned Size = getRegSizeInBits(*SRC);
4105 assert(ARC &&
"Invalid register class size");
4111 unsigned Size = getRegSizeInBits(*SRC);
4113 assert(ARC &&
"Invalid register class size");
4119 unsigned Size = getRegSizeInBits(*VRC);
4121 return &AMDGPU::SGPR_32RegClass;
4123 assert(SRC &&
"Invalid register class size");
4130 unsigned SubIdx)
const {
4133 getMatchingSuperRegClass(SuperRC, SubRC, SubIdx);
4134 return MatchRC && MatchRC->
hasSubClassEq(SuperRC) ? MatchRC :
nullptr;
4140 return !ST.hasMFMAInlineLiteralBug();
4161 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::VCC_HI;
4164 if (ReserveHighestRegister) {
4187 unsigned EltSize)
const {
4189 assert(RegBitWidth >= 32 && RegBitWidth <= 1024 && EltSize >= 2);
4191 const unsigned RegHalves = RegBitWidth / 16;
4192 const unsigned EltHalves = EltSize / 2;
4193 assert(RegSplitParts.size() + 1 >= EltHalves);
4195 const std::vector<int16_t> &Parts = RegSplitParts[EltHalves - 1];
4196 const unsigned NumParts = RegHalves / EltHalves;
4198 return ArrayRef(Parts.data(), NumParts);
4204 return Reg.isVirtual() ? MRI.
getRegClass(Reg) : getPhysRegBaseClass(Reg);
4211 return getSubRegisterClass(SrcRC, MO.
getSubReg());
4231 unsigned MinOcc = ST.getOccupancyWithWorkGroupSizes(MF).first;
4232 switch (RC->
getID()) {
4234 return AMDGPUGenRegisterInfo::getRegPressureLimit(RC, MF);
4235 case AMDGPU::VGPR_32RegClassID:
4240 ST.getMaxNumVGPRs(MF));
4241 case AMDGPU::SGPR_32RegClassID:
4242 case AMDGPU::SGPR_LO16RegClassID:
4243 return std::min(ST.getMaxNumSGPRs(MinOcc,
true), ST.getMaxNumSGPRs(MF));
4248 unsigned Idx)
const {
4249 switch (
static_cast<AMDGPU::RegisterPressureSets
>(Idx)) {
4250 case AMDGPU::RegisterPressureSets::VGPR_32:
4251 case AMDGPU::RegisterPressureSets::AGPR_32:
4254 case AMDGPU::RegisterPressureSets::SReg_32:
4263 static const int Empty[] = { -1 };
4265 if (RegPressureIgnoredUnits[
static_cast<unsigned>(RegUnit)])
4268 return AMDGPUGenRegisterInfo::getRegUnitPressureSets(RegUnit);
4283 switch (Hint.first) {
4290 getMatchingSuperReg(Paired, AMDGPU::lo16, &AMDGPU::VGPR_32RegClass);
4291 }
else if (VRM && VRM->
hasPhys(Paired)) {
4292 PairedPhys = getMatchingSuperReg(VRM->
getPhys(Paired), AMDGPU::lo16,
4293 &AMDGPU::VGPR_32RegClass);
4300 Hints.
insert(PairedPhys);
4308 PairedPhys =
TRI->getSubReg(Paired, AMDGPU::lo16);
4309 }
else if (VRM && VRM->
hasPhys(Paired)) {
4310 PairedPhys =
TRI->getSubReg(VRM->
getPhys(Paired), AMDGPU::lo16);
4315 Hints.
insert(PairedPhys);
4325 if (AMDGPU::VGPR_16RegClass.
contains(PhysReg) &&
4340 unsigned &MaxVGPRsForCurrentOccupancy)
const {
4345 unsigned CurrentOccupancy =
4346 ST.getOccupancyWithNumVGPRs(NumAllocatedVGPRs, DynamicVGPRBlockSize);
4347 MaxVGPRsForCurrentOccupancy =
4348 ST.getMaxNumVGPRs(CurrentOccupancy, DynamicVGPRBlockSize);
4350 LLVM_DEBUG(
dbgs() <<
"anti-hints: VGPRs allocated = " << NumAllocatedVGPRs
4351 <<
", RecordedMaxOccupancy = " << RecordedMaxOccupancy
4352 <<
", current occupancy = " << CurrentOccupancy <<
'\n');
4356 if (CurrentOccupancy == 1)
4365 unsigned MaxVGPRsCutOffForRecordedMaxOccupancy =
4366 (ST.getMaxNumVGPRs(RecordedMaxOccupancy, DynamicVGPRBlockSize) * 80) /
4368 unsigned MaxVGPRsCutOffForCurrentOccupancy =
4369 (MaxVGPRsForCurrentOccupancy * 95) / 100;
4371 if (NumAllocatedVGPRs >= MaxVGPRsCutOffForRecordedMaxOccupancy) {
4373 << MaxVGPRsCutOffForRecordedMaxOccupancy
4374 <<
" VGPR cutoff for RecordedMaxOccupancy\n");
4378 if (NumAllocatedVGPRs >= MaxVGPRsCutOffForCurrentOccupancy) {
4380 << MaxVGPRsCutOffForCurrentOccupancy
4381 <<
" VGPR cutoff for current occupancy\n");
4389bool SIRegisterInfo::isRegWithinOccupancyBudget(
4390 MCPhysReg Reg,
unsigned NumVGPRs,
unsigned NumAGPRs,
4391 unsigned MaxVGPRsForCurrentOccupancy)
const {
4398 unsigned RegEndIndex =
4401 unsigned MaxVGPR = NumVGPRs;
4402 unsigned MaxAGPR = NumAGPRs;
4405 MaxAGPR = std::max(MaxAGPR, RegEndIndex);
4407 MaxVGPR = std::max(MaxVGPR, RegEndIndex);
4409 return static_cast<unsigned>(
4411 MaxVGPRsForCurrentOccupancy;
4420 return isAntiHintedReg(Reg, AntiHintedRegUnits);
4426 "SGPR anti-hints are not handled");
4427 unsigned NumVGPRs = 0;
4428 unsigned NumAGPRs = 0;
4430 assert(
Matrix &&
"LiveRegMatrix required to compute occupancy");
4431 assert(RegClassInfo &&
"RegClassInfo required to compute occupancy");
4433 if (
Matrix->isPhysRegUsed(Reg) ||
4439 if (
Matrix->isPhysRegUsed(Reg) ||
4444 unsigned NumAllocatedVGPRs =
4448 unsigned MaxVGPRsForCurrentOccupancy = 0;
4454 auto *BeyondBudgetStart = std::stable_partition(
4456 return isRegWithinOccupancyBudget(Reg, NumVGPRs, NumAGPRs,
4457 MaxVGPRsForCurrentOccupancy);
4460 [[maybe_unused]]
auto *PartitionPoint = std::stable_partition(
4461 CustomOrder.
begin(), BeyondBudgetStart,
4462 [&](
MCPhysReg Reg) { return !isAntiHintedReg(Reg, AntiHintedRegUnits); });
4465 size_t NonAntiHintedCount =
4466 std::distance(CustomOrder.
begin(), PartitionPoint);
4467 size_t AntiHintedCount = std::distance(PartitionPoint, BeyondBudgetStart);
4468 size_t BeyondBudgetCount =
4469 std::distance(BeyondBudgetStart, CustomOrder.
end());
4470 dbgs() <<
"Added " << NonAntiHintedCount
4471 <<
" non-anti-hinted registers first\n"
4472 <<
"Added " << AntiHintedCount
4473 <<
" anti-hinted registers at the end\n"
4474 <<
"Beyond current occupancy budget, left: " << BeyondBudgetCount
4481 return AMDGPU::SGPR30_SGPR31;
4487 switch (RB.
getID()) {
4488 case AMDGPU::VGPRRegBankID:
4490 std::max(ST.useRealTrue16Insts() ? 16u : 32u,
Size));
4491 case AMDGPU::VCCRegBankID:
4494 case AMDGPU::SGPRRegBankID:
4496 case AMDGPU::AGPRRegBankID:
4510 return getAllocatableClass(RC);
4516 return isWave32 ? AMDGPU::VCC_LO : AMDGPU::VCC;
4520 return isWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
4525 return ST.needsAlignedVGPRs() ? &AMDGPU::VReg_64_Align2RegClass
4526 : &AMDGPU::VReg_64RegClass;
4538 if (Reg.isVirtual()) {
4542 LaneBitmask SubLanes = SubReg ? getSubRegIndexLaneMask(SubReg)
4547 if ((S.LaneMask & SubLanes) == SubLanes) {
4548 V = S.getVNInfoAt(UseIdx);
4560 for (MCRegUnit Unit : regunits(Reg.asMCReg())) {
4575 if (!Def || !MDT.dominates(Def, &
Use))
4578 assert(Def->modifiesRegister(Reg,
this));
4584 assert(getRegSizeInBits(*getPhysRegBaseClass(Reg)) <= 32);
4587 {&AMDGPU::VGPR_32RegClass, &AMDGPU::SReg_32RegClass,
4588 &AMDGPU::AGPR_32RegClass}) {
4589 if (
MCPhysReg Super = getMatchingSuperReg(Reg, AMDGPU::lo16, RC))
4592 if (
MCPhysReg Super = getMatchingSuperReg(Reg, AMDGPU::hi16,
4593 &AMDGPU::VGPR_32RegClass)) {
4597 return AMDGPU::NoRegister;
4601 if (!ST.needsAlignedVGPRs())
4612 assert(&RC != &AMDGPU::VS_64RegClass);
4619 return ArrayRef(AMDGPU::SGPR_128RegClass.begin(), ST.getMaxNumSGPRs(MF) / 4);
4624 return ArrayRef(AMDGPU::SGPR_64RegClass.begin(), ST.getMaxNumSGPRs(MF) / 2);
4629 return ArrayRef(AMDGPU::SGPR_32RegClass.begin(), ST.getMaxNumSGPRs(MF));
4634 unsigned SubReg)
const {
4637 return std::min(128u, getSubRegIdxSize(SubReg));
4641 return std::min(32u, getSubRegIdxSize(SubReg));
4650 bool IncludeCalls)
const {
4651 unsigned NumArchVGPRs = ST.getAddressableNumArchVGPRs();
4653 (RC.
getID() == AMDGPU::VGPR_32RegClassID)
4657 if (Reg != AMDGPU::VCC_LO && Reg != AMDGPU::VCC_HI &&
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
AMDGPU Reserve WWM Registers
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static const Function * getParent(const Value *V)
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
static DebugLoc getDebugLoc(MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
Return the first DebugLoc that has line number information, given a range of instructions.
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
This file declares the machine register scavenger class.
static MachineInstrBuilder spillVGPRtoAGPR(const GCNSubtarget &ST, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, int Index, unsigned Lane, unsigned ValueReg, bool IsKill, bool NeedsCFI)
static int getOffenMUBUFStore(unsigned Opc)
static bool wrapsAround32(int64_t LHS, int64_t RHS)
static const TargetRegisterClass * getAnyAGPRClassForBitWidth(unsigned BitWidth)
static bool isSCCLiveInto(const RegScavenger &RS, const MachineInstr &MI)
static int getOffsetMUBUFLoad(unsigned Opc)
static const std::array< unsigned, 17 > SubRegFromChannelTableWidthMap
static unsigned getNumSubRegsForSpillOp(const MachineInstr &MI, const SIInstrInfo *TII)
static cl::opt< bool > EnableSpillCFISavedRegs("amdgpu-spill-cfi-saved-regs", cl::desc("Enable spilling the registers required for CFI emission"), cl::ReallyHidden, cl::init(false))
static void emitUnsupportedError(const Function &Fn, const MachineInstr &MI, const Twine &ErrMsg)
static const TargetRegisterClass * getAlignedAGPRClassForBitWidth(unsigned BitWidth)
static bool buildMUBUFOffsetLoadStore(const GCNSubtarget &ST, MachineFrameInfo &MFI, MachineBasicBlock::iterator MI, int Index, int64_t Offset)
static unsigned getFlatScratchSpillOpcode(const SIInstrInfo *TII, unsigned LoadStoreOp, unsigned EltSize)
static const TargetRegisterClass * getAlignedVGPRClassForBitWidth(unsigned BitWidth)
static int getOffsetMUBUFStore(unsigned Opc)
static const TargetRegisterClass * getAnyVGPRClassForBitWidth(unsigned BitWidth)
static cl::opt< unsigned > StressSGPRLimit("amdgpu-stress-sgpr", cl::Hidden, cl::init(0), cl::desc("Limit SGPRs to N registers by reserving the rest"))
static cl::opt< bool > EnableSpillSGPRToVGPR("amdgpu-spill-sgpr-to-vgpr", cl::desc("Enable spilling SGPRs to VGPRs"), cl::ReallyHidden, cl::init(true))
static const TargetRegisterClass * getAlignedVectorSuperClassForBitWidth(unsigned BitWidth)
static const TargetRegisterClass * getAnyVectorSuperClassForBitWidth(unsigned BitWidth)
static cl::opt< unsigned > StressAGPRLimit("amdgpu-stress-agpr", cl::Hidden, cl::init(0), cl::desc("Limit AGPRs to N registers by reserving the rest"))
static cl::opt< unsigned > StressVGPRLimit("amdgpu-stress-vgpr", cl::Hidden, cl::init(0), cl::desc("Limit VGPRs to N registers by reserving the rest"))
static bool foldingOffsetChangesCarry(const MachineOperand &OtherOp, int64_t Offset, Register FrameReg)
static bool isFIPlusImmOrVGPR(const SIRegisterInfo &TRI, const MachineInstr &MI)
static int getOffenMUBUFLoad(unsigned Opc)
Interface definition for SIRegisterInfo.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
LocallyHashedType DenseMapInfo< LocallyHashedType >::Empty
static const char * getRegisterName(MCRegister Reg)
bool isBottomOfStack() const
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool empty() const
Returns whether there are no bits in this bitvector.
Diagnostic information for unsupported feature in backend.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
void removeAllRegUnitsForPhysReg(MCRegister Reg)
Remove associated live ranges for the register units associated with Reg.
bool hasInterval(Register Reg) const
MachineInstr * getInstructionFromIndex(SlotIndex index) const
Returns the instruction associated with the given index.
MachineDominatorTree & getDomTree()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LiveRange & getRegUnit(MCRegUnit Unit)
Return the live range for register unit Unit.
This class represents the liveness of a register, stack slot, etc.
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
Describe properties that are true of each instruction in the target description file.
MCRegAliasIterator enumerates all registers aliasing Reg.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
unsigned getID() const
getID() - Return the register class ID number.
ArrayRef< MCPhysReg > getRegisters() const
const uint8_t TSFlags
Configurable target specific flags.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
Wrapper class representing physical registers. Should be passed by value.
static MCRegister from(unsigned Val)
Check the provided unsigned value is a valid MCRegister.
Generic base class for all target subtargets.
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool hasCalls() const
Return true if the current function has any function calls.
Align getObjectAlign(int ObjectIdx) const
Return the alignment of the specified stack object.
bool hasStackObjects() const
Return true if there are any stack objects in this function.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
void setAsmPrinterFlag(AsmPrinterFlagTy Flag)
Set a flag for the AsmPrinter.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
const MachinePointerInfo & getPointerInfo() const
Flags getFlags() const
Return the raw flags of the source value,.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
void setImm(int64_t immVal)
LLVM_ABI void setIsRenamable(bool Val=true)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
const RegClassOrRegBank & getRegClassOrRegBank(Register Reg) const
Return the register bank or register class of Reg.
bool isReserved(MCRegister PhysReg) const
isReserved - Returns true when PhysReg is a reserved register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool isAllocatable(MCRegister PhysReg) const
isAllocatable - Returns true when PhysReg belongs to an allocatable register class and it hasn't been...
std::pair< unsigned, Register > getRegAllocationHint(Register VReg) const
getRegAllocationHint - Return the register allocation hint for the specified virtual register.
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI LaneBitmask getMaxLaneMaskForVReg(Register Reg) const
Returns a mask covering all bits that can appear in lane masks of subregisters of the virtual registe...
LLVM_ABI bool isPhysRegUsed(MCRegister PhysReg, bool SkipRegMaskTest=false) const
Return true if the specified register is modified or read in this function.
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Holds all the information related to register banks.
virtual bool isDivergentRegBank(const RegisterBank *RB) const
Returns true if the register bank is considered divergent.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
ArrayRef< MCPhysReg > getOrder(const TargetRegisterClass *RC) const
getOrder - Returns the preferred allocation order for RC.
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
MachineInstr * buildCFIForSGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister SGPR, int64_t Offset) const
Create a CFI index describing a spill of a SGPR to VMEM and build a MachineInstr around it.
MachineInstr * buildCFIForVRegToVRegSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister Reg, const MCRegister RegCopy) const
Create a CFI index describing a spill of the VGPR/AGPR Reg to another VGPR/AGPR RegCopy and build a M...
MachineInstr * buildCFIForVGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister VGPR, int64_t Offset) const
Create a CFI index describing a spill of a VGPR to VMEM and build a MachineInstr around it.
MachineInstr * buildCFIForSGPRToVGPRSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister SGPR, const MCRegister VGPR, const int Lane) const
Create a CFI index describing a spill of an SGPR to a single lane of a VGPR and build a MachineInstr ...
static bool isFLATScratch(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
static bool isVOP3(const MCInstrDesc &Desc)
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
ArrayRef< MCPhysReg > getAGPRSpillVGPRs() const
unsigned getOccupancy() const
MCPhysReg getVGPRToAGPRSpill(int FrameIndex, unsigned Lane) const
Register getLongBranchReservedReg() const
unsigned getDynamicVGPRBlockSize() const
Register getStackPtrOffsetReg() const
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
ArrayRef< MCPhysReg > getVGPRSpillAGPRs() const
ArrayRef< SIRegisterInfo::SpilledReg > getSGPRSpillToVirtualVGPRLanes(int FrameIndex) const
uint32_t getMaskForVGPRBlockOps(Register RegisterBlock) const
Register getSGPRForEXECCopy() const
ArrayRef< SIRegisterInfo::SpilledReg > getSGPRSpillToPhysicalVGPRLanes(int FrameIndex) const
Register getVGPRForAGPRCopy() const
Register getFrameOffsetReg() const
BitVector getPerLaneVGPRMask() const
bool checkFlag(Register Reg, uint8_t Flag) const
void addToSpilledVGPRs(unsigned num)
const ReservedRegSet & getWWMReservedRegs() const
void addToSpilledSGPRs(unsigned num)
Register materializeFrameBaseRegister(MachineBasicBlock *MBB, int FrameIdx, int64_t Offset) const override
int64_t getScratchInstrOffset(const MachineInstr *MI) const
bool isFrameOffsetLegal(const MachineInstr *MI, Register BaseReg, int64_t Offset) const override
const TargetRegisterClass * getCompatibleSubRegClass(const TargetRegisterClass *SuperRC, const TargetRegisterClass *SubRC, unsigned SubIdx) const
Returns a register class which is compatible with SuperRC, such that a subregister exists with class ...
ArrayRef< MCPhysReg > getAllSGPR64(const MachineFunction &MF) const
Return all SGPR64 which satisfy the waves per execution unit requirement of the subtarget.
MCRegister findUnusedRegister(const MachineRegisterInfo &MRI, const TargetRegisterClass *RC, const MachineFunction &MF, bool ReserveHighestVGPR=false) const
Returns a lowest register that is not used at any point in the function.
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
MCPhysReg get32BitRegister(MCPhysReg Reg) const
const uint32_t * getCallPreservedMask(const MachineFunction &MF, CallingConv::ID) const override
void buildSpillLoadStore(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, unsigned LoadStoreOp, int Index, Register ValueReg, bool ValueIsKill, MCRegister ScratchOffsetReg, int64_t InstrOffset, MachineMemOperand *MMO, RegScavenger *RS, LiveRegUnits *LiveUnits=nullptr, bool NeedsCFI=false) const
bool requiresFrameIndexReplacementScavenging(const MachineFunction &MF) const override
bool shouldRealignStack(const MachineFunction &MF) const override
bool restoreSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool OnlyToVGPR=false, bool SpillToPhysVGPRLane=false) const
bool isProperlyAlignedRC(const TargetRegisterClass &RC) const
static bool hasVectorRegisters(const TargetRegisterClass *RC)
const TargetRegisterClass * getEquivalentVGPRClass(const TargetRegisterClass *SRC) const
Register getFrameRegister(const MachineFunction &MF) const override
LLVM_READONLY const TargetRegisterClass * getVectorSuperClassForBitWidth(unsigned BitWidth) const
bool spillEmergencySGPR(MachineBasicBlock::iterator MI, MachineBasicBlock &RestoreMBB, Register SGPR, RegScavenger *RS) const
SIRegisterInfo(const GCNSubtarget &ST)
const uint32_t * getAllVGPRRegMask() const
MCRegister getReturnAddressReg(const MachineFunction &MF) const
const MCPhysReg * getCalleeSavedRegs(const MachineFunction *MF) const override
bool hasBasePointer(const MachineFunction &MF) const
const TargetRegisterClass * getCrossCopyRegClass(const TargetRegisterClass *RC) const override
Returns a legal register class to copy a register in the specified class to or from.
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
ArrayRef< MCPhysReg > getAllSGPR32(const MachineFunction &MF) const
Return all SGPR32 which satisfy the waves per execution unit requirement of the subtarget.
const TargetRegisterClass * getLargestLegalSuperClass(const TargetRegisterClass *RC, const MachineFunction &MF) const override
MCRegister reservedPrivateSegmentBufferReg(const MachineFunction &MF) const
Return the end register initially reserved for the scratch buffer in case spilling is needed.
bool eliminateSGPRToVGPRSpillFrameIndex(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool SpillToPhysVGPRLane=false) const
Special case of eliminateFrameIndex.
bool isVGPR(const MachineRegisterInfo &MRI, Register Reg) const
bool getRegAllocationHints(Register VirtReg, ArrayRef< MCPhysReg > Order, SmallSetVector< MCPhysReg, 16 > &Hints, const MachineFunction &MF, const VirtRegMap *VRM, const LiveRegMatrix *Matrix) const override
bool isAsmClobberable(const MachineFunction &MF, MCRegister PhysReg) const override
LLVM_READONLY const TargetRegisterClass * getAGPRClassForBitWidth(unsigned BitWidth) const
static bool isChainScratchRegister(Register VGPR)
bool requiresRegisterScavenging(const MachineFunction &Fn) const override
bool opCanUseInlineConstant(unsigned OpType) const
const TargetRegisterClass * getRegClassForSizeOnBank(unsigned Size, const RegisterBank &Bank) const
bool isUniformReg(const MachineRegisterInfo &MRI, const RegisterBankInfo &RBI, Register Reg) const override
const uint32_t * getNoPreservedMask() const override
bool shouldApplyAntiHints(const MachineFunction &MF, unsigned NumAllocatedVGPRs, unsigned &MaxVGPRsForCurrentOccupancy) const
StringRef getRegAsmName(MCRegister Reg) const override
MCRegister getAlignedHighSGPRForRC(const MachineFunction &MF, const unsigned Align, const TargetRegisterClass *RC) const
Return the largest available SGPR aligned to Align for the register class RC.
void buildCFIForBlockCSRStore(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register BlockReg, int64_t Offset) const
const TargetRegisterClass * getRegClassForReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getHWRegIndex(MCRegister Reg) const
const MCPhysReg * getCalleeSavedRegsViaCopy(const MachineFunction *MF) const
const uint32_t * getAllVectorRegMask() const
const TargetRegisterClass * getEquivalentAGPRClass(const TargetRegisterClass *SRC) const
void filterAndSortForAntiHintedRegs(Register VirtReg, MutableArrayRef< MCPhysReg > CustomOrder, const BitVector &AntiHintedRegUnits, const MachineFunction &MF, const LiveRegMatrix *Matrix=nullptr, const RegisterClassInfo *RegClassInfo=nullptr) const override
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
const TargetRegisterClass * getRegClassForTypeOnBank(LLT Ty, const RegisterBank &Bank) const
bool opCanUseLiteralConstant(unsigned OpType) const
Register getBaseRegister() const
LLVM_READONLY const TargetRegisterClass * getAlignedLo256VGPRClassForBitWidth(unsigned BitWidth) const
LLVM_READONLY const TargetRegisterClass * getVGPRClassForBitWidth(unsigned BitWidth) const
const TargetRegisterClass * getEquivalentAVClass(const TargetRegisterClass *SRC) const
bool requiresFrameIndexScavenging(const MachineFunction &MF) const override
static bool isVGPRClass(const TargetRegisterClass *RC)
MachineInstr * findReachingDef(Register Reg, unsigned SubReg, MachineInstr &Use, MachineRegisterInfo &MRI, LiveIntervals *LIS) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
const TargetRegisterClass * getEquivalentSGPRClass(const TargetRegisterClass *VRC) const
SmallVector< StringLiteral > getVRegFlagsOfReg(Register Reg, const MachineFunction &MF) const override
LLVM_READONLY const TargetRegisterClass * getDefaultVectorSuperClassForBitWidth(unsigned BitWidth) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
ArrayRef< MCPhysReg > getAllSGPR128(const MachineFunction &MF) const
Return all SGPR128 which satisfy the waves per execution unit requirement of the subtarget.
unsigned getRegPressureSetLimit(const MachineFunction &MF, unsigned Idx) const override
BitVector getReservedRegs(const MachineFunction &MF) const override
bool needsFrameBaseReg(MachineInstr *MI, int64_t Offset) const override
const TargetRegisterClass * getRegClassForOperandReg(const MachineRegisterInfo &MRI, const MachineOperand &MO) const
void addImplicitUsesForBlockCSRLoad(MachineInstrBuilder &MIB, Register BlockReg) const
unsigned getNumUsedPhysRegs(const MachineRegisterInfo &MRI, const TargetRegisterClass &RC, bool IncludeCalls=true) const
const uint32_t * getAllAGPRRegMask() const
const int * getRegUnitPressureSets(MCRegUnit RegUnit) const override
bool isAGPR(const MachineRegisterInfo &MRI, Register Reg) const
bool eliminateFrameIndex(MachineBasicBlock::iterator MI, int SPAdj, unsigned FIOperandNum, RegScavenger *RS) const override
bool spillSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool OnlyToVGPR=false, bool SpillToPhysVGPRLane=false, bool NeedsCFI=false) const
If OnlyToVGPR is true, this will only succeed if this manages to find a free VGPR lane to spill.
MCRegister getExec() const
MCRegister getVCC() const
int64_t getFrameIndexInstrOffset(const MachineInstr *MI, int Idx) const override
bool isVectorSuperClass(const TargetRegisterClass *RC) const
const TargetRegisterClass * getWaveMaskRegClass() const
unsigned getSubRegAlignmentNumBits(const TargetRegisterClass *RC, unsigned SubReg) const
void resolveFrameIndex(MachineInstr &MI, Register BaseReg, int64_t Offset) const override
bool requiresVirtualBaseRegisters(const MachineFunction &Fn) const override
const TargetRegisterClass * getVGPR64Class() const
void buildVGPRSpillLoadStore(SGPRSpillBuilder &SB, int Index, int Offset, bool IsLoad, bool IsKill=true) const
bool isCFISavedRegsSpillEnabled() const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
const TargetRegisterClass * getConstrainedRegClassForReg(Register Reg, const MachineRegisterInfo &MRI) const override
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex - An opaque wrapper around machine indexes.
bool isValid() const
Returns true if this is a valid index.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
SlotIndex replaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
ReplaceMachineInstrInMaps - Replacing a machine instr with a new one in maps used by register allocat...
A SetVector that performs no allocations if smaller than a certain size.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
virtual const TargetRegisterClass * getLargestLegalSuperClass(const TargetRegisterClass *RC, const MachineFunction &) const
Returns the largest super class of RC that is legal to use in the current sub-target and has the same...
virtual bool shouldRealignStack(const MachineFunction &MF) const
True if storage within the function requires the stack pointer to be aligned more than the normal cal...
virtual bool getRegAllocationHints(Register VirtReg, ArrayRef< MCPhysReg > Order, SmallSetVector< MCPhysReg, 16 > &Hints, const MachineFunction &MF, const VirtRegMap *VRM=nullptr, const LiveRegMatrix *Matrix=nullptr) const
Get a list of 'hint' registers that the register allocator should try first when allocating a physica...
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
A Use represents the edge between a Value definition and its users.
VNInfo - Value Number Information.
MCRegister getPhys(Register virtReg) const
returns the physical register mapped to the specified virtual register
bool hasPhys(Register virtReg) const
returns true if the specified virtual register is mapped to a physical register
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ PRIVATE_ADDRESS
Address space for private memory.
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
@ OPERAND_REG_INLINE_AC_FIRST
@ OPERAND_REG_INLINE_AC_LAST
LLVM_READONLY int32_t getFlatScratchInstSVfromSVS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSTfromSS(uint32_t Opcode)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
@ Fast
Attempts to make calls as fast as possible (e.g.
@ C
The default llvm calling convention, compatible with C.
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
@ Renamable
Register that may be renamed.
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
auto reverse(ContainerTy &&C)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
constexpr RegState getDefRegState(bool B)
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
constexpr bool hasRegState(RegState Value, RegState Test)
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ Sub
Subtraction of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
void call_once(once_flag &flag, Function &&F, Args &&... ArgList)
Execute the function specified as a parameter once.
constexpr unsigned BitWidth
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
This class contains a discriminated union of information about pointers in memory operands,...
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
void setMI(MachineBasicBlock *NewMBB, MachineBasicBlock::iterator NewMI)
ArrayRef< int16_t > SplitParts
SIMachineFunctionInfo & MFI
SGPRSpillBuilder(const SIRegisterInfo &TRI, const SIInstrInfo &TII, bool IsWave32, MachineBasicBlock::iterator MI, int Index, RegScavenger *RS)
SGPRSpillBuilder(const SIRegisterInfo &TRI, const SIInstrInfo &TII, bool IsWave32, MachineBasicBlock::iterator MI, Register Reg, bool IsKill, int Index, RegScavenger *RS)
PerVGPRData getPerVGPRData()
MachineBasicBlock::iterator MI
void readWriteTmpVGPR(unsigned Offset, bool IsLoad)
const SIRegisterInfo & TRI
The llvm::once_flag structure.