87#define DEBUG_TYPE "si-wqm"
96 StateStrict = StateStrictWWM | StateStrictWQM,
103 explicit PrintState(
int State) : State(State) {}
109 static const std::pair<char, const char *> Mapping[] = {
110 std::pair(StateWQM,
"WQM"), std::pair(StateStrictWWM,
"StrictWWM"),
111 std::pair(StateStrictWQM,
"StrictWQM"), std::pair(StateExact,
"Exact")};
112 char State = PS.State;
113 for (
auto M : Mapping) {
114 if (State & M.first) {
131 char MarkedStates = 0;
138 char InitialState = 0;
139 bool NeedsLowering =
false;
151class SIWholeQuadMode {
188 std::vector<WorkItem> &Worklist);
191 std::vector<WorkItem> &Worklist);
193 std::vector<WorkItem> &Worklist);
195 std::vector<WorkItem> &Worklist);
196 char scanInstructions(
MachineFunction &MF, std::vector<WorkItem> &Worklist,
198 void propagateInstruction(
MachineInstr &
MI, std::vector<WorkItem> &Worklist);
213 Register SaveOrig,
char StrictStateNeeded);
216 char NonStrictState,
char CurrentStrictState);
225 bool lowerLiveMaskQueries();
226 bool lowerCopyInstrs();
227 bool lowerKillInstrs(
bool IsWQM);
241 StringRef getPassName()
const override {
return "SI Whole Quad Mode"; }
258char SIWholeQuadModeLegacy::ID = 0;
272 for (
const auto &BII : Blocks) {
275 <<
" InNeeds = " << PrintState(BII.second.InNeeds)
276 <<
", Needs = " << PrintState(BII.second.Needs)
277 <<
", OutNeeds = " << PrintState(BII.second.OutNeeds) <<
"\n\n";
280 auto III = Instructions.find(&
MI);
281 if (III != Instructions.end()) {
282 dbgs() <<
" " <<
MI <<
" Needs = " << PrintState(III->second.Needs)
283 <<
", OutNeeds = " << PrintState(III->second.OutNeeds) <<
'\n';
291 std::vector<WorkItem> &Worklist) {
292 InstrInfo &
II = Instructions[&
MI];
294 assert(!(Flag & StateExact) && Flag != 0);
297 II.MarkedStates |= Flag;
303 Flag &=
~II.Disabled;
307 if ((
II.Needs & Flag) == Flag)
312 Worklist.emplace_back(&
MI);
316void SIWholeQuadMode::markDefs(
const MachineInstr &
UseMI,
LiveRange &LR,
317 VirtRegOrUnit VRegOrUnit,
unsigned SubReg,
318 char Flag, std::vector<WorkItem> &Worklist) {
328 const LaneBitmask UseLanes =
329 SubReg ?
TRI->getSubRegIndexLaneMask(SubReg)
340 LaneBitmask DefinedLanes;
342 PhiEntry(
const VNInfo *Phi,
unsigned PredIdx, LaneBitmask DefinedLanes)
343 :
Phi(
Phi), PredIdx(PredIdx), DefinedLanes(DefinedLanes) {}
345 using VisitKey = std::pair<const VNInfo *, LaneBitmask>;
347 SmallSet<VisitKey, 4> Visited;
348 LaneBitmask DefinedLanes;
349 unsigned NextPredIdx = 0;
351 const VNInfo *NextValue =
nullptr;
352 const VisitKey
Key(
Value, DefinedLanes);
359 if (
Value->isPHIDef()) {
362 assert(
MBB &&
"Phi-def has no defining MBB");
365 unsigned Idx = NextPredIdx;
368 for (; PI != PE && !NextValue; ++PI, ++
Idx) {
370 if (!Visited.
count(VisitKey(VN, DefinedLanes)))
380 assert(
MI &&
"Def has no defining instruction");
385 for (
const MachineOperand &
Op :
MI->all_defs()) {
390 LaneBitmask OpLanes =
392 :
TRI->getSubRegIndexLaneMask(
Op.getSubReg());
393 LaneBitmask Overlap = (UseLanes & OpLanes);
396 HasDef |= Overlap.
any();
399 DefinedLanes |= OpLanes;
403 if ((DefinedLanes & UseLanes) != UseLanes) {
406 if (
const VNInfo *VN = LRQ.
valueIn()) {
407 if (!Visited.
count(VisitKey(VN, DefinedLanes)))
414 markInstruction(*
MI, Flag, Worklist);
417 markInstruction(*
MI, Flag, Worklist);
421 if (!NextValue && !PhiStack.
empty()) {
424 NextValue =
Entry.Phi;
425 NextPredIdx =
Entry.PredIdx;
426 DefinedLanes =
Entry.DefinedLanes;
434void SIWholeQuadMode::markOperand(
const MachineInstr &
MI,
435 const MachineOperand &
Op,
char Flag,
436 std::vector<WorkItem> &Worklist) {
443 case AMDGPU::EXEC_LO:
453 markDefs(
MI, LR, VirtRegOrUnit(
Reg),
Op.getSubReg(), Flag, Worklist);
462 markDefs(
MI, LR, VirtRegOrUnit(Unit), AMDGPU::NoSubRegister, Flag,
469void SIWholeQuadMode::markInstructionUses(
const MachineInstr &
MI,
char Flag,
470 std::vector<WorkItem> &Worklist) {
471 LLVM_DEBUG(
dbgs() <<
"markInstructionUses " << PrintState(Flag) <<
": "
474 for (
const MachineOperand &Use :
MI.all_uses())
475 markOperand(
MI, Use, Flag, Worklist);
480char SIWholeQuadMode::scanInstructions(
483 char GlobalFlags = 0;
485 SmallVector<MachineInstr *, 4> SoftWQMInstrs;
486 bool HasImplicitDerivatives =
493 ReversePostOrderTraversal<MachineFunction *> RPOT(&MF);
494 for (MachineBasicBlock *
MBB : RPOT) {
495 BlockInfo &BBI = Blocks[
MBB];
497 for (MachineInstr &
MI : *
MBB) {
498 InstrInfo &III = Instructions[&
MI];
499 unsigned Opcode =
MI.getOpcode();
502 if (
TII->isWQM(Opcode)) {
507 if (ST->hasExtendedImageInsts() && HasImplicitDerivatives) {
511 markInstructionUses(
MI, StateWQM, Worklist);
512 GlobalFlags |= StateWQM;
514 }
else if (Opcode == AMDGPU::WQM) {
518 LowerToCopyInstrs.insert(&
MI);
519 }
else if (Opcode == AMDGPU::SOFT_WQM) {
520 LowerToCopyInstrs.insert(&
MI);
522 }
else if (Opcode == AMDGPU::STRICT_WWM) {
526 markInstructionUses(
MI, StateStrictWWM, Worklist);
527 GlobalFlags |= StateStrictWWM;
529 }
else if (Opcode == AMDGPU::STRICT_WQM ||
530 TII->isDualSourceBlendEXP(
MI)) {
534 markInstructionUses(
MI, StateStrictWQM, Worklist);
535 GlobalFlags |= StateStrictWQM;
537 if (Opcode == AMDGPU::STRICT_WQM) {
543 BBI.Needs |= StateExact;
544 if (!(BBI.InNeeds & StateExact)) {
545 BBI.InNeeds |= StateExact;
546 Worklist.emplace_back(
MBB);
548 GlobalFlags |= StateExact;
549 III.Disabled = StateWQM | StateStrict;
551 }
else if (Opcode == AMDGPU::LDS_PARAM_LOAD ||
552 Opcode == AMDGPU::DS_PARAM_LOAD ||
553 Opcode == AMDGPU::LDS_DIRECT_LOAD ||
554 Opcode == AMDGPU::DS_DIRECT_LOAD) {
557 III.Needs |= StateStrictWQM;
558 GlobalFlags |= StateStrictWQM;
559 }
else if (Opcode == AMDGPU::V_SET_INACTIVE_B32) {
561 III.Disabled = StateStrict;
562 MachineOperand &Inactive =
MI.getOperand(4);
563 if (Inactive.
isReg()) {
564 if (Inactive.
isUndef() &&
MI.getOperand(3).getImm() == 0)
565 LowerToCopyInstrs.insert(&
MI);
567 markOperand(
MI, Inactive, StateStrictWWM, Worklist);
570 BBI.NeedsLowering =
true;
571 }
else if (
TII->isDisableWQM(
MI)) {
572 BBI.Needs |= StateExact;
573 if (!(BBI.InNeeds & StateExact)) {
574 BBI.InNeeds |= StateExact;
575 Worklist.emplace_back(
MBB);
577 GlobalFlags |= StateExact;
578 III.Disabled = StateWQM | StateStrict;
579 }
else if (Opcode == AMDGPU::SI_PS_LIVE ||
580 Opcode == AMDGPU::SI_LIVE_MASK) {
582 }
else if (Opcode == AMDGPU::SI_KILL_I1_TERMINATOR ||
583 Opcode == AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR ||
584 Opcode == AMDGPU::SI_DEMOTE_I1) {
586 BBI.NeedsLowering =
true;
587 }
else if (Opcode == AMDGPU::SI_INIT_EXEC ||
588 Opcode == AMDGPU::SI_INIT_EXEC_FROM_INPUT ||
589 Opcode == AMDGPU::SI_INIT_WHOLE_WAVE) {
591 }
else if (WQMOutputs) {
596 for (
const MachineOperand &MO :
MI.defs()) {
599 TRI->hasVectorRegisters(
TRI->getPhysRegBaseClass(
Reg))) {
606 if (
TII->hasUnwantedEffectsWhenEXECEmpty(
MI)) {
607 for (
auto &
Op :
MI.uses()) {
610 if (!
TRI->isVectorRegister(*MRI,
Op.getReg()))
619 markInstruction(
MI, Flags, Worklist);
620 GlobalFlags |=
Flags;
629 if (GlobalFlags & StateWQM) {
630 for (MachineInstr *
MI : SetInactiveInstrs)
631 markInstruction(*
MI, StateWQM, Worklist);
632 for (MachineInstr *
MI : SoftWQMInstrs)
633 markInstruction(*
MI, StateWQM, Worklist);
639void SIWholeQuadMode::propagateInstruction(MachineInstr &
MI,
640 std::vector<WorkItem>& Worklist) {
642 InstrInfo
II = Instructions[&
MI];
643 BlockInfo &BI = Blocks[
MBB];
647 if ((
II.OutNeeds & StateWQM) && !(
II.Disabled & StateWQM) &&
648 (
MI.isTerminator() || (
TII->usesVM_CNT(
MI) &&
MI.mayStore()))) {
649 Instructions[&
MI].Needs = StateWQM;
654 if (
II.Needs & StateWQM) {
655 BI.Needs |= StateWQM;
656 if (!(BI.InNeeds & StateWQM)) {
657 BI.InNeeds |= StateWQM;
658 Worklist.emplace_back(
MBB);
663 if (MachineInstr *PrevMI =
MI.getPrevNode()) {
664 char InNeeds = (
II.Needs & ~StateStrict) |
II.OutNeeds;
665 if (!PrevMI->isPHI()) {
666 InstrInfo &PrevII = Instructions[PrevMI];
667 if ((PrevII.OutNeeds | InNeeds) != PrevII.OutNeeds) {
668 PrevII.OutNeeds |= InNeeds;
669 Worklist.emplace_back(PrevMI);
678 markInstructionUses(
MI,
II.Needs, Worklist);
682 if (
II.Needs & StateStrictWWM)
683 BI.Needs |= StateStrictWWM;
684 if (
II.Needs & StateStrictWQM)
685 BI.Needs |= StateStrictWQM;
688void SIWholeQuadMode::propagateBlock(MachineBasicBlock &
MBB,
689 std::vector<WorkItem>& Worklist) {
690 BlockInfo BI = Blocks[&
MBB];
695 InstrInfo &LastII = Instructions[LastMI];
696 if ((LastII.OutNeeds | BI.OutNeeds) != LastII.OutNeeds) {
697 LastII.OutNeeds |= BI.OutNeeds;
698 Worklist.emplace_back(LastMI);
704 BlockInfo &PredBI = Blocks[Pred];
705 if ((PredBI.OutNeeds | BI.InNeeds) == PredBI.OutNeeds)
708 PredBI.OutNeeds |= BI.InNeeds;
709 PredBI.InNeeds |= BI.InNeeds;
710 Worklist.emplace_back(Pred);
715 BlockInfo &SuccBI = Blocks[Succ];
716 if ((SuccBI.InNeeds | BI.OutNeeds) == SuccBI.InNeeds)
719 SuccBI.InNeeds |= BI.OutNeeds;
720 Worklist.emplace_back(Succ);
725 std::vector<WorkItem> Worklist;
727 char GlobalFlags = scanInstructions(MF, Worklist, ExeczSideEffectInstrs);
729 while (!Worklist.empty()) {
730 WorkItem WI = Worklist.back();
734 propagateInstruction(*WI.MI, Worklist);
736 propagateBlock(*WI.MBB, Worklist);
738 if (Worklist.empty()) {
744 for (
auto *
MI : ExeczSideEffectInstrs) {
745 InstrInfo
II = Instructions[
MI];
746 if (
II.OutNeeds & StateWQM)
747 markInstructionUses(*
MI, StateWQM, Worklist);
751 ExeczSideEffectInstrs.clear();
759SIWholeQuadMode::saveSCC(MachineBasicBlock &
MBB,
766 MachineInstr *Restore =
777void SIWholeQuadMode::splitBlock(MachineInstr *TermMI) {
778 MachineBasicBlock *BB = TermMI->
getParent();
782 MachineBasicBlock *SplitBB =
783 BB->
splitAt(*TermMI,
true, LIS);
787 unsigned NewOpcode = 0;
789 case AMDGPU::S_AND_B32:
790 NewOpcode = AMDGPU::S_AND_B32_term;
792 case AMDGPU::S_AND_B64:
793 NewOpcode = AMDGPU::S_AND_B64_term;
795 case AMDGPU::S_MOV_B32:
796 NewOpcode = AMDGPU::S_MOV_B32_term;
798 case AMDGPU::S_MOV_B64:
799 NewOpcode = AMDGPU::S_MOV_B64_term;
801 case AMDGPU::S_ANDN2_B32:
802 NewOpcode = AMDGPU::S_ANDN2_B32_term;
804 case AMDGPU::S_ANDN2_B64:
805 NewOpcode = AMDGPU::S_ANDN2_B64_term;
819 for (MachineBasicBlock *Succ : SplitBB->
successors()) {
820 DTUpdates.
push_back({DomTreeT::Insert, SplitBB, Succ});
821 DTUpdates.
push_back({DomTreeT::Delete, BB, Succ});
823 DTUpdates.
push_back({DomTreeT::Insert, BB, SplitBB});
831MachineInstr *SIWholeQuadMode::lowerKillF32(MachineInstr &
MI) {
846 switch (
MI.getOperand(2).getImm()) {
848 Opcode = AMDGPU::V_CMP_LG_F32_e64;
851 Opcode = AMDGPU::V_CMP_GE_F32_e64;
854 Opcode = AMDGPU::V_CMP_GT_F32_e64;
857 Opcode = AMDGPU::V_CMP_LE_F32_e64;
860 Opcode = AMDGPU::V_CMP_LT_F32_e64;
863 Opcode = AMDGPU::V_CMP_EQ_F32_e64;
866 Opcode = AMDGPU::V_CMP_O_F32_e64;
869 Opcode = AMDGPU::V_CMP_U_F32_e64;
873 Opcode = AMDGPU::V_CMP_NEQ_F32_e64;
877 Opcode = AMDGPU::V_CMP_NLT_F32_e64;
881 Opcode = AMDGPU::V_CMP_NLE_F32_e64;
885 Opcode = AMDGPU::V_CMP_NGT_F32_e64;
889 Opcode = AMDGPU::V_CMP_NGE_F32_e64;
893 Opcode = AMDGPU::V_CMP_NLG_F32_e64;
902 MachineInstr *VcmpMI;
903 const MachineOperand &Op0 =
MI.getOperand(0);
904 const MachineOperand &Op1 =
MI.getOperand(1);
920 MachineInstr *MaskUpdateMI =
927 MachineInstr *EarlyTermMI =
930 MachineInstr *ExecMaskMI =
951MachineInstr *SIWholeQuadMode::lowerKillI1(MachineInstr &
MI,
bool IsWQM) {
957 MachineInstr *MaskUpdateMI =
nullptr;
959 const bool IsDemote = IsWQM && (
MI.getOpcode() == AMDGPU::SI_DEMOTE_I1);
960 const MachineOperand &
Op =
MI.getOperand(0);
961 int64_t KillVal =
MI.getOperand(1).getImm();
962 MachineInstr *ComputeKilledMaskMI =
nullptr;
968 if (
Op.getImm() == KillVal) {
975 bool IsLastTerminator = std::next(
MI.getIterator()) ==
MBB.
end();
976 if (!IsLastTerminator) {
1009 MachineInstr *EarlyTermMI =
1014 MachineInstr *NewTerm;
1015 MachineInstr *WQMMaskMI =
nullptr;
1032 }
else if (!IsWQM) {
1052 if (ComputeKilledMaskMI)
1075void SIWholeQuadMode::lowerBlock(MachineBasicBlock &
MBB, BlockInfo &BI) {
1076 if (!BI.NeedsLowering)
1081 SmallVector<MachineInstr *, 4> SplitPoints;
1083 char State = BI.InitialState;
1087 auto MIState = StateTransition.find(&
MI);
1088 if (MIState != StateTransition.end())
1089 State = MIState->second;
1091 MachineInstr *SplitPoint =
nullptr;
1092 switch (
MI.getOpcode()) {
1093 case AMDGPU::SI_DEMOTE_I1:
1094 case AMDGPU::SI_KILL_I1_TERMINATOR:
1095 SplitPoint = lowerKillI1(
MI, State == StateWQM);
1097 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
1098 SplitPoint = lowerKillF32(
MI);
1100 case AMDGPU::ENTER_STRICT_WWM:
1101 ActiveLanesReg =
MI.getOperand(0).getReg();
1103 case AMDGPU::EXIT_STRICT_WWM:
1106 case AMDGPU::V_SET_INACTIVE_B32:
1107 if (ActiveLanesReg) {
1108 LiveInterval &LI = LIS->
getInterval(
MI.getOperand(5).getReg());
1110 MI.getOperand(5).setReg(ActiveLanesReg);
1113 assert(State == StateExact || State == StateWQM);
1124 for (MachineInstr *
MI : SplitPoints)
1144 SlotIndex FirstIdx = FirstNonDbg != MBBE
1149 SlotIndex
Idx = PreferLast ? LastIdx : FirstIdx;
1150 const LiveRange::Segment *S;
1159 if (
Next < FirstIdx)
1164 assert(EndMI &&
"Segment does not end on valid instruction");
1188 bool IsExecDef =
false;
1189 for (
const MachineOperand &MO :
MBBI->all_defs()) {
1191 MO.getReg() == AMDGPU::EXEC_LO || MO.getReg() == AMDGPU::EXEC;
1205void SIWholeQuadMode::toExact(MachineBasicBlock &
MBB,
1210 bool IsTerminator = Before ==
MBB.
end();
1211 if (!IsTerminator) {
1213 if (FirstTerm !=
MBB.
end()) {
1216 IsTerminator = BeforeIdx > FirstTermIdx;
1239 StateTransition[
MI] = StateExact;
1242void SIWholeQuadMode::toWQM(MachineBasicBlock &
MBB,
1258 StateTransition[
MI] = StateWQM;
1261void SIWholeQuadMode::toStrictMode(MachineBasicBlock &
MBB,
1263 Register SaveOrig,
char StrictStateNeeded) {
1266 assert(StrictStateNeeded == StateStrictWWM ||
1267 StrictStateNeeded == StateStrictWQM);
1271 if (StrictStateNeeded == StateStrictWWM) {
1281 StateTransition[
MI] = StrictStateNeeded;
1284void SIWholeQuadMode::fromStrictMode(MachineBasicBlock &
MBB,
1286 Register SavedOrig,
char NonStrictState,
1287 char CurrentStrictState) {
1291 assert(CurrentStrictState == StateStrictWWM ||
1292 CurrentStrictState == StateStrictWQM);
1296 if (CurrentStrictState == StateStrictWWM) {
1306 StateTransition[
MI] = NonStrictState;
1309void SIWholeQuadMode::processBlock(MachineBasicBlock &
MBB, BlockInfo &BI,
1313 if (!IsEntry && BI.Needs == StateWQM && BI.OutNeeds != StateExact) {
1314 BI.InitialState = StateWQM;
1323 bool WQMFromExec = IsEntry;
1324 char State = (IsEntry || !(BI.InNeeds & StateWQM)) ? StateExact : StateWQM;
1325 char NonStrictState = 0;
1331 if (
II != IE &&
II->getOpcode() == AMDGPU::COPY &&
1332 II->getOperand(1).getReg() == LMC.
ExecReg)
1347 BI.InitialState = State;
1349 for (
unsigned Idx = 0;; ++
Idx) {
1351 char Needs = StateExact | StateWQM;
1357 if (FirstStrict == IE)
1361 if (IsEntry && Idx == 0 && (BI.InNeeds & StateWQM))
1367 MachineInstr &
MI = *
II;
1369 if (
MI.isTerminator() ||
TII->mayReadEXEC(*MRI,
MI)) {
1370 auto III = Instructions.find(&
MI);
1371 if (III != Instructions.end()) {
1372 if (III->second.Needs & StateStrictWWM)
1373 Needs = StateStrictWWM;
1374 else if (III->second.Needs & StateStrictWQM)
1375 Needs = StateStrictWQM;
1376 else if (III->second.Needs & StateWQM)
1379 Needs &= ~III->second.Disabled;
1380 OutNeeds = III->second.OutNeeds;
1385 Needs = StateExact | StateWQM | StateStrict;
1389 if (
MI.isBranch() && OutNeeds == StateExact)
1395 if (BI.OutNeeds & StateWQM)
1397 else if (BI.OutNeeds == StateExact)
1400 Needs = StateWQM | StateExact;
1404 if (!(Needs & State)) {
1406 if (State == StateStrictWWM || Needs == StateStrictWWM ||
1407 State == StateStrictWQM || Needs == StateStrictWQM) {
1409 First = FirstStrict;
1416 bool SaveSCC =
false;
1419 case StateStrictWWM:
1420 case StateStrictWQM:
1424 SaveSCC = (Needs & StateStrict) || ((Needs & StateWQM) && WQMFromExec);
1428 SaveSCC = !(Needs & StateWQM);
1434 char StartState = State & StateStrict ? NonStrictState : State;
1436 StartState == StateWQM && (Needs & StateExact) && !(Needs & StateWQM);
1437 bool ExactToWQM = StartState == StateExact && (Needs & StateWQM) &&
1438 !(Needs & StateExact);
1439 bool PreferLast = Needs == StateWQM;
1444 if ((WQMToExact && (OutNeeds & StateWQM)) || ExactToWQM) {
1446 if (
TII->hasUnwantedEffectsWhenEXECEmpty(*
I)) {
1447 PreferLast = WQMToExact;
1453 prepareInsertion(
MBB,
First,
II, PreferLast, SaveSCC);
1455 if (State & StateStrict) {
1456 assert(State == StateStrictWWM || State == StateStrictWQM);
1457 assert(SavedNonStrictReg);
1458 fromStrictMode(
MBB, Before, SavedNonStrictReg, NonStrictState, State);
1461 SavedNonStrictReg = 0;
1462 State = NonStrictState;
1465 if (Needs & StateStrict) {
1466 NonStrictState = State;
1467 assert(Needs == StateStrictWWM || Needs == StateStrictWQM);
1468 assert(!SavedNonStrictReg);
1471 toStrictMode(
MBB, Before, SavedNonStrictReg, Needs);
1475 if (!WQMFromExec && (OutNeeds & StateWQM)) {
1480 toExact(
MBB, Before, SavedWQMReg);
1482 }
else if (ExactToWQM) {
1483 assert(WQMFromExec == (SavedWQMReg == 0));
1485 toWQM(
MBB, Before, SavedWQMReg);
1501 if (Needs != (StateExact | StateWQM | StateStrict)) {
1502 if (Needs != (StateExact | StateWQM))
1513 assert(!SavedNonStrictReg);
1516bool SIWholeQuadMode::lowerLiveMaskQueries() {
1517 for (MachineInstr *
MI : LiveMaskQueries) {
1521 MachineInstr *
Copy =
1526 MI->eraseFromParent();
1528 return !LiveMaskQueries.empty();
1531bool SIWholeQuadMode::lowerCopyInstrs() {
1532 for (MachineInstr *
MI : LowerToMovInstrs) {
1533 assert(
MI->getNumExplicitOperands() == 2);
1538 TRI->getRegClassForOperandReg(*MRI,
MI->getOperand(0));
1539 if (
TRI->isVGPRClass(regClass)) {
1540 const unsigned MovOp =
TII->getMovOpcode(regClass);
1541 MI->setDesc(
TII->get(MovOp));
1545 assert(
any_of(
MI->implicit_operands(), [](
const MachineOperand &MO) {
1546 return MO.isUse() && MO.getReg() == AMDGPU::EXEC;
1552 if (
MI->getOperand(0).isEarlyClobber()) {
1554 MI->getOperand(0).setIsEarlyClobber(
false);
1557 int Index =
MI->findRegisterUseOperandIdx(AMDGPU::EXEC,
nullptr);
1558 while (Index >= 0) {
1559 MI->removeOperand(Index);
1560 Index =
MI->findRegisterUseOperandIdx(AMDGPU::EXEC,
nullptr);
1562 MI->setDesc(
TII->get(AMDGPU::COPY));
1566 for (MachineInstr *
MI : LowerToCopyInstrs) {
1569 if (
MI->getOpcode() == AMDGPU::V_SET_INACTIVE_B32) {
1570 assert(
MI->getNumExplicitOperands() == 6);
1572 LiveInterval *RecomputeLI =
nullptr;
1573 if (
MI->getOperand(4).isReg())
1574 RecomputeLI = &LIS->
getInterval(
MI->getOperand(4).getReg());
1576 MI->removeOperand(5);
1577 MI->removeOperand(4);
1578 MI->removeOperand(3);
1579 MI->removeOperand(1);
1584 assert(
MI->getNumExplicitOperands() == 2);
1587 unsigned CopyOp =
MI->getOperand(1).isReg()
1588 ? (unsigned)AMDGPU::COPY
1589 :
TII->getMovOpcode(
TRI->getRegClassForOperandReg(
1590 *MRI,
MI->getOperand(0)));
1591 MI->setDesc(
TII->get(CopyOp));
1594 return !LowerToCopyInstrs.empty() || !LowerToMovInstrs.empty();
1597bool SIWholeQuadMode::lowerKillInstrs(
bool IsWQM) {
1598 for (MachineInstr *
MI : KillInstrs) {
1599 MachineInstr *SplitPoint =
nullptr;
1600 switch (
MI->getOpcode()) {
1601 case AMDGPU::SI_DEMOTE_I1:
1602 case AMDGPU::SI_KILL_I1_TERMINATOR:
1603 SplitPoint = lowerKillI1(*
MI, IsWQM);
1605 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
1606 SplitPoint = lowerKillF32(*
MI);
1612 return !KillInstrs.empty();
1615void SIWholeQuadMode::lowerInitExec(MachineInstr &
MI) {
1618 if (
MI.getOpcode() == AMDGPU::SI_INIT_WHOLE_WAVE) {
1620 "init whole wave not in entry block");
1634 MI.eraseFromParent();
1643 if (
MI.getOpcode() == AMDGPU::SI_INIT_EXEC) {
1647 .
addImm(
MI.getOperand(0).getImm());
1652 MI.eraseFromParent();
1663 Register InputReg =
MI.getOperand(0).getReg();
1664 MachineInstr *FirstMI = &*
MBB->
begin();
1666 MachineInstr *DefInstr = MRI->
getVRegDef(InputReg);
1669 if (DefInstr != FirstMI) {
1688 auto BfeMI =
BuildMI(*
MBB, FirstMI,
DL,
TII->get(AMDGPU::S_BFE_U32), CountReg)
1690 .
addImm((
MI.getOperand(1).getImm() & Mask) | 0x70000)
1695 auto CmpMI =
BuildMI(*
MBB, FirstMI,
DL,
TII->get(AMDGPU::S_CMP_EQ_U32))
1696 .
addReg(CountReg, RegState::Kill)
1702 MI.eraseFromParent();
1707 MI.eraseFromParent();
1722SIWholeQuadMode::lowerInitExecInstrs(MachineBasicBlock &Entry,
bool &
Changed) {
1725 for (MachineInstr *
MI : InitExecInstrs) {
1729 if (
MI->getParent() == &Entry)
1730 InsertPt = std::next(
MI->getIterator());
1741 <<
" ------------- \n");
1744 Instructions.clear();
1746 LiveMaskQueries.clear();
1747 LowerToCopyInstrs.clear();
1748 LowerToMovInstrs.clear();
1750 InitExecInstrs.clear();
1751 SetInactiveInstrs.
clear();
1752 StateTransition.clear();
1763 const bool HasLiveMaskQueries = !LiveMaskQueries.empty();
1764 const bool HasWaveModes = GlobalFlags & ~StateExact;
1765 const bool HasKills = !KillInstrs.empty();
1766 const bool UsesWQM = GlobalFlags & StateWQM;
1767 if (HasKills || UsesWQM || (HasWaveModes && HasLiveMaskQueries)) {
1778 for (MachineInstr *
MI : SetInactiveInstrs) {
1779 if (LowerToCopyInstrs.contains(
MI))
1781 auto &
Info = Instructions[
MI];
1782 if (
Info.MarkedStates & StateStrict) {
1783 Info.Needs |= StateStrictWWM;
1784 Info.Disabled &= ~StateStrictWWM;
1785 Blocks[
MI->getParent()].Needs |= StateStrictWWM;
1788 LowerToCopyInstrs.insert(
MI);
1794 Changed |= lowerLiveMaskQueries();
1797 if (!HasWaveModes) {
1799 Changed |= lowerKillInstrs(
false);
1800 }
else if (GlobalFlags == StateWQM) {
1807 lowerKillInstrs(
true);
1811 if (GlobalFlags & StateWQM)
1812 Blocks[&
Entry].InNeeds |= StateWQM;
1814 for (
auto &BII : Blocks)
1815 processBlock(*BII.first, BII.second, BII.first == &Entry);
1817 for (
auto &BII : Blocks)
1818 lowerBlock(*BII.first, BII.second);
1823 if (LiveMaskReg != LMC.
ExecReg)
1832 if (!KillInstrs.empty() || !InitExecInstrs.empty())
1838bool SIWholeQuadModeLegacy::runOnMachineFunction(
MachineFunction &MF) {
1839 LiveIntervals *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
1840 auto *MDTWrapper = getAnalysisIfAvailable<MachineDominatorTreeWrapperPass>();
1841 MachineDominatorTree *MDT = MDTWrapper ? &MDTWrapper->getDomTree() :
nullptr;
1843 getAnalysisIfAvailable<MachinePostDominatorTreeWrapperPass>();
1844 MachinePostDominatorTree *PDT =
1845 PDTWrapper ? &PDTWrapper->getPostDomTree() :
nullptr;
1846 SIWholeQuadMode Impl(MF, LIS, MDT, PDT);
1847 return Impl.run(MF);
1860 SIWholeQuadMode Impl(MF, LIS, MDT, PDT);
MachineInstrBuilder & UseMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static void analyzeFunction(Function &Fn, const DataLayout &Layout, FunctionVarLocsBuilder *FnVarLocs)
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
Register const TargetRegisterInfo * TRI
Promote Memory to Register
uint64_t IntrinsicInst * II
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
static void splitBlock(MachineBasicBlock &MBB, MachineInstr &MI, MachineDominatorTree *MDT, MachineLoopInfo *MLI)
SI Optimize VGPR LiveRange
unsigned getWavefrontSize() const
const unsigned AndSaveExecTermOpc
const unsigned AndTermOpc
static const LaneMaskConstants & get(const GCNSubtarget &ST)
const unsigned OrSaveExecOpc
const unsigned AndSaveExecOpc
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
void applyUpdates(ArrayRef< UpdateType > Updates)
Inform the dominator tree about a sequence of CFG edge insertions and deletions and perform a batch u...
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
void removeAllRegUnitsForPhysReg(MCRegister Reg)
Remove associated live ranges for the register units associated with Reg.
MachineInstr * getInstructionFromIndex(SlotIndex index) const
Returns the instruction associated with the given index.
SlotIndex InsertMachineInstrInMaps(MachineInstr &MI)
LLVM_ABI void handleMove(MachineInstr &MI, bool UpdateFlags=false)
Call this method to notify LiveIntervals that instruction MI has been moved within a basic block.
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
void RemoveMachineInstrFromMaps(MachineInstr &MI)
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
void removeInterval(Register Reg)
Interval removal.
LiveRange & getRegUnit(MCRegUnit Unit)
Return the live range for register unit Unit.
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
LiveInterval & createAndComputeVirtRegInterval(Register Reg)
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
VNInfo * valueIn() const
Return the value that is live-in to the instruction.
This class represents the liveness of a register, stack slot, etc.
const Segment * getSegmentContaining(SlotIndex Idx) const
Return the segment that contains the specified index, or null if there is none.
LiveQueryResult Query(SlotIndex Idx) const
Query Liveness at Idx.
VNInfo * getVNInfoBefore(SlotIndex Idx) const
getVNInfoBefore - Return the VNInfo that is live up to but not necessarily including Idx,...
static MCRegister from(unsigned Val)
Check the provided unsigned value is a valid MCRegister.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
succ_iterator succ_begin()
MachineInstr * remove(MachineInstr *I)
Remove the unbundled instruction from the instruction list without deleting it.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
unsigned succ_size() const
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI DebugLoc findDebugLoc(instr_iterator MBBI)
Find the next valid DebugLoc starting at MBBI, skipping any debug instructions.
pred_iterator pred_begin()
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
instr_iterator instr_end()
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
reverse_iterator rbegin()
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
Analysis pass which computes a MachineDominatorTree.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
void dump() const
dump - Print the current MachineFunction to cerr, useful for debugger use.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineBasicBlock & front() const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI MachineInstr * removeFromParent()
Unlink 'this' from the containing basic block, and return it without deleting it.
const MachineBasicBlock * getParent() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
Register getReg() const
getReg - Returns the register number.
MachinePostDominatorTree - an analysis pass wrapper for DominatorTree used to compute the post-domina...
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI LaneBitmask getMaxLaneMaskForVReg(Register Reg) const
Returns a mask covering all bits that can appear in lane masks of subregisters of the virtual registe...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
This class implements a map that also provides access to all stored values in a deterministic order.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Wrapper class representing virtual and physical registers.
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
SlotIndex getBaseIndex() const
Returns the base index for associated with this index.
A SetVector that performs no allocations if smaller than a certain size.
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Wrapper class representing a virtual register or register unit.
constexpr bool isVirtualReg() const
constexpr Register asVirtualReg() const
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char WavefrontSize[]
Key for Kernel::CodeProps::Metadata::mWavefrontSize.
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
NodeAddr< PhiNode * > Phi
This is an optimization pass for GlobalISel generic memory operations.
IterT next_nodbg(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It, then continue incrementing it while it points to a debug instruction.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
DominatorTreeBase< T, false > DomTreeBase
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
@ Disabled
Don't do any conversion of .debug_str_offsets tables.
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
WorkItem(const BasicBlock *BB, int St)
static constexpr LaneBitmask getAll()
constexpr bool any() const
static constexpr LaneBitmask getNone()