LLVM 24.0.0git
SIWholeQuadMode.cpp
Go to the documentation of this file.
1//===-- SIWholeQuadMode.cpp - enter and suspend whole quad mode -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This pass adds instructions to enable whole quad mode (strict or non-strict)
11/// for pixel shaders, and strict whole wavefront mode for all programs.
12///
13/// The "strict" prefix indicates that inactive lanes do not take part in
14/// control flow, specifically an inactive lane enabled by a strict WQM/WWM will
15/// always be enabled irrespective of control flow decisions. Conversely in
16/// non-strict WQM inactive lanes may control flow decisions.
17///
18/// Whole quad mode is required for derivative computations, but it interferes
19/// with shader side effects (stores and atomics). It ensures that WQM is
20/// enabled when necessary, but disabled around stores and atomics.
21///
22/// When necessary, this pass creates a function prolog
23///
24/// S_MOV_B64 LiveMask, EXEC
25/// S_WQM_B64 EXEC, EXEC
26///
27/// to enter WQM at the top of the function and surrounds blocks of Exact
28/// instructions by
29///
30/// S_AND_SAVEEXEC_B64 Tmp, LiveMask
31/// ...
32/// S_MOV_B64 EXEC, Tmp
33///
34/// We also compute when a sequence of instructions requires strict whole
35/// wavefront mode (StrictWWM) and insert instructions to save and restore it:
36///
37/// S_OR_SAVEEXEC_B64 Tmp, -1
38/// ...
39/// S_MOV_B64 EXEC, Tmp
40///
41/// When a sequence of instructions requires strict whole quad mode (StrictWQM)
42/// we use a similar save and restore mechanism and force whole quad mode for
43/// those instructions:
44///
45/// S_MOV_B64 Tmp, EXEC
46/// S_WQM_B64 EXEC, EXEC
47/// ...
48/// S_MOV_B64 EXEC, Tmp
49///
50/// In order to avoid excessive switching during sequences of Exact
51/// instructions, the pass first analyzes which instructions must be run in WQM
52/// (aka which instructions produce values that lead to derivative
53/// computations).
54///
55/// Basic blocks are always exited in WQM as long as some successor needs WQM.
56///
57/// There is room for improvement given better control flow analysis:
58///
59/// (1) at the top level (outside of control flow statements, and as long as
60/// kill hasn't been used), one SGPR can be saved by recovering WQM from
61/// the LiveMask (this is implemented for the entry block).
62///
63/// (2) when entire regions (e.g. if-else blocks or entire loops) only
64/// consist of exact and don't-care instructions, the switch only has to
65/// be done at the entry and exit points rather than potentially in each
66/// block of the region.
67///
68//===----------------------------------------------------------------------===//
69
70#include "SIWholeQuadMode.h"
71#include "AMDGPU.h"
72#include "AMDGPULaneMaskUtils.h"
73#include "GCNSubtarget.h"
81#include "llvm/IR/CallingConv.h"
84
85using namespace llvm;
86
87#define DEBUG_TYPE "si-wqm"
88
89namespace {
90
91enum {
92 StateWQM = 0x1,
93 StateStrictWWM = 0x2,
94 StateStrictWQM = 0x4,
95 StateExact = 0x8,
96 StateStrict = StateStrictWWM | StateStrictWQM,
97};
98
99struct PrintState {
100public:
101 int State;
102
103 explicit PrintState(int State) : State(State) {}
104};
105
106#ifndef NDEBUG
107static raw_ostream &operator<<(raw_ostream &OS, const PrintState &PS) {
108
109 static const std::pair<char, const char *> Mapping[] = {
110 std::pair(StateWQM, "WQM"), std::pair(StateStrictWWM, "StrictWWM"),
111 std::pair(StateStrictWQM, "StrictWQM"), std::pair(StateExact, "Exact")};
112 char State = PS.State;
113 for (auto M : Mapping) {
114 if (State & M.first) {
115 OS << M.second;
116 State &= ~M.first;
117
118 if (State)
119 OS << '|';
120 }
121 }
122 assert(State == 0);
123 return OS;
124}
125#endif
126
127struct InstrInfo {
128 char Needs = 0;
129 char Disabled = 0;
130 char OutNeeds = 0;
131 char MarkedStates = 0;
132};
133
134struct BlockInfo {
135 char Needs = 0;
136 char InNeeds = 0;
137 char OutNeeds = 0;
138 char InitialState = 0;
139 bool NeedsLowering = false;
140};
141
142struct WorkItem {
143 MachineBasicBlock *MBB = nullptr;
144 MachineInstr *MI = nullptr;
145
146 WorkItem() = default;
149};
150
151class SIWholeQuadMode {
152public:
153 SIWholeQuadMode(MachineFunction &MF, LiveIntervals *LIS,
155 : ST(&MF.getSubtarget<GCNSubtarget>()), TII(ST->getInstrInfo()),
156 TRI(&TII->getRegisterInfo()), MRI(&MF.getRegInfo()), LIS(LIS), MDT(MDT),
157 PDT(PDT), LMC(AMDGPU::LaneMaskConstants::get(*ST)) {}
158 bool run(MachineFunction &MF);
159
160private:
161 const GCNSubtarget *ST;
162 const SIInstrInfo *TII;
163 const SIRegisterInfo *TRI;
165 LiveIntervals *LIS;
168 const AMDGPU::LaneMaskConstants &LMC;
169
170 Register LiveMaskReg;
171
174
175 // Tracks state (WQM/StrictWWM/StrictWQM/Exact) after a given instruction
177
178 SmallVector<MachineInstr *, 2> LiveMaskQueries;
179 SmallVector<MachineInstr *, 4> LowerToMovInstrs;
180 SmallSetVector<MachineInstr *, 4> LowerToCopyInstrs;
182 SmallVector<MachineInstr *, 4> InitExecInstrs;
183 SmallVector<MachineInstr *, 4> SetInactiveInstrs;
184
185 void printInfo();
186
187 void markInstruction(MachineInstr &MI, char Flag,
188 std::vector<WorkItem> &Worklist);
189 void markDefs(const MachineInstr &UseMI, LiveRange &LR,
190 VirtRegOrUnit VRegOrUnit, unsigned SubReg, char Flag,
191 std::vector<WorkItem> &Worklist);
192 void markOperand(const MachineInstr &MI, const MachineOperand &Op, char Flag,
193 std::vector<WorkItem> &Worklist);
194 void markInstructionUses(const MachineInstr &MI, char Flag,
195 std::vector<WorkItem> &Worklist);
196 char scanInstructions(MachineFunction &MF, std::vector<WorkItem> &Worklist,
197 SmallVector<MachineInstr *> &ExeczSideEffectInstrs);
198 void propagateInstruction(MachineInstr &MI, std::vector<WorkItem> &Worklist);
199 void propagateBlock(MachineBasicBlock &MBB, std::vector<WorkItem> &Worklist);
201
206 MachineBasicBlock::iterator Last, bool PreferLast,
207 bool SaveSCC);
209 Register SaveWQM);
211 Register SavedWQM);
212 void toStrictMode(MachineBasicBlock &MBB, MachineBasicBlock::iterator Before,
213 Register SaveOrig, char StrictStateNeeded);
214 void fromStrictMode(MachineBasicBlock &MBB,
215 MachineBasicBlock::iterator Before, Register SavedOrig,
216 char NonStrictState, char CurrentStrictState);
217
218 void splitBlock(MachineInstr *TermMI);
219 MachineInstr *lowerKillI1(MachineInstr &MI, bool IsWQM);
220 MachineInstr *lowerKillF32(MachineInstr &MI);
221
222 void lowerBlock(MachineBasicBlock &MBB, BlockInfo &BI);
223 void processBlock(MachineBasicBlock &MBB, BlockInfo &BI, bool IsEntry);
224
225 bool lowerLiveMaskQueries();
226 bool lowerCopyInstrs();
227 bool lowerKillInstrs(bool IsWQM);
228 void lowerInitExec(MachineInstr &MI);
229 MachineBasicBlock::iterator lowerInitExecInstrs(MachineBasicBlock &Entry,
230 bool &Changed);
231};
232
233class SIWholeQuadModeLegacy : public MachineFunctionPass {
234public:
235 static char ID;
236
237 SIWholeQuadModeLegacy() : MachineFunctionPass(ID) {}
238
239 bool runOnMachineFunction(MachineFunction &MF) override;
240
241 StringRef getPassName() const override { return "SI Whole Quad Mode"; }
242
243 void getAnalysisUsage(AnalysisUsage &AU) const override {
250 }
251
252 MachineFunctionProperties getClearedProperties() const override {
253 return MachineFunctionProperties().setIsSSA();
254 }
255};
256} // end anonymous namespace
257
258char SIWholeQuadModeLegacy::ID = 0;
259
260INITIALIZE_PASS_BEGIN(SIWholeQuadModeLegacy, DEBUG_TYPE, "SI Whole Quad Mode",
261 false, false)
265INITIALIZE_PASS_END(SIWholeQuadModeLegacy, DEBUG_TYPE, "SI Whole Quad Mode",
267
268char &llvm::SIWholeQuadModeID = SIWholeQuadModeLegacy::ID;
269
270#ifndef NDEBUG
271LLVM_DUMP_METHOD void SIWholeQuadMode::printInfo() {
272 for (const auto &BII : Blocks) {
273 dbgs() << "\n"
274 << printMBBReference(*BII.first) << ":\n"
275 << " InNeeds = " << PrintState(BII.second.InNeeds)
276 << ", Needs = " << PrintState(BII.second.Needs)
277 << ", OutNeeds = " << PrintState(BII.second.OutNeeds) << "\n\n";
278
279 for (const MachineInstr &MI : *BII.first) {
280 auto III = Instructions.find(&MI);
281 if (III != Instructions.end()) {
282 dbgs() << " " << MI << " Needs = " << PrintState(III->second.Needs)
283 << ", OutNeeds = " << PrintState(III->second.OutNeeds) << '\n';
284 }
285 }
286 }
287}
288#endif
289
290void SIWholeQuadMode::markInstruction(MachineInstr &MI, char Flag,
291 std::vector<WorkItem> &Worklist) {
292 InstrInfo &II = Instructions[&MI];
293
294 assert(!(Flag & StateExact) && Flag != 0);
295
296 // Capture all states requested in marking including disabled ones.
297 II.MarkedStates |= Flag;
298
299 // Remove any disabled states from the flag. The user that required it gets
300 // an undefined value in the helper lanes. For example, this can happen if
301 // the result of an atomic is used by instruction that requires WQM, where
302 // ignoring the request for WQM is correct as per the relevant specs.
303 Flag &= ~II.Disabled;
304
305 // Ignore if the flag is already encompassed by the existing needs, or we
306 // just disabled everything.
307 if ((II.Needs & Flag) == Flag)
308 return;
309
310 LLVM_DEBUG(dbgs() << "markInstruction " << PrintState(Flag) << ": " << MI);
311 II.Needs |= Flag;
312 Worklist.emplace_back(&MI);
313}
314
315/// Mark all relevant definitions of register \p Reg in usage \p UseMI.
316void SIWholeQuadMode::markDefs(const MachineInstr &UseMI, LiveRange &LR,
317 VirtRegOrUnit VRegOrUnit, unsigned SubReg,
318 char Flag, std::vector<WorkItem> &Worklist) {
319 LLVM_DEBUG(dbgs() << "markDefs " << PrintState(Flag) << ": " << UseMI);
320
321 LiveQueryResult UseLRQ = LR.Query(LIS->getInstructionIndex(UseMI));
322 const VNInfo *Value = UseLRQ.valueIn();
323 if (!Value)
324 return;
325
326 // Note: this code assumes that lane masks on AMDGPU completely
327 // cover registers.
328 const LaneBitmask UseLanes =
329 SubReg ? TRI->getSubRegIndexLaneMask(SubReg)
330 : (VRegOrUnit.isVirtualReg()
331 ? MRI->getMaxLaneMaskForVReg(VRegOrUnit.asVirtualReg())
333
334 // Perform a depth-first iteration of the LiveRange graph marking defs.
335 // Stop processing of a given branch when all use lanes have been defined.
336 // The first definition stops processing for a physical register.
337 struct PhiEntry {
338 const VNInfo *Phi;
339 unsigned PredIdx;
340 LaneBitmask DefinedLanes;
341
342 PhiEntry(const VNInfo *Phi, unsigned PredIdx, LaneBitmask DefinedLanes)
343 : Phi(Phi), PredIdx(PredIdx), DefinedLanes(DefinedLanes) {}
344 };
345 using VisitKey = std::pair<const VNInfo *, LaneBitmask>;
347 SmallSet<VisitKey, 4> Visited;
348 LaneBitmask DefinedLanes;
349 unsigned NextPredIdx = 0; // Only used for processing phi nodes
350 do {
351 const VNInfo *NextValue = nullptr;
352 const VisitKey Key(Value, DefinedLanes);
353
354 if (Visited.insert(Key).second) {
355 // On first visit to a phi then start processing first predecessor
356 NextPredIdx = 0;
357 }
358
359 if (Value->isPHIDef()) {
360 // Each predecessor node in the phi must be processed as a subgraph
361 const MachineBasicBlock *MBB = LIS->getMBBFromIndex(Value->def);
362 assert(MBB && "Phi-def has no defining MBB");
363
364 // Find next predecessor to process
365 unsigned Idx = NextPredIdx;
366 const auto *PI = MBB->pred_begin() + Idx;
367 const auto *PE = MBB->pred_end();
368 for (; PI != PE && !NextValue; ++PI, ++Idx) {
369 if (const VNInfo *VN = LR.getVNInfoBefore(LIS->getMBBEndIdx(*PI))) {
370 if (!Visited.count(VisitKey(VN, DefinedLanes)))
371 NextValue = VN;
372 }
373 }
374
375 // If there are more predecessors to process; add phi to stack
376 if (PI != PE)
377 PhiStack.emplace_back(Value, Idx, DefinedLanes);
378 } else {
379 MachineInstr *MI = LIS->getInstructionFromIndex(Value->def);
380 assert(MI && "Def has no defining instruction");
381
382 if (VRegOrUnit.isVirtualReg()) {
383 // Iterate over all operands to find relevant definitions
384 bool HasDef = false;
385 for (const MachineOperand &Op : MI->all_defs()) {
386 if (Op.getReg() != VRegOrUnit.asVirtualReg())
387 continue;
388
389 // Compute lanes defined and overlap with use
390 LaneBitmask OpLanes =
391 Op.isUndef() ? LaneBitmask::getAll()
392 : TRI->getSubRegIndexLaneMask(Op.getSubReg());
393 LaneBitmask Overlap = (UseLanes & OpLanes);
394
395 // Record if this instruction defined any of use
396 HasDef |= Overlap.any();
397
398 // Mark any lanes defined
399 DefinedLanes |= OpLanes;
400 }
401
402 // Check if all lanes of use have been defined
403 if ((DefinedLanes & UseLanes) != UseLanes) {
404 // Definition not complete; need to process input value
405 LiveQueryResult LRQ = LR.Query(LIS->getInstructionIndex(*MI));
406 if (const VNInfo *VN = LRQ.valueIn()) {
407 if (!Visited.count(VisitKey(VN, DefinedLanes)))
408 NextValue = VN;
409 }
410 }
411
412 // Only mark the instruction if it defines some part of the use
413 if (HasDef)
414 markInstruction(*MI, Flag, Worklist);
415 } else {
416 // For physical registers simply mark the defining instruction
417 markInstruction(*MI, Flag, Worklist);
418 }
419 }
420
421 if (!NextValue && !PhiStack.empty()) {
422 // Reach end of chain; revert to processing last phi
423 PhiEntry &Entry = PhiStack.back();
424 NextValue = Entry.Phi;
425 NextPredIdx = Entry.PredIdx;
426 DefinedLanes = Entry.DefinedLanes;
427 PhiStack.pop_back();
428 }
429
430 Value = NextValue;
431 } while (Value);
432}
433
434void SIWholeQuadMode::markOperand(const MachineInstr &MI,
435 const MachineOperand &Op, char Flag,
436 std::vector<WorkItem> &Worklist) {
437 assert(Op.isReg());
438 Register Reg = Op.getReg();
439
440 // Ignore some hardware registers
441 switch (Reg) {
442 case AMDGPU::EXEC:
443 case AMDGPU::EXEC_LO:
444 return;
445 default:
446 break;
447 }
448
449 LLVM_DEBUG(dbgs() << "markOperand " << PrintState(Flag) << ": " << Op
450 << " for " << MI);
451 if (Reg.isVirtual()) {
452 LiveRange &LR = LIS->getInterval(Reg);
453 markDefs(MI, LR, VirtRegOrUnit(Reg), Op.getSubReg(), Flag, Worklist);
454 } else {
455 // Handle physical registers that we need to track; this is mostly relevant
456 // for VCC, which can appear as the (implicit) input of a uniform branch,
457 // e.g. when a loop counter is stored in a VGPR.
458 for (MCRegUnit Unit : TRI->regunits(Reg.asMCReg())) {
459 LiveRange &LR = LIS->getRegUnit(Unit);
460 const VNInfo *Value = LR.Query(LIS->getInstructionIndex(MI)).valueIn();
461 if (Value)
462 markDefs(MI, LR, VirtRegOrUnit(Unit), AMDGPU::NoSubRegister, Flag,
463 Worklist);
464 }
465 }
466}
467
468/// Mark all instructions defining the uses in \p MI with \p Flag.
469void SIWholeQuadMode::markInstructionUses(const MachineInstr &MI, char Flag,
470 std::vector<WorkItem> &Worklist) {
471 LLVM_DEBUG(dbgs() << "markInstructionUses " << PrintState(Flag) << ": "
472 << MI);
473
474 for (const MachineOperand &Use : MI.all_uses())
475 markOperand(MI, Use, Flag, Worklist);
476}
477
478// Scan instructions to determine which ones require an Exact execmask and
479// which ones seed WQM requirements.
480char SIWholeQuadMode::scanInstructions(
481 MachineFunction &MF, std::vector<WorkItem> &Worklist,
482 SmallVector<MachineInstr *> &ExeczSideEffectInstrs) {
483 char GlobalFlags = 0;
484 bool WQMOutputs = MF.getFunction().hasFnAttribute("amdgpu-ps-wqm-outputs");
485 SmallVector<MachineInstr *, 4> SoftWQMInstrs;
486 bool HasImplicitDerivatives =
487 MF.getFunction().getCallingConv() == CallingConv::AMDGPU_PS;
488
489 // We need to visit the basic blocks in reverse post-order so that we visit
490 // defs before uses, in particular so that we don't accidentally mark an
491 // instruction as needing e.g. WQM before visiting it and realizing it needs
492 // WQM disabled.
493 ReversePostOrderTraversal<MachineFunction *> RPOT(&MF);
494 for (MachineBasicBlock *MBB : RPOT) {
495 BlockInfo &BBI = Blocks[MBB];
496
497 for (MachineInstr &MI : *MBB) {
498 InstrInfo &III = Instructions[&MI];
499 unsigned Opcode = MI.getOpcode();
500 char Flags = 0;
501
502 if (TII->isWQM(Opcode)) {
503 // If LOD is not supported WQM is not needed.
504 // Only generate implicit WQM if implicit derivatives are required.
505 // This avoids inserting unintended WQM if a shader type without
506 // implicit derivatives uses an image sampling instruction.
507 if (ST->hasExtendedImageInsts() && HasImplicitDerivatives) {
508 // Sampling instructions don't need to produce results for all pixels
509 // in a quad, they just require all inputs of a quad to have been
510 // computed for derivatives.
511 markInstructionUses(MI, StateWQM, Worklist);
512 GlobalFlags |= StateWQM;
513 }
514 } else if (Opcode == AMDGPU::WQM) {
515 // The WQM intrinsic requires its output to have all the helper lanes
516 // correct, so we need it to be in WQM.
517 Flags = StateWQM;
518 LowerToCopyInstrs.insert(&MI);
519 } else if (Opcode == AMDGPU::SOFT_WQM) {
520 LowerToCopyInstrs.insert(&MI);
521 SoftWQMInstrs.push_back(&MI);
522 } else if (Opcode == AMDGPU::STRICT_WWM) {
523 // The STRICT_WWM intrinsic doesn't make the same guarantee, and plus
524 // it needs to be executed in WQM or Exact so that its copy doesn't
525 // clobber inactive lanes.
526 markInstructionUses(MI, StateStrictWWM, Worklist);
527 GlobalFlags |= StateStrictWWM;
528 LowerToMovInstrs.push_back(&MI);
529 } else if (Opcode == AMDGPU::STRICT_WQM ||
530 TII->isDualSourceBlendEXP(MI)) {
531 // STRICT_WQM is similar to STRICTWWM, but instead of enabling all
532 // threads of the wave like STRICTWWM, STRICT_WQM enables all threads in
533 // quads that have at least one active thread.
534 markInstructionUses(MI, StateStrictWQM, Worklist);
535 GlobalFlags |= StateStrictWQM;
536
537 if (Opcode == AMDGPU::STRICT_WQM) {
538 LowerToMovInstrs.push_back(&MI);
539 } else {
540 // Dual source blend export acts as implicit strict-wqm, its sources
541 // need to be shuffled in strict wqm, but the export itself needs to
542 // run in exact mode.
543 BBI.Needs |= StateExact;
544 if (!(BBI.InNeeds & StateExact)) {
545 BBI.InNeeds |= StateExact;
546 Worklist.emplace_back(MBB);
547 }
548 GlobalFlags |= StateExact;
549 III.Disabled = StateWQM | StateStrict;
550 }
551 } else if (Opcode == AMDGPU::LDS_PARAM_LOAD ||
552 Opcode == AMDGPU::DS_PARAM_LOAD ||
553 Opcode == AMDGPU::LDS_DIRECT_LOAD ||
554 Opcode == AMDGPU::DS_DIRECT_LOAD) {
555 // Mark these STRICTWQM, but only for the instruction, not its operands.
556 // This avoid unnecessarily marking M0 as requiring WQM.
557 III.Needs |= StateStrictWQM;
558 GlobalFlags |= StateStrictWQM;
559 } else if (Opcode == AMDGPU::V_SET_INACTIVE_B32) {
560 // Disable strict states; StrictWQM will be added as required later.
561 III.Disabled = StateStrict;
562 MachineOperand &Inactive = MI.getOperand(4);
563 if (Inactive.isReg()) {
564 if (Inactive.isUndef() && MI.getOperand(3).getImm() == 0)
565 LowerToCopyInstrs.insert(&MI);
566 else
567 markOperand(MI, Inactive, StateStrictWWM, Worklist);
568 }
569 SetInactiveInstrs.push_back(&MI);
570 BBI.NeedsLowering = true;
571 } else if (TII->isDisableWQM(MI)) {
572 BBI.Needs |= StateExact;
573 if (!(BBI.InNeeds & StateExact)) {
574 BBI.InNeeds |= StateExact;
575 Worklist.emplace_back(MBB);
576 }
577 GlobalFlags |= StateExact;
578 III.Disabled = StateWQM | StateStrict;
579 } else if (Opcode == AMDGPU::SI_PS_LIVE ||
580 Opcode == AMDGPU::SI_LIVE_MASK) {
581 LiveMaskQueries.push_back(&MI);
582 } else if (Opcode == AMDGPU::SI_KILL_I1_TERMINATOR ||
583 Opcode == AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR ||
584 Opcode == AMDGPU::SI_DEMOTE_I1) {
585 KillInstrs.push_back(&MI);
586 BBI.NeedsLowering = true;
587 } else if (Opcode == AMDGPU::SI_INIT_EXEC ||
588 Opcode == AMDGPU::SI_INIT_EXEC_FROM_INPUT ||
589 Opcode == AMDGPU::SI_INIT_WHOLE_WAVE) {
590 InitExecInstrs.push_back(&MI);
591 } else if (WQMOutputs) {
592 // The function is in machine SSA form, which means that physical
593 // VGPRs correspond to shader inputs and outputs. Inputs are
594 // only used, outputs are only defined.
595 // FIXME: is this still valid?
596 for (const MachineOperand &MO : MI.defs()) {
597 Register Reg = MO.getReg();
598 if (Reg.isPhysical() &&
599 TRI->hasVectorRegisters(TRI->getPhysRegBaseClass(Reg))) {
600 Flags = StateWQM;
601 break;
602 }
603 }
604 }
605
606 if (TII->hasUnwantedEffectsWhenEXECEmpty(MI)) {
607 for (auto &Op : MI.uses()) {
608 if (!Op.isReg())
609 continue;
610 if (!TRI->isVectorRegister(*MRI, Op.getReg()))
611 continue;
612
613 ExeczSideEffectInstrs.push_back(&MI);
614 break;
615 }
616 }
617
618 if (Flags) {
619 markInstruction(MI, Flags, Worklist);
620 GlobalFlags |= Flags;
621 }
622 }
623 }
624
625 // Mark sure that any SET_INACTIVE instructions are computed in WQM if WQM is
626 // ever used anywhere in the function. This implements the corresponding
627 // semantics of @llvm.amdgcn.set.inactive.
628 // Similarly for SOFT_WQM instructions, implementing @llvm.amdgcn.softwqm.
629 if (GlobalFlags & StateWQM) {
630 for (MachineInstr *MI : SetInactiveInstrs)
631 markInstruction(*MI, StateWQM, Worklist);
632 for (MachineInstr *MI : SoftWQMInstrs)
633 markInstruction(*MI, StateWQM, Worklist);
634 }
635
636 return GlobalFlags;
637}
638
639void SIWholeQuadMode::propagateInstruction(MachineInstr &MI,
640 std::vector<WorkItem>& Worklist) {
641 MachineBasicBlock *MBB = MI.getParent();
642 InstrInfo II = Instructions[&MI]; // take a copy to prevent dangling references
643 BlockInfo &BI = Blocks[MBB];
644
645 // Control flow-type instructions and stores to temporary memory that are
646 // followed by WQM computations must themselves be in WQM.
647 if ((II.OutNeeds & StateWQM) && !(II.Disabled & StateWQM) &&
648 (MI.isTerminator() || (TII->usesVM_CNT(MI) && MI.mayStore()))) {
649 Instructions[&MI].Needs = StateWQM;
650 II.Needs = StateWQM;
651 }
652
653 // Propagate to block level
654 if (II.Needs & StateWQM) {
655 BI.Needs |= StateWQM;
656 if (!(BI.InNeeds & StateWQM)) {
657 BI.InNeeds |= StateWQM;
658 Worklist.emplace_back(MBB);
659 }
660 }
661
662 // Propagate backwards within block
663 if (MachineInstr *PrevMI = MI.getPrevNode()) {
664 char InNeeds = (II.Needs & ~StateStrict) | II.OutNeeds;
665 if (!PrevMI->isPHI()) {
666 InstrInfo &PrevII = Instructions[PrevMI];
667 if ((PrevII.OutNeeds | InNeeds) != PrevII.OutNeeds) {
668 PrevII.OutNeeds |= InNeeds;
669 Worklist.emplace_back(PrevMI);
670 }
671 }
672 }
673
674 // Propagate WQM flag to instruction inputs
675 assert(!(II.Needs & StateExact));
676
677 if (II.Needs != 0)
678 markInstructionUses(MI, II.Needs, Worklist);
679
680 // Ensure we process a block containing StrictWWM/StrictWQM, even if it does
681 // not require any WQM transitions.
682 if (II.Needs & StateStrictWWM)
683 BI.Needs |= StateStrictWWM;
684 if (II.Needs & StateStrictWQM)
685 BI.Needs |= StateStrictWQM;
686}
687
688void SIWholeQuadMode::propagateBlock(MachineBasicBlock &MBB,
689 std::vector<WorkItem>& Worklist) {
690 BlockInfo BI = Blocks[&MBB]; // Make a copy to prevent dangling references.
691
692 // Propagate through instructions
693 if (!MBB.empty()) {
694 MachineInstr *LastMI = &*MBB.rbegin();
695 InstrInfo &LastII = Instructions[LastMI];
696 if ((LastII.OutNeeds | BI.OutNeeds) != LastII.OutNeeds) {
697 LastII.OutNeeds |= BI.OutNeeds;
698 Worklist.emplace_back(LastMI);
699 }
700 }
701
702 // Predecessor blocks must provide for our WQM/Exact needs.
703 for (MachineBasicBlock *Pred : MBB.predecessors()) {
704 BlockInfo &PredBI = Blocks[Pred];
705 if ((PredBI.OutNeeds | BI.InNeeds) == PredBI.OutNeeds)
706 continue;
707
708 PredBI.OutNeeds |= BI.InNeeds;
709 PredBI.InNeeds |= BI.InNeeds;
710 Worklist.emplace_back(Pred);
711 }
712
713 // All successors must be prepared to accept the same set of WQM/Exact data.
714 for (MachineBasicBlock *Succ : MBB.successors()) {
715 BlockInfo &SuccBI = Blocks[Succ];
716 if ((SuccBI.InNeeds | BI.OutNeeds) == SuccBI.InNeeds)
717 continue;
718
719 SuccBI.InNeeds |= BI.OutNeeds;
720 Worklist.emplace_back(Succ);
721 }
722}
723
724char SIWholeQuadMode::analyzeFunction(MachineFunction &MF) {
725 std::vector<WorkItem> Worklist;
726 SmallVector<MachineInstr *> ExeczSideEffectInstrs;
727 char GlobalFlags = scanInstructions(MF, Worklist, ExeczSideEffectInstrs);
728
729 while (!Worklist.empty()) {
730 WorkItem WI = Worklist.back();
731 Worklist.pop_back();
732
733 if (WI.MI)
734 propagateInstruction(*WI.MI, Worklist);
735 else
736 propagateBlock(*WI.MBB, Worklist);
737
738 if (Worklist.empty()) {
739 // Currently we let the instructions having sideeffect when execz to run
740 // under wqm, this avoids unwanted side-effect with exact mode if only
741 // helper lanes execute the parent block. At the same time, the wqm
742 // property should be back-propagated along the data-flow of their sources
743 // to ensure their sources have correct data for helper lanes.
744 for (auto *MI : ExeczSideEffectInstrs) {
745 InstrInfo II = Instructions[MI];
746 if (II.OutNeeds & StateWQM)
747 markInstructionUses(*MI, StateWQM, Worklist);
748 }
749 // The side-effect backward propagation should not expand the wqm-region.
750 // So we only need to run the propagation once.
751 ExeczSideEffectInstrs.clear();
752 }
753 }
754
755 return GlobalFlags;
756}
757
759SIWholeQuadMode::saveSCC(MachineBasicBlock &MBB,
761 Register SaveReg = MRI->createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
762
763 MachineInstr *Save =
764 BuildMI(MBB, Before, DebugLoc(), TII->get(AMDGPU::COPY), SaveReg)
765 .addReg(AMDGPU::SCC);
766 MachineInstr *Restore =
767 BuildMI(MBB, Before, DebugLoc(), TII->get(AMDGPU::COPY), AMDGPU::SCC)
768 .addReg(SaveReg);
769
770 LIS->InsertMachineInstrInMaps(*Save);
771 LIS->InsertMachineInstrInMaps(*Restore);
773
774 return Restore;
775}
776
777void SIWholeQuadMode::splitBlock(MachineInstr *TermMI) {
778 MachineBasicBlock *BB = TermMI->getParent();
779 LLVM_DEBUG(dbgs() << "Split block " << printMBBReference(*BB) << " @ "
780 << *TermMI << "\n");
781
782 MachineBasicBlock *SplitBB =
783 BB->splitAt(*TermMI, /*UpdateLiveIns*/ true, LIS);
784
785 // Convert last instruction in block to a terminator.
786 // Note: this only covers the expected patterns
787 unsigned NewOpcode = 0;
788 switch (TermMI->getOpcode()) {
789 case AMDGPU::S_AND_B32:
790 NewOpcode = AMDGPU::S_AND_B32_term;
791 break;
792 case AMDGPU::S_AND_B64:
793 NewOpcode = AMDGPU::S_AND_B64_term;
794 break;
795 case AMDGPU::S_MOV_B32:
796 NewOpcode = AMDGPU::S_MOV_B32_term;
797 break;
798 case AMDGPU::S_MOV_B64:
799 NewOpcode = AMDGPU::S_MOV_B64_term;
800 break;
801 case AMDGPU::S_ANDN2_B32:
802 NewOpcode = AMDGPU::S_ANDN2_B32_term;
803 break;
804 case AMDGPU::S_ANDN2_B64:
805 NewOpcode = AMDGPU::S_ANDN2_B64_term;
806 break;
807 default:
808 llvm_unreachable("Unexpected instruction");
809 }
810
811 // These terminators fallthrough to the next block, no need to add an
812 // unconditional branch to the next block (SplitBB).
813 TermMI->setDesc(TII->get(NewOpcode));
814
815 if (SplitBB != BB) {
816 // Update dominator trees
817 using DomTreeT = DomTreeBase<MachineBasicBlock>;
819 for (MachineBasicBlock *Succ : SplitBB->successors()) {
820 DTUpdates.push_back({DomTreeT::Insert, SplitBB, Succ});
821 DTUpdates.push_back({DomTreeT::Delete, BB, Succ});
822 }
823 DTUpdates.push_back({DomTreeT::Insert, BB, SplitBB});
824 if (MDT)
825 MDT->applyUpdates(DTUpdates);
826 if (PDT)
827 PDT->applyUpdates(DTUpdates);
828 }
829}
830
831MachineInstr *SIWholeQuadMode::lowerKillF32(MachineInstr &MI) {
832 assert(LiveMaskReg.isVirtual());
833
834 const DebugLoc &DL = MI.getDebugLoc();
835 unsigned Opcode = 0;
836
837 assert(MI.getOperand(0).isReg());
838
839 // Comparison is for live lanes; however here we compute the inverse
840 // (killed lanes). This is because VCMP will always generate 0 bits
841 // for inactive lanes so a mask of live lanes would not be correct
842 // inside control flow.
843 // Invert the comparison by swapping the operands and adjusting
844 // the comparison codes.
845
846 switch (MI.getOperand(2).getImm()) {
847 case ISD::SETUEQ:
848 Opcode = AMDGPU::V_CMP_LG_F32_e64;
849 break;
850 case ISD::SETUGT:
851 Opcode = AMDGPU::V_CMP_GE_F32_e64;
852 break;
853 case ISD::SETUGE:
854 Opcode = AMDGPU::V_CMP_GT_F32_e64;
855 break;
856 case ISD::SETULT:
857 Opcode = AMDGPU::V_CMP_LE_F32_e64;
858 break;
859 case ISD::SETULE:
860 Opcode = AMDGPU::V_CMP_LT_F32_e64;
861 break;
862 case ISD::SETUNE:
863 Opcode = AMDGPU::V_CMP_EQ_F32_e64;
864 break;
865 case ISD::SETO:
866 Opcode = AMDGPU::V_CMP_O_F32_e64;
867 break;
868 case ISD::SETUO:
869 Opcode = AMDGPU::V_CMP_U_F32_e64;
870 break;
871 case ISD::SETOEQ:
872 case ISD::SETEQ:
873 Opcode = AMDGPU::V_CMP_NEQ_F32_e64;
874 break;
875 case ISD::SETOGT:
876 case ISD::SETGT:
877 Opcode = AMDGPU::V_CMP_NLT_F32_e64;
878 break;
879 case ISD::SETOGE:
880 case ISD::SETGE:
881 Opcode = AMDGPU::V_CMP_NLE_F32_e64;
882 break;
883 case ISD::SETOLT:
884 case ISD::SETLT:
885 Opcode = AMDGPU::V_CMP_NGT_F32_e64;
886 break;
887 case ISD::SETOLE:
888 case ISD::SETLE:
889 Opcode = AMDGPU::V_CMP_NGE_F32_e64;
890 break;
891 case ISD::SETONE:
892 case ISD::SETNE:
893 Opcode = AMDGPU::V_CMP_NLG_F32_e64;
894 break;
895 default:
896 llvm_unreachable("invalid ISD:SET cond code");
897 }
898
899 MachineBasicBlock &MBB = *MI.getParent();
900
901 // Pick opcode based on comparison type.
902 MachineInstr *VcmpMI;
903 const MachineOperand &Op0 = MI.getOperand(0);
904 const MachineOperand &Op1 = MI.getOperand(1);
905
906 // VCC represents lanes killed.
907 if (TRI->isVGPR(*MRI, Op0.getReg())) {
908 Opcode = AMDGPU::getVOPe32(Opcode);
909 VcmpMI = BuildMI(MBB, &MI, DL, TII->get(Opcode)).add(Op1).add(Op0);
910 } else {
911 VcmpMI = BuildMI(MBB, &MI, DL, TII->get(Opcode))
912 .addReg(LMC.VccReg, RegState::Define)
913 .addImm(0) // src0 modifiers
914 .add(Op1)
915 .addImm(0) // src1 modifiers
916 .add(Op0)
917 .addImm(0); // omod
918 }
919
920 MachineInstr *MaskUpdateMI =
921 BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), LiveMaskReg)
922 .addReg(LiveMaskReg)
923 .addReg(LMC.VccReg);
924
925 // State of SCC represents whether any lanes are live in mask,
926 // if SCC is 0 then no lanes will be alive anymore.
927 MachineInstr *EarlyTermMI =
928 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_EARLY_TERMINATE_SCC0));
929
930 MachineInstr *ExecMaskMI =
931 BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), LMC.ExecReg)
932 .addReg(LMC.ExecReg)
933 .addReg(LMC.VccReg)
934 .setOperandDead(3);
935
936 assert(MBB.succ_size() == 1);
937
938 // Update live intervals
940
941 LIS->ReplaceMachineInstrInMaps(MI, *VcmpMI);
942 MBB.remove(&MI);
943
944 LIS->InsertMachineInstrInMaps(*MaskUpdateMI);
945 LIS->InsertMachineInstrInMaps(*EarlyTermMI);
946 LIS->InsertMachineInstrInMaps(*ExecMaskMI);
947
948 return ExecMaskMI;
949}
950
951MachineInstr *SIWholeQuadMode::lowerKillI1(MachineInstr &MI, bool IsWQM) {
952 assert(LiveMaskReg.isVirtual());
953
954 MachineBasicBlock &MBB = *MI.getParent();
955
956 const DebugLoc &DL = MI.getDebugLoc();
957 MachineInstr *MaskUpdateMI = nullptr;
958
959 const bool IsDemote = IsWQM && (MI.getOpcode() == AMDGPU::SI_DEMOTE_I1);
960 const MachineOperand &Op = MI.getOperand(0);
961 int64_t KillVal = MI.getOperand(1).getImm();
962 MachineInstr *ComputeKilledMaskMI = nullptr;
963 Register CndReg = !Op.isImm() ? Op.getReg() : Register();
964 Register TmpReg;
965
966 // Is this a static or dynamic kill?
967 if (Op.isImm()) {
968 if (Op.getImm() == KillVal) {
969 // Static: all active lanes are killed
970 MaskUpdateMI = BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), LiveMaskReg)
971 .addReg(LiveMaskReg)
972 .addReg(LMC.ExecReg);
973 } else {
974 // Static: kill does nothing
975 bool IsLastTerminator = std::next(MI.getIterator()) == MBB.end();
976 if (!IsLastTerminator) {
978 } else {
979 assert(MBB.succ_size() == 1);
980 MachineInstr *NewTerm = BuildMI(MBB, MI, DL, TII->get(AMDGPU::S_BRANCH))
981 .addMBB(*MBB.succ_begin());
982 LIS->ReplaceMachineInstrInMaps(MI, *NewTerm);
983 }
984 MBB.remove(&MI);
985 return nullptr;
986 }
987 } else {
988 if (!KillVal) {
989 // Op represents live lanes after kill,
990 // so exec mask needs to be factored in.
991 TmpReg = MRI->createVirtualRegister(TRI->getBoolRC());
992 ComputeKilledMaskMI = BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), TmpReg)
993 .addReg(LMC.ExecReg)
994 .add(Op)
995 .setOperandDead(3);
996 MaskUpdateMI = BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), LiveMaskReg)
997 .addReg(LiveMaskReg)
998 .addReg(TmpReg);
999 } else {
1000 // Op represents lanes to kill
1001 MaskUpdateMI = BuildMI(MBB, MI, DL, TII->get(LMC.AndN2Opc), LiveMaskReg)
1002 .addReg(LiveMaskReg)
1003 .add(Op);
1004 }
1005 }
1006
1007 // State of SCC represents whether any lanes are live in mask,
1008 // if SCC is 0 then no lanes will be alive anymore.
1009 MachineInstr *EarlyTermMI =
1010 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_EARLY_TERMINATE_SCC0));
1011
1012 // In the case we got this far some lanes are still live,
1013 // update EXEC to deactivate lanes as appropriate.
1014 MachineInstr *NewTerm;
1015 MachineInstr *WQMMaskMI = nullptr;
1016 Register LiveMaskWQM;
1017 if (IsDemote) {
1018 // Demote - deactivate quads with only helper lanes
1019 LiveMaskWQM = MRI->createVirtualRegister(TRI->getBoolRC());
1020 WQMMaskMI = BuildMI(MBB, MI, DL, TII->get(LMC.WQMOpc), LiveMaskWQM)
1021 .addReg(LiveMaskReg)
1022 .setOperandDead(2);
1023 NewTerm = BuildMI(MBB, MI, DL, TII->get(LMC.AndOpc), LMC.ExecReg)
1024 .addReg(LMC.ExecReg)
1025 .addReg(LiveMaskWQM)
1026 .setOperandDead(3);
1027 } else {
1028 // Kill - deactivate lanes no longer in live mask
1029 if (Op.isImm()) {
1030 NewTerm =
1031 BuildMI(MBB, &MI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addImm(0);
1032 } else if (!IsWQM) {
1033 NewTerm = BuildMI(MBB, &MI, DL, TII->get(LMC.AndOpc), LMC.ExecReg)
1034 .addReg(LMC.ExecReg)
1035 .addReg(LiveMaskReg)
1036 .setOperandDead(3);
1037 } else {
1038 unsigned Opcode = KillVal ? LMC.AndN2Opc : LMC.AndOpc;
1039 NewTerm = BuildMI(MBB, &MI, DL, TII->get(Opcode), LMC.ExecReg)
1040 .addReg(LMC.ExecReg)
1041 .add(Op)
1042 .setOperandDead(3);
1043 }
1044 }
1045
1046 // Update live intervals
1048 MBB.remove(&MI);
1049 assert(EarlyTermMI);
1050 assert(MaskUpdateMI);
1051 assert(NewTerm);
1052 if (ComputeKilledMaskMI)
1053 LIS->InsertMachineInstrInMaps(*ComputeKilledMaskMI);
1054 LIS->InsertMachineInstrInMaps(*MaskUpdateMI);
1055 LIS->InsertMachineInstrInMaps(*EarlyTermMI);
1056 if (WQMMaskMI)
1057 LIS->InsertMachineInstrInMaps(*WQMMaskMI);
1058 LIS->InsertMachineInstrInMaps(*NewTerm);
1059
1060 if (CndReg) {
1061 LIS->removeInterval(CndReg);
1063 }
1064 if (TmpReg)
1066 if (LiveMaskWQM)
1067 LIS->createAndComputeVirtRegInterval(LiveMaskWQM);
1068
1069 return NewTerm;
1070}
1071
1072// Replace (or supplement) instructions accessing live mask.
1073// This can only happen once all the live mask registers have been created
1074// and the execute state (WQM/StrictWWM/Exact) of instructions is known.
1075void SIWholeQuadMode::lowerBlock(MachineBasicBlock &MBB, BlockInfo &BI) {
1076 if (!BI.NeedsLowering)
1077 return;
1078
1079 LLVM_DEBUG(dbgs() << "\nLowering block " << printMBBReference(MBB) << ":\n");
1080
1081 SmallVector<MachineInstr *, 4> SplitPoints;
1082 Register ActiveLanesReg = 0;
1083 char State = BI.InitialState;
1084
1085 for (MachineInstr &MI : llvm::make_early_inc_range(
1087 auto MIState = StateTransition.find(&MI);
1088 if (MIState != StateTransition.end())
1089 State = MIState->second;
1090
1091 MachineInstr *SplitPoint = nullptr;
1092 switch (MI.getOpcode()) {
1093 case AMDGPU::SI_DEMOTE_I1:
1094 case AMDGPU::SI_KILL_I1_TERMINATOR:
1095 SplitPoint = lowerKillI1(MI, State == StateWQM);
1096 break;
1097 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
1098 SplitPoint = lowerKillF32(MI);
1099 break;
1100 case AMDGPU::ENTER_STRICT_WWM:
1101 ActiveLanesReg = MI.getOperand(0).getReg();
1102 break;
1103 case AMDGPU::EXIT_STRICT_WWM:
1104 ActiveLanesReg = 0;
1105 break;
1106 case AMDGPU::V_SET_INACTIVE_B32:
1107 if (ActiveLanesReg) {
1108 LiveInterval &LI = LIS->getInterval(MI.getOperand(5).getReg());
1109 MRI->constrainRegClass(ActiveLanesReg, TRI->getWaveMaskRegClass());
1110 MI.getOperand(5).setReg(ActiveLanesReg);
1111 LIS->shrinkToUses(&LI);
1112 } else {
1113 assert(State == StateExact || State == StateWQM);
1114 }
1115 break;
1116 default:
1117 break;
1118 }
1119 if (SplitPoint)
1120 SplitPoints.push_back(SplitPoint);
1121 }
1122
1123 // Perform splitting after instruction scan to simplify iteration.
1124 for (MachineInstr *MI : SplitPoints)
1125 splitBlock(MI);
1126}
1127
1128// Return an iterator in the (inclusive) range [First, Last] at which
1129// instructions can be safely inserted, keeping in mind that some of the
1130// instructions we want to add necessarily clobber SCC.
1131MachineBasicBlock::iterator SIWholeQuadMode::prepareInsertion(
1132 MachineBasicBlock &MBB, MachineBasicBlock::iterator First,
1133 MachineBasicBlock::iterator Last, bool PreferLast, bool SaveSCC) {
1134 if (!SaveSCC)
1135 return PreferLast ? Last : First;
1136
1137 LiveRange &LR =
1138 LIS->getRegUnit(*TRI->regunits(MCRegister::from(AMDGPU::SCC)).begin());
1139 auto MBBE = MBB.end();
1140 // Skip debug instructions when getting slot indices, as they don't have
1141 // entries in the slot index map.
1142 auto FirstNonDbg = skipDebugInstructionsForward(First, MBBE);
1143 auto LastNonDbg = skipDebugInstructionsForward(Last, MBBE);
1144 SlotIndex FirstIdx = FirstNonDbg != MBBE
1145 ? LIS->getInstructionIndex(*FirstNonDbg)
1146 : LIS->getMBBEndIdx(&MBB);
1147 SlotIndex LastIdx = LastNonDbg != MBBE ? LIS->getInstructionIndex(*LastNonDbg)
1148 : LIS->getMBBEndIdx(&MBB);
1149 SlotIndex Idx = PreferLast ? LastIdx : FirstIdx;
1150 const LiveRange::Segment *S;
1151
1152 for (;;) {
1153 S = LR.getSegmentContaining(Idx);
1154 if (!S)
1155 break;
1156
1157 if (PreferLast) {
1158 SlotIndex Next = S->start.getBaseIndex();
1159 if (Next < FirstIdx)
1160 break;
1161 Idx = Next;
1162 } else {
1163 MachineInstr *EndMI = LIS->getInstructionFromIndex(S->end.getBaseIndex());
1164 assert(EndMI && "Segment does not end on valid instruction");
1165 auto NextI = next_nodbg(EndMI->getIterator(), MBB.instr_end());
1166 if (NextI == MBB.instr_end())
1167 break;
1168 SlotIndex Next = LIS->getInstructionIndex(*NextI);
1169 if (Next > LastIdx)
1170 break;
1171 Idx = Next;
1172 }
1173 }
1174
1176
1177 if (MachineInstr *MI = LIS->getInstructionFromIndex(Idx))
1178 MBBI = MI;
1179 else {
1180 assert(Idx == LIS->getMBBEndIdx(&MBB));
1181 MBBI = MBB.end();
1182 }
1183
1184 // Move insertion point past any operations modifying EXEC.
1185 // This assumes that the value of SCC defined by any of these operations
1186 // does not need to be preserved.
1187 while (MBBI != Last) {
1188 bool IsExecDef = false;
1189 for (const MachineOperand &MO : MBBI->all_defs()) {
1190 IsExecDef |=
1191 MO.getReg() == AMDGPU::EXEC_LO || MO.getReg() == AMDGPU::EXEC;
1192 }
1193 if (!IsExecDef)
1194 break;
1195 MBBI++;
1196 S = nullptr;
1197 }
1198
1199 if (S)
1200 MBBI = saveSCC(MBB, MBBI);
1201
1202 return MBBI;
1203}
1204
1205void SIWholeQuadMode::toExact(MachineBasicBlock &MBB,
1207 Register SaveWQM) {
1208 assert(LiveMaskReg.isVirtual());
1209
1210 bool IsTerminator = Before == MBB.end();
1211 if (!IsTerminator) {
1212 auto FirstTerm = MBB.getFirstTerminator();
1213 if (FirstTerm != MBB.end()) {
1214 SlotIndex FirstTermIdx = LIS->getInstructionIndex(*FirstTerm);
1215 SlotIndex BeforeIdx = LIS->getInstructionIndex(*Before);
1216 IsTerminator = BeforeIdx > FirstTermIdx;
1217 }
1218 }
1219
1220 const DebugLoc &DL = MBB.findDebugLoc(Before);
1221 MachineInstr *MI;
1222
1223 if (SaveWQM) {
1224 unsigned Opcode =
1225 IsTerminator ? LMC.AndSaveExecTermOpc : LMC.AndSaveExecOpc;
1226 MI = BuildMI(MBB, Before, DL, TII->get(Opcode), SaveWQM)
1227 .addReg(LiveMaskReg)
1228 .setOperandDead(3);
1229 } else {
1230 unsigned Opcode = IsTerminator ? LMC.AndTermOpc : LMC.AndOpc;
1231 MI = BuildMI(MBB, Before, DL, TII->get(Opcode), LMC.ExecReg)
1232 .addReg(LMC.ExecReg)
1233 .addReg(LiveMaskReg)
1234 .setOperandDead(3);
1235 }
1236
1238 LIS->removeAllRegUnitsForPhysReg(AMDGPU::EXEC);
1239 StateTransition[MI] = StateExact;
1240}
1241
1242void SIWholeQuadMode::toWQM(MachineBasicBlock &MBB,
1244 Register SavedWQM) {
1245 const DebugLoc &DL = MBB.findDebugLoc(Before);
1246 MachineInstr *MI;
1247
1248 if (SavedWQM) {
1249 MI = BuildMI(MBB, Before, DL, TII->get(AMDGPU::COPY), LMC.ExecReg)
1250 .addReg(SavedWQM);
1251 } else {
1252 MI = BuildMI(MBB, Before, DL, TII->get(LMC.WQMOpc), LMC.ExecReg)
1253 .addReg(LMC.ExecReg)
1254 .setOperandDead(2);
1255 }
1256
1258 StateTransition[MI] = StateWQM;
1259}
1260
1261void SIWholeQuadMode::toStrictMode(MachineBasicBlock &MBB,
1263 Register SaveOrig, char StrictStateNeeded) {
1264 MachineInstr *MI;
1265 assert(SaveOrig);
1266 assert(StrictStateNeeded == StateStrictWWM ||
1267 StrictStateNeeded == StateStrictWQM);
1268
1269 const DebugLoc &DL = MBB.findDebugLoc(Before);
1270
1271 if (StrictStateNeeded == StateStrictWWM) {
1272 MI = BuildMI(MBB, Before, DL, TII->get(AMDGPU::ENTER_STRICT_WWM), SaveOrig)
1273 .addImm(-1)
1274 .setOperandDead(3);
1275 } else {
1276 MI = BuildMI(MBB, Before, DL, TII->get(AMDGPU::ENTER_STRICT_WQM), SaveOrig)
1277 .addImm(-1)
1278 .setOperandDead(3);
1279 }
1281 StateTransition[MI] = StrictStateNeeded;
1282}
1283
1284void SIWholeQuadMode::fromStrictMode(MachineBasicBlock &MBB,
1286 Register SavedOrig, char NonStrictState,
1287 char CurrentStrictState) {
1288 MachineInstr *MI;
1289
1290 assert(SavedOrig);
1291 assert(CurrentStrictState == StateStrictWWM ||
1292 CurrentStrictState == StateStrictWQM);
1293
1294 const DebugLoc &DL = MBB.findDebugLoc(Before);
1295
1296 if (CurrentStrictState == StateStrictWWM) {
1297 MI =
1298 BuildMI(MBB, Before, DL, TII->get(AMDGPU::EXIT_STRICT_WWM), LMC.ExecReg)
1299 .addReg(SavedOrig);
1300 } else {
1301 MI =
1302 BuildMI(MBB, Before, DL, TII->get(AMDGPU::EXIT_STRICT_WQM), LMC.ExecReg)
1303 .addReg(SavedOrig);
1304 }
1306 StateTransition[MI] = NonStrictState;
1307}
1308
1309void SIWholeQuadMode::processBlock(MachineBasicBlock &MBB, BlockInfo &BI,
1310 bool IsEntry) {
1311 // This is a non-entry block that is WQM throughout, so no need to do
1312 // anything.
1313 if (!IsEntry && BI.Needs == StateWQM && BI.OutNeeds != StateExact) {
1314 BI.InitialState = StateWQM;
1315 return;
1316 }
1317
1318 LLVM_DEBUG(dbgs() << "\nProcessing block " << printMBBReference(MBB)
1319 << ":\n");
1320
1321 Register SavedWQMReg;
1322 Register SavedNonStrictReg;
1323 bool WQMFromExec = IsEntry;
1324 char State = (IsEntry || !(BI.InNeeds & StateWQM)) ? StateExact : StateWQM;
1325 char NonStrictState = 0;
1326 const TargetRegisterClass *BoolRC = TRI->getBoolRC();
1327
1328 auto II = MBB.getFirstNonPHI(), IE = MBB.end();
1329 if (IsEntry) {
1330 // Skip the instruction that saves LiveMask
1331 if (II != IE && II->getOpcode() == AMDGPU::COPY &&
1332 II->getOperand(1).getReg() == LMC.ExecReg)
1333 ++II;
1334 }
1335
1336 // This stores the first instruction where it's safe to switch from WQM to
1337 // Exact or vice versa.
1339
1340 // This stores the first instruction where it's safe to switch from Strict
1341 // mode to Exact/WQM or to switch to Strict mode. It must always be the same
1342 // as, or after, FirstWQM since if it's safe to switch to/from Strict, it must
1343 // be safe to switch to/from WQM as well.
1344 MachineBasicBlock::iterator FirstStrict = IE;
1345
1346 // Record initial state is block information.
1347 BI.InitialState = State;
1348
1349 for (unsigned Idx = 0;; ++Idx) {
1351 char Needs = StateExact | StateWQM; // Strict mode is disabled by default.
1352 char OutNeeds = 0;
1353
1354 if (FirstWQM == IE)
1355 FirstWQM = II;
1356
1357 if (FirstStrict == IE)
1358 FirstStrict = II;
1359
1360 // Adjust needs if this is first instruction of WQM requiring shader.
1361 if (IsEntry && Idx == 0 && (BI.InNeeds & StateWQM))
1362 Needs = StateWQM;
1363
1364 // First, figure out the allowed states (Needs) based on the propagated
1365 // flags.
1366 if (II != IE) {
1367 MachineInstr &MI = *II;
1368
1369 if (MI.isTerminator() || TII->mayReadEXEC(*MRI, MI)) {
1370 auto III = Instructions.find(&MI);
1371 if (III != Instructions.end()) {
1372 if (III->second.Needs & StateStrictWWM)
1373 Needs = StateStrictWWM;
1374 else if (III->second.Needs & StateStrictWQM)
1375 Needs = StateStrictWQM;
1376 else if (III->second.Needs & StateWQM)
1377 Needs = StateWQM;
1378 else
1379 Needs &= ~III->second.Disabled;
1380 OutNeeds = III->second.OutNeeds;
1381 }
1382 } else {
1383 // If the instruction doesn't actually need a correct EXEC, then we can
1384 // safely leave Strict mode enabled.
1385 Needs = StateExact | StateWQM | StateStrict;
1386 }
1387
1388 // Exact mode exit can occur in terminators, but must be before branches.
1389 if (MI.isBranch() && OutNeeds == StateExact)
1390 Needs = StateExact;
1391
1392 ++Next;
1393 } else {
1394 // End of basic block
1395 if (BI.OutNeeds & StateWQM)
1396 Needs = StateWQM;
1397 else if (BI.OutNeeds == StateExact)
1398 Needs = StateExact;
1399 else
1400 Needs = StateWQM | StateExact;
1401 }
1402
1403 // Now, transition if necessary.
1404 if (!(Needs & State)) {
1406 if (State == StateStrictWWM || Needs == StateStrictWWM ||
1407 State == StateStrictWQM || Needs == StateStrictWQM) {
1408 // We must switch to or from Strict mode.
1409 First = FirstStrict;
1410 } else {
1411 // We only need to switch to/from WQM, so we can use FirstWQM.
1412 First = FirstWQM;
1413 }
1414
1415 // Whether we need to save SCC depends on start and end states.
1416 bool SaveSCC = false;
1417 switch (State) {
1418 case StateExact:
1419 case StateStrictWWM:
1420 case StateStrictWQM:
1421 // Exact/Strict -> Strict: save SCC
1422 // Exact/Strict -> WQM: save SCC if WQM mask is generated from exec
1423 // Exact/Strict -> Exact: no save
1424 SaveSCC = (Needs & StateStrict) || ((Needs & StateWQM) && WQMFromExec);
1425 break;
1426 case StateWQM:
1427 // WQM -> Exact/Strict: save SCC
1428 SaveSCC = !(Needs & StateWQM);
1429 break;
1430 default:
1431 llvm_unreachable("Unknown state");
1432 break;
1433 }
1434 char StartState = State & StateStrict ? NonStrictState : State;
1435 bool WQMToExact =
1436 StartState == StateWQM && (Needs & StateExact) && !(Needs & StateWQM);
1437 bool ExactToWQM = StartState == StateExact && (Needs & StateWQM) &&
1438 !(Needs & StateExact);
1439 bool PreferLast = Needs == StateWQM;
1440 // Exact regions in divergent control flow may run at EXEC=0, so try to
1441 // exclude instructions with unexpected effects from them.
1442 // FIXME: ideally we would branch over these when EXEC=0,
1443 // but this requires updating implicit values, live intervals and CFG.
1444 if ((WQMToExact && (OutNeeds & StateWQM)) || ExactToWQM) {
1445 for (MachineBasicBlock::iterator I = First; I != II; ++I) {
1446 if (TII->hasUnwantedEffectsWhenEXECEmpty(*I)) {
1447 PreferLast = WQMToExact;
1448 break;
1449 }
1450 }
1451 }
1453 prepareInsertion(MBB, First, II, PreferLast, SaveSCC);
1454
1455 if (State & StateStrict) {
1456 assert(State == StateStrictWWM || State == StateStrictWQM);
1457 assert(SavedNonStrictReg);
1458 fromStrictMode(MBB, Before, SavedNonStrictReg, NonStrictState, State);
1459
1460 LIS->createAndComputeVirtRegInterval(SavedNonStrictReg);
1461 SavedNonStrictReg = 0;
1462 State = NonStrictState;
1463 }
1464
1465 if (Needs & StateStrict) {
1466 NonStrictState = State;
1467 assert(Needs == StateStrictWWM || Needs == StateStrictWQM);
1468 assert(!SavedNonStrictReg);
1469 SavedNonStrictReg = MRI->createVirtualRegister(BoolRC);
1470
1471 toStrictMode(MBB, Before, SavedNonStrictReg, Needs);
1472 State = Needs;
1473 } else {
1474 if (WQMToExact) {
1475 if (!WQMFromExec && (OutNeeds & StateWQM)) {
1476 assert(!SavedWQMReg);
1477 SavedWQMReg = MRI->createVirtualRegister(BoolRC);
1478 }
1479 Before = skipDebugInstructionsForward(Before, MBB.end());
1480 toExact(MBB, Before, SavedWQMReg);
1481 State = StateExact;
1482 } else if (ExactToWQM) {
1483 assert(WQMFromExec == (SavedWQMReg == 0));
1484
1485 toWQM(MBB, Before, SavedWQMReg);
1486
1487 if (SavedWQMReg) {
1488 LIS->createAndComputeVirtRegInterval(SavedWQMReg);
1489 SavedWQMReg = 0;
1490 }
1491 State = StateWQM;
1492 } else {
1493 // We can get here if we transitioned from StrictWWM to a
1494 // non-StrictWWM state that already matches our needs, but we
1495 // shouldn't need to do anything.
1496 assert(Needs & State);
1497 }
1498 }
1499 }
1500
1501 if (Needs != (StateExact | StateWQM | StateStrict)) {
1502 if (Needs != (StateExact | StateWQM))
1503 FirstWQM = IE;
1504 FirstStrict = IE;
1505 }
1506
1507 if (II == IE)
1508 break;
1509
1510 II = Next;
1511 }
1512 assert(!SavedWQMReg);
1513 assert(!SavedNonStrictReg);
1514}
1515
1516bool SIWholeQuadMode::lowerLiveMaskQueries() {
1517 for (MachineInstr *MI : LiveMaskQueries) {
1518 const DebugLoc &DL = MI->getDebugLoc();
1519 Register Dest = MI->getOperand(0).getReg();
1520
1521 MachineInstr *Copy =
1522 BuildMI(*MI->getParent(), MI, DL, TII->get(AMDGPU::COPY), Dest)
1523 .addReg(LiveMaskReg);
1524
1525 LIS->ReplaceMachineInstrInMaps(*MI, *Copy);
1526 MI->eraseFromParent();
1527 }
1528 return !LiveMaskQueries.empty();
1529}
1530
1531bool SIWholeQuadMode::lowerCopyInstrs() {
1532 for (MachineInstr *MI : LowerToMovInstrs) {
1533 assert(MI->getNumExplicitOperands() == 2);
1534
1535 const Register Reg = MI->getOperand(0).getReg();
1536
1537 const TargetRegisterClass *regClass =
1538 TRI->getRegClassForOperandReg(*MRI, MI->getOperand(0));
1539 if (TRI->isVGPRClass(regClass)) {
1540 const unsigned MovOp = TII->getMovOpcode(regClass);
1541 MI->setDesc(TII->get(MovOp));
1542
1543 // Check that it already implicitly depends on exec (like all VALU movs
1544 // should do).
1545 assert(any_of(MI->implicit_operands(), [](const MachineOperand &MO) {
1546 return MO.isUse() && MO.getReg() == AMDGPU::EXEC;
1547 }));
1548 } else {
1549 // Remove early-clobber and exec dependency from simple SGPR copies.
1550 // This allows some to be eliminated during/post RA.
1551 LLVM_DEBUG(dbgs() << "simplify SGPR copy: " << *MI);
1552 if (MI->getOperand(0).isEarlyClobber()) {
1553 LIS->removeInterval(Reg);
1554 MI->getOperand(0).setIsEarlyClobber(false);
1556 }
1557 int Index = MI->findRegisterUseOperandIdx(AMDGPU::EXEC, /*TRI=*/nullptr);
1558 while (Index >= 0) {
1559 MI->removeOperand(Index);
1560 Index = MI->findRegisterUseOperandIdx(AMDGPU::EXEC, /*TRI=*/nullptr);
1561 }
1562 MI->setDesc(TII->get(AMDGPU::COPY));
1563 LLVM_DEBUG(dbgs() << " -> " << *MI);
1564 }
1565 }
1566 for (MachineInstr *MI : LowerToCopyInstrs) {
1567 LLVM_DEBUG(dbgs() << "simplify: " << *MI);
1568
1569 if (MI->getOpcode() == AMDGPU::V_SET_INACTIVE_B32) {
1570 assert(MI->getNumExplicitOperands() == 6);
1571
1572 LiveInterval *RecomputeLI = nullptr;
1573 if (MI->getOperand(4).isReg())
1574 RecomputeLI = &LIS->getInterval(MI->getOperand(4).getReg());
1575
1576 MI->removeOperand(5);
1577 MI->removeOperand(4);
1578 MI->removeOperand(3);
1579 MI->removeOperand(1);
1580
1581 if (RecomputeLI)
1582 LIS->shrinkToUses(RecomputeLI);
1583 } else {
1584 assert(MI->getNumExplicitOperands() == 2);
1585 }
1586
1587 unsigned CopyOp = MI->getOperand(1).isReg()
1588 ? (unsigned)AMDGPU::COPY
1589 : TII->getMovOpcode(TRI->getRegClassForOperandReg(
1590 *MRI, MI->getOperand(0)));
1591 MI->setDesc(TII->get(CopyOp));
1592 LLVM_DEBUG(dbgs() << " -> " << *MI);
1593 }
1594 return !LowerToCopyInstrs.empty() || !LowerToMovInstrs.empty();
1595}
1596
1597bool SIWholeQuadMode::lowerKillInstrs(bool IsWQM) {
1598 for (MachineInstr *MI : KillInstrs) {
1599 MachineInstr *SplitPoint = nullptr;
1600 switch (MI->getOpcode()) {
1601 case AMDGPU::SI_DEMOTE_I1:
1602 case AMDGPU::SI_KILL_I1_TERMINATOR:
1603 SplitPoint = lowerKillI1(*MI, IsWQM);
1604 break;
1605 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
1606 SplitPoint = lowerKillF32(*MI);
1607 break;
1608 }
1609 if (SplitPoint)
1610 splitBlock(SplitPoint);
1611 }
1612 return !KillInstrs.empty();
1613}
1614
1615void SIWholeQuadMode::lowerInitExec(MachineInstr &MI) {
1616 MachineBasicBlock *MBB = MI.getParent();
1617
1618 if (MI.getOpcode() == AMDGPU::SI_INIT_WHOLE_WAVE) {
1619 assert(MBB == &MBB->getParent()->front() &&
1620 "init whole wave not in entry block");
1621 Register EntryExec = MRI->createVirtualRegister(TRI->getBoolRC());
1622 MachineInstr *SaveExec = BuildMI(*MBB, MBB->begin(), MI.getDebugLoc(),
1623 TII->get(LMC.OrSaveExecOpc), EntryExec)
1624 .addImm(-1)
1625 .setOperandDead(3);
1626
1627 // Replace all uses of MI's destination reg with EntryExec.
1628 MRI->replaceRegWith(MI.getOperand(0).getReg(), EntryExec);
1629
1630 if (LIS) {
1632 }
1633
1634 MI.eraseFromParent();
1635
1636 if (LIS) {
1637 LIS->InsertMachineInstrInMaps(*SaveExec);
1638 LIS->createAndComputeVirtRegInterval(EntryExec);
1639 }
1640 return;
1641 }
1642
1643 if (MI.getOpcode() == AMDGPU::SI_INIT_EXEC) {
1644 // This should be before all vector instructions.
1645 MachineInstr *InitMI = BuildMI(*MBB, MBB->begin(), MI.getDebugLoc(),
1646 TII->get(LMC.MovOpc), LMC.ExecReg)
1647 .addImm(MI.getOperand(0).getImm());
1648 if (LIS) {
1650 LIS->InsertMachineInstrInMaps(*InitMI);
1651 }
1652 MI.eraseFromParent();
1653 return;
1654 }
1655
1656 // Extract the thread count from an SGPR input and set EXEC accordingly.
1657 // Since BFM can't shift by 64, handle that case with CMP + CMOV.
1658 //
1659 // S_BFE_U32 count, input, {shift, 7}
1660 // S_BFM_B64 exec, count, 0
1661 // S_CMP_EQ_U32 count, 64
1662 // S_CMOV_B64 exec, -1
1663 Register InputReg = MI.getOperand(0).getReg();
1664 MachineInstr *FirstMI = &*MBB->begin();
1665 if (InputReg.isVirtual()) {
1666 MachineInstr *DefInstr = MRI->getVRegDef(InputReg);
1667 assert(DefInstr && DefInstr->isCopy());
1668 if (DefInstr->getParent() == MBB) {
1669 if (DefInstr != FirstMI) {
1670 // If the `InputReg` is defined in current block, we also need to
1671 // move that instruction to the beginning of the block.
1672 DefInstr->removeFromParent();
1673 MBB->insert(FirstMI, DefInstr);
1674 if (LIS)
1675 LIS->handleMove(*DefInstr);
1676 } else {
1677 // If first instruction is definition then move pointer after it.
1678 FirstMI = &*std::next(FirstMI->getIterator());
1679 }
1680 }
1681 }
1682
1683 // Insert instruction sequence at block beginning (before vector operations).
1684 const DebugLoc &DL = MI.getDebugLoc();
1685 const unsigned WavefrontSize = ST->getWavefrontSize();
1686 const unsigned Mask = (WavefrontSize << 1) - 1;
1687 Register CountReg = MRI->createVirtualRegister(&AMDGPU::SGPR_32RegClass);
1688 auto BfeMI = BuildMI(*MBB, FirstMI, DL, TII->get(AMDGPU::S_BFE_U32), CountReg)
1689 .addReg(InputReg)
1690 .addImm((MI.getOperand(1).getImm() & Mask) | 0x70000)
1691 .setOperandDead(3);
1692 auto BfmMI = BuildMI(*MBB, FirstMI, DL, TII->get(LMC.BfmOpc), LMC.ExecReg)
1693 .addReg(CountReg)
1694 .addImm(0);
1695 auto CmpMI = BuildMI(*MBB, FirstMI, DL, TII->get(AMDGPU::S_CMP_EQ_U32))
1696 .addReg(CountReg, RegState::Kill)
1697 .addImm(WavefrontSize);
1698 auto CmovMI =
1699 BuildMI(*MBB, FirstMI, DL, TII->get(LMC.CMovOpc), LMC.ExecReg).addImm(-1);
1700
1701 if (!LIS) {
1702 MI.eraseFromParent();
1703 return;
1704 }
1705
1707 MI.eraseFromParent();
1708
1709 LIS->InsertMachineInstrInMaps(*BfeMI);
1710 LIS->InsertMachineInstrInMaps(*BfmMI);
1711 LIS->InsertMachineInstrInMaps(*CmpMI);
1712 LIS->InsertMachineInstrInMaps(*CmovMI);
1713
1714 LIS->removeInterval(InputReg);
1715 LIS->createAndComputeVirtRegInterval(InputReg);
1716 LIS->createAndComputeVirtRegInterval(CountReg);
1717}
1718
1719/// Lower INIT_EXEC instructions. Return a suitable insert point in \p Entry
1720/// for instructions that depend on EXEC.
1722SIWholeQuadMode::lowerInitExecInstrs(MachineBasicBlock &Entry, bool &Changed) {
1723 MachineBasicBlock::iterator InsertPt = Entry.getFirstNonPHI();
1724
1725 for (MachineInstr *MI : InitExecInstrs) {
1726 // Try to handle undefined cases gracefully:
1727 // - multiple INIT_EXEC instructions
1728 // - INIT_EXEC instructions not in the entry block
1729 if (MI->getParent() == &Entry)
1730 InsertPt = std::next(MI->getIterator());
1731
1732 lowerInitExec(*MI);
1733 Changed = true;
1734 }
1735
1736 return InsertPt;
1737}
1738
1739bool SIWholeQuadMode::run(MachineFunction &MF) {
1740 LLVM_DEBUG(dbgs() << "SI Whole Quad Mode on " << MF.getName()
1741 << " ------------- \n");
1742 LLVM_DEBUG(MF.dump(););
1743
1744 Instructions.clear();
1745 Blocks.clear();
1746 LiveMaskQueries.clear();
1747 LowerToCopyInstrs.clear();
1748 LowerToMovInstrs.clear();
1749 KillInstrs.clear();
1750 InitExecInstrs.clear();
1751 SetInactiveInstrs.clear();
1752 StateTransition.clear();
1753
1754 const char GlobalFlags = analyzeFunction(MF);
1755 bool Changed = false;
1756
1757 LiveMaskReg = LMC.ExecReg;
1758
1759 MachineBasicBlock &Entry = MF.front();
1760 MachineBasicBlock::iterator EntryMI = lowerInitExecInstrs(Entry, Changed);
1761
1762 // Store a copy of the original live mask when required
1763 const bool HasLiveMaskQueries = !LiveMaskQueries.empty();
1764 const bool HasWaveModes = GlobalFlags & ~StateExact;
1765 const bool HasKills = !KillInstrs.empty();
1766 const bool UsesWQM = GlobalFlags & StateWQM;
1767 if (HasKills || UsesWQM || (HasWaveModes && HasLiveMaskQueries)) {
1768 LiveMaskReg = MRI->createVirtualRegister(TRI->getBoolRC());
1769 MachineInstr *MI =
1770 BuildMI(Entry, EntryMI, DebugLoc(), TII->get(AMDGPU::COPY), LiveMaskReg)
1771 .addReg(LMC.ExecReg);
1773 Changed = true;
1774 }
1775
1776 // Check if V_SET_INACTIVE was touched by a strict state mode.
1777 // If so, promote to WWM; otherwise lower to COPY.
1778 for (MachineInstr *MI : SetInactiveInstrs) {
1779 if (LowerToCopyInstrs.contains(MI))
1780 continue;
1781 auto &Info = Instructions[MI];
1782 if (Info.MarkedStates & StateStrict) {
1783 Info.Needs |= StateStrictWWM;
1784 Info.Disabled &= ~StateStrictWWM;
1785 Blocks[MI->getParent()].Needs |= StateStrictWWM;
1786 } else {
1787 LLVM_DEBUG(dbgs() << "Has no WWM marking: " << *MI);
1788 LowerToCopyInstrs.insert(MI);
1789 }
1790 }
1791
1792 LLVM_DEBUG(printInfo());
1793
1794 Changed |= lowerLiveMaskQueries();
1795 Changed |= lowerCopyInstrs();
1796
1797 if (!HasWaveModes) {
1798 // No wave mode execution
1799 Changed |= lowerKillInstrs(false);
1800 } else if (GlobalFlags == StateWQM) {
1801 // Shader only needs WQM
1802 auto MI =
1803 BuildMI(Entry, EntryMI, DebugLoc(), TII->get(LMC.WQMOpc), LMC.ExecReg)
1804 .addReg(LMC.ExecReg)
1805 .setOperandDead(2);
1807 lowerKillInstrs(true);
1808 Changed = true;
1809 } else {
1810 // Mark entry for WQM if required.
1811 if (GlobalFlags & StateWQM)
1812 Blocks[&Entry].InNeeds |= StateWQM;
1813 // Wave mode switching requires full lowering pass.
1814 for (auto &BII : Blocks)
1815 processBlock(*BII.first, BII.second, BII.first == &Entry);
1816 // Lowering blocks causes block splitting so perform as a second pass.
1817 for (auto &BII : Blocks)
1818 lowerBlock(*BII.first, BII.second);
1819 Changed = true;
1820 }
1821
1822 // Compute live range for live mask
1823 if (LiveMaskReg != LMC.ExecReg)
1824 LIS->createAndComputeVirtRegInterval(LiveMaskReg);
1825
1826 // Physical registers like SCC aren't tracked by default anyway, so just
1827 // removing the ranges we computed is the simplest option for maintaining
1828 // the analysis results.
1829 LIS->removeAllRegUnitsForPhysReg(AMDGPU::SCC);
1830
1831 // If we performed any kills then recompute EXEC
1832 if (!KillInstrs.empty() || !InitExecInstrs.empty())
1833 LIS->removeAllRegUnitsForPhysReg(AMDGPU::EXEC);
1834
1835 return Changed;
1836}
1837
1838bool SIWholeQuadModeLegacy::runOnMachineFunction(MachineFunction &MF) {
1839 LiveIntervals *LIS = &getAnalysis<LiveIntervalsWrapperPass>().getLIS();
1840 auto *MDTWrapper = getAnalysisIfAvailable<MachineDominatorTreeWrapperPass>();
1841 MachineDominatorTree *MDT = MDTWrapper ? &MDTWrapper->getDomTree() : nullptr;
1842 auto *PDTWrapper =
1843 getAnalysisIfAvailable<MachinePostDominatorTreeWrapperPass>();
1844 MachinePostDominatorTree *PDT =
1845 PDTWrapper ? &PDTWrapper->getPostDomTree() : nullptr;
1846 SIWholeQuadMode Impl(MF, LIS, MDT, PDT);
1847 return Impl.run(MF);
1848}
1849
1850PreservedAnalyses
1853 MFPropsModifier _(*this, MF);
1854
1860 SIWholeQuadMode Impl(MF, LIS, MDT, PDT);
1861 bool Changed = Impl.run(MF);
1862 if (!Changed)
1863 return PreservedAnalyses::all();
1864
1870 return PA;
1871}
MachineInstrBuilder & UseMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static void analyzeFunction(Function &Fn, const DataLayout &Layout, FunctionVarLocsBuilder *FnVarLocs)
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:686
AMD GCN specific subclass of TargetSubtarget.
#define DEBUG_TYPE
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t IntrinsicInst * II
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
static void splitBlock(MachineBasicBlock &MBB, MachineInstr &MI, MachineDominatorTree *MDT, MachineLoopInfo *MLI)
SI Optimize VGPR LiveRange
#define LLVM_DEBUG(...)
Definition Debug.h:119
unsigned getWavefrontSize() const
static const LaneMaskConstants & get(const GCNSubtarget &ST)
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
void applyUpdates(ArrayRef< UpdateType > Updates)
Inform the dominator tree about a sequence of CFG edge insertions and deletions and perform a batch u...
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
void removeAllRegUnitsForPhysReg(MCRegister Reg)
Remove associated live ranges for the register units associated with Reg.
MachineInstr * getInstructionFromIndex(SlotIndex index) const
Returns the instruction associated with the given index.
SlotIndex InsertMachineInstrInMaps(MachineInstr &MI)
LLVM_ABI void handleMove(MachineInstr &MI, bool UpdateFlags=false)
Call this method to notify LiveIntervals that instruction MI has been moved within a basic block.
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
void RemoveMachineInstrFromMaps(MachineInstr &MI)
SlotIndex getMBBEndIdx(const MachineBasicBlock *mbb) const
Return the last index in the given basic block.
LiveInterval & getInterval(Register Reg)
void removeInterval(Register Reg)
Interval removal.
LiveRange & getRegUnit(MCRegUnit Unit)
Return the live range for register unit Unit.
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
MachineBasicBlock * getMBBFromIndex(SlotIndex index) const
LiveInterval & createAndComputeVirtRegInterval(Register Reg)
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
VNInfo * valueIn() const
Return the value that is live-in to the instruction.
This class represents the liveness of a register, stack slot, etc.
const Segment * getSegmentContaining(SlotIndex Idx) const
Return the segment that contains the specified index, or null if there is none.
LiveQueryResult Query(SlotIndex Idx) const
Query Liveness at Idx.
VNInfo * getVNInfoBefore(SlotIndex Idx) const
getVNInfoBefore - Return the VNInfo that is live up to but not necessarily including Idx,...
static MCRegister from(unsigned Val)
Check the provided unsigned value is a valid MCRegister.
Definition MCRegister.h:77
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
MachineInstr * remove(MachineInstr *I)
Remove the unbundled instruction from the instruction list without deleting it.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI DebugLoc findDebugLoc(instr_iterator MBBI)
Find the next valid DebugLoc starting at MBBI, skipping any debug instructions.
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
reverse_iterator rbegin()
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
Analysis pass which computes a MachineDominatorTree.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
void dump() const
dump - Print the current MachineFunction to cerr, useful for debugger use.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineBasicBlock & front() const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isCopy() const
LLVM_ABI MachineInstr * removeFromParent()
Unlink 'this' from the containing basic block, and return it without deleting it.
const MachineBasicBlock * getParent() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
Register getReg() const
getReg - Returns the register number.
MachinePostDominatorTree - an analysis pass wrapper for DominatorTree used to compute the post-domina...
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI LaneBitmask getMaxLaneMaskForVReg(Register Reg) const
Returns a mask covering all bits that can appear in lane masks of subregisters of the virtual registe...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
This class implements a map that also provides access to all stored values in a deterministic order.
Definition MapVector.h:38
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Definition Analysis.h:132
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
SlotIndex getBaseIndex() const
Returns the base index for associated with this index.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
Definition SmallSet.h:176
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Wrapper class representing a virtual register or register unit.
Definition Register.h:175
constexpr bool isVirtualReg() const
Definition Register.h:191
constexpr Register asVirtualReg() const
Definition Register.h:200
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char WavefrontSize[]
Key for Kernel::CodeProps::Metadata::mWavefrontSize.
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ Entry
Definition COFF.h:862
NodeAddr< PhiNode * > Phi
Definition RDFGraph.h:390
This is an optimization pass for GlobalISel generic memory operations.
IterT next_nodbg(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It, then continue incrementing it while it points to a debug instruction.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
DominatorTreeBase< T, false > DomTreeBase
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
char & SIWholeQuadModeID
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
@ Disabled
Don't do any conversion of .debug_str_offsets tables.
Definition DWP.h:30
LLVM_ABI Printable printMBBReference(const MachineBasicBlock &MBB)
Prints a machine basic block reference.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
WorkItem(const BasicBlock *BB, int St)
static constexpr LaneBitmask getAll()
Definition LaneBitmask.h:82
constexpr bool any() const
Definition LaneBitmask.h:53
static constexpr LaneBitmask getNone()
Definition LaneBitmask.h:81