LLVM 24.0.0git
SIInstrInfo.h
Go to the documentation of this file.
1//===- SIInstrInfo.h - SI Instruction Info Interface ------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Interface definition for SIInstrInfo.
11//
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
15#define LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
16
17#include "AMDGPUMIRFormatter.h"
19#include "SIRegisterInfo.h"
21#include "llvm/ADT/SetVector.h"
25
26#define GET_INSTRINFO_HEADER
27#include "AMDGPUGenInstrInfo.inc"
28
29namespace llvm {
30
31class APInt;
32class GCNSubtarget;
33class MachineDominatorTree;
34class MachineRegisterInfo;
35class RegScavenger;
36class SIMachineFunctionInfo;
37class MCRegisterClass;
38using TargetRegisterClass = MCRegisterClass;
39class ScheduleHazardRecognizer;
40
41constexpr unsigned DefaultMemoryClusterDWordsLimit = 8;
42
43/// Mark the MMO of a uniform load if there are no potentially clobbering stores
44/// on any path from the start of an entry function to this load.
47
48/// Mark the MMO of a load as the last use.
51
52/// Mark the MMO of cooperative load/store atomics.
55
57 // Operands that need to replaced by waterfall
59 // Target physical registers replacing the MOs
61};
62/// Mark the MMO of accesses to memory locations that are
63/// never written to by other threads.
66
67/// Utility to store machine instructions worklist.
69 SIInstrWorklist() = default;
70
71 void insert(MachineInstr *MI);
72
73 MachineInstr *top() const { return InstrList[Front]; }
74
75 void erase_top() {
76 InSet.erase(InstrList[Front]);
77 ++Front;
78 }
79
80 bool empty() const { return Front == InstrList.size(); }
81
82 void clear() {
83 InstrList.clear();
84 Front = 0;
85 InSet.clear();
86 DeferredList.clear();
87 }
88
90
91 SetVector<MachineInstr *> &getDeferredList() { return DeferredList; }
92
93private:
94 /// InstrList contains the MachineInstrs.
97 unsigned Front = 0;
98 /// Deferred instructions are specific MachineInstr
99 /// that will be added by insert method.
100 SetVector<MachineInstr *> DeferredList;
101};
102
103// In namespace llvm so ADL finds it when SIInstrFlags predicates are
104// instantiated with MachineInstr (MachineInstr is in namespace llvm).
106 return MI.getDesc().TSFlags;
107}
108
109class SIInstrInfo final : public AMDGPUGenInstrInfo {
110 struct ThreeAddressUpdates;
111
112private:
113 const SIRegisterInfo RI;
114 const GCNSubtarget &ST;
115 TargetSchedModel SchedModel;
116 mutable std::unique_ptr<AMDGPUMIRFormatter> Formatter;
117
118 // The inverse predicate should have the negative value.
119 enum BranchPredicate {
120 INVALID_BR = 0,
121 SCC_TRUE = 1,
122 SCC_FALSE = -1,
123 VCCNZ = 2,
124 VCCZ = -2,
125 EXECNZ = -3,
126 EXECZ = 3
127 };
128
129 using SetVectorType = SmallSetVector<MachineInstr *, 32>;
130
131 static unsigned getBranchOpcode(BranchPredicate Cond);
132 static BranchPredicate getBranchPredicate(unsigned Opcode);
133
134public:
137 const MachineOperand &SuperReg,
138 const TargetRegisterClass *SuperRC,
139 unsigned SubIdx,
140 const TargetRegisterClass *SubRC) const;
143 const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC,
144 unsigned SubIdx, const TargetRegisterClass *SubRC) const;
145
146private:
147 bool optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
148 bool NeedInversion) const;
149
150 bool invertSCCUse(MachineInstr *SCCDef) const;
151
152 void swapOperands(MachineInstr &Inst) const;
153
154 std::pair<bool, MachineBasicBlock *>
155 moveScalarAddSub(SIInstrWorklist &Worklist, MachineInstr &Inst,
156 MachineDominatorTree *MDT = nullptr) const;
157
158 void lowerSelect(SIInstrWorklist &Worklist, MachineInstr &Inst,
159 MachineDominatorTree *MDT = nullptr) const;
160
161 void lowerScalarAbs(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
162
163 void lowerScalarAbsDiff(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
164
165 void lowerScalarXnor(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
166
167 void splitScalarNotBinop(SIInstrWorklist &Worklist, MachineInstr &Inst,
168 unsigned Opcode) const;
169
170 void splitScalarBinOpN2(SIInstrWorklist &Worklist, MachineInstr &Inst,
171 unsigned Opcode) const;
172
173 void splitScalar64BitUnaryOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
174 unsigned Opcode, bool Swap = false) const;
175
176 void splitScalar64BitBinaryOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
177 unsigned Opcode,
178 MachineDominatorTree *MDT = nullptr) const;
179
180 void splitScalarSMulU64(SIInstrWorklist &Worklist, MachineInstr &Inst,
181 MachineDominatorTree *MDT) const;
182
183 void splitScalarSMulPseudo(SIInstrWorklist &Worklist, MachineInstr &Inst,
184 MachineDominatorTree *MDT) const;
185
186 void splitScalar64BitXnor(SIInstrWorklist &Worklist, MachineInstr &Inst,
187 MachineDominatorTree *MDT = nullptr) const;
188
189 void splitScalar64BitBCNT(SIInstrWorklist &Worklist,
190 MachineInstr &Inst) const;
191 void splitScalar64BitBFE(SIInstrWorklist &Worklist, MachineInstr &Inst) const;
192 void splitScalar64BitCountOp(SIInstrWorklist &Worklist, MachineInstr &Inst,
193 unsigned Opcode,
194 MachineDominatorTree *MDT = nullptr) const;
195 void movePackToVALU(SIInstrWorklist &Worklist, MachineRegisterInfo &MRI,
196 MachineInstr &Inst) const;
197
198 void addUsersToMoveToVALUWorklist(Register Reg, MachineRegisterInfo &MRI,
199 SIInstrWorklist &Worklist) const;
200
201 void addSCCDefUsersToVALUWorklist(const MachineOperand &Op,
202 MachineInstr &SCCDefInst,
203 SIInstrWorklist &Worklist,
204 Register NewCond = Register()) const;
205 void addSCCDefsToVALUWorklist(MachineInstr *SCCUseInst,
206 SIInstrWorklist &Worklist) const;
207
208 const TargetRegisterClass *
209 getDestEquivalentVGPRClass(const MachineInstr &Inst) const;
210
211 bool checkInstOffsetsDoNotOverlap(const MachineInstr &MIa,
212 const MachineInstr &MIb) const;
213
214 Register findUsedSGPR(const MachineInstr &MI, int OpIndices[3]) const;
215
216 bool verifyCopy(const MachineInstr &MI, const MachineRegisterInfo &MRI,
217 StringRef &ErrInfo) const;
218
219 bool resultDependsOnExec(const MachineInstr &MI) const;
220
221 MachineInstr *convertToThreeAddressImpl(MachineInstr &MI,
222 ThreeAddressUpdates &Updates) const;
223
224protected:
225 /// If the specific machine instruction is a instruction that moves/copies
226 /// value from one register to another register return destination and source
227 /// registers as machine operands.
228 std::optional<DestSourcePair>
229 isCopyInstrImpl(const MachineInstr &MI) const override;
230
232 AMDGPU::OpName Src0OpName, MachineOperand &Src1,
233 AMDGPU::OpName Src1OpName) const;
234 bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx,
235 unsigned toIdx) const;
236 bool isNonCommutableDPP(const MachineInstr &MI) const;
238 unsigned OpIdx0,
239 unsigned OpIdx1) const override;
240
241public:
243 MO_MASK = 0xf,
244
246 // MO_GOTPCREL -> symbol@GOTPCREL -> R_AMDGPU_GOTPCREL.
248 // MO_GOTPCREL32_LO -> symbol@gotpcrel32@lo -> R_AMDGPU_GOTPCREL32_LO.
251 // MO_GOTPCREL32_HI -> symbol@gotpcrel32@hi -> R_AMDGPU_GOTPCREL32_HI.
253 // MO_GOTPCREL64 -> symbol@GOTPCREL -> R_AMDGPU_GOTPCREL.
255 // MO_REL32_LO -> symbol@rel32@lo -> R_AMDGPU_REL32_LO.
258 // MO_REL32_HI -> symbol@rel32@hi -> R_AMDGPU_REL32_HI.
261
263
267 };
268
269 explicit SIInstrInfo(const GCNSubtarget &ST);
270
272 return RI;
273 }
274
275 // FIXME: This is inaccurate and needs to account for use context. Normal asm
276 // constraints should use 64-bit pointers.
278 InlineAsm::ConstraintCode C) const override {
279 return &AMDGPU::VGPR_32RegClass;
280 }
281
282 const GCNSubtarget &getSubtarget() const {
283 return ST;
284 }
285
286 bool isReMaterializableImpl(const MachineInstr &MI) const override;
287
288 bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override;
289
290 bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo,
291 MachineCycleInfo *CI) const override;
292
293 bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0,
294 int64_t &Offset1) const override;
295
296 bool isGlobalMemoryObject(const MachineInstr *MI) const override;
297
299 const MachineInstr &LdSt,
301 bool &OffsetIsScalable, LocationSize &Width) const final;
302
304 int64_t Offset1, bool OffsetIsScalable1,
306 int64_t Offset2, bool OffsetIsScalable2,
307 unsigned ClusterSize,
308 unsigned NumBytes) const override;
309
310 bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0,
311 int64_t Offset1, unsigned NumLoads) const override;
312
314 const DebugLoc &DL, Register DestReg, Register SrcReg,
315 bool KillSrc, bool RenamableDest = false,
316 bool RenamableSrc = false) const override;
317
318private:
319 void storeRegToStackSlotImpl(MachineBasicBlock &MBB,
321 bool isKill, int FrameIndex,
322 const TargetRegisterClass *RC, Register VReg,
323 MachineInstr::MIFlag Flags, bool NeedsCFI) const;
324
325public:
328 bool isKill, int FrameIndex,
329 const TargetRegisterClass *RC) const;
330
332 int64_t &ImmVal) const override;
333
334 std::optional<int64_t>
336 const MachineOperand &Op,
337 MachineInstr **DefMI = nullptr) const;
338 std::optional<int64_t>
340 MachineInstr **DefMI = nullptr) const;
341
343 const TargetRegisterClass *RC,
344 unsigned Size,
345 const SIMachineFunctionInfo &MFI,
346 bool NeedsCFI) const;
347 unsigned
349 unsigned Size,
350 const SIMachineFunctionInfo &MFI) const;
351
354 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
355 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
356
359 int FrameIndex, const TargetRegisterClass *RC, Register VReg,
360 unsigned SubReg = 0,
361 MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
362
363 bool expandPostRAPseudo(MachineInstr &MI) const override;
364
365 void
367 Register DestReg, unsigned SubIdx, const MachineInstr &Orig,
368 LaneBitmask UsedLanes = LaneBitmask::getAll()) const override;
369
370 // Splits a V_MOV_B64_DPP_PSEUDO opcode into a pair of v_mov_b32_dpp
371 // instructions. Returns a pair of generated instructions.
372 // Can split either post-RA with physical registers or pre-RA with
373 // virtual registers. In latter case IR needs to be in SSA form and
374 // and a REG_SEQUENCE is produced to define original register.
375 std::pair<MachineInstr*, MachineInstr*>
377
378 // Returns an opcode that can be used to move a value to a \p DstRC
379 // register. If there is no hardware instruction that can store to \p
380 // DstRC, then AMDGPU::COPY is returned.
381 unsigned getMovOpcode(const TargetRegisterClass *DstRC) const;
382
383 const MCInstrDesc &getIndirectRegWriteMovRelPseudo(unsigned VecSize,
384 unsigned EltSize,
385 bool IsSGPR) const;
386
387 const MCInstrDesc &getIndirectGPRIDXPseudo(unsigned VecSize,
388 bool IsIndirectSrc) const;
390 int commuteOpcode(unsigned Opc) const;
391
393 inline int commuteOpcode(const MachineInstr &MI) const {
394 return commuteOpcode(MI.getOpcode());
395 }
396
397 bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0,
398 unsigned &SrcOpIdx1) const override;
399
400 bool findCommutedOpIndices(const MCInstrDesc &Desc, unsigned &SrcOpIdx0,
401 unsigned &SrcOpIdx1) const;
402
403 bool isBranchOffsetInRange(unsigned BranchOpc,
404 int64_t BrOffset) const override;
405
406 MachineBasicBlock *getBranchDestBlock(const MachineInstr &MI) const override;
407
408 /// Return whether the block terminate with divergent branch.
409 /// Note this only work before lowering the pseudo control flow instructions.
410 bool hasDivergentBranch(const MachineBasicBlock *MBB) const;
411
413 MachineBasicBlock &NewDestBB,
414 MachineBasicBlock &RestoreBB, const DebugLoc &DL,
415 int64_t BrOffset, RegScavenger *RS) const override;
416
420 MachineBasicBlock *&FBB,
422 bool AllowModify) const;
423
425 MachineBasicBlock *&FBB,
427 bool AllowModify = false) const override;
428
430 int *BytesRemoved = nullptr) const override;
431
434 const DebugLoc &DL,
435 int *BytesAdded = nullptr) const override;
436
438 SmallVectorImpl<MachineOperand> &Cond) const override;
439
440 std::unique_ptr<PipelinerLoopInfo>
441 analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override;
442
445 Register TrueReg, Register FalseReg, int &CondCycles,
446 int &TrueCycles, int &FalseCycles) const override;
447
451 Register TrueReg, Register FalseReg) const override;
452
453 bool analyzeCompare(const MachineInstr &MI, Register &SrcReg,
454 Register &SrcReg2, int64_t &CmpMask,
455 int64_t &CmpValue) const override;
456
457 bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg,
458 Register SrcReg2, int64_t CmpMask, int64_t CmpValue,
459 const MachineRegisterInfo *MRI) const override;
460
461 bool
463 const MachineInstr &MIb) const override;
464
465 static bool isFoldableCopy(const MachineInstr &MI);
466 static unsigned getFoldableCopySrcIdx(const MachineInstr &MI);
467
468 void removeModOperands(MachineInstr &MI) const;
469
471 const MCInstrDesc &NewDesc) const;
472
473 /// Return the extracted immediate value in a subregister use from a constant
474 /// materialized in a super register.
475 ///
476 /// e.g. %imm = S_MOV_B64 K[0:63]
477 /// USE %imm.sub1
478 /// This will return K[32:63]
479 static std::optional<int64_t> extractSubregFromImm(int64_t ImmVal,
480 unsigned SubRegIndex);
481
483 MachineRegisterInfo *MRI) const final;
484
485 unsigned getMachineCSELookAheadLimit() const override { return 500; }
486
488 LiveIntervals *LIS) const override;
489
491 const MachineBasicBlock *MBB,
492 const MachineFunction &MF) const override;
493
494 static bool isSALU(const MachineInstr &MI) {
495 return SIInstrFlags::isSALU(MI);
496 }
497
498 bool isSALU(uint32_t Opcode) const {
499 return SIInstrFlags::isSALU(get(Opcode));
500 }
501
502 static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA) {
503 if (!AllowLDSDMA && isLDSDMA(MI))
504 return false;
505
506 return SIInstrFlags::isVALU(MI);
507 }
508
509 /// LDSDMA instructions act as both VALU and memory instructions, thus
510 /// we also tag them as VALU. However, in many places, we do not actually want
511 /// to include LDSDMA instructions in this query. By setting \p AllowLDSDMA to
512 /// false, this will return false for LDSDMA instructions.
513 bool isVALU(uint32_t Opcode, bool AllowLDSDMA) const {
514 if (!AllowLDSDMA && isLDSDMA(Opcode))
515 return false;
516
517 return SIInstrFlags::isVALU(get(Opcode));
518 }
519
520 static bool isImage(const MachineInstr &MI) {
522 }
523
524 bool isImage(uint32_t Opcode) const {
525 return SIInstrFlags::isImage(get(Opcode));
526 }
527
528 static bool isVMEM(const MachineInstr &MI) {
529 return SIInstrFlags::isVMEM(MI);
530 }
531
532 bool isVMEM(uint32_t Opcode) const {
533 return SIInstrFlags::isVMEM(get(Opcode));
534 }
535
536 /// True if MI implicitly drains XCNT.
537 static bool isXcntDrain(const MachineInstr &MI);
538
539 static bool isSOP1(const MachineInstr &MI) {
540 return SIInstrFlags::isSOP1(MI);
541 }
542
543 bool isSOP1(uint32_t Opcode) const {
544 return SIInstrFlags::isSOP1(get(Opcode));
545 }
546
547 static bool isSOP2(const MachineInstr &MI) {
548 return SIInstrFlags::isSOP2(MI);
549 }
550
551 bool isSOP2(uint32_t Opcode) const {
552 return SIInstrFlags::isSOP2(get(Opcode));
553 }
554
555 static bool isSOPC(const MachineInstr &MI) {
556 return SIInstrFlags::isSOPC(MI);
557 }
558
559 bool isSOPC(uint32_t Opcode) const {
560 return SIInstrFlags::isSOPC(get(Opcode));
561 }
562
563 static bool isSOPK(const MachineInstr &MI) {
564 return SIInstrFlags::isSOPK(MI);
565 }
566
567 bool isSOPK(uint32_t Opcode) const {
568 return SIInstrFlags::isSOPK(get(Opcode));
569 }
570
571 static bool isSOPP(const MachineInstr &MI) {
572 return SIInstrFlags::isSOPP(MI);
573 }
574
575 bool isSOPP(uint32_t Opcode) const {
576 return SIInstrFlags::isSOPP(get(Opcode));
577 }
578
579 static bool isPacked(const MachineInstr &MI) {
581 }
582
583 bool isPacked(uint32_t Opcode) const {
584 return SIInstrFlags::isPacked(get(Opcode));
585 }
586
587 static bool isVOP1(const MachineInstr &MI) {
588 return SIInstrFlags::isVOP1(MI);
589 }
590
591 bool isVOP1(uint32_t Opcode) const {
592 return SIInstrFlags::isVOP1(get(Opcode));
593 }
594
595 static bool isVOP2(const MachineInstr &MI) {
596 return SIInstrFlags::isVOP2(MI);
597 }
598
599 bool isVOP2(uint32_t Opcode) const {
600 return SIInstrFlags::isVOP2(get(Opcode));
601 }
602
603 static bool isVOP3(const MCInstrDesc &Desc) {
605 }
606
607 static bool isVOP3(const MachineInstr &MI) { return isVOP3(MI.getDesc()); }
608
609 bool isVOP3(uint32_t Opcode) const { return isVOP3(get(Opcode)); }
610
611 static bool isSDWA(const MachineInstr &MI) {
612 return SIInstrFlags::isSDWA(MI);
613 }
614
615 bool isSDWA(uint32_t Opcode) const {
616 return SIInstrFlags::isSDWA(get(Opcode));
617 }
618
619 static bool isVOPC(const MachineInstr &MI) {
620 return SIInstrFlags::isVOPC(MI);
621 }
622
623 bool isVOPC(uint32_t Opcode) const {
624 return SIInstrFlags::isVOPC(get(Opcode));
625 }
626
627 static bool isMUBUF(const MachineInstr &MI) {
629 }
630
631 bool isMUBUF(uint32_t Opcode) const {
632 return SIInstrFlags::isMUBUF(get(Opcode));
633 }
634
635 static bool isMTBUF(const MachineInstr &MI) {
637 }
638
639 bool isMTBUF(uint32_t Opcode) const {
640 return SIInstrFlags::isMTBUF(get(Opcode));
641 }
642
643 static bool isBUF(const MachineInstr &MI) {
644 return isMUBUF(MI) || isMTBUF(MI);
645 }
646
647 static bool isSMRD(const MachineInstr &MI) {
648 return SIInstrFlags::isSMRD(MI);
649 }
650
651 bool isSMRD(uint32_t Opcode) const {
652 return SIInstrFlags::isSMRD(get(Opcode));
653 }
654
655 bool isBufferSMRD(const MachineInstr &MI) const;
656
657 static bool isDS(const MachineInstr &MI) { return SIInstrFlags::isDS(MI); }
658
659 bool isDS(uint32_t Opcode) const { return SIInstrFlags::isDS(get(Opcode)); }
660
661 static bool isLDSDMA(const MachineInstr &MI) {
662 return (SIInstrFlags::isVALU(MI) && (isMUBUF(MI) || isFLAT(MI))) ||
664 }
665
666 bool isLDSDMA(uint32_t Opcode) const {
667 return (SIInstrFlags::isVALU(get(Opcode)) &&
668 (isMUBUF(Opcode) || isFLAT(Opcode))) ||
670 }
671
672 static bool isGWS(const MachineInstr &MI) { return SIInstrFlags::isGWS(MI); }
673
674 bool isGWS(uint32_t Opcode) const { return SIInstrFlags::isGWS(get(Opcode)); }
675
676 bool isAlwaysGDS(uint32_t Opcode) const;
677
678 static bool isMIMG(const MachineInstr &MI) {
679 return SIInstrFlags::isMIMG(MI);
680 }
681
682 bool isMIMG(uint32_t Opcode) const {
683 return SIInstrFlags::isMIMG(get(Opcode));
684 }
685
686 static bool isVIMAGE(const MachineInstr &MI) {
688 }
689
690 bool isVIMAGE(uint32_t Opcode) const {
691 return SIInstrFlags::isVIMAGE(get(Opcode));
692 }
693
694 static bool isVSAMPLE(const MachineInstr &MI) {
696 }
697
698 bool isVSAMPLE(uint32_t Opcode) const {
699 return SIInstrFlags::isVSAMPLE(get(Opcode));
700 }
701
702 static bool isGather4(const MachineInstr &MI) {
704 }
705
706 bool isGather4(uint32_t Opcode) const {
707 return SIInstrFlags::isGather4(get(Opcode));
708 }
709
710 static bool isFLAT(const MachineInstr &MI) {
711 return SIInstrFlags::isFLAT(MI);
712 }
713
714 // Is a FLAT encoded instruction which accesses a specific segment,
715 // i.e. global_* or scratch_*.
719
720 bool isSegmentSpecificFLAT(uint32_t Opcode) const {
722 }
723
724 static bool isFLATGlobal(const MachineInstr &MI) {
726 }
727
728 bool isFLATGlobal(uint32_t Opcode) const {
729 return SIInstrFlags::isFlatGlobal(get(Opcode));
730 }
731
732 static bool isFLATScratch(const MachineInstr &MI) {
734 }
735
736 bool isFLATScratch(uint32_t Opcode) const {
737 return SIInstrFlags::isFlatScratch(get(Opcode));
738 }
739
740 // Any FLAT encoded instruction, including global_* and scratch_*.
741 bool isFLAT(uint32_t Opcode) const {
742 return SIInstrFlags::isFLAT(get(Opcode));
743 }
744
745 /// \returns true for SCRATCH_ instructions, or FLAT/BUF instructions unless
746 /// the MMOs do not include scratch.
747 /// Conservatively correct; will return true if \p MI cannot be proven
748 /// to not hit scratch.
749 bool mayAccessScratch(const MachineInstr &MI) const;
750
751 /// \returns true for FLAT instructions that can access VMEM.
752 bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const;
753
754 /// \returns true for FLAT instructions that can access LDS.
755 bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const;
756
757 static bool isBlockLoadStore(uint32_t Opcode) {
758 switch (Opcode) {
759 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
760 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
761 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE:
762 case AMDGPU::SCRATCH_STORE_BLOCK_SADDR:
763 case AMDGPU::SCRATCH_LOAD_BLOCK_SADDR:
764 case AMDGPU::SCRATCH_STORE_BLOCK_SVS:
765 case AMDGPU::SCRATCH_LOAD_BLOCK_SVS:
766 return true;
767 default:
768 return false;
769 }
770 }
771
773 switch (MI.getOpcode()) {
774 case AMDGPU::S_ABSDIFF_I32:
775 case AMDGPU::S_ABS_I32:
776 case AMDGPU::S_AND_B32:
777 case AMDGPU::S_AND_B64:
778 case AMDGPU::S_ANDN2_B32:
779 case AMDGPU::S_ANDN2_B64:
780 case AMDGPU::S_ASHR_I32:
781 case AMDGPU::S_ASHR_I64:
782 case AMDGPU::S_BCNT0_I32_B32:
783 case AMDGPU::S_BCNT0_I32_B64:
784 case AMDGPU::S_BCNT1_I32_B32:
785 case AMDGPU::S_BCNT1_I32_B64:
786 case AMDGPU::S_BFE_I32:
787 case AMDGPU::S_BFE_I64:
788 case AMDGPU::S_BFE_U32:
789 case AMDGPU::S_BFE_U64:
790 case AMDGPU::S_LSHL_B32:
791 case AMDGPU::S_LSHL_B64:
792 case AMDGPU::S_LSHR_B32:
793 case AMDGPU::S_LSHR_B64:
794 case AMDGPU::S_NAND_B32:
795 case AMDGPU::S_NAND_B64:
796 case AMDGPU::S_NOR_B32:
797 case AMDGPU::S_NOR_B64:
798 case AMDGPU::S_NOT_B32:
799 case AMDGPU::S_NOT_B64:
800 case AMDGPU::S_OR_B32:
801 case AMDGPU::S_OR_B64:
802 case AMDGPU::S_ORN2_B32:
803 case AMDGPU::S_ORN2_B64:
804 case AMDGPU::S_QUADMASK_B32:
805 case AMDGPU::S_QUADMASK_B64:
806 case AMDGPU::S_WQM_B32:
807 case AMDGPU::S_WQM_B64:
808 case AMDGPU::S_XNOR_B32:
809 case AMDGPU::S_XNOR_B64:
810 case AMDGPU::S_XOR_B32:
811 case AMDGPU::S_XOR_B64:
812 return true;
813 default:
814 return false;
815 }
816 }
817
818 static bool isEXP(const MachineInstr &MI) { return SIInstrFlags::isEXP(MI); }
819
821 if (!isEXP(MI))
822 return false;
823 unsigned Target = MI.getOperand(0).getImm();
826 }
827
828 bool isEXP(uint32_t Opcode) const { return SIInstrFlags::isEXP(get(Opcode)); }
829
830 static bool isAtomicNoRet(const MachineInstr &MI) {
832 }
833
834 bool isAtomicNoRet(uint32_t Opcode) const {
835 return SIInstrFlags::isAtomicNoRet(get(Opcode));
836 }
837
838 static bool isAtomicRet(const MachineInstr &MI) {
840 }
841
842 bool isAtomicRet(uint32_t Opcode) const {
843 return SIInstrFlags::isAtomicRet(get(Opcode));
844 }
845
846 static bool isAtomic(const MachineInstr &MI) {
848 }
849
850 bool isAtomic(uint32_t Opcode) const {
851 return SIInstrFlags::isAtomic(get(Opcode));
852 }
853
855 unsigned Opc = MI.getOpcode();
856 // Exclude instructions that read FROM LDS (not write to it)
857 return isLDSDMA(MI) && Opc != AMDGPU::BUFFER_STORE_LDS_DWORD &&
858 Opc != AMDGPU::TENSOR_STORE_FROM_LDS_d2 &&
859 Opc != AMDGPU::TENSOR_STORE_FROM_LDS_d4;
860 }
861
862 static bool isSBarrierSCCWrite(unsigned Opcode) {
863 return Opcode == AMDGPU::S_BARRIER_LEAVE ||
864 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM ||
865 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0;
866 }
867
868 static bool isCBranchVCCZRead(const MachineInstr &MI) {
869 unsigned Opc = MI.getOpcode();
870 return (Opc == AMDGPU::S_CBRANCH_VCCNZ || Opc == AMDGPU::S_CBRANCH_VCCZ) &&
871 !MI.getOperand(1).isUndef();
872 }
873
874 static bool isWQM(const MachineInstr &MI) { return SIInstrFlags::isWQM(MI); }
875
876 bool isWQM(uint32_t Opcode) const { return SIInstrFlags::isWQM(get(Opcode)); }
877
878 static bool isDisableWQM(const MachineInstr &MI) {
880 }
881
882 bool isDisableWQM(uint32_t Opcode) const {
883 return SIInstrFlags::isDisableWQM(get(Opcode));
884 }
885
886 // SI_SPILL_S32_TO_VGPR and SI_RESTORE_S32_FROM_VGPR form a special case of
887 // SGPRs spilling to VGPRs which are SGPR spills but from VALU instructions
888 // therefore we need an explicit check for them since just checking if the
889 // Spill bit is set and what instruction type it came from misclassifies
890 // them.
891 static bool isVGPRSpill(const MachineInstr &MI) {
892 return MI.getOpcode() != AMDGPU::SI_SPILL_S32_TO_VGPR &&
893 MI.getOpcode() != AMDGPU::SI_RESTORE_S32_FROM_VGPR &&
894 (isSpill(MI) && isVALU(MI, /*AllowLDSDMA=*/false));
895 }
896
897 bool isVGPRSpill(uint32_t Opcode) const {
898 return Opcode != AMDGPU::SI_SPILL_S32_TO_VGPR &&
899 Opcode != AMDGPU::SI_RESTORE_S32_FROM_VGPR &&
900 (isSpill(Opcode) && isVALU(Opcode, /*AllowLDSDMA=*/false));
901 }
902
903 static bool isSGPRSpill(const MachineInstr &MI) {
904 return MI.getOpcode() == AMDGPU::SI_SPILL_S32_TO_VGPR ||
905 MI.getOpcode() == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
906 (isSpill(MI) && isSALU(MI));
907 }
908
909 bool isSGPRSpill(uint32_t Opcode) const {
910 return Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR ||
911 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
912 (isSpill(Opcode) && isSALU(Opcode));
913 }
914
915 bool isSpill(uint32_t Opcode) const {
916 return SIInstrFlags::isSpill(get(Opcode));
917 }
918
919 static bool isSpill(const MCInstrDesc &Desc) {
921 }
922
923 static bool isSpill(const MachineInstr &MI) { return isSpill(MI.getDesc()); }
924
925 static bool isWWMRegSpillOpcode(uint32_t Opcode) {
926 return Opcode == AMDGPU::SI_SPILL_WWM_V32_SAVE ||
927 Opcode == AMDGPU::SI_SPILL_WWM_AV32_SAVE ||
928 Opcode == AMDGPU::SI_SPILL_WWM_V32_RESTORE ||
929 Opcode == AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
930 }
931
932 static bool isChainCallOpcode(uint64_t Opcode) {
933 return Opcode == AMDGPU::SI_CS_CHAIN_TC_W32 ||
934 Opcode == AMDGPU::SI_CS_CHAIN_TC_W64;
935 }
936
937 static bool isDPP(const MachineInstr &MI) { return SIInstrFlags::isDPP(MI); }
938
939 bool isDPP(uint32_t Opcode) const { return SIInstrFlags::isDPP(get(Opcode)); }
940
941 // Some opcodes use Src1 for DPP instead of Src0, because the sequencer
942 // transforms them and reverse the order of their operands at runtime.
943 //
944 // Documentation is incomplete on which instructions are effected, so
945 // the implementation is derived from experimentation.
946 //
947 // Listed as target-independent pseudos; the per-subtarget MC opcodes
948 // (V_SUBREV_NC_U32_e32_gfx11 and friends) are all reached through these.
949 // Defined out of line because GCNSubtarget is incomplete here.
950 static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode);
951
952 static bool isTRANS(const MachineInstr &MI) {
954 }
955
956 bool isTRANS(uint32_t Opcode) const {
957 return SIInstrFlags::isTRANS(get(Opcode));
958 }
959
960 static bool isVOP3P(const MachineInstr &MI) {
962 }
963
964 bool isVOP3P(uint32_t Opcode) const {
965 return SIInstrFlags::isVOP3P(get(Opcode));
966 }
967
968 bool isVOP3PMix(const MachineInstr &MI) const {
969 return isVOP3PMix(MI.getOpcode());
970 }
971
972 bool isVOP3PMix(uint16_t Opcode) const {
973 switch (Opcode) {
974 case AMDGPU::V_FMA_MIXHI_F16:
975 case AMDGPU::V_FMA_MIXLO_F16:
976 case AMDGPU::V_FMA_MIX_F32:
977 case AMDGPU::V_MAD_MIXHI_F16:
978 case AMDGPU::V_MAD_MIXLO_F16:
979 case AMDGPU::V_MAD_MIX_F32:
980 return true;
981 default:
982 return false;
983 }
984 }
985
986 static bool isVINTRP(const MachineInstr &MI) {
988 }
989
990 bool isVINTRP(uint32_t Opcode) const {
991 return SIInstrFlags::isVINTRP(get(Opcode));
992 }
993
994 static bool isMAI(const MCInstrDesc &Desc) {
996 }
997
998 static bool isMAI(const MachineInstr &MI) { return isMAI(MI.getDesc()); }
999
1000 bool isMAI(uint32_t Opcode) const { return isMAI(get(Opcode)); }
1001
1002 static bool isMFMA(const MachineInstr &MI) {
1003 return isMAI(MI) && MI.getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 &&
1004 MI.getOpcode() != AMDGPU::V_ACCVGPR_READ_B32_e64;
1005 }
1006
1007 bool isMFMA(uint32_t Opcode) const {
1008 return isMAI(Opcode) && Opcode != AMDGPU::V_ACCVGPR_WRITE_B32_e64 &&
1009 Opcode != AMDGPU::V_ACCVGPR_READ_B32_e64;
1010 }
1011
1012 static bool isDOT(const MachineInstr &MI) { return SIInstrFlags::isDOT(MI); }
1013
1014 static bool isWMMA(const MachineInstr &MI) {
1015 return SIInstrFlags::isWMMA(MI);
1016 }
1017
1018 bool isWMMA(uint32_t Opcode) const {
1019 return SIInstrFlags::isWMMA(get(Opcode));
1020 }
1021
1022 static bool isMFMAorWMMA(const MachineInstr &MI) {
1023 return isMFMA(MI) || isWMMA(MI) || isSWMMAC(MI);
1024 }
1025
1026 bool isMFMAorWMMA(uint32_t Opcode) const {
1027 return isMFMA(Opcode) || isWMMA(Opcode) || isSWMMAC(Opcode);
1028 }
1029
1030 static bool isSWMMAC(const MachineInstr &MI) {
1031 return SIInstrFlags::isSWMMAC(MI);
1032 }
1033
1034 bool isSWMMAC(uint32_t Opcode) const {
1035 return SIInstrFlags::isSWMMAC(get(Opcode));
1036 }
1037
1038 bool isDOT(uint32_t Opcode) const { return SIInstrFlags::isDOT(get(Opcode)); }
1039
1040 bool isXDLWMMA(const MachineInstr &MI) const;
1041
1042 bool isXDL(const MachineInstr &MI) const;
1043
1044 static bool isDGEMM(unsigned Opcode) { return AMDGPU::getMAIIsDGEMM(Opcode); }
1045
1046 static bool isLDSDIR(const MachineInstr &MI) {
1047 return SIInstrFlags::isLDSDIR(MI);
1048 }
1049
1050 bool isLDSDIR(uint32_t Opcode) const {
1051 return SIInstrFlags::isLDSDIR(get(Opcode));
1052 }
1053
1054 static bool isVINTERP(const MachineInstr &MI) {
1056 }
1057
1058 bool isVINTERP(uint32_t Opcode) const {
1059 return SIInstrFlags::isVINTERP(get(Opcode));
1060 }
1061
1062 static bool isScalarUnit(const MachineInstr &MI) {
1064 }
1065
1066 static bool usesVM_CNT(const MachineInstr &MI) {
1068 }
1069
1070 static bool usesLGKM_CNT(const MachineInstr &MI) {
1072 }
1073
1074 static bool usesASYNC_CNT(const MachineInstr &MI) {
1076 }
1077
1078 bool usesASYNC_CNT(uint32_t Opcode) const {
1079 return SIInstrFlags::usesASYNC_CNT(get(Opcode));
1080 }
1081
1082 static bool usesTENSOR_CNT(const MachineInstr &MI) {
1084 }
1085
1086 bool usesTENSOR_CNT(uint32_t Opcode) const {
1087 return SIInstrFlags::usesTENSOR_CNT(get(Opcode));
1088 }
1089
1090 // Most sopk treat the immediate as a signed 16-bit, however some
1091 // use it as unsigned.
1092 static bool sopkIsZext(unsigned Opcode) {
1093 return Opcode == AMDGPU::S_CMPK_EQ_U32 || Opcode == AMDGPU::S_CMPK_LG_U32 ||
1094 Opcode == AMDGPU::S_CMPK_GT_U32 || Opcode == AMDGPU::S_CMPK_GE_U32 ||
1095 Opcode == AMDGPU::S_CMPK_LT_U32 || Opcode == AMDGPU::S_CMPK_LE_U32 ||
1096 Opcode == AMDGPU::S_GETREG_B32 ||
1097 Opcode == AMDGPU::S_GETREG_B32_const;
1098 }
1099
1100 /// \returns true if this is an s_store_dword* instruction. This is more
1101 /// specific than isSMEM && mayStore.
1102 static bool isScalarStore(const MachineInstr &MI) {
1104 }
1105
1106 bool isScalarStore(uint32_t Opcode) const {
1107 return SIInstrFlags::isScalarStore(get(Opcode));
1108 }
1109
1110 static bool isFixedSize(const MachineInstr &MI) {
1112 }
1113
1114 bool isFixedSize(uint32_t Opcode) const {
1115 return SIInstrFlags::isFixedSize(get(Opcode));
1116 }
1117
1118 static bool hasFPClamp(const MachineInstr &MI) {
1120 }
1121
1122 bool hasFPClamp(uint32_t Opcode) const {
1123 return SIInstrFlags::hasFPClamp(get(Opcode));
1124 }
1125
1126 static bool hasIntClamp(const MachineInstr &MI) {
1128 }
1129
1130 static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B) {
1131 const MCInstrDesc &DA = A.getDesc(), &DB = B.getDesc();
1136 }
1137
1138 static bool usesFPDPRounding(const MachineInstr &MI) {
1140 }
1141
1142 bool usesFPDPRounding(uint32_t Opcode) const {
1143 return SIInstrFlags::usesFPDPRounding(get(Opcode));
1144 }
1145
1146 static bool isFPAtomic(const MachineInstr &MI) {
1148 }
1149
1150 bool isFPAtomic(uint32_t Opcode) const {
1151 return SIInstrFlags::isFPAtomic(get(Opcode));
1152 }
1153
1154 static bool isNeverUniform(const MachineInstr &MI) {
1156 }
1157
1158 // Check to see if opcode is for a barrier start. Pre gfx12 this is just the
1159 // S_BARRIER, but after support for S_BARRIER_SIGNAL* / S_BARRIER_WAIT we want
1160 // to check for the barrier start (S_BARRIER_SIGNAL*)
1161 bool isBarrierStart(unsigned Opcode) const {
1162 return Opcode == AMDGPU::S_BARRIER ||
1163 Opcode == AMDGPU::S_BARRIER_SIGNAL_M0 ||
1164 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0 ||
1165 Opcode == AMDGPU::S_BARRIER_SIGNAL_IMM ||
1166 Opcode == AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM;
1167 }
1168
1169 bool isBarrier(unsigned Opcode) const {
1170 return isBarrierStart(Opcode) || Opcode == AMDGPU::S_BARRIER_WAIT ||
1171 Opcode == AMDGPU::S_BARRIER_INIT_M0 ||
1172 Opcode == AMDGPU::S_BARRIER_INIT_IMM ||
1173 Opcode == AMDGPU::S_BARRIER_JOIN_IMM ||
1174 Opcode == AMDGPU::S_BARRIER_LEAVE || Opcode == AMDGPU::DS_GWS_INIT ||
1175 Opcode == AMDGPU::DS_GWS_BARRIER;
1176 }
1177
1178 static bool isLoadMonitor(unsigned Opc) {
1179 switch (Opc) {
1180 case AMDGPU::GLOBAL_LOAD_MONITOR_B32:
1181 case AMDGPU::GLOBAL_LOAD_MONITOR_B32_SADDR:
1182 case AMDGPU::GLOBAL_LOAD_MONITOR_B64:
1183 case AMDGPU::GLOBAL_LOAD_MONITOR_B64_SADDR:
1184 case AMDGPU::GLOBAL_LOAD_MONITOR_B128:
1185 case AMDGPU::GLOBAL_LOAD_MONITOR_B128_SADDR:
1186 case AMDGPU::FLAT_LOAD_MONITOR_B32:
1187 case AMDGPU::FLAT_LOAD_MONITOR_B64:
1188 case AMDGPU::FLAT_LOAD_MONITOR_B128:
1189 return true;
1190 default:
1191 return false;
1192 }
1193 }
1194
1195 static bool isGFX12CacheInvOrWBInst(unsigned Opc) {
1196 return Opc == AMDGPU::GLOBAL_INV || Opc == AMDGPU::GLOBAL_WB ||
1197 Opc == AMDGPU::GLOBAL_WBINV;
1198 }
1199
1200 static bool isF16PseudoScalarTrans(unsigned Opcode) {
1201 return Opcode == AMDGPU::V_S_EXP_F16_e64 ||
1202 Opcode == AMDGPU::V_S_LOG_F16_e64 ||
1203 Opcode == AMDGPU::V_S_RCP_F16_e64 ||
1204 Opcode == AMDGPU::V_S_RSQ_F16_e64 ||
1205 Opcode == AMDGPU::V_S_SQRT_F16_e64;
1206 }
1207
1208 static bool isPseudoScalarTrans(unsigned Opcode) {
1209 return isF16PseudoScalarTrans(Opcode) ||
1210 Opcode == AMDGPU::V_S_EXP_F32_e64 ||
1211 Opcode == AMDGPU::V_S_LOG_F32_e64 ||
1212 Opcode == AMDGPU::V_S_RCP_F32_e64 ||
1213 Opcode == AMDGPU::V_S_RSQ_F32_e64 ||
1214 Opcode == AMDGPU::V_S_SQRT_F32_e64;
1215 }
1216
1217 static bool isVPermPk16(unsigned Opcode) {
1218 return Opcode == AMDGPU::V_PERM_PK16_B4_U4_e64 ||
1219 Opcode == AMDGPU::V_PERM_PK16_B6_U4_e64 ||
1220 Opcode == AMDGPU::V_PERM_PK16_B8_U4_e64;
1221 }
1222
1223 // \returns true if \p MI clears the V_PERM_PK16 hazard when it immediately
1224 // follows a V_PERM_PK16 (i.e. \p MI is a "safe" instruction).
1226 unsigned Opc = MI.getOpcode();
1227
1228 // Only VALU ops issue on the pipe that clears the V_PERM_PK16 hazard.
1229 if (!isVALU(MI, /*AllowLDSDMA=*/false))
1230 return false;
1231 // OP_XDL: matrix (WMMA/SWMMAC/DOT) ops clear the hazard.
1232 if (isXDL(MI))
1233 return true;
1234 // Pseudo-scalar transcendentals (OP32_SCL_T) do NOT clear the hazard.
1236 return false;
1237
1238 // Use the table lookup, not getBlockingCycles(): occupancy is gated off on
1239 // gfx1251 but the hazard applies to both gfx1250 and gfx1251. Gfx1250 table
1240 // is valid enough for gfx1251 w.r.t. v_perm_pk16 safety check here.
1241 // OP_32_T is in the table at 2 and is safe; other table entries (>= 2) are
1242 // not.
1243 unsigned Cycles = getGFX1250BlockingCyclesTable(MI);
1244 return Cycles < 2 || (Cycles == 2 && isTRANS(MI));
1245 }
1246
1250
1251 bool doesNotReadTiedSource(uint32_t Opcode) const {
1252 return SIInstrFlags::isTiedSourceNotRead(get(Opcode));
1253 }
1254
1255 bool isIGLP(unsigned Opcode) const {
1256 return Opcode == AMDGPU::SCHED_BARRIER ||
1257 Opcode == AMDGPU::SCHED_GROUP_BARRIER || Opcode == AMDGPU::IGLP_OPT;
1258 }
1259
1260 bool isIGLP(const MachineInstr &MI) const { return isIGLP(MI.getOpcode()); }
1261
1262 // Return true if the instruction is mutually exclusive with all non-IGLP DAG
1263 // mutations, requiring all other mutations to be disabled.
1264 bool isIGLPMutationOnly(unsigned Opcode) const {
1265 return Opcode == AMDGPU::SCHED_GROUP_BARRIER || Opcode == AMDGPU::IGLP_OPT;
1266 }
1267
1268 static unsigned getNonSoftWaitcntOpcode(unsigned Opcode) {
1269 switch (Opcode) {
1270 case AMDGPU::S_WAITCNT_soft:
1271 return AMDGPU::S_WAITCNT;
1272 case AMDGPU::S_WAITCNT_VSCNT_soft:
1273 return AMDGPU::S_WAITCNT_VSCNT;
1274 case AMDGPU::S_WAIT_LOADCNT_soft:
1275 return AMDGPU::S_WAIT_LOADCNT;
1276 case AMDGPU::S_WAIT_STORECNT_soft:
1277 return AMDGPU::S_WAIT_STORECNT;
1278 case AMDGPU::S_WAIT_SAMPLECNT_soft:
1279 return AMDGPU::S_WAIT_SAMPLECNT;
1280 case AMDGPU::S_WAIT_BVHCNT_soft:
1281 return AMDGPU::S_WAIT_BVHCNT;
1282 case AMDGPU::S_WAIT_DSCNT_soft:
1283 return AMDGPU::S_WAIT_DSCNT;
1284 case AMDGPU::S_WAIT_KMCNT_soft:
1285 return AMDGPU::S_WAIT_KMCNT;
1286 case AMDGPU::S_WAIT_XCNT_soft:
1287 return AMDGPU::S_WAIT_XCNT;
1288 default:
1289 return Opcode;
1290 }
1291 }
1292
1293 static bool isWaitcnt(unsigned Opcode) {
1294 switch (getNonSoftWaitcntOpcode(Opcode)) {
1295 case AMDGPU::S_WAITCNT:
1296 case AMDGPU::S_WAITCNT_VSCNT:
1297 case AMDGPU::S_WAITCNT_VMCNT:
1298 case AMDGPU::S_WAITCNT_EXPCNT:
1299 case AMDGPU::S_WAITCNT_LGKMCNT:
1300 case AMDGPU::S_WAIT_LOADCNT:
1301 case AMDGPU::S_WAIT_LOADCNT_DSCNT:
1302 case AMDGPU::S_WAIT_STORECNT:
1303 case AMDGPU::S_WAIT_STORECNT_DSCNT:
1304 case AMDGPU::S_WAIT_SAMPLECNT:
1305 case AMDGPU::S_WAIT_BVHCNT:
1306 case AMDGPU::S_WAIT_EXPCNT:
1307 case AMDGPU::S_WAIT_DSCNT:
1308 case AMDGPU::S_WAIT_KMCNT:
1309 case AMDGPU::S_WAIT_XCNT:
1310 case AMDGPU::S_WAIT_IDLE:
1311 return true;
1312 default:
1313 return false;
1314 }
1315 }
1316
1317 bool isVGPRCopy(const MachineInstr &MI) const {
1318 assert(isCopyInstr(MI));
1319 Register Dest = MI.getOperand(0).getReg();
1320 const MachineFunction &MF = *MI.getMF();
1321 const MachineRegisterInfo &MRI = MF.getRegInfo();
1322 return !RI.isSGPRReg(MRI, Dest);
1323 }
1324
1325 bool hasVGPRUses(const MachineInstr &MI) const {
1326 const MachineFunction &MF = *MI.getMF();
1327 const MachineRegisterInfo &MRI = MF.getRegInfo();
1328 return llvm::any_of(MI.explicit_uses(),
1329 [&MRI, this](const MachineOperand &MO) {
1330 return MO.isReg() && RI.isVGPR(MRI, MO.getReg());});
1331 }
1332
1333 /// Return true if the instruction modifies the mode register.q
1334 static bool modifiesModeRegister(const MachineInstr &MI);
1335
1336 /// This function is used to determine if an instruction can be safely
1337 /// executed under EXEC = 0 without hardware error, indeterminate results,
1338 /// and/or visible effects on future vector execution or outside the shader.
1339 /// Note: as of 2024 the only use of this is SIPreEmitPeephole where it is
1340 /// used in removing branches over short EXEC = 0 sequences.
1341 /// As such it embeds certain assumptions which may not apply to every case
1342 /// of EXEC = 0 execution.
1344
1345 /// Returns true if the instruction could potentially depend on the value of
1346 /// exec. If false, exec dependencies may safely be ignored.
1347 bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const;
1348
1349 bool isInlineConstant(const APInt &Imm) const;
1350
1351 bool isInlineConstant(const APFloat &Imm) const;
1352
1353 // Returns true if this non-register operand definitely does not need to be
1354 // encoded as a 32-bit literal. Note that this function handles all kinds of
1355 // operands, not just immediates.
1356 //
1357 // Some operands like FrameIndexes could resolve to an inline immediate value
1358 // that will not require an additional 4-bytes; this function assumes that it
1359 // will.
1360 bool isInlineConstant(const MachineOperand &MO, uint8_t OperandType) const {
1361 if (!MO.isImm())
1362 return false;
1363 return isInlineConstant(MO.getImm(), OperandType);
1364 }
1365 bool isInlineConstant(int64_t ImmVal, uint8_t OperandType) const;
1366
1368 const MCOperandInfo &OpInfo) const {
1369 return isInlineConstant(MO, OpInfo.OperandType);
1370 }
1371
1372 /// \p returns true if \p UseMO is substituted with \p DefMO in \p MI it would
1373 /// be an inline immediate.
1375 const MachineOperand &UseMO,
1376 const MachineOperand &DefMO) const {
1377 assert(UseMO.getParent() == &MI);
1378 int OpIdx = UseMO.getOperandNo();
1379 if (OpIdx >= MI.getDesc().NumOperands)
1380 return false;
1381
1382 return isInlineConstant(DefMO, MI.getDesc().operands()[OpIdx]);
1383 }
1384
1385 /// \p returns true if the operand \p OpIdx in \p MI is a valid inline
1386 /// immediate.
1387 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx) const {
1388 const MachineOperand &MO = MI.getOperand(OpIdx);
1389 return isInlineConstant(MO, MI.getDesc().operands()[OpIdx].OperandType);
1390 }
1391
1392 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx,
1393 int64_t ImmVal) const {
1394 if (OpIdx >= MI.getDesc().NumOperands)
1395 return false;
1396
1397 if (isCopyInstr(MI)) {
1398 unsigned Size = getOpSize(MI, OpIdx);
1399 assert(Size == 8 || Size == 4);
1400
1401 uint8_t OpType = (Size == 8) ?
1403 return isInlineConstant(ImmVal, OpType);
1404 }
1405
1406 return isInlineConstant(ImmVal, MI.getDesc().operands()[OpIdx].OperandType);
1407 }
1408
1409 bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx,
1410 const MachineOperand &MO) const {
1411 return isInlineConstant(MI, OpIdx, MO.getImm());
1412 }
1413
1414 bool isInlineConstant(const MachineOperand &MO) const {
1415 return isInlineConstant(*MO.getParent(), MO.getOperandNo());
1416 }
1417
1418 bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
1419 const MachineOperand &MO) const;
1420
1421 bool isLiteralOperandLegal(const MCInstrDesc &InstDesc,
1422 const MCOperandInfo &OpInfo) const;
1423
1424 bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
1425 int64_t ImmVal) const;
1426
1427 bool isImmOperandLegal(const MachineInstr &MI, unsigned OpNo,
1428 const MachineOperand &MO) const {
1429 return isImmOperandLegal(MI.getDesc(), OpNo, MO);
1430 }
1431
1432 bool isNeverCoissue(MachineInstr &MI) const;
1433
1434 /// Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
1435 bool isLegalAV64PseudoImm(uint64_t Imm) const;
1436
1437 /// Return true if this 64-bit VALU instruction has a 32-bit encoding.
1438 /// This function will return false if you pass it a 32-bit instruction.
1439 bool hasVALU32BitEncoding(unsigned Opcode) const;
1440
1441 /// Return true if \p Reg is a lane mask that already has 0 in every bit
1442 /// corresponding to a lane that is inactive in EXEC where \p Use executes,
1443 /// so that ANDing it with EXEC there would be a no-op. Requires SSA form.
1445 const MachineRegisterInfo &MRI, unsigned Depth = 0) const;
1446
1447 bool physRegUsesConstantBus(const MachineOperand &Reg) const;
1449 const MachineRegisterInfo &MRI) const;
1450
1451 /// Returns true if this operand uses the constant bus.
1452 bool usesConstantBus(const MachineRegisterInfo &MRI,
1453 const MachineOperand &MO,
1454 const MCOperandInfo &OpInfo) const;
1455
1457 int OpIdx) const {
1458 return usesConstantBus(MRI, MI.getOperand(OpIdx),
1459 MI.getDesc().operands()[OpIdx]);
1460 }
1461
1462 /// Return true if this instruction has any modifiers.
1463 /// e.g. src[012]_mod, omod, clamp.
1464 bool hasModifiers(unsigned Opcode) const;
1465
1466 bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const;
1467 bool hasAnyModifiersSet(const MachineInstr &MI) const;
1468
1469 bool canShrink(const MachineInstr &MI,
1470 const MachineRegisterInfo &MRI) const;
1471
1473 unsigned NewOpcode) const;
1474
1475 bool verifyInstruction(const MachineInstr &MI,
1476 StringRef &ErrInfo) const override;
1477
1478 unsigned getVALUOp(const MachineInstr &MI) const;
1479 unsigned getVALUOp(unsigned Opc) const;
1480
1483 const DebugLoc &DL, Register Reg, bool IsSCCLive,
1484 SlotIndexes *Indexes = nullptr) const;
1485
1488 Register Reg, SlotIndexes *Indexes = nullptr) const;
1489
1491
1492 /// Return the correct register class for \p OpNo. For target-specific
1493 /// instructions, this will return the register class that has been defined
1494 /// in tablegen. For generic instructions, like REG_SEQUENCE it will return
1495 /// the register class of its machine operand.
1496 /// to infer the correct register class base on the other operands.
1498 unsigned OpNo) const;
1499
1500 /// Return the size in bytes of the operand OpNo on the given
1501 // instruction opcode.
1502 unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const {
1503 const MCOperandInfo &OpInfo = get(Opcode).operands()[OpNo];
1504
1505 if (OpInfo.RegClass == -1) {
1506 // If this is an immediate operand, this must be a 32-bit literal.
1507 assert(OpInfo.OperandType == MCOI::OPERAND_IMMEDIATE);
1508 return 4;
1509 }
1510
1511 return RI.getRegSizeInBits(*RI.getRegClass(getOpRegClassID(OpInfo))) / 8;
1512 }
1513
1514 /// This form should usually be preferred since it handles operands
1515 /// with unknown register classes.
1516 unsigned getOpSize(const MachineInstr &MI, unsigned OpNo) const {
1517 const MachineOperand &MO = MI.getOperand(OpNo);
1518 if (MO.isReg()) {
1519 if (unsigned SubReg = MO.getSubReg()) {
1520 return RI.getSubRegIdxSize(SubReg) / 8;
1521 }
1522 }
1523 return RI.getRegSizeInBits(*getOpRegClass(MI, OpNo)) / 8;
1524 }
1525
1526 /// Legalize the \p OpIndex operand of this instruction by inserting
1527 /// a MOV. For example:
1528 /// ADD_I32_e32 VGPR0, 15
1529 /// to
1530 /// MOV VGPR1, 15
1531 /// ADD_I32_e32 VGPR0, VGPR1
1532 ///
1533 /// If the operand being legalized is a register, then a COPY will be used
1534 /// instead of MOV.
1535 void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const;
1536
1537 /// Check if \p MO is a legal operand if it was the \p OpIdx Operand
1538 /// for \p MI.
1539 bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx,
1540 const MachineOperand *MO = nullptr) const;
1541
1542 /// Check if \p MO would be a valid operand for the given operand
1543 /// definition \p OpInfo. Note this does not attempt to validate constant bus
1544 /// restrictions (e.g. literal constant usage).
1546 const MCOperandInfo &OpInfo,
1547 const MachineOperand &MO) const;
1548
1549 /// Check if \p MO (a register operand) is a legal register for the
1550 /// given operand description or operand index.
1551 /// The operand index version provide more legality checks
1552 bool isLegalRegOperand(const MachineRegisterInfo &MRI,
1553 const MCOperandInfo &OpInfo,
1554 const MachineOperand &MO) const;
1555 bool isLegalRegOperand(const MachineInstr &MI, unsigned OpIdx,
1556 const MachineOperand &MO) const;
1557
1558 /// Check if \p MO would be a legal operand for a single-SGPR-read
1559 /// instruction.
1560 ///
1561 /// Single-SGPR-read instructions typically accept VGPRs, SGPRs, or immediates
1562 /// as source operands. On gfx12+, if a source operand uses SGPRs, the HW can
1563 /// only read the first SGPR and replicate the value across all lanes. \p SrcN
1564 /// can be 0, 1, or 2, representing src0, src1, and src2, respectively. If \p
1565 /// MO is nullptr, the operand corresponding to \p SrcN will be used. Non-SGPR
1566 /// operands are always considered legal.
1567 bool
1569 const MachineInstr &MI, unsigned SrcN,
1570 const MachineOperand *MO = nullptr) const;
1571
1572 /// Legalize operands in \p MI by either commuting it or inserting a
1573 /// copy of src1.
1575
1576 /// Fix operands in \p MI to satisfy constant bus requirements.
1578
1579 /// Copy a value from a VGPR (\p SrcReg) to SGPR. The desired register class
1580 /// for the dst register (\p DstRC) can be optionally supplied. This function
1581 /// can only be used when it is know that the value in SrcReg is same across
1582 /// all threads in the wave.
1583 /// \returns The SGPR register that \p SrcReg was copied to.
1586 const TargetRegisterClass *DstRC = nullptr) const;
1587
1590
1593 const TargetRegisterClass *DstRC,
1595 const DebugLoc &DL) const;
1596
1597 /// Legalize all operands in this instruction. This function may create new
1598 /// instructions and control-flow around \p MI. If present, \p MDT is
1599 /// updated.
1600 /// \returns A new basic block that contains \p MI if new blocks were created.
1602 legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT = nullptr) const;
1603
1604 /// Change SADDR form of a FLAT \p Inst to its VADDR form if saddr operand
1605 /// was moved to VGPR. \returns true if succeeded.
1606 bool moveFlatAddrToVGPR(MachineInstr &Inst) const;
1607
1608 /// Fix operands in Inst to fix 16bit SALU to VALU lowering.
1610 MachineRegisterInfo &MRI) const;
1611 void legalizeOperandsVALUt16(MachineInstr &Inst, unsigned OpIdx,
1612 MachineRegisterInfo &MRI) const;
1613
1614 /// Replace the instructions opcode with the equivalent VALU
1615 /// opcode. This function will also move the users of MachineInstruntions
1616 /// in the \p WorkList to the VALU if necessary. If present, \p MDT is
1617 /// updated.
1618 void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const;
1619
1620 void
1622 MachineInstr &Inst,
1624 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
1625 /// Wrapper function for generating waterfall for instruction \p MI
1626 /// This function take into consideration of related pre & succ instructions
1627 /// (e.g. calling process) into consideratioin
1630 ArrayRef<Register> PhySGPRs = {}) const;
1631
1632 void insertNoop(MachineBasicBlock &MBB,
1633 MachineBasicBlock::iterator MI) const override;
1634
1635 void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI,
1636 unsigned Quantity) const override;
1637
1638 /// Build instructions that simulate the behavior of a `s_trap 2` instructions
1639 /// for hardware (namely, gfx11) that runs in PRIV=1 mode. There, s_trap is
1640 /// interpreted as a nop.
1641 MachineBasicBlock *insertSimulatedTrap(MachineRegisterInfo &MRI,
1642 MachineBasicBlock &MBB,
1643 MachineInstr &MI,
1644 const DebugLoc &DL) const;
1645
1646 /// Return the number of wait states that result from executing this
1647 /// instruction.
1648 static unsigned getNumWaitStates(const MachineInstr &MI);
1649
1650 /// Returns the operand named \p Op. If \p MI does not have an
1651 /// operand named \c Op, this function returns nullptr.
1653 MachineOperand *getNamedOperand(MachineInstr &MI,
1654 AMDGPU::OpName OperandName) const;
1655
1658 AMDGPU::OpName OperandName) const {
1659 return getNamedOperand(const_cast<MachineInstr &>(MI), OperandName);
1660 }
1661
1662 /// Get required immediate operand
1664 AMDGPU::OpName OperandName) const {
1665 int Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OperandName);
1666 return MI.getOperand(Idx).getImm();
1667 }
1668
1671
1672 bool isLowLatencyInstruction(const MachineInstr &MI) const;
1673 bool isHighLatencyDef(int Opc) const override;
1674
1675 /// Return the descriptor of the target-specific machine instruction
1676 /// that corresponds to the specified pseudo or native opcode.
1677 const MCInstrDesc &getMCOpcodeFromPseudo(unsigned Opcode) const {
1678 return get(pseudoToMCOpcode(Opcode));
1679 }
1680
1681 Register isStackAccess(const MachineInstr &MI, int &FrameIndex,
1682 TypeSize &MemBytes) const;
1683 Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex,
1684 TypeSize &MemBytes) const;
1685
1687 int &FrameIndex) const override {
1688 TypeSize MemBytes = TypeSize::getZero();
1689 return isLoadFromStackSlot(MI, FrameIndex, MemBytes);
1690 }
1691
1692 Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex,
1693 TypeSize &MemBytes) const override;
1694
1696 int &FrameIndex) const override {
1697 TypeSize MemBytes = TypeSize::getZero();
1698 return isStoreToStackSlot(MI, FrameIndex, MemBytes);
1699 }
1700
1701 Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex,
1702 TypeSize &MemBytes) const override;
1703
1704 unsigned getInstSizeInBytes(const MachineInstr &MI) const override;
1705
1706 InstSizeVerifyMode
1707 getInstSizeVerifyMode(const MachineInstr &MI) const override;
1708
1709 bool mayAccessFlatAddressSpace(const MachineInstr &MI) const;
1710
1711 std::pair<unsigned, unsigned>
1712 decomposeMachineOperandsTargetFlags(unsigned TF) const override;
1713
1715 getSerializableTargetIndices() const override;
1716
1719
1722
1725 const ScheduleDAG *DAG) const override;
1726
1729 MachineLoopInfo *MLI) const override;
1730
1733 const ScheduleDAGMI *DAG) const override;
1734
1736 const MachineFunction &MF) const override;
1737
1739 Register Reg = Register()) const override;
1740
1741 bool canAddToBBProlog(const MachineInstr &MI) const;
1742
1745 const DebugLoc &DL, Register Src,
1746 Register Dst) const override;
1747
1750 const DebugLoc &DL, Register Src,
1751 unsigned SrcSubReg,
1752 Register Dst) const override;
1753
1754 bool isWave32() const;
1755
1756 bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const;
1757
1758 bool hasRAWDependency(const MachineInstr &FirstMI,
1759 const MachineInstr &SecondMI) const;
1760
1761 /// Return a partially built integer add instruction without carry.
1762 /// Caller must add source operands.
1763 /// For pre-GFX9 it will generate unused carry destination operand.
1764 /// TODO: After GFX9 it should return a no-carry operation.
1767 const DebugLoc &DL,
1768 Register DestReg) const;
1769
1772 const DebugLoc &DL,
1773 Register DestReg,
1774 RegScavenger &RS) const;
1775
1776 static bool isKillTerminator(unsigned Opcode);
1777 const MCInstrDesc &getKillTerminatorFromPseudo(unsigned Opcode) const;
1778
1779 bool isLegalMUBUFImmOffset(unsigned Imm) const;
1780
1781 static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST);
1782
1783 bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset,
1784 Align Alignment = Align(4)) const;
1785
1786 /// Returns if \p Offset is legal for the subtarget as the offset to a FLAT
1787 /// encoded instruction with the given \p FlatVariant.
1788 bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace,
1789 AMDGPU::FlatAddrSpace FlatVariant) const;
1790
1791 /// Split \p COffsetVal into {immediate offset field, remainder offset}
1792 /// values.
1793 std::pair<int64_t, int64_t>
1794 splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace,
1795 AMDGPU::FlatAddrSpace FlatVariant) const;
1796
1797 /// Returns true if negative offsets are allowed for the given \p FlatVariant.
1798 bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const;
1799
1800 /// \brief Return a target-specific opcode if Opcode is a pseudo instruction.
1801 /// Return -1 if the target-specific opcode for the pseudo instruction does
1802 /// not exist. If Opcode is not a pseudo instruction, this is identity.
1803 int pseudoToMCOpcode(int Opcode) const;
1804
1805 /// \brief Check if this instruction should only be used by assembler.
1806 /// Return true if this opcode should not be used by codegen.
1807 bool isAsmOnlyOpcode(int MCOp) const;
1808
1809 void fixImplicitOperands(MachineInstr &MI) const;
1810
1812 ArrayRef<unsigned> Ops, int FrameIndex,
1813 MachineInstr *&CopyMI,
1814 LiveIntervals *LIS = nullptr,
1815 VirtRegMap *VRM = nullptr) const override;
1816
1817 unsigned getInstrLatency(const InstrItineraryData *ItinData,
1818 const MachineInstr &MI,
1819 unsigned *PredCost = nullptr) const override;
1820
1821 unsigned getBlockingCycles(const MachineInstr &MI) const;
1822
1823 /// GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
1824 /// Returns 0 if \p MI is not in the table. Used as a multi-pass VALU denylist
1825 /// (e.g. V_PERM_PK16 hazard) on both gfx1250 and gfx1251.
1826 unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const;
1827
1828 const MachineOperand &getCalleeOperand(const MachineInstr &MI) const override;
1829
1831
1833
1834 const MIRFormatter *getMIRFormatter() const override;
1835
1836 static unsigned getDSShaderTypeValue(const MachineFunction &MF);
1837
1838 const TargetSchedModel &getSchedModel() const { return SchedModel; }
1839
1841 Register DstReg,
1842 MachineInstr &Inst) const;
1843
1845 SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
1848 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const;
1849
1850 // FIXME: This should be removed
1851 // Enforce operand's \p OpName even alignment if required by target.
1852 // This is used if an operand is a 32 bit register but needs to be aligned
1853 // regardless.
1854 void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const;
1855
1856 /// Get the repeat rate for a VALU instruction from the scheduling model.
1857 /// Returns 1 for regular VALU, >1 for long-latency VALU (packed, F64, etc.)
1858 unsigned getRepeatRate(const MachineInstr &MI) const;
1859};
1860
1861/// \brief Returns true if a reg:subreg pair P has a TRC class
1863 const TargetRegisterClass &TRC,
1864 MachineRegisterInfo &MRI) {
1865 auto *RC = MRI.getRegClass(P.Reg);
1866 if (!P.SubReg)
1867 return RC == &TRC;
1868 auto *TRI = MRI.getTargetRegisterInfo();
1869 return RC == TRI->getMatchingSuperRegClass(RC, &TRC, P.SubReg);
1870}
1871
1872/// \brief Create RegSubRegPair from a register MachineOperand
1873inline
1875 assert(O.isReg());
1876 return TargetInstrInfo::RegSubRegPair(O.getReg(), O.getSubReg());
1877}
1878
1879/// \brief Return the SubReg component from REG_SEQUENCE
1880TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI,
1881 unsigned SubReg);
1882
1883/// \brief Return the defining instruction for a given reg:subreg pair
1884/// skipping copy like instructions and subreg-manipulation pseudos.
1885/// Following another subreg of a reg:subreg isn't supported.
1886MachineInstr *getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P,
1887 const MachineRegisterInfo &MRI);
1888
1889/// \brief Return false if EXEC is not changed between the def of \p VReg at \p
1890/// DefMI and the use at \p UseMI. Should be run on SSA. Currently does not
1891/// attempt to track between blocks.
1892bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI,
1893 Register VReg,
1894 const MachineInstr &DefMI,
1895 const MachineInstr &UseMI);
1896
1897/// \brief Return false if EXEC is not changed between the def of \p VReg at \p
1898/// DefMI and all its uses. Should be run on SSA. Currently does not attempt to
1899/// track between blocks.
1900bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI,
1901 Register VReg,
1902 const MachineInstr &DefMI);
1903
1904namespace AMDGPU {
1905
1907 int32_t getVOPe64(uint32_t Opcode);
1908
1910 int32_t getVOPe32(uint32_t Opcode);
1911
1913 int32_t getSDWAOp(uint32_t Opcode);
1914
1916 int32_t getDPPOp32(uint32_t Opcode);
1917
1919 int32_t getDPPOp64(uint32_t Opcode);
1920
1923
1925 int32_t getCommuteRev(uint32_t Opcode);
1926
1928 int32_t getCommuteOrig(uint32_t Opcode);
1929
1931 int32_t getAddr64Inst(uint32_t Opcode);
1932
1933 /// Check if \p Opcode is an Addr64 opcode.
1934 ///
1935 /// \returns \p Opcode if it is an Addr64 opcode, otherwise -1.
1937 int32_t getIfAddr64Inst(uint32_t Opcode);
1938
1940 int32_t getSOPKOp(uint32_t Opcode);
1941
1942 /// \returns SADDR form of a FLAT Global instruction given an \p Opcode
1943 /// of a VADDR form.
1946
1947 /// \returns VADDR form of a FLAT Global instruction given an \p Opcode
1948 /// of a SADDR form.
1951
1952 /// \returns ST form with only immediate offset of a FLAT Scratch instruction
1953 /// given an \p Opcode of an SS (SADDR) form.
1956
1957 /// \returns SV (VADDR) form of a FLAT Scratch instruction given an \p Opcode
1958 /// of an SVS (SADDR + VADDR) form.
1961
1962 /// \returns SS (SADDR) form of a FLAT Scratch instruction given an \p Opcode
1963 /// of an SV (VADDR) form.
1966
1967 /// \returns SV (VADDR) form of a FLAT Scratch instruction given an \p Opcode
1968 /// of an SS (SADDR) form.
1971
1972 /// \returns earlyclobber version of a MAC MFMA is exists.
1975
1976 /// \returns Version of an instruction which uses AGPRs for coupled operands
1977 /// given an \p Opcode which uses VGPRs for coupled operands.
1979 int32_t getAGPRFormOp(uint32_t Opcode);
1980
1981 /// \returns v_cmpx version of a v_cmp instruction.
1984
1985 const uint64_t RSRC_DATA_FORMAT = 0xf00000000000LL;
1988 const uint64_t RSRC_TID_ENABLE = UINT64_C(1) << (32 + 23);
1989
1990} // end namespace AMDGPU
1991
1992namespace AMDGPU {
1994 // For sgpr to vgpr spill instructions
1996};
1997} // namespace AMDGPU
1998
1999namespace SI {
2001
2002/// Offsets in bytes from the start of the input buffer
2014
2015} // end namespace KernelInputOffsets
2016} // end namespace SI
2017
2018} // end namespace llvm
2019
2020#endif // LLVM_LIB_TARGET_AMDGPU_SIINSTRINFO_H
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Provides AMDGPU specific target descriptions.
AMDGPU specific overrides of MIRFormatter.
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READONLY
Definition Compiler.h:338
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t IntrinsicInst * II
#define P(N)
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
Interface definition for SIRegisterInfo.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static unsigned getBranchOpcode(ISD::CondCode Cond)
Class for arbitrary precision integers.
Definition APInt.h:78
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
A debug info location.
Definition DebugLoc.h:126
Itinerary data supplied by a subtarget to be used by a target.
Describe properties that are true of each instruction in the target description file.
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
MIRFormater - Interface to format MIR operand based on target.
MachineInstrBundleIterator< MachineInstr > iterator
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Representation of each machine instruction.
Flags
Flags values. These may be or'd together.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
const TargetRegisterInfo * getTargetRegisterInfo() const
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Represents one node in the SelectionDAG.
bool usesFPDPRounding(uint32_t Opcode) const
static bool isCBranchVCCZRead(const MachineInstr &MI)
bool isLegalMUBUFImmOffset(unsigned Imm) const
bool isInlineConstant(const APInt &Imm) const
static bool isMAI(const MachineInstr &MI)
void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const
Fix operands in MI to satisfy constant bus requirements.
bool canAddToBBProlog(const MachineInstr &MI) const
static bool isDS(const MachineInstr &MI)
static bool isVMEM(const MachineInstr &MI)
MachineBasicBlock * legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT=nullptr) const
Legalize all operands in this instruction.
bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0, int64_t &Offset1) const override
bool isWQM(uint32_t Opcode) const
static bool isVOP3(const MachineInstr &MI)
bool isMTBUF(uint32_t Opcode) const
unsigned getLiveRangeSplitOpcode(Register Reg, const MachineFunction &MF) const override
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
static bool isNeverUniform(const MachineInstr &MI)
bool isXDLWMMA(const MachineInstr &MI) const
bool isBasicBlockPrologue(const MachineInstr &MI, Register Reg=Register()) const override
bool isSpill(uint32_t Opcode) const
bool isMUBUF(uint32_t Opcode) const
uint64_t getDefaultRsrcDataFormat() const
static bool isSOPP(const MachineInstr &MI)
bool hasVGPRUses(const MachineInstr &MI) const
bool mayAccessScratch(const MachineInstr &MI) const
bool isIGLP(unsigned Opcode) const
static bool isFLATScratch(const MachineInstr &MI)
bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Returns if Offset is legal for the subtarget as the offset to a FLAT encoded instruction with the giv...
const MCInstrDesc & getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize, bool IsSGPR) const
static bool isSpill(const MachineInstr &MI)
MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg) const
Return a partially built integer add instruction without carry.
bool isVOP3PMix(uint16_t Opcode) const
bool mayAccessFlatAddressSpace(const MachineInstr &MI) const
bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0, int64_t Offset1, unsigned NumLoads) const override
bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset, Align Alignment=Align(4)) const
bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const
Replace the instructions opcode with the equivalent VALU opcode.
bool isDisableWQM(uint32_t Opcode) const
static bool isSMRD(const MachineInstr &MI)
unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const
GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, SlotIndexes *Indexes=nullptr) const
void storeRegToStackSlotCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC) const
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineOperand &MO, const MCOperandInfo &OpInfo) const
Returns true if this operand uses the constant bus.
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
bool doesNotReadTiedSource(uint32_t Opcode) const
bool isSOPK(uint32_t Opcode) const
static unsigned getFoldableCopySrcIdx(const MachineInstr &MI)
unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const
Return the size in bytes of the operand OpNo on the given.
bool isLDSDMA(uint32_t Opcode) const
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const final
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineInstr &MI, int OpIdx) const
const TargetRegisterClass * getInlineAsmMemoryOperandRegClass(InlineAsm::ConstraintCode C) const override
bool isAtomicRet(uint32_t Opcode) const
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool isVINTRP(uint32_t Opcode) const
bool isSOP1(uint32_t Opcode) const
static bool isMTBUF(const MachineInstr &MI)
const MCInstrDesc & getIndirectGPRIDXPseudo(unsigned VecSize, bool IsIndirectSrc) const
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool setsSCCIfResultIsNonZero(const MachineInstr &MI)
const MIRFormatter * getMIRFormatter() const override
bool isSGPRSpill(uint32_t Opcode) const
static bool isXcntDrain(const MachineInstr &MI)
True if MI implicitly drains XCNT.
void legalizeGenericOperand(MachineBasicBlock &InsertMBB, MachineBasicBlock::iterator I, const TargetRegisterClass *DstRC, MachineOperand &Op, MachineRegisterInfo &MRI, const DebugLoc &DL) const
MachineInstr * buildShrunkInst(MachineInstr &MI, unsigned NewOpcode) const
static bool isVOP2(const MachineInstr &MI)
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
static bool isSDWA(const MachineInstr &MI)
const MCInstrDesc & getKillTerminatorFromPseudo(unsigned Opcode) const
static bool mayWriteLDSThroughDMA(const MachineInstr &MI)
void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, unsigned Quantity) const override
bool isDOT(uint32_t Opcode) const
static bool isVINTRP(const MachineInstr &MI)
bool isIGLPMutationOnly(unsigned Opcode) const
static bool isGather4(const MachineInstr &MI)
bool isDS(uint32_t Opcode) const
MachineInstr * getWholeWaveFunctionSetup(MachineFunction &MF) const
static bool isMFMAorWMMA(const MachineInstr &MI)
static bool isWQM(const MachineInstr &MI)
static bool doesNotReadTiedSource(const MachineInstr &MI)
bool isLegalVSrcOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO would be a valid operand for the given operand definition OpInfo.
static bool isDOT(const MachineInstr &MI)
std::unique_ptr< PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
InstSizeVerifyMode getInstSizeVerifyMode(const MachineInstr &MI) const override
bool isGather4(uint32_t Opcode) const
static bool usesFPDPRounding(const MachineInstr &MI)
bool isVIMAGE(uint32_t Opcode) const
MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const override
bool isImage(uint32_t Opcode) const
static bool usesTENSOR_CNT(const MachineInstr &MI)
bool isInlineConstant(const MachineOperand &MO) const
bool hasModifiers(unsigned Opcode) const
Return true if this instruction has any modifiers.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
static bool isSWMMAC(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
bool isDPP(uint32_t Opcode) const
bool isWave32() const
bool isHighLatencyDef(int Opc) const override
void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const
Legalize the OpIndex operand of this instruction by inserting a MOV.
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
static bool isVOPC(const MachineInstr &MI)
void removeModOperands(MachineInstr &MI) const
bool hasFPClamp(uint32_t Opcode) const
bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx, int64_t ImmVal) const
bool isGWS(uint32_t Opcode) const
unsigned getRepeatRate(const MachineInstr &MI) const
Get the repeat rate for a VALU instruction from the scheduling model.
unsigned getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI) const
bool isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN, const MachineOperand *MO=nullptr) const
Check if MO would be a legal operand for a single-SGPR-read instruction.
bool isXDL(const MachineInstr &MI) const
Register isStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
static bool isVIMAGE(const MachineInstr &MI)
static bool isLDSDIR(const MachineInstr &MI)
void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const
static bool isSOP2(const MachineInstr &MI)
LLVM_READONLY const MachineOperand * getNamedOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
static bool isGWS(const MachineInstr &MI)
bool hasRAWDependency(const MachineInstr &FirstMI, const MachineInstr &SecondMI) const
bool isLegalAV64PseudoImm(uint64_t Imm) const
Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
bool isNeverCoissue(MachineInstr &MI) const
const TargetSchedModel & getSchedModel() const
static bool isBUF(const MachineInstr &MI)
bool isInlineConstant(const MachineInstr &MI, const MachineOperand &UseMO, const MachineOperand &DefMO) const
returns true if UseMO is substituted with DefMO in MI it would be an inline immediate.
bool isSOPC(uint32_t Opcode) const
bool isMAI(uint32_t Opcode) const
bool isNonCommutableDPP(const MachineInstr &MI) const
void handleCopyToPhysHelper(SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst, MachineRegisterInfo &MRI, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const
bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx, unsigned toIdx) const
bool isFixedSize(uint32_t Opcode) const
static bool isFLATGlobal(const MachineInstr &MI)
unsigned getMachineCSELookAheadLimit() const override
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
bool isGlobalMemoryObject(const MachineInstr *MI) const override
static bool isVSAMPLE(const MachineInstr &MI)
static bool isAtomicRet(const MachineInstr &MI)
bool isBufferSMRD(const MachineInstr &MI) const
static bool isKillTerminator(unsigned Opcode)
bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const
If OpX is multicycle, anti-dependencies are not allowed.
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
const GCNSubtarget & getSubtarget() const
void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, bool IsSCCLive, SlotIndexes *Indexes=nullptr) const
bool hasVALU32BitEncoding(unsigned Opcode) const
Return true if this 64-bit VALU instruction has a 32-bit encoding.
static bool isDisableWQM(const MachineInstr &MI)
unsigned getBlockingCycles(const MachineInstr &MI) const
bool isVOP3PMix(const MachineInstr &MI) const
unsigned getMovOpcode(const TargetRegisterClass *DstRC) const
Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
bool isMFMAorWMMA(uint32_t Opcode) const
unsigned buildExtractSubReg(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const
Legalize operands in MI by either commuting it or inserting a copy of src1.
static bool isVPermPk16(unsigned Opcode)
bool isFLATGlobal(uint32_t Opcode) const
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const final
static bool isTRANS(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isSOPK(const MachineInstr &MI)
const TargetRegisterClass * getOpRegClass(const MachineInstr &MI, unsigned OpNo) const
Return the correct register class for OpNo.
MachineBasicBlock * insertSimulatedTrap(MachineRegisterInfo &MRI, MachineBasicBlock &MBB, MachineInstr &MI, const DebugLoc &DL) const
Build instructions that simulate the behavior of a s_trap 2 instructions for hardware (namely,...
bool isMIMG(uint32_t Opcode) const
static unsigned getNonSoftWaitcntOpcode(unsigned Opcode)
bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx) const
returns true if the operand OpIdx in MI is a valid inline immediate.
bool isVGPRSpill(uint32_t Opcode) const
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static bool isFoldableCopy(const MachineInstr &MI)
bool isAtomic(uint32_t Opcode) const
static bool isVINTERP(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
bool expandPostRAPseudo(MachineInstr &MI) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={}) const
Wrapper function for generating waterfall for instruction MI This function take into consideration of...
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool hasFPClamp(const MachineInstr &MI)
bool isVOP2(uint32_t Opcode) const
static bool isGFX12CacheInvOrWBInst(unsigned Opc)
static bool isSegmentSpecificFLAT(const MachineInstr &MI)
static bool isWaitcnt(unsigned Opcode)
bool isReMaterializableImpl(const MachineInstr &MI) const override
bool isFLATScratch(uint32_t Opcode) const
static bool isVOP3(const MCInstrDesc &Desc)
unsigned getOpSize(const MachineInstr &MI, unsigned OpNo) const
This form should usually be preferred since it handles operands with unknown register classes.
bool isSegmentSpecificFLAT(uint32_t Opcode) const
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool physRegUsesConstantBus(const MachineOperand &Reg) const
bool isInlineConstant(const MachineOperand &MO, const MCOperandInfo &OpInfo) const
static bool isF16PseudoScalarTrans(unsigned Opcode)
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const
static bool isChainCallOpcode(uint64_t Opcode)
bool isVOPC(uint32_t Opcode) const
static bool isDPP(const MachineInstr &MI)
bool analyzeBranchImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const
bool isPacked(uint32_t Opcode) const
static bool isMFMA(const MachineInstr &MI)
bool isLowLatencyInstruction(const MachineInstr &MI) const
bool isIGLP(const MachineInstr &MI) const
static bool isScalarStore(const MachineInstr &MI)
bool isFLAT(uint32_t Opcode) const
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is a instruction that moves/copies value from one register to ano...
bool isVALU(uint32_t Opcode, bool AllowLDSDMA) const
LDSDMA instructions act as both VALU and memory instructions, thus we also tag them as VALU.
void mutateAndCleanupImplicit(MachineInstr &MI, const MCInstrDesc &NewDesc) const
ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const
static bool isMAI(const MCInstrDesc &Desc)
static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode)
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
static bool isFPAtomic(const MachineInstr &MI)
static bool usesLGKM_CNT(const MachineInstr &MI)
bool isFPAtomic(uint32_t Opcode) const
void legalizeOperandsVALUt16(MachineInstr &Inst, MachineRegisterInfo &MRI) const
Fix operands in Inst to fix 16bit SALU to VALU lowering.
bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo, const MachineOperand &MO) const
static bool isPacked(const MachineInstr &MI)
bool canShrink(const MachineInstr &MI, const MachineRegisterInfo &MRI) const
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool isVPermPk16SafeInstr(const MachineInstr &MI) const
bool isAsmOnlyOpcode(int MCOp) const
Check if this instruction should only be used by assembler.
bool isAlwaysGDS(uint32_t Opcode) const
static bool isVGPRSpill(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
This is used by the post-RA scheduler (SchedulePostRAList.cpp).
bool verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const override
static bool isSBarrierSCCWrite(unsigned Opcode)
unsigned getInstrLatency(const InstrItineraryData *ItinData, const MachineInstr &MI, unsigned *PredCost=nullptr) const override
unsigned getVectorRegSpillSaveOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI, bool NeedsCFI) const
int64_t getNamedImmOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
Get required immediate operand.
bool isSOP2(uint32_t Opcode) const
ArrayRef< std::pair< int, const char * > > getSerializableTargetIndices() const override
bool isVGPRCopy(const MachineInstr &MI) const
bool isVOP1(uint32_t Opcode) const
bool isVINTERP(uint32_t Opcode) const
bool regUsesConstantBus(const MachineOperand &Reg, const MachineRegisterInfo &MRI) const
static bool isMIMG(const MachineInstr &MI)
MachineOperand buildExtractSubRegOrImm(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
bool isLegalRegOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO (a register operand) is a legal register for the given operand description or operand ind...
LLVM_READONLY int commuteOpcode(const MachineInstr &MI) const
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
static bool isVOP3P(const MachineInstr &MI)
unsigned getVALUOp(const MachineInstr &MI) const
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI, MachineRegisterInfo &MRI, const TargetRegisterClass *DstRC=nullptr) const
Copy a value from a VGPR (SrcReg) to SGPR.
bool hasDivergentBranch(const MachineBasicBlock *MBB) const
Return whether the block terminate with divergent branch.
bool isInlineConstant(const MachineOperand &MO, uint8_t OperandType) const
std::pair< int64_t, int64_t > splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Split COffsetVal into {immediate offset field, remainder offset} values.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
bool isVOP3P(uint32_t Opcode) const
void fixImplicitOperands(MachineInstr &MI) const
bool moveFlatAddrToVGPR(MachineInstr &Inst) const
Change SADDR form of a FLAT Inst to its VADDR form if saddr operand was moved to VGPR.
bool isSWMMAC(uint32_t Opcode) const
static bool usesASYNC_CNT(const MachineInstr &MI)
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool isMaskedByExec(Register Reg, const MachineInstr &Use, const MachineRegisterInfo &MRI, unsigned Depth=0) const
Return true if Reg is a lane mask that already has 0 in every bit corresponding to a lane that is ina...
void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const
bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0, AMDGPU::OpName Src0OpName, MachineOperand &Src1, AMDGPU::OpName Src1OpName) const
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
static bool isDualSourceBlendEXP(const MachineInstr &MI)
bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const
This function is used to determine if an instruction can be safely executed under EXEC = 0 without ha...
bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg, int64_t &ImmVal) const override
static bool isAtomic(const MachineInstr &MI)
bool canInsertSelect(const MachineBasicBlock &MBB, ArrayRef< MachineOperand > Cond, Register DstReg, Register TrueReg, Register FalseReg, int &CondCycles, int &TrueCycles, int &FalseCycles) const override
bool isLiteralOperandLegal(const MCInstrDesc &InstDesc, const MCOperandInfo &OpInfo) const
static bool isWWMRegSpillOpcode(uint32_t Opcode)
static bool sopkIsZext(unsigned Opcode)
static bool isSGPRSpill(const MachineInstr &MI)
static bool isWMMA(const MachineInstr &MI)
bool isMFMA(uint32_t Opcode) const
ArrayRef< std::pair< MachineMemOperand::Flags, const char * > > getSerializableMachineMemOperandTargetFlags() const override
bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const
Returns true if the instruction could potentially depend on the value of exec.
bool isSALU(uint32_t Opcode) const
bool isSMRD(uint32_t Opcode) const
bool isWMMA(uint32_t Opcode) const
void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool isVSAMPLE(uint32_t Opcode) const
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
std::pair< MachineInstr *, MachineInstr * > expandMovDPP64(MachineInstr &MI) const
static bool isLoadMonitor(unsigned Opc)
static bool isSOP1(const MachineInstr &MI)
static bool isSOPC(const MachineInstr &MI)
bool isSOPP(uint32_t Opcode) const
static bool isFLAT(const MachineInstr &MI)
const SIRegisterInfo & getRegisterInfo() const
bool isLDSDIR(uint32_t Opcode) const
bool isBarrier(unsigned Opcode) const
bool isAtomicNoRet(uint32_t Opcode) const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx0, unsigned OpIdx1) const override
bool isVOP3(uint32_t Opcode) const
bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const
static bool hasIntClamp(const MachineInstr &MI)
static bool isSpill(const MCInstrDesc &Desc)
int pseudoToMCOpcode(int Opcode) const
Return a target-specific opcode if Opcode is a pseudo instruction.
const MCInstrDesc & getMCOpcodeFromPseudo(unsigned Opcode) const
Return the descriptor of the target-specific machine instruction that corresponds to the specified ps...
bool isImmOperandLegal(const MachineInstr &MI, unsigned OpNo, const MachineOperand &MO) const
bool isEXP(uint32_t Opcode) const
bool isInlineConstant(const MachineInstr &MI, unsigned OpIdx, const MachineOperand &MO) const
static bool isScalarUnit(const MachineInstr &MI)
static bool usesVM_CNT(const MachineInstr &MI)
MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const override
static bool isFixedSize(const MachineInstr &MI)
bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo, MachineCycleInfo *CI) const override
static bool isPseudoScalarTrans(unsigned Opcode)
bool usesTENSOR_CNT(uint32_t Opcode) const
LLVM_READONLY int commuteOpcode(unsigned Opc) const
bool isScalarStore(uint32_t Opcode) const
static bool isBlockLoadStore(uint32_t Opcode)
ValueUniformity getValueUniformity(const MachineInstr &MI) const final
uint64_t getScratchRsrcWords23() const
bool usesASYNC_CNT(uint32_t Opcode) const
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
MachineInstr * convertToThreeAddress(MachineInstr &MI, LiveIntervals *LIS) const override
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool isVMEM(uint32_t Opcode) const
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx, const MachineOperand *MO=nullptr) const
Check if MO is a legal operand if it was the OpIdx Operand for MI.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const
Returns true if negative offsets are allowed for the given FlatVariant.
bool isBarrierStart(unsigned Opcode) const
void moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
static bool isLDSDMA(const MachineInstr &MI)
static bool isAtomicNoRet(const MachineInstr &MI)
static bool isVOP1(const MachineInstr &MI)
SIInstrInfo(const GCNSubtarget &ST)
std::optional< int64_t > getImmOrMaterializedImm(const MachineRegisterInfo &MRI, const MachineOperand &Op, MachineInstr **DefMI=nullptr) const
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
bool isTRANS(uint32_t Opcode) const
bool isSDWA(uint32_t Opcode) const
bool hasAnyModifiersSet(const MachineInstr &MI) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
A vector that has set insertion semantics.
Definition SetVector.h:57
SlotIndexes pass.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Provide an instruction scheduling machine model to CodeGen passes.
Target - Wrapper for Target specific information.
static constexpr TypeSize getZero()
Definition TypeSize.h:345
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
const uint64_t RSRC_DATA_FORMAT
LLVM_READONLY int32_t getSOPKOp(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteRev(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteOrig(uint32_t Opcode)
LLVM_READONLY int32_t getDPPOp32(uint32_t Opcode)
LLVM_READONLY int32_t getGlobalVaddrOp(uint32_t Opcode)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
const uint64_t RSRC_ELEMENT_SIZE_SHIFT
LLVM_READONLY int32_t getIfAddr64Inst(uint32_t Opcode)
Check if Opcode is an Addr64 opcode.
const uint64_t RSRC_TID_ENABLE
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
LLVM_READONLY int32_t getSDWAOp(uint32_t Opcode)
LLVM_READONLY int32_t getVCMPXOpFromVCMP(uint32_t Opcode)
LLVM_READONLY int32_t getAGPRFormOp(uint32_t Opcode)
LLVM_READONLY int32_t getGlobalSaddrOp(uint32_t Opcode)
LLVM_READONLY int32_t getAddr64Inst(uint32_t Opcode)
LLVM_READONLY int32_t getVOPe64(uint32_t Opcode)
LLVM_READONLY int32_t getDPPOp64(uint32_t Opcode)
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
LLVM_READONLY int32_t getBasicFromSDWAOp(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
const uint64_t RSRC_INDEX_STRIDE_SHIFT
bool getMAIIsDGEMM(unsigned Opc)
Returns true if MAI operation is a double precision GEMM.
LLVM_READONLY int32_t getFlatScratchInstSVfromSVS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSTfromSS(uint32_t Opcode)
@ OPERAND_IMMEDIATE
Definition MCInstrDesc.h:61
constexpr bool isAtomicRet(const T &...O)
Definition SIDefines.h:361
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:233
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:236
constexpr bool isScalarStore(const T &...O)
Definition SIDefines.h:310
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:227
constexpr bool isFPAtomic(const T &...O)
Definition SIDefines.h:346
constexpr bool usesVM_CNT(const T &...O)
Definition SIDefines.h:382
constexpr bool usesTENSOR_CNT(const T &...O)
Definition SIDefines.h:307
constexpr bool isMAI(const T &...O)
Definition SIDefines.h:349
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:230
constexpr bool isSWMMAC(const T &...O)
Definition SIDefines.h:376
constexpr bool isTRANS(const T &...O)
Definition SIDefines.h:255
constexpr bool isSOP2(const T &...O)
Definition SIDefines.h:215
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:283
constexpr bool isVOP3P(const T &...O)
Definition SIDefines.h:239
constexpr bool usesFPDPRounding(const T &...O)
Definition SIDefines.h:343
constexpr bool isDisableWQM(const T &...O)
Definition SIDefines.h:301
constexpr bool hasIntClamp(const T &...O)
Definition SIDefines.h:328
constexpr bool isAtomicNoRet(const T &...O)
Definition SIDefines.h:358
constexpr bool isMTBUF(const T &...O)
Definition SIDefines.h:261
constexpr bool isVIMAGE(const T &...O)
Definition SIDefines.h:274
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:268
constexpr bool isFlatScratch(const T &...O)
Definition SIDefines.h:355
constexpr bool isSpill(const T &...O)
Definition SIDefines.h:289
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:271
constexpr bool isVMEM(const T &...O)
Definition SIDefines.h:401
constexpr bool hasFPClamp(const T &...O)
Definition SIDefines.h:325
constexpr bool isNeverUniform(const T &...O)
Definition SIDefines.h:370
constexpr bool isImage(const T &...O)
Definition SIDefines.h:397
constexpr bool isWMMA(const T &...O)
Definition SIDefines.h:364
constexpr bool isVALU(const T &...O)
Definition SIDefines.h:209
constexpr bool isWQM(const T &...O)
Definition SIDefines.h:298
constexpr bool hasClampLo(const T &...O)
Definition SIDefines.h:331
constexpr bool isGWS(const T &...O)
Definition SIDefines.h:373
constexpr bool isFlatGlobal(const T &...O)
Definition SIDefines.h:340
constexpr bool usesASYNC_CNT(const T &...O)
Definition SIDefines.h:316
constexpr bool isMUBUF(const T &...O)
Definition SIDefines.h:258
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
constexpr bool isTiedSourceNotRead(const T &...O)
Definition SIDefines.h:367
constexpr bool isEXP(const T &...O)
Definition SIDefines.h:280
constexpr bool usesLGKM_CNT(const T &...O)
Definition SIDefines.h:385
constexpr bool isSOPK(const T &...O)
Definition SIDefines.h:221
constexpr bool isSOPC(const T &...O)
Definition SIDefines.h:218
constexpr bool isSOPP(const T &...O)
Definition SIDefines.h:224
constexpr bool isDOT(const T &...O)
Definition SIDefines.h:352
constexpr bool isVINTRP(const T &...O)
Definition SIDefines.h:246
constexpr bool isVINTERP(const T &...O)
Definition SIDefines.h:295
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:277
constexpr bool isDS(const T &...O)
Definition SIDefines.h:286
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:390
constexpr bool isLDSDIR(const T &...O)
Definition SIDefines.h:292
constexpr bool isSALU(const T &...O)
Definition SIDefines.h:206
constexpr bool isGather4(const T &...O)
Definition SIDefines.h:304
constexpr bool isPacked(const T &...O)
Definition SIDefines.h:337
constexpr bool hasClampHi(const T &...O)
Definition SIDefines.h:334
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:252
constexpr bool isSOP1(const T &...O)
Definition SIDefines.h:212
constexpr bool isSegmentSpecificFLAT(const T &...O)
Definition SIDefines.h:393
constexpr bool isFixedSize(const T &...O)
Definition SIDefines.h:313
Offsets
Offsets in bytes from the start of the input buffer.
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
uint64_t getTSFlags(const MachineInstr &MI)
Op::Description Desc
TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI, unsigned SubReg)
Return the SubReg component from REG_SEQUENCE.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
Definition SIInstrInfo.h:45
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
MachineInstr * getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P, const MachineRegisterInfo &MRI)
Return the defining instruction for a given reg:subreg pair skipping copy like instructions and subre...
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
Definition SIInstrInfo.h:53
DWARFExpression::Operation Op
constexpr unsigned DefaultMemoryClusterDWordsLimit
Definition SIInstrInfo.h:41
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Definition SIInstrInfo.h:49
bool isOfRegClass(const TargetInstrInfo::RegSubRegPair &P, const TargetRegisterClass &TRC, MachineRegisterInfo &MRI)
Returns true if a reg:subreg pair P has a TRC class.
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
Definition Uniformity.h:18
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
Definition SIInstrInfo.h:64
bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI)
Return false if EXEC is not changed between the def of VReg at DefMI and all its uses.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
Helper struct for the implementation of 3-address conversion to communicate updates made to instructi...
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static constexpr LaneBitmask getAll()
Definition LaneBitmask.h:82
Utility to store machine instructions worklist.
Definition SIInstrInfo.h:68
MachineInstr * top() const
Definition SIInstrInfo.h:73
bool isDeferred(MachineInstr *MI)
SetVector< MachineInstr * > & getDeferredList()
Definition SIInstrInfo.h:91
void insert(MachineInstr *MI)
A pair composed of a register and a sub-register index.
SmallVector< MachineOperand * > MOs
Definition SIInstrInfo.h:58
SmallVector< Register > SGPRs
Definition SIInstrInfo.h:60