LLVM 24.0.0git
SIInstrInfo.cpp
Go to the documentation of this file.
1//===- SIInstrInfo.cpp - SI Instruction Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// SI Implementation of TargetInstrInfo.
11//
12//===----------------------------------------------------------------------===//
13
14#include "SIInstrInfo.h"
15#include "AMDGPU.h"
16#include "AMDGPUInstrInfo.h"
17#include "AMDGPULaneMaskUtils.h"
18#include "GCNHazardRecognizer.h"
19#include "GCNSubtarget.h"
22#include "llvm/ADT/STLExtras.h"
33#include "llvm/IR/IntrinsicsAMDGPU.h"
34#include "llvm/MC/MCContext.h"
37#include <tuple>
38
39using namespace llvm;
40
41#define DEBUG_TYPE "si-instr-info"
42
43#define GET_INSTRINFO_CTOR_DTOR
44#include "AMDGPUGenInstrInfo.inc"
45
46namespace llvm::AMDGPU {
47#define GET_ImageDimIntrinsicTable_IMPL
48#define GET_RsrcIntrinsics_IMPL
49#define GET_GFX1250BlockingCyclesTable_DECL
50#define GET_GFX1250BlockingCyclesTable_IMPL
51
56
57#include "AMDGPUGenSearchableTables.inc"
58} // namespace llvm::AMDGPU
59
60// Must be at least 4 to be able to branch over minimum unconditional branch
61// code. This is only for making it possible to write reasonably small tests for
62// long branches.
64BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16),
65 cl::desc("Restrict range of branch instructions (DEBUG)"));
66
68 "amdgpu-fix-16-bit-physreg-copies",
69 cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"),
70 cl::init(true),
72
74 : AMDGPUGenInstrInfo(ST, RI, AMDGPU::ADJCALLSTACKUP,
75 AMDGPU::ADJCALLSTACKDOWN),
76 RI(ST), ST(ST) {
77 SchedModel.init(&ST);
78}
79
80//===----------------------------------------------------------------------===//
81// TargetInstrInfo callbacks
82//===----------------------------------------------------------------------===//
83
84static unsigned getNumOperandsNoGlue(SDNode *Node) {
85 unsigned N = Node->getNumOperands();
86 while (N && Node->getOperand(N - 1).getValueType() == MVT::Glue)
87 --N;
88 return N;
89}
90
91/// Returns true if both nodes have the same value for the given
92/// operand \p Op, or if both nodes do not have this operand.
94 AMDGPU::OpName OpName) {
95 unsigned Opc0 = N0->getMachineOpcode();
96 unsigned Opc1 = N1->getMachineOpcode();
97
98 int Op0Idx = AMDGPU::getNamedOperandIdx(Opc0, OpName);
99 int Op1Idx = AMDGPU::getNamedOperandIdx(Opc1, OpName);
100
101 if (Op0Idx == -1 && Op1Idx == -1)
102 return true;
103
104
105 if ((Op0Idx == -1 && Op1Idx != -1) ||
106 (Op1Idx == -1 && Op0Idx != -1))
107 return false;
108
109 // getNamedOperandIdx returns the index for the MachineInstr's operands,
110 // which includes the result as the first operand. We are indexing into the
111 // MachineSDNode's operands, so we need to skip the result operand to get
112 // the real index.
113 --Op0Idx;
114 --Op1Idx;
115
116 return N0->getOperand(Op0Idx) == N1->getOperand(Op1Idx);
117}
118
119static bool canRemat(const MachineInstr &MI) {
120
124 return true;
125
126 if (SIInstrInfo::isSMRD(MI)) {
127 return !MI.memoperands_empty() &&
128 llvm::all_of(MI.memoperands(), [](const MachineMemOperand *MMO) {
129 return MMO->isLoad() && MMO->isInvariant();
130 });
131 }
132
133 return false;
134}
135
136// Split relocation flags for 64-bit global-address materialization into a
137// common base and the hi/lo relocation variants.
138static std::tuple<unsigned, unsigned, unsigned>
140 const MachineOperand &SrcOp) {
141 const unsigned BaseFlags = SrcOp.getTargetFlags() & ~SIInstrInfo::MO_MASK;
142 const unsigned Reloc = SrcOp.getTargetFlags() & SIInstrInfo::MO_MASK;
143
144 // Infer the relocation type from the existing flags on the global operand.
145 // The relocation type should have been determined earlier in the pipeline.
146 unsigned LoReloc, HiReloc;
147 switch (Reloc) {
151 LoReloc = SIInstrInfo::MO_REL32_LO;
152 HiReloc = SIInstrInfo::MO_REL32_HI;
153 break;
158 break;
161 // For 64-bit GOT-relative, use the 64-bit relocation.
164 break;
168 LoReloc = SIInstrInfo::MO_ABS32_LO;
169 HiReloc = SIInstrInfo::MO_ABS32_HI;
170 break;
171 default:
172 llvm_unreachable("unknown relocation type for global address");
173 break;
174 }
175
176 return {BaseFlags, LoReloc, HiReloc};
177}
178
180 const MachineInstr &MI) const {
181
182 if (canRemat(MI)) {
183 // Normally VALU use of exec would block the rematerialization, but that
184 // is OK in this case to have an implicit exec read as all VALU do.
185 // We really want all of the generic logic for this except for this.
186
187 // Another potential implicit use is mode register. The core logic of
188 // the RA will not attempt rematerialization if mode is set anywhere
189 // in the function, otherwise it is safe since mode is not changed.
190
191 // There is difference to generic method which does not allow
192 // rematerialization if there are virtual register uses. We allow this,
193 // therefore this method includes SOP instructions as well.
194 if (!MI.hasImplicitDef() &&
195 MI.getNumImplicitOperands() == MI.getDesc().implicit_uses().size() &&
196 !MI.mayRaiseFPException())
197 return true;
198 }
199
200 // Everything below copied from TargetInstrInfo::isReMaterializableImpl. The
201 // only difference is that we allow operations that perform read-modify-write
202 // on sub-registers.
203
204 // Remat clients assume operand 0 is the defined register.
205 if (!MI.getNumOperands() || !MI.getOperand(0).isReg())
206 return false;
207 Register DefReg = MI.getOperand(0).getReg();
208
209 const MachineFunction &MF = *MI.getMF();
210
211 // A load from a fixed stack slot can be rematerialized. This may be
212 // redundant with subsequent checks, but it's target-independent,
213 // simple, and a common case.
214 int FrameIdx = 0;
215 if (isLoadFromStackSlot(MI, FrameIdx) &&
217 return true;
218
219 // Avoid instructions obviously unsafe for remat.
220 if (MI.isNotDuplicable() || MI.mayStore() || MI.mayRaiseFPException() ||
221 MI.hasUnmodeledSideEffects())
222 return false;
223
224 // Don't remat inline asm. We have no idea how expensive it is
225 // even if it's side effect free.
226 if (MI.isInlineAsm())
227 return false;
228
229 // Avoid instructions which load from potentially varying memory.
230 if (MI.mayLoad() && !MI.isDereferenceableInvariantLoad())
231 return false;
232
233 const MachineRegisterInfo &MRI = MF.getRegInfo();
234
235 // If any of the registers accessed are non-constant, conservatively assume
236 // the instruction is not rematerializable.
237 for (const MachineOperand &MO : MI.operands()) {
238 if (!MO.isReg())
239 continue;
240 Register Reg = MO.getReg();
241 if (Reg == 0)
242 continue;
243
244 // Check for a well-behaved physical register.
245 if (Reg.isPhysical()) {
246 if (MO.isUse()) {
247 // If the physreg has no defs anywhere, it's just an ambient register
248 // and we can freely move its uses. Alternatively, if it's allocatable,
249 // it could get allocated to something with a def during allocation.
250 if (!MRI.isConstantPhysReg(Reg))
251 return false;
252 } else {
253 // A physreg def. We can't remat it.
254 return false;
255 }
256 continue;
257 }
258
259 // Only allow one virtual-register def. There may be multiple defs of the
260 // same virtual register, though.
261 if (MO.isDef() && Reg != DefReg)
262 return false;
263 }
264
265 return true;
266}
267
269 switch (Opcode) {
270 // v_subrev_u16 (gfx9)
271 case AMDGPU::V_SUBREV_U16_e32:
272 case AMDGPU::V_SUBREV_U16_e64:
273 // v_subrev_u32 (gfx9) / v_subrev_nc_u32 (gfx10+)
274 case AMDGPU::V_SUBREV_U32_e32:
275 case AMDGPU::V_SUBREV_U32_e64:
276 // v_subrev_co_u32
277 case AMDGPU::V_SUBREV_CO_U32_e32:
278 case AMDGPU::V_SUBREV_CO_U32_e64:
279 // v_subbrev_u32 (gfx9) / v_subrev_co_ci_u32 (gfx10+)
280 case AMDGPU::V_SUBBREV_U32_e32:
281 case AMDGPU::V_SUBBREV_U32_e64:
282 return true;
283 // REV shift opcodes worked this way before GFX11, verified on hardware
284 case AMDGPU::V_ASHRREV_I16_e32:
285 case AMDGPU::V_ASHRREV_I16_e64:
286 case AMDGPU::V_ASHRREV_I32_e32:
287 case AMDGPU::V_ASHRREV_I32_e64:
288 case AMDGPU::V_ASHRREV_I64_e64:
289 case AMDGPU::V_LSHLREV_B16_e32:
290 case AMDGPU::V_LSHLREV_B16_e64:
291 case AMDGPU::V_LSHLREV_B32_e32:
292 case AMDGPU::V_LSHLREV_B32_e64:
293 case AMDGPU::V_LSHLREV_B64_e64:
294 case AMDGPU::V_LSHRREV_B16_e32:
295 case AMDGPU::V_LSHRREV_B16_e64:
296 case AMDGPU::V_LSHRREV_B32_e32:
297 case AMDGPU::V_LSHRREV_B32_e64:
298 case AMDGPU::V_LSHRREV_B64_e64:
299 return !ST.hasGFX11Insts();
300 default:
301 return false;
302 }
303}
304
305// Returns true if the result of a VALU instruction depends on exec.
306bool SIInstrInfo::resultDependsOnExec(const MachineInstr &MI) const {
307 assert(isVALU(MI, /*AllowLDSDMA=*/true));
308
309 // If it is convergent it depends on EXEC.
310 if (MI.isConvergent())
311 return true;
312
313 // If it defines an SGPR it depends on EXEC, unless it's dead.
314 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
315 for (const MachineOperand &Def : MI.defs()) {
316 if (Def.isDead())
317 continue;
318
319 Register Reg = Def.getReg();
320 if (Reg && RI.isSGPRReg(MRI, Reg))
321 return true;
322 }
323
324 return false;
325}
326
327bool SIInstrInfo::isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const {
328 const MachineOperand &MO = MI.getOperand(OpIdx);
329 // Any implicit use of exec by VALU is not a real register read.
330 return MO.getReg() == AMDGPU::EXEC && MO.isImplicit() &&
331 isVALU(MI, /*AllowLDSDMA=*/true) && !resultDependsOnExec(MI);
332}
333
335 MachineBasicBlock *SuccToSinkTo,
336 MachineCycleInfo *CI) const {
337 // Allow sinking if MI edits lane mask (divergent i1 in sgpr).
338 if (MI.getOpcode() == AMDGPU::SI_IF_BREAK)
339 return true;
340
341 MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
342 // Check if sinking of MI would create temporal divergent use.
343 for (auto Op : MI.uses()) {
344 if (Op.isReg() && Op.getReg().isVirtual() &&
345 RI.isSGPRClass(MRI.getRegClass(Op.getReg()))) {
346 MachineInstr *SgprDef = MRI.getVRegDef(Op.getReg());
347 if (!SgprDef)
348 continue;
349
350 // SgprDef defined inside cycle
351 CycleRef FromCycle = CI->getCycle(SgprDef->getParent());
352 if (!FromCycle)
353 continue;
354
355 CycleRef ToCycle = CI->getCycle(SuccToSinkTo);
356 // Check if there is a FromCycle that contains SgprDef's basic block but
357 // does not contain SuccToSinkTo and also has divergent exit condition.
358 while (FromCycle && !(ToCycle && CI->contains(FromCycle, ToCycle))) {
360 CI->getExitingBlocks(FromCycle, ExitingBlocks);
361
362 // FromCycle has divergent exit condition.
363 for (MachineBasicBlock *ExitingBlock : ExitingBlocks) {
364 if (hasDivergentBranch(ExitingBlock))
365 return false;
366 }
367
368 FromCycle = CI->getParentCycle(FromCycle);
369 }
370 }
371 }
372
373 return true;
374}
375
377 int64_t &Offset0,
378 int64_t &Offset1) const {
379 if (!Load0->isMachineOpcode() || !Load1->isMachineOpcode())
380 return false;
381
382 unsigned Opc0 = Load0->getMachineOpcode();
383 unsigned Opc1 = Load1->getMachineOpcode();
384
385 // Make sure both are actually loads.
386 if (!get(Opc0).mayLoad() || !get(Opc1).mayLoad())
387 return false;
388
389 // A mayLoad instruction without a def is not a load. Likely a prefetch.
390 if (!get(Opc0).getNumDefs() || !get(Opc1).getNumDefs())
391 return false;
392
393 if (isDS(Opc0) && isDS(Opc1)) {
394
395 // FIXME: Handle this case:
396 if (getNumOperandsNoGlue(Load0) != getNumOperandsNoGlue(Load1))
397 return false;
398
399 // Check base reg.
400 if (Load0->getOperand(0) != Load1->getOperand(0))
401 return false;
402
403 // Skip read2 / write2 variants for simplicity.
404 // TODO: We should report true if the used offsets are adjacent (excluded
405 // st64 versions).
406 int Offset0Idx = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
407 int Offset1Idx = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
408 if (Offset0Idx == -1 || Offset1Idx == -1)
409 return false;
410
411 // XXX - be careful of dataless loads
412 // getNamedOperandIdx returns the index for MachineInstrs. Since they
413 // include the output in the operand list, but SDNodes don't, we need to
414 // subtract the index by one.
415 Offset0Idx -= get(Opc0).NumDefs;
416 Offset1Idx -= get(Opc1).NumDefs;
417 Offset0 = Load0->getConstantOperandVal(Offset0Idx);
418 Offset1 = Load1->getConstantOperandVal(Offset1Idx);
419 return true;
420 }
421
422 if (isSMRD(Opc0) && isSMRD(Opc1)) {
423 // Skip time and cache invalidation instructions.
424 if (!AMDGPU::hasNamedOperand(Opc0, AMDGPU::OpName::sbase) ||
425 !AMDGPU::hasNamedOperand(Opc1, AMDGPU::OpName::sbase))
426 return false;
427
428 unsigned NumOps = getNumOperandsNoGlue(Load0);
429 if (NumOps != getNumOperandsNoGlue(Load1))
430 return false;
431
432 // Check base reg.
433 if (Load0->getOperand(0) != Load1->getOperand(0))
434 return false;
435
436 // Match register offsets, if both register and immediate offsets present.
437 assert(NumOps == 4 || NumOps == 5);
438 if (NumOps == 5 && Load0->getOperand(1) != Load1->getOperand(1))
439 return false;
440
441 const ConstantSDNode *Load0Offset =
443 const ConstantSDNode *Load1Offset =
445
446 if (!Load0Offset || !Load1Offset)
447 return false;
448
449 Offset0 = Load0Offset->getZExtValue();
450 Offset1 = Load1Offset->getZExtValue();
451 return true;
452 }
453
454 // MUBUF and MTBUF can access the same addresses.
455 if ((isMUBUF(Opc0) || isMTBUF(Opc0)) && (isMUBUF(Opc1) || isMTBUF(Opc1))) {
456
457 // MUBUF and MTBUF have vaddr at different indices.
458 if (!nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::soffset) ||
459 !nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::vaddr) ||
460 !nodesHaveSameOperandValue(Load0, Load1, AMDGPU::OpName::srsrc))
461 return false;
462
463 int OffIdx0 = AMDGPU::getNamedOperandIdx(Opc0, AMDGPU::OpName::offset);
464 int OffIdx1 = AMDGPU::getNamedOperandIdx(Opc1, AMDGPU::OpName::offset);
465
466 if (OffIdx0 == -1 || OffIdx1 == -1)
467 return false;
468
469 // getNamedOperandIdx returns the index for MachineInstrs. Since they
470 // include the output in the operand list, but SDNodes don't, we need to
471 // subtract the index by one.
472 OffIdx0 -= get(Opc0).NumDefs;
473 OffIdx1 -= get(Opc1).NumDefs;
474
475 SDValue Off0 = Load0->getOperand(OffIdx0);
476 SDValue Off1 = Load1->getOperand(OffIdx1);
477
478 // The offset might be a FrameIndexSDNode.
479 if (!isa<ConstantSDNode>(Off0) || !isa<ConstantSDNode>(Off1))
480 return false;
481
482 Offset0 = Off0->getAsZExtVal();
483 Offset1 = Off1->getAsZExtVal();
484 return true;
485 }
486
487 return false;
488}
489
490static bool isStride64(unsigned Opc) {
491 switch (Opc) {
492 case AMDGPU::DS_READ2ST64_B32:
493 case AMDGPU::DS_READ2ST64_B64:
494 case AMDGPU::DS_WRITE2ST64_B32:
495 case AMDGPU::DS_WRITE2ST64_B64:
496 return true;
497 default:
498 return false;
499 }
500}
501
504 int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const {
505 if (!LdSt.mayLoadOrStore())
506 return false;
507
508 unsigned Opc = LdSt.getOpcode();
509 OffsetIsScalable = false;
510 const MachineOperand *BaseOp, *OffsetOp;
511 int DataOpIdx;
512
513 if (isDS(LdSt)) {
514 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::addr);
515 OffsetOp = getNamedOperand(LdSt, AMDGPU::OpName::offset);
516 if (OffsetOp) {
517 // Normal, single offset LDS instruction.
518 if (!BaseOp) {
519 // DS_CONSUME/DS_APPEND use M0 for the base address.
520 // TODO: find the implicit use operand for M0 and use that as BaseOp?
521 return false;
522 }
523 BaseOps.push_back(BaseOp);
524 Offset = OffsetOp->getImm();
525 // Get appropriate operand, and compute width accordingly.
526 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
527 if (DataOpIdx == -1)
528 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
529 if (Opc == AMDGPU::DS_ATOMIC_ASYNC_BARRIER_ARRIVE_B64)
530 Width = LocationSize::precise(64);
531 else
532 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
533 } else {
534 // The 2 offset instructions use offset0 and offset1 instead. We can treat
535 // these as a load with a single offset if the 2 offsets are consecutive.
536 // We will use this for some partially aligned loads.
537 const MachineOperand *Offset0Op =
538 getNamedOperand(LdSt, AMDGPU::OpName::offset0);
539 const MachineOperand *Offset1Op =
540 getNamedOperand(LdSt, AMDGPU::OpName::offset1);
541
542 unsigned Offset0 = Offset0Op->getImm() & 0xff;
543 unsigned Offset1 = Offset1Op->getImm() & 0xff;
544 if (Offset0 + 1 != Offset1)
545 return false;
546
547 // Each of these offsets is in element sized units, so we need to convert
548 // to bytes of the individual reads.
549
550 unsigned EltSize;
551 if (LdSt.mayLoad())
552 EltSize = RI.getRegSizeInBits(*getOpRegClass(LdSt, 0)) / 16;
553 else {
554 assert(LdSt.mayStore());
555 int Data0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
556 EltSize = RI.getRegSizeInBits(*getOpRegClass(LdSt, Data0Idx)) / 8;
557 }
558
559 if (isStride64(Opc))
560 EltSize *= 64;
561
562 BaseOps.push_back(BaseOp);
563 Offset = EltSize * Offset0;
564 // Get appropriate operand(s), and compute width accordingly.
565 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
566 if (DataOpIdx == -1) {
567 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
568 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
569 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data1);
570 Width = LocationSize::precise(
571 Width.getValue() + TypeSize::getFixed(getOpSize(LdSt, DataOpIdx)));
572 } else {
573 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
574 }
575 }
576 return true;
577 }
578
579 if (isMUBUF(LdSt) || isMTBUF(LdSt)) {
580 const MachineOperand *RSrc = getNamedOperand(LdSt, AMDGPU::OpName::srsrc);
581 if (!RSrc) // e.g. BUFFER_WBINVL1_VOL
582 return false;
583 BaseOps.push_back(RSrc);
584 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::vaddr);
585 if (BaseOp && !BaseOp->isFI())
586 BaseOps.push_back(BaseOp);
587 const MachineOperand *OffsetImm =
588 getNamedOperand(LdSt, AMDGPU::OpName::offset);
589 Offset = OffsetImm->getImm();
590 const MachineOperand *SOffset =
591 getNamedOperand(LdSt, AMDGPU::OpName::soffset);
592 if (SOffset) {
593 if (SOffset->isReg())
594 BaseOps.push_back(SOffset);
595 else
596 Offset += SOffset->getImm();
597 }
598 // Get appropriate operand, and compute width accordingly.
599 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
600 if (DataOpIdx == -1)
601 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
602 if (DataOpIdx == -1) // LDS DMA
603 return false;
604 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
605 return true;
606 }
607
608 if (isImage(LdSt)) {
609 auto RsrcOpName =
610 isMIMG(LdSt) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
611 int SRsrcIdx = AMDGPU::getNamedOperandIdx(Opc, RsrcOpName);
612 BaseOps.push_back(&LdSt.getOperand(SRsrcIdx));
613 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
614 if (VAddr0Idx >= 0) {
615 // GFX10 possible NSA encoding.
616 for (int I = VAddr0Idx; I < SRsrcIdx; ++I)
617 BaseOps.push_back(&LdSt.getOperand(I));
618 } else {
619 BaseOps.push_back(getNamedOperand(LdSt, AMDGPU::OpName::vaddr));
620 }
621 Offset = 0;
622 // Get appropriate operand, and compute width accordingly.
623 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
624 if (DataOpIdx == -1)
625 return false; // no return sampler
626 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
627 return true;
628 }
629
630 if (isSMRD(LdSt)) {
631 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::sbase);
632 if (!BaseOp) // e.g. S_MEMTIME
633 return false;
634 BaseOps.push_back(BaseOp);
635 OffsetOp = getNamedOperand(LdSt, AMDGPU::OpName::offset);
636 Offset = OffsetOp ? OffsetOp->getImm() : 0;
637 // Get appropriate operand, and compute width accordingly.
638 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::sdst);
639 if (DataOpIdx == -1)
640 return false;
641 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
642 return true;
643 }
644
645 if (isFLAT(LdSt)) {
646 // Instructions have either vaddr or saddr or both or none.
647 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::vaddr);
648 if (BaseOp)
649 BaseOps.push_back(BaseOp);
650 BaseOp = getNamedOperand(LdSt, AMDGPU::OpName::saddr);
651 if (BaseOp)
652 BaseOps.push_back(BaseOp);
653 Offset = getNamedOperand(LdSt, AMDGPU::OpName::offset)->getImm();
654 // Get appropriate operand, and compute width accordingly.
655 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
656 if (DataOpIdx == -1)
657 DataOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
658 if (DataOpIdx == -1) // LDS DMA
659 return false;
660 Width = LocationSize::precise(getOpSize(LdSt, DataOpIdx));
661 return true;
662 }
663
664 return false;
665}
666
667static bool memOpsHaveSameBasePtr(const MachineInstr &MI1,
669 const MachineInstr &MI2,
671 // Only examine the first "base" operand of each instruction, on the
672 // assumption that it represents the real base address of the memory access.
673 // Other operands are typically offsets or indices from this base address.
674 if (BaseOps1.front()->isIdenticalTo(*BaseOps2.front()))
675 return true;
676
677 if (!MI1.hasOneMemOperand() || !MI2.hasOneMemOperand())
678 return false;
679
680 auto *MO1 = *MI1.memoperands_begin();
681 auto *MO2 = *MI2.memoperands_begin();
682 if (MO1->getAddrSpace() != MO2->getAddrSpace())
683 return false;
684
685 const auto *Base1 = MO1->getValue();
686 const auto *Base2 = MO2->getValue();
687 if (!Base1 || !Base2)
688 return false;
689 Base1 = getUnderlyingObject(Base1);
690 Base2 = getUnderlyingObject(Base2);
691
692 if (isa<UndefValue>(Base1) || isa<UndefValue>(Base2))
693 return false;
694
695 return Base1 == Base2;
696}
697
699 int64_t Offset1, bool OffsetIsScalable1,
701 int64_t Offset2, bool OffsetIsScalable2,
702 unsigned ClusterSize,
703 unsigned NumBytes) const {
704 // If the mem ops (to be clustered) do not have the same base ptr, then they
705 // should not be clustered
706 unsigned MaxMemoryClusterDWords = DefaultMemoryClusterDWordsLimit;
707 if (!BaseOps1.empty() && !BaseOps2.empty()) {
708 const MachineInstr &FirstLdSt = *BaseOps1.front()->getParent();
709 const MachineInstr &SecondLdSt = *BaseOps2.front()->getParent();
710 if (!memOpsHaveSameBasePtr(FirstLdSt, BaseOps1, SecondLdSt, BaseOps2))
711 return false;
712
713 const SIMachineFunctionInfo *MFI =
714 FirstLdSt.getMF()->getInfo<SIMachineFunctionInfo>();
715 MaxMemoryClusterDWords = MFI->getMaxMemoryClusterDWords();
716 } else if (!BaseOps1.empty() || !BaseOps2.empty()) {
717 // If only one base op is empty, they do not have the same base ptr
718 return false;
719 }
720
721 // In order to avoid register pressure, on an average, the number of DWORDS
722 // loaded together by all clustered mem ops should not exceed
723 // MaxMemoryClusterDWords. This is an empirical value based on certain
724 // observations and performance related experiments.
725 // The good thing about this heuristic is - it avoids clustering of too many
726 // sub-word loads, and also avoids clustering of wide loads. Below is the
727 // brief summary of how the heuristic behaves for various `LoadSize` when
728 // MaxMemoryClusterDWords is 8.
729 //
730 // (1) 1 <= LoadSize <= 4: cluster at max 8 mem ops
731 // (2) 5 <= LoadSize <= 8: cluster at max 4 mem ops
732 // (3) 9 <= LoadSize <= 12: cluster at max 2 mem ops
733 // (4) 13 <= LoadSize <= 16: cluster at max 2 mem ops
734 // (5) LoadSize >= 17: do not cluster
735 const unsigned LoadSize = NumBytes / ClusterSize;
736 const unsigned NumDWords = ((LoadSize + 3) / 4) * ClusterSize;
737 return NumDWords <= MaxMemoryClusterDWords;
738}
739
740// FIXME: This behaves strangely. If, for example, you have 32 load + stores,
741// the first 16 loads will be interleaved with the stores, and the next 16 will
742// be clustered as expected. It should really split into 2 16 store batches.
743//
744// Loads are clustered until this returns false, rather than trying to schedule
745// groups of stores. This also means we have to deal with saying different
746// address space loads should be clustered, and ones which might cause bank
747// conflicts.
748//
749// This might be deprecated so it might not be worth that much effort to fix.
751 int64_t Offset0, int64_t Offset1,
752 unsigned NumLoads) const {
753 assert(Offset1 > Offset0 &&
754 "Second offset should be larger than first offset!");
755 // If we have less than 16 loads in a row, and the offsets are within 64
756 // bytes, then schedule together.
757
758 // A cacheline is 64 bytes (for global memory).
759 return (NumLoads <= 16 && (Offset1 - Offset0) < 64);
760}
761
764 const DebugLoc &DL, MCRegister DestReg,
765 MCRegister SrcReg, bool KillSrc,
766 const char *Msg = "illegal VGPR to SGPR copy") {
767 MachineFunction *MF = MBB.getParent();
768
771
772 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_ILLEGAL_COPY), DestReg)
773 .addReg(SrcReg, getKillRegState(KillSrc));
774}
775
776/// Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908. It is not
777/// possible to have a direct copy in these cases on GFX908, so an intermediate
778/// VGPR copy is required.
781 const DebugLoc &DL, MCRegister DestReg,
782 MCRegister SrcReg, bool KillSrc,
783 RegScavenger &RS, bool RegsOverlap,
784 Register ImpUseSuperReg = Register()) {
785 assert((TII.getSubtarget().hasMAIInsts() &&
786 !TII.getSubtarget().hasGFX90AInsts()) &&
787 "Expected GFX908 subtarget.");
788
789 assert((AMDGPU::SReg_32RegClass.contains(SrcReg) ||
790 AMDGPU::AGPR_32RegClass.contains(SrcReg)) &&
791 "Source register of the copy should be either an SGPR or an AGPR.");
792
793 assert(AMDGPU::AGPR_32RegClass.contains(DestReg) &&
794 "Destination register of the copy should be an AGPR.");
795
796 const SIRegisterInfo &RI = TII.getRegisterInfo();
797
798 // First try to find defining accvgpr_write to avoid temporary registers.
799 // In the case of copies of overlapping AGPRs, we conservatively do not
800 // reuse previous accvgpr_writes. Otherwise, we may incorrectly pick up
801 // an accvgpr_write used for this same copy due to implicit-defs
802 if (!RegsOverlap) {
803 for (auto Def = MI, E = MBB.begin(); Def != E; ) {
804 --Def;
805
806 if (!Def->modifiesRegister(SrcReg, &RI))
807 continue;
808
809 if (Def->getOpcode() != AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
810 Def->getOperand(0).getReg() != SrcReg)
811 break;
812
813 MachineOperand &DefOp = Def->getOperand(1);
814 assert(DefOp.isReg() || DefOp.isImm());
815
816 if (DefOp.isReg()) {
817 bool SafeToPropagate = true;
818 // Check that register source operand is not clobbered before MI.
819 // Immediate operands are always safe to propagate.
820 for (auto I = Def; I != MI && SafeToPropagate; ++I)
821 if (I->modifiesRegister(DefOp.getReg(), &RI))
822 SafeToPropagate = false;
823
824 if (!SafeToPropagate)
825 break;
826
827 for (auto I = Def; I != MI; ++I)
828 I->clearRegisterKills(DefOp.getReg(), &RI);
829 }
830
831 MachineInstrBuilder Builder =
832 BuildMI(MBB, MI, DL, TII.get(AMDGPU::V_ACCVGPR_WRITE_B32_e64),
833 DestReg)
834 .add(DefOp);
835
836 if (ImpUseSuperReg) {
837 Builder.addReg(ImpUseSuperReg,
839 }
840
841 return;
842 }
843 }
844
845 RS.enterBasicBlockEnd(MBB);
846 RS.backward(std::next(MI));
847
848 // Ideally we want to have three registers for a long reg_sequence copy
849 // to hide 2 waitstates between v_mov_b32 and accvgpr_write.
850 unsigned MaxVGPRs = RI.getRegPressureLimit(&AMDGPU::VGPR_32RegClass,
851 *MBB.getParent());
852
853 // Registers in the sequence are allocated contiguously so we can just
854 // use register number to pick one of three round-robin temps.
855 unsigned RegNo = (DestReg - AMDGPU::AGPR0) % 3;
856 Register Tmp =
857 MBB.getParent()->getInfo<SIMachineFunctionInfo>()->getVGPRForAGPRCopy();
858 assert(MBB.getParent()->getRegInfo().isReserved(Tmp) &&
859 "VGPR used for an intermediate copy should have been reserved.");
860
861 // Only loop through if there are any free registers left. We don't want to
862 // spill.
863 while (RegNo--) {
864 Register Tmp2 = RS.scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass, MI,
865 /* RestoreAfter */ false, 0,
866 /* AllowSpill */ false);
867 if (!Tmp2 || RI.getHWRegIndex(Tmp2) >= MaxVGPRs)
868 break;
869 Tmp = Tmp2;
870 RS.setRegUsed(Tmp);
871 }
872
873 // Insert copy to temporary VGPR.
874 unsigned TmpCopyOp = AMDGPU::V_MOV_B32_e32;
875 if (AMDGPU::AGPR_32RegClass.contains(SrcReg)) {
876 TmpCopyOp = AMDGPU::V_ACCVGPR_READ_B32_e64;
877 } else {
878 assert(AMDGPU::SReg_32RegClass.contains(SrcReg));
879 }
880
881 MachineInstrBuilder UseBuilder = BuildMI(MBB, MI, DL, TII.get(TmpCopyOp), Tmp)
882 .addReg(SrcReg, getKillRegState(KillSrc));
883 if (ImpUseSuperReg) {
884 UseBuilder.addReg(ImpUseSuperReg,
886 }
887
888 BuildMI(MBB, MI, DL, TII.get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DestReg)
889 .addReg(Tmp, RegState::Kill);
890}
891
894 MCRegister DestReg, MCRegister SrcReg, bool KillSrc,
895 const TargetRegisterClass *RC, bool Forward) {
896 const SIRegisterInfo &RI = TII.getRegisterInfo();
897 ArrayRef<int16_t> BaseIndices = RI.getRegSplitParts(RC, 4);
899 MachineInstr *FirstMI = nullptr, *LastMI = nullptr;
900
901 for (unsigned Idx = 0; Idx < BaseIndices.size(); ++Idx) {
902 int16_t SubIdx = BaseIndices[Idx];
903 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
904 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
905 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
906 unsigned Opcode = AMDGPU::S_MOV_B32;
907
908 // Is SGPR aligned? If so try to combine with next.
909 bool AlignedDest = ((DestSubReg - AMDGPU::SGPR0) % 2) == 0;
910 bool AlignedSrc = ((SrcSubReg - AMDGPU::SGPR0) % 2) == 0;
911 if (AlignedDest && AlignedSrc && (Idx + 1 < BaseIndices.size())) {
912 // Can use SGPR64 copy
913 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
914 SubIdx = RI.getSubRegFromChannel(Channel, 2);
915 DestSubReg = RI.getSubReg(DestReg, SubIdx);
916 SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
917 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
918 Opcode = AMDGPU::S_MOV_B64;
919 Idx++;
920 }
921
922 LastMI = BuildMI(MBB, I, DL, TII.get(Opcode), DestSubReg)
923 .addReg(SrcSubReg)
924 .addReg(SrcReg, RegState::Implicit);
925
926 if (!FirstMI)
927 FirstMI = LastMI;
928
929 if (!Forward)
930 I--;
931 }
932
933 assert(FirstMI && LastMI);
934 if (!Forward)
935 std::swap(FirstMI, LastMI);
936
937 if (KillSrc)
938 LastMI->addRegisterKilled(SrcReg, &RI);
939}
940
943 const DebugLoc &DL, Register DestReg,
944 Register SrcReg, bool KillSrc, bool RenamableDest,
945 bool RenamableSrc) const {
946 const TargetRegisterClass *RC = RI.getPhysRegBaseClass(DestReg);
947 unsigned Size = RI.getRegSizeInBits(*RC);
948 const TargetRegisterClass *SrcRC = RI.getPhysRegBaseClass(SrcReg);
949 unsigned SrcSize = RI.getRegSizeInBits(*SrcRC);
950
951 // The rest of copyPhysReg assumes Src and Dst size are the same size.
952 // TODO-GFX11_16BIT If all true 16 bit instruction patterns are completed can
953 // we remove Fix16BitCopies and this code block?
954 if (Fix16BitCopies) {
955 if (((Size == 16) != (SrcSize == 16))) {
956 // Non-VGPR Src and Dst will later be expanded back to 32 bits.
957 assert(ST.useRealTrue16Insts());
958 Register &RegToFix = (Size == 32) ? DestReg : SrcReg;
959 MCRegister SubReg = RI.getSubReg(RegToFix, AMDGPU::lo16);
960 RegToFix = SubReg;
961
962 if (DestReg == SrcReg) {
963 // Identity copy. Insert empty bundle since ExpandPostRA expects an
964 // instruction here.
965 BuildMI(MBB, MI, DL, get(AMDGPU::BUNDLE));
966 return;
967 }
968 RC = RI.getPhysRegBaseClass(DestReg);
969 Size = RI.getRegSizeInBits(*RC);
970 SrcRC = RI.getPhysRegBaseClass(SrcReg);
971 SrcSize = RI.getRegSizeInBits(*SrcRC);
972 }
973 }
974
975 if (RC == &AMDGPU::VGPR_32RegClass) {
976 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg) ||
977 AMDGPU::SReg_32RegClass.contains(SrcReg) ||
978 AMDGPU::AGPR_32RegClass.contains(SrcReg));
979 unsigned Opc = AMDGPU::AGPR_32RegClass.contains(SrcReg) ?
980 AMDGPU::V_ACCVGPR_READ_B32_e64 : AMDGPU::V_MOV_B32_e32;
981 BuildMI(MBB, MI, DL, get(Opc), DestReg)
982 .addReg(SrcReg, getKillRegState(KillSrc));
983 return;
984 }
985
986 if (RC == &AMDGPU::SReg_32_XM0RegClass ||
987 RC == &AMDGPU::SReg_32RegClass) {
988 if (SrcReg == AMDGPU::SCC) {
989 BuildMI(MBB, MI, DL, get(AMDGPU::S_CSELECT_B32), DestReg)
990 .addImm(1)
991 .addImm(0);
992 return;
993 }
994
995 if (!AMDGPU::SReg_32RegClass.contains(SrcReg)) {
996 if (DestReg == AMDGPU::VCC_LO) {
997 // FIXME: Hack until VReg_1 removed.
998 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg));
999 BuildMI(MBB, MI, DL, get(AMDGPU::V_CMP_NE_U32_e32))
1000 .addImm(0)
1001 .addReg(SrcReg, getKillRegState(KillSrc));
1002 return;
1003 }
1004
1005 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1006 return;
1007 }
1008
1009 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DestReg)
1010 .addReg(SrcReg, getKillRegState(KillSrc));
1011 return;
1012 }
1013
1014 if (RC == &AMDGPU::SReg_64RegClass) {
1015 if (SrcReg == AMDGPU::SCC) {
1016 BuildMI(MBB, MI, DL, get(AMDGPU::S_CSELECT_B64), DestReg)
1017 .addImm(1)
1018 .addImm(0);
1019 return;
1020 }
1021
1022 if (!AMDGPU::SReg_64_EncodableRegClass.contains(SrcReg)) {
1023 if (DestReg == AMDGPU::VCC) {
1024 // FIXME: Hack until VReg_1 removed.
1025 assert(AMDGPU::VGPR_32RegClass.contains(SrcReg));
1026 BuildMI(MBB, MI, DL, get(AMDGPU::V_CMP_NE_U32_e32))
1027 .addImm(0)
1028 .addReg(SrcReg, getKillRegState(KillSrc));
1029 return;
1030 }
1031
1032 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1033 return;
1034 }
1035
1036 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B64), DestReg)
1037 .addReg(SrcReg, getKillRegState(KillSrc));
1038 return;
1039 }
1040
1041 if (DestReg == AMDGPU::SCC) {
1042 // Copying 64-bit or 32-bit sources to SCC barely makes sense,
1043 // but SelectionDAG emits such copies for i1 sources.
1044 if (AMDGPU::SReg_64RegClass.contains(SrcReg)) {
1045 // This copy can only be produced by patterns
1046 // with explicit SCC, which are known to be enabled
1047 // only for subtargets with S_CMP_LG_U64 present.
1048 assert(ST.hasScalarCompareEq64());
1049 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U64))
1050 .addReg(SrcReg, getKillRegState(KillSrc))
1051 .addImm(0);
1052 } else {
1053 assert(AMDGPU::SReg_32RegClass.contains(SrcReg));
1054 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U32))
1055 .addReg(SrcReg, getKillRegState(KillSrc))
1056 .addImm(0);
1057 }
1058
1059 return;
1060 }
1061
1062 if (RC == &AMDGPU::AGPR_32RegClass) {
1063 if (AMDGPU::VGPR_32RegClass.contains(SrcReg) ||
1064 (ST.hasGFX90AInsts() && AMDGPU::SReg_32RegClass.contains(SrcReg))) {
1065 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DestReg)
1066 .addReg(SrcReg, getKillRegState(KillSrc));
1067 return;
1068 }
1069
1070 if (AMDGPU::AGPR_32RegClass.contains(SrcReg) && ST.hasGFX90AInsts()) {
1071 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_MOV_B32), DestReg)
1072 .addReg(SrcReg, getKillRegState(KillSrc));
1073 return;
1074 }
1075
1076 // FIXME: Pass should maintain scavenger to avoid scan through the block on
1077 // every AGPR spill.
1078 RegScavenger RS;
1079 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1080 indirectCopyToAGPR(*this, MBB, MI, DL, DestReg, SrcReg, KillSrc, RS, Overlap);
1081 return;
1082 }
1083
1084 if (Size == 16) {
1085 assert(AMDGPU::VGPR_16RegClass.contains(SrcReg) ||
1086 AMDGPU::SReg_LO16RegClass.contains(SrcReg) ||
1087 AMDGPU::AGPR_LO16RegClass.contains(SrcReg));
1088
1089 bool IsSGPRDst = AMDGPU::SReg_LO16RegClass.contains(DestReg);
1090 bool IsSGPRSrc = AMDGPU::SReg_LO16RegClass.contains(SrcReg);
1091 bool IsAGPRDst = AMDGPU::AGPR_LO16RegClass.contains(DestReg);
1092 bool IsAGPRSrc = AMDGPU::AGPR_LO16RegClass.contains(SrcReg);
1093 bool DstLow = !AMDGPU::isHi16Reg(DestReg, RI);
1094 bool SrcLow = !AMDGPU::isHi16Reg(SrcReg, RI);
1095 MCRegister NewDestReg = RI.get32BitRegister(DestReg);
1096 MCRegister NewSrcReg = RI.get32BitRegister(SrcReg);
1097
1098 if (IsSGPRDst) {
1099 if (!IsSGPRSrc) {
1100 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1101 return;
1102 }
1103
1104 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), NewDestReg)
1105 .addReg(NewSrcReg, getKillRegState(KillSrc));
1106 return;
1107 }
1108
1109 if (IsAGPRDst || IsAGPRSrc) {
1110 if (!DstLow || !SrcLow) {
1111 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc,
1112 "Cannot use hi16 subreg with an AGPR!");
1113 }
1114
1115 copyPhysReg(MBB, MI, DL, NewDestReg, NewSrcReg, KillSrc);
1116 return;
1117 }
1118
1119 if (ST.useRealTrue16Insts()) {
1120 if (IsSGPRSrc) {
1121 assert(SrcLow);
1122 SrcReg = NewSrcReg;
1123 }
1124 // Use the smaller instruction encoding if possible.
1125 if (AMDGPU::VGPR_16_Lo128RegClass.contains(DestReg) &&
1126 (IsSGPRSrc || AMDGPU::VGPR_16_Lo128RegClass.contains(SrcReg))) {
1127 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B16_t16_e32), DestReg)
1128 .addReg(SrcReg);
1129 } else {
1130 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B16_t16_e64), DestReg)
1131 .addImm(0) // src0_modifiers
1132 .addReg(SrcReg)
1133 .addImm(0); // op_sel
1134 }
1135 return;
1136 }
1137
1138 if (IsSGPRSrc && !ST.hasSDWAScalar()) {
1139 if (!DstLow || !SrcLow) {
1140 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc,
1141 "Cannot use hi16 subreg on VI!");
1142 }
1143
1144 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), NewDestReg)
1145 .addReg(NewSrcReg, getKillRegState(KillSrc));
1146 return;
1147 }
1148
1149 auto MIB = BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_sdwa), NewDestReg)
1150 .addImm(0) // src0_modifiers
1151 .addReg(NewSrcReg)
1152 .addImm(0) // clamp
1159 // First implicit operand is $exec.
1160 MIB->tieOperands(0, MIB->getNumOperands() - 1);
1161 return;
1162 }
1163
1164 // Returns true if Dst and Src are in Opc's HwMode-resolved destination and
1165 // source operand classes.
1166 auto CanCopyWith = [&](unsigned Opc, MCRegister Dst, MCRegister Src,
1167 unsigned SrcOp = 1) {
1168 const MCInstrDesc &Desc = get(Opc);
1169 const TargetRegisterClass *DstOpRC = getRegClass(Desc, 0);
1170 const TargetRegisterClass *SrcOpRC = getRegClass(Desc, SrcOp);
1171 return DstOpRC && SrcOpRC && DstOpRC->contains(Dst) &&
1172 SrcOpRC->contains(Src);
1173 };
1174
1175 if (RC == RI.getVGPR64Class() && (SrcRC == RC || RI.isSGPRClass(SrcRC))) {
1176 if (ST.hasVMovB64Inst() &&
1177 CanCopyWith(AMDGPU::V_MOV_B64_e32, DestReg, SrcReg)) {
1178 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B64_e32), DestReg)
1179 .addReg(SrcReg, getKillRegState(KillSrc));
1180 return;
1181 }
1182 if (ST.hasPkMovB32() &&
1183 CanCopyWith(AMDGPU::V_PK_MOV_B32, DestReg, SrcReg, /*SrcOp=*/2)) {
1184 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), DestReg)
1186 .addReg(SrcReg)
1188 .addReg(SrcReg)
1189 .addImm(0) // op_sel_lo
1190 .addImm(0) // op_sel_hi
1191 .addImm(0) // neg_lo
1192 .addImm(0) // neg_hi
1193 .addImm(0) // clamp
1194 .addReg(SrcReg, getKillRegState(KillSrc) | RegState::Implicit);
1195 return;
1196 }
1197 }
1198
1199 const bool Forward = RI.getHWRegIndex(DestReg) <= RI.getHWRegIndex(SrcReg);
1200 if (RI.isSGPRClass(RC)) {
1201 if (!RI.isSGPRClass(SrcRC)) {
1202 reportIllegalCopy(this, MBB, MI, DL, DestReg, SrcReg, KillSrc);
1203 return;
1204 }
1205 const bool CanKillSuperReg = KillSrc && !RI.regsOverlap(SrcReg, DestReg);
1206 expandSGPRCopy(*this, MBB, MI, DL, DestReg, SrcReg, CanKillSuperReg, RC,
1207 Forward);
1208 return;
1209 }
1210
1211 unsigned Opcode = AMDGPU::V_MOV_B32_e32;
1212 unsigned WideOpcode = AMDGPU::INSTRUCTION_LIST_END;
1213 if (RI.isAGPRClass(RC)) {
1214 if (ST.hasGFX90AInsts() && RI.isAGPRClass(SrcRC))
1215 Opcode = AMDGPU::V_ACCVGPR_MOV_B32;
1216 else if (RI.hasVGPRs(SrcRC) ||
1217 (ST.hasGFX90AInsts() && RI.isSGPRClass(SrcRC)))
1218 Opcode = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
1219 else
1220 Opcode = AMDGPU::INSTRUCTION_LIST_END;
1221 } else if (RI.hasVGPRs(RC) && RI.isAGPRClass(SrcRC)) {
1222 Opcode = AMDGPU::V_ACCVGPR_READ_B32_e64;
1223 } else if (RI.isVGPRClass(RC)) {
1224 if (ST.hasVMovB64Inst())
1225 WideOpcode = AMDGPU::V_MOV_B64_e32;
1226 else if (ST.hasPkMovB32())
1227 WideOpcode = AMDGPU::V_PK_MOV_B32;
1228 }
1229
1230 const TargetRegisterClass *WideDstRC{}, *WideSrcRC{};
1231 if (WideOpcode != AMDGPU::INSTRUCTION_LIST_END) {
1232 const MCInstrDesc &Desc = get(WideOpcode);
1233 unsigned SrcOp = WideOpcode == AMDGPU::V_PK_MOV_B32 ? 2 : 1;
1234 WideDstRC = getRegClass(Desc, 0);
1235 WideSrcRC = getRegClass(Desc, SrcOp);
1236 }
1237
1238 // If there is an overlap, we can't kill the super-register on the last
1239 // instruction, since it will also kill the components made live by this def.
1240 const bool Overlap = RI.regsOverlap(SrcReg, DestReg);
1241 const bool CanKillSuperReg = KillSrc && !Overlap;
1242
1243 // For the cases where we need an intermediate instruction/temporary register
1244 // (destination is an AGPR), we need a scavenger.
1245 //
1246 // FIXME: The pass should maintain this for us so we don't have to re-scan the
1247 // whole block for every handled copy.
1248 std::unique_ptr<RegScavenger> RS;
1249 if (Opcode == AMDGPU::INSTRUCTION_LIST_END)
1250 RS = std::make_unique<RegScavenger>();
1251
1252 ArrayRef<int16_t> SubIndices = RI.getRegSplitParts(RC, 4);
1253
1254 for (unsigned Idx{}; Idx < SubIndices.size();) {
1255 unsigned NumRegs = 1;
1256 unsigned ThisOpcode = Opcode;
1257 unsigned SubIdx =
1258 Forward ? SubIndices[Idx] : SubIndices[SubIndices.size() - Idx - 1];
1259
1260 if (WideDstRC && WideSrcRC && Idx + 1 < SubIndices.size()) {
1261 unsigned Channel = RI.getChannelFromSubReg(SubIdx);
1262 if (!Forward)
1263 --Channel;
1264
1265 unsigned WideSubIdx = RI.getSubRegFromChannel(Channel, 2);
1266 Register WideDst = RI.getSubReg(DestReg, WideSubIdx);
1267 Register WideSrc = RI.getSubReg(SrcReg, WideSubIdx);
1268
1269 if (WideDst && WideSrc && WideDstRC->contains(WideDst) &&
1270 WideSrcRC->contains(WideSrc)) {
1271 SubIdx = WideSubIdx;
1272 NumRegs = 2;
1273 ThisOpcode = WideOpcode;
1274 }
1275 }
1276
1277 Register DestSubReg = RI.getSubReg(DestReg, SubIdx);
1278 Register SrcSubReg = RI.getSubReg(SrcReg, SubIdx);
1279 assert(DestSubReg && SrcSubReg && "Failed to find subregs!");
1280
1281 Idx += NumRegs;
1282 bool UseKill = CanKillSuperReg && Idx == SubIndices.size();
1283
1284 if (ThisOpcode == AMDGPU::INSTRUCTION_LIST_END) {
1285 Register ImpUseSuper = SrcReg;
1286 indirectCopyToAGPR(*this, MBB, MI, DL, DestSubReg, SrcSubReg, UseKill,
1287 *RS, Overlap, ImpUseSuper);
1288 } else if (ThisOpcode == AMDGPU::V_PK_MOV_B32) {
1289 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), DestSubReg)
1291 .addReg(SrcSubReg)
1293 .addReg(SrcSubReg)
1294 .addImm(0) // op_sel_lo
1295 .addImm(0) // op_sel_hi
1296 .addImm(0) // neg_lo
1297 .addImm(0) // neg_hi
1298 .addImm(0) // clamp
1299 .addReg(SrcReg, getKillRegState(UseKill) | RegState::Implicit);
1300 } else {
1301 MachineInstrBuilder Builder =
1302 BuildMI(MBB, MI, DL, get(ThisOpcode), DestSubReg).addReg(SrcSubReg);
1303
1304 Builder.addReg(SrcReg, getKillRegState(UseKill) | RegState::Implicit);
1305 }
1306 }
1307}
1308
1309int SIInstrInfo::commuteOpcode(unsigned Opcode) const {
1310 int32_t NewOpc;
1311
1312 // Try to map original to commuted opcode
1313 NewOpc = AMDGPU::getCommuteRev(Opcode);
1314 if (NewOpc != -1)
1315 // Check if the commuted (REV) opcode exists on the target.
1316 return pseudoToMCOpcode(NewOpc) != -1 ? NewOpc : -1;
1317
1318 // Try to map commuted to original opcode
1319 NewOpc = AMDGPU::getCommuteOrig(Opcode);
1320 if (NewOpc != -1)
1321 // Check if the original (non-REV) opcode exists on the target.
1322 return pseudoToMCOpcode(NewOpc) != -1 ? NewOpc : -1;
1323
1324 return Opcode;
1325}
1326
1328 const Register Reg,
1329 int64_t &ImmVal) const {
1330 switch (MI.getOpcode()) {
1331 case AMDGPU::V_MOV_B32_e32:
1332 case AMDGPU::S_MOV_B32:
1333 case AMDGPU::S_MOVK_I32:
1334 case AMDGPU::S_MOV_B64:
1335 case AMDGPU::V_MOV_B64_e32:
1336 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
1337 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
1338 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
1339 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
1340 case AMDGPU::V_MOV_B64_PSEUDO:
1341 case AMDGPU::V_MOV_B16_t16_e32: {
1342 const MachineOperand &Src0 = MI.getOperand(1);
1343 if (Src0.isImm()) {
1344 ImmVal = Src0.getImm();
1345 return MI.getOperand(0).getReg() == Reg;
1346 }
1347
1348 return false;
1349 }
1350 case AMDGPU::V_MOV_B16_t16_e64: {
1351 const MachineOperand &Src0 = MI.getOperand(2);
1352 if (Src0.isImm() && !MI.getOperand(1).getImm()) {
1353 ImmVal = Src0.getImm();
1354 return MI.getOperand(0).getReg() == Reg;
1355 }
1356
1357 return false;
1358 }
1359 case AMDGPU::S_BREV_B32:
1360 case AMDGPU::V_BFREV_B32_e32:
1361 case AMDGPU::V_BFREV_B32_e64: {
1362 const MachineOperand &Src0 = MI.getOperand(1);
1363 if (Src0.isImm()) {
1364 ImmVal = static_cast<int64_t>(reverseBits<int32_t>(Src0.getImm()));
1365 return MI.getOperand(0).getReg() == Reg;
1366 }
1367
1368 return false;
1369 }
1370 case AMDGPU::S_NOT_B32:
1371 case AMDGPU::V_NOT_B32_e32:
1372 case AMDGPU::V_NOT_B32_e64: {
1373 const MachineOperand &Src0 = MI.getOperand(1);
1374 if (Src0.isImm()) {
1375 ImmVal = static_cast<int64_t>(~static_cast<int32_t>(Src0.getImm()));
1376 return MI.getOperand(0).getReg() == Reg;
1377 }
1378
1379 return false;
1380 }
1381 default:
1382 return false;
1383 }
1384}
1385
1386std::optional<int64_t>
1388 const MachineOperand &Op,
1389 MachineInstr **DefMI) const {
1390 if (DefMI)
1391 *DefMI = nullptr;
1392
1393 if (Op.isImm())
1394 return Op.getImm();
1395
1396 if (!Op.isReg() || !Op.getReg().isVirtual())
1397 return std::nullopt;
1398 MachineInstr *Def = MRI.getUniqueVRegDef(Op.getReg());
1399 if (Def && Def->isMoveImmediate()) {
1400 const MachineOperand &ImmSrc = Def->getOperand(1);
1401 if (ImmSrc.isImm()) {
1402 if (DefMI)
1403 *DefMI = Def;
1404 return extractSubregFromImm(ImmSrc.getImm(), Op.getSubReg());
1405 }
1406 }
1407
1408 return std::nullopt;
1409}
1410
1411std::optional<int64_t>
1417
1419
1420 if (RI.isAGPRClass(DstRC))
1421 return AMDGPU::COPY;
1422 if (RI.getRegSizeInBits(*DstRC) == 16) {
1423 // Assume hi bits are unneeded. Only _e64 true16 instructions are legal
1424 // before RA.
1425 return RI.isSGPRClass(DstRC) ? AMDGPU::COPY : AMDGPU::V_MOV_B16_t16_e64;
1426 }
1427 if (RI.getRegSizeInBits(*DstRC) == 32)
1428 return RI.isSGPRClass(DstRC) ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1429 if (RI.getRegSizeInBits(*DstRC) == 64 && RI.isSGPRClass(DstRC))
1430 return AMDGPU::S_MOV_B64;
1431 if (RI.getRegSizeInBits(*DstRC) == 64 && !RI.isSGPRClass(DstRC))
1432 return AMDGPU::V_MOV_B64_PSEUDO;
1433 return AMDGPU::COPY;
1434}
1435
1436const MCInstrDesc &
1438 bool IsIndirectSrc) const {
1439 if (IsIndirectSrc) {
1440 if (VecSize <= 32) // 4 bytes
1441 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1);
1442 if (VecSize <= 64) // 8 bytes
1443 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2);
1444 if (VecSize <= 96) // 12 bytes
1445 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3);
1446 if (VecSize <= 128) // 16 bytes
1447 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4);
1448 if (VecSize <= 160) // 20 bytes
1449 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5);
1450 if (VecSize <= 192) // 24 bytes
1451 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6);
1452 if (VecSize <= 224) // 28 bytes
1453 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7);
1454 if (VecSize <= 256) // 32 bytes
1455 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8);
1456 if (VecSize <= 288) // 36 bytes
1457 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9);
1458 if (VecSize <= 320) // 40 bytes
1459 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10);
1460 if (VecSize <= 352) // 44 bytes
1461 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11);
1462 if (VecSize <= 384) // 48 bytes
1463 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12);
1464 if (VecSize <= 512) // 64 bytes
1465 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16);
1466 if (VecSize <= 1024) // 128 bytes
1467 return get(AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32);
1468
1469 llvm_unreachable("unsupported size for IndirectRegReadGPRIDX pseudos");
1470 }
1471
1472 if (VecSize <= 32) // 4 bytes
1473 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1);
1474 if (VecSize <= 64) // 8 bytes
1475 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2);
1476 if (VecSize <= 96) // 12 bytes
1477 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3);
1478 if (VecSize <= 128) // 16 bytes
1479 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4);
1480 if (VecSize <= 160) // 20 bytes
1481 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5);
1482 if (VecSize <= 192) // 24 bytes
1483 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6);
1484 if (VecSize <= 224) // 28 bytes
1485 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7);
1486 if (VecSize <= 256) // 32 bytes
1487 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8);
1488 if (VecSize <= 288) // 36 bytes
1489 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9);
1490 if (VecSize <= 320) // 40 bytes
1491 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10);
1492 if (VecSize <= 352) // 44 bytes
1493 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11);
1494 if (VecSize <= 384) // 48 bytes
1495 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12);
1496 if (VecSize <= 512) // 64 bytes
1497 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16);
1498 if (VecSize <= 1024) // 128 bytes
1499 return get(AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32);
1500
1501 llvm_unreachable("unsupported size for IndirectRegWriteGPRIDX pseudos");
1502}
1503
1504static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize) {
1505 if (VecSize <= 32) // 4 bytes
1506 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1507 if (VecSize <= 64) // 8 bytes
1508 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1509 if (VecSize <= 96) // 12 bytes
1510 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1511 if (VecSize <= 128) // 16 bytes
1512 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1513 if (VecSize <= 160) // 20 bytes
1514 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1515 if (VecSize <= 192) // 24 bytes
1516 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1517 if (VecSize <= 224) // 28 bytes
1518 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1519 if (VecSize <= 256) // 32 bytes
1520 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1521 if (VecSize <= 288) // 36 bytes
1522 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1523 if (VecSize <= 320) // 40 bytes
1524 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1525 if (VecSize <= 352) // 44 bytes
1526 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1527 if (VecSize <= 384) // 48 bytes
1528 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1529 if (VecSize <= 512) // 64 bytes
1530 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1531 if (VecSize <= 1024) // 128 bytes
1532 return AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1533
1534 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1535}
1536
1537static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize) {
1538 if (VecSize <= 32) // 4 bytes
1539 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1;
1540 if (VecSize <= 64) // 8 bytes
1541 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2;
1542 if (VecSize <= 96) // 12 bytes
1543 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3;
1544 if (VecSize <= 128) // 16 bytes
1545 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4;
1546 if (VecSize <= 160) // 20 bytes
1547 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5;
1548 if (VecSize <= 192) // 24 bytes
1549 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6;
1550 if (VecSize <= 224) // 28 bytes
1551 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7;
1552 if (VecSize <= 256) // 32 bytes
1553 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8;
1554 if (VecSize <= 288) // 36 bytes
1555 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9;
1556 if (VecSize <= 320) // 40 bytes
1557 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10;
1558 if (VecSize <= 352) // 44 bytes
1559 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11;
1560 if (VecSize <= 384) // 48 bytes
1561 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12;
1562 if (VecSize <= 512) // 64 bytes
1563 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16;
1564 if (VecSize <= 1024) // 128 bytes
1565 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32;
1566
1567 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1568}
1569
1570static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize) {
1571 if (VecSize <= 64) // 8 bytes
1572 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1;
1573 if (VecSize <= 128) // 16 bytes
1574 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2;
1575 if (VecSize <= 256) // 32 bytes
1576 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4;
1577 if (VecSize <= 512) // 64 bytes
1578 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8;
1579 if (VecSize <= 1024) // 128 bytes
1580 return AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16;
1581
1582 llvm_unreachable("unsupported size for IndirectRegWrite pseudos");
1583}
1584
1585const MCInstrDesc &
1586SIInstrInfo::getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize,
1587 bool IsSGPR) const {
1588 if (IsSGPR) {
1589 switch (EltSize) {
1590 case 32:
1591 return get(getIndirectSGPRWriteMovRelPseudo32(VecSize));
1592 case 64:
1593 return get(getIndirectSGPRWriteMovRelPseudo64(VecSize));
1594 default:
1595 llvm_unreachable("invalid reg indexing elt size");
1596 }
1597 }
1598
1599 assert(EltSize == 32 && "invalid reg indexing elt size");
1601}
1602
1603static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1604 switch (Size) {
1605 case 4:
1606 return NeedsCFI ? AMDGPU::SI_SPILL_S32_CFI_SAVE : AMDGPU::SI_SPILL_S32_SAVE;
1607 case 8:
1608 return NeedsCFI ? AMDGPU::SI_SPILL_S64_CFI_SAVE : AMDGPU::SI_SPILL_S64_SAVE;
1609 case 12:
1610 return NeedsCFI ? AMDGPU::SI_SPILL_S96_CFI_SAVE : AMDGPU::SI_SPILL_S96_SAVE;
1611 case 16:
1612 return NeedsCFI ? AMDGPU::SI_SPILL_S128_CFI_SAVE
1613 : AMDGPU::SI_SPILL_S128_SAVE;
1614 case 20:
1615 return NeedsCFI ? AMDGPU::SI_SPILL_S160_CFI_SAVE
1616 : AMDGPU::SI_SPILL_S160_SAVE;
1617 case 24:
1618 return NeedsCFI ? AMDGPU::SI_SPILL_S192_CFI_SAVE
1619 : AMDGPU::SI_SPILL_S192_SAVE;
1620 case 28:
1621 return NeedsCFI ? AMDGPU::SI_SPILL_S224_CFI_SAVE
1622 : AMDGPU::SI_SPILL_S224_SAVE;
1623 case 32:
1624 return AMDGPU::SI_SPILL_S256_SAVE;
1625 case 36:
1626 return AMDGPU::SI_SPILL_S288_SAVE;
1627 case 40:
1628 return AMDGPU::SI_SPILL_S320_SAVE;
1629 case 44:
1630 return AMDGPU::SI_SPILL_S352_SAVE;
1631 case 48:
1632 return AMDGPU::SI_SPILL_S384_SAVE;
1633 case 64:
1634 return NeedsCFI ? AMDGPU::SI_SPILL_S512_CFI_SAVE
1635 : AMDGPU::SI_SPILL_S512_SAVE;
1636 case 128:
1637 return NeedsCFI ? AMDGPU::SI_SPILL_S1024_CFI_SAVE
1638 : AMDGPU::SI_SPILL_S1024_SAVE;
1639 default:
1640 llvm_unreachable("unknown register size");
1641 }
1642}
1643
1644static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1645 switch (Size) {
1646 case 2:
1647 return AMDGPU::SI_SPILL_V16_SAVE;
1648 case 4:
1649 return NeedsCFI ? AMDGPU::SI_SPILL_V32_CFI_SAVE : AMDGPU::SI_SPILL_V32_SAVE;
1650 case 8:
1651 return NeedsCFI ? AMDGPU::SI_SPILL_V64_CFI_SAVE : AMDGPU::SI_SPILL_V64_SAVE;
1652 case 12:
1653 return NeedsCFI ? AMDGPU::SI_SPILL_V96_CFI_SAVE : AMDGPU::SI_SPILL_V96_SAVE;
1654 case 16:
1655 return NeedsCFI ? AMDGPU::SI_SPILL_V128_CFI_SAVE
1656 : AMDGPU::SI_SPILL_V128_SAVE;
1657 case 20:
1658 return NeedsCFI ? AMDGPU::SI_SPILL_V160_CFI_SAVE
1659 : AMDGPU::SI_SPILL_V160_SAVE;
1660 case 24:
1661 return NeedsCFI ? AMDGPU::SI_SPILL_V192_CFI_SAVE
1662 : AMDGPU::SI_SPILL_V192_SAVE;
1663 case 28:
1664 return NeedsCFI ? AMDGPU::SI_SPILL_V224_CFI_SAVE
1665 : AMDGPU::SI_SPILL_V224_SAVE;
1666 case 32:
1667 return NeedsCFI ? AMDGPU::SI_SPILL_V256_CFI_SAVE
1668 : AMDGPU::SI_SPILL_V256_SAVE;
1669 case 36:
1670 return NeedsCFI ? AMDGPU::SI_SPILL_V288_CFI_SAVE
1671 : AMDGPU::SI_SPILL_V288_SAVE;
1672 case 40:
1673 return NeedsCFI ? AMDGPU::SI_SPILL_V320_CFI_SAVE
1674 : AMDGPU::SI_SPILL_V320_SAVE;
1675 case 44:
1676 return NeedsCFI ? AMDGPU::SI_SPILL_V352_CFI_SAVE
1677 : AMDGPU::SI_SPILL_V352_SAVE;
1678 case 48:
1679 return NeedsCFI ? AMDGPU::SI_SPILL_V384_CFI_SAVE
1680 : AMDGPU::SI_SPILL_V384_SAVE;
1681 case 64:
1682 return NeedsCFI ? AMDGPU::SI_SPILL_V512_CFI_SAVE
1683 : AMDGPU::SI_SPILL_V512_SAVE;
1684 case 128:
1685 return NeedsCFI ? AMDGPU::SI_SPILL_V1024_CFI_SAVE
1686 : AMDGPU::SI_SPILL_V1024_SAVE;
1687 default:
1688 llvm_unreachable("unknown register size");
1689 }
1690}
1691
1692static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI) {
1693 switch (Size) {
1694 case 4:
1695 return NeedsCFI ? AMDGPU::SI_SPILL_AV32_CFI_SAVE
1696 : AMDGPU::SI_SPILL_AV32_SAVE;
1697 case 8:
1698 return NeedsCFI ? AMDGPU::SI_SPILL_AV64_CFI_SAVE
1699 : AMDGPU::SI_SPILL_AV64_SAVE;
1700 case 12:
1701 return NeedsCFI ? AMDGPU::SI_SPILL_AV96_CFI_SAVE
1702 : AMDGPU::SI_SPILL_AV96_SAVE;
1703 case 16:
1704 return NeedsCFI ? AMDGPU::SI_SPILL_AV128_CFI_SAVE
1705 : AMDGPU::SI_SPILL_AV128_SAVE;
1706 case 20:
1707 return NeedsCFI ? AMDGPU::SI_SPILL_AV160_CFI_SAVE
1708 : AMDGPU::SI_SPILL_AV160_SAVE;
1709 case 24:
1710 return NeedsCFI ? AMDGPU::SI_SPILL_AV192_CFI_SAVE
1711 : AMDGPU::SI_SPILL_AV192_SAVE;
1712 case 28:
1713 return NeedsCFI ? AMDGPU::SI_SPILL_AV224_CFI_SAVE
1714 : AMDGPU::SI_SPILL_AV224_SAVE;
1715 case 32:
1716 return NeedsCFI ? AMDGPU::SI_SPILL_AV256_CFI_SAVE
1717 : AMDGPU::SI_SPILL_AV256_SAVE;
1718 case 36:
1719 return AMDGPU::SI_SPILL_AV288_SAVE;
1720 case 40:
1721 return AMDGPU::SI_SPILL_AV320_SAVE;
1722 case 44:
1723 return AMDGPU::SI_SPILL_AV352_SAVE;
1724 case 48:
1725 return AMDGPU::SI_SPILL_AV384_SAVE;
1726 case 64:
1727 return NeedsCFI ? AMDGPU::SI_SPILL_AV512_CFI_SAVE
1728 : AMDGPU::SI_SPILL_AV512_SAVE;
1729 case 128:
1730 return NeedsCFI ? AMDGPU::SI_SPILL_AV1024_CFI_SAVE
1731 : AMDGPU::SI_SPILL_AV1024_SAVE;
1732 default:
1733 llvm_unreachable("unknown register size");
1734 }
1735}
1736
1737static unsigned getWWMRegSpillSaveOpcode(unsigned Size,
1738 bool IsVectorSuperClass) {
1739 // Currently, there is only 32-bit WWM register spills needed.
1740 if (Size != 4)
1741 llvm_unreachable("unknown wwm register spill size");
1742
1743 if (IsVectorSuperClass)
1744 return AMDGPU::SI_SPILL_WWM_AV32_SAVE;
1745
1746 return AMDGPU::SI_SPILL_WWM_V32_SAVE;
1747}
1748
1750 Register Reg, const TargetRegisterClass *RC, unsigned Size,
1751 const SIMachineFunctionInfo &MFI, bool NeedsCFI) const {
1752 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1753
1754 // Choose the right opcode if spilling a WWM register.
1756 return getWWMRegSpillSaveOpcode(Size, IsVectorSuperClass);
1757
1758 // TODO: Check if AGPRs are available
1759 if (ST.hasMAIInsts())
1760 return getAVSpillSaveOpcode(Size, NeedsCFI);
1761
1762 return getVGPRSpillSaveOpcode(Size, NeedsCFI);
1763}
1764
1765void SIInstrInfo::storeRegToStackSlotImpl(
1767 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
1768 MachineInstr::MIFlag Flags, bool NeedsCFI) const {
1769 MachineFunction *MF = MBB.getParent();
1771 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
1772 const DebugLoc &DL = MBB.findDebugLoc(MI);
1773
1774 MachinePointerInfo PtrInfo
1775 = MachinePointerInfo::getFixedStack(*MF, FrameIndex);
1777 PtrInfo, MachineMemOperand::MOStore, FrameInfo.getObjectSize(FrameIndex),
1778 FrameInfo.getObjectAlign(FrameIndex));
1779 unsigned SpillSize = RI.getSpillSize(*RC);
1780
1781 MachineRegisterInfo &MRI = MF->getRegInfo();
1782 if (RI.isSGPRClass(RC)) {
1783 if (FrameInfo.getStackID(FrameIndex) == TargetStackID::SGPRSpill)
1784 MFI->setHasSpilledSGPRs();
1785 assert(SrcReg != AMDGPU::M0 && "m0 should not be spilled");
1786 assert(SrcReg != AMDGPU::EXEC_LO && SrcReg != AMDGPU::EXEC_HI &&
1787 SrcReg != AMDGPU::EXEC && "exec should not be spilled");
1788
1789 // We are only allowed to create one new instruction when spilling
1790 // registers, so we need to use pseudo instruction for spilling SGPRs.
1791 const MCInstrDesc &OpDesc =
1792 get(getSGPRSpillSaveOpcode(SpillSize, NeedsCFI));
1793
1794 // The SGPR spill/restore instructions only work on number sgprs, so we need
1795 // to make sure we are using the correct register class.
1796 if (SrcReg.isVirtual() && SpillSize == 4) {
1797 MRI.constrainRegClass(SrcReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
1798 }
1799
1800 BuildMI(MBB, MI, DL, OpDesc)
1801 .addReg(SrcReg, getKillRegState(isKill)) // data
1802 .addFrameIndex(FrameIndex) // addr
1803 .addMemOperand(MMO)
1805
1806 return;
1807 }
1808
1809 unsigned Opcode = getVectorRegSpillSaveOpcode(VReg ? VReg : SrcReg, RC,
1810 SpillSize, *MFI, NeedsCFI);
1811 MFI->setHasSpilledVGPRs();
1812
1813 BuildMI(MBB, MI, DL, get(Opcode))
1814 .addReg(SrcReg, getKillRegState(isKill)) // data
1815 .addFrameIndex(FrameIndex) // addr
1816 .addReg(MFI->getStackPtrOffsetReg()) // scratch_offset
1817 .addImm(0) // offset
1818 .addMemOperand(MMO);
1819}
1820
1823 bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
1824 MachineInstr::MIFlag Flags) const {
1825 storeRegToStackSlotImpl(MBB, MI, SrcReg, isKill, FrameIndex, RC, VReg, Flags,
1826 false);
1827}
1828
1831 Register SrcReg, bool isKill,
1832 int FrameIndex,
1833 const TargetRegisterClass *RC) const {
1834 storeRegToStackSlotImpl(MBB, MI, SrcReg, isKill, FrameIndex, RC, Register(),
1835 MachineInstr::NoFlags, true);
1836}
1837
1838static unsigned getSGPRSpillRestoreOpcode(unsigned Size) {
1839 switch (Size) {
1840 case 4:
1841 return AMDGPU::SI_SPILL_S32_RESTORE;
1842 case 8:
1843 return AMDGPU::SI_SPILL_S64_RESTORE;
1844 case 12:
1845 return AMDGPU::SI_SPILL_S96_RESTORE;
1846 case 16:
1847 return AMDGPU::SI_SPILL_S128_RESTORE;
1848 case 20:
1849 return AMDGPU::SI_SPILL_S160_RESTORE;
1850 case 24:
1851 return AMDGPU::SI_SPILL_S192_RESTORE;
1852 case 28:
1853 return AMDGPU::SI_SPILL_S224_RESTORE;
1854 case 32:
1855 return AMDGPU::SI_SPILL_S256_RESTORE;
1856 case 36:
1857 return AMDGPU::SI_SPILL_S288_RESTORE;
1858 case 40:
1859 return AMDGPU::SI_SPILL_S320_RESTORE;
1860 case 44:
1861 return AMDGPU::SI_SPILL_S352_RESTORE;
1862 case 48:
1863 return AMDGPU::SI_SPILL_S384_RESTORE;
1864 case 64:
1865 return AMDGPU::SI_SPILL_S512_RESTORE;
1866 case 128:
1867 return AMDGPU::SI_SPILL_S1024_RESTORE;
1868 default:
1869 llvm_unreachable("unknown register size");
1870 }
1871}
1872
1873static unsigned getVGPRSpillRestoreOpcode(unsigned Size) {
1874 switch (Size) {
1875 case 2:
1876 return AMDGPU::SI_SPILL_V16_RESTORE;
1877 case 4:
1878 return AMDGPU::SI_SPILL_V32_RESTORE;
1879 case 8:
1880 return AMDGPU::SI_SPILL_V64_RESTORE;
1881 case 12:
1882 return AMDGPU::SI_SPILL_V96_RESTORE;
1883 case 16:
1884 return AMDGPU::SI_SPILL_V128_RESTORE;
1885 case 20:
1886 return AMDGPU::SI_SPILL_V160_RESTORE;
1887 case 24:
1888 return AMDGPU::SI_SPILL_V192_RESTORE;
1889 case 28:
1890 return AMDGPU::SI_SPILL_V224_RESTORE;
1891 case 32:
1892 return AMDGPU::SI_SPILL_V256_RESTORE;
1893 case 36:
1894 return AMDGPU::SI_SPILL_V288_RESTORE;
1895 case 40:
1896 return AMDGPU::SI_SPILL_V320_RESTORE;
1897 case 44:
1898 return AMDGPU::SI_SPILL_V352_RESTORE;
1899 case 48:
1900 return AMDGPU::SI_SPILL_V384_RESTORE;
1901 case 64:
1902 return AMDGPU::SI_SPILL_V512_RESTORE;
1903 case 128:
1904 return AMDGPU::SI_SPILL_V1024_RESTORE;
1905 default:
1906 llvm_unreachable("unknown register size");
1907 }
1908}
1909
1910static unsigned getAVSpillRestoreOpcode(unsigned Size) {
1911 switch (Size) {
1912 case 4:
1913 return AMDGPU::SI_SPILL_AV32_RESTORE;
1914 case 8:
1915 return AMDGPU::SI_SPILL_AV64_RESTORE;
1916 case 12:
1917 return AMDGPU::SI_SPILL_AV96_RESTORE;
1918 case 16:
1919 return AMDGPU::SI_SPILL_AV128_RESTORE;
1920 case 20:
1921 return AMDGPU::SI_SPILL_AV160_RESTORE;
1922 case 24:
1923 return AMDGPU::SI_SPILL_AV192_RESTORE;
1924 case 28:
1925 return AMDGPU::SI_SPILL_AV224_RESTORE;
1926 case 32:
1927 return AMDGPU::SI_SPILL_AV256_RESTORE;
1928 case 36:
1929 return AMDGPU::SI_SPILL_AV288_RESTORE;
1930 case 40:
1931 return AMDGPU::SI_SPILL_AV320_RESTORE;
1932 case 44:
1933 return AMDGPU::SI_SPILL_AV352_RESTORE;
1934 case 48:
1935 return AMDGPU::SI_SPILL_AV384_RESTORE;
1936 case 64:
1937 return AMDGPU::SI_SPILL_AV512_RESTORE;
1938 case 128:
1939 return AMDGPU::SI_SPILL_AV1024_RESTORE;
1940 default:
1941 llvm_unreachable("unknown register size");
1942 }
1943}
1944
1945static unsigned getWWMRegSpillRestoreOpcode(unsigned Size,
1946 bool IsVectorSuperClass) {
1947 // Currently, there is only 32-bit WWM register spills needed.
1948 if (Size != 4)
1949 llvm_unreachable("unknown wwm register spill size");
1950
1951 if (IsVectorSuperClass) // TODO: Always use this if there are AGPRs
1952 return AMDGPU::SI_SPILL_WWM_AV32_RESTORE;
1953
1954 return AMDGPU::SI_SPILL_WWM_V32_RESTORE;
1955}
1956
1958 Register Reg, const TargetRegisterClass *RC, unsigned Size,
1959 const SIMachineFunctionInfo &MFI) const {
1960 bool IsVectorSuperClass = RI.isVectorSuperClass(RC);
1961
1962 // Choose the right opcode if restoring a WWM register.
1964 return getWWMRegSpillRestoreOpcode(Size, IsVectorSuperClass);
1965
1966 // TODO: Check if AGPRs are available
1967 if (ST.hasMAIInsts())
1969
1970 assert(!RI.isAGPRClass(RC));
1972}
1973
1976 Register DestReg, int FrameIndex,
1977 const TargetRegisterClass *RC,
1978 Register VReg, unsigned SubReg,
1979 MachineInstr::MIFlag Flags) const {
1980 MachineFunction *MF = MBB.getParent();
1982 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
1983 const DebugLoc &DL = MBB.findDebugLoc(MI);
1984 unsigned SpillSize = RI.getSpillSize(*RC);
1985
1986 MachinePointerInfo PtrInfo
1987 = MachinePointerInfo::getFixedStack(*MF, FrameIndex);
1988
1990 PtrInfo, MachineMemOperand::MOLoad, FrameInfo.getObjectSize(FrameIndex),
1991 FrameInfo.getObjectAlign(FrameIndex));
1992
1993 if (RI.isSGPRClass(RC)) {
1994 if (FrameInfo.getStackID(FrameIndex) == TargetStackID::SGPRSpill)
1995 MFI->setHasSpilledSGPRs();
1996 assert(DestReg != AMDGPU::M0 && "m0 should not be reloaded into");
1997 assert(DestReg != AMDGPU::EXEC_LO && DestReg != AMDGPU::EXEC_HI &&
1998 DestReg != AMDGPU::EXEC && "exec should not be spilled");
1999
2000 // FIXME: Maybe this should not include a memoperand because it will be
2001 // lowered to non-memory instructions.
2002 const MCInstrDesc &OpDesc = get(getSGPRSpillRestoreOpcode(SpillSize));
2003 if (DestReg.isVirtual() && SpillSize == 4) {
2004 MachineRegisterInfo &MRI = MF->getRegInfo();
2005 MRI.constrainRegClass(DestReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
2006 }
2007
2008 BuildMI(MBB, MI, DL, OpDesc, DestReg)
2009 .addFrameIndex(FrameIndex) // addr
2010 .addMemOperand(MMO)
2012
2013 return;
2014 }
2015
2016 unsigned Opcode = getVectorRegSpillRestoreOpcode(VReg ? VReg : DestReg, RC,
2017 SpillSize, *MFI);
2018 BuildMI(MBB, MI, DL, get(Opcode), DestReg)
2019 .addFrameIndex(FrameIndex) // vaddr
2020 .addReg(MFI->getStackPtrOffsetReg()) // scratch_offset
2021 .addImm(0) // offset
2022 .addMemOperand(MMO);
2023}
2024
2029
2032 unsigned Quantity) const {
2033 DebugLoc DL = MBB.findDebugLoc(MI);
2034 unsigned MaxSNopCount = 1u << ST.getSNopBits();
2035 while (Quantity > 0) {
2036 unsigned Arg = std::min(Quantity, MaxSNopCount);
2037 Quantity -= Arg;
2038 BuildMI(MBB, MI, DL, get(AMDGPU::S_NOP)).addImm(Arg - 1);
2039 }
2040}
2041
2045 const DebugLoc &DL) const {
2046 MachineFunction *MF = MBB.getParent();
2047 constexpr unsigned DoorbellIDMask = 0x3ff;
2048 constexpr unsigned ECQueueWaveAbort = 0x400;
2049
2050 MachineBasicBlock *TrapBB = &MBB;
2051 MachineBasicBlock *HaltLoopBB = MF->CreateMachineBasicBlock();
2052
2053 if (!MBB.succ_empty() || std::next(MI.getIterator()) != MBB.end()) {
2054 MBB.splitAt(MI, /*UpdateLiveIns=*/false);
2055 TrapBB = MF->CreateMachineBasicBlock();
2056 BuildMI(MBB, MI, DL, get(AMDGPU::S_CBRANCH_EXECNZ)).addMBB(TrapBB);
2057 MF->push_back(TrapBB);
2058 MBB.addSuccessor(TrapBB);
2059 }
2060 // Start with a `s_trap 2`, if we're in PRIV=1 and we need the workaround this
2061 // will be a nop.
2062 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_TRAP))
2063 .addImm(static_cast<unsigned>(GCNSubtarget::TrapID::LLVMAMDHSATrap));
2064 Register DoorbellReg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2065 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_SENDMSG_RTN_B32),
2066 DoorbellReg)
2068 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::TTMP2)
2069 .addUse(AMDGPU::M0);
2070 Register DoorbellRegMasked =
2071 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2072 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_AND_B32), DoorbellRegMasked)
2073 .addUse(DoorbellReg)
2074 .addImm(DoorbellIDMask)
2075 .setOperandDead(3); // implicit-def $scc
2076 Register SetWaveAbortBit =
2077 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
2078 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_OR_B32), SetWaveAbortBit)
2079 .addUse(DoorbellRegMasked)
2080 .addImm(ECQueueWaveAbort)
2081 .setOperandDead(3); // implicit-def $scc
2082 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2083 .addUse(SetWaveAbortBit);
2084 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_SENDMSG))
2086 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_MOV_B32), AMDGPU::M0)
2087 .addUse(AMDGPU::TTMP2);
2088 BuildMI(*TrapBB, TrapBB->end(), DL, get(AMDGPU::S_BRANCH)).addMBB(HaltLoopBB);
2089 TrapBB->addSuccessor(HaltLoopBB);
2090
2091 BuildMI(*HaltLoopBB, HaltLoopBB->end(), DL, get(AMDGPU::S_SETHALT)).addImm(5);
2092 BuildMI(*HaltLoopBB, HaltLoopBB->end(), DL, get(AMDGPU::S_BRANCH))
2093 .addMBB(HaltLoopBB);
2094 MF->push_back(HaltLoopBB);
2095 HaltLoopBB->addSuccessor(HaltLoopBB);
2096
2097 return MBB.getNextNode();
2098}
2099
2101 switch (MI.getOpcode()) {
2102 default:
2103 if (MI.isMetaInstruction())
2104 return 0;
2105 return 1; // FIXME: Do wait states equal cycles?
2106
2107 case AMDGPU::S_NOP:
2108 return MI.getOperand(0).getImm() + 1;
2109 // SI_RETURN_TO_EPILOG is a fallthrough to code outside of the function. The
2110 // hazard, even if one exist, won't really be visible. Should we handle it?
2111 }
2112}
2113
2115 MachineBasicBlock &MBB = *MI.getParent();
2116 DebugLoc DL = MBB.findDebugLoc(MI);
2118
2119 switch (MI.getOpcode()) {
2120 default: return TargetInstrInfo::expandPostRAPseudo(MI);
2121 case AMDGPU::S_MOV_B64_term:
2122 // This is only a terminator to get the correct spill code placement during
2123 // register allocation.
2124 MI.setDesc(get(AMDGPU::S_MOV_B64));
2125 break;
2126
2127 case AMDGPU::S_MOV_B32_term:
2128 // This is only a terminator to get the correct spill code placement during
2129 // register allocation.
2130 MI.setDesc(get(AMDGPU::S_MOV_B32));
2131 break;
2132
2133 case AMDGPU::S_XOR_B64_term:
2134 // This is only a terminator to get the correct spill code placement during
2135 // register allocation.
2136 MI.setDesc(get(AMDGPU::S_XOR_B64));
2137 break;
2138
2139 case AMDGPU::S_XOR_B32_term:
2140 // This is only a terminator to get the correct spill code placement during
2141 // register allocation.
2142 MI.setDesc(get(AMDGPU::S_XOR_B32));
2143 break;
2144 case AMDGPU::S_OR_B64_term:
2145 // This is only a terminator to get the correct spill code placement during
2146 // register allocation.
2147 MI.setDesc(get(AMDGPU::S_OR_B64));
2148 break;
2149 case AMDGPU::S_OR_B32_term:
2150 // This is only a terminator to get the correct spill code placement during
2151 // register allocation.
2152 MI.setDesc(get(AMDGPU::S_OR_B32));
2153 break;
2154
2155 case AMDGPU::S_ANDN2_B64_term:
2156 // This is only a terminator to get the correct spill code placement during
2157 // register allocation.
2158 MI.setDesc(get(AMDGPU::S_ANDN2_B64));
2159 break;
2160
2161 case AMDGPU::S_ANDN2_B32_term:
2162 // This is only a terminator to get the correct spill code placement during
2163 // register allocation.
2164 MI.setDesc(get(AMDGPU::S_ANDN2_B32));
2165 break;
2166
2167 case AMDGPU::S_AND_B64_term:
2168 // This is only a terminator to get the correct spill code placement during
2169 // register allocation.
2170 MI.setDesc(get(AMDGPU::S_AND_B64));
2171 break;
2172
2173 case AMDGPU::S_AND_B32_term:
2174 // This is only a terminator to get the correct spill code placement during
2175 // register allocation.
2176 MI.setDesc(get(AMDGPU::S_AND_B32));
2177 break;
2178
2179 case AMDGPU::S_AND_SAVEEXEC_B64_term:
2180 // This is only a terminator to get the correct spill code placement during
2181 // register allocation.
2182 MI.setDesc(get(AMDGPU::S_AND_SAVEEXEC_B64));
2183 break;
2184
2185 case AMDGPU::S_AND_SAVEEXEC_B32_term:
2186 // This is only a terminator to get the correct spill code placement during
2187 // register allocation.
2188 MI.setDesc(get(AMDGPU::S_AND_SAVEEXEC_B32));
2189 break;
2190
2191 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
2192 MI.setDesc(get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32));
2193 break;
2194 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
2195 MI.setDesc(get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32));
2196 break;
2197
2198 case AMDGPU::SI_SPILL_S32_TO_VGPR:
2199 MI.setDesc(get(AMDGPU::V_WRITELANE_B32));
2200 break;
2201
2202 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
2203 MI.setDesc(get(AMDGPU::V_READLANE_B32));
2204 break;
2205 case AMDGPU::AV_MOV_B32_IMM_PSEUDO: {
2206 Register Dst = MI.getOperand(0).getReg();
2207 bool IsAGPR = SIRegisterInfo::isAGPRClass(RI.getPhysRegBaseClass(Dst));
2208 MI.setDesc(
2209 get(IsAGPR ? AMDGPU::V_ACCVGPR_WRITE_B32_e64 : AMDGPU::V_MOV_B32_e32));
2210 break;
2211 }
2212 case AMDGPU::AV_MOV_B64_IMM_PSEUDO: {
2213 Register Dst = MI.getOperand(0).getReg();
2214 if (SIRegisterInfo::isAGPRClass(RI.getPhysRegBaseClass(Dst))) {
2215 int64_t Imm = MI.getOperand(1).getImm();
2216
2217 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2218 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2219 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DstLo)
2221 BuildMI(MBB, MI, DL, get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), DstHi)
2222 .addImm(SignExtend64<32>(Imm >> 32));
2223 MI.eraseFromParent();
2224 break;
2225 }
2226
2227 [[fallthrough]];
2228 }
2229 case AMDGPU::V_MOV_B64_PSEUDO: {
2230 Register Dst = MI.getOperand(0).getReg();
2231 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2232 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2233
2234 const MCInstrDesc &Mov64Desc = get(AMDGPU::V_MOV_B64_e32);
2235 const TargetRegisterClass *Mov64RC = getRegClass(Mov64Desc, /*OpNum=*/0);
2236
2237 const MachineOperand &SrcOp = MI.getOperand(1);
2238 // FIXME: Will this work for 64-bit floating point immediates?
2239 assert(!SrcOp.isFPImm());
2240 if (ST.hasVMovB64Inst() && Mov64RC->contains(Dst)) {
2241 MI.setDesc(Mov64Desc);
2242 if (SrcOp.isReg() || isInlineConstant(MI, 1) ||
2243 (SrcOp.isImm() &&
2244 (isUInt<32>(SrcOp.getImm()) || ST.has64BitLiterals())) ||
2245 (SrcOp.isGlobal() && ST.has64BitLiterals()))
2246 break;
2247 }
2248 if (SrcOp.isGlobal()) {
2249 // The address is unknown until link time, so the PK_MOV inline-constant
2250 // shortcut cannot apply.
2251 const GlobalValue *GV = SrcOp.getGlobal();
2252 int64_t Offset = SrcOp.getOffset();
2253 unsigned BaseFlags, LoReloc, HiReloc;
2254 std::tie(BaseFlags, LoReloc, HiReloc) =
2256
2257 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2258 .addGlobalAddress(GV, Offset, BaseFlags | LoReloc);
2259 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2260 .addGlobalAddress(GV, Offset, BaseFlags | HiReloc);
2261 } else if (SrcOp.isImm()) {
2262 APInt Imm(64, SrcOp.getImm());
2263 APInt Lo(32, Imm.getLoBits(32).getZExtValue());
2264 APInt Hi(32, Imm.getHiBits(32).getZExtValue());
2265 const MCInstrDesc &PkMovDesc = get(AMDGPU::V_PK_MOV_B32);
2266 const TargetRegisterClass *PkMovRC = getRegClass(PkMovDesc, /*OpNum=*/0);
2267
2268 if (ST.hasPkMovB32() && Lo == Hi && isInlineConstant(Lo) &&
2269 PkMovRC->contains(Dst)) {
2270 BuildMI(MBB, MI, DL, PkMovDesc, Dst)
2272 .addImm(Lo.getSExtValue())
2274 .addImm(Lo.getSExtValue())
2275 .addImm(0) // op_sel_lo
2276 .addImm(0) // op_sel_hi
2277 .addImm(0) // neg_lo
2278 .addImm(0) // neg_hi
2279 .addImm(0); // clamp
2280 } else {
2281 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2282 .addImm(Lo.getSExtValue());
2283 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2284 .addImm(Hi.getSExtValue());
2285 }
2286 } else {
2287 assert(SrcOp.isReg());
2288 if (ST.hasPkMovB32() &&
2289 !RI.isAGPR(MBB.getParent()->getRegInfo(), SrcOp.getReg())) {
2290 BuildMI(MBB, MI, DL, get(AMDGPU::V_PK_MOV_B32), Dst)
2291 .addImm(SISrcMods::OP_SEL_1) // src0_mod
2292 .addReg(SrcOp.getReg())
2294 .addReg(SrcOp.getReg())
2295 .addImm(0) // op_sel_lo
2296 .addImm(0) // op_sel_hi
2297 .addImm(0) // neg_lo
2298 .addImm(0) // neg_hi
2299 .addImm(0); // clamp
2300 } else {
2301 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstLo)
2302 .addReg(RI.getSubReg(SrcOp.getReg(), AMDGPU::sub0));
2303 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_e32), DstHi)
2304 .addReg(RI.getSubReg(SrcOp.getReg(), AMDGPU::sub1));
2305 }
2306 }
2307 MI.eraseFromParent();
2308 break;
2309 }
2310 case AMDGPU::V_MOV_B64_DPP_PSEUDO: {
2312 break;
2313 }
2314 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2315 const MachineOperand &SrcOp = MI.getOperand(1);
2316 assert(!SrcOp.isFPImm());
2317
2318 if (ST.has64BitLiterals()) {
2319 MI.setDesc(get(AMDGPU::S_MOV_B64));
2320 break;
2321 }
2322
2323 if (SrcOp.isGlobal()) {
2324 Register Dst = MI.getOperand(0).getReg();
2325 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2326 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2327 const GlobalValue *GV = SrcOp.getGlobal();
2328 int64_t Offset = SrcOp.getOffset();
2329 unsigned BaseFlags, LoReloc, HiReloc;
2330 std::tie(BaseFlags, LoReloc, HiReloc) =
2332
2333 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstLo)
2334 .addGlobalAddress(GV, Offset, BaseFlags | LoReloc);
2335 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstHi)
2336 .addGlobalAddress(GV, Offset, BaseFlags | HiReloc);
2337 MI.eraseFromParent();
2338 break;
2339 }
2340
2341 // SrcOp is immediate
2342 APInt Imm(64, SrcOp.getImm());
2343 if (Imm.isIntN(32) || isInlineConstant(Imm)) {
2344 MI.setDesc(get(AMDGPU::S_MOV_B64));
2345 break;
2346 }
2347
2348 Register Dst = MI.getOperand(0).getReg();
2349 Register DstLo = RI.getSubReg(Dst, AMDGPU::sub0);
2350 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2351
2352 APInt Lo(32, Imm.getLoBits(32).getZExtValue());
2353 APInt Hi(32, Imm.getHiBits(32).getZExtValue());
2354 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstLo)
2355 .addImm(Lo.getSExtValue());
2356 BuildMI(MBB, MI, DL, get(AMDGPU::S_MOV_B32), DstHi)
2357 .addImm(Hi.getSExtValue());
2358 MI.eraseFromParent();
2359 break;
2360 }
2361 case AMDGPU::V_SET_INACTIVE_B32: {
2362 // Lower V_SET_INACTIVE_B32 to V_CNDMASK_B32.
2363 Register DstReg = MI.getOperand(0).getReg();
2364 BuildMI(MBB, MI, DL, get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
2365 .add(MI.getOperand(3))
2366 .add(MI.getOperand(4))
2367 .add(MI.getOperand(1))
2368 .add(MI.getOperand(2))
2369 .add(MI.getOperand(5));
2370 MI.eraseFromParent();
2371 break;
2372 }
2373 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2374 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2375 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2376 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2377 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2378 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2379 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2380 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2381 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2382 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2383 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2384 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2385 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2386 case AMDGPU::V_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2387 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V1:
2388 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V2:
2389 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V3:
2390 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V4:
2391 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V5:
2392 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V6:
2393 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V7:
2394 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V8:
2395 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V9:
2396 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V10:
2397 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V11:
2398 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V12:
2399 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V16:
2400 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B32_V32:
2401 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V1:
2402 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V2:
2403 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V4:
2404 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V8:
2405 case AMDGPU::S_INDIRECT_REG_WRITE_MOVREL_B64_V16: {
2406 const TargetRegisterClass *EltRC = getOpRegClass(MI, 2);
2407
2408 unsigned Opc;
2409 if (RI.hasVGPRs(EltRC)) {
2410 Opc = AMDGPU::V_MOVRELD_B32_e32;
2411 } else {
2412 Opc = RI.getRegSizeInBits(*EltRC) == 64 ? AMDGPU::S_MOVRELD_B64
2413 : AMDGPU::S_MOVRELD_B32;
2414 }
2415
2416 const MCInstrDesc &OpDesc = get(Opc);
2417 Register VecReg = MI.getOperand(0).getReg();
2418 bool IsUndef = MI.getOperand(1).isUndef();
2419 unsigned SubReg = MI.getOperand(3).getImm();
2420 assert(VecReg == MI.getOperand(1).getReg());
2421
2423 BuildMI(MBB, MI, DL, OpDesc)
2424 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2425 .add(MI.getOperand(2))
2427 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2428
2429 const int ImpDefIdx =
2430 OpDesc.getNumOperands() + OpDesc.implicit_uses().size();
2431 const int ImpUseIdx = ImpDefIdx + 1;
2432 MIB->tieOperands(ImpDefIdx, ImpUseIdx);
2433 MI.eraseFromParent();
2434 break;
2435 }
2436 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V1:
2437 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V2:
2438 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V3:
2439 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V4:
2440 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V5:
2441 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V6:
2442 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V7:
2443 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V8:
2444 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V9:
2445 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V10:
2446 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V11:
2447 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V12:
2448 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V16:
2449 case AMDGPU::V_INDIRECT_REG_WRITE_GPR_IDX_B32_V32: {
2450 assert(ST.useVGPRIndexMode());
2451 Register VecReg = MI.getOperand(0).getReg();
2452 bool IsUndef = MI.getOperand(1).isUndef();
2453 MachineOperand &Idx = MI.getOperand(3);
2454 Register SubReg = MI.getOperand(4).getImm();
2455
2456 MachineInstr *SetOn = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_ON))
2457 .add(Idx)
2459 SetOn->getOperand(3).setIsUndef();
2460
2461 const MCInstrDesc &OpDesc = get(AMDGPU::V_MOV_B32_indirect_write);
2463 BuildMI(MBB, MI, DL, OpDesc)
2464 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2465 .add(MI.getOperand(2))
2467 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2468
2469 const int ImpDefIdx =
2470 OpDesc.getNumOperands() + OpDesc.implicit_uses().size();
2471 const int ImpUseIdx = ImpDefIdx + 1;
2472 MIB->tieOperands(ImpDefIdx, ImpUseIdx);
2473
2474 MachineInstr *SetOff = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_OFF));
2475
2476 finalizeBundle(MBB, SetOn->getIterator(), std::next(SetOff->getIterator()));
2477
2478 MI.eraseFromParent();
2479 break;
2480 }
2481 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V1:
2482 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V2:
2483 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V3:
2484 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V4:
2485 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V5:
2486 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V6:
2487 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V7:
2488 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V8:
2489 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V9:
2490 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V10:
2491 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V11:
2492 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V12:
2493 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V16:
2494 case AMDGPU::V_INDIRECT_REG_READ_GPR_IDX_B32_V32: {
2495 assert(ST.useVGPRIndexMode());
2496 Register Dst = MI.getOperand(0).getReg();
2497 Register VecReg = MI.getOperand(1).getReg();
2498 bool IsUndef = MI.getOperand(1).isUndef();
2499 Register SubReg = MI.getOperand(3).getImm();
2500
2501 MachineInstr *SetOn = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_ON))
2502 .add(MI.getOperand(2))
2504 SetOn->getOperand(3).setIsUndef();
2505
2506 BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_indirect_read))
2507 .addDef(Dst)
2508 .addReg(RI.getSubReg(VecReg, SubReg), RegState::Undef)
2509 .addReg(VecReg, RegState::Implicit | getUndefRegState(IsUndef));
2510
2511 MachineInstr *SetOff = BuildMI(MBB, MI, DL, get(AMDGPU::S_SET_GPR_IDX_OFF));
2512
2513 finalizeBundle(MBB, SetOn->getIterator(), std::next(SetOff->getIterator()));
2514
2515 MI.eraseFromParent();
2516 break;
2517 }
2518 case AMDGPU::SI_PC_ADD_REL_OFFSET: {
2519 MachineFunction &MF = *MBB.getParent();
2520 Register Reg = MI.getOperand(0).getReg();
2521 Register RegLo = RI.getSubReg(Reg, AMDGPU::sub0);
2522 Register RegHi = RI.getSubReg(Reg, AMDGPU::sub1);
2523 MachineOperand OpLo = MI.getOperand(1);
2524 MachineOperand OpHi = MI.getOperand(2);
2525
2526 // Create a bundle so these instructions won't be re-ordered by the
2527 // post-RA scheduler.
2528 MIBundleBuilder Bundler(MBB, MI);
2529 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_GETPC_B64), Reg));
2530
2531 // What we want here is an offset from the value returned by s_getpc (which
2532 // is the address of the s_add_u32 instruction) to the global variable, but
2533 // since the encoding of $symbol starts 4 bytes after the start of the
2534 // s_add_u32 instruction, we end up with an offset that is 4 bytes too
2535 // small. This requires us to add 4 to the global variable offset in order
2536 // to compute the correct address. Similarly for the s_addc_u32 instruction,
2537 // the encoding of $symbol starts 12 bytes after the start of the s_add_u32
2538 // instruction.
2539
2540 int64_t Adjust = 0;
2541 if (ST.hasGetPCZeroExtension()) {
2542 // Fix up hardware that does not sign-extend the 48-bit PC value by
2543 // inserting: s_sext_i32_i16 reghi, reghi
2544 Bundler.append(
2545 BuildMI(MF, DL, get(AMDGPU::S_SEXT_I32_I16), RegHi).addReg(RegHi));
2546 Adjust += 4;
2547 }
2548
2549 if (OpLo.isGlobal())
2550 OpLo.setOffset(OpLo.getOffset() + Adjust + 4);
2551 Bundler.append(
2552 BuildMI(MF, DL, get(AMDGPU::S_ADD_U32), RegLo).addReg(RegLo).add(OpLo));
2553
2554 if (OpHi.isGlobal())
2555 OpHi.setOffset(OpHi.getOffset() + Adjust + 12);
2556 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_ADDC_U32), RegHi)
2557 .addReg(RegHi)
2558 .add(OpHi));
2559
2560 finalizeBundle(MBB, Bundler.begin());
2561
2562 MI.eraseFromParent();
2563 break;
2564 }
2565 case AMDGPU::SI_PC_ADD_REL_OFFSET64: {
2566 MachineFunction &MF = *MBB.getParent();
2567 Register Reg = MI.getOperand(0).getReg();
2568 MachineOperand Op = MI.getOperand(1);
2569
2570 // Create a bundle so these instructions won't be re-ordered by the
2571 // post-RA scheduler.
2572 MIBundleBuilder Bundler(MBB, MI);
2573 Bundler.append(BuildMI(MF, DL, get(AMDGPU::S_GETPC_B64), Reg));
2574 if (Op.isGlobal())
2575 Op.setOffset(Op.getOffset() + 4);
2576 Bundler.append(
2577 BuildMI(MF, DL, get(AMDGPU::S_ADD_U64), Reg).addReg(Reg).add(Op));
2578
2579 finalizeBundle(MBB, Bundler.begin());
2580
2581 MI.eraseFromParent();
2582 break;
2583 }
2584 case AMDGPU::ENTER_STRICT_WWM: {
2585 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2586 // Whole Wave Mode is entered.
2587 MI.setDesc(get(LMC.OrSaveExecOpc));
2588 break;
2589 }
2590 case AMDGPU::ENTER_STRICT_WQM: {
2591 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2592 // STRICT_WQM is entered.
2593 BuildMI(MBB, MI, DL, get(LMC.MovOpc), MI.getOperand(0).getReg())
2594 .addReg(LMC.ExecReg);
2595 BuildMI(MBB, MI, DL, get(LMC.WQMOpc), LMC.ExecReg).addReg(LMC.ExecReg);
2596
2597 MI.eraseFromParent();
2598 break;
2599 }
2600 case AMDGPU::EXIT_STRICT_WWM:
2601 case AMDGPU::EXIT_STRICT_WQM: {
2602 // This only gets its own opcode so that SIPreAllocateWWMRegs can tell when
2603 // WWM/STICT_WQM is exited.
2604 MI.setDesc(get(LMC.MovOpc));
2605 break;
2606 }
2607 case AMDGPU::SI_RETURN: {
2608 const MachineFunction *MF = MBB.getParent();
2609 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
2610 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2611 // Hiding the return address use with SI_RETURN may lead to extra kills in
2612 // the function and missing live-ins. We are fine in practice because callee
2613 // saved register handling ensures the register value is restored before
2614 // RET, but we need the undef flag here to appease the MachineVerifier
2615 // liveness checks.
2617 BuildMI(MBB, MI, DL, get(AMDGPU::S_SETPC_B64_return))
2618 .addReg(TRI->getReturnAddressReg(*MF), RegState::Undef);
2619
2620 MIB.copyImplicitOps(MI);
2621 MI.eraseFromParent();
2622 break;
2623 }
2624
2625 case AMDGPU::S_MUL_U64_U32_PSEUDO:
2626 case AMDGPU::S_MUL_I64_I32_PSEUDO:
2627 MI.setDesc(get(AMDGPU::S_MUL_U64));
2628 break;
2629
2630 case AMDGPU::S_GETPC_B64_pseudo:
2631 MI.setDesc(get(AMDGPU::S_GETPC_B64));
2632 if (ST.hasGetPCZeroExtension()) {
2633 Register Dst = MI.getOperand(0).getReg();
2634 Register DstHi = RI.getSubReg(Dst, AMDGPU::sub1);
2635 // Fix up hardware that does not sign-extend the 48-bit PC value by
2636 // inserting: s_sext_i32_i16 dsthi, dsthi
2637 BuildMI(MBB, std::next(MI.getIterator()), DL, get(AMDGPU::S_SEXT_I32_I16),
2638 DstHi)
2639 .addReg(DstHi);
2640 }
2641 break;
2642
2643 case AMDGPU::V_MAX_BF16_PSEUDO_e64: {
2644 assert(ST.hasBF16PackedInsts());
2645 MI.setDesc(get(AMDGPU::V_PK_MAX_NUM_BF16));
2646 MI.addOperand(MachineOperand::CreateImm(0)); // op_sel
2647 MI.addOperand(MachineOperand::CreateImm(0)); // neg_lo
2648 MI.addOperand(MachineOperand::CreateImm(0)); // neg_hi
2649 auto Op0 = getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
2650 Op0->setImm(Op0->getImm() | SISrcMods::OP_SEL_1);
2651 auto Op1 = getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
2652 Op1->setImm(Op1->getImm() | SISrcMods::OP_SEL_1);
2653 break;
2654 }
2655
2656 case AMDGPU::GET_STACK_BASE:
2657 // The stack starts at offset 0 unless we need to reserve some space at the
2658 // bottom.
2659 if (ST.getFrameLowering()->mayReserveScratchForCWSR(*MBB.getParent())) {
2660 // When CWSR is used in dynamic VGPR mode, the trap handler needs to save
2661 // some of the VGPRs. The size of the required scratch space has already
2662 // been computed by prolog epilog insertion.
2663 const SIMachineFunctionInfo *MFI =
2664 MBB.getParent()->getInfo<SIMachineFunctionInfo>();
2665 unsigned VGPRSize = MFI->getScratchReservedForDynamicVGPRs();
2666 Register DestReg = MI.getOperand(0).getReg();
2667 BuildMI(MBB, MI, DL, get(AMDGPU::S_GETREG_B32), DestReg)
2670 // The MicroEngine ID is 0 for the graphics queue, and 1 or 2 for compute
2671 // (3 is unused, so we ignore it). Unfortunately, S_GETREG doesn't set
2672 // SCC, so we need to check for 0 manually.
2673 BuildMI(MBB, MI, DL, get(AMDGPU::S_CMP_LG_U32)).addImm(0).addReg(DestReg);
2674 // Change the implicif-def of SCC to an explicit use (but first remove
2675 // the dead flag if present).
2676 MI.getOperand(MI.getNumExplicitOperands()).setIsDead(false);
2677 MI.getOperand(MI.getNumExplicitOperands()).setIsUse();
2678 MI.setDesc(get(AMDGPU::S_CMOVK_I32));
2679 MI.addOperand(MachineOperand::CreateImm(VGPRSize));
2680 } else {
2681 MI.setDesc(get(AMDGPU::S_MOV_B32));
2682 MI.addOperand(MachineOperand::CreateImm(0));
2683 MI.removeOperand(
2684 MI.getNumExplicitOperands()); // Drop implicit def of SCC.
2685 }
2686 break;
2687 }
2688
2689 return true;
2690}
2691
2694 unsigned SubIdx, const MachineInstr &Orig,
2695 LaneBitmask UsedLanes) const {
2696
2697 // Try shrinking the instruction to remat only the part needed for current
2698 // context.
2699 // TODO: Handle more cases.
2700 unsigned Opcode = Orig.getOpcode();
2701 switch (Opcode) {
2702 case AMDGPU::S_MOV_B64:
2703 case AMDGPU::S_MOV_B64_IMM_PSEUDO: {
2704 if (SubIdx != 0)
2705 break;
2706
2707 if (!Orig.getOperand(1).isImm())
2708 break;
2709
2710 // Shrink S_MOV_B64 to S_MOV_B32 when UsedLanes indicates only a single
2711 // 32-bit lane of the 64-bit value is live at the rematerialization point.
2712 if (UsedLanes.all())
2713 break;
2714
2715 // Determine which half of the 64-bit immediate corresponds to the use.
2716 unsigned OrigSubReg = Orig.getOperand(0).getSubReg();
2717 unsigned LoSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub0);
2718 unsigned HiSubReg = RI.composeSubRegIndices(OrigSubReg, AMDGPU::sub1);
2719
2720 bool NeedLo = (UsedLanes & RI.getSubRegIndexLaneMask(LoSubReg)).any();
2721 bool NeedHi = (UsedLanes & RI.getSubRegIndexLaneMask(HiSubReg)).any();
2722
2723 if (NeedLo && NeedHi)
2724 break;
2725
2726 int64_t Imm64 = Orig.getOperand(1).getImm();
2727 int32_t Imm32 = NeedLo ? Lo_32(Imm64) : Hi_32(Imm64);
2728
2729 unsigned UseSubReg = NeedLo ? LoSubReg : HiSubReg;
2730
2731 // Emit S_MOV_B32 defining just the needed 32-bit subreg of DestReg.
2732 BuildMI(MBB, I, Orig.getDebugLoc(), get(AMDGPU::S_MOV_B32))
2733 .addReg(DestReg, RegState::Define | RegState::Undef, UseSubReg)
2734 .addImm(Imm32);
2735 return;
2736 }
2737
2738 case AMDGPU::S_LOAD_DWORDX16_IMM:
2739 case AMDGPU::S_LOAD_DWORDX8_IMM: {
2740 if (SubIdx != 0)
2741 break;
2742
2743 if (I == MBB.end())
2744 break;
2745
2746 if (I->isBundled())
2747 break;
2748
2749 // Look for a single use of the register that is also a subreg.
2750 Register RegToFind = Orig.getOperand(0).getReg();
2751 MachineOperand *UseMO = nullptr;
2752 for (auto &CandMO : I->operands()) {
2753 if (!CandMO.isReg() || CandMO.getReg() != RegToFind || CandMO.isDef())
2754 continue;
2755 if (UseMO) {
2756 UseMO = nullptr;
2757 break;
2758 }
2759 UseMO = &CandMO;
2760 }
2761 if (!UseMO || UseMO->getSubReg() == AMDGPU::NoSubRegister)
2762 break;
2763
2764 unsigned Offset = RI.getSubRegIdxOffset(UseMO->getSubReg());
2765 unsigned SubregSize = RI.getSubRegIdxSize(UseMO->getSubReg());
2766
2767 MachineFunction *MF = MBB.getParent();
2768 MachineRegisterInfo &MRI = MF->getRegInfo();
2769 assert(MRI.use_nodbg_empty(DestReg) && "DestReg should have no users yet.");
2770
2771 unsigned NewOpcode = -1;
2772 if (SubregSize == 256)
2773 NewOpcode = AMDGPU::S_LOAD_DWORDX8_IMM;
2774 else if (SubregSize == 128)
2775 NewOpcode = AMDGPU::S_LOAD_DWORDX4_IMM;
2776 else
2777 break;
2778
2779 const MCInstrDesc &TID = get(NewOpcode);
2780 const TargetRegisterClass *NewRC =
2781 RI.getAllocatableClass(getRegClass(TID, 0));
2782 MRI.setRegClass(DestReg, NewRC);
2783
2784 UseMO->setReg(DestReg);
2785 UseMO->setSubReg(AMDGPU::NoSubRegister);
2786
2787 // Use a smaller load with the desired size, possibly with updated offset.
2788 MachineInstr *MI = MF->CloneMachineInstr(&Orig);
2789 MI->setDesc(TID);
2790 MI->getOperand(0).setReg(DestReg);
2791 MI->getOperand(0).setSubReg(AMDGPU::NoSubRegister);
2792 if (Offset) {
2793 MachineOperand *OffsetMO = getNamedOperand(*MI, AMDGPU::OpName::offset);
2794 int64_t FinalOffset = OffsetMO->getImm() + Offset / 8;
2795 OffsetMO->setImm(FinalOffset);
2796 }
2798 for (const MachineMemOperand *MemOp : Orig.memoperands())
2799 NewMMOs.push_back(MF->getMachineMemOperand(MemOp, MemOp->getPointerInfo(),
2800 SubregSize / 8));
2801 MI->setMemRefs(*MF, NewMMOs);
2802
2803 MBB.insert(I, MI);
2804 return;
2805 }
2806
2807 default:
2808 break;
2809 }
2810
2811 TargetInstrInfo::reMaterialize(MBB, I, DestReg, SubIdx, Orig, UsedLanes);
2812}
2813
2814std::pair<MachineInstr*, MachineInstr*>
2816 assert (MI.getOpcode() == AMDGPU::V_MOV_B64_DPP_PSEUDO);
2817
2818 if (ST.hasVMovB64Inst() && ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
2820 ST, getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl)->getImm())) {
2821 MI.setDesc(get(AMDGPU::V_MOV_B64_dpp));
2822 return std::pair(&MI, nullptr);
2823 }
2824
2825 MachineBasicBlock &MBB = *MI.getParent();
2826 DebugLoc DL = MBB.findDebugLoc(MI);
2827 MachineFunction *MF = MBB.getParent();
2828 MachineRegisterInfo &MRI = MF->getRegInfo();
2829 Register Dst = MI.getOperand(0).getReg();
2830 unsigned Part = 0;
2831 MachineInstr *Split[2];
2832
2833 for (auto Sub : { AMDGPU::sub0, AMDGPU::sub1 }) {
2834 auto MovDPP = BuildMI(MBB, MI, DL, get(AMDGPU::V_MOV_B32_dpp));
2835 if (Dst.isPhysical()) {
2836 MovDPP.addDef(RI.getSubReg(Dst, Sub));
2837 } else {
2838 assert(MRI.isSSA());
2839 auto Tmp = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2840 MovDPP.addDef(Tmp);
2841 }
2842
2843 for (unsigned I = 1; I <= 2; ++I) { // old and src operands.
2844 const MachineOperand &SrcOp = MI.getOperand(I);
2845 assert(!SrcOp.isFPImm());
2846 if (SrcOp.isImm()) {
2847 APInt Imm(64, SrcOp.getImm());
2848 Imm.ashrInPlace(Part * 32);
2849 MovDPP.addImm(Imm.getLoBits(32).getZExtValue());
2850 } else {
2851 assert(SrcOp.isReg());
2852 Register Src = SrcOp.getReg();
2853 if (Src.isPhysical())
2854 MovDPP.addReg(RI.getSubReg(Src, Sub));
2855 else
2856 MovDPP.addReg(Src, getUndefRegState(SrcOp.isUndef()), Sub);
2857 }
2858 }
2859
2860 for (const MachineOperand &MO : llvm::drop_begin(MI.explicit_operands(), 3))
2861 MovDPP.addImm(MO.getImm());
2862
2863 Split[Part] = MovDPP;
2864 ++Part;
2865 }
2866
2867 if (Dst.isVirtual())
2868 BuildMI(MBB, MI, DL, get(AMDGPU::REG_SEQUENCE), Dst)
2869 .addReg(Split[0]->getOperand(0).getReg())
2870 .addImm(AMDGPU::sub0)
2871 .addReg(Split[1]->getOperand(0).getReg())
2872 .addImm(AMDGPU::sub1);
2873
2874 MI.eraseFromParent();
2875 return std::pair(Split[0], Split[1]);
2876}
2877
2878std::optional<DestSourcePair>
2880 if (MI.getOpcode() == AMDGPU::WWM_COPY)
2881 return DestSourcePair{MI.getOperand(0), MI.getOperand(1)};
2882
2883 return std::nullopt;
2884}
2885
2887 AMDGPU::OpName Src0OpName,
2888 MachineOperand &Src1,
2889 AMDGPU::OpName Src1OpName) const {
2890 MachineOperand *Src0Mods = getNamedOperand(MI, Src0OpName);
2891 if (!Src0Mods)
2892 return false;
2893
2894 MachineOperand *Src1Mods = getNamedOperand(MI, Src1OpName);
2895 assert(Src1Mods &&
2896 "All commutable instructions have both src0 and src1 modifiers");
2897
2898 int Src0ModsVal = Src0Mods->getImm();
2899 int Src1ModsVal = Src1Mods->getImm();
2900
2901 Src1Mods->setImm(Src0ModsVal);
2902 Src0Mods->setImm(Src1ModsVal);
2903 return true;
2904}
2905
2907 MachineOperand &RegOp,
2908 MachineOperand &NonRegOp) {
2909 Register Reg = RegOp.getReg();
2910 unsigned SubReg = RegOp.getSubReg();
2911 bool IsKill = RegOp.isKill();
2912 bool IsDead = RegOp.isDead();
2913 bool IsUndef = RegOp.isUndef();
2914 bool IsDebug = RegOp.isDebug();
2915
2916 if (NonRegOp.isImm())
2917 RegOp.ChangeToImmediate(NonRegOp.getImm());
2918 else if (NonRegOp.isFI())
2919 RegOp.ChangeToFrameIndex(NonRegOp.getIndex());
2920 else if (NonRegOp.isGlobal()) {
2921 RegOp.ChangeToGA(NonRegOp.getGlobal(), NonRegOp.getOffset(),
2922 NonRegOp.getTargetFlags());
2923 } else
2924 return nullptr;
2925
2926 // Make sure we don't reinterpret a subreg index in the target flags.
2927 RegOp.setTargetFlags(NonRegOp.getTargetFlags());
2928
2929 NonRegOp.ChangeToRegister(Reg, false, false, IsKill, IsDead, IsUndef, IsDebug);
2930 NonRegOp.setSubReg(SubReg);
2931
2932 return &MI;
2933}
2934
2936 MachineOperand &NonRegOp1,
2937 MachineOperand &NonRegOp2) {
2938 unsigned TargetFlags = NonRegOp1.getTargetFlags();
2939 int64_t NonRegVal = NonRegOp1.getImm();
2940
2941 NonRegOp1.setImm(NonRegOp2.getImm());
2942 NonRegOp2.setImm(NonRegVal);
2943 NonRegOp1.setTargetFlags(NonRegOp2.getTargetFlags());
2944 NonRegOp2.setTargetFlags(TargetFlags);
2945 return &MI;
2946}
2947
2948bool SIInstrInfo::isLegalToSwap(const MachineInstr &MI, unsigned OpIdx0,
2949 unsigned OpIdx1) const {
2950 const MCInstrDesc &InstDesc = MI.getDesc();
2951 const MCOperandInfo &OpInfo0 = InstDesc.operands()[OpIdx0];
2952 const MCOperandInfo &OpInfo1 = InstDesc.operands()[OpIdx1];
2953
2954 unsigned Opc = MI.getOpcode();
2955 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
2956
2957 const MachineOperand &MO0 = MI.getOperand(OpIdx0);
2958 const MachineOperand &MO1 = MI.getOperand(OpIdx1);
2959
2960 // Swap doesn't breach constant bus or literal limits
2961 // It may move literal to position other than src0, this is not allowed
2962 // pre-gfx10 However, most test cases need literals in Src0 for VOP
2963 // FIXME: After gfx9, literal can be in place other than Src0
2964 if (isVALU(MI, /*AllowLDSDMA=*/false)) {
2965 if ((int)OpIdx0 == Src0Idx && !MO0.isReg() &&
2966 !isInlineConstant(MO0, OpInfo1))
2967 return false;
2968 if ((int)OpIdx1 == Src0Idx && !MO1.isReg() &&
2969 !isInlineConstant(MO1, OpInfo0))
2970 return false;
2971 }
2972
2973 if ((int)OpIdx1 != Src0Idx && MO0.isReg()) {
2974 if (OpInfo1.RegClass == -1)
2975 return OpInfo1.OperandType == MCOI::OPERAND_UNKNOWN;
2976 return isLegalRegOperand(MI, OpIdx1, MO0) &&
2977 (!MO1.isReg() || isLegalRegOperand(MI, OpIdx0, MO1));
2978 }
2979 if ((int)OpIdx0 != Src0Idx && MO1.isReg()) {
2980 if (OpInfo0.RegClass == -1)
2981 return OpInfo0.OperandType == MCOI::OPERAND_UNKNOWN;
2982 return (!MO0.isReg() || isLegalRegOperand(MI, OpIdx1, MO0)) &&
2983 isLegalRegOperand(MI, OpIdx0, MO1);
2984 }
2985
2986 // No need to check 64-bit literals since swapping does not bring new
2987 // 64-bit literals into current instruction to fold to 32-bit
2988
2989 return isImmOperandLegal(MI, OpIdx1, MO0);
2990}
2991
2993 if (!isDPP(MI))
2994 return false;
2995 const MachineOperand *DppCtrl = getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl);
2996 return !DppCtrl || DppCtrl->getImm() != AMDGPU::DPP::QUAD_PERM_ID;
2997}
2998
3000 unsigned Src0Idx,
3001 unsigned Src1Idx) const {
3002 assert(!NewMI && "this should never be used");
3003
3005 return nullptr;
3006
3007 unsigned Opc = MI.getOpcode();
3008 int CommutedOpcode = commuteOpcode(Opc);
3009 if (CommutedOpcode == -1)
3010 return nullptr;
3011
3012 if (Src0Idx > Src1Idx)
3013 std::swap(Src0Idx, Src1Idx);
3014
3015 assert(AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0) ==
3016 static_cast<int>(Src0Idx) &&
3017 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1) ==
3018 static_cast<int>(Src1Idx) &&
3019 "inconsistency with findCommutedOpIndices");
3020
3021 if (!isLegalToSwap(MI, Src0Idx, Src1Idx))
3022 return nullptr;
3023
3024 MachineInstr *CommutedMI = nullptr;
3025 MachineOperand &Src0 = MI.getOperand(Src0Idx);
3026 MachineOperand &Src1 = MI.getOperand(Src1Idx);
3027 if (Src0.isReg() && Src1.isReg()) {
3028 // Be sure to copy the source modifiers to the right place.
3029 CommutedMI =
3030 TargetInstrInfo::commuteInstructionImpl(MI, NewMI, Src0Idx, Src1Idx);
3031 } else if (Src0.isReg() && !Src1.isReg()) {
3032 CommutedMI = swapRegAndNonRegOperand(MI, Src0, Src1);
3033 } else if (!Src0.isReg() && Src1.isReg()) {
3034 CommutedMI = swapRegAndNonRegOperand(MI, Src1, Src0);
3035 } else if (Src0.isImm() && Src1.isImm()) {
3036 CommutedMI = swapImmOperands(MI, Src0, Src1);
3037 } else {
3038 // FIXME: Found two non registers to commute. This does happen.
3039 return nullptr;
3040 }
3041
3042 if (CommutedMI) {
3043 swapSourceModifiers(MI, Src0, AMDGPU::OpName::src0_modifiers,
3044 Src1, AMDGPU::OpName::src1_modifiers);
3045
3046 swapSourceModifiers(MI, Src0, AMDGPU::OpName::src0_sel, Src1,
3047 AMDGPU::OpName::src1_sel);
3048
3049 CommutedMI->setDesc(get(CommutedOpcode));
3050 }
3051
3052 return CommutedMI;
3053}
3054
3055// This needs to be implemented because the source modifiers may be inserted
3056// between the true commutable operands, and the base
3057// TargetInstrInfo::commuteInstruction uses it.
3059 unsigned &SrcOpIdx0,
3060 unsigned &SrcOpIdx1) const {
3062 return false;
3063
3064 return findCommutedOpIndices(MI.getDesc(), SrcOpIdx0, SrcOpIdx1);
3065}
3066
3068 unsigned &SrcOpIdx0,
3069 unsigned &SrcOpIdx1) const {
3070 if (!Desc.isCommutable())
3071 return false;
3072
3073 unsigned Opc = Desc.getOpcode();
3074 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
3075 if (Src0Idx == -1)
3076 return false;
3077
3078 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
3079 if (Src1Idx == -1)
3080 return false;
3081
3082 return fixCommutedOpIndices(SrcOpIdx0, SrcOpIdx1, Src0Idx, Src1Idx);
3083}
3084
3086 int64_t BrOffset) const {
3087 // BranchRelaxation should never have to check s_setpc_b64 or s_add_pc_i64
3088 // because its dest block is unanalyzable.
3089 assert(isSOPP(BranchOp) || isSOPK(BranchOp));
3090
3091 // Convert to dwords.
3092 BrOffset /= 4;
3093
3094 // The branch instructions do PC += signext(SIMM16 * 4) + 4, so the offset is
3095 // from the next instruction.
3096 BrOffset -= 1;
3097
3098 return isIntN(BranchOffsetBits, BrOffset);
3099}
3100
3103 return MI.getOperand(0).getMBB();
3104}
3105
3107 for (const MachineInstr &MI : MBB->terminators()) {
3108 if (MI.getOpcode() == AMDGPU::SI_IF || MI.getOpcode() == AMDGPU::SI_ELSE ||
3109 MI.getOpcode() == AMDGPU::SI_LOOP ||
3110 MI.getOpcode() == AMDGPU::SI_WATERFALL_LOOP)
3111 return true;
3112 }
3113 return false;
3114}
3115
3117 MachineBasicBlock &DestBB,
3118 MachineBasicBlock &RestoreBB,
3119 const DebugLoc &DL, int64_t BrOffset,
3120 RegScavenger *RS) const {
3121 assert(MBB.empty() &&
3122 "new block should be inserted for expanding unconditional branch");
3123 assert(MBB.pred_size() == 1);
3124 assert(RestoreBB.empty() &&
3125 "restore block should be inserted for restoring clobbered registers");
3126
3127 MachineFunction *MF = MBB.getParent();
3128 MachineRegisterInfo &MRI = MF->getRegInfo();
3130 auto I = MBB.end();
3131 auto &MCCtx = MF->getContext();
3132
3133 if (ST.useAddPC64Inst()) {
3134 MCSymbol *Offset =
3135 MCCtx.createTempSymbol("offset", /*AlwaysAddSuffix=*/true);
3136 auto AddPC = BuildMI(MBB, I, DL, get(AMDGPU::S_ADD_PC_I64))
3138 MCSymbol *PostAddPCLabel =
3139 MCCtx.createTempSymbol("post_addpc", /*AlwaysAddSuffix=*/true);
3140 AddPC->setPostInstrSymbol(*MF, PostAddPCLabel);
3141 auto *OffsetExpr = MCBinaryExpr::createSub(
3142 MCSymbolRefExpr::create(DestBB.getSymbol(), MCCtx),
3143 MCSymbolRefExpr::create(PostAddPCLabel, MCCtx), MCCtx);
3144 Offset->setVariableValue(OffsetExpr);
3145 return;
3146 }
3147
3148 assert(RS && "RegScavenger required for long branching");
3149
3150 // FIXME: Virtual register workaround for RegScavenger not working with empty
3151 // blocks.
3152 Register PCReg = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
3153
3154 // Note: as this is used after hazard recognizer we need to apply some hazard
3155 // workarounds directly.
3156 const bool FlushSGPRWrites = (ST.isWave64() && ST.hasVALUMaskWriteHazard()) ||
3157 ST.hasVALUReadSGPRHazard();
3158 auto ApplyHazardWorkarounds = [this, &MBB, &I, &DL, FlushSGPRWrites]() {
3159 if (FlushSGPRWrites)
3160 BuildMI(MBB, I, DL, get(AMDGPU::S_WAITCNT_DEPCTR))
3162 };
3163
3164 // We need to compute the offset relative to the instruction immediately after
3165 // s_getpc_b64. Insert pc arithmetic code before last terminator.
3166 MachineInstr *GetPC = BuildMI(MBB, I, DL, get(AMDGPU::S_GETPC_B64), PCReg);
3167 ApplyHazardWorkarounds();
3168
3169 MCSymbol *PostGetPCLabel =
3170 MCCtx.createTempSymbol("post_getpc", /*AlwaysAddSuffix=*/true);
3171 GetPC->setPostInstrSymbol(*MF, PostGetPCLabel);
3172
3173 MCSymbol *OffsetLo =
3174 MCCtx.createTempSymbol("offset_lo", /*AlwaysAddSuffix=*/true);
3175 MCSymbol *OffsetHi =
3176 MCCtx.createTempSymbol("offset_hi", /*AlwaysAddSuffix=*/true);
3177 BuildMI(MBB, I, DL, get(AMDGPU::S_ADD_U32))
3178 .addReg(PCReg, RegState::Define, AMDGPU::sub0)
3179 .addReg(PCReg, {}, AMDGPU::sub0)
3180 .addSym(OffsetLo, MO_FAR_BRANCH_OFFSET);
3181 BuildMI(MBB, I, DL, get(AMDGPU::S_ADDC_U32))
3182 .addReg(PCReg, RegState::Define, AMDGPU::sub1)
3183 .addReg(PCReg, {}, AMDGPU::sub1)
3184 .addSym(OffsetHi, MO_FAR_BRANCH_OFFSET);
3185 ApplyHazardWorkarounds();
3186
3187 // Insert the indirect branch after the other terminator.
3188 BuildMI(&MBB, DL, get(AMDGPU::S_SETPC_B64))
3189 .addReg(PCReg);
3190
3191 // If a spill is needed for the pc register pair, we need to insert a spill
3192 // restore block right before the destination block, and insert a short branch
3193 // into the old destination block's fallthrough predecessor.
3194 // e.g.:
3195 //
3196 // s_cbranch_scc0 skip_long_branch:
3197 //
3198 // long_branch_bb:
3199 // spill s[8:9]
3200 // s_getpc_b64 s[8:9]
3201 // s_add_u32 s8, s8, restore_bb
3202 // s_addc_u32 s9, s9, 0
3203 // s_setpc_b64 s[8:9]
3204 //
3205 // skip_long_branch:
3206 // foo;
3207 //
3208 // .....
3209 //
3210 // dest_bb_fallthrough_predecessor:
3211 // bar;
3212 // s_branch dest_bb
3213 //
3214 // restore_bb:
3215 // restore s[8:9]
3216 // fallthrough dest_bb
3217 ///
3218 // dest_bb:
3219 // buzz;
3220
3221 Register LongBranchReservedReg = MFI->getLongBranchReservedReg();
3222 Register Scav;
3223
3224 // If we've previously reserved a register for long branches
3225 // avoid running the scavenger and just use those registers
3226 if (LongBranchReservedReg) {
3227 RS->enterBasicBlock(MBB);
3228 Scav = LongBranchReservedReg;
3229 } else {
3230 RS->enterBasicBlockEnd(MBB);
3231 Scav = RS->scavengeRegisterBackwards(
3232 AMDGPU::SReg_64RegClass, MachineBasicBlock::iterator(GetPC),
3233 /* RestoreAfter */ false, 0, /* AllowSpill */ false);
3234 }
3235 if (Scav) {
3236 RS->setRegUsed(Scav);
3237 MRI.replaceRegWith(PCReg, Scav);
3238 MRI.clearVirtRegs();
3239 } else {
3240 // As SGPR needs VGPR to be spilled, we reuse the slot of temporary VGPR for
3241 // SGPR spill.
3242 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
3243 const SIRegisterInfo *TRI = ST.getRegisterInfo();
3244 TRI->spillEmergencySGPR(GetPC, RestoreBB, AMDGPU::SGPR0_SGPR1, RS);
3245 MRI.replaceRegWith(PCReg, AMDGPU::SGPR0_SGPR1);
3246 MRI.clearVirtRegs();
3247 }
3248
3249 MCSymbol *DestLabel = Scav ? DestBB.getSymbol() : RestoreBB.getSymbol();
3250 // Now, the distance could be defined.
3252 MCSymbolRefExpr::create(DestLabel, MCCtx),
3253 MCSymbolRefExpr::create(PostGetPCLabel, MCCtx), MCCtx);
3254 // Add offset assignments.
3255 auto *Mask = MCConstantExpr::create(0xFFFFFFFFULL, MCCtx);
3256 OffsetLo->setVariableValue(MCBinaryExpr::createAnd(Offset, Mask, MCCtx));
3257 auto *ShAmt = MCConstantExpr::create(32, MCCtx);
3258 OffsetHi->setVariableValue(MCBinaryExpr::createAShr(Offset, ShAmt, MCCtx));
3259}
3260
3261unsigned SIInstrInfo::getBranchOpcode(SIInstrInfo::BranchPredicate Cond) {
3262 switch (Cond) {
3263 case SIInstrInfo::SCC_TRUE:
3264 return AMDGPU::S_CBRANCH_SCC1;
3265 case SIInstrInfo::SCC_FALSE:
3266 return AMDGPU::S_CBRANCH_SCC0;
3267 case SIInstrInfo::VCCNZ:
3268 return AMDGPU::S_CBRANCH_VCCNZ;
3269 case SIInstrInfo::VCCZ:
3270 return AMDGPU::S_CBRANCH_VCCZ;
3271 case SIInstrInfo::EXECNZ:
3272 return AMDGPU::S_CBRANCH_EXECNZ;
3273 case SIInstrInfo::EXECZ:
3274 return AMDGPU::S_CBRANCH_EXECZ;
3275 default:
3276 llvm_unreachable("invalid branch predicate");
3277 }
3278}
3279
3280SIInstrInfo::BranchPredicate SIInstrInfo::getBranchPredicate(unsigned Opcode) {
3281 switch (Opcode) {
3282 case AMDGPU::S_CBRANCH_SCC0:
3283 return SCC_FALSE;
3284 case AMDGPU::S_CBRANCH_SCC1:
3285 return SCC_TRUE;
3286 case AMDGPU::S_CBRANCH_VCCNZ:
3287 return VCCNZ;
3288 case AMDGPU::S_CBRANCH_VCCZ:
3289 return VCCZ;
3290 case AMDGPU::S_CBRANCH_EXECNZ:
3291 return EXECNZ;
3292 case AMDGPU::S_CBRANCH_EXECZ:
3293 return EXECZ;
3294 default:
3295 return INVALID_BR;
3296 }
3297}
3298
3302 MachineBasicBlock *&FBB,
3304 bool AllowModify) const {
3305 if (I->getOpcode() == AMDGPU::S_BRANCH) {
3306 // Unconditional Branch
3307 TBB = I->getOperand(0).getMBB();
3308 return false;
3309 }
3310
3311 BranchPredicate Pred = getBranchPredicate(I->getOpcode());
3312 if (Pred == INVALID_BR)
3313 return true;
3314
3315 MachineBasicBlock *CondBB = I->getOperand(0).getMBB();
3316 Cond.push_back(MachineOperand::CreateImm(Pred));
3317 Cond.push_back(I->getOperand(1)); // Save the branch register.
3318
3319 ++I;
3320
3321 if (I == MBB.end()) {
3322 // Conditional branch followed by fall-through.
3323 TBB = CondBB;
3324 return false;
3325 }
3326
3327 if (I->getOpcode() == AMDGPU::S_BRANCH) {
3328 TBB = CondBB;
3329 FBB = I->getOperand(0).getMBB();
3330 return false;
3331 }
3332
3333 return true;
3334}
3335
3337 MachineBasicBlock *&FBB,
3339 bool AllowModify) const {
3340 MachineBasicBlock::iterator I = MBB.getFirstTerminator();
3341 auto E = MBB.end();
3342 if (I == E)
3343 return false;
3344
3345 // Skip over the instructions that are artificially terminators for special
3346 // exec management.
3347 while (I != E && !I->isBranch() && !I->isReturn()) {
3348 switch (I->getOpcode()) {
3349 case AMDGPU::S_MOV_B64_term:
3350 case AMDGPU::S_XOR_B64_term:
3351 case AMDGPU::S_OR_B64_term:
3352 case AMDGPU::S_ANDN2_B64_term:
3353 case AMDGPU::S_AND_B64_term:
3354 case AMDGPU::S_AND_SAVEEXEC_B64_term:
3355 case AMDGPU::S_MOV_B32_term:
3356 case AMDGPU::S_XOR_B32_term:
3357 case AMDGPU::S_OR_B32_term:
3358 case AMDGPU::S_ANDN2_B32_term:
3359 case AMDGPU::S_AND_B32_term:
3360 case AMDGPU::S_AND_SAVEEXEC_B32_term:
3361 case AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term:
3362 case AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term:
3363 break;
3364 case AMDGPU::SI_IF:
3365 case AMDGPU::SI_ELSE:
3366 case AMDGPU::SI_KILL_I1_TERMINATOR:
3367 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
3368 // FIXME: It's messy that these need to be considered here at all.
3369 return true;
3370 default:
3371 llvm_unreachable("unexpected non-branch terminator inst");
3372 }
3373
3374 ++I;
3375 }
3376
3377 if (I == E)
3378 return false;
3379
3380 return analyzeBranchImpl(MBB, I, TBB, FBB, Cond, AllowModify);
3381}
3382
3384 int *BytesRemoved) const {
3385 unsigned Count = 0;
3386 unsigned RemovedSize = 0;
3387 for (MachineInstr &MI : llvm::make_early_inc_range(MBB.terminators())) {
3388 // Skip over artificial terminators when removing instructions.
3389 if (MI.isBranch() || MI.isReturn()) {
3390 RemovedSize += getInstSizeInBytes(MI);
3391 MI.eraseFromParent();
3392 ++Count;
3393 }
3394 }
3395
3396 if (BytesRemoved)
3397 *BytesRemoved = RemovedSize;
3398
3399 return Count;
3400}
3401
3402// Copy the flags onto the implicit condition register operand.
3404 const MachineOperand &OrigCond) {
3405 CondReg.setIsUndef(OrigCond.isUndef());
3406 CondReg.setIsKill(OrigCond.isKill());
3407}
3408
3411 MachineBasicBlock *FBB,
3413 const DebugLoc &DL,
3414 int *BytesAdded) const {
3415 if (!FBB && Cond.empty()) {
3416 BuildMI(&MBB, DL, get(AMDGPU::S_BRANCH))
3417 .addMBB(TBB);
3418 if (BytesAdded)
3419 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3420 return 1;
3421 }
3422
3423 assert(TBB && Cond[0].isImm());
3424
3425 unsigned Opcode
3426 = getBranchOpcode(static_cast<BranchPredicate>(Cond[0].getImm()));
3427
3428 if (!FBB) {
3429 MachineInstr *CondBr =
3430 BuildMI(&MBB, DL, get(Opcode))
3431 .addMBB(TBB);
3432
3433 // Copy the flags onto the implicit condition register operand.
3434 preserveCondRegFlags(CondBr->getOperand(1), Cond[1]);
3435 fixImplicitOperands(*CondBr);
3436
3437 if (BytesAdded)
3438 *BytesAdded = ST.hasOffset3fBug() ? 8 : 4;
3439 return 1;
3440 }
3441
3442 assert(TBB && FBB);
3443
3444 MachineInstr *CondBr =
3445 BuildMI(&MBB, DL, get(Opcode))
3446 .addMBB(TBB);
3447 fixImplicitOperands(*CondBr);
3448 BuildMI(&MBB, DL, get(AMDGPU::S_BRANCH))
3449 .addMBB(FBB);
3450
3451 MachineOperand &CondReg = CondBr->getOperand(1);
3452 CondReg.setIsUndef(Cond[1].isUndef());
3453 CondReg.setIsKill(Cond[1].isKill());
3454
3455 if (BytesAdded)
3456 *BytesAdded = ST.hasOffset3fBug() ? 16 : 8;
3457
3458 return 2;
3459}
3460
3463 if (Cond.size() != 2) {
3464 return true;
3465 }
3466
3467 if (Cond[0].isImm()) {
3468 Cond[0].setImm(-Cond[0].getImm());
3469 return false;
3470 }
3471
3472 return true;
3473}
3474
3475namespace {
3476class AMDGPUPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
3477private:
3478 /// The compare instruction for loop control
3479 const MachineInstr *CmpInst = nullptr;
3480 /// The normalized condition used by createTripCountGreaterCondition()
3482
3483public:
3484 AMDGPUPipelinerLoopInfo(MachineInstr *CmpInst,
3486 : CmpInst(CmpInst), Cond(Cond.begin(), Cond.end()) {}
3487
3488 bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
3489 return CmpInst && MI == CmpInst;
3490 }
3491
3492 std::optional<bool> createTripCountGreaterCondition(
3493 int TC, MachineBasicBlock &MBB,
3494 SmallVectorImpl<MachineOperand> &CondParam) override {
3495 CondParam = this->Cond;
3496 return {};
3497 }
3498
3499 void adjustTripCount(int TripCountAdjust) override {}
3500
3501 void setPreheader(MachineBasicBlock *NewPreheader) override {}
3502};
3503} // namespace
3504
3505std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
3507 MachineBasicBlock *TBB = nullptr, *FBB = nullptr;
3509 // Unanalyzable terminator.
3510 if (analyzeBranch(*LoopBB, TBB, FBB, Cond, /*AllowModify=*/false))
3511 return nullptr;
3512
3513 // Infinite loops are not supported.
3514 if (TBB == LoopBB && FBB == LoopBB)
3515 return nullptr;
3516
3517 // Must be conditional branch.
3518 if (FBB == nullptr)
3519 return nullptr;
3520
3521 assert((TBB == LoopBB || FBB == LoopBB) &&
3522 "The Loop must be a single-basic-block loop");
3523
3524 // Divergent (VCC/EXEC) back-edge is not supported.
3525 BranchPredicate Pred = static_cast<BranchPredicate>(Cond[0].getImm());
3526 if (Pred != SCC_TRUE && Pred != SCC_FALSE)
3527 return nullptr;
3528
3529 // Calls and inline assembly are not supported.
3530 for (const MachineInstr &MI : *LoopBB)
3531 if (MI.isCall() || MI.isInlineAsm())
3532 return nullptr;
3533
3534 // Normalization for createTripCountGreaterCondition(): make Cond mean
3535 // "exit the loop" so the expander emits correct prolog guard branches.
3536 if (TBB == LoopBB)
3538
3539 auto Instructions = make_range(
3541 LoopBB->rend());
3542 auto CmpI = llvm::find_if(Instructions, [&](const MachineInstr &MI) {
3543 return MI.modifiesRegister(Cond[1].getReg(), &RI);
3544 });
3545
3546 if (CmpI == Instructions.end() || CmpI->isPHI())
3547 return nullptr;
3548 MachineInstr *CmpInst = &*CmpI;
3549
3550 return std::make_unique<AMDGPUPipelinerLoopInfo>(CmpInst, Cond);
3551}
3552
3555 Register DstReg, Register TrueReg,
3556 Register FalseReg, int &CondCycles,
3557 int &TrueCycles, int &FalseCycles) const {
3558 switch (Cond[0].getImm()) {
3559 case VCCNZ:
3560 case VCCZ: {
3561 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3562 const TargetRegisterClass *RC = MRI.getRegClass(TrueReg);
3563 if (MRI.getRegClass(FalseReg) != RC)
3564 return false;
3565
3566 int NumInsts = AMDGPU::getRegBitWidth(*RC) / 32;
3567 CondCycles = TrueCycles = FalseCycles = NumInsts; // ???
3568
3569 // Limit to equal cost for branch vs. N v_cndmask_b32s.
3570 return RI.hasVGPRs(RC) && NumInsts <= 6;
3571 }
3572 case SCC_TRUE:
3573 case SCC_FALSE: {
3574 // FIXME: We could insert for VGPRs if we could replace the original compare
3575 // with a vector one.
3576 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3577 const TargetRegisterClass *RC = MRI.getRegClass(TrueReg);
3578 if (MRI.getRegClass(FalseReg) != RC)
3579 return false;
3580
3581 int NumInsts = AMDGPU::getRegBitWidth(*RC) / 32;
3582
3583 // Multiples of 8 can do s_cselect_b64
3584 if (NumInsts % 2 == 0)
3585 NumInsts /= 2;
3586
3587 CondCycles = TrueCycles = FalseCycles = NumInsts; // ???
3588 return RI.isSGPRClass(RC);
3589 }
3590 default:
3591 return false;
3592 }
3593}
3594
3598 Register TrueReg, Register FalseReg) const {
3599 BranchPredicate Pred = static_cast<BranchPredicate>(Cond[0].getImm());
3600 if (Pred == VCCZ || Pred == SCC_FALSE) {
3601 Pred = static_cast<BranchPredicate>(-Pred);
3602 std::swap(TrueReg, FalseReg);
3603 }
3604
3605 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
3606 const TargetRegisterClass *DstRC = MRI.getRegClass(DstReg);
3607 unsigned DstSize = RI.getRegSizeInBits(*DstRC);
3608
3609 if (DstSize == 32) {
3611 if (Pred == SCC_TRUE) {
3612 Select = BuildMI(MBB, I, DL, get(AMDGPU::S_CSELECT_B32), DstReg)
3613 .addReg(TrueReg)
3614 .addReg(FalseReg);
3615 } else {
3616 // Instruction's operands are backwards from what is expected.
3617 Select = BuildMI(MBB, I, DL, get(AMDGPU::V_CNDMASK_B32_e32), DstReg)
3618 .addReg(FalseReg)
3619 .addReg(TrueReg);
3620 }
3621
3622 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3623 return;
3624 }
3625
3626 if (DstSize == 64 && Pred == SCC_TRUE) {
3628 BuildMI(MBB, I, DL, get(AMDGPU::S_CSELECT_B64), DstReg)
3629 .addReg(TrueReg)
3630 .addReg(FalseReg);
3631
3632 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3633 return;
3634 }
3635
3636 static const int16_t Sub0_15[] = {
3637 AMDGPU::sub0, AMDGPU::sub1, AMDGPU::sub2, AMDGPU::sub3,
3638 AMDGPU::sub4, AMDGPU::sub5, AMDGPU::sub6, AMDGPU::sub7,
3639 AMDGPU::sub8, AMDGPU::sub9, AMDGPU::sub10, AMDGPU::sub11,
3640 AMDGPU::sub12, AMDGPU::sub13, AMDGPU::sub14, AMDGPU::sub15,
3641 };
3642
3643 static const int16_t Sub0_15_64[] = {
3644 AMDGPU::sub0_sub1, AMDGPU::sub2_sub3,
3645 AMDGPU::sub4_sub5, AMDGPU::sub6_sub7,
3646 AMDGPU::sub8_sub9, AMDGPU::sub10_sub11,
3647 AMDGPU::sub12_sub13, AMDGPU::sub14_sub15,
3648 };
3649
3650 unsigned SelOp = AMDGPU::V_CNDMASK_B32_e32;
3651 const TargetRegisterClass *EltRC = &AMDGPU::VGPR_32RegClass;
3652 const int16_t *SubIndices = Sub0_15;
3653 int NElts = DstSize / 32;
3654
3655 // 64-bit select is only available for SALU.
3656 // TODO: Split 96-bit into 64-bit and 32-bit, not 3x 32-bit.
3657 if (Pred == SCC_TRUE) {
3658 if (NElts % 2) {
3659 SelOp = AMDGPU::S_CSELECT_B32;
3660 EltRC = &AMDGPU::SGPR_32RegClass;
3661 } else {
3662 SelOp = AMDGPU::S_CSELECT_B64;
3663 EltRC = &AMDGPU::SGPR_64RegClass;
3664 SubIndices = Sub0_15_64;
3665 NElts /= 2;
3666 }
3667 }
3668
3670 MBB, I, DL, get(AMDGPU::REG_SEQUENCE), DstReg);
3671
3672 I = MIB->getIterator();
3673
3675 for (int Idx = 0; Idx != NElts; ++Idx) {
3676 Register DstElt = MRI.createVirtualRegister(EltRC);
3677 Regs.push_back(DstElt);
3678
3679 unsigned SubIdx = SubIndices[Idx];
3680
3682 if (SelOp == AMDGPU::V_CNDMASK_B32_e32) {
3683 Select = BuildMI(MBB, I, DL, get(SelOp), DstElt)
3684 .addReg(FalseReg, {}, SubIdx)
3685 .addReg(TrueReg, {}, SubIdx);
3686 } else {
3687 Select = BuildMI(MBB, I, DL, get(SelOp), DstElt)
3688 .addReg(TrueReg, {}, SubIdx)
3689 .addReg(FalseReg, {}, SubIdx);
3690 }
3691
3692 preserveCondRegFlags(Select->getOperand(3), Cond[1]);
3694
3695 MIB.addReg(DstElt)
3696 .addImm(SubIdx);
3697 }
3698}
3699
3701
3702 if (MI.isBranch() || MI.isCall() || MI.isReturn() || MI.isIndirectBranch())
3703 return true;
3704
3705 switch (MI.getOpcode()) {
3706 case AMDGPU::S_ENDPGM:
3707 case AMDGPU::S_ENDPGM_SAVED:
3708 case AMDGPU::S_TRAP:
3709 case AMDGPU::S_GETREG_B32:
3710 case AMDGPU::S_SETREG_B32:
3711 case AMDGPU::S_SETREG_B32_mode:
3712 case AMDGPU::S_SETREG_IMM32_B32:
3713 case AMDGPU::S_SETREG_IMM32_B32_mode:
3714 case AMDGPU::S_SENDMSG:
3715 case AMDGPU::S_SENDMSGHALT:
3716 case AMDGPU::S_SENDMSG_RTN_B32:
3717 case AMDGPU::S_SENDMSG_RTN_B64:
3718 case AMDGPU::S_BARRIER_WAIT:
3719 case AMDGPU::S_BARRIER_SIGNAL_M0:
3720 case AMDGPU::S_BARRIER_SIGNAL_IMM:
3721 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_M0:
3722 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM:
3723 return true;
3724 default:
3725 return false;
3726 }
3727}
3728
3730 switch (MI.getOpcode()) {
3731 case AMDGPU::V_MOV_B16_t16_e32:
3732 case AMDGPU::V_MOV_B16_t16_e64:
3733 case AMDGPU::V_MOV_B32_e32:
3734 case AMDGPU::V_MOV_B32_e64:
3735 case AMDGPU::V_MOV_B64_PSEUDO:
3736 case AMDGPU::V_MOV_B64_e32:
3737 case AMDGPU::V_MOV_B64_e64:
3738 case AMDGPU::S_MOV_B32:
3739 case AMDGPU::S_MOV_B64:
3740 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3741 case AMDGPU::COPY:
3742 case AMDGPU::WWM_COPY:
3743 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3744 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3745 case AMDGPU::V_ACCVGPR_MOV_B32:
3746 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3747 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3748 return true;
3749 default:
3750 return false;
3751 }
3752}
3753
3755 switch (MI.getOpcode()) {
3756 case AMDGPU::V_MOV_B16_t16_e32:
3757 case AMDGPU::V_MOV_B16_t16_e64:
3758 return 2;
3759 case AMDGPU::V_MOV_B32_e32:
3760 case AMDGPU::V_MOV_B32_e64:
3761 case AMDGPU::V_MOV_B64_PSEUDO:
3762 case AMDGPU::V_MOV_B64_e32:
3763 case AMDGPU::V_MOV_B64_e64:
3764 case AMDGPU::S_MOV_B32:
3765 case AMDGPU::S_MOV_B64:
3766 case AMDGPU::S_MOV_B64_IMM_PSEUDO:
3767 case AMDGPU::COPY:
3768 case AMDGPU::WWM_COPY:
3769 case AMDGPU::V_ACCVGPR_WRITE_B32_e64:
3770 case AMDGPU::V_ACCVGPR_READ_B32_e64:
3771 case AMDGPU::V_ACCVGPR_MOV_B32:
3772 case AMDGPU::AV_MOV_B32_IMM_PSEUDO:
3773 case AMDGPU::AV_MOV_B64_IMM_PSEUDO:
3774 return 1;
3775 default:
3776 llvm_unreachable("MI is not a foldable copy");
3777 }
3778}
3779
3780static constexpr AMDGPU::OpName ModifierOpNames[] = {
3781 AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src1_modifiers,
3782 AMDGPU::OpName::src2_modifiers, AMDGPU::OpName::clamp,
3783 AMDGPU::OpName::omod, AMDGPU::OpName::op_sel};
3784
3786 unsigned Opc = MI.getOpcode();
3787 for (AMDGPU::OpName Name : reverse(ModifierOpNames)) {
3788 int Idx = AMDGPU::getNamedOperandIdx(Opc, Name);
3789 if (Idx >= 0)
3790 MI.removeOperand(Idx);
3791 }
3792}
3793
3795 const MCInstrDesc &NewDesc) const {
3796 MI.setDesc(NewDesc);
3797
3798 // Remove any leftover implicit operands from mutating the instruction. e.g.
3799 // if we replace an s_and_b32 with a copy, we don't need the implicit scc def
3800 // anymore.
3801 const MCInstrDesc &Desc = MI.getDesc();
3802 unsigned NumOps = Desc.getNumOperands() + Desc.implicit_uses().size() +
3803 Desc.implicit_defs().size();
3804
3805 for (unsigned I = MI.getNumOperands() - 1; I >= NumOps; --I)
3806 MI.removeOperand(I);
3807}
3808
3809std::optional<int64_t> SIInstrInfo::extractSubregFromImm(int64_t Imm,
3810 unsigned SubRegIndex) {
3811 switch (SubRegIndex) {
3812 case AMDGPU::NoSubRegister:
3813 return Imm;
3814 case AMDGPU::sub0:
3815 return SignExtend64<32>(Imm);
3816 case AMDGPU::sub1:
3817 return SignExtend64<32>(Imm >> 32);
3818 case AMDGPU::lo16:
3819 return SignExtend64<16>(Imm);
3820 case AMDGPU::hi16:
3821 return SignExtend64<16>(Imm >> 16);
3822 case AMDGPU::sub1_lo16:
3823 return SignExtend64<16>(Imm >> 32);
3824 case AMDGPU::sub1_hi16:
3825 return SignExtend64<16>(Imm >> 48);
3826 default:
3827 return std::nullopt;
3828 }
3829
3830 llvm_unreachable("covered subregister switch");
3831}
3832
3833static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc) {
3834 switch (Opc) {
3835 case AMDGPU::V_MAC_F16_e32:
3836 case AMDGPU::V_MAC_F16_e64:
3837 case AMDGPU::V_MAD_F16_e64:
3838 return AMDGPU::V_MADAK_F16;
3839 case AMDGPU::V_MAC_F32_e32:
3840 case AMDGPU::V_MAC_F32_e64:
3841 case AMDGPU::V_MAD_F32_e64:
3842 return AMDGPU::V_MADAK_F32;
3843 case AMDGPU::V_FMAC_F32_e32:
3844 case AMDGPU::V_FMAC_F32_e64:
3845 case AMDGPU::V_FMA_F32_e64:
3846 return AMDGPU::V_FMAAK_F32;
3847 case AMDGPU::V_FMAC_F16_e32:
3848 case AMDGPU::V_FMAC_F16_e64:
3849 case AMDGPU::V_FMAC_F16_t16_e64:
3850 case AMDGPU::V_FMAC_F16_fake16_e64:
3851 case AMDGPU::V_FMAC_F16_t16_e32:
3852 case AMDGPU::V_FMAC_F16_fake16_e32:
3853 case AMDGPU::V_FMA_F16_e64:
3854 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3855 ? AMDGPU::V_FMAAK_F16_t16
3856 : AMDGPU::V_FMAAK_F16_fake16
3857 : AMDGPU::V_FMAAK_F16;
3858 case AMDGPU::V_FMAC_F64_e32:
3859 case AMDGPU::V_FMAC_F64_e64:
3860 case AMDGPU::V_FMA_F64_e64:
3861 return AMDGPU::V_FMAAK_F64;
3862 default:
3863 llvm_unreachable("invalid instruction");
3864 }
3865}
3866
3867static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc) {
3868 switch (Opc) {
3869 case AMDGPU::V_MAC_F16_e32:
3870 case AMDGPU::V_MAC_F16_e64:
3871 case AMDGPU::V_MAD_F16_e64:
3872 return AMDGPU::V_MADMK_F16;
3873 case AMDGPU::V_MAC_F32_e32:
3874 case AMDGPU::V_MAC_F32_e64:
3875 case AMDGPU::V_MAD_F32_e64:
3876 return AMDGPU::V_MADMK_F32;
3877 case AMDGPU::V_FMAC_F32_e32:
3878 case AMDGPU::V_FMAC_F32_e64:
3879 case AMDGPU::V_FMA_F32_e64:
3880 return AMDGPU::V_FMAMK_F32;
3881 case AMDGPU::V_FMAC_F16_e32:
3882 case AMDGPU::V_FMAC_F16_e64:
3883 case AMDGPU::V_FMAC_F16_t16_e64:
3884 case AMDGPU::V_FMAC_F16_fake16_e64:
3885 case AMDGPU::V_FMAC_F16_t16_e32:
3886 case AMDGPU::V_FMAC_F16_fake16_e32:
3887 case AMDGPU::V_FMA_F16_e64:
3888 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
3889 ? AMDGPU::V_FMAMK_F16_t16
3890 : AMDGPU::V_FMAMK_F16_fake16
3891 : AMDGPU::V_FMAMK_F16;
3892 case AMDGPU::V_FMAC_F64_e32:
3893 case AMDGPU::V_FMAC_F64_e64:
3894 case AMDGPU::V_FMA_F64_e64:
3895 return AMDGPU::V_FMAMK_F64;
3896 default:
3897 llvm_unreachable("invalid instruction");
3898 }
3899}
3900
3902 Register Reg, MachineRegisterInfo *MRI) const {
3903 int64_t Imm;
3904 if (!getConstValDefinedInReg(DefMI, Reg, Imm))
3905 return false;
3906
3907 const bool HasMultipleUses = !MRI->hasOneNonDBGUse(Reg);
3908
3909 assert(!DefMI.getOperand(0).getSubReg() && "Expected SSA form");
3910
3911 unsigned Opc = UseMI.getOpcode();
3912 if (Opc == AMDGPU::COPY) {
3913 assert(!UseMI.getOperand(0).getSubReg() && "Expected SSA form");
3914
3915 Register DstReg = UseMI.getOperand(0).getReg();
3916 Register UseSubReg = UseMI.getOperand(1).getSubReg();
3917
3918 const TargetRegisterClass *DstRC = RI.getRegClassForReg(*MRI, DstReg);
3919
3920 if (HasMultipleUses) {
3921 // TODO: This should fold in more cases with multiple use, but we need to
3922 // more carefully consider what those uses are.
3923 unsigned ImmDefSize = RI.getRegSizeInBits(*MRI->getRegClass(Reg));
3924
3925 // Avoid breaking up a 64-bit inline immediate into a subregister extract.
3926 if (UseSubReg != AMDGPU::NoSubRegister && ImmDefSize == 64)
3927 return false;
3928
3929 // Most of the time folding a 32-bit inline constant is free (though this
3930 // might not be true if we can't later fold it into a real user).
3931 //
3932 // FIXME: This isInlineConstant check is imprecise if
3933 // getConstValDefinedInReg handled the tricky non-mov cases.
3934 if (ImmDefSize == 32 &&
3936 return false;
3937 }
3938
3939 bool Is16Bit = UseSubReg != AMDGPU::NoSubRegister &&
3940 RI.getSubRegIdxSize(UseSubReg) == 16;
3941
3942 if (Is16Bit) {
3943 if (RI.hasVGPRs(DstRC))
3944 return false; // Do not clobber vgpr_hi16
3945
3946 if (DstReg.isVirtual() && UseSubReg != AMDGPU::lo16)
3947 return false;
3948 }
3949
3950 MachineFunction *MF = UseMI.getMF();
3951
3952 unsigned NewOpc = AMDGPU::INSTRUCTION_LIST_END;
3953 MCRegister MovDstPhysReg =
3954 DstReg.isPhysical() ? DstReg.asMCReg() : MCRegister();
3955
3956 std::optional<int64_t> SubRegImm = extractSubregFromImm(Imm, UseSubReg);
3957
3958 // TODO: Try to fold with AMDGPU::V_MOV_B16_t16_e64
3959 for (unsigned MovOp :
3960 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
3961 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_ACCVGPR_WRITE_B32_e64}) {
3962 const MCInstrDesc &MovDesc = get(MovOp);
3963
3964 const TargetRegisterClass *MovDstRC = getRegClass(MovDesc, 0);
3965 if (Is16Bit) {
3966 // We just need to find a correctly sized register class, so the
3967 // subregister index compatibility doesn't matter since we're statically
3968 // extracting the immediate value.
3969 MovDstRC = RI.getMatchingSuperRegClass(MovDstRC, DstRC, AMDGPU::lo16);
3970 if (!MovDstRC)
3971 continue;
3972
3973 if (MovDstPhysReg) {
3974 // FIXME: We probably should not do this. If there is a live value in
3975 // the high half of the register, it will be corrupted.
3976 MovDstPhysReg =
3977 RI.getMatchingSuperReg(MovDstPhysReg, AMDGPU::lo16, MovDstRC);
3978 if (!MovDstPhysReg)
3979 continue;
3980 }
3981 }
3982
3983 // Result class isn't the right size, try the next instruction.
3984 if (MovDstPhysReg) {
3985 if (!MovDstRC->contains(MovDstPhysReg))
3986 return false;
3987 } else if (!MRI->constrainRegClass(DstReg, MovDstRC)) {
3988 // TODO: This will be overly conservative in the case of 16-bit virtual
3989 // SGPRs. We could hack up the virtual register uses to use a compatible
3990 // 32-bit class.
3991 continue;
3992 }
3993
3994 const MCOperandInfo &OpInfo = MovDesc.operands()[1];
3995
3996 // Ensure the interpreted immediate value is a valid operand in the new
3997 // mov.
3998 //
3999 // FIXME: isImmOperandLegal should have form that doesn't require existing
4000 // MachineInstr or MachineOperand
4001 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType) &&
4002 !isInlineConstant(*SubRegImm, OpInfo.OperandType))
4003 break;
4004
4005 NewOpc = MovOp;
4006 break;
4007 }
4008
4009 if (NewOpc == AMDGPU::INSTRUCTION_LIST_END)
4010 return false;
4011
4012 if (Is16Bit) {
4013 UseMI.getOperand(0).setSubReg(AMDGPU::NoSubRegister);
4014 if (MovDstPhysReg)
4015 UseMI.getOperand(0).setReg(MovDstPhysReg);
4016 assert(UseMI.getOperand(1).getReg().isVirtual());
4017 }
4018
4019 const MCInstrDesc &NewMCID = get(NewOpc);
4020 UseMI.setDesc(NewMCID);
4021 UseMI.getOperand(1).ChangeToImmediate(*SubRegImm);
4022 UseMI.addImplicitDefUseOperands(*MF);
4023 return true;
4024 }
4025
4026 if (HasMultipleUses)
4027 return false;
4028
4029 if (Opc == AMDGPU::V_MAD_F32_e64 || Opc == AMDGPU::V_MAC_F32_e64 ||
4030 Opc == AMDGPU::V_MAD_F16_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4031 Opc == AMDGPU::V_FMA_F32_e64 || Opc == AMDGPU::V_FMAC_F32_e64 ||
4032 Opc == AMDGPU::V_FMA_F16_e64 || Opc == AMDGPU::V_FMAC_F16_e64 ||
4033 Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4034 Opc == AMDGPU::V_FMAC_F16_fake16_e64 || Opc == AMDGPU::V_FMA_F64_e64 ||
4035 Opc == AMDGPU::V_FMAC_F64_e64) {
4036 // Don't fold if we are using source or output modifiers. The new VOP2
4037 // instructions don't have them.
4039 return false;
4040
4041 // If this is a free constant, there's no reason to do this.
4042 // TODO: We could fold this here instead of letting SIFoldOperands do it
4043 // later.
4044 int Src0Idx = getNamedOperandIdx(UseMI.getOpcode(), AMDGPU::OpName::src0);
4045
4046 // Any src operand can be used for the legality check.
4047 if (isInlineConstant(UseMI, Src0Idx, Imm))
4048 return false;
4049
4050 MachineOperand *Src0 = &UseMI.getOperand(Src0Idx);
4051
4052 MachineOperand *Src1 = getNamedOperand(UseMI, AMDGPU::OpName::src1);
4053 MachineOperand *Src2 = getNamedOperand(UseMI, AMDGPU::OpName::src2);
4054
4055 auto CopyRegOperandToNarrowerRC =
4056 [MRI, this](MachineInstr &MI, unsigned OpNo,
4057 const TargetRegisterClass *NewRC) -> void {
4058 if (!MI.getOperand(OpNo).isReg())
4059 return;
4060 Register Reg = MI.getOperand(OpNo).getReg();
4061 const TargetRegisterClass *RC = RI.getRegClassForReg(*MRI, Reg);
4062 if (RI.getCommonSubClass(RC, NewRC) != NewRC)
4063 return;
4064 Register Tmp = MRI->createVirtualRegister(NewRC);
4065 BuildMI(*MI.getParent(), MI.getIterator(), MI.getDebugLoc(),
4066 get(AMDGPU::COPY), Tmp)
4067 .addReg(Reg);
4068 MI.getOperand(OpNo).setReg(Tmp);
4069 MI.getOperand(OpNo).setIsKill();
4070 };
4071
4072 // Multiplied part is the constant: Use v_madmk_{f16, f32}.
4073 if ((Src0->isReg() && Src0->getReg() == Reg) ||
4074 (Src1->isReg() && Src1->getReg() == Reg)) {
4075 MachineOperand *RegSrc =
4076 Src1->isReg() && Src1->getReg() == Reg ? Src0 : Src1;
4077 if (!RegSrc->isReg())
4078 return false;
4079 if (RI.isSGPRClass(MRI->getRegClass(RegSrc->getReg())) &&
4080 ST.getConstantBusLimit(Opc) < 2)
4081 return false;
4082
4083 if (!Src2->isReg() || RI.isSGPRClass(MRI->getRegClass(Src2->getReg())))
4084 return false;
4085
4086 // If src2 is also a literal constant then we have to choose which one to
4087 // fold. In general it is better to choose madak so that the other literal
4088 // can be materialized in an sgpr instead of a vgpr:
4089 // s_mov_b32 s0, literal
4090 // v_madak_f32 v0, s0, v0, literal
4091 // Instead of:
4092 // v_mov_b32 v1, literal
4093 // v_madmk_f32 v0, v0, literal, v1
4094 MachineInstr *Def = MRI->getUniqueVRegDef(Src2->getReg());
4095 if (Def && Def->isMoveImmediate() &&
4096 !isInlineConstant(Def->getOperand(1)))
4097 return false;
4098
4099 unsigned NewOpc = getNewFMAMKInst(ST, Opc);
4100 if (pseudoToMCOpcode(NewOpc) == -1)
4101 return false;
4102
4103 const std::optional<int64_t> SubRegImm = extractSubregFromImm(
4104 Imm, RegSrc == Src1 ? Src0->getSubReg() : Src1->getSubReg());
4105
4106 // FIXME: This would be a lot easier if we could return a new instruction
4107 // instead of having to modify in place.
4108
4109 Register SrcReg = RegSrc->getReg();
4110 unsigned SrcSubReg = RegSrc->getSubReg();
4111 Src0->setReg(SrcReg);
4112 Src0->setSubReg(SrcSubReg);
4113 Src0->setIsKill(RegSrc->isKill());
4114
4115 if (Opc == AMDGPU::V_MAC_F32_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4116 Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4117 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4118 Opc == AMDGPU::V_FMAC_F16_e64 || Opc == AMDGPU::V_FMAC_F64_e64)
4119 UseMI.untieRegOperand(
4120 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2));
4121
4122 Src1->ChangeToImmediate(*SubRegImm);
4123
4125 UseMI.setDesc(get(NewOpc));
4126
4127 if (NewOpc == AMDGPU::V_FMAMK_F16_t16 ||
4128 NewOpc == AMDGPU::V_FMAMK_F16_fake16) {
4129 const TargetRegisterClass *NewRC = getRegClass(get(NewOpc), 0);
4130 Register Tmp = MRI->createVirtualRegister(NewRC);
4131 BuildMI(*UseMI.getParent(), std::next(UseMI.getIterator()),
4132 UseMI.getDebugLoc(), get(AMDGPU::COPY),
4133 UseMI.getOperand(0).getReg())
4134 .addReg(Tmp, RegState::Kill);
4135 UseMI.getOperand(0).setReg(Tmp);
4136 CopyRegOperandToNarrowerRC(UseMI, 1, NewRC);
4137 CopyRegOperandToNarrowerRC(UseMI, 3, NewRC);
4138 }
4139
4140 bool DeleteDef = MRI->use_nodbg_empty(Reg);
4141 if (DeleteDef)
4142 DefMI.eraseFromParent();
4143
4144 return true;
4145 }
4146
4147 // Added part is the constant: Use v_madak_{f16, f32}.
4148 if (Src2->isReg() && Src2->getReg() == Reg) {
4149 if (ST.getConstantBusLimit(Opc) < 2) {
4150 // Not allowed to use constant bus for another operand.
4151 // We can however allow an inline immediate as src0.
4152 bool Src0Inlined = false;
4153 if (Src0->isReg()) {
4154 // Try to inline constant if possible.
4155 // If the Def moves immediate and the use is single
4156 // We are saving VGPR here.
4157 MachineInstr *Def = MRI->getUniqueVRegDef(Src0->getReg());
4158 if (Def && Def->isMoveImmediate() &&
4159 isInlineConstant(Def->getOperand(1)) &&
4160 MRI->hasOneNonDBGUse(Src0->getReg())) {
4161 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4162 Src0Inlined = true;
4163 } else if (ST.getConstantBusLimit(Opc) <= 1 &&
4164 RI.isSGPRReg(*MRI, Src0->getReg())) {
4165 return false;
4166 }
4167 // VGPR is okay as Src0 - fallthrough
4168 }
4169
4170 if (Src1->isReg() && !Src0Inlined) {
4171 // We have one slot for inlinable constant so far - try to fill it
4172 MachineInstr *Def = MRI->getUniqueVRegDef(Src1->getReg());
4173 if (Def && Def->isMoveImmediate() &&
4174 isInlineConstant(Def->getOperand(1)) &&
4175 MRI->hasOneNonDBGUse(Src1->getReg()) && commuteInstruction(UseMI))
4176 Src0->ChangeToImmediate(Def->getOperand(1).getImm());
4177 else if (RI.isSGPRReg(*MRI, Src1->getReg()))
4178 return false;
4179 // VGPR is okay as Src1 - fallthrough
4180 }
4181 }
4182
4183 unsigned NewOpc = getNewFMAAKInst(ST, Opc);
4184 if (pseudoToMCOpcode(NewOpc) == -1)
4185 return false;
4186
4187 // FIXME: This would be a lot easier if we could return a new instruction
4188 // instead of having to modify in place.
4189
4190 if (Opc == AMDGPU::V_MAC_F32_e64 || Opc == AMDGPU::V_MAC_F16_e64 ||
4191 Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_t16_e64 ||
4192 Opc == AMDGPU::V_FMAC_F16_fake16_e64 ||
4193 Opc == AMDGPU::V_FMAC_F16_e64 || Opc == AMDGPU::V_FMAC_F64_e64)
4194 UseMI.untieRegOperand(
4195 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2));
4196
4197 const std::optional<int64_t> SubRegImm =
4198 extractSubregFromImm(Imm, Src2->getSubReg());
4199
4200 // ChangingToImmediate adds Src2 back to the instruction.
4201 Src2->ChangeToImmediate(*SubRegImm);
4202
4203 // These come before src2.
4205 UseMI.setDesc(get(NewOpc));
4206
4207 if (NewOpc == AMDGPU::V_FMAAK_F16_t16 ||
4208 NewOpc == AMDGPU::V_FMAAK_F16_fake16) {
4209 const TargetRegisterClass *NewRC = getRegClass(get(NewOpc), 0);
4210 Register Tmp = MRI->createVirtualRegister(NewRC);
4211 BuildMI(*UseMI.getParent(), std::next(UseMI.getIterator()),
4212 UseMI.getDebugLoc(), get(AMDGPU::COPY),
4213 UseMI.getOperand(0).getReg())
4214 .addReg(Tmp, RegState::Kill);
4215 UseMI.getOperand(0).setReg(Tmp);
4216 CopyRegOperandToNarrowerRC(UseMI, 1, NewRC);
4217 CopyRegOperandToNarrowerRC(UseMI, 2, NewRC);
4218 }
4219
4220 // It might happen that UseMI was commuted
4221 // and we now have SGPR as SRC1. If so 2 inlined
4222 // constant and SGPR are illegal.
4224
4225 int NewSrc0Idx =
4226 AMDGPU::getNamedOperandIdx(UseMI.getOpcode(), AMDGPU::OpName::src0);
4227 if (!isOperandLegal(UseMI, NewSrc0Idx))
4228 legalizeOpWithMove(UseMI, NewSrc0Idx);
4229
4230 bool DeleteDef = MRI->use_nodbg_empty(Reg);
4231 if (DeleteDef)
4232 DefMI.eraseFromParent();
4233
4234 return true;
4235 }
4236 }
4237
4238 return false;
4239}
4240
4241static bool
4244 if (BaseOps1.size() != BaseOps2.size())
4245 return false;
4246 for (size_t I = 0, E = BaseOps1.size(); I < E; ++I) {
4247 if (!BaseOps1[I]->isIdenticalTo(*BaseOps2[I]))
4248 return false;
4249 }
4250 return true;
4251}
4252
4253static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA,
4254 LocationSize WidthB, int OffsetB) {
4255 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
4256 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
4257 LocationSize LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
4258 return LowWidth.hasValue() &&
4259 LowOffset + (int)LowWidth.getValue() <= HighOffset;
4260}
4261
4262bool SIInstrInfo::checkInstOffsetsDoNotOverlap(const MachineInstr &MIa,
4263 const MachineInstr &MIb) const {
4264 SmallVector<const MachineOperand *, 4> BaseOps0, BaseOps1;
4265 int64_t Offset0, Offset1;
4266 LocationSize Dummy0 = LocationSize::precise(0);
4267 LocationSize Dummy1 = LocationSize::precise(0);
4268 bool Offset0IsScalable, Offset1IsScalable;
4269 if (!getMemOperandsWithOffsetWidth(MIa, BaseOps0, Offset0, Offset0IsScalable,
4270 Dummy0) ||
4271 !getMemOperandsWithOffsetWidth(MIb, BaseOps1, Offset1, Offset1IsScalable,
4272 Dummy1))
4273 return false;
4274
4275 if (!memOpsHaveSameBaseOperands(BaseOps0, BaseOps1))
4276 return false;
4277
4278 if (!MIa.hasOneMemOperand() || !MIb.hasOneMemOperand()) {
4279 // FIXME: Handle ds_read2 / ds_write2.
4280 return false;
4281 }
4282 LocationSize Width0 = MIa.memoperands().front()->getSize();
4283 LocationSize Width1 = MIb.memoperands().front()->getSize();
4284 return offsetsDoNotOverlap(Width0, Offset0, Width1, Offset1);
4285}
4286
4288 const MachineInstr &MIb) const {
4289 assert(MIa.mayLoadOrStore() &&
4290 "MIa must load from or modify a memory location");
4291 assert(MIb.mayLoadOrStore() &&
4292 "MIb must load from or modify a memory location");
4293
4295 return false;
4296
4297 // XXX - Can we relax this between address spaces?
4298 if (MIa.hasOrderedMemoryRef() || MIb.hasOrderedMemoryRef())
4299 return false;
4300
4301 if (isLDSDMA(MIa) || isLDSDMA(MIb))
4302 return false;
4303
4304 if (MIa.isBundle() || MIb.isBundle())
4305 return false;
4306
4307 // TODO: Should we check the address space from the MachineMemOperand? That
4308 // would allow us to distinguish objects we know don't alias based on the
4309 // underlying address space, even if it was lowered to a different one,
4310 // e.g. private accesses lowered to use MUBUF instructions on a scratch
4311 // buffer.
4312 if (isDS(MIa)) {
4313 if (isDS(MIb))
4314 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4315
4316 return !isFLAT(MIb) || isSegmentSpecificFLAT(MIb);
4317 }
4318
4319 if (isMUBUF(MIa) || isMTBUF(MIa)) {
4320 if (isMUBUF(MIb) || isMTBUF(MIb))
4321 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4322
4323 if (isFLAT(MIb))
4324 return isFLATScratch(MIb);
4325
4326 return !isSMRD(MIb);
4327 }
4328
4329 if (isSMRD(MIa)) {
4330 if (isSMRD(MIb))
4331 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4332
4333 if (isFLAT(MIb))
4334 return isFLATScratch(MIb);
4335
4336 return !isMUBUF(MIb) && !isMTBUF(MIb);
4337 }
4338
4339 if (isFLAT(MIa)) {
4340 if (isFLAT(MIb)) {
4341 if ((isFLATScratch(MIa) && isFLATGlobal(MIb)) ||
4342 (isFLATGlobal(MIa) && isFLATScratch(MIb)))
4343 return true;
4344
4345 return checkInstOffsetsDoNotOverlap(MIa, MIb);
4346 }
4347
4348 return false;
4349 }
4350
4351 return false;
4352}
4353
4354static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc) {
4355 switch (Opc) {
4356 case AMDGPU::V_MAC_F16_e32:
4357 case AMDGPU::V_MAC_F16_e64:
4358 return AMDGPU::V_MAD_F16_e64;
4359 case AMDGPU::V_MAC_F32_e32:
4360 case AMDGPU::V_MAC_F32_e64:
4361 return AMDGPU::V_MAD_F32_e64;
4362 case AMDGPU::V_MAC_LEGACY_F32_e32:
4363 case AMDGPU::V_MAC_LEGACY_F32_e64:
4364 return AMDGPU::V_MAD_LEGACY_F32_e64;
4365 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4366 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4367 return AMDGPU::V_FMA_LEGACY_F32_e64;
4368 case AMDGPU::V_FMAC_F16_e32:
4369 case AMDGPU::V_FMAC_F16_e64:
4370 case AMDGPU::V_FMAC_F16_t16_e64:
4371 case AMDGPU::V_FMAC_F16_fake16_e64:
4372 return ST.hasTrue16BitInsts() ? ST.useRealTrue16Insts()
4373 ? AMDGPU::V_FMA_F16_gfx9_t16_e64
4374 : AMDGPU::V_FMA_F16_gfx9_fake16_e64
4375 : AMDGPU::V_FMA_F16_gfx9_e64;
4376 case AMDGPU::V_FMAC_F32_e32:
4377 case AMDGPU::V_FMAC_F32_e64:
4378 return AMDGPU::V_FMA_F32_e64;
4379 case AMDGPU::V_FMAC_F64_e32:
4380 case AMDGPU::V_FMAC_F64_e64:
4381 return AMDGPU::V_FMA_F64_e64;
4382 default:
4383 llvm_unreachable("invalid instruction");
4384 }
4385}
4386
4387/// Helper struct for the implementation of 3-address conversion to communicate
4388/// updates made to instruction operands.
4390 /// Other instruction whose def is no longer used by the converted
4391 /// instruction.
4393};
4394
4396 LiveIntervals *LIS) const {
4397 MachineBasicBlock &MBB = *MI.getParent();
4398 MachineInstr *CandidateMI = &MI;
4399
4400 if (MI.isBundle()) {
4401 // This is a temporary placeholder for bundle handling that enables us to
4402 // exercise the relevant code paths in the two-address instruction pass.
4403 if (MI.getBundleSize() != 1)
4404 return nullptr;
4405 CandidateMI = MI.getNextNode();
4406 }
4407
4409 MachineInstr *NewMI = convertToThreeAddressImpl(*CandidateMI, U);
4410 if (!NewMI)
4411 return nullptr;
4412
4413 if (MI.isBundle()) {
4414 CandidateMI->eraseFromBundle();
4415
4416 for (MachineOperand &MO : MI.all_defs()) {
4417 if (MO.isTied())
4418 MI.untieRegOperand(MO.getOperandNo());
4419 }
4420 } else {
4421 if (LIS) {
4422 LIS->ReplaceMachineInstrInMaps(MI, *NewMI);
4423 // SlotIndex of defs needs to be updated when converting to early-clobber
4424 MachineOperand &Def = NewMI->getOperand(0);
4425 if (Def.isEarlyClobber() && Def.isReg() &&
4426 LIS->hasInterval(Def.getReg())) {
4427 SlotIndex OldIndex = LIS->getInstructionIndex(*NewMI).getRegSlot(false);
4428 SlotIndex NewIndex = LIS->getInstructionIndex(*NewMI).getRegSlot(true);
4429 auto &LI = LIS->getInterval(Def.getReg());
4430 auto UpdateDefIndex = [&](LiveRange &LR) {
4431 auto *S = LR.find(OldIndex);
4432 if (S != LR.end() && S->start == OldIndex) {
4433 assert(S->valno && S->valno->def == OldIndex);
4434 S->start = NewIndex;
4435 S->valno->def = NewIndex;
4436 }
4437 };
4438 UpdateDefIndex(LI);
4439 for (auto &SR : LI.subranges())
4440 UpdateDefIndex(SR);
4441 }
4442 }
4443 }
4444
4445 if (U.RemoveMIUse) {
4446 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4447 // The only user is the instruction which will be killed.
4448 Register DefReg = U.RemoveMIUse->getOperand(0).getReg();
4449
4450 if (MRI.hasOneNonDBGUse(DefReg)) {
4451 // We cannot just remove the DefMI here, calling pass will crash.
4452 U.RemoveMIUse->setDesc(get(AMDGPU::IMPLICIT_DEF));
4453 U.RemoveMIUse->getOperand(0).setIsDead(true);
4454 for (unsigned I = U.RemoveMIUse->getNumOperands() - 1; I != 0; --I)
4455 U.RemoveMIUse->removeOperand(I);
4456 }
4457
4458 if (MI.isBundle()) {
4459 VirtRegInfo VRI = AnalyzeVirtRegInBundle(MI, DefReg);
4460 if (!VRI.Reads && !VRI.Writes) {
4461 for (MachineOperand &MO : MI.all_uses()) {
4462 if (MO.isReg() && MO.getReg() == DefReg) {
4463 assert(MO.getSubReg() == 0 &&
4464 "tied sub-registers in bundles currently not supported");
4465 MI.removeOperand(MO.getOperandNo());
4466 break;
4467 }
4468 }
4469
4470 if (LIS)
4471 LIS->shrinkToUses(&LIS->getInterval(DefReg));
4472 }
4473 } else if (LIS) {
4474 LiveInterval &DefLI = LIS->getInterval(DefReg);
4475
4476 // We cannot delete the original instruction here, so hack out the use
4477 // in the original instruction with a dummy register so we can use
4478 // shrinkToUses to deal with any multi-use edge cases. Other targets do
4479 // not have the complexity of deleting a use to consider here.
4480 Register DummyReg = MRI.cloneVirtualRegister(DefReg);
4481 for (MachineOperand &MIOp : MI.uses()) {
4482 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4483 MIOp.setIsUndef(true);
4484 MIOp.setReg(DummyReg);
4485 }
4486 }
4487
4488 if (MI.isBundle()) {
4489 VirtRegInfo VRI = AnalyzeVirtRegInBundle(MI, DefReg);
4490 if (!VRI.Reads && !VRI.Writes) {
4491 for (MachineOperand &MIOp : MI.uses()) {
4492 if (MIOp.isReg() && MIOp.getReg() == DefReg) {
4493 MIOp.setIsUndef(true);
4494 MIOp.setReg(DummyReg);
4495 }
4496 }
4497 }
4498
4499 MI.addOperand(MachineOperand::CreateReg(DummyReg, false, false, false,
4500 false, /*isUndef=*/true));
4501 }
4502
4503 LIS->shrinkToUses(&DefLI);
4504 }
4505 }
4506
4507 return MI.isBundle() ? &MI : NewMI;
4508}
4509
4511SIInstrInfo::convertToThreeAddressImpl(MachineInstr &MI,
4512 ThreeAddressUpdates &U) const {
4513 MachineBasicBlock &MBB = *MI.getParent();
4514 unsigned Opc = MI.getOpcode();
4515
4516 // Handle MFMA.
4517 int NewMFMAOpc = AMDGPU::getMFMAEarlyClobberOp(Opc);
4518 if (NewMFMAOpc != -1) {
4520 BuildMI(MBB, MI, MI.getDebugLoc(), get(NewMFMAOpc));
4521 for (unsigned I = 0, E = MI.getNumExplicitOperands(); I != E; ++I)
4522 MIB.add(MI.getOperand(I));
4523 return MIB;
4524 }
4525
4526 if (SIInstrInfo::isWMMA(MI)) {
4527 unsigned NewOpc = AMDGPU::mapWMMA2AddrTo3AddrOpcode(MI.getOpcode());
4528 MachineInstrBuilder MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4529 .setMIFlags(MI.getFlags());
4530 for (unsigned I = 0, E = MI.getNumExplicitOperands(); I != E; ++I)
4531 MIB->addOperand(MI.getOperand(I));
4532 return MIB;
4533 }
4534
4535 assert(Opc != AMDGPU::V_FMAC_F16_t16_e32 &&
4536 Opc != AMDGPU::V_FMAC_F16_fake16_e32 &&
4537 "V_FMAC_F16_t16/fake16_e32 is not supported and not expected to be "
4538 "present pre-RA");
4539
4540 // Handle MAC/FMAC.
4541 bool IsF64 = Opc == AMDGPU::V_FMAC_F64_e32 || Opc == AMDGPU::V_FMAC_F64_e64;
4542 bool IsLegacy = Opc == AMDGPU::V_MAC_LEGACY_F32_e32 ||
4543 Opc == AMDGPU::V_MAC_LEGACY_F32_e64 ||
4544 Opc == AMDGPU::V_FMAC_LEGACY_F32_e32 ||
4545 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64;
4546 bool Src0Literal = false;
4547
4548 switch (Opc) {
4549 default:
4550 return nullptr;
4551 case AMDGPU::V_MAC_F16_e64:
4552 case AMDGPU::V_FMAC_F16_e64:
4553 case AMDGPU::V_FMAC_F16_t16_e64:
4554 case AMDGPU::V_FMAC_F16_fake16_e64:
4555 case AMDGPU::V_MAC_F32_e64:
4556 case AMDGPU::V_MAC_LEGACY_F32_e64:
4557 case AMDGPU::V_FMAC_F32_e64:
4558 case AMDGPU::V_FMAC_LEGACY_F32_e64:
4559 case AMDGPU::V_FMAC_F64_e64:
4560 break;
4561 case AMDGPU::V_MAC_F16_e32:
4562 case AMDGPU::V_FMAC_F16_e32:
4563 case AMDGPU::V_MAC_F32_e32:
4564 case AMDGPU::V_MAC_LEGACY_F32_e32:
4565 case AMDGPU::V_FMAC_F32_e32:
4566 case AMDGPU::V_FMAC_LEGACY_F32_e32:
4567 case AMDGPU::V_FMAC_F64_e32: {
4568 int Src0Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
4569 AMDGPU::OpName::src0);
4570 const MachineOperand *Src0 = &MI.getOperand(Src0Idx);
4571 if (!Src0->isReg() && !Src0->isImm())
4572 return nullptr;
4573
4574 if (Src0->isImm() && !isInlineConstant(MI, Src0Idx, *Src0))
4575 Src0Literal = true;
4576
4577 break;
4578 }
4579 }
4580
4581 MachineInstrBuilder MIB;
4582 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
4583 const MachineOperand *Src0 = getNamedOperand(MI, AMDGPU::OpName::src0);
4584 const MachineOperand *Src0Mods =
4585 getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
4586 const MachineOperand *Src1 = getNamedOperand(MI, AMDGPU::OpName::src1);
4587 const MachineOperand *Src1Mods =
4588 getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
4589 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
4590 const MachineOperand *Src2Mods =
4591 getNamedOperand(MI, AMDGPU::OpName::src2_modifiers);
4592 const MachineOperand *Clamp = getNamedOperand(MI, AMDGPU::OpName::clamp);
4593 const MachineOperand *Omod = getNamedOperand(MI, AMDGPU::OpName::omod);
4594 const MachineOperand *OpSel = getNamedOperand(MI, AMDGPU::OpName::op_sel);
4595
4596 if (!Src0Mods && !Src1Mods && !Src2Mods && !Clamp && !Omod && !IsLegacy &&
4597 (!IsF64 || ST.hasFmaakFmamkF64Insts()) &&
4598 // If we have an SGPR input, we will violate the constant bus restriction.
4599 (ST.getConstantBusLimit(Opc) > 1 || !Src0->isReg() ||
4600 !RI.isSGPRReg(MBB.getParent()->getRegInfo(), Src0->getReg()))) {
4601 MachineInstr *DefMI = nullptr;
4602 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4603 std::optional<int64_t> ImmOpt;
4604 int64_t Imm;
4605
4606 if (!Src0Literal &&
4607 (ImmOpt = getImmOrMaterializedImm(MRI, *Src2, &DefMI))) {
4608 unsigned NewOpc = getNewFMAAKInst(ST, Opc);
4609 if (pseudoToMCOpcode(NewOpc) != -1) {
4610 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4611 .add(*Dst)
4612 .add(*Src0)
4613 .add(*Src1)
4614 .addImm(*ImmOpt)
4615 .setMIFlags(MI.getFlags());
4616 U.RemoveMIUse = DefMI;
4617 return MIB;
4618 }
4619 }
4620 unsigned NewOpc = getNewFMAMKInst(ST, Opc);
4621 if (!Src0Literal &&
4622 (ImmOpt = getImmOrMaterializedImm(MRI, *Src1, &DefMI))) {
4623 if (pseudoToMCOpcode(NewOpc) != -1) {
4624 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4625 .add(*Dst)
4626 .add(*Src0)
4627 .addImm(*ImmOpt)
4628 .add(*Src2)
4629 .setMIFlags(MI.getFlags());
4630 U.RemoveMIUse = DefMI;
4631 return MIB;
4632 }
4633 }
4634 if ((ImmOpt = getImmOrMaterializedImm(MRI, *Src0, &DefMI))) {
4635 Imm = *ImmOpt;
4636 if (pseudoToMCOpcode(NewOpc) != -1 &&
4638 MI, AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::src0),
4639 Src1)) {
4640 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4641 .add(*Dst)
4642 .add(*Src1)
4643 .addImm(Imm)
4644 .add(*Src2)
4645 .setMIFlags(MI.getFlags());
4646 U.RemoveMIUse = DefMI;
4647 return MIB;
4648 }
4649 }
4650 }
4651
4652 // VOP2 mac/fmac with a literal operand cannot be converted to VOP3 mad/fma
4653 // if VOP3 does not allow a literal operand.
4654 if (Src0Literal && !ST.hasVOP3Literal())
4655 return nullptr;
4656
4657 unsigned NewOpc = getNewFMAInst(ST, Opc);
4658
4659 if (pseudoToMCOpcode(NewOpc) == -1)
4660 return nullptr;
4661
4662 MIB = BuildMI(MBB, MI, MI.getDebugLoc(), get(NewOpc))
4663 .add(*Dst)
4664 .addImm(Src0Mods ? Src0Mods->getImm() : 0)
4665 .add(*Src0)
4666 .addImm(Src1Mods ? Src1Mods->getImm() : 0)
4667 .add(*Src1)
4668 .addImm(Src2Mods ? Src2Mods->getImm() : 0)
4669 .add(*Src2)
4670 .addImm(Clamp ? Clamp->getImm() : 0)
4671 .addImm(Omod ? Omod->getImm() : 0)
4672 .setMIFlags(MI.getFlags());
4673 if (AMDGPU::hasNamedOperand(NewOpc, AMDGPU::OpName::op_sel))
4674 MIB.addImm(OpSel ? OpSel->getImm() : 0);
4675 return MIB;
4676}
4677
4678// It's not generally safe to move VALU instructions across these since it will
4679// start using the register as a base index rather than directly.
4680// XXX - Why isn't hasSideEffects sufficient for these?
4682 switch (MI.getOpcode()) {
4683 case AMDGPU::S_SET_GPR_IDX_ON:
4684 case AMDGPU::S_SET_GPR_IDX_MODE:
4685 case AMDGPU::S_SET_GPR_IDX_OFF:
4686 return true;
4687 default:
4688 return false;
4689 }
4690}
4691
4693 const MachineBasicBlock *MBB,
4694 const MachineFunction &MF) const {
4695 // Skipping the check for SP writes in the base implementation. The reason it
4696 // was added was apparently due to compile time concerns.
4697 //
4698 // TODO: Do we really want this barrier? It triggers unnecessary hazard nops
4699 // but is probably avoidable.
4700
4701 // Copied from base implementation.
4702 // Terminators and labels can't be scheduled around.
4703 if (MI.isTerminator() || MI.isPosition())
4704 return true;
4705
4706 // INLINEASM_BR can jump to another block
4707 if (MI.getOpcode() == TargetOpcode::INLINEASM_BR)
4708 return true;
4709
4710 if (MI.getOpcode() == AMDGPU::SCHED_BARRIER && MI.getOperand(0).getImm() == 0)
4711 return true;
4712
4713 // Target-independent instructions do not have an implicit-use of EXEC, even
4714 // when they operate on VGPRs. Treating EXEC modifications as scheduling
4715 // boundaries prevents incorrect movements of such instructions.
4716 return MI.modifiesRegister(AMDGPU::EXEC, &RI) ||
4717 MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32 ||
4718 MI.getOpcode() == AMDGPU::S_SETREG_B32 ||
4719 MI.getOpcode() == AMDGPU::S_SETPRIO ||
4720 MI.getOpcode() == AMDGPU::S_SETPRIO_INC_WG ||
4722}
4723
4725 return Opcode == AMDGPU::DS_ORDERED_COUNT ||
4726 Opcode == AMDGPU::DS_ADD_GS_REG_RTN ||
4727 Opcode == AMDGPU::DS_SUB_GS_REG_RTN || isGWS(Opcode);
4728}
4729
4731 // Instructions that access scratch use FLAT encoding or BUF encodings.
4732 if ((!isFLAT(MI) || isFLATGlobal(MI)) && !isBUF(MI))
4733 return false;
4734
4735 // SCRATCH instructions always access scratch.
4736 if (isFLATScratch(MI))
4737 return true;
4738
4739 // If FLAT_SCRATCH registers are not initialized, we can never access scratch
4740 // via the aperture.
4741 if (MI.getMF()->getFunction().hasFnAttribute("amdgpu-no-flat-scratch-init"))
4742 return false;
4743
4744 // If there are no memory operands then conservatively assume the flat
4745 // operation may access scratch.
4746 if (MI.memoperands_empty())
4747 return true;
4748
4749 // See if any memory operand specifies an address space that involves scratch.
4750 return any_of(MI.memoperands(), [](const MachineMemOperand *Memop) {
4751 unsigned AS = Memop->getAddrSpace();
4752 if (AS == AMDGPUAS::FLAT_ADDRESS) {
4753 const MDNode *MD = Memop->getAAInfo().NoAliasAddrSpace;
4754 return !MD || !AMDGPU::hasValueInRangeLikeMetadata(
4755 *MD, AMDGPUAS::PRIVATE_ADDRESS);
4756 }
4757 return AS == AMDGPUAS::PRIVATE_ADDRESS;
4758 });
4759}
4760
4762 assert(isFLAT(MI));
4763
4764 // All flat instructions use the VMEM counter except prefetch.
4765 if (!usesVM_CNT(MI))
4766 return false;
4767
4768 // If there are no memory operands then conservatively assume the flat
4769 // operation may access VMEM.
4770 if (MI.memoperands_empty())
4771 return true;
4772
4773 // See if any memory operand specifies an address space that involves VMEM.
4774 // Flat operations only supported FLAT, LOCAL (LDS), or address spaces
4775 // involving VMEM such as GLOBAL, CONSTANT, PRIVATE (SCRATCH), etc. The REGION
4776 // (GDS) address space is not supported by flat operations. Therefore, simply
4777 // return true unless only the LDS address space is found.
4778 for (const MachineMemOperand *Memop : MI.memoperands()) {
4779 unsigned AS = Memop->getAddrSpace();
4781 if (AS != AMDGPUAS::LOCAL_ADDRESS)
4782 return true;
4783 }
4784
4785 return false;
4786}
4787
4789 bool TgSplit) const {
4790 assert(isFLAT(MI));
4791
4792 // Flat instruction such as SCRATCH and GLOBAL do not use the lgkm counter.
4793 if (!usesLGKM_CNT(MI))
4794 return false;
4795
4796 // If in tgsplit mode then there can be no use of LDS.
4797 if (TgSplit)
4798 return false;
4799
4800 // If there are no memory operands then conservatively assume the flat
4801 // operation may access LDS.
4802 if (MI.memoperands_empty())
4803 return true;
4804
4805 // See if any memory operand specifies an address space that involves LDS.
4806 for (const MachineMemOperand *Memop : MI.memoperands()) {
4807 unsigned AS = Memop->getAddrSpace();
4809 return true;
4810 }
4811
4812 return false;
4813}
4814
4816 // Skip the full operand and register alias search modifiesRegister
4817 // does. There's only a handful of instructions that touch this, it's only an
4818 // implicit def, and doesn't alias any other registers.
4819 return is_contained(MI.getDesc().implicit_defs(), AMDGPU::MODE);
4820}
4821
4823 unsigned Opcode = MI.getOpcode();
4824
4825 if (MI.mayStore() && isSMRD(MI))
4826 return true; // scalar store or atomic
4827
4828 // This will terminate the function when other lanes may need to continue.
4829 if (MI.isReturn())
4830 return true;
4831
4832 // These instructions cause shader I/O that may cause hardware lockups
4833 // when executed with an empty EXEC mask.
4834 //
4835 // Note: exp with VM = DONE = 0 is automatically skipped by hardware when
4836 // EXEC = 0, but checking for that case here seems not worth it
4837 // given the typical code patterns.
4838 if (Opcode == AMDGPU::S_SENDMSG || Opcode == AMDGPU::S_SENDMSGHALT ||
4839 isEXP(Opcode) || Opcode == AMDGPU::DS_ORDERED_COUNT ||
4840 Opcode == AMDGPU::S_TRAP || Opcode == AMDGPU::S_WAIT_EVENT ||
4841 Opcode == AMDGPU::S_SETHALT)
4842 return true;
4843
4844 if (MI.isCall() || MI.isInlineAsm())
4845 return true; // conservative assumption
4846
4847 // V_PERM_PK16 must issue with EXEC != 0 so its follower (or an inserted
4848 // V_NOP) actually runs on the VALU pipe. Returning true here keeps the
4849 // s_cbranch_execz that skips this region when EXEC is empty.
4850 if (ST.hasVPermPk16Hazard() && isVPermPk16(Opcode))
4851 return true;
4852
4853 // Assume that barrier interactions are only intended with active lanes.
4854 if (isBarrier(Opcode))
4855 return true;
4856
4857 // A mode change is a scalar operation that influences vector instructions.
4859 return true;
4860
4861 // These are like SALU instructions in terms of effects, so it's questionable
4862 // whether we should return true for those.
4863 //
4864 // However, executing them with EXEC = 0 causes them to operate on undefined
4865 // data, which we avoid by returning true here.
4866 if (Opcode == AMDGPU::V_READFIRSTLANE_B32 ||
4867 Opcode == AMDGPU::V_READLANE_B32 || Opcode == AMDGPU::V_WRITELANE_B32 ||
4868 Opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR ||
4869 Opcode == AMDGPU::SI_SPILL_S32_TO_VGPR)
4870 return true;
4871
4872 return false;
4873}
4874
4876 const MachineInstr &MI) const {
4877 if (MI.isMetaInstruction())
4878 return false;
4879
4880 // This won't read exec if this is an SGPR->SGPR copy.
4881 if (MI.isCopyLike()) {
4882 if (!RI.isSGPRReg(MRI, MI.getOperand(0).getReg()))
4883 return true;
4884
4885 // Make sure this isn't copying exec as a normal operand
4886 return MI.readsRegister(AMDGPU::EXEC, &RI);
4887 }
4888
4889 // Make a conservative assumption about the callee.
4890 if (MI.isCall())
4891 return true;
4892
4893 // Be conservative with any unhandled generic opcodes.
4894 if (!isTargetSpecificOpcode(MI.getOpcode()))
4895 return true;
4896
4897 return !isSALU(MI) || MI.readsRegister(AMDGPU::EXEC, &RI);
4898}
4899
4901 switch (Imm.getBitWidth()) {
4902 case 1: // This likely will be a condition code mask.
4903 return true;
4904
4905 case 32:
4906 return AMDGPU::isInlinableLiteral32(Imm.getSExtValue(),
4907 ST.hasInv2PiInlineImm());
4908 case 64:
4909 return AMDGPU::isInlinableLiteral64(Imm.getSExtValue(),
4910 ST.hasInv2PiInlineImm());
4911 case 16:
4912 return ST.has16BitInsts() &&
4913 AMDGPU::isInlinableLiteralI16(Imm.getSExtValue(),
4914 ST.hasInv2PiInlineImm());
4915 default:
4916 llvm_unreachable("invalid bitwidth");
4917 }
4918}
4919
4921 APInt IntImm = Imm.bitcastToAPInt();
4922 int64_t IntImmVal = IntImm.getSExtValue();
4923 bool HasInv2Pi = ST.hasInv2PiInlineImm();
4924 switch (APFloat::SemanticsToEnum(Imm.getSemantics())) {
4925 default:
4926 llvm_unreachable("invalid fltSemantics");
4929 return isInlineConstant(IntImm);
4931 return ST.has16BitInsts() &&
4932 AMDGPU::isInlinableLiteralBF16(IntImmVal, HasInv2Pi);
4934 return ST.has16BitInsts() &&
4935 AMDGPU::isInlinableLiteralFP16(IntImmVal, HasInv2Pi);
4936 }
4937}
4938
4939bool SIInstrInfo::isInlineConstant(int64_t Imm, uint8_t OperandType) const {
4940 // MachineOperand provides no way to tell the true operand size, since it only
4941 // records a 64-bit value. We need to know the size to determine if a 32-bit
4942 // floating point immediate bit pattern is legal for an integer immediate. It
4943 // would be for any 32-bit integer operand, but would not be for a 64-bit one.
4944 switch (OperandType) {
4954 int32_t Trunc = static_cast<int32_t>(Imm);
4955 return AMDGPU::isInlinableLiteral32(Trunc, ST.hasInv2PiInlineImm());
4956 }
4964 return AMDGPU::isInlinableLiteral64(Imm, ST.hasInv2PiInlineImm());
4967 // We would expect inline immediates to not be concerned with an integer/fp
4968 // distinction. However, in the case of 16-bit integer operations, the
4969 // "floating point" values appear to not work. It seems read the low 16-bits
4970 // of 32-bit immediates, which happens to always work for the integer
4971 // values.
4972 //
4973 // See llvm bugzilla 46302.
4974 //
4975 // TODO: Theoretically we could use op-sel to use the high bits of the
4976 // 32-bit FP values.
4985 return AMDGPU::isPKFMACF16InlineConstant(Imm, ST.isGFX11Plus());
4990 return false;
4993 if (isInt<16>(Imm) || isUInt<16>(Imm)) {
4994 // A few special case instructions have 16-bit operands on subtargets
4995 // where 16-bit instructions are not legal.
4996 // TODO: Do the 32-bit immediates work? We shouldn't really need to handle
4997 // constants in these cases
4998 int16_t Trunc = static_cast<int16_t>(Imm);
4999 return ST.has16BitInsts() &&
5000 AMDGPU::isInlinableLiteralFP16(Trunc, ST.hasInv2PiInlineImm());
5001 }
5002
5003 return false;
5004 }
5007 if (isInt<16>(Imm) || isUInt<16>(Imm)) {
5008 int16_t Trunc = static_cast<int16_t>(Imm);
5009 return ST.has16BitInsts() &&
5010 AMDGPU::isInlinableLiteralBF16(Trunc, ST.hasInv2PiInlineImm());
5011 }
5012 return false;
5013 }
5018 return false;
5020 return isLegalAV64PseudoImm(Imm);
5023 // Always embedded in the instruction for free.
5024 return true;
5034 // Just ignore anything else.
5035 return false;
5036 default:
5037 llvm_unreachable("invalid operand type");
5038 }
5039}
5040
5041static bool compareMachineOp(const MachineOperand &Op0,
5042 const MachineOperand &Op1) {
5043 if (Op0.getType() != Op1.getType())
5044 return false;
5045
5046 switch (Op0.getType()) {
5048 return Op0.getReg() == Op1.getReg();
5050 return Op0.getImm() == Op1.getImm();
5051 default:
5052 llvm_unreachable("Didn't expect to be comparing these operand types");
5053 }
5054}
5055
5057 const MCOperandInfo &OpInfo) const {
5058 if (OpInfo.OperandType == MCOI::OPERAND_IMMEDIATE)
5059 return true;
5060
5061 if (!RI.opCanUseLiteralConstant(OpInfo.OperandType))
5062 return false;
5063
5064 if (!isVOP3(InstDesc) || !AMDGPU::isSISrcOperand(OpInfo))
5065 return true;
5066
5067 return ST.hasVOP3Literal();
5068}
5069
5070bool SIInstrInfo::isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
5071 int64_t ImmVal) const {
5072 const unsigned Opc = InstDesc.getOpcode();
5073 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
5074 if (Src1Idx != -1 && isDPP(Opc) && !ST.hasDPPSrc1SGPR() &&
5075 OpNo == static_cast<unsigned>(Src1Idx))
5076 return false;
5077
5078 const MCOperandInfo &OpInfo = InstDesc.operands()[OpNo];
5079 if (isInlineConstant(ImmVal, OpInfo.OperandType)) {
5080 if (isMAI(InstDesc) && ST.hasMFMAInlineLiteralBug() &&
5081 OpNo == (unsigned)AMDGPU::getNamedOperandIdx(InstDesc.getOpcode(),
5082 AMDGPU::OpName::src2))
5083 return false;
5084
5085 if (ST.hasBF16InlineConstFromUpperFP32() && isVOP1(Opc)) {
5086 if ((OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_BF16 ||
5087 OpInfo.OperandType == AMDGPU::OPERAND_REG_INLINE_C_BF16) &&
5088 isInlineConstant(ImmVal, OpInfo.OperandType))
5089 return false;
5090 }
5091
5092 return RI.opCanUseInlineConstant(OpInfo.OperandType);
5093 }
5094
5095 return isLiteralOperandLegal(InstDesc, OpInfo);
5096}
5097
5098bool SIInstrInfo::isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo,
5099 const MachineOperand &MO) const {
5100 if (MO.isImm())
5101 return isImmOperandLegal(InstDesc, OpNo, MO.getImm());
5102
5103 assert((MO.isTargetIndex() || MO.isFI() || MO.isGlobal()) &&
5104 "unexpected imm-like operand kind");
5105 const MCOperandInfo &OpInfo = InstDesc.operands()[OpNo];
5106 return isLiteralOperandLegal(InstDesc, OpInfo);
5107}
5108
5110 // 2 32-bit inline constants packed into one.
5111 return AMDGPU::isInlinableLiteral32(Lo_32(Imm), ST.hasInv2PiInlineImm()) &&
5112 AMDGPU::isInlinableLiteral32(Hi_32(Imm), ST.hasInv2PiInlineImm());
5113}
5114
5115bool SIInstrInfo::hasVALU32BitEncoding(unsigned Opcode) const {
5116 // GFX90A does not have V_MUL_LEGACY_F32_e32.
5117 if (Opcode == AMDGPU::V_MUL_LEGACY_F32_e64 && ST.hasGFX90AInsts())
5118 return false;
5119
5120 int Op32 = AMDGPU::getVOPe32(Opcode);
5121 if (Op32 == -1)
5122 return false;
5123
5124 return pseudoToMCOpcode(Op32) != -1;
5125}
5126
5127/// Return true if \p MI is a VALU comparison, i.e. an instruction that writes
5128/// a lane mask with one bit per lane, and zeroes the bits of lanes that were
5129/// inactive when it executed.
5130///
5131/// TODO: Also handle the sdst result of V_ADD_CO_U32 and V_SUB_CO_U32 and
5132/// V_DIV_SCALE_F32.
5133static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI) {
5134 if (TII.isVOPC(MI))
5135 return true;
5136 int Op32 = AMDGPU::getVOPe32(MI.getOpcode());
5137 return Op32 != -1 && TII.isVOPC(Op32);
5138}
5139
5141 const MachineRegisterInfo &MRI,
5142 unsigned Depth) const {
5143 assert(MRI.isSSA() && "isMaskedByExec requires SSA form");
5145 const MachineBasicBlock *MBB = Use.getParent();
5146
5147 // EXEC itself is trivially masked by EXEC.
5148 if (Reg == LMC.ExecReg)
5149 return true;
5150
5151 // Maximum depth of the def-use walk.
5152 constexpr unsigned MaxDepth = 6;
5153 if (Depth >= MaxDepth || !Reg.isVirtual())
5154 return false;
5155
5156 // Only look at definitions that can execute under the same EXEC mask as the
5157 // use.
5158 const MachineInstr *Def = MRI.getVRegDef(Reg);
5159 if (!Def || Def->getParent() != MBB)
5160 return false;
5161
5162 if (isVCmp(*this, *Def))
5163 return true;
5164
5165 // Recurse into an operand, which must be a whole register to say anything
5166 // about the whole lane mask.
5167 auto Recurse = [&](unsigned OpIdx) {
5168 const MachineOperand &MO = Def->getOperand(OpIdx);
5169 return MO.isReg() && !MO.getSubReg() &&
5170 isMaskedByExec(MO.getReg(), Use, MRI, Depth + 1);
5171 };
5172
5173 unsigned Opc = Def->getOpcode();
5174 if (Opc == AMDGPU::COPY && Recurse(1))
5175 return true;
5176 if (Opc == LMC.AndOpc && (Recurse(1) || Recurse(2)))
5177 return true;
5178 if (Opc == LMC.AndN2Opc && Recurse(1))
5179 return true;
5180 if ((Opc == LMC.OrOpc || Opc == LMC.XorOpc) && Recurse(1) && Recurse(2))
5181 return true;
5182 // TODO: Sometimes we encounter "reg = S_CSELECT -1, 0". If Reg has no other
5183 // uses this could be optimized to "reg = S_CSELECT $exec, 0".
5184
5185 return false;
5186}
5187
5188bool SIInstrInfo::hasModifiers(unsigned Opcode) const {
5189 // The src0_modifier operand is present on all instructions
5190 // that have modifiers.
5191
5192 return AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src0_modifiers);
5193}
5194
5196 AMDGPU::OpName OpName) const {
5197 const MachineOperand *Mods = getNamedOperand(MI, OpName);
5198 return Mods && Mods->getImm();
5199}
5200
5202 return any_of(ModifierOpNames,
5203 [&](AMDGPU::OpName Name) { return hasModifiersSet(MI, Name); });
5204}
5205
5207 const MachineRegisterInfo &MRI) const {
5208 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
5209 // Can't shrink instruction with three operands.
5210 if (Src2) {
5211 switch (MI.getOpcode()) {
5212 default: return false;
5213
5214 case AMDGPU::V_ADDC_U32_e64:
5215 case AMDGPU::V_SUBB_U32_e64:
5216 case AMDGPU::V_SUBBREV_U32_e64: {
5217 const MachineOperand *Src1
5218 = getNamedOperand(MI, AMDGPU::OpName::src1);
5219 if (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()))
5220 return false;
5221 // Additional verification is needed for sdst/src2.
5222 return true;
5223 }
5224 case AMDGPU::V_MAC_F16_e64:
5225 case AMDGPU::V_MAC_F32_e64:
5226 case AMDGPU::V_MAC_LEGACY_F32_e64:
5227 case AMDGPU::V_FMAC_F16_e64:
5228 case AMDGPU::V_FMAC_F16_t16_e64:
5229 case AMDGPU::V_FMAC_F16_fake16_e64:
5230 case AMDGPU::V_FMAC_F32_e64:
5231 case AMDGPU::V_FMAC_F64_e64:
5232 case AMDGPU::V_FMAC_LEGACY_F32_e64:
5233 if (!Src2->isReg() || !RI.isVGPR(MRI, Src2->getReg()) ||
5234 hasModifiersSet(MI, AMDGPU::OpName::src2_modifiers))
5235 return false;
5236 break;
5237
5238 case AMDGPU::V_CNDMASK_B32_e64:
5239 break;
5240 }
5241 }
5242
5243 const MachineOperand *Src1 = getNamedOperand(MI, AMDGPU::OpName::src1);
5244 if (Src1 && (!Src1->isReg() || !RI.isVGPR(MRI, Src1->getReg()) ||
5245 hasModifiersSet(MI, AMDGPU::OpName::src1_modifiers)))
5246 return false;
5247
5248 // Make sure src0 isn't using any modifiers.
5249 if (hasModifiersSet(MI, AMDGPU::OpName::src0_modifiers))
5250 return false;
5251
5252 // Can it be shrunk to a valid 32 bit opcode?
5253 if (!hasVALU32BitEncoding(MI.getOpcode()))
5254 return false;
5255
5256 const MachineOperand *Src0 = getNamedOperand(MI, AMDGPU::OpName::src0);
5257 if (Src0 && Src0->isImm()) {
5258 unsigned Op32 = AMDGPU::getVOPe32(MI.getOpcode());
5259 if (!isImmOperandLegal(
5260 get(Op32), AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src0),
5261 *Src0))
5262 return false;
5263 }
5264
5265 // Check output modifiers
5266 return !hasModifiersSet(MI, AMDGPU::OpName::omod) &&
5267 !hasModifiersSet(MI, AMDGPU::OpName::clamp) &&
5268 !hasModifiersSet(MI, AMDGPU::OpName::byte_sel) &&
5269 // TODO: Can we avoid checking bound_ctrl/fi here?
5270 // They are only used by permlane*_swap special case.
5271 !hasModifiersSet(MI, AMDGPU::OpName::bound_ctrl) &&
5272 !hasModifiersSet(MI, AMDGPU::OpName::fi);
5273}
5274
5275// Set VCC operand with all flags from \p Orig, except for setting it as
5276// implicit.
5278 const MachineOperand &Orig) {
5279
5280 for (MachineOperand &Use : MI.implicit_operands()) {
5281 if (Use.isUse() &&
5282 (Use.getReg() == AMDGPU::VCC || Use.getReg() == AMDGPU::VCC_LO)) {
5283 Use.setIsUndef(Orig.isUndef());
5284 Use.setIsKill(Orig.isKill());
5285 return;
5286 }
5287 }
5288}
5289
5291 unsigned Op32) const {
5292 MachineBasicBlock *MBB = MI.getParent();
5293
5294 const MCInstrDesc &Op32Desc = get(Op32);
5295 MachineInstrBuilder Inst32 =
5296 BuildMI(*MBB, MI, MI.getDebugLoc(), Op32Desc)
5297 .setMIFlags(MI.getFlags());
5298
5299 // Add the dst operand if the 32-bit encoding also has an explicit $vdst.
5300 // For VOPC instructions, this is replaced by an implicit def of vcc.
5301
5302 // We assume the defs of the shrunk opcode are in the same order, and the
5303 // shrunk opcode loses the last def (SGPR def, in the VOP3->VOPC case).
5304 for (int I = 0, E = Op32Desc.getNumDefs(); I != E; ++I)
5305 Inst32.add(MI.getOperand(I));
5306
5307 const MachineOperand *Src2 = getNamedOperand(MI, AMDGPU::OpName::src2);
5308
5309 int Idx = MI.getNumExplicitDefs();
5310 for (const MachineOperand &Use : MI.explicit_uses()) {
5311 int OpTy = MI.getDesc().operands()[Idx++].OperandType;
5313 continue;
5314
5315 if (&Use == Src2) {
5316 if (AMDGPU::getNamedOperandIdx(Op32, AMDGPU::OpName::src2) == -1) {
5317 // In the case of V_CNDMASK_B32_e32, the explicit operand src2 is
5318 // replaced with an implicit read of vcc or vcc_lo. The implicit read
5319 // of vcc was already added during the initial BuildMI, but we
5320 // 1) may need to change vcc to vcc_lo to preserve the original register
5321 // 2) have to preserve the original flags.
5322 copyFlagsToImplicitVCC(*Inst32, *Src2);
5323 continue;
5324 }
5325 }
5326
5327 Inst32.add(Use);
5328 }
5329
5330 // FIXME: Losing implicit operands
5331 fixImplicitOperands(*Inst32);
5332
5333 // The explicit carry/result def is dropped in favor of an implicit VCC def;
5334 // preserve the dead flag.
5335 const MachineOperand *OldSDst = getNamedOperand(MI, AMDGPU::OpName::sdst);
5336 if (OldSDst && OldSDst->isDead()) {
5337 if (MachineOperand *NewVCC =
5338 Inst32->findRegisterDefOperand(RI.getVCC(), &RI))
5339 NewVCC->setIsDead();
5340 }
5341
5342 return Inst32;
5343}
5344
5346 // Null is free
5347 Register Reg = RegOp.getReg();
5348 if (Reg == AMDGPU::SGPR_NULL || Reg == AMDGPU::SGPR_NULL64)
5349 return false;
5350
5351 // SGPRs use the constant bus
5352
5353 // FIXME: implicit registers that are not part of the MCInstrDesc's implicit
5354 // physical register operands should also count, except for exec.
5355 if (RegOp.isImplicit())
5356 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::M0;
5357
5358 // SGPRs use the constant bus
5359 return AMDGPU::SReg_32RegClass.contains(Reg) ||
5360 AMDGPU::SReg_64RegClass.contains(Reg);
5361}
5362
5364 const MachineRegisterInfo &MRI) const {
5365 Register Reg = RegOp.getReg();
5366 return Reg.isVirtual() ? RI.isSGPRClass(MRI.getRegClass(Reg))
5367 : physRegUsesConstantBus(RegOp);
5368}
5369
5371 const MachineOperand &MO,
5372 const MCOperandInfo &OpInfo) const {
5373 // Literal constants use the constant bus.
5374 if (!MO.isReg())
5375 return !isInlineConstant(MO, OpInfo);
5376
5377 Register Reg = MO.getReg();
5378 return Reg.isVirtual() ? RI.isSGPRClass(MRI.getRegClass(Reg))
5380}
5381
5383 for (const MachineOperand &MO : MI.implicit_operands()) {
5384 // We only care about reads.
5385 if (MO.isDef())
5386 continue;
5387
5388 switch (MO.getReg()) {
5389 case AMDGPU::VCC:
5390 case AMDGPU::VCC_LO:
5391 case AMDGPU::VCC_HI:
5392 case AMDGPU::M0:
5393 case AMDGPU::FLAT_SCR:
5394 return MO.getReg();
5395
5396 default:
5397 break;
5398 }
5399 }
5400
5401 return Register();
5402}
5403
5404static bool shouldReadExec(const MachineInstr &MI) {
5405 if (SIInstrInfo::isVALU(MI, /*AllowLDSDMA=*/true)) {
5406 switch (MI.getOpcode()) {
5407 case AMDGPU::V_READLANE_B32:
5408 case AMDGPU::SI_RESTORE_S32_FROM_VGPR:
5409 case AMDGPU::V_WRITELANE_B32:
5410 case AMDGPU::SI_SPILL_S32_TO_VGPR:
5411 return false;
5412 }
5413
5414 return true;
5415 }
5416
5417 if (MI.isPreISelOpcode() ||
5418 SIInstrInfo::isGenericOpcode(MI.getOpcode()) ||
5421 return false;
5422
5423 return true;
5424}
5425
5426static bool isRegOrFI(const MachineOperand &MO) {
5427 return MO.isReg() || MO.isFI();
5428}
5429
5430static bool isSubRegOf(const SIRegisterInfo &TRI,
5431 const MachineOperand &SuperVec,
5432 const MachineOperand &SubReg) {
5433 if (SubReg.getReg().isPhysical())
5434 return TRI.isSubRegister(SuperVec.getReg(), SubReg.getReg());
5435
5436 return SubReg.getSubReg() != AMDGPU::NoSubRegister &&
5437 SubReg.getReg() == SuperVec.getReg();
5438}
5439
5440// Verify the illegal copy from vector register to SGPR for generic opcode COPY
5441bool SIInstrInfo::verifyCopy(const MachineInstr &MI,
5442 const MachineRegisterInfo &MRI,
5443 StringRef &ErrInfo) const {
5444 Register DstReg = MI.getOperand(0).getReg();
5445 Register SrcReg = MI.getOperand(1).getReg();
5446 // This is a check for copy from vector register to SGPR
5447 if (RI.isVectorRegister(MRI, SrcReg) && RI.isSGPRReg(MRI, DstReg)) {
5448 ErrInfo = "illegal copy from vector register to SGPR";
5449 return false;
5450 }
5451 return true;
5452}
5453
5455 StringRef &ErrInfo) const {
5456 uint32_t Opcode = MI.getOpcode();
5457 const MachineFunction *MF = MI.getMF();
5458 const MachineRegisterInfo &MRI = MF->getRegInfo();
5459
5460 // FIXME: At this point the COPY verify is done only for non-ssa forms.
5461 // Find a better property to recognize the point where instruction selection
5462 // is just done.
5463 // We can only enforce this check after SIFixSGPRCopies pass so that the
5464 // illegal copies are legalized and thereafter we don't expect a pass
5465 // inserting similar copies.
5466 if (!MRI.isSSA() && MI.isCopy())
5467 return verifyCopy(MI, MRI, ErrInfo);
5468
5469 if (SIInstrInfo::isGenericOpcode(Opcode))
5470 return true;
5471
5472 int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
5473 int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
5474 int Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2);
5475 int Src3Idx = -1;
5476 if (Src0Idx == -1) {
5477 // VOPD V_DUAL_* instructions use different operand names.
5478 Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0X);
5479 Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1X);
5480 Src2Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0Y);
5481 Src3Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vsrc1Y);
5482 }
5483
5484 // Make sure the number of operands is correct.
5485 const MCInstrDesc &Desc = get(Opcode);
5486 if (!Desc.isVariadic() &&
5487 Desc.getNumOperands() != MI.getNumExplicitOperands()) {
5488 ErrInfo = "Instruction has wrong number of operands.";
5489 return false;
5490 }
5491
5492 if (MI.isInlineAsm()) {
5493 // Verify register classes for inlineasm constraints.
5494 for (unsigned I = InlineAsm::MIOp_FirstOperand, E = MI.getNumOperands();
5495 I != E; ++I) {
5496 const TargetRegisterClass *RC = MI.getRegClassConstraint(I, this, &RI);
5497 if (!RC)
5498 continue;
5499
5500 const MachineOperand &Op = MI.getOperand(I);
5501 if (!Op.isReg())
5502 continue;
5503
5504 Register Reg = Op.getReg();
5505 if (!Reg.isVirtual() && !RC->contains(Reg)) {
5506 ErrInfo = "inlineasm operand has incorrect register class.";
5507 return false;
5508 }
5509 }
5510
5511 return true;
5512 }
5513
5514 if (isImage(MI) && MI.memoperands_empty() && MI.mayLoadOrStore()) {
5515 ErrInfo = "missing memory operand from image instruction.";
5516 return false;
5517 }
5518
5519 // Make sure the register classes are correct.
5520 for (int i = 0, e = Desc.getNumOperands(); i != e; ++i) {
5521 const MachineOperand &MO = MI.getOperand(i);
5522 if (MO.isFPImm()) {
5523 ErrInfo = "FPImm Machine Operands are not supported. ISel should bitcast "
5524 "all fp values to integers.";
5525 return false;
5526 }
5527
5528 const MCOperandInfo &OpInfo = Desc.operands()[i];
5529
5530 switch (OpInfo.OperandType) {
5532 if (MI.getOperand(i).isImm() || MI.getOperand(i).isGlobal()) {
5533 ErrInfo = "Illegal immediate value for operand.";
5534 return false;
5535 }
5536 break;
5550 break;
5564 if (!MO.isReg() && (!MO.isImm() || !isInlineConstant(MI, i))) {
5565 ErrInfo = "Illegal immediate value for operand.";
5566 return false;
5567 }
5568 break;
5569 }
5574 if (ST.has64BitLiterals() && Desc.getSize() != 4 && MO.isImm() &&
5575 !isInlineConstant(MI, i) &&
5577 OpInfo.OperandType ==
5579 ErrInfo = "illegal 64-bit immediate value for operand.";
5580 return false;
5581 }
5582 break;
5585 if (!MI.getOperand(i).isImm() || !isInlineConstant(MI, i)) {
5586 ErrInfo = "Expected inline constant for operand.";
5587 return false;
5588 }
5589 break;
5592 break;
5597 // Check if this operand is an immediate.
5598 // FrameIndex operands will be replaced by immediates, so they are
5599 // allowed.
5600 if (!MI.getOperand(i).isImm() && !MI.getOperand(i).isFI()) {
5601 ErrInfo = "Expected immediate, but got non-immediate";
5602 return false;
5603 }
5604 break;
5608 break;
5609 default:
5610 if (OpInfo.isGenericType())
5611 continue;
5612 break;
5613 }
5614 }
5615
5616 // Verify SDWA
5617 if (isSDWA(MI)) {
5618 if (!ST.hasSDWA()) {
5619 ErrInfo = "SDWA is not supported on this target";
5620 return false;
5621 }
5622
5623 for (auto Op : {AMDGPU::OpName::src0_sel, AMDGPU::OpName::src1_sel,
5624 AMDGPU::OpName::dst_sel}) {
5625 const MachineOperand *MO = getNamedOperand(MI, Op);
5626 if (!MO)
5627 continue;
5628 int64_t Imm = MO->getImm();
5630 ErrInfo = "Invalid SDWA selection";
5631 return false;
5632 }
5633 }
5634
5635 int DstIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdst);
5636
5637 for (int OpIdx : {DstIdx, Src0Idx, Src1Idx, Src2Idx}) {
5638 if (OpIdx == -1)
5639 continue;
5640 const MachineOperand &MO = MI.getOperand(OpIdx);
5641
5642 if (!ST.hasSDWAScalar()) {
5643 // Only VGPRS on VI
5644 if (!MO.isReg() || !RI.hasVGPRs(RI.getRegClassForReg(MRI, MO.getReg()))) {
5645 ErrInfo = "Only VGPRs allowed as operands in SDWA instructions on VI";
5646 return false;
5647 }
5648 } else {
5649 // No immediates on GFX9
5650 if (!MO.isReg()) {
5651 ErrInfo =
5652 "Only reg allowed as operands in SDWA instructions on GFX9+";
5653 return false;
5654 }
5655 }
5656 }
5657
5658 if (!ST.hasSDWAOmod()) {
5659 // No omod allowed on VI
5660 const MachineOperand *OMod = getNamedOperand(MI, AMDGPU::OpName::omod);
5661 if (OMod != nullptr &&
5662 (!OMod->isImm() || OMod->getImm() != 0)) {
5663 ErrInfo = "OMod not allowed in SDWA instructions on VI";
5664 return false;
5665 }
5666 }
5667
5668 if (Opcode == AMDGPU::V_CVT_F32_FP8_sdwa ||
5669 Opcode == AMDGPU::V_CVT_F32_BF8_sdwa ||
5670 Opcode == AMDGPU::V_CVT_PK_F32_FP8_sdwa ||
5671 Opcode == AMDGPU::V_CVT_PK_F32_BF8_sdwa) {
5672 const MachineOperand *Src0ModsMO =
5673 getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
5674 unsigned Mods = Src0ModsMO->getImm();
5675 if (Mods & SISrcMods::ABS || Mods & SISrcMods::NEG ||
5676 Mods & SISrcMods::SEXT) {
5677 ErrInfo = "sext, abs and neg are not allowed on this instruction";
5678 return false;
5679 }
5680 }
5681
5682 uint32_t BasicOpcode = AMDGPU::getBasicFromSDWAOp(Opcode);
5683 if (isVOPC(BasicOpcode)) {
5684 if (!ST.hasSDWASdst() && DstIdx != -1) {
5685 // Only vcc allowed as dst on VI for VOPC
5686 const MachineOperand &Dst = MI.getOperand(DstIdx);
5687 if (!Dst.isReg() || Dst.getReg() != AMDGPU::VCC) {
5688 ErrInfo = "Only VCC allowed as dst in SDWA instructions on VI";
5689 return false;
5690 }
5691 } else if (!ST.hasSDWAOutModsVOPC()) {
5692 // No clamp allowed on GFX9 for VOPC
5693 const MachineOperand *Clamp = getNamedOperand(MI, AMDGPU::OpName::clamp);
5694 if (Clamp && (!Clamp->isImm() || Clamp->getImm() != 0)) {
5695 ErrInfo = "Clamp not allowed in VOPC SDWA instructions on VI";
5696 return false;
5697 }
5698
5699 // No omod allowed on GFX9 for VOPC
5700 const MachineOperand *OMod = getNamedOperand(MI, AMDGPU::OpName::omod);
5701 if (OMod && (!OMod->isImm() || OMod->getImm() != 0)) {
5702 ErrInfo = "OMod not allowed in VOPC SDWA instructions on VI";
5703 return false;
5704 }
5705 }
5706 }
5707
5708 const MachineOperand *DstUnused = getNamedOperand(MI, AMDGPU::OpName::dst_unused);
5709 if (DstUnused && DstUnused->isImm() &&
5710 DstUnused->getImm() == AMDGPU::SDWA::UNUSED_PRESERVE) {
5711 const MachineOperand &Dst = MI.getOperand(DstIdx);
5712 if (!Dst.isReg() || !Dst.isTied()) {
5713 ErrInfo = "Dst register should have tied register";
5714 return false;
5715 }
5716
5717 const MachineOperand &TiedMO =
5718 MI.getOperand(MI.findTiedOperandIdx(DstIdx));
5719 if (!TiedMO.isReg() || !TiedMO.isImplicit() || !TiedMO.isUse()) {
5720 ErrInfo =
5721 "Dst register should be tied to implicit use of preserved register";
5722 return false;
5723 }
5724 if (TiedMO.getReg().isPhysical() && Dst.getReg() != TiedMO.getReg()) {
5725 ErrInfo = "Dst register should use same physical register as preserved";
5726 return false;
5727 }
5728 }
5729 }
5730
5731 if (isDPP(MI) && !ST.hasDPPSrc1SGPR() && Src1Idx != -1) {
5732 const MachineOperand &Src1MO = MI.getOperand(Src1Idx);
5733 if (Src1MO.isReg() && RI.isSGPRReg(MRI, Src1MO.getReg())) {
5734 ErrInfo = "DPP src1 cannot be SGPR on this subtarget";
5735 return false;
5736 }
5737 if (Src1MO.isImm()) {
5738 ErrInfo = "DPP src1 cannot be an immediate on this subtarget";
5739 return false;
5740 }
5741 }
5742
5743 // Verify MIMG / VIMAGE / VSAMPLE
5744 if (isImage(Opcode) && !MI.mayStore()) {
5745 // Ensure that the return type used is large enough for all the options
5746 // being used TFE/LWE require an extra result register.
5747 const MachineOperand *DMask = getNamedOperand(MI, AMDGPU::OpName::dmask);
5748 if (DMask) {
5749 uint64_t DMaskImm = DMask->getImm();
5750 uint32_t RegCount = isGather4(Opcode) ? 4 : llvm::popcount(DMaskImm);
5751 const MachineOperand *TFE = getNamedOperand(MI, AMDGPU::OpName::tfe);
5752 const MachineOperand *LWE = getNamedOperand(MI, AMDGPU::OpName::lwe);
5753 const MachineOperand *D16 = getNamedOperand(MI, AMDGPU::OpName::d16);
5754
5755 // Adjust for packed 16 bit values
5756 if (D16 && D16->getImm() && !ST.hasUnpackedD16VMem())
5757 RegCount = divideCeil(RegCount, 2);
5758
5759 // Adjust if using LWE or TFE
5760 if ((LWE && LWE->getImm()) || (TFE && TFE->getImm()))
5761 RegCount += 1;
5762
5763 const uint32_t DstIdx =
5764 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::vdata);
5765 const MachineOperand &Dst = MI.getOperand(DstIdx);
5766 if (Dst.isReg()) {
5767 const TargetRegisterClass *DstRC = getOpRegClass(MI, DstIdx);
5768 uint32_t DstSize = RI.getRegSizeInBits(*DstRC) / 32;
5769 if (RegCount > DstSize) {
5770 ErrInfo = "Image instruction returns too many registers for dst "
5771 "register class";
5772 return false;
5773 }
5774 }
5775 }
5776 }
5777
5778 // Verify VOP*. Ignore multiple sgpr operands on writelane.
5779 if (isVALU(MI, /*AllowLDSDMA=*/false) &&
5780 Desc.getOpcode() != AMDGPU::V_WRITELANE_B32) {
5781 unsigned ConstantBusCount = 0;
5782 bool UsesLiteral = false;
5783 const MachineOperand *LiteralVal = nullptr;
5784
5785 int ImmIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::imm);
5786 if (ImmIdx != -1) {
5787 ++ConstantBusCount;
5788 UsesLiteral = true;
5789 LiteralVal = &MI.getOperand(ImmIdx);
5790 }
5791
5792 SmallVector<Register, 2> SGPRsUsed;
5793 Register SGPRUsed;
5794
5795 // Only look at the true operands. Only a real operand can use the constant
5796 // bus, and we don't want to check pseudo-operands like the source modifier
5797 // flags.
5798 for (int OpIdx : {Src0Idx, Src1Idx, Src2Idx, Src3Idx}) {
5799 if (OpIdx == -1)
5800 continue;
5801 const MachineOperand &MO = MI.getOperand(OpIdx);
5802 if (usesConstantBus(MRI, MO, MI.getDesc().operands()[OpIdx])) {
5803 if (MO.isReg()) {
5804 SGPRUsed = MO.getReg();
5805 if (!llvm::is_contained(SGPRsUsed, SGPRUsed)) {
5806 ++ConstantBusCount;
5807 SGPRsUsed.push_back(SGPRUsed);
5808 }
5809 } else if (!MO.isFI()) { // Treat FI like a register.
5810 if (!UsesLiteral) {
5811 ++ConstantBusCount;
5812 UsesLiteral = true;
5813 LiteralVal = &MO;
5814 } else if (!MO.isIdenticalTo(*LiteralVal)) {
5815 assert(isVOP2(MI) || isVOP3(MI));
5816 ErrInfo = "VOP2/VOP3 instruction uses more than one literal";
5817 return false;
5818 }
5819 }
5820 }
5821 }
5822
5823 SGPRUsed = findImplicitSGPRRead(MI);
5824 if (SGPRUsed) {
5825 // Implicit uses may safely overlap true operands
5826 if (llvm::all_of(SGPRsUsed, [this, SGPRUsed](unsigned SGPR) {
5827 return !RI.regsOverlap(SGPRUsed, SGPR);
5828 })) {
5829 ++ConstantBusCount;
5830 SGPRsUsed.push_back(SGPRUsed);
5831 }
5832 }
5833
5834 // v_writelane_b32 is an exception from constant bus restriction:
5835 // vsrc0 can be sgpr, const or m0 and lane select sgpr, m0 or inline-const
5836 if (ConstantBusCount > ST.getConstantBusLimit(Opcode) &&
5837 Opcode != AMDGPU::V_WRITELANE_B32) {
5838 ErrInfo = "VOP* instruction violates constant bus restriction";
5839 return false;
5840 }
5841
5842 if (isVOP3(MI) && UsesLiteral && !ST.hasVOP3Literal()) {
5843 ErrInfo = "VOP3 instruction uses literal";
5844 return false;
5845 }
5846 }
5847
5848 // Special case for writelane - this can break the multiple constant bus rule,
5849 // but still can't use more than one SGPR register
5850 if (Desc.getOpcode() == AMDGPU::V_WRITELANE_B32) {
5851 unsigned SGPRCount = 0;
5852 Register SGPRUsed;
5853
5854 for (int OpIdx : {Src0Idx, Src1Idx}) {
5855 if (OpIdx == -1)
5856 break;
5857
5858 const MachineOperand &MO = MI.getOperand(OpIdx);
5859
5860 if (usesConstantBus(MRI, MO, MI.getDesc().operands()[OpIdx])) {
5861 if (MO.isReg() && MO.getReg() != AMDGPU::M0) {
5862 if (MO.getReg() != SGPRUsed)
5863 ++SGPRCount;
5864 SGPRUsed = MO.getReg();
5865 }
5866 }
5867 if (SGPRCount > ST.getConstantBusLimit(Opcode)) {
5868 ErrInfo = "WRITELANE instruction violates constant bus restriction";
5869 return false;
5870 }
5871 }
5872 }
5873
5874 // Verify misc. restrictions on specific instructions.
5875 if (Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F32_e64 ||
5876 Desc.getOpcode() == AMDGPU::V_DIV_SCALE_F64_e64) {
5877 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5878 const MachineOperand &Src1 = MI.getOperand(Src1Idx);
5879 const MachineOperand &Src2 = MI.getOperand(Src2Idx);
5880 if (Src0.isReg() && Src1.isReg() && Src2.isReg()) {
5881 if (!compareMachineOp(Src0, Src1) &&
5882 !compareMachineOp(Src0, Src2)) {
5883 ErrInfo = "v_div_scale_{f32|f64} require src0 = src1 or src2";
5884 return false;
5885 }
5886 }
5887 if ((getNamedOperand(MI, AMDGPU::OpName::src0_modifiers)->getImm() &
5888 SISrcMods::ABS) ||
5889 (getNamedOperand(MI, AMDGPU::OpName::src1_modifiers)->getImm() &
5890 SISrcMods::ABS) ||
5891 (getNamedOperand(MI, AMDGPU::OpName::src2_modifiers)->getImm() &
5892 SISrcMods::ABS)) {
5893 ErrInfo = "ABS not allowed in VOP3B instructions";
5894 return false;
5895 }
5896 }
5897
5898 if (isSOP2(MI) || isSOPC(MI)) {
5899 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5900 const MachineOperand &Src1 = MI.getOperand(Src1Idx);
5901
5902 if (!isRegOrFI(Src0) && !isRegOrFI(Src1) &&
5903 !isInlineConstant(Src0, Desc.operands()[Src0Idx]) &&
5904 !isInlineConstant(Src1, Desc.operands()[Src1Idx]) &&
5905 !Src0.isIdenticalTo(Src1)) {
5906 ErrInfo = "SOP2/SOPC instruction requires too many immediate constants";
5907 return false;
5908 }
5909 }
5910
5911 if (isSOPK(MI)) {
5912 const auto *Op = getNamedOperand(MI, AMDGPU::OpName::simm16);
5913 if (Desc.isBranch()) {
5914 if (!Op->isMBB()) {
5915 ErrInfo = "invalid branch target for SOPK instruction";
5916 return false;
5917 }
5918 } else {
5919 uint64_t Imm = Op->getImm();
5920 if (sopkIsZext(Opcode)) {
5921 if (!isUInt<16>(Imm)) {
5922 ErrInfo = "invalid immediate for SOPK instruction";
5923 return false;
5924 }
5925 } else {
5926 if (!isInt<16>(Imm)) {
5927 ErrInfo = "invalid immediate for SOPK instruction";
5928 return false;
5929 }
5930 }
5931 }
5932 }
5933
5934 if (Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e32 ||
5935 Desc.getOpcode() == AMDGPU::V_MOVRELS_B32_e64 ||
5936 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5937 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64) {
5938 const bool IsDst = Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e32 ||
5939 Desc.getOpcode() == AMDGPU::V_MOVRELD_B32_e64;
5940
5941 const unsigned StaticNumOps =
5942 Desc.getNumOperands() + Desc.implicit_uses().size();
5943 const unsigned NumImplicitOps = IsDst ? 2 : 1;
5944
5945 // Require additional implicit operands. This allows a fixup done by the
5946 // post RA scheduler where the main implicit operand is killed and
5947 // implicit-defs are added for sub-registers that remain live after this
5948 // instruction.
5949 if (MI.getNumOperands() < StaticNumOps + NumImplicitOps) {
5950 ErrInfo = "missing implicit register operands";
5951 return false;
5952 }
5953
5954 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
5955 if (IsDst) {
5956 if (!Dst->isUse()) {
5957 ErrInfo = "v_movreld_b32 vdst should be a use operand";
5958 return false;
5959 }
5960
5961 unsigned UseOpIdx;
5962 if (!MI.isRegTiedToUseOperand(StaticNumOps, &UseOpIdx) ||
5963 UseOpIdx != StaticNumOps + 1) {
5964 ErrInfo = "movrel implicit operands should be tied";
5965 return false;
5966 }
5967 }
5968
5969 const MachineOperand &Src0 = MI.getOperand(Src0Idx);
5970 const MachineOperand &ImpUse
5971 = MI.getOperand(StaticNumOps + NumImplicitOps - 1);
5972 if (!ImpUse.isReg() || !ImpUse.isUse() ||
5973 !isSubRegOf(RI, ImpUse, IsDst ? *Dst : Src0)) {
5974 ErrInfo = "src0 should be subreg of implicit vector use";
5975 return false;
5976 }
5977 }
5978
5979 // Make sure we aren't losing exec uses in the td files. This mostly requires
5980 // being careful when using let Uses to try to add other use registers.
5981 if (shouldReadExec(MI)) {
5982 if (!MI.hasRegisterImplicitUseOperand(AMDGPU::EXEC)) {
5983 ErrInfo = "VALU instruction does not implicitly read exec mask";
5984 return false;
5985 }
5986 }
5987
5988 if (isSMRD(MI)) {
5989 if (MI.mayStore() &&
5990 ST.getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS) {
5991 // The register offset form of scalar stores may only use m0 as the
5992 // soffset register.
5993 const MachineOperand *Soff = getNamedOperand(MI, AMDGPU::OpName::soffset);
5994 if (Soff && Soff->getReg() != AMDGPU::M0) {
5995 ErrInfo = "scalar stores must use m0 as offset register";
5996 return false;
5997 }
5998 }
5999 }
6000
6001 if (isFLAT(MI) && !ST.hasFlatInstOffsets()) {
6002 const MachineOperand *Offset = getNamedOperand(MI, AMDGPU::OpName::offset);
6003 if (Offset->getImm() != 0) {
6004 ErrInfo = "subtarget does not support offsets in flat instructions";
6005 return false;
6006 }
6007 }
6008
6009 if (isDS(MI) && !ST.hasGDS()) {
6010 const MachineOperand *GDSOp = getNamedOperand(MI, AMDGPU::OpName::gds);
6011 if (GDSOp && GDSOp->getImm() != 0) {
6012 ErrInfo = "GDS is not supported on this subtarget";
6013 return false;
6014 }
6015 }
6016
6017 if (isImage(MI)) {
6018 const MachineOperand *DimOp = getNamedOperand(MI, AMDGPU::OpName::dim);
6019 if (DimOp) {
6020 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opcode,
6021 AMDGPU::OpName::vaddr0);
6022 AMDGPU::OpName RSrcOpName =
6023 isMIMG(MI) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
6024 int RsrcIdx = AMDGPU::getNamedOperandIdx(Opcode, RSrcOpName);
6025 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opcode);
6026 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
6027 AMDGPU::getMIMGBaseOpcodeInfo(Info->BaseOpcode);
6028 const AMDGPU::MIMGDimInfo *Dim =
6030
6031 if (!Dim) {
6032 ErrInfo = "dim is out of range";
6033 return false;
6034 }
6035
6036 bool IsA16 = false;
6037 if (ST.hasR128A16()) {
6038 const MachineOperand *R128A16 = getNamedOperand(MI, AMDGPU::OpName::r128);
6039 IsA16 = R128A16->getImm() != 0;
6040 } else if (ST.hasA16()) {
6041 const MachineOperand *A16 = getNamedOperand(MI, AMDGPU::OpName::a16);
6042 IsA16 = A16->getImm() != 0;
6043 }
6044
6045 bool IsNSA = RsrcIdx - VAddr0Idx > 1;
6046
6047 unsigned AddrWords =
6048 AMDGPU::getAddrSizeMIMGOp(BaseOpcode, Dim, IsA16, ST.hasG16());
6049
6050 unsigned VAddrWords;
6051 if (IsNSA) {
6052 VAddrWords = RsrcIdx - VAddr0Idx;
6053 if (ST.hasPartialNSAEncoding() &&
6054 AddrWords > ST.getNSAMaxSize(isVSAMPLE(MI))) {
6055 unsigned LastVAddrIdx = RsrcIdx - 1;
6056 VAddrWords += getOpSize(MI, LastVAddrIdx) / 4 - 1;
6057 }
6058 } else {
6059 VAddrWords = getOpSize(MI, VAddr0Idx) / 4;
6060 if (AddrWords > 12)
6061 AddrWords = 16;
6062 }
6063
6064 if (VAddrWords != AddrWords) {
6065 LLVM_DEBUG(dbgs() << "bad vaddr size, expected " << AddrWords
6066 << " but got " << VAddrWords << "\n");
6067 ErrInfo = "bad vaddr size";
6068 return false;
6069 }
6070 }
6071 }
6072
6073 const MachineOperand *DppCt = getNamedOperand(MI, AMDGPU::OpName::dpp_ctrl);
6074 if (DppCt) {
6075 using namespace AMDGPU::DPP;
6076
6077 unsigned DC = DppCt->getImm();
6078 if (DC == DppCtrl::DPP_UNUSED1 || DC == DppCtrl::DPP_UNUSED2 ||
6079 DC == DppCtrl::DPP_UNUSED3 || DC > DppCtrl::DPP_LAST ||
6080 (DC >= DppCtrl::DPP_UNUSED4_FIRST && DC <= DppCtrl::DPP_UNUSED4_LAST) ||
6081 (DC >= DppCtrl::DPP_UNUSED5_FIRST && DC <= DppCtrl::DPP_UNUSED5_LAST) ||
6082 (DC >= DppCtrl::DPP_UNUSED6_FIRST && DC <= DppCtrl::DPP_UNUSED6_LAST) ||
6083 (DC >= DppCtrl::DPP_UNUSED7_FIRST && DC <= DppCtrl::DPP_UNUSED7_LAST) ||
6084 (DC >= DppCtrl::DPP_UNUSED8_FIRST && DC <= DppCtrl::DPP_UNUSED8_LAST)) {
6085 ErrInfo = "Invalid dpp_ctrl value";
6086 return false;
6087 }
6088 if (DC >= DppCtrl::WAVE_SHL1 && DC <= DppCtrl::WAVE_ROR1 &&
6089 !ST.hasDPPWavefrontShifts()) {
6090 ErrInfo = "Invalid dpp_ctrl value: "
6091 "wavefront shifts are not supported on GFX10+";
6092 return false;
6093 }
6094 if (DC >= DppCtrl::BCAST15 && DC <= DppCtrl::BCAST31 &&
6095 !ST.hasDPPBroadcasts()) {
6096 ErrInfo = "Invalid dpp_ctrl value: "
6097 "broadcasts are not supported on GFX10+";
6098 return false;
6099 }
6100 if (DC >= DppCtrl::ROW_SHARE_FIRST && DC <= DppCtrl::ROW_XMASK_LAST &&
6101 ST.getGeneration() < AMDGPUSubtarget::GFX10) {
6102 if (DC >= DppCtrl::ROW_NEWBCAST_FIRST &&
6103 DC <= DppCtrl::ROW_NEWBCAST_LAST &&
6104 !ST.hasGFX90AInsts()) {
6105 ErrInfo = "Invalid dpp_ctrl value: "
6106 "row_newbroadcast/row_share is not supported before "
6107 "GFX90A/GFX10";
6108 return false;
6109 }
6110 if (DC > DppCtrl::ROW_NEWBCAST_LAST || !ST.hasGFX90AInsts()) {
6111 ErrInfo = "Invalid dpp_ctrl value: "
6112 "row_share and row_xmask are not supported before GFX10";
6113 return false;
6114 }
6115 }
6116
6117 if (Opcode != AMDGPU::V_MOV_B64_DPP_PSEUDO &&
6119 ST.hasFeature(AMDGPU::FeatureDPALU_DPP) &&
6120 AMDGPU::isDPALU_DPP(Desc, *this, ST)) {
6121 ErrInfo = "Invalid dpp_ctrl value: "
6122 "DP ALU dpp only support row_newbcast";
6123 return false;
6124 }
6125 }
6126
6127 if ((MI.mayStore() || MI.mayLoad()) && !isVGPRSpill(MI)) {
6128 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::vdst);
6129 AMDGPU::OpName DataName =
6130 isDS(Opcode) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata;
6131 const MachineOperand *Data = getNamedOperand(MI, DataName);
6132 const MachineOperand *Data2 = getNamedOperand(MI, AMDGPU::OpName::data1);
6133 if (Data && !Data->isReg())
6134 Data = nullptr;
6135
6136 if (!ST.hasGFX90AInsts()) {
6137 if ((Dst && RI.isAGPR(MRI, Dst->getReg())) ||
6138 (Data && RI.isAGPR(MRI, Data->getReg())) ||
6139 (Data2 && RI.isAGPR(MRI, Data2->getReg()))) {
6140 ErrInfo = "Invalid register class: "
6141 "agpr loads and stores not supported on this GPU";
6142 return false;
6143 }
6144 }
6145 }
6146
6147 if (ST.needsAlignedVGPRs()) {
6148 const auto isAlignedReg = [&MI, &MRI, this](AMDGPU::OpName OpName) -> bool {
6150 if (!Op)
6151 return true;
6152 Register Reg = Op->getReg();
6153 if (Reg.isPhysical())
6154 return !(RI.getHWRegIndex(Reg) & 1);
6155 const TargetRegisterClass &RC = *MRI.getRegClass(Reg);
6156 return RI.getRegSizeInBits(RC) > 32 && RI.isProperlyAlignedRC(RC) &&
6157 !(RI.getChannelFromSubReg(Op->getSubReg()) & 1);
6158 };
6159
6160 if (isMIMG(MI)) {
6161 if (!isAlignedReg(AMDGPU::OpName::vaddr)) {
6162 ErrInfo = "Subtarget requires even aligned vector registers "
6163 "for vaddr operand of image instructions";
6164 return false;
6165 }
6166 }
6167 }
6168
6169 if (Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts()) {
6170 const MachineOperand *Src = getNamedOperand(MI, AMDGPU::OpName::src0);
6171 if (Src->isReg() && RI.isSGPRReg(MRI, Src->getReg())) {
6172 ErrInfo = "Invalid register class: "
6173 "v_accvgpr_write with an SGPR is not supported on this GPU";
6174 return false;
6175 }
6176 }
6177
6178 if (Desc.getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS) {
6179 const MachineOperand &SrcOp = MI.getOperand(1);
6180 if (!SrcOp.isReg() || SrcOp.getReg().isVirtual()) {
6181 ErrInfo = "pseudo expects only physical SGPRs";
6182 return false;
6183 }
6184 }
6185
6186 if (const MachineOperand *CPol = getNamedOperand(MI, AMDGPU::OpName::cpol)) {
6187 if (CPol->getImm() & AMDGPU::CPol::SCAL) {
6188 if (!ST.hasScaleOffset()) {
6189 ErrInfo = "Subtarget does not support offset scaling";
6190 return false;
6191 }
6192 if (!AMDGPU::supportsScaleOffset(*this, MI.getOpcode())) {
6193 ErrInfo = "Instruction does not support offset scaling";
6194 return false;
6195 }
6196 }
6197 }
6198
6199 // See SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
6200 if (AMDGPU::isSingleSGPRReadInst(Opcode)) {
6201 for (unsigned I = 0; I < 3; ++I) {
6203 return false;
6204 }
6205 }
6206
6207 if (ST.hasFlatScratchHiInB64InstHazard() && isSALU(MI) &&
6208 MI.readsRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI, nullptr)) {
6209 const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::sdst);
6210 if ((Dst && RI.getRegClassForReg(MRI, Dst->getReg()) ==
6211 &AMDGPU::SReg_64RegClass) ||
6212 Opcode == AMDGPU::S_BITCMP0_B64 || Opcode == AMDGPU::S_BITCMP1_B64) {
6213 ErrInfo = "Instruction cannot read flat_scratch_base_hi";
6214 return false;
6215 }
6216 }
6217
6218 return true;
6219}
6220
6222 if (MI.getOpcode() == AMDGPU::S_MOV_B32) {
6223 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6224 return MI.getOperand(1).isReg() || RI.isAGPR(MRI, MI.getOperand(0).getReg())
6225 ? AMDGPU::COPY
6226 : AMDGPU::V_MOV_B32_e32;
6227 }
6228 return getVALUOp(MI.getOpcode());
6229}
6230
6231// It is more readable to list mapped opcodes on the same line.
6232// clang-format off
6233
6234unsigned SIInstrInfo::getVALUOp(unsigned Opc) const {
6235 switch (Opc) {
6236 default: return AMDGPU::INSTRUCTION_LIST_END;
6237 case AMDGPU::REG_SEQUENCE: return AMDGPU::REG_SEQUENCE;
6238 case AMDGPU::COPY: return AMDGPU::COPY;
6239 case AMDGPU::PHI: return AMDGPU::PHI;
6240 case AMDGPU::INSERT_SUBREG: return AMDGPU::INSERT_SUBREG;
6241 case AMDGPU::WQM: return AMDGPU::WQM;
6242 case AMDGPU::SOFT_WQM: return AMDGPU::SOFT_WQM;
6243 case AMDGPU::STRICT_WWM: return AMDGPU::STRICT_WWM;
6244 case AMDGPU::STRICT_WQM: return AMDGPU::STRICT_WQM;
6245 case AMDGPU::S_ADD_I32:
6246 return ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
6247 case AMDGPU::S_ADDC_U32:
6248 return AMDGPU::V_ADDC_U32_e32;
6249 case AMDGPU::S_SUB_I32:
6250 return ST.hasAddNoCarryInsts() ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_SUB_CO_U32_e32;
6251 // FIXME: These are not consistently handled, and selected when the carry is
6252 // used.
6253 case AMDGPU::S_ADD_U32:
6254 return AMDGPU::V_ADD_CO_U32_e32;
6255 case AMDGPU::S_SUB_U32:
6256 return AMDGPU::V_SUB_CO_U32_e32;
6257 case AMDGPU::S_ADD_U64_PSEUDO:
6258 return AMDGPU::V_ADD_U64_PSEUDO;
6259 case AMDGPU::S_SUB_U64_PSEUDO:
6260 return AMDGPU::V_SUB_U64_PSEUDO;
6261 case AMDGPU::S_SUBB_U32: return AMDGPU::V_SUBB_U32_e32;
6262 case AMDGPU::S_MUL_I32: return AMDGPU::V_MUL_LO_U32_e64;
6263 case AMDGPU::S_MUL_HI_U32: return AMDGPU::V_MUL_HI_U32_e64;
6264 case AMDGPU::S_MUL_HI_I32: return AMDGPU::V_MUL_HI_I32_e64;
6265 case AMDGPU::S_AND_B32: return AMDGPU::V_AND_B32_e64;
6266 case AMDGPU::S_OR_B32: return AMDGPU::V_OR_B32_e64;
6267 case AMDGPU::S_XOR_B32: return AMDGPU::V_XOR_B32_e64;
6268 case AMDGPU::S_XNOR_B32:
6269 return ST.hasDLInsts() ? AMDGPU::V_XNOR_B32_e64 : AMDGPU::INSTRUCTION_LIST_END;
6270 case AMDGPU::S_MIN_I32: return AMDGPU::V_MIN_I32_e64;
6271 case AMDGPU::S_MIN_U32: return AMDGPU::V_MIN_U32_e64;
6272 case AMDGPU::S_MAX_I32: return AMDGPU::V_MAX_I32_e64;
6273 case AMDGPU::S_MAX_U32: return AMDGPU::V_MAX_U32_e64;
6274 case AMDGPU::S_ASHR_I32: return AMDGPU::V_ASHR_I32_e32;
6275 case AMDGPU::S_ASHR_I64: return AMDGPU::V_ASHR_I64_e64;
6276 case AMDGPU::S_LSHL_B32: return AMDGPU::V_LSHL_B32_e32;
6277 case AMDGPU::S_LSHL_B64: return AMDGPU::V_LSHL_B64_e64;
6278 case AMDGPU::S_LSHR_B32: return AMDGPU::V_LSHR_B32_e32;
6279 case AMDGPU::S_LSHR_B64: return AMDGPU::V_LSHR_B64_e64;
6280 case AMDGPU::S_SEXT_I32_I8: return AMDGPU::V_BFE_I32_e64;
6281 case AMDGPU::S_SEXT_I32_I16: return AMDGPU::V_BFE_I32_e64;
6282 case AMDGPU::S_BFE_U32: return AMDGPU::V_BFE_U32_e64;
6283 case AMDGPU::S_BFE_I32: return AMDGPU::V_BFE_I32_e64;
6284 case AMDGPU::S_BFM_B32: return AMDGPU::V_BFM_B32_e64;
6285 case AMDGPU::S_BREV_B32: return AMDGPU::V_BFREV_B32_e32;
6286 case AMDGPU::S_NOT_B32: return AMDGPU::V_NOT_B32_e32;
6287 case AMDGPU::S_NOT_B64: return AMDGPU::V_NOT_B32_e32;
6288 case AMDGPU::S_CMP_EQ_I32: return AMDGPU::V_CMP_EQ_I32_e64;
6289 case AMDGPU::S_CMP_LG_I32: return AMDGPU::V_CMP_NE_I32_e64;
6290 case AMDGPU::S_CMP_GT_I32: return AMDGPU::V_CMP_GT_I32_e64;
6291 case AMDGPU::S_CMP_GE_I32: return AMDGPU::V_CMP_GE_I32_e64;
6292 case AMDGPU::S_CMP_LT_I32: return AMDGPU::V_CMP_LT_I32_e64;
6293 case AMDGPU::S_CMP_LE_I32: return AMDGPU::V_CMP_LE_I32_e64;
6294 case AMDGPU::S_CMP_EQ_U32: return AMDGPU::V_CMP_EQ_U32_e64;
6295 case AMDGPU::S_CMP_LG_U32: return AMDGPU::V_CMP_NE_U32_e64;
6296 case AMDGPU::S_CMP_GT_U32: return AMDGPU::V_CMP_GT_U32_e64;
6297 case AMDGPU::S_CMP_GE_U32: return AMDGPU::V_CMP_GE_U32_e64;
6298 case AMDGPU::S_CMP_LT_U32: return AMDGPU::V_CMP_LT_U32_e64;
6299 case AMDGPU::S_CMP_LE_U32: return AMDGPU::V_CMP_LE_U32_e64;
6300 case AMDGPU::S_CMP_EQ_U64: return AMDGPU::V_CMP_EQ_U64_e64;
6301 case AMDGPU::S_CMP_LG_U64: return AMDGPU::V_CMP_NE_U64_e64;
6302 case AMDGPU::S_BCNT1_I32_B32: return AMDGPU::V_BCNT_U32_B32_e64;
6303 case AMDGPU::S_FF1_I32_B32: return AMDGPU::V_FFBL_B32_e32;
6304 case AMDGPU::S_FLBIT_I32_B32: return AMDGPU::V_FFBH_U32_e32;
6305 case AMDGPU::S_FLBIT_I32: return AMDGPU::V_FFBH_I32_e64;
6306 case AMDGPU::S_CBRANCH_SCC0: return AMDGPU::S_CBRANCH_VCCZ;
6307 case AMDGPU::S_CBRANCH_SCC1: return AMDGPU::S_CBRANCH_VCCNZ;
6308 case AMDGPU::S_CVT_F32_I32: return AMDGPU::V_CVT_F32_I32_e64;
6309 case AMDGPU::S_CVT_F32_U32: return AMDGPU::V_CVT_F32_U32_e64;
6310 case AMDGPU::S_CVT_I32_F32: return AMDGPU::V_CVT_I32_F32_e64;
6311 case AMDGPU::S_CVT_U32_F32: return AMDGPU::V_CVT_U32_F32_e64;
6312 case AMDGPU::S_CVT_F32_F16:
6313 case AMDGPU::S_CVT_HI_F32_F16:
6314 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F32_F16_t16_e64
6315 : AMDGPU::V_CVT_F32_F16_fake16_e64;
6316 case AMDGPU::S_CVT_F16_F32:
6317 return ST.useRealTrue16Insts() ? AMDGPU::V_CVT_F16_F32_t16_e64
6318 : AMDGPU::V_CVT_F16_F32_fake16_e64;
6319 case AMDGPU::S_CEIL_F32: return AMDGPU::V_CEIL_F32_e64;
6320 case AMDGPU::S_FLOOR_F32: return AMDGPU::V_FLOOR_F32_e64;
6321 case AMDGPU::S_TRUNC_F32: return AMDGPU::V_TRUNC_F32_e64;
6322 case AMDGPU::S_RNDNE_F32: return AMDGPU::V_RNDNE_F32_e64;
6323 case AMDGPU::S_CEIL_F16:
6324 return ST.useRealTrue16Insts() ? AMDGPU::V_CEIL_F16_t16_e64
6325 : AMDGPU::V_CEIL_F16_fake16_e64;
6326 case AMDGPU::S_FLOOR_F16:
6327 return ST.useRealTrue16Insts() ? AMDGPU::V_FLOOR_F16_t16_e64
6328 : AMDGPU::V_FLOOR_F16_fake16_e64;
6329 case AMDGPU::S_TRUNC_F16:
6330 return ST.useRealTrue16Insts() ? AMDGPU::V_TRUNC_F16_t16_e64
6331 : AMDGPU::V_TRUNC_F16_fake16_e64;
6332 case AMDGPU::S_RNDNE_F16:
6333 return ST.useRealTrue16Insts() ? AMDGPU::V_RNDNE_F16_t16_e64
6334 : AMDGPU::V_RNDNE_F16_fake16_e64;
6335 case AMDGPU::S_ADD_F32: return AMDGPU::V_ADD_F32_e64;
6336 case AMDGPU::S_SUB_F32: return AMDGPU::V_SUB_F32_e64;
6337 case AMDGPU::S_MIN_F32: return AMDGPU::V_MIN_F32_e64;
6338 case AMDGPU::S_MAX_F32: return AMDGPU::V_MAX_F32_e64;
6339 case AMDGPU::S_MINIMUM_F32: return AMDGPU::V_MINIMUM_F32_e64;
6340 case AMDGPU::S_MAXIMUM_F32: return AMDGPU::V_MAXIMUM_F32_e64;
6341 case AMDGPU::S_MUL_F32: return AMDGPU::V_MUL_F32_e64;
6342 case AMDGPU::S_ADD_F16:
6343 return ST.useRealTrue16Insts() ? AMDGPU::V_ADD_F16_t16_e64
6344 : AMDGPU::V_ADD_F16_fake16_e64;
6345 case AMDGPU::S_SUB_F16:
6346 return ST.useRealTrue16Insts() ? AMDGPU::V_SUB_F16_t16_e64
6347 : AMDGPU::V_SUB_F16_fake16_e64;
6348 case AMDGPU::S_MIN_F16:
6349 return ST.useRealTrue16Insts() ? AMDGPU::V_MIN_F16_t16_e64
6350 : AMDGPU::V_MIN_F16_fake16_e64;
6351 case AMDGPU::S_MAX_F16:
6352 return ST.useRealTrue16Insts() ? AMDGPU::V_MAX_F16_t16_e64
6353 : AMDGPU::V_MAX_F16_fake16_e64;
6354 case AMDGPU::S_MINIMUM_F16:
6355 return ST.useRealTrue16Insts() ? AMDGPU::V_MINIMUM_F16_t16_e64
6356 : AMDGPU::V_MINIMUM_F16_fake16_e64;
6357 case AMDGPU::S_MAXIMUM_F16:
6358 return ST.useRealTrue16Insts() ? AMDGPU::V_MAXIMUM_F16_t16_e64
6359 : AMDGPU::V_MAXIMUM_F16_fake16_e64;
6360 case AMDGPU::S_MUL_F16:
6361 return ST.useRealTrue16Insts() ? AMDGPU::V_MUL_F16_t16_e64
6362 : AMDGPU::V_MUL_F16_fake16_e64;
6363 case AMDGPU::S_CVT_PK_RTZ_F16_F32: return AMDGPU::V_CVT_PKRTZ_F16_F32_e64;
6364 case AMDGPU::S_FMAC_F32: return AMDGPU::V_FMAC_F32_e64;
6365 case AMDGPU::S_FMAC_F16:
6366 return ST.useRealTrue16Insts() ? AMDGPU::V_FMAC_F16_t16_e64
6367 : AMDGPU::V_FMAC_F16_fake16_e64;
6368 case AMDGPU::S_FMAMK_F32: return AMDGPU::V_FMAMK_F32;
6369 case AMDGPU::S_FMAAK_F32: return AMDGPU::V_FMAAK_F32;
6370 case AMDGPU::S_CMP_LT_F32: return AMDGPU::V_CMP_LT_F32_e64;
6371 case AMDGPU::S_CMP_EQ_F32: return AMDGPU::V_CMP_EQ_F32_e64;
6372 case AMDGPU::S_CMP_LE_F32: return AMDGPU::V_CMP_LE_F32_e64;
6373 case AMDGPU::S_CMP_GT_F32: return AMDGPU::V_CMP_GT_F32_e64;
6374 case AMDGPU::S_CMP_LG_F32: return AMDGPU::V_CMP_LG_F32_e64;
6375 case AMDGPU::S_CMP_GE_F32: return AMDGPU::V_CMP_GE_F32_e64;
6376 case AMDGPU::S_CMP_O_F32: return AMDGPU::V_CMP_O_F32_e64;
6377 case AMDGPU::S_CMP_U_F32: return AMDGPU::V_CMP_U_F32_e64;
6378 case AMDGPU::S_CMP_NGE_F32: return AMDGPU::V_CMP_NGE_F32_e64;
6379 case AMDGPU::S_CMP_NLG_F32: return AMDGPU::V_CMP_NLG_F32_e64;
6380 case AMDGPU::S_CMP_NGT_F32: return AMDGPU::V_CMP_NGT_F32_e64;
6381 case AMDGPU::S_CMP_NLE_F32: return AMDGPU::V_CMP_NLE_F32_e64;
6382 case AMDGPU::S_CMP_NEQ_F32: return AMDGPU::V_CMP_NEQ_F32_e64;
6383 case AMDGPU::S_CMP_NLT_F32: return AMDGPU::V_CMP_NLT_F32_e64;
6384 case AMDGPU::S_CMP_LT_F16:
6385 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LT_F16_t16_e64
6386 : AMDGPU::V_CMP_LT_F16_fake16_e64;
6387 case AMDGPU::S_CMP_EQ_F16:
6388 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_EQ_F16_t16_e64
6389 : AMDGPU::V_CMP_EQ_F16_fake16_e64;
6390 case AMDGPU::S_CMP_LE_F16:
6391 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LE_F16_t16_e64
6392 : AMDGPU::V_CMP_LE_F16_fake16_e64;
6393 case AMDGPU::S_CMP_GT_F16:
6394 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GT_F16_t16_e64
6395 : AMDGPU::V_CMP_GT_F16_fake16_e64;
6396 case AMDGPU::S_CMP_LG_F16:
6397 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_LG_F16_t16_e64
6398 : AMDGPU::V_CMP_LG_F16_fake16_e64;
6399 case AMDGPU::S_CMP_GE_F16:
6400 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_GE_F16_t16_e64
6401 : AMDGPU::V_CMP_GE_F16_fake16_e64;
6402 case AMDGPU::S_CMP_O_F16:
6403 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_O_F16_t16_e64
6404 : AMDGPU::V_CMP_O_F16_fake16_e64;
6405 case AMDGPU::S_CMP_U_F16:
6406 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_U_F16_t16_e64
6407 : AMDGPU::V_CMP_U_F16_fake16_e64;
6408 case AMDGPU::S_CMP_NGE_F16:
6409 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGE_F16_t16_e64
6410 : AMDGPU::V_CMP_NGE_F16_fake16_e64;
6411 case AMDGPU::S_CMP_NLG_F16:
6412 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLG_F16_t16_e64
6413 : AMDGPU::V_CMP_NLG_F16_fake16_e64;
6414 case AMDGPU::S_CMP_NGT_F16:
6415 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NGT_F16_t16_e64
6416 : AMDGPU::V_CMP_NGT_F16_fake16_e64;
6417 case AMDGPU::S_CMP_NLE_F16:
6418 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLE_F16_t16_e64
6419 : AMDGPU::V_CMP_NLE_F16_fake16_e64;
6420 case AMDGPU::S_CMP_NEQ_F16:
6421 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NEQ_F16_t16_e64
6422 : AMDGPU::V_CMP_NEQ_F16_fake16_e64;
6423 case AMDGPU::S_CMP_NLT_F16:
6424 return ST.useRealTrue16Insts() ? AMDGPU::V_CMP_NLT_F16_t16_e64
6425 : AMDGPU::V_CMP_NLT_F16_fake16_e64;
6426 case AMDGPU::V_S_EXP_F32_e64: return AMDGPU::V_EXP_F32_e64;
6427 case AMDGPU::V_S_EXP_F16_e64:
6428 return ST.useRealTrue16Insts() ? AMDGPU::V_EXP_F16_t16_e64
6429 : AMDGPU::V_EXP_F16_fake16_e64;
6430 case AMDGPU::V_S_LOG_F32_e64: return AMDGPU::V_LOG_F32_e64;
6431 case AMDGPU::V_S_LOG_F16_e64:
6432 return ST.useRealTrue16Insts() ? AMDGPU::V_LOG_F16_t16_e64
6433 : AMDGPU::V_LOG_F16_fake16_e64;
6434 case AMDGPU::V_S_RCP_F32_e64: return AMDGPU::V_RCP_F32_e64;
6435 case AMDGPU::V_S_RCP_F16_e64:
6436 return ST.useRealTrue16Insts() ? AMDGPU::V_RCP_F16_t16_e64
6437 : AMDGPU::V_RCP_F16_fake16_e64;
6438 case AMDGPU::V_S_RSQ_F32_e64: return AMDGPU::V_RSQ_F32_e64;
6439 case AMDGPU::V_S_RSQ_F16_e64:
6440 return ST.useRealTrue16Insts() ? AMDGPU::V_RSQ_F16_t16_e64
6441 : AMDGPU::V_RSQ_F16_fake16_e64;
6442 case AMDGPU::V_S_SQRT_F32_e64: return AMDGPU::V_SQRT_F32_e64;
6443 case AMDGPU::V_S_SQRT_F16_e64:
6444 return ST.useRealTrue16Insts() ? AMDGPU::V_SQRT_F16_t16_e64
6445 : AMDGPU::V_SQRT_F16_fake16_e64;
6446 }
6448 "Unexpected scalar opcode without corresponding vector one!");
6449}
6450
6451// clang-format on
6452
6456 const DebugLoc &DL, Register Reg,
6457 bool IsSCCLive,
6458 SlotIndexes *Indexes) const {
6459 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
6460 const SIInstrInfo *TII = ST.getInstrInfo();
6462 if (IsSCCLive) {
6463 // Insert two move instructions, one to save the original value of EXEC and
6464 // the other to turn on all bits in EXEC. This is required as we can't use
6465 // the single instruction S_OR_SAVEEXEC that clobbers SCC.
6466 auto StoreExecMI = BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), Reg)
6468 auto FlipExecMI =
6469 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addImm(-1);
6470 if (Indexes) {
6471 Indexes->insertMachineInstrInMaps(*StoreExecMI);
6472 Indexes->insertMachineInstrInMaps(*FlipExecMI);
6473 }
6474 } else {
6475 auto SaveExec =
6476 BuildMI(MBB, MBBI, DL, TII->get(LMC.OrSaveExecOpc), Reg).addImm(-1);
6477 SaveExec->getOperand(3).setIsDead(); // Mark SCC as dead.
6478 if (Indexes)
6479 Indexes->insertMachineInstrInMaps(*SaveExec);
6480 }
6481}
6482
6485 const DebugLoc &DL, Register Reg,
6486 SlotIndexes *Indexes) const {
6488 auto ExecRestoreMI = BuildMI(MBB, MBBI, DL, get(LMC.MovOpc), LMC.ExecReg)
6489 .addReg(Reg, RegState::Kill);
6490 if (Indexes)
6491 Indexes->insertMachineInstrInMaps(*ExecRestoreMI);
6492}
6493
6497 "Not a whole wave func");
6498 MachineBasicBlock &MBB = *MF.begin();
6499 for (MachineInstr &MI : MBB)
6500 if (MI.getOpcode() == AMDGPU::SI_WHOLE_WAVE_FUNC_SETUP ||
6501 MI.getOpcode() == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
6502 return &MI;
6503
6504 llvm_unreachable("Couldn't find SI_SETUP_WHOLE_WAVE_FUNC instruction");
6505}
6506
6508 unsigned OpNo) const {
6509 const MCInstrDesc &Desc = get(MI.getOpcode());
6510 if (MI.isVariadic() || OpNo >= Desc.getNumOperands() ||
6511 Desc.operands()[OpNo].RegClass == -1) {
6512 Register Reg = MI.getOperand(OpNo).getReg();
6513
6514 if (Reg.isVirtual()) {
6515 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6516 return MRI.getRegClass(Reg);
6517 }
6518 return RI.getPhysRegBaseClass(Reg);
6519 }
6520
6521 int16_t RegClass = getOpRegClassID(Desc.operands()[OpNo]);
6522 return RegClass < 0 ? nullptr : RI.getRegClass(RegClass);
6523}
6524
6525// Convert VOP3 operand index to source number.
6526static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx) {
6527 constexpr AMDGPU::OpName OpNames[] = {
6528 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2};
6529
6530 for (auto [I, OpName] : enumerate(OpNames)) {
6531 int SrcIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[I]);
6532 if (static_cast<unsigned>(SrcIdx) == OpIdx)
6533 return I;
6534 }
6535
6536 return UINT_MAX;
6537}
6538
6541 MachineBasicBlock *MBB = MI.getParent();
6542 MachineOperand &MO = MI.getOperand(OpIdx);
6543 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo();
6544 unsigned RCID = getOpRegClassID(get(MI.getOpcode()).operands()[OpIdx]);
6545 const TargetRegisterClass *RC = RI.getRegClass(RCID);
6546 unsigned Size = RI.getRegSizeInBits(*RC);
6547 unsigned Opcode = (Size == 64) ? AMDGPU::V_MOV_B64_PSEUDO
6548 : Size == 16 ? AMDGPU::V_MOV_B16_t16_e64
6549 : AMDGPU::V_MOV_B32_e32;
6550 if (MO.isReg())
6551 Opcode = AMDGPU::COPY;
6552 else if (RI.isSGPRClass(RC))
6553 Opcode = (Size == 64) ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
6554
6555 const TargetRegisterClass *VRC = RI.getEquivalentVGPRClass(RC);
6556 Register Reg = MRI.createVirtualRegister(VRC);
6557 DebugLoc DL = MBB->findDebugLoc(I);
6558
6559 if (Size == 128 && AMDGPU::isPackedSingleSGPR64BitInst(MI.getOpcode()) &&
6561 // Special case for V_PK_*64 instructions: these do not have OPSEL but SGPR
6562 // sources behave like OPSEL is set replicating low 64-bits into high. VGPR
6563 // sources in turn read actual 4 registers. To move operand from an SGPR to
6564 // a VGPR we need to replicate low half.
6565 // We also do not select immediates for these instructions so it always has
6566 // to be an SGPR register here.
6567 // Operands which are not legal as per isLegalSingleSGPRReadInstOperand()
6568 // sent here specifically to fix a non-splat SGPR and shall perform a full
6569 // copy.
6570
6571 const TargetRegisterClass *VRC64 = RI.getVGPRClassForBitWidth(64);
6572 Register Low64 = MRI.createVirtualRegister(VRC64);
6573 assert(MO.isReg() && RI.isSGPRReg(MRI, MO.getReg()));
6574 BuildMI(*MBB, I, DL, get(TargetOpcode::COPY), Low64)
6575 .addReg(MO.getReg(), {}, AMDGPU::sub0_sub1);
6576 BuildMI(*MBB, I, DL, get(TargetOpcode::REG_SEQUENCE), Reg)
6577 .addReg(Low64)
6578 .addImm(AMDGPU::sub0_sub1)
6579 .addReg(Low64, RegState::Kill)
6580 .addImm(AMDGPU::sub2_sub3);
6581 } else if (Opcode == AMDGPU::V_MOV_B16_t16_e64) {
6582 BuildMI(*MBB, I, DL, get(Opcode), Reg)
6583 .addImm(0) // src0_modifiers
6584 .add(MO)
6585 .addImm(0); // op_sel
6586 } else {
6587 BuildMI(*MBB, I, DL, get(Opcode), Reg).add(MO);
6588 }
6589
6590 MO.ChangeToRegister(Reg, false);
6591}
6592
6595 const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC,
6596 unsigned SubIdx, const TargetRegisterClass *SubRC) const {
6597 if (!SuperReg.getReg().isVirtual())
6598 return RI.getSubReg(SuperReg.getReg(), SubIdx);
6599
6600 MachineBasicBlock *MBB = MI->getParent();
6601 const DebugLoc &DL = MI->getDebugLoc();
6602 Register SubReg = MRI.createVirtualRegister(SubRC);
6603
6604 unsigned NewSubIdx = RI.composeSubRegIndices(SuperReg.getSubReg(), SubIdx);
6605 BuildMI(*MBB, MI, DL, get(TargetOpcode::COPY), SubReg)
6606 .addReg(SuperReg.getReg(), {}, NewSubIdx);
6607 return SubReg;
6608}
6609
6612 const MachineOperand &Op, const TargetRegisterClass *SuperRC,
6613 unsigned SubIdx, const TargetRegisterClass *SubRC) const {
6614 if (Op.isImm()) {
6615 if (SubIdx == AMDGPU::sub0)
6616 return MachineOperand::CreateImm(static_cast<int32_t>(Op.getImm()));
6617 if (SubIdx == AMDGPU::sub1)
6618 return MachineOperand::CreateImm(static_cast<int32_t>(Op.getImm() >> 32));
6619
6620 llvm_unreachable("Unhandled register index for immediate");
6621 }
6622
6623 unsigned SubReg = buildExtractSubReg(MII, MRI, Op, SuperRC,
6624 SubIdx, SubRC);
6625 return MachineOperand::CreateReg(SubReg, false);
6626}
6627
6628// Change the order of operands from (0, 1, 2) to (0, 2, 1)
6629void SIInstrInfo::swapOperands(MachineInstr &Inst) const {
6630 assert(Inst.getNumExplicitOperands() == 3);
6631 MachineOperand Op1 = Inst.getOperand(1);
6632 Inst.removeOperand(1);
6633 Inst.addOperand(Op1);
6634}
6635
6637 const MCOperandInfo &OpInfo,
6638 const MachineOperand &MO) const {
6639 if (!MO.isReg())
6640 return false;
6641
6642 Register Reg = MO.getReg();
6643
6644 const TargetRegisterClass *DRC = RI.getRegClass(getOpRegClassID(OpInfo));
6645 if (Reg.isPhysical())
6646 return DRC->contains(Reg);
6647
6648 const TargetRegisterClass *RC = MRI.getRegClass(Reg);
6649
6650 if (MO.getSubReg()) {
6651 const TargetRegisterClass *SuperRC =
6652 RI.getLargestLegalSuperClass(RC, MRI.getMF());
6653 if (!SuperRC)
6654 return false;
6655 return RI.getMatchingSuperRegClass(SuperRC, DRC, MO.getSubReg()) != nullptr;
6656 }
6657
6658 return RI.getCommonSubClass(DRC, RC) != nullptr;
6659}
6660
6662 const MachineOperand &MO) const {
6663 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
6664 const MCOperandInfo OpInfo = MI.getDesc().operands()[OpIdx];
6665 unsigned Opc = MI.getOpcode();
6666
6667 // See SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
6668 if (MO.isReg() && RI.isSGPRReg(MRI, MO.getReg()) &&
6669 AMDGPU::isSingleSGPRReadInst(MI.getOpcode()) &&
6671 &MO))
6672 return false;
6673
6674 if (!isLegalRegOperand(MRI, OpInfo, MO))
6675 return false;
6676
6677 // check Accumulate GPR operand
6678 bool IsAGPR = RI.isAGPR(MRI, MO.getReg());
6679 if (IsAGPR && !ST.hasMAIInsts())
6680 return false;
6681 if (IsAGPR && (!ST.hasGFX90AInsts() || !MRI.reservedRegsFrozen()) &&
6682 (MI.mayLoad() || MI.mayStore() || isDS(Opc) || isMIMG(Opc)))
6683 return false;
6684 // Atomics should have both vdst and vdata either vgpr or agpr.
6685 const int VDstIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
6686 const int DataIdx = AMDGPU::getNamedOperandIdx(
6687 Opc, isDS(Opc) ? AMDGPU::OpName::data0 : AMDGPU::OpName::vdata);
6688 if ((int)OpIdx == VDstIdx && DataIdx != -1 &&
6689 MI.getOperand(DataIdx).isReg() &&
6690 RI.isAGPR(MRI, MI.getOperand(DataIdx).getReg()) != IsAGPR)
6691 return false;
6692 if ((int)OpIdx == DataIdx) {
6693 if (VDstIdx != -1 &&
6694 RI.isAGPR(MRI, MI.getOperand(VDstIdx).getReg()) != IsAGPR)
6695 return false;
6696 // DS instructions with 2 src operands also must have tied RC.
6697 const int Data1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data1);
6698 if (Data1Idx != -1 && MI.getOperand(Data1Idx).isReg() &&
6699 RI.isAGPR(MRI, MI.getOperand(Data1Idx).getReg()) != IsAGPR)
6700 return false;
6701 }
6702
6703 // Check V_ACCVGPR_WRITE_B32_e64
6704 if (Opc == AMDGPU::V_ACCVGPR_WRITE_B32_e64 && !ST.hasGFX90AInsts() &&
6705 (int)OpIdx == AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0) &&
6706 RI.isSGPRReg(MRI, MO.getReg()))
6707 return false;
6708
6709 if (ST.hasFlatScratchHiInB64InstHazard() &&
6710 MO.getReg() == AMDGPU::SRC_FLAT_SCRATCH_BASE_HI && isSALU(MI)) {
6711 if (const MachineOperand *Dst = getNamedOperand(MI, AMDGPU::OpName::sdst)) {
6712 if (AMDGPU::getRegBitWidth(*RI.getRegClassForReg(MRI, Dst->getReg())) ==
6713 64)
6714 return false;
6715 }
6716 if (Opc == AMDGPU::S_BITCMP0_B64 || Opc == AMDGPU::S_BITCMP1_B64)
6717 return false;
6718 }
6719 if (!ST.hasDPPSrc1SGPR() && isDPP(MI) && RI.isSGPRReg(MRI, MO.getReg()) &&
6720 (int)OpIdx == AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1))
6721 return false;
6722
6723 return true;
6724}
6725
6727 const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN,
6728 const MachineOperand *MO) const {
6729 constexpr unsigned NumOps = 3;
6730 constexpr AMDGPU::OpName OpNames[NumOps * 2] = {
6731 AMDGPU::OpName::src0, AMDGPU::OpName::src1,
6732 AMDGPU::OpName::src2, AMDGPU::OpName::src0_modifiers,
6733 AMDGPU::OpName::src1_modifiers, AMDGPU::OpName::src2_modifiers};
6734
6735 assert(SrcN < NumOps);
6736
6737 if (!MO) {
6738 int SrcIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[SrcN]);
6739 if (SrcIdx == -1)
6740 return true;
6741 MO = &MI.getOperand(SrcIdx);
6742 }
6743
6744 if (!MO->isReg() || !RI.isSGPRReg(MRI, MO->getReg()))
6745 return true;
6746
6747 int ModsIdx =
6748 AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpNames[NumOps + SrcN]);
6749 if (ModsIdx == -1)
6750 return false;
6751
6752 unsigned Mods = MI.getOperand(ModsIdx).getImm();
6753 bool OpSel = Mods & SISrcMods::OP_SEL_0;
6754 bool OpSelHi = Mods & SISrcMods::OP_SEL_1;
6755
6756 return !OpSel && !OpSelHi;
6757}
6758
6759bool SIInstrInfo::isOperandLegal(const MachineInstr &MI, unsigned OpIdx,
6760 const MachineOperand *MO) const {
6761 const MachineFunction &MF = *MI.getMF();
6762 const MachineRegisterInfo &MRI = MF.getRegInfo();
6763 const MCInstrDesc &InstDesc = MI.getDesc();
6764 const MCOperandInfo &OpInfo = InstDesc.operands()[OpIdx];
6765 int64_t RegClass = getOpRegClassID(OpInfo);
6766 const TargetRegisterClass *DefinedRC =
6767 RegClass != -1 ? RI.getRegClass(RegClass) : nullptr;
6768 if (!MO)
6769 MO = &MI.getOperand(OpIdx);
6770
6771 const bool IsInlineConst = !MO->isReg() && isInlineConstant(*MO, OpInfo);
6772
6773 if (isVALU(MI, /*AllowLDSDMA=*/false) && !IsInlineConst &&
6774 usesConstantBus(MRI, *MO, OpInfo)) {
6775 const MachineOperand *UsedLiteral = nullptr;
6776
6777 int ConstantBusLimit = ST.getConstantBusLimit(MI.getOpcode());
6778 int LiteralLimit = !isVOP3(MI) || ST.hasVOP3Literal() ? 1 : 0;
6779
6780 // TODO: Be more permissive with frame indexes.
6781 if (!MO->isReg() && !isInlineConstant(*MO, OpInfo)) {
6782 if (!LiteralLimit--)
6783 return false;
6784
6785 UsedLiteral = MO;
6786 }
6787
6789 if (MO->isReg())
6790 SGPRsUsed.insert(RegSubRegPair(MO->getReg(), MO->getSubReg()));
6791
6792 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
6793 if (i == OpIdx)
6794 continue;
6795 const MachineOperand &Op = MI.getOperand(i);
6796 if (Op.isReg()) {
6797 if (Op.isUse()) {
6798 RegSubRegPair SGPR(Op.getReg(), Op.getSubReg());
6799 if (regUsesConstantBus(Op, MRI) && SGPRsUsed.insert(SGPR).second) {
6800 if (--ConstantBusLimit <= 0)
6801 return false;
6802 }
6803 }
6804 } else if (AMDGPU::isSISrcOperand(InstDesc.operands()[i]) &&
6805 !isInlineConstant(Op, InstDesc.operands()[i])) {
6806 // The same literal may be used multiple times.
6807 if (!UsedLiteral)
6808 UsedLiteral = &Op;
6809 else if (UsedLiteral->isIdenticalTo(Op))
6810 continue;
6811
6812 if (!LiteralLimit--)
6813 return false;
6814 if (--ConstantBusLimit <= 0)
6815 return false;
6816 }
6817 }
6818 } else if (!IsInlineConst && !MO->isReg() && isSALU(MI)) {
6819 // There can be at most one literal operand, but it can be repeated.
6820 for (unsigned i = 0, e = MI.getNumOperands(); i != e; ++i) {
6821 if (i == OpIdx)
6822 continue;
6823 const MachineOperand &Op = MI.getOperand(i);
6824 if (!Op.isReg() && !Op.isFI() && !Op.isRegMask() &&
6825 !isInlineConstant(Op, InstDesc.operands()[i]) &&
6826 !Op.isIdenticalTo(*MO))
6827 return false;
6828
6829 // Do not fold a non-inlineable and non-register operand into an
6830 // instruction that already has a frame index. The frame index handling
6831 // code could not handle well when a frame index co-exists with another
6832 // non-register operand, unless that operand is an inlineable immediate.
6833 if (Op.isFI())
6834 return false;
6835 }
6836 }
6837
6838 if (MO->isReg()) {
6839 if (!DefinedRC)
6840 return OpInfo.OperandType == MCOI::OPERAND_UNKNOWN;
6841 return isLegalRegOperand(MI, OpIdx, *MO);
6842 }
6843
6844 if (MO->isImm()) {
6845 uint64_t Imm = MO->getImm();
6846 bool Is64BitFPOp = OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_FP64 ||
6847 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2FP64;
6848 bool Is64BitOp = Is64BitFPOp ||
6849 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_INT64 ||
6850 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2INT32 ||
6851 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2FP32 ||
6852 OpInfo.OperandType == AMDGPU::OPERAND_REG_IMM_V2INT64;
6853 if (Is64BitOp &&
6854 !AMDGPU::isInlinableLiteral64(Imm, ST.hasInv2PiInlineImm())) {
6855 if (!AMDGPU::isValid32BitLiteral(Imm, Is64BitFPOp) &&
6856 (!ST.has64BitLiterals() || InstDesc.getSize() != 4))
6857 return false;
6858
6859 // FIXME: We can use sign extended 64-bit literals, but only for signed
6860 // operands. At the moment we do not know if an operand is signed.
6861 // Such operand will be encoded as its low 32 bits and then either
6862 // correctly sign extended or incorrectly zero extended by HW.
6863 // If 64-bit literals are supported and the literal will be encoded
6864 // as full 64 bit we still can use it.
6865 if (!Is64BitFPOp && (int32_t)Imm < 0 &&
6866 (!ST.has64BitLiterals() || AMDGPU::isValid32BitLiteral(Imm, false)))
6867 return false;
6868 }
6869 }
6870
6871 // Handle non-register types that are treated like immediates.
6872 assert(MO->isImm() || MO->isTargetIndex() || MO->isFI() || MO->isGlobal());
6873
6874 if (!DefinedRC) {
6875 // This operand expects an immediate.
6876 return true;
6877 }
6878
6879 return isImmOperandLegal(MI, OpIdx, *MO);
6880}
6881
6883 bool IsGFX950Only = ST.hasGFX950Insts();
6884 bool IsGFX940Only = ST.hasGFX940Insts();
6885
6886 if (!IsGFX950Only && !IsGFX940Only)
6887 return false;
6888
6889 if (!isVALU(MI, /*AllowLDSDMA=*/false))
6890 return false;
6891
6892 // V_COS, V_EXP, V_RCP, etc.
6893 if (isTRANS(MI))
6894 return true;
6895
6896 // DOT2, DOT2C, DOT4, etc.
6897 if (isDOT(MI))
6898 return true;
6899
6900 // MFMA, SMFMA
6901 if (isMFMA(MI))
6902 return true;
6903
6904 unsigned Opcode = MI.getOpcode();
6905 switch (Opcode) {
6906 case AMDGPU::V_CVT_PK_BF8_F32_e64:
6907 case AMDGPU::V_CVT_PK_FP8_F32_e64:
6908 case AMDGPU::V_MQSAD_PK_U16_U8_e64:
6909 case AMDGPU::V_MQSAD_U32_U8_e64:
6910 case AMDGPU::V_PK_ADD_F16:
6911 case AMDGPU::V_PK_ADD_F32:
6912 case AMDGPU::V_PK_ADD_I16:
6913 case AMDGPU::V_PK_ADD_U16:
6914 case AMDGPU::V_PK_ASHRREV_I16:
6915 case AMDGPU::V_PK_FMA_F16:
6916 case AMDGPU::V_PK_FMA_F32:
6917 case AMDGPU::V_PK_FMAC_F16_e32:
6918 case AMDGPU::V_PK_FMAC_F16_e64:
6919 case AMDGPU::V_PK_LSHLREV_B16:
6920 case AMDGPU::V_PK_LSHRREV_B16:
6921 case AMDGPU::V_PK_MAD_I16:
6922 case AMDGPU::V_PK_MAD_U16:
6923 case AMDGPU::V_PK_MAX_F16:
6924 case AMDGPU::V_PK_MAX_I16:
6925 case AMDGPU::V_PK_MAX_U16:
6926 case AMDGPU::V_PK_MIN_F16:
6927 case AMDGPU::V_PK_MIN_I16:
6928 case AMDGPU::V_PK_MIN_U16:
6929 case AMDGPU::V_PK_MOV_B32:
6930 case AMDGPU::V_PK_MUL_F16:
6931 case AMDGPU::V_PK_MUL_F32:
6932 case AMDGPU::V_PK_MUL_LO_U16:
6933 case AMDGPU::V_PK_SUB_I16:
6934 case AMDGPU::V_PK_SUB_U16:
6935 case AMDGPU::V_QSAD_PK_U16_U8_e64:
6936 return true;
6937 default:
6938 return false;
6939 }
6940}
6941
6943 MachineInstr &MI) const {
6944 unsigned Opc = MI.getOpcode();
6945 const MCInstrDesc &InstrDesc = get(Opc);
6946
6947 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
6948 MachineOperand &Src0 = MI.getOperand(Src0Idx);
6949
6950 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
6951 MachineOperand &Src1 = MI.getOperand(Src1Idx);
6952
6953 // If there is an implicit SGPR use such as VCC use for v_addc_u32/v_subb_u32
6954 // we need to only have one constant bus use before GFX10.
6955 bool HasImplicitSGPR = findImplicitSGPRRead(MI);
6956 if (HasImplicitSGPR && ST.getConstantBusLimit(Opc) <= 1 && Src0.isReg() &&
6957 RI.isSGPRReg(MRI, Src0.getReg()))
6958 legalizeOpWithMove(MI, Src0Idx);
6959
6960 // Special case: V_WRITELANE_B32 accepts only immediate or SGPR operands for
6961 // both the value to write (src0) and lane select (src1). Fix up non-SGPR
6962 // src0/src1 with V_READFIRSTLANE.
6963 if (Opc == AMDGPU::V_WRITELANE_B32) {
6964 const DebugLoc &DL = MI.getDebugLoc();
6965 if (Src0.isReg() && RI.isVGPR(MRI, Src0.getReg())) {
6966 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6967 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
6968 .add(Src0);
6969 Src0.ChangeToRegister(Reg, false);
6970 }
6971 if (Src1.isReg() && RI.isVGPR(MRI, Src1.getReg())) {
6972 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6973 const DebugLoc &DL = MI.getDebugLoc();
6974 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
6975 .add(Src1);
6976 Src1.ChangeToRegister(Reg, false);
6977 }
6978 return;
6979 }
6980
6981 // Special case: V_FMAC_F32 and V_FMAC_F16 have src2.
6982 if (Opc == AMDGPU::V_FMAC_F32_e32 || Opc == AMDGPU::V_FMAC_F16_e32) {
6983 int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
6984 if (!RI.isVGPR(MRI, MI.getOperand(Src2Idx).getReg()))
6985 legalizeOpWithMove(MI, Src2Idx);
6986 }
6987
6988 // VOP2 src0 instructions support all operand types, so we don't need to check
6989 // their legality. If src1 is already legal, we don't need to do anything.
6990 if (isLegalRegOperand(MRI, InstrDesc.operands()[Src1Idx], Src1))
6991 return;
6992
6993 // Special case: V_READLANE_B32 accepts only immediate or SGPR operands for
6994 // lane select. Fix up using V_READFIRSTLANE, since we assume that the lane
6995 // select is uniform.
6996 if (Opc == AMDGPU::V_READLANE_B32 && Src1.isReg() &&
6997 RI.isVGPR(MRI, Src1.getReg())) {
6998 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6999 const DebugLoc &DL = MI.getDebugLoc();
7000 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7001 .add(Src1);
7002 Src1.ChangeToRegister(Reg, false);
7003 return;
7004 }
7005
7006 // We do not use commuteInstruction here because it is too aggressive and will
7007 // commute if it is possible. We only want to commute here if it improves
7008 // legality. This can be called a fairly large number of times so don't waste
7009 // compile time pointlessly swapping and checking legality again.
7010 if (HasImplicitSGPR || !MI.isCommutable()) {
7011 legalizeOpWithMove(MI, Src1Idx);
7012 return;
7013 }
7014
7015 // If src0 can be used as src1, commuting will make the operands legal.
7016 // Otherwise we have to give up and insert a move.
7017 //
7018 // TODO: Other immediate-like operand kinds could be commuted if there was a
7019 // MachineOperand::ChangeTo* for them.
7020 if ((!Src1.isImm() && !Src1.isReg()) ||
7021 !isLegalRegOperand(MRI, InstrDesc.operands()[Src1Idx], Src0)) {
7022 legalizeOpWithMove(MI, Src1Idx);
7023 return;
7024 }
7025
7026 int CommutedOpc = commuteOpcode(MI);
7027 if (CommutedOpc == -1) {
7028 legalizeOpWithMove(MI, Src1Idx);
7029 return;
7030 }
7031
7032 MI.setDesc(get(CommutedOpc));
7033
7034 Register Src0Reg = Src0.getReg();
7035 unsigned Src0SubReg = Src0.getSubReg();
7036 bool Src0Kill = Src0.isKill();
7037
7038 if (Src1.isImm())
7039 Src0.ChangeToImmediate(Src1.getImm());
7040 else if (Src1.isReg()) {
7041 Src0.ChangeToRegister(Src1.getReg(), false, false, Src1.isKill());
7042 Src0.setSubReg(Src1.getSubReg());
7043 } else
7044 llvm_unreachable("Should only have register or immediate operands");
7045
7046 Src1.ChangeToRegister(Src0Reg, false, false, Src0Kill);
7047 Src1.setSubReg(Src0SubReg);
7049}
7050
7051// Legalize VOP3 operands. All operand types are supported for any operand
7052// but only one literal constant and only starting from GFX10.
7054 MachineInstr &MI) const {
7055 unsigned Opc = MI.getOpcode();
7056
7057 int VOP3Idx[3] = {
7058 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0),
7059 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1),
7060 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2)
7061 };
7062
7063 if (Opc == AMDGPU::V_PERMLANE16_B32_e64 ||
7064 Opc == AMDGPU::V_PERMLANEX16_B32_e64 ||
7065 Opc == AMDGPU::V_PERMLANE_BCAST_B32_e64 ||
7066 Opc == AMDGPU::V_PERMLANE_UP_B32_e64 ||
7067 Opc == AMDGPU::V_PERMLANE_DOWN_B32_e64 ||
7068 Opc == AMDGPU::V_PERMLANE_XOR_B32_e64 ||
7069 Opc == AMDGPU::V_PERMLANE_IDX_GEN_B32_e64) {
7070 // src1 and src2 must be scalar
7071 MachineOperand &Src1 = MI.getOperand(VOP3Idx[1]);
7072 const DebugLoc &DL = MI.getDebugLoc();
7073 if (Src1.isReg() && !RI.isSGPRClass(MRI.getRegClass(Src1.getReg()))) {
7074 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7075 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7076 .add(Src1);
7077 Src1.ChangeToRegister(Reg, false);
7078 }
7079 if (VOP3Idx[2] != -1) {
7080 MachineOperand &Src2 = MI.getOperand(VOP3Idx[2]);
7081 if (Src2.isReg() && !RI.isSGPRClass(MRI.getRegClass(Src2.getReg()))) {
7082 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7083 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7084 .add(Src2);
7085 Src2.ChangeToRegister(Reg, false);
7086 }
7087 }
7088 }
7089
7090 // Find the one SGPR operand we are allowed to use.
7091 int ConstantBusLimit = ST.getConstantBusLimit(Opc);
7092 int LiteralLimit = ST.hasVOP3Literal() ? 1 : 0;
7093 SmallDenseSet<unsigned> SGPRsUsed;
7094 Register SGPRReg = findUsedSGPR(MI, VOP3Idx);
7095 if (SGPRReg) {
7096 SGPRsUsed.insert(SGPRReg);
7097 --ConstantBusLimit;
7098 }
7099
7100 for (int Idx : VOP3Idx) {
7101 if (Idx == -1)
7102 break;
7103 MachineOperand &MO = MI.getOperand(Idx);
7104
7105 if (!MO.isReg()) {
7106 if (isInlineConstant(MO, get(Opc).operands()[Idx]))
7107 continue;
7108
7109 if (LiteralLimit > 0 && ConstantBusLimit > 0) {
7110 --LiteralLimit;
7111 --ConstantBusLimit;
7112 continue;
7113 }
7114
7115 --LiteralLimit;
7116 --ConstantBusLimit;
7117 legalizeOpWithMove(MI, Idx);
7118 continue;
7119 }
7120
7121 if (!RI.isSGPRClass(RI.getRegClassForReg(MRI, MO.getReg())))
7122 continue; // VGPRs are legal
7123
7124 // We can use one SGPR in each VOP3 instruction prior to GFX10
7125 // and two starting from GFX10.
7126 if (SGPRsUsed.count(MO.getReg()))
7127 continue;
7128 if (ConstantBusLimit > 0) {
7129 SGPRsUsed.insert(MO.getReg());
7130 --ConstantBusLimit;
7131 continue;
7132 }
7133
7134 // If we make it this far, then the operand is not legal and we must
7135 // legalize it.
7136 legalizeOpWithMove(MI, Idx);
7137 }
7138
7139 // Special case: V_FMAC_F32 and V_FMAC_F16 have src2 tied to vdst.
7140 if ((Opc == AMDGPU::V_FMAC_F32_e64 || Opc == AMDGPU::V_FMAC_F16_e64) &&
7141 !RI.isVGPR(MRI, MI.getOperand(VOP3Idx[2]).getReg()))
7142 legalizeOpWithMove(MI, VOP3Idx[2]);
7143
7144 // Fix the register class of single-sgpr-read instructions on gfx12+. See
7145 // SIInstrInfo::isLegalSingleSGPRReadInstOperand for more information.
7147 for (unsigned I = 0; I < 3; ++I) {
7148 if (!isLegalSingleSGPRReadInstOperand(MRI, MI, /*SrcN=*/I))
7149 legalizeOpWithMove(MI, VOP3Idx[I]);
7150 }
7151 }
7152}
7153
7156 const TargetRegisterClass *DstRC /*=nullptr*/) const {
7157 const TargetRegisterClass *VRC = MRI.getRegClass(SrcReg);
7158 const TargetRegisterClass *SRC = RI.getEquivalentSGPRClass(VRC);
7159 if (DstRC)
7160 SRC = RI.getCommonSubClass(SRC, DstRC);
7161
7162 Register DstReg = MRI.createVirtualRegister(SRC);
7163 unsigned SubRegs = RI.getRegSizeInBits(*VRC) / 32;
7164
7165 if (RI.hasAGPRs(VRC)) {
7166 VRC = RI.getEquivalentVGPRClass(VRC);
7167 Register NewSrcReg = MRI.createVirtualRegister(VRC);
7168 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7169 get(TargetOpcode::COPY), NewSrcReg)
7170 .addReg(SrcReg);
7171 SrcReg = NewSrcReg;
7172 }
7173
7174 if (SubRegs == 1) {
7175 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7176 get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
7177 .addReg(SrcReg);
7178 return DstReg;
7179 }
7180
7182 for (unsigned i = 0; i < SubRegs; ++i) {
7183 Register SGPR = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7184 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7185 get(AMDGPU::V_READFIRSTLANE_B32), SGPR)
7186 .addReg(SrcReg, {}, RI.getSubRegFromChannel(i));
7187 SRegs.push_back(SGPR);
7188 }
7189
7191 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(),
7192 get(AMDGPU::REG_SEQUENCE), DstReg);
7193 for (unsigned i = 0; i < SubRegs; ++i) {
7194 MIB.addReg(SRegs[i]);
7195 MIB.addImm(RI.getSubRegFromChannel(i));
7196 }
7197 return DstReg;
7198}
7199
7201 MachineInstr &MI) const {
7202
7203 // If the pointer is store in VGPRs, then we need to move them to
7204 // SGPRs using v_readfirstlane. This is safe because we only select
7205 // loads with uniform pointers to SMRD instruction so we know the
7206 // pointer value is uniform.
7207 MachineOperand *SBase = getNamedOperand(MI, AMDGPU::OpName::sbase);
7208 if (SBase && !RI.isSGPRClass(MRI.getRegClass(SBase->getReg()))) {
7209 Register SGPR = readlaneVGPRToSGPR(SBase->getReg(), MI, MRI);
7210 SBase->setReg(SGPR);
7211 }
7212 MachineOperand *SOff = getNamedOperand(MI, AMDGPU::OpName::soffset);
7213 if (SOff && !RI.isSGPRReg(MRI, SOff->getReg())) {
7214 Register SGPR = readlaneVGPRToSGPR(SOff->getReg(), MI, MRI);
7215 SOff->setReg(SGPR);
7216 }
7217}
7218
7220 unsigned Opc = Inst.getOpcode();
7221 int OldSAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::saddr);
7222 if (OldSAddrIdx < 0)
7223 return false;
7224
7225 assert(isSegmentSpecificFLAT(Inst) || (isFLAT(Inst) && ST.hasFlatGVSMode()));
7226
7227 int NewOpc = AMDGPU::getGlobalVaddrOp(Opc);
7228 if (NewOpc < 0)
7230 if (NewOpc < 0)
7231 return false;
7232
7233 MachineRegisterInfo &MRI = Inst.getMF()->getRegInfo();
7234 MachineOperand &SAddr = Inst.getOperand(OldSAddrIdx);
7235 if (RI.isSGPRReg(MRI, SAddr.getReg()))
7236 return false;
7237
7238 int NewVAddrIdx = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vaddr);
7239 if (NewVAddrIdx < 0)
7240 return false;
7241
7242 int OldVAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
7243
7244 // Check vaddr, it shall be zero or absent.
7245 MachineInstr *VAddrDef = nullptr;
7246 if (OldVAddrIdx >= 0) {
7247 MachineOperand &VAddr = Inst.getOperand(OldVAddrIdx);
7248 VAddrDef = MRI.getUniqueVRegDef(VAddr.getReg());
7249 if (!VAddrDef || !VAddrDef->isMoveImmediate() ||
7250 !VAddrDef->getOperand(1).isImm() ||
7251 VAddrDef->getOperand(1).getImm() != 0)
7252 return false;
7253 }
7254
7255 const MCInstrDesc &NewDesc = get(NewOpc);
7256 Inst.setDesc(NewDesc);
7257
7258 // Callers expect iterator to be valid after this call, so modify the
7259 // instruction in place.
7260 if (OldVAddrIdx == NewVAddrIdx) {
7261 MachineOperand &NewVAddr = Inst.getOperand(NewVAddrIdx);
7262 // Clear use list from the old vaddr holding a zero register.
7263 MRI.removeRegOperandFromUseList(&NewVAddr);
7264 MRI.moveOperands(&NewVAddr, &SAddr, 1);
7265 Inst.removeOperand(OldSAddrIdx);
7266 // Update the use list with the pointer we have just moved from vaddr to
7267 // saddr position. Otherwise new vaddr will be missing from the use list.
7268 MRI.removeRegOperandFromUseList(&NewVAddr);
7269 MRI.addRegOperandToUseList(&NewVAddr);
7270 } else {
7271 assert(OldSAddrIdx == NewVAddrIdx);
7272
7273 if (OldVAddrIdx >= 0) {
7274 int NewVDstIn = AMDGPU::getNamedOperandIdx(NewOpc,
7275 AMDGPU::OpName::vdst_in);
7276
7277 // removeOperand doesn't try to fixup tied operand indexes at it goes, so
7278 // it asserts. Untie the operands for now and retie them afterwards.
7279 if (NewVDstIn != -1) {
7280 int OldVDstIn = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
7281 Inst.untieRegOperand(OldVDstIn);
7282 }
7283
7284 Inst.removeOperand(OldVAddrIdx);
7285
7286 if (NewVDstIn != -1) {
7287 int NewVDst = AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
7288 Inst.tieOperands(NewVDst, NewVDstIn);
7289 }
7290 }
7291 }
7292
7293 if (VAddrDef && MRI.use_nodbg_empty(VAddrDef->getOperand(0).getReg()))
7294 VAddrDef->eraseFromParent();
7295
7296 return true;
7297}
7298
7299// FIXME: Remove this when SelectionDAG is obsoleted.
7301 MachineInstr &MI) const {
7302 if (!isSegmentSpecificFLAT(MI) && !ST.hasFlatGVSMode())
7303 return;
7304
7305 // Fixup SGPR operands in VGPRs. We only select these when the DAG divergence
7306 // thinks they are uniform, so a readfirstlane should be valid.
7307 MachineOperand *SAddr = getNamedOperand(MI, AMDGPU::OpName::saddr);
7308 if (!SAddr || RI.isSGPRClass(MRI.getRegClass(SAddr->getReg())))
7309 return;
7310
7312 return;
7313
7314 const TargetRegisterClass *DeclaredRC =
7315 getRegClass(MI.getDesc(), SAddr->getOperandNo());
7316
7317 Register ToSGPR = readlaneVGPRToSGPR(SAddr->getReg(), MI, MRI, DeclaredRC);
7318 SAddr->setReg(ToSGPR);
7319}
7320
7323 const TargetRegisterClass *DstRC,
7326 const DebugLoc &DL) const {
7327 Register OpReg = Op.getReg();
7328 unsigned OpSubReg = Op.getSubReg();
7329
7330 const TargetRegisterClass *OpRC = RI.getSubClassWithSubReg(
7331 RI.getRegClassForReg(MRI, OpReg), OpSubReg);
7332
7333 // Check if operand is already the correct register class.
7334 if (DstRC == OpRC)
7335 return;
7336
7337 Register DstReg = MRI.createVirtualRegister(DstRC);
7338 auto Copy = BuildMI(InsertMBB, I, DL, get(AMDGPU::COPY), DstReg)
7339 .addReg(OpReg, {}, OpSubReg);
7340 Op.setReg(DstReg);
7341 Op.setSubReg(AMDGPU::NoSubRegister);
7342
7343 MachineInstr *Def = MRI.getVRegDef(OpReg);
7344 if (!Def)
7345 return;
7346
7347 // Try to eliminate the copy if it is copying an immediate value.
7348 if (Def->isMoveImmediate() && DstRC != &AMDGPU::VReg_1RegClass)
7349 foldImmediate(*Copy, *Def, OpReg, &MRI);
7350
7351 bool ImpDef = Def->isImplicitDef();
7352 while (!ImpDef && Def && Def->isCopy()) {
7353 if (Def->getOperand(1).getReg().isPhysical())
7354 break;
7355 Def = MRI.getUniqueVRegDef(Def->getOperand(1).getReg());
7356 ImpDef = Def && Def->isImplicitDef();
7357 }
7358 if (!RI.isSGPRClass(DstRC) && !Copy->readsRegister(AMDGPU::EXEC, &RI) &&
7359 !ImpDef)
7360 Copy.addReg(AMDGPU::EXEC, RegState::Implicit);
7361}
7362
7363// Emit the actual waterfall loop, executing the wrapped instruction for each
7364// unique value of \p ScalarOps across all lanes. In the best case we execute 1
7365// iteration, in the worst case we execute 64 (once per lane).
7368 MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL,
7369 ArrayRef<MachineOperand *> ScalarOps, ArrayRef<Register> PhySGPRs = {}) {
7370 MachineFunction &MF = *LoopBB.getParent();
7372 const SIRegisterInfo *TRI = ST.getRegisterInfo();
7374 const auto *BoolXExecRC = TRI->getWaveMaskRegClass();
7375
7376 // Emit v_cmpx_eq and s_andn2_wrexec when both instructions are
7377 // available. Otherwise, use the previous pattern of v_cmp_eq,
7378 // s_and_saveexec, and s_xor.
7379 bool UseNewExecInstructions =
7380 ST.hasNoSdstCMPX() && TII.pseudoToMCOpcode(LMC.AndN2WrExecOpc) != -1;
7381
7383 Register CondReg;
7384
7385 Register PhiExec;
7386 Register NewExec;
7387
7388 if (UseNewExecInstructions) {
7389 PhiExec = MRI.createVirtualRegister(BoolXExecRC);
7390 NewExec = MRI.createVirtualRegister(BoolXExecRC);
7391 Register InitExec = MRI.createVirtualRegister(BoolXExecRC);
7392 BuildMI(PredBB, PredBB.end(), DL, TII.get(LMC.MovOpc), InitExec)
7393 .addReg(LMC.ExecReg);
7394
7395 BuildMI(LoopBB, I, DL, TII.get(TargetOpcode::PHI), PhiExec)
7396 .addReg(InitExec)
7397 .addMBB(&PredBB)
7398 .addReg(NewExec)
7399 .addMBB(&BodyBB);
7400 }
7401
7402 // Placement of v_cmpx instructions (when index is longer than 64 bit)
7403 // involves a trade-off between register pressure and latency:
7404 // (a) Defering all v_cmpx after all v_readfirstlane may increase
7405 // register pressure because arguments and results of all
7406 // v_readfirstlane instructions must stay live until deferred v_cmpx use them.
7407 // (b) Interleaving v_cmpx with v_readfirstlanes may reduce live ranges and
7408 // increase latency by placing v_readfirstlane instructions
7409 // immediately before v_cmpx instruction that directly depend on it.
7410 ///
7411 // Emitting interleaved v_cmpx and v_readfirstlane requires
7412 // block splitting because v_cmpx changes EXEC mask and therefore for safety
7413 // v_cmpx needs to be treated as terminator until after register allocation
7414 // (spill placement) and instruction reordering.
7415 //
7416 // Current implementation defers v_cmpx and leaves other instruction
7417 // scheduling decisions to later passes, where register pressure is known or
7418 // easier to approximate.
7419 // Non-terminators (V_READFIRSTLANE and REG_SEQUENCE) are inserted before I;
7420 // v_cmpx instructions are inserted at the end of LoopBB.
7421 // After the first v_cmpx is emitted, I is updated to point to it
7422 // so subsequent non-terminators are inserted before all v_cmpx instructions.
7423 for (auto [Idx, ScalarOp] : enumerate(ScalarOps)) {
7424 unsigned RegSize = TRI->getRegSizeInBits(ScalarOp->getReg(), MRI);
7425 unsigned NumSubRegs = RegSize / 32;
7426 Register VScalarOp = ScalarOp->getReg();
7427
7428 const TargetRegisterClass *RFLSrcRC =
7429 TII.getRegClass(TII.get(AMDGPU::V_READFIRSTLANE_B32), 1);
7430
7431 if (NumSubRegs == 1) {
7432 const TargetRegisterClass *VScalarOpRC = MRI.getRegClass(VScalarOp);
7433 if (const TargetRegisterClass *Common =
7434 TRI->getCommonSubClass(VScalarOpRC, RFLSrcRC);
7435 Common != VScalarOpRC) {
7436 Register VRReg = MRI.createVirtualRegister(Common);
7437 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::COPY), VRReg).addReg(VScalarOp);
7438 VScalarOp = VRReg;
7439 }
7440 Register CurReg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7441
7442 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurReg)
7443 .addReg(VScalarOp);
7444
7445 if (UseNewExecInstructions) {
7446 auto CmpxMI = BuildMI(LoopBB, LoopBB.end(), DL,
7447 TII.get(AMDGPU::V_CMPX_EQ_U32_nosdst_e32_term))
7448 .addReg(CurReg)
7449 .addReg(VScalarOp);
7450 if (I == LoopBB.end())
7451 I = CmpxMI.getInstr()->getIterator();
7452 } else {
7453 Register NewCondReg = MRI.createVirtualRegister(BoolXExecRC);
7454
7455 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_CMP_EQ_U32_e64), NewCondReg)
7456 .addReg(CurReg)
7457 .addReg(VScalarOp);
7458
7459 // Combine the comparison results with AND.
7460 if (!CondReg) { // First.
7461 CondReg = NewCondReg;
7462 } else { // If not the first, we create an AND.
7463 Register AndReg = MRI.createVirtualRegister(BoolXExecRC);
7464 BuildMI(LoopBB, I, DL, TII.get(LMC.AndOpc), AndReg)
7465 .addReg(CondReg)
7466 .addReg(NewCondReg)
7467 .setOperandDead(3);
7468 CondReg = AndReg;
7469 }
7470 }
7471
7472 // Update ScalarOp operand to use the SGPR ScalarOp.
7473 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7474 ScalarOp->setReg(CurReg);
7475 else {
7476 // Insert into the same block of use
7477 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(), DL,
7478 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7479 .addReg(CurReg);
7480 ScalarOp->setReg(PhySGPRs[Idx]);
7481 }
7482 ScalarOp->setIsKill();
7483 } else {
7484 SmallVector<Register, 8> ReadlanePieces;
7485 RegState VScalarOpUndef = getUndefRegState(ScalarOp->isUndef());
7486 assert(NumSubRegs % 2 == 0 && NumSubRegs <= 32 &&
7487 "Unhandled register size");
7488
7489 for (unsigned Idx = 0; Idx < NumSubRegs; Idx += 2) {
7490 Register CurRegLo =
7491 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7492 Register CurRegHi =
7493 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7494
7495 // Read the next variant <- also loop target.
7496 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegLo)
7497 .addReg(VScalarOp, VScalarOpUndef, TRI->getSubRegFromChannel(Idx));
7498
7499 // Read the next variant <- also loop target.
7500 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), CurRegHi)
7501 .addReg(VScalarOp, VScalarOpUndef,
7502 TRI->getSubRegFromChannel(Idx + 1));
7503
7504 ReadlanePieces.push_back(CurRegLo);
7505 ReadlanePieces.push_back(CurRegHi);
7506
7507 // Comparison is to be done as 64-bit.
7508 Register CurReg = MRI.createVirtualRegister(&AMDGPU::SGPR_64RegClass);
7509 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), CurReg)
7510 .addReg(CurRegLo)
7511 .addImm(AMDGPU::sub0)
7512 .addReg(CurRegHi)
7513 .addImm(AMDGPU::sub1);
7514
7515 unsigned SubReg =
7516 NumSubRegs <= 2 ? 0 : TRI->getSubRegFromChannel(Idx, 2);
7517
7518 if (UseNewExecInstructions) {
7519 auto CmpxMI = BuildMI(LoopBB, LoopBB.end(), DL,
7520 TII.get(AMDGPU::V_CMPX_EQ_U64_nosdst_e32_term))
7521 .addReg(CurReg)
7522 .addReg(VScalarOp, VScalarOpUndef, SubReg);
7523 if (I == LoopBB.end())
7524 I = CmpxMI.getInstr()->getIterator();
7525 } else {
7526 Register NewCondReg = MRI.createVirtualRegister(BoolXExecRC);
7527 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::V_CMP_EQ_U64_e64), NewCondReg)
7528 .addReg(CurReg)
7529 .addReg(VScalarOp, VScalarOpUndef, SubReg);
7530
7531 // Combine the comparison results with AND.
7532 if (!CondReg) { // First.
7533 CondReg = NewCondReg;
7534 } else { // If not the first, we create an AND.
7535 Register AndReg = MRI.createVirtualRegister(BoolXExecRC);
7536 BuildMI(LoopBB, I, DL, TII.get(LMC.AndOpc), AndReg)
7537 .addReg(CondReg)
7538 .addReg(NewCondReg)
7539 .setOperandDead(3);
7540 CondReg = AndReg;
7541 }
7542 }
7543 } // End for loop.
7544
7545 const auto *SScalarOpRC =
7546 TRI->getEquivalentSGPRClass(MRI.getRegClass(VScalarOp));
7547 Register SScalarOp = MRI.createVirtualRegister(SScalarOpRC);
7548
7549 // Build scalar ScalarOp.
7550 auto Merge =
7551 BuildMI(LoopBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), SScalarOp);
7552 unsigned Channel = 0;
7553 for (Register Piece : ReadlanePieces) {
7554 Merge.addReg(Piece).addImm(TRI->getSubRegFromChannel(Channel++));
7555 }
7556
7557 // Update ScalarOp operand to use the SGPR ScalarOp.
7558 if (PhySGPRs.empty() || !PhySGPRs[Idx].isValid())
7559 ScalarOp->setReg(SScalarOp);
7560 else {
7561 BuildMI(*ScalarOp->getParent()->getParent(), ScalarOp->getParent(), DL,
7562 TII.get(AMDGPU::COPY), PhySGPRs[Idx])
7563 .addReg(SScalarOp);
7564 ScalarOp->setReg(PhySGPRs[Idx]);
7565 }
7566 ScalarOp->setIsKill();
7567 }
7568 }
7569
7570 // AndSaveExecOpc modifies EXEC but can't be isTerminator=1: terminators
7571 // that define virtual registers aren't supported.
7572 Register SaveExec;
7573 if (!UseNewExecInstructions) {
7574 SaveExec = MRI.createVirtualRegister(BoolXExecRC);
7575 MRI.setSimpleHint(SaveExec, CondReg);
7576
7577 // Update EXEC to matching lanes, saving original to SaveExec.
7578 BuildMI(LoopBB, I, DL, TII.get(LMC.AndSaveExecOpc), SaveExec)
7579 .addReg(CondReg, RegState::Kill)
7580 .setOperandDead(3);
7581 }
7582
7583 // The original instruction is here; we insert the terminators after it.
7584 I = BodyBB.end();
7585
7586 if (UseNewExecInstructions) {
7587 // Compute the remaining lanes into a plain virtual register and write EXEC
7588 // from a terminator, so spill code for NewExec is placed before EXEC
7589 // changes. SIOptimizeExecMasking opportunistically folds the pair back
7590 // into S_ANDN2_WREXEC after register allocation.
7591 MRI.setSimpleHint(NewExec, PhiExec);
7592 BuildMI(BodyBB, I, DL, TII.get(LMC.AndN2Opc), NewExec)
7593 .addReg(PhiExec)
7594 .addReg(LMC.ExecReg)
7595 .setOperandDead(3);
7596 BuildMI(BodyBB, I, DL, TII.get(LMC.MovTermOpc), LMC.ExecReg)
7597 .addReg(NewExec);
7598 } else {
7599 // Update EXEC, switch all done bits to 0 and all todo bits to 1.
7600 BuildMI(BodyBB, I, DL, TII.get(LMC.XorTermOpc), LMC.ExecReg)
7601 .addReg(LMC.ExecReg)
7602 .addReg(SaveExec)
7603 .setOperandDead(3);
7604 }
7605
7606 BuildMI(BodyBB, I, DL, TII.get(AMDGPU::SI_WATERFALL_LOOP)).addMBB(&LoopBB);
7607}
7608
7609// Build a waterfall loop around \p MI, replacing the VGPR \p ScalarOp register
7610// with SGPRs by iterating over all unique values across all lanes.
7611// Returns the loop basic block that now contains \p MI.
7612static MachineBasicBlock *
7616 MachineBasicBlock::iterator Begin = nullptr,
7617 MachineBasicBlock::iterator End = nullptr,
7618 ArrayRef<Register> PhySGPRs = {}) {
7619 assert((PhySGPRs.empty() || PhySGPRs.size() == ScalarOps.size()) &&
7620 "Physical SGPRs must be empty or match the number of scalar operands");
7622 MachineFunction &MF = *MBB.getParent();
7624 const SIRegisterInfo *TRI = ST.getRegisterInfo();
7625 MachineRegisterInfo &MRI = MF.getRegInfo();
7626 if (!Begin.isValid())
7627 Begin = &MI;
7628 if (!End.isValid()) {
7629 End = &MI;
7630 ++End;
7631 }
7632 const DebugLoc &DL = MI.getDebugLoc();
7634 const auto *BoolXExecRC = TRI->getWaveMaskRegClass();
7635
7636 // Save SCC. Waterfall Loop may overwrite SCC.
7637 Register SaveSCCReg;
7638
7639 // FIXME: We should maintain SCC liveness while doing the FixSGPRCopies walk
7640 // rather than unlimited scan everywhere
7641 bool SCCNotDead =
7642 MBB.computeRegisterLiveness(TRI, AMDGPU::SCC, MI,
7643 std::numeric_limits<unsigned>::max()) !=
7645 if (SCCNotDead) {
7646 SaveSCCReg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7647 BuildMI(MBB, Begin, DL, TII.get(AMDGPU::S_CSELECT_B32), SaveSCCReg)
7648 .addImm(1)
7649 .addImm(0);
7650 }
7651
7652 Register SaveExec = MRI.createVirtualRegister(BoolXExecRC);
7653
7654 // Save the EXEC mask
7655 BuildMI(MBB, Begin, DL, TII.get(LMC.MovOpc), SaveExec).addReg(LMC.ExecReg);
7656
7657 // Killed uses in the instruction we are waterfalling around will be
7658 // incorrect due to the added control-flow.
7660 ++AfterMI;
7661 for (auto I = Begin; I != AfterMI; I++) {
7662 for (auto &MO : I->all_uses())
7663 MRI.clearKillFlags(MO.getReg());
7664 }
7665
7666 // To insert the loop we need to split the block. Move everything after this
7667 // point to a new block, and insert a new empty block between the two.
7670 MachineBasicBlock *RemainderBB = MF.CreateMachineBasicBlock();
7672 ++MBBI;
7673
7674 MF.insert(MBBI, LoopBB);
7675 MF.insert(MBBI, BodyBB);
7676 MF.insert(MBBI, RemainderBB);
7677
7678 LoopBB->addSuccessor(BodyBB);
7679 BodyBB->addSuccessor(LoopBB);
7680 BodyBB->addSuccessor(RemainderBB);
7681
7682 // Move Begin to MI to the BodyBB, and the remainder of the block to
7683 // RemainderBB.
7684 RemainderBB->transferSuccessorsAndUpdatePHIs(&MBB);
7685 RemainderBB->splice(RemainderBB->begin(), &MBB, End, MBB.end());
7686 BodyBB->splice(BodyBB->begin(), &MBB, Begin, MBB.end());
7687
7688 MBB.addSuccessor(LoopBB);
7689
7690 // Update dominators. We know that MBB immediately dominates LoopBB, that
7691 // LoopBB immediately dominates BodyBB, and BodyBB immediately dominates
7692 // RemainderBB. RemainderBB immediately dominates all of the successors
7693 // transferred to it from MBB that MBB used to properly dominate.
7694 if (MDT) {
7695 MDT->addNewBlock(LoopBB, &MBB);
7696 MDT->addNewBlock(BodyBB, LoopBB);
7697 MDT->addNewBlock(RemainderBB, BodyBB);
7698 for (auto &Succ : RemainderBB->successors()) {
7699 if (MDT->properlyDominates(&MBB, Succ)) {
7700 MDT->changeImmediateDominator(Succ, RemainderBB);
7701 }
7702 }
7703 }
7704
7705 emitLoadScalarOpsFromVGPRLoop(TII, MRI, MBB, *LoopBB, *BodyBB, DL, ScalarOps,
7706 PhySGPRs);
7707
7708 MachineBasicBlock::iterator First = RemainderBB->begin();
7709 // Restore SCC
7710 if (SCCNotDead) {
7711 BuildMI(*RemainderBB, First, DL, TII.get(AMDGPU::S_CMP_LG_U32))
7712 .addReg(SaveSCCReg, RegState::Kill)
7713 .addImm(0);
7714 }
7715
7716 // Restore the EXEC mask
7717 BuildMI(*RemainderBB, First, DL, TII.get(LMC.MovOpc), LMC.ExecReg)
7718 .addReg(SaveExec);
7719 return BodyBB;
7720}
7721
7722// Extract pointer from Rsrc and return a zero-value Rsrc replacement.
7723static std::tuple<unsigned, unsigned>
7725 MachineBasicBlock &MBB = *MI.getParent();
7726 MachineFunction &MF = *MBB.getParent();
7727 MachineRegisterInfo &MRI = MF.getRegInfo();
7728
7729 // Extract the ptr from the resource descriptor.
7730 unsigned RsrcPtr =
7731 TII.buildExtractSubReg(MI, MRI, Rsrc, &AMDGPU::VReg_128RegClass,
7732 AMDGPU::sub0_sub1, &AMDGPU::VReg_64RegClass);
7733
7734 // Create an empty resource descriptor
7735 Register Zero64 = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
7736 Register SRsrcFormatLo = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7737 Register SRsrcFormatHi = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
7738 Register NewSRsrc = MRI.createVirtualRegister(&AMDGPU::SGPR_128RegClass);
7739 uint64_t RsrcDataFormat = TII.getDefaultRsrcDataFormat();
7740
7741 // Zero64 = 0
7742 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B64), Zero64)
7743 .addImm(0);
7744
7745 // SRsrcFormatLo = RSRC_DATA_FORMAT{31-0}
7746 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), SRsrcFormatLo)
7747 .addImm(Lo_32(RsrcDataFormat));
7748
7749 // SRsrcFormatHi = RSRC_DATA_FORMAT{63-32}
7750 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), SRsrcFormatHi)
7751 .addImm(Hi_32(RsrcDataFormat));
7752
7753 // NewSRsrc = {Zero64, SRsrcFormat}
7754 BuildMI(MBB, MI, MI.getDebugLoc(), TII.get(AMDGPU::REG_SEQUENCE), NewSRsrc)
7755 .addReg(Zero64)
7756 .addImm(AMDGPU::sub0_sub1)
7757 .addReg(SRsrcFormatLo)
7758 .addImm(AMDGPU::sub2)
7759 .addReg(SRsrcFormatHi)
7760 .addImm(AMDGPU::sub3);
7761
7762 return std::tuple(RsrcPtr, NewSRsrc);
7763}
7764
7767 MachineDominatorTree *MDT) const {
7768 MachineFunction &MF = *MI.getMF();
7769 MachineRegisterInfo &MRI = MF.getRegInfo();
7770 MachineBasicBlock *CreatedBB = nullptr;
7771
7772 // Legalize True16
7773 if (ST.useRealTrue16Insts())
7775
7776 // Legalize VOP2
7777 if (isVOP2(MI) || isVOPC(MI)) {
7779 return CreatedBB;
7780 }
7781
7782 // Legalize VOP3
7783 if (isVOP3(MI)) {
7785 return CreatedBB;
7786 }
7787
7788 // Legalize SMRD
7789 if (isSMRD(MI)) {
7791 return CreatedBB;
7792 }
7793
7794 // Legalize FLAT
7795 if (isFLAT(MI)) {
7797 return CreatedBB;
7798 }
7799
7800 // Legalize PHI
7801 // The register class of the operands must be the same type as the register
7802 // class of the output.
7803 if (MI.getOpcode() == AMDGPU::PHI) {
7804 const TargetRegisterClass *VRC = getOpRegClass(MI, 0);
7805 assert(!RI.isSGPRClass(VRC));
7806
7807 // Update all the operands so they have the same type.
7808 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
7809 MachineOperand &Op = MI.getOperand(I);
7810 if (!Op.isReg() || !Op.getReg().isVirtual())
7811 continue;
7812
7813 // MI is a PHI instruction.
7814 MachineBasicBlock *InsertBB = MI.getOperand(I + 1).getMBB();
7816
7817 // Avoid creating no-op copies with the same src and dst reg class. These
7818 // confuse some of the machine passes.
7819 legalizeGenericOperand(*InsertBB, Insert, VRC, Op, MRI, MI.getDebugLoc());
7820 }
7821 }
7822
7823 // REG_SEQUENCE doesn't really require operand legalization, but if one has a
7824 // VGPR dest type and SGPR sources, insert copies so all operands are
7825 // VGPRs. This seems to help operand folding / the register coalescer.
7826 if (MI.getOpcode() == AMDGPU::REG_SEQUENCE) {
7827 MachineBasicBlock *MBB = MI.getParent();
7828 const TargetRegisterClass *DstRC = getOpRegClass(MI, 0);
7829 if (RI.hasVGPRs(DstRC)) {
7830 // Update all the operands so they are VGPR register classes. These may
7831 // not be the same register class because REG_SEQUENCE supports mixing
7832 // subregister index types e.g. sub0_sub1 + sub2 + sub3
7833 for (unsigned I = 1, E = MI.getNumOperands(); I != E; I += 2) {
7834 MachineOperand &Op = MI.getOperand(I);
7835 if (!Op.isReg() || !Op.getReg().isVirtual())
7836 continue;
7837
7838 const TargetRegisterClass *OpRC = MRI.getRegClass(Op.getReg());
7839 const TargetRegisterClass *VRC = RI.getEquivalentVGPRClass(OpRC);
7840 if (VRC == OpRC)
7841 continue;
7842
7843 legalizeGenericOperand(*MBB, MI, VRC, Op, MRI, MI.getDebugLoc());
7844 Op.setIsKill();
7845 }
7846 }
7847
7848 return CreatedBB;
7849 }
7850
7851 // Legalize INSERT_SUBREG
7852 // src0 must have the same register class as dst
7853 if (MI.getOpcode() == AMDGPU::INSERT_SUBREG) {
7854 Register Dst = MI.getOperand(0).getReg();
7855 Register Src0 = MI.getOperand(1).getReg();
7856 const TargetRegisterClass *DstRC = MRI.getRegClass(Dst);
7857 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0);
7858 if (DstRC != Src0RC) {
7859 MachineBasicBlock *MBB = MI.getParent();
7860 MachineOperand &Op = MI.getOperand(1);
7861 legalizeGenericOperand(*MBB, MI, DstRC, Op, MRI, MI.getDebugLoc());
7862 }
7863 return CreatedBB;
7864 }
7865
7866 // Legalize SI_INIT_M0
7867 if (MI.getOpcode() == AMDGPU::SI_INIT_M0) {
7868 MachineOperand &Src = MI.getOperand(0);
7869 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7870 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7871 return CreatedBB;
7872 }
7873
7874 // Legalize S_BITREPLICATE, S_QUADMASK and S_WQM
7875 if (MI.getOpcode() == AMDGPU::S_BITREPLICATE_B64_B32 ||
7876 MI.getOpcode() == AMDGPU::S_QUADMASK_B32 ||
7877 MI.getOpcode() == AMDGPU::S_QUADMASK_B64 ||
7878 MI.getOpcode() == AMDGPU::S_WQM_B32 ||
7879 MI.getOpcode() == AMDGPU::S_WQM_B64 ||
7880 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U32 ||
7881 MI.getOpcode() == AMDGPU::S_INVERSE_BALLOT_U64) {
7882 MachineOperand &Src = MI.getOperand(1);
7883 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7884 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7885 return CreatedBB;
7886 }
7887
7888 // Legalize MIMG/VIMAGE/VSAMPLE and MUBUF/MTBUF for shaders.
7889 //
7890 // Shaders only generate MUBUF/MTBUF instructions via intrinsics or via
7891 // scratch memory access. In both cases, the legalization never involves
7892 // conversion to the addr64 form.
7894 (isMUBUF(MI) || isMTBUF(MI)))) {
7895 AMDGPU::OpName RSrcOpName = (isVIMAGE(MI) || isVSAMPLE(MI))
7896 ? AMDGPU::OpName::rsrc
7897 : AMDGPU::OpName::srsrc;
7898 MachineOperand *SRsrc = getNamedOperand(MI, RSrcOpName);
7899 if (SRsrc && !RI.isSGPRClass(MRI.getRegClass(SRsrc->getReg())))
7900 CreatedBB = generateWaterFallLoop(*this, MI, {SRsrc}, MDT);
7901
7902 AMDGPU::OpName SampOpName =
7903 isMIMG(MI) ? AMDGPU::OpName::ssamp : AMDGPU::OpName::samp;
7904 MachineOperand *SSamp = getNamedOperand(MI, SampOpName);
7905 if (SSamp && !RI.isSGPRClass(MRI.getRegClass(SSamp->getReg())))
7906 CreatedBB = generateWaterFallLoop(*this, MI, {SSamp}, MDT);
7907
7908 return CreatedBB;
7909 }
7910
7911 // Legalize SI_CALL
7912 if (MI.getOpcode() == AMDGPU::SI_CALL_ISEL) {
7913 MachineOperand *Dest = &MI.getOperand(0);
7914 if (!RI.isSGPRClass(MRI.getRegClass(Dest->getReg()))) {
7915 createWaterFallForSiCall(&MI, MDT, {Dest});
7916 }
7917 }
7918
7919 // Legalize s_sleep_var.
7920 if (MI.getOpcode() == AMDGPU::S_SLEEP_VAR) {
7921 const DebugLoc &DL = MI.getDebugLoc();
7922 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7923 int Src0Idx =
7924 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
7925 MachineOperand &Src0 = MI.getOperand(Src0Idx);
7926 BuildMI(*MI.getParent(), MI, DL, get(AMDGPU::V_READFIRSTLANE_B32), Reg)
7927 .add(Src0);
7928 Src0.ChangeToRegister(Reg, false);
7929 return nullptr;
7930 }
7931
7932 // Legalize TENSOR_LOAD_TO_LDS_d2/_d4, TENSOR_STORE_FROM_LDS_d2/_d4. All their
7933 // operands are scalar.
7934 if (MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d2 ||
7935 MI.getOpcode() == AMDGPU::TENSOR_LOAD_TO_LDS_d4 ||
7936 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d2 ||
7937 MI.getOpcode() == AMDGPU::TENSOR_STORE_FROM_LDS_d4) {
7938 for (MachineOperand &Src : MI.explicit_operands()) {
7939 if (Src.isReg() && RI.hasVectorRegisters(MRI.getRegClass(Src.getReg())))
7940 Src.setReg(readlaneVGPRToSGPR(Src.getReg(), MI, MRI));
7941 }
7942 return CreatedBB;
7943 }
7944
7945 // Legalize MUBUF instructions.
7946 bool isSoffsetLegal = true;
7947 int SoffsetIdx =
7948 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::soffset);
7949 if (SoffsetIdx != -1) {
7950 MachineOperand *Soffset = &MI.getOperand(SoffsetIdx);
7951 if (Soffset->isReg() && Soffset->getReg().isVirtual() &&
7952 !RI.isSGPRClass(MRI.getRegClass(Soffset->getReg()))) {
7953 isSoffsetLegal = false;
7954 }
7955 }
7956
7957 bool isRsrcLegal = true;
7958 int RsrcIdx =
7959 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::srsrc);
7960 if (RsrcIdx != -1) {
7961 MachineOperand *Rsrc = &MI.getOperand(RsrcIdx);
7962 if (Rsrc->isReg() && !RI.isSGPRReg(MRI, Rsrc->getReg()))
7963 isRsrcLegal = false;
7964 }
7965
7966 // The operands are legal.
7967 if (isRsrcLegal && isSoffsetLegal)
7968 return CreatedBB;
7969
7970 if (!isRsrcLegal) {
7971 // Legalize a VGPR Rsrc
7972 //
7973 // If the instruction is _ADDR64, we can avoid a waterfall by extracting
7974 // the base pointer from the VGPR Rsrc, adding it to the VAddr, then using
7975 // a zero-value SRsrc.
7976 //
7977 // If the instruction is _OFFSET (both idxen and offen disabled), and we
7978 // support ADDR64 instructions, we can convert to ADDR64 and do the same as
7979 // above.
7980 //
7981 // Otherwise we are on non-ADDR64 hardware, and/or we have
7982 // idxen/offen/bothen and we fall back to a waterfall loop.
7983
7984 MachineOperand *Rsrc = &MI.getOperand(RsrcIdx);
7985 MachineBasicBlock &MBB = *MI.getParent();
7986
7987 MachineOperand *VAddr = getNamedOperand(MI, AMDGPU::OpName::vaddr);
7988 if (VAddr && AMDGPU::getIfAddr64Inst(MI.getOpcode()) != -1) {
7989 // This is already an ADDR64 instruction so we need to add the pointer
7990 // extracted from the resource descriptor to the current value of VAddr.
7991 Register NewVAddrLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
7992 Register NewVAddrHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
7993 Register NewVAddr = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
7994
7995 const auto *BoolXExecRC = RI.getWaveMaskRegClass();
7996 Register CondReg0 = MRI.createVirtualRegister(BoolXExecRC);
7997 Register CondReg1 = MRI.createVirtualRegister(BoolXExecRC);
7998
7999 unsigned RsrcPtr, NewSRsrc;
8000 std::tie(RsrcPtr, NewSRsrc) = extractRsrcPtr(*this, MI, *Rsrc);
8001
8002 // NewVaddrLo = RsrcPtr:sub0 + VAddr:sub0
8003 const DebugLoc &DL = MI.getDebugLoc();
8004 BuildMI(MBB, MI, DL, get(AMDGPU::V_ADD_CO_U32_e64), NewVAddrLo)
8005 .addDef(CondReg0)
8006 .addReg(RsrcPtr, {}, AMDGPU::sub0)
8007 .addReg(VAddr->getReg(), {}, AMDGPU::sub0)
8008 .addImm(0);
8009
8010 // NewVaddrHi = RsrcPtr:sub1 + VAddr:sub1
8011 BuildMI(MBB, MI, DL, get(AMDGPU::V_ADDC_U32_e64), NewVAddrHi)
8012 .addDef(CondReg1, RegState::Dead)
8013 .addReg(RsrcPtr, {}, AMDGPU::sub1)
8014 .addReg(VAddr->getReg(), {}, AMDGPU::sub1)
8015 .addReg(CondReg0, RegState::Kill)
8016 .addImm(0);
8017
8018 // NewVaddr = {NewVaddrHi, NewVaddrLo}
8019 BuildMI(MBB, MI, MI.getDebugLoc(), get(AMDGPU::REG_SEQUENCE), NewVAddr)
8020 .addReg(NewVAddrLo)
8021 .addImm(AMDGPU::sub0)
8022 .addReg(NewVAddrHi)
8023 .addImm(AMDGPU::sub1);
8024
8025 VAddr->setReg(NewVAddr);
8026 Rsrc->setReg(NewSRsrc);
8027 } else if (!VAddr && ST.hasAddr64()) {
8028 // This instructions is the _OFFSET variant, so we need to convert it to
8029 // ADDR64.
8030 assert(ST.getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS &&
8031 "FIXME: Need to emit flat atomics here");
8032
8033 unsigned RsrcPtr, NewSRsrc;
8034 std::tie(RsrcPtr, NewSRsrc) = extractRsrcPtr(*this, MI, *Rsrc);
8035
8036 Register NewVAddr = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
8037 MachineOperand *VData = getNamedOperand(MI, AMDGPU::OpName::vdata);
8038 MachineOperand *Offset = getNamedOperand(MI, AMDGPU::OpName::offset);
8039 MachineOperand *SOffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8040 unsigned Addr64Opcode = AMDGPU::getAddr64Inst(MI.getOpcode());
8041
8042 // Atomics with return have an additional tied operand and are
8043 // missing some of the special bits.
8044 MachineOperand *VDataIn = getNamedOperand(MI, AMDGPU::OpName::vdata_in);
8045 MachineInstr *Addr64;
8046
8047 if (!VDataIn) {
8048 // Regular buffer load / store.
8050 BuildMI(MBB, MI, MI.getDebugLoc(), get(Addr64Opcode))
8051 .add(*VData)
8052 .addReg(NewVAddr)
8053 .addReg(NewSRsrc)
8054 .add(*SOffset)
8055 .add(*Offset);
8056
8057 if (const MachineOperand *CPol =
8058 getNamedOperand(MI, AMDGPU::OpName::cpol)) {
8059 MIB.addImm(CPol->getImm());
8060 }
8061
8062 if (const MachineOperand *TFE =
8063 getNamedOperand(MI, AMDGPU::OpName::tfe)) {
8064 MIB.addImm(TFE->getImm());
8065 }
8066
8067 MIB.addImm(getNamedImmOperand(MI, AMDGPU::OpName::swz));
8068
8069 MIB.cloneMemRefs(MI);
8070 Addr64 = MIB;
8071 } else {
8072 // Atomics with return.
8073 Addr64 = BuildMI(MBB, MI, MI.getDebugLoc(), get(Addr64Opcode))
8074 .add(*VData)
8075 .add(*VDataIn)
8076 .addReg(NewVAddr)
8077 .addReg(NewSRsrc)
8078 .add(*SOffset)
8079 .add(*Offset)
8080 .addImm(getNamedImmOperand(MI, AMDGPU::OpName::cpol))
8081 .cloneMemRefs(MI);
8082 }
8083
8084 MI.removeFromParent();
8085
8086 // NewVaddr = {NewVaddrHi, NewVaddrLo}
8087 BuildMI(MBB, Addr64, Addr64->getDebugLoc(), get(AMDGPU::REG_SEQUENCE),
8088 NewVAddr)
8089 .addReg(RsrcPtr, {}, AMDGPU::sub0)
8090 .addImm(AMDGPU::sub0)
8091 .addReg(RsrcPtr, {}, AMDGPU::sub1)
8092 .addImm(AMDGPU::sub1);
8093 } else {
8094 // Legalize a VGPR Rsrc and soffset together.
8095 if (!isSoffsetLegal) {
8096 MachineOperand *Soffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8097 CreatedBB = generateWaterFallLoop(*this, MI, {Rsrc, Soffset}, MDT);
8098 return CreatedBB;
8099 }
8100 CreatedBB = generateWaterFallLoop(*this, MI, {Rsrc}, MDT);
8101 return CreatedBB;
8102 }
8103 }
8104
8105 // Legalize a VGPR soffset.
8106 if (!isSoffsetLegal) {
8107 MachineOperand *Soffset = getNamedOperand(MI, AMDGPU::OpName::soffset);
8108 CreatedBB = generateWaterFallLoop(*this, MI, {Soffset}, MDT);
8109 return CreatedBB;
8110 }
8111 return CreatedBB;
8112}
8113
8115 if (InSet.insert(MI).second)
8116 InstrList.push_back(MI);
8117 // Add MBUF instructiosn to deferred list.
8118 int RsrcIdx =
8119 AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::srsrc);
8120 if (RsrcIdx != -1) {
8121 DeferredList.insert(MI);
8122 }
8123}
8124
8126 return DeferredList.contains(MI);
8127}
8128
8129// Legalize size mismatches between 16bit and 32bit registers in v2s copy
8130// lowering (change sgpr to vgpr).
8131// This is mainly caused by 16bit SALU and 16bit VALU using reg with different
8132// size. Need to legalize the size of the operands during the vgpr lowering
8133// chain. This can be removed after we have sgpr16 in place
8135 MachineRegisterInfo &MRI) const {
8136 if (!ST.useRealTrue16Insts())
8137 return;
8138
8139 unsigned Opcode = MI.getOpcode();
8140 MachineBasicBlock *MBB = MI.getParent();
8141 // Legalize operands and check for size mismatch
8142 if (OpIdx >= MI.getNumExplicitOperands() ||
8143 OpIdx >= get(Opcode).getNumOperands() ||
8144 get(Opcode).operands()[OpIdx].RegClass == -1)
8145 return;
8146
8147 MachineOperand &Op = MI.getOperand(OpIdx);
8148 if (!Op.isReg() || !Op.getReg().isVirtual() || Op.isDef())
8149 return;
8150
8151 const TargetRegisterClass *CurrRC = MRI.getRegClass(Op.getReg());
8152 if (!RI.isVGPRClass(CurrRC))
8153 return;
8154
8155 int16_t RCID = getOpRegClassID(get(Opcode).operands()[OpIdx]);
8156 const TargetRegisterClass *ExpectedRC = RI.getRegClass(RCID);
8157 if (RI.getMatchingSuperRegClass(CurrRC, ExpectedRC, AMDGPU::lo16)) {
8158 // Default to the lo16 only if the subregister is not specified.
8159 if (Op.getSubReg() == AMDGPU::NoSubRegister)
8160 Op.setSubReg(AMDGPU::lo16);
8161 return;
8162 }
8163
8164 const TargetRegisterClass *CurrSRC =
8165 RI.getSubRegisterClass(CurrRC, Op.getSubReg());
8166 if (RI.getMatchingSuperRegClass(ExpectedRC, CurrSRC, AMDGPU::lo16)) {
8167 const DebugLoc &DL = MI.getDebugLoc();
8168 Register NewDstReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8169 Register Undef = MRI.createVirtualRegister(&AMDGPU::VGPR_16RegClass);
8170 BuildMI(*MBB, MI, DL, get(AMDGPU::IMPLICIT_DEF), Undef);
8171 BuildMI(*MBB, MI, DL, get(AMDGPU::REG_SEQUENCE), NewDstReg)
8172 .addReg(Op.getReg(), {}, Op.getSubReg())
8173 .addImm(AMDGPU::lo16)
8174 .addReg(Undef)
8175 .addImm(AMDGPU::hi16);
8176 Op.setReg(NewDstReg);
8177 Op.setSubReg(AMDGPU::NoSubRegister);
8178 }
8179}
8181 MachineRegisterInfo &MRI) const {
8182 for (unsigned OpIdx = 0; OpIdx < MI.getNumExplicitOperands(); OpIdx++)
8183 legalizeOperandsVALUt16(MI, OpIdx, MRI);
8184}
8185
8189 ArrayRef<Register> PhySGPRs) const {
8190 assert(MI->getOpcode() == AMDGPU::SI_CALL_ISEL &&
8191 "This only handle waterfall for SI_CALL_ISEL");
8192 // Move everything between ADJCALLSTACKUP and ADJCALLSTACKDOWN and
8193 // following copies, we also need to move copies from and to physical
8194 // registers into the loop block.
8195 // Also move the copies to physical registers into the loop block
8196 MachineBasicBlock &MBB = *MI->getParent();
8198 while (Start->getOpcode() != AMDGPU::ADJCALLSTACKUP)
8199 --Start;
8201 while (End->getOpcode() != AMDGPU::ADJCALLSTACKDOWN)
8202 ++End;
8203
8204 // Also include following copies of the return value
8205 ++End;
8206 while (End != MBB.end() && End->isCopy() &&
8207 MI->definesRegister(End->getOperand(1).getReg(), &RI))
8208 ++End;
8209
8210 generateWaterFallLoop(*this, *MI, ScalarOps, MDT, Start, End, PhySGPRs);
8211}
8212
8214 MachineDominatorTree *MDT) const {
8216 DenseMap<MachineInstr *, bool> V2SPhyCopiesToErase;
8217 while (!Worklist.empty()) {
8218 MachineInstr &Inst = *Worklist.top();
8219 Worklist.erase_top();
8220 // Skip MachineInstr in the deferred list.
8221 if (Worklist.isDeferred(&Inst))
8222 continue;
8223 moveToVALUImpl(Worklist, MDT, Inst, WaterFalls, V2SPhyCopiesToErase);
8224 }
8225
8226 // Deferred list of instructions will be processed once
8227 // all the MachineInstr in the worklist are done.
8228 for (MachineInstr *Inst : Worklist.getDeferredList()) {
8229 moveToVALUImpl(Worklist, MDT, *Inst, WaterFalls, V2SPhyCopiesToErase);
8230 assert(Worklist.empty() &&
8231 "Deferred MachineInstr are not supposed to re-populate worklist");
8232 }
8233
8234 for (auto &Entry : WaterFalls) {
8235 if (Entry.first->getOpcode() == AMDGPU::SI_CALL_ISEL)
8236 createWaterFallForSiCall(Entry.first, MDT, Entry.second.MOs,
8237 Entry.second.SGPRs);
8238 }
8239
8240 for (std::pair<MachineInstr *, bool> Entry : V2SPhyCopiesToErase)
8241 if (Entry.second)
8242 Entry.first->eraseFromParent();
8243}
8245 MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const {
8246 // If it's a copy of a VGPR to a physical SGPR, insert a V_READFIRSTLANE and
8247 // hope for the best.
8248 const TargetRegisterClass *DstRC = RI.getRegClassForReg(MRI, DstReg);
8249 ArrayRef<int16_t> SubRegIndices = RI.getRegSplitParts(DstRC, 4);
8250 if (SubRegIndices.size() <= 1) {
8251 Register NewDst = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
8252 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8253 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8254 .add(Inst.getOperand(1));
8255 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(), get(AMDGPU::COPY),
8256 DstReg)
8257 .addReg(NewDst);
8258 } else {
8260 for (int16_t Indice : SubRegIndices) {
8261 Register NewDst = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
8262 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8263 get(AMDGPU::V_READFIRSTLANE_B32), NewDst)
8264 .addReg(Inst.getOperand(1).getReg(), {}, Indice);
8265
8266 DstRegs.push_back(NewDst);
8267 }
8269 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8270 get(AMDGPU::REG_SEQUENCE), DstReg);
8271 for (unsigned i = 0; i < SubRegIndices.size(); ++i) {
8272 MIB.addReg(DstRegs[i]);
8273 MIB.addImm(RI.getSubRegFromChannel(i));
8274 }
8275 }
8276}
8277
8279 SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst,
8282 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
8283 if (DstReg == AMDGPU::M0) {
8284 createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
8285 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8286 return;
8287 }
8288 Register SrcReg = Inst.getOperand(1).getReg();
8291 // Only search current block since phyreg's def & use cannot cross
8292 // blocks when MF.NoPhi = false.
8293 while (++I != E) {
8294 // For SI_CALL_ISEL users, replace the phys SGPR with the VGPR source
8295 // and record the operand for later waterfall loop generation.
8296 if (I->getOpcode() == AMDGPU::SI_CALL_ISEL) {
8297 MachineInstr *UseMI = &*I;
8298 for (unsigned i = 0; i < UseMI->getNumOperands(); ++i) {
8299 if (UseMI->getOperand(i).isReg() &&
8300 UseMI->getOperand(i).getReg() == DstReg) {
8301 MachineOperand *MO = &UseMI->getOperand(i);
8302 MO->setReg(SrcReg);
8303 V2PhysSCopyInfo &V2SCopyInfo = WaterFalls[UseMI];
8304 V2SCopyInfo.MOs.push_back(MO);
8305 V2SCopyInfo.SGPRs.push_back(DstReg);
8306 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8307 }
8308 }
8309 } else if (I->getOpcode() == AMDGPU::SI_RETURN_TO_EPILOG &&
8310 I->getOperand(0).isReg() &&
8311 I->getOperand(0).getReg() == DstReg) {
8312 createReadFirstLaneFromCopyToPhysReg(MRI, DstReg, Inst);
8313 V2SPhyCopiesToErase.try_emplace(&Inst, true);
8314 } else if (I->readsRegister(DstReg, &RI)) {
8315 // COPY cannot be erased if other type of inst uses it.
8316 V2SPhyCopiesToErase[&Inst] = false;
8317 }
8318 if (I->findRegisterDefOperand(DstReg, &RI))
8319 break;
8320 }
8321}
8322
8324 SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst,
8326 DenseMap<MachineInstr *, bool> &V2SPhyCopiesToErase) const {
8327
8329 if (!MBB)
8330 return;
8331 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo();
8332 unsigned Opcode = Inst.getOpcode();
8333 unsigned NewOpcode = getVALUOp(Inst);
8334 const DebugLoc &DL = Inst.getDebugLoc();
8335
8336 // Handle some special cases
8337 switch (Opcode) {
8338 default:
8339 break;
8340 case AMDGPU::S_ADD_I32:
8341 case AMDGPU::S_SUB_I32: {
8342 // FIXME: The u32 versions currently selected use the carry.
8343 bool Changed;
8344 MachineBasicBlock *CreatedBBTmp = nullptr;
8345 std::tie(Changed, CreatedBBTmp) = moveScalarAddSub(Worklist, Inst, MDT);
8346 if (Changed)
8347 return;
8348
8349 // Default handling
8350 break;
8351 }
8352
8353 case AMDGPU::S_MUL_U64:
8354 if (ST.useVMulU64Inst()) {
8355 NewOpcode = AMDGPU::V_MUL_U64_e64;
8356 break;
8357 }
8358 // Split s_mul_u64 in 32-bit vector multiplications.
8359 splitScalarSMulU64(Worklist, Inst, MDT);
8360 Inst.eraseFromParent();
8361 return;
8362
8363 case AMDGPU::S_MUL_U64_U32_PSEUDO:
8364 case AMDGPU::S_MUL_I64_I32_PSEUDO:
8365 // This is a special case of s_mul_u64 where all the operands are either
8366 // zero extended or sign extended.
8367 splitScalarSMulPseudo(Worklist, Inst, MDT);
8368 Inst.eraseFromParent();
8369 return;
8370
8371 case AMDGPU::S_AND_B64:
8372 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_AND_B32, MDT);
8373 Inst.eraseFromParent();
8374 return;
8375
8376 case AMDGPU::S_OR_B64:
8377 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_OR_B32, MDT);
8378 Inst.eraseFromParent();
8379 return;
8380
8381 case AMDGPU::S_XOR_B64:
8382 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XOR_B32, MDT);
8383 Inst.eraseFromParent();
8384 return;
8385
8386 case AMDGPU::S_NAND_B64:
8387 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NAND_B32, MDT);
8388 Inst.eraseFromParent();
8389 return;
8390
8391 case AMDGPU::S_NOR_B64:
8392 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_NOR_B32, MDT);
8393 Inst.eraseFromParent();
8394 return;
8395
8396 case AMDGPU::S_XNOR_B64:
8397 if (ST.hasDLInsts())
8398 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_XNOR_B32, MDT);
8399 else
8400 splitScalar64BitXnor(Worklist, Inst, MDT);
8401 Inst.eraseFromParent();
8402 return;
8403
8404 case AMDGPU::S_ANDN2_B64:
8405 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ANDN2_B32, MDT);
8406 Inst.eraseFromParent();
8407 return;
8408
8409 case AMDGPU::S_ORN2_B64:
8410 splitScalar64BitBinaryOp(Worklist, Inst, AMDGPU::S_ORN2_B32, MDT);
8411 Inst.eraseFromParent();
8412 return;
8413
8414 case AMDGPU::S_BREV_B64:
8415 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_BREV_B32, true);
8416 Inst.eraseFromParent();
8417 return;
8418
8419 case AMDGPU::S_NOT_B64:
8420 splitScalar64BitUnaryOp(Worklist, Inst, AMDGPU::S_NOT_B32);
8421 Inst.eraseFromParent();
8422 return;
8423
8424 case AMDGPU::S_BCNT1_I32_B64:
8425 splitScalar64BitBCNT(Worklist, Inst);
8426 Inst.eraseFromParent();
8427 return;
8428
8429 case AMDGPU::S_BFE_I64:
8430 splitScalar64BitBFE(Worklist, Inst);
8431 Inst.eraseFromParent();
8432 return;
8433
8434 case AMDGPU::S_FLBIT_I32_B64:
8435 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBH_U32_e32);
8436 Inst.eraseFromParent();
8437 return;
8438 case AMDGPU::S_FF1_I32_B64:
8439 splitScalar64BitCountOp(Worklist, Inst, AMDGPU::V_FFBL_B32_e32);
8440 Inst.eraseFromParent();
8441 return;
8442
8443 case AMDGPU::S_LSHL_B32:
8444 if (ST.hasOnlyRevVALUShifts()) {
8445 NewOpcode = AMDGPU::V_LSHLREV_B32_e64;
8446 swapOperands(Inst);
8447 }
8448 break;
8449 case AMDGPU::S_ASHR_I32:
8450 if (ST.hasOnlyRevVALUShifts()) {
8451 NewOpcode = AMDGPU::V_ASHRREV_I32_e64;
8452 swapOperands(Inst);
8453 }
8454 break;
8455 case AMDGPU::S_LSHR_B32:
8456 if (ST.hasOnlyRevVALUShifts()) {
8457 NewOpcode = AMDGPU::V_LSHRREV_B32_e64;
8458 swapOperands(Inst);
8459 }
8460 break;
8461 case AMDGPU::S_LSHL_B64:
8462 if (ST.hasOnlyRevVALUShifts()) {
8463 NewOpcode = ST.getGeneration() >= AMDGPUSubtarget::GFX12
8464 ? AMDGPU::V_LSHLREV_B64_pseudo_e64
8465 : AMDGPU::V_LSHLREV_B64_e64;
8466 swapOperands(Inst);
8467 }
8468 break;
8469 case AMDGPU::S_ASHR_I64:
8470 if (ST.hasOnlyRevVALUShifts()) {
8471 NewOpcode = AMDGPU::V_ASHRREV_I64_e64;
8472 swapOperands(Inst);
8473 }
8474 break;
8475 case AMDGPU::S_LSHR_B64:
8476 if (ST.hasOnlyRevVALUShifts()) {
8477 NewOpcode = AMDGPU::V_LSHRREV_B64_e64;
8478 swapOperands(Inst);
8479 }
8480 break;
8481
8482 case AMDGPU::S_ABS_I32:
8483 lowerScalarAbs(Worklist, Inst);
8484 Inst.eraseFromParent();
8485 return;
8486
8487 case AMDGPU::S_ABSDIFF_I32:
8488 lowerScalarAbsDiff(Worklist, Inst);
8489 Inst.eraseFromParent();
8490 return;
8491
8492 case AMDGPU::S_CBRANCH_SCC0:
8493 case AMDGPU::S_CBRANCH_SCC1: {
8494 // Clear unused bits of vcc
8495 Register CondReg = Inst.getOperand(1).getReg();
8496 bool IsSCC = CondReg == AMDGPU::SCC;
8498 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(LMC.AndOpc), LMC.VccReg)
8499 .addReg(LMC.ExecReg)
8500 .addReg(IsSCC ? LMC.VccReg : CondReg)
8501 .setOperandDead(3); // implicit-def $scc
8502 Inst.removeOperand(1);
8503 } break;
8504
8505 case AMDGPU::S_BFE_U64:
8506 case AMDGPU::S_BFM_B64:
8507 llvm_unreachable("Moving this op to VALU not implemented");
8508
8509 case AMDGPU::S_PACK_LL_B32_B16:
8510 case AMDGPU::S_PACK_LH_B32_B16:
8511 case AMDGPU::S_PACK_HL_B32_B16:
8512 case AMDGPU::S_PACK_HH_B32_B16:
8513 movePackToVALU(Worklist, MRI, Inst);
8514 Inst.eraseFromParent();
8515 return;
8516
8517 case AMDGPU::S_XNOR_B32:
8518 lowerScalarXnor(Worklist, Inst);
8519 Inst.eraseFromParent();
8520 return;
8521
8522 case AMDGPU::S_NAND_B32:
8523 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_AND_B32);
8524 Inst.eraseFromParent();
8525 return;
8526
8527 case AMDGPU::S_NOR_B32:
8528 splitScalarNotBinop(Worklist, Inst, AMDGPU::S_OR_B32);
8529 Inst.eraseFromParent();
8530 return;
8531
8532 case AMDGPU::S_ANDN2_B32:
8533 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_AND_B32);
8534 Inst.eraseFromParent();
8535 return;
8536
8537 case AMDGPU::S_ORN2_B32:
8538 splitScalarBinOpN2(Worklist, Inst, AMDGPU::S_OR_B32);
8539 Inst.eraseFromParent();
8540 return;
8541
8542 // TODO: remove as soon as everything is ready
8543 // to replace VGPR to SGPR copy with V_READFIRSTLANEs.
8544 // S_ADD/SUB_CO_PSEUDO as well as S_UADDO/USUBO_PSEUDO
8545 // can only be selected from the uniform SDNode.
8546 case AMDGPU::S_ADD_CO_PSEUDO:
8547 case AMDGPU::S_SUB_CO_PSEUDO: {
8548 unsigned Opc = (Inst.getOpcode() == AMDGPU::S_ADD_CO_PSEUDO)
8549 ? AMDGPU::V_ADDC_U32_e64
8550 : AMDGPU::V_SUBB_U32_e64;
8551 const auto *CarryRC = RI.getWaveMaskRegClass();
8552
8553 Register CarryInReg = Inst.getOperand(4).getReg();
8554 if (!MRI.constrainRegClass(CarryInReg, CarryRC)) {
8555 Register NewCarryReg = MRI.createVirtualRegister(CarryRC);
8556 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(AMDGPU::COPY), NewCarryReg)
8557 .addReg(CarryInReg);
8558 }
8559
8560 Register CarryOutReg = Inst.getOperand(1).getReg();
8561
8562 Register DestReg = MRI.createVirtualRegister(RI.getEquivalentVGPRClass(
8563 MRI.getRegClass(Inst.getOperand(0).getReg())));
8564 MachineInstr *CarryOp =
8565 BuildMI(*MBB, &Inst, Inst.getDebugLoc(), get(Opc), DestReg)
8566 .addReg(CarryOutReg, RegState::Define)
8567 .add(Inst.getOperand(2))
8568 .add(Inst.getOperand(3))
8569 .addReg(CarryInReg)
8570 .addImm(0);
8571 legalizeOperands(*CarryOp);
8572 MRI.replaceRegWith(Inst.getOperand(0).getReg(), DestReg);
8573 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8574 Inst.eraseFromParent();
8575 }
8576 return;
8577 case AMDGPU::S_UADDO_PSEUDO:
8578 case AMDGPU::S_USUBO_PSEUDO: {
8579 MachineOperand &Dest0 = Inst.getOperand(0);
8580 MachineOperand &Dest1 = Inst.getOperand(1);
8581 MachineOperand &Src0 = Inst.getOperand(2);
8582 MachineOperand &Src1 = Inst.getOperand(3);
8583
8584 unsigned Opc = (Inst.getOpcode() == AMDGPU::S_UADDO_PSEUDO)
8585 ? AMDGPU::V_ADD_CO_U32_e64
8586 : AMDGPU::V_SUB_CO_U32_e64;
8587 const TargetRegisterClass *NewRC =
8588 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest0.getReg()));
8589 Register DestReg = MRI.createVirtualRegister(NewRC);
8590 MachineInstr *NewInstr = BuildMI(*MBB, &Inst, DL, get(Opc), DestReg)
8591 .addReg(Dest1.getReg(), RegState::Define)
8592 .add(Src0)
8593 .add(Src1)
8594 .addImm(0); // clamp bit
8595
8596 legalizeOperands(*NewInstr, MDT);
8597 MRI.replaceRegWith(Dest0.getReg(), DestReg);
8598 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8599 Inst.eraseFromParent();
8600 }
8601 return;
8602 case AMDGPU::S_LSHL1_ADD_U32:
8603 case AMDGPU::S_LSHL2_ADD_U32:
8604 case AMDGPU::S_LSHL3_ADD_U32:
8605 case AMDGPU::S_LSHL4_ADD_U32: {
8606 MachineOperand &Dest = Inst.getOperand(0);
8607 MachineOperand &Src0 = Inst.getOperand(1);
8608 MachineOperand &Src1 = Inst.getOperand(2);
8609 unsigned ShiftAmt = (Opcode == AMDGPU::S_LSHL1_ADD_U32 ? 1
8610 : Opcode == AMDGPU::S_LSHL2_ADD_U32 ? 2
8611 : Opcode == AMDGPU::S_LSHL3_ADD_U32 ? 3
8612 : 4);
8613
8614 const TargetRegisterClass *NewRC =
8615 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest.getReg()));
8616 Register DestReg = MRI.createVirtualRegister(NewRC);
8617 MachineInstr *NewInstr =
8618 BuildMI(*MBB, &Inst, DL, get(AMDGPU::V_LSHL_ADD_U32_e64), DestReg)
8619 .add(Src0)
8620 .addImm(ShiftAmt)
8621 .add(Src1);
8622
8623 legalizeOperands(*NewInstr, MDT);
8624 MRI.replaceRegWith(Dest.getReg(), DestReg);
8625 addUsersToMoveToVALUWorklist(DestReg, MRI, Worklist);
8626 Inst.eraseFromParent();
8627 }
8628 return;
8629 case AMDGPU::S_CSELECT_B32:
8630 case AMDGPU::S_CSELECT_B64:
8631 lowerSelect(Worklist, Inst, MDT);
8632 Inst.eraseFromParent();
8633 return;
8634 case AMDGPU::S_CMP_EQ_I32:
8635 case AMDGPU::S_CMP_LG_I32:
8636 case AMDGPU::S_CMP_GT_I32:
8637 case AMDGPU::S_CMP_GE_I32:
8638 case AMDGPU::S_CMP_LT_I32:
8639 case AMDGPU::S_CMP_LE_I32:
8640 case AMDGPU::S_CMP_EQ_U32:
8641 case AMDGPU::S_CMP_LG_U32:
8642 case AMDGPU::S_CMP_GT_U32:
8643 case AMDGPU::S_CMP_GE_U32:
8644 case AMDGPU::S_CMP_LT_U32:
8645 case AMDGPU::S_CMP_LE_U32:
8646 case AMDGPU::S_CMP_EQ_U64:
8647 case AMDGPU::S_CMP_LG_U64:
8648 case AMDGPU::S_CMP_LT_F32:
8649 case AMDGPU::S_CMP_EQ_F32:
8650 case AMDGPU::S_CMP_LE_F32:
8651 case AMDGPU::S_CMP_GT_F32:
8652 case AMDGPU::S_CMP_LG_F32:
8653 case AMDGPU::S_CMP_GE_F32:
8654 case AMDGPU::S_CMP_O_F32:
8655 case AMDGPU::S_CMP_U_F32:
8656 case AMDGPU::S_CMP_NGE_F32:
8657 case AMDGPU::S_CMP_NLG_F32:
8658 case AMDGPU::S_CMP_NGT_F32:
8659 case AMDGPU::S_CMP_NLE_F32:
8660 case AMDGPU::S_CMP_NEQ_F32:
8661 case AMDGPU::S_CMP_NLT_F32: {
8662 Register CondReg = MRI.createVirtualRegister(RI.getWaveMaskRegClass());
8663 auto NewInstr =
8664 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode), CondReg)
8665 .setMIFlags(Inst.getFlags());
8666 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src0_modifiers) >=
8667 0) {
8668 NewInstr
8669 .addImm(0) // src0_modifiers
8670 .add(Inst.getOperand(0)) // src0
8671 .addImm(0) // src1_modifiers
8672 .add(Inst.getOperand(1)) // src1
8673 .addImm(0); // clamp
8674 } else {
8675 NewInstr.add(Inst.getOperand(0)).add(Inst.getOperand(1));
8676 }
8677 legalizeOperands(*NewInstr, MDT);
8678 int SCCIdx = Inst.findRegisterDefOperandIdx(AMDGPU::SCC, /*TRI=*/nullptr);
8679 const MachineOperand &SCCOp = Inst.getOperand(SCCIdx);
8680 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8681 Inst.eraseFromParent();
8682 return;
8683 }
8684 case AMDGPU::S_CMP_LT_F16:
8685 case AMDGPU::S_CMP_EQ_F16:
8686 case AMDGPU::S_CMP_LE_F16:
8687 case AMDGPU::S_CMP_GT_F16:
8688 case AMDGPU::S_CMP_LG_F16:
8689 case AMDGPU::S_CMP_GE_F16:
8690 case AMDGPU::S_CMP_O_F16:
8691 case AMDGPU::S_CMP_U_F16:
8692 case AMDGPU::S_CMP_NGE_F16:
8693 case AMDGPU::S_CMP_NLG_F16:
8694 case AMDGPU::S_CMP_NGT_F16:
8695 case AMDGPU::S_CMP_NLE_F16:
8696 case AMDGPU::S_CMP_NEQ_F16:
8697 case AMDGPU::S_CMP_NLT_F16: {
8698 Register CondReg = MRI.createVirtualRegister(RI.getWaveMaskRegClass());
8699 auto NewInstr =
8700 BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode), CondReg)
8701 .setMIFlags(Inst.getFlags());
8702 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::src0_modifiers)) {
8703 NewInstr
8704 .addImm(0) // src0_modifiers
8705 .add(Inst.getOperand(0)) // src0
8706 .addImm(0) // src1_modifiers
8707 .add(Inst.getOperand(1)) // src1
8708 .addImm(0); // clamp
8709 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::op_sel))
8710 NewInstr.addImm(0); // op_sel0
8711 } else {
8712 NewInstr
8713 .add(Inst.getOperand(0))
8714 .add(Inst.getOperand(1));
8715 }
8716 legalizeOperands(*NewInstr, MDT);
8717 int SCCIdx = Inst.findRegisterDefOperandIdx(AMDGPU::SCC, /*TRI=*/nullptr);
8718 const MachineOperand &SCCOp = Inst.getOperand(SCCIdx);
8719 addSCCDefUsersToVALUWorklist(SCCOp, Inst, Worklist, CondReg);
8720 Inst.eraseFromParent();
8721 return;
8722 }
8723 case AMDGPU::S_CVT_HI_F32_F16: {
8724 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8725 Register NewDst = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8726 if (ST.useRealTrue16Insts()) {
8727 BuildMI(*MBB, Inst, DL, get(AMDGPU::COPY), TmpReg)
8728 .add(Inst.getOperand(1));
8729 BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8730 .addImm(0) // src0_modifiers
8731 .addReg(TmpReg, {}, AMDGPU::hi16)
8732 .addImm(0) // clamp
8733 .addImm(0) // omod
8734 .addImm(0); // op_sel0
8735 } else {
8736 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
8737 .addImm(16)
8738 .add(Inst.getOperand(1));
8739 BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8740 .addImm(0) // src0_modifiers
8741 .addReg(TmpReg)
8742 .addImm(0) // clamp
8743 .addImm(0); // omod
8744 }
8745
8746 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8747 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8748 Inst.eraseFromParent();
8749 return;
8750 }
8751 case AMDGPU::S_MINIMUM_F32:
8752 case AMDGPU::S_MAXIMUM_F32: {
8753 Register NewDst = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
8754 MachineInstr *NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8755 .addImm(0) // src0_modifiers
8756 .add(Inst.getOperand(1))
8757 .addImm(0) // src1_modifiers
8758 .add(Inst.getOperand(2))
8759 .addImm(0) // clamp
8760 .addImm(0); // omod
8761 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8762
8763 legalizeOperands(*NewInstr, MDT);
8764 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8765 Inst.eraseFromParent();
8766 return;
8767 }
8768 case AMDGPU::S_MINIMUM_F16:
8769 case AMDGPU::S_MAXIMUM_F16: {
8770 Register NewDst = MRI.createVirtualRegister(ST.useRealTrue16Insts()
8771 ? &AMDGPU::VGPR_16RegClass
8772 : &AMDGPU::VGPR_32RegClass);
8773 MachineInstr *NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8774 .addImm(0) // src0_modifiers
8775 .add(Inst.getOperand(1))
8776 .addImm(0) // src1_modifiers
8777 .add(Inst.getOperand(2))
8778 .addImm(0) // clamp
8779 .addImm(0) // omod
8780 .addImm(0); // opsel0
8781 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8782 legalizeOperands(*NewInstr, MDT);
8783 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8784 Inst.eraseFromParent();
8785 return;
8786 }
8787 case AMDGPU::V_S_EXP_F16_e64:
8788 case AMDGPU::V_S_LOG_F16_e64:
8789 case AMDGPU::V_S_RCP_F16_e64:
8790 case AMDGPU::V_S_RSQ_F16_e64:
8791 case AMDGPU::V_S_SQRT_F16_e64: {
8792 Register NewDst = MRI.createVirtualRegister(ST.useRealTrue16Insts()
8793 ? &AMDGPU::VGPR_16RegClass
8794 : &AMDGPU::VGPR_32RegClass);
8795 auto NewInstr = BuildMI(*MBB, Inst, DL, get(NewOpcode), NewDst)
8796 .add(Inst.getOperand(1)) // src0_modifiers
8797 .add(Inst.getOperand(2))
8798 .add(Inst.getOperand(3)) // clamp
8799 .add(Inst.getOperand(4)) // omod
8800 .setMIFlags(Inst.getFlags());
8801 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::op_sel))
8802 NewInstr.addImm(0); // opsel0
8803 MRI.replaceRegWith(Inst.getOperand(0).getReg(), NewDst);
8804 legalizeOperands(*NewInstr, MDT);
8805 addUsersToMoveToVALUWorklist(NewDst, MRI, Worklist);
8806 Inst.eraseFromParent();
8807 return;
8808 }
8809 }
8810
8811 if (NewOpcode == AMDGPU::INSTRUCTION_LIST_END) {
8812 // We cannot move this instruction to the VALU, so we should try to
8813 // legalize its operands instead.
8814 legalizeOperands(Inst, MDT);
8815 return;
8816 }
8817 // Handle converting generic instructions like COPY-to-SGPR into
8818 // COPY-to-VGPR.
8819 if (NewOpcode == Opcode) {
8820 Register DstReg = Inst.getOperand(0).getReg();
8821 const TargetRegisterClass *NewDstRC = getDestEquivalentVGPRClass(Inst);
8822
8823 if (Inst.isCopy() && DstReg.isPhysical() &&
8824 Inst.getOperand(1).getReg().isVirtual()) {
8825 handleCopyToPhysHelper(Worklist, DstReg, Inst, MRI, WaterFalls,
8826 V2SPhyCopiesToErase);
8827 return;
8828 }
8829
8830 if (Inst.isCopy() && Inst.getOperand(1).getReg().isVirtual()) {
8831 Register NewDstReg = Inst.getOperand(1).getReg();
8832 const TargetRegisterClass *SrcRC = RI.getRegClassForReg(MRI, NewDstReg);
8833 if (const TargetRegisterClass *CommonRC =
8834 RI.getCommonSubClass(NewDstRC, SrcRC)) {
8835 // Instead of creating a copy where src and dst are the same register
8836 // class, we just replace all uses of dst with src. These kinds of
8837 // copies interfere with the heuristics MachineSink uses to decide
8838 // whether or not to split a critical edge. Since the pass assumes
8839 // that copies will end up as machine instructions and not be
8840 // eliminated.
8841 addUsersToMoveToVALUWorklist(DstReg, MRI, Worklist);
8842 unsigned SrcSubReg = Inst.getOperand(1).getSubReg();
8843 bool IsUndef = Inst.getOperand(1).isUndef();
8844 for (MachineOperand &UseMO :
8845 make_early_inc_range(MRI.use_operands(DstReg))) {
8846 UseMO.setSubReg(
8847 RI.composeSubRegIndices(SrcSubReg, UseMO.getSubReg()));
8848 UseMO.setReg(NewDstReg);
8849 if (IsUndef)
8850 UseMO.setIsUndef();
8851 }
8852 MRI.clearKillFlags(NewDstReg);
8853
8854 if (!MRI.constrainRegClass(NewDstReg, CommonRC))
8855 llvm_unreachable("failed to constrain register");
8856
8857 Inst.eraseFromParent();
8858
8859 for (MachineOperand &UseMO :
8860 make_early_inc_range(MRI.use_operands(NewDstReg))) {
8861 MachineInstr &UseMI = *UseMO.getParent();
8862
8863 // Legalize t16 operands since replaceReg is called after
8864 // addUsersToVALU.
8866
8867 unsigned OpIdx = UseMI.getOperandNo(&UseMO);
8868 if (const TargetRegisterClass *OpRC =
8869 getRegClass(UseMI.getDesc(), OpIdx))
8870 MRI.constrainRegClass(NewDstReg, OpRC);
8871 }
8872
8873 return;
8874 }
8875 }
8876
8877 // If this is a v2s copy between 16bit and 32bit reg,
8878 // replace vgpr copy to reg_sequence/extract_subreg
8879 // This can be remove after we have sgpr16 in place
8880 if (ST.useRealTrue16Insts() && Inst.isCopy() &&
8881 Inst.getOperand(1).getReg().isVirtual() &&
8882 RI.isVGPR(MRI, Inst.getOperand(1).getReg())) {
8883 const TargetRegisterClass *SrcRegRC = getOpRegClass(Inst, 1);
8884 if (RI.getMatchingSuperRegClass(NewDstRC, SrcRegRC, AMDGPU::lo16)) {
8885 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8886 Register Undef = MRI.createVirtualRegister(&AMDGPU::VGPR_16RegClass);
8887 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8888 get(AMDGPU::IMPLICIT_DEF), Undef);
8889 BuildMI(*Inst.getParent(), &Inst, Inst.getDebugLoc(),
8890 get(AMDGPU::REG_SEQUENCE), NewDstReg)
8891 .addReg(Inst.getOperand(1).getReg())
8892 .addImm(AMDGPU::lo16)
8893 .addReg(Undef)
8894 .addImm(AMDGPU::hi16);
8895 Inst.eraseFromParent();
8896 MRI.replaceRegWith(DstReg, NewDstReg);
8897 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8898 return;
8899 } else if (RI.getMatchingSuperRegClass(SrcRegRC, NewDstRC,
8900 AMDGPU::lo16)) {
8901 Inst.getOperand(1).setSubReg(AMDGPU::lo16);
8902 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8903 MRI.replaceRegWith(DstReg, NewDstReg);
8904 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8905 return;
8906 }
8907 }
8908
8909 Register NewDstReg = MRI.createVirtualRegister(NewDstRC);
8910 MRI.replaceRegWith(DstReg, NewDstReg);
8911 legalizeOperands(Inst, MDT);
8912 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
8913 return;
8914 }
8915
8916 // Use the new VALU Opcode.
8917 auto NewInstr = BuildMI(*MBB, Inst, Inst.getDebugLoc(), get(NewOpcode))
8918 .setMIFlags(Inst.getFlags());
8919 if (isVOP3(NewOpcode) && !isVOP3(Opcode)) {
8920 // Intersperse VOP3 modifiers among the SALU operands.
8921 NewInstr->addOperand(Inst.getOperand(0));
8922 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8923 AMDGPU::OpName::src0_modifiers) >= 0)
8924 NewInstr.addImm(0);
8925 if (AMDGPU::hasNamedOperand(NewOpcode, AMDGPU::OpName::src0)) {
8926 const MachineOperand &Src = Inst.getOperand(1);
8927 NewInstr->addOperand(Src);
8928 }
8929
8930 if (Opcode == AMDGPU::S_SEXT_I32_I8 || Opcode == AMDGPU::S_SEXT_I32_I16) {
8931 // We are converting these to a BFE, so we need to add the missing
8932 // operands for the size and offset.
8933 unsigned Size = (Opcode == AMDGPU::S_SEXT_I32_I8) ? 8 : 16;
8934 NewInstr.addImm(0);
8935 NewInstr.addImm(Size);
8936 } else if (Opcode == AMDGPU::S_BCNT1_I32_B32) {
8937 // The VALU version adds the second operand to the result, so insert an
8938 // extra 0 operand.
8939 NewInstr.addImm(0);
8940 } else if (Opcode == AMDGPU::S_BFE_I32 || Opcode == AMDGPU::S_BFE_U32) {
8941 const MachineOperand &OffsetWidthOp = Inst.getOperand(2);
8942 // If we need to move this to VGPRs, we need to unpack the second
8943 // operand back into the 2 separate ones for bit offset and width.
8944 assert(OffsetWidthOp.isImm() &&
8945 "Scalar BFE is only implemented for constant width and offset");
8946 uint32_t Imm = OffsetWidthOp.getImm();
8947
8948 uint32_t Offset = Imm & 0x3f; // Extract bits [5:0].
8949 uint32_t BitWidth = (Imm & 0x7f0000) >> 16; // Extract bits [22:16].
8950 NewInstr.addImm(Offset);
8951 NewInstr.addImm(BitWidth);
8952 } else {
8953 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8954 AMDGPU::OpName::src1_modifiers) >= 0)
8955 NewInstr.addImm(0);
8956 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src1) >= 0)
8957 NewInstr->addOperand(Inst.getOperand(2));
8958 if (AMDGPU::getNamedOperandIdx(NewOpcode,
8959 AMDGPU::OpName::src2_modifiers) >= 0)
8960 NewInstr.addImm(0);
8961 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::src2) >= 0)
8962 NewInstr->addOperand(Inst.getOperand(3));
8963 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::clamp) >= 0)
8964 NewInstr.addImm(0);
8965 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::omod) >= 0)
8966 NewInstr.addImm(0);
8967 if (AMDGPU::getNamedOperandIdx(NewOpcode, AMDGPU::OpName::op_sel) >= 0)
8968 NewInstr.addImm(0);
8969 }
8970 } else {
8971 // Just copy the SALU operands.
8972 for (const MachineOperand &Op : Inst.explicit_operands())
8973 NewInstr->addOperand(Op);
8974 }
8975
8976 // Remove any references to SCC. Vector instructions can't read from it, and
8977 // We're just about to add the implicit use / defs of VCC, and we don't want
8978 // both.
8979 bool DeadSCCDef = false;
8980 for (MachineOperand &Op : Inst.implicit_operands()) {
8981 if (Op.getReg() == AMDGPU::SCC) {
8982 // Only propagate through live-def of SCC.
8983 if (Op.isDef()) {
8984 if (Op.isDead())
8985 DeadSCCDef = true;
8986 else
8987 addSCCDefUsersToVALUWorklist(Op, Inst, Worklist);
8988 continue;
8989 }
8990
8991 addSCCDefsToVALUWorklist(NewInstr, Worklist);
8992 }
8993 }
8994 Inst.eraseFromParent();
8995 Register NewDstReg;
8996 if (NewInstr->getOperand(0).isReg() && NewInstr->getOperand(0).isDef()) {
8997 Register DstReg = NewInstr->getOperand(0).getReg();
8998 assert(DstReg.isVirtual());
8999 // Update the destination register class.
9000 const TargetRegisterClass *NewDstRC = getDestEquivalentVGPRClass(*NewInstr);
9001 assert(NewDstRC);
9002 NewDstReg = MRI.createVirtualRegister(NewDstRC);
9003 MRI.replaceRegWith(DstReg, NewDstReg);
9004 }
9005 fixImplicitOperands(*NewInstr);
9006
9007 if (DeadSCCDef) {
9008 // A scalar op with a dead SCC def lowers to a VALU op whose VCC def will
9009 // also be dead.
9010 if (MachineOperand *VCCDef =
9011 NewInstr->findRegisterDefOperand(RI.getVCC(), &RI))
9012 VCCDef->setIsDead();
9013 }
9014
9015 // Legalize the operands
9016 legalizeOperands(*NewInstr, MDT);
9017 if (NewDstReg)
9018 addUsersToMoveToVALUWorklist(NewDstReg, MRI, Worklist);
9019}
9020
9021// Add/sub require special handling to deal with carry outs.
9022std::pair<bool, MachineBasicBlock *>
9023SIInstrInfo::moveScalarAddSub(SIInstrWorklist &Worklist, MachineInstr &Inst,
9024 MachineDominatorTree *MDT) const {
9025 if (ST.hasAddNoCarryInsts()) {
9026 // Assume there is no user of scc since we don't select this in that case.
9027 // Since scc isn't used, it doesn't really matter if the i32 or u32 variant
9028 // is used.
9029
9030 MachineBasicBlock &MBB = *Inst.getParent();
9031 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9032
9033 Register OldDstReg = Inst.getOperand(0).getReg();
9034 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9035
9036 unsigned Opc = Inst.getOpcode();
9037 assert(Opc == AMDGPU::S_ADD_I32 || Opc == AMDGPU::S_SUB_I32);
9038
9039 unsigned NewOpc = Opc == AMDGPU::S_ADD_I32 ?
9040 AMDGPU::V_ADD_U32_e64 : AMDGPU::V_SUB_U32_e64;
9041
9042 assert(Inst.getOperand(3).getReg() == AMDGPU::SCC);
9043 Inst.removeOperand(3);
9044
9045 Inst.setDesc(get(NewOpc));
9046 Inst.addOperand(MachineOperand::CreateImm(0)); // clamp bit
9047 Inst.addImplicitDefUseOperands(*MBB.getParent());
9048 MRI.replaceRegWith(OldDstReg, ResultReg);
9049 MachineBasicBlock *NewBB = legalizeOperands(Inst, MDT);
9050
9051 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9052 return std::pair(true, NewBB);
9053 }
9054
9055 return std::pair(false, nullptr);
9056}
9057
9058void SIInstrInfo::lowerSelect(SIInstrWorklist &Worklist, MachineInstr &Inst,
9059 MachineDominatorTree *MDT) const {
9060
9061 MachineBasicBlock &MBB = *Inst.getParent();
9062 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9063 MachineBasicBlock::iterator MII = Inst;
9064 const DebugLoc &DL = Inst.getDebugLoc();
9065
9066 MachineOperand &Dest = Inst.getOperand(0);
9067 MachineOperand &Src0 = Inst.getOperand(1);
9068 MachineOperand &Src1 = Inst.getOperand(2);
9069 MachineOperand &Cond = Inst.getOperand(3);
9070
9071 Register CondReg = Cond.getReg();
9072 bool IsSCC = (CondReg == AMDGPU::SCC);
9073
9074 // Remove S_CSELECT instructions that we previously inserted to feed the SCC
9075 // condition output from S_CMP into the SGPR condition input of V_CNDMASK. If
9076 // the S_CMP has been promoted to V_CMP then we can feed its SGPR condition
9077 // output directly into the V_CNDMASK.
9078 if (!IsSCC && Src0.isImm() && (Src0.getImm() == -1) && Src1.isImm() &&
9079 (Src1.getImm() == 0)) {
9080 for (MachineOperand &UseMO :
9082 MachineInstr &UseMI = *UseMO.getParent();
9083 switch (UseMI.getOpcode()) {
9084 case AMDGPU::V_CNDMASK_B16_fake16_e32:
9085 case AMDGPU::V_CNDMASK_B16_fake16_e64:
9086 case AMDGPU::V_CNDMASK_B16_t16_e32:
9087 case AMDGPU::V_CNDMASK_B16_t16_e64:
9088 case AMDGPU::V_CNDMASK_B32_e32:
9089 case AMDGPU::V_CNDMASK_B32_e64:
9090 case AMDGPU::V_CNDMASK_B64_PSEUDO:
9091 if (UseMO.isImplicit() ||
9092 &UseMO == getNamedOperand(UseMI, AMDGPU::OpName::src2))
9093 UseMO.setReg(CondReg);
9094 }
9095 }
9096 if (MRI.use_nodbg_empty(Dest.getReg()))
9097 return;
9098 }
9099
9100 Register NewCondReg = CondReg;
9101 if (IsSCC) {
9102 const TargetRegisterClass *TC = RI.getWaveMaskRegClass();
9103 NewCondReg = MRI.createVirtualRegister(TC);
9104
9105 // Now look for the closest SCC def if it is a copy
9106 // replacing the CondReg with the COPY source register
9107 bool CopyFound = false;
9108 for (MachineInstr &CandI :
9110 Inst.getParent()->rend())) {
9111 if (CandI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI, false, false) !=
9112 -1) {
9113 if (CandI.isCopy() && CandI.getOperand(0).getReg() == AMDGPU::SCC) {
9114 BuildMI(MBB, MII, DL, get(AMDGPU::COPY), NewCondReg)
9115 .addReg(CandI.getOperand(1).getReg());
9116 CopyFound = true;
9117 }
9118 break;
9119 }
9120 }
9121 if (!CopyFound) {
9122 // SCC def is not a copy
9123 // Insert a trivial select instead of creating a copy, because a copy from
9124 // SCC would semantically mean just copying a single bit, but we may need
9125 // the result to be a vector condition mask that needs preserving.
9126 unsigned Opcode =
9127 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
9128 auto NewSelect =
9129 BuildMI(MBB, MII, DL, get(Opcode), NewCondReg).addImm(-1).addImm(0);
9130 NewSelect->getOperand(3).setIsUndef(Cond.isUndef());
9131 }
9132 }
9133
9134 Register NewDestReg = MRI.createVirtualRegister(
9135 RI.getEquivalentVGPRClass(MRI.getRegClass(Dest.getReg())));
9136 MachineInstr *NewInst;
9137 if (Inst.getOpcode() == AMDGPU::S_CSELECT_B32) {
9138 NewInst = BuildMI(MBB, MII, DL, get(AMDGPU::V_CNDMASK_B32_e64), NewDestReg)
9139 .addImm(0)
9140 .add(Src1) // False
9141 .addImm(0)
9142 .add(Src0) // True
9143 .addReg(NewCondReg);
9144 } else {
9145 NewInst =
9146 BuildMI(MBB, MII, DL, get(AMDGPU::V_CNDMASK_B64_PSEUDO), NewDestReg)
9147 .add(Src1) // False
9148 .add(Src0) // True
9149 .addReg(NewCondReg);
9150 }
9151 MRI.replaceRegWith(Dest.getReg(), NewDestReg);
9152 legalizeOperands(*NewInst, MDT);
9153 addUsersToMoveToVALUWorklist(NewDestReg, MRI, Worklist);
9154}
9155
9156void SIInstrInfo::lowerScalarAbs(SIInstrWorklist &Worklist,
9157 MachineInstr &Inst) const {
9158 MachineBasicBlock &MBB = *Inst.getParent();
9159 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9160 MachineBasicBlock::iterator MII = Inst;
9161 const DebugLoc &DL = Inst.getDebugLoc();
9162
9163 MachineOperand &Dest = Inst.getOperand(0);
9164 MachineOperand &Src = Inst.getOperand(1);
9165 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9166 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9167
9168 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9169 unsigned SubOp =
9170 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9171
9172 MachineInstrBuilder Sub =
9173 BuildMI(MBB, MII, DL, get(SubOp), TmpReg).addImm(0).addReg(Src.getReg());
9174 if (HasCarryOut)
9175 Sub.setOperandDead(3); // Dead vcc
9176
9177 BuildMI(MBB, MII, DL, get(AMDGPU::V_MAX_I32_e64), ResultReg)
9178 .addReg(Src.getReg())
9179 .addReg(TmpReg);
9180
9181 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9182 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9183}
9184
9185void SIInstrInfo::lowerScalarAbsDiff(SIInstrWorklist &Worklist,
9186 MachineInstr &Inst) const {
9187 MachineBasicBlock &MBB = *Inst.getParent();
9188 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9189 MachineBasicBlock::iterator MII = Inst;
9190 const DebugLoc &DL = Inst.getDebugLoc();
9191
9192 MachineOperand &Dest = Inst.getOperand(0);
9193 MachineOperand &Src1 = Inst.getOperand(1);
9194 MachineOperand &Src2 = Inst.getOperand(2);
9195 Register SubResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9196 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9197 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9198
9199 bool HasCarryOut = !ST.hasAddNoCarryInsts();
9200 unsigned SubOp =
9201 HasCarryOut ? AMDGPU::V_SUB_CO_U32_e32 : AMDGPU::V_SUB_U32_e32;
9202
9203 MachineInstrBuilder Sub1 = BuildMI(MBB, MII, DL, get(SubOp), SubResultReg)
9204 .addReg(Src1.getReg())
9205 .addReg(Src2.getReg());
9206
9207 MachineInstrBuilder Sub2 =
9208 BuildMI(MBB, MII, DL, get(SubOp), TmpReg).addImm(0).addReg(SubResultReg);
9209
9210 if (HasCarryOut) {
9211 Sub1.setOperandDead(3); // Dead vcc
9212 Sub2.setOperandDead(3); // Dead vcc
9213 }
9214
9215 BuildMI(MBB, MII, DL, get(AMDGPU::V_MAX_I32_e64), ResultReg)
9216 .addReg(SubResultReg)
9217 .addReg(TmpReg);
9218
9219 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9220 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9221}
9222
9223void SIInstrInfo::lowerScalarXnor(SIInstrWorklist &Worklist,
9224 MachineInstr &Inst) const {
9225 MachineBasicBlock &MBB = *Inst.getParent();
9226 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9227 MachineBasicBlock::iterator MII = Inst;
9228 const DebugLoc &DL = Inst.getDebugLoc();
9229
9230 MachineOperand &Dest = Inst.getOperand(0);
9231 MachineOperand &Src0 = Inst.getOperand(1);
9232 MachineOperand &Src1 = Inst.getOperand(2);
9233
9234 if (ST.hasDLInsts()) {
9235 Register NewDest = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9236 legalizeGenericOperand(MBB, MII, &AMDGPU::VGPR_32RegClass, Src0, MRI, DL);
9237 legalizeGenericOperand(MBB, MII, &AMDGPU::VGPR_32RegClass, Src1, MRI, DL);
9238
9239 BuildMI(MBB, MII, DL, get(AMDGPU::V_XNOR_B32_e64), NewDest)
9240 .add(Src0)
9241 .add(Src1);
9242
9243 MRI.replaceRegWith(Dest.getReg(), NewDest);
9244 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9245 } else {
9246 // Using the identity !(x ^ y) == (!x ^ y) == (x ^ !y), we can
9247 // invert either source and then perform the XOR. If either source is a
9248 // scalar register, then we can leave the inversion on the scalar unit to
9249 // achieve a better distribution of scalar and vector instructions.
9250 bool Src0IsSGPR = Src0.isReg() &&
9251 RI.isSGPRClass(MRI.getRegClass(Src0.getReg()));
9252 bool Src1IsSGPR = Src1.isReg() &&
9253 RI.isSGPRClass(MRI.getRegClass(Src1.getReg()));
9254 MachineInstr *Xor;
9255 Register Temp = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9256 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9257
9258 // Build a pair of scalar instructions and add them to the work list.
9259 // The next iteration over the work list will lower these to the vector
9260 // unit as necessary.
9261 if (Src0IsSGPR) {
9262 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Temp).add(Src0);
9263 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), NewDest)
9264 .addReg(Temp)
9265 .add(Src1);
9266 } else if (Src1IsSGPR) {
9267 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Temp).add(Src1);
9268 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), NewDest)
9269 .add(Src0)
9270 .addReg(Temp);
9271 } else {
9272 Xor = BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B32), Temp)
9273 .add(Src0)
9274 .add(Src1);
9275 MachineInstr *Not =
9276 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), NewDest).addReg(Temp);
9277 Worklist.insert(Not);
9278 }
9279
9280 MRI.replaceRegWith(Dest.getReg(), NewDest);
9281
9282 Worklist.insert(Xor);
9283
9284 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9285 }
9286}
9287
9288void SIInstrInfo::splitScalarNotBinop(SIInstrWorklist &Worklist,
9289 MachineInstr &Inst,
9290 unsigned Opcode) const {
9291 MachineBasicBlock &MBB = *Inst.getParent();
9292 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9293 MachineBasicBlock::iterator MII = Inst;
9294 const DebugLoc &DL = Inst.getDebugLoc();
9295
9296 MachineOperand &Dest = Inst.getOperand(0);
9297 MachineOperand &Src0 = Inst.getOperand(1);
9298 MachineOperand &Src1 = Inst.getOperand(2);
9299
9300 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9301 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
9302
9303 MachineInstr &Op = *BuildMI(MBB, MII, DL, get(Opcode), Interm)
9304 .add(Src0)
9305 .add(Src1);
9306
9307 MachineInstr &Not = *BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), NewDest)
9308 .addReg(Interm);
9309
9310 Worklist.insert(&Op);
9311 Worklist.insert(&Not);
9312
9313 MRI.replaceRegWith(Dest.getReg(), NewDest);
9314 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9315}
9316
9317void SIInstrInfo::splitScalarBinOpN2(SIInstrWorklist &Worklist,
9318 MachineInstr &Inst,
9319 unsigned Opcode) const {
9320 MachineBasicBlock &MBB = *Inst.getParent();
9321 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9322 MachineBasicBlock::iterator MII = Inst;
9323 const DebugLoc &DL = Inst.getDebugLoc();
9324
9325 MachineOperand &Dest = Inst.getOperand(0);
9326 MachineOperand &Src0 = Inst.getOperand(1);
9327 MachineOperand &Src1 = Inst.getOperand(2);
9328
9329 Register NewDest = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
9330 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
9331
9332 MachineInstr &Not = *BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B32), Interm)
9333 .add(Src1);
9334
9335 MachineInstr &Op = *BuildMI(MBB, MII, DL, get(Opcode), NewDest)
9336 .add(Src0)
9337 .addReg(Interm);
9338
9339 Worklist.insert(&Not);
9340 Worklist.insert(&Op);
9341
9342 MRI.replaceRegWith(Dest.getReg(), NewDest);
9343 addUsersToMoveToVALUWorklist(NewDest, MRI, Worklist);
9344}
9345
9346void SIInstrInfo::splitScalar64BitUnaryOp(SIInstrWorklist &Worklist,
9347 MachineInstr &Inst, unsigned Opcode,
9348 bool Swap) const {
9349 MachineBasicBlock &MBB = *Inst.getParent();
9350 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9351
9352 MachineOperand &Dest = Inst.getOperand(0);
9353 MachineOperand &Src0 = Inst.getOperand(1);
9354 const DebugLoc &DL = Inst.getDebugLoc();
9355
9356 MachineBasicBlock::iterator MII = Inst;
9357
9358 const MCInstrDesc &InstDesc = get(Opcode);
9359 const TargetRegisterClass *Src0RC = Src0.isReg() ?
9360 MRI.getRegClass(Src0.getReg()) :
9361 &AMDGPU::SGPR_32RegClass;
9362
9363 const TargetRegisterClass *Src0SubRC =
9364 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9365
9366 MachineOperand SrcReg0Sub0 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9367 AMDGPU::sub0, Src0SubRC);
9368
9369 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9370 const TargetRegisterClass *NewDestRC = RI.getEquivalentVGPRClass(DestRC);
9371 const TargetRegisterClass *NewDestSubRC =
9372 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9373
9374 Register DestSub0 = MRI.createVirtualRegister(NewDestSubRC);
9375 MachineInstr &LoHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub0).add(SrcReg0Sub0);
9376
9377 MachineOperand SrcReg0Sub1 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9378 AMDGPU::sub1, Src0SubRC);
9379
9380 Register DestSub1 = MRI.createVirtualRegister(NewDestSubRC);
9381 MachineInstr &HiHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub1).add(SrcReg0Sub1);
9382
9383 if (Swap)
9384 std::swap(DestSub0, DestSub1);
9385
9386 Register FullDestReg = MRI.createVirtualRegister(NewDestRC);
9387 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9388 .addReg(DestSub0)
9389 .addImm(AMDGPU::sub0)
9390 .addReg(DestSub1)
9391 .addImm(AMDGPU::sub1);
9392
9393 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9394
9395 Worklist.insert(&LoHalf);
9396 Worklist.insert(&HiHalf);
9397
9398 // We don't need to legalizeOperands here because for a single operand, src0
9399 // will support any kind of input.
9400
9401 // Move all users of this moved value.
9402 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9403}
9404
9405// There is not a vector equivalent of s_mul_u64. For this reason, we need to
9406// split the s_mul_u64 in 32-bit vector multiplications.
9407void SIInstrInfo::splitScalarSMulU64(SIInstrWorklist &Worklist,
9408 MachineInstr &Inst,
9409 MachineDominatorTree *MDT) const {
9410 MachineBasicBlock &MBB = *Inst.getParent();
9411 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9412
9413 Register FullDestReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9414 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9415 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9416
9417 MachineOperand &Dest = Inst.getOperand(0);
9418 MachineOperand &Src0 = Inst.getOperand(1);
9419 MachineOperand &Src1 = Inst.getOperand(2);
9420 const DebugLoc &DL = Inst.getDebugLoc();
9421 MachineBasicBlock::iterator MII = Inst;
9422
9423 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0.getReg());
9424 const TargetRegisterClass *Src1RC = MRI.getRegClass(Src1.getReg());
9425 const TargetRegisterClass *Src0SubRC =
9426 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9427 if (RI.isSGPRClass(Src0SubRC))
9428 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9429 const TargetRegisterClass *Src1SubRC =
9430 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9431 if (RI.isSGPRClass(Src1SubRC))
9432 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9433
9434 // First, we extract the low 32-bit and high 32-bit values from each of the
9435 // operands.
9436 MachineOperand Op0L =
9437 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
9438 MachineOperand Op1L =
9439 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
9440 MachineOperand Op0H =
9441 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
9442 MachineOperand Op1H =
9443 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
9444
9445 // The multilication is done as follows:
9446 //
9447 // Op1H Op1L
9448 // * Op0H Op0L
9449 // --------------------
9450 // Op1H*Op0L Op1L*Op0L
9451 // + Op1H*Op0H Op1L*Op0H
9452 // -----------------------------------------
9453 // (Op1H*Op0L + Op1L*Op0H + carry) Op1L*Op0L
9454 //
9455 // We drop Op1H*Op0H because the result of the multiplication is a 64-bit
9456 // value and that would overflow.
9457 // The low 32-bit value is Op1L*Op0L.
9458 // The high 32-bit value is Op1H*Op0L + Op1L*Op0H + carry (from Op1L*Op0L).
9459
9460 Register Op1L_Op0H_Reg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9461 MachineInstr *Op1L_Op0H =
9462 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), Op1L_Op0H_Reg)
9463 .add(Op1L)
9464 .add(Op0H);
9465
9466 Register Op1H_Op0L_Reg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9467 MachineInstr *Op1H_Op0L =
9468 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), Op1H_Op0L_Reg)
9469 .add(Op1H)
9470 .add(Op0L);
9471
9472 Register CarryReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9473 MachineInstr *Carry =
9474 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_HI_U32_e64), CarryReg)
9475 .add(Op1L)
9476 .add(Op0L);
9477
9478 MachineInstr *LoHalf =
9479 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), DestSub0)
9480 .add(Op1L)
9481 .add(Op0L);
9482
9483 Register AddReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9484 MachineInstr *Add = BuildMI(MBB, MII, DL, get(AMDGPU::V_ADD_U32_e32), AddReg)
9485 .addReg(Op1L_Op0H_Reg)
9486 .addReg(Op1H_Op0L_Reg);
9487
9488 MachineInstr *HiHalf =
9489 BuildMI(MBB, MII, DL, get(AMDGPU::V_ADD_U32_e32), DestSub1)
9490 .addReg(AddReg)
9491 .addReg(CarryReg);
9492
9493 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9494 .addReg(DestSub0)
9495 .addImm(AMDGPU::sub0)
9496 .addReg(DestSub1)
9497 .addImm(AMDGPU::sub1);
9498
9499 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9500
9501 // Try to legalize the operands in case we need to swap the order to keep it
9502 // valid.
9503 legalizeOperands(*Op1L_Op0H, MDT);
9504 legalizeOperands(*Op1H_Op0L, MDT);
9505 legalizeOperands(*Carry, MDT);
9506 legalizeOperands(*LoHalf, MDT);
9507 legalizeOperands(*Add, MDT);
9508 legalizeOperands(*HiHalf, MDT);
9509
9510 // Move all users of this moved value.
9511 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9512}
9513
9514// Lower S_MUL_U64_U32_PSEUDO/S_MUL_I64_I32_PSEUDO in two 32-bit vector
9515// multiplications.
9516void SIInstrInfo::splitScalarSMulPseudo(SIInstrWorklist &Worklist,
9517 MachineInstr &Inst,
9518 MachineDominatorTree *MDT) const {
9519 MachineBasicBlock &MBB = *Inst.getParent();
9520 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9521
9522 Register FullDestReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9523 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9524 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9525
9526 MachineOperand &Dest = Inst.getOperand(0);
9527 MachineOperand &Src0 = Inst.getOperand(1);
9528 MachineOperand &Src1 = Inst.getOperand(2);
9529 const DebugLoc &DL = Inst.getDebugLoc();
9530 MachineBasicBlock::iterator MII = Inst;
9531
9532 const TargetRegisterClass *Src0RC = MRI.getRegClass(Src0.getReg());
9533 const TargetRegisterClass *Src1RC = MRI.getRegClass(Src1.getReg());
9534 const TargetRegisterClass *Src0SubRC =
9535 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9536 if (RI.isSGPRClass(Src0SubRC))
9537 Src0SubRC = RI.getEquivalentVGPRClass(Src0SubRC);
9538 const TargetRegisterClass *Src1SubRC =
9539 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9540 if (RI.isSGPRClass(Src1SubRC))
9541 Src1SubRC = RI.getEquivalentVGPRClass(Src1SubRC);
9542
9543 // First, we extract the low 32-bit and high 32-bit values from each of the
9544 // operands.
9545 MachineOperand Op0L =
9546 buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
9547 MachineOperand Op1L =
9548 buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
9549
9550 unsigned Opc = Inst.getOpcode();
9551 unsigned NewOpc = Opc == AMDGPU::S_MUL_U64_U32_PSEUDO
9552 ? AMDGPU::V_MUL_HI_U32_e64
9553 : AMDGPU::V_MUL_HI_I32_e64;
9554 MachineInstr *HiHalf =
9555 BuildMI(MBB, MII, DL, get(NewOpc), DestSub1).add(Op1L).add(Op0L);
9556
9557 MachineInstr *LoHalf =
9558 BuildMI(MBB, MII, DL, get(AMDGPU::V_MUL_LO_U32_e64), DestSub0)
9559 .add(Op1L)
9560 .add(Op0L);
9561
9562 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9563 .addReg(DestSub0)
9564 .addImm(AMDGPU::sub0)
9565 .addReg(DestSub1)
9566 .addImm(AMDGPU::sub1);
9567
9568 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9569
9570 // Try to legalize the operands in case we need to swap the order to keep it
9571 // valid.
9572 legalizeOperands(*HiHalf, MDT);
9573 legalizeOperands(*LoHalf, MDT);
9574
9575 // Move all users of this moved value.
9576 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9577}
9578
9579void SIInstrInfo::splitScalar64BitBinaryOp(SIInstrWorklist &Worklist,
9580 MachineInstr &Inst, unsigned Opcode,
9581 MachineDominatorTree *MDT) const {
9582 MachineBasicBlock &MBB = *Inst.getParent();
9583 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9584
9585 MachineOperand &Dest = Inst.getOperand(0);
9586 MachineOperand &Src0 = Inst.getOperand(1);
9587 MachineOperand &Src1 = Inst.getOperand(2);
9588 const DebugLoc &DL = Inst.getDebugLoc();
9589
9590 MachineBasicBlock::iterator MII = Inst;
9591
9592 const MCInstrDesc &InstDesc = get(Opcode);
9593 const TargetRegisterClass *Src0RC = Src0.isReg() ?
9594 MRI.getRegClass(Src0.getReg()) :
9595 &AMDGPU::SGPR_32RegClass;
9596
9597 const TargetRegisterClass *Src0SubRC =
9598 RI.getSubRegisterClass(Src0RC, AMDGPU::sub0);
9599 const TargetRegisterClass *Src1RC = Src1.isReg() ?
9600 MRI.getRegClass(Src1.getReg()) :
9601 &AMDGPU::SGPR_32RegClass;
9602
9603 const TargetRegisterClass *Src1SubRC =
9604 RI.getSubRegisterClass(Src1RC, AMDGPU::sub0);
9605
9606 MachineOperand SrcReg0Sub0 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9607 AMDGPU::sub0, Src0SubRC);
9608 MachineOperand SrcReg1Sub0 = buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC,
9609 AMDGPU::sub0, Src1SubRC);
9610 MachineOperand SrcReg0Sub1 = buildExtractSubRegOrImm(MII, MRI, Src0, Src0RC,
9611 AMDGPU::sub1, Src0SubRC);
9612 MachineOperand SrcReg1Sub1 = buildExtractSubRegOrImm(MII, MRI, Src1, Src1RC,
9613 AMDGPU::sub1, Src1SubRC);
9614
9615 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9616 const TargetRegisterClass *NewDestRC = RI.getEquivalentVGPRClass(DestRC);
9617 const TargetRegisterClass *NewDestSubRC =
9618 RI.getSubRegisterClass(NewDestRC, AMDGPU::sub0);
9619
9620 Register DestSub0 = MRI.createVirtualRegister(NewDestSubRC);
9621 MachineInstr &LoHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub0)
9622 .add(SrcReg0Sub0)
9623 .add(SrcReg1Sub0);
9624
9625 Register DestSub1 = MRI.createVirtualRegister(NewDestSubRC);
9626 MachineInstr &HiHalf = *BuildMI(MBB, MII, DL, InstDesc, DestSub1)
9627 .add(SrcReg0Sub1)
9628 .add(SrcReg1Sub1);
9629
9630 Register FullDestReg = MRI.createVirtualRegister(NewDestRC);
9631 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), FullDestReg)
9632 .addReg(DestSub0)
9633 .addImm(AMDGPU::sub0)
9634 .addReg(DestSub1)
9635 .addImm(AMDGPU::sub1);
9636
9637 MRI.replaceRegWith(Dest.getReg(), FullDestReg);
9638
9639 Worklist.insert(&LoHalf);
9640 Worklist.insert(&HiHalf);
9641
9642 // Move all users of this moved value.
9643 addUsersToMoveToVALUWorklist(FullDestReg, MRI, Worklist);
9644}
9645
9646void SIInstrInfo::splitScalar64BitXnor(SIInstrWorklist &Worklist,
9647 MachineInstr &Inst,
9648 MachineDominatorTree *MDT) const {
9649 MachineBasicBlock &MBB = *Inst.getParent();
9650 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9651
9652 MachineOperand &Dest = Inst.getOperand(0);
9653 MachineOperand &Src0 = Inst.getOperand(1);
9654 MachineOperand &Src1 = Inst.getOperand(2);
9655 const DebugLoc &DL = Inst.getDebugLoc();
9656
9657 MachineBasicBlock::iterator MII = Inst;
9658
9659 const TargetRegisterClass *DestRC = MRI.getRegClass(Dest.getReg());
9660
9661 Register Interm = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
9662
9663 MachineOperand* Op0;
9664 MachineOperand* Op1;
9665
9666 if (Src0.isReg() && RI.isSGPRReg(MRI, Src0.getReg())) {
9667 Op0 = &Src0;
9668 Op1 = &Src1;
9669 } else {
9670 Op0 = &Src1;
9671 Op1 = &Src0;
9672 }
9673
9674 BuildMI(MBB, MII, DL, get(AMDGPU::S_NOT_B64), Interm)
9675 .add(*Op0);
9676
9677 Register NewDest = MRI.createVirtualRegister(DestRC);
9678
9679 MachineInstr &Xor = *BuildMI(MBB, MII, DL, get(AMDGPU::S_XOR_B64), NewDest)
9680 .addReg(Interm)
9681 .add(*Op1);
9682
9683 MRI.replaceRegWith(Dest.getReg(), NewDest);
9684
9685 Worklist.insert(&Xor);
9686}
9687
9688void SIInstrInfo::splitScalar64BitBCNT(SIInstrWorklist &Worklist,
9689 MachineInstr &Inst) const {
9690 MachineBasicBlock &MBB = *Inst.getParent();
9691 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9692
9693 MachineBasicBlock::iterator MII = Inst;
9694 const DebugLoc &DL = Inst.getDebugLoc();
9695
9696 MachineOperand &Dest = Inst.getOperand(0);
9697 MachineOperand &Src = Inst.getOperand(1);
9698
9699 const MCInstrDesc &InstDesc = get(AMDGPU::V_BCNT_U32_B32_e64);
9700 const TargetRegisterClass *SrcRC = Src.isReg() ?
9701 MRI.getRegClass(Src.getReg()) :
9702 &AMDGPU::SGPR_32RegClass;
9703
9704 Register MidReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9705 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9706
9707 const TargetRegisterClass *SrcSubRC =
9708 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9709
9710 MachineOperand SrcRegSub0 = buildExtractSubRegOrImm(MII, MRI, Src, SrcRC,
9711 AMDGPU::sub0, SrcSubRC);
9712 MachineOperand SrcRegSub1 = buildExtractSubRegOrImm(MII, MRI, Src, SrcRC,
9713 AMDGPU::sub1, SrcSubRC);
9714
9715 BuildMI(MBB, MII, DL, InstDesc, MidReg).add(SrcRegSub0).addImm(0);
9716
9717 BuildMI(MBB, MII, DL, InstDesc, ResultReg).add(SrcRegSub1).addReg(MidReg);
9718
9719 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9720
9721 // We don't need to legalize operands here. src0 for either instruction can be
9722 // an SGPR, and the second input is unused or determined here.
9723 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9724}
9725
9726void SIInstrInfo::splitScalar64BitBFE(SIInstrWorklist &Worklist,
9727 MachineInstr &Inst) const {
9728 MachineBasicBlock &MBB = *Inst.getParent();
9729 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9730 MachineBasicBlock::iterator MII = Inst;
9731 const DebugLoc &DL = Inst.getDebugLoc();
9732
9733 MachineOperand &Dest = Inst.getOperand(0);
9734 uint32_t Imm = Inst.getOperand(2).getImm();
9735 uint32_t Offset = Imm & 0x3f; // Extract bits [5:0].
9736 uint32_t BitWidth = (Imm & 0x7f0000) >> 16; // Extract bits [22:16].
9737
9738 (void) Offset;
9739
9740 // Only sext_inreg cases handled.
9741 assert(Inst.getOpcode() == AMDGPU::S_BFE_I64 && BitWidth <= 32 &&
9742 Offset == 0 && "Not implemented");
9743
9744 if (BitWidth < 32) {
9745 Register MidRegLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9746 Register MidRegHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9747 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9748
9749 BuildMI(MBB, MII, DL, get(AMDGPU::V_BFE_I32_e64), MidRegLo)
9750 .addReg(Inst.getOperand(1).getReg(), {}, AMDGPU::sub0)
9751 .addImm(0)
9752 .addImm(BitWidth);
9753
9754 BuildMI(MBB, MII, DL, get(AMDGPU::V_ASHRREV_I32_e32), MidRegHi)
9755 .addImm(31)
9756 .addReg(MidRegLo);
9757
9758 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), ResultReg)
9759 .addReg(MidRegLo)
9760 .addImm(AMDGPU::sub0)
9761 .addReg(MidRegHi)
9762 .addImm(AMDGPU::sub1);
9763
9764 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9765 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9766 return;
9767 }
9768
9769 MachineOperand &Src = Inst.getOperand(1);
9770 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9771 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VReg_64RegClass);
9772
9773 BuildMI(MBB, MII, DL, get(AMDGPU::V_ASHRREV_I32_e64), TmpReg)
9774 .addImm(31)
9775 .addReg(Src.getReg(), {}, AMDGPU::sub0);
9776
9777 BuildMI(MBB, MII, DL, get(TargetOpcode::REG_SEQUENCE), ResultReg)
9778 .addReg(Src.getReg(), {}, AMDGPU::sub0)
9779 .addImm(AMDGPU::sub0)
9780 .addReg(TmpReg)
9781 .addImm(AMDGPU::sub1);
9782
9783 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9784 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9785}
9786
9787void SIInstrInfo::splitScalar64BitCountOp(SIInstrWorklist &Worklist,
9788 MachineInstr &Inst, unsigned Opcode,
9789 MachineDominatorTree *MDT) const {
9790 // (S_FLBIT_I32_B64 hi:lo) ->
9791 // -> (umin (V_FFBH_U32_e32 hi), (or (V_FFBH_U32_e32 lo), 32))
9792 // (S_FF1_I32_B64 hi:lo) ->
9793 // ->(umin (or (V_FFBL_B32_e32 hi), 32) (V_FFBL_B32_e32 lo))
9794
9795 MachineBasicBlock &MBB = *Inst.getParent();
9796 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9797 MachineBasicBlock::iterator MII = Inst;
9798 const DebugLoc &DL = Inst.getDebugLoc();
9799
9800 MachineOperand &Dest = Inst.getOperand(0);
9801 MachineOperand &Src = Inst.getOperand(1);
9802
9803 const MCInstrDesc &InstDesc = get(Opcode);
9804
9805 bool IsCtlz = Opcode == AMDGPU::V_FFBH_U32_e32;
9806
9807 const TargetRegisterClass *SrcRC =
9808 Src.isReg() ? MRI.getRegClass(Src.getReg()) : &AMDGPU::SGPR_32RegClass;
9809 const TargetRegisterClass *SrcSubRC =
9810 RI.getSubRegisterClass(SrcRC, AMDGPU::sub0);
9811
9812 MachineOperand SrcRegSub0 =
9813 buildExtractSubRegOrImm(MII, MRI, Src, SrcRC, AMDGPU::sub0, SrcSubRC);
9814 MachineOperand SrcRegSub1 =
9815 buildExtractSubRegOrImm(MII, MRI, Src, SrcRC, AMDGPU::sub1, SrcSubRC);
9816
9817 Register MidReg1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9818 Register MidReg2 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9819 Register MidReg3 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9820 Register MidReg4 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9821
9822 BuildMI(MBB, MII, DL, InstDesc, MidReg1).add(SrcRegSub0);
9823
9824 BuildMI(MBB, MII, DL, InstDesc, MidReg2).add(SrcRegSub1);
9825
9826 BuildMI(MBB, MII, DL, get(AMDGPU::V_OR_B32_e32), MidReg3)
9827 .addImm(32)
9828 .addReg(IsCtlz ? MidReg1 : MidReg2);
9829
9830 BuildMI(MBB, MII, DL, get(AMDGPU::V_MIN_U32_e64), MidReg4)
9831 .addReg(MidReg3)
9832 .addReg(IsCtlz ? MidReg2 : MidReg1);
9833
9834 MRI.replaceRegWith(Dest.getReg(), MidReg4);
9835
9836 addUsersToMoveToVALUWorklist(MidReg4, MRI, Worklist);
9837}
9838
9839void SIInstrInfo::addUsersToMoveToVALUWorklist(
9840 Register DstReg, MachineRegisterInfo &MRI,
9841 SIInstrWorklist &Worklist) const {
9842 for (MachineOperand &MO : make_early_inc_range(MRI.use_operands(DstReg))) {
9843 MachineInstr &UseMI = *MO.getParent();
9844
9845 unsigned OpNo = 0;
9846
9847 switch (UseMI.getOpcode()) {
9848 case AMDGPU::COPY:
9849 case AMDGPU::WQM:
9850 case AMDGPU::SOFT_WQM:
9851 case AMDGPU::STRICT_WWM:
9852 case AMDGPU::STRICT_WQM:
9853 case AMDGPU::REG_SEQUENCE:
9854 case AMDGPU::PHI:
9855 case AMDGPU::INSERT_SUBREG:
9856 break;
9857 default:
9858 OpNo = MO.getOperandNo();
9859 break;
9860 }
9861
9862 const TargetRegisterClass *OpRC = getOpRegClass(UseMI, OpNo);
9863 MRI.constrainRegClass(DstReg, OpRC);
9864
9865 if (!RI.hasVectorRegisters(OpRC))
9866 Worklist.insert(&UseMI);
9867 else
9868 // Legalization could change user list.
9869 legalizeOperandsVALUt16(UseMI, OpNo, MRI);
9870 }
9871}
9872
9873void SIInstrInfo::movePackToVALU(SIInstrWorklist &Worklist,
9875 MachineInstr &Inst) const {
9876 Register ResultReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9877 MachineBasicBlock *MBB = Inst.getParent();
9878 MachineOperand &Src0 = Inst.getOperand(1);
9879 MachineOperand &Src1 = Inst.getOperand(2);
9880 const DebugLoc &DL = Inst.getDebugLoc();
9881
9882 if (ST.useRealTrue16Insts()) {
9883 Register SrcReg0, SrcReg1;
9884 if (!Src0.isReg() || !RI.isVGPR(MRI, Src0.getReg())) {
9885 SrcReg0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9886 BuildMI(*MBB, Inst, DL,
9887 get(Src0.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg0)
9888 .add(Src0);
9889 } else {
9890 SrcReg0 = Src0.getReg();
9891 }
9892
9893 if (!Src1.isReg() || !RI.isVGPR(MRI, Src1.getReg())) {
9894 SrcReg1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9895 BuildMI(*MBB, Inst, DL,
9896 get(Src1.isImm() ? AMDGPU::V_MOV_B32_e32 : AMDGPU::COPY), SrcReg1)
9897 .add(Src1);
9898 } else {
9899 SrcReg1 = Src1.getReg();
9900 }
9901
9902 bool isSrc0Reg16 = MRI.constrainRegClass(SrcReg0, &AMDGPU::VGPR_16RegClass);
9903 bool isSrc1Reg16 = MRI.constrainRegClass(SrcReg1, &AMDGPU::VGPR_16RegClass);
9904
9905 auto NewMI = BuildMI(*MBB, Inst, DL, get(AMDGPU::REG_SEQUENCE), ResultReg);
9906 switch (Inst.getOpcode()) {
9907 case AMDGPU::S_PACK_LL_B32_B16:
9908 NewMI
9909 .addReg(SrcReg0, {},
9910 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9911 .addImm(AMDGPU::lo16)
9912 .addReg(SrcReg1, {},
9913 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9914 .addImm(AMDGPU::hi16);
9915 break;
9916 case AMDGPU::S_PACK_LH_B32_B16:
9917 NewMI
9918 .addReg(SrcReg0, {},
9919 isSrc0Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9920 .addImm(AMDGPU::lo16)
9921 .addReg(SrcReg1, {}, AMDGPU::hi16)
9922 .addImm(AMDGPU::hi16);
9923 break;
9924 case AMDGPU::S_PACK_HL_B32_B16:
9925 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9926 .addImm(AMDGPU::lo16)
9927 .addReg(SrcReg1, {},
9928 isSrc1Reg16 ? AMDGPU::NoSubRegister : AMDGPU::lo16)
9929 .addImm(AMDGPU::hi16);
9930 break;
9931 case AMDGPU::S_PACK_HH_B32_B16:
9932 NewMI.addReg(SrcReg0, {}, AMDGPU::hi16)
9933 .addImm(AMDGPU::lo16)
9934 .addReg(SrcReg1, {}, AMDGPU::hi16)
9935 .addImm(AMDGPU::hi16);
9936 break;
9937 default:
9938 llvm_unreachable("unhandled s_pack_* instruction");
9939 }
9940
9941 MachineOperand &Dest = Inst.getOperand(0);
9942 MRI.replaceRegWith(Dest.getReg(), ResultReg);
9943 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
9944 return;
9945 }
9946
9947 switch (Inst.getOpcode()) {
9948 case AMDGPU::S_PACK_LL_B32_B16: {
9949 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9950 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9951
9952 // FIXME: Can do a lot better if we know the high bits of src0 or src1 are
9953 // 0.
9954 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
9955 .addImm(0xffff);
9956
9957 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_AND_B32_e64), TmpReg)
9958 .addReg(ImmReg, RegState::Kill)
9959 .add(Src0);
9960
9961 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHL_OR_B32_e64), ResultReg)
9962 .add(Src1)
9963 .addImm(16)
9964 .addReg(TmpReg, RegState::Kill);
9965 break;
9966 }
9967 case AMDGPU::S_PACK_LH_B32_B16: {
9968 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9969 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
9970 .addImm(0xffff);
9971 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_BFI_B32_e64), ResultReg)
9972 .addReg(ImmReg, RegState::Kill)
9973 .add(Src0)
9974 .add(Src1);
9975 break;
9976 }
9977 case AMDGPU::S_PACK_HL_B32_B16: {
9978 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9979 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
9980 .addImm(16)
9981 .add(Src0);
9982 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHL_OR_B32_e64), ResultReg)
9983 .add(Src1)
9984 .addImm(16)
9985 .addReg(TmpReg, RegState::Kill);
9986 break;
9987 }
9988 case AMDGPU::S_PACK_HH_B32_B16: {
9989 Register ImmReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9990 Register TmpReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
9991 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_LSHRREV_B32_e64), TmpReg)
9992 .addImm(16)
9993 .add(Src0);
9994 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_MOV_B32_e32), ImmReg)
9995 .addImm(0xffff0000);
9996 BuildMI(*MBB, Inst, DL, get(AMDGPU::V_AND_OR_B32_e64), ResultReg)
9997 .add(Src1)
9998 .addReg(ImmReg, RegState::Kill)
9999 .addReg(TmpReg, RegState::Kill);
10000 break;
10001 }
10002 default:
10003 llvm_unreachable("unhandled s_pack_* instruction");
10004 }
10005
10006 MachineOperand &Dest = Inst.getOperand(0);
10007 MRI.replaceRegWith(Dest.getReg(), ResultReg);
10008 addUsersToMoveToVALUWorklist(ResultReg, MRI, Worklist);
10009}
10010
10011void SIInstrInfo::addSCCDefUsersToVALUWorklist(const MachineOperand &Op,
10012 MachineInstr &SCCDefInst,
10013 SIInstrWorklist &Worklist,
10014 Register NewCond) const {
10015
10016 // Ensure that def inst defines SCC, which is still live.
10017 assert(Op.isReg() && Op.getReg() == AMDGPU::SCC && Op.isDef() &&
10018 !Op.isDead() && Op.getParent() == &SCCDefInst);
10019 SmallVector<MachineInstr *, 4> CopyToDelete;
10020 // This assumes that all the users of SCC are in the same block
10021 // as the SCC def.
10022 for (MachineInstr &MI : // Skip the def inst itself.
10023 make_range(std::next(MachineBasicBlock::iterator(SCCDefInst)),
10024 SCCDefInst.getParent()->end())) {
10025 // Check if SCC is used first.
10026 int SCCIdx = MI.findRegisterUseOperandIdx(AMDGPU::SCC, &RI, false);
10027 if (SCCIdx != -1) {
10028 if (MI.isCopy()) {
10029 MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
10030 Register DestReg = MI.getOperand(0).getReg();
10031
10032 MRI.replaceRegWith(DestReg, NewCond);
10033 CopyToDelete.push_back(&MI);
10034 } else {
10035
10036 if (NewCond.isValid())
10037 MI.getOperand(SCCIdx).setReg(NewCond);
10038
10039 Worklist.insert(&MI);
10040 }
10041 }
10042 // Exit if we find another SCC def.
10043 if (MI.findRegisterDefOperandIdx(AMDGPU::SCC, &RI, false, false) != -1)
10044 break;
10045 }
10046 for (auto &Copy : CopyToDelete)
10047 Copy->eraseFromParent();
10048}
10049
10050// Instructions that use SCC may be converted to VALU instructions. When that
10051// happens, the SCC register is changed to VCC_LO. The instruction that defines
10052// SCC must be changed to an instruction that defines VCC. This function makes
10053// sure that the instruction that defines SCC is added to the moveToVALU
10054// worklist.
10055void SIInstrInfo::addSCCDefsToVALUWorklist(MachineInstr *SCCUseInst,
10056 SIInstrWorklist &Worklist) const {
10057 // Look for a preceding instruction that either defines VCC or SCC. If VCC
10058 // then there is nothing to do because the defining instruction has been
10059 // converted to a VALU already. If SCC then that instruction needs to be
10060 // converted to a VALU.
10061 for (MachineInstr &MI :
10062 make_range(std::next(MachineBasicBlock::reverse_iterator(SCCUseInst)),
10063 SCCUseInst->getParent()->rend())) {
10064 if (MI.modifiesRegister(AMDGPU::VCC, &RI))
10065 break;
10066 if (MI.definesRegister(AMDGPU::SCC, &RI)) {
10067 Worklist.insert(&MI);
10068 break;
10069 }
10070 }
10071}
10072
10073const TargetRegisterClass *SIInstrInfo::getDestEquivalentVGPRClass(
10074 const MachineInstr &Inst) const {
10075 const TargetRegisterClass *NewDstRC = getOpRegClass(Inst, 0);
10076
10077 switch (Inst.getOpcode()) {
10078 // For target instructions, getOpRegClass just returns the virtual register
10079 // class associated with the operand, so we need to find an equivalent VGPR
10080 // register class in order to move the instruction to the VALU.
10081 case AMDGPU::COPY:
10082 case AMDGPU::PHI:
10083 case AMDGPU::REG_SEQUENCE:
10084 case AMDGPU::INSERT_SUBREG:
10085 case AMDGPU::WQM:
10086 case AMDGPU::SOFT_WQM:
10087 case AMDGPU::STRICT_WWM:
10088 case AMDGPU::STRICT_WQM: {
10089 const TargetRegisterClass *SrcRC = getOpRegClass(Inst, 1);
10090 if (RI.isAGPRClass(SrcRC)) {
10091 if (RI.isAGPRClass(NewDstRC))
10092 return nullptr;
10093
10094 switch (Inst.getOpcode()) {
10095 case AMDGPU::PHI:
10096 case AMDGPU::REG_SEQUENCE:
10097 case AMDGPU::INSERT_SUBREG:
10098 NewDstRC = RI.getEquivalentAGPRClass(NewDstRC);
10099 break;
10100 default:
10101 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10102 }
10103
10104 if (!NewDstRC)
10105 return nullptr;
10106 } else {
10107 if (!RI.isSGPRClass(NewDstRC) || NewDstRC == &AMDGPU::VReg_1RegClass)
10108 return nullptr;
10109
10110 NewDstRC = RI.getEquivalentVGPRClass(NewDstRC);
10111 if (!NewDstRC)
10112 return nullptr;
10113 }
10114
10115 return NewDstRC;
10116 }
10117 default:
10118 return NewDstRC;
10119 }
10120}
10121
10122// Find the one SGPR operand we are allowed to use.
10123Register SIInstrInfo::findUsedSGPR(const MachineInstr &MI,
10124 int OpIndices[3]) const {
10125 const MCInstrDesc &Desc = MI.getDesc();
10126
10127 // Find the one SGPR operand we are allowed to use.
10128 //
10129 // First we need to consider the instruction's operand requirements before
10130 // legalizing. Some operands are required to be SGPRs, such as implicit uses
10131 // of VCC, but we are still bound by the constant bus requirement to only use
10132 // one.
10133 //
10134 // If the operand's class is an SGPR, we can never move it.
10135
10136 Register SGPRReg = findImplicitSGPRRead(MI);
10137 if (SGPRReg)
10138 return SGPRReg;
10139
10140 Register UsedSGPRs[3] = {Register()};
10141 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
10142
10143 for (unsigned i = 0; i < 3; ++i) {
10144 int Idx = OpIndices[i];
10145 if (Idx == -1)
10146 break;
10147
10148 const MachineOperand &MO = MI.getOperand(Idx);
10149 if (!MO.isReg())
10150 continue;
10151
10152 // Is this operand statically required to be an SGPR based on the operand
10153 // constraints?
10154 const TargetRegisterClass *OpRC =
10155 RI.getRegClass(getOpRegClassID(Desc.operands()[Idx]));
10156 bool IsRequiredSGPR = RI.isSGPRClass(OpRC);
10157 if (IsRequiredSGPR)
10158 return MO.getReg();
10159
10160 // If this could be a VGPR or an SGPR, Check the dynamic register class.
10161 Register Reg = MO.getReg();
10162 const TargetRegisterClass *RegRC = MRI.getRegClass(Reg);
10163 if (RI.isSGPRClass(RegRC))
10164 UsedSGPRs[i] = Reg;
10165 }
10166
10167 // We don't have a required SGPR operand, so we have a bit more freedom in
10168 // selecting operands to move.
10169
10170 // Try to select the most used SGPR. If an SGPR is equal to one of the
10171 // others, we choose that.
10172 //
10173 // e.g.
10174 // V_FMA_F32 v0, s0, s0, s0 -> No moves
10175 // V_FMA_F32 v0, s0, s1, s0 -> Move s1
10176
10177 // TODO: If some of the operands are 64-bit SGPRs and some 32, we should
10178 // prefer those.
10179
10180 if (UsedSGPRs[0]) {
10181 if (UsedSGPRs[0] == UsedSGPRs[1] || UsedSGPRs[0] == UsedSGPRs[2])
10182 SGPRReg = UsedSGPRs[0];
10183 }
10184
10185 if (!SGPRReg && UsedSGPRs[1]) {
10186 if (UsedSGPRs[1] == UsedSGPRs[2])
10187 SGPRReg = UsedSGPRs[1];
10188 }
10189
10190 return SGPRReg;
10191}
10192
10194 AMDGPU::OpName OperandName) const {
10195 if (OperandName == AMDGPU::OpName::NUM_OPERAND_NAMES)
10196 return nullptr;
10197
10198 int Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OperandName);
10199 if (Idx == -1)
10200 return nullptr;
10201
10202 return &MI.getOperand(Idx);
10203}
10204
10206 if (ST.getGeneration() >= AMDGPUSubtarget::GFX10) {
10207 int64_t Format = ST.getGeneration() >= AMDGPUSubtarget::GFX11
10210 return (Format << 44) |
10211 (1ULL << 56) | // RESOURCE_LEVEL = 1
10212 (3ULL << 60); // OOB_SELECT = 3
10213 }
10214
10215 uint64_t RsrcDataFormat = AMDGPU::RSRC_DATA_FORMAT;
10216 if (ST.isAmdHsaOS()) {
10217 // Set ATC = 1. GFX9 doesn't have this bit.
10218 if (ST.getGeneration() <= AMDGPUSubtarget::VOLCANIC_ISLANDS)
10219 RsrcDataFormat |= (1ULL << 56);
10220
10221 // Set MTYPE = 2 (MTYPE_UC = uncached). GFX9 doesn't have this.
10222 // BTW, it disables TC L2 and therefore decreases performance.
10223 if (ST.getGeneration() == AMDGPUSubtarget::VOLCANIC_ISLANDS)
10224 RsrcDataFormat |= (2ULL << 59);
10225 }
10226
10227 return RsrcDataFormat;
10228}
10229
10231 uint64_t Rsrc23 = getDefaultRsrcDataFormat() |
10233 0xffffffff; // Size;
10234
10235 // GFX9 doesn't have ELEMENT_SIZE.
10236 if (ST.getGeneration() <= AMDGPUSubtarget::VOLCANIC_ISLANDS) {
10237 uint64_t EltSizeValue = Log2_32(ST.getMaxPrivateElementSize(true)) - 1;
10238 Rsrc23 |= EltSizeValue << AMDGPU::RSRC_ELEMENT_SIZE_SHIFT;
10239 }
10240
10241 // IndexStride = 64 / 32.
10242 uint64_t IndexStride = ST.isWave64() ? 3 : 2;
10243 Rsrc23 |= IndexStride << AMDGPU::RSRC_INDEX_STRIDE_SHIFT;
10244
10245 // If TID_ENABLE is set, DATA_FORMAT specifies stride bits [14:17].
10246 // Clear them unless we want a huge stride.
10247 if (ST.getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS &&
10248 ST.getGeneration() <= AMDGPUSubtarget::GFX9)
10249 Rsrc23 &= ~AMDGPU::RSRC_DATA_FORMAT;
10250
10251 return Rsrc23;
10252}
10253
10255 unsigned Opc = MI.getOpcode();
10256
10257 return isSMRD(Opc);
10258}
10259
10261 return get(Opc).mayLoad() &&
10262 (isMUBUF(Opc) || isMTBUF(Opc) || isMIMG(Opc) || isFLAT(Opc));
10263}
10264
10266 TypeSize &MemBytes) const {
10267 const MachineOperand *Addr = getNamedOperand(MI, AMDGPU::OpName::vaddr);
10268 if (!Addr || !Addr->isFI())
10269 return Register();
10270
10271 assert(!MI.memoperands_empty() &&
10272 (*MI.memoperands_begin())->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS);
10273
10274 FrameIndex = Addr->getIndex();
10275
10276 int VDataIdx =
10277 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdata);
10278 MemBytes = TypeSize::getFixed(getOpSize(MI.getOpcode(), VDataIdx));
10279 return MI.getOperand(VDataIdx).getReg();
10280}
10281
10283 TypeSize &MemBytes) const {
10284 const MachineOperand *Addr = getNamedOperand(MI, AMDGPU::OpName::addr);
10285 assert(Addr && Addr->isFI());
10286 FrameIndex = Addr->getIndex();
10287
10288 int DataIdx =
10289 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::data);
10290 MemBytes = TypeSize::getFixed(getOpSize(MI.getOpcode(), DataIdx));
10291 return MI.getOperand(DataIdx).getReg();
10292}
10293
10295 int &FrameIndex,
10296 TypeSize &MemBytes) const {
10297 if (!MI.mayLoad())
10298 return Register();
10299
10300 if (isMUBUF(MI) || isVGPRSpill(MI))
10301 return isStackAccess(MI, FrameIndex, MemBytes);
10302
10303 if (isSGPRSpill(MI))
10304 return isSGPRStackAccess(MI, FrameIndex, MemBytes);
10305
10306 return Register();
10307}
10308
10310 int &FrameIndex,
10311 TypeSize &MemBytes) const {
10312 if (!MI.mayStore())
10313 return Register();
10314
10315 if (isMUBUF(MI) || isVGPRSpill(MI))
10316 return isStackAccess(MI, FrameIndex, MemBytes);
10317
10318 if (isSGPRSpill(MI))
10319 return isSGPRStackAccess(MI, FrameIndex, MemBytes);
10320
10321 return Register();
10322}
10323
10325 unsigned Opc = MI.getOpcode();
10327 unsigned DescSize = Desc.getSize();
10328
10329 // If we have a definitive size, we can use it. Otherwise we need to inspect
10330 // the operands to know the size.
10331 if (isFixedSize(MI)) {
10332 unsigned Size = DescSize;
10333
10334 // If we hit the buggy offset, an extra nop will be inserted in MC so
10335 // estimate the worst case.
10336 if (MI.isBranch() && ST.hasOffset3fBug())
10337 Size += 4;
10338
10339 return Size;
10340 }
10341
10342 // Instructions may have a 32-bit literal encoded after them. Check
10343 // operands that could ever be literals.
10344 if (isVALU(MI, /*AllowLDSDMA=*/false) || isSALU(MI)) {
10345 if (isDPP(MI))
10346 return DescSize;
10347 bool HasLiteral = false;
10348 unsigned LiteralSize = 4;
10349 for (int I = 0, E = MI.getNumExplicitOperands(); I != E; ++I) {
10350 const MachineOperand &Op = MI.getOperand(I);
10351 const MCOperandInfo &OpInfo = Desc.operands()[I];
10352 if (!Op.isReg() && !isInlineConstant(Op, OpInfo)) {
10353 HasLiteral = true;
10354 if (ST.has64BitLiterals()) {
10355 switch (OpInfo.OperandType) {
10356 default:
10357 break;
10360 if (!AMDGPU::isValid32BitLiteral(Op.getImm(), true))
10361 LiteralSize = 8;
10362 break;
10365 // A 32-bit literal is only valid when the value fits in BOTH signed
10366 // and unsigned 32-bit ranges [0, 2^31-1], matching the MC code
10367 // emitter's getLit64Encoding logic. This is because of the lack of
10368 // abilility to tell signedness of the literal, therefore we need to
10369 // be conservative and assume values outside this range require a
10370 // 64-bit literal encoding (8 bytes).
10371 if (!Op.isImm() || !isInt<32>(Op.getImm()) ||
10372 !isUInt<32>(Op.getImm()))
10373 LiteralSize = 8;
10374 break;
10375 }
10376 }
10377 break;
10378 }
10379 }
10380 return HasLiteral ? DescSize + LiteralSize : DescSize;
10381 }
10382
10383 // Check whether we have extra NSA words.
10384 if (isMIMG(MI)) {
10385 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
10386 if (VAddr0Idx < 0)
10387 return 8;
10388
10389 int RSrcIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::srsrc);
10390 return 8 + 4 * ((RSrcIdx - VAddr0Idx + 2) / 4);
10391 }
10392
10393 switch (Opc) {
10394 case TargetOpcode::BUNDLE:
10395 return getInstBundleSize(MI);
10396 case TargetOpcode::INLINEASM:
10397 case TargetOpcode::INLINEASM_BR: {
10398 const MachineFunction *MF = MI.getMF();
10399 const char *AsmStr = MI.getOperand(0).getSymbolName();
10400 return getInlineAsmLength(AsmStr, MF->getTarget().getMCAsmInfo(), &ST);
10401 }
10402 default:
10403 if (MI.isMetaInstruction())
10404 return 0;
10405
10406 // If D16 Pseudo inst, get correct MC code size
10407 const auto *D16Info = AMDGPU::getT16D16Helper(Opc);
10408 if (D16Info) {
10409 // Assume d16_lo/hi inst are always in same size
10410 unsigned LoInstOpcode = D16Info->LoOp;
10411 const MCInstrDesc &Desc = getMCOpcodeFromPseudo(LoInstOpcode);
10412 DescSize = Desc.getSize();
10413 }
10414
10415 // If FMA Pseudo inst, get correct MC code size
10416 if (Opc == AMDGPU::V_FMA_MIX_F16_t16 || Opc == AMDGPU::V_FMA_MIX_BF16_t16) {
10417 // All potential lowerings are the same size; arbitrarily pick one.
10418 const MCInstrDesc &Desc = getMCOpcodeFromPseudo(AMDGPU::V_FMA_MIXLO_F16);
10419 DescSize = Desc.getSize();
10420 }
10421
10422 return DescSize;
10423 }
10424}
10425
10428 if (MI.isBranch() && ST.hasOffset3fBug())
10429 return InstSizeVerifyMode::NoVerify;
10430 return InstSizeVerifyMode::ExactSize;
10431}
10432
10435 static const std::pair<int, const char *> TargetIndices[] = {
10436 {AMDGPU::TI_CONSTDATA_START, "amdgpu-constdata-start"},
10437 {AMDGPU::TI_SCRATCH_RSRC_DWORD0, "amdgpu-scratch-rsrc-dword0"},
10438 {AMDGPU::TI_SCRATCH_RSRC_DWORD1, "amdgpu-scratch-rsrc-dword1"},
10439 {AMDGPU::TI_SCRATCH_RSRC_DWORD2, "amdgpu-scratch-rsrc-dword2"},
10440 {AMDGPU::TI_SCRATCH_RSRC_DWORD3, "amdgpu-scratch-rsrc-dword3"}};
10441 return ArrayRef(TargetIndices);
10442}
10443
10444/// This is used by the post-RA scheduler (SchedulePostRAList.cpp). The
10445/// post-RA version of misched uses CreateTargetMIHazardRecognizer.
10448 const ScheduleDAG *DAG) const {
10449 return new GCNHazardRecognizer(DAG->MF);
10450}
10451
10452/// This is the hazard recognizer used at -O0 by the PostRAHazardRecognizer
10453/// pass.
10460
10461// Called during:
10462// - pre-RA scheduling and post-RA scheduling
10465 const ScheduleDAGMI *DAG) const {
10466 // Borrowed from Arm Target
10467 // We would like to restrict this hazard recognizer to only
10468 // post-RA scheduling; we can tell that we're post-RA because we don't
10469 // track VRegLiveness.
10470 if (!DAG->hasVRegLiveness())
10471 return new GCNHazardRecognizer(DAG->MF);
10473}
10474
10475std::pair<unsigned, unsigned>
10477 return std::pair(TF & MO_MASK, TF & ~MO_MASK);
10478}
10479
10482 static const std::pair<unsigned, const char *> TargetFlags[] = {
10483 {MO_GOTPCREL, "amdgpu-gotprel"},
10484 {MO_GOTPCREL32_LO, "amdgpu-gotprel32-lo"},
10485 {MO_GOTPCREL32_HI, "amdgpu-gotprel32-hi"},
10486 {MO_GOTPCREL64, "amdgpu-gotprel64"},
10487 {MO_REL32_LO, "amdgpu-rel32-lo"},
10488 {MO_REL32_HI, "amdgpu-rel32-hi"},
10489 {MO_REL64, "amdgpu-rel64"},
10490 {MO_ABS32_LO, "amdgpu-abs32-lo"},
10491 {MO_ABS32_HI, "amdgpu-abs32-hi"},
10492 {MO_ABS64, "amdgpu-abs64"},
10493 };
10494
10495 return ArrayRef(TargetFlags);
10496}
10497
10500 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
10501 {
10502 {MONoClobber, "amdgpu-noclobber"},
10503 {MOLastUse, "amdgpu-last-use"},
10504 {MOCooperative, "amdgpu-cooperative"},
10505 {MOThreadPrivate, "amdgpu-thread-private"},
10506 };
10507
10508 return ArrayRef(TargetFlags);
10509}
10510
10512 const MachineFunction &MF) const {
10514 assert(SrcReg.isVirtual());
10515 if (MFI->checkFlag(SrcReg, AMDGPU::VirtRegFlag::WWM_REG))
10516 return AMDGPU::WWM_COPY;
10517
10518 return AMDGPU::COPY;
10519}
10520
10522 uint32_t Opcode = MI.getOpcode();
10523 // Check if it is SGPR spill or wwm-register spill Opcode.
10524 if (isSGPRSpill(Opcode) || isWWMRegSpillOpcode(Opcode))
10525 return true;
10526
10527 const MachineFunction *MF = MI.getMF();
10528 const MachineRegisterInfo &MRI = MF->getRegInfo();
10530
10531 // See if this is Liverange split instruction inserted for SGPR or
10532 // wwm-register. The implicit def inserted for wwm-registers should also be
10533 // included as they can appear at the bb begin.
10534 bool IsLRSplitInst = MI.getFlag(MachineInstr::LRSplit);
10535 if (!IsLRSplitInst && Opcode != AMDGPU::IMPLICIT_DEF)
10536 return false;
10537
10538 Register Reg = MI.getOperand(0).getReg();
10539 if (RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg)))
10540 return IsLRSplitInst;
10541
10542 return MFI->isWWMReg(Reg);
10543}
10544
10546 Register Reg) const {
10547 // We need to handle instructions which may be inserted during register
10548 // allocation to handle the prolog. The initial prolog instruction may have
10549 // been separated from the start of the block by spills and copies inserted
10550 // needed by the prolog. However, the insertions for scalar registers can
10551 // always be placed at the BB top as they are independent of the exec mask
10552 // value.
10553 bool IsNullOrVectorRegister = true;
10554 if (Reg) {
10555 const MachineFunction *MF = MI.getMF();
10556 const MachineRegisterInfo &MRI = MF->getRegInfo();
10557 IsNullOrVectorRegister = !RI.isSGPRClass(RI.getRegClassForReg(MRI, Reg));
10558 }
10559
10560 return IsNullOrVectorRegister &&
10561 (canAddToBBProlog(MI) ||
10562 (!MI.isTerminator() && MI.getOpcode() != AMDGPU::COPY &&
10563 MI.modifiesRegister(AMDGPU::EXEC, &RI)));
10564}
10565
10569 const DebugLoc &DL,
10570 Register DestReg) const {
10571 if (ST.hasAddNoCarryInsts())
10572 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_U32_e64), DestReg);
10573
10574 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
10575 Register UnusedCarry = MRI.createVirtualRegister(RI.getBoolRC());
10576 MRI.setRegAllocationHint(UnusedCarry, 0, RI.getVCC());
10577
10578 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_CO_U32_e64), DestReg)
10579 .addReg(UnusedCarry, RegState::Define | RegState::Dead);
10580}
10581
10584 const DebugLoc &DL,
10585 Register DestReg,
10586 RegScavenger &RS) const {
10587 if (ST.hasAddNoCarryInsts())
10588 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_U32_e32), DestReg);
10589
10590 // If available, prefer to use vcc.
10591 Register UnusedCarry = !RS.isRegUsed(AMDGPU::VCC)
10592 ? Register(RI.getVCC())
10593 : RS.scavengeRegisterBackwards(
10594 *RI.getBoolRC(), I, /* RestoreAfter */ false,
10595 0, /* AllowSpill */ false);
10596
10597 // TODO: Users need to deal with this.
10598 if (!UnusedCarry.isValid())
10599 return MachineInstrBuilder();
10600
10601 return BuildMI(MBB, I, DL, get(AMDGPU::V_ADD_CO_U32_e64), DestReg)
10602 .addReg(UnusedCarry, RegState::Define | RegState::Dead);
10603}
10604
10605bool SIInstrInfo::isKillTerminator(unsigned Opcode) {
10606 switch (Opcode) {
10607 case AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR:
10608 case AMDGPU::SI_KILL_I1_TERMINATOR:
10609 return true;
10610 default:
10611 return false;
10612 }
10613}
10614
10616 switch (Opcode) {
10617 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
10618 return get(AMDGPU::SI_KILL_F32_COND_IMM_TERMINATOR);
10619 case AMDGPU::SI_KILL_I1_PSEUDO:
10620 return get(AMDGPU::SI_KILL_I1_TERMINATOR);
10621 default:
10622 llvm_unreachable("invalid opcode, expected SI_KILL_*_PSEUDO");
10623 }
10624}
10625
10627 return Imm <= getMaxMUBUFImmOffset(ST);
10628}
10629
10631 // GFX12 field is non-negative 24-bit signed byte offset.
10632 const unsigned OffsetBits =
10633 ST.getGeneration() >= AMDGPUSubtarget::GFX12 ? 23 : 12;
10634 return (1 << OffsetBits) - 1;
10635}
10636
10638 if (!ST.isWave32())
10639 return;
10640
10641 if (MI.isInlineAsm())
10642 return;
10643
10644 if (MI.getNumOperands() < MI.getDesc().getNumOperands())
10645 return;
10646
10647 for (auto &Op : MI.implicit_operands()) {
10648 if (Op.isReg() && Op.getReg() == AMDGPU::VCC)
10649 Op.setReg(AMDGPU::VCC_LO);
10650 }
10651}
10652
10654 if (!isSMRD(MI))
10655 return false;
10656
10657 // Check that it is using a buffer resource.
10658 int Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::sbase);
10659 if (Idx == -1) // e.g. s_memtime
10660 return false;
10661
10662 const int16_t RCID = getOpRegClassID(MI.getDesc().operands()[Idx]);
10663 return RI.getRegClass(RCID)->hasSubClassEq(&AMDGPU::SGPR_128RegClass);
10664}
10665
10666// Given Imm, split it into the values to put into the SOffset and ImmOffset
10667// fields in an MUBUF instruction. Return false if it is not possible (due to a
10668// hardware bug needing a workaround).
10669//
10670// The required alignment ensures that individual address components remain
10671// aligned if they are aligned to begin with. It also ensures that additional
10672// offsets within the given alignment can be added to the resulting ImmOffset.
10674 uint32_t &ImmOffset, Align Alignment) const {
10675 const uint64_t MaxOffset = SIInstrInfo::getMaxMUBUFImmOffset(ST);
10676 const uint32_t MaxImm = alignDown(MaxOffset, Alignment.value());
10677 uint32_t Overflow = 0;
10678
10679 if (Imm > MaxImm) {
10680 if (Imm <= MaxImm + 64) {
10681 // Use an SOffset inline constant for 4..64
10682 Overflow = Imm - MaxImm;
10683 Imm = MaxImm;
10684 } else {
10685 // Try to keep the same value in SOffset for adjacent loads, so that
10686 // the corresponding register contents can be re-used.
10687 //
10688 // Load values with all low-bits (except for alignment bits) set into
10689 // SOffset, so that a larger range of values can be covered using
10690 // s_movk_i32.
10691 //
10692 // Atomic operations fail to work correctly when individual address
10693 // components are unaligned, even if their sum is aligned.
10694 uint32_t High = (Imm + Alignment.value()) & ~MaxOffset;
10695 uint32_t Low = (Imm + Alignment.value()) & MaxOffset;
10696 Imm = Low;
10697 Overflow = High - Alignment.value();
10698 }
10699 }
10700
10701 if (Overflow > 0) {
10702 // There is a hardware bug in SI and CI which prevents address clamping in
10703 // MUBUF instructions from working correctly with SOffsets. The immediate
10704 // offset is unaffected.
10705 if (ST.getGeneration() <= AMDGPUSubtarget::SEA_ISLANDS)
10706 return false;
10707
10708 // It is not possible to set immediate in SOffset field on some targets.
10709 if (ST.hasRestrictedSOffset())
10710 return false;
10711 }
10712
10713 ImmOffset = Imm;
10714 SOffset = Overflow;
10715 return true;
10716}
10717
10718// Depending on the used address space and instructions, some immediate offsets
10719// are allowed and some are not.
10720// Pre-GFX12, flat instruction offsets can only be non-negative, global and
10721// scratch instruction offsets can also be negative. On GFX12, offsets can be
10722// negative for all variants.
10723//
10724// There are several bugs related to these offsets:
10725// On gfx10.1, flat instructions that go into the global address space cannot
10726// use an offset.
10727//
10728// For scratch instructions, the address can be either an SGPR or a VGPR.
10729// The following offsets can be used, depending on the architecture (x means
10730// cannot be used):
10731// +----------------------------+------+------+
10732// | Address-Mode | SGPR | VGPR |
10733// +----------------------------+------+------+
10734// | gfx9 | | |
10735// | negative, 4-aligned offset | x | ok |
10736// | negative, unaligned offset | x | ok |
10737// +----------------------------+------+------+
10738// | gfx10 | | |
10739// | negative, 4-aligned offset | ok | ok |
10740// | negative, unaligned offset | ok | x |
10741// +----------------------------+------+------+
10742// | gfx10.3 | | |
10743// | negative, 4-aligned offset | ok | ok |
10744// | negative, unaligned offset | ok | ok |
10745// +----------------------------+------+------+
10746//
10747// This function ignores the addressing mode, so if an offset cannot be used in
10748// one addressing mode, it is considered illegal.
10749bool SIInstrInfo::isLegalFLATOffset(int64_t Offset, unsigned AddrSpace,
10750 AMDGPU::FlatAddrSpace FlatVariant) const {
10751 // TODO: Should 0 be special cased?
10752 if (!ST.hasFlatInstOffsets())
10753 return false;
10754
10756 if (ST.hasFlatSegmentOffsetBug() && FlatVariant == FlatAddrSpace::FLAT &&
10757 (AddrSpace == AMDGPUAS::FLAT_ADDRESS ||
10758 AddrSpace == AMDGPUAS::GLOBAL_ADDRESS))
10759 return false;
10760
10761 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10762 FlatVariant == FlatAddrSpace::FlatScratch && Offset < 0 &&
10763 (Offset % 4) != 0) {
10764 return false;
10765 }
10766
10767 bool AllowNegative = allowNegativeFlatOffset(FlatVariant);
10768 unsigned N = AMDGPU::getNumFlatOffsetBits(ST);
10769 return isIntN(N, Offset) && (AllowNegative || Offset >= 0);
10770}
10771
10772// See comment on SIInstrInfo::isLegalFLATOffset for what is legal and what not.
10773std::pair<int64_t, int64_t>
10774SIInstrInfo::splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace,
10775 AMDGPU::FlatAddrSpace FlatVariant) const {
10776 int64_t RemainderOffset = COffsetVal;
10777 int64_t ImmField = 0;
10778
10779 bool AllowNegative = allowNegativeFlatOffset(FlatVariant);
10780 const unsigned NumBits = AMDGPU::getNumFlatOffsetBits(ST) - 1;
10781
10782 if (AllowNegative) {
10783 // Use signed division by a power of two to truncate towards 0.
10784 int64_t D = 1LL << NumBits;
10785 RemainderOffset = (COffsetVal / D) * D;
10786 ImmField = COffsetVal - RemainderOffset;
10787
10788 if (ST.hasNegativeUnalignedScratchOffsetBug() &&
10789 FlatVariant == AMDGPU::FlatAddrSpace::FlatScratch && ImmField < 0 &&
10790 (ImmField % 4) != 0) {
10791 // Make ImmField a multiple of 4
10792 RemainderOffset += ImmField % 4;
10793 ImmField -= ImmField % 4;
10794 }
10795 } else if (COffsetVal >= 0) {
10796 ImmField = COffsetVal & maskTrailingOnes<uint64_t>(NumBits);
10797 RemainderOffset = COffsetVal - ImmField;
10798 }
10799
10800 assert(isLegalFLATOffset(ImmField, AddrSpace, FlatVariant));
10801 assert(RemainderOffset + ImmField == COffsetVal);
10802 return {ImmField, RemainderOffset};
10803}
10804
10806 AMDGPU::FlatAddrSpace FlatVariant) const {
10807 if (ST.hasNegativeScratchOffsetBug() &&
10809 return false;
10810
10811 return FlatVariant != AMDGPU::FlatAddrSpace::FLAT || AMDGPU::isGFX12Plus(ST);
10812}
10813
10814static unsigned subtargetEncodingFamily(const GCNSubtarget &ST) {
10815 switch (ST.getGeneration()) {
10816 default:
10817 break;
10820 return SIEncodingFamily::SI;
10822 // The GFX80 encoding family only contains buffer instructions with unpacked
10823 // D16 data; pseudoToMCOpcode falls back on VI for everything else.
10824 // TODO: remove this when we discard GFX80 encoding.
10825 return ST.hasUnpackedD16VMem() ? SIEncodingFamily::GFX80
10828 return SIEncodingFamily::VI;
10832 return ST.hasGFX11_7Insts() ? SIEncodingFamily::GFX1170
10835 return ST.hasGFX1250Insts() ? SIEncodingFamily::GFX1250
10839 }
10840 llvm_unreachable("Unknown subtarget generation!");
10841}
10842
10843bool SIInstrInfo::isAsmOnlyOpcode(int MCOp) const {
10844 switch(MCOp) {
10845 // These opcodes use indirect register addressing so
10846 // they need special handling by codegen (currently missing).
10847 // Therefore it is too risky to allow these opcodes
10848 // to be selected by dpp combiner or sdwa peepholer.
10849 case AMDGPU::V_MOVRELS_B32_dpp_gfx10:
10850 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
10851 case AMDGPU::V_MOVRELD_B32_dpp_gfx10:
10852 case AMDGPU::V_MOVRELD_B32_sdwa_gfx10:
10853 case AMDGPU::V_MOVRELSD_B32_dpp_gfx10:
10854 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
10855 case AMDGPU::V_MOVRELSD_2_B32_dpp_gfx10:
10856 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
10857 return true;
10858 default:
10859 return false;
10860 }
10861}
10862
10863#define GENERATE_RENAMED_GFX9_CASES(OPCODE) \
10864 case OPCODE##_dpp: \
10865 case OPCODE##_e32: \
10866 case OPCODE##_e64: \
10867 case OPCODE##_e64_dpp: \
10868 case OPCODE##_sdwa:
10869
10870static bool isRenamedInGFX9(int Opcode) {
10871 switch (Opcode) {
10872 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADDC_U32)
10873 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADD_CO_U32)
10874 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_ADD_U32)
10875 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBBREV_U32)
10876 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBB_U32)
10877 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBREV_CO_U32)
10878 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUBREV_U32)
10879 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUB_CO_U32)
10880 GENERATE_RENAMED_GFX9_CASES(AMDGPU::V_SUB_U32)
10881 //
10882 case AMDGPU::V_DIV_FIXUP_F16_gfx9_e64:
10883 case AMDGPU::V_DIV_FIXUP_F16_gfx9_fake16_e64:
10884 case AMDGPU::V_FMA_F16_gfx9_e64:
10885 case AMDGPU::V_FMA_F16_gfx9_fake16_e64:
10886 case AMDGPU::V_INTERP_P2_F16:
10887 case AMDGPU::V_MAD_F16_e64:
10888 case AMDGPU::V_MAD_U16_e64:
10889 case AMDGPU::V_MAD_I16_e64:
10890 return true;
10891 default:
10892 return false;
10893 }
10894}
10895
10896int SIInstrInfo::pseudoToMCOpcode(int Opcode) const {
10897 assert(Opcode == (int)SIInstrInfo::getNonSoftWaitcntOpcode(Opcode) &&
10898 "SIInsertWaitcnts should have promoted soft waitcnt instructions!");
10899
10900 unsigned Gen = subtargetEncodingFamily(ST);
10901
10902 if (ST.getGeneration() == AMDGPUSubtarget::GFX9 && isRenamedInGFX9(Opcode))
10904
10905 if (SIInstrFlags::isSDWA(get(Opcode))) {
10906 switch (ST.getGeneration()) {
10907 default:
10909 break;
10912 break;
10915 break;
10916 }
10917 }
10918
10919 if (isMAI(Opcode)) {
10920 int MFMAOp = AMDGPU::getMFMAEarlyClobberOp(Opcode);
10921 if (MFMAOp != -1)
10922 Opcode = MFMAOp;
10923 }
10924
10925 int32_t MCOp = AMDGPU::getMCOpcode(Opcode, Gen);
10926
10927 // Only buffer instructions with unpacked D16 data have a GFX80 encoding.
10928 // Anything else on such a subtarget uses the plain VI encoding.
10929 // TODO: remove this when we discard GFX80 encoding.
10930 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && Gen == SIEncodingFamily::GFX80)
10932
10933 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX11_7Insts())
10935
10936 if (MCOp == AMDGPU::INSTRUCTION_LIST_END && ST.hasGFX1250Insts())
10938
10939 // -1 means that Opcode is already a native instruction.
10940 if (MCOp == -1)
10941 return Opcode;
10942
10943 if (ST.hasGFX90AInsts()) {
10944 uint32_t NMCOp = AMDGPU::INSTRUCTION_LIST_END;
10945 if (ST.hasGFX940Insts())
10947 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10949 if (NMCOp == AMDGPU::INSTRUCTION_LIST_END)
10951 if (NMCOp != AMDGPU::INSTRUCTION_LIST_END)
10952 MCOp = NMCOp;
10953 }
10954
10955 // INSTRUCTION_LIST_END means that Opcode is a pseudo instruction that has no
10956 // encoding in the given subtarget generation.
10957 if (MCOp == AMDGPU::INSTRUCTION_LIST_END)
10958 return -1;
10959
10960 if (isAsmOnlyOpcode(MCOp))
10961 return -1;
10962
10963 return MCOp;
10964}
10965
10966static
10968 assert(RegOpnd.isReg());
10969 return RegOpnd.isUndef() ? TargetInstrInfo::RegSubRegPair() :
10970 getRegSubRegPair(RegOpnd);
10971}
10972
10975 assert(MI.isRegSequence());
10976 for (unsigned I = 0, E = (MI.getNumOperands() - 1)/ 2; I < E; ++I)
10977 if (MI.getOperand(1 + 2 * I + 1).getImm() == SubReg) {
10978 auto &RegOp = MI.getOperand(1 + 2 * I);
10979 return getRegOrUndef(RegOp);
10980 }
10982}
10983
10984// Try to find the definition of reg:subreg in subreg-manipulation pseudos
10985// Following a subreg of reg:subreg isn't supported
10988 if (!RSR.SubReg)
10989 return false;
10990 switch (MI.getOpcode()) {
10991 default: break;
10992 case AMDGPU::REG_SEQUENCE:
10993 RSR = getRegSequenceSubReg(MI, RSR.SubReg);
10994 return true;
10995 // EXTRACT_SUBREG ins't supported as this would follow a subreg of subreg
10996 case AMDGPU::INSERT_SUBREG:
10997 if (RSR.SubReg == (unsigned)MI.getOperand(3).getImm())
10998 // inserted the subreg we're looking for
10999 RSR = getRegOrUndef(MI.getOperand(2));
11000 else { // the subreg in the rest of the reg
11001 auto R1 = getRegOrUndef(MI.getOperand(1));
11002 if (R1.SubReg) // subreg of subreg isn't supported
11003 return false;
11004 RSR.Reg = R1.Reg;
11005 }
11006 return true;
11007 }
11008 return false;
11009}
11010
11012 const MachineRegisterInfo &MRI) {
11013 assert(MRI.isSSA());
11014 if (!P.Reg.isVirtual())
11015 return nullptr;
11016
11017 auto RSR = P;
11018 auto *DefInst = MRI.getVRegDef(RSR.Reg);
11019 while (auto *MI = DefInst) {
11020 DefInst = nullptr;
11021 switch (MI->getOpcode()) {
11022 case AMDGPU::COPY:
11023 case AMDGPU::V_MOV_B32_e32: {
11024 auto &Op1 = MI->getOperand(1);
11025 if (Op1.isReg() && Op1.getReg().isVirtual()) {
11026 if (Op1.isUndef())
11027 return nullptr;
11028 RSR = getRegSubRegPair(Op1);
11029 DefInst = MRI.getVRegDef(RSR.Reg);
11030 }
11031 break;
11032 }
11033 default:
11034 if (followSubRegDef(*MI, RSR)) {
11035 if (!RSR.Reg)
11036 return nullptr;
11037 DefInst = MRI.getVRegDef(RSR.Reg);
11038 }
11039 }
11040 if (!DefInst)
11041 return MI;
11042 }
11043 return nullptr;
11044}
11045
11047 Register VReg,
11048 const MachineInstr &DefMI,
11049 const MachineInstr &UseMI) {
11050 assert(MRI.isSSA() && "Must be run on SSA");
11051
11052 auto *TRI = MRI.getTargetRegisterInfo();
11053 auto *DefBB = DefMI.getParent();
11054
11055 // Don't bother searching between blocks, although it is possible this block
11056 // doesn't modify exec.
11057 if (UseMI.getParent() != DefBB)
11058 return true;
11059
11060 const int MaxInstScan = 20;
11061 int NumInst = 0;
11062
11063 // Stop scan at the use.
11064 auto E = UseMI.getIterator();
11065 for (auto I = std::next(DefMI.getIterator()); I != E; ++I) {
11066 if (I->isDebugInstr())
11067 continue;
11068
11069 if (++NumInst > MaxInstScan)
11070 return true;
11071
11072 if (I->modifiesRegister(AMDGPU::EXEC, TRI))
11073 return true;
11074 }
11075
11076 return false;
11077}
11078
11080 Register VReg,
11081 const MachineInstr &DefMI) {
11082 assert(MRI.isSSA() && "Must be run on SSA");
11083
11084 auto *TRI = MRI.getTargetRegisterInfo();
11085 auto *DefBB = DefMI.getParent();
11086
11087 const int MaxUseScan = 10;
11088 int NumUse = 0;
11089
11090 for (auto &Use : MRI.use_nodbg_operands(VReg)) {
11091 auto &UseInst = *Use.getParent();
11092 // Don't bother searching between blocks, although it is possible this block
11093 // doesn't modify exec.
11094 if (UseInst.getParent() != DefBB || UseInst.isPHI())
11095 return true;
11096
11097 if (++NumUse > MaxUseScan)
11098 return true;
11099 }
11100
11101 if (NumUse == 0)
11102 return false;
11103
11104 const int MaxInstScan = 20;
11105 int NumInst = 0;
11106
11107 // Stop scan when we have seen all the uses.
11108 for (auto I = std::next(DefMI.getIterator()); ; ++I) {
11109 assert(I != DefBB->end());
11110
11111 if (I->isDebugInstr())
11112 continue;
11113
11114 if (++NumInst > MaxInstScan)
11115 return true;
11116
11117 for (const MachineOperand &Op : I->operands()) {
11118 // We don't check reg masks here as they're used only on calls:
11119 // 1. EXEC is only considered const within one BB
11120 // 2. Call should be a terminator instruction if present in a BB
11121
11122 if (!Op.isReg())
11123 continue;
11124
11125 Register Reg = Op.getReg();
11126 if (Op.isUse()) {
11127 if (Reg == VReg && --NumUse == 0)
11128 return false;
11129 } else if (TRI->regsOverlap(Reg, AMDGPU::EXEC))
11130 return true;
11131 }
11132 }
11133}
11134
11137 const DebugLoc &DL, Register Src, Register Dst) const {
11138 auto Cur = MBB.begin();
11139 if (Cur != MBB.end())
11140 do {
11141 if (!Cur->isPHI() && Cur->readsRegister(Dst, /*TRI=*/nullptr))
11142 return BuildMI(MBB, Cur, DL, get(TargetOpcode::COPY), Dst).addReg(Src);
11143 ++Cur;
11144 } while (Cur != MBB.end() && Cur != LastPHIIt);
11145
11146 return TargetInstrInfo::createPHIDestinationCopy(MBB, LastPHIIt, DL, Src,
11147 Dst);
11148}
11149
11152 const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const {
11153 if (InsPt != MBB.end() &&
11154 (InsPt->getOpcode() == AMDGPU::SI_IF ||
11155 InsPt->getOpcode() == AMDGPU::SI_ELSE ||
11156 InsPt->getOpcode() == AMDGPU::SI_IF_BREAK) &&
11157 InsPt->definesRegister(Src, /*TRI=*/nullptr)) {
11158 InsPt++;
11159 return BuildMI(MBB, InsPt, DL,
11160 get(AMDGPU::LaneMaskConstants::get(ST).MovTermOpc), Dst)
11161 .addReg(Src, {}, SrcSubReg)
11162 .addReg(AMDGPU::EXEC, RegState::Implicit);
11163 }
11164 return TargetInstrInfo::createPHISourceCopy(MBB, InsPt, DL, Src, SrcSubReg,
11165 Dst);
11166}
11167
11168bool llvm::SIInstrInfo::isWave32() const { return ST.isWave32(); }
11169
11171 const MachineInstr &SecondMI) const {
11172 for (const auto &Use : SecondMI.all_uses()) {
11173 if (Use.isReg() && FirstMI.modifiesRegister(Use.getReg(), &RI))
11174 return true;
11175 }
11176 return false;
11177}
11178
11179/// If OpX is multicycle, anti-dependencies are not allowed.
11180/// isDPMACCInstruction was not designed for VOPD, but it is fit for the
11181/// purpose.
11183 const MachineInstr &OpX) const {
11185}
11186
11189 ArrayRef<unsigned> Ops, int FrameIndex,
11190 MachineInstr *&CopyMI, LiveIntervals *LIS,
11191 VirtRegMap *VRM) const {
11192 // This is a bit of a hack (copied from AArch64). Consider this instruction:
11193 //
11194 // %0:sreg_32 = COPY $m0
11195 //
11196 // We explicitly chose SReg_32 for the virtual register so such a copy might
11197 // be eliminated by RegisterCoalescer. However, that may not be possible, and
11198 // %0 may even spill. We can't spill $m0 normally (it would require copying to
11199 // a numbered SGPR anyway), and since it is in the SReg_32 register class,
11200 // TargetInstrInfo::foldMemoryOperand() is going to try.
11201 // A similar issue also exists with spilling and reloading $exec registers.
11202 //
11203 // To prevent that, constrain the %0 register class here.
11204 if (isFullCopyInstr(MI)) {
11205 Register DstReg = MI.getOperand(0).getReg();
11206 Register SrcReg = MI.getOperand(1).getReg();
11207 if ((DstReg.isVirtual() || SrcReg.isVirtual()) &&
11208 (DstReg.isVirtual() != SrcReg.isVirtual())) {
11209 MachineRegisterInfo &MRI = MF.getRegInfo();
11210 Register VirtReg = DstReg.isVirtual() ? DstReg : SrcReg;
11211 const TargetRegisterClass *RC = MRI.getRegClass(VirtReg);
11212 if (RC->hasSuperClassEq(&AMDGPU::SReg_32RegClass)) {
11213 MRI.constrainRegClass(VirtReg, &AMDGPU::SReg_32_XM0_XEXECRegClass);
11214 return nullptr;
11215 }
11216 if (RC->hasSuperClassEq(&AMDGPU::SReg_64RegClass)) {
11217 MRI.constrainRegClass(VirtReg, &AMDGPU::SReg_64_XEXECRegClass);
11218 return nullptr;
11219 }
11220 }
11221 }
11222
11223 return nullptr;
11224}
11225
11227 const MachineInstr &MI,
11228 unsigned *PredCost) const {
11229 if (MI.isBundle()) {
11231 MachineBasicBlock::const_instr_iterator E(MI.getParent()->instr_end());
11232 unsigned Lat = 0, Count = 0;
11233 for (++I; I != E && I->isBundledWithPred(); ++I) {
11234 ++Count;
11235 Lat = std::max(Lat, SchedModel.computeInstrLatency(&*I));
11236 }
11237 return Lat + Count - 1;
11238 }
11239
11240 return SchedModel.computeInstrLatency(&MI);
11241}
11242
11244 if (!ST.hasGFX1250VALUBlockingCycles())
11245 return 0;
11247}
11248
11249unsigned
11251 if (const auto *Entry = AMDGPU::getGFX1250BlockingCyclesInfo(MI.getOpcode()))
11252 return Entry->GFX1250BlockingCycles;
11253 return 0;
11254}
11255
11256const MachineOperand &
11258 if (const MachineOperand *CallAddrOp =
11259 getNamedOperand(MI, AMDGPU::OpName::src0))
11260 return *CallAddrOp;
11262}
11263
11266 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
11267 unsigned Opcode = MI.getOpcode();
11268
11269 auto HandleAddrSpaceCast = [this, &MRI](const MachineInstr &MI) {
11270 Register Dst = MI.getOperand(0).getReg();
11271 Register Src = MI.getOperand(1).getReg();
11272 LLT DstTy = MRI.getType(Dst);
11273 LLT SrcTy = MRI.getType(Src);
11274 unsigned DstAS = DstTy.getAddressSpace();
11275 unsigned SrcAS = SrcTy.getAddressSpace();
11276 return SrcAS == AMDGPUAS::PRIVATE_ADDRESS &&
11277 DstAS == AMDGPUAS::FLAT_ADDRESS &&
11278 ST.hasGloballyAddressableScratch()
11281 };
11282
11283 // If the target supports globally addressable scratch, the mapping from
11284 // scratch memory to the flat aperture changes therefore an address space cast
11285 // is no longer uniform.
11286 if (Opcode == TargetOpcode::G_ADDRSPACE_CAST)
11287 return HandleAddrSpaceCast(MI);
11288
11289 if (auto *GI = dyn_cast<GIntrinsic>(&MI)) {
11290 auto IID = GI->getIntrinsicID();
11295
11296 switch (IID) {
11297 case Intrinsic::amdgcn_if:
11298 case Intrinsic::amdgcn_else:
11299 // FIXME: Uniform if second result
11300 break;
11301 }
11302
11304 }
11305
11306 // Loads from the private and flat address spaces are divergent, because
11307 // threads can execute the load instruction with the same inputs and get
11308 // different results.
11309 //
11310 // All other loads are not divergent, because if threads issue loads with the
11311 // same arguments, they will always get the same result.
11312 if (Opcode == AMDGPU::G_LOAD || Opcode == AMDGPU::G_ZEXTLOAD ||
11313 Opcode == AMDGPU::G_SEXTLOAD) {
11314 if (MI.memoperands_empty())
11315 return ValueUniformity::NeverUniform; // conservative assumption
11316
11317 if (llvm::any_of(MI.memoperands(), [](const MachineMemOperand *mmo) {
11318 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11319 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11320 })) {
11321 // At least one MMO in a non-global address space.
11323 }
11325 }
11326
11327 if (SIInstrInfo::isGenericAtomicRMWOpcode(Opcode) ||
11328 Opcode == AMDGPU::G_ATOMIC_CMPXCHG ||
11329 Opcode == AMDGPU::G_ATOMIC_CMPXCHG_WITH_SUCCESS ||
11330 AMDGPU::isGenericAtomic(Opcode)) {
11332 }
11333
11334 // Result is computed from uniform SP and uniform wave-wide max size.
11335 if (Opcode == TargetOpcode::G_DYN_STACKALLOC)
11337
11338 if (Opcode == AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_SETUP)
11340
11342}
11343
11345 if (!Formatter)
11346 Formatter = std::make_unique<AMDGPUMIRFormatter>(ST);
11347 return Formatter.get();
11348}
11349
11351
11352 if (isNeverUniform(MI))
11354
11355 unsigned opcode = MI.getOpcode();
11356 if (opcode == AMDGPU::V_READLANE_B32 ||
11357 opcode == AMDGPU::V_READFIRSTLANE_B32 ||
11358 opcode == AMDGPU::SI_RESTORE_S32_FROM_VGPR)
11360
11361 // If any of defs is divergent, report as NeverUniform. isUniformReg will
11362 // calculate in more detail for each def from its reg class, if available.
11363 if (MI.isInlineAsm()) {
11364 for (const MachineOperand &MO : MI.operands()) {
11365 if (!MO.isReg() || !MO.isDef())
11366 continue;
11367 const TargetRegisterClass *RC =
11368 MI.getRegClassConstraint(MO.getOperandNo(), this, &RI);
11369 if (!RC || !RI.isSGPRClass(RC))
11371 }
11372 }
11373
11374 if (isCopyInstr(MI)) {
11375 const MachineOperand &srcOp = MI.getOperand(1);
11376 if (srcOp.isReg() && srcOp.getReg().isPhysical()) {
11377 const TargetRegisterClass *regClass =
11378 RI.getPhysRegBaseClass(srcOp.getReg());
11379 return RI.isSGPRClass(regClass) ? ValueUniformity::AlwaysUniform
11381 }
11383 }
11384
11385 // GMIR handling
11386 if (MI.isPreISelOpcode())
11388
11389 // Atomics are divergent because they are executed sequentially: when an
11390 // atomic operation refers to the same address in each thread, then each
11391 // thread after the first sees the value written by the previous thread as
11392 // original value.
11393
11394 if (isAtomic(MI))
11396
11397 // Loads from the private and flat address spaces are divergent, because
11398 // threads can execute the load instruction with the same inputs and get
11399 // different results.
11400 if (isFLAT(MI) && MI.mayLoad()) {
11401 if (MI.memoperands_empty())
11402 return ValueUniformity::NeverUniform; // conservative assumption
11403
11404 if (llvm::any_of(MI.memoperands(), [](const MachineMemOperand *mmo) {
11405 return mmo->getAddrSpace() == AMDGPUAS::PRIVATE_ADDRESS ||
11406 mmo->getAddrSpace() == AMDGPUAS::FLAT_ADDRESS;
11407 })) {
11408 // At least one MMO in a non-global address space.
11410 }
11411
11413 }
11414
11415 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
11416 const AMDGPURegisterBankInfo *RBI = ST.getRegBankInfo();
11417
11418 // FIXME: It's conceptually broken to report this for an instruction, and not
11419 // a specific def operand. For inline asm in particular, there could be mixed
11420 // uniform and divergent results.
11421 for (unsigned I = 0, E = MI.getNumOperands(); I != E; ++I) {
11422 const MachineOperand &SrcOp = MI.getOperand(I);
11423 if (!SrcOp.isReg())
11424 continue;
11425
11426 Register Reg = SrcOp.getReg();
11427 if (!Reg || !SrcOp.readsReg())
11428 continue;
11429
11430 // If RegBank is null, this is unassigned or an unallocatable special
11431 // register, which are all scalars.
11432 const RegisterBank *RegBank = RBI->getRegBank(Reg, MRI, RI);
11433 if (RegBank && RegBank->getID() != AMDGPU::SGPRRegBankID)
11435 }
11436
11437 // TODO: Uniformity check condtions above can be rearranged for more
11438 // redability
11439
11440 // TODO: amdgcn.{ballot, [if]cmp} should be AlwaysUniform, but they are
11441 // currently turned into no-op COPYs by SelectionDAG ISel and are
11442 // therefore no longer recognizable.
11443
11445}
11446
11448 switch (MF.getFunction().getCallingConv()) {
11450 return 1;
11452 return 2;
11454 return 3;
11458 const Function &F = MF.getFunction();
11459 F.getContext().diagnose(DiagnosticInfoUnsupported(
11460 F, "ds_ordered_count unsupported for this calling conv"));
11461 [[fallthrough]];
11462 }
11465 case CallingConv::C:
11466 case CallingConv::Fast:
11467 default:
11468 // Assume other calling conventions are various compute callable functions
11469 return 0;
11470 }
11471}
11472
11474 Register &SrcReg2, int64_t &CmpMask,
11475 int64_t &CmpValue) const {
11476 if (!MI.getOperand(0).isReg() || MI.getOperand(0).getSubReg())
11477 return false;
11478
11479 switch (MI.getOpcode()) {
11480 default:
11481 break;
11482 case AMDGPU::S_CMP_EQ_U32:
11483 case AMDGPU::S_CMP_EQ_I32:
11484 case AMDGPU::S_CMP_LG_U32:
11485 case AMDGPU::S_CMP_LG_I32:
11486 case AMDGPU::S_CMP_LT_U32:
11487 case AMDGPU::S_CMP_LT_I32:
11488 case AMDGPU::S_CMP_GT_U32:
11489 case AMDGPU::S_CMP_GT_I32:
11490 case AMDGPU::S_CMP_LE_U32:
11491 case AMDGPU::S_CMP_LE_I32:
11492 case AMDGPU::S_CMP_GE_U32:
11493 case AMDGPU::S_CMP_GE_I32:
11494 case AMDGPU::S_CMP_EQ_U64:
11495 case AMDGPU::S_CMP_LG_U64:
11496 SrcReg = MI.getOperand(0).getReg();
11497 if (MI.getOperand(1).isReg()) {
11498 if (MI.getOperand(1).getSubReg())
11499 return false;
11500 SrcReg2 = MI.getOperand(1).getReg();
11501 CmpValue = 0;
11502 } else if (MI.getOperand(1).isImm()) {
11503 SrcReg2 = Register();
11504 CmpValue = MI.getOperand(1).getImm();
11505 } else {
11506 return false;
11507 }
11508 CmpMask = ~0;
11509 return true;
11510 case AMDGPU::S_CMPK_EQ_U32:
11511 case AMDGPU::S_CMPK_EQ_I32:
11512 case AMDGPU::S_CMPK_LG_U32:
11513 case AMDGPU::S_CMPK_LG_I32:
11514 case AMDGPU::S_CMPK_LT_U32:
11515 case AMDGPU::S_CMPK_LT_I32:
11516 case AMDGPU::S_CMPK_GT_U32:
11517 case AMDGPU::S_CMPK_GT_I32:
11518 case AMDGPU::S_CMPK_LE_U32:
11519 case AMDGPU::S_CMPK_LE_I32:
11520 case AMDGPU::S_CMPK_GE_U32:
11521 case AMDGPU::S_CMPK_GE_I32:
11522 SrcReg = MI.getOperand(0).getReg();
11523 SrcReg2 = Register();
11524 CmpValue = MI.getOperand(1).getImm();
11525 CmpMask = ~0;
11526 return true;
11527 }
11528
11529 return false;
11530}
11531
11533 for (MachineBasicBlock *S : MBB->successors()) {
11534 if (S->isLiveIn(AMDGPU::SCC))
11535 return false;
11536 }
11537 return true;
11538}
11539
11540// Invert all uses of SCC following SCCDef because SCCDef may be deleted and
11541// (incoming SCC) = !(SCC defined by SCCDef).
11542// Return true if all uses can be re-written, false otherwise.
11543bool SIInstrInfo::invertSCCUse(MachineInstr *SCCDef) const {
11544 MachineBasicBlock *MBB = SCCDef->getParent();
11545 SmallVector<MachineInstr *> InvertInstr;
11546 bool SCCIsDead = false;
11547
11548 // Scan instructions for SCC uses that need to be inverted until SCC is dead.
11549 constexpr unsigned ScanLimit = 12;
11550 unsigned Count = 0;
11551 for (MachineInstr &MI : instructionsWithoutDebug(
11552 std::next(MachineBasicBlock::iterator(SCCDef)), MBB->end())) {
11553 if (++Count > ScanLimit)
11554 return false;
11555 if (MI.readsRegister(AMDGPU::SCC, &RI)) {
11556 if (MI.getOpcode() == AMDGPU::S_CSELECT_B32 ||
11557 MI.getOpcode() == AMDGPU::S_CSELECT_B64 ||
11558 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11559 MI.getOpcode() == AMDGPU::S_CBRANCH_SCC1)
11560 InvertInstr.push_back(&MI);
11561 else
11562 return false;
11563 }
11564 if (MI.definesRegister(AMDGPU::SCC, &RI)) {
11565 SCCIsDead = true;
11566 break;
11567 }
11568 }
11569 if (!SCCIsDead && isSCCDeadOnExit(MBB))
11570 SCCIsDead = true;
11571
11572 // SCC may have more uses. Can't invert all of them.
11573 if (!SCCIsDead)
11574 return false;
11575
11576 // Invert uses
11577 for (MachineInstr *MI : InvertInstr) {
11578 if (MI->getOpcode() == AMDGPU::S_CSELECT_B32 ||
11579 MI->getOpcode() == AMDGPU::S_CSELECT_B64) {
11580 swapOperands(*MI);
11581 } else if (MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0 ||
11582 MI->getOpcode() == AMDGPU::S_CBRANCH_SCC1) {
11583 MI->setDesc(get(MI->getOpcode() == AMDGPU::S_CBRANCH_SCC0
11584 ? AMDGPU::S_CBRANCH_SCC1
11585 : AMDGPU::S_CBRANCH_SCC0));
11586 } else {
11587 llvm_unreachable("SCC used but no inversion handling");
11588 }
11589 }
11590 return true;
11591}
11592
11593// SCC is already valid after SCCValid.
11594// SCCRedefine will redefine SCC to the same value already available after
11595// SCCValid. If there are no intervening SCC conflicts delete SCCRedefine and
11596// update kill/dead flags if necessary.
11597bool SIInstrInfo::optimizeSCC(MachineInstr *SCCValid, MachineInstr *SCCRedefine,
11598 bool NeedInversion) const {
11599 MachineInstr *KillsSCC = nullptr;
11600 if (SCCValid->getParent() != SCCRedefine->getParent())
11601 return false;
11602 for (MachineInstr &MI : make_range(std::next(SCCValid->getIterator()),
11603 SCCRedefine->getIterator())) {
11604 if (MI.modifiesRegister(AMDGPU::SCC, &RI))
11605 return false;
11606 if (MI.killsRegister(AMDGPU::SCC, &RI))
11607 KillsSCC = &MI;
11608 }
11609 if (NeedInversion && !invertSCCUse(SCCRedefine))
11610 return false;
11611 if (MachineOperand *SccDef =
11612 SCCValid->findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr))
11613 SccDef->setIsDead(false);
11614 if (KillsSCC)
11615 KillsSCC->clearRegisterKills(AMDGPU::SCC, /*TRI=*/nullptr);
11616 SCCRedefine->eraseFromParent();
11617 return true;
11618}
11619
11620/// If \p Sel is an S_CSELECT* of two different constants A and B, return them,
11621/// truncated to the width of the select.
11622static std::optional<std::pair<int64_t, int64_t>>
11624 const MachineInstr &Sel) {
11625 unsigned Opc = Sel.getOpcode();
11626 if (Opc != AMDGPU::S_CSELECT_B32 && Opc != AMDGPU::S_CSELECT_B64)
11627 return {};
11628 std::optional<int64_t> A =
11629 TII.getImmOrMaterializedImm(MRI, Sel.getOperand(1));
11630 if (!A)
11631 return {};
11632 std::optional<int64_t> B =
11633 TII.getImmOrMaterializedImm(MRI, Sel.getOperand(2));
11634 if (!B)
11635 return {};
11636 if (Opc == AMDGPU::S_CSELECT_B32) {
11637 A = Lo_32(*A);
11638 B = Lo_32(*B);
11639 }
11640 if (*A == *B)
11641 return {};
11642 return std::pair(*A, *B);
11643}
11644
11645static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion,
11646 unsigned &NewDefOpc) {
11647 // S_ADD_U32 X, 1 sets SCC on carryout which can only happen if result==0.
11648 // S_ADD_I32 X, 1 can be converted to S_ADD_U32 X, 1 if SCC is dead.
11649 if (Def.getOpcode() != AMDGPU::S_ADD_I32 &&
11650 Def.getOpcode() != AMDGPU::S_ADD_U32)
11651 return false;
11652 const MachineOperand &AddSrc1 = Def.getOperand(1);
11653 const MachineOperand &AddSrc2 = Def.getOperand(2);
11654 const MachineRegisterInfo &MRI = Def.getMF()->getRegInfo();
11655 const SIInstrInfo *TII = static_cast<const SIInstrInfo *>(
11656 Def.getMF()->getSubtarget().getInstrInfo());
11657
11658 auto Imm1 = TII->getImmOrMaterializedImm(MRI, AddSrc1);
11659 auto Imm2 = TII->getImmOrMaterializedImm(MRI, AddSrc2);
11660 if ((!Imm1 || *Imm1 != 1) && (!Imm2 || *Imm2 != 1))
11661 return false;
11662
11663 if (Def.getOpcode() == AMDGPU::S_ADD_I32) {
11664 const MachineOperand *SccDef =
11665 Def.findRegisterDefOperand(AMDGPU::SCC, /*TRI=*/nullptr);
11666 if (!SccDef->isDead())
11667 return false;
11668 NewDefOpc = AMDGPU::S_ADD_U32;
11669 }
11670 NeedInversion = !NeedInversion;
11671 return true;
11672}
11673
11675 Register SrcReg2, int64_t CmpMask,
11676 int64_t CmpValue,
11677 const MachineRegisterInfo *MRI) const {
11678 if (!SrcReg || SrcReg.isPhysical())
11679 return false;
11680
11681 if (SrcReg2) {
11682 auto ImmOpt = getImmOrMaterializedImm(*MRI, SrcReg2);
11683 if (!ImmOpt)
11684 return false;
11685 CmpValue = *ImmOpt;
11686 }
11687
11688 const auto optimizeCmpSelect = [&CmpInstr, SrcReg, CmpValue, MRI,
11689 this](bool NeedInversion) -> bool {
11690 MachineInstr *Def = MRI->getVRegDef(SrcReg);
11691 if (!Def)
11692 return false;
11693
11694 unsigned NewDefOpc = Def->getOpcode();
11695 if (auto Consts = getSelectConstants(*this, *MRI, *Def)) {
11696 // sX = S_CSELECT* A, B with A != B, so sX == A exactly when SCC was set.
11697 // Comparing sX with A or B recomputes SCC or its inverse:
11698 //
11699 // s_cmp_eq_* sX, A => SCC s_cmp_lg_* sX, A => !SCC
11700 // s_cmp_eq_* sX, B => !SCC s_cmp_lg_* sX, B => SCC
11701 auto [A, B] = *Consts;
11702 int64_t C = Def->getOpcode() == AMDGPU::S_CSELECT_B32 ? Lo_32(CmpValue)
11703 : CmpValue;
11704 if (C == A)
11705 NeedInversion = !NeedInversion;
11706 else if (C != B)
11707 return false;
11708 } else {
11709 // For S_OP that set SCC = DST!=0, do the transformation
11710 //
11711 // s_cmp_[lg|eq]_* (S_OP ...), 0 => (S_OP ...)
11712 //
11713 // For (S_OP ...) that set SCC = DST==0, invert NeedInversion and
11714 // do the transformation:
11715 //
11716 // s_cmp_[lg|eq]_* (S_OP ...), 0 => (S_OP ...)
11717 if (CmpValue != 0 ||
11718 (!setsSCCIfResultIsNonZero(*Def) &&
11719 !setsSCCIfResultIsZero(*Def, NeedInversion, NewDefOpc)))
11720 return false;
11721 }
11722
11723 if (!optimizeSCC(Def, &CmpInstr, NeedInversion))
11724 return false;
11725
11726 if (NewDefOpc != Def->getOpcode())
11727 Def->setDesc(get(NewDefOpc));
11728
11729 // If s_or_b32 result, sY, is unused (i.e. it is effectively a 64-bit
11730 // s_cmp_lg of a register pair) and the inputs are the hi and lo-halves of a
11731 // 64-bit select then delete s_or_b32 in the sequence:
11732 // sX = s_cselect_b64 A, B (A != B, one of them 0)
11733 // sLo = copy sX.sub0
11734 // sHi = copy sX.sub1
11735 // sY = s_or_b32 sLo, sHi
11736 if (Def->getOpcode() == AMDGPU::S_OR_B32 &&
11737 MRI->use_nodbg_empty(Def->getOperand(0).getReg())) {
11738 const MachineOperand &OrOpnd1 = Def->getOperand(1);
11739 const MachineOperand &OrOpnd2 = Def->getOperand(2);
11740 if (OrOpnd1.isReg() && OrOpnd2.isReg()) {
11741 MachineInstr *Def1 = MRI->getVRegDef(OrOpnd1.getReg());
11742 MachineInstr *Def2 = MRI->getVRegDef(OrOpnd2.getReg());
11743 if (Def1 && Def1->getOpcode() == AMDGPU::COPY && Def2 &&
11744 Def2->getOpcode() == AMDGPU::COPY && Def1->getOperand(1).isReg() &&
11745 Def2->getOperand(1).isReg() &&
11746 Def1->getOperand(1).getSubReg() == AMDGPU::sub0 &&
11747 Def2->getOperand(1).getSubReg() == AMDGPU::sub1 &&
11748 Def1->getOperand(1).getReg() == Def2->getOperand(1).getReg()) {
11749 if (MachineInstr *Select =
11750 MRI->getVRegDef(Def1->getOperand(1).getReg())) {
11751 if (auto Consts = getSelectConstants(*this, *MRI, *Select)) {
11752 auto [A, B] = *Consts;
11753 if (A == 0 || B == 0)
11754 optimizeSCC(Select, Def, /*NeedInversion=*/A == 0);
11755 }
11756 }
11757 }
11758 }
11759 }
11760 return true;
11761 };
11762
11763 const auto optimizeCmpAnd = [&CmpInstr, SrcReg, CmpValue, MRI,
11764 this](int64_t ExpectedValue, unsigned SrcSize,
11765 bool IsReversible, bool IsSigned) -> bool {
11766 // s_cmp_eq_u32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11767 // s_cmp_eq_i32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11768 // s_cmp_ge_u32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11769 // s_cmp_ge_i32 (s_and_b32 $src, 1 << n), 1 << n => s_and_b32 $src, 1 << n
11770 // s_cmp_eq_u64 (s_and_b64 $src, 1 << n), 1 << n => s_and_b64 $src, 1 << n
11771 // s_cmp_lg_u32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11772 // s_cmp_lg_i32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11773 // s_cmp_gt_u32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11774 // s_cmp_gt_i32 (s_and_b32 $src, 1 << n), 0 => s_and_b32 $src, 1 << n
11775 // s_cmp_lg_u64 (s_and_b64 $src, 1 << n), 0 => s_and_b64 $src, 1 << n
11776 //
11777 // Signed ge/gt are not used for the sign bit.
11778 //
11779 // If result of the AND is unused except in the compare:
11780 // s_and_b(32|64) $src, 1 << n => s_bitcmp1_b(32|64) $src, n
11781 //
11782 // s_cmp_eq_u32 (s_and_b32 $src, 1 << n), 0 => s_bitcmp0_b32 $src, n
11783 // s_cmp_eq_i32 (s_and_b32 $src, 1 << n), 0 => s_bitcmp0_b32 $src, n
11784 // s_cmp_eq_u64 (s_and_b64 $src, 1 << n), 0 => s_bitcmp0_b64 $src, n
11785 // s_cmp_lg_u32 (s_and_b32 $src, 1 << n), 1 << n => s_bitcmp0_b32 $src, n
11786 // s_cmp_lg_i32 (s_and_b32 $src, 1 << n), 1 << n => s_bitcmp0_b32 $src, n
11787 // s_cmp_lg_u64 (s_and_b64 $src, 1 << n), 1 << n => s_bitcmp0_b64 $src, n
11788
11789 MachineInstr *Def = MRI->getVRegDef(SrcReg);
11790 if (!Def)
11791 return false;
11792
11793 if (Def->getOpcode() != AMDGPU::S_AND_B32 &&
11794 Def->getOpcode() != AMDGPU::S_AND_B64)
11795 return false;
11796
11797 int64_t Mask;
11798 const auto isMask = [&Mask, SrcSize, MRI,
11799 this](const MachineOperand *MO) -> bool {
11800 auto ImmOpt = this->getImmOrMaterializedImm(*MRI, *MO);
11801 if (!ImmOpt)
11802 return false;
11803 Mask = *ImmOpt;
11804 Mask &= maxUIntN(SrcSize);
11805 return isPowerOf2_64(Mask);
11806 };
11807
11808 MachineOperand *SrcOp = &Def->getOperand(1);
11809 if (isMask(SrcOp))
11810 SrcOp = &Def->getOperand(2);
11811 else if (isMask(&Def->getOperand(2)))
11812 SrcOp = &Def->getOperand(1);
11813 else
11814 return false;
11815
11816 // A valid Mask is required to have a single bit set, hence a non-zero and
11817 // power-of-two value. This verifies that we will not do 64-bit shift below.
11818 assert(llvm::has_single_bit<uint64_t>(Mask) && "Invalid mask.");
11819 unsigned BitNo = llvm::countr_zero((uint64_t)Mask);
11820 if (IsSigned && BitNo == SrcSize - 1)
11821 return false;
11822
11823 ExpectedValue <<= BitNo;
11824
11825 bool IsReversedCC = false;
11826 if (CmpValue != ExpectedValue) {
11827 if (!IsReversible)
11828 return false;
11829 IsReversedCC = CmpValue == (ExpectedValue ^ Mask);
11830 if (!IsReversedCC)
11831 return false;
11832 }
11833
11834 Register DefReg = Def->getOperand(0).getReg();
11835 if (IsReversedCC && !MRI->hasOneNonDBGUse(DefReg))
11836 return false;
11837
11838 if (!optimizeSCC(Def, &CmpInstr, /*NeedInversion=*/false))
11839 return false;
11840
11841 if (!MRI->use_nodbg_empty(DefReg)) {
11842 assert(!IsReversedCC);
11843 return true;
11844 }
11845
11846 // Replace AND with unused result with a S_BITCMP.
11847 MachineBasicBlock *MBB = Def->getParent();
11848
11849 unsigned NewOpc = (SrcSize == 32) ? IsReversedCC ? AMDGPU::S_BITCMP0_B32
11850 : AMDGPU::S_BITCMP1_B32
11851 : IsReversedCC ? AMDGPU::S_BITCMP0_B64
11852 : AMDGPU::S_BITCMP1_B64;
11853
11854 BuildMI(*MBB, Def, Def->getDebugLoc(), get(NewOpc))
11855 .add(*SrcOp)
11856 .addImm(BitNo);
11857 Def->eraseFromParent();
11858
11859 return true;
11860 };
11861
11862 switch (CmpInstr.getOpcode()) {
11863 default:
11864 break;
11865 case AMDGPU::S_CMP_EQ_U32:
11866 case AMDGPU::S_CMP_EQ_I32:
11867 case AMDGPU::S_CMPK_EQ_U32:
11868 case AMDGPU::S_CMPK_EQ_I32:
11869 return optimizeCmpAnd(1, 32, true, false) ||
11870 optimizeCmpSelect(/*NeedInversion=*/true);
11871 case AMDGPU::S_CMP_GE_U32:
11872 case AMDGPU::S_CMPK_GE_U32:
11873 return optimizeCmpAnd(1, 32, false, false);
11874 case AMDGPU::S_CMP_GE_I32:
11875 case AMDGPU::S_CMPK_GE_I32:
11876 return optimizeCmpAnd(1, 32, false, true);
11877 case AMDGPU::S_CMP_EQ_U64:
11878 return optimizeCmpAnd(1, 64, true, false) ||
11879 optimizeCmpSelect(/*NeedInversion=*/true);
11880 case AMDGPU::S_CMP_LG_U32:
11881 case AMDGPU::S_CMP_LG_I32:
11882 case AMDGPU::S_CMPK_LG_U32:
11883 case AMDGPU::S_CMPK_LG_I32:
11884 return optimizeCmpAnd(0, 32, true, false) ||
11885 optimizeCmpSelect(/*NeedInversion=*/false);
11886 case AMDGPU::S_CMP_GT_U32:
11887 case AMDGPU::S_CMPK_GT_U32:
11888 return optimizeCmpAnd(0, 32, false, false);
11889 case AMDGPU::S_CMP_GT_I32:
11890 case AMDGPU::S_CMPK_GT_I32:
11891 return optimizeCmpAnd(0, 32, false, true);
11892 case AMDGPU::S_CMP_LG_U64:
11893 return optimizeCmpAnd(0, 64, true, false) ||
11894 optimizeCmpSelect(/*NeedInversion=*/false);
11895 }
11896
11897 return false;
11898}
11899
11901 AMDGPU::OpName OpName) const {
11902 if (!ST.needsAlignedVGPRs())
11903 return;
11904
11905 int OpNo = AMDGPU::getNamedOperandIdx(MI.getOpcode(), OpName);
11906 if (OpNo < 0)
11907 return;
11908 MachineOperand &Op = MI.getOperand(OpNo);
11909 if (getOpSize(MI, OpNo) > 4)
11910 return;
11911
11912 // Add implicit aligned super-reg to force alignment on the data operand.
11913 const DebugLoc &DL = MI.getDebugLoc();
11914 MachineBasicBlock *BB = MI.getParent();
11916 Register DataReg = Op.getReg();
11917 bool IsAGPR = RI.isAGPR(MRI, DataReg);
11919 IsAGPR ? &AMDGPU::AGPR_32RegClass : &AMDGPU::VGPR_32RegClass);
11920 BuildMI(*BB, MI, DL, get(AMDGPU::IMPLICIT_DEF), Undef);
11921 Register NewVR =
11922 MRI.createVirtualRegister(IsAGPR ? &AMDGPU::AReg_64_Align2RegClass
11923 : &AMDGPU::VReg_64_Align2RegClass);
11924 BuildMI(*BB, MI, DL, get(AMDGPU::REG_SEQUENCE), NewVR)
11925 .addReg(DataReg, {}, Op.getSubReg())
11926 .addImm(AMDGPU::sub0)
11927 .addReg(Undef)
11928 .addImm(AMDGPU::sub1);
11929 Op.setReg(NewVR);
11930 Op.setSubReg(AMDGPU::sub0);
11931 MI.addOperand(MachineOperand::CreateReg(NewVR, false, true));
11932}
11933
11935 if (!SchedModel.hasInstrSchedModel())
11936 return 0;
11937
11938 // The repeat rate is the throughput-limiting resource occupancy: the largest
11939 // number of cycles any written processor resource is held.
11940 const MCSchedClassDesc *SCDesc = SchedModel.resolveSchedClass(&MI);
11941 unsigned RepeatRate = 0;
11943 PI = SchedModel.getWriteProcResBegin(SCDesc),
11944 PE = SchedModel.getWriteProcResEnd(SCDesc);
11945 PI != PE; ++PI) {
11946 RepeatRate = std::max(RepeatRate, (unsigned)PI->ReleaseAtCycle);
11947 }
11948
11949 return RepeatRate;
11950}
11951
11953 if (isIGLP(*MI))
11954 return false;
11955
11957}
11958
11960 if (!isWMMA(MI) && !isSWMMAC(MI))
11961 return false;
11962
11963 if (ST.hasGFX1250Insts())
11964 return AMDGPU::getWMMAIsXDL(MI.getOpcode());
11965
11966 return true;
11967}
11968
11970 unsigned Opcode = MI.getOpcode();
11971
11972 if (AMDGPU::isGFX12Plus(ST))
11973 return isDOT(MI) || isXDLWMMA(MI);
11974
11975 if (!isMAI(MI) || isDGEMM(Opcode) ||
11976 Opcode == AMDGPU::V_ACCVGPR_WRITE_B32_e64 ||
11977 Opcode == AMDGPU::V_ACCVGPR_READ_B32_e64)
11978 return false;
11979
11980 if (!ST.hasGFX940Insts())
11981 return true;
11982
11983 return AMDGPU::getMAIIsGFX940XDL(Opcode);
11984}
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
AMDGPU Register Bank Select
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool isUndef(const MachineInstr &MI)
TargetInstrInfo::RegSubRegPair RegSubRegPair
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
R600 Clause Merge
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file declares the machine register scavenger class.
static cl::opt< bool > Fix16BitCopies("amdgpu-fix-16-bit-physreg-copies", cl::desc("Fix copies between 32 and 16 bit registers by extending to 32 bit"), cl::init(true), cl::ReallyHidden)
static void expandSGPRCopy(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const TargetRegisterClass *RC, bool Forward)
static std::optional< std::pair< int64_t, int64_t > > getSelectConstants(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, const MachineInstr &Sel)
If Sel is an S_CSELECT* of two different constants A and B, return them, truncated to the width of th...
static unsigned getNewFMAInst(const GCNSubtarget &ST, unsigned Opc)
static unsigned getIndirectSGPRWriteMovRelPseudo32(unsigned VecSize)
static bool compareMachineOp(const MachineOperand &Op0, const MachineOperand &Op1)
static bool isStride64(unsigned Opc)
static MachineBasicBlock * generateWaterFallLoop(const SIInstrInfo &TII, MachineInstr &MI, ArrayRef< MachineOperand * > ScalarOps, MachineDominatorTree *MDT, MachineBasicBlock::iterator Begin=nullptr, MachineBasicBlock::iterator End=nullptr, ArrayRef< Register > PhySGPRs={})
#define GENERATE_RENAMED_GFX9_CASES(OPCODE)
static std::tuple< unsigned, unsigned > extractRsrcPtr(const SIInstrInfo &TII, MachineInstr &MI, MachineOperand &Rsrc)
static unsigned VOP3OpIdxToSrcN(const MachineInstr &MI, unsigned OpIdx)
static bool followSubRegDef(MachineInstr &MI, TargetInstrInfo::RegSubRegPair &RSR)
static unsigned getIndirectSGPRWriteMovRelPseudo64(unsigned VecSize)
static MachineInstr * swapImmOperands(MachineInstr &MI, MachineOperand &NonRegOp1, MachineOperand &NonRegOp2)
static void copyFlagsToImplicitVCC(MachineInstr &MI, const MachineOperand &Orig)
static bool offsetsDoNotOverlap(LocationSize WidthA, int OffsetA, LocationSize WidthB, int OffsetB)
static void indirectCopyToAGPR(const SIInstrInfo &TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, RegScavenger &RS, bool RegsOverlap, Register ImpUseSuperReg=Register())
Handle copying from SGPR to AGPR, or from AGPR to AGPR on GFX908.
static unsigned getWWMRegSpillSaveOpcode(unsigned Size, bool IsVectorSuperClass)
static bool memOpsHaveSameBaseOperands(ArrayRef< const MachineOperand * > BaseOps1, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getWWMRegSpillRestoreOpcode(unsigned Size, bool IsVectorSuperClass)
static unsigned getSGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool setsSCCIfResultIsZero(const MachineInstr &Def, bool &NeedInversion, unsigned &NewDefOpc)
static bool isSCCDeadOnExit(MachineBasicBlock *MBB)
static unsigned getIndirectVGPRWriteMovRelPseudoOpc(unsigned VecSize)
static unsigned subtargetEncodingFamily(const GCNSubtarget &ST)
static void preserveCondRegFlags(MachineOperand &CondReg, const MachineOperand &OrigCond)
static Register findImplicitSGPRRead(const MachineInstr &MI)
static unsigned getNewFMAAKInst(const GCNSubtarget &ST, unsigned Opc)
static cl::opt< unsigned > BranchOffsetBits("amdgpu-s-branch-bits", cl::ReallyHidden, cl::init(16), cl::desc("Restrict range of branch instructions (DEBUG)"))
static unsigned getAVSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static bool memOpsHaveSameBasePtr(const MachineInstr &MI1, ArrayRef< const MachineOperand * > BaseOps1, const MachineInstr &MI2, ArrayRef< const MachineOperand * > BaseOps2)
static unsigned getSGPRSpillRestoreOpcode(unsigned Size)
static bool isRegOrFI(const MachineOperand &MO)
static bool isVCmp(const SIInstrInfo &TII, const MachineInstr &MI)
Return true if MI is a VALU comparison, i.e.
static unsigned getVGPRSpillSaveOpcode(unsigned Size, bool NeedsCFI)
static constexpr AMDGPU::OpName ModifierOpNames[]
static void reportIllegalCopy(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, const char *Msg="illegal VGPR to SGPR copy")
static MachineInstr * swapRegAndNonRegOperand(MachineInstr &MI, MachineOperand &RegOp, MachineOperand &NonRegOp)
static bool shouldReadExec(const MachineInstr &MI)
static unsigned getNewFMAMKInst(const GCNSubtarget &ST, unsigned Opc)
static bool isRenamedInGFX9(int Opcode)
static TargetInstrInfo::RegSubRegPair getRegOrUndef(const MachineOperand &RegOpnd)
static std::tuple< unsigned, unsigned, unsigned > splitGlobalAddressRelocFlags(const GCNSubtarget &ST, const MachineOperand &SrcOp)
static bool changesVGPRIndexingMode(const MachineInstr &MI)
static bool isSubRegOf(const SIRegisterInfo &TRI, const MachineOperand &SuperVec, const MachineOperand &SubReg)
static bool nodesHaveSameOperandValue(SDNode *N0, SDNode *N1, AMDGPU::OpName OpName)
Returns true if both nodes have the same value for the given operand Op, or if both nodes do not have...
static unsigned getNumOperandsNoGlue(SDNode *Node)
static bool canRemat(const MachineInstr &MI)
static unsigned getAVSpillRestoreOpcode(unsigned Size)
static void emitLoadScalarOpsFromVGPRLoop(const SIInstrInfo &TII, MachineRegisterInfo &MRI, MachineBasicBlock &PredBB, MachineBasicBlock &LoopBB, MachineBasicBlock &BodyBB, const DebugLoc &DL, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={})
static unsigned getVGPRSpillRestoreOpcode(unsigned Size)
Interface definition for SIInstrInfo.
bool IsDead
const char * Msg
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
#define LLVM_DEBUG(...)
Definition Debug.h:119
static const LaneMaskConstants & get(const GCNSubtarget &ST)
static LLVM_ABI Semantics SemanticsToEnum(const llvm::fltSemantics &Sem)
Definition APFloat.cpp:185
Class for arbitrary precision integers.
Definition APInt.h:78
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class is the base class for the comparison instructions.
Definition InstrTypes.h:728
uint64_t getZExtValue() const
Opaque handle to a cycle within a GenericCycleInfo that wraps the cycle's preorder index.
A debug info location.
Definition DebugLoc.h:126
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:857
Diagnostic information for unsupported feature in backend.
void changeImmediateDominator(DomTreeNodeBase< NodeT > *N, DomTreeNodeBase< NodeT > *NewIDom)
changeImmediateDominator - This method is used to update the dominator tree information when a node's...
DomTreeNodeBase< NodeT > * addNewBlock(NodeT *BB, NodeT *DomBB)
Add a new node to the dominator tree information.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
void getExitingBlocks(CycleRef C, SmallVectorImpl< BlockT * > &TmpStorage) const
Return all blocks of C that have a successor outside of C.
CycleRef getParentCycle(CycleRef C) const
bool contains(CycleRef Outer, CycleRef Inner) const
Returns true iff Outer contains Inner. O(1). Non-strict.
CycleRef getCycle(const BlockT *Block) const
Find the innermost cycle containing Block.
Itinerary data supplied by a subtarget to be used by a target.
constexpr unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasInterval(Register Reg) const
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LLVM_ABI bool shrinkToUses(LiveInterval *li, SmallVectorImpl< MachineInstr * > *dead=nullptr)
After removing some uses of a register, shrink its live range to just the remaining uses.
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
This class represents the liveness of a register, stack slot, etc.
bool hasValue() const
static LocationSize precise(uint64_t Value)
TypeSize getValue() const
static const MCBinaryExpr * createAnd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:347
static const MCBinaryExpr * createAShr(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:417
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
unsigned getSize() const
Return the number of bytes in the encoding of this instruction, or zero if the encoding size cannot b...
ArrayRef< MCPhysReg > implicit_uses() const
Return a list of registers that are potentially read by any instance of this machine instruction.
unsigned getOpcode() const
Return the opcode number for this descriptor.
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
uint8_t OperandType
Information about the type of the operand.
int16_t RegClass
This specifies the register class enumeration of the operand if the operand is a register.
Definition MCInstrDesc.h:94
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
LLVM_ABI void setVariableValue(const MCExpr *Value)
Definition MCSymbol.cpp:50
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
MIBundleBuilder & append(MachineInstr *MI)
Insert MI into MBB by appending it to the instructions in the bundle.
MIRFormater - Interface to format MIR operand based on target.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI MCSymbol * getSymbol() const
Return the MCSymbol for this basic block.
void push_back(MachineInstr *MI)
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
Instructions::const_iterator const_instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< succ_iterator > successors()
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool isImmutableObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to an immutable object.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & copyImplicitOps(const MachineInstr &OtherMI) const
Copy all the implicit operands from OtherMI onto this one.
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
bool isCopy() const
const MachineBasicBlock * getParent() const
LLVM_ABI void addImplicitDefUseOperands(MachineFunction &MF)
Add all implicit def and use operands to this instruction.
bool isBundle() const
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
bool modifiesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr modifies (fully define or partially define) the specified register.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
LLVM_ABI bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by mayLoad / mayStore,...
void untieRegOperand(unsigned OpIdx)
Break any tie involving OpIdx.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
LLVM_ABI void eraseFromBundle()
Unlink 'this' from its basic block and delete it.
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mop_range explicit_operands()
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
LLVM_ABI bool hasOrderedMemoryRef() const
Return true if this instruction may have an ordered or volatile memory reference, or if the informati...
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
ArrayRef< MachineMemOperand * > memoperands() const
Access to memory operands of the instruction.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
filtered_mop_range all_uses()
Returns an iterator range over all operands that are (explicit or implicit) register uses.
LLVM_ABI void setPostInstrSymbol(MachineFunction &MF, MCSymbol *Symbol)
Set a symbol that will be emitted just after the instruction itself.
LLVM_ABI void clearRegisterKills(Register Reg, const TargetRegisterInfo *RegInfo)
Clear all kill flags affecting Reg.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI int findRegisterDefOperandIdx(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false) const
Returns the operand index that is a def of the specified register or -1 if it is not found.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand * findRegisterDefOperand(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false)
Wrapper for findRegisterDefOperandIdx, it returns a pointer to the MachineOperand rather than an inde...
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
const GlobalValue * getGlobal() const
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
void setOffset(int64_t Offset)
unsigned getTargetFlags() const
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isTargetIndex() const
isTargetIndex - Tests if this is a MO_TargetIndex operand.
void setTargetFlags(unsigned F)
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_Register
Register operand.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
int64_t getOffset() const
Return the offset from the symbol in this operand.
bool isFPImm() const
isFPImm - Tests if this is a MO_FPImmediate operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI void moveOperands(MachineOperand *Dst, MachineOperand *Src, unsigned NumOps)
Move NumOps operands from Src to Dst, updating use-def lists as needed.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool reservedRegsFrozen() const
reservedRegsFrozen - Returns true after freezeReservedRegs() was called to ensure the set of reserved...
LLVM_ABI void clearVirtRegs()
clearVirtRegs - Remove all virtual registers (after physreg assignment).
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
const MachineFunction & getMF() const
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI bool isConstantPhysReg(MCRegister PhysReg) const
Returns true if PhysReg is unallocatable and constant throughout the function.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
iterator_range< use_iterator > use_operands(Register Reg) const
LLVM_ABI void removeRegOperandFromUseList(MachineOperand *MO)
Remove MO from its use-def list.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
LLVM_ABI void addRegOperandToUseList(MachineOperand *MO)
Add MO to the linked list of operands for its register.
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isValid() const
Definition Register.h:112
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isLegalMUBUFImmOffset(unsigned Imm) const
bool isInlineConstant(const APInt &Imm) const
void legalizeOperandsVOP3(MachineRegisterInfo &MRI, MachineInstr &MI) const
Fix operands in MI to satisfy constant bus requirements.
bool canAddToBBProlog(const MachineInstr &MI) const
static bool isDS(const MachineInstr &MI)
MachineBasicBlock * legalizeOperands(MachineInstr &MI, MachineDominatorTree *MDT=nullptr) const
Legalize all operands in this instruction.
bool areLoadsFromSameBasePtr(SDNode *Load0, SDNode *Load1, int64_t &Offset0, int64_t &Offset1) const override
unsigned getLiveRangeSplitOpcode(Register Reg, const MachineFunction &MF) const override
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
static bool isNeverUniform(const MachineInstr &MI)
bool isXDLWMMA(const MachineInstr &MI) const
bool isBasicBlockPrologue(const MachineInstr &MI, Register Reg=Register()) const override
uint64_t getDefaultRsrcDataFormat() const
static bool isSOPP(const MachineInstr &MI)
bool mayAccessScratch(const MachineInstr &MI) const
bool isIGLP(unsigned Opcode) const
static bool isFLATScratch(const MachineInstr &MI)
bool isLegalFLATOffset(int64_t Offset, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Returns if Offset is legal for the subtarget as the offset to a FLAT encoded instruction with the giv...
const MCInstrDesc & getIndirectRegWriteMovRelPseudo(unsigned VecSize, unsigned EltSize, bool IsSGPR) const
MachineInstrBuilder getAddNoCarry(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg) const
Return a partially built integer add instruction without carry.
bool shouldScheduleLoadsNear(SDNode *Load0, SDNode *Load1, int64_t Offset0, int64_t Offset1, unsigned NumLoads) const override
bool splitMUBUFOffset(uint32_t Imm, uint32_t &SOffset, uint32_t &ImmOffset, Align Alignment=Align(4)) const
bool isIgnorableUse(const MachineInstr &MI, unsigned OpIdx) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
void moveToVALU(SIInstrWorklist &Worklist, MachineDominatorTree *MDT) const
Replace the instructions opcode with the equivalent VALU opcode.
static bool isSMRD(const MachineInstr &MI)
unsigned getGFX1250BlockingCyclesTable(const MachineInstr &MI) const
GFX1250 blocking-cycles table lookup with no occupancy subtarget gate.
void restoreExec(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, SlotIndexes *Indexes=nullptr) const
void storeRegToStackSlotCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC) const
bool usesConstantBus(const MachineRegisterInfo &MRI, const MachineOperand &MO, const MCOperandInfo &OpInfo) const
Returns true if this operand uses the constant bus.
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getFoldableCopySrcIdx(const MachineInstr &MI)
unsigned getOpSize(uint32_t Opcode, unsigned OpNo) const
Return the size in bytes of the operand OpNo on the given.
void legalizeOperandsFLAT(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &LdSt, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const final
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
static bool isMTBUF(const MachineInstr &MI)
const MCInstrDesc & getIndirectGPRIDXPseudo(unsigned VecSize, bool IsIndirectSrc) const
static bool isDGEMM(unsigned Opcode)
static bool isEXP(const MachineInstr &MI)
static bool isSALU(const MachineInstr &MI)
static bool setsSCCIfResultIsNonZero(const MachineInstr &MI)
const MIRFormatter * getMIRFormatter() const override
static bool isXcntDrain(const MachineInstr &MI)
True if MI implicitly drains XCNT.
void legalizeGenericOperand(MachineBasicBlock &InsertMBB, MachineBasicBlock::iterator I, const TargetRegisterClass *DstRC, MachineOperand &Op, MachineRegisterInfo &MRI, const DebugLoc &DL) const
MachineInstr * buildShrunkInst(MachineInstr &MI, unsigned NewOpcode) const
static bool isVOP2(const MachineInstr &MI)
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
static bool isSDWA(const MachineInstr &MI)
const MCInstrDesc & getKillTerminatorFromPseudo(unsigned Opcode) const
void insertNoops(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, unsigned Quantity) const override
static bool isGather4(const MachineInstr &MI)
MachineInstr * getWholeWaveFunctionSetup(MachineFunction &MF) const
static bool isDOT(const MachineInstr &MI)
std::unique_ptr< PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
InstSizeVerifyMode getInstSizeVerifyMode(const MachineInstr &MI) const override
MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const override
bool hasModifiers(unsigned Opcode) const
Return true if this instruction has any modifiers.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
static bool isSWMMAC(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
bool isWave32() const
bool isHighLatencyDef(int Opc) const override
void legalizeOpWithMove(MachineInstr &MI, unsigned OpIdx) const
Legalize the OpIndex operand of this instruction by inserting a MOV.
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
static bool isVOPC(const MachineInstr &MI)
void removeModOperands(MachineInstr &MI) const
unsigned getRepeatRate(const MachineInstr &MI) const
Get the repeat rate for a VALU instruction from the scheduling model.
unsigned getVectorRegSpillRestoreOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI) const
bool isLegalSingleSGPRReadInstOperand(const MachineRegisterInfo &MRI, const MachineInstr &MI, unsigned SrcN, const MachineOperand *MO=nullptr) const
Check if MO would be a legal operand for a single-SGPR-read instruction.
bool isXDL(const MachineInstr &MI) const
Register isStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
static bool isVIMAGE(const MachineInstr &MI)
void enforceOperandRCAlignment(MachineInstr &MI, AMDGPU::OpName OpName) const
static bool isSOP2(const MachineInstr &MI)
static bool isGWS(const MachineInstr &MI)
bool hasRAWDependency(const MachineInstr &FirstMI, const MachineInstr &SecondMI) const
bool isLegalAV64PseudoImm(uint64_t Imm) const
Check if this immediate value can be used for AV_MOV_B64_IMM_PSEUDO.
bool isNeverCoissue(MachineInstr &MI) const
static bool isBUF(const MachineInstr &MI)
bool isNonCommutableDPP(const MachineInstr &MI) const
void handleCopyToPhysHelper(SIInstrWorklist &Worklist, Register DstReg, MachineInstr &Inst, MachineRegisterInfo &MRI, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
bool hasModifiersSet(const MachineInstr &MI, AMDGPU::OpName OpName) const
bool isLegalToSwap(const MachineInstr &MI, unsigned fromIdx, unsigned toIdx) const
static bool isFLATGlobal(const MachineInstr &MI)
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
bool isGlobalMemoryObject(const MachineInstr *MI) const override
static bool isVSAMPLE(const MachineInstr &MI)
bool isBufferSMRD(const MachineInstr &MI) const
static bool isKillTerminator(unsigned Opcode)
bool isVOPDAntidependencyAllowed(const MachineInstr &MI) const
If OpX is multicycle, anti-dependencies are not allowed.
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
void insertScratchExecCopy(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, Register Reg, bool IsSCCLive, SlotIndexes *Indexes=nullptr) const
bool hasVALU32BitEncoding(unsigned Opcode) const
Return true if this 64-bit VALU instruction has a 32-bit encoding.
unsigned getBlockingCycles(const MachineInstr &MI) const
unsigned getMovOpcode(const TargetRegisterClass *DstRC) const
Register isSGPRStackAccess(const MachineInstr &MI, int &FrameIndex, TypeSize &MemBytes) const
unsigned buildExtractSubReg(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
void legalizeOperandsVOP2(MachineRegisterInfo &MRI, MachineInstr &MI) const
Legalize operands in MI by either commuting it or inserting a copy of src1.
static bool isVPermPk16(unsigned Opcode)
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const final
static bool isTRANS(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isSOPK(const MachineInstr &MI)
const TargetRegisterClass * getOpRegClass(const MachineInstr &MI, unsigned OpNo) const
Return the correct register class for OpNo.
MachineBasicBlock * insertSimulatedTrap(MachineRegisterInfo &MRI, MachineBasicBlock &MBB, MachineInstr &MI, const DebugLoc &DL) const
Build instructions that simulate the behavior of a s_trap 2 instructions for hardware (namely,...
static unsigned getNonSoftWaitcntOpcode(unsigned Opcode)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static bool isFoldableCopy(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
bool expandPostRAPseudo(MachineInstr &MI) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
void createWaterFallForSiCall(MachineInstr *MI, MachineDominatorTree *MDT, ArrayRef< MachineOperand * > ScalarOps, ArrayRef< Register > PhySGPRs={}) const
Wrapper function for generating waterfall for instruction MI This function take into consideration of...
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isSegmentSpecificFLAT(const MachineInstr &MI)
bool isReMaterializableImpl(const MachineInstr &MI) const override
static bool isVOP3(const MCInstrDesc &Desc)
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool physRegUsesConstantBus(const MachineOperand &Reg) const
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
bool mayAccessVMEMThroughFlat(const MachineInstr &MI) const
static bool isDPP(const MachineInstr &MI)
bool analyzeBranchImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify) const
static bool isMFMA(const MachineInstr &MI)
bool isLowLatencyInstruction(const MachineInstr &MI) const
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is a instruction that moves/copies value from one register to ano...
void mutateAndCleanupImplicit(MachineInstr &MI, const MCInstrDesc &NewDesc) const
ValueUniformity getGenericValueUniformity(const MachineInstr &MI) const
static bool isMAI(const MCInstrDesc &Desc)
static bool isSrc1DPPRevOpcode(const GCNSubtarget &ST, uint32_t Opcode)
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
static bool usesLGKM_CNT(const MachineInstr &MI)
void legalizeOperandsVALUt16(MachineInstr &Inst, MachineRegisterInfo &MRI) const
Fix operands in Inst to fix 16bit SALU to VALU lowering.
bool isImmOperandLegal(const MCInstrDesc &InstDesc, unsigned OpNo, const MachineOperand &MO) const
bool canShrink(const MachineInstr &MI, const MachineRegisterInfo &MRI) const
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool isAsmOnlyOpcode(int MCOp) const
Check if this instruction should only be used by assembler.
bool isAlwaysGDS(uint32_t Opcode) const
static bool isVGPRSpill(const MachineInstr &MI)
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
This is used by the post-RA scheduler (SchedulePostRAList.cpp).
bool verifyInstruction(const MachineInstr &MI, StringRef &ErrInfo) const override
unsigned getInstrLatency(const InstrItineraryData *ItinData, const MachineInstr &MI, unsigned *PredCost=nullptr) const override
unsigned getVectorRegSpillSaveOpcode(Register Reg, const TargetRegisterClass *RC, unsigned Size, const SIMachineFunctionInfo &MFI, bool NeedsCFI) const
int64_t getNamedImmOperand(const MachineInstr &MI, AMDGPU::OpName OperandName) const
Get required immediate operand.
ArrayRef< std::pair< int, const char * > > getSerializableTargetIndices() const override
bool regUsesConstantBus(const MachineOperand &Reg, const MachineRegisterInfo &MRI) const
static bool isMIMG(const MachineInstr &MI)
MachineOperand buildExtractSubRegOrImm(MachineBasicBlock::iterator MI, MachineRegisterInfo &MRI, const MachineOperand &SuperReg, const TargetRegisterClass *SuperRC, unsigned SubIdx, const TargetRegisterClass *SubRC) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
bool isLegalRegOperand(const MachineRegisterInfo &MRI, const MCOperandInfo &OpInfo, const MachineOperand &MO) const
Check if MO (a register operand) is a legal register for the given operand description or operand ind...
static unsigned getNumWaitStates(const MachineInstr &MI)
Return the number of wait states that result from executing this instruction.
unsigned getVALUOp(const MachineInstr &MI) const
static bool modifiesModeRegister(const MachineInstr &MI)
Return true if the instruction modifies the mode register.q.
Register readlaneVGPRToSGPR(Register SrcReg, MachineInstr &UseMI, MachineRegisterInfo &MRI, const TargetRegisterClass *DstRC=nullptr) const
Copy a value from a VGPR (SrcReg) to SGPR.
bool hasDivergentBranch(const MachineBasicBlock *MBB) const
Return whether the block terminate with divergent branch.
std::pair< int64_t, int64_t > splitFlatOffset(int64_t COffsetVal, unsigned AddrSpace, AMDGPU::FlatAddrSpace FlatVariant) const
Split COffsetVal into {immediate offset field, remainder offset} values.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
void fixImplicitOperands(MachineInstr &MI) const
bool moveFlatAddrToVGPR(MachineInstr &Inst) const
Change SADDR form of a FLAT Inst to its VADDR form if saddr operand was moved to VGPR.
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool isMaskedByExec(Register Reg, const MachineInstr &Use, const MachineRegisterInfo &MRI, unsigned Depth=0) const
Return true if Reg is a lane mask that already has 0 in every bit corresponding to a lane that is ina...
void createReadFirstLaneFromCopyToPhysReg(MachineRegisterInfo &MRI, Register DstReg, MachineInstr &Inst) const
bool swapSourceModifiers(MachineInstr &MI, MachineOperand &Src0, AMDGPU::OpName Src0OpName, MachineOperand &Src1, AMDGPU::OpName Src1OpName) const
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
bool hasUnwantedEffectsWhenEXECEmpty(const MachineInstr &MI) const
This function is used to determine if an instruction can be safely executed under EXEC = 0 without ha...
bool getConstValDefinedInReg(const MachineInstr &MI, const Register Reg, int64_t &ImmVal) const override
static bool isAtomic(const MachineInstr &MI)
bool canInsertSelect(const MachineBasicBlock &MBB, ArrayRef< MachineOperand > Cond, Register DstReg, Register TrueReg, Register FalseReg, int &CondCycles, int &TrueCycles, int &FalseCycles) const override
bool isLiteralOperandLegal(const MCInstrDesc &InstDesc, const MCOperandInfo &OpInfo) const
static bool isWWMRegSpillOpcode(uint32_t Opcode)
static bool sopkIsZext(unsigned Opcode)
static bool isSGPRSpill(const MachineInstr &MI)
static bool isWMMA(const MachineInstr &MI)
ArrayRef< std::pair< MachineMemOperand::Flags, const char * > > getSerializableMachineMemOperandTargetFlags() const override
bool mayReadEXEC(const MachineRegisterInfo &MRI, const MachineInstr &MI) const
Returns true if the instruction could potentially depend on the value of exec.
void legalizeOperandsSMRD(MachineRegisterInfo &MRI, MachineInstr &MI) const
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
std::pair< MachineInstr *, MachineInstr * > expandMovDPP64(MachineInstr &MI) const
static bool isSOPC(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
bool isBarrier(unsigned Opcode) const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx0, unsigned OpIdx1) const override
bool mayAccessLDSThroughFlat(const MachineInstr &MI, bool TgSplit) const
int pseudoToMCOpcode(int Opcode) const
Return a target-specific opcode if Opcode is a pseudo instruction.
const MCInstrDesc & getMCOpcodeFromPseudo(unsigned Opcode) const
Return the descriptor of the target-specific machine instruction that corresponds to the specified ps...
static bool usesVM_CNT(const MachineInstr &MI)
MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const override
static bool isFixedSize(const MachineInstr &MI)
bool isSafeToSink(MachineInstr &MI, MachineBasicBlock *SuccToSinkTo, MachineCycleInfo *CI) const override
LLVM_READONLY int commuteOpcode(unsigned Opc) const
ValueUniformity getValueUniformity(const MachineInstr &MI) const final
uint64_t getScratchRsrcWords23() const
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
MachineInstr * convertToThreeAddress(MachineInstr &MI, LiveIntervals *LIS) const override
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
bool isOperandLegal(const MachineInstr &MI, unsigned OpIdx, const MachineOperand *MO=nullptr) const
Check if MO is a legal operand if it was the OpIdx Operand for MI.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool allowNegativeFlatOffset(AMDGPU::FlatAddrSpace FlatVariant) const
Returns true if negative offsets are allowed for the given FlatVariant.
void moveToVALUImpl(SIInstrWorklist &Worklist, MachineDominatorTree *MDT, MachineInstr &Inst, DenseMap< MachineInstr *, V2PhysSCopyInfo > &WaterFalls, DenseMap< MachineInstr *, bool > &V2SPhyCopiesToErase) const
static bool isLDSDMA(const MachineInstr &MI)
static bool isVOP1(const MachineInstr &MI)
SIInstrInfo(const GCNSubtarget &ST)
std::optional< int64_t > getImmOrMaterializedImm(const MachineRegisterInfo &MRI, const MachineOperand &Op, MachineInstr **DefMI=nullptr) const
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
bool hasAnyModifiersSet(const MachineInstr &MI) const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
void setHasSpilledVGPRs(bool Spill=true)
bool isWWMReg(Register Reg) const
bool checkFlag(Register Reg, uint8_t Flag) const
void setHasSpilledSGPRs(bool Spill=true)
unsigned getScratchReservedForDynamicVGPRs() const
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
unsigned getHWRegIndex(MCRegister Reg) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
unsigned getChannelFromSubReg(unsigned SubReg) const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
MachineFunction & MF
Machine function.
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
SlotIndex - An opaque wrapper around machine indexes.
Definition SlotIndexes.h:66
SlotIndex getRegSlot(bool EC=false) const
Returns the register use/def slot in the current instruction for a normal or early-clobber def.
SlotIndexes pass.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
int64_t getImm() const
Register getReg() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Object returned by analyzeLoopForPipelining.
virtual ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *, const ScheduleDAGMI *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual MachineInstr * createPHIDestinationCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual const MachineOperand & getCalleeOperand(const MachineInstr &MI) const
Returns the callee operand from the given MI.
virtual void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const
Re-issue the specified 'original' instruction at the specific location targeting a new destination re...
virtual MachineInstr * createPHISourceCopy(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsPt, const DebugLoc &DL, Register Src, unsigned SrcSubReg, Register Dst) const
During PHI eleimination lets target to make necessary checks and insert the copy to the PHI destinati...
virtual MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const
This method commutes the operands of the given machine instruction MI.
virtual bool isGlobalMemoryObject(const MachineInstr *MI) const
Returns true if MI is an instruction we are unable to reason about (like a call or something with unm...
virtual bool expandPostRAPseudo(MachineInstr &MI) const
This function is called for all pseudo instructions that remain after register allocation.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
const MCWriteProcResEntry * ProcResIter
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
Definition DenseSet.h:187
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
const uint64_t RSRC_DATA_FORMAT
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool getWMMAIsXDL(unsigned Opc)
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_READONLY int32_t getCommuteRev(uint32_t Opcode)
LLVM_READONLY int32_t getCommuteOrig(uint32_t Opcode)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool isGFX12Plus(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_READONLY int32_t getGlobalVaddrOp(uint32_t Opcode)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
bool getMAIIsGFX940XDL(unsigned Opc)
const uint64_t RSRC_ELEMENT_SIZE_SHIFT
bool isIntrinsicAlwaysUniform(unsigned IntrID)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getIfAddr64Inst(uint32_t Opcode)
Check if Opcode is an Addr64 opcode.
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
const uint64_t RSRC_TID_ENABLE
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
bool isGenericAtomic(unsigned Opc)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
LLVM_READONLY int32_t getAddr64Inst(uint32_t Opcode)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:459
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:432
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
Definition SIDefines.h:444
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:429
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:439
@ OPERAND_SDWA_VOPC_DST
Definition SIDefines.h:476
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:428
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_INLINE_C_AV64_PSEUDO
Definition SIDefines.h:470
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:448
@ OPERAND_REG_IMM_INT16
Definition SIDefines.h:427
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:456
LLVM_READONLY int32_t getBasicFromSDWAOp(uint32_t Opcode)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
@ TI_SCRATCH_RSRC_DWORD1
Definition AMDGPU.h:667
@ TI_SCRATCH_RSRC_DWORD3
Definition AMDGPU.h:669
@ TI_SCRATCH_RSRC_DWORD0
Definition AMDGPU.h:666
@ TI_SCRATCH_RSRC_DWORD2
Definition AMDGPU.h:668
@ TI_CONSTDATA_START
Definition AMDGPU.h:665
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
const uint64_t RSRC_INDEX_STRIDE_SHIFT
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ OPERAND_GENERIC_4
Definition MCInstrDesc.h:71
@ OPERAND_GENERIC_2
Definition MCInstrDesc.h:69
@ OPERAND_GENERIC_1
Definition MCInstrDesc.h:68
@ OPERAND_GENERIC_3
Definition MCInstrDesc.h:70
@ OPERAND_IMMEDIATE
Definition MCInstrDesc.h:61
@ OPERAND_GENERIC_0
Definition MCInstrDesc.h:67
@ OPERAND_GENERIC_5
Definition MCInstrDesc.h:72
Not(const Pred &P) -> Not< Pred >
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
constexpr uint64_t maxUIntN(uint64_t N)
Gets the maximum value for a N-bit unsigned integer.
Definition MathExtras.h:208
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
TargetInstrInfo::RegSubRegPair getRegSequenceSubReg(MachineInstr &MI, unsigned SubReg)
Return the SubReg component from REG_SEQUENCE.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
Definition SIInstrInfo.h:45
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
MachineInstr * getVRegSubRegDef(const TargetInstrInfo::RegSubRegPair &P, const MachineRegisterInfo &MRI)
Return the defining instruction for a given reg:subreg pair skipping copy like instructions and subre...
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
auto instructionsWithoutDebug(IterT It, IterT End, bool SkipPseudoOp=true)
Construct a range iterator which begins at It and moves forwards until End is reached,...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI VirtRegInfo AnalyzeVirtRegInBundle(MachineInstr &MI, Register Reg, SmallVectorImpl< std::pair< MachineInstr *, unsigned > > *Ops=nullptr)
AnalyzeVirtRegInBundle - Analyze how the current instruction or bundle uses a virtual register.
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
Definition SIInstrInfo.h:53
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ Xor
Bitwise or logical XOR of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
bool isTargetSpecificOpcode(unsigned Opcode)
Check whether the given Opcode is a target-specific opcode.
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned DefaultMemoryClusterDWordsLimit
Definition SIInstrInfo.h:41
constexpr unsigned BitWidth
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Definition SIInstrInfo.h:49
constexpr T reverseBits(T Val)
Reverse the bits in Val.
Definition MathExtras.h:119
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
constexpr RegState getUndefRegState(bool B)
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
Definition Uniformity.h:18
@ AlwaysUniform
The result value is always uniform.
Definition Uniformity.h:23
@ NeverUniform
The result value can never be assumed to be uniform.
Definition Uniformity.h:26
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
Definition SIInstrInfo.h:64
bool execMayBeModifiedBeforeAnyUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI)
Return false if EXEC is not changed between the def of VReg at DefMI and all its uses.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Helper struct for the implementation of 3-address conversion to communicate updates made to instructi...
MachineInstr * RemoveMIUse
Other instruction whose def is no longer used by the converted instruction.
static constexpr uint64_t encode(Fields... Values)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr bool all() const
Definition LaneBitmask.h:54
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Definition MCSchedule.h:129
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
Utility to store machine instructions worklist.
Definition SIInstrInfo.h:68
MachineInstr * top() const
Definition SIInstrInfo.h:73
bool isDeferred(MachineInstr *MI)
SetVector< MachineInstr * > & getDeferredList()
Definition SIInstrInfo.h:91
void insert(MachineInstr *MI)
A pair composed of a register and a sub-register index.
VirtRegInfo - Information about a virtual register used by a set of operands.
bool Reads
Reads - One of the operands read the virtual register.
bool Writes
Writes - One of the operands writes the virtual register.