LLVM 24.0.0git
SIRegisterInfo.cpp
Go to the documentation of this file.
1//===-- SIRegisterInfo.cpp - SI Register Information ---------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// SI implementation of the TargetRegisterInfo class.
11//
12//===----------------------------------------------------------------------===//
13
14#include "SIRegisterInfo.h"
15#include "AMDGPU.h"
17#include "GCNSubtarget.h"
25#include "llvm/Support/Debug.h"
27
28#define DEBUG_TYPE "amdgpu-si-register-info"
29
30using namespace llvm;
31
32#define GET_REGINFO_TARGET_DESC
33#include "AMDGPUGenRegisterInfo.inc"
34
36 "amdgpu-spill-sgpr-to-vgpr",
37 cl::desc("Enable spilling SGPRs to VGPRs"),
39 cl::init(true));
40
42 "amdgpu-spill-cfi-saved-regs",
43 cl::desc("Enable spilling the registers required for CFI emission"),
45
47 "amdgpu-stress-vgpr", cl::Hidden, cl::init(0),
48 cl::desc("Limit VGPRs to N registers by reserving the rest"));
49
51 "amdgpu-stress-agpr", cl::Hidden, cl::init(0),
52 cl::desc("Limit AGPRs to N registers by reserving the rest"));
53
55 "amdgpu-stress-sgpr", cl::Hidden, cl::init(0),
56 cl::desc("Limit SGPRs to N registers by reserving the rest"));
57
58std::array<std::vector<int16_t>, 32> SIRegisterInfo::RegSplitParts;
59std::array<std::array<uint16_t, 32>, 9> SIRegisterInfo::SubRegFromChannelTable;
60
61// Map numbers of DWORDs to indexes in SubRegFromChannelTable.
62// Valid indexes are shifted 1, such that a 0 mapping means unsupported.
63// e.g. for 8 DWORDs (256-bit), SubRegFromChannelTableWidthMap[8] = 8,
64// meaning index 7 in SubRegFromChannelTable.
65static const std::array<unsigned, 17> SubRegFromChannelTableWidthMap = {
66 0, 1, 2, 3, 4, 5, 6, 7, 8, 0, 0, 0, 0, 0, 0, 0, 9};
67
68static void emitUnsupportedError(const Function &Fn, const MachineInstr &MI,
69 const Twine &ErrMsg) {
71 DiagnosticInfoUnsupported(Fn, ErrMsg, MI.getDebugLoc()));
72}
73
74namespace llvm {
75
76// A temporary struct to spill SGPRs.
77// This is mostly to spill SGPRs to memory. Spilling SGPRs into VGPR lanes emits
78// just v_writelane and v_readlane.
79//
80// When spilling to memory, the SGPRs are written into VGPR lanes and the VGPR
81// is saved to scratch (or the other way around for loads).
82// For this, a VGPR is required where the needed lanes can be clobbered. The
83// RegScavenger can provide a VGPR where currently active lanes can be
84// clobbered, but we still need to save inactive lanes.
85// The high-level steps are:
86// - Try to scavenge SGPR(s) to save exec
87// - Try to scavenge VGPR
88// - Save needed, all or inactive lanes of a TmpVGPR
89// - Spill/Restore SGPRs using TmpVGPR
90// - Restore TmpVGPR
91//
92// To save all lanes of TmpVGPR, exec needs to be saved and modified. If we
93// cannot scavenge temporary SGPRs to save exec, we use the following code:
94// buffer_store_dword TmpVGPR ; only if active lanes need to be saved
95// s_not exec, exec
96// buffer_store_dword TmpVGPR ; save inactive lanes
97// s_not exec, exec
99 struct PerVGPRData {
100 unsigned PerVGPR;
101 unsigned NumVGPRs;
102 int64_t VGPRLanes;
103 };
104
105 // The SGPR to save
109 unsigned NumSubRegs;
110 bool IsKill;
111 const DebugLoc &DL;
112
113 /* When spilling to stack */
114 // The SGPRs are written into this VGPR, which is then written to scratch
115 // (or vice versa for loads).
116 Register TmpVGPR = AMDGPU::NoRegister;
117 // Temporary spill slot to save TmpVGPR to.
119 // If TmpVGPR is live before the spill or if it is scavenged.
120 bool TmpVGPRLive = false;
121 // Scavenged SGPR to save EXEC.
122 Register SavedExecReg = AMDGPU::NoRegister;
123 // Stack index to write the SGPRs to.
124 int Index;
125 unsigned EltSize = 4;
126
135 unsigned MovOpc;
136 unsigned NotOpc;
137
141 : SGPRSpillBuilder(TRI, TII, IsWave32, MI, MI->getOperand(0).getReg(),
142 MI->getOperand(0).isKill(), Index, RS) {}
143
146 bool IsKill, int Index, RegScavenger *RS)
147 : SuperReg(Reg), MI(MI), IsKill(IsKill), DL(MI->getDebugLoc()),
148 Index(Index), RS(RS), MBB(MI->getParent()), MF(*MBB->getParent()),
149 MFI(*MF.getInfo<SIMachineFunctionInfo>()), TII(TII), TRI(TRI),
151 const TargetRegisterClass *RC = TRI.getPhysRegBaseClass(SuperReg);
152 SplitParts = TRI.getRegSplitParts(RC, EltSize);
153 NumSubRegs = SplitParts.empty() ? 1 : SplitParts.size();
154
155 if (IsWave32) {
156 ExecReg = AMDGPU::EXEC_LO;
157 MovOpc = AMDGPU::S_MOV_B32;
158 NotOpc = AMDGPU::S_NOT_B32;
159 } else {
160 ExecReg = AMDGPU::EXEC;
161 MovOpc = AMDGPU::S_MOV_B64;
162 NotOpc = AMDGPU::S_NOT_B64;
163 }
164
165 assert(SuperReg != AMDGPU::M0 && "m0 should never spill");
166 assert(SuperReg != AMDGPU::EXEC_LO && SuperReg != AMDGPU::EXEC_HI &&
167 SuperReg != AMDGPU::EXEC && "exec should never spill");
168 }
169
172 Data.PerVGPR = IsWave32 ? 32 : 64;
173 Data.NumVGPRs = (NumSubRegs + (Data.PerVGPR - 1)) / Data.PerVGPR;
174 Data.VGPRLanes = (1LL << std::min(Data.PerVGPR, NumSubRegs)) - 1LL;
175 return Data;
176 }
177
178 // Tries to scavenge SGPRs to save EXEC and a VGPR. Uses v0 if no VGPR is
179 // free.
180 // Writes these instructions if an SGPR can be scavenged:
181 // s_mov_b64 s[6:7], exec ; Save exec
182 // s_mov_b64 exec, 3 ; Wanted lanemask
183 // buffer_store_dword v1 ; Write scavenged VGPR to emergency slot
184 //
185 // Writes these instructions if no SGPR can be scavenged:
186 // buffer_store_dword v0 ; Only if no free VGPR was found
187 // s_not_b64 exec, exec
188 // buffer_store_dword v0 ; Save inactive lanes
189 // ; exec stays inverted, it is flipped back in
190 // ; restore.
191 void prepare() {
192 // Scavenged temporary VGPR to use. It must be scavenged once for any number
193 // of spilled subregs.
194 // FIXME: The liveness analysis is limited and does not tell if a register
195 // is in use in lanes that are currently inactive. We can never be sure if
196 // a register as actually in use in another lane, so we need to save all
197 // used lanes of the chosen VGPR.
198 assert(RS && "Cannot spill SGPR to memory without RegScavenger");
199 TmpVGPR = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass, MI, false,
200 0, false);
201
202 // Reserve temporary stack slot
203 TmpVGPRIndex = MFI.getScavengeFI(MF.getFrameInfo(), TRI);
204 if (TmpVGPR) {
205 // Found a register that is dead in the currently active lanes, we only
206 // need to spill inactive lanes.
207 TmpVGPRLive = false;
208 } else {
209 // Pick v0 because it doesn't make a difference.
210 TmpVGPR = AMDGPU::VGPR0;
211 TmpVGPRLive = true;
212 }
213
214 if (TmpVGPRLive) {
215 // We need to inform the scavenger that this index is already in use until
216 // we're done with the custom emergency spill.
217 RS->assignRegToScavengingIndex(TmpVGPRIndex, TmpVGPR);
218 }
219
220 // We may end up recursively calling the scavenger, and don't want to re-use
221 // the same register.
222 RS->setRegUsed(TmpVGPR);
223
224 // Try to scavenge SGPRs to save exec
225 assert(!SavedExecReg && "Exec is already saved, refuse to save again");
226 const TargetRegisterClass &RC =
227 IsWave32 ? AMDGPU::SGPR_32RegClass : AMDGPU::SGPR_64RegClass;
228 RS->setRegUsed(SuperReg);
229 SavedExecReg = RS->scavengeRegisterBackwards(RC, MI, false, 0, false);
230
231 int64_t VGPRLanes = getPerVGPRData().VGPRLanes;
232
233 if (SavedExecReg) {
234 RS->setRegUsed(SavedExecReg);
235 // Set exec to needed lanes
237 auto I =
238 BuildMI(*MBB, MI, DL, TII.get(MovOpc), ExecReg).addImm(VGPRLanes);
239 if (!TmpVGPRLive)
241 // Spill needed lanes
242 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ false);
243 } else {
244 // The modify and restore of exec clobber SCC, which we would have to save
245 // and restore. FIXME: We probably would need to reserve a register for
246 // this.
247 if (RS->isRegUsed(AMDGPU::SCC))
248 emitUnsupportedError(MF.getFunction(), *MI,
249 "unhandled SGPR spill to memory");
250
251 // Spill active lanes
252 if (TmpVGPRLive)
253 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ false,
254 /*IsKill*/ false);
255 // Spill inactive lanes
256 auto I = BuildMI(*MBB, MI, DL, TII.get(NotOpc), ExecReg).addReg(ExecReg);
257 if (!TmpVGPRLive)
259 I->getOperand(2).setIsDead(); // Mark SCC as dead.
260 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ false);
261 }
262 }
263
264 // Writes these instructions if an SGPR can be scavenged:
265 // buffer_load_dword v1 ; Write scavenged VGPR to emergency slot
266 // s_waitcnt vmcnt(0) ; If a free VGPR was found
267 // s_mov_b64 exec, s[6:7] ; Save exec
268 //
269 // Writes these instructions if no SGPR can be scavenged:
270 // buffer_load_dword v0 ; Restore inactive lanes
271 // s_waitcnt vmcnt(0) ; If a free VGPR was found
272 // s_not_b64 exec, exec
273 // buffer_load_dword v0 ; Only if no free VGPR was found
274 void restore() {
275 if (SavedExecReg) {
276 // Restore used lanes
277 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ true,
278 /*IsKill*/ false);
279 // Restore exec
280 auto I = BuildMI(*MBB, MI, DL, TII.get(MovOpc), ExecReg)
282 // Add an implicit use of the load so it is not dead.
283 // FIXME This inserts an unnecessary waitcnt
284 if (!TmpVGPRLive) {
286 }
287 } else {
288 // Restore inactive lanes
289 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ true,
290 /*IsKill*/ false);
291 auto I = BuildMI(*MBB, MI, DL, TII.get(NotOpc), ExecReg).addReg(ExecReg);
292 if (!TmpVGPRLive)
294 I->getOperand(2).setIsDead(); // Mark SCC as dead.
295
296 // Restore active lanes
297 if (TmpVGPRLive)
298 TRI.buildVGPRSpillLoadStore(*this, TmpVGPRIndex, 0, /*IsLoad*/ true);
299 }
300
301 // Inform the scavenger where we're releasing our custom scavenged register.
302 if (TmpVGPRLive) {
303 MachineBasicBlock::iterator RestorePt = std::prev(MI);
304 RS->assignRegToScavengingIndex(TmpVGPRIndex, TmpVGPR, &*RestorePt);
305 }
306 }
307
308 // Write TmpVGPR to memory or read TmpVGPR from memory.
309 // Either using a single buffer_load/store if exec is set to the needed mask
310 // or using
311 // buffer_load
312 // s_not exec, exec
313 // buffer_load
314 // s_not exec, exec
315 void readWriteTmpVGPR(unsigned Offset, bool IsLoad) {
316 if (SavedExecReg) {
317 // Spill needed lanes
318 TRI.buildVGPRSpillLoadStore(*this, Index, Offset, IsLoad);
319 } else {
320 // The modify and restore of exec clobber SCC, which we would have to save
321 // and restore. FIXME: We probably would need to reserve a register for
322 // this.
323 if (RS->isRegUsed(AMDGPU::SCC))
324 emitUnsupportedError(MF.getFunction(), *MI,
325 "unhandled SGPR spill to memory");
326
327 // Spill active lanes
328 TRI.buildVGPRSpillLoadStore(*this, Index, Offset, IsLoad,
329 /*IsKill*/ false);
330 // Spill inactive lanes
331 auto Not0 = BuildMI(*MBB, MI, DL, TII.get(NotOpc), ExecReg).addReg(ExecReg);
332 Not0->getOperand(2).setIsDead(); // Mark SCC as dead.
333 TRI.buildVGPRSpillLoadStore(*this, Index, Offset, IsLoad);
334 auto Not1 = BuildMI(*MBB, MI, DL, TII.get(NotOpc), ExecReg).addReg(ExecReg);
335 Not1->getOperand(2).setIsDead(); // Mark SCC as dead.
336 }
337 }
338
340 assert(MBB->getParent() == &MF);
341 MI = NewMI;
342 MBB = NewMBB;
343 }
344};
345
346} // namespace llvm
347
349 : AMDGPUGenRegisterInfo(AMDGPU::PC_REG, ST.getAMDGPUDwarfFlavour(),
350 ST.getAMDGPUDwarfFlavour(),
351 /*PC=*/0,
352 ST.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
353 ST(ST), SpillSGPRToVGPR(EnableSpillSGPRToVGPR), isWave32(ST.isWave32()) {
354
355 assert(getSubRegIndexLaneMask(AMDGPU::sub0).getAsInteger() == 3 &&
356 getSubRegIndexLaneMask(AMDGPU::sub31).getAsInteger() == (3ULL << 62) &&
357 (getSubRegIndexLaneMask(AMDGPU::lo16) |
358 getSubRegIndexLaneMask(AMDGPU::hi16)).getAsInteger() ==
359 getSubRegIndexLaneMask(AMDGPU::sub0).getAsInteger() &&
360 "getNumCoveredRegs() will not work with generated subreg masks!");
361
362 RegPressureIgnoredUnits.resize(getNumRegUnits());
363 RegPressureIgnoredUnits.set(
364 static_cast<unsigned>(*regunits(MCRegister::from(AMDGPU::M0)).begin()));
365 for (auto Reg : AMDGPU::VGPR_16RegClass) {
366 if (AMDGPU::isHi16Reg(Reg, *this))
367 RegPressureIgnoredUnits.set(
368 static_cast<unsigned>(*regunits(Reg).begin()));
369 }
370
371 // HACK: Until this is fully tablegen'd.
372 static llvm::once_flag InitializeRegSplitPartsFlag;
373
374 static auto InitializeRegSplitPartsOnce = [this]() {
375 for (unsigned Idx = 1, E = getNumSubRegIndices() - 1; Idx < E; ++Idx) {
376 unsigned Size = getSubRegIdxSize(Idx);
377 if (Size & 15)
378 continue;
379 std::vector<int16_t> &Vec = RegSplitParts[Size / 16 - 1];
380 unsigned Pos = getSubRegIdxOffset(Idx);
381 if (Pos % Size)
382 continue;
383 Pos /= Size;
384 if (Vec.empty()) {
385 unsigned MaxNumParts = 1024 / Size; // Maximum register is 1024 bits.
386 Vec.resize(MaxNumParts);
387 }
388 Vec[Pos] = Idx;
389 }
390 };
391
392 static llvm::once_flag InitializeSubRegFromChannelTableFlag;
393
394 static auto InitializeSubRegFromChannelTableOnce = [this]() {
395 for (auto &Row : SubRegFromChannelTable)
396 Row.fill(AMDGPU::NoSubRegister);
397 for (unsigned Idx = 1; Idx < getNumSubRegIndices(); ++Idx) {
398 unsigned Width = getSubRegIdxSize(Idx) / 32;
399 unsigned Offset = getSubRegIdxOffset(Idx) / 32;
401 Width = SubRegFromChannelTableWidthMap[Width];
402 if (Width == 0)
403 continue;
404 unsigned TableIdx = Width - 1;
405 assert(TableIdx < SubRegFromChannelTable.size());
406 assert(Offset < SubRegFromChannelTable[TableIdx].size());
407 SubRegFromChannelTable[TableIdx][Offset] = Idx;
408 }
409 };
410
411 llvm::call_once(InitializeRegSplitPartsFlag, InitializeRegSplitPartsOnce);
412 llvm::call_once(InitializeSubRegFromChannelTableFlag,
413 InitializeSubRegFromChannelTableOnce);
414}
415
416void SIRegisterInfo::reserveRegisterTuples(BitVector &Reserved,
417 MCRegister Reg) const {
418 for (MCRegAliasIterator R(Reg, this, true); R.isValid(); ++R)
419 Reserved.set(*R);
420}
421
422// Forced to be here by one .inc
424 const MachineFunction *MF) const {
426 switch (CC) {
427 case CallingConv::C:
430 return ST.hasGFX90AInsts() ? CSR_AMDGPU_GFX90AInsts_SaveList
431 : CSR_AMDGPU_SaveList;
434 return ST.hasGFX90AInsts() ? CSR_AMDGPU_SI_Gfx_GFX90AInsts_SaveList
435 : CSR_AMDGPU_SI_Gfx_SaveList;
437 return CSR_AMDGPU_CS_ChainPreserve_SaveList;
438 default: {
439 // Dummy to not crash RegisterClassInfo.
440 static const MCPhysReg NoCalleeSavedReg = AMDGPU::NoRegister;
441 return &NoCalleeSavedReg;
442 }
443 }
444}
445
446const MCPhysReg *
448 return nullptr;
449}
450
452 CallingConv::ID CC) const {
453 switch (CC) {
454 case CallingConv::C:
457 return ST.hasGFX90AInsts() ? CSR_AMDGPU_GFX90AInsts_RegMask
458 : CSR_AMDGPU_RegMask;
461 return ST.hasGFX90AInsts() ? CSR_AMDGPU_SI_Gfx_GFX90AInsts_RegMask
462 : CSR_AMDGPU_SI_Gfx_RegMask;
465 // Calls to these functions never return, so we can pretend everything is
466 // preserved.
467 return AMDGPU_AllVGPRs_RegMask;
468 default:
469 return nullptr;
470 }
471}
472
474 return CSR_AMDGPU_NoRegs_RegMask;
475}
476
478 return VGPR >= AMDGPU::VGPR0 && VGPR < AMDGPU::VGPR8;
479}
480
483 const MachineFunction &MF) const {
484 // FIXME: Should have a helper function like getEquivalentVGPRClass to get the
485 // equivalent AV class. If used one, the verifier will crash after
486 // RegBankSelect in the GISel flow. The aligned regclasses are not fully given
487 // until Instruction selection.
488 if (ST.hasMAIInsts() && (isVGPRClass(RC) || isAGPRClass(RC))) {
489 if (RC == &AMDGPU::VGPR_32RegClass || RC == &AMDGPU::AGPR_32RegClass)
490 return &AMDGPU::AV_32RegClass;
491 if (RC == &AMDGPU::VReg_64RegClass || RC == &AMDGPU::AReg_64RegClass)
492 return &AMDGPU::AV_64RegClass;
493 if (RC == &AMDGPU::VReg_64_Align2RegClass ||
494 RC == &AMDGPU::AReg_64_Align2RegClass)
495 return &AMDGPU::AV_64_Align2RegClass;
496 if (RC == &AMDGPU::VReg_96RegClass || RC == &AMDGPU::AReg_96RegClass)
497 return &AMDGPU::AV_96RegClass;
498 if (RC == &AMDGPU::VReg_96_Align2RegClass ||
499 RC == &AMDGPU::AReg_96_Align2RegClass)
500 return &AMDGPU::AV_96_Align2RegClass;
501 if (RC == &AMDGPU::VReg_128RegClass || RC == &AMDGPU::AReg_128RegClass)
502 return &AMDGPU::AV_128RegClass;
503 if (RC == &AMDGPU::VReg_128_Align2RegClass ||
504 RC == &AMDGPU::AReg_128_Align2RegClass)
505 return &AMDGPU::AV_128_Align2RegClass;
506 if (RC == &AMDGPU::VReg_160RegClass || RC == &AMDGPU::AReg_160RegClass)
507 return &AMDGPU::AV_160RegClass;
508 if (RC == &AMDGPU::VReg_160_Align2RegClass ||
509 RC == &AMDGPU::AReg_160_Align2RegClass)
510 return &AMDGPU::AV_160_Align2RegClass;
511 if (RC == &AMDGPU::VReg_192RegClass || RC == &AMDGPU::AReg_192RegClass)
512 return &AMDGPU::AV_192RegClass;
513 if (RC == &AMDGPU::VReg_192_Align2RegClass ||
514 RC == &AMDGPU::AReg_192_Align2RegClass)
515 return &AMDGPU::AV_192_Align2RegClass;
516 if (RC == &AMDGPU::VReg_256RegClass || RC == &AMDGPU::AReg_256RegClass)
517 return &AMDGPU::AV_256RegClass;
518 if (RC == &AMDGPU::VReg_256_Align2RegClass ||
519 RC == &AMDGPU::AReg_256_Align2RegClass)
520 return &AMDGPU::AV_256_Align2RegClass;
521 if (RC == &AMDGPU::VReg_512RegClass || RC == &AMDGPU::AReg_512RegClass)
522 return &AMDGPU::AV_512RegClass;
523 if (RC == &AMDGPU::VReg_512_Align2RegClass ||
524 RC == &AMDGPU::AReg_512_Align2RegClass)
525 return &AMDGPU::AV_512_Align2RegClass;
526 if (RC == &AMDGPU::VReg_1024RegClass || RC == &AMDGPU::AReg_1024RegClass)
527 return &AMDGPU::AV_1024RegClass;
528 if (RC == &AMDGPU::VReg_1024_Align2RegClass ||
529 RC == &AMDGPU::AReg_1024_Align2RegClass)
530 return &AMDGPU::AV_1024_Align2RegClass;
531 }
532
534}
535
537 const SIFrameLowering *TFI = ST.getFrameLowering();
539
540 // During ISel lowering we always reserve the stack pointer in entry and chain
541 // functions, but never actually want to reference it when accessing our own
542 // frame. If we need a frame pointer we use it, but otherwise we can just use
543 // an immediate "0" which we represent by returning NoRegister.
544 if (FuncInfo->isBottomOfStack()) {
545 return TFI->hasFP(MF) ? FuncInfo->getFrameOffsetReg() : Register();
546 }
547 return TFI->hasFP(MF) ? FuncInfo->getFrameOffsetReg()
548 : FuncInfo->getStackPtrOffsetReg();
549}
550
552 // When we need stack realignment, we can't reference off of the
553 // stack pointer, so we reserve a base pointer.
554 return shouldRealignStack(MF);
555}
556
557Register SIRegisterInfo::getBaseRegister() const { return AMDGPU::SGPR34; }
558
560 return AMDGPU_AllVGPRs_RegMask;
561}
562
564 return AMDGPU_AllAGPRs_RegMask;
565}
566
568 return AMDGPU_AllVectorRegs_RegMask;
569}
570
571unsigned SIRegisterInfo::getSubRegFromChannel(unsigned Channel,
572 unsigned NumRegs) {
573 assert(NumRegs < SubRegFromChannelTableWidthMap.size());
574 unsigned NumRegIndex = SubRegFromChannelTableWidthMap[NumRegs];
575 assert(NumRegIndex && "Not implemented");
576 assert(Channel < SubRegFromChannelTable[NumRegIndex - 1].size());
577 return SubRegFromChannelTable[NumRegIndex - 1][Channel];
578}
579
583
586 const unsigned Align,
587 const TargetRegisterClass *RC) const {
588 unsigned BaseIdx = alignDown(ST.getMaxNumSGPRs(MF), Align) - Align;
589 MCRegister BaseReg(AMDGPU::SGPR_32RegClass.getRegister(BaseIdx));
590 return getMatchingSuperReg(BaseReg, AMDGPU::sub0, RC);
591}
592
594 const MachineFunction &MF) const {
595 return getAlignedHighSGPRForRC(MF, /*Align=*/4, &AMDGPU::SGPR_128RegClass);
596}
597
599 BitVector Reserved(getNumRegs());
600 Reserved.set(AMDGPU::MODE);
601
603
604 // Reserve special purpose registers.
605 //
606 // EXEC_LO and EXEC_HI could be allocated and used as regular register, but
607 // this seems likely to result in bugs, so I'm marking them as reserved.
608 reserveRegisterTuples(Reserved, AMDGPU::EXEC);
609 reserveRegisterTuples(Reserved, AMDGPU::FLAT_SCR);
610
611 // M0 has to be reserved so that llvm accepts it as a live-in into a block.
612 reserveRegisterTuples(Reserved, AMDGPU::M0);
613
614 // Reserve src_vccz, src_execz, src_scc.
615 reserveRegisterTuples(Reserved, AMDGPU::SRC_VCCZ);
616 reserveRegisterTuples(Reserved, AMDGPU::SRC_EXECZ);
617 reserveRegisterTuples(Reserved, AMDGPU::SRC_SCC);
618
619 // Reserve the memory aperture registers
620 reserveRegisterTuples(Reserved, AMDGPU::SRC_SHARED_BASE);
621 reserveRegisterTuples(Reserved, AMDGPU::SRC_SHARED_LIMIT);
622 reserveRegisterTuples(Reserved, AMDGPU::SRC_PRIVATE_BASE);
623 reserveRegisterTuples(Reserved, AMDGPU::SRC_PRIVATE_LIMIT);
624 reserveRegisterTuples(Reserved, AMDGPU::SRC_FLAT_SCRATCH_BASE_LO);
625 reserveRegisterTuples(Reserved, AMDGPU::SRC_FLAT_SCRATCH_BASE_HI);
626
627 // Reserve async counters pseudo registers
628 reserveRegisterTuples(Reserved, AMDGPU::ASYNCcnt);
629 reserveRegisterTuples(Reserved, AMDGPU::TENSORcnt);
630
631 // Reserve src_pops_exiting_wave_id - support is not implemented in Codegen.
632 reserveRegisterTuples(Reserved, AMDGPU::SRC_POPS_EXITING_WAVE_ID);
633
634 // Reserve xnack_mask registers - support is not implemented in Codegen.
635 reserveRegisterTuples(Reserved, AMDGPU::XNACK_MASK);
636
637 // Reserve lds_direct register - support is not implemented in Codegen.
638 reserveRegisterTuples(Reserved, AMDGPU::LDS_DIRECT);
639
640 // Reserve Trap Handler registers - support is not implemented in Codegen.
641 reserveRegisterTuples(Reserved, AMDGPU::TBA);
642 reserveRegisterTuples(Reserved, AMDGPU::TMA);
643 reserveRegisterTuples(Reserved, AMDGPU::TTMP0_TTMP1);
644 reserveRegisterTuples(Reserved, AMDGPU::TTMP2_TTMP3);
645 reserveRegisterTuples(Reserved, AMDGPU::TTMP4_TTMP5);
646 reserveRegisterTuples(Reserved, AMDGPU::TTMP6_TTMP7);
647 reserveRegisterTuples(Reserved, AMDGPU::TTMP8_TTMP9);
648 reserveRegisterTuples(Reserved, AMDGPU::TTMP10_TTMP11);
649 reserveRegisterTuples(Reserved, AMDGPU::TTMP12_TTMP13);
650 reserveRegisterTuples(Reserved, AMDGPU::TTMP14_TTMP15);
651
652 // Reserve null register - it shall never be allocated
653 reserveRegisterTuples(Reserved, AMDGPU::SGPR_NULL64);
654
655 // Reserve SGPRs.
656 //
657 unsigned MaxNumSGPRs = ST.getMaxNumSGPRs(MF);
658 if (StressSGPRLimit.getNumOccurrences() && StressSGPRLimit < MaxNumSGPRs)
659 MaxNumSGPRs = StressSGPRLimit;
660 unsigned TotalNumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
661 for (const TargetRegisterClass &RC : regclasses()) {
662 if (RC.isBaseClass() && isSGPRClass(&RC)) {
663 unsigned NumRegs = divideCeil(getRegSizeInBits(RC), 32);
664 for (MCPhysReg Reg : RC) {
665 unsigned Index = getHWRegIndex(Reg);
666 if (Index + NumRegs > MaxNumSGPRs && Index < TotalNumSGPRs &&
667 Reg != AMDGPU::VCC_LO && Reg != AMDGPU::VCC_HI &&
668 Reg != AMDGPU::VCC)
669 Reserved.set(Reg);
670 }
671 }
672 }
673
674 Register ScratchRSrcReg = MFI->getScratchRSrcReg();
675 if (ScratchRSrcReg != AMDGPU::NoRegister) {
676 // Reserve 4 SGPRs for the scratch buffer resource descriptor in case we
677 // need to spill.
678 // TODO: May need to reserve a VGPR if doing LDS spilling.
679 reserveRegisterTuples(Reserved, ScratchRSrcReg);
680 }
681
682 Register LongBranchReservedReg = MFI->getLongBranchReservedReg();
683 if (LongBranchReservedReg)
684 reserveRegisterTuples(Reserved, LongBranchReservedReg);
685
686 // We have to assume the SP is needed in case there are calls in the function,
687 // which is detected after the function is lowered. If we aren't really going
688 // to need SP, don't bother reserving it.
689 MCRegister StackPtrReg = MFI->getStackPtrOffsetReg();
690 if (StackPtrReg) {
691 reserveRegisterTuples(Reserved, StackPtrReg);
692 assert(!isSubRegister(ScratchRSrcReg, StackPtrReg));
693 }
694
695 MCRegister FrameReg = MFI->getFrameOffsetReg();
696 if (FrameReg) {
697 reserveRegisterTuples(Reserved, FrameReg);
698 assert(!isSubRegister(ScratchRSrcReg, FrameReg));
699 }
700
701 if (hasBasePointer(MF)) {
702 MCRegister BasePtrReg = getBaseRegister();
703 reserveRegisterTuples(Reserved, BasePtrReg);
704 assert(!isSubRegister(ScratchRSrcReg, BasePtrReg));
705 }
706
707 // FIXME: Use same reserved register introduced in D149775
708 // SGPR used to preserve EXEC MASK around WWM spill/copy instructions.
709 Register ExecCopyReg = MFI->getSGPRForEXECCopy();
710 if (ExecCopyReg)
711 reserveRegisterTuples(Reserved, ExecCopyReg);
712
713 // Reserve VGPRs/AGPRs.
714 //
715 auto [MaxNumVGPRs, MaxNumAGPRs] = ST.getMaxNumVectorRegs(MF.getFunction());
716
717 // Stress test: override VGPR/AGPR limits.
718 if (StressVGPRLimit.getNumOccurrences() && StressVGPRLimit < MaxNumVGPRs)
719 MaxNumVGPRs = StressVGPRLimit;
720 if (StressAGPRLimit.getNumOccurrences() && StressAGPRLimit < MaxNumAGPRs)
721 MaxNumAGPRs = StressAGPRLimit;
722
723 for (const TargetRegisterClass &RC : regclasses()) {
724 if (RC.isBaseClass() && isVGPRClass(&RC)) {
725 unsigned NumRegs = divideCeil(getRegSizeInBits(RC), 32);
726 for (MCPhysReg Reg : RC) {
727 unsigned Index = getHWRegIndex(Reg);
728 if (Index + NumRegs > MaxNumVGPRs)
729 Reserved.set(Reg);
730 }
731 }
732 }
733
734 // Reserve all the AGPRs if there are no instructions to use it.
735 if (!ST.hasMAIInsts())
736 MaxNumAGPRs = 0;
737 for (const TargetRegisterClass &RC : regclasses()) {
738 if (RC.isBaseClass() && isAGPRClass(&RC)) {
739 unsigned NumRegs = divideCeil(getRegSizeInBits(RC), 32);
740 for (MCPhysReg Reg : RC) {
741 unsigned Index = getHWRegIndex(Reg);
742 if (Index + NumRegs > MaxNumAGPRs)
743 Reserved.set(Reg);
744 }
745 }
746 }
747
748 // On GFX908, in order to guarantee copying between AGPRs, we need a scratch
749 // VGPR available at all times.
750 if (ST.hasMAIInsts() && !ST.hasGFX90AInsts()) {
751 reserveRegisterTuples(Reserved, MFI->getVGPRForAGPRCopy());
752 }
753
754 // During wwm-regalloc, reserve the registers for per-lane VGPR allocation.
755 // The MFI->getPerLaneVGPRMask() field will have a valid bitmask only during
756 // wwm-regalloc and it would be empty otherwise.
757 BitVector PerLaneVGPRMask = MFI->getPerLaneVGPRMask();
758 if (!PerLaneVGPRMask.empty()) {
759 for (unsigned RegI = AMDGPU::VGPR0, RegE = AMDGPU::VGPR0 + MaxNumVGPRs;
760 RegI < RegE; ++RegI) {
761 if (PerLaneVGPRMask.test(RegI))
762 reserveRegisterTuples(Reserved, RegI);
763 }
764 }
765
766 for (Register Reg : MFI->getWWMReservedRegs())
767 reserveRegisterTuples(Reserved, Reg);
768
769 // FIXME: Stop using reserved registers for this.
770 for (MCPhysReg Reg : MFI->getAGPRSpillVGPRs())
771 reserveRegisterTuples(Reserved, Reg);
772
773 for (MCPhysReg Reg : MFI->getVGPRSpillAGPRs())
774 reserveRegisterTuples(Reserved, Reg);
775
776 return Reserved;
777}
778
780 MCRegister PhysReg) const {
781 return !MF.getRegInfo().isReserved(PhysReg);
782}
783
786 // On entry or in chain functions, the base address is 0, so it can't possibly
787 // need any more alignment.
788
789 // FIXME: Should be able to specify the entry frame alignment per calling
790 // convention instead.
791 if (Info->isBottomOfStack())
792 return false;
793
795}
796
799 if (Info->isEntryFunction()) {
800 const MachineFrameInfo &MFI = Fn.getFrameInfo();
801 return MFI.hasStackObjects() || MFI.hasCalls();
802 }
803
804 // May need scavenger for dealing with callee saved registers.
805 return true;
806}
807
809 const MachineFunction &MF) const {
810 // Do not use frame virtual registers. They used to be used for SGPRs, but
811 // once we reach PrologEpilogInserter, we can no longer spill SGPRs. If the
812 // scavenger fails, we can increment/decrement the necessary SGPRs to avoid a
813 // spill.
814 return false;
815}
816
818 const MachineFunction &MF) const {
819 const MachineFrameInfo &MFI = MF.getFrameInfo();
820 return MFI.hasStackObjects();
821}
822
824 const MachineFunction &) const {
825 // There are no special dedicated stack or frame pointers.
826 return true;
827}
828
831
832 int OffIdx = AMDGPU::getNamedOperandIdx(MI->getOpcode(),
833 AMDGPU::OpName::offset);
834 return MI->getOperand(OffIdx).getImm();
835}
836
838 int Idx) const {
839 switch (MI->getOpcode()) {
840 case AMDGPU::V_ADD_U32_e32:
841 case AMDGPU::V_ADD_U32_e64:
842 case AMDGPU::V_ADD_CO_U32_e32: {
843 int OtherIdx = Idx == 1 ? 2 : 1;
844 const MachineOperand &OtherOp = MI->getOperand(OtherIdx);
845 return OtherOp.isImm() ? OtherOp.getImm() : 0;
846 }
847 case AMDGPU::V_ADD_CO_U32_e64: {
848 int OtherIdx = Idx == 2 ? 3 : 2;
849 const MachineOperand &OtherOp = MI->getOperand(OtherIdx);
850 return OtherOp.isImm() ? OtherOp.getImm() : 0;
851 }
852 default:
853 break;
854 }
855
857 return 0;
858
859 assert((Idx == AMDGPU::getNamedOperandIdx(MI->getOpcode(),
860 AMDGPU::OpName::vaddr) ||
861 (Idx == AMDGPU::getNamedOperandIdx(MI->getOpcode(),
862 AMDGPU::OpName::saddr))) &&
863 "Should never see frame index on non-address operand");
864
866}
867
869 const MachineInstr &MI) {
870 assert(MI.getDesc().isAdd());
871 const MachineOperand &Src0 = MI.getOperand(1);
872 const MachineOperand &Src1 = MI.getOperand(2);
873
874 if (Src0.isFI()) {
875 return Src1.isImm() || (Src1.isReg() && TRI.isVGPR(MI.getMF()->getRegInfo(),
876 Src1.getReg()));
877 }
878
879 if (Src1.isFI()) {
880 return Src0.isImm() || (Src0.isReg() && TRI.isVGPR(MI.getMF()->getRegInfo(),
881 Src0.getReg()));
882 }
883
884 return false;
885}
886
888 // TODO: Handle v_add_co_u32, v_or_b32, v_and_b32 and scalar opcodes.
889 switch (MI->getOpcode()) {
890 case AMDGPU::V_ADD_U32_e32: {
891 // TODO: We could handle this but it requires work to avoid violating
892 // operand restrictions.
893 if (ST.getConstantBusLimit(AMDGPU::V_ADD_U32_e32) < 2 &&
894 !isFIPlusImmOrVGPR(*this, *MI))
895 return false;
896 [[fallthrough]];
897 }
898 case AMDGPU::V_ADD_U32_e64:
899 // FIXME: This optimization is barely profitable hasFlatScratchEnabled
900 // as-is.
901 //
902 // Much of the benefit with the MUBUF handling is we avoid duplicating the
903 // shift of the frame register, which isn't needed with scratch.
904 //
905 // materializeFrameBaseRegister doesn't know the register classes of the
906 // uses, and unconditionally uses an s_add_i32, which will end up using a
907 // copy for the vector uses.
908 return !ST.hasFlatScratchEnabled();
909 case AMDGPU::V_ADD_CO_U32_e32:
910 if (ST.getConstantBusLimit(AMDGPU::V_ADD_CO_U32_e32) < 2 &&
911 !isFIPlusImmOrVGPR(*this, *MI))
912 return false;
913 // We can't deal with the case where the carry out has a use (though this
914 // should never happen)
915 return MI->getOperand(3).isDead();
916 case AMDGPU::V_ADD_CO_U32_e64:
917 // TODO: Should we check use_empty instead?
918 return MI->getOperand(1).isDead();
919 default:
920 break;
921 }
922
924 return false;
925
926 int64_t FullOffset = Offset + getScratchInstrOffset(MI);
927
928 const SIInstrInfo *TII = ST.getInstrInfo();
930 return !TII->isLegalMUBUFImmOffset(FullOffset);
931
932 return !TII->isLegalFLATOffset(FullOffset, AMDGPUAS::PRIVATE_ADDRESS,
934}
935
937 int FrameIdx,
938 int64_t Offset) const {
939 MachineBasicBlock::iterator Ins = MBB->begin();
940 DebugLoc DL; // Defaults to "unknown"
941
942 if (Ins != MBB->end())
943 DL = Ins->getDebugLoc();
944
945 MachineFunction *MF = MBB->getParent();
946 const SIInstrInfo *TII = ST.getInstrInfo();
947 MachineRegisterInfo &MRI = MF->getRegInfo();
948 unsigned MovOpc =
949 ST.hasFlatScratchEnabled() ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
950
951 Register BaseReg = MRI.createVirtualRegister(
952 ST.hasFlatScratchEnabled() ? &AMDGPU::SReg_32_XEXEC_HIRegClass
953 : &AMDGPU::VGPR_32RegClass);
954
955 if (Offset == 0) {
956 BuildMI(*MBB, Ins, DL, TII->get(MovOpc), BaseReg)
957 .addFrameIndex(FrameIdx);
958 return BaseReg;
959 }
960
961 Register OffsetReg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
962
963 Register FIReg = MRI.createVirtualRegister(ST.hasFlatScratchEnabled()
964 ? &AMDGPU::SReg_32_XM0RegClass
965 : &AMDGPU::VGPR_32RegClass);
966
967 BuildMI(*MBB, Ins, DL, TII->get(AMDGPU::S_MOV_B32), OffsetReg)
968 .addImm(Offset);
969 BuildMI(*MBB, Ins, DL, TII->get(MovOpc), FIReg)
970 .addFrameIndex(FrameIdx);
971
972 if (ST.hasFlatScratchEnabled()) {
973 // FIXME: Make sure scc isn't live in.
974 BuildMI(*MBB, Ins, DL, TII->get(AMDGPU::S_ADD_I32), BaseReg)
975 .addReg(OffsetReg, RegState::Kill)
976 .addReg(FIReg)
977 .setOperandDead(3); // scc
978 return BaseReg;
979 }
980
981 TII->getAddNoCarry(*MBB, Ins, DL, BaseReg)
982 .addReg(OffsetReg, RegState::Kill)
983 .addReg(FIReg)
984 .addImm(0); // clamp bit
985
986 return BaseReg;
987}
988
990 int64_t Offset) const {
991 const SIInstrInfo *TII = ST.getInstrInfo();
992
993 switch (MI.getOpcode()) {
994 case AMDGPU::V_ADD_U32_e32:
995 case AMDGPU::V_ADD_CO_U32_e32: {
996 MachineOperand *FIOp = &MI.getOperand(2);
997 MachineOperand *ImmOp = &MI.getOperand(1);
998 if (!FIOp->isFI())
999 std::swap(FIOp, ImmOp);
1000
1001 if (!ImmOp->isImm()) {
1002 assert(Offset == 0);
1003 FIOp->ChangeToRegister(BaseReg, false);
1004 TII->legalizeOperandsVOP2(MI.getMF()->getRegInfo(), MI);
1005 return;
1006 }
1007
1008 int64_t TotalOffset = ImmOp->getImm() + Offset;
1009 if (TotalOffset == 0) {
1010 MI.setDesc(TII->get(AMDGPU::COPY));
1011 for (unsigned I = MI.getNumOperands() - 1; I != 1; --I)
1012 MI.removeOperand(I);
1013
1014 MI.getOperand(1).ChangeToRegister(BaseReg, false);
1015 return;
1016 }
1017
1018 ImmOp->setImm(TotalOffset);
1019
1020 MachineBasicBlock *MBB = MI.getParent();
1021 MachineFunction *MF = MBB->getParent();
1022 MachineRegisterInfo &MRI = MF->getRegInfo();
1023
1024 // FIXME: materializeFrameBaseRegister does not know the register class of
1025 // the uses of the frame index, and assumes SGPR for hasFlatScratchEnabled.
1026 // Emit a copy so we have a legal operand and hope the register coalescer
1027 // can clean it up.
1028 if (isSGPRReg(MRI, BaseReg)) {
1029 Register BaseRegVGPR =
1030 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1031 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), BaseRegVGPR)
1032 .addReg(BaseReg);
1033 MI.getOperand(2).ChangeToRegister(BaseRegVGPR, false);
1034 } else {
1035 MI.getOperand(2).ChangeToRegister(BaseReg, false);
1036 }
1037 return;
1038 }
1039 case AMDGPU::V_ADD_U32_e64:
1040 case AMDGPU::V_ADD_CO_U32_e64: {
1041 int Src0Idx = MI.getNumExplicitDefs();
1042 MachineOperand *FIOp = &MI.getOperand(Src0Idx);
1043 MachineOperand *ImmOp = &MI.getOperand(Src0Idx + 1);
1044 if (!FIOp->isFI())
1045 std::swap(FIOp, ImmOp);
1046
1047 if (!ImmOp->isImm()) {
1048 FIOp->ChangeToRegister(BaseReg, false);
1049 TII->legalizeOperandsVOP3(MI.getMF()->getRegInfo(), MI);
1050 return;
1051 }
1052
1053 int64_t TotalOffset = ImmOp->getImm() + Offset;
1054 if (TotalOffset == 0) {
1055 MI.setDesc(TII->get(AMDGPU::COPY));
1056
1057 for (unsigned I = MI.getNumOperands() - 1; I != 1; --I)
1058 MI.removeOperand(I);
1059
1060 MI.getOperand(1).ChangeToRegister(BaseReg, false);
1061 } else {
1062 FIOp->ChangeToRegister(BaseReg, false);
1063 ImmOp->setImm(TotalOffset);
1064 }
1065
1066 return;
1067 }
1068 default:
1069 break;
1070 }
1071
1072 bool IsFlat = TII->isFLATScratch(MI);
1073
1074#ifndef NDEBUG
1075 // FIXME: Is it possible to be storing a frame index to itself?
1076 bool SeenFI = false;
1077 for (const MachineOperand &MO: MI.operands()) {
1078 if (MO.isFI()) {
1079 if (SeenFI)
1080 llvm_unreachable("should not see multiple frame indices");
1081
1082 SeenFI = true;
1083 }
1084 }
1085#endif
1086
1087 MachineOperand *FIOp =
1088 TII->getNamedOperand(MI, IsFlat ? AMDGPU::OpName::saddr
1089 : AMDGPU::OpName::vaddr);
1090
1091 MachineOperand *OffsetOp = TII->getNamedOperand(MI, AMDGPU::OpName::offset);
1092 int64_t NewOffset = OffsetOp->getImm() + Offset;
1093
1094 assert(FIOp && FIOp->isFI() && "frame index must be address operand");
1095 assert(TII->isMUBUF(MI) || TII->isFLATScratch(MI));
1096
1097 if (IsFlat) {
1098 assert(TII->isLegalFLATOffset(NewOffset, AMDGPUAS::PRIVATE_ADDRESS,
1100 "offset should be legal");
1101 FIOp->ChangeToRegister(BaseReg, false);
1102 OffsetOp->setImm(NewOffset);
1103 return;
1104 }
1105
1106#ifndef NDEBUG
1107 MachineOperand *SOffset = TII->getNamedOperand(MI, AMDGPU::OpName::soffset);
1108 assert(SOffset->isImm() && SOffset->getImm() == 0);
1109#endif
1110
1111 assert(TII->isLegalMUBUFImmOffset(NewOffset) && "offset should be legal");
1112
1113 FIOp->ChangeToRegister(BaseReg, false);
1114 OffsetOp->setImm(NewOffset);
1115}
1116
1118 Register BaseReg,
1119 int64_t Offset) const {
1120
1121 switch (MI->getOpcode()) {
1122 case AMDGPU::V_ADD_U32_e32:
1123 case AMDGPU::V_ADD_CO_U32_e32:
1124 return true;
1125 case AMDGPU::V_ADD_U32_e64:
1126 case AMDGPU::V_ADD_CO_U32_e64:
1127 return ST.hasVOP3Literal() || AMDGPU::isInlinableIntLiteral(Offset);
1128 default:
1129 break;
1130 }
1131
1133 return false;
1134
1135 int64_t NewOffset = Offset + getScratchInstrOffset(MI);
1136
1137 const SIInstrInfo *TII = ST.getInstrInfo();
1139 return TII->isLegalMUBUFImmOffset(NewOffset);
1140
1141 return TII->isLegalFLATOffset(NewOffset, AMDGPUAS::PRIVATE_ADDRESS,
1143}
1144
1145const TargetRegisterClass *
1147 return RC == &AMDGPU::SCC_CLASSRegClass ? &AMDGPU::SReg_32RegClass : RC;
1148}
1149
1151 const SIInstrInfo *TII) {
1152
1153 unsigned Op = MI.getOpcode();
1154 switch (Op) {
1155 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
1156 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
1157 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE:
1158 // FIXME: This assumes the mask is statically known and not computed at
1159 // runtime. However, some ABIs may want to compute the mask dynamically and
1160 // this will need to be updated.
1161 return llvm::popcount(
1162 (uint64_t)TII->getNamedOperand(MI, AMDGPU::OpName::mask)->getImm());
1163 case AMDGPU::SI_SPILL_S1024_SAVE:
1164 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
1165 case AMDGPU::SI_SPILL_S1024_RESTORE:
1166 case AMDGPU::SI_SPILL_V1024_SAVE:
1167 case AMDGPU::SI_SPILL_V1024_CFI_SAVE:
1168 case AMDGPU::SI_SPILL_V1024_RESTORE:
1169 case AMDGPU::SI_SPILL_A1024_SAVE:
1170 case AMDGPU::SI_SPILL_A1024_CFI_SAVE:
1171 case AMDGPU::SI_SPILL_A1024_RESTORE:
1172 case AMDGPU::SI_SPILL_AV1024_SAVE:
1173 case AMDGPU::SI_SPILL_AV1024_CFI_SAVE:
1174 case AMDGPU::SI_SPILL_AV1024_RESTORE:
1175 return 32;
1176 case AMDGPU::SI_SPILL_S512_SAVE:
1177 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
1178 case AMDGPU::SI_SPILL_S512_RESTORE:
1179 case AMDGPU::SI_SPILL_V512_SAVE:
1180 case AMDGPU::SI_SPILL_V512_CFI_SAVE:
1181 case AMDGPU::SI_SPILL_V512_RESTORE:
1182 case AMDGPU::SI_SPILL_A512_SAVE:
1183 case AMDGPU::SI_SPILL_A512_CFI_SAVE:
1184 case AMDGPU::SI_SPILL_A512_RESTORE:
1185 case AMDGPU::SI_SPILL_AV512_SAVE:
1186 case AMDGPU::SI_SPILL_AV512_CFI_SAVE:
1187 case AMDGPU::SI_SPILL_AV512_RESTORE:
1188 return 16;
1189 case AMDGPU::SI_SPILL_S384_SAVE:
1190 case AMDGPU::SI_SPILL_S384_RESTORE:
1191 case AMDGPU::SI_SPILL_V384_SAVE:
1192 case AMDGPU::SI_SPILL_V384_RESTORE:
1193 case AMDGPU::SI_SPILL_A384_SAVE:
1194 case AMDGPU::SI_SPILL_A384_RESTORE:
1195 case AMDGPU::SI_SPILL_AV384_SAVE:
1196 case AMDGPU::SI_SPILL_AV384_RESTORE:
1197 return 12;
1198 case AMDGPU::SI_SPILL_S352_SAVE:
1199 case AMDGPU::SI_SPILL_S352_RESTORE:
1200 case AMDGPU::SI_SPILL_V352_SAVE:
1201 case AMDGPU::SI_SPILL_V352_RESTORE:
1202 case AMDGPU::SI_SPILL_A352_SAVE:
1203 case AMDGPU::SI_SPILL_A352_RESTORE:
1204 case AMDGPU::SI_SPILL_AV352_SAVE:
1205 case AMDGPU::SI_SPILL_AV352_RESTORE:
1206 return 11;
1207 case AMDGPU::SI_SPILL_S320_SAVE:
1208 case AMDGPU::SI_SPILL_S320_RESTORE:
1209 case AMDGPU::SI_SPILL_V320_SAVE:
1210 case AMDGPU::SI_SPILL_V320_RESTORE:
1211 case AMDGPU::SI_SPILL_A320_SAVE:
1212 case AMDGPU::SI_SPILL_A320_RESTORE:
1213 case AMDGPU::SI_SPILL_AV320_SAVE:
1214 case AMDGPU::SI_SPILL_AV320_RESTORE:
1215 return 10;
1216 case AMDGPU::SI_SPILL_S288_SAVE:
1217 case AMDGPU::SI_SPILL_S288_RESTORE:
1218 case AMDGPU::SI_SPILL_V288_SAVE:
1219 case AMDGPU::SI_SPILL_V288_RESTORE:
1220 case AMDGPU::SI_SPILL_A288_SAVE:
1221 case AMDGPU::SI_SPILL_A288_RESTORE:
1222 case AMDGPU::SI_SPILL_AV288_SAVE:
1223 case AMDGPU::SI_SPILL_AV288_RESTORE:
1224 return 9;
1225 case AMDGPU::SI_SPILL_S256_SAVE:
1226 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
1227 case AMDGPU::SI_SPILL_S256_RESTORE:
1228 case AMDGPU::SI_SPILL_V256_SAVE:
1229 case AMDGPU::SI_SPILL_V256_CFI_SAVE:
1230 case AMDGPU::SI_SPILL_V256_RESTORE:
1231 case AMDGPU::SI_SPILL_A256_SAVE:
1232 case AMDGPU::SI_SPILL_A256_CFI_SAVE:
1233 case AMDGPU::SI_SPILL_A256_RESTORE:
1234 case AMDGPU::SI_SPILL_AV256_SAVE:
1235 case AMDGPU::SI_SPILL_AV256_CFI_SAVE:
1236 case AMDGPU::SI_SPILL_AV256_RESTORE:
1237 return 8;
1238 case AMDGPU::SI_SPILL_S224_SAVE:
1239 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
1240 case AMDGPU::SI_SPILL_S224_RESTORE:
1241 case AMDGPU::SI_SPILL_V224_SAVE:
1242 case AMDGPU::SI_SPILL_V224_CFI_SAVE:
1243 case AMDGPU::SI_SPILL_V224_RESTORE:
1244 case AMDGPU::SI_SPILL_A224_SAVE:
1245 case AMDGPU::SI_SPILL_A224_CFI_SAVE:
1246 case AMDGPU::SI_SPILL_A224_RESTORE:
1247 case AMDGPU::SI_SPILL_AV224_SAVE:
1248 case AMDGPU::SI_SPILL_AV224_CFI_SAVE:
1249 case AMDGPU::SI_SPILL_AV224_RESTORE:
1250 return 7;
1251 case AMDGPU::SI_SPILL_S192_SAVE:
1252 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
1253 case AMDGPU::SI_SPILL_S192_RESTORE:
1254 case AMDGPU::SI_SPILL_V192_SAVE:
1255 case AMDGPU::SI_SPILL_V192_CFI_SAVE:
1256 case AMDGPU::SI_SPILL_V192_RESTORE:
1257 case AMDGPU::SI_SPILL_A192_SAVE:
1258 case AMDGPU::SI_SPILL_A192_CFI_SAVE:
1259 case AMDGPU::SI_SPILL_A192_RESTORE:
1260 case AMDGPU::SI_SPILL_AV192_SAVE:
1261 case AMDGPU::SI_SPILL_AV192_CFI_SAVE:
1262 case AMDGPU::SI_SPILL_AV192_RESTORE:
1263 return 6;
1264 case AMDGPU::SI_SPILL_S160_SAVE:
1265 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
1266 case AMDGPU::SI_SPILL_S160_RESTORE:
1267 case AMDGPU::SI_SPILL_V160_SAVE:
1268 case AMDGPU::SI_SPILL_V160_CFI_SAVE:
1269 case AMDGPU::SI_SPILL_V160_RESTORE:
1270 case AMDGPU::SI_SPILL_A160_SAVE:
1271 case AMDGPU::SI_SPILL_A160_CFI_SAVE:
1272 case AMDGPU::SI_SPILL_A160_RESTORE:
1273 case AMDGPU::SI_SPILL_AV160_SAVE:
1274 case AMDGPU::SI_SPILL_AV160_CFI_SAVE:
1275 case AMDGPU::SI_SPILL_AV160_RESTORE:
1276 return 5;
1277 case AMDGPU::SI_SPILL_S128_SAVE:
1278 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
1279 case AMDGPU::SI_SPILL_S128_RESTORE:
1280 case AMDGPU::SI_SPILL_V128_SAVE:
1281 case AMDGPU::SI_SPILL_V128_CFI_SAVE:
1282 case AMDGPU::SI_SPILL_V128_RESTORE:
1283 case AMDGPU::SI_SPILL_A128_SAVE:
1284 case AMDGPU::SI_SPILL_A128_CFI_SAVE:
1285 case AMDGPU::SI_SPILL_A128_RESTORE:
1286 case AMDGPU::SI_SPILL_AV128_SAVE:
1287 case AMDGPU::SI_SPILL_AV128_CFI_SAVE:
1288 case AMDGPU::SI_SPILL_AV128_RESTORE:
1289 return 4;
1290 case AMDGPU::SI_SPILL_S96_SAVE:
1291 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
1292 case AMDGPU::SI_SPILL_S96_RESTORE:
1293 case AMDGPU::SI_SPILL_V96_SAVE:
1294 case AMDGPU::SI_SPILL_V96_CFI_SAVE:
1295 case AMDGPU::SI_SPILL_V96_RESTORE:
1296 case AMDGPU::SI_SPILL_A96_SAVE:
1297 case AMDGPU::SI_SPILL_A96_CFI_SAVE:
1298 case AMDGPU::SI_SPILL_A96_RESTORE:
1299 case AMDGPU::SI_SPILL_AV96_SAVE:
1300 case AMDGPU::SI_SPILL_AV96_CFI_SAVE:
1301 case AMDGPU::SI_SPILL_AV96_RESTORE:
1302 return 3;
1303 case AMDGPU::SI_SPILL_S64_SAVE:
1304 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
1305 case AMDGPU::SI_SPILL_S64_RESTORE:
1306 case AMDGPU::SI_SPILL_V64_SAVE:
1307 case AMDGPU::SI_SPILL_V64_CFI_SAVE:
1308 case AMDGPU::SI_SPILL_V64_RESTORE:
1309 case AMDGPU::SI_SPILL_A64_SAVE:
1310 case AMDGPU::SI_SPILL_A64_CFI_SAVE:
1311 case AMDGPU::SI_SPILL_A64_RESTORE:
1312 case AMDGPU::SI_SPILL_AV64_SAVE:
1313 case AMDGPU::SI_SPILL_AV64_CFI_SAVE:
1314 case AMDGPU::SI_SPILL_AV64_RESTORE:
1315 return 2;
1316 case AMDGPU::SI_SPILL_S32_SAVE:
1317 case AMDGPU::SI_SPILL_S32_CFI_SAVE:
1318 case AMDGPU::SI_SPILL_S32_RESTORE:
1319 case AMDGPU::SI_SPILL_V32_SAVE:
1320 case AMDGPU::SI_SPILL_V32_CFI_SAVE:
1321 case AMDGPU::SI_SPILL_V32_RESTORE:
1322 case AMDGPU::SI_SPILL_A32_SAVE:
1323 case AMDGPU::SI_SPILL_A32_CFI_SAVE:
1324 case AMDGPU::SI_SPILL_A32_RESTORE:
1325 case AMDGPU::SI_SPILL_AV32_SAVE:
1326 case AMDGPU::SI_SPILL_AV32_CFI_SAVE:
1327 case AMDGPU::SI_SPILL_AV32_RESTORE:
1328 case AMDGPU::SI_SPILL_WWM_V32_SAVE:
1329 case AMDGPU::SI_SPILL_WWM_V32_RESTORE:
1330 case AMDGPU::SI_SPILL_WWM_AV32_SAVE:
1331 case AMDGPU::SI_SPILL_WWM_AV32_RESTORE:
1332 case AMDGPU::SI_SPILL_V16_SAVE:
1333 case AMDGPU::SI_SPILL_V16_RESTORE:
1334 return 1;
1335 default: llvm_unreachable("Invalid spill opcode");
1336 }
1337}
1338
1339static int getOffsetMUBUFStore(unsigned Opc) {
1340 switch (Opc) {
1341 case AMDGPU::BUFFER_STORE_DWORD_OFFEN:
1342 return AMDGPU::BUFFER_STORE_DWORD_OFFSET;
1343 case AMDGPU::BUFFER_STORE_BYTE_OFFEN:
1344 return AMDGPU::BUFFER_STORE_BYTE_OFFSET;
1345 case AMDGPU::BUFFER_STORE_SHORT_OFFEN:
1346 return AMDGPU::BUFFER_STORE_SHORT_OFFSET;
1347 case AMDGPU::BUFFER_STORE_DWORDX2_OFFEN:
1348 return AMDGPU::BUFFER_STORE_DWORDX2_OFFSET;
1349 case AMDGPU::BUFFER_STORE_DWORDX3_OFFEN:
1350 return AMDGPU::BUFFER_STORE_DWORDX3_OFFSET;
1351 case AMDGPU::BUFFER_STORE_DWORDX4_OFFEN:
1352 return AMDGPU::BUFFER_STORE_DWORDX4_OFFSET;
1353 case AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFEN:
1354 return AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFSET;
1355 case AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFEN:
1356 return AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFSET;
1357 default:
1358 return -1;
1359 }
1360}
1361
1362static int getOffsetMUBUFLoad(unsigned Opc) {
1363 switch (Opc) {
1364 case AMDGPU::BUFFER_LOAD_DWORD_OFFEN:
1365 return AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
1366 case AMDGPU::BUFFER_LOAD_UBYTE_OFFEN:
1367 return AMDGPU::BUFFER_LOAD_UBYTE_OFFSET;
1368 case AMDGPU::BUFFER_LOAD_SBYTE_OFFEN:
1369 return AMDGPU::BUFFER_LOAD_SBYTE_OFFSET;
1370 case AMDGPU::BUFFER_LOAD_USHORT_OFFEN:
1371 return AMDGPU::BUFFER_LOAD_USHORT_OFFSET;
1372 case AMDGPU::BUFFER_LOAD_SSHORT_OFFEN:
1373 return AMDGPU::BUFFER_LOAD_SSHORT_OFFSET;
1374 case AMDGPU::BUFFER_LOAD_DWORDX2_OFFEN:
1375 return AMDGPU::BUFFER_LOAD_DWORDX2_OFFSET;
1376 case AMDGPU::BUFFER_LOAD_DWORDX3_OFFEN:
1377 return AMDGPU::BUFFER_LOAD_DWORDX3_OFFSET;
1378 case AMDGPU::BUFFER_LOAD_DWORDX4_OFFEN:
1379 return AMDGPU::BUFFER_LOAD_DWORDX4_OFFSET;
1380 case AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFEN:
1381 return AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFSET;
1382 case AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFEN:
1383 return AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFSET;
1384 case AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFEN:
1385 return AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFSET;
1386 case AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFEN:
1387 return AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFSET;
1388 case AMDGPU::BUFFER_LOAD_SHORT_D16_OFFEN:
1389 return AMDGPU::BUFFER_LOAD_SHORT_D16_OFFSET;
1390 case AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFEN:
1391 return AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFSET;
1392 default:
1393 return -1;
1394 }
1395}
1396
1397static int getOffenMUBUFStore(unsigned Opc) {
1398 switch (Opc) {
1399 case AMDGPU::BUFFER_STORE_DWORD_OFFSET:
1400 return AMDGPU::BUFFER_STORE_DWORD_OFFEN;
1401 case AMDGPU::BUFFER_STORE_BYTE_OFFSET:
1402 return AMDGPU::BUFFER_STORE_BYTE_OFFEN;
1403 case AMDGPU::BUFFER_STORE_SHORT_OFFSET:
1404 return AMDGPU::BUFFER_STORE_SHORT_OFFEN;
1405 case AMDGPU::BUFFER_STORE_DWORDX2_OFFSET:
1406 return AMDGPU::BUFFER_STORE_DWORDX2_OFFEN;
1407 case AMDGPU::BUFFER_STORE_DWORDX3_OFFSET:
1408 return AMDGPU::BUFFER_STORE_DWORDX3_OFFEN;
1409 case AMDGPU::BUFFER_STORE_DWORDX4_OFFSET:
1410 return AMDGPU::BUFFER_STORE_DWORDX4_OFFEN;
1411 case AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFSET:
1412 return AMDGPU::BUFFER_STORE_SHORT_D16_HI_OFFEN;
1413 case AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFSET:
1414 return AMDGPU::BUFFER_STORE_BYTE_D16_HI_OFFEN;
1415 default:
1416 return -1;
1417 }
1418}
1419
1420static int getOffenMUBUFLoad(unsigned Opc) {
1421 switch (Opc) {
1422 case AMDGPU::BUFFER_LOAD_DWORD_OFFSET:
1423 return AMDGPU::BUFFER_LOAD_DWORD_OFFEN;
1424 case AMDGPU::BUFFER_LOAD_UBYTE_OFFSET:
1425 return AMDGPU::BUFFER_LOAD_UBYTE_OFFEN;
1426 case AMDGPU::BUFFER_LOAD_SBYTE_OFFSET:
1427 return AMDGPU::BUFFER_LOAD_SBYTE_OFFEN;
1428 case AMDGPU::BUFFER_LOAD_USHORT_OFFSET:
1429 return AMDGPU::BUFFER_LOAD_USHORT_OFFEN;
1430 case AMDGPU::BUFFER_LOAD_SSHORT_OFFSET:
1431 return AMDGPU::BUFFER_LOAD_SSHORT_OFFEN;
1432 case AMDGPU::BUFFER_LOAD_DWORDX2_OFFSET:
1433 return AMDGPU::BUFFER_LOAD_DWORDX2_OFFEN;
1434 case AMDGPU::BUFFER_LOAD_DWORDX3_OFFSET:
1435 return AMDGPU::BUFFER_LOAD_DWORDX3_OFFEN;
1436 case AMDGPU::BUFFER_LOAD_DWORDX4_OFFSET:
1437 return AMDGPU::BUFFER_LOAD_DWORDX4_OFFEN;
1438 case AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFSET:
1439 return AMDGPU::BUFFER_LOAD_UBYTE_D16_OFFEN;
1440 case AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFSET:
1441 return AMDGPU::BUFFER_LOAD_UBYTE_D16_HI_OFFEN;
1442 case AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFSET:
1443 return AMDGPU::BUFFER_LOAD_SBYTE_D16_OFFEN;
1444 case AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFSET:
1445 return AMDGPU::BUFFER_LOAD_SBYTE_D16_HI_OFFEN;
1446 case AMDGPU::BUFFER_LOAD_SHORT_D16_OFFSET:
1447 return AMDGPU::BUFFER_LOAD_SHORT_D16_OFFEN;
1448 case AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFSET:
1449 return AMDGPU::BUFFER_LOAD_SHORT_D16_HI_OFFEN;
1450 default:
1451 return -1;
1452 }
1453}
1454
1457 MachineBasicBlock::iterator MI, int Index, unsigned Lane,
1458 unsigned ValueReg, bool IsKill, bool NeedsCFI) {
1459 MachineFunction *MF = MBB.getParent();
1461 const SIInstrInfo *TII = ST.getInstrInfo();
1462 const SIFrameLowering *TFL = ST.getFrameLowering();
1463
1464 MCPhysReg Reg = MFI->getVGPRToAGPRSpill(Index, Lane);
1465
1466 if (Reg == AMDGPU::NoRegister)
1467 return MachineInstrBuilder();
1468
1469 bool IsStore = MI->mayStore();
1470 MachineRegisterInfo &MRI = MF->getRegInfo();
1471 auto *TRI = static_cast<const SIRegisterInfo*>(MRI.getTargetRegisterInfo());
1472
1473 unsigned Dst = IsStore ? Reg : ValueReg;
1474 unsigned Src = IsStore ? ValueReg : Reg;
1475 bool IsVGPR = TRI->isVGPR(MRI, Reg);
1476 const DebugLoc &DL = MI->getDebugLoc();
1477 if (IsVGPR == TRI->isVGPR(MRI, ValueReg)) {
1478 // Spiller during regalloc may restore a spilled register to its superclass.
1479 // It could result in AGPR spills restored to VGPRs or the other way around,
1480 // making the src and dst with identical regclasses at this point. It just
1481 // needs a copy in such cases.
1482 auto CopyMIB = BuildMI(MBB, MI, DL, TII->get(AMDGPU::COPY), Dst)
1483 .addReg(Src, getKillRegState(IsKill));
1485 if (NeedsCFI)
1486 TFL->buildCFIForVRegToVRegSpill(MBB, MI, DL, Src, Dst);
1487 return CopyMIB;
1488 }
1489 unsigned Opc = (IsStore ^ IsVGPR) ? AMDGPU::V_ACCVGPR_WRITE_B32_e64
1490 : AMDGPU::V_ACCVGPR_READ_B32_e64;
1491
1492 auto MIB = BuildMI(MBB, MI, DL, TII->get(Opc), Dst)
1493 .addReg(Src, getKillRegState(IsKill));
1495 if (NeedsCFI)
1496 TFL->buildCFIForVRegToVRegSpill(MBB, MI, DL, Src, Dst);
1497 return MIB;
1498}
1499
1500// This differs from buildSpillLoadStore by only scavenging a VGPR. It does not
1501// need to handle the case where an SGPR may need to be spilled while spilling.
1503 MachineFrameInfo &MFI,
1505 int Index,
1506 int64_t Offset) {
1507 const SIInstrInfo *TII = ST.getInstrInfo();
1508 MachineBasicBlock *MBB = MI->getParent();
1509 const DebugLoc &DL = MI->getDebugLoc();
1510 bool IsStore = MI->mayStore();
1511
1512 unsigned Opc = MI->getOpcode();
1513 int LoadStoreOp = IsStore ?
1515 if (LoadStoreOp == -1)
1516 return false;
1517
1518 const MachineOperand *Reg = TII->getNamedOperand(*MI, AMDGPU::OpName::vdata);
1519 if (spillVGPRtoAGPR(ST, *MBB, MI, Index, 0, Reg->getReg(), false, false)
1520 .getInstr())
1521 return true;
1522
1523 MachineInstrBuilder NewMI =
1524 BuildMI(*MBB, MI, DL, TII->get(LoadStoreOp))
1525 .add(*Reg)
1526 .add(*TII->getNamedOperand(*MI, AMDGPU::OpName::srsrc))
1527 .add(*TII->getNamedOperand(*MI, AMDGPU::OpName::soffset))
1528 .addImm(Offset)
1529 .addImm(0) // cpol
1530 .addImm(0) // swz
1531 .cloneMemRefs(*MI);
1532
1533 const MachineOperand *VDataIn = TII->getNamedOperand(*MI,
1534 AMDGPU::OpName::vdata_in);
1535 if (VDataIn)
1536 NewMI.add(*VDataIn);
1537 return true;
1538}
1539
1541 unsigned LoadStoreOp,
1542 unsigned EltSize) {
1543 bool IsStore = TII->get(LoadStoreOp).mayStore();
1544 bool HasVAddr = AMDGPU::hasNamedOperand(LoadStoreOp, AMDGPU::OpName::vaddr);
1545 bool UseST =
1546 !HasVAddr && !AMDGPU::hasNamedOperand(LoadStoreOp, AMDGPU::OpName::saddr);
1547
1548 // Handle block load/store first.
1549 if (TII->isBlockLoadStore(LoadStoreOp))
1550 return LoadStoreOp;
1551
1552 switch (EltSize) {
1553 case 4:
1554 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
1555 : AMDGPU::SCRATCH_LOAD_DWORD_SADDR;
1556 break;
1557 case 8:
1558 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX2_SADDR
1559 : AMDGPU::SCRATCH_LOAD_DWORDX2_SADDR;
1560 break;
1561 case 12:
1562 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX3_SADDR
1563 : AMDGPU::SCRATCH_LOAD_DWORDX3_SADDR;
1564 break;
1565 case 16:
1566 LoadStoreOp = IsStore ? AMDGPU::SCRATCH_STORE_DWORDX4_SADDR
1567 : AMDGPU::SCRATCH_LOAD_DWORDX4_SADDR;
1568 break;
1569 default:
1570 llvm_unreachable("Unexpected spill load/store size!");
1571 }
1572
1573 if (HasVAddr)
1574 LoadStoreOp = AMDGPU::getFlatScratchInstSVfromSS(LoadStoreOp);
1575 else if (UseST)
1576 LoadStoreOp = AMDGPU::getFlatScratchInstSTfromSS(LoadStoreOp);
1577
1578 return LoadStoreOp;
1579}
1580
1583 unsigned LoadStoreOp, int Index, Register ValueReg, bool IsKill,
1584 MCRegister ScratchOffsetReg, int64_t InstOffset, MachineMemOperand *MMO,
1585 RegScavenger *RS, LiveRegUnits *LiveUnits, bool NeedsCFI) const {
1586 assert((!RS || !LiveUnits) && "Only RS or LiveUnits can be set but not both");
1587
1588 MachineFunction *MF = MBB.getParent();
1589 const SIInstrInfo *TII = ST.getInstrInfo();
1590 const MachineFrameInfo &MFI = MF->getFrameInfo();
1591 const SIFrameLowering *TFL = ST.getFrameLowering();
1592 const SIMachineFunctionInfo *FuncInfo = MF->getInfo<SIMachineFunctionInfo>();
1593
1594 const MCInstrDesc *Desc = &TII->get(LoadStoreOp);
1595 bool IsStore = Desc->mayStore();
1596 bool IsFlat = TII->isFLATScratch(LoadStoreOp);
1597 bool IsBlock = TII->isBlockLoadStore(LoadStoreOp);
1598
1599 bool CanClobberSCC = false;
1600 bool Scavenged = false;
1601 MCRegister SOffset = ScratchOffsetReg;
1602
1603 const TargetRegisterClass *RC = getRegClassForReg(MF->getRegInfo(), ValueReg);
1604 // On gfx90a+ AGPR is a regular VGPR acceptable for loads and stores.
1605 const bool IsAGPR = !ST.hasGFX90AInsts() && isAGPRClass(RC);
1606 unsigned RegWidth = AMDGPU::getRegBitWidth(*RC) / 8;
1607
1608 // On targets with register tuple alignment requirements,
1609 // for unaligned tuples, spill the first sub-reg as a 32-bit spill,
1610 // and spill the rest as a regular aligned tuple.
1611 // eg: SPILL_V224 $vgpr1_vgpr2_vgpr3_vgpr4_vgpr5_vgpr6_vgpr7
1612 // will be spilt as:
1613 // SPILL_SCRATCH_DWORD $vgpr1
1614 // SPILL_SCRATCH_DWORDx4 $vgpr2_vgpr3_vgpr4_vgpr5
1615 // SPILL_SCRATCH_DWORDx2 $vgpr6_vgpr7
1616 bool IsRegMisaligned = false;
1617 if (!IsBlock && !IsAGPR && RegWidth > 4 && IsFlat) {
1618 unsigned SpillOpcode =
1619 getFlatScratchSpillOpcode(TII, LoadStoreOp, std::min(RegWidth, 16u));
1620 int VDataIdx =
1621 IsStore ? AMDGPU::getNamedOperandIdx(SpillOpcode, AMDGPU::OpName::vdata)
1622 : 0; // Restore Ops have data reg as the first (output) operand.
1623 const TargetRegisterClass *ExpectedRC =
1624 TII->getRegClass(TII->get(SpillOpcode), VDataIdx);
1625 if (!ExpectedRC->contains(ValueReg)) {
1626 unsigned NumRegs = std::min(AMDGPU::getRegBitWidth(*ExpectedRC) / 4, 4u);
1627 unsigned SubIdx = getSubRegFromChannel(0, NumRegs);
1628 const TargetRegisterClass *MatchRC =
1629 getMatchingSuperRegClass(RC, ExpectedRC, SubIdx);
1630 if (!MatchRC || !MatchRC->contains(ValueReg))
1631 IsRegMisaligned = true;
1632 }
1633 }
1634 // The first sub-register will be spilled as a 32-bit value
1635 if (IsRegMisaligned)
1636 RegWidth -= 4u;
1637 // Always use 4 byte operations for AGPRs because we need to scavenge
1638 // a temporary VGPR.
1639 // If we're using a block operation, the element should be the whole block.
1640 unsigned EltSize = IsBlock ? RegWidth
1641 : (IsFlat && !IsAGPR) ? std::min(RegWidth, 16u)
1642 : 4u;
1643 unsigned NumSubRegs = RegWidth / EltSize;
1644 unsigned Size = NumSubRegs * EltSize;
1645 unsigned RemSize = RegWidth - Size;
1646 unsigned NumRemSubRegs = RemSize ? 1 : 0;
1647 // An additional sub-register is needed to spill the misaligned component.
1648 if (IsRegMisaligned)
1649 NumSubRegs += 1;
1650 int64_t Offset = InstOffset + MFI.getObjectOffset(Index);
1651 int64_t MaterializedOffset = Offset;
1652
1653 // Maxoffset is the starting offset for the last chunk to be spilled.
1654 // In case of non-zero remainder element, max offset will be the
1655 // last address(offset + Size) after spilling all the EltSize chunks.
1656 int64_t MaxOffset = Offset + Size - (RemSize ? 0 : EltSize);
1657 int64_t ScratchOffsetRegDelta = 0;
1658 int64_t AdditionalCFIOffset = 0;
1659
1660 if (IsFlat && EltSize > 4) {
1661 LoadStoreOp = getFlatScratchSpillOpcode(TII, LoadStoreOp, EltSize);
1662 Desc = &TII->get(LoadStoreOp);
1663 }
1664
1665 Align Alignment = MFI.getObjectAlign(Index);
1666 const MachinePointerInfo &BasePtrInfo = MMO->getPointerInfo();
1667
1668 assert((IsFlat || ((Offset % EltSize) == 0)) &&
1669 "unexpected VGPR spill offset");
1670
1671 // Track a VGPR to use for a constant offset we need to materialize.
1672 Register TmpOffsetVGPR;
1673
1674 // Track a VGPR to use as an intermediate value.
1675 Register TmpIntermediateVGPR;
1676 bool UseVGPROffset = false;
1677
1678 // Materialize a VGPR offset required for the given SGPR/VGPR/Immediate
1679 // combination.
1680 auto MaterializeVOffset = [&](Register SGPRBase, Register TmpVGPR,
1681 int64_t VOffset) {
1682 // We are using a VGPR offset
1683 if (IsFlat && SGPRBase) {
1684 // We only have 1 VGPR offset, or 1 SGPR offset. We don't have a free
1685 // SGPR, so perform the add as vector.
1686 // We don't need a base SGPR in the kernel.
1687
1688 if (ST.getConstantBusLimit(AMDGPU::V_ADD_U32_e64) >= 2) {
1689 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_ADD_U32_e64), TmpVGPR)
1690 .addReg(SGPRBase)
1691 .addImm(VOffset)
1692 .addImm(0); // clamp
1693 } else {
1694 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), TmpVGPR)
1695 .addReg(SGPRBase);
1696 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_ADD_U32_e32), TmpVGPR)
1697 .addImm(VOffset)
1698 .addReg(TmpOffsetVGPR);
1699 }
1700 } else {
1701 assert(TmpOffsetVGPR);
1702 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), TmpVGPR)
1703 .addImm(VOffset);
1704 }
1705 };
1706
1707 bool IsOffsetLegal =
1708 IsFlat ? TII->isLegalFLATOffset(MaxOffset, AMDGPUAS::PRIVATE_ADDRESS,
1710 : TII->isLegalMUBUFImmOffset(MaxOffset);
1711 if (!IsOffsetLegal || (IsFlat && !SOffset && !ST.hasFlatScratchSTMode())) {
1712 SOffset = MCRegister();
1713
1714 // We don't have access to the register scavenger if this function is called
1715 // during PEI::scavengeFrameVirtualRegs() so use LiveUnits in this case.
1716 // TODO: Clobbering SCC is not necessary for scratch instructions in the
1717 // entry.
1718 if (RS) {
1719 SOffset = RS->scavengeRegisterBackwards(AMDGPU::SGPR_32RegClass, MI, false, 0, false);
1720
1721 // Piggy back on the liveness scan we just did see if SCC is dead.
1722 CanClobberSCC = !RS->isRegUsed(AMDGPU::SCC);
1723 } else if (LiveUnits) {
1724 CanClobberSCC = LiveUnits->available(AMDGPU::SCC);
1725 for (MCRegister Reg : AMDGPU::SGPR_32RegClass) {
1726 if (LiveUnits->available(Reg) && !MF->getRegInfo().isReserved(Reg)) {
1727 SOffset = Reg;
1728 break;
1729 }
1730 }
1731 }
1732
1733 if (ScratchOffsetReg != AMDGPU::NoRegister && !CanClobberSCC)
1734 SOffset = Register();
1735
1736 if (!SOffset) {
1737 UseVGPROffset = true;
1738
1739 if (RS) {
1740 TmpOffsetVGPR = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass, MI, false, 0);
1741 } else {
1742 assert(LiveUnits);
1743 for (MCRegister Reg : AMDGPU::VGPR_32RegClass) {
1744 if (LiveUnits->available(Reg) && !MF->getRegInfo().isReserved(Reg)) {
1745 TmpOffsetVGPR = Reg;
1746 break;
1747 }
1748 }
1749 }
1750
1751 assert(TmpOffsetVGPR);
1752 } else if (!SOffset && CanClobberSCC) {
1753 // There are no free SGPRs, and since we are in the process of spilling
1754 // VGPRs too. Since we need a VGPR in order to spill SGPRs (this is true
1755 // on SI/CI and on VI it is true until we implement spilling using scalar
1756 // stores), we have no way to free up an SGPR. Our solution here is to
1757 // add the offset directly to the ScratchOffset or StackPtrOffset
1758 // register, and then subtract the offset after the spill to return the
1759 // register to it's original value.
1760
1761 // TODO: If we don't have to do an emergency stack slot spill, converting
1762 // to use the VGPR offset is fewer instructions.
1763 if (!ScratchOffsetReg)
1764 ScratchOffsetReg = FuncInfo->getStackPtrOffsetReg();
1765 SOffset = ScratchOffsetReg;
1766 ScratchOffsetRegDelta = Offset;
1767 } else {
1768 Scavenged = true;
1769 }
1770
1771 AdditionalCFIOffset = Offset;
1772 // We currently only support spilling VGPRs to EltSize boundaries, meaning
1773 // we can simplify the adjustment of Offset here to just scale with
1774 // WavefrontSize.
1775 if (!IsFlat && !UseVGPROffset)
1776 Offset *= ST.getWavefrontSize();
1777
1778 if (!UseVGPROffset && !SOffset)
1779 report_fatal_error("could not scavenge SGPR to spill in entry function");
1780
1781 if (UseVGPROffset) {
1782 // We are using a VGPR offset
1783 MaterializeVOffset(ScratchOffsetReg, TmpOffsetVGPR, Offset);
1784 } else if (ScratchOffsetReg == AMDGPU::NoRegister) {
1785 BuildMI(MBB, MI, DL, TII->get(AMDGPU::S_MOV_B32), SOffset).addImm(Offset);
1786 } else {
1787 assert(Offset != 0);
1788 auto Add = BuildMI(MBB, MI, DL, TII->get(AMDGPU::S_ADD_I32), SOffset)
1789 .addReg(ScratchOffsetReg)
1790 .addImm(Offset);
1791 Add->getOperand(3).setIsDead(); // Mark SCC as dead.
1792 }
1793
1794 Offset = 0;
1795 }
1796
1797 if (IsFlat && SOffset == AMDGPU::NoRegister) {
1798 assert(AMDGPU::getNamedOperandIdx(LoadStoreOp, AMDGPU::OpName::vaddr) < 0
1799 && "Unexpected vaddr for flat scratch with a FI operand");
1800
1801 if (UseVGPROffset) {
1802 LoadStoreOp = AMDGPU::getFlatScratchInstSVfromSS(LoadStoreOp);
1803 } else {
1804 assert(ST.hasFlatScratchSTMode());
1805 assert(!TII->isBlockLoadStore(LoadStoreOp) && "Block ops don't have ST");
1806 LoadStoreOp = AMDGPU::getFlatScratchInstSTfromSS(LoadStoreOp);
1807 }
1808
1809 Desc = &TII->get(LoadStoreOp);
1810 }
1811
1812 // Save a copy of the original element size before its potentially changed for
1813 // misaligned tuples.
1814 unsigned OrigEltSize = EltSize;
1815 for (unsigned i = 0, e = NumSubRegs + NumRemSubRegs, RegOffset = 0; i != e;
1816 ++i, RegOffset += EltSize) {
1817 if (IsRegMisaligned) {
1818 if (i == 0) {
1819 // For misaligned register tuples, spill only the first sub-reg in the
1820 // first iteration.
1821 EltSize = 4u;
1822 } else {
1823 // The misaligned register was spilt. Now the rest of the tuple is
1824 // properly aligned.
1825 IsRegMisaligned = false;
1826 EltSize = OrigEltSize;
1827 }
1828 LoadStoreOp = getFlatScratchSpillOpcode(TII, LoadStoreOp, EltSize);
1829 }
1830 if (i == NumSubRegs) {
1831 EltSize = RemSize;
1832 LoadStoreOp = getFlatScratchSpillOpcode(TII, LoadStoreOp, EltSize);
1833 }
1834 Desc = &TII->get(LoadStoreOp);
1835
1836 if (!IsFlat && UseVGPROffset) {
1837 int NewLoadStoreOp = IsStore ? getOffenMUBUFStore(LoadStoreOp)
1838 : getOffenMUBUFLoad(LoadStoreOp);
1839 Desc = &TII->get(NewLoadStoreOp);
1840 }
1841
1842 if (UseVGPROffset && TmpOffsetVGPR == TmpIntermediateVGPR) {
1843 // If we are spilling an AGPR beyond the range of the memory instruction
1844 // offset and need to use a VGPR offset, we ideally have at least 2
1845 // scratch VGPRs. If we don't have a second free VGPR without spilling,
1846 // recycle the VGPR used for the offset which requires resetting after
1847 // each subregister.
1848
1849 MaterializeVOffset(ScratchOffsetReg, TmpOffsetVGPR, MaterializedOffset);
1850 }
1851
1852 unsigned NumRegs = EltSize / 4;
1853 Register SubReg = e == 1
1854 ? ValueReg
1855 : Register(getSubReg(ValueReg,
1856 getSubRegFromChannel(RegOffset / 4, NumRegs)));
1857
1858 RegState SOffsetRegState = {};
1859 RegState SrcDstRegState = getDefRegState(!IsStore);
1860 const bool IsLastSubReg = i + 1 == e;
1861 const bool IsFirstSubReg = i == 0;
1862 if (IsLastSubReg) {
1863 SOffsetRegState |= getKillRegState(Scavenged);
1864 // The last implicit use carries the "Kill" flag.
1865 SrcDstRegState |= getKillRegState(IsKill);
1866 }
1867
1868 // Make sure the whole register is defined if there are undef components by
1869 // adding an implicit def of the super-reg on the first instruction.
1870 bool NeedSuperRegDef = e > 1 && IsStore && IsFirstSubReg;
1871 bool NeedSuperRegImpOperand = e > 1;
1872
1873 // Remaining element size to spill into memory after some parts of it
1874 // spilled into either AGPRs or VGPRs.
1875 unsigned RemEltSize = EltSize;
1876
1877 // AGPRs to spill VGPRs and vice versa are allocated in a reverse order,
1878 // starting from the last lane. In case if a register cannot be completely
1879 // spilled into another register that will ensure its alignment does not
1880 // change. For targets with VGPR alignment requirement this is important
1881 // in case of flat scratch usage as we might get a scratch_load or
1882 // scratch_store of an unaligned register otherwise.
1883 for (int LaneS = (RegOffset + EltSize) / 4 - 1, Lane = LaneS,
1884 LaneE = RegOffset / 4;
1885 Lane >= LaneE; --Lane) {
1886 bool IsSubReg = e > 1 || EltSize > 4;
1887 Register Sub = IsSubReg
1888 ? Register(getSubReg(ValueReg, getSubRegFromChannel(Lane)))
1889 : ValueReg;
1890 auto MIB =
1891 spillVGPRtoAGPR(ST, MBB, MI, Index, Lane, Sub, IsKill, NeedsCFI);
1892 if (!MIB.getInstr())
1893 break;
1894 if (NeedSuperRegDef || (IsSubReg && IsStore && Lane == LaneS && IsFirstSubReg)) {
1895 MIB.addReg(ValueReg, RegState::ImplicitDefine);
1896 NeedSuperRegDef = false;
1897 }
1898 if ((IsSubReg || NeedSuperRegImpOperand) && (IsFirstSubReg || IsLastSubReg)) {
1899 NeedSuperRegImpOperand = true;
1900 RegState State = SrcDstRegState;
1901 if (!IsLastSubReg || (Lane != LaneE))
1902 State &= ~RegState::Kill;
1903 if (!IsFirstSubReg || (Lane != LaneS))
1904 State &= ~RegState::Define;
1905 MIB.addReg(ValueReg, RegState::Implicit | State);
1906 }
1907 RemEltSize -= 4;
1908 }
1909
1910 if (!RemEltSize) // Fully spilled into AGPRs.
1911 continue;
1912
1913 if (RemEltSize != EltSize) { // Partially spilled to AGPRs
1914 assert(IsFlat && EltSize > 4);
1915
1916 unsigned NumRegs = RemEltSize / 4;
1917 SubReg = Register(getSubReg(ValueReg,
1918 getSubRegFromChannel(RegOffset / 4, NumRegs)));
1919 unsigned Opc = getFlatScratchSpillOpcode(TII, LoadStoreOp, RemEltSize);
1920 Desc = &TII->get(Opc);
1921 }
1922
1923 unsigned FinalReg = SubReg;
1924
1925 if (IsAGPR) {
1926 assert(EltSize == 4);
1927
1928 if (!TmpIntermediateVGPR) {
1929 TmpIntermediateVGPR = FuncInfo->getVGPRForAGPRCopy();
1930 assert(MF->getRegInfo().isReserved(TmpIntermediateVGPR));
1931 }
1932 if (IsStore) {
1933 auto AccRead = BuildMI(MBB, MI, DL,
1934 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64),
1935 TmpIntermediateVGPR)
1936 .addReg(SubReg, getKillRegState(IsKill));
1937 if (NeedSuperRegDef)
1938 AccRead.addReg(ValueReg, RegState::ImplicitDefine);
1939 if (NeedSuperRegImpOperand && (IsFirstSubReg || IsLastSubReg))
1940 AccRead.addReg(ValueReg, RegState::Implicit);
1942 }
1943 SubReg = TmpIntermediateVGPR;
1944 } else if (UseVGPROffset) {
1945 if (!TmpOffsetVGPR) {
1946 TmpOffsetVGPR = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
1947 MI, false, 0);
1948 RS->setRegUsed(TmpOffsetVGPR);
1949 }
1950 }
1951
1952 Register FinalValueReg = ValueReg;
1953 if (LoadStoreOp == AMDGPU::SCRATCH_LOAD_USHORT_SADDR ||
1954 LoadStoreOp == AMDGPU::SCRATCH_LOAD_USHORT_ST) {
1955 // If we are loading 16-bit value with SRAMECC endabled we need a temp
1956 // 32-bit VGPR to load and extract 16-bits into the final register.
1957 ValueReg =
1958 RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass, MI, false, 0);
1959 SubReg = ValueReg;
1960 IsKill = false;
1961 }
1962
1963 // Create the MMO, additional set the NonVolatile flag as scratch memory
1964 // used for spills will not be used outside the thread.
1965 MachinePointerInfo PInfo = BasePtrInfo.getWithOffset(RegOffset);
1967 PInfo, MMO->getFlags() | MOThreadPrivate, RemEltSize,
1968 commonAlignment(Alignment, RegOffset));
1969
1970 auto MIB =
1971 BuildMI(MBB, MI, DL, *Desc)
1972 .addReg(SubReg, getDefRegState(!IsStore) | getKillRegState(IsKill));
1973
1974 if (UseVGPROffset) {
1975 // For an AGPR spill, we reuse the same temp VGPR for the offset and the
1976 // intermediate accvgpr_write.
1977 MIB.addReg(TmpOffsetVGPR, getKillRegState(IsLastSubReg && !IsAGPR));
1978 }
1979
1980 if (!IsFlat)
1981 MIB.addReg(FuncInfo->getScratchRSrcReg());
1982
1983 if (SOffset == AMDGPU::NoRegister) {
1984 if (!IsFlat) {
1985 if (UseVGPROffset && ScratchOffsetReg) {
1986 MIB.addReg(ScratchOffsetReg);
1987 } else {
1988 assert(FuncInfo->isBottomOfStack());
1989 MIB.addImm(0);
1990 }
1991 }
1992 } else {
1993 MIB.addReg(SOffset, SOffsetRegState);
1994 }
1995
1996 MIB.addImm(Offset + RegOffset);
1997
1998 bool LastUse = MMO->getFlags() & MOLastUse;
1999 MIB.addImm(LastUse ? AMDGPU::CPol::TH_LU : 0); // cpol
2000
2001 if (!IsFlat)
2002 MIB.addImm(0); // swz
2003 MIB.addMemOperand(NewMMO);
2004
2005 if (FinalValueReg != ValueReg) {
2006 // Extract 16-bit from the loaded 32-bit value.
2007 ValueReg = getSubReg(ValueReg, AMDGPU::lo16);
2008 MIB = BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_MOV_B16_t16_e64))
2009 .addReg(FinalValueReg, getDefRegState(true))
2010 .addImm(0)
2011 .addReg(ValueReg, getKillRegState(true))
2012 .addImm(0);
2013 ValueReg = FinalValueReg;
2014 }
2015
2016 if (IsStore && NeedsCFI) {
2017 if (TII->isBlockLoadStore(LoadStoreOp)) {
2018 assert(RegOffset == 0 &&
2019 "expected whole register block to be treated as single element");
2021 } else {
2023 MBB, MI, DebugLoc(), SubReg,
2024 (Offset + RegOffset) * ST.getWavefrontSize() + AdditionalCFIOffset);
2025 }
2026 }
2027
2028 if (!IsAGPR && NeedSuperRegDef)
2029 MIB.addReg(ValueReg, RegState::ImplicitDefine);
2030
2031 if (!IsStore && IsAGPR && TmpIntermediateVGPR != AMDGPU::NoRegister) {
2032 MIB = BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64),
2033 FinalReg)
2034 .addReg(TmpIntermediateVGPR, RegState::Kill);
2036 }
2037
2038 bool IsSrcDstDef = hasRegState(SrcDstRegState, RegState::Define);
2039 bool PartialReloadCopy = (RemEltSize != EltSize) && !IsStore;
2040 if (NeedSuperRegImpOperand &&
2041 (IsFirstSubReg || (IsLastSubReg && !IsSrcDstDef))) {
2042 MIB.addReg(ValueReg, RegState::Implicit | SrcDstRegState);
2043 if (PartialReloadCopy)
2044 MIB.addReg(ValueReg, RegState::Implicit);
2045 }
2046
2047 // The epilog restore of a wwm-scratch register can cause undesired
2048 // optimization during machine-cp post PrologEpilogInserter if the same
2049 // register was assigned for return value ABI lowering with a COPY
2050 // instruction. As given below, with the epilog reload, the earlier COPY
2051 // appeared to be dead during machine-cp.
2052 // ...
2053 // v0 in WWM operation, needs the WWM spill at prolog/epilog.
2054 // $vgpr0 = V_WRITELANE_B32 $sgpr20, 0, $vgpr0
2055 // ...
2056 // Epilog block:
2057 // $vgpr0 = COPY $vgpr1 // outgoing value moved to v0
2058 // ...
2059 // WWM spill restore to preserve the inactive lanes of v0.
2060 // $sgpr4_sgpr5 = S_XOR_SAVEEXEC_B64 -1
2061 // $vgpr0 = BUFFER_LOAD $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 0, 0, 0
2062 // $exec = S_MOV_B64 killed $sgpr4_sgpr5
2063 // ...
2064 // SI_RETURN implicit $vgpr0
2065 // ...
2066 // To fix it, mark the same reg as a tied op for such restore instructions
2067 // so that it marks a usage for the preceding COPY.
2068 if (!IsStore && MI != MBB.end() && MI->isReturn() &&
2069 MI->readsRegister(SubReg, this)) {
2070 MIB.addReg(SubReg, RegState::Implicit);
2071 MIB->tieOperands(0, MIB->getNumOperands() - 1);
2072 }
2073
2074 // If we're building a block load, we should add artificial uses for the
2075 // CSR VGPRs that are *not* being transferred. This is because liveness
2076 // analysis is not aware of the mask, so we need to somehow inform it that
2077 // those registers are not available before the load and they should not be
2078 // scavenged.
2079 if (!IsStore && TII->isBlockLoadStore(LoadStoreOp))
2080 addImplicitUsesForBlockCSRLoad(MIB, ValueReg);
2081 }
2082
2083 if (ScratchOffsetRegDelta != 0) {
2084 // Subtract the offset we added to the ScratchOffset register.
2085 BuildMI(MBB, MI, DL, TII->get(AMDGPU::S_ADD_I32), SOffset)
2086 .addReg(SOffset)
2087 .addImm(-ScratchOffsetRegDelta);
2088 }
2089}
2090
2092 Register BlockReg) const {
2093 const MachineFunction *MF = MIB->getMF();
2094 const SIMachineFunctionInfo *FuncInfo = MF->getInfo<SIMachineFunctionInfo>();
2095 uint32_t Mask = FuncInfo->getMaskForVGPRBlockOps(BlockReg);
2096 Register BaseVGPR = getSubReg(BlockReg, AMDGPU::sub0);
2097 for (unsigned RegOffset = 1; RegOffset < 32; ++RegOffset)
2098 if (!(Mask & (1 << RegOffset)) &&
2099 isCalleeSavedPhysReg(BaseVGPR + RegOffset, *MF))
2100 MIB.addUse(BaseVGPR + RegOffset, RegState::Implicit);
2101}
2102
2105 Register BlockReg,
2106 int64_t Offset) const {
2107 const MachineFunction *MF = MBB.getParent();
2108 const SIMachineFunctionInfo *FuncInfo = MF->getInfo<SIMachineFunctionInfo>();
2109 uint32_t Mask = FuncInfo->getMaskForVGPRBlockOps(BlockReg);
2110 Register BaseVGPR = getSubReg(BlockReg, AMDGPU::sub0);
2111 for (unsigned RegOffset = 0; RegOffset < 32; ++RegOffset) {
2112 Register VGPR = BaseVGPR + RegOffset;
2113 if (Mask & (1 << RegOffset)) {
2114 assert(isCalleeSavedPhysReg(VGPR, *MF));
2115 ST.getFrameLowering()->buildCFIForVGPRToVMEMSpill(
2116 MBB, MBBI, DebugLoc(), VGPR,
2117 (Offset + RegOffset) * ST.getWavefrontSize());
2118 } else if (isCalleeSavedPhysReg(VGPR, *MF)) {
2119 // FIXME: This is a workaround for the fact that FrameLowering's
2120 // emitPrologueEntryCFI considers the block load to clobber all registers
2121 // in the block.
2122 ST.getFrameLowering()->buildCFIForSameValue(MBB, MBBI, DebugLoc(),
2123 BaseVGPR + RegOffset);
2124 }
2125 }
2126}
2127
2129 int Offset, bool IsLoad,
2130 bool IsKill) const {
2131 // Load/store VGPR
2132 MachineFrameInfo &FrameInfo = SB.MF.getFrameInfo();
2133 assert(FrameInfo.getStackID(Index) != TargetStackID::SGPRSpill);
2134
2135 Register FrameReg =
2136 FrameInfo.isFixedObjectIndex(Index) && hasBasePointer(SB.MF)
2137 ? getBaseRegister()
2138 : getFrameRegister(SB.MF);
2139
2140 Align Alignment = FrameInfo.getObjectAlign(Index);
2144 SB.EltSize, Alignment);
2145
2146 if (IsLoad) {
2147 unsigned Opc = ST.hasFlatScratchEnabled()
2148 ? AMDGPU::SCRATCH_LOAD_DWORD_SADDR
2149 : AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
2150 buildSpillLoadStore(*SB.MBB, SB.MI, SB.DL, Opc, Index, SB.TmpVGPR, false,
2151 FrameReg, (int64_t)Offset * SB.EltSize, MMO, SB.RS);
2152 } else {
2153 unsigned Opc = ST.hasFlatScratchEnabled()
2154 ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
2155 : AMDGPU::BUFFER_STORE_DWORD_OFFSET;
2156 buildSpillLoadStore(*SB.MBB, SB.MI, SB.DL, Opc, Index, SB.TmpVGPR, IsKill,
2157 FrameReg, (int64_t)Offset * SB.EltSize, MMO, SB.RS);
2158 // This only ever adds one VGPR spill
2159 SB.MFI.addToSpilledVGPRs(1);
2160 }
2161}
2162
2164 RegScavenger *RS, SlotIndexes *Indexes,
2165 LiveIntervals *LIS, bool OnlyToVGPR,
2166 bool SpillToPhysVGPRLane, bool NeedsCFI) const {
2167 assert(!MI->getOperand(0).isUndef() &&
2168 "undef spill should have been deleted earlier");
2169
2170 SGPRSpillBuilder SB(*this, *ST.getInstrInfo(), isWave32, MI, Index, RS);
2171
2172 ArrayRef<SpilledReg> VGPRSpills =
2173 SpillToPhysVGPRLane ? SB.MFI.getSGPRSpillToPhysicalVGPRLanes(Index)
2175 bool SpillToVGPR = !VGPRSpills.empty();
2176 if (OnlyToVGPR && !SpillToVGPR)
2177 return false;
2178
2179 const SIFrameLowering *TFL = ST.getFrameLowering();
2180
2181 assert(SpillToVGPR || (SB.SuperReg != SB.MFI.getStackPtrOffsetReg() &&
2182 SB.SuperReg != SB.MFI.getFrameOffsetReg()));
2183
2184 if (SpillToVGPR) {
2185
2186 // Since stack slot coloring pass is trying to optimize SGPR spills,
2187 // VGPR lanes (mapped from spill stack slot) may be shared for SGPR
2188 // spills of different sizes. This accounts for number of VGPR lanes alloted
2189 // equal to the largest SGPR being spilled in them.
2190 assert(SB.NumSubRegs <= VGPRSpills.size() &&
2191 "Num of SGPRs spilled should be less than or equal to num of "
2192 "the VGPR lanes.");
2193
2194 for (unsigned i = 0, e = SB.NumSubRegs; i < e; ++i) {
2195 Register SubReg =
2196 SB.NumSubRegs == 1
2197 ? SB.SuperReg
2198 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2199 SpilledReg Spill = VGPRSpills[i];
2200
2201 bool IsFirstSubreg = i == 0;
2202 bool IsLastSubreg = i == SB.NumSubRegs - 1;
2203 bool UseKill = SB.IsKill && IsLastSubreg;
2204
2205
2206 // Mark the "old value of vgpr" input undef only if this is the first sgpr
2207 // spill to this specific vgpr in the first basic block.
2208 auto MIB = BuildMI(*SB.MBB, MI, SB.DL,
2209 SB.TII.get(AMDGPU::SI_SPILL_S32_TO_VGPR), Spill.VGPR)
2210 .addReg(SubReg, getKillRegState(UseKill))
2211 .addImm(Spill.Lane)
2212 .addReg(Spill.VGPR);
2213
2214 MachineInstr *CFI = nullptr;
2215 if (NeedsCFI) {
2216 if (SB.SuperReg == SB.TRI.getReturnAddressReg(SB.MF)) {
2217 if (i == e - 1)
2218 CFI = TFL->buildCFIForSGPRToVGPRSpill(*SB.MBB, MI, DebugLoc(),
2219 AMDGPU::PC_REG, VGPRSpills);
2220 } else {
2221 CFI = TFL->buildCFIForSGPRToVGPRSpill(*SB.MBB, MI, DebugLoc(), SubReg,
2222 Spill.VGPR, Spill.Lane);
2223 }
2224 }
2225
2226 if (Indexes) {
2227 if (IsFirstSubreg)
2228 Indexes->replaceMachineInstrInMaps(*MI, *MIB);
2229 else
2230 Indexes->insertMachineInstrInMaps(*MIB);
2231
2232 if (CFI)
2233 Indexes->insertMachineInstrInMaps(*CFI);
2234 }
2235
2236 if (IsFirstSubreg && SB.NumSubRegs > 1) {
2237 // We may be spilling a super-register which is only partially defined,
2238 // and need to ensure later spills think the value is defined.
2239 MIB.addReg(SB.SuperReg, RegState::ImplicitDefine);
2240 }
2241
2242 if (SB.NumSubRegs > 1 && (IsFirstSubreg || IsLastSubreg))
2243 MIB.addReg(SB.SuperReg, getKillRegState(UseKill) | RegState::Implicit);
2244
2245 // FIXME: Since this spills to another register instead of an actual
2246 // frame index, we should delete the frame index when all references to
2247 // it are fixed.
2248 }
2249 } else {
2250 SB.prepare();
2251
2252 // SubReg carries the "Kill" flag when SubReg == SB.SuperReg.
2253 RegState SubKillState = getKillRegState((SB.NumSubRegs == 1) && SB.IsKill);
2254
2255 // Per VGPR helper data
2256 auto PVD = SB.getPerVGPRData();
2257
2258 for (unsigned Offset = 0; Offset < PVD.NumVGPRs; ++Offset) {
2259 RegState TmpVGPRFlags = RegState::Undef;
2260
2261 // Write sub registers into the VGPR
2262 for (unsigned i = Offset * PVD.PerVGPR,
2263 e = std::min((Offset + 1) * PVD.PerVGPR, SB.NumSubRegs);
2264 i < e; ++i) {
2265 Register SubReg =
2266 SB.NumSubRegs == 1
2267 ? SB.SuperReg
2268 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2269
2270 MachineInstrBuilder WriteLane =
2271 BuildMI(*SB.MBB, MI, SB.DL,
2272 SB.TII.get(AMDGPU::SI_SPILL_S32_TO_VGPR), SB.TmpVGPR)
2273 .addReg(SubReg, SubKillState)
2274 .addImm(i % PVD.PerVGPR)
2275 .addReg(SB.TmpVGPR, TmpVGPRFlags);
2276 TmpVGPRFlags = {};
2277
2278 if (Indexes) {
2279 if (i == 0)
2280 Indexes->replaceMachineInstrInMaps(*MI, *WriteLane);
2281 else
2282 Indexes->insertMachineInstrInMaps(*WriteLane);
2283 }
2284
2285 // There could be undef components of a spilled super register.
2286 // TODO: Can we detect this and skip the spill?
2287 if (SB.NumSubRegs > 1) {
2288 // The last implicit use of the SB.SuperReg carries the "Kill" flag.
2289 RegState SuperKillState = {};
2290 if (i + 1 == SB.NumSubRegs)
2291 SuperKillState |= getKillRegState(SB.IsKill);
2292 WriteLane.addReg(SB.SuperReg, RegState::Implicit | SuperKillState);
2293 }
2294 }
2295
2296 // Write out VGPR
2297 SB.readWriteTmpVGPR(Offset, /*IsLoad*/ false);
2298
2299 // TODO: Implement CFI for SpillToVMEM for all scenarios.
2300 MachineInstr *CFI = nullptr;
2301 if (NeedsCFI && SB.SuperReg == SB.TRI.getReturnAddressReg(SB.MF)) {
2302 int64_t CFIOffset = (Offset * SB.EltSize +
2303 SB.MF.getFrameInfo().getObjectOffset(Index)) *
2304 ST.getWavefrontSize();
2305 CFI = TFL->buildCFIForSGPRToVMEMSpill(*SB.MBB, MI, DebugLoc(),
2306 AMDGPU::PC_REG, CFIOffset);
2307 }
2308 if (Indexes && CFI)
2309 Indexes->insertMachineInstrInMaps(*CFI);
2310 }
2311
2312 SB.restore();
2313 }
2314
2315 MI->eraseFromParent();
2317
2318 if (LIS)
2320
2321 return true;
2322}
2323
2325 RegScavenger *RS, SlotIndexes *Indexes,
2326 LiveIntervals *LIS, bool OnlyToVGPR,
2327 bool SpillToPhysVGPRLane) const {
2328 SGPRSpillBuilder SB(*this, *ST.getInstrInfo(), isWave32, MI, Index, RS);
2329
2330 ArrayRef<SpilledReg> VGPRSpills =
2331 SpillToPhysVGPRLane ? SB.MFI.getSGPRSpillToPhysicalVGPRLanes(Index)
2333 bool SpillToVGPR = !VGPRSpills.empty();
2334 if (OnlyToVGPR && !SpillToVGPR)
2335 return false;
2336
2337 if (SpillToVGPR) {
2338 for (unsigned i = 0, e = SB.NumSubRegs; i < e; ++i) {
2339 Register SubReg =
2340 SB.NumSubRegs == 1
2341 ? SB.SuperReg
2342 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2343
2344 SpilledReg Spill = VGPRSpills[i];
2345 auto MIB = BuildMI(*SB.MBB, MI, SB.DL,
2346 SB.TII.get(AMDGPU::SI_RESTORE_S32_FROM_VGPR), SubReg)
2347 .addReg(Spill.VGPR)
2348 .addImm(Spill.Lane);
2349 if (SB.NumSubRegs > 1 && i == 0)
2351 if (Indexes) {
2352 if (i == e - 1)
2353 Indexes->replaceMachineInstrInMaps(*MI, *MIB);
2354 else
2355 Indexes->insertMachineInstrInMaps(*MIB);
2356 }
2357 }
2358 } else {
2359 SB.prepare();
2360
2361 // Per VGPR helper data
2362 auto PVD = SB.getPerVGPRData();
2363
2364 for (unsigned Offset = 0; Offset < PVD.NumVGPRs; ++Offset) {
2365 // Load in VGPR data
2366 SB.readWriteTmpVGPR(Offset, /*IsLoad*/ true);
2367
2368 // Unpack lanes
2369 for (unsigned i = Offset * PVD.PerVGPR,
2370 e = std::min((Offset + 1) * PVD.PerVGPR, SB.NumSubRegs);
2371 i < e; ++i) {
2372 Register SubReg =
2373 SB.NumSubRegs == 1
2374 ? SB.SuperReg
2375 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2376
2377 bool LastSubReg = (i + 1 == e);
2378 auto MIB = BuildMI(*SB.MBB, MI, SB.DL,
2379 SB.TII.get(AMDGPU::SI_RESTORE_S32_FROM_VGPR), SubReg)
2380 .addReg(SB.TmpVGPR, getKillRegState(LastSubReg))
2381 .addImm(i);
2382 if (SB.NumSubRegs > 1 && i == 0)
2384 if (Indexes) {
2385 if (i == e - 1)
2386 Indexes->replaceMachineInstrInMaps(*MI, *MIB);
2387 else
2388 Indexes->insertMachineInstrInMaps(*MIB);
2389 }
2390 }
2391 }
2392
2393 SB.restore();
2394 }
2395
2396 MI->eraseFromParent();
2397
2398 if (LIS)
2400
2401 return true;
2402}
2403
2405 MachineBasicBlock &RestoreMBB,
2406 Register SGPR, RegScavenger *RS) const {
2407 SGPRSpillBuilder SB(*this, *ST.getInstrInfo(), isWave32, MI, SGPR, false, 0,
2408 RS);
2409 SB.prepare();
2410 // Generate the spill of SGPR to SB.TmpVGPR.
2411 RegState SubKillState = getKillRegState((SB.NumSubRegs == 1) && SB.IsKill);
2412 auto PVD = SB.getPerVGPRData();
2413 for (unsigned Offset = 0; Offset < PVD.NumVGPRs; ++Offset) {
2414 RegState TmpVGPRFlags = RegState::Undef;
2415 // Write sub registers into the VGPR
2416 for (unsigned i = Offset * PVD.PerVGPR,
2417 e = std::min((Offset + 1) * PVD.PerVGPR, SB.NumSubRegs);
2418 i < e; ++i) {
2419 Register SubReg =
2420 SB.NumSubRegs == 1
2421 ? SB.SuperReg
2422 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2423
2424 MachineInstrBuilder WriteLane =
2425 BuildMI(*SB.MBB, MI, SB.DL, SB.TII.get(AMDGPU::V_WRITELANE_B32),
2426 SB.TmpVGPR)
2427 .addReg(SubReg, SubKillState)
2428 .addImm(i % PVD.PerVGPR)
2429 .addReg(SB.TmpVGPR, TmpVGPRFlags);
2430 TmpVGPRFlags = {};
2431 // There could be undef components of a spilled super register.
2432 // TODO: Can we detect this and skip the spill?
2433 if (SB.NumSubRegs > 1) {
2434 // The last implicit use of the SB.SuperReg carries the "Kill" flag.
2435 RegState SuperKillState = {};
2436 if (i + 1 == SB.NumSubRegs)
2437 SuperKillState |= getKillRegState(SB.IsKill);
2438 WriteLane.addReg(SB.SuperReg, RegState::Implicit | SuperKillState);
2439 }
2440 }
2441 // Don't need to write VGPR out.
2442 }
2443
2444 // Restore clobbered registers in the specified restore block.
2445 MI = RestoreMBB.end();
2446 SB.setMI(&RestoreMBB, MI);
2447 // Generate the restore of SGPR from SB.TmpVGPR.
2448 for (unsigned Offset = 0; Offset < PVD.NumVGPRs; ++Offset) {
2449 // Don't need to load VGPR in.
2450 // Unpack lanes
2451 for (unsigned i = Offset * PVD.PerVGPR,
2452 e = std::min((Offset + 1) * PVD.PerVGPR, SB.NumSubRegs);
2453 i < e; ++i) {
2454 Register SubReg =
2455 SB.NumSubRegs == 1
2456 ? SB.SuperReg
2457 : Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
2458
2459 assert(SubReg.isPhysical());
2460 bool LastSubReg = (i + 1 == e);
2461 auto MIB = BuildMI(*SB.MBB, MI, SB.DL, SB.TII.get(AMDGPU::V_READLANE_B32),
2462 SubReg)
2463 .addReg(SB.TmpVGPR, getKillRegState(LastSubReg))
2464 .addImm(i);
2465 if (SB.NumSubRegs > 1 && i == 0)
2467 }
2468 }
2469 SB.restore();
2470
2472 return false;
2473}
2474
2475/// Special case of eliminateFrameIndex. Returns true if the SGPR was spilled to
2476/// a VGPR and the stack slot can be safely eliminated when all other users are
2477/// handled.
2480 SlotIndexes *Indexes, LiveIntervals *LIS, bool SpillToPhysVGPRLane) const {
2481 bool NeedsCFI = false;
2482 switch (MI->getOpcode()) {
2483 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
2484 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
2485 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
2486 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
2487 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
2488 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
2489 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
2490 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
2491 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
2492 case AMDGPU::SI_SPILL_S32_CFI_SAVE:
2493 NeedsCFI = true;
2494 [[fallthrough]];
2495 case AMDGPU::SI_SPILL_S1024_SAVE:
2496 case AMDGPU::SI_SPILL_S512_SAVE:
2497 case AMDGPU::SI_SPILL_S384_SAVE:
2498 case AMDGPU::SI_SPILL_S352_SAVE:
2499 case AMDGPU::SI_SPILL_S320_SAVE:
2500 case AMDGPU::SI_SPILL_S288_SAVE:
2501 case AMDGPU::SI_SPILL_S256_SAVE:
2502 case AMDGPU::SI_SPILL_S224_SAVE:
2503 case AMDGPU::SI_SPILL_S192_SAVE:
2504 case AMDGPU::SI_SPILL_S160_SAVE:
2505 case AMDGPU::SI_SPILL_S128_SAVE:
2506 case AMDGPU::SI_SPILL_S96_SAVE:
2507 case AMDGPU::SI_SPILL_S64_SAVE:
2508 case AMDGPU::SI_SPILL_S32_SAVE:
2509 return spillSGPR(MI, FI, RS, Indexes, LIS, true, SpillToPhysVGPRLane,
2510 NeedsCFI);
2511 case AMDGPU::SI_SPILL_S1024_RESTORE:
2512 case AMDGPU::SI_SPILL_S512_RESTORE:
2513 case AMDGPU::SI_SPILL_S384_RESTORE:
2514 case AMDGPU::SI_SPILL_S352_RESTORE:
2515 case AMDGPU::SI_SPILL_S320_RESTORE:
2516 case AMDGPU::SI_SPILL_S288_RESTORE:
2517 case AMDGPU::SI_SPILL_S256_RESTORE:
2518 case AMDGPU::SI_SPILL_S224_RESTORE:
2519 case AMDGPU::SI_SPILL_S192_RESTORE:
2520 case AMDGPU::SI_SPILL_S160_RESTORE:
2521 case AMDGPU::SI_SPILL_S128_RESTORE:
2522 case AMDGPU::SI_SPILL_S96_RESTORE:
2523 case AMDGPU::SI_SPILL_S64_RESTORE:
2524 case AMDGPU::SI_SPILL_S32_RESTORE:
2525 return restoreSGPR(MI, FI, RS, Indexes, LIS, true, SpillToPhysVGPRLane);
2526 default:
2527 llvm_unreachable("not an SGPR spill instruction");
2528 }
2529}
2530
2531// Does adding the low 32 bits of \p LHS and \p RHS carry out?
2532static bool wrapsAround32(int64_t LHS, int64_t RHS) {
2533 return static_cast<uint64_t>(static_cast<uint32_t>(LHS)) +
2534 static_cast<uint32_t>(RHS) >
2535 UINT32_MAX;
2536}
2537
2538// Would folding Offset into OtherOp (in place of a separate frame-base add)
2539// use a different carry-out than the unfolded add?
2541 int64_t Offset, Register FrameReg) {
2542 return OtherOp.isImm() ? wrapsAround32(OtherOp.getImm(), Offset)
2543 : FrameReg.isValid();
2544}
2545
2546// Is SCC live into MI, so that frame index lowering must not clobber it?
2547static bool isSCCLiveInto(const RegScavenger &RS, const MachineInstr &MI) {
2548 return (RS.isRegUsed(AMDGPU::SCC) &&
2549 !MI.definesRegister(AMDGPU::SCC, /*TRI=*/nullptr)) ||
2550 MI.readsRegister(AMDGPU::SCC, /*TRI=*/nullptr);
2551}
2552
2554 int SPAdj, unsigned FIOperandNum,
2555 RegScavenger *RS) const {
2556 MachineFunction *MF = MI->getMF();
2557 MachineBasicBlock *MBB = MI->getParent();
2559 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
2560 const SIInstrInfo *TII = ST.getInstrInfo();
2561 const DebugLoc &DL = MI->getDebugLoc();
2562
2563 assert(SPAdj == 0 && "unhandled SP adjustment in call sequence?");
2564
2566 "unreserved scratch RSRC register");
2567
2568 MachineOperand *FIOp = &MI->getOperand(FIOperandNum);
2569 int Index = MI->getOperand(FIOperandNum).getIndex();
2570
2571 Register FrameReg = FrameInfo.isFixedObjectIndex(Index) && hasBasePointer(*MF)
2572 ? getBaseRegister()
2573 : getFrameRegister(*MF);
2574
2575 bool NeedsCFI = false;
2576
2577 switch (MI->getOpcode()) {
2578 // SGPR register spill
2579 case AMDGPU::SI_SPILL_S1024_CFI_SAVE:
2580 case AMDGPU::SI_SPILL_S512_CFI_SAVE:
2581 case AMDGPU::SI_SPILL_S256_CFI_SAVE:
2582 case AMDGPU::SI_SPILL_S224_CFI_SAVE:
2583 case AMDGPU::SI_SPILL_S192_CFI_SAVE:
2584 case AMDGPU::SI_SPILL_S160_CFI_SAVE:
2585 case AMDGPU::SI_SPILL_S128_CFI_SAVE:
2586 case AMDGPU::SI_SPILL_S96_CFI_SAVE:
2587 case AMDGPU::SI_SPILL_S64_CFI_SAVE:
2588 case AMDGPU::SI_SPILL_S32_CFI_SAVE: {
2589 NeedsCFI = true;
2590 [[fallthrough]];
2591 }
2592 case AMDGPU::SI_SPILL_S1024_SAVE:
2593 case AMDGPU::SI_SPILL_S512_SAVE:
2594 case AMDGPU::SI_SPILL_S384_SAVE:
2595 case AMDGPU::SI_SPILL_S352_SAVE:
2596 case AMDGPU::SI_SPILL_S320_SAVE:
2597 case AMDGPU::SI_SPILL_S288_SAVE:
2598 case AMDGPU::SI_SPILL_S256_SAVE:
2599 case AMDGPU::SI_SPILL_S224_SAVE:
2600 case AMDGPU::SI_SPILL_S192_SAVE:
2601 case AMDGPU::SI_SPILL_S160_SAVE:
2602 case AMDGPU::SI_SPILL_S128_SAVE:
2603 case AMDGPU::SI_SPILL_S96_SAVE:
2604 case AMDGPU::SI_SPILL_S64_SAVE:
2605 case AMDGPU::SI_SPILL_S32_SAVE: {
2606 return spillSGPR(MI, Index, RS, nullptr, nullptr,
2607 FrameInfo.getStackID(Index) == TargetStackID::SGPRSpill,
2608 false, NeedsCFI);
2609 }
2610
2611 // SGPR register restore
2612 case AMDGPU::SI_SPILL_S1024_RESTORE:
2613 case AMDGPU::SI_SPILL_S512_RESTORE:
2614 case AMDGPU::SI_SPILL_S384_RESTORE:
2615 case AMDGPU::SI_SPILL_S352_RESTORE:
2616 case AMDGPU::SI_SPILL_S320_RESTORE:
2617 case AMDGPU::SI_SPILL_S288_RESTORE:
2618 case AMDGPU::SI_SPILL_S256_RESTORE:
2619 case AMDGPU::SI_SPILL_S224_RESTORE:
2620 case AMDGPU::SI_SPILL_S192_RESTORE:
2621 case AMDGPU::SI_SPILL_S160_RESTORE:
2622 case AMDGPU::SI_SPILL_S128_RESTORE:
2623 case AMDGPU::SI_SPILL_S96_RESTORE:
2624 case AMDGPU::SI_SPILL_S64_RESTORE:
2625 case AMDGPU::SI_SPILL_S32_RESTORE: {
2626 return restoreSGPR(MI, Index, RS, nullptr, nullptr,
2627 FrameInfo.getStackID(Index) ==
2629 }
2630
2631 // VGPR register spill
2632 case AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE:
2633 case AMDGPU::SI_SPILL_V1024_CFI_SAVE:
2634 case AMDGPU::SI_SPILL_V512_CFI_SAVE:
2635 case AMDGPU::SI_SPILL_V256_CFI_SAVE:
2636 case AMDGPU::SI_SPILL_V224_CFI_SAVE:
2637 case AMDGPU::SI_SPILL_V192_CFI_SAVE:
2638 case AMDGPU::SI_SPILL_V160_CFI_SAVE:
2639 case AMDGPU::SI_SPILL_V128_CFI_SAVE:
2640 case AMDGPU::SI_SPILL_V96_CFI_SAVE:
2641 case AMDGPU::SI_SPILL_V64_CFI_SAVE:
2642 case AMDGPU::SI_SPILL_V32_CFI_SAVE:
2643 case AMDGPU::SI_SPILL_A1024_CFI_SAVE:
2644 case AMDGPU::SI_SPILL_A512_CFI_SAVE:
2645 case AMDGPU::SI_SPILL_A256_CFI_SAVE:
2646 case AMDGPU::SI_SPILL_A224_CFI_SAVE:
2647 case AMDGPU::SI_SPILL_A192_CFI_SAVE:
2648 case AMDGPU::SI_SPILL_A160_CFI_SAVE:
2649 case AMDGPU::SI_SPILL_A128_CFI_SAVE:
2650 case AMDGPU::SI_SPILL_A96_CFI_SAVE:
2651 case AMDGPU::SI_SPILL_A64_CFI_SAVE:
2652 case AMDGPU::SI_SPILL_A32_CFI_SAVE:
2653 case AMDGPU::SI_SPILL_AV1024_CFI_SAVE:
2654 case AMDGPU::SI_SPILL_AV512_CFI_SAVE:
2655 case AMDGPU::SI_SPILL_AV256_CFI_SAVE:
2656 case AMDGPU::SI_SPILL_AV224_CFI_SAVE:
2657 case AMDGPU::SI_SPILL_AV192_CFI_SAVE:
2658 case AMDGPU::SI_SPILL_AV160_CFI_SAVE:
2659 case AMDGPU::SI_SPILL_AV128_CFI_SAVE:
2660 case AMDGPU::SI_SPILL_AV96_CFI_SAVE:
2661 case AMDGPU::SI_SPILL_AV64_CFI_SAVE:
2662 case AMDGPU::SI_SPILL_AV32_CFI_SAVE:
2663 NeedsCFI = true;
2664 [[fallthrough]];
2665 case AMDGPU::SI_BLOCK_SPILL_V1024_SAVE:
2666 case AMDGPU::SI_SPILL_V1024_SAVE:
2667 case AMDGPU::SI_SPILL_V512_SAVE:
2668 case AMDGPU::SI_SPILL_V384_SAVE:
2669 case AMDGPU::SI_SPILL_V352_SAVE:
2670 case AMDGPU::SI_SPILL_V320_SAVE:
2671 case AMDGPU::SI_SPILL_V288_SAVE:
2672 case AMDGPU::SI_SPILL_V256_SAVE:
2673 case AMDGPU::SI_SPILL_V224_SAVE:
2674 case AMDGPU::SI_SPILL_V192_SAVE:
2675 case AMDGPU::SI_SPILL_V160_SAVE:
2676 case AMDGPU::SI_SPILL_V128_SAVE:
2677 case AMDGPU::SI_SPILL_V96_SAVE:
2678 case AMDGPU::SI_SPILL_V64_SAVE:
2679 case AMDGPU::SI_SPILL_V32_SAVE:
2680 case AMDGPU::SI_SPILL_V16_SAVE:
2681 case AMDGPU::SI_SPILL_A1024_SAVE:
2682 case AMDGPU::SI_SPILL_A512_SAVE:
2683 case AMDGPU::SI_SPILL_A384_SAVE:
2684 case AMDGPU::SI_SPILL_A352_SAVE:
2685 case AMDGPU::SI_SPILL_A320_SAVE:
2686 case AMDGPU::SI_SPILL_A288_SAVE:
2687 case AMDGPU::SI_SPILL_A256_SAVE:
2688 case AMDGPU::SI_SPILL_A224_SAVE:
2689 case AMDGPU::SI_SPILL_A192_SAVE:
2690 case AMDGPU::SI_SPILL_A160_SAVE:
2691 case AMDGPU::SI_SPILL_A128_SAVE:
2692 case AMDGPU::SI_SPILL_A96_SAVE:
2693 case AMDGPU::SI_SPILL_A64_SAVE:
2694 case AMDGPU::SI_SPILL_A32_SAVE:
2695 case AMDGPU::SI_SPILL_AV1024_SAVE:
2696 case AMDGPU::SI_SPILL_AV512_SAVE:
2697 case AMDGPU::SI_SPILL_AV384_SAVE:
2698 case AMDGPU::SI_SPILL_AV352_SAVE:
2699 case AMDGPU::SI_SPILL_AV320_SAVE:
2700 case AMDGPU::SI_SPILL_AV288_SAVE:
2701 case AMDGPU::SI_SPILL_AV256_SAVE:
2702 case AMDGPU::SI_SPILL_AV224_SAVE:
2703 case AMDGPU::SI_SPILL_AV192_SAVE:
2704 case AMDGPU::SI_SPILL_AV160_SAVE:
2705 case AMDGPU::SI_SPILL_AV128_SAVE:
2706 case AMDGPU::SI_SPILL_AV96_SAVE:
2707 case AMDGPU::SI_SPILL_AV64_SAVE:
2708 case AMDGPU::SI_SPILL_AV32_SAVE:
2709 case AMDGPU::SI_SPILL_WWM_V32_SAVE:
2710 case AMDGPU::SI_SPILL_WWM_AV32_SAVE: {
2711 assert(
2712 MI->getOpcode() != AMDGPU::SI_BLOCK_SPILL_V1024_SAVE &&
2713 "block spill does not currenty support spilling non-CSR registers");
2714
2715 if (MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE)
2716 // Put mask into M0.
2717 BuildMI(*MBB, MI, MI->getDebugLoc(), TII->get(AMDGPU::S_MOV_B32),
2718 AMDGPU::M0)
2719 .add(*TII->getNamedOperand(*MI, AMDGPU::OpName::mask));
2720
2721 const MachineOperand *VData = TII->getNamedOperand(*MI,
2722 AMDGPU::OpName::vdata);
2723 if (VData->isUndef()) {
2724 MI->eraseFromParent();
2725 return true;
2726 }
2727
2728 assert(TII->getNamedOperand(*MI, AMDGPU::OpName::soffset)->getReg() ==
2729 MFI->getStackPtrOffsetReg());
2730
2731 unsigned Opc;
2732 if (MI->getOpcode() == AMDGPU::SI_SPILL_V16_SAVE) {
2733 assert(ST.hasFlatScratchEnabled() && "Flat Scratch is not enabled!");
2734 Opc = AMDGPU::SCRATCH_STORE_SHORT_SADDR_t16;
2735 } else {
2736 Opc = MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE
2737 ? AMDGPU::SCRATCH_STORE_BLOCK_SADDR
2738 : ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
2739 : AMDGPU::BUFFER_STORE_DWORD_OFFSET;
2740 }
2741
2742 auto *MBB = MI->getParent();
2743 bool IsWWMRegSpill = TII->isWWMRegSpillOpcode(MI->getOpcode());
2744 if (IsWWMRegSpill) {
2745 TII->insertScratchExecCopy(*MF, *MBB, MI, DL, MFI->getSGPRForEXECCopy(),
2746 RS->isRegUsed(AMDGPU::SCC));
2747 }
2749 *MBB, MI, DL, Opc, Index, VData->getReg(), VData->isKill(), FrameReg,
2750 TII->getNamedOperand(*MI, AMDGPU::OpName::offset)->getImm(),
2751 *MI->memoperands_begin(), RS, nullptr, NeedsCFI);
2753 if (IsWWMRegSpill)
2754 TII->restoreExec(*MF, *MBB, MI, DL, MFI->getSGPRForEXECCopy());
2755
2756 MI->eraseFromParent();
2757 return true;
2758 }
2759 case AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE: {
2760 // Put mask into M0.
2761 BuildMI(*MBB, MI, MI->getDebugLoc(), TII->get(AMDGPU::S_MOV_B32),
2762 AMDGPU::M0)
2763 .add(*TII->getNamedOperand(*MI, AMDGPU::OpName::mask));
2764 [[fallthrough]];
2765 }
2766 case AMDGPU::SI_SPILL_V16_RESTORE:
2767 case AMDGPU::SI_SPILL_V32_RESTORE:
2768 case AMDGPU::SI_SPILL_V64_RESTORE:
2769 case AMDGPU::SI_SPILL_V96_RESTORE:
2770 case AMDGPU::SI_SPILL_V128_RESTORE:
2771 case AMDGPU::SI_SPILL_V160_RESTORE:
2772 case AMDGPU::SI_SPILL_V192_RESTORE:
2773 case AMDGPU::SI_SPILL_V224_RESTORE:
2774 case AMDGPU::SI_SPILL_V256_RESTORE:
2775 case AMDGPU::SI_SPILL_V288_RESTORE:
2776 case AMDGPU::SI_SPILL_V320_RESTORE:
2777 case AMDGPU::SI_SPILL_V352_RESTORE:
2778 case AMDGPU::SI_SPILL_V384_RESTORE:
2779 case AMDGPU::SI_SPILL_V512_RESTORE:
2780 case AMDGPU::SI_SPILL_V1024_RESTORE:
2781 case AMDGPU::SI_SPILL_A32_RESTORE:
2782 case AMDGPU::SI_SPILL_A64_RESTORE:
2783 case AMDGPU::SI_SPILL_A96_RESTORE:
2784 case AMDGPU::SI_SPILL_A128_RESTORE:
2785 case AMDGPU::SI_SPILL_A160_RESTORE:
2786 case AMDGPU::SI_SPILL_A192_RESTORE:
2787 case AMDGPU::SI_SPILL_A224_RESTORE:
2788 case AMDGPU::SI_SPILL_A256_RESTORE:
2789 case AMDGPU::SI_SPILL_A288_RESTORE:
2790 case AMDGPU::SI_SPILL_A320_RESTORE:
2791 case AMDGPU::SI_SPILL_A352_RESTORE:
2792 case AMDGPU::SI_SPILL_A384_RESTORE:
2793 case AMDGPU::SI_SPILL_A512_RESTORE:
2794 case AMDGPU::SI_SPILL_A1024_RESTORE:
2795 case AMDGPU::SI_SPILL_AV32_RESTORE:
2796 case AMDGPU::SI_SPILL_AV64_RESTORE:
2797 case AMDGPU::SI_SPILL_AV96_RESTORE:
2798 case AMDGPU::SI_SPILL_AV128_RESTORE:
2799 case AMDGPU::SI_SPILL_AV160_RESTORE:
2800 case AMDGPU::SI_SPILL_AV192_RESTORE:
2801 case AMDGPU::SI_SPILL_AV224_RESTORE:
2802 case AMDGPU::SI_SPILL_AV256_RESTORE:
2803 case AMDGPU::SI_SPILL_AV288_RESTORE:
2804 case AMDGPU::SI_SPILL_AV320_RESTORE:
2805 case AMDGPU::SI_SPILL_AV352_RESTORE:
2806 case AMDGPU::SI_SPILL_AV384_RESTORE:
2807 case AMDGPU::SI_SPILL_AV512_RESTORE:
2808 case AMDGPU::SI_SPILL_AV1024_RESTORE:
2809 case AMDGPU::SI_SPILL_WWM_V32_RESTORE:
2810 case AMDGPU::SI_SPILL_WWM_AV32_RESTORE: {
2811 const MachineOperand *VData = TII->getNamedOperand(*MI,
2812 AMDGPU::OpName::vdata);
2813 assert(TII->getNamedOperand(*MI, AMDGPU::OpName::soffset)->getReg() ==
2814 MFI->getStackPtrOffsetReg());
2815
2816 unsigned Opc;
2817 if (MI->getOpcode() == AMDGPU::SI_SPILL_V16_RESTORE) {
2818 assert(ST.hasFlatScratchEnabled() && "Flat Scratch is not enabled!");
2819 Opc = ST.d16PreservesUnusedBits()
2820 ? AMDGPU::SCRATCH_LOAD_SHORT_D16_SADDR_t16
2821 : AMDGPU::SCRATCH_LOAD_USHORT_SADDR;
2822 } else {
2823 Opc = MI->getOpcode() == AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE
2824 ? AMDGPU::SCRATCH_LOAD_BLOCK_SADDR
2825 : ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_LOAD_DWORD_SADDR
2826 : AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
2827 }
2828
2829 auto *MBB = MI->getParent();
2830 bool IsWWMRegSpill = TII->isWWMRegSpillOpcode(MI->getOpcode());
2831 if (IsWWMRegSpill) {
2832 TII->insertScratchExecCopy(*MF, *MBB, MI, DL, MFI->getSGPRForEXECCopy(),
2833 RS->isRegUsed(AMDGPU::SCC));
2834 }
2835
2837 *MBB, MI, DL, Opc, Index, VData->getReg(), VData->isKill(), FrameReg,
2838 TII->getNamedOperand(*MI, AMDGPU::OpName::offset)->getImm(),
2839 *MI->memoperands_begin(), RS);
2840
2841 if (IsWWMRegSpill)
2842 TII->restoreExec(*MF, *MBB, MI, DL, MFI->getSGPRForEXECCopy());
2843
2844 MI->eraseFromParent();
2845 return true;
2846 }
2847 case AMDGPU::V_ADD_U32_e32:
2848 case AMDGPU::V_ADD_U32_e64:
2849 case AMDGPU::V_ADD_CO_U32_e32:
2850 case AMDGPU::V_ADD_CO_U32_e64: {
2851 // TODO: Handle sub, and, or.
2852 unsigned NumDefs = MI->getNumExplicitDefs();
2853 unsigned Src0Idx = NumDefs;
2854
2855 bool HasClamp = false;
2856 MachineOperand *VCCOp = nullptr;
2857
2858 switch (MI->getOpcode()) {
2859 case AMDGPU::V_ADD_U32_e32:
2860 break;
2861 case AMDGPU::V_ADD_U32_e64:
2862 HasClamp = MI->getOperand(3).getImm();
2863 break;
2864 case AMDGPU::V_ADD_CO_U32_e32:
2865 VCCOp = &MI->getOperand(3);
2866 break;
2867 case AMDGPU::V_ADD_CO_U32_e64:
2868 VCCOp = &MI->getOperand(1);
2869 HasClamp = MI->getOperand(4).getImm();
2870 break;
2871 default:
2872 break;
2873 }
2874 bool DeadVCC = !VCCOp || VCCOp->isDead();
2875 MachineOperand &DstOp = MI->getOperand(0);
2876 Register DstReg = DstOp.getReg();
2877
2878 unsigned OtherOpIdx =
2879 FIOperandNum == Src0Idx ? FIOperandNum + 1 : Src0Idx;
2880 MachineOperand *OtherOp = &MI->getOperand(OtherOpIdx);
2881
2882 unsigned Src1Idx = Src0Idx + 1;
2883 Register MaterializedReg = FrameReg;
2884 Register ScavengedVGPR;
2885
2886 int64_t Offset = FrameInfo.getObjectOffset(Index);
2887
2888 // A split or wrapping fold add carries out of the wrong sum, and clamp
2889 // does not distribute.
2890 if ((!DeadVCC || HasClamp) &&
2891 foldingOffsetChangesCarry(*OtherOp, Offset, FrameReg))
2892 break;
2893
2894 // For the non-immediate case, we could fall through to the default
2895 // handling, but we do an in-place update of the result register here to
2896 // avoid scavenging another register.
2897 if (OtherOp->isImm()) {
2898 int64_t TotalOffset = OtherOp->getImm() + Offset;
2899
2900 if (!ST.hasVOP3Literal() && SIInstrInfo::isVOP3(*MI) &&
2901 !AMDGPU::isInlinableIntLiteral(TotalOffset)) {
2902 // If we can't support a VOP3 literal in the VALU instruction, we
2903 // can't specially fold into the add.
2904 // TODO: Handle VOP3->VOP2 shrink to support the fold.
2905 break;
2906 }
2907
2908 OtherOp->setImm(TotalOffset);
2909 Offset = 0;
2910 }
2911
2912 if (FrameReg && !ST.hasFlatScratchEnabled()) {
2913 // We should just do an in-place update of the result register. However,
2914 // the value there may also be used by the add, in which case we need a
2915 // temporary register.
2916 //
2917 // FIXME: The scavenger is not finding the result register in the
2918 // common case where the add does not read the register.
2919
2920 ScavengedVGPR = RS->scavengeRegisterBackwards(
2921 AMDGPU::VGPR_32RegClass, MI, /*RestoreAfter=*/false, /*SPAdj=*/0);
2922
2923 // TODO: If we have a free SGPR, it's sometimes better to use a scalar
2924 // shift.
2925 BuildMI(*MBB, *MI, DL, TII->get(AMDGPU::V_LSHRREV_B32_e64))
2926 .addDef(ScavengedVGPR, RegState::Renamable)
2927 .addImm(ST.getWavefrontSizeLog2())
2928 .addReg(FrameReg);
2929 MaterializedReg = ScavengedVGPR;
2930 }
2931
2932 if ((!OtherOp->isImm() || OtherOp->getImm() != 0) && MaterializedReg) {
2933 if (OtherOp->isImm()) {
2934 FIOp->ChangeToRegister(MaterializedReg, false);
2935 FIOp->setIsKill(MaterializedReg != FrameReg);
2936 } else {
2937 if (ST.hasFlatScratchEnabled() &&
2938 !TII->isOperandLegal(*MI, Src1Idx, OtherOp)) {
2939 // We didn't need the shift above, so we have an SGPR for the frame
2940 // register, but may have a VGPR only operand.
2941 //
2942 // TODO: On gfx10+, we can easily change the opcode to the e64
2943 // version and use the higher constant bus restriction to avoid this
2944 // copy.
2945
2946 if (!ScavengedVGPR) {
2947 ScavengedVGPR = RS->scavengeRegisterBackwards(
2948 AMDGPU::VGPR_32RegClass, MI, /*RestoreAfter=*/false,
2949 /*SPAdj=*/0);
2950 }
2951
2952 assert(ScavengedVGPR != DstReg);
2953
2954 BuildMI(*MBB, *MI, DL, TII->get(AMDGPU::V_MOV_B32_e32),
2955 ScavengedVGPR)
2956 .addReg(MaterializedReg,
2957 getKillRegState(MaterializedReg != FrameReg));
2958 MaterializedReg = ScavengedVGPR;
2959 }
2960
2961 // TODO: In the flat scratch case, if this is an add of an SGPR, and
2962 // SCC is not live, we could use a scalar add + vector add instead of
2963 // 2 vector adds.
2964 auto AddI32 = BuildMI(*MBB, *MI, DL, TII->get(MI->getOpcode()))
2965 .addDef(DstReg, RegState::Renamable);
2966 if (NumDefs == 2)
2967 AddI32.add(MI->getOperand(1));
2968
2969 RegState MaterializedRegFlags =
2970 getKillRegState(MaterializedReg != FrameReg);
2971
2972 if (isVGPRClass(getPhysRegBaseClass(MaterializedReg))) {
2973 // If we know we have a VGPR already, it's more likely the other
2974 // operand is a legal vsrc0.
2975 AddI32.add(*OtherOp).addReg(MaterializedReg, MaterializedRegFlags);
2976 } else {
2977 // Commute operands to avoid violating VOP2 restrictions. This will
2978 // typically happen when using scratch.
2979 AddI32.addReg(MaterializedReg, MaterializedRegFlags).add(*OtherOp);
2980 }
2981
2982 if (MI->getOpcode() == AMDGPU::V_ADD_CO_U32_e64 ||
2983 MI->getOpcode() == AMDGPU::V_ADD_U32_e64)
2984 AddI32.addImm(0); // clamp
2985
2986 if (MI->getOpcode() == AMDGPU::V_ADD_CO_U32_e32)
2987 AddI32.setOperandDead(3); // Dead vcc
2988
2989 MaterializedReg = DstReg;
2990
2991 OtherOp->ChangeToRegister(MaterializedReg, false);
2992 OtherOp->setIsKill(true);
2994 Offset = 0;
2995 }
2996 } else if (Offset != 0) {
2997 assert(!MaterializedReg);
2999 Offset = 0;
3000 } else {
3001 if (DeadVCC && !HasClamp) {
3002 assert(Offset == 0);
3003
3004 // TODO: Losing kills and implicit operands. Just mutate to copy and
3005 // let lowerCopy deal with it?
3006 if (OtherOp->isReg() && OtherOp->getReg() == DstReg) {
3007 // Folded to an identity copy.
3008 MI->eraseFromParent();
3009 return true;
3010 }
3011
3012 // The immediate value should be in OtherOp
3013 MI->setDesc(TII->get(AMDGPU::V_MOV_B32_e32));
3014 MI->removeOperand(FIOperandNum);
3015
3016 unsigned NumOps = MI->getNumOperands();
3017 for (unsigned I = NumOps - 2; I >= NumDefs + 1; --I)
3018 MI->removeOperand(I);
3019
3020 if (NumDefs == 2)
3021 MI->removeOperand(1);
3022
3023 // The code below can't deal with a mov.
3024 return true;
3025 }
3026
3027 // This folded to a constant, but we have to keep the add around for
3028 // pointless implicit defs or clamp modifier.
3029 FIOp->ChangeToImmediate(0);
3030 }
3031
3032 // Try to improve legality by commuting.
3033 if (!TII->isOperandLegal(*MI, Src1Idx) && TII->commuteInstruction(*MI)) {
3034 std::swap(FIOp, OtherOp);
3035 std::swap(FIOperandNum, OtherOpIdx);
3036 }
3037
3038 // We need at most one mov to satisfy the operand constraints. Prefer to
3039 // move the FI operand first, as it may be a literal in a VOP3
3040 // instruction.
3041 for (unsigned SrcIdx : {FIOperandNum, OtherOpIdx}) {
3042 if (!TII->isOperandLegal(*MI, SrcIdx)) {
3043 // If commuting didn't make the operands legal, we need to materialize
3044 // in a register.
3045 // TODO: Can use SGPR on gfx10+ in some cases.
3046 if (!ScavengedVGPR) {
3047 ScavengedVGPR = RS->scavengeRegisterBackwards(
3048 AMDGPU::VGPR_32RegClass, MI, /*RestoreAfter=*/false,
3049 /*SPAdj=*/0);
3050 }
3051
3052 assert(ScavengedVGPR != DstReg);
3053
3054 MachineOperand &Src = MI->getOperand(SrcIdx);
3055 BuildMI(*MBB, *MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), ScavengedVGPR)
3056 .add(Src);
3057
3058 Src.ChangeToRegister(ScavengedVGPR, false);
3059 Src.setIsKill(true);
3060 break;
3061 }
3062 }
3063
3064 // Fold out add of 0 case that can appear in kernels.
3065 if (FIOp->isImm() && FIOp->getImm() == 0 && DeadVCC && !HasClamp) {
3066 if (OtherOp->isReg() && OtherOp->getReg() != DstReg) {
3067 BuildMI(*MBB, *MI, DL, TII->get(AMDGPU::COPY), DstReg).add(*OtherOp);
3068 }
3069
3070 MI->eraseFromParent();
3071 }
3072
3073 return true;
3074 }
3075 case AMDGPU::S_ADD_I32:
3076 case AMDGPU::S_ADD_U32: {
3077 // TODO: Handle s_or_b32, s_and_b32.
3078 unsigned OtherOpIdx = FIOperandNum == 1 ? 2 : 1;
3079 MachineOperand &OtherOp = MI->getOperand(OtherOpIdx);
3080
3081 assert(FrameReg || MFI->isBottomOfStack());
3082
3083 MachineOperand &DstOp = MI->getOperand(0);
3084 const DebugLoc &DL = MI->getDebugLoc();
3085 Register MaterializedReg = FrameReg;
3086
3087 int64_t Offset = FrameInfo.getObjectOffset(Index);
3088
3089 // See the VALU adds above, with SCC in place of the carry-out.
3090 bool DeadSCC = MI->getOperand(3).isDead();
3091 if (!DeadSCC && foldingOffsetChangesCarry(OtherOp, Offset, FrameReg))
3092 break;
3093
3094 Register TmpReg;
3095
3096 // FIXME: Scavenger should figure out that the result register is
3097 // available. Also should do this for the v_add case.
3098 if (OtherOp.isReg() && OtherOp.getReg() != DstOp.getReg())
3099 TmpReg = DstOp.getReg();
3100
3101 if (FrameReg && !ST.hasFlatScratchEnabled()) {
3102 // FIXME: In the common case where the add does not also read its result
3103 // (i.e. this isn't a reg += fi), it's not finding the dest reg as
3104 // available.
3105 if (!TmpReg)
3106 TmpReg = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3107 MI, /*RestoreAfter=*/false, 0,
3108 /*AllowSpill=*/false);
3109 if (TmpReg) {
3110 BuildMI(*MBB, *MI, DL, TII->get(AMDGPU::S_LSHR_B32))
3111 .addDef(TmpReg, RegState::Renamable)
3112 .addReg(FrameReg)
3113 .addImm(ST.getWavefrontSizeLog2())
3114 .setOperandDead(3); // Set SCC dead
3115 }
3116 MaterializedReg = TmpReg;
3117 }
3118
3119 // For the non-immediate case, we could fall through to the default
3120 // handling, but we do an in-place update of the result register here to
3121 // avoid scavenging another register.
3122 if (OtherOp.isImm()) {
3123 OtherOp.setImm(OtherOp.getImm() + Offset);
3124 Offset = 0;
3125
3126 if (MaterializedReg)
3127 FIOp->ChangeToRegister(MaterializedReg, false);
3128 else
3129 FIOp->ChangeToImmediate(0);
3130 } else if (MaterializedReg) {
3131 // If we can't fold the other operand, do another increment.
3132 Register DstReg = DstOp.getReg();
3133
3134 if (!TmpReg && MaterializedReg == FrameReg) {
3135 TmpReg = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3136 MI, /*RestoreAfter=*/false, 0,
3137 /*AllowSpill=*/false);
3138 DstReg = TmpReg;
3139 }
3140
3141 if (TmpReg) {
3142 auto AddI32 = BuildMI(*MBB, *MI, DL, MI->getDesc())
3143 .addDef(DstReg, RegState::Renamable)
3144 .addReg(MaterializedReg, RegState::Kill)
3145 .add(OtherOp);
3146 if (DeadSCC)
3147 AddI32.setOperandDead(3);
3148
3149 MaterializedReg = DstReg;
3150
3151 OtherOp.ChangeToRegister(MaterializedReg, false);
3152 OtherOp.setIsKill(true);
3153 OtherOp.setIsRenamable(true);
3154 }
3156 } else {
3157 // If we don't have any other offset to apply, we can just directly
3158 // interpret the frame index as the offset.
3160 }
3161
3162 if (DeadSCC && OtherOp.isImm() && OtherOp.getImm() == 0) {
3163 assert(Offset == 0);
3164 MI->removeOperand(3);
3165 MI->removeOperand(OtherOpIdx);
3166 MachineOperand &Src = MI->getOperand(1);
3167 MI->setDesc(TII->get(Src.isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32));
3168 } else if (DeadSCC && FIOp->isImm() && FIOp->getImm() == 0) {
3169 assert(Offset == 0);
3170 MI->removeOperand(3);
3171 MI->removeOperand(FIOperandNum);
3172 MachineOperand &Src = MI->getOperand(1);
3173 MI->setDesc(TII->get(Src.isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32));
3174 }
3175
3176 assert(!FIOp->isFI());
3177 return true;
3178 }
3179 default: {
3180 break;
3181 }
3182 }
3183
3184 int64_t Offset = FrameInfo.getObjectOffset(Index);
3185 if (ST.hasFlatScratchEnabled()) {
3186 if (TII->isFLATScratch(*MI)) {
3187 assert(
3188 (int16_t)FIOperandNum ==
3189 AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::saddr));
3190
3191 // The offset is always swizzled, just replace it
3192 if (FrameReg)
3193 FIOp->ChangeToRegister(FrameReg, false);
3194
3196 TII->getNamedOperand(*MI, AMDGPU::OpName::offset);
3197 int64_t NewOffset = Offset + OffsetOp->getImm();
3198 if (TII->isLegalFLATOffset(NewOffset, AMDGPUAS::PRIVATE_ADDRESS,
3200 OffsetOp->setImm(NewOffset);
3201 if (FrameReg)
3202 return false;
3203 Offset = 0;
3204 }
3205
3206 if (!Offset) {
3207 unsigned Opc = MI->getOpcode();
3208 int NewOpc = -1;
3209 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vaddr)) {
3211 } else if (ST.hasFlatScratchSTMode()) {
3212 // On GFX10 we have ST mode to use no registers for an address.
3213 // Otherwise we need to materialize 0 into an SGPR.
3215 }
3216
3217 if (NewOpc != -1) {
3218 // removeOperand doesn't fixup tied operand indexes as it goes, so
3219 // it asserts. Untie vdst_in for now and retie them afterwards.
3220 int VDstIn =
3221 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
3222 bool TiedVDst = VDstIn != -1 && MI->getOperand(VDstIn).isReg() &&
3223 MI->getOperand(VDstIn).isTied();
3224 if (TiedVDst)
3225 MI->untieRegOperand(VDstIn);
3226
3227 MI->removeOperand(
3228 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::saddr));
3229
3230 if (TiedVDst) {
3231 int NewVDst =
3232 AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst);
3233 int NewVDstIn =
3234 AMDGPU::getNamedOperandIdx(NewOpc, AMDGPU::OpName::vdst_in);
3235 assert(NewVDst != -1 && NewVDstIn != -1 && "Must be tied!");
3236 MI->tieOperands(NewVDst, NewVDstIn);
3237 }
3238 MI->setDesc(TII->get(NewOpc));
3239 return false;
3240 }
3241 }
3242 }
3243
3244 if (!FrameReg) {
3246 if (TII->isOperandLegal(*MI, FIOperandNum, FIOp))
3247 return false;
3248 }
3249
3250 // We need to use register here. Check if we can use an SGPR or need
3251 // a VGPR.
3252 FIOp->ChangeToRegister(AMDGPU::M0, false);
3253 bool UseSGPR = TII->isOperandLegal(*MI, FIOperandNum, FIOp);
3254
3255 if (!Offset && FrameReg && UseSGPR) {
3256 FIOp->setReg(FrameReg);
3257 return false;
3258 }
3259
3260 const TargetRegisterClass *RC =
3261 UseSGPR ? &AMDGPU::SReg_32_XM0RegClass : &AMDGPU::VGPR_32RegClass;
3262
3263 Register TmpReg =
3264 RS->scavengeRegisterBackwards(*RC, MI, false, 0, !UseSGPR);
3265 FIOp->setReg(TmpReg);
3266 FIOp->setIsKill();
3267
3268 if ((!FrameReg || !Offset) && TmpReg) {
3269 unsigned Opc = UseSGPR ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
3270 auto MIB = BuildMI(*MBB, MI, DL, TII->get(Opc), TmpReg);
3271 if (FrameReg)
3272 MIB.addReg(FrameReg);
3273 else
3274 MIB.addImm(Offset);
3275
3276 return false;
3277 }
3278
3279 bool NeedSaveSCC = isSCCLiveInto(*RS, *MI);
3280
3281 Register TmpSReg =
3282 UseSGPR ? TmpReg
3283 : RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3284 MI, false, 0, !UseSGPR);
3285
3286 // If no SGPR was scavenged but a frame register is available, fall
3287 // through to reuse it as the temporary (computed in place, restored
3288 // after). Only bail out when there is no frame register, or a VGPR
3289 // operand is needed but none could be scavenged.
3290 if ((!TmpSReg && !FrameReg) || (!TmpReg && !UseSGPR)) {
3291 int SVfromSSOpcode =
3293 int SVfromSVSOpcode =
3295 int SVOpcode = SVfromSSOpcode != -1 ? SVfromSSOpcode : SVfromSVSOpcode;
3296 if (ST.hasFlatScratchSVSMode() && SVOpcode != -1) {
3297 // SV form encodes only the offset in vaddr; an SS-form scratch op
3298 // keeps its FI in the SGPR saddr, so this is only reached with no
3299 // frame register. SVS form has both vaddr and saddr but still depends
3300 // on the FI being in the SGPR saddr so it is also possible to end up
3301 // here through SVS form without frame register and scavenged SGPR.
3302 assert(!FrameReg &&
3303 "SV-form fallback cannot encode a frame register");
3304
3305 // Fold as much of the constant offset as possible into the SV form
3306 // instruction's immediate offset field, and materialize the
3307 // remainder (plus the frame register, if any) into the scavenged
3308 // VGPR used as the vaddr.
3309 int64_t FullOffset =
3310 Offset +
3311 TII->getNamedOperand(*MI, AMDGPU::OpName::offset)->getImm();
3312 auto [ImmOffset, RemainderOffset] =
3313 TII->splitFlatOffset(FullOffset, AMDGPUAS::PRIVATE_ADDRESS,
3315
3316 Register UsedVAddr;
3317 if (MachineOperand *VAddr =
3318 TII->getNamedOperand(*MI, AMDGPU::OpName::vaddr)) {
3319 MachineOperand *VData =
3320 TII->getNamedOperand(*MI, AMDGPU::OpName::vdata);
3321
3322 // SVS form: add RemainderOffset to vaddr.
3323 Register Src = VAddr->getReg();
3324 bool CanReuseVAddr = VAddr->isKill() &&
3325 !(VData && regsOverlap(Src, VData->getReg()));
3326 Register Dst = CanReuseVAddr ? Src
3327 : RS->scavengeRegisterBackwards(
3328 AMDGPU::VGPR_32RegClass, MI,
3329 false, 0, /*AllowSpill=*/true);
3330 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_ADD_U32_e32), Dst)
3331 .addImm(RemainderOffset)
3332 .addReg(Src, getKillRegState(CanReuseVAddr));
3333 UsedVAddr = Dst;
3334 } else {
3335 // SS form: no vaddr, materialize remainder as vgpr.
3336 UsedVAddr = RS->scavengeRegisterBackwards(
3337 AMDGPU::VGPR_32RegClass, MI, false, 0, /*AllowSpill=*/true);
3338 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), UsedVAddr)
3339 .addImm(RemainderOffset);
3340 }
3341 BuildMI(*MBB, MI, DL, TII->get(SVOpcode))
3342 .add(MI->getOperand(0)) // $vdata
3343 .addReg(UsedVAddr, RegState::Kill) // $vaddr
3344 .addImm(ImmOffset) // $offset
3345 .add(*TII->getNamedOperand(*MI, AMDGPU::OpName::cpol));
3346 MI->eraseFromParent();
3347 return true;
3348 }
3349 report_fatal_error("Cannot scavenge register in FI elimination!");
3350 }
3351
3352 if (!TmpSReg) {
3353 // Use frame register and restore it after.
3354 TmpSReg = FrameReg;
3355 FIOp->setReg(FrameReg);
3356 FIOp->setIsKill(false);
3357 }
3358
3359 if (NeedSaveSCC) {
3360 assert(!(Offset & 0x1) && "Flat scratch offset must be aligned!");
3361 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_ADDC_U32), TmpSReg)
3362 .addReg(FrameReg)
3363 .addImm(Offset);
3364 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_BITCMP1_B32))
3365 .addReg(TmpSReg)
3366 .addImm(0);
3367 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_BITSET0_B32), TmpSReg)
3368 .addImm(0)
3369 .addReg(TmpSReg);
3370 } else {
3371 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_ADD_I32), TmpSReg)
3372 .addReg(FrameReg)
3373 .addImm(Offset);
3374 }
3375
3376 if (!UseSGPR)
3377 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), TmpReg)
3378 .addReg(TmpSReg, RegState::Kill);
3379
3380 if (TmpSReg == FrameReg) {
3381 // Undo frame register modification.
3382 if (NeedSaveSCC &&
3383 !MI->registerDefIsDead(AMDGPU::SCC, /*TRI=*/nullptr)) {
3385 BuildMI(*MBB, std::next(MI), DL, TII->get(AMDGPU::S_ADDC_U32),
3386 TmpSReg)
3387 .addReg(FrameReg)
3388 .addImm(-Offset);
3389 I = BuildMI(*MBB, std::next(I), DL, TII->get(AMDGPU::S_BITCMP1_B32))
3390 .addReg(TmpSReg)
3391 .addImm(0);
3392 BuildMI(*MBB, std::next(I), DL, TII->get(AMDGPU::S_BITSET0_B32),
3393 TmpSReg)
3394 .addImm(0)
3395 .addReg(TmpSReg);
3396 } else {
3397 BuildMI(*MBB, std::next(MI), DL, TII->get(AMDGPU::S_ADD_I32),
3398 FrameReg)
3399 .addReg(FrameReg)
3400 .addImm(-Offset);
3401 }
3402 }
3403
3404 return false;
3405 }
3406
3407 bool IsMUBUF = TII->isMUBUF(*MI);
3408
3409 if (!IsMUBUF && !MFI->isBottomOfStack()) {
3410 // Convert to a swizzled stack address by scaling by the wave size.
3411 // In an entry function/kernel the offset is already swizzled.
3412 bool IsSALU = isSGPRClass(TII->getRegClass(MI->getDesc(), FIOperandNum));
3413 bool LiveSCC = isSCCLiveInto(*RS, *MI);
3414 // The scavenger is positioned at the liveness state immediately after MI,
3415 // so we need only check if SCC is used.
3416 bool SCCLiveAfterMI = RS->isRegUsed(AMDGPU::SCC);
3417 const TargetRegisterClass *RC = IsSALU && !LiveSCC
3418 ? &AMDGPU::SReg_32RegClass
3419 : &AMDGPU::VGPR_32RegClass;
3420 bool IsCopy = MI->getOpcode() == AMDGPU::V_MOV_B32_e32 ||
3421 MI->getOpcode() == AMDGPU::V_MOV_B32_e64 ||
3422 MI->getOpcode() == AMDGPU::S_MOV_B32;
3423
3424 int64_t Offset = FrameInfo.getObjectOffset(Index);
3425
3426 // Scaling FrameReg in place is the last resort when there is nothing to
3427 // scavenge. It has to be undone after MI, which is only possible while MI
3428 // does not use FrameReg for anything besides the frame index.
3429 bool CanUseFrameRegAsSGPRScratch = IsSALU && FrameReg &&
3430 !MI->readsRegister(FrameReg, this) &&
3431 !MI->modifiesRegister(FrameReg, this);
3432 // ResultReg is a VGPR while SCC is live, so FrameReg cannot stand in for
3433 // it there.
3434 bool CanUseFrameRegAsScratch = CanUseFrameRegAsSGPRScratch && !LiveSCC;
3435
3436 bool RestoreFrameReg = false;
3437
3438 Register ResultReg;
3439 if (IsCopy) {
3440 ResultReg = MI->getOperand(0).getReg();
3441 } else {
3442 ResultReg = RS->scavengeRegisterBackwards(*RC, MI, false, 0,
3443 /*AllowSpill=*/false);
3444 if (!ResultReg) {
3445 if (CanUseFrameRegAsScratch) {
3446 // Spilling an SGPR here instead would flip EXEC with S_NOT, and
3447 // that clobbers the SCC MI may be defining for a later use.
3448 ResultReg = FrameReg;
3449 RestoreFrameReg = true;
3450 } else {
3451 ResultReg = RS->scavengeRegisterBackwards(*RC, MI, false, 0);
3452 }
3453 }
3454 }
3455
3456 // The carry-out lane of Add is unused, so it is safe to write with
3457 // S_MOV_B32 even into a VGPR.
3458 auto MaterializeCarryOutOffset = [&](MachineInstrBuilder &Add) {
3459 Register ConstOffsetReg =
3460 isWave32 ? Add.getReg(1)
3461 : Register(getSubReg(Add.getReg(1), AMDGPU::sub0));
3462 BuildMI(*MBB, *Add, DL, TII->get(AMDGPU::S_MOV_B32), ConstOffsetReg)
3463 .addImm(Offset);
3464 return ConstOffsetReg;
3465 };
3466
3467 if (Offset == 0) {
3468 unsigned OpCode =
3469 IsSALU && !LiveSCC ? AMDGPU::S_LSHR_B32 : AMDGPU::V_LSHRREV_B32_e64;
3470 Register TmpResultReg = ResultReg;
3471 if (IsSALU && LiveSCC) {
3472 TmpResultReg = RS->scavengeRegisterBackwards(AMDGPU::VGPR_32RegClass,
3473 MI, false, 0);
3474 }
3475
3476 auto Shift = BuildMI(*MBB, MI, DL, TII->get(OpCode), TmpResultReg);
3477 if (OpCode == AMDGPU::V_LSHRREV_B32_e64)
3478 // For V_LSHRREV, the operands are reversed (the shift count goes
3479 // first).
3480 Shift.addImm(ST.getWavefrontSizeLog2()).addReg(FrameReg);
3481 else
3482 Shift.addReg(FrameReg).addImm(ST.getWavefrontSizeLog2());
3483 if (IsSALU && !LiveSCC)
3484 Shift.getInstr()->getOperand(3).setIsDead(); // Mark SCC as dead.
3485 if (IsSALU && LiveSCC) {
3486 Register NewDest;
3487 if (IsCopy) {
3488 assert(ResultReg.isPhysical());
3489 NewDest = ResultReg;
3490 } else {
3491 // Spilling an SGPR here would flip EXEC with S_NOT, and that
3492 // clobbers the SCC this path exists to preserve, so scale FrameReg
3493 // in place instead.
3494 NewDest = RS->scavengeRegisterBackwards(AMDGPU::SReg_32_XM0RegClass,
3495 Shift, false, 0,
3496 /*AllowSpill=*/false);
3497 if (!NewDest) {
3498 if (CanUseFrameRegAsSGPRScratch) {
3499 NewDest = FrameReg;
3500 RestoreFrameReg = true;
3501 } else {
3502 // Nothing is left to scale in place, so fall back to the SGPR
3503 // spill even though it clobbers SCC.
3505 "unhandled SGPR spill to memory");
3506 NewDest = RS->scavengeRegisterBackwards(
3507 AMDGPU::SReg_32_XM0RegClass, Shift, false, 0);
3508 }
3509 }
3510 }
3511 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), NewDest)
3512 .addReg(TmpResultReg);
3513 ResultReg = NewDest;
3514 }
3515 } else {
3517 if (!IsSALU) {
3518 if ((MIB = TII->getAddNoCarry(*MBB, MI, DL, ResultReg, *RS)) !=
3519 nullptr) {
3520 // Reuse ResultReg in intermediate step.
3521 Register ScaledReg = ResultReg;
3522
3523 BuildMI(*MBB, *MIB, DL, TII->get(AMDGPU::V_LSHRREV_B32_e64),
3524 ScaledReg)
3525 .addImm(ST.getWavefrontSizeLog2())
3526 .addReg(FrameReg);
3527
3528 const bool IsVOP2 = MIB->getOpcode() == AMDGPU::V_ADD_U32_e32;
3529
3530 // TODO: Fold if use instruction is another add of a constant.
3531 if (IsVOP2 ||
3532 AMDGPU::isInlinableLiteral32(Offset, ST.hasInv2PiInlineImm())) {
3533 // FIXME: This can fail
3534 MIB.addImm(Offset);
3535 MIB.addReg(ScaledReg, RegState::Kill);
3536 if (!IsVOP2)
3537 MIB.addImm(0); // clamp bit
3538 } else {
3539 assert(MIB->getOpcode() == AMDGPU::V_ADD_CO_U32_e64 &&
3540 "Need to reuse carry out register");
3541
3542 MIB.addReg(MaterializeCarryOutOffset(MIB), RegState::Kill);
3543 MIB.addReg(ScaledReg, RegState::Kill);
3544 MIB.addImm(0); // clamp bit
3545 }
3546 }
3547 }
3548 if (!MIB || IsSALU) {
3549 // We have to produce a carry out, and there isn't a free SGPR pair
3550 // for it. We can keep the whole computation on the SALU to avoid
3551 // clobbering an additional register at the cost of an extra mov.
3552
3553 // We may have 1 free scratch SGPR even though a carry out is
3554 // unavailable. Only one additional mov is needed.
3555 Register TmpScaledReg = IsCopy && IsSALU
3556 ? ResultReg
3557 : RS->scavengeRegisterBackwards(
3558 AMDGPU::SReg_32_XM0RegClass, MI,
3559 false, 0, /*AllowSpill=*/false);
3560 // A scalar result is already materialized in ResultReg, which holds
3561 // the scavenged register, or FrameReg itself if nothing was free.
3562 Register ScaledReg = TmpScaledReg;
3563 if (!ScaledReg.isValid())
3564 ScaledReg = IsSALU ? ResultReg : FrameReg;
3565 Register TmpResultReg = ScaledReg;
3566
3567 if (!LiveSCC) {
3568 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_LSHR_B32), TmpResultReg)
3569 .addReg(FrameReg)
3570 .addImm(ST.getWavefrontSizeLog2());
3571 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_ADD_I32), TmpResultReg)
3572 .addReg(TmpResultReg, RegState::Kill)
3573 .addImm(Offset);
3574 } else {
3575 TmpResultReg = RS->scavengeRegisterBackwards(
3576 AMDGPU::VGPR_32RegClass, MI, false, 0, /*AllowSpill=*/true);
3577
3579 if ((Add = TII->getAddNoCarry(*MBB, MI, DL, TmpResultReg, *RS))) {
3580 BuildMI(*MBB, *Add, DL, TII->get(AMDGPU::V_LSHRREV_B32_e64),
3581 TmpResultReg)
3582 .addImm(ST.getWavefrontSizeLog2())
3583 .addReg(FrameReg);
3584 if (Add->getOpcode() == AMDGPU::V_ADD_CO_U32_e64) {
3585 Add.addReg(MaterializeCarryOutOffset(Add), RegState::Kill)
3586 .addReg(TmpResultReg, RegState::Kill)
3587 .addImm(0);
3588 } else
3589 Add.addImm(Offset).addReg(TmpResultReg, RegState::Kill);
3590 } else {
3591 assert(Offset > 0 && isUInt<24>(2 * ST.getMaxWaveScratchSize()) &&
3592 "offset is unsafe for v_mad_u32_u24");
3593
3594 // We start with a frame pointer with a wave space value, and
3595 // an offset in lane-space. We are materializing a lane space
3596 // value. We can either do a right shift of the frame pointer
3597 // to get to lane space, or a left shift of the offset to get
3598 // to wavespace. We can right shift after the computation to
3599 // get back to the desired per-lane value. We are using the
3600 // mad_u32_u24 primarily as an add with no carry out clobber.
3601 bool IsInlinableLiteral =
3602 AMDGPU::isInlinableLiteral32(Offset, ST.hasInv2PiInlineImm());
3603 if (!IsInlinableLiteral) {
3604 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32),
3605 TmpResultReg)
3606 .addImm(Offset);
3607 }
3608
3609 Add = BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_MAD_U32_U24_e64),
3610 TmpResultReg);
3611
3612 if (!IsInlinableLiteral) {
3613 Add.addReg(TmpResultReg, RegState::Kill);
3614 } else {
3615 // We fold the offset into mad itself if its inlinable.
3616 Add.addImm(Offset);
3617 }
3618 Add.addImm(ST.getWavefrontSize()).addReg(FrameReg).addImm(0);
3619 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_LSHRREV_B32_e64),
3620 TmpResultReg)
3621 .addImm(ST.getWavefrontSizeLog2())
3622 .addReg(TmpResultReg);
3623 }
3624
3625 Register NewDest;
3626 if (IsCopy) {
3627 NewDest = ResultReg;
3628 } else {
3629 // As above, an SGPR spill here would clobber the live SCC.
3630 NewDest = RS->scavengeRegisterBackwards(
3631 AMDGPU::SReg_32_XM0RegClass, *Add, false, 0,
3632 /*AllowSpill=*/false);
3633 if (!NewDest) {
3634 if (CanUseFrameRegAsSGPRScratch) {
3635 NewDest = FrameReg;
3636 RestoreFrameReg = true;
3637 } else {
3638 // As above, fall back to the SCC-clobbering SGPR spill.
3640 "unhandled SGPR spill to memory");
3641 NewDest = RS->scavengeRegisterBackwards(
3642 AMDGPU::SReg_32_XM0RegClass, *Add, false, 0);
3643 }
3644 }
3645 }
3646
3647 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
3648 NewDest)
3649 .addReg(TmpResultReg);
3650 ResultReg = NewDest;
3651 }
3652 // A scalar result still reads FrameReg at MI, so FrameReg is
3653 // restored after MI instead.
3654 if (!IsSALU) {
3655 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::COPY), ResultReg)
3656 .addReg(TmpResultReg, RegState::Kill);
3657 // If there were truly no free SGPRs, we need to undo everything.
3658 if (!TmpScaledReg.isValid()) {
3659 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_ADD_I32), ScaledReg)
3660 .addReg(ScaledReg, RegState::Kill)
3661 .addImm(-Offset);
3662 BuildMI(*MBB, MI, DL, TII->get(AMDGPU::S_LSHL_B32), ScaledReg)
3663 .addReg(FrameReg)
3664 .addImm(ST.getWavefrontSizeLog2());
3665 }
3666 }
3667 }
3668 }
3669
3670 if (RestoreFrameReg) {
3671 // Put FrameReg back now that MI has consumed the scaled address.
3672 // S_MUL_I32 undoes the scaling without writing SCC, which S_LSHL_B32
3673 // would. When MI leaves SCC live, fold the offset back with the carry
3674 // sequence that smuggles SCC through bit 0, which the scaling has just
3675 // cleared.
3676 MachineBasicBlock::iterator InsPt = std::next(MI);
3677 BuildMI(*MBB, InsPt, DL, TII->get(AMDGPU::S_MUL_I32), FrameReg)
3678 .addReg(FrameReg)
3679 .addImm(ST.getWavefrontSize());
3680
3681 if (Offset) {
3682 int64_t ScaledOffset = -Offset * ST.getWavefrontSize();
3683 if (!SCCLiveAfterMI) {
3684 BuildMI(*MBB, InsPt, DL, TII->get(AMDGPU::S_ADD_I32), FrameReg)
3685 .addReg(FrameReg)
3686 .addImm(ScaledOffset);
3687 } else {
3688 BuildMI(*MBB, InsPt, DL, TII->get(AMDGPU::S_ADDC_U32), FrameReg)
3689 .addReg(FrameReg)
3690 .addImm(ScaledOffset);
3691 BuildMI(*MBB, InsPt, DL, TII->get(AMDGPU::S_BITCMP1_B32))
3692 .addReg(FrameReg)
3693 .addImm(0);
3694 BuildMI(*MBB, InsPt, DL, TII->get(AMDGPU::S_BITSET0_B32), FrameReg)
3695 .addImm(0)
3696 .addReg(FrameReg);
3697 }
3698 }
3699 }
3700
3701 // Don't introduce an extra copy if we're just materializing in a mov.
3702 if (IsCopy) {
3703 MI->eraseFromParent();
3704 return true;
3705 }
3706 // FrameReg is restored after MI, so MI does not kill it.
3707 FIOp->ChangeToRegister(ResultReg, false, false, !RestoreFrameReg);
3708 return false;
3709 }
3710
3711 if (IsMUBUF) {
3712 // Disable offen so we don't need a 0 vgpr base.
3713 assert(
3714 static_cast<int>(FIOperandNum) ==
3715 AMDGPU::getNamedOperandIdx(MI->getOpcode(), AMDGPU::OpName::vaddr));
3716
3717 auto &SOffset = *TII->getNamedOperand(*MI, AMDGPU::OpName::soffset);
3718 assert((SOffset.isImm() && SOffset.getImm() == 0));
3719
3720 if (FrameReg != AMDGPU::NoRegister)
3721 SOffset.ChangeToRegister(FrameReg, false);
3722
3723 int64_t Offset = FrameInfo.getObjectOffset(Index);
3724 int64_t OldImm =
3725 TII->getNamedOperand(*MI, AMDGPU::OpName::offset)->getImm();
3726 int64_t NewOffset = OldImm + Offset;
3727
3728 if (TII->isLegalMUBUFImmOffset(NewOffset) &&
3729 buildMUBUFOffsetLoadStore(ST, FrameInfo, MI, Index, NewOffset)) {
3730 MI->eraseFromParent();
3731 return true;
3732 }
3733 }
3734
3735 // If the offset is simply too big, don't convert to a scratch wave offset
3736 // relative index.
3737
3739
3740 // Not isImmOperandLegal: a SALU user may already have a literal.
3741 if (!TII->isOperandLegal(*MI, FIOperandNum, FIOp)) {
3742 const TargetRegisterClass *OpRC =
3743 TII->getRegClass(MI->getDesc(), FIOperandNum);
3744 bool UseSGPR = OpRC && isSGPRClass(OpRC);
3745
3746 const TargetRegisterClass *RC =
3747 UseSGPR ? &AMDGPU::SReg_32_XM0RegClass : &AMDGPU::VGPR_32RegClass;
3748 Register TmpReg = RS->scavengeRegisterBackwards(*RC, MI, false, 0);
3749 BuildMI(*MBB, MI, DL,
3750 TII->get(UseSGPR ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32),
3751 TmpReg)
3752 .addImm(Offset);
3753 FIOp->ChangeToRegister(TmpReg, false, false, true);
3754 }
3755
3756 return false;
3757}
3758
3762
3764 return getEncodingValue(Reg) & AMDGPU::HWEncoding::REG_IDX_MASK;
3765}
3766
3767static const TargetRegisterClass *
3769 if (BitWidth == 64)
3770 return &AMDGPU::VReg_64RegClass;
3771 if (BitWidth == 96)
3772 return &AMDGPU::VReg_96RegClass;
3773 if (BitWidth == 128)
3774 return &AMDGPU::VReg_128RegClass;
3775 if (BitWidth == 160)
3776 return &AMDGPU::VReg_160RegClass;
3777 if (BitWidth == 192)
3778 return &AMDGPU::VReg_192RegClass;
3779 if (BitWidth == 224)
3780 return &AMDGPU::VReg_224RegClass;
3781 if (BitWidth == 256)
3782 return &AMDGPU::VReg_256RegClass;
3783 if (BitWidth == 288)
3784 return &AMDGPU::VReg_288RegClass;
3785 if (BitWidth == 320)
3786 return &AMDGPU::VReg_320RegClass;
3787 if (BitWidth == 352)
3788 return &AMDGPU::VReg_352RegClass;
3789 if (BitWidth == 384)
3790 return &AMDGPU::VReg_384RegClass;
3791 if (BitWidth == 512)
3792 return &AMDGPU::VReg_512RegClass;
3793 if (BitWidth == 1024)
3794 return &AMDGPU::VReg_1024RegClass;
3795
3796 return nullptr;
3797}
3798
3799static const TargetRegisterClass *
3801 if (BitWidth == 64)
3802 return &AMDGPU::VReg_64_Align2RegClass;
3803 if (BitWidth == 96)
3804 return &AMDGPU::VReg_96_Align2RegClass;
3805 if (BitWidth == 128)
3806 return &AMDGPU::VReg_128_Align2RegClass;
3807 if (BitWidth == 160)
3808 return &AMDGPU::VReg_160_Align2RegClass;
3809 if (BitWidth == 192)
3810 return &AMDGPU::VReg_192_Align2RegClass;
3811 if (BitWidth == 224)
3812 return &AMDGPU::VReg_224_Align2RegClass;
3813 if (BitWidth == 256)
3814 return &AMDGPU::VReg_256_Align2RegClass;
3815 if (BitWidth == 288)
3816 return &AMDGPU::VReg_288_Align2RegClass;
3817 if (BitWidth == 320)
3818 return &AMDGPU::VReg_320_Align2RegClass;
3819 if (BitWidth == 352)
3820 return &AMDGPU::VReg_352_Align2RegClass;
3821 if (BitWidth == 384)
3822 return &AMDGPU::VReg_384_Align2RegClass;
3823 if (BitWidth == 512)
3824 return &AMDGPU::VReg_512_Align2RegClass;
3825 if (BitWidth == 1024)
3826 return &AMDGPU::VReg_1024_Align2RegClass;
3827
3828 return nullptr;
3829}
3830
3831const TargetRegisterClass *
3833 if (BitWidth == 1)
3834 return &AMDGPU::VReg_1RegClass;
3835 if (BitWidth == 16)
3836 return &AMDGPU::VGPR_16RegClass;
3837 if (BitWidth == 32)
3838 return &AMDGPU::VGPR_32RegClass;
3839 return ST.needsAlignedVGPRs() ? getAlignedVGPRClassForBitWidth(BitWidth)
3841}
3842
3843const TargetRegisterClass *
3845 if (BitWidth <= 32)
3846 return &AMDGPU::VGPR_32_Lo256RegClass;
3847 if (BitWidth <= 64)
3848 return &AMDGPU::VReg_64_Lo256_Align2RegClass;
3849 if (BitWidth <= 96)
3850 return &AMDGPU::VReg_96_Lo256_Align2RegClass;
3851 if (BitWidth <= 128)
3852 return &AMDGPU::VReg_128_Lo256_Align2RegClass;
3853 if (BitWidth <= 160)
3854 return &AMDGPU::VReg_160_Lo256_Align2RegClass;
3855 if (BitWidth <= 192)
3856 return &AMDGPU::VReg_192_Lo256_Align2RegClass;
3857 if (BitWidth <= 224)
3858 return &AMDGPU::VReg_224_Lo256_Align2RegClass;
3859 if (BitWidth <= 256)
3860 return &AMDGPU::VReg_256_Lo256_Align2RegClass;
3861 if (BitWidth <= 288)
3862 return &AMDGPU::VReg_288_Lo256_Align2RegClass;
3863 if (BitWidth <= 320)
3864 return &AMDGPU::VReg_320_Lo256_Align2RegClass;
3865 if (BitWidth <= 352)
3866 return &AMDGPU::VReg_352_Lo256_Align2RegClass;
3867 if (BitWidth <= 384)
3868 return &AMDGPU::VReg_384_Lo256_Align2RegClass;
3869 if (BitWidth <= 512)
3870 return &AMDGPU::VReg_512_Lo256_Align2RegClass;
3871 if (BitWidth <= 1024)
3872 return &AMDGPU::VReg_1024_Lo256_Align2RegClass;
3873
3874 return nullptr;
3875}
3876
3877static const TargetRegisterClass *
3879 if (BitWidth == 64)
3880 return &AMDGPU::AReg_64RegClass;
3881 if (BitWidth == 96)
3882 return &AMDGPU::AReg_96RegClass;
3883 if (BitWidth == 128)
3884 return &AMDGPU::AReg_128RegClass;
3885 if (BitWidth == 160)
3886 return &AMDGPU::AReg_160RegClass;
3887 if (BitWidth == 192)
3888 return &AMDGPU::AReg_192RegClass;
3889 if (BitWidth == 224)
3890 return &AMDGPU::AReg_224RegClass;
3891 if (BitWidth == 256)
3892 return &AMDGPU::AReg_256RegClass;
3893 if (BitWidth == 288)
3894 return &AMDGPU::AReg_288RegClass;
3895 if (BitWidth == 320)
3896 return &AMDGPU::AReg_320RegClass;
3897 if (BitWidth == 352)
3898 return &AMDGPU::AReg_352RegClass;
3899 if (BitWidth == 384)
3900 return &AMDGPU::AReg_384RegClass;
3901 if (BitWidth == 512)
3902 return &AMDGPU::AReg_512RegClass;
3903 if (BitWidth == 1024)
3904 return &AMDGPU::AReg_1024RegClass;
3905
3906 return nullptr;
3907}
3908
3909static const TargetRegisterClass *
3911 if (BitWidth == 64)
3912 return &AMDGPU::AReg_64_Align2RegClass;
3913 if (BitWidth == 96)
3914 return &AMDGPU::AReg_96_Align2RegClass;
3915 if (BitWidth == 128)
3916 return &AMDGPU::AReg_128_Align2RegClass;
3917 if (BitWidth == 160)
3918 return &AMDGPU::AReg_160_Align2RegClass;
3919 if (BitWidth == 192)
3920 return &AMDGPU::AReg_192_Align2RegClass;
3921 if (BitWidth == 224)
3922 return &AMDGPU::AReg_224_Align2RegClass;
3923 if (BitWidth == 256)
3924 return &AMDGPU::AReg_256_Align2RegClass;
3925 if (BitWidth == 288)
3926 return &AMDGPU::AReg_288_Align2RegClass;
3927 if (BitWidth == 320)
3928 return &AMDGPU::AReg_320_Align2RegClass;
3929 if (BitWidth == 352)
3930 return &AMDGPU::AReg_352_Align2RegClass;
3931 if (BitWidth == 384)
3932 return &AMDGPU::AReg_384_Align2RegClass;
3933 if (BitWidth == 512)
3934 return &AMDGPU::AReg_512_Align2RegClass;
3935 if (BitWidth == 1024)
3936 return &AMDGPU::AReg_1024_Align2RegClass;
3937
3938 return nullptr;
3939}
3940
3941const TargetRegisterClass *
3943 if (BitWidth == 16)
3944 return &AMDGPU::AGPR_LO16RegClass;
3945 if (BitWidth == 32)
3946 return &AMDGPU::AGPR_32RegClass;
3947 return ST.needsAlignedVGPRs() ? getAlignedAGPRClassForBitWidth(BitWidth)
3949}
3950
3951static const TargetRegisterClass *
3953 if (BitWidth == 64)
3954 return &AMDGPU::AV_64RegClass;
3955 if (BitWidth == 96)
3956 return &AMDGPU::AV_96RegClass;
3957 if (BitWidth == 128)
3958 return &AMDGPU::AV_128RegClass;
3959 if (BitWidth == 160)
3960 return &AMDGPU::AV_160RegClass;
3961 if (BitWidth == 192)
3962 return &AMDGPU::AV_192RegClass;
3963 if (BitWidth == 224)
3964 return &AMDGPU::AV_224RegClass;
3965 if (BitWidth == 256)
3966 return &AMDGPU::AV_256RegClass;
3967 if (BitWidth == 288)
3968 return &AMDGPU::AV_288RegClass;
3969 if (BitWidth == 320)
3970 return &AMDGPU::AV_320RegClass;
3971 if (BitWidth == 352)
3972 return &AMDGPU::AV_352RegClass;
3973 if (BitWidth == 384)
3974 return &AMDGPU::AV_384RegClass;
3975 if (BitWidth == 512)
3976 return &AMDGPU::AV_512RegClass;
3977 if (BitWidth == 1024)
3978 return &AMDGPU::AV_1024RegClass;
3979
3980 return nullptr;
3981}
3982
3983static const TargetRegisterClass *
3985 if (BitWidth == 64)
3986 return &AMDGPU::AV_64_Align2RegClass;
3987 if (BitWidth == 96)
3988 return &AMDGPU::AV_96_Align2RegClass;
3989 if (BitWidth == 128)
3990 return &AMDGPU::AV_128_Align2RegClass;
3991 if (BitWidth == 160)
3992 return &AMDGPU::AV_160_Align2RegClass;
3993 if (BitWidth == 192)
3994 return &AMDGPU::AV_192_Align2RegClass;
3995 if (BitWidth == 224)
3996 return &AMDGPU::AV_224_Align2RegClass;
3997 if (BitWidth == 256)
3998 return &AMDGPU::AV_256_Align2RegClass;
3999 if (BitWidth == 288)
4000 return &AMDGPU::AV_288_Align2RegClass;
4001 if (BitWidth == 320)
4002 return &AMDGPU::AV_320_Align2RegClass;
4003 if (BitWidth == 352)
4004 return &AMDGPU::AV_352_Align2RegClass;
4005 if (BitWidth == 384)
4006 return &AMDGPU::AV_384_Align2RegClass;
4007 if (BitWidth == 512)
4008 return &AMDGPU::AV_512_Align2RegClass;
4009 if (BitWidth == 1024)
4010 return &AMDGPU::AV_1024_Align2RegClass;
4011
4012 return nullptr;
4013}
4014
4015const TargetRegisterClass *
4017 if (BitWidth == 32)
4018 return &AMDGPU::AV_32RegClass;
4019 return ST.needsAlignedVGPRs()
4022}
4023
4024const TargetRegisterClass *
4026 // TODO: In principle this should use AV classes for gfx908 too. This is
4027 // limited to 90a+ to avoid regressing special case copy optimizations which
4028 // need new handling. The core issue is that it's not possible to directly
4029 // copy between AGPRs on gfx908, and the current optimizations around that
4030 // expect to see copies to VGPR.
4031 return ST.hasGFX90AInsts() ? getVectorSuperClassForBitWidth(BitWidth)
4033}
4034
4035const TargetRegisterClass *
4037 if (BitWidth == 16 || BitWidth == 32)
4038 return &AMDGPU::SReg_32RegClass;
4039 if (BitWidth == 64)
4040 return &AMDGPU::SReg_64RegClass;
4041 if (BitWidth == 96)
4042 return &AMDGPU::SGPR_96RegClass;
4043 if (BitWidth == 128)
4044 return &AMDGPU::SGPR_128RegClass;
4045 if (BitWidth == 160)
4046 return &AMDGPU::SGPR_160RegClass;
4047 if (BitWidth == 192)
4048 return &AMDGPU::SGPR_192RegClass;
4049 if (BitWidth == 224)
4050 return &AMDGPU::SGPR_224RegClass;
4051 if (BitWidth == 256)
4052 return &AMDGPU::SGPR_256RegClass;
4053 if (BitWidth == 288)
4054 return &AMDGPU::SGPR_288RegClass;
4055 if (BitWidth == 320)
4056 return &AMDGPU::SGPR_320RegClass;
4057 if (BitWidth == 352)
4058 return &AMDGPU::SGPR_352RegClass;
4059 if (BitWidth == 384)
4060 return &AMDGPU::SGPR_384RegClass;
4061 if (BitWidth == 512)
4062 return &AMDGPU::SGPR_512RegClass;
4063 if (BitWidth == 1024)
4064 return &AMDGPU::SGPR_1024RegClass;
4065
4066 return nullptr;
4067}
4068
4070 Register Reg) const {
4071 const TargetRegisterClass *RC;
4072 if (Reg.isVirtual())
4073 RC = MRI.getRegClass(Reg);
4074 else
4075 RC = getPhysRegBaseClass(Reg);
4076 return RC && isSGPRClass(RC);
4077}
4078
4079const TargetRegisterClass *
4081 unsigned Size = getRegSizeInBits(*SRC);
4082
4083 switch (SRC->getID()) {
4084 default:
4085 break;
4086 case AMDGPU::VS_16_Lo128RegClassID:
4087 return getAllocatableClass(&AMDGPU::VGPR_16_Lo128RegClass);
4088 case AMDGPU::VS_32_Lo128RegClassID:
4089 return getAllocatableClass(&AMDGPU::VGPR_32_Lo128RegClass);
4090 case AMDGPU::VS_32_Lo256RegClassID:
4091 case AMDGPU::VS_64_Lo256RegClassID:
4092 return getAllocatableClass(getAlignedLo256VGPRClassForBitWidth(Size));
4093 }
4094
4095 const TargetRegisterClass *VRC =
4096 getAllocatableClass(getVGPRClassForBitWidth(Size));
4097 assert(VRC && "Invalid register class size");
4098 return VRC;
4099}
4100
4101const TargetRegisterClass *
4103 unsigned Size = getRegSizeInBits(*SRC);
4105 assert(ARC && "Invalid register class size");
4106 return ARC;
4107}
4108
4109const TargetRegisterClass *
4111 unsigned Size = getRegSizeInBits(*SRC);
4113 assert(ARC && "Invalid register class size");
4114 return ARC;
4115}
4116
4117const TargetRegisterClass *
4119 unsigned Size = getRegSizeInBits(*VRC);
4120 if (Size == 32)
4121 return &AMDGPU::SGPR_32RegClass;
4123 assert(SRC && "Invalid register class size");
4124 return SRC;
4125}
4126
4127const TargetRegisterClass *
4129 const TargetRegisterClass *SubRC,
4130 unsigned SubIdx) const {
4131 // Ensure this subregister index is aligned in the super register.
4132 const TargetRegisterClass *MatchRC =
4133 getMatchingSuperRegClass(SuperRC, SubRC, SubIdx);
4134 return MatchRC && MatchRC->hasSubClassEq(SuperRC) ? MatchRC : nullptr;
4135}
4136
4137bool SIRegisterInfo::opCanUseInlineConstant(unsigned OpType) const {
4140 return !ST.hasMFMAInlineLiteralBug();
4141
4142 return OpType >= AMDGPU::OPERAND_SRC_FIRST &&
4143 OpType <= AMDGPU::OPERAND_SRC_LAST;
4144}
4145
4146bool SIRegisterInfo::opCanUseLiteralConstant(unsigned OpType) const {
4147 // TODO: 64-bit operands have extending behavior from 32-bit literal.
4148 return OpType >= AMDGPU::OPERAND_REG_IMM_FIRST &&
4150}
4151
4152/// Returns a lowest register that is not used at any point in the function.
4153/// If all registers are used, then this function will return
4154/// AMDGPU::NoRegister. If \p ReserveHighestRegister = true, then return
4155/// highest unused register.
4157 const MachineRegisterInfo &MRI, const TargetRegisterClass *RC,
4158 const MachineFunction &MF, bool ReserveHighestRegister) const {
4159 // Never offer VCC as an unused register.
4160 auto isVCC = [](MCRegister Reg) {
4161 return Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::VCC_HI;
4162 };
4163
4164 if (ReserveHighestRegister) {
4165 for (MCRegister Reg : reverse(*RC))
4166 if (MRI.isAllocatable(Reg) && !MRI.isPhysRegUsed(Reg) && !isVCC(Reg))
4167 return Reg;
4168 } else {
4169 for (MCRegister Reg : *RC)
4170 if (MRI.isAllocatable(Reg) && !MRI.isPhysRegUsed(Reg) && !isVCC(Reg))
4171 return Reg;
4172 }
4173 return MCRegister();
4174}
4175
4177 const RegisterBankInfo &RBI,
4178 Register Reg) const {
4179 auto *RB = RBI.getRegBank(Reg, MRI, *this);
4180 if (!RB)
4181 return false;
4182
4183 return !RBI.isDivergentRegBank(RB);
4184}
4185
4187 unsigned EltSize) const {
4188 const unsigned RegBitWidth = AMDGPU::getRegBitWidth(*RC);
4189 assert(RegBitWidth >= 32 && RegBitWidth <= 1024 && EltSize >= 2);
4190
4191 const unsigned RegHalves = RegBitWidth / 16;
4192 const unsigned EltHalves = EltSize / 2;
4193 assert(RegSplitParts.size() + 1 >= EltHalves);
4194
4195 const std::vector<int16_t> &Parts = RegSplitParts[EltHalves - 1];
4196 const unsigned NumParts = RegHalves / EltHalves;
4197
4198 return ArrayRef(Parts.data(), NumParts);
4199}
4200
4203 Register Reg) const {
4204 return Reg.isVirtual() ? MRI.getRegClass(Reg) : getPhysRegBaseClass(Reg);
4205}
4206
4207const TargetRegisterClass *
4209 const MachineOperand &MO) const {
4210 const TargetRegisterClass *SrcRC = getRegClassForReg(MRI, MO.getReg());
4211 return getSubRegisterClass(SrcRC, MO.getSubReg());
4212}
4213
4215 Register Reg) const {
4216 const TargetRegisterClass *RC = getRegClassForReg(MRI, Reg);
4217 // Registers without classes are unaddressable, SGPR-like registers.
4218 return RC && isVGPRClass(RC);
4219}
4220
4222 Register Reg) const {
4223 const TargetRegisterClass *RC = getRegClassForReg(MRI, Reg);
4224
4225 // Registers without classes are unaddressable, SGPR-like registers.
4226 return RC && isAGPRClass(RC);
4227}
4228
4230 MachineFunction &MF) const {
4231 unsigned MinOcc = ST.getOccupancyWithWorkGroupSizes(MF).first;
4232 switch (RC->getID()) {
4233 default:
4234 return AMDGPUGenRegisterInfo::getRegPressureLimit(RC, MF);
4235 case AMDGPU::VGPR_32RegClassID:
4236 return std::min(
4237 ST.getMaxNumVGPRs(
4238 MinOcc,
4240 ST.getMaxNumVGPRs(MF));
4241 case AMDGPU::SGPR_32RegClassID:
4242 case AMDGPU::SGPR_LO16RegClassID:
4243 return std::min(ST.getMaxNumSGPRs(MinOcc, true), ST.getMaxNumSGPRs(MF));
4244 }
4245}
4246
4248 unsigned Idx) const {
4249 switch (static_cast<AMDGPU::RegisterPressureSets>(Idx)) {
4250 case AMDGPU::RegisterPressureSets::VGPR_32:
4251 case AMDGPU::RegisterPressureSets::AGPR_32:
4252 return getRegPressureLimit(&AMDGPU::VGPR_32RegClass,
4253 const_cast<MachineFunction &>(MF));
4254 case AMDGPU::RegisterPressureSets::SReg_32:
4255 return getRegPressureLimit(&AMDGPU::SGPR_32RegClass,
4256 const_cast<MachineFunction &>(MF));
4257 }
4258
4259 llvm_unreachable("Unexpected register pressure set!");
4260}
4261
4262const int *SIRegisterInfo::getRegUnitPressureSets(MCRegUnit RegUnit) const {
4263 static const int Empty[] = { -1 };
4264
4265 if (RegPressureIgnoredUnits[static_cast<unsigned>(RegUnit)])
4266 return Empty;
4267
4268 return AMDGPUGenRegisterInfo::getRegUnitPressureSets(RegUnit);
4269}
4270
4272 ArrayRef<MCPhysReg> Order,
4274 const MachineFunction &MF,
4275 const VirtRegMap *VRM,
4276 const LiveRegMatrix *Matrix) const {
4277
4278 const MachineRegisterInfo &MRI = MF.getRegInfo();
4279 const SIRegisterInfo *TRI = ST.getRegisterInfo();
4280
4281 std::pair<unsigned, Register> Hint = MRI.getRegAllocationHint(VirtReg);
4282
4283 switch (Hint.first) {
4284 case AMDGPURI::Size32: {
4285 Register Paired = Hint.second;
4286 assert(Paired);
4287 Register PairedPhys;
4288 if (Paired.isPhysical()) {
4289 PairedPhys =
4290 getMatchingSuperReg(Paired, AMDGPU::lo16, &AMDGPU::VGPR_32RegClass);
4291 } else if (VRM && VRM->hasPhys(Paired)) {
4292 PairedPhys = getMatchingSuperReg(VRM->getPhys(Paired), AMDGPU::lo16,
4293 &AMDGPU::VGPR_32RegClass);
4294 }
4295
4296 // Prefer the paired physreg.
4297 if (PairedPhys)
4298 // isLo(Paired) is implicitly true here from the API of
4299 // getMatchingSuperReg.
4300 Hints.insert(PairedPhys);
4301 return false;
4302 }
4303 case AMDGPURI::Size16: {
4304 Register Paired = Hint.second;
4305 assert(Paired);
4306 Register PairedPhys;
4307 if (Paired.isPhysical()) {
4308 PairedPhys = TRI->getSubReg(Paired, AMDGPU::lo16);
4309 } else if (VRM && VRM->hasPhys(Paired)) {
4310 PairedPhys = TRI->getSubReg(VRM->getPhys(Paired), AMDGPU::lo16);
4311 }
4312
4313 // First prefer the paired physreg.
4314 if (PairedPhys)
4315 Hints.insert(PairedPhys);
4316 else {
4317 // Add all the lo16 physregs.
4318 // When the Paired operand has not yet been assigned a physreg it is
4319 // better to try putting VirtReg in a lo16 register, because possibly
4320 // later Paired can be assigned to the overlapping register and the COPY
4321 // can be eliminated.
4322 for (MCPhysReg PhysReg : Order) {
4323 if (PhysReg == PairedPhys || AMDGPU::isHi16Reg(PhysReg, *this))
4324 continue;
4325 if (AMDGPU::VGPR_16RegClass.contains(PhysReg) &&
4326 !MRI.isReserved(PhysReg))
4327 Hints.insert(PhysReg);
4328 }
4329 }
4330 return false;
4331 }
4332 default:
4333 return TargetRegisterInfo::getRegAllocationHints(VirtReg, Order, Hints, MF,
4334 VRM);
4335 }
4336}
4337
4339 const MachineFunction &MF, unsigned NumAllocatedVGPRs,
4340 unsigned &MaxVGPRsForCurrentOccupancy) const {
4341
4343 unsigned DynamicVGPRBlockSize = MFI->getDynamicVGPRBlockSize();
4344 unsigned RecordedMaxOccupancy = MFI->getOccupancy();
4345 unsigned CurrentOccupancy =
4346 ST.getOccupancyWithNumVGPRs(NumAllocatedVGPRs, DynamicVGPRBlockSize);
4347 MaxVGPRsForCurrentOccupancy =
4348 ST.getMaxNumVGPRs(CurrentOccupancy, DynamicVGPRBlockSize);
4349
4350 LLVM_DEBUG(dbgs() << "anti-hints: VGPRs allocated = " << NumAllocatedVGPRs
4351 << ", RecordedMaxOccupancy = " << RecordedMaxOccupancy
4352 << ", current occupancy = " << CurrentOccupancy << '\n');
4353
4354 // If we are already at lowest occupancy, then there is no need to protect
4355 // against occupancy regression.
4356 if (CurrentOccupancy == 1)
4357 return true;
4358
4359 // Do not apply anti-hints if we are reaching close to the VGPR budget. For
4360 // recorded max occupancy, the 80% cutoff is a conservative: anti-hints are
4361 // disabled early enough that later registers still have headroom to stay at
4362 // recorded max occupancy. For current occupancy, the 95% cutoff margin is
4363 // used to not apply anti-hints close to the limit of the current occupancy
4364 // budget.
4365 unsigned MaxVGPRsCutOffForRecordedMaxOccupancy =
4366 (ST.getMaxNumVGPRs(RecordedMaxOccupancy, DynamicVGPRBlockSize) * 80) /
4367 100;
4368 unsigned MaxVGPRsCutOffForCurrentOccupancy =
4369 (MaxVGPRsForCurrentOccupancy * 95) / 100;
4370
4371 if (NumAllocatedVGPRs >= MaxVGPRsCutOffForRecordedMaxOccupancy) {
4372 LLVM_DEBUG(dbgs() << "anti-hints: not applied, at or above the "
4373 << MaxVGPRsCutOffForRecordedMaxOccupancy
4374 << " VGPR cutoff for RecordedMaxOccupancy\n");
4375 return false;
4376 }
4377
4378 if (NumAllocatedVGPRs >= MaxVGPRsCutOffForCurrentOccupancy) {
4379 LLVM_DEBUG(dbgs() << "anti-hints: not applied, at or above the "
4380 << MaxVGPRsCutOffForCurrentOccupancy
4381 << " VGPR cutoff for current occupancy\n");
4382 return false;
4383 }
4384
4385 return true;
4386}
4387
4388// Returns true if Reg fits within the current occupancy VGPR budget.
4389bool SIRegisterInfo::isRegWithinOccupancyBudget(
4390 MCPhysReg Reg, unsigned NumVGPRs, unsigned NumAGPRs,
4391 unsigned MaxVGPRsForCurrentOccupancy) const {
4392 const TargetRegisterClass *RC = getPhysRegBaseClass(Reg);
4393
4394 // No VGPR or AGPR usage.
4395 if (!RC || !hasVectorRegisters(RC))
4396 return true;
4397
4398 unsigned RegEndIndex =
4399 getHWRegIndex(Reg) + divideCeil(getRegSizeInBits(*RC), 32);
4400
4401 unsigned MaxVGPR = NumVGPRs;
4402 unsigned MaxAGPR = NumAGPRs;
4403
4404 if (isAGPRClass(RC))
4405 MaxAGPR = std::max(MaxAGPR, RegEndIndex);
4406 else
4407 MaxVGPR = std::max(MaxVGPR, RegEndIndex);
4408
4409 return static_cast<unsigned>(
4410 AMDGPU::getTotalNumVGPRs(ST.hasGFX90AInsts(), MaxAGPR, MaxVGPR)) <=
4411 MaxVGPRsForCurrentOccupancy;
4412}
4413
4415 Register VirtReg, MutableArrayRef<MCPhysReg> CustomOrder,
4416 const BitVector &AntiHintedRegUnits, const MachineFunction &MF,
4417 const LiveRegMatrix *Matrix, const RegisterClassInfo *RegClassInfo) const {
4418
4419 if (none_of(CustomOrder, [&](MCPhysReg Reg) {
4420 return isAntiHintedReg(Reg, AntiHintedRegUnits);
4421 }))
4422 return;
4423
4424 const MachineRegisterInfo &MRI = MF.getRegInfo();
4425 assert(hasVectorRegisters(MRI.getRegClass(VirtReg)) &&
4426 "SGPR anti-hints are not handled");
4427 unsigned NumVGPRs = 0;
4428 unsigned NumAGPRs = 0;
4429
4430 assert(Matrix && "LiveRegMatrix required to compute occupancy");
4431 assert(RegClassInfo && "RegClassInfo required to compute occupancy");
4432 for (MCPhysReg Reg : RegClassInfo->getOrder(&AMDGPU::VGPR_32RegClass)) {
4433 if (Matrix->isPhysRegUsed(Reg) ||
4434 MRI.isPhysRegUsed(Reg, /*SkipRegMaskTest=*/true))
4435 NumVGPRs = std::max(NumVGPRs, getHWRegIndex(Reg) + 1);
4436 }
4437
4438 for (MCPhysReg Reg : RegClassInfo->getOrder(&AMDGPU::AGPR_32RegClass)) {
4439 if (Matrix->isPhysRegUsed(Reg) ||
4440 MRI.isPhysRegUsed(Reg, /*SkipRegMaskTest=*/true))
4441 NumAGPRs = std::max(NumAGPRs, getHWRegIndex(Reg) + 1);
4442 }
4443
4444 unsigned NumAllocatedVGPRs =
4445 AMDGPU::getTotalNumVGPRs(ST.hasGFX90AInsts(), NumAGPRs, NumVGPRs);
4446
4447 // Early exit if we should not apply anti-hints.
4448 unsigned MaxVGPRsForCurrentOccupancy = 0;
4449 if (!shouldApplyAntiHints(MF, NumAllocatedVGPRs, MaxVGPRsForCurrentOccupancy))
4450 return;
4451
4452 // Reorder all in-budget first so the anti-hinted partition covers
4453 // both VGPRs and AGPRs.
4454 auto *BeyondBudgetStart = std::stable_partition(
4455 CustomOrder.begin(), CustomOrder.end(), [&](MCPhysReg Reg) {
4456 return isRegWithinOccupancyBudget(Reg, NumVGPRs, NumAGPRs,
4457 MaxVGPRsForCurrentOccupancy);
4458 });
4459
4460 [[maybe_unused]] auto *PartitionPoint = std::stable_partition(
4461 CustomOrder.begin(), BeyondBudgetStart,
4462 [&](MCPhysReg Reg) { return !isAntiHintedReg(Reg, AntiHintedRegUnits); });
4463
4464 LLVM_DEBUG({
4465 size_t NonAntiHintedCount =
4466 std::distance(CustomOrder.begin(), PartitionPoint);
4467 size_t AntiHintedCount = std::distance(PartitionPoint, BeyondBudgetStart);
4468 size_t BeyondBudgetCount =
4469 std::distance(BeyondBudgetStart, CustomOrder.end());
4470 dbgs() << "Added " << NonAntiHintedCount
4471 << " non-anti-hinted registers first\n"
4472 << "Added " << AntiHintedCount
4473 << " anti-hinted registers at the end\n"
4474 << "Beyond current occupancy budget, left: " << BeyondBudgetCount
4475 << '\n';
4476 });
4477}
4478
4480 // Not a callee saved register.
4481 return AMDGPU::SGPR30_SGPR31;
4482}
4483
4484const TargetRegisterClass *
4486 const RegisterBank &RB) const {
4487 switch (RB.getID()) {
4488 case AMDGPU::VGPRRegBankID:
4490 std::max(ST.useRealTrue16Insts() ? 16u : 32u, Size));
4491 case AMDGPU::VCCRegBankID:
4492 assert(Size == 1);
4493 return getWaveMaskRegClass();
4494 case AMDGPU::SGPRRegBankID:
4495 return getSGPRClassForBitWidth(std::max(32u, Size));
4496 case AMDGPU::AGPRRegBankID:
4497 return getAGPRClassForBitWidth(std::max(32u, Size));
4498 default:
4499 llvm_unreachable("unknown register bank");
4500 }
4501}
4502
4504 Register Reg, const MachineRegisterInfo &MRI) const {
4505 const RegClassOrRegBank &RCOrRB = MRI.getRegClassOrRegBank(Reg);
4506 if (const RegisterBank *RB = dyn_cast<const RegisterBank *>(RCOrRB))
4507 return getRegClassForTypeOnBank(MRI.getType(Reg), *RB);
4508
4509 if (const auto *RC = dyn_cast<const TargetRegisterClass *>(RCOrRB))
4510 return getAllocatableClass(RC);
4511
4512 return nullptr;
4513}
4514
4516 return isWave32 ? AMDGPU::VCC_LO : AMDGPU::VCC;
4517}
4518
4520 return isWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
4521}
4522
4524 // VGPR tuples have an alignment requirement on gfx90a variants.
4525 return ST.needsAlignedVGPRs() ? &AMDGPU::VReg_64_Align2RegClass
4526 : &AMDGPU::VReg_64RegClass;
4527}
4528
4529// Find reaching register definition
4533 LiveIntervals *LIS) const {
4534 auto &MDT = LIS->getDomTree();
4535 SlotIndex UseIdx = LIS->getInstructionIndex(Use);
4536 SlotIndex DefIdx;
4537
4538 if (Reg.isVirtual()) {
4539 if (!LIS->hasInterval(Reg))
4540 return nullptr;
4541 LiveInterval &LI = LIS->getInterval(Reg);
4542 LaneBitmask SubLanes = SubReg ? getSubRegIndexLaneMask(SubReg)
4543 : MRI.getMaxLaneMaskForVReg(Reg);
4544 VNInfo *V = nullptr;
4545 if (LI.hasSubRanges()) {
4546 for (auto &S : LI.subranges()) {
4547 if ((S.LaneMask & SubLanes) == SubLanes) {
4548 V = S.getVNInfoAt(UseIdx);
4549 break;
4550 }
4551 }
4552 } else {
4553 V = LI.getVNInfoAt(UseIdx);
4554 }
4555 if (!V)
4556 return nullptr;
4557 DefIdx = V->def;
4558 } else {
4559 // Find last def.
4560 for (MCRegUnit Unit : regunits(Reg.asMCReg())) {
4561 LiveRange &LR = LIS->getRegUnit(Unit);
4562 if (VNInfo *V = LR.getVNInfoAt(UseIdx)) {
4563 if (!DefIdx.isValid() ||
4564 MDT.dominates(LIS->getInstructionFromIndex(DefIdx),
4565 LIS->getInstructionFromIndex(V->def)))
4566 DefIdx = V->def;
4567 } else {
4568 return nullptr;
4569 }
4570 }
4571 }
4572
4573 MachineInstr *Def = LIS->getInstructionFromIndex(DefIdx);
4574
4575 if (!Def || !MDT.dominates(Def, &Use))
4576 return nullptr;
4577
4578 assert(Def->modifiesRegister(Reg, this));
4579
4580 return Def;
4581}
4582
4584 assert(getRegSizeInBits(*getPhysRegBaseClass(Reg)) <= 32);
4585
4586 for (const TargetRegisterClass *RC :
4587 {&AMDGPU::VGPR_32RegClass, &AMDGPU::SReg_32RegClass,
4588 &AMDGPU::AGPR_32RegClass}) {
4589 if (MCPhysReg Super = getMatchingSuperReg(Reg, AMDGPU::lo16, RC))
4590 return Super;
4591 }
4592 if (MCPhysReg Super = getMatchingSuperReg(Reg, AMDGPU::hi16,
4593 &AMDGPU::VGPR_32RegClass)) {
4594 return Super;
4595 }
4596
4597 return AMDGPU::NoRegister;
4598}
4599
4601 if (!ST.needsAlignedVGPRs())
4602 return true;
4603
4604 if (isVGPRClass(&RC))
4605 return RC.hasSuperClassEq(getVGPRClassForBitWidth(getRegSizeInBits(RC)));
4606 if (isAGPRClass(&RC))
4607 return RC.hasSuperClassEq(getAGPRClassForBitWidth(getRegSizeInBits(RC)));
4608 if (isVectorSuperClass(&RC))
4609 return RC.hasSuperClassEq(
4610 getVectorSuperClassForBitWidth(getRegSizeInBits(RC)));
4611
4612 assert(&RC != &AMDGPU::VS_64RegClass);
4613
4614 return true;
4615}
4616
4619 return ArrayRef(AMDGPU::SGPR_128RegClass.begin(), ST.getMaxNumSGPRs(MF) / 4);
4620}
4621
4624 return ArrayRef(AMDGPU::SGPR_64RegClass.begin(), ST.getMaxNumSGPRs(MF) / 2);
4625}
4626
4629 return ArrayRef(AMDGPU::SGPR_32RegClass.begin(), ST.getMaxNumSGPRs(MF));
4630}
4631
4632unsigned
4634 unsigned SubReg) const {
4635 switch (RC->TSFlags & SIRCFlags::RegKindMask) {
4636 case SIRCFlags::HasSGPR:
4637 return std::min(128u, getSubRegIdxSize(SubReg));
4638 case SIRCFlags::HasAGPR:
4639 case SIRCFlags::HasVGPR:
4641 return std::min(32u, getSubRegIdxSize(SubReg));
4642 default:
4643 break;
4644 }
4645 return 0;
4646}
4647
4649 const TargetRegisterClass &RC,
4650 bool IncludeCalls) const {
4651 unsigned NumArchVGPRs = ST.getAddressableNumArchVGPRs();
4653 (RC.getID() == AMDGPU::VGPR_32RegClassID)
4654 ? RC.getRegisters().take_front(NumArchVGPRs)
4655 : RC.getRegisters();
4656 for (MCPhysReg Reg : reverse(Registers)) {
4657 if (Reg != AMDGPU::VCC_LO && Reg != AMDGPU::VCC_HI &&
4658 MRI.isPhysRegUsed(Reg, /*SkipRegMaskTest=*/!IncludeCalls))
4659 return getHWRegIndex(Reg) + 1;
4660 }
4661 return 0;
4662}
4663
4666 const MachineFunction &MF) const {
4668 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
4669 if (FuncInfo->checkFlag(Reg, AMDGPU::VirtRegFlag::WWM_REG))
4670 RegFlags.push_back("WWM_REG");
4671 return RegFlags;
4672}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
AMDGPU Reserve WWM Registers
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static const Function * getParent(const Value *V)
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
Live Register Matrix
A set of register units.
#define I(x, y, z)
Definition MD5.cpp:57
static DebugLoc getDebugLoc(MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
Return the first DebugLoc that has line number information, given a range of instructions.
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
if(PassOpts->AAPipeline)
This file declares the machine register scavenger class.
static MachineInstrBuilder spillVGPRtoAGPR(const GCNSubtarget &ST, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, int Index, unsigned Lane, unsigned ValueReg, bool IsKill, bool NeedsCFI)
static int getOffenMUBUFStore(unsigned Opc)
static bool wrapsAround32(int64_t LHS, int64_t RHS)
static const TargetRegisterClass * getAnyAGPRClassForBitWidth(unsigned BitWidth)
static bool isSCCLiveInto(const RegScavenger &RS, const MachineInstr &MI)
static int getOffsetMUBUFLoad(unsigned Opc)
static const std::array< unsigned, 17 > SubRegFromChannelTableWidthMap
static unsigned getNumSubRegsForSpillOp(const MachineInstr &MI, const SIInstrInfo *TII)
static cl::opt< bool > EnableSpillCFISavedRegs("amdgpu-spill-cfi-saved-regs", cl::desc("Enable spilling the registers required for CFI emission"), cl::ReallyHidden, cl::init(false))
static void emitUnsupportedError(const Function &Fn, const MachineInstr &MI, const Twine &ErrMsg)
static const TargetRegisterClass * getAlignedAGPRClassForBitWidth(unsigned BitWidth)
static bool buildMUBUFOffsetLoadStore(const GCNSubtarget &ST, MachineFrameInfo &MFI, MachineBasicBlock::iterator MI, int Index, int64_t Offset)
static unsigned getFlatScratchSpillOpcode(const SIInstrInfo *TII, unsigned LoadStoreOp, unsigned EltSize)
static const TargetRegisterClass * getAlignedVGPRClassForBitWidth(unsigned BitWidth)
static int getOffsetMUBUFStore(unsigned Opc)
static const TargetRegisterClass * getAnyVGPRClassForBitWidth(unsigned BitWidth)
static cl::opt< unsigned > StressSGPRLimit("amdgpu-stress-sgpr", cl::Hidden, cl::init(0), cl::desc("Limit SGPRs to N registers by reserving the rest"))
static cl::opt< bool > EnableSpillSGPRToVGPR("amdgpu-spill-sgpr-to-vgpr", cl::desc("Enable spilling SGPRs to VGPRs"), cl::ReallyHidden, cl::init(true))
static const TargetRegisterClass * getAlignedVectorSuperClassForBitWidth(unsigned BitWidth)
static const TargetRegisterClass * getAnyVectorSuperClassForBitWidth(unsigned BitWidth)
static cl::opt< unsigned > StressAGPRLimit("amdgpu-stress-agpr", cl::Hidden, cl::init(0), cl::desc("Limit AGPRs to N registers by reserving the rest"))
static cl::opt< unsigned > StressVGPRLimit("amdgpu-stress-vgpr", cl::Hidden, cl::init(0), cl::desc("Limit VGPRs to N registers by reserving the rest"))
static bool foldingOffsetChangesCarry(const MachineOperand &OtherOp, int64_t Offset, Register FrameReg)
static bool isFIPlusImmOrVGPR(const SIRegisterInfo &TRI, const MachineInstr &MI)
static int getOffenMUBUFLoad(unsigned Opc)
Interface definition for SIRegisterInfo.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
#define LLVM_DEBUG(...)
Definition Debug.h:119
LocallyHashedType DenseMapInfo< LocallyHashedType >::Empty
Value * RHS
Value * LHS
static const char * getRegisterName(MCRegister Reg)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
bool test(unsigned Idx) const
Returns true if bit Idx is set.
Definition BitVector.h:482
bool empty() const
Returns whether there are no bits in this bitvector.
Definition BitVector.h:175
A debug info location.
Definition DebugLoc.h:126
Diagnostic information for unsupported feature in backend.
Register getReg() const
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
LiveInterval - This class represents the liveness of a register, or stack slot.
bool hasSubRanges() const
Returns true if subregister liveness information is available.
iterator_range< subrange_iterator > subranges()
void removeAllRegUnitsForPhysReg(MCRegister Reg)
Remove associated live ranges for the register units associated with Reg.
bool hasInterval(Register Reg) const
MachineInstr * getInstructionFromIndex(SlotIndex index) const
Returns the instruction associated with the given index.
MachineDominatorTree & getDomTree()
SlotIndex getInstructionIndex(const MachineInstr &Instr) const
Returns the base index of the given instruction.
LiveInterval & getInterval(Register Reg)
LiveRange & getRegUnit(MCRegUnit Unit)
Return the live range for register unit Unit.
This class represents the liveness of a register, stack slot, etc.
VNInfo * getVNInfoAt(SlotIndex Idx) const
getVNInfoAt - Return the VNInfo that is live at Idx, or NULL.
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
Describe properties that are true of each instruction in the target description file.
MCRegAliasIterator enumerates all registers aliasing Reg.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
unsigned getID() const
getID() - Return the register class ID number.
ArrayRef< MCPhysReg > getRegisters() const
const uint8_t TSFlags
Configurable target specific flags.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
static MCRegister from(unsigned Val)
Check the provided unsigned value is a valid MCRegister.
Definition MCRegister.h:77
Generic base class for all target subtargets.
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool hasCalls() const
Return true if the current function has any function calls.
Align getObjectAlign(int ObjectIdx) const
Return the alignment of the specified stack object.
bool hasStackObjects() const
Return true if there are any stack objects in this function.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
void setAsmPrinterFlag(AsmPrinterFlagTy Flag)
Set a flag for the AsmPrinter.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
const MachinePointerInfo & getPointerInfo() const
Flags getFlags() const
Return the raw flags of the source value,.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
void setImm(int64_t immVal)
int64_t getImm() const
LLVM_ABI void setIsRenamable(bool Val=true)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
const RegClassOrRegBank & getRegClassOrRegBank(Register Reg) const
Return the register bank or register class of Reg.
bool isReserved(MCRegister PhysReg) const
isReserved - Returns true when PhysReg is a reserved register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool isAllocatable(MCRegister PhysReg) const
isAllocatable - Returns true when PhysReg belongs to an allocatable register class and it hasn't been...
std::pair< unsigned, Register > getRegAllocationHint(Register VReg) const
getRegAllocationHint - Return the register allocation hint for the specified virtual register.
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI LaneBitmask getMaxLaneMaskForVReg(Register Reg) const
Returns a mask covering all bits that can appear in lane masks of subregisters of the virtual registe...
LLVM_ABI bool isPhysRegUsed(MCRegister PhysReg, bool SkipRegMaskTest=false) const
Return true if the specified register is modified or read in this function.
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:294
iterator end() const
Definition ArrayRef.h:339
iterator begin() const
Definition ArrayRef.h:338
Holds all the information related to register banks.
virtual bool isDivergentRegBank(const RegisterBank *RB) const
Returns true if the register bank is considered divergent.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
ArrayRef< MCPhysReg > getOrder(const TargetRegisterClass *RC) const
getOrder - Returns the preferred allocation order for RC.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isValid() const
Definition Register.h:112
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
MachineInstr * buildCFIForSGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister SGPR, int64_t Offset) const
Create a CFI index describing a spill of a SGPR to VMEM and build a MachineInstr around it.
MachineInstr * buildCFIForVRegToVRegSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister Reg, const MCRegister RegCopy) const
Create a CFI index describing a spill of the VGPR/AGPR Reg to another VGPR/AGPR RegCopy and build a M...
MachineInstr * buildCFIForVGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister VGPR, int64_t Offset) const
Create a CFI index describing a spill of a VGPR to VMEM and build a MachineInstr around it.
MachineInstr * buildCFIForSGPRToVGPRSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister SGPR, const MCRegister VGPR, const int Lane) const
Create a CFI index describing a spill of an SGPR to a single lane of a VGPR and build a MachineInstr ...
static bool isFLATScratch(const MachineInstr &MI)
static bool isMUBUF(const MachineInstr &MI)
static bool isVOP3(const MCInstrDesc &Desc)
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
ArrayRef< MCPhysReg > getAGPRSpillVGPRs() const
MCPhysReg getVGPRToAGPRSpill(int FrameIndex, unsigned Lane) const
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
ArrayRef< MCPhysReg > getVGPRSpillAGPRs() const
ArrayRef< SIRegisterInfo::SpilledReg > getSGPRSpillToVirtualVGPRLanes(int FrameIndex) const
uint32_t getMaskForVGPRBlockOps(Register RegisterBlock) const
ArrayRef< SIRegisterInfo::SpilledReg > getSGPRSpillToPhysicalVGPRLanes(int FrameIndex) const
bool checkFlag(Register Reg, uint8_t Flag) const
const ReservedRegSet & getWWMReservedRegs() const
Register materializeFrameBaseRegister(MachineBasicBlock *MBB, int FrameIdx, int64_t Offset) const override
int64_t getScratchInstrOffset(const MachineInstr *MI) const
bool isFrameOffsetLegal(const MachineInstr *MI, Register BaseReg, int64_t Offset) const override
const TargetRegisterClass * getCompatibleSubRegClass(const TargetRegisterClass *SuperRC, const TargetRegisterClass *SubRC, unsigned SubIdx) const
Returns a register class which is compatible with SuperRC, such that a subregister exists with class ...
ArrayRef< MCPhysReg > getAllSGPR64(const MachineFunction &MF) const
Return all SGPR64 which satisfy the waves per execution unit requirement of the subtarget.
MCRegister findUnusedRegister(const MachineRegisterInfo &MRI, const TargetRegisterClass *RC, const MachineFunction &MF, bool ReserveHighestVGPR=false) const
Returns a lowest register that is not used at any point in the function.
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
MCPhysReg get32BitRegister(MCPhysReg Reg) const
const uint32_t * getCallPreservedMask(const MachineFunction &MF, CallingConv::ID) const override
void buildSpillLoadStore(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, unsigned LoadStoreOp, int Index, Register ValueReg, bool ValueIsKill, MCRegister ScratchOffsetReg, int64_t InstrOffset, MachineMemOperand *MMO, RegScavenger *RS, LiveRegUnits *LiveUnits=nullptr, bool NeedsCFI=false) const
bool requiresFrameIndexReplacementScavenging(const MachineFunction &MF) const override
bool shouldRealignStack(const MachineFunction &MF) const override
bool restoreSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool OnlyToVGPR=false, bool SpillToPhysVGPRLane=false) const
bool isProperlyAlignedRC(const TargetRegisterClass &RC) const
static bool hasVectorRegisters(const TargetRegisterClass *RC)
const TargetRegisterClass * getEquivalentVGPRClass(const TargetRegisterClass *SRC) const
Register getFrameRegister(const MachineFunction &MF) const override
LLVM_READONLY const TargetRegisterClass * getVectorSuperClassForBitWidth(unsigned BitWidth) const
bool spillEmergencySGPR(MachineBasicBlock::iterator MI, MachineBasicBlock &RestoreMBB, Register SGPR, RegScavenger *RS) const
SIRegisterInfo(const GCNSubtarget &ST)
const uint32_t * getAllVGPRRegMask() const
MCRegister getReturnAddressReg(const MachineFunction &MF) const
const MCPhysReg * getCalleeSavedRegs(const MachineFunction *MF) const override
bool hasBasePointer(const MachineFunction &MF) const
const TargetRegisterClass * getCrossCopyRegClass(const TargetRegisterClass *RC) const override
Returns a legal register class to copy a register in the specified class to or from.
ArrayRef< int16_t > getRegSplitParts(const TargetRegisterClass *RC, unsigned EltSize) const
ArrayRef< MCPhysReg > getAllSGPR32(const MachineFunction &MF) const
Return all SGPR32 which satisfy the waves per execution unit requirement of the subtarget.
const TargetRegisterClass * getLargestLegalSuperClass(const TargetRegisterClass *RC, const MachineFunction &MF) const override
MCRegister reservedPrivateSegmentBufferReg(const MachineFunction &MF) const
Return the end register initially reserved for the scratch buffer in case spilling is needed.
bool eliminateSGPRToVGPRSpillFrameIndex(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool SpillToPhysVGPRLane=false) const
Special case of eliminateFrameIndex.
bool isVGPR(const MachineRegisterInfo &MRI, Register Reg) const
bool getRegAllocationHints(Register VirtReg, ArrayRef< MCPhysReg > Order, SmallSetVector< MCPhysReg, 16 > &Hints, const MachineFunction &MF, const VirtRegMap *VRM, const LiveRegMatrix *Matrix) const override
bool isAsmClobberable(const MachineFunction &MF, MCRegister PhysReg) const override
LLVM_READONLY const TargetRegisterClass * getAGPRClassForBitWidth(unsigned BitWidth) const
static bool isChainScratchRegister(Register VGPR)
bool requiresRegisterScavenging(const MachineFunction &Fn) const override
bool opCanUseInlineConstant(unsigned OpType) const
const TargetRegisterClass * getRegClassForSizeOnBank(unsigned Size, const RegisterBank &Bank) const
bool isUniformReg(const MachineRegisterInfo &MRI, const RegisterBankInfo &RBI, Register Reg) const override
const uint32_t * getNoPreservedMask() const override
bool shouldApplyAntiHints(const MachineFunction &MF, unsigned NumAllocatedVGPRs, unsigned &MaxVGPRsForCurrentOccupancy) const
StringRef getRegAsmName(MCRegister Reg) const override
MCRegister getAlignedHighSGPRForRC(const MachineFunction &MF, const unsigned Align, const TargetRegisterClass *RC) const
Return the largest available SGPR aligned to Align for the register class RC.
void buildCFIForBlockCSRStore(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register BlockReg, int64_t Offset) const
const TargetRegisterClass * getRegClassForReg(const MachineRegisterInfo &MRI, Register Reg) const
unsigned getHWRegIndex(MCRegister Reg) const
const MCPhysReg * getCalleeSavedRegsViaCopy(const MachineFunction *MF) const
const uint32_t * getAllVectorRegMask() const
const TargetRegisterClass * getEquivalentAGPRClass(const TargetRegisterClass *SRC) const
void filterAndSortForAntiHintedRegs(Register VirtReg, MutableArrayRef< MCPhysReg > CustomOrder, const BitVector &AntiHintedRegUnits, const MachineFunction &MF, const LiveRegMatrix *Matrix=nullptr, const RegisterClassInfo *RegClassInfo=nullptr) const override
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
const TargetRegisterClass * getRegClassForTypeOnBank(LLT Ty, const RegisterBank &Bank) const
bool opCanUseLiteralConstant(unsigned OpType) const
Register getBaseRegister() const
LLVM_READONLY const TargetRegisterClass * getAlignedLo256VGPRClassForBitWidth(unsigned BitWidth) const
LLVM_READONLY const TargetRegisterClass * getVGPRClassForBitWidth(unsigned BitWidth) const
const TargetRegisterClass * getEquivalentAVClass(const TargetRegisterClass *SRC) const
bool requiresFrameIndexScavenging(const MachineFunction &MF) const override
static bool isVGPRClass(const TargetRegisterClass *RC)
MachineInstr * findReachingDef(Register Reg, unsigned SubReg, MachineInstr &Use, MachineRegisterInfo &MRI, LiveIntervals *LIS) const
bool isSGPRReg(const MachineRegisterInfo &MRI, Register Reg) const
const TargetRegisterClass * getEquivalentSGPRClass(const TargetRegisterClass *VRC) const
SmallVector< StringLiteral > getVRegFlagsOfReg(Register Reg, const MachineFunction &MF) const override
LLVM_READONLY const TargetRegisterClass * getDefaultVectorSuperClassForBitWidth(unsigned BitWidth) const
unsigned getRegPressureLimit(const TargetRegisterClass *RC, MachineFunction &MF) const override
ArrayRef< MCPhysReg > getAllSGPR128(const MachineFunction &MF) const
Return all SGPR128 which satisfy the waves per execution unit requirement of the subtarget.
unsigned getRegPressureSetLimit(const MachineFunction &MF, unsigned Idx) const override
BitVector getReservedRegs(const MachineFunction &MF) const override
bool needsFrameBaseReg(MachineInstr *MI, int64_t Offset) const override
const TargetRegisterClass * getRegClassForOperandReg(const MachineRegisterInfo &MRI, const MachineOperand &MO) const
void addImplicitUsesForBlockCSRLoad(MachineInstrBuilder &MIB, Register BlockReg) const
unsigned getNumUsedPhysRegs(const MachineRegisterInfo &MRI, const TargetRegisterClass &RC, bool IncludeCalls=true) const
const uint32_t * getAllAGPRRegMask() const
const int * getRegUnitPressureSets(MCRegUnit RegUnit) const override
bool isAGPR(const MachineRegisterInfo &MRI, Register Reg) const
bool eliminateFrameIndex(MachineBasicBlock::iterator MI, int SPAdj, unsigned FIOperandNum, RegScavenger *RS) const override
bool spillSGPR(MachineBasicBlock::iterator MI, int FI, RegScavenger *RS, SlotIndexes *Indexes=nullptr, LiveIntervals *LIS=nullptr, bool OnlyToVGPR=false, bool SpillToPhysVGPRLane=false, bool NeedsCFI=false) const
If OnlyToVGPR is true, this will only succeed if this manages to find a free VGPR lane to spill.
MCRegister getExec() const
MCRegister getVCC() const
int64_t getFrameIndexInstrOffset(const MachineInstr *MI, int Idx) const override
bool isVectorSuperClass(const TargetRegisterClass *RC) const
const TargetRegisterClass * getWaveMaskRegClass() const
unsigned getSubRegAlignmentNumBits(const TargetRegisterClass *RC, unsigned SubReg) const
void resolveFrameIndex(MachineInstr &MI, Register BaseReg, int64_t Offset) const override
bool requiresVirtualBaseRegisters(const MachineFunction &Fn) const override
const TargetRegisterClass * getVGPR64Class() const
void buildVGPRSpillLoadStore(SGPRSpillBuilder &SB, int Index, int Offset, bool IsLoad, bool IsKill=true) const
bool isCFISavedRegsSpillEnabled() const
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
const TargetRegisterClass * getConstrainedRegClassForReg(Register Reg, const MachineRegisterInfo &MRI) const override
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
SlotIndex - An opaque wrapper around machine indexes.
Definition SlotIndexes.h:66
bool isValid() const
Returns true if this is a valid index.
SlotIndexes pass.
SlotIndex insertMachineInstrInMaps(MachineInstr &MI, bool Late=false)
Insert the given machine instruction into the mapping.
SlotIndex replaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
ReplaceMachineInstrInMaps - Replacing a machine instr with a new one in maps used by register allocat...
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
virtual const TargetRegisterClass * getLargestLegalSuperClass(const TargetRegisterClass *RC, const MachineFunction &) const
Returns the largest super class of RC that is legal to use in the current sub-target and has the same...
virtual bool shouldRealignStack(const MachineFunction &MF) const
True if storage within the function requires the stack pointer to be aligned more than the normal cal...
virtual bool getRegAllocationHints(Register VirtReg, ArrayRef< MCPhysReg > Order, SmallSetVector< MCPhysReg, 16 > &Hints, const MachineFunction &MF, const VirtRegMap *VRM=nullptr, const LiveRegMatrix *Matrix=nullptr) const
Get a list of 'hint' registers that the register allocator should try first when allocating a physica...
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
VNInfo - Value Number Information.
MCRegister getPhys(Register virtReg) const
returns the physical register mapped to the specified virtual register
Definition VirtRegMap.h:91
bool hasPhys(Register virtReg) const
returns true if the specified virtual register is mapped to a physical register
Definition VirtRegMap.h:87
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ PRIVATE_ADDRESS
Address space for private memory.
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
@ OPERAND_REG_IMM_FIRST
Definition SIDefines.h:478
@ OPERAND_REG_INLINE_AC_FIRST
Definition SIDefines.h:484
@ OPERAND_REG_INLINE_AC_LAST
Definition SIDefines.h:485
@ OPERAND_REG_IMM_LAST
Definition SIDefines.h:479
LLVM_READONLY int32_t getFlatScratchInstSVfromSVS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSVfromSS(uint32_t Opcode)
LLVM_READONLY int32_t getFlatScratchInstSTfromSS(uint32_t Opcode)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
@ Renamable
Register that may be renamed.
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
@ HasSGPR
Definition SIDefines.h:29
@ HasVGPR
Definition SIDefines.h:27
@ RegKindMask
Definition SIDefines.h:32
@ HasAGPR
Definition SIDefines.h:28
constexpr RegState getDefRegState(bool B)
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr bool hasRegState(RegState Value, RegState Test)
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
void call_once(once_flag &flag, Function &&F, Args &&... ArgList)
Execute the function specified as a parameter once.
Definition Threading.h:86
constexpr unsigned BitWidth
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Definition SIInstrInfo.h:49
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:206
static const MachineMemOperand::Flags MOThreadPrivate
Mark the MMO of accesses to memory locations that are never written to by other threads.
Definition SIInstrInfo.h:64
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
This class contains a discriminated union of information about pointers in memory operands,...
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
void setMI(MachineBasicBlock *NewMBB, MachineBasicBlock::iterator NewMI)
ArrayRef< int16_t > SplitParts
SIMachineFunctionInfo & MFI
SGPRSpillBuilder(const SIRegisterInfo &TRI, const SIInstrInfo &TII, bool IsWave32, MachineBasicBlock::iterator MI, int Index, RegScavenger *RS)
SGPRSpillBuilder(const SIRegisterInfo &TRI, const SIInstrInfo &TII, bool IsWave32, MachineBasicBlock::iterator MI, Register Reg, bool IsKill, int Index, RegScavenger *RS)
MachineBasicBlock::iterator MI
void readWriteTmpVGPR(unsigned Offset, bool IsLoad)
const SIRegisterInfo & TRI
MachineBasicBlock * MBB
const SIInstrInfo & TII
The llvm::once_flag structure.
Definition Threading.h:67