LLVM 24.0.0git
AMDGPUInstructionSelector.cpp
Go to the documentation of this file.
1//===- AMDGPUInstructionSelector.cpp ----------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the InstructionSelector class for
10/// AMDGPU.
11/// \todo This should be generated by TableGen.
12//===----------------------------------------------------------------------===//
13
15#include "AMDGPU.h"
17#include "AMDGPUInstrInfo.h"
28#include "llvm/IR/IntrinsicsAMDGPU.h"
29#include <optional>
30
31#define DEBUG_TYPE "amdgpu-isel"
32
33using namespace llvm;
34using namespace MIPatternMatch;
35
36#define GET_GLOBALISEL_IMPL
37#define AMDGPUSubtarget GCNSubtarget
38#include "AMDGPUGenGlobalISel.inc"
39#undef GET_GLOBALISEL_IMPL
40#undef AMDGPUSubtarget
41
43 const GCNSubtarget &STI, const AMDGPURegisterBankInfo &RBI)
44 : TII(*STI.getInstrInfo()), TRI(*STI.getRegisterInfo()), RBI(RBI), STI(STI),
46#include "AMDGPUGenGlobalISel.inc"
49#include "AMDGPUGenGlobalISel.inc"
51{
52}
53
54const char *AMDGPUInstructionSelector::getName() { return DEBUG_TYPE; }
55
66
67// Return the wave level SGPR base address if this is a wave address.
69 return Def->getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS
70 ? Def->getOperand(1).getReg()
71 : Register();
72}
73
74bool AMDGPUInstructionSelector::isVCC(Register Reg,
75 const MachineRegisterInfo &MRI) const {
76 // The verifier is oblivious to s1 being a valid value for wavesize registers.
77 if (Reg.isPhysical())
78 return false;
79
80 auto &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
81 const TargetRegisterClass *RC =
83 if (RC) {
84 const LLT Ty = MRI.getType(Reg);
85 if (!Ty.isValid() || Ty.getSizeInBits() != 1)
86 return false;
87 // G_TRUNC s1 result is never vcc.
88 return !mi_match(Reg, MRI, m_GTrunc(m_Reg())) &&
89 RC->hasSuperClassEq(TRI.getBoolRC());
90 }
91
92 const RegisterBank *RB = cast<const RegisterBank *>(RegClassOrBank);
93 return RB->getID() == AMDGPU::VCCRegBankID;
94}
95
96bool AMDGPUInstructionSelector::constrainCopyLikeIntrin(MachineInstr &MI,
97 unsigned NewOpc) const {
98 MI.setDesc(TII.get(NewOpc));
99 MI.removeOperand(1); // Remove intrinsic ID.
100 MI.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
101
102 Register DstReg = MI.getOperand(0).getReg();
103 Register SrcReg = MI.getOperand(1).getReg();
104
105 // TODO: This should be legalized to s32 if needed
106 if (MRI->getType(DstReg) == LLT::scalar(1))
107 return false;
108
109 const TargetRegisterClass *DstRC =
110 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
111 const TargetRegisterClass *SrcRC =
112 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
113 if (!DstRC || DstRC != SrcRC)
114 return false;
115
116 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
117 !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
118 return false;
119 const MCInstrDesc &MCID = MI.getDesc();
120 if (MCID.getOperandConstraint(0, MCOI::EARLY_CLOBBER) != -1) {
121 MI.getOperand(0).setIsEarlyClobber(true);
122 }
123 return true;
124}
125
126bool AMDGPUInstructionSelector::selectCOPY(MachineInstr &I) const {
127 const DebugLoc &DL = I.getDebugLoc();
128 MachineBasicBlock *BB = I.getParent();
129 I.setDesc(TII.get(TargetOpcode::COPY));
130
131 Register DstReg = I.getOperand(0).getReg();
132 Register SrcReg = I.getOperand(1).getReg();
133
134 if (isVCC(DstReg, *MRI)) {
135 if (SrcReg == AMDGPU::SCC) {
136 const TargetRegisterClass *RC =
137 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
138 if (!RC)
139 return true;
140 return RBI.constrainGenericRegister(DstReg, *RC, *MRI);
141 }
142
143 if (!isVCC(SrcReg, *MRI)) {
144 // TODO: Should probably leave the copy and let copyPhysReg expand it.
145 if (!RBI.constrainGenericRegister(DstReg, *TRI.getBoolRC(), *MRI))
146 return false;
147
148 const TargetRegisterClass *SrcRC =
149 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
150
151 std::optional<ValueAndVReg> ConstVal =
152 getIConstantVRegValWithLookThrough(SrcReg, *MRI, true);
153 if (ConstVal) {
154 unsigned MovOpc =
155 STI.isWave64() ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
156 BuildMI(*BB, &I, DL, TII.get(MovOpc), DstReg)
157 .addImm(ConstVal->Value.getBoolValue() ? -1 : 0);
158 } else {
159 Register MaskedReg = MRI->createVirtualRegister(SrcRC);
160
161 // We can't trust the high bits at this point, so clear them.
162
163 // TODO: Skip masking high bits if def is known boolean.
164
165 if (AMDGPU::getRegBitWidth(SrcRC->getID()) == 16) {
166 assert(Subtarget->useRealTrue16Insts());
167 const int64_t NoMods = 0;
168 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_AND_B16_t16_e64), MaskedReg)
169 .addImm(NoMods)
170 .addImm(1)
171 .addImm(NoMods)
172 .addReg(SrcReg)
173 .addImm(NoMods);
174 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CMP_NE_U16_t16_e64), DstReg)
175 .addImm(NoMods)
176 .addImm(0)
177 .addImm(NoMods)
178 .addReg(MaskedReg)
179 .addImm(NoMods);
180 } else {
181 bool IsSGPR = TRI.isSGPRClass(SrcRC);
182 unsigned AndOpc = IsSGPR ? AMDGPU::S_AND_B32 : AMDGPU::V_AND_B32_e32;
183 auto And = BuildMI(*BB, &I, DL, TII.get(AndOpc), MaskedReg)
184 .addImm(1)
185 .addReg(SrcReg);
186 if (IsSGPR)
187 And.setOperandDead(3); // Dead scc
188
189 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CMP_NE_U32_e64), DstReg)
190 .addImm(0)
191 .addReg(MaskedReg);
192 }
193 }
194
195 if (!MRI->getRegClassOrNull(SrcReg))
196 MRI->setRegClass(SrcReg, SrcRC);
197 I.eraseFromParent();
198 return true;
199 }
200
201 const TargetRegisterClass *RC =
202 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
203 if (RC && !RBI.constrainGenericRegister(DstReg, *RC, *MRI))
204 return false;
205
206 return true;
207 }
208
209 for (const MachineOperand &MO : I.operands()) {
210 if (MO.getReg().isPhysical())
211 continue;
212
213 const TargetRegisterClass *RC =
214 TRI.getConstrainedRegClassForReg(MO.getReg(), *MRI);
215 if (!RC)
216 continue;
217 RBI.constrainGenericRegister(MO.getReg(), *RC, *MRI);
218 }
219 return true;
220}
221
222bool AMDGPUInstructionSelector::selectCOPY_SCC_VCC(MachineInstr &I) const {
223 const DebugLoc &DL = I.getDebugLoc();
224 MachineBasicBlock *BB = I.getParent();
225 Register VCCReg = I.getOperand(1).getReg();
226 MachineInstr *Cmp;
227
228 // Set SCC as a side effect with S_CMP or S_OR.
229 if (STI.hasScalarCompareEq64()) {
230 unsigned CmpOpc =
231 STI.isWave64() ? AMDGPU::S_CMP_LG_U64 : AMDGPU::S_CMP_LG_U32;
232 Cmp = BuildMI(*BB, &I, DL, TII.get(CmpOpc)).addReg(VCCReg).addImm(0);
233 } else {
234 Register DeadDst = MRI->createVirtualRegister(&AMDGPU::SReg_64RegClass);
235 Cmp = BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_OR_B64), DeadDst)
236 .addReg(VCCReg)
237 .addReg(VCCReg);
238 }
239
240 constrainSelectedInstRegOperands(*Cmp, TII, TRI, RBI);
241
242 Register DstReg = I.getOperand(0).getReg();
243 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), DstReg).addReg(AMDGPU::SCC);
244
245 I.eraseFromParent();
246 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
247}
248
249bool AMDGPUInstructionSelector::selectCOPY_VCC_SCC(MachineInstr &I) const {
250 const DebugLoc &DL = I.getDebugLoc();
251 MachineBasicBlock *BB = I.getParent();
252
253 Register DstReg = I.getOperand(0).getReg();
254 Register SrcReg = I.getOperand(1).getReg();
255 std::optional<ValueAndVReg> Arg =
256 getIConstantVRegValWithLookThrough(I.getOperand(1).getReg(), *MRI);
257
258 if (Arg) {
259 const int64_t Value = Arg->Value.getZExtValue();
260 if (Value == 0) {
261 unsigned Opcode = STI.isWave64() ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
262 BuildMI(*BB, &I, DL, TII.get(Opcode), DstReg).addImm(0);
263 } else {
264 assert(Value == 1);
265 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), DstReg).addReg(TRI.getExec());
266 }
267 I.eraseFromParent();
268 return RBI.constrainGenericRegister(DstReg, *TRI.getBoolRC(), *MRI);
269 }
270
271 // RegBankLegalize ensures that SrcReg is bool in reg (high bits are 0).
272 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC).addReg(SrcReg);
273
274 unsigned SelectOpcode =
275 STI.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
276 MachineInstr *Select = BuildMI(*BB, &I, DL, TII.get(SelectOpcode), DstReg)
277 .addReg(TRI.getExec())
278 .addImm(0);
279
280 I.eraseFromParent();
282 return true;
283}
284
285bool AMDGPUInstructionSelector::selectReadAnyLane(MachineInstr &I) const {
286 Register DstReg = I.getOperand(0).getReg();
287 Register SrcReg = I.getOperand(1).getReg();
288
289 const DebugLoc &DL = I.getDebugLoc();
290 MachineBasicBlock *BB = I.getParent();
291
292 auto RFL = BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
293 .addReg(SrcReg);
294
295 I.eraseFromParent();
296 constrainSelectedInstRegOperands(*RFL, TII, TRI, RBI);
297 return true;
298}
299
300bool AMDGPUInstructionSelector::selectPHI(MachineInstr &I) const {
301 const Register DefReg = I.getOperand(0).getReg();
302 const LLT DefTy = MRI->getType(DefReg);
303
304 // S1 G_PHIs should not be selected in instruction-select, instead:
305 // - divergent S1 G_PHI should go through lane mask merging algorithm
306 // and be fully inst-selected in AMDGPUGlobalISelDivergenceLowering
307 // - uniform S1 G_PHI should be lowered into S32 G_PHI in AMDGPURegBankSelect
308 if (DefTy == LLT::scalar(1))
309 return false;
310
311 // TODO: Verify this doesn't have insane operands (i.e. VGPR to SGPR copy)
312
313 const RegClassOrRegBank &RegClassOrBank =
314 MRI->getRegClassOrRegBank(DefReg);
315
316 const TargetRegisterClass *DefRC =
318 if (!DefRC) {
319 if (!DefTy.isValid()) {
320 LLVM_DEBUG(dbgs() << "PHI operand has no type, not a gvreg?\n");
321 return false;
322 }
323
324 const RegisterBank &RB = *cast<const RegisterBank *>(RegClassOrBank);
325 DefRC = TRI.getRegClassForTypeOnBank(DefTy, RB);
326 if (!DefRC) {
327 LLVM_DEBUG(dbgs() << "PHI operand has unexpected size/bank\n");
328 return false;
329 }
330 }
331
332 // If inputs have register bank, assign corresponding reg class.
333 // Note: registers don't need to have the same reg bank.
334 for (unsigned i = 1; i != I.getNumOperands(); i += 2) {
335 const Register SrcReg = I.getOperand(i).getReg();
336
337 const RegisterBank *RB = MRI->getRegBankOrNull(SrcReg);
338 if (RB) {
339 const LLT SrcTy = MRI->getType(SrcReg);
340 const TargetRegisterClass *SrcRC =
341 TRI.getRegClassForTypeOnBank(SrcTy, *RB);
342 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
343 return false;
344 }
345 }
346
347 I.setDesc(TII.get(TargetOpcode::PHI));
348 return RBI.constrainGenericRegister(DefReg, *DefRC, *MRI);
349}
350
352AMDGPUInstructionSelector::getSubOperand64(MachineOperand &MO,
353 const TargetRegisterClass &SubRC,
354 unsigned SubIdx) const {
355
356 MachineInstr *MI = MO.getParent();
357 MachineBasicBlock *BB = MO.getParent()->getParent();
358 Register DstReg = MRI->createVirtualRegister(&SubRC);
359
360 if (MO.isReg()) {
361 unsigned ComposedSubIdx = TRI.composeSubRegIndices(MO.getSubReg(), SubIdx);
362 Register Reg = MO.getReg();
363 BuildMI(*BB, MI, MI->getDebugLoc(), TII.get(AMDGPU::COPY), DstReg)
364 .addReg(Reg, {}, ComposedSubIdx);
365
366 return MachineOperand::CreateReg(DstReg, MO.isDef(), MO.isImplicit(),
367 MO.isKill(), MO.isDead(), MO.isUndef(),
368 MO.isEarlyClobber(), 0, MO.isDebug(),
369 MO.isInternalRead());
370 }
371
372 assert(MO.isImm());
373
374 APInt Imm(64, MO.getImm());
375
376 switch (SubIdx) {
377 default:
378 llvm_unreachable("do not know to split immediate with this sub index.");
379 case AMDGPU::sub0:
380 return MachineOperand::CreateImm(Imm.getLoBits(32).getSExtValue());
381 case AMDGPU::sub1:
382 return MachineOperand::CreateImm(Imm.getHiBits(32).getSExtValue());
383 }
384}
385
386static unsigned getLogicalBitOpcode(unsigned Opc, bool Is64) {
387 switch (Opc) {
388 case AMDGPU::G_AND:
389 return Is64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
390 case AMDGPU::G_OR:
391 return Is64 ? AMDGPU::S_OR_B64 : AMDGPU::S_OR_B32;
392 case AMDGPU::G_XOR:
393 return Is64 ? AMDGPU::S_XOR_B64 : AMDGPU::S_XOR_B32;
394 default:
395 llvm_unreachable("not a bit op");
396 }
397}
398
399bool AMDGPUInstructionSelector::selectG_AND_OR_XOR(MachineInstr &I) const {
400 Register DstReg = I.getOperand(0).getReg();
401 unsigned Size = RBI.getSizeInBits(DstReg, *MRI, TRI);
402
403 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
404 if (DstRB->getID() != AMDGPU::SGPRRegBankID &&
405 DstRB->getID() != AMDGPU::VCCRegBankID)
406 return false;
407
408 bool Is64 = Size > 32 || (DstRB->getID() == AMDGPU::VCCRegBankID &&
409 STI.isWave64());
410 I.setDesc(TII.get(getLogicalBitOpcode(I.getOpcode(), Is64)));
411
412 // Dead implicit-def of scc
413 I.addOperand(MachineOperand::CreateReg(AMDGPU::SCC, true, // isDef
414 true, // isImp
415 false, // isKill
416 true)); // isDead
417 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
418 return true;
419}
420
421bool AMDGPUInstructionSelector::selectG_ADD_SUB(MachineInstr &I) const {
422 MachineBasicBlock *BB = I.getParent();
424 Register DstReg = I.getOperand(0).getReg();
425 const DebugLoc &DL = I.getDebugLoc();
426 LLT Ty = MRI->getType(DstReg);
427 if (Ty.isVector())
428 return false;
429
430 unsigned Size = Ty.getSizeInBits();
431 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
432 const bool IsSALU = DstRB->getID() == AMDGPU::SGPRRegBankID;
433 const bool Sub = I.getOpcode() == TargetOpcode::G_SUB;
434
435 if (Size == 32) {
436 if (IsSALU) {
437 const unsigned Opc = Sub ? AMDGPU::S_SUB_U32 : AMDGPU::S_ADD_U32;
438 MachineInstr *Add =
439 BuildMI(*BB, &I, DL, TII.get(Opc), DstReg)
440 .add(I.getOperand(1))
441 .add(I.getOperand(2))
442 .setOperandDead(3); // Dead scc
443 I.eraseFromParent();
444 constrainSelectedInstRegOperands(*Add, TII, TRI, RBI);
445 return true;
446 }
447
448 if (STI.hasAddNoCarryInsts()) {
449 const unsigned Opc = Sub ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_ADD_U32_e64;
450 I.setDesc(TII.get(Opc));
451 I.addOperand(*MF, MachineOperand::CreateImm(0));
452 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
453 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
454 return true;
455 }
456
457 const unsigned Opc = Sub ? AMDGPU::V_SUB_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e64;
458
459 Register UnusedCarry = MRI->createVirtualRegister(TRI.getWaveMaskRegClass());
460 MachineInstr *Add
461 = BuildMI(*BB, &I, DL, TII.get(Opc), DstReg)
462 .addDef(UnusedCarry, RegState::Dead)
463 .add(I.getOperand(1))
464 .add(I.getOperand(2))
465 .addImm(0);
466 I.eraseFromParent();
467 constrainSelectedInstRegOperands(*Add, TII, TRI, RBI);
468 return true;
469 }
470
471 assert(!Sub && "illegal sub should not reach here");
472
473 const TargetRegisterClass &RC
474 = IsSALU ? AMDGPU::SReg_64_XEXECRegClass : AMDGPU::VReg_64RegClass;
475 const TargetRegisterClass &HalfRC
476 = IsSALU ? AMDGPU::SReg_32RegClass : AMDGPU::VGPR_32RegClass;
477
478 MachineOperand Lo1(getSubOperand64(I.getOperand(1), HalfRC, AMDGPU::sub0));
479 MachineOperand Lo2(getSubOperand64(I.getOperand(2), HalfRC, AMDGPU::sub0));
480 MachineOperand Hi1(getSubOperand64(I.getOperand(1), HalfRC, AMDGPU::sub1));
481 MachineOperand Hi2(getSubOperand64(I.getOperand(2), HalfRC, AMDGPU::sub1));
482
483 Register DstLo = MRI->createVirtualRegister(&HalfRC);
484 Register DstHi = MRI->createVirtualRegister(&HalfRC);
485
486 if (IsSALU) {
487 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADD_U32), DstLo)
488 .add(Lo1)
489 .add(Lo2);
490 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADDC_U32), DstHi)
491 .add(Hi1)
492 .add(Hi2)
493 .setOperandDead(3); // Dead scc
494 } else {
495 const TargetRegisterClass *CarryRC = TRI.getWaveMaskRegClass();
496 Register CarryReg = MRI->createVirtualRegister(CarryRC);
497 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_ADD_CO_U32_e64), DstLo)
498 .addDef(CarryReg)
499 .add(Lo1)
500 .add(Lo2)
501 .addImm(0);
502 MachineInstr *Addc = BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_ADDC_U32_e64), DstHi)
503 .addDef(MRI->createVirtualRegister(CarryRC), RegState::Dead)
504 .add(Hi1)
505 .add(Hi2)
506 .addReg(CarryReg, RegState::Kill)
507 .addImm(0);
508
509 constrainSelectedInstRegOperands(*Addc, TII, TRI, RBI);
510 }
511
512 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
513 .addReg(DstLo)
514 .addImm(AMDGPU::sub0)
515 .addReg(DstHi)
516 .addImm(AMDGPU::sub1);
517
518
519 if (!RBI.constrainGenericRegister(DstReg, RC, *MRI))
520 return false;
521
522 I.eraseFromParent();
523 return true;
524}
525
526bool AMDGPUInstructionSelector::selectG_UADDO_USUBO_UADDE_USUBE(
527 MachineInstr &I) const {
528 MachineBasicBlock *BB = I.getParent();
530 const DebugLoc &DL = I.getDebugLoc();
531 Register Dst0Reg = I.getOperand(0).getReg();
532 Register Dst1Reg = I.getOperand(1).getReg();
533 const bool IsAdd = I.getOpcode() == AMDGPU::G_UADDO ||
534 I.getOpcode() == AMDGPU::G_UADDE;
535 const bool HasCarryIn = I.getOpcode() == AMDGPU::G_UADDE ||
536 I.getOpcode() == AMDGPU::G_USUBE;
537
538 if (isVCC(Dst1Reg, *MRI)) {
539 unsigned NoCarryOpc =
540 IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
541 unsigned CarryOpc = IsAdd ? AMDGPU::V_ADDC_U32_e64 : AMDGPU::V_SUBB_U32_e64;
542 I.setDesc(TII.get(HasCarryIn ? CarryOpc : NoCarryOpc));
543 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
544 I.addOperand(*MF, MachineOperand::CreateImm(0));
545 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
546 return true;
547 }
548
549 Register Src0Reg = I.getOperand(2).getReg();
550 Register Src1Reg = I.getOperand(3).getReg();
551
552 if (HasCarryIn) {
553 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC)
554 .addReg(I.getOperand(4).getReg());
555 }
556
557 unsigned NoCarryOpc = IsAdd ? AMDGPU::S_ADD_U32 : AMDGPU::S_SUB_U32;
558 unsigned CarryOpc = IsAdd ? AMDGPU::S_ADDC_U32 : AMDGPU::S_SUBB_U32;
559
560 auto CarryInst = BuildMI(*BB, &I, DL, TII.get(HasCarryIn ? CarryOpc : NoCarryOpc), Dst0Reg)
561 .add(I.getOperand(2))
562 .add(I.getOperand(3));
563
564 if (MRI->use_nodbg_empty(Dst1Reg)) {
565 CarryInst.setOperandDead(3); // Dead scc
566 } else {
567 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst1Reg)
568 .addReg(AMDGPU::SCC);
569 if (!MRI->getRegClassOrNull(Dst1Reg))
570 MRI->setRegClass(Dst1Reg, &AMDGPU::SReg_32RegClass);
571 }
572
573 if (!RBI.constrainGenericRegister(Dst0Reg, AMDGPU::SReg_32RegClass, *MRI) ||
574 !RBI.constrainGenericRegister(Src0Reg, AMDGPU::SReg_32RegClass, *MRI) ||
575 !RBI.constrainGenericRegister(Src1Reg, AMDGPU::SReg_32RegClass, *MRI))
576 return false;
577
578 if (HasCarryIn &&
579 !RBI.constrainGenericRegister(I.getOperand(4).getReg(),
580 AMDGPU::SReg_32RegClass, *MRI))
581 return false;
582
583 I.eraseFromParent();
584 return true;
585}
586
587bool AMDGPUInstructionSelector::selectG_AMDGPU_MAD_64_32(
588 MachineInstr &I) const {
589 MachineBasicBlock *BB = I.getParent();
591 const bool IsUnsigned = I.getOpcode() == AMDGPU::G_AMDGPU_MAD_U64_U32;
592 bool UseNoCarry = Subtarget->hasMadNC64_32Insts() &&
593 MRI->use_nodbg_empty(I.getOperand(1).getReg());
594
595 unsigned Opc;
596 if (Subtarget->hasMADIntraFwdBug())
597 Opc = IsUnsigned ? AMDGPU::V_MAD_U64_U32_gfx11_e64
598 : AMDGPU::V_MAD_I64_I32_gfx11_e64;
599 else if (UseNoCarry)
600 Opc = IsUnsigned ? AMDGPU::V_MAD_NC_U64_U32_e64
601 : AMDGPU::V_MAD_NC_I64_I32_e64;
602 else
603 Opc = IsUnsigned ? AMDGPU::V_MAD_U64_U32_e64 : AMDGPU::V_MAD_I64_I32_e64;
604
605 if (UseNoCarry)
606 I.removeOperand(1);
607
608 I.setDesc(TII.get(Opc));
609 I.addOperand(*MF, MachineOperand::CreateImm(0));
610 I.addImplicitDefUseOperands(*MF);
611 I.getOperand(0).setIsEarlyClobber(true);
612 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
613 return true;
614}
615
616// TODO: We should probably legalize these to only using 32-bit results.
617bool AMDGPUInstructionSelector::selectG_EXTRACT(MachineInstr &I) const {
618 MachineBasicBlock *BB = I.getParent();
619 Register DstReg = I.getOperand(0).getReg();
620 Register SrcReg = I.getOperand(1).getReg();
621 LLT DstTy = MRI->getType(DstReg);
622 LLT SrcTy = MRI->getType(SrcReg);
623 const unsigned SrcSize = SrcTy.getSizeInBits();
624 unsigned DstSize = DstTy.getSizeInBits();
625
626 // TODO: Should handle any multiple of 32 offset.
627 unsigned Offset = I.getOperand(2).getImm();
628 if (Offset % 32 != 0 || DstSize > 128)
629 return false;
630
631 // 16-bit operations really use 32-bit registers.
632 // FIXME: Probably should not allow 16-bit G_EXTRACT results.
633 if (DstSize == 16)
634 DstSize = 32;
635
636 const TargetRegisterClass *DstRC =
637 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
638 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
639 return false;
640
641 const RegisterBank *SrcBank = RBI.getRegBank(SrcReg, *MRI, TRI);
642 const TargetRegisterClass *SrcRC =
643 TRI.getRegClassForSizeOnBank(SrcSize, *SrcBank);
644 if (!SrcRC)
645 return false;
646 unsigned SubReg = SIRegisterInfo::getSubRegFromChannel(Offset / 32,
647 DstSize / 32);
648 SrcRC = TRI.getSubClassWithSubReg(SrcRC, SubReg);
649 if (!SrcRC)
650 return false;
651
652 SrcReg = constrainOperandRegClass(*MF, TRI, *MRI, TII, RBI, I,
653 *SrcRC, I.getOperand(1));
654 const DebugLoc &DL = I.getDebugLoc();
655 BuildMI(*BB, &I, DL, TII.get(TargetOpcode::COPY), DstReg)
656 .addReg(SrcReg, {}, SubReg);
657
658 I.eraseFromParent();
659 return true;
660}
661
662bool AMDGPUInstructionSelector::selectS16MergeToS32(MachineInstr &MI) const {
663 Register Dst = MI.getOperand(0).getReg();
664 Register Src0 = MI.getOperand(1).getReg();
665 Register Src1 = MI.getOperand(2).getReg();
666
667 LLT Src0Ty = MRI->getType(Src0);
668 LLT Src1Ty = MRI->getType(Src1);
669
670 const RegisterBank *DstBank = RBI.getRegBank(Dst, *MRI, TRI);
671 const RegisterBank *Src0Bank = RBI.getRegBank(Src0, *MRI, TRI);
672 const RegisterBank *Src1Bank = RBI.getRegBank(Src1, *MRI, TRI);
673 const bool IsVector = DstBank->getID() == AMDGPU::VGPRRegBankID;
674
675 Register ShiftSrc0;
676 Register ShiftSrc1;
677
678 const DebugLoc &DL = MI.getDebugLoc();
679 MachineBasicBlock *BB = MI.getParent();
680
681 // VGPR case
682 if (IsVector) {
683 // If source are both VGPR16, use REG_SEQUENCE with lo16/hi16 subregisters
684 if (Src0Bank->getID() == AMDGPU::VGPRRegBankID &&
685 Src1Bank->getID() == AMDGPU::VGPRRegBankID &&
686 Src0Ty == LLT::scalar(16) && Src1Ty == LLT::scalar(16)) {
687 BuildMI(*BB, MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), Dst)
688 .addReg(Src0)
689 .addImm(AMDGPU::lo16)
690 .addReg(Src1)
691 .addImm(AMDGPU::hi16);
692
693 if (!RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI))
694 return false;
695
696 MI.eraseFromParent();
697 return true;
698 }
699
700 // Otherwise, use V_LSHL_OR_B32_e64
701 Register TmpReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
702 auto MIB = BuildMI(*BB, MI, DL, TII.get(AMDGPU::V_AND_B32_e32), TmpReg)
703 .addImm(0xFFFF)
704 .addReg(Src0);
705 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
706
707 MIB = BuildMI(*BB, MI, DL, TII.get(AMDGPU::V_LSHL_OR_B32_e64), Dst)
708 .addReg(Src1)
709 .addImm(16)
710 .addReg(TmpReg);
711 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
712
713 MI.eraseFromParent();
714 return true;
715 }
716
717 // SGPR case -> S_PACK_*_B32_B16
718 // With multiple uses of the shift, this will duplicate the shift and
719 // increase register pressure.
720 //
721 // (merge (lshr_oneuse $src0, 16), (lshr_oneuse $src1, 16)
722 // => (S_PACK_HH_B32_B16 $src0, $src1)
723 // (merge (lshr_oneuse SReg_32:$src0, 16), $src1)
724 // => (S_PACK_HL_B32_B16 $src0, $src1)
725 // (merge $src0, (lshr_oneuse SReg_32:$src1, 16))
726 // => (S_PACK_LH_B32_B16 $src0, $src1)
727 // (merge $src0, $src1)
728 // => (S_PACK_LL_B32_B16 $src0, $src1)
729
730 bool Shift0 = mi_match(
731 Src0, *MRI, m_OneUse(m_GLShr(m_Reg(ShiftSrc0), m_SpecificICst(16))));
732
733 bool Shift1 = mi_match(
734 Src1, *MRI, m_OneUse(m_GLShr(m_Reg(ShiftSrc1), m_SpecificICst(16))));
735
736 unsigned Opc = AMDGPU::S_PACK_LL_B32_B16;
737 if (Shift0 && Shift1) {
738 Opc = AMDGPU::S_PACK_HH_B32_B16;
739 MI.getOperand(1).setReg(ShiftSrc0);
740 MI.getOperand(2).setReg(ShiftSrc1);
741 } else if (Shift1) {
742 Opc = AMDGPU::S_PACK_LH_B32_B16;
743 MI.getOperand(2).setReg(ShiftSrc1);
744 } else if (Shift0) {
745 auto ConstSrc1 =
746 getAnyConstantVRegValWithLookThrough(Src1, *MRI, true, true);
747 if (ConstSrc1 && ConstSrc1->Value == 0) {
748 // build_vector_trunc (lshr $src0, 16), 0 -> s_lshr_b32 $src0, 16
749 auto MIB = BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_LSHR_B32), Dst)
750 .addReg(ShiftSrc0)
751 .addImm(16)
752 .setOperandDead(3); // Dead scc
753
754 MI.eraseFromParent();
755 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
756 return true;
757 }
758 if (STI.hasSPackHL()) {
759 Opc = AMDGPU::S_PACK_HL_B32_B16;
760 MI.getOperand(1).setReg(ShiftSrc0);
761 }
762 }
763
764 MI.setDesc(TII.get(Opc));
765 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
766 return true;
767}
768
769// Pack each pair of s16 into an s32 with S_PACK_LL_B32_B16, then combine the
770// s32 pieces into the destination with a REG_SEQUENCE.
771bool AMDGPUInstructionSelector::selectS16MergeToWide(MachineInstr &MI) const {
772 MachineBasicBlock *BB = MI.getParent();
773 const DebugLoc &DL = MI.getDebugLoc();
774 Register DstReg = MI.getOperand(0).getReg();
775 const unsigned DstSize = MRI->getType(DstReg).getSizeInBits();
776 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
777 const unsigned NumSrc = MI.getNumOperands() - 1;
778
779 // Pack each pair of s16 sources into an s32.
781 for (unsigned I = 0; I != NumSrc; I += 2) {
782 Register S32 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
783 auto Pack = BuildMI(*BB, MI, DL, TII.get(AMDGPU::S_PACK_LL_B32_B16), S32)
784 .addReg(MI.getOperand(I + 1).getReg())
785 .addReg(MI.getOperand(I + 2).getReg());
786 constrainSelectedInstRegOperands(*Pack, TII, TRI, RBI);
787 S32Regs.push_back(S32);
788 }
789
790 // Combine the s32 pieces into the destination with a REG_SEQUENCE.
791 const TargetRegisterClass *DstRC =
792 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
793 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
794 return false;
795 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(DstRC, /*EltSize=*/4);
796 auto MIB = BuildMI(*BB, MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), DstReg);
797 for (unsigned I = 0, E = S32Regs.size(); I != E; ++I)
798 MIB.addReg(S32Regs[I]).addImm(SubRegs[I]);
799
800 MI.eraseFromParent();
801 return true;
802}
803
804bool AMDGPUInstructionSelector::selectG_MERGE_VALUES(MachineInstr &MI) const {
805 MachineBasicBlock *BB = MI.getParent();
806 Register DstReg = MI.getOperand(0).getReg();
807 LLT DstTy = MRI->getType(DstReg);
808 LLT SrcTy = MRI->getType(MI.getOperand(1).getReg());
809
810 const unsigned SrcSize = SrcTy.getSizeInBits();
811 if (SrcSize < 32) {
812 // Handle s32 <- G_MERGE_VALUES s16, s16
813 if (SrcSize == 16 && DstTy.getSizeInBits() == 32 &&
814 MI.getNumOperands() == 3) {
815 return selectS16MergeToS32(MI);
816 }
817 // With true16 a scalar s16 is a register type, so a scalar wider than 32
818 // bits can be built from s16 pieces.
819 bool IsWideS16Merge = SrcSize == 16 && DstTy.getSizeInBits() > 32 &&
820 DstTy.getSizeInBits() % 32 == 0;
821
822 // SGPRs have no 16-bit subregisters, so pack pairs of s16 with S_PACK.
823 if (IsWideS16Merge &&
824 RBI.getRegBank(DstReg, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID)
825 return selectS16MergeToWide(MI);
826
827 // A VGPR wide s16 merge falls through to the generic path below.
828 if (!IsWideS16Merge)
829 return selectImpl(MI, *CoverageInfo);
830 }
831
832 const DebugLoc &DL = MI.getDebugLoc();
833 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
834 const unsigned DstSize = DstTy.getSizeInBits();
835 const TargetRegisterClass *DstRC =
836 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
837 if (!DstRC)
838 return false;
839
840 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(DstRC, SrcSize / 8);
841 MachineInstrBuilder MIB =
842 BuildMI(*BB, &MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), DstReg);
843 for (int I = 0, E = MI.getNumOperands() - 1; I != E; ++I) {
844 MachineOperand &Src = MI.getOperand(I + 1);
845 Register SrcReg = Src.getReg();
846 MIB.addReg(SrcReg, getUndefRegState(Src.isUndef()));
847 MIB.addImm(SubRegs[I]);
848
849 const TargetRegisterClass *SrcRC =
850 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
851 if (SrcRC && !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
852 return false;
853 }
854
855 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
856 return false;
857
858 MI.eraseFromParent();
859 return true;
860}
861
862bool AMDGPUInstructionSelector::selectG_UNMERGE_VALUES(MachineInstr &MI) const {
863 MachineBasicBlock *BB = MI.getParent();
864 const int NumDst = MI.getNumOperands() - 1;
865
866 MachineOperand &Src = MI.getOperand(NumDst);
867
868 Register SrcReg = Src.getReg();
869 Register DstReg0 = MI.getOperand(0).getReg();
870 LLT DstTy = MRI->getType(DstReg0);
871 LLT SrcTy = MRI->getType(SrcReg);
872
873 const unsigned DstSize = DstTy.getSizeInBits();
874 const unsigned SrcSize = SrcTy.getSizeInBits();
875 const DebugLoc &DL = MI.getDebugLoc();
876 const RegisterBank *SrcBank = RBI.getRegBank(SrcReg, *MRI, TRI);
877
878 const TargetRegisterClass *SrcRC =
879 TRI.getRegClassForSizeOnBank(SrcSize, *SrcBank);
880
881 // Note we could have mixed SGPR and VGPR destination banks for an SGPR
882 // source, and this relies on the fact that the same subregister indices are
883 // used for both.
884 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(SrcRC, DstSize / 8);
885 for (int I = 0, E = NumDst; I != E; ++I) {
886 Register DstReg = MI.getOperand(I).getReg();
887 // hi16:sreg_32 is not allowed so explicitly shift upper 16-bits.
888 if (SrcBank->getID() == AMDGPU::SGPRRegBankID &&
889 SubRegs[I] == AMDGPU::hi16) {
890 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_LSHR_B32), DstReg)
891 .addReg(SrcReg)
892 .addImm(16);
893 } else {
894 BuildMI(*BB, &MI, DL, TII.get(TargetOpcode::COPY), DstReg)
895 .addReg(SrcReg, {}, SubRegs[I]);
896
897 // Make sure the subregister index is valid for the source register.
898 SrcRC = TRI.getSubClassWithSubReg(SrcRC, SubRegs[I]);
899 }
900
901 if (!SrcRC || !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
902 return false;
903
904 const TargetRegisterClass *DstRC =
905 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
906 if (DstRC && !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
907 return false;
908 }
909
910 MI.eraseFromParent();
911 return true;
912}
913
914bool AMDGPUInstructionSelector::selectG_BUILD_VECTOR(MachineInstr &MI) const {
915 assert(MI.getOpcode() == AMDGPU::G_BUILD_VECTOR_TRUNC ||
916 MI.getOpcode() == AMDGPU::G_BUILD_VECTOR);
917
918 Register Src0 = MI.getOperand(1).getReg();
919 Register Src1 = MI.getOperand(2).getReg();
920 LLT SrcTy = MRI->getType(Src0);
921 const unsigned SrcSize = SrcTy.getSizeInBits();
922
923 // BUILD_VECTOR with >=32 bits source is handled by MERGE_VALUE.
924 if (MI.getOpcode() == AMDGPU::G_BUILD_VECTOR && SrcSize >= 32) {
925 return selectG_MERGE_VALUES(MI);
926 }
927
928 // Selection logic below is for V2S16 only.
929 // For G_BUILD_VECTOR_TRUNC, additionally check that the operands are s32.
930 Register Dst = MI.getOperand(0).getReg();
931 if (MRI->getType(Dst) != LLT::fixed_vector(2, 16) ||
932 (MI.getOpcode() == AMDGPU::G_BUILD_VECTOR_TRUNC &&
933 SrcTy != LLT::scalar(32)))
934 return selectImpl(MI, *CoverageInfo);
935
936 const RegisterBank *DstBank = RBI.getRegBank(Dst, *MRI, TRI);
937 if (DstBank->getID() == AMDGPU::AGPRRegBankID)
938 return false;
939
940 assert(DstBank->getID() == AMDGPU::SGPRRegBankID ||
941 DstBank->getID() == AMDGPU::VGPRRegBankID);
942 const bool IsVector = DstBank->getID() == AMDGPU::VGPRRegBankID;
943
944 const DebugLoc &DL = MI.getDebugLoc();
945 MachineBasicBlock *BB = MI.getParent();
946
947 // First, before trying TableGen patterns, check if both sources are
948 // constants. In those cases, we can trivially compute the final constant
949 // and emit a simple move.
950 auto ConstSrc1 = getAnyConstantVRegValWithLookThrough(Src1, *MRI, true, true);
951 if (ConstSrc1) {
952 auto ConstSrc0 =
953 getAnyConstantVRegValWithLookThrough(Src0, *MRI, true, true);
954 if (ConstSrc0) {
955 const int64_t K0 = ConstSrc0->Value.getSExtValue();
956 const int64_t K1 = ConstSrc1->Value.getSExtValue();
957 uint32_t Lo16 = static_cast<uint32_t>(K0) & 0xffff;
958 uint32_t Hi16 = static_cast<uint32_t>(K1) & 0xffff;
959 uint32_t Imm = Lo16 | (Hi16 << 16);
960
961 // VALU
962 if (IsVector) {
963 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::V_MOV_B32_e32), Dst).addImm(Imm);
964 MI.eraseFromParent();
965 return RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI);
966 }
967
968 // SALU
969 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), Dst).addImm(Imm);
970 MI.eraseFromParent();
971 return RBI.constrainGenericRegister(Dst, AMDGPU::SReg_32RegClass, *MRI);
972 }
973 }
974
975 // Now try TableGen patterns.
976 if (selectImpl(MI, *CoverageInfo))
977 return true;
978
979 // TODO: This should probably be a combine somewhere
980 // (build_vector $src0, undef) -> copy $src0
981 MachineInstr *Src1Def = getDefIgnoringCopies(Src1, *MRI);
982 if (Src1Def->getOpcode() == AMDGPU::G_IMPLICIT_DEF) {
983 MI.setDesc(TII.get(AMDGPU::COPY));
984 MI.removeOperand(2);
985 const auto &RC =
986 IsVector ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
987 return RBI.constrainGenericRegister(Dst, RC, *MRI) &&
988 RBI.constrainGenericRegister(Src0, RC, *MRI);
989 }
990
991 return selectS16MergeToS32(MI);
992}
993
994bool AMDGPUInstructionSelector::selectG_IMPLICIT_DEF(MachineInstr &I) const {
995 const MachineOperand &MO = I.getOperand(0);
996
997 // FIXME: Interface for getConstrainedRegClassForReg needs work. The
998 // regbank check here is to know why getConstrainedRegClassForReg failed.
999 const TargetRegisterClass *RC =
1000 TRI.getConstrainedRegClassForReg(MO.getReg(), *MRI);
1001 if ((!RC && !MRI->getRegBankOrNull(MO.getReg())) ||
1002 (RC && RBI.constrainGenericRegister(MO.getReg(), *RC, *MRI))) {
1003 I.setDesc(TII.get(TargetOpcode::IMPLICIT_DEF));
1004 return true;
1005 }
1006
1007 return false;
1008}
1009
1010bool AMDGPUInstructionSelector::selectG_INSERT(MachineInstr &I) const {
1011 MachineBasicBlock *BB = I.getParent();
1012
1013 Register DstReg = I.getOperand(0).getReg();
1014 Register Src0Reg = I.getOperand(1).getReg();
1015 Register Src1Reg = I.getOperand(2).getReg();
1016 LLT Src1Ty = MRI->getType(Src1Reg);
1017
1018 unsigned DstSize = MRI->getType(DstReg).getSizeInBits();
1019 unsigned InsSize = Src1Ty.getSizeInBits();
1020
1021 int64_t Offset = I.getOperand(3).getImm();
1022
1023 // FIXME: These cases should have been illegal and unnecessary to check here.
1024 if (Offset % 32 != 0 || InsSize % 32 != 0)
1025 return false;
1026
1027 // Currently not handled by getSubRegFromChannel.
1028 if (InsSize > 128)
1029 return false;
1030
1031 unsigned SubReg = TRI.getSubRegFromChannel(Offset / 32, InsSize / 32);
1032 if (SubReg == AMDGPU::NoSubRegister)
1033 return false;
1034
1035 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
1036 const TargetRegisterClass *DstRC =
1037 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
1038 if (!DstRC)
1039 return false;
1040
1041 const RegisterBank *Src0Bank = RBI.getRegBank(Src0Reg, *MRI, TRI);
1042 const RegisterBank *Src1Bank = RBI.getRegBank(Src1Reg, *MRI, TRI);
1043 const TargetRegisterClass *Src0RC =
1044 TRI.getRegClassForSizeOnBank(DstSize, *Src0Bank);
1045 const TargetRegisterClass *Src1RC =
1046 TRI.getRegClassForSizeOnBank(InsSize, *Src1Bank);
1047
1048 // Deal with weird cases where the class only partially supports the subreg
1049 // index.
1050 Src0RC = TRI.getSubClassWithSubReg(Src0RC, SubReg);
1051 if (!Src0RC || !Src1RC)
1052 return false;
1053
1054 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
1055 !RBI.constrainGenericRegister(Src0Reg, *Src0RC, *MRI) ||
1056 !RBI.constrainGenericRegister(Src1Reg, *Src1RC, *MRI))
1057 return false;
1058
1059 const DebugLoc &DL = I.getDebugLoc();
1060 BuildMI(*BB, &I, DL, TII.get(TargetOpcode::INSERT_SUBREG), DstReg)
1061 .addReg(Src0Reg)
1062 .addReg(Src1Reg)
1063 .addImm(SubReg);
1064
1065 I.eraseFromParent();
1066 return true;
1067}
1068
1069bool AMDGPUInstructionSelector::selectG_SBFX_UBFX(MachineInstr &MI) const {
1070 Register DstReg = MI.getOperand(0).getReg();
1071 Register SrcReg = MI.getOperand(1).getReg();
1072 Register OffsetReg = MI.getOperand(2).getReg();
1073 Register WidthReg = MI.getOperand(3).getReg();
1074
1075 assert(RBI.getRegBank(DstReg, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID &&
1076 "scalar BFX instructions are expanded in regbankselect");
1077 assert(MRI->getType(MI.getOperand(0).getReg()).getSizeInBits() == 32 &&
1078 "64-bit vector BFX instructions are expanded in regbankselect");
1079
1080 const DebugLoc &DL = MI.getDebugLoc();
1081 MachineBasicBlock *MBB = MI.getParent();
1082
1083 bool IsSigned = MI.getOpcode() == TargetOpcode::G_SBFX;
1084 unsigned Opc = IsSigned ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
1085 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), DstReg)
1086 .addReg(SrcReg)
1087 .addReg(OffsetReg)
1088 .addReg(WidthReg);
1089 MI.eraseFromParent();
1090 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1091 return true;
1092}
1093
1094bool AMDGPUInstructionSelector::selectInterpP1F16(MachineInstr &MI) const {
1095 if (STI.getLDSBankCount() != 16)
1096 return selectImpl(MI, *CoverageInfo);
1097
1098 Register Dst = MI.getOperand(0).getReg();
1099 Register Src0 = MI.getOperand(2).getReg();
1100 Register M0Val = MI.getOperand(6).getReg();
1101 if (!RBI.constrainGenericRegister(M0Val, AMDGPU::SReg_32RegClass, *MRI) ||
1102 !RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI) ||
1103 !RBI.constrainGenericRegister(Src0, AMDGPU::VGPR_32RegClass, *MRI))
1104 return false;
1105
1106 // This requires 2 instructions. It is possible to write a pattern to support
1107 // this, but the generated isel emitter doesn't correctly deal with multiple
1108 // output instructions using the same physical register input. The copy to m0
1109 // is incorrectly placed before the second instruction.
1110 //
1111 // TODO: Match source modifiers.
1112
1113 Register InterpMov = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1114 const DebugLoc &DL = MI.getDebugLoc();
1115 MachineBasicBlock *MBB = MI.getParent();
1116
1117 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1118 .addReg(M0Val);
1119 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_INTERP_MOV_F32), InterpMov)
1120 .addImm(2)
1121 .addImm(MI.getOperand(4).getImm()) // $attr
1122 .addImm(MI.getOperand(3).getImm()); // $attrchan
1123
1124 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_INTERP_P1LV_F16), Dst)
1125 .addImm(0) // $src0_modifiers
1126 .addReg(Src0) // $src0
1127 .addImm(MI.getOperand(4).getImm()) // $attr
1128 .addImm(MI.getOperand(3).getImm()) // $attrchan
1129 .addImm(0) // $src2_modifiers
1130 .addReg(InterpMov) // $src2 - 2 f16 values selected by high
1131 .addImm(MI.getOperand(5).getImm()) // $high
1132 .addImm(0) // $clamp
1133 .addImm(0); // $omod
1134
1135 MI.eraseFromParent();
1136 return true;
1137}
1138
1139// Writelane is special in that it can use SGPR and M0 (which would normally
1140// count as using the constant bus twice - but in this case it is allowed since
1141// the lane selector doesn't count as a use of the constant bus). However, it is
1142// still required to abide by the 1 SGPR rule. Fix this up if we might have
1143// multiple SGPRs.
1144bool AMDGPUInstructionSelector::selectWritelane(MachineInstr &MI) const {
1145 // With a constant bus limit of at least 2, there's no issue.
1146 if (STI.getConstantBusLimit(AMDGPU::V_WRITELANE_B32) > 1)
1147 return selectImpl(MI, *CoverageInfo);
1148
1149 MachineBasicBlock *MBB = MI.getParent();
1150 const DebugLoc &DL = MI.getDebugLoc();
1151 Register VDst = MI.getOperand(0).getReg();
1152 Register Val = MI.getOperand(2).getReg();
1153 Register LaneSelect = MI.getOperand(3).getReg();
1154 Register VDstIn = MI.getOperand(4).getReg();
1155
1156 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_WRITELANE_B32), VDst);
1157
1158 std::optional<ValueAndVReg> ConstSelect =
1159 getIConstantVRegValWithLookThrough(LaneSelect, *MRI);
1160 if (ConstSelect) {
1161 // The selector has to be an inline immediate, so we can use whatever for
1162 // the other operands.
1163 MIB.addReg(Val);
1164 MIB.addImm(ConstSelect->Value.getSExtValue() &
1165 maskTrailingOnes<uint64_t>(STI.getWavefrontSizeLog2()));
1166 } else {
1167 std::optional<ValueAndVReg> ConstVal =
1169
1170 // If the value written is an inline immediate, we can get away without a
1171 // copy to m0.
1172 if (ConstVal && AMDGPU::isInlinableLiteral32(ConstVal->Value.getSExtValue(),
1173 STI.hasInv2PiInlineImm())) {
1174 MIB.addImm(ConstVal->Value.getSExtValue());
1175 MIB.addReg(LaneSelect);
1176 } else {
1177 MIB.addReg(Val);
1178
1179 // If the lane selector was originally in a VGPR and copied with
1180 // readfirstlane, there's a hazard to read the same SGPR from the
1181 // VALU. Constrain to a different SGPR to help avoid needing a nop later.
1182 RBI.constrainGenericRegister(LaneSelect, AMDGPU::SReg_32_XM0RegClass, *MRI);
1183
1184 BuildMI(*MBB, *MIB, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1185 .addReg(LaneSelect);
1186 MIB.addReg(AMDGPU::M0);
1187 }
1188 }
1189
1190 MIB.addReg(VDstIn);
1191
1192 MI.eraseFromParent();
1193 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1194 return true;
1195}
1196
1197// We need to handle this here because tablegen doesn't support matching
1198// instructions with multiple outputs.
1199bool AMDGPUInstructionSelector::selectDivScale(MachineInstr &MI) const {
1200 Register Dst0 = MI.getOperand(0).getReg();
1201 Register Dst1 = MI.getOperand(1).getReg();
1202
1203 LLT Ty = MRI->getType(Dst0);
1204 unsigned Opc;
1205 if (Ty == LLT::scalar(32))
1206 Opc = AMDGPU::V_DIV_SCALE_F32_e64;
1207 else if (Ty == LLT::scalar(64))
1208 Opc = AMDGPU::V_DIV_SCALE_F64_e64;
1209 else
1210 return false;
1211
1212 // TODO: Match source modifiers.
1213
1214 const DebugLoc &DL = MI.getDebugLoc();
1215 MachineBasicBlock *MBB = MI.getParent();
1216
1217 Register Numer = MI.getOperand(3).getReg();
1218 Register Denom = MI.getOperand(4).getReg();
1219 unsigned ChooseDenom = MI.getOperand(5).getImm();
1220
1221 Register Src0 = ChooseDenom != 0 ? Numer : Denom;
1222
1223 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), Dst0)
1224 .addDef(Dst1)
1225 .addImm(0) // $src0_modifiers
1226 .addUse(Src0) // $src0
1227 .addImm(0) // $src1_modifiers
1228 .addUse(Denom) // $src1
1229 .addImm(0) // $src2_modifiers
1230 .addUse(Numer) // $src2
1231 .addImm(0) // $clamp
1232 .addImm(0); // $omod
1233
1234 MI.eraseFromParent();
1235 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1236 return true;
1237}
1238
1239bool AMDGPUInstructionSelector::selectG_INTRINSIC(MachineInstr &I) const {
1240 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(I).getIntrinsicID();
1241 switch (IntrinsicID) {
1242 case Intrinsic::amdgcn_if_break: {
1243 MachineBasicBlock *BB = I.getParent();
1244
1245 // FIXME: Manually selecting to avoid dealing with the SReg_1 trick
1246 // SelectionDAG uses for wave32 vs wave64.
1247 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::SI_IF_BREAK))
1248 .add(I.getOperand(0))
1249 .add(I.getOperand(2))
1250 .add(I.getOperand(3))
1251 .setOperandDead(3); // implicit-def $scc
1252
1253 Register DstReg = I.getOperand(0).getReg();
1254 Register Src0Reg = I.getOperand(2).getReg();
1255 Register Src1Reg = I.getOperand(3).getReg();
1256
1257 I.eraseFromParent();
1258
1259 for (Register Reg : { DstReg, Src0Reg, Src1Reg })
1260 MRI->setRegClass(Reg, TRI.getWaveMaskRegClass());
1261
1262 return true;
1263 }
1264 case Intrinsic::amdgcn_interp_p1_f16:
1265 return selectInterpP1F16(I);
1266 case Intrinsic::amdgcn_wqm:
1267 return constrainCopyLikeIntrin(I, AMDGPU::WQM);
1268 case Intrinsic::amdgcn_softwqm:
1269 return constrainCopyLikeIntrin(I, AMDGPU::SOFT_WQM);
1270 case Intrinsic::amdgcn_strict_wwm:
1271 case Intrinsic::amdgcn_wwm:
1272 return constrainCopyLikeIntrin(I, AMDGPU::STRICT_WWM);
1273 case Intrinsic::amdgcn_strict_wqm:
1274 return constrainCopyLikeIntrin(I, AMDGPU::STRICT_WQM);
1275 case Intrinsic::amdgcn_writelane:
1276 return selectWritelane(I);
1277 case Intrinsic::amdgcn_div_scale:
1278 return selectDivScale(I);
1279 case Intrinsic::amdgcn_ballot:
1280 return selectBallot(I);
1281 case Intrinsic::amdgcn_reloc_constant:
1282 return selectRelocConstant(I);
1283 case Intrinsic::amdgcn_groupstaticsize:
1284 return selectGroupStaticSize(I);
1285 case Intrinsic::returnaddress:
1286 return selectReturnAddress(I);
1287 case Intrinsic::amdgcn_smfmac_f32_16x16x32_f16:
1288 case Intrinsic::amdgcn_smfmac_f32_32x32x16_f16:
1289 case Intrinsic::amdgcn_smfmac_f32_16x16x32_bf16:
1290 case Intrinsic::amdgcn_smfmac_f32_32x32x16_bf16:
1291 case Intrinsic::amdgcn_smfmac_i32_16x16x64_i8:
1292 case Intrinsic::amdgcn_smfmac_i32_32x32x32_i8:
1293 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_bf8:
1294 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_fp8:
1295 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_bf8:
1296 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_fp8:
1297 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_bf8:
1298 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_fp8:
1299 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_bf8:
1300 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_fp8:
1301 case Intrinsic::amdgcn_smfmac_f32_16x16x64_f16:
1302 case Intrinsic::amdgcn_smfmac_f32_32x32x32_f16:
1303 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf16:
1304 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf16:
1305 case Intrinsic::amdgcn_smfmac_i32_16x16x128_i8:
1306 case Intrinsic::amdgcn_smfmac_i32_32x32x64_i8:
1307 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_bf8:
1308 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_fp8:
1309 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_bf8:
1310 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_fp8:
1311 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_bf8:
1312 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_fp8:
1313 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_bf8:
1314 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_fp8:
1315 return selectSMFMACIntrin(I);
1316 case Intrinsic::amdgcn_permlane16_swap:
1317 case Intrinsic::amdgcn_permlane32_swap:
1318 return selectPermlaneSwapIntrin(I, IntrinsicID);
1319 case Intrinsic::amdgcn_wave_shuffle:
1320 return selectWaveShuffleIntrin(I);
1321 default:
1322 return selectImpl(I, *CoverageInfo);
1323 }
1324}
1325
1327 const GCNSubtarget &ST) {
1328 if (Size != 16 && Size != 32 && Size != 64)
1329 return -1;
1330
1331 if (Size == 16 && !ST.has16BitInsts())
1332 return -1;
1333
1334 const auto Select = [&](unsigned S16Opc, unsigned TrueS16Opc,
1335 unsigned FakeS16Opc, unsigned S32Opc,
1336 unsigned S64Opc) {
1337 if (Size == 16)
1338 return ST.hasTrue16BitInsts()
1339 ? ST.useRealTrue16Insts() ? TrueS16Opc : FakeS16Opc
1340 : S16Opc;
1341 if (Size == 32)
1342 return S32Opc;
1343 return S64Opc;
1344 };
1345
1346 switch (P) {
1347 default:
1348 llvm_unreachable("Unknown condition code!");
1349 case CmpInst::ICMP_NE:
1350 return Select(AMDGPU::V_CMP_NE_U16_e64, AMDGPU::V_CMP_NE_U16_t16_e64,
1351 AMDGPU::V_CMP_NE_U16_fake16_e64, AMDGPU::V_CMP_NE_U32_e64,
1352 AMDGPU::V_CMP_NE_U64_e64);
1353 case CmpInst::ICMP_EQ:
1354 return Select(AMDGPU::V_CMP_EQ_U16_e64, AMDGPU::V_CMP_EQ_U16_t16_e64,
1355 AMDGPU::V_CMP_EQ_U16_fake16_e64, AMDGPU::V_CMP_EQ_U32_e64,
1356 AMDGPU::V_CMP_EQ_U64_e64);
1357 case CmpInst::ICMP_SGT:
1358 return Select(AMDGPU::V_CMP_GT_I16_e64, AMDGPU::V_CMP_GT_I16_t16_e64,
1359 AMDGPU::V_CMP_GT_I16_fake16_e64, AMDGPU::V_CMP_GT_I32_e64,
1360 AMDGPU::V_CMP_GT_I64_e64);
1361 case CmpInst::ICMP_SGE:
1362 return Select(AMDGPU::V_CMP_GE_I16_e64, AMDGPU::V_CMP_GE_I16_t16_e64,
1363 AMDGPU::V_CMP_GE_I16_fake16_e64, AMDGPU::V_CMP_GE_I32_e64,
1364 AMDGPU::V_CMP_GE_I64_e64);
1365 case CmpInst::ICMP_SLT:
1366 return Select(AMDGPU::V_CMP_LT_I16_e64, AMDGPU::V_CMP_LT_I16_t16_e64,
1367 AMDGPU::V_CMP_LT_I16_fake16_e64, AMDGPU::V_CMP_LT_I32_e64,
1368 AMDGPU::V_CMP_LT_I64_e64);
1369 case CmpInst::ICMP_SLE:
1370 return Select(AMDGPU::V_CMP_LE_I16_e64, AMDGPU::V_CMP_LE_I16_t16_e64,
1371 AMDGPU::V_CMP_LE_I16_fake16_e64, AMDGPU::V_CMP_LE_I32_e64,
1372 AMDGPU::V_CMP_LE_I64_e64);
1373 case CmpInst::ICMP_UGT:
1374 return Select(AMDGPU::V_CMP_GT_U16_e64, AMDGPU::V_CMP_GT_U16_t16_e64,
1375 AMDGPU::V_CMP_GT_U16_fake16_e64, AMDGPU::V_CMP_GT_U32_e64,
1376 AMDGPU::V_CMP_GT_U64_e64);
1377 case CmpInst::ICMP_UGE:
1378 return Select(AMDGPU::V_CMP_GE_U16_e64, AMDGPU::V_CMP_GE_U16_t16_e64,
1379 AMDGPU::V_CMP_GE_U16_fake16_e64, AMDGPU::V_CMP_GE_U32_e64,
1380 AMDGPU::V_CMP_GE_U64_e64);
1381 case CmpInst::ICMP_ULT:
1382 return Select(AMDGPU::V_CMP_LT_U16_e64, AMDGPU::V_CMP_LT_U16_t16_e64,
1383 AMDGPU::V_CMP_LT_U16_fake16_e64, AMDGPU::V_CMP_LT_U32_e64,
1384 AMDGPU::V_CMP_LT_U64_e64);
1385 case CmpInst::ICMP_ULE:
1386 return Select(AMDGPU::V_CMP_LE_U16_e64, AMDGPU::V_CMP_LE_U16_t16_e64,
1387 AMDGPU::V_CMP_LE_U16_fake16_e64, AMDGPU::V_CMP_LE_U32_e64,
1388 AMDGPU::V_CMP_LE_U64_e64);
1389
1390 case CmpInst::FCMP_OEQ:
1391 return Select(AMDGPU::V_CMP_EQ_F16_e64, AMDGPU::V_CMP_EQ_F16_t16_e64,
1392 AMDGPU::V_CMP_EQ_F16_fake16_e64, AMDGPU::V_CMP_EQ_F32_e64,
1393 AMDGPU::V_CMP_EQ_F64_e64);
1394 case CmpInst::FCMP_OGT:
1395 return Select(AMDGPU::V_CMP_GT_F16_e64, AMDGPU::V_CMP_GT_F16_t16_e64,
1396 AMDGPU::V_CMP_GT_F16_fake16_e64, AMDGPU::V_CMP_GT_F32_e64,
1397 AMDGPU::V_CMP_GT_F64_e64);
1398 case CmpInst::FCMP_OGE:
1399 return Select(AMDGPU::V_CMP_GE_F16_e64, AMDGPU::V_CMP_GE_F16_t16_e64,
1400 AMDGPU::V_CMP_GE_F16_fake16_e64, AMDGPU::V_CMP_GE_F32_e64,
1401 AMDGPU::V_CMP_GE_F64_e64);
1402 case CmpInst::FCMP_OLT:
1403 return Select(AMDGPU::V_CMP_LT_F16_e64, AMDGPU::V_CMP_LT_F16_t16_e64,
1404 AMDGPU::V_CMP_LT_F16_fake16_e64, AMDGPU::V_CMP_LT_F32_e64,
1405 AMDGPU::V_CMP_LT_F64_e64);
1406 case CmpInst::FCMP_OLE:
1407 return Select(AMDGPU::V_CMP_LE_F16_e64, AMDGPU::V_CMP_LE_F16_t16_e64,
1408 AMDGPU::V_CMP_LE_F16_fake16_e64, AMDGPU::V_CMP_LE_F32_e64,
1409 AMDGPU::V_CMP_LE_F64_e64);
1410 case CmpInst::FCMP_ONE:
1411 return Select(AMDGPU::V_CMP_NEQ_F16_e64, AMDGPU::V_CMP_NEQ_F16_t16_e64,
1412 AMDGPU::V_CMP_NEQ_F16_fake16_e64, AMDGPU::V_CMP_NEQ_F32_e64,
1413 AMDGPU::V_CMP_NEQ_F64_e64);
1414 case CmpInst::FCMP_ORD:
1415 return Select(AMDGPU::V_CMP_O_F16_e64, AMDGPU::V_CMP_O_F16_t16_e64,
1416 AMDGPU::V_CMP_O_F16_fake16_e64, AMDGPU::V_CMP_O_F32_e64,
1417 AMDGPU::V_CMP_O_F64_e64);
1418 case CmpInst::FCMP_UNO:
1419 return Select(AMDGPU::V_CMP_U_F16_e64, AMDGPU::V_CMP_U_F16_t16_e64,
1420 AMDGPU::V_CMP_U_F16_fake16_e64, AMDGPU::V_CMP_U_F32_e64,
1421 AMDGPU::V_CMP_U_F64_e64);
1422 case CmpInst::FCMP_UEQ:
1423 return Select(AMDGPU::V_CMP_NLG_F16_e64, AMDGPU::V_CMP_NLG_F16_t16_e64,
1424 AMDGPU::V_CMP_NLG_F16_fake16_e64, AMDGPU::V_CMP_NLG_F32_e64,
1425 AMDGPU::V_CMP_NLG_F64_e64);
1426 case CmpInst::FCMP_UGT:
1427 return Select(AMDGPU::V_CMP_NLE_F16_e64, AMDGPU::V_CMP_NLE_F16_t16_e64,
1428 AMDGPU::V_CMP_NLE_F16_fake16_e64, AMDGPU::V_CMP_NLE_F32_e64,
1429 AMDGPU::V_CMP_NLE_F64_e64);
1430 case CmpInst::FCMP_UGE:
1431 return Select(AMDGPU::V_CMP_NLT_F16_e64, AMDGPU::V_CMP_NLT_F16_t16_e64,
1432 AMDGPU::V_CMP_NLT_F16_fake16_e64, AMDGPU::V_CMP_NLT_F32_e64,
1433 AMDGPU::V_CMP_NLT_F64_e64);
1434 case CmpInst::FCMP_ULT:
1435 return Select(AMDGPU::V_CMP_NGE_F16_e64, AMDGPU::V_CMP_NGE_F16_t16_e64,
1436 AMDGPU::V_CMP_NGE_F16_fake16_e64, AMDGPU::V_CMP_NGE_F32_e64,
1437 AMDGPU::V_CMP_NGE_F64_e64);
1438 case CmpInst::FCMP_ULE:
1439 return Select(AMDGPU::V_CMP_NGT_F16_e64, AMDGPU::V_CMP_NGT_F16_t16_e64,
1440 AMDGPU::V_CMP_NGT_F16_fake16_e64, AMDGPU::V_CMP_NGT_F32_e64,
1441 AMDGPU::V_CMP_NGT_F64_e64);
1442 case CmpInst::FCMP_UNE:
1443 return Select(AMDGPU::V_CMP_NEQ_F16_e64, AMDGPU::V_CMP_NEQ_F16_t16_e64,
1444 AMDGPU::V_CMP_NEQ_F16_fake16_e64, AMDGPU::V_CMP_NEQ_F32_e64,
1445 AMDGPU::V_CMP_NEQ_F64_e64);
1446 case CmpInst::FCMP_TRUE:
1447 return Select(AMDGPU::V_CMP_TRU_F16_e64, AMDGPU::V_CMP_TRU_F16_t16_e64,
1448 AMDGPU::V_CMP_TRU_F16_fake16_e64, AMDGPU::V_CMP_TRU_F32_e64,
1449 AMDGPU::V_CMP_TRU_F64_e64);
1451 return Select(AMDGPU::V_CMP_F_F16_e64, AMDGPU::V_CMP_F_F16_t16_e64,
1452 AMDGPU::V_CMP_F_F16_fake16_e64, AMDGPU::V_CMP_F_F32_e64,
1453 AMDGPU::V_CMP_F_F64_e64);
1454 }
1455}
1456
1457int AMDGPUInstructionSelector::getS_CMPOpcode(CmpInst::Predicate P,
1458 unsigned Size) const {
1459 if (Size == 64) {
1460 if (!STI.hasScalarCompareEq64())
1461 return -1;
1462
1463 switch (P) {
1464 case CmpInst::ICMP_NE:
1465 return AMDGPU::S_CMP_LG_U64;
1466 case CmpInst::ICMP_EQ:
1467 return AMDGPU::S_CMP_EQ_U64;
1468 default:
1469 return -1;
1470 }
1471 }
1472
1473 if (Size == 32) {
1474 switch (P) {
1475 case CmpInst::ICMP_NE:
1476 return AMDGPU::S_CMP_LG_U32;
1477 case CmpInst::ICMP_EQ:
1478 return AMDGPU::S_CMP_EQ_U32;
1479 case CmpInst::ICMP_SGT:
1480 return AMDGPU::S_CMP_GT_I32;
1481 case CmpInst::ICMP_SGE:
1482 return AMDGPU::S_CMP_GE_I32;
1483 case CmpInst::ICMP_SLT:
1484 return AMDGPU::S_CMP_LT_I32;
1485 case CmpInst::ICMP_SLE:
1486 return AMDGPU::S_CMP_LE_I32;
1487 case CmpInst::ICMP_UGT:
1488 return AMDGPU::S_CMP_GT_U32;
1489 case CmpInst::ICMP_UGE:
1490 return AMDGPU::S_CMP_GE_U32;
1491 case CmpInst::ICMP_ULT:
1492 return AMDGPU::S_CMP_LT_U32;
1493 case CmpInst::ICMP_ULE:
1494 return AMDGPU::S_CMP_LE_U32;
1495 case CmpInst::FCMP_OEQ:
1496 return AMDGPU::S_CMP_EQ_F32;
1497 case CmpInst::FCMP_OGT:
1498 return AMDGPU::S_CMP_GT_F32;
1499 case CmpInst::FCMP_OGE:
1500 return AMDGPU::S_CMP_GE_F32;
1501 case CmpInst::FCMP_OLT:
1502 return AMDGPU::S_CMP_LT_F32;
1503 case CmpInst::FCMP_OLE:
1504 return AMDGPU::S_CMP_LE_F32;
1505 case CmpInst::FCMP_ONE:
1506 return AMDGPU::S_CMP_LG_F32;
1507 case CmpInst::FCMP_ORD:
1508 return AMDGPU::S_CMP_O_F32;
1509 case CmpInst::FCMP_UNO:
1510 return AMDGPU::S_CMP_U_F32;
1511 case CmpInst::FCMP_UEQ:
1512 return AMDGPU::S_CMP_NLG_F32;
1513 case CmpInst::FCMP_UGT:
1514 return AMDGPU::S_CMP_NLE_F32;
1515 case CmpInst::FCMP_UGE:
1516 return AMDGPU::S_CMP_NLT_F32;
1517 case CmpInst::FCMP_ULT:
1518 return AMDGPU::S_CMP_NGE_F32;
1519 case CmpInst::FCMP_ULE:
1520 return AMDGPU::S_CMP_NGT_F32;
1521 case CmpInst::FCMP_UNE:
1522 return AMDGPU::S_CMP_NEQ_F32;
1523 default:
1524 llvm_unreachable("Unknown condition code!");
1525 }
1526 }
1527
1528 if (Size == 16) {
1529 if (!STI.hasSALUFloatInsts())
1530 return -1;
1531
1532 switch (P) {
1533 case CmpInst::FCMP_OEQ:
1534 return AMDGPU::S_CMP_EQ_F16;
1535 case CmpInst::FCMP_OGT:
1536 return AMDGPU::S_CMP_GT_F16;
1537 case CmpInst::FCMP_OGE:
1538 return AMDGPU::S_CMP_GE_F16;
1539 case CmpInst::FCMP_OLT:
1540 return AMDGPU::S_CMP_LT_F16;
1541 case CmpInst::FCMP_OLE:
1542 return AMDGPU::S_CMP_LE_F16;
1543 case CmpInst::FCMP_ONE:
1544 return AMDGPU::S_CMP_LG_F16;
1545 case CmpInst::FCMP_ORD:
1546 return AMDGPU::S_CMP_O_F16;
1547 case CmpInst::FCMP_UNO:
1548 return AMDGPU::S_CMP_U_F16;
1549 case CmpInst::FCMP_UEQ:
1550 return AMDGPU::S_CMP_NLG_F16;
1551 case CmpInst::FCMP_UGT:
1552 return AMDGPU::S_CMP_NLE_F16;
1553 case CmpInst::FCMP_UGE:
1554 return AMDGPU::S_CMP_NLT_F16;
1555 case CmpInst::FCMP_ULT:
1556 return AMDGPU::S_CMP_NGE_F16;
1557 case CmpInst::FCMP_ULE:
1558 return AMDGPU::S_CMP_NGT_F16;
1559 case CmpInst::FCMP_UNE:
1560 return AMDGPU::S_CMP_NEQ_F16;
1561 default:
1562 llvm_unreachable("Unknown condition code!");
1563 }
1564 }
1565
1566 return -1;
1567}
1568
1569bool AMDGPUInstructionSelector::selectG_ICMP_or_FCMP(MachineInstr &I) const {
1570
1571 MachineBasicBlock *BB = I.getParent();
1572 const DebugLoc &DL = I.getDebugLoc();
1573
1574 Register SrcReg = I.getOperand(2).getReg();
1575 unsigned Size = RBI.getSizeInBits(SrcReg, *MRI, TRI);
1576
1577 auto Pred = (CmpInst::Predicate)I.getOperand(1).getPredicate();
1578
1579 Register CCReg = I.getOperand(0).getReg();
1580 if (!isVCC(CCReg, *MRI)) {
1581 int Opcode = getS_CMPOpcode(Pred, Size);
1582 if (Opcode == -1)
1583 return false;
1584 MachineInstr *ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode))
1585 .add(I.getOperand(2))
1586 .add(I.getOperand(3));
1587 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), CCReg)
1588 .addReg(AMDGPU::SCC);
1589 constrainSelectedInstRegOperands(*ICmp, TII, TRI, RBI);
1590 bool Ret =
1591 RBI.constrainGenericRegister(CCReg, AMDGPU::SReg_32RegClass, *MRI);
1592 I.eraseFromParent();
1593 return Ret;
1594 }
1595
1596 if (I.getOpcode() == AMDGPU::G_FCMP)
1597 return false;
1598
1599 int Opcode = getV_CMPOpcode(Pred, Size, *Subtarget);
1600 if (Opcode == -1)
1601 return false;
1602
1603 MachineInstrBuilder ICmp;
1604 // t16 instructions
1605 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src0_modifiers)) {
1606 ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode), I.getOperand(0).getReg())
1607 .addImm(0)
1608 .add(I.getOperand(2))
1609 .addImm(0)
1610 .add(I.getOperand(3))
1611 .addImm(0); // op_sel
1612 } else {
1613 ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode), I.getOperand(0).getReg())
1614 .add(I.getOperand(2))
1615 .add(I.getOperand(3));
1616 }
1617
1618 RBI.constrainGenericRegister(ICmp->getOperand(0).getReg(),
1619 *TRI.getBoolRC(), *MRI);
1620 constrainSelectedInstRegOperands(*ICmp, TII, TRI, RBI);
1621 I.eraseFromParent();
1622 return true;
1623}
1624
1625// Ballot has to zero bits in input lane-mask that are zero in current exec,
1626// Done as AND with exec. For inputs that are results of instruction that
1627// implicitly use same exec, for example compares in same basic block or SCC to
1628// VCC copy, use copy.
1631 MachineInstr *MI = MRI.getVRegDef(Reg);
1632 if (MI->getParent() != MBB)
1633 return false;
1634
1635 // Lane mask generated by SCC to VCC copy.
1636 if (MI->getOpcode() == AMDGPU::COPY) {
1637 auto DstRB = MRI.getRegBankOrNull(MI->getOperand(0).getReg());
1638 auto SrcRB = MRI.getRegBankOrNull(MI->getOperand(1).getReg());
1639 if (DstRB && SrcRB && DstRB->getID() == AMDGPU::VCCRegBankID &&
1640 SrcRB->getID() == AMDGPU::SGPRRegBankID)
1641 return true;
1642 }
1643
1644 // Lane mask generated by SCC to VCC copy
1645 if (MI->getOpcode() == AMDGPU::G_AMDGPU_COPY_VCC_SCC)
1646 return true;
1647
1648 // Lane mask generated using compare with same exec.
1649 if (isa<GAnyCmp>(MI))
1650 return true;
1651
1652 Register LHS, RHS;
1653 // Look through AND.
1654 if (mi_match(Reg, MRI, m_GAnd(m_Reg(LHS), m_Reg(RHS))))
1655 return isLaneMaskFromSameBlock(LHS, MRI, MBB) ||
1657
1658 return false;
1659}
1660
1661bool AMDGPUInstructionSelector::selectBallot(MachineInstr &I) const {
1662 MachineBasicBlock *BB = I.getParent();
1663 const DebugLoc &DL = I.getDebugLoc();
1664 Register DstReg = I.getOperand(0).getReg();
1665 Register SrcReg = I.getOperand(2).getReg();
1666 const unsigned BallotSize = MRI->getType(DstReg).getSizeInBits();
1667 const unsigned WaveSize = STI.getWavefrontSize();
1668
1669 // In the common case, the return type matches the wave size.
1670 // However we also support emitting i64 ballots in wave32 mode.
1671 if (BallotSize != WaveSize && (BallotSize != 64 || WaveSize != 32))
1672 return false;
1673
1674 std::optional<ValueAndVReg> Arg =
1676
1677 Register Dst = DstReg;
1678 // i64 ballot on Wave32: new Dst(i32) for WaveSize ballot.
1679 if (BallotSize != WaveSize) {
1680 Dst = MRI->createVirtualRegister(TRI.getBoolRC());
1681 }
1682
1683 if (Arg) {
1684 const int64_t Value = Arg->Value.getZExtValue();
1685 if (Value == 0) {
1686 // Dst = S_MOV 0
1687 unsigned Opcode = WaveSize == 64 ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
1688 BuildMI(*BB, &I, DL, TII.get(Opcode), Dst).addImm(0);
1689 } else {
1690 // Dst = COPY EXEC
1691 assert(Value == 1);
1692 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst).addReg(TRI.getExec());
1693 }
1694 if (!RBI.constrainGenericRegister(Dst, *TRI.getBoolRC(), *MRI))
1695 return false;
1696 } else {
1697 if (isLaneMaskFromSameBlock(SrcReg, *MRI, BB)) {
1698 // Dst = COPY SrcReg
1699 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst).addReg(SrcReg);
1700 if (!RBI.constrainGenericRegister(Dst, *TRI.getBoolRC(), *MRI))
1701 return false;
1702 } else {
1703 // Dst = S_AND SrcReg, EXEC
1704 unsigned AndOpc = WaveSize == 64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
1705 auto And = BuildMI(*BB, &I, DL, TII.get(AndOpc), Dst)
1706 .addReg(SrcReg)
1707 .addReg(TRI.getExec())
1708 .setOperandDead(3); // Dead scc
1709 constrainSelectedInstRegOperands(*And, TII, TRI, RBI);
1710 }
1711 }
1712
1713 // i64 ballot on Wave32: zero-extend i32 ballot to i64.
1714 if (BallotSize != WaveSize) {
1715 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1716 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_MOV_B32), HiReg).addImm(0);
1717 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
1718 .addReg(Dst)
1719 .addImm(AMDGPU::sub0)
1720 .addReg(HiReg)
1721 .addImm(AMDGPU::sub1);
1722 }
1723
1724 I.eraseFromParent();
1725 return true;
1726}
1727
1728bool AMDGPUInstructionSelector::selectRelocConstant(MachineInstr &I) const {
1729 Register DstReg = I.getOperand(0).getReg();
1730 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
1731 const TargetRegisterClass *DstRC = TRI.getRegClassForSizeOnBank(32, *DstBank);
1732 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
1733 return false;
1734
1735 const bool IsVALU = DstBank->getID() == AMDGPU::VGPRRegBankID;
1736
1737 Module *M = MF->getFunction().getParent();
1738 const MDNode *Metadata = I.getOperand(2).getMetadata();
1739 auto SymbolName = cast<MDString>(Metadata->getOperand(0))->getString();
1740 auto *RelocSymbol = cast<GlobalVariable>(
1741 M->getOrInsertGlobal(SymbolName, Type::getInt32Ty(M->getContext())));
1742
1743 MachineBasicBlock *BB = I.getParent();
1744 BuildMI(*BB, &I, I.getDebugLoc(),
1745 TII.get(IsVALU ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32), DstReg)
1747
1748 I.eraseFromParent();
1749 return true;
1750}
1751
1752bool AMDGPUInstructionSelector::selectGroupStaticSize(MachineInstr &I) const {
1753 Triple::OSType OS = MF->getTarget().getTargetTriple().getOS();
1754
1755 Register DstReg = I.getOperand(0).getReg();
1756 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
1757 unsigned Mov = DstRB->getID() == AMDGPU::SGPRRegBankID ?
1758 AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1759
1760 MachineBasicBlock *MBB = I.getParent();
1761 const DebugLoc &DL = I.getDebugLoc();
1762
1763 auto MIB = BuildMI(*MBB, &I, DL, TII.get(Mov), DstReg);
1764
1765 if (OS == Triple::AMDHSA || OS == Triple::AMDPAL) {
1766 const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
1767 MIB.addImm(MFI->getLDSSize());
1768 } else {
1769 Module *M = MF->getFunction().getParent();
1770 const GlobalValue *GV =
1771 Intrinsic::getOrInsertDeclaration(M, Intrinsic::amdgcn_groupstaticsize);
1773 }
1774
1775 I.eraseFromParent();
1776 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1777 return true;
1778}
1779
1780bool AMDGPUInstructionSelector::selectReturnAddress(MachineInstr &I) const {
1781 MachineBasicBlock *MBB = I.getParent();
1783 const DebugLoc &DL = I.getDebugLoc();
1784
1785 Register DstReg = I.getOperand(0).getReg();
1786 unsigned Depth = I.getOperand(2).getImm();
1787
1788 const TargetRegisterClass *RC =
1789 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
1790 if (!RC->hasSubClassEq(&AMDGPU::SGPR_64RegClass) ||
1791 !RBI.constrainGenericRegister(DstReg, *RC, *MRI))
1792 return false;
1793
1794 // Check for kernel and shader functions
1795 if (Depth != 0 ||
1796 MF.getInfo<SIMachineFunctionInfo>()->isEntryFunction()) {
1797 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_MOV_B64), DstReg)
1798 .addImm(0);
1799 I.eraseFromParent();
1800 return true;
1801 }
1802
1803 MachineFrameInfo &MFI = MF.getFrameInfo();
1804 // There is a call to @llvm.returnaddress in this function
1805 MFI.setReturnAddressIsTaken(true);
1806
1807 // Get the return address reg and mark it as an implicit live-in
1808 Register ReturnAddrReg = TRI.getReturnAddressReg(MF);
1809 Register LiveIn = getFunctionLiveInPhysReg(MF, TII, ReturnAddrReg,
1810 AMDGPU::SReg_64RegClass, DL);
1811 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), DstReg)
1812 .addReg(LiveIn);
1813 I.eraseFromParent();
1814 return true;
1815}
1816
1817bool AMDGPUInstructionSelector::selectEndCfIntrinsic(MachineInstr &MI) const {
1818 // FIXME: Manually selecting to avoid dealing with the SReg_1 trick
1819 // SelectionDAG uses for wave32 vs wave64.
1820 MachineBasicBlock *BB = MI.getParent();
1821 BuildMI(*BB, &MI, MI.getDebugLoc(), TII.get(AMDGPU::SI_END_CF))
1822 .add(MI.getOperand(1))
1823 .setOperandDead(2); // implicit-def $scc
1824
1825 Register Reg = MI.getOperand(1).getReg();
1826 MI.eraseFromParent();
1827
1828 if (!MRI->getRegClassOrNull(Reg))
1829 MRI->setRegClass(Reg, TRI.getWaveMaskRegClass());
1830 return true;
1831}
1832
1833bool AMDGPUInstructionSelector::selectDSOrderedIntrinsic(
1834 MachineInstr &MI, Intrinsic::ID IntrID) const {
1835 MachineBasicBlock *MBB = MI.getParent();
1837 const DebugLoc &DL = MI.getDebugLoc();
1838
1839 unsigned IndexOperand = MI.getOperand(7).getImm();
1840 bool WaveRelease = MI.getOperand(8).getImm() != 0;
1841 bool WaveDone = MI.getOperand(9).getImm() != 0;
1842
1843 if (WaveDone && !WaveRelease) {
1844 // TODO: Move this to IR verifier
1845 const Function &Fn = MF->getFunction();
1846 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1847 Fn, "ds_ordered_count: wave_done requires wave_release", DL));
1848 }
1849
1850 unsigned OrderedCountIndex = IndexOperand & 0x3f;
1851 IndexOperand &= ~0x3f;
1852 unsigned CountDw = 0;
1853
1854 if (STI.getGeneration() >= AMDGPUSubtarget::GFX10) {
1855 CountDw = (IndexOperand >> 24) & 0xf;
1856 IndexOperand &= ~(0xf << 24);
1857
1858 if (CountDw < 1 || CountDw > 4) {
1859 const Function &Fn = MF->getFunction();
1860 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1861 Fn, "ds_ordered_count: dword count must be between 1 and 4", DL));
1862 CountDw = 1;
1863 }
1864 }
1865
1866 if (IndexOperand) {
1867 const Function &Fn = MF->getFunction();
1868 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1869 Fn, "ds_ordered_count: bad index operand", DL));
1870 }
1871
1872 unsigned Instruction = IntrID == Intrinsic::amdgcn_ds_ordered_add ? 0 : 1;
1873 unsigned ShaderType = SIInstrInfo::getDSShaderTypeValue(*MF);
1874
1875 unsigned Offset0 = OrderedCountIndex << 2;
1876 unsigned Offset1 = WaveRelease | (WaveDone << 1) | (Instruction << 4);
1877
1878 if (STI.getGeneration() >= AMDGPUSubtarget::GFX10)
1879 Offset1 |= (CountDw - 1) << 6;
1880
1881 if (STI.getGeneration() < AMDGPUSubtarget::GFX11)
1882 Offset1 |= ShaderType << 2;
1883
1884 unsigned Offset = Offset0 | (Offset1 << 8);
1885
1886 Register M0Val = MI.getOperand(2).getReg();
1887 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1888 .addReg(M0Val);
1889
1890 Register DstReg = MI.getOperand(0).getReg();
1891 Register ValReg = MI.getOperand(3).getReg();
1892 MachineInstrBuilder DS =
1893 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::DS_ORDERED_COUNT), DstReg)
1894 .addReg(ValReg)
1895 .addImm(Offset)
1896 .cloneMemRefs(MI);
1897
1898 if (!RBI.constrainGenericRegister(M0Val, AMDGPU::SReg_32RegClass, *MRI))
1899 return false;
1900
1901 constrainSelectedInstRegOperands(*DS, TII, TRI, RBI);
1902 MI.eraseFromParent();
1903 return true;
1904}
1905
1906static unsigned gwsIntrinToOpcode(unsigned IntrID) {
1907 switch (IntrID) {
1908 case Intrinsic::amdgcn_ds_gws_init:
1909 return AMDGPU::DS_GWS_INIT;
1910 case Intrinsic::amdgcn_ds_gws_barrier:
1911 return AMDGPU::DS_GWS_BARRIER;
1912 case Intrinsic::amdgcn_ds_gws_sema_v:
1913 return AMDGPU::DS_GWS_SEMA_V;
1914 case Intrinsic::amdgcn_ds_gws_sema_br:
1915 return AMDGPU::DS_GWS_SEMA_BR;
1916 case Intrinsic::amdgcn_ds_gws_sema_p:
1917 return AMDGPU::DS_GWS_SEMA_P;
1918 case Intrinsic::amdgcn_ds_gws_sema_release_all:
1919 return AMDGPU::DS_GWS_SEMA_RELEASE_ALL;
1920 default:
1921 llvm_unreachable("not a gws intrinsic");
1922 }
1923}
1924
1925bool AMDGPUInstructionSelector::selectDSGWSIntrinsic(MachineInstr &MI,
1926 Intrinsic::ID IID) const {
1927 if (!STI.hasGWS() || (IID == Intrinsic::amdgcn_ds_gws_sema_release_all &&
1928 !STI.hasGWSSemaReleaseAll()))
1929 return false;
1930
1931 // intrinsic ID, vsrc, offset
1932 const bool HasVSrc = MI.getNumOperands() == 3;
1933 assert(HasVSrc || MI.getNumOperands() == 2);
1934
1935 Register BaseOffset = MI.getOperand(HasVSrc ? 2 : 1).getReg();
1936 const RegisterBank *OffsetRB = RBI.getRegBank(BaseOffset, *MRI, TRI);
1937 if (OffsetRB->getID() != AMDGPU::SGPRRegBankID)
1938 return false;
1939
1940 MachineInstr *OffsetDef = getDefIgnoringCopies(BaseOffset, *MRI);
1941 unsigned ImmOffset;
1942
1943 MachineBasicBlock *MBB = MI.getParent();
1944 const DebugLoc &DL = MI.getDebugLoc();
1945
1946 MachineInstr *Readfirstlane = nullptr;
1947
1948 // If we legalized the VGPR input, strip out the readfirstlane to analyze the
1949 // incoming offset, in case there's an add of a constant. We'll have to put it
1950 // back later.
1951 if (OffsetDef->getOpcode() == AMDGPU::V_READFIRSTLANE_B32) {
1952 Readfirstlane = OffsetDef;
1953 BaseOffset = OffsetDef->getOperand(1).getReg();
1954 OffsetDef = getDefIgnoringCopies(BaseOffset, *MRI);
1955 }
1956
1957 if (OffsetDef->getOpcode() == AMDGPU::G_CONSTANT) {
1958 // If we have a constant offset, try to use the 0 in m0 as the base.
1959 // TODO: Look into changing the default m0 initialization value. If the
1960 // default -1 only set the low 16-bits, we could leave it as-is and add 1 to
1961 // the immediate offset.
1962
1963 ImmOffset = OffsetDef->getOperand(1).getCImm()->getZExtValue();
1964 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), AMDGPU::M0)
1965 .addImm(0);
1966 } else {
1967 std::tie(BaseOffset, ImmOffset) =
1968 AMDGPU::getBaseWithConstantOffset(*MRI, BaseOffset, VT);
1969
1970 if (Readfirstlane) {
1971 // We have the constant offset now, so put the readfirstlane back on the
1972 // variable component.
1973 if (!RBI.constrainGenericRegister(BaseOffset, AMDGPU::VGPR_32RegClass, *MRI))
1974 return false;
1975
1976 Readfirstlane->getOperand(1).setReg(BaseOffset);
1977 BaseOffset = Readfirstlane->getOperand(0).getReg();
1978 } else {
1979 if (!RBI.constrainGenericRegister(BaseOffset,
1980 AMDGPU::SReg_32RegClass, *MRI))
1981 return false;
1982 }
1983
1984 Register M0Base = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1985 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::S_LSHL_B32), M0Base)
1986 .addReg(BaseOffset)
1987 .addImm(16)
1988 .setOperandDead(3); // Dead scc
1989
1990 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1991 .addReg(M0Base);
1992 }
1993
1994 // The resource id offset is computed as (<isa opaque base> + M0[21:16] +
1995 // offset field) % 64. Some versions of the programming guide omit the m0
1996 // part, or claim it's from offset 0.
1997
1998 unsigned Opc = gwsIntrinToOpcode(IID);
1999 const MCInstrDesc &InstrDesc = TII.get(Opc);
2000
2001 if (HasVSrc) {
2002 Register VSrc = MI.getOperand(1).getReg();
2003
2004 int Data0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
2005 const TargetRegisterClass *DataRC = TII.getRegClass(InstrDesc, Data0Idx);
2006 const TargetRegisterClass *SubRC =
2007 TRI.getSubRegisterClass(DataRC, AMDGPU::sub0);
2008
2009 if (!SubRC) {
2010 // 32-bit normal case.
2011 if (!RBI.constrainGenericRegister(VSrc, *DataRC, *MRI))
2012 return false;
2013
2014 BuildMI(*MBB, &MI, DL, InstrDesc)
2015 .addReg(VSrc)
2016 .addImm(ImmOffset)
2017 .cloneMemRefs(MI);
2018 } else {
2019 // Requires even register alignment, so create 64-bit value and pad the
2020 // top half with undef.
2021 Register DataReg = MRI->createVirtualRegister(DataRC);
2022 if (!RBI.constrainGenericRegister(VSrc, *SubRC, *MRI))
2023 return false;
2024
2025 Register UndefReg = MRI->createVirtualRegister(SubRC);
2026 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2027 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), DataReg)
2028 .addReg(VSrc)
2029 .addImm(AMDGPU::sub0)
2030 .addReg(UndefReg)
2031 .addImm(AMDGPU::sub1);
2032
2033 BuildMI(*MBB, &MI, DL, InstrDesc)
2034 .addReg(DataReg)
2035 .addImm(ImmOffset)
2036 .cloneMemRefs(MI);
2037 }
2038 } else {
2039 BuildMI(*MBB, &MI, DL, InstrDesc)
2040 .addImm(ImmOffset)
2041 .cloneMemRefs(MI);
2042 }
2043
2044 MI.eraseFromParent();
2045 return true;
2046}
2047
2048bool AMDGPUInstructionSelector::selectDSAppendConsume(MachineInstr &MI,
2049 bool IsAppend) const {
2050 Register PtrBase = MI.getOperand(2).getReg();
2051 LLT PtrTy = MRI->getType(PtrBase);
2052 bool IsGDS = PtrTy.getAddressSpace() == AMDGPUAS::REGION_ADDRESS;
2053
2054 unsigned Offset;
2055 std::tie(PtrBase, Offset) = selectDS1Addr1OffsetImpl(MI.getOperand(2));
2056
2057 // TODO: Should this try to look through readfirstlane like GWS?
2058 if (!isDSOffsetLegal(PtrBase, Offset)) {
2059 PtrBase = MI.getOperand(2).getReg();
2060 Offset = 0;
2061 }
2062
2063 MachineBasicBlock *MBB = MI.getParent();
2064 const DebugLoc &DL = MI.getDebugLoc();
2065 const unsigned Opc = IsAppend ? AMDGPU::DS_APPEND : AMDGPU::DS_CONSUME;
2066
2067 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
2068 .addReg(PtrBase);
2069 if (!RBI.constrainGenericRegister(PtrBase, AMDGPU::SReg_32RegClass, *MRI))
2070 return false;
2071
2072 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), MI.getOperand(0).getReg())
2073 .addImm(Offset)
2074 .addImm(IsGDS ? -1 : 0)
2075 .cloneMemRefs(MI);
2076 MI.eraseFromParent();
2077 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2078 return true;
2079}
2080
2081bool AMDGPUInstructionSelector::selectInitWholeWave(MachineInstr &MI) const {
2082 MachineFunction *MF = MI.getMF();
2083 SIMachineFunctionInfo *MFInfo = MF->getInfo<SIMachineFunctionInfo>();
2084
2085 MFInfo->setInitWholeWave();
2086 return selectImpl(MI, *CoverageInfo);
2087}
2088
2089static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE,
2090 bool &IsTexFail) {
2091 if (TexFailCtrl)
2092 IsTexFail = true;
2093
2094 TFE = TexFailCtrl & 0x1;
2095 TexFailCtrl &= ~(uint64_t)0x1;
2096 LWE = TexFailCtrl & 0x2;
2097 TexFailCtrl &= ~(uint64_t)0x2;
2098
2099 return TexFailCtrl == 0;
2100}
2101
2102bool AMDGPUInstructionSelector::selectImageIntrinsic(
2103 MachineInstr &MI, const AMDGPU::ImageDimIntrinsicInfo *Intr) const {
2104 MachineBasicBlock *MBB = MI.getParent();
2105 const DebugLoc &DL = MI.getDebugLoc();
2106 unsigned IntrOpcode = Intr->BaseOpcode;
2107
2108 // For image atomic: use no-return opcode if result is unused.
2109 if (Intr->AtomicNoRetBaseOpcode != Intr->BaseOpcode) {
2110 Register ResultDef = MI.getOperand(0).getReg();
2111 if (MRI->use_nodbg_empty(ResultDef))
2112 IntrOpcode = Intr->AtomicNoRetBaseOpcode;
2113 }
2114
2115 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
2117
2118 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfo(Intr->Dim);
2119 const bool IsGFX10Plus = AMDGPU::isGFX10Plus(STI);
2120 const bool IsGFX11Plus = AMDGPU::isGFX11Plus(STI);
2121 const bool IsGFX12Plus = AMDGPU::isGFX12Plus(STI);
2122 const bool IsGFX13Plus = AMDGPU::isGFX13Plus(STI);
2123
2124 const unsigned ArgOffset = MI.getNumExplicitDefs() + 1;
2125
2126 Register VDataIn = AMDGPU::NoRegister;
2127 Register VDataOut = AMDGPU::NoRegister;
2128 LLT VDataTy;
2129 int NumVDataDwords = -1;
2130 bool IsD16 = MI.getOpcode() == AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16 ||
2131 MI.getOpcode() == AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16;
2132
2133 bool Unorm;
2134 if (!BaseOpcode->Sampler)
2135 Unorm = true;
2136 else
2137 Unorm = MI.getOperand(ArgOffset + Intr->UnormIndex).getImm() != 0;
2138
2139 bool TFE;
2140 bool LWE;
2141 bool IsTexFail = false;
2142 if (!parseTexFail(MI.getOperand(ArgOffset + Intr->TexFailCtrlIndex).getImm(),
2143 TFE, LWE, IsTexFail))
2144 return false;
2145
2146 const int Flags = MI.getOperand(ArgOffset + Intr->NumArgs).getImm();
2147 const bool IsA16 = (Flags & 1) != 0;
2148 const bool IsG16 = (Flags & 2) != 0;
2149
2150 // A16 implies 16 bit gradients if subtarget doesn't support G16
2151 if (IsA16 && !STI.hasG16() && !IsG16)
2152 return false;
2153
2154 unsigned DMask = 0;
2155 unsigned DMaskLanes = 0;
2156
2157 if (BaseOpcode->Atomic) {
2158 if (!BaseOpcode->NoReturn)
2159 VDataOut = MI.getOperand(0).getReg();
2160 VDataIn = MI.getOperand(2).getReg();
2161 LLT Ty = MRI->getType(VDataIn);
2162
2163 // Be careful to allow atomic swap on 16-bit element vectors.
2164 const bool Is64Bit = BaseOpcode->AtomicX2 ?
2165 Ty.getSizeInBits() == 128 :
2166 Ty.getSizeInBits() == 64;
2167
2168 if (BaseOpcode->AtomicX2) {
2169 assert(MI.getOperand(3).getReg() == AMDGPU::NoRegister);
2170
2171 DMask = Is64Bit ? 0xf : 0x3;
2172 NumVDataDwords = Is64Bit ? 4 : 2;
2173 } else {
2174 DMask = Is64Bit ? 0x3 : 0x1;
2175 NumVDataDwords = Is64Bit ? 2 : 1;
2176 }
2177 } else {
2178 DMask = MI.getOperand(ArgOffset + Intr->DMaskIndex).getImm();
2179 DMaskLanes = BaseOpcode->Gather4 ? 4 : llvm::popcount(DMask);
2180
2181 if (BaseOpcode->Store) {
2182 VDataIn = MI.getOperand(1).getReg();
2183 VDataTy = MRI->getType(VDataIn);
2184 NumVDataDwords = (VDataTy.getSizeInBits() + 31) / 32;
2185 } else if (BaseOpcode->NoReturn) {
2186 NumVDataDwords = 0;
2187 } else {
2188 VDataOut = MI.getOperand(0).getReg();
2189 VDataTy = MRI->getType(VDataOut);
2190 NumVDataDwords = DMaskLanes;
2191
2192 if (IsD16 && !STI.hasUnpackedD16VMem())
2193 NumVDataDwords = (DMaskLanes + 1) / 2;
2194 }
2195 }
2196
2197 // Set G16 opcode
2198 if (Subtarget->hasG16() && IsG16) {
2199 const AMDGPU::MIMGG16MappingInfo *G16MappingInfo =
2201 assert(G16MappingInfo);
2202 IntrOpcode = G16MappingInfo->G16; // set opcode to variant with _g16
2203 }
2204
2205 // TODO: Check this in verifier.
2206 assert((!IsTexFail || DMaskLanes >= 1) && "should have legalized this");
2207
2208 unsigned CPol = MI.getOperand(ArgOffset + Intr->CachePolicyIndex).getImm();
2209 // Keep GLC only when the atomic's result is actually used.
2210 if (BaseOpcode->Atomic && !BaseOpcode->NoReturn)
2212 if (CPol & ~((IsGFX12Plus ? AMDGPU::CPol::ALL : AMDGPU::CPol::ALL_pregfx12) |
2214 return false;
2215
2216 int NumVAddrRegs = 0;
2217 int NumVAddrDwords = 0;
2218 for (unsigned I = Intr->VAddrStart; I < Intr->VAddrEnd; I++) {
2219 // Skip the $noregs and 0s inserted during legalization.
2220 MachineOperand &AddrOp = MI.getOperand(ArgOffset + I);
2221 if (!AddrOp.isReg())
2222 continue; // XXX - Break?
2223
2224 Register Addr = AddrOp.getReg();
2225 if (!Addr)
2226 break;
2227
2228 ++NumVAddrRegs;
2229 NumVAddrDwords += (MRI->getType(Addr).getSizeInBits() + 31) / 32;
2230 }
2231
2232 // The legalizer preprocessed the intrinsic arguments. If we aren't using
2233 // NSA, these should have been packed into a single value in the first
2234 // address register
2235 const bool UseNSA =
2236 NumVAddrRegs != 1 &&
2237 (STI.hasPartialNSAEncoding() ? NumVAddrDwords >= NumVAddrRegs
2238 : NumVAddrDwords == NumVAddrRegs);
2239 if (UseNSA && !STI.hasFeature(AMDGPU::FeatureNSAEncoding)) {
2240 LLVM_DEBUG(dbgs() << "Trying to use NSA on non-NSA target\n");
2241 return false;
2242 }
2243
2244 if (IsTexFail)
2245 ++NumVDataDwords;
2246
2247 int Opcode = -1;
2248 if (IsGFX13Plus) {
2249 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx13,
2250 NumVDataDwords, NumVAddrDwords);
2251 } else if (IsGFX12Plus) {
2252 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx12,
2253 NumVDataDwords, NumVAddrDwords);
2254 } else if (IsGFX11Plus) {
2255 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
2256 UseNSA ? AMDGPU::MIMGEncGfx11NSA
2257 : AMDGPU::MIMGEncGfx11Default,
2258 NumVDataDwords, NumVAddrDwords);
2259 } else if (IsGFX10Plus) {
2260 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
2261 UseNSA ? AMDGPU::MIMGEncGfx10NSA
2262 : AMDGPU::MIMGEncGfx10Default,
2263 NumVDataDwords, NumVAddrDwords);
2264 } else {
2265 if (Subtarget->hasGFX90AInsts()) {
2266 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx90a,
2267 NumVDataDwords, NumVAddrDwords);
2268 if (Opcode == -1) {
2269 LLVM_DEBUG(
2270 dbgs()
2271 << "requested image instruction is not supported on this GPU\n");
2272 return false;
2273 }
2274 }
2275 if (Opcode == -1 &&
2276 STI.getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
2277 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx8,
2278 NumVDataDwords, NumVAddrDwords);
2279 if (Opcode == -1)
2280 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx6,
2281 NumVDataDwords, NumVAddrDwords);
2282 }
2283 if (Opcode == -1)
2284 return false;
2285
2286 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opcode))
2287 .cloneMemRefs(MI);
2288
2289 if (VDataOut) {
2290 if (BaseOpcode->AtomicX2) {
2291 const bool Is64 = MRI->getType(VDataOut).getSizeInBits() == 64;
2292
2293 Register TmpReg = MRI->createVirtualRegister(
2294 Is64 ? &AMDGPU::VReg_128RegClass : &AMDGPU::VReg_64RegClass);
2295 unsigned SubReg = Is64 ? AMDGPU::sub0_sub1 : AMDGPU::sub0;
2296
2297 MIB.addDef(TmpReg);
2298 if (!MRI->use_empty(VDataOut)) {
2299 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), VDataOut)
2300 .addReg(TmpReg, RegState::Kill, SubReg);
2301 }
2302
2303 } else {
2304 MIB.addDef(VDataOut); // vdata output
2305 }
2306 }
2307
2308 if (VDataIn)
2309 MIB.addReg(VDataIn); // vdata input
2310
2311 for (int I = 0; I != NumVAddrRegs; ++I) {
2312 MachineOperand &SrcOp = MI.getOperand(ArgOffset + Intr->VAddrStart + I);
2313 if (SrcOp.isReg()) {
2314 assert(SrcOp.getReg() != 0);
2315 MIB.addReg(SrcOp.getReg());
2316 }
2317 }
2318
2319 MIB.addReg(MI.getOperand(ArgOffset + Intr->RsrcIndex).getReg());
2320 if (BaseOpcode->Sampler)
2321 MIB.addReg(MI.getOperand(ArgOffset + Intr->SampIndex).getReg());
2322
2323 MIB.addImm(DMask); // dmask
2324
2325 if (IsGFX10Plus)
2326 MIB.addImm(DimInfo->Encoding);
2327 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::unorm))
2328 MIB.addImm(Unorm);
2329
2330 MIB.addImm(CPol);
2331 MIB.addImm(IsA16 && // a16 or r128
2332 STI.hasFeature(AMDGPU::FeatureR128A16) ? -1 : 0);
2333 if (IsGFX10Plus)
2334 MIB.addImm(IsA16 ? -1 : 0);
2335
2336 if (!Subtarget->hasGFX90AInsts()) {
2337 MIB.addImm(TFE); // tfe
2338 } else if (TFE) {
2339 LLVM_DEBUG(dbgs() << "TFE is not supported on this GPU\n");
2340 return false;
2341 }
2342
2343 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::lwe))
2344 MIB.addImm(LWE); // lwe
2345 if (!IsGFX10Plus)
2346 MIB.addImm(DimInfo->DA ? -1 : 0);
2347 if (BaseOpcode->HasD16)
2348 MIB.addImm(IsD16 ? -1 : 0);
2349
2350 MI.eraseFromParent();
2351 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2352 TII.enforceOperandRCAlignment(*MIB, AMDGPU::OpName::vaddr);
2353 return true;
2354}
2355
2356// We need to handle this here because tablegen doesn't support matching
2357// instructions with multiple outputs.
2358bool AMDGPUInstructionSelector::selectDSBvhStackIntrinsic(
2359 MachineInstr &MI) const {
2360 Register Dst0 = MI.getOperand(0).getReg();
2361 Register Dst1 = MI.getOperand(1).getReg();
2362
2363 const DebugLoc &DL = MI.getDebugLoc();
2364 MachineBasicBlock *MBB = MI.getParent();
2365
2366 Register Addr = MI.getOperand(3).getReg();
2367 Register Data0 = MI.getOperand(4).getReg();
2368 Register Data1 = MI.getOperand(5).getReg();
2369 unsigned Offset = MI.getOperand(6).getImm();
2370
2371 unsigned Opc;
2372 switch (cast<GIntrinsic>(MI).getIntrinsicID()) {
2373 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
2374 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
2375 Opc = AMDGPU::DS_BVH_STACK_RTN_B32;
2376 break;
2377 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
2378 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP1_RTN_B32;
2379 break;
2380 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
2381 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP2_RTN_B64;
2382 break;
2383 }
2384
2385 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), Dst0)
2386 .addDef(Dst1)
2387 .addUse(Addr)
2388 .addUse(Data0)
2389 .addUse(Data1)
2390 .addImm(Offset)
2391 .cloneMemRefs(MI);
2392
2393 MI.eraseFromParent();
2394 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2395 return true;
2396}
2397
2398bool AMDGPUInstructionSelector::selectG_INTRINSIC_W_SIDE_EFFECTS(
2399 MachineInstr &I) const {
2400 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(I).getIntrinsicID();
2401 switch (IntrinsicID) {
2402 case Intrinsic::amdgcn_end_cf:
2403 return selectEndCfIntrinsic(I);
2404 case Intrinsic::amdgcn_ds_ordered_add:
2405 case Intrinsic::amdgcn_ds_ordered_swap:
2406 return selectDSOrderedIntrinsic(I, IntrinsicID);
2407 case Intrinsic::amdgcn_ds_gws_init:
2408 case Intrinsic::amdgcn_ds_gws_barrier:
2409 case Intrinsic::amdgcn_ds_gws_sema_v:
2410 case Intrinsic::amdgcn_ds_gws_sema_br:
2411 case Intrinsic::amdgcn_ds_gws_sema_p:
2412 case Intrinsic::amdgcn_ds_gws_sema_release_all:
2413 return selectDSGWSIntrinsic(I, IntrinsicID);
2414 case Intrinsic::amdgcn_ds_append:
2415 return selectDSAppendConsume(I, true);
2416 case Intrinsic::amdgcn_ds_consume:
2417 return selectDSAppendConsume(I, false);
2418 case Intrinsic::amdgcn_init_whole_wave:
2419 return selectInitWholeWave(I);
2420 case Intrinsic::amdgcn_raw_buffer_load_lds:
2421 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
2422 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
2423 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
2424 case Intrinsic::amdgcn_struct_buffer_load_lds:
2425 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
2426 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
2427 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
2428 return selectBufferLoadLds(I);
2429 // Until we can store both the address space of the global and the LDS
2430 // arguments by having tto MachineMemOperands on an intrinsic, we just trust
2431 // that the argument is a global pointer (buffer pointers have been handled by
2432 // a LLVM IR-level lowering).
2433 case Intrinsic::amdgcn_load_to_lds:
2434 case Intrinsic::amdgcn_load_async_to_lds:
2435 case Intrinsic::amdgcn_global_load_lds:
2436 case Intrinsic::amdgcn_global_load_async_lds:
2437 return selectGlobalLoadLds(I);
2438 case Intrinsic::amdgcn_tensor_load_to_lds:
2439 case Intrinsic::amdgcn_tensor_store_from_lds:
2440 return selectTensorLoadStore(I, IntrinsicID);
2441 case Intrinsic::amdgcn_asyncmark:
2442 case Intrinsic::amdgcn_wait_asyncmark:
2443 if (!Subtarget->hasAsyncMark())
2444 return false;
2445 break;
2446 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
2447 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
2448 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
2449 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
2450 return selectDSBvhStackIntrinsic(I);
2451 case Intrinsic::amdgcn_s_alloc_vgpr: {
2452 // S_ALLOC_VGPR doesn't have a destination register, it just implicitly sets
2453 // SCC. We then need to COPY it into the result vreg.
2454 MachineBasicBlock *MBB = I.getParent();
2455 const DebugLoc &DL = I.getDebugLoc();
2456
2457 Register ResReg = I.getOperand(0).getReg();
2458
2459 MachineInstr *AllocMI = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_ALLOC_VGPR))
2460 .add(I.getOperand(2));
2461 (void)BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), ResReg)
2462 .addReg(AMDGPU::SCC);
2463 I.eraseFromParent();
2464 constrainSelectedInstRegOperands(*AllocMI, TII, TRI, RBI);
2465 return RBI.constrainGenericRegister(ResReg, AMDGPU::SReg_32RegClass, *MRI);
2466 }
2467 case Intrinsic::amdgcn_s_barrier_init:
2468 case Intrinsic::amdgcn_s_barrier_signal_var:
2469 return selectNamedBarrierInit(I, IntrinsicID);
2470 case Intrinsic::amdgcn_s_wakeup_barrier:
2471 case Intrinsic::amdgcn_s_barrier_join:
2472 case Intrinsic::amdgcn_s_get_named_barrier_state:
2473 return selectNamedBarrierInst(I, IntrinsicID);
2474 case Intrinsic::amdgcn_s_get_barrier_state:
2475 return selectSGetBarrierState(I, IntrinsicID);
2476 case Intrinsic::amdgcn_s_barrier_signal_isfirst:
2477 return selectSBarrierSignalIsfirst(I, IntrinsicID);
2478 }
2479 return selectImpl(I, *CoverageInfo);
2480}
2481
2482bool AMDGPUInstructionSelector::selectG_SELECT(MachineInstr &I) const {
2483 if (selectImpl(I, *CoverageInfo))
2484 return true;
2485
2486 MachineBasicBlock *BB = I.getParent();
2487 const DebugLoc &DL = I.getDebugLoc();
2488
2489 Register DstReg = I.getOperand(0).getReg();
2490 unsigned Size = RBI.getSizeInBits(DstReg, *MRI, TRI);
2491 assert(Size <= 32 || Size == 64);
2492 const MachineOperand &CCOp = I.getOperand(1);
2493 Register CCReg = CCOp.getReg();
2494 if (!isVCC(CCReg, *MRI)) {
2495 unsigned SelectOpcode = Size == 64 ? AMDGPU::S_CSELECT_B64 :
2496 AMDGPU::S_CSELECT_B32;
2497 MachineInstr *CopySCC = BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC)
2498 .addReg(CCReg);
2499
2500 // The generic constrainSelectedInstRegOperands doesn't work for the scc register
2501 // bank, because it does not cover the register class that we used to represent
2502 // for it. So we need to manually set the register class here.
2503 if (!MRI->getRegClassOrNull(CCReg))
2504 MRI->setRegClass(CCReg, TRI.getConstrainedRegClassForReg(CCReg, *MRI));
2505 MachineInstr *Select = BuildMI(*BB, &I, DL, TII.get(SelectOpcode), DstReg)
2506 .add(I.getOperand(2))
2507 .add(I.getOperand(3));
2508
2510 constrainSelectedInstRegOperands(*CopySCC, TII, TRI, RBI);
2511 I.eraseFromParent();
2512 return true;
2513 }
2514
2515 // Wide VGPR select should have been split in RegBankSelect.
2516 if (Size > 32)
2517 return false;
2518
2519 MachineInstr *Select =
2520 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
2521 .addImm(0)
2522 .add(I.getOperand(3))
2523 .addImm(0)
2524 .add(I.getOperand(2))
2525 .add(I.getOperand(1));
2526
2528 I.eraseFromParent();
2529 return true;
2530}
2531
2532bool AMDGPUInstructionSelector::selectG_TRUNC(MachineInstr &I) const {
2533 Register DstReg = I.getOperand(0).getReg();
2534 Register SrcReg = I.getOperand(1).getReg();
2535 const LLT DstTy = MRI->getType(DstReg);
2536 const LLT SrcTy = MRI->getType(SrcReg);
2537 const LLT S1 = LLT::scalar(1);
2538
2539 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
2540 const RegisterBank *DstRB;
2541 if (DstTy == S1) {
2542 // This is a special case. We don't treat s1 for legalization artifacts as
2543 // vcc booleans.
2544 DstRB = SrcRB;
2545 } else {
2546 DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
2547 if (SrcRB != DstRB)
2548 return false;
2549 }
2550
2551 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
2552
2553 unsigned DstSize = DstTy.getSizeInBits();
2554 unsigned SrcSize = SrcTy.getSizeInBits();
2555
2556 const TargetRegisterClass *SrcRC =
2557 TRI.getRegClassForSizeOnBank(SrcSize, *SrcRB);
2558 const TargetRegisterClass *DstRC =
2559 TRI.getRegClassForSizeOnBank(DstSize, *DstRB);
2560 if (!SrcRC || !DstRC)
2561 return false;
2562
2563 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
2564 !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI)) {
2565 LLVM_DEBUG(dbgs() << "Failed to constrain G_TRUNC\n");
2566 return false;
2567 }
2568
2569 if (DstRC == &AMDGPU::VGPR_16RegClass && SrcSize == 32) {
2570 assert(STI.useRealTrue16Insts());
2571 const DebugLoc &DL = I.getDebugLoc();
2572 MachineBasicBlock *MBB = I.getParent();
2573 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), DstReg)
2574 .addReg(SrcReg, {}, AMDGPU::lo16);
2575 I.eraseFromParent();
2576 return true;
2577 }
2578
2579 if (DstTy == LLT::fixed_vector(2, 16) && SrcTy == LLT::fixed_vector(2, 32)) {
2580 MachineBasicBlock *MBB = I.getParent();
2581 const DebugLoc &DL = I.getDebugLoc();
2582
2583 Register LoReg = MRI->createVirtualRegister(DstRC);
2584 Register HiReg = MRI->createVirtualRegister(DstRC);
2585 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), LoReg)
2586 .addReg(SrcReg, {}, AMDGPU::sub0);
2587 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), HiReg)
2588 .addReg(SrcReg, {}, AMDGPU::sub1);
2589
2590 if (IsVALU && STI.hasSDWA()) {
2591 // Write the low 16-bits of the high element into the high 16-bits of the
2592 // low element.
2593 MachineInstr *MovSDWA =
2594 BuildMI(*MBB, I, DL, TII.get(AMDGPU::V_MOV_B32_sdwa), DstReg)
2595 .addImm(0) // $src0_modifiers
2596 .addReg(HiReg) // $src0
2597 .addImm(0) // $clamp
2598 .addImm(AMDGPU::SDWA::WORD_1) // $dst_sel
2599 .addImm(AMDGPU::SDWA::UNUSED_PRESERVE) // $dst_unused
2600 .addImm(AMDGPU::SDWA::WORD_0) // $src0_sel
2601 .addReg(LoReg, RegState::Implicit);
2602 MovSDWA->tieOperands(0, MovSDWA->getNumOperands() - 1);
2603 } else {
2604 Register TmpReg0 = MRI->createVirtualRegister(DstRC);
2605 Register TmpReg1 = MRI->createVirtualRegister(DstRC);
2606 Register ImmReg = MRI->createVirtualRegister(DstRC);
2607 if (IsVALU) {
2608 BuildMI(*MBB, I, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), TmpReg0)
2609 .addImm(16)
2610 .addReg(HiReg);
2611 } else {
2612 BuildMI(*MBB, I, DL, TII.get(AMDGPU::S_LSHL_B32), TmpReg0)
2613 .addReg(HiReg)
2614 .addImm(16)
2615 .setOperandDead(3); // Dead scc
2616 }
2617
2618 unsigned MovOpc = IsVALU ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32;
2619 unsigned AndOpc = IsVALU ? AMDGPU::V_AND_B32_e64 : AMDGPU::S_AND_B32;
2620 unsigned OrOpc = IsVALU ? AMDGPU::V_OR_B32_e64 : AMDGPU::S_OR_B32;
2621
2622 BuildMI(*MBB, I, DL, TII.get(MovOpc), ImmReg)
2623 .addImm(0xffff);
2624 auto And = BuildMI(*MBB, I, DL, TII.get(AndOpc), TmpReg1)
2625 .addReg(LoReg)
2626 .addReg(ImmReg);
2627 auto Or = BuildMI(*MBB, I, DL, TII.get(OrOpc), DstReg)
2628 .addReg(TmpReg0)
2629 .addReg(TmpReg1);
2630
2631 if (!IsVALU) {
2632 And.setOperandDead(3); // Dead scc
2633 Or.setOperandDead(3); // Dead scc
2634 }
2635 }
2636
2637 I.eraseFromParent();
2638 return true;
2639 }
2640
2641 if (!DstTy.isScalar())
2642 return false;
2643
2644 if (SrcSize > 32) {
2645 unsigned SubRegIdx = DstSize < 32
2646 ? static_cast<unsigned>(AMDGPU::sub0)
2647 : TRI.getSubRegFromChannel(0, DstSize / 32);
2648 if (SubRegIdx == AMDGPU::NoSubRegister)
2649 return false;
2650
2651 // Deal with weird cases where the class only partially supports the subreg
2652 // index.
2653 const TargetRegisterClass *SrcWithSubRC
2654 = TRI.getSubClassWithSubReg(SrcRC, SubRegIdx);
2655 if (!SrcWithSubRC)
2656 return false;
2657
2658 if (SrcWithSubRC != SrcRC) {
2659 if (!RBI.constrainGenericRegister(SrcReg, *SrcWithSubRC, *MRI))
2660 return false;
2661 }
2662
2663 I.getOperand(1).setSubReg(SubRegIdx);
2664 }
2665
2666 I.setDesc(TII.get(TargetOpcode::COPY));
2667 return true;
2668}
2669
2670/// \returns true if a bitmask for \p Size bits will be an inline immediate.
2671static bool shouldUseAndMask(unsigned Size, unsigned &Mask) {
2673 int SignedMask = static_cast<int>(Mask);
2674 return SignedMask >= -16 && SignedMask <= 64;
2675}
2676
2677// Like RegisterBankInfo::getRegBank, but don't assume vcc for s1.
2678const RegisterBank *AMDGPUInstructionSelector::getArtifactRegBank(
2679 Register Reg, const MachineRegisterInfo &MRI,
2680 const TargetRegisterInfo &TRI) const {
2681 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
2682 if (auto *RB = dyn_cast<const RegisterBank *>(RegClassOrBank))
2683 return RB;
2684
2685 // Ignore the type, since we don't use vcc in artifacts.
2686 if (auto *RC = dyn_cast<const TargetRegisterClass *>(RegClassOrBank))
2687 return &RBI.getRegBankFromRegClass(*RC, LLT());
2688 return nullptr;
2689}
2690
2691bool AMDGPUInstructionSelector::selectG_SZA_EXT(MachineInstr &I) const {
2692 bool InReg = I.getOpcode() == AMDGPU::G_SEXT_INREG;
2693 bool Signed = I.getOpcode() == AMDGPU::G_SEXT || InReg;
2694 const DebugLoc &DL = I.getDebugLoc();
2695 MachineBasicBlock &MBB = *I.getParent();
2696 const Register DstReg = I.getOperand(0).getReg();
2697 const Register SrcReg = I.getOperand(1).getReg();
2698
2699 const LLT DstTy = MRI->getType(DstReg);
2700 const LLT SrcTy = MRI->getType(SrcReg);
2701 const unsigned SrcSize = I.getOpcode() == AMDGPU::G_SEXT_INREG ?
2702 I.getOperand(2).getImm() : SrcTy.getSizeInBits();
2703 const unsigned DstSize = DstTy.getSizeInBits();
2704 if (!DstTy.isScalar())
2705 return false;
2706
2707 // Artifact casts should never use vcc.
2708 const RegisterBank *SrcBank = getArtifactRegBank(SrcReg, *MRI, TRI);
2709
2710 // FIXME: This should probably be illegal and split earlier.
2711 if (I.getOpcode() == AMDGPU::G_ANYEXT) {
2712 if (DstSize <= 32)
2713 return selectCOPY(I);
2714
2715 const TargetRegisterClass *SrcRC =
2716 TRI.getRegClassForTypeOnBank(SrcTy, *SrcBank);
2717 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
2718 const TargetRegisterClass *DstRC =
2719 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
2720
2721 Register UndefReg = MRI->createVirtualRegister(SrcRC);
2722 BuildMI(MBB, I, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2723 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
2724 .addReg(SrcReg)
2725 .addImm(AMDGPU::sub0)
2726 .addReg(UndefReg)
2727 .addImm(AMDGPU::sub1);
2728 I.eraseFromParent();
2729
2730 return RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) &&
2731 RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI);
2732 }
2733
2734 if (SrcBank->getID() == AMDGPU::VGPRRegBankID && DstSize <= 32) {
2735 // 64-bit should have been split up in RegBankSelect
2736
2737 // Try to use an and with a mask if it will save code size.
2738 unsigned Mask;
2739 if (!Signed && shouldUseAndMask(SrcSize, Mask)) {
2740 MachineInstr *ExtI =
2741 BuildMI(MBB, I, DL, TII.get(AMDGPU::V_AND_B32_e32), DstReg)
2742 .addImm(Mask)
2743 .addReg(SrcReg);
2744 I.eraseFromParent();
2745 constrainSelectedInstRegOperands(*ExtI, TII, TRI, RBI);
2746 return true;
2747 }
2748
2749 const unsigned BFE = Signed ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
2750 MachineInstr *ExtI =
2751 BuildMI(MBB, I, DL, TII.get(BFE), DstReg)
2752 .addReg(SrcReg)
2753 .addImm(0) // Offset
2754 .addImm(SrcSize); // Width
2755 I.eraseFromParent();
2756 constrainSelectedInstRegOperands(*ExtI, TII, TRI, RBI);
2757 return true;
2758 }
2759
2760 if (SrcBank->getID() == AMDGPU::SGPRRegBankID && DstSize <= 64) {
2761 const TargetRegisterClass &SrcRC = InReg && DstSize > 32 ?
2762 AMDGPU::SReg_64RegClass : AMDGPU::SReg_32RegClass;
2763 if (!RBI.constrainGenericRegister(SrcReg, SrcRC, *MRI))
2764 return false;
2765
2766 if (Signed && DstSize == 32 && (SrcSize == 8 || SrcSize == 16)) {
2767 const unsigned SextOpc = SrcSize == 8 ?
2768 AMDGPU::S_SEXT_I32_I8 : AMDGPU::S_SEXT_I32_I16;
2769 BuildMI(MBB, I, DL, TII.get(SextOpc), DstReg)
2770 .addReg(SrcReg);
2771 I.eraseFromParent();
2772 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
2773 }
2774
2775 // Using a single 32-bit SALU to calculate the high half is smaller than
2776 // S_BFE with a literal constant operand.
2777 if (DstSize > 32 && SrcSize == 32) {
2778 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2779 unsigned SubReg = InReg ? AMDGPU::sub0 : AMDGPU::NoSubRegister;
2780 if (Signed) {
2781 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_ASHR_I32), HiReg)
2782 .addReg(SrcReg, {}, SubReg)
2783 .addImm(31)
2784 .setOperandDead(3); // Dead scc
2785 } else {
2786 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_MOV_B32), HiReg)
2787 .addImm(0);
2788 }
2789 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
2790 .addReg(SrcReg, {}, SubReg)
2791 .addImm(AMDGPU::sub0)
2792 .addReg(HiReg)
2793 .addImm(AMDGPU::sub1);
2794 I.eraseFromParent();
2795 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_64RegClass,
2796 *MRI);
2797 }
2798
2799 const unsigned BFE64 = Signed ? AMDGPU::S_BFE_I64 : AMDGPU::S_BFE_U64;
2800 const unsigned BFE32 = Signed ? AMDGPU::S_BFE_I32 : AMDGPU::S_BFE_U32;
2801
2802 // Scalar BFE is encoded as S1[5:0] = offset, S1[22:16]= width.
2803 if (DstSize > 32 && (SrcSize <= 32 || InReg)) {
2804 // We need a 64-bit register source, but the high bits don't matter.
2805 Register ExtReg = MRI->createVirtualRegister(&AMDGPU::SReg_64RegClass);
2806 Register UndefReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2807 unsigned SubReg = InReg ? AMDGPU::sub0 : AMDGPU::NoSubRegister;
2808
2809 BuildMI(MBB, I, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2810 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), ExtReg)
2811 .addReg(SrcReg, {}, SubReg)
2812 .addImm(AMDGPU::sub0)
2813 .addReg(UndefReg)
2814 .addImm(AMDGPU::sub1);
2815
2816 BuildMI(MBB, I, DL, TII.get(BFE64), DstReg)
2817 .addReg(ExtReg)
2818 .addImm(SrcSize << 16);
2819
2820 I.eraseFromParent();
2821 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_64RegClass, *MRI);
2822 }
2823
2824 unsigned Mask;
2825 if (!Signed && shouldUseAndMask(SrcSize, Mask)) {
2826 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_AND_B32), DstReg)
2827 .addReg(SrcReg)
2828 .addImm(Mask)
2829 .setOperandDead(3); // Dead scc
2830 } else {
2831 BuildMI(MBB, I, DL, TII.get(BFE32), DstReg)
2832 .addReg(SrcReg)
2833 .addImm(SrcSize << 16);
2834 }
2835
2836 I.eraseFromParent();
2837 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
2838 }
2839
2840 return false;
2841}
2842
2846
2848 Register BitcastSrc;
2849 if (mi_match(Reg, MRI, m_GBitcast(m_Reg(BitcastSrc))))
2850 Reg = BitcastSrc;
2851 return Reg;
2852}
2853
2855 Register &Out) {
2856 // When unmerging a register that is composed of 2 x 16-bit values allow to
2857 // use an extract hi instruction for the upper 16 bits. We only need to check
2858 // the size of `In` as all defs are guaranteed to be the same type for
2859 // GUnmerge.
2860 GUnmerge *Unmerge;
2861 if (mi_match(In, MRI, m_GUnmerge(Unmerge))) {
2862 if (Unmerge->getNumDefs() == 2 && Unmerge->getOperand(1).getReg() == In &&
2863 MRI.getType(In).getSizeInBits() == 16) {
2864 Out = stripBitCast(Unmerge->getSourceReg(), MRI);
2865 return true;
2866 }
2867 }
2868
2869 Register Trunc;
2870 if (!mi_match(In, MRI, m_GTrunc(m_Reg(Trunc))))
2871 return false;
2872
2873 Register LShlSrc;
2874 Register Cst;
2875 if (mi_match(Trunc, MRI, m_GLShr(m_Reg(LShlSrc), m_Reg(Cst)))) {
2876 Cst = stripCopy(Cst, MRI);
2877 if (mi_match(Cst, MRI, m_SpecificICst(16))) {
2878 Out = stripBitCast(LShlSrc, MRI);
2879 return true;
2880 }
2881 }
2882
2883 ArrayRef<int> Mask;
2884 Register Src1;
2885 if (!mi_match(Trunc, MRI, m_GShuffleVector(m_Reg(Src1), m_Reg(), Mask)))
2886 return false;
2887
2888 assert(MRI.getType(Src1) == LLT::fixed_vector(2, 16));
2889 assert(Mask.size() == 2);
2890
2891 if (Mask[0] == 1 && Mask[1] <= 1) {
2892 Out = Trunc;
2893 return true;
2894 }
2895
2896 return false;
2897}
2898
2900 Register &Out) {
2901 // There could be a bitcast between the extraction and its use.
2902 In = stripBitCast(In, MRI);
2903
2904 // The first def of a 2 x 16-bit unmerge is the low half of its source.
2905 if (auto *Unmerge = dyn_cast<GUnmerge>(MRI.getVRegDef(In))) {
2906 if (Unmerge->getNumDefs() == 2 && Unmerge->getOperand(0).getReg() == In &&
2907 MRI.getType(In).getSizeInBits() == 16) {
2908 Out = stripBitCast(Unmerge->getSourceReg(), MRI);
2909 return true;
2910 }
2911 }
2912
2913 // A truncation from 32 to 16 bits keeps the low half in place.
2914 Register Trunc;
2915 if (mi_match(In, MRI, m_GTrunc(m_Reg(Trunc))) &&
2916 MRI.getType(Trunc).getSizeInBits() == 32) {
2917 Out = stripBitCast(Trunc, MRI);
2918 return true;
2919 }
2920
2921 return false;
2922}
2923
2924bool AMDGPUInstructionSelector::selectG_FPEXT(MachineInstr &I) const {
2925 if (!Subtarget->hasSALUFloatInsts())
2926 return false;
2927
2928 Register Dst = I.getOperand(0).getReg();
2929 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
2930 if (DstRB->getID() != AMDGPU::SGPRRegBankID)
2931 return false;
2932
2933 Register Src = I.getOperand(1).getReg();
2934
2935 if (MRI->getType(Dst) == LLT::scalar(32) &&
2936 MRI->getType(Src) == LLT::scalar(16)) {
2937 if (isExtractHiElt(*MRI, Src, Src)) {
2938 MachineBasicBlock *BB = I.getParent();
2939 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::S_CVT_HI_F32_F16), Dst)
2940 .addUse(Src);
2941 I.eraseFromParent();
2942 return RBI.constrainGenericRegister(Dst, AMDGPU::SReg_32RegClass, *MRI);
2943 }
2944 }
2945
2946 return false;
2947}
2948
2949bool AMDGPUInstructionSelector::selectG_FNEG(MachineInstr &MI) const {
2950 // Only manually handle the f64 SGPR case.
2951 //
2952 // FIXME: This is a workaround for 2.5 different tablegen problems. Because
2953 // the bit ops theoretically have a second result due to the implicit def of
2954 // SCC, the GlobalISelEmitter is overly conservative and rejects it. Fixing
2955 // that is easy by disabling the check. The result works, but uses a
2956 // nonsensical sreg32orlds_and_sreg_1 regclass.
2957 //
2958 // The DAG emitter is more problematic, and incorrectly adds both S_XOR_B32 to
2959 // the variadic REG_SEQUENCE operands.
2960
2961 Register Dst = MI.getOperand(0).getReg();
2962 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
2963 if (DstRB->getID() != AMDGPU::SGPRRegBankID ||
2964 MRI->getType(Dst) != LLT::scalar(64))
2965 return false;
2966
2967 Register Src = MI.getOperand(1).getReg();
2968 MachineInstr *Fabs = getOpcodeDef(TargetOpcode::G_FABS, Src, *MRI);
2969 if (Fabs)
2970 Src = Fabs->getOperand(1).getReg();
2971
2972 if (!RBI.constrainGenericRegister(Src, AMDGPU::SReg_64RegClass, *MRI) ||
2973 !RBI.constrainGenericRegister(Dst, AMDGPU::SReg_64RegClass, *MRI))
2974 return false;
2975
2976 MachineBasicBlock *BB = MI.getParent();
2977 const DebugLoc &DL = MI.getDebugLoc();
2978 Register LoReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2979 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2980 Register ConstReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2981 Register OpReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2982
2983 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), LoReg)
2984 .addReg(Src, {}, AMDGPU::sub0);
2985 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), HiReg)
2986 .addReg(Src, {}, AMDGPU::sub1);
2987 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), ConstReg)
2988 .addImm(0x80000000);
2989
2990 // Set or toggle sign bit.
2991 unsigned Opc = Fabs ? AMDGPU::S_OR_B32 : AMDGPU::S_XOR_B32;
2992 BuildMI(*BB, &MI, DL, TII.get(Opc), OpReg)
2993 .addReg(HiReg)
2994 .addReg(ConstReg)
2995 .setOperandDead(3); // Dead scc
2996 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), Dst)
2997 .addReg(LoReg)
2998 .addImm(AMDGPU::sub0)
2999 .addReg(OpReg)
3000 .addImm(AMDGPU::sub1);
3001 MI.eraseFromParent();
3002 return true;
3003}
3004
3005// FIXME: This is a workaround for the same tablegen problems as G_FNEG
3006bool AMDGPUInstructionSelector::selectG_FABS(MachineInstr &MI) const {
3007 Register Dst = MI.getOperand(0).getReg();
3008 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
3009 if (DstRB->getID() != AMDGPU::SGPRRegBankID ||
3010 MRI->getType(Dst) != LLT::scalar(64))
3011 return false;
3012
3013 Register Src = MI.getOperand(1).getReg();
3014 MachineBasicBlock *BB = MI.getParent();
3015 const DebugLoc &DL = MI.getDebugLoc();
3016 Register LoReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3017 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3018 Register ConstReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3019 Register OpReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3020
3021 if (!RBI.constrainGenericRegister(Src, AMDGPU::SReg_64RegClass, *MRI) ||
3022 !RBI.constrainGenericRegister(Dst, AMDGPU::SReg_64RegClass, *MRI))
3023 return false;
3024
3025 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), LoReg)
3026 .addReg(Src, {}, AMDGPU::sub0);
3027 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), HiReg)
3028 .addReg(Src, {}, AMDGPU::sub1);
3029 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), ConstReg)
3030 .addImm(0x7fffffff);
3031
3032 // Clear sign bit.
3033 // TODO: Should this used S_BITSET0_*?
3034 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_AND_B32), OpReg)
3035 .addReg(HiReg)
3036 .addReg(ConstReg)
3037 .setOperandDead(3); // Dead scc
3038 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), Dst)
3039 .addReg(LoReg)
3040 .addImm(AMDGPU::sub0)
3041 .addReg(OpReg)
3042 .addImm(AMDGPU::sub1);
3043
3044 MI.eraseFromParent();
3045 return true;
3046}
3047
3048static bool isConstant(const MachineInstr &MI) {
3049 return MI.getOpcode() == TargetOpcode::G_CONSTANT;
3050}
3051
3052void AMDGPUInstructionSelector::getAddrModeInfo(const MachineInstr &Load,
3053 const MachineRegisterInfo &MRI, SmallVectorImpl<GEPInfo> &AddrInfo) const {
3054
3055 unsigned OpNo = Load.getOpcode() == AMDGPU::G_PREFETCH ? 0 : 1;
3056 const MachineInstr *PtrMI =
3057 MRI.getUniqueVRegDef(Load.getOperand(OpNo).getReg());
3058
3059 assert(PtrMI);
3060
3061 if (PtrMI->getOpcode() != TargetOpcode::G_PTR_ADD)
3062 return;
3063
3064 GEPInfo GEPInfo;
3065
3066 for (unsigned i = 1; i != 3; ++i) {
3067 const MachineOperand &GEPOp = PtrMI->getOperand(i);
3068 const MachineInstr *OpDef = MRI.getUniqueVRegDef(GEPOp.getReg());
3069 assert(OpDef);
3070 if (i == 2 && isConstant(*OpDef)) {
3071 // TODO: Could handle constant base + variable offset, but a combine
3072 // probably should have commuted it.
3073 assert(GEPInfo.Imm == 0);
3074 GEPInfo.Imm = OpDef->getOperand(1).getCImm()->getSExtValue();
3075 continue;
3076 }
3077 const RegisterBank *OpBank = RBI.getRegBank(GEPOp.getReg(), MRI, TRI);
3078 if (OpBank->getID() == AMDGPU::SGPRRegBankID)
3079 GEPInfo.SgprParts.push_back(GEPOp.getReg());
3080 else
3081 GEPInfo.VgprParts.push_back(GEPOp.getReg());
3082 }
3083
3084 AddrInfo.push_back(GEPInfo);
3085 getAddrModeInfo(*PtrMI, MRI, AddrInfo);
3086}
3087
3088bool AMDGPUInstructionSelector::isSGPR(Register Reg) const {
3089 return RBI.getRegBank(Reg, *MRI, TRI)->getID() == AMDGPU::SGPRRegBankID;
3090}
3091
3092bool AMDGPUInstructionSelector::isInstrUniform(const MachineInstr &MI) const {
3093 if (!MI.hasOneMemOperand())
3094 return false;
3095
3096 const MachineMemOperand *MMO = *MI.memoperands_begin();
3097 const Value *Ptr = MMO->getValue();
3098
3099 // UndefValue means this is a load of a kernel input. These are uniform.
3100 // Sometimes LDS instructions have constant pointers.
3101 // If Ptr is null, then that means this mem operand contains a
3102 // PseudoSourceValue like GOT.
3104 return true;
3105
3107 return true;
3108
3109 if (MI.getOpcode() == AMDGPU::G_PREFETCH)
3110 return RBI.getRegBank(MI.getOperand(0).getReg(), *MRI, TRI)->getID() ==
3111 AMDGPU::SGPRRegBankID;
3112
3113 const Instruction *I = dyn_cast<Instruction>(Ptr);
3114 return I && I->getMetadata("amdgpu.uniform");
3115}
3116
3117bool AMDGPUInstructionSelector::hasVgprParts(ArrayRef<GEPInfo> AddrInfo) const {
3118 for (const GEPInfo &GEPInfo : AddrInfo) {
3119 if (!GEPInfo.VgprParts.empty())
3120 return true;
3121 }
3122 return false;
3123}
3124
3125void AMDGPUInstructionSelector::initM0(MachineInstr &I) const {
3126 const LLT PtrTy = MRI->getType(I.getOperand(1).getReg());
3127 unsigned AS = PtrTy.getAddressSpace();
3129 STI.ldsRequiresM0Init()) {
3130 MachineBasicBlock *BB = I.getParent();
3131
3132 // If DS instructions require M0 initialization, insert it before selecting.
3133 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), AMDGPU::M0)
3134 .addImm(-1);
3135 }
3136}
3137
3138bool AMDGPUInstructionSelector::selectG_LOAD_STORE_ATOMICRMW(
3139 MachineInstr &I) const {
3140 initM0(I);
3141 return selectImpl(I, *CoverageInfo);
3142}
3143
3145 if (Reg.isPhysical())
3146 return false;
3147
3149 const unsigned Opcode = MI.getOpcode();
3150
3151 if (Opcode == AMDGPU::COPY)
3152 return isVCmpResult(MI.getOperand(1).getReg(), MRI);
3153
3154 if (Opcode == AMDGPU::G_AND || Opcode == AMDGPU::G_OR ||
3155 Opcode == AMDGPU::G_XOR)
3156 return isVCmpResult(MI.getOperand(1).getReg(), MRI) &&
3157 isVCmpResult(MI.getOperand(2).getReg(), MRI);
3158
3159 if (auto *GI = dyn_cast<GIntrinsic>(&MI))
3160 return GI->is(Intrinsic::amdgcn_class);
3161
3162 return Opcode == AMDGPU::G_ICMP || Opcode == AMDGPU::G_FCMP;
3163}
3164
3165bool AMDGPUInstructionSelector::selectG_BRCOND(MachineInstr &I) const {
3166 MachineBasicBlock *BB = I.getParent();
3167 MachineOperand &CondOp = I.getOperand(0);
3168 Register CondReg = CondOp.getReg();
3169 const DebugLoc &DL = I.getDebugLoc();
3170
3171 unsigned BrOpcode;
3172 Register CondPhysReg;
3173 const TargetRegisterClass *ConstrainRC;
3174
3175 // In SelectionDAG, we inspect the IR block for uniformity metadata to decide
3176 // whether the branch is uniform when selecting the instruction. In
3177 // GlobalISel, we should push that decision into RegBankSelect. Assume for now
3178 // RegBankSelect knows what it's doing if the branch condition is scc, even
3179 // though it currently does not.
3180 if (!isVCC(CondReg, *MRI)) {
3181 if (MRI->getType(CondReg) != LLT::scalar(32))
3182 return false;
3183
3184 CondPhysReg = AMDGPU::SCC;
3185 BrOpcode = AMDGPU::S_CBRANCH_SCC1;
3186 ConstrainRC = &AMDGPU::SReg_32RegClass;
3187 } else {
3188 // FIXME: Should scc->vcc copies and with exec?
3189
3190 // Unless the value of CondReg is a result of a V_CMP* instruction then we
3191 // need to insert an and with exec.
3192 if (!isVCmpResult(CondReg, *MRI)) {
3193 const bool Is64 = STI.isWave64();
3194 const unsigned Opcode = Is64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
3195 const Register Exec = Is64 ? AMDGPU::EXEC : AMDGPU::EXEC_LO;
3196
3197 Register TmpReg = MRI->createVirtualRegister(TRI.getBoolRC());
3198 BuildMI(*BB, &I, DL, TII.get(Opcode), TmpReg)
3199 .addReg(CondReg)
3200 .addReg(Exec)
3201 .setOperandDead(3); // Dead scc
3202 CondReg = TmpReg;
3203 }
3204
3205 CondPhysReg = TRI.getVCC();
3206 BrOpcode = AMDGPU::S_CBRANCH_VCCNZ;
3207 ConstrainRC = TRI.getBoolRC();
3208 }
3209
3210 if (!MRI->getRegClassOrNull(CondReg))
3211 MRI->setRegClass(CondReg, ConstrainRC);
3212
3213 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), CondPhysReg)
3214 .addReg(CondReg);
3215 BuildMI(*BB, &I, DL, TII.get(BrOpcode))
3216 .addMBB(I.getOperand(1).getMBB());
3217
3218 I.eraseFromParent();
3219 return true;
3220}
3221
3222bool AMDGPUInstructionSelector::selectG_GLOBAL_VALUE(
3223 MachineInstr &I) const {
3224 Register DstReg = I.getOperand(0).getReg();
3225 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3226 const bool IsVGPR = DstRB->getID() == AMDGPU::VGPRRegBankID;
3227 I.setDesc(TII.get(IsVGPR ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32));
3228 if (IsVGPR)
3229 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
3230
3231 return RBI.constrainGenericRegister(
3232 DstReg, IsVGPR ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass, *MRI);
3233}
3234
3235bool AMDGPUInstructionSelector::selectG_PTRMASK(MachineInstr &I) const {
3236 Register DstReg = I.getOperand(0).getReg();
3237 Register SrcReg = I.getOperand(1).getReg();
3238 Register MaskReg = I.getOperand(2).getReg();
3239 LLT Ty = MRI->getType(DstReg);
3240 LLT MaskTy = MRI->getType(MaskReg);
3241 MachineBasicBlock *BB = I.getParent();
3242 const DebugLoc &DL = I.getDebugLoc();
3243
3244 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3245 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
3246 const RegisterBank *MaskRB = RBI.getRegBank(MaskReg, *MRI, TRI);
3247 const bool IsVGPR = DstRB->getID() == AMDGPU::VGPRRegBankID;
3248 if (DstRB != SrcRB) // Should only happen for hand written MIR.
3249 return false;
3250
3251 // Try to avoid emitting a bit operation when we only need to touch half of
3252 // the 64-bit pointer.
3253 APInt MaskOnes = VT->getKnownOnes(MaskReg).zext(64);
3254 const APInt MaskHi32 = APInt::getHighBitsSet(64, 32);
3255 const APInt MaskLo32 = APInt::getLowBitsSet(64, 32);
3256
3257 const bool CanCopyLow32 = (MaskOnes & MaskLo32) == MaskLo32;
3258 const bool CanCopyHi32 = (MaskOnes & MaskHi32) == MaskHi32;
3259
3260 if (!IsVGPR && Ty.getSizeInBits() == 64 &&
3261 !CanCopyLow32 && !CanCopyHi32) {
3262 auto MIB = BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_AND_B64), DstReg)
3263 .addReg(SrcReg)
3264 .addReg(MaskReg)
3265 .setOperandDead(3); // Dead scc
3266 I.eraseFromParent();
3267 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3268 return true;
3269 }
3270
3271 unsigned NewOpc = IsVGPR ? AMDGPU::V_AND_B32_e64 : AMDGPU::S_AND_B32;
3272 const TargetRegisterClass &RegRC
3273 = IsVGPR ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
3274
3275 const TargetRegisterClass *DstRC = TRI.getRegClassForTypeOnBank(Ty, *DstRB);
3276 const TargetRegisterClass *SrcRC = TRI.getRegClassForTypeOnBank(Ty, *SrcRB);
3277 const TargetRegisterClass *MaskRC =
3278 TRI.getRegClassForTypeOnBank(MaskTy, *MaskRB);
3279
3280 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
3281 !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
3282 !RBI.constrainGenericRegister(MaskReg, *MaskRC, *MRI))
3283 return false;
3284
3285 if (Ty.getSizeInBits() == 32) {
3286 assert(MaskTy.getSizeInBits() == 32 &&
3287 "ptrmask should have been narrowed during legalize");
3288
3289 auto NewOp = BuildMI(*BB, &I, DL, TII.get(NewOpc), DstReg)
3290 .addReg(SrcReg)
3291 .addReg(MaskReg);
3292
3293 if (!IsVGPR)
3294 NewOp.setOperandDead(3); // Dead scc
3295 I.eraseFromParent();
3296 return true;
3297 }
3298
3299 Register HiReg = MRI->createVirtualRegister(&RegRC);
3300 Register LoReg = MRI->createVirtualRegister(&RegRC);
3301
3302 // Extract the subregisters from the source pointer.
3303 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), LoReg)
3304 .addReg(SrcReg, {}, AMDGPU::sub0);
3305 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), HiReg)
3306 .addReg(SrcReg, {}, AMDGPU::sub1);
3307
3308 Register MaskedLo, MaskedHi;
3309
3310 if (CanCopyLow32) {
3311 // If all the bits in the low half are 1, we only need a copy for it.
3312 MaskedLo = LoReg;
3313 } else {
3314 // Extract the mask subregister and apply the and.
3315 Register MaskLo = MRI->createVirtualRegister(&RegRC);
3316 MaskedLo = MRI->createVirtualRegister(&RegRC);
3317
3318 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), MaskLo)
3319 .addReg(MaskReg, {}, AMDGPU::sub0);
3320 BuildMI(*BB, &I, DL, TII.get(NewOpc), MaskedLo)
3321 .addReg(LoReg)
3322 .addReg(MaskLo);
3323 }
3324
3325 if (CanCopyHi32) {
3326 // If all the bits in the high half are 1, we only need a copy for it.
3327 MaskedHi = HiReg;
3328 } else {
3329 Register MaskHi = MRI->createVirtualRegister(&RegRC);
3330 MaskedHi = MRI->createVirtualRegister(&RegRC);
3331
3332 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), MaskHi)
3333 .addReg(MaskReg, {}, AMDGPU::sub1);
3334 BuildMI(*BB, &I, DL, TII.get(NewOpc), MaskedHi)
3335 .addReg(HiReg)
3336 .addReg(MaskHi);
3337 }
3338
3339 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
3340 .addReg(MaskedLo)
3341 .addImm(AMDGPU::sub0)
3342 .addReg(MaskedHi)
3343 .addImm(AMDGPU::sub1);
3344 I.eraseFromParent();
3345 return true;
3346}
3347
3348/// Return the register to use for the index value, and the subregister to use
3349/// for the indirectly accessed register.
3350static std::pair<Register, unsigned>
3352 const TargetRegisterClass *SuperRC, Register IdxReg,
3353 unsigned EltSize, GISelValueTracking &ValueTracking) {
3354 Register IdxBaseReg;
3355 int Offset;
3356
3357 std::tie(IdxBaseReg, Offset) =
3358 AMDGPU::getBaseWithConstantOffset(MRI, IdxReg, &ValueTracking);
3359 if (IdxBaseReg == AMDGPU::NoRegister) {
3360 // This will happen if the index is a known constant. This should ordinarily
3361 // be legalized out, but handle it as a register just in case.
3362 assert(Offset == 0);
3363 IdxBaseReg = IdxReg;
3364 }
3365
3366 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(SuperRC, EltSize);
3367
3368 // Skip out of bounds offsets, or else we would end up using an undefined
3369 // register.
3370 if (static_cast<unsigned>(Offset) >= SubRegs.size())
3371 return std::pair(IdxReg, SubRegs[0]);
3372 return std::pair(IdxBaseReg, SubRegs[Offset]);
3373}
3374
3375bool AMDGPUInstructionSelector::selectG_EXTRACT_VECTOR_ELT(
3376 MachineInstr &MI) const {
3377 Register DstReg = MI.getOperand(0).getReg();
3378 Register SrcReg = MI.getOperand(1).getReg();
3379 Register IdxReg = MI.getOperand(2).getReg();
3380
3381 LLT DstTy = MRI->getType(DstReg);
3382 LLT SrcTy = MRI->getType(SrcReg);
3383
3384 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3385 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
3386 const RegisterBank *IdxRB = RBI.getRegBank(IdxReg, *MRI, TRI);
3387
3388 // The index must be scalar. If it wasn't RegBankSelect should have moved this
3389 // into a waterfall loop.
3390 if (IdxRB->getID() != AMDGPU::SGPRRegBankID)
3391 return false;
3392
3393 const TargetRegisterClass *SrcRC =
3394 TRI.getRegClassForTypeOnBank(SrcTy, *SrcRB);
3395 const TargetRegisterClass *DstRC =
3396 TRI.getRegClassForTypeOnBank(DstTy, *DstRB);
3397 if (!SrcRC || !DstRC)
3398 return false;
3399 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
3400 !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
3401 !RBI.constrainGenericRegister(IdxReg, AMDGPU::SReg_32RegClass, *MRI))
3402 return false;
3403
3404 MachineBasicBlock *BB = MI.getParent();
3405 const DebugLoc &DL = MI.getDebugLoc();
3406 const bool Is64 = DstTy.getSizeInBits() == 64;
3407
3408 unsigned SubReg;
3409 std::tie(IdxReg, SubReg) = computeIndirectRegIndex(
3410 *MRI, TRI, SrcRC, IdxReg, DstTy.getSizeInBits() / 8, *VT);
3411
3412 if (SrcRB->getID() == AMDGPU::SGPRRegBankID) {
3413 if (DstTy.getSizeInBits() != 32 && !Is64)
3414 return false;
3415
3416 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3417 .addReg(IdxReg);
3418
3419 unsigned Opc = Is64 ? AMDGPU::S_MOVRELS_B64 : AMDGPU::S_MOVRELS_B32;
3420 BuildMI(*BB, &MI, DL, TII.get(Opc), DstReg)
3421 .addReg(SrcReg, {}, SubReg)
3422 .addReg(SrcReg, RegState::Implicit);
3423 MI.eraseFromParent();
3424 return true;
3425 }
3426
3427 if (SrcRB->getID() != AMDGPU::VGPRRegBankID || DstTy.getSizeInBits() != 32)
3428 return false;
3429
3430 if (!STI.useVGPRIndexMode()) {
3431 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3432 .addReg(IdxReg);
3433 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::V_MOVRELS_B32_e32), DstReg)
3434 .addReg(SrcReg, {}, SubReg)
3435 .addReg(SrcReg, RegState::Implicit);
3436 MI.eraseFromParent();
3437 return true;
3438 }
3439
3440 const MCInstrDesc &GPRIDXDesc =
3441 TII.getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*SrcRC), true);
3442 BuildMI(*BB, MI, DL, GPRIDXDesc, DstReg)
3443 .addReg(SrcReg)
3444 .addReg(IdxReg)
3445 .addImm(SubReg);
3446
3447 MI.eraseFromParent();
3448 return true;
3449}
3450
3451// TODO: Fold insert_vector_elt (extract_vector_elt) into movrelsd
3452bool AMDGPUInstructionSelector::selectG_INSERT_VECTOR_ELT(
3453 MachineInstr &MI) const {
3454 Register DstReg = MI.getOperand(0).getReg();
3455 Register VecReg = MI.getOperand(1).getReg();
3456 Register ValReg = MI.getOperand(2).getReg();
3457 Register IdxReg = MI.getOperand(3).getReg();
3458
3459 LLT VecTy = MRI->getType(DstReg);
3460 LLT ValTy = MRI->getType(ValReg);
3461 unsigned VecSize = VecTy.getSizeInBits();
3462 unsigned ValSize = ValTy.getSizeInBits();
3463
3464 const RegisterBank *VecRB = RBI.getRegBank(VecReg, *MRI, TRI);
3465 const RegisterBank *ValRB = RBI.getRegBank(ValReg, *MRI, TRI);
3466 const RegisterBank *IdxRB = RBI.getRegBank(IdxReg, *MRI, TRI);
3467
3468 assert(VecTy.getElementType() == ValTy);
3469
3470 // The index must be scalar. If it wasn't RegBankSelect should have moved this
3471 // into a waterfall loop.
3472 if (IdxRB->getID() != AMDGPU::SGPRRegBankID)
3473 return false;
3474
3475 const TargetRegisterClass *VecRC =
3476 TRI.getRegClassForTypeOnBank(VecTy, *VecRB);
3477 const TargetRegisterClass *ValRC =
3478 TRI.getRegClassForTypeOnBank(ValTy, *ValRB);
3479
3480 if (!RBI.constrainGenericRegister(VecReg, *VecRC, *MRI) ||
3481 !RBI.constrainGenericRegister(DstReg, *VecRC, *MRI) ||
3482 !RBI.constrainGenericRegister(ValReg, *ValRC, *MRI) ||
3483 !RBI.constrainGenericRegister(IdxReg, AMDGPU::SReg_32RegClass, *MRI))
3484 return false;
3485
3486 if (VecRB->getID() == AMDGPU::VGPRRegBankID && ValSize != 32)
3487 return false;
3488
3489 unsigned SubReg;
3490 std::tie(IdxReg, SubReg) =
3491 computeIndirectRegIndex(*MRI, TRI, VecRC, IdxReg, ValSize / 8, *VT);
3492
3493 const bool IndexMode = VecRB->getID() == AMDGPU::VGPRRegBankID &&
3494 STI.useVGPRIndexMode();
3495
3496 MachineBasicBlock *BB = MI.getParent();
3497 const DebugLoc &DL = MI.getDebugLoc();
3498
3499 if (!IndexMode) {
3500 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3501 .addReg(IdxReg);
3502
3503 const MCInstrDesc &RegWriteOp = TII.getIndirectRegWriteMovRelPseudo(
3504 VecSize, ValSize, VecRB->getID() == AMDGPU::SGPRRegBankID);
3505 BuildMI(*BB, MI, DL, RegWriteOp, DstReg)
3506 .addReg(VecReg)
3507 .addReg(ValReg)
3508 .addImm(SubReg);
3509 MI.eraseFromParent();
3510 return true;
3511 }
3512
3513 const MCInstrDesc &GPRIDXDesc =
3514 TII.getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), false);
3515 BuildMI(*BB, MI, DL, GPRIDXDesc, DstReg)
3516 .addReg(VecReg)
3517 .addReg(ValReg)
3518 .addReg(IdxReg)
3519 .addImm(SubReg);
3520
3521 MI.eraseFromParent();
3522 return true;
3523}
3524
3525static bool isAsyncLDSDMA(Intrinsic::ID Intr) {
3526 switch (Intr) {
3527 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
3528 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
3529 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
3530 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
3531 case Intrinsic::amdgcn_load_async_to_lds:
3532 case Intrinsic::amdgcn_global_load_async_lds:
3533 return true;
3534 }
3535 return false;
3536}
3537
3538bool AMDGPUInstructionSelector::selectBufferLoadLds(MachineInstr &MI) const {
3539 if (!Subtarget->hasVMemToLDSLoad())
3540 return false;
3541 unsigned Opc;
3542 unsigned Size = MI.getOperand(3).getImm();
3543 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
3544
3545 // The struct intrinsic variants add one additional operand over raw.
3546 const bool HasVIndex = MI.getNumOperands() == 9;
3547 Register VIndex;
3548 int OpOffset = 0;
3549 if (HasVIndex) {
3550 VIndex = MI.getOperand(4).getReg();
3551 OpOffset = 1;
3552 }
3553
3554 Register VOffset = MI.getOperand(4 + OpOffset).getReg();
3555 std::optional<ValueAndVReg> MaybeVOffset =
3557 const bool HasVOffset = !MaybeVOffset || MaybeVOffset->Value.getZExtValue();
3558
3559 switch (Size) {
3560 default:
3561 return false;
3562 case 1:
3563 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_BOTHEN
3564 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_IDXEN
3565 : HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFEN
3566 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFSET;
3567 break;
3568 case 2:
3569 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_BOTHEN
3570 : AMDGPU::BUFFER_LOAD_USHORT_LDS_IDXEN
3571 : HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFEN
3572 : AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFSET;
3573 break;
3574 case 4:
3575 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_BOTHEN
3576 : AMDGPU::BUFFER_LOAD_DWORD_LDS_IDXEN
3577 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFEN
3578 : AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFSET;
3579 break;
3580 case 12:
3581 if (!Subtarget->hasLDSLoadB96_B128())
3582 return false;
3583
3584 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_BOTHEN
3585 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_IDXEN
3586 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFEN
3587 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFSET;
3588 break;
3589 case 16:
3590 if (!Subtarget->hasLDSLoadB96_B128())
3591 return false;
3592
3593 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_BOTHEN
3594 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_IDXEN
3595 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFEN
3596 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFSET;
3597 break;
3598 }
3599
3600 MachineBasicBlock *MBB = MI.getParent();
3601 const DebugLoc &DL = MI.getDebugLoc();
3602 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3603 .add(MI.getOperand(2));
3604
3605 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc));
3606
3607 if (HasVIndex && HasVOffset) {
3608 Register IdxReg = MRI->createVirtualRegister(TRI.getVGPR64Class());
3609 BuildMI(*MBB, &*MIB, DL, TII.get(AMDGPU::REG_SEQUENCE), IdxReg)
3610 .addReg(VIndex)
3611 .addImm(AMDGPU::sub0)
3612 .addReg(VOffset)
3613 .addImm(AMDGPU::sub1);
3614
3615 MIB.addReg(IdxReg);
3616 } else if (HasVIndex) {
3617 MIB.addReg(VIndex);
3618 } else if (HasVOffset) {
3619 MIB.addReg(VOffset);
3620 }
3621
3622 MIB.add(MI.getOperand(1)); // rsrc
3623 MIB.add(MI.getOperand(5 + OpOffset)); // soffset
3624 MIB.add(MI.getOperand(6 + OpOffset)); // imm offset
3625 bool IsGFX12Plus = AMDGPU::isGFX12Plus(STI);
3626 unsigned Aux = MI.getOperand(7 + OpOffset).getImm();
3627 MIB.addImm(Aux & (IsGFX12Plus ? AMDGPU::CPol::ALL
3628 : AMDGPU::CPol::ALL_pregfx12)); // cpol
3629 MIB.addImm(
3630 Aux & (IsGFX12Plus ? AMDGPU::CPol::SWZ : AMDGPU::CPol::SWZ_pregfx12)
3631 ? 1
3632 : 0); // swz
3633 MIB.addImm(isAsyncLDSDMA(IntrinsicID));
3634
3635 MachineMemOperand *LoadMMO = *MI.memoperands_begin();
3636 // Don't set the offset value here because the pointer points to the base of
3637 // the buffer.
3638 MachinePointerInfo LoadPtrI = LoadMMO->getPointerInfo();
3639
3640 MachinePointerInfo StorePtrI = LoadPtrI;
3641 LoadPtrI.V = PoisonValue::get(PointerType::get(MF->getFunction().getContext(),
3645
3646 auto F = LoadMMO->getFlags() &
3648 LoadMMO = MF->getMachineMemOperand(LoadPtrI, F | MachineMemOperand::MOLoad,
3649 Size, LoadMMO->getBaseAlign());
3650
3651 MachineMemOperand *StoreMMO =
3652 MF->getMachineMemOperand(StorePtrI, F | MachineMemOperand::MOStore,
3653 sizeof(int32_t), LoadMMO->getBaseAlign());
3654
3655 MIB.setMemRefs({LoadMMO, StoreMMO});
3656
3657 MI.eraseFromParent();
3658 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3659 return true;
3660}
3661
3662/// Match a zero extend from a 32-bit value to 64-bits.
3663Register AMDGPUInstructionSelector::matchZeroExtendFromS32(Register Reg) const {
3664 Register ZExtSrc;
3665 if (mi_match(Reg, *MRI, m_GZExt(m_Reg(ZExtSrc))))
3666 return MRI->getType(ZExtSrc) == LLT::scalar(32) ? ZExtSrc : Register();
3667
3668 // Match legalized form %zext = G_MERGE_VALUES (s32 %x), (s32 0)
3669 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3670 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3671 return Register();
3672
3673 assert(Def->getNumOperands() == 3 &&
3674 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3675 if (mi_match(Def->getOperand(2).getReg(), *MRI, m_ZeroInt())) {
3676 return Def->getOperand(1).getReg();
3677 }
3678
3679 return Register();
3680}
3681
3682/// Match a sign extend from a 32-bit value to 64-bits.
3683Register AMDGPUInstructionSelector::matchSignExtendFromS32(Register Reg) const {
3684 Register SExtSrc;
3685 if (mi_match(Reg, *MRI, m_GSExt(m_Reg(SExtSrc))))
3686 return MRI->getType(SExtSrc) == LLT::scalar(32) ? SExtSrc : Register();
3687
3688 // Match legalized form %sext = G_MERGE_VALUES (s32 %x), G_ASHR((S32 %x, 31))
3689 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3690 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3691 return Register();
3692
3693 assert(Def->getNumOperands() == 3 &&
3694 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3695 if (mi_match(Def->getOperand(2).getReg(), *MRI,
3696 m_GAShr(m_SpecificReg(Def->getOperand(1).getReg()),
3697 m_SpecificICst(31))))
3698 return Def->getOperand(1).getReg();
3699
3700 Register ZextSrc = matchZeroExtendFromS32(Reg);
3701 if (ZextSrc && VT->signBitIsZero(ZextSrc))
3702 return ZextSrc;
3703
3704 return Register();
3705}
3706
3707/// Match a zero extend from a 32-bit value to 64-bits, or \p Reg itself if it
3708/// is 32-bit.
3710AMDGPUInstructionSelector::matchZeroExtendFromS32OrS32(Register Reg) const {
3711 return MRI->getType(Reg) == LLT::scalar(32) ? Reg
3712 : matchZeroExtendFromS32(Reg);
3713}
3714
3715/// Match a sign extend from a 32-bit value to 64-bits, or \p Reg itself if it
3716/// is 32-bit.
3718AMDGPUInstructionSelector::matchSignExtendFromS32OrS32(Register Reg) const {
3719 return MRI->getType(Reg) == LLT::scalar(32) ? Reg
3720 : matchSignExtendFromS32(Reg);
3721}
3722
3724AMDGPUInstructionSelector::matchExtendFromS32OrS32(Register Reg,
3725 bool IsSigned) const {
3726 if (IsSigned)
3727 return matchSignExtendFromS32OrS32(Reg);
3728
3729 return matchZeroExtendFromS32OrS32(Reg);
3730}
3731
3732Register AMDGPUInstructionSelector::matchAnyExtendFromS32(Register Reg) const {
3733 Register AnyExtSrc;
3734 if (mi_match(Reg, *MRI, m_GAnyExt(m_Reg(AnyExtSrc))))
3735 return MRI->getType(AnyExtSrc) == LLT::scalar(32) ? AnyExtSrc : Register();
3736
3737 // Match legalized form %zext = G_MERGE_VALUES (s32 %x), (s32 G_IMPLICIT_DEF)
3738 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3739 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3740 return Register();
3741
3742 assert(Def->getNumOperands() == 3 &&
3743 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3744
3745 if (mi_match(Def->getOperand(2).getReg(), *MRI, m_GImplicitDef()))
3746 return Def->getOperand(1).getReg();
3747
3748 return Register();
3749}
3750
3751bool AMDGPUInstructionSelector::selectGlobalLoadLds(MachineInstr &MI) const{
3752 if (!Subtarget->hasVMemToLDSLoad())
3753 return false;
3754
3755 unsigned Opc;
3756 unsigned Size = MI.getOperand(3).getImm();
3757 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
3758
3759 switch (Size) {
3760 default:
3761 return false;
3762 case 1:
3763 Opc = AMDGPU::GLOBAL_LOAD_LDS_UBYTE;
3764 break;
3765 case 2:
3766 Opc = AMDGPU::GLOBAL_LOAD_LDS_USHORT;
3767 break;
3768 case 4:
3769 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORD;
3770 break;
3771 case 12:
3772 if (!Subtarget->hasLDSLoadB96_B128())
3773 return false;
3774 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX3;
3775 break;
3776 case 16:
3777 if (!Subtarget->hasLDSLoadB96_B128())
3778 return false;
3779 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX4;
3780 break;
3781 }
3782
3783 MachineBasicBlock *MBB = MI.getParent();
3784 const DebugLoc &DL = MI.getDebugLoc();
3785 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3786 .add(MI.getOperand(2));
3787
3788 Register Addr = MI.getOperand(1).getReg();
3789 Register VOffset;
3790 // Try to split SAddr and VOffset. Global and LDS pointers share the same
3791 // immediate offset, so we cannot use a regular SelectGlobalSAddr().
3792 if (!isSGPR(Addr)) {
3793 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
3794 if (isSGPR(AddrDef->Reg)) {
3795 Addr = AddrDef->Reg;
3796 } else if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
3797 Register SAddr =
3798 getSrcRegIgnoringCopies(AddrDef->MI->getOperand(1).getReg(), *MRI);
3799 if (isSGPR(SAddr)) {
3800 Register PtrBaseOffset = AddrDef->MI->getOperand(2).getReg();
3801 if (Register Off = matchZeroExtendFromS32(PtrBaseOffset)) {
3802 Addr = SAddr;
3803 VOffset = Off;
3804 }
3805 }
3806 }
3807 }
3808
3809 if (isSGPR(Addr)) {
3811 if (!VOffset) {
3812 VOffset = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
3813 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_MOV_B32_e32), VOffset)
3814 .addImm(0);
3815 }
3816 }
3817
3818 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc))
3819 .addReg(Addr);
3820
3821 if (isSGPR(Addr))
3822 MIB.addReg(VOffset);
3823
3824 MIB.add(MI.getOperand(4)); // offset
3825
3826 unsigned Aux = MI.getOperand(5).getImm();
3827 MIB.addImm(Aux & ~AMDGPU::CPol::VIRTUAL_BITS); // cpol
3828 MIB.addImm(isAsyncLDSDMA(IntrinsicID));
3829
3830 MachineMemOperand *LoadMMO = *MI.memoperands_begin();
3831 MachinePointerInfo LoadPtrI = LoadMMO->getPointerInfo();
3832 LoadPtrI.Offset = MI.getOperand(4).getImm();
3833 MachinePointerInfo StorePtrI = LoadPtrI;
3834 LoadPtrI.V = PoisonValue::get(PointerType::get(MF->getFunction().getContext(),
3838 auto F = LoadMMO->getFlags() &
3840 LoadMMO = MF->getMachineMemOperand(LoadPtrI, F | MachineMemOperand::MOLoad,
3841 Size, LoadMMO->getBaseAlign());
3842 MachineMemOperand *StoreMMO =
3843 MF->getMachineMemOperand(StorePtrI, F | MachineMemOperand::MOStore,
3844 sizeof(int32_t), Align(4));
3845
3846 MIB.setMemRefs({LoadMMO, StoreMMO});
3847
3848 MI.eraseFromParent();
3849 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3850 return true;
3851}
3852
3853bool AMDGPUInstructionSelector::selectTensorLoadStore(MachineInstr &MI,
3854 Intrinsic::ID IID) const {
3855 bool IsLoad = IID == Intrinsic::amdgcn_tensor_load_to_lds;
3856 unsigned Opc =
3857 IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d4 : AMDGPU::TENSOR_STORE_FROM_LDS_d4;
3858 int NumGroups = 4;
3859
3860 // A lamda function to check whether an operand is a vector of all 0s.
3861 const auto isAllZeros = [&](MachineOperand &Opnd) {
3862 const MachineInstr *DefMI = MRI->getVRegDef(Opnd.getReg());
3863 if (!DefMI)
3864 return false;
3865 return llvm::isBuildVectorAllZeros(*DefMI, *MRI, true);
3866 };
3867
3868 // Use _D2 version if both group 2 and 3 are zero-initialized.
3869 if (isAllZeros(MI.getOperand(3)) && isAllZeros(MI.getOperand(4))) {
3870 NumGroups = 2;
3871 Opc = IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d2
3872 : AMDGPU::TENSOR_STORE_FROM_LDS_d2;
3873 }
3874
3875 // TODO: Handle the fifth group: MI.getOpetand(5), which is silently ignored
3876 // for now because all existing targets only support up to 4 groups.
3877 MachineBasicBlock *MBB = MI.getParent();
3878 auto MIB = BuildMI(*MBB, &MI, MI.getDebugLoc(), TII.get(Opc))
3879 .add(MI.getOperand(1)) // D# group 0
3880 .add(MI.getOperand(2)); // D# group 1
3881
3882 if (NumGroups >= 4) { // Has at least 4 groups
3883 MIB.add(MI.getOperand(3)) // D# group 2
3884 .add(MI.getOperand(4)); // D# group 3
3885 }
3886
3887 MIB.addImm(0) // r128
3888 .add(MI.getOperand(6)); // cpol
3889
3890 MI.eraseFromParent();
3891 return true;
3892}
3893
3894bool AMDGPUInstructionSelector::selectBVHIntersectRayIntrinsic(
3895 MachineInstr &MI) const {
3896 unsigned OpcodeOpIdx =
3897 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY ? 1 : 3;
3898 MI.setDesc(TII.get(MI.getOperand(OpcodeOpIdx).getImm()));
3899 MI.removeOperand(OpcodeOpIdx);
3900 MI.addImplicitDefUseOperands(*MI.getMF());
3901 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
3902 return true;
3903}
3904
3905// FIXME: This should be removed and let the patterns select. We just need the
3906// AGPR/VGPR combination versions.
3907bool AMDGPUInstructionSelector::selectSMFMACIntrin(MachineInstr &MI) const {
3908 unsigned Opc;
3909 switch (cast<GIntrinsic>(MI).getIntrinsicID()) {
3910 case Intrinsic::amdgcn_smfmac_f32_16x16x32_f16:
3911 Opc = AMDGPU::V_SMFMAC_F32_16X16X32_F16_e64;
3912 break;
3913 case Intrinsic::amdgcn_smfmac_f32_32x32x16_f16:
3914 Opc = AMDGPU::V_SMFMAC_F32_32X32X16_F16_e64;
3915 break;
3916 case Intrinsic::amdgcn_smfmac_f32_16x16x32_bf16:
3917 Opc = AMDGPU::V_SMFMAC_F32_16X16X32_BF16_e64;
3918 break;
3919 case Intrinsic::amdgcn_smfmac_f32_32x32x16_bf16:
3920 Opc = AMDGPU::V_SMFMAC_F32_32X32X16_BF16_e64;
3921 break;
3922 case Intrinsic::amdgcn_smfmac_i32_16x16x64_i8:
3923 Opc = AMDGPU::V_SMFMAC_I32_16X16X64_I8_e64;
3924 break;
3925 case Intrinsic::amdgcn_smfmac_i32_32x32x32_i8:
3926 Opc = AMDGPU::V_SMFMAC_I32_32X32X32_I8_e64;
3927 break;
3928 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_bf8:
3929 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF8_BF8_e64;
3930 break;
3931 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_fp8:
3932 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF8_FP8_e64;
3933 break;
3934 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_bf8:
3935 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_FP8_BF8_e64;
3936 break;
3937 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_fp8:
3938 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_FP8_FP8_e64;
3939 break;
3940 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_bf8:
3941 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF8_BF8_e64;
3942 break;
3943 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_fp8:
3944 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF8_FP8_e64;
3945 break;
3946 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_bf8:
3947 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_FP8_BF8_e64;
3948 break;
3949 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_fp8:
3950 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_FP8_FP8_e64;
3951 break;
3952 case Intrinsic::amdgcn_smfmac_f32_16x16x64_f16:
3953 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_F16_e64;
3954 break;
3955 case Intrinsic::amdgcn_smfmac_f32_32x32x32_f16:
3956 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_F16_e64;
3957 break;
3958 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf16:
3959 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF16_e64;
3960 break;
3961 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf16:
3962 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF16_e64;
3963 break;
3964 case Intrinsic::amdgcn_smfmac_i32_16x16x128_i8:
3965 Opc = AMDGPU::V_SMFMAC_I32_16X16X128_I8_e64;
3966 break;
3967 case Intrinsic::amdgcn_smfmac_i32_32x32x64_i8:
3968 Opc = AMDGPU::V_SMFMAC_I32_32X32X64_I8_e64;
3969 break;
3970 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_bf8:
3971 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_BF8_BF8_e64;
3972 break;
3973 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_fp8:
3974 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_BF8_FP8_e64;
3975 break;
3976 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_bf8:
3977 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_FP8_BF8_e64;
3978 break;
3979 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_fp8:
3980 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_FP8_FP8_e64;
3981 break;
3982 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_bf8:
3983 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_BF8_BF8_e64;
3984 break;
3985 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_fp8:
3986 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_BF8_FP8_e64;
3987 break;
3988 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_bf8:
3989 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_FP8_BF8_e64;
3990 break;
3991 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_fp8:
3992 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_FP8_FP8_e64;
3993 break;
3994 default:
3995 llvm_unreachable("unhandled smfmac intrinsic");
3996 }
3997
3998 auto VDst_In = MI.getOperand(4);
3999
4000 MI.setDesc(TII.get(Opc));
4001 MI.removeOperand(4); // VDst_In
4002 MI.removeOperand(1); // Intrinsic ID
4003 MI.addOperand(VDst_In); // Readd VDst_In to the end
4004 MI.addImplicitDefUseOperands(*MI.getMF());
4005 const MCInstrDesc &MCID = MI.getDesc();
4006 if (MCID.getOperandConstraint(0, MCOI::EARLY_CLOBBER) != -1) {
4007 MI.getOperand(0).setIsEarlyClobber(true);
4008 }
4009 return true;
4010}
4011
4012bool AMDGPUInstructionSelector::selectPermlaneSwapIntrin(
4013 MachineInstr &MI, Intrinsic::ID IntrID) const {
4014 if (IntrID == Intrinsic::amdgcn_permlane16_swap &&
4015 !Subtarget->hasPermlane16Swap())
4016 return false;
4017 if (IntrID == Intrinsic::amdgcn_permlane32_swap &&
4018 !Subtarget->hasPermlane32Swap())
4019 return false;
4020
4021 unsigned Opcode = IntrID == Intrinsic::amdgcn_permlane16_swap
4022 ? AMDGPU::V_PERMLANE16_SWAP_B32_e64
4023 : AMDGPU::V_PERMLANE32_SWAP_B32_e64;
4024
4025 MI.removeOperand(2);
4026 MI.setDesc(TII.get(Opcode));
4027 MI.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
4028
4029 MachineOperand &FI = MI.getOperand(4);
4031
4032 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
4033 return true;
4034}
4035
4036bool AMDGPUInstructionSelector::selectWaveAddress(MachineInstr &MI) const {
4037 Register DstReg = MI.getOperand(0).getReg();
4038 Register SrcReg = MI.getOperand(1).getReg();
4039 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4040 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
4041 MachineBasicBlock *MBB = MI.getParent();
4042 const DebugLoc &DL = MI.getDebugLoc();
4043
4044 if (IsVALU) {
4045 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHRREV_B32_e64), DstReg)
4046 .addImm(Subtarget->getWavefrontSizeLog2())
4047 .addReg(SrcReg);
4048 } else {
4049 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::S_LSHR_B32), DstReg)
4050 .addReg(SrcReg)
4051 .addImm(Subtarget->getWavefrontSizeLog2())
4052 .setOperandDead(3); // Dead scc
4053 }
4054
4055 const TargetRegisterClass &RC =
4056 IsVALU ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
4057 if (!RBI.constrainGenericRegister(DstReg, RC, *MRI))
4058 return false;
4059
4060 MI.eraseFromParent();
4061 return true;
4062}
4063
4064bool AMDGPUInstructionSelector::selectWaveShuffleIntrin(
4065 MachineInstr &MI) const {
4066 assert(MI.getNumOperands() == 4);
4067 MachineBasicBlock *MBB = MI.getParent();
4068 const DebugLoc &DL = MI.getDebugLoc();
4069
4070 Register DstReg = MI.getOperand(0).getReg();
4071 Register ValReg = MI.getOperand(2).getReg();
4072 Register IdxReg = MI.getOperand(3).getReg();
4073
4074 const LLT DstTy = MRI->getType(DstReg);
4075 unsigned DstSize = DstTy.getSizeInBits();
4076 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4077 const TargetRegisterClass *DstRC =
4078 TRI.getRegClassForSizeOnBank(DstSize, *DstRB);
4079
4080 if (DstTy != LLT::scalar(32))
4081 return false;
4082
4083 if (!Subtarget->supportsBPermute())
4084 return false;
4085
4086 // If we can bpermute across the whole wave, then just do that
4087 if (Subtarget->supportsWaveWideBPermute()) {
4088 Register ShiftIdxReg = MRI->createVirtualRegister(DstRC);
4089 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), ShiftIdxReg)
4090 .addImm(2)
4091 .addReg(IdxReg);
4092
4093 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), DstReg)
4094 .addReg(ShiftIdxReg)
4095 .addReg(ValReg)
4096 .addImm(0);
4097 } else {
4098 // Otherwise, we need to make use of whole wave mode
4099 assert(Subtarget->isWave64());
4100
4101 // Set inactive lanes to poison
4102 Register UndefValReg =
4103 MRI->createVirtualRegister(TRI.getRegClass(AMDGPU::SReg_32RegClassID));
4104 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefValReg);
4105
4106 Register UndefExecReg = MRI->createVirtualRegister(
4107 TRI.getRegClass(AMDGPU::SReg_64_XEXECRegClassID));
4108 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefExecReg);
4109
4110 Register PoisonValReg = MRI->createVirtualRegister(DstRC);
4111 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32), PoisonValReg)
4112 .addImm(0)
4113 .addReg(ValReg)
4114 .addImm(0)
4115 .addReg(UndefValReg)
4116 .addReg(UndefExecReg);
4117
4118 // ds_bpermute requires index to be multiplied by 4
4119 Register ShiftIdxReg = MRI->createVirtualRegister(DstRC);
4120 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), ShiftIdxReg)
4121 .addImm(2)
4122 .addReg(IdxReg);
4123
4124 Register PoisonIdxReg = MRI->createVirtualRegister(DstRC);
4125 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32), PoisonIdxReg)
4126 .addImm(0)
4127 .addReg(ShiftIdxReg)
4128 .addImm(0)
4129 .addReg(UndefValReg)
4130 .addReg(UndefExecReg);
4131
4132 Register PoisonUnshiftedIdxReg = MRI->createVirtualRegister(DstRC);
4133 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32),
4134 PoisonUnshiftedIdxReg)
4135 .addImm(0)
4136 .addReg(IdxReg)
4137 .addImm(0)
4138 .addReg(UndefValReg)
4139 .addReg(UndefExecReg);
4140
4141 // Get permutation of each half, then we'll select which one to use
4142 Register SameSidePermReg = MRI->createVirtualRegister(DstRC);
4143 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), SameSidePermReg)
4144 .addReg(PoisonIdxReg)
4145 .addReg(PoisonValReg)
4146 .addImm(0);
4147
4148 Register SwappedValReg = MRI->createVirtualRegister(DstRC);
4149 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_PERMLANE64_B32), SwappedValReg)
4150 .addReg(PoisonValReg);
4151
4152 Register OppSidePermReg = MRI->createVirtualRegister(DstRC);
4153 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), OppSidePermReg)
4154 .addReg(PoisonIdxReg)
4155 .addReg(SwappedValReg)
4156 .addImm(0);
4157
4158 Register WWMSwapPermReg = MRI->createVirtualRegister(DstRC);
4159 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::STRICT_WWM), WWMSwapPermReg)
4160 .addReg(OppSidePermReg);
4161
4162 // Select which side to take the permute from
4163 // We can get away with only using mbcnt_lo here since we're only
4164 // trying to detect which side of 32 each lane is on, and mbcnt_lo
4165 // returns 32 for lanes 32-63.
4166 Register ThreadIDReg = MRI->createVirtualRegister(DstRC);
4167 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_MBCNT_LO_U32_B32_e64), ThreadIDReg)
4168 .addImm(-1)
4169 .addImm(0);
4170
4171 Register XORReg = MRI->createVirtualRegister(DstRC);
4172 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_XOR_B32_e64), XORReg)
4173 .addReg(ThreadIDReg)
4174 .addReg(PoisonUnshiftedIdxReg);
4175
4176 Register ANDReg = MRI->createVirtualRegister(DstRC);
4177 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_AND_B32_e64), ANDReg)
4178 .addReg(XORReg)
4179 .addImm(32);
4180
4181 Register CompareReg = MRI->createVirtualRegister(
4182 TRI.getRegClass(AMDGPU::SReg_64_XEXECRegClassID));
4183 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_CMP_EQ_U32_e64), CompareReg)
4184 .addReg(ANDReg)
4185 .addImm(0);
4186
4187 // Finally do the selection
4188 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
4189 .addImm(0)
4190 .addReg(WWMSwapPermReg)
4191 .addImm(0)
4192 .addReg(SameSidePermReg)
4193 .addReg(CompareReg);
4194 }
4195
4196 MI.eraseFromParent();
4197 return true;
4198}
4199
4200// Match BITOP3 operation and return a number of matched instructions plus
4201// truth table.
4202static std::pair<unsigned, uint8_t> BitOp3_Op(Register R,
4204 const MachineRegisterInfo &MRI) {
4205 unsigned NumOpcodes = 0;
4206 uint8_t LHSBits, RHSBits;
4207
4208 auto getOperandBits = [&Src, R, &MRI](Register Op, uint8_t &Bits) -> bool {
4209 // Define truth table given Src0, Src1, Src2 bits permutations:
4210 // 0 0 0
4211 // 0 0 1
4212 // 0 1 0
4213 // 0 1 1
4214 // 1 0 0
4215 // 1 0 1
4216 // 1 1 0
4217 // 1 1 1
4218 const uint8_t SrcBits[3] = { 0xf0, 0xcc, 0xaa };
4219
4220 if (mi_match(Op, MRI, m_AllOnesInt())) {
4221 Bits = 0xff;
4222 return true;
4223 }
4224 if (mi_match(Op, MRI, m_ZeroInt())) {
4225 Bits = 0;
4226 return true;
4227 }
4228
4229 for (unsigned I = 0; I < Src.size(); ++I) {
4230 // Try to find existing reused operand
4231 if (Src[I] == Op) {
4232 Bits = SrcBits[I];
4233 return true;
4234 }
4235 // Try to replace parent operator
4236 if (Src[I] == R) {
4237 Bits = SrcBits[I];
4238 Src[I] = Op;
4239 return true;
4240 }
4241 }
4242
4243 if (Src.size() == 3) {
4244 // No room left for operands. Try one last time, there can be a 'not' of
4245 // one of our source operands. In this case we can compute the bits
4246 // without growing Src vector.
4247 Register LHS;
4248 if (mi_match(Op, MRI, m_Not(m_Reg(LHS)))) {
4250 for (unsigned I = 0; I < Src.size(); ++I) {
4251 if (Src[I] == LHS) {
4252 Bits = ~SrcBits[I];
4253 return true;
4254 }
4255 }
4256 }
4257
4258 return false;
4259 }
4260
4261 Bits = SrcBits[Src.size()];
4262 Src.push_back(Op);
4263 return true;
4264 };
4265
4266 MachineInstr *MI = MRI.getVRegDef(R);
4267 switch (MI->getOpcode()) {
4268 case TargetOpcode::G_AND:
4269 case TargetOpcode::G_OR:
4270 case TargetOpcode::G_XOR: {
4271 Register LHS = getSrcRegIgnoringCopies(MI->getOperand(1).getReg(), MRI);
4272 Register RHS = getSrcRegIgnoringCopies(MI->getOperand(2).getReg(), MRI);
4273
4274 SmallVector<Register, 3> Backup(Src.begin(), Src.end());
4275 if (!getOperandBits(LHS, LHSBits) ||
4276 !getOperandBits(RHS, RHSBits)) {
4277 Src = std::move(Backup);
4278 return std::make_pair(0, 0);
4279 }
4280
4281 // Recursion is naturally limited by the size of the operand vector.
4282 //
4283 // When LHS and RHS share a common sub-expression, one side's recursion
4284 // may decompose that sub-expression and replace the Src slot the other
4285 // side occupies with sub-operands via the "replace parent" path in
4286 // getOperandBits. The other side's cached bit-pattern then refers to a
4287 // slot whose contents changed, producing a wrong truth table.
4288 //
4289 // We detect this in three ways:
4290 // (A) If LHS recursed, its truth table is valid against the Src state
4291 // when LHS recursion completed (SrcAfterLHS). If RHS recursion
4292 // then mutates a Src slot that LHSBits depends on, LHSBits is
4293 // stale.
4294 // (B) If RHS did not recurse, RHSBits came from getOperandBits and
4295 // refers to a specific Src slot. If that slot's contents changed
4296 // (by either recursion), RHSBits is stale.
4297 // (C) Symmetrically for LHS if it did not recurse.
4298 SmallVector<Register, 3> SrcBeforeRecurse(Src.begin(), Src.end());
4299 uint8_t LHSBitsOrig = LHSBits;
4300 uint8_t RHSBitsOrig = RHSBits;
4301
4302 auto LHSOp = BitOp3_Op(LHS, Src, MRI);
4303 if (LHSOp.first) {
4304 NumOpcodes += LHSOp.first;
4305 LHSBits = LHSOp.second;
4306 }
4307
4308 SmallVector<Register, 3> SrcAfterLHS(Src.begin(), Src.end());
4309
4310 auto RHSOp = BitOp3_Op(RHS, Src, MRI);
4311 if (RHSOp.first) {
4312 NumOpcodes += RHSOp.first;
4313 RHSBits = RHSOp.second;
4314 }
4315
4316 // dependsOnSlot: true iff the truth table TT varies with slot Slot.
4317 auto dependsOnSlot = [](uint8_t TT, int Slot) -> bool {
4318 if (Slot < 0 || Slot > 2)
4319 return false;
4320 const uint8_t Masks[3] = {0x0f, 0x33, 0x55};
4321 const int Shifts[3] = {4, 2, 1};
4322 return ((TT ^ (TT >> Shifts[Slot])) & Masks[Slot]) != 0;
4323 };
4324
4325 // findSlot: locate the Src slot a getOperandBits result depends on,
4326 // including negated (NOT) patterns that getOperandBits resolves via
4327 // the ~SrcBits[I] shortcut.
4328 const uint8_t SrcBitsConst[3] = {0xf0, 0xcc, 0xaa};
4329 auto findSlot = [&](uint8_t Bits, Register Op,
4330 const SmallVectorImpl<Register> &S) -> int {
4331 Register NegatedInner;
4332 bool IsNegationOp = mi_match(Op, MRI, m_Not(m_Reg(NegatedInner)));
4333 if (IsNegationOp)
4334 NegatedInner = getSrcRegIgnoringCopies(NegatedInner, MRI);
4335 for (int I = 0; I < (int)S.size(); I++) {
4336 if (Bits == SrcBitsConst[I] && S[I] == Op)
4337 return I;
4338 if (IsNegationOp && Bits == (uint8_t)~SrcBitsConst[I] &&
4339 S[I] == NegatedInner)
4340 return I;
4341 }
4342 return -1;
4343 };
4344
4345 bool Stale = false;
4346
4347 // (A) LHS recursed: its truth table is against SrcAfterLHS.
4348 // Check if RHS recursion mutated a slot that LHSBits uses.
4349 if (LHSOp.first) {
4350 for (int I = 0; I < (int)SrcAfterLHS.size() && I < 3; I++) {
4351 if (I < (int)Src.size() && Src[I] != SrcAfterLHS[I] &&
4352 dependsOnSlot(LHSBits, I)) {
4353 Stale = true;
4354 break;
4355 }
4356 }
4357 }
4358
4359 // (B) RHS did not recurse: RHSBits from getOperandBits is against
4360 // SrcBeforeRecurse. Check if that slot was mutated since then.
4361 if (!Stale && !RHSOp.first) {
4362 int Slot = findSlot(RHSBitsOrig, RHS, SrcBeforeRecurse);
4363 if (Slot >= 0 &&
4364 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4365 Stale = true;
4366 }
4367
4368 // (C) LHS did not recurse: LHSBits from getOperandBits is against
4369 // SrcBeforeRecurse. Check if that slot was mutated since then.
4370 if (!Stale && !LHSOp.first) {
4371 int Slot = findSlot(LHSBitsOrig, LHS, SrcBeforeRecurse);
4372 if (Slot >= 0 &&
4373 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4374 Stale = true;
4375 }
4376
4377 if (Stale) {
4378 Src = std::move(SrcBeforeRecurse);
4379 LHSBits = LHSBitsOrig;
4380 RHSBits = RHSBitsOrig;
4381 NumOpcodes = 0;
4382 }
4383 break;
4384 }
4385 default:
4386 return std::make_pair(0, 0);
4387 }
4388
4389 uint8_t TTbl;
4390 switch (MI->getOpcode()) {
4391 case TargetOpcode::G_AND:
4392 TTbl = LHSBits & RHSBits;
4393 break;
4394 case TargetOpcode::G_OR:
4395 TTbl = LHSBits | RHSBits;
4396 break;
4397 case TargetOpcode::G_XOR:
4398 TTbl = LHSBits ^ RHSBits;
4399 break;
4400 default:
4401 break;
4402 }
4403
4404 return std::make_pair(NumOpcodes + 1, TTbl);
4405}
4406
4407bool AMDGPUInstructionSelector::selectBITOP3(MachineInstr &MI) const {
4408 if (!Subtarget->hasBitOp3Insts())
4409 return false;
4410
4411 Register DstReg = MI.getOperand(0).getReg();
4412 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4413 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
4414 if (!IsVALU)
4415 return false;
4416
4418 uint8_t TTbl;
4419 unsigned NumOpcodes;
4420
4421 std::tie(NumOpcodes, TTbl) = BitOp3_Op(DstReg, Src, *MRI);
4422
4423 // Src.empty() case can happen if all operands are all zero or all ones.
4424 // Normally it shall be optimized out before reaching this.
4425 if (NumOpcodes < 2 || Src.empty())
4426 return false;
4427
4428 // RegBankSelect splits wider VALU logic ops and widens 1-bit ones, so only
4429 // 16 and 32 bit types reach here. Note that <2 x i16> is 32 bits wide.
4430 unsigned Size = MRI->getType(DstReg).getSizeInBits();
4431 assert((Size == 16 || Size == 32) && "unexpected VALU logic op size");
4432 const bool IsB32 = Size == 32;
4433 if (NumOpcodes == 2 && IsB32) {
4434 // Avoid using BITOP3 for OR3, XOR3, AND_OR. This is not faster but makes
4435 // asm more readable. This cannot be modeled with AddedComplexity because
4436 // selector does not know how many operations did we match.
4437 if (mi_match(MI, *MRI, m_GXor(m_GXor(m_Reg(), m_Reg()), m_Reg())) ||
4438 mi_match(MI, *MRI, m_GOr(m_GOr(m_Reg(), m_Reg()), m_Reg())) ||
4439 mi_match(MI, *MRI, m_GOr(m_GAnd(m_Reg(), m_Reg()), m_Reg())))
4440 return false;
4441 } else if (NumOpcodes < 4) {
4442 // For a uniform case threshold should be higher to account for moves
4443 // between VGPRs and SGPRs. It needs one operand in a VGPR, rest two can be
4444 // in SGPRs and a readtfirstlane after.
4445 return false;
4446 }
4447
4448 unsigned Opc = IsB32 ? AMDGPU::V_BITOP3_B32_e64 : AMDGPU::V_BITOP3_B16_e64;
4449 if (!IsB32 && STI.hasTrue16BitInsts())
4450 Opc = STI.useRealTrue16Insts() ? AMDGPU::V_BITOP3_B16_gfx1250_t16_e64
4451 : AMDGPU::V_BITOP3_B16_gfx1250_fake16_e64;
4452 unsigned CBL = STI.getConstantBusLimit(Opc);
4453 MachineBasicBlock *MBB = MI.getParent();
4454 const DebugLoc &DL = MI.getDebugLoc();
4455
4456 for (unsigned I = 0; I < Src.size(); ++I) {
4457 const RegisterBank *RB = RBI.getRegBank(Src[I], *MRI, TRI);
4458 if (RB->getID() != AMDGPU::SGPRRegBankID)
4459 continue;
4460 if (CBL > 0) {
4461 --CBL;
4462 continue;
4463 }
4464 Register NewReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
4465 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::COPY), NewReg)
4466 .addReg(Src[I]);
4467 Src[I] = NewReg;
4468 }
4469
4470 // Last operand can be ignored, turning a ternary operation into a binary.
4471 // For example: (~a & b & c) | (~a & b & ~c) -> (~a & b). We can replace
4472 // 'c' with 'a' here without changing the answer. In some pathological
4473 // cases it should be possible to get an operation with a single operand
4474 // too if optimizer would not catch it.
4475 while (Src.size() < 3)
4476 Src.push_back(Src[0]);
4477
4478 auto MIB = BuildMI(*MBB, MI, DL, TII.get(Opc), DstReg);
4479 if (!IsB32)
4480 MIB.addImm(0); // src_mod0
4481 MIB.addReg(Src[0]);
4482 if (!IsB32)
4483 MIB.addImm(0); // src_mod1
4484 MIB.addReg(Src[1]);
4485 if (!IsB32)
4486 MIB.addImm(0); // src_mod2
4487 MIB.addReg(Src[2])
4488 .addImm(TTbl);
4489 if (!IsB32)
4490 MIB.addImm(0); // op_sel
4491
4492 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
4493 MI.eraseFromParent();
4494
4495 return true;
4496}
4497
4498bool AMDGPUInstructionSelector::selectWriteRegister(MachineInstr &MI) const {
4499 const MDString *RegStr =
4500 cast<MDString>(MI.getOperand(0).getMetadata()->getOperand(0));
4501 Register SrcReg = MI.getOperand(1).getReg();
4502 LLT Ty = MRI->getType(SrcReg);
4503
4504 Register PhysReg = Subtarget->getTargetLowering()->getRegisterByName(
4505 RegStr->getString().data(), Ty, *MF);
4506 if (!PhysReg) {
4507 const Function &Fn = MF->getFunction();
4508 Fn.getContext().diagnose(DiagnosticInfoGenericWithLoc(
4509 "invalid register \"" + Twine(RegStr->getString()) +
4510 "\" for llvm.write_register",
4511 Fn, MI.getDebugLoc()));
4512 MI.eraseFromParent();
4513 return true;
4514 }
4515
4516 if (!RBI.constrainGenericRegister(
4517 SrcReg, *TRI.getSGPRClassForBitWidth(Ty.getSizeInBits()), *MRI))
4518 return false;
4519
4520 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), TII.get(AMDGPU::COPY), PhysReg)
4521 .addReg(SrcReg);
4522 MI.eraseFromParent();
4523 return true;
4524}
4525
4526bool AMDGPUInstructionSelector::selectStackRestore(MachineInstr &MI) const {
4527 Register SrcReg = MI.getOperand(0).getReg();
4528 if (!RBI.constrainGenericRegister(SrcReg, AMDGPU::SReg_32RegClass, *MRI))
4529 return false;
4530
4531 MachineInstr *DefMI = MRI->getVRegDef(SrcReg);
4532 Register SP =
4533 Subtarget->getTargetLowering()->getStackPointerRegisterToSaveRestore();
4534 Register WaveAddr = getWaveAddress(DefMI);
4535 MachineBasicBlock *MBB = MI.getParent();
4536 const DebugLoc &DL = MI.getDebugLoc();
4537
4538 if (!WaveAddr) {
4539 WaveAddr = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
4540 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::S_LSHR_B32), WaveAddr)
4541 .addReg(SrcReg)
4542 .addImm(Subtarget->getWavefrontSizeLog2())
4543 .setOperandDead(3); // Dead scc
4544 }
4545
4546 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), SP)
4547 .addReg(WaveAddr);
4548
4549 MI.eraseFromParent();
4550 return true;
4551}
4552
4554
4555 if (!I.isPreISelOpcode()) {
4556 if (I.isCopy())
4557 return selectCOPY(I);
4558 return true;
4559 }
4560
4561 switch (I.getOpcode()) {
4562 case TargetOpcode::G_AND:
4563 case TargetOpcode::G_OR:
4564 case TargetOpcode::G_XOR:
4565 if (selectBITOP3(I))
4566 return true;
4567 if (selectImpl(I, *CoverageInfo))
4568 return true;
4569 return selectG_AND_OR_XOR(I);
4570 case TargetOpcode::G_ADD:
4571 case TargetOpcode::G_SUB:
4572 case TargetOpcode::G_PTR_ADD:
4573 if (selectImpl(I, *CoverageInfo))
4574 return true;
4575 return selectG_ADD_SUB(I);
4576 case TargetOpcode::G_UADDO:
4577 case TargetOpcode::G_USUBO:
4578 case TargetOpcode::G_UADDE:
4579 case TargetOpcode::G_USUBE:
4580 return selectG_UADDO_USUBO_UADDE_USUBE(I);
4581 case AMDGPU::G_AMDGPU_MAD_U64_U32:
4582 case AMDGPU::G_AMDGPU_MAD_I64_I32:
4583 return selectG_AMDGPU_MAD_64_32(I);
4584 case TargetOpcode::G_INTTOPTR:
4585 case TargetOpcode::G_BITCAST:
4586 case TargetOpcode::G_PTRTOINT:
4587 case TargetOpcode::G_FREEZE:
4588 return selectCOPY(I);
4589 case TargetOpcode::G_FNEG:
4590 if (selectImpl(I, *CoverageInfo))
4591 return true;
4592 return selectG_FNEG(I);
4593 case TargetOpcode::G_FABS:
4594 if (selectImpl(I, *CoverageInfo))
4595 return true;
4596 return selectG_FABS(I);
4597 case TargetOpcode::G_EXTRACT:
4598 return selectG_EXTRACT(I);
4599 case TargetOpcode::G_MERGE_VALUES:
4600 case TargetOpcode::G_CONCAT_VECTORS:
4601 return selectG_MERGE_VALUES(I);
4602 case TargetOpcode::G_UNMERGE_VALUES:
4603 return selectG_UNMERGE_VALUES(I);
4604 case TargetOpcode::G_BUILD_VECTOR:
4605 case TargetOpcode::G_BUILD_VECTOR_TRUNC:
4606 return selectG_BUILD_VECTOR(I);
4607 case TargetOpcode::G_IMPLICIT_DEF:
4608 return selectG_IMPLICIT_DEF(I);
4609 case TargetOpcode::G_INSERT:
4610 return selectG_INSERT(I);
4611 case TargetOpcode::G_INTRINSIC:
4612 case TargetOpcode::G_INTRINSIC_CONVERGENT:
4613 return selectG_INTRINSIC(I);
4614 case TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS:
4615 case TargetOpcode::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS:
4616 return selectG_INTRINSIC_W_SIDE_EFFECTS(I);
4617 case TargetOpcode::G_ICMP:
4618 case TargetOpcode::G_FCMP:
4619 if (selectG_ICMP_or_FCMP(I))
4620 return true;
4621 return selectImpl(I, *CoverageInfo);
4622 case TargetOpcode::G_LOAD:
4623 case TargetOpcode::G_ZEXTLOAD:
4624 case TargetOpcode::G_SEXTLOAD:
4625 case TargetOpcode::G_STORE:
4626 case TargetOpcode::G_ATOMIC_CMPXCHG:
4627 case TargetOpcode::G_ATOMICRMW_XCHG:
4628 case TargetOpcode::G_ATOMICRMW_ADD:
4629 case TargetOpcode::G_ATOMICRMW_SUB:
4630 case TargetOpcode::G_ATOMICRMW_AND:
4631 case TargetOpcode::G_ATOMICRMW_OR:
4632 case TargetOpcode::G_ATOMICRMW_XOR:
4633 case TargetOpcode::G_ATOMICRMW_MIN:
4634 case TargetOpcode::G_ATOMICRMW_MAX:
4635 case TargetOpcode::G_ATOMICRMW_UMIN:
4636 case TargetOpcode::G_ATOMICRMW_UMAX:
4637 case TargetOpcode::G_ATOMICRMW_UINC_WRAP:
4638 case TargetOpcode::G_ATOMICRMW_UDEC_WRAP:
4639 case TargetOpcode::G_ATOMICRMW_USUB_COND:
4640 case TargetOpcode::G_ATOMICRMW_USUB_SAT:
4641 case TargetOpcode::G_ATOMICRMW_FADD:
4642 case TargetOpcode::G_ATOMICRMW_FMIN:
4643 case TargetOpcode::G_ATOMICRMW_FMAX:
4644 return selectG_LOAD_STORE_ATOMICRMW(I);
4645 case TargetOpcode::G_SELECT:
4646 return selectG_SELECT(I);
4647 case TargetOpcode::G_TRUNC:
4648 return selectG_TRUNC(I);
4649 case TargetOpcode::G_SEXT:
4650 case TargetOpcode::G_ZEXT:
4651 case TargetOpcode::G_ANYEXT:
4652 case TargetOpcode::G_SEXT_INREG:
4653 // This is a workaround. For extension from type i1, `selectImpl()` uses
4654 // patterns from TD file and generates an illegal VGPR to SGPR COPY as type
4655 // i1 can only be hold in a SGPR class.
4656 if (MRI->getType(I.getOperand(1).getReg()) != LLT::scalar(1) &&
4657 selectImpl(I, *CoverageInfo))
4658 return true;
4659 return selectG_SZA_EXT(I);
4660 case TargetOpcode::G_FPEXT:
4661 if (selectG_FPEXT(I))
4662 return true;
4663 return selectImpl(I, *CoverageInfo);
4664 case TargetOpcode::G_BRCOND:
4665 return selectG_BRCOND(I);
4666 case TargetOpcode::G_GLOBAL_VALUE:
4667 return selectG_GLOBAL_VALUE(I);
4668 case TargetOpcode::G_PTRMASK:
4669 return selectG_PTRMASK(I);
4670 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
4671 return selectG_EXTRACT_VECTOR_ELT(I);
4672 case TargetOpcode::G_INSERT_VECTOR_ELT:
4673 return selectG_INSERT_VECTOR_ELT(I);
4674 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD:
4675 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16:
4676 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_NORET:
4677 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE:
4678 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16: {
4679 const AMDGPU::ImageDimIntrinsicInfo *Intr =
4681 assert(Intr && "not an image intrinsic with image pseudo");
4682 return selectImageIntrinsic(I, Intr);
4683 }
4684 case AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY:
4685 case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
4686 case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
4687 return selectBVHIntersectRayIntrinsic(I);
4688 case AMDGPU::G_SBFX:
4689 case AMDGPU::G_UBFX:
4690 return selectG_SBFX_UBFX(I);
4691 case AMDGPU::G_SI_CALL:
4692 I.setDesc(TII.get(AMDGPU::SI_CALL));
4693 return true;
4694 case AMDGPU::G_AMDGPU_WAVE_ADDRESS:
4695 return selectWaveAddress(I);
4696 case AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_RETURN: {
4697 I.setDesc(TII.get(AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN));
4698 return true;
4699 }
4700 case AMDGPU::G_STACKRESTORE:
4701 return selectStackRestore(I);
4702 case TargetOpcode::G_WRITE_REGISTER:
4703 return selectWriteRegister(I);
4704 case AMDGPU::G_PHI:
4705 return selectPHI(I);
4706 case AMDGPU::G_AMDGPU_COPY_SCC_VCC:
4707 return selectCOPY_SCC_VCC(I);
4708 case AMDGPU::G_AMDGPU_COPY_VCC_SCC:
4709 return selectCOPY_VCC_SCC(I);
4710 case AMDGPU::G_AMDGPU_READANYLANE:
4711 return selectReadAnyLane(I);
4712 case TargetOpcode::G_CONSTANT:
4713 case TargetOpcode::G_FCONSTANT:
4714 default:
4715 return selectImpl(I, *CoverageInfo);
4716 }
4717 return false;
4718}
4719
4721AMDGPUInstructionSelector::selectVCSRC(MachineOperand &Root) const {
4722 return {{
4723 [=](MachineInstrBuilder &MIB) { MIB.add(Root); }
4724 }};
4725
4726}
4727
4728std::pair<Register, unsigned> AMDGPUInstructionSelector::selectVOP3ModsImpl(
4729 Register Src, bool IsCanonicalizing, bool AllowAbs, bool OpSel) const {
4730 unsigned Mods = 0;
4731 MachineInstr *MI = getDefIgnoringCopies(Src, *MRI);
4732
4733 if (MI->getOpcode() == AMDGPU::G_FNEG) {
4734 Src = MI->getOperand(1).getReg();
4735 Mods |= SISrcMods::NEG;
4736 MI = getDefIgnoringCopies(Src, *MRI);
4737 } else if (MI->getOpcode() == AMDGPU::G_FSUB && IsCanonicalizing) {
4738 // Fold fsub [+-]0 into fneg. This may not have folded depending on the
4739 // denormal mode, but we're implicitly canonicalizing in a source operand.
4740 const ConstantFP *LHS =
4741 getConstantFPVRegVal(MI->getOperand(1).getReg(), *MRI);
4742 if (LHS && LHS->isZero()) {
4743 Mods |= SISrcMods::NEG;
4744 Src = MI->getOperand(2).getReg();
4745 }
4746 }
4747
4748 if (AllowAbs && MI->getOpcode() == AMDGPU::G_FABS) {
4749 Src = MI->getOperand(1).getReg();
4750 Mods |= SISrcMods::ABS;
4751 }
4752
4753 if (OpSel)
4754 Mods |= SISrcMods::OP_SEL_0;
4755
4756 return std::pair(Src, Mods);
4757}
4758
4759std::pair<Register, unsigned>
4760AMDGPUInstructionSelector::selectVOP3PModsF32Impl(Register Src) const {
4761 unsigned Mods;
4762 std::tie(Src, Mods) = selectVOP3ModsImpl(Src);
4763 Mods |= SISrcMods::OP_SEL_1;
4764 return std::pair(Src, Mods);
4765}
4766
4767Register AMDGPUInstructionSelector::copyToVGPRIfSrcFolded(
4768 Register Src, unsigned Mods, MachineOperand Root, MachineInstr *InsertPt,
4769 bool ForceVGPR) const {
4770 if ((Mods != 0 || ForceVGPR) &&
4771 RBI.getRegBank(Src, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID) {
4772
4773 // If we looked through copies to find source modifiers on an SGPR operand,
4774 // we now have an SGPR register source. To avoid potentially violating the
4775 // constant bus restriction, we need to insert a copy to a VGPR.
4776 Register VGPRSrc = MRI->cloneVirtualRegister(Root.getReg());
4777 BuildMI(*InsertPt->getParent(), InsertPt, InsertPt->getDebugLoc(),
4778 TII.get(AMDGPU::COPY), VGPRSrc)
4779 .addReg(Src);
4780 Src = VGPRSrc;
4781 }
4782
4783 return Src;
4784}
4785
4786/// Some instructions must have 32-bit sources. With real true16 instructions a
4787/// 16-bit VALU value lives in a VGPR_16, which they cannot read, so place it in
4788/// the low half of a new 32-bit VGPR.
4790AMDGPUInstructionSelector::widenSrcIfVGPR16(Register Src,
4791 MachineInstr *InsertPt) const {
4792 if (!Subtarget->useRealTrue16Insts() || MRI->getType(Src) != LLT::scalar(16))
4793 return Src;
4794
4795 const RegisterBank *SrcRB = RBI.getRegBank(Src, *MRI, TRI);
4796 if (!SrcRB || SrcRB->getID() != AMDGPU::VGPRRegBankID)
4797 return Src;
4798
4799 MachineIRBuilder B(*InsertPt);
4800
4801 Register ImpDefReg = MRI->createVirtualRegister(&AMDGPU::VGPR_16RegClass);
4802 B.buildInstr(TargetOpcode::IMPLICIT_DEF).addDef(ImpDefReg);
4803
4804 Register DstReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
4805 B.buildInstr(AMDGPU::REG_SEQUENCE)
4806 .addDef(DstReg)
4807 .addReg(Src)
4808 .addImm(AMDGPU::lo16)
4809 .addReg(ImpDefReg)
4810 .addImm(AMDGPU::hi16);
4811
4812 return DstReg;
4813}
4814
4815///
4816/// This will select either an SGPR or VGPR operand and will save us from
4817/// having to write an extra tablegen pattern.
4819AMDGPUInstructionSelector::selectVSRC0(MachineOperand &Root) const {
4820 return {{
4821 [=](MachineInstrBuilder &MIB) { MIB.add(Root); }
4822 }};
4823}
4824
4826AMDGPUInstructionSelector::selectVOP3Mods0(MachineOperand &Root) const {
4827 Register Src;
4828 unsigned Mods;
4829 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
4830
4831 return {{
4832 [=](MachineInstrBuilder &MIB) {
4833 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4834 },
4835 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
4836 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4837 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4838 }};
4839}
4840
4842AMDGPUInstructionSelector::selectVOP3BMods0(MachineOperand &Root) const {
4843 Register Src;
4844 unsigned Mods;
4845 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
4846 /*IsCanonicalizing=*/true,
4847 /*AllowAbs=*/false);
4848
4849 return {{
4850 [=](MachineInstrBuilder &MIB) {
4851 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4852 },
4853 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
4854 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4855 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4856 }};
4857}
4858
4860AMDGPUInstructionSelector::selectVOP3OMods(MachineOperand &Root) const {
4861 return {{
4862 [=](MachineInstrBuilder &MIB) { MIB.add(Root); },
4863 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4864 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4865 }};
4866}
4867
4869AMDGPUInstructionSelector::selectVOP3Mods(MachineOperand &Root) const {
4870 Register Src;
4871 unsigned Mods;
4872 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
4873
4874 return {{
4875 [=](MachineInstrBuilder &MIB) {
4876 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4877 },
4878 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4879 }};
4880}
4881
4883AMDGPUInstructionSelector::selectVOP3ModsNonCanonicalizing(
4884 MachineOperand &Root) const {
4885 Register Src;
4886 unsigned Mods;
4887 std::tie(Src, Mods) =
4888 selectVOP3ModsImpl(Root.getReg(), /*IsCanonicalizing=*/false);
4889
4890 return {{
4891 [=](MachineInstrBuilder &MIB) {
4892 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4893 },
4894 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4895 }};
4896}
4897
4899AMDGPUInstructionSelector::selectVOP3BMods(MachineOperand &Root) const {
4900 Register Src;
4901 unsigned Mods;
4902 std::tie(Src, Mods) =
4903 selectVOP3ModsImpl(Root.getReg(), /*IsCanonicalizing=*/true,
4904 /*AllowAbs=*/false);
4905
4906 return {{
4907 [=](MachineInstrBuilder &MIB) {
4908 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4909 },
4910 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4911 }};
4912}
4913
4915AMDGPUInstructionSelector::selectVOP3NoMods(MachineOperand &Root) const {
4916 Register Reg = Root.getReg();
4917 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
4918 if (Def->getOpcode() == AMDGPU::G_FNEG || Def->getOpcode() == AMDGPU::G_FABS)
4919 return {};
4920 return {{
4921 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
4922 }};
4923}
4924
4925enum class SrcStatus {
4930 // This means current op = [op_upper, op_lower] and src = -op_lower.
4933 // This means current op = [op_upper, op_lower] and src = [op_upper,
4934 // -op_lower].
4942};
4943/// Test if the MI is truncating to half, such as `%reg0:n = G_TRUNC %reg1:2n`
4944static bool isTruncHalf(const MachineInstr *MI,
4945 const MachineRegisterInfo &MRI) {
4946 if (MI->getOpcode() != AMDGPU::G_TRUNC)
4947 return false;
4948
4949 unsigned DstSize = MRI.getType(MI->getOperand(0).getReg()).getSizeInBits();
4950 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4951 return DstSize * 2 == SrcSize;
4952}
4953
4954/// Test if the MI is logic shift right with half bits,
4955/// such as `%reg0:2n =G_LSHR %reg1:2n, CONST(n)`
4956static bool isLshrHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI) {
4957 if (MI->getOpcode() != AMDGPU::G_LSHR)
4958 return false;
4959
4960 Register ShiftSrc;
4961 std::optional<ValueAndVReg> ShiftAmt;
4962 if (mi_match(MI->getOperand(0).getReg(), MRI,
4963 m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt)))) {
4964 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4965 unsigned Shift = ShiftAmt->Value.getZExtValue();
4966 return Shift * 2 == SrcSize;
4967 }
4968 return false;
4969}
4970
4971/// Test if the MI is shift left with half bits,
4972/// such as `%reg0:2n =G_SHL %reg1:2n, CONST(n)`
4973static bool isShlHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI) {
4974 if (MI->getOpcode() != AMDGPU::G_SHL)
4975 return false;
4976
4977 Register ShiftSrc;
4978 std::optional<ValueAndVReg> ShiftAmt;
4979 if (mi_match(MI->getOperand(0).getReg(), MRI,
4980 m_GShl(m_Reg(ShiftSrc), m_GCst(ShiftAmt)))) {
4981 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4982 unsigned Shift = ShiftAmt->Value.getZExtValue();
4983 return Shift * 2 == SrcSize;
4984 }
4985 return false;
4986}
4987
4988/// Test function, if the MI is `%reg0:n, %reg1:n = G_UNMERGE_VALUES %reg2:2n`
4989static bool isUnmergeHalf(const MachineInstr *MI,
4990 const MachineRegisterInfo &MRI) {
4991 if (MI->getOpcode() != AMDGPU::G_UNMERGE_VALUES)
4992 return false;
4993 return MI->getNumOperands() == 3 && MI->getOperand(0).isDef() &&
4994 MI->getOperand(1).isDef() && !MI->getOperand(2).isDef();
4995}
4996
4998
5000 const MachineRegisterInfo &MRI) {
5001 LLT OpTy = MRI.getType(Reg);
5002 if (OpTy.isScalar())
5003 return TypeClass::SCALAR;
5004 if (OpTy.isVector() && OpTy.getNumElements() == 2)
5007}
5008
5010 const MachineRegisterInfo &MRI) {
5011 TypeClass NegType = isVectorOfTwoOrScalar(Reg, MRI);
5012 if (NegType != TypeClass::VECTOR_OF_TWO && NegType != TypeClass::SCALAR)
5013 return SrcStatus::INVALID;
5014
5015 switch (S) {
5016 case SrcStatus::IS_SAME:
5017 if (NegType == TypeClass::VECTOR_OF_TWO) {
5018 // Vector of 2:
5019 // [SrcHi, SrcLo] = [CurrHi, CurrLo]
5020 // [CurrHi, CurrLo] = neg [OpHi, OpLo](2 x Type)
5021 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5022 // [SrcHi, SrcLo] = [-OpHi, -OpLo]
5024 }
5025 if (NegType == TypeClass::SCALAR) {
5026 // Scalar:
5027 // [SrcHi, SrcLo] = [CurrHi, CurrLo]
5028 // [CurrHi, CurrLo] = neg [OpHi, OpLo](Type)
5029 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5030 // [SrcHi, SrcLo] = [-OpHi, OpLo]
5031 return SrcStatus::IS_HI_NEG;
5032 }
5033 break;
5035 if (NegType == TypeClass::VECTOR_OF_TWO) {
5036 // Vector of 2:
5037 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5038 // [CurrHi, CurrLo] = neg [OpHi, OpLo](2 x Type)
5039 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5040 // [SrcHi, SrcLo] = [-(-OpHi), -OpLo] = [OpHi, -OpLo]
5041 return SrcStatus::IS_LO_NEG;
5042 }
5043 if (NegType == TypeClass::SCALAR) {
5044 // Scalar:
5045 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5046 // [CurrHi, CurrLo] = neg [OpHi, OpLo](Type)
5047 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5048 // [SrcHi, SrcLo] = [-(-OpHi), OpLo] = [OpHi, OpLo]
5049 return SrcStatus::IS_SAME;
5050 }
5051 break;
5053 if (NegType == TypeClass::VECTOR_OF_TWO) {
5054 // Vector of 2:
5055 // [SrcHi, SrcLo] = [CurrHi, -CurrLo]
5056 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](2 x Type)
5057 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5058 // [SrcHi, SrcLo] = [-OpHi, -(-OpLo)] = [-OpHi, OpLo]
5059 return SrcStatus::IS_HI_NEG;
5060 }
5061 if (NegType == TypeClass::SCALAR) {
5062 // Scalar:
5063 // [SrcHi, SrcLo] = [CurrHi, -CurrLo]
5064 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](Type)
5065 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5066 // [SrcHi, SrcLo] = [-OpHi, -OpLo]
5068 }
5069 break;
5071 if (NegType == TypeClass::VECTOR_OF_TWO) {
5072 // Vector of 2:
5073 // [SrcHi, SrcLo] = [-CurrHi, -CurrLo]
5074 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](2 x Type)
5075 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5076 // [SrcHi, SrcLo] = [OpHi, OpLo]
5077 return SrcStatus::IS_SAME;
5078 }
5079 if (NegType == TypeClass::SCALAR) {
5080 // Scalar:
5081 // [SrcHi, SrcLo] = [-CurrHi, -CurrLo]
5082 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](Type)
5083 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5084 // [SrcHi, SrcLo] = [OpHi, -OpLo]
5085 return SrcStatus::IS_LO_NEG;
5086 }
5087 break;
5089 // Vector of 2:
5090 // Src = CurrUpper
5091 // Curr = [CurrUpper, CurrLower]
5092 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5093 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5094 // Src = -OpUpper
5095 //
5096 // Scalar:
5097 // Src = CurrUpper
5098 // Curr = [CurrUpper, CurrLower]
5099 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5100 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5101 // Src = -OpUpper
5104 if (NegType == TypeClass::VECTOR_OF_TWO) {
5105 // Vector of 2:
5106 // Src = CurrLower
5107 // Curr = [CurrUpper, CurrLower]
5108 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5109 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5110 // Src = -OpLower
5112 }
5113 if (NegType == TypeClass::SCALAR) {
5114 // Scalar:
5115 // Src = CurrLower
5116 // Curr = [CurrUpper, CurrLower]
5117 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5118 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5119 // Src = OpLower
5121 }
5122 break;
5124 // Vector of 2:
5125 // Src = -CurrUpper
5126 // Curr = [CurrUpper, CurrLower]
5127 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5128 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5129 // Src = -(-OpUpper) = OpUpper
5130 //
5131 // Scalar:
5132 // Src = -CurrUpper
5133 // Curr = [CurrUpper, CurrLower]
5134 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5135 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5136 // Src = -(-OpUpper) = OpUpper
5139 if (NegType == TypeClass::VECTOR_OF_TWO) {
5140 // Vector of 2:
5141 // Src = -CurrLower
5142 // Curr = [CurrUpper, CurrLower]
5143 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5144 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5145 // Src = -(-OpLower) = OpLower
5147 }
5148 if (NegType == TypeClass::SCALAR) {
5149 // Scalar:
5150 // Src = -CurrLower
5151 // Curr = [CurrUpper, CurrLower]
5152 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5153 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5154 // Src = -OpLower
5156 }
5157 break;
5158 default:
5159 break;
5160 }
5161 llvm_unreachable("unexpected SrcStatus & NegType combination");
5162}
5163
5164static std::optional<std::pair<Register, SrcStatus>>
5165calcNextStatus(std::pair<Register, SrcStatus> Curr,
5166 const MachineRegisterInfo &MRI) {
5167 const MachineInstr *MI = MRI.getVRegDef(Curr.first);
5168
5169 unsigned Opc = MI->getOpcode();
5170
5171 // Handle general Opc cases.
5172 switch (Opc) {
5173 case AMDGPU::G_BITCAST:
5174 return std::optional<std::pair<Register, SrcStatus>>(
5175 {MI->getOperand(1).getReg(), Curr.second});
5176 case AMDGPU::COPY:
5177 if (MI->getOperand(1).getReg().isPhysical())
5178 return std::nullopt;
5179 return std::optional<std::pair<Register, SrcStatus>>(
5180 {MI->getOperand(1).getReg(), Curr.second});
5181 case AMDGPU::G_FNEG: {
5182 SrcStatus Stat = getNegStatus(Curr.first, Curr.second, MRI);
5183 if (Stat == SrcStatus::INVALID)
5184 return std::nullopt;
5185 return std::optional<std::pair<Register, SrcStatus>>(
5186 {MI->getOperand(1).getReg(), Stat});
5187 }
5188 default:
5189 break;
5190 }
5191
5192 // Calc next Stat from current Stat.
5193 switch (Curr.second) {
5194 case SrcStatus::IS_SAME:
5195 if (isTruncHalf(MI, MRI))
5196 return std::optional<std::pair<Register, SrcStatus>>(
5197 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF});
5198 else if (isUnmergeHalf(MI, MRI)) {
5199 if (Curr.first == MI->getOperand(0).getReg())
5200 return std::optional<std::pair<Register, SrcStatus>>(
5201 {MI->getOperand(2).getReg(), SrcStatus::IS_LOWER_HALF});
5202 return std::optional<std::pair<Register, SrcStatus>>(
5203 {MI->getOperand(2).getReg(), SrcStatus::IS_UPPER_HALF});
5204 }
5205 break;
5207 if (isTruncHalf(MI, MRI)) {
5208 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5209 // [CurrHi, CurrLo] = trunc [OpUpper, OpLower] = OpLower
5210 // = [OpLowerHi, OpLowerLo]
5211 // Src = [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5212 // = [-OpLowerHi, OpLowerLo]
5213 // = -OpLower
5214 return std::optional<std::pair<Register, SrcStatus>>(
5215 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5216 }
5217 if (isUnmergeHalf(MI, MRI)) {
5218 if (Curr.first == MI->getOperand(0).getReg())
5219 return std::optional<std::pair<Register, SrcStatus>>(
5220 {MI->getOperand(2).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5221 return std::optional<std::pair<Register, SrcStatus>>(
5222 {MI->getOperand(2).getReg(), SrcStatus::IS_UPPER_HALF_NEG});
5223 }
5224 break;
5226 if (isShlHalf(MI, MRI))
5227 return std::optional<std::pair<Register, SrcStatus>>(
5228 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF});
5229 break;
5231 if (isLshrHalf(MI, MRI))
5232 return std::optional<std::pair<Register, SrcStatus>>(
5233 {MI->getOperand(1).getReg(), SrcStatus::IS_UPPER_HALF});
5234 break;
5236 if (isShlHalf(MI, MRI))
5237 return std::optional<std::pair<Register, SrcStatus>>(
5238 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5239 break;
5241 if (isLshrHalf(MI, MRI))
5242 return std::optional<std::pair<Register, SrcStatus>>(
5243 {MI->getOperand(1).getReg(), SrcStatus::IS_UPPER_HALF_NEG});
5244 break;
5245 default:
5246 break;
5247 }
5248 return std::nullopt;
5249}
5250
5251/// This is used to control valid status that current MI supports. For example,
5252/// non floating point intrinsic such as @llvm.amdgcn.sdot2 does not support NEG
5253/// bit on VOP3P.
5254/// The class can be further extended to recognize support on SEL, NEG, ABS bit
5255/// for different MI on different arch
5257private:
5258 bool HasNeg = false;
5259 // Assume all complex pattern of VOP3P have opsel.
5260 bool HasOpsel = true;
5261
5262public:
5264 const MachineInstr *MI = MRI.getVRegDef(Reg);
5265 unsigned Opc = MI->getOpcode();
5266
5267 if (Opc == TargetOpcode::G_INTRINSIC) {
5268 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(*MI).getIntrinsicID();
5269 // Only float point intrinsic has neg & neg_hi bits.
5270 if (IntrinsicID == Intrinsic::amdgcn_fdot2)
5271 HasNeg = true;
5273 // Keep same for generic op.
5274 HasNeg = true;
5275 }
5276 }
5277 bool checkOptions(SrcStatus Stat) const {
5278 if (!HasNeg &&
5279 (Stat >= SrcStatus::NEG_START && Stat <= SrcStatus::NEG_END)) {
5280 return false;
5281 }
5282 if (!HasOpsel &&
5283 (Stat >= SrcStatus::HALF_START && Stat <= SrcStatus::HALF_END)) {
5284 return false;
5285 }
5286 return true;
5287 }
5288};
5289
5292 int MaxDepth = 3) {
5293 int Depth = 0;
5294 auto Curr = calcNextStatus({Reg, SrcStatus::IS_SAME}, MRI);
5296
5297 while (Depth <= MaxDepth && Curr.has_value()) {
5298 Depth++;
5299 if (SO.checkOptions(Curr.value().second))
5300 Statlist.push_back(Curr.value());
5301 Curr = calcNextStatus(Curr.value(), MRI);
5302 }
5303
5304 return Statlist;
5305}
5306
5307static std::pair<Register, SrcStatus>
5309 int MaxDepth = 3) {
5310 int Depth = 0;
5311 std::pair<Register, SrcStatus> LastSameOrNeg = {Reg, SrcStatus::IS_SAME};
5312 auto Curr = calcNextStatus(LastSameOrNeg, MRI);
5313
5314 while (Depth <= MaxDepth && Curr.has_value()) {
5315 Depth++;
5316 SrcStatus Stat = Curr.value().second;
5317 if (SO.checkOptions(Stat)) {
5318 if (Stat == SrcStatus::IS_SAME || Stat == SrcStatus::IS_HI_NEG ||
5320 LastSameOrNeg = Curr.value();
5321 }
5322 Curr = calcNextStatus(Curr.value(), MRI);
5323 }
5324
5325 return LastSameOrNeg;
5326}
5327
5328static bool isSameBitWidth(Register Reg1, Register Reg2,
5329 const MachineRegisterInfo &MRI) {
5330 unsigned Width1 = MRI.getType(Reg1).getSizeInBits();
5331 unsigned Width2 = MRI.getType(Reg2).getSizeInBits();
5332 return Width1 == Width2;
5333}
5334
5335static unsigned updateMods(SrcStatus HiStat, SrcStatus LoStat, unsigned Mods) {
5336 // SrcStatus::IS_LOWER_HALF remain 0.
5337 if (HiStat == SrcStatus::IS_UPPER_HALF_NEG) {
5338 Mods ^= SISrcMods::NEG_HI;
5339 Mods |= SISrcMods::OP_SEL_1;
5340 } else if (HiStat == SrcStatus::IS_UPPER_HALF)
5341 Mods |= SISrcMods::OP_SEL_1;
5342 else if (HiStat == SrcStatus::IS_LOWER_HALF_NEG)
5343 Mods ^= SISrcMods::NEG_HI;
5344 else if (HiStat == SrcStatus::IS_HI_NEG)
5345 Mods ^= SISrcMods::NEG_HI;
5346
5347 if (LoStat == SrcStatus::IS_UPPER_HALF_NEG) {
5348 Mods ^= SISrcMods::NEG;
5349 Mods |= SISrcMods::OP_SEL_0;
5350 } else if (LoStat == SrcStatus::IS_UPPER_HALF)
5351 Mods |= SISrcMods::OP_SEL_0;
5352 else if (LoStat == SrcStatus::IS_LOWER_HALF_NEG)
5353 Mods |= SISrcMods::NEG;
5354 else if (LoStat == SrcStatus::IS_HI_NEG)
5355 Mods ^= SISrcMods::NEG;
5356
5357 return Mods;
5358}
5359
5360static bool isValidToPack(SrcStatus HiStat, SrcStatus LoStat, Register NewReg,
5361 Register RootReg, const SIInstrInfo &TII,
5362 const MachineRegisterInfo &MRI) {
5363 auto IsHalfState = [](SrcStatus S) {
5366 };
5367 return isSameBitWidth(NewReg, RootReg, MRI) && IsHalfState(LoStat) &&
5368 IsHalfState(HiStat);
5369}
5370
5371std::pair<Register, unsigned> AMDGPUInstructionSelector::selectVOP3PModsImpl(
5372 Register RootReg, const MachineRegisterInfo &MRI, bool IsDOT) const {
5373 unsigned Mods = 0;
5374 // No modification if Root type is not form of <2 x Type>.
5375 if (isVectorOfTwoOrScalar(RootReg, MRI) != TypeClass::VECTOR_OF_TWO) {
5376 Mods |= SISrcMods::OP_SEL_1;
5377 return {RootReg, Mods};
5378 }
5379
5380 SearchOptions SO(RootReg, MRI);
5381
5382 std::pair<Register, SrcStatus> Stat = getLastSameOrNeg(RootReg, MRI, SO);
5383
5384 if (Stat.second == SrcStatus::IS_BOTH_NEG)
5386 else if (Stat.second == SrcStatus::IS_HI_NEG)
5387 Mods ^= SISrcMods::NEG_HI;
5388 else if (Stat.second == SrcStatus::IS_LO_NEG)
5389 Mods ^= SISrcMods::NEG;
5390
5391 // 64-bit VOP3P instructions do not have OPSEL or ABS. Bail on v2f64 or v2i64.
5392 // TODO: Select NEG_LO and NEG_HI modifiers from BUILD_VECTOR.
5393 if (MRI.getType(RootReg).getSizeInBits() == 128) {
5394 Mods |= SISrcMods::OP_SEL_1; // Just the default, OPSEL unsupported.
5395 return {Stat.first, Mods};
5396 }
5397
5398 GBuildVector *MI;
5399 if (!mi_match(Stat.first, MRI, m_GBuildVector(MI)) ||
5400 MI->getNumOperands() != 3 || (IsDOT && Subtarget->hasDOTOpSelHazard())) {
5401 Mods |= SISrcMods::OP_SEL_1;
5402 return {Stat.first, Mods};
5403 }
5404
5406 getSrcStats(MI->getOperand(2).getReg(), MRI, SO);
5407
5408 if (StatlistHi.empty()) {
5409 Mods |= SISrcMods::OP_SEL_1;
5410 return {Stat.first, Mods};
5411 }
5412
5414 getSrcStats(MI->getOperand(1).getReg(), MRI, SO);
5415
5416 if (StatlistLo.empty()) {
5417 Mods |= SISrcMods::OP_SEL_1;
5418 return {Stat.first, Mods};
5419 }
5420
5421 for (int I = StatlistHi.size() - 1; I >= 0; I--) {
5422 for (int J = StatlistLo.size() - 1; J >= 0; J--) {
5423 if (StatlistHi[I].first == StatlistLo[J].first &&
5424 isValidToPack(StatlistHi[I].second, StatlistLo[J].second,
5425 StatlistHi[I].first, RootReg, TII, MRI))
5426 return {StatlistHi[I].first,
5427 updateMods(StatlistHi[I].second, StatlistLo[J].second, Mods)};
5428 }
5429 }
5430 // Packed instructions do not have abs modifiers.
5431 Mods |= SISrcMods::OP_SEL_1;
5432
5433 return {Stat.first, Mods};
5434}
5435
5436// Removed unused function `getAllKindImm` to eliminate dead code.
5437
5438static bool checkRB(Register Reg, unsigned int RBNo,
5439 const AMDGPURegisterBankInfo &RBI,
5440 const MachineRegisterInfo &MRI,
5441 const TargetRegisterInfo &TRI) {
5442 const RegisterBank *RB = RBI.getRegBank(Reg, MRI, TRI);
5443 return RB->getID() == RBNo;
5444}
5445
5446// This function is used to get the correct register bank for returned reg.
5447// Assume:
5448// 1. VOP3P is always legal for VGPR.
5449// 2. RootOp's regbank is legal.
5450// Thus
5451// 1. If RootOp is SGPR, then NewOp can be SGPR or VGPR.
5452// 2. If RootOp is VGPR, then NewOp must be VGPR.
5453static Register
5456 const TargetRegisterInfo &TRI, const SIInstrInfo &TII) {
5457 // RootOp can only be VGPR or SGPR (some hand written cases such as.
5458 // inst-select-ashr.v2s16.mir::ashr_v2s16_vs).
5459 if (checkRB(RootReg, AMDGPU::SGPRRegBankID, RBI, MRI, TRI) ||
5460 checkRB(NewReg, AMDGPU::VGPRRegBankID, RBI, MRI, TRI))
5461 return NewReg;
5462
5463 if (mi_match(RootReg, MRI, m_Copy(m_SpecificReg(NewReg)))) {
5464 // RootOp is VGPR, NewOp is not VGPR, but RootOp = COPY NewOp.
5465 return RootReg;
5466 }
5467
5468 Register DstReg = MRI.cloneVirtualRegister(RootReg);
5469 MachineInstrBuilder MIB = BuildMI(*Use.getParent(), Use, Use.getDebugLoc(),
5470 TII.get(AMDGPU::COPY), DstReg)
5471 .addReg(NewReg);
5472
5473 // Only accept VGPR.
5474 return MIB->getOperand(0).getReg();
5475}
5476
5478AMDGPUInstructionSelector::selectVOP3PRetHelper(MachineOperand &Root,
5479 bool IsDOT) const {
5480 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
5481 Register Reg;
5482 unsigned Mods;
5483 std::tie(Reg, Mods) = selectVOP3PModsImpl(Root.getReg(), MRI, IsDOT);
5484
5485 Reg = getLegalRegBank(Reg, Root.getReg(), *Root.getParent(), RBI, MRI, TRI,
5486 TII);
5487 return {{
5488 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
5489 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5490 }};
5491}
5492
5494AMDGPUInstructionSelector::selectVOP3PMods(MachineOperand &Root) const {
5495
5496 return selectVOP3PRetHelper(Root);
5497}
5498
5500AMDGPUInstructionSelector::selectVOP3PModsDOT(MachineOperand &Root) const {
5501
5502 return selectVOP3PRetHelper(Root, true);
5503}
5504
5506AMDGPUInstructionSelector::selectVOP3PNoModsDOT(MachineOperand &Root) const {
5507 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
5508 Register Src;
5509 unsigned Mods;
5510 std::tie(Src, Mods) = selectVOP3PModsImpl(Root.getReg(), MRI, true /*IsDOT*/);
5511 if (Mods != SISrcMods::OP_SEL_1)
5512 return {};
5513
5514 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); }}};
5515}
5516
5518AMDGPUInstructionSelector::selectVOP3PModsF32(MachineOperand &Root) const {
5519 Register Src;
5520 unsigned Mods;
5521 std::tie(Src, Mods) = selectVOP3PModsF32Impl(Root.getReg());
5522
5523 return {{
5524 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5525 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5526 }};
5527}
5528
5530AMDGPUInstructionSelector::selectVOP3PNoModsF32(MachineOperand &Root) const {
5531 Register Src;
5532 unsigned Mods;
5533 std::tie(Src, Mods) = selectVOP3PModsF32Impl(Root.getReg());
5534 if (Mods != SISrcMods::OP_SEL_1)
5535 return {};
5536
5537 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); }}};
5538}
5539
5541AMDGPUInstructionSelector::selectWMMAOpSelVOP3PMods(
5542 MachineOperand &Root) const {
5543 assert((Root.isImm() && (Root.getImm() == -1 || Root.getImm() == 0)) &&
5544 "expected i1 value");
5545 unsigned Mods = SISrcMods::OP_SEL_1;
5546 if (Root.getImm() != 0)
5547 Mods |= SISrcMods::OP_SEL_0;
5548
5549 return {{
5550 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5551 }};
5552}
5553
5555 MachineInstr *InsertPt,
5556 MachineRegisterInfo &MRI) {
5557 const TargetRegisterClass *DstRegClass;
5558 switch (Elts.size()) {
5559 case 8:
5560 DstRegClass = &AMDGPU::VReg_256RegClass;
5561 break;
5562 case 4:
5563 DstRegClass = &AMDGPU::VReg_128RegClass;
5564 break;
5565 case 2:
5566 DstRegClass = &AMDGPU::VReg_64RegClass;
5567 break;
5568 default:
5569 llvm_unreachable("unhandled Reg sequence size");
5570 }
5571
5572 MachineIRBuilder B(*InsertPt);
5573 auto MIB = B.buildInstr(AMDGPU::REG_SEQUENCE)
5574 .addDef(MRI.createVirtualRegister(DstRegClass));
5575 for (unsigned i = 0; i < Elts.size(); ++i) {
5576 MIB.addReg(Elts[i]);
5578 }
5579 return MIB->getOperand(0).getReg();
5580}
5581
5582static void selectWMMAModsNegAbs(unsigned ModOpcode, unsigned &Mods,
5584 MachineInstr *InsertPt,
5585 MachineRegisterInfo &MRI) {
5586 if (ModOpcode == TargetOpcode::G_FNEG) {
5587 Mods |= SISrcMods::NEG;
5588 // Check if all elements also have abs modifier
5589 SmallVector<Register, 8> NegAbsElts;
5590 for (auto El : Elts) {
5591 Register FabsSrc;
5592 if (!mi_match(El, MRI, m_GFabs(m_Reg(FabsSrc))))
5593 break;
5594 NegAbsElts.push_back(FabsSrc);
5595 }
5596 if (Elts.size() != NegAbsElts.size()) {
5597 // Neg
5598 Src = buildRegSequence(Elts, InsertPt, MRI);
5599 } else {
5600 // Neg and Abs
5601 Mods |= SISrcMods::NEG_HI;
5602 Src = buildRegSequence(NegAbsElts, InsertPt, MRI);
5603 }
5604 } else {
5605 assert(ModOpcode == TargetOpcode::G_FABS);
5606 // Abs
5607 Mods |= SISrcMods::NEG_HI;
5608 Src = buildRegSequence(Elts, InsertPt, MRI);
5609 }
5610}
5611
5613AMDGPUInstructionSelector::selectWMMAModsF32NegAbs(MachineOperand &Root) const {
5614 Register Src = Root.getReg();
5615 unsigned Mods = SISrcMods::OP_SEL_1;
5617
5618 GBuildVector *BV;
5619 if (mi_match(Src, *MRI, m_GBuildVector(BV))) {
5620 assert(BV->getNumSources() > 0);
5621 // Based on first element decide which mod we match, neg or abs
5622 MachineInstr *ElF32 = MRI->getVRegDef(BV->getSourceReg(0));
5623 unsigned ModOpcode = (ElF32->getOpcode() == AMDGPU::G_FNEG)
5624 ? AMDGPU::G_FNEG
5625 : AMDGPU::G_FABS;
5626 for (unsigned i = 0; i < BV->getNumSources(); ++i) {
5627 ElF32 = MRI->getVRegDef(BV->getSourceReg(i));
5628 if (ElF32->getOpcode() != ModOpcode)
5629 break;
5630 EltsF32.push_back(ElF32->getOperand(1).getReg());
5631 }
5632
5633 // All elements had ModOpcode modifier
5634 if (BV->getNumSources() == EltsF32.size()) {
5635 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF32, Src, Root.getParent(),
5636 *MRI);
5637 }
5638 }
5639
5640 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5641 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5642}
5643
5645AMDGPUInstructionSelector::selectWMMAModsF16Neg(MachineOperand &Root) const {
5646 Register Src = Root.getReg();
5647 unsigned Mods = SISrcMods::OP_SEL_1;
5648 SmallVector<Register, 8> EltsV2F16;
5649
5650 GConcatVectors *CV;
5651 if (mi_match(Src, *MRI, m_GConcatVectors(CV))) {
5652 for (unsigned i = 0; i < CV->getNumSources(); ++i) {
5653 Register FNegSrc;
5654 if (!mi_match(CV->getSourceReg(i), *MRI, m_GFNeg(m_Reg(FNegSrc))))
5655 break;
5656 EltsV2F16.push_back(FNegSrc);
5657 }
5658
5659 // All elements had ModOpcode modifier
5660 if (CV->getNumSources() == EltsV2F16.size()) {
5661 Mods |= SISrcMods::NEG;
5662 Mods |= SISrcMods::NEG_HI;
5663 Src = buildRegSequence(EltsV2F16, Root.getParent(), *MRI);
5664 }
5665 }
5666
5667 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5668 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5669}
5670
5672AMDGPUInstructionSelector::selectWMMAModsF16NegAbs(MachineOperand &Root) const {
5673 Register Src = Root.getReg();
5674 unsigned Mods = SISrcMods::OP_SEL_1;
5675 SmallVector<Register, 8> EltsV2F16;
5676
5677 GConcatVectors *CV;
5678 if (mi_match(Src, *MRI, m_GConcatVectors(CV))) {
5679 assert(CV->getNumSources() > 0);
5680 MachineInstr *ElV2F16 = MRI->getVRegDef(CV->getSourceReg(0));
5681 // Based on first element decide which mod we match, neg or abs
5682 unsigned ModOpcode = (ElV2F16->getOpcode() == AMDGPU::G_FNEG)
5683 ? AMDGPU::G_FNEG
5684 : AMDGPU::G_FABS;
5685
5686 for (unsigned i = 0; i < CV->getNumSources(); ++i) {
5687 ElV2F16 = MRI->getVRegDef(CV->getSourceReg(i));
5688 if (ElV2F16->getOpcode() != ModOpcode)
5689 break;
5690 EltsV2F16.push_back(ElV2F16->getOperand(1).getReg());
5691 }
5692
5693 // All elements had ModOpcode modifier
5694 if (CV->getNumSources() == EltsV2F16.size()) {
5695 MachineIRBuilder B(*Root.getParent());
5696 selectWMMAModsNegAbs(ModOpcode, Mods, EltsV2F16, Src, Root.getParent(),
5697 *MRI);
5698 }
5699 }
5700
5701 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5702 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5703}
5704
5706AMDGPUInstructionSelector::selectWMMAVISrc(MachineOperand &Root) const {
5707 std::optional<FPValueAndVReg> FPValReg;
5708 if (mi_match(Root.getReg(), *MRI, m_GFCstOrSplat(FPValReg))) {
5709 if (TII.isInlineConstant(FPValReg->Value)) {
5710 return {{[=](MachineInstrBuilder &MIB) {
5711 MIB.addImm(FPValReg->Value.bitcastToAPInt().getSExtValue());
5712 }}};
5713 }
5714 // Non-inlineable splat floats should not fall-through for integer immediate
5715 // checks.
5716 return {};
5717 }
5718
5719 APInt ICst;
5720 if (mi_match(Root.getReg(), *MRI, m_ICstOrSplat(ICst))) {
5721 if (TII.isInlineConstant(ICst)) {
5722 return {
5723 {[=](MachineInstrBuilder &MIB) { MIB.addImm(ICst.getSExtValue()); }}};
5724 }
5725 }
5726
5727 return {};
5728}
5729
5731AMDGPUInstructionSelector::selectSWMMACIndex8(MachineOperand &Root) const {
5732 Register Src =
5733 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5734 unsigned Key = 0;
5735
5736 Register ShiftSrc;
5737 std::optional<ValueAndVReg> ShiftAmt;
5738 if (mi_match(Src, *MRI, m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt))) &&
5739 MRI->getType(ShiftSrc).getSizeInBits() == 32 &&
5740 ShiftAmt->Value.getZExtValue() % 8 == 0) {
5741 Key = ShiftAmt->Value.getZExtValue() / 8;
5742 Src = ShiftSrc;
5743 }
5744
5745 return {{
5746 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5747 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5748 }};
5749}
5750
5752AMDGPUInstructionSelector::selectSWMMACIndex16(MachineOperand &Root) const {
5753
5754 Register Src =
5755 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5756 unsigned Key = 0;
5757
5758 Register ShiftSrc;
5759 std::optional<ValueAndVReg> ShiftAmt;
5760 if (mi_match(Src, *MRI, m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt))) &&
5761 MRI->getType(ShiftSrc).getSizeInBits() == 32 &&
5762 ShiftAmt->Value.getZExtValue() == 16) {
5763 Src = ShiftSrc;
5764 Key = 1;
5765 }
5766
5767 return {{
5768 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5769 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5770 }};
5771}
5772
5774AMDGPUInstructionSelector::selectSWMMACIndex32(MachineOperand &Root) const {
5775 Register Src =
5776 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5777 unsigned Key = 0;
5778
5779 Register S32 = matchZeroExtendFromS32(Src);
5780 if (!S32)
5781 S32 = matchAnyExtendFromS32(Src);
5782
5783 if (S32) {
5784 const MachineInstr *Def = getDefIgnoringCopies(S32, *MRI);
5785 if (Def->getOpcode() == TargetOpcode::G_UNMERGE_VALUES) {
5786 assert(Def->getNumOperands() == 3);
5787 Register DstReg1 = Def->getOperand(1).getReg();
5788 if (mi_match(S32, *MRI,
5789 m_any_of(m_SpecificReg(DstReg1), m_Copy(m_Reg(DstReg1))))) {
5790 Src = Def->getOperand(2).getReg();
5791 Key = 1;
5792 }
5793 }
5794 }
5795
5796 return {{
5797 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5798 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5799 }};
5800}
5801
5803AMDGPUInstructionSelector::selectVOP3OpSelMods(MachineOperand &Root) const {
5804 Register Src;
5805 unsigned Mods;
5806 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
5807
5808 Register ExtractSrc;
5809 if (!Subtarget->useRealTrue16Insts() &&
5810 MRI->getType(Root.getReg()).getSizeInBits() == 16 &&
5811 isExtractHiElt(*MRI, Src, ExtractSrc)) {
5812 Src = ExtractSrc;
5813 Mods |= SISrcMods::OP_SEL_0;
5814 }
5815
5816 return {{
5817 [=](MachineInstrBuilder &MIB) {
5818 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
5819 },
5820 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5821 }};
5822}
5823
5824// FIXME-TRUE16 remove when fake16 is removed
5826AMDGPUInstructionSelector::selectVINTERPMods(MachineOperand &Root) const {
5827 Register Src;
5828 unsigned Mods;
5829 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
5830 /*IsCanonicalizing=*/true,
5831 /*AllowAbs=*/false,
5832 /*OpSel=*/false);
5833
5834 return {{
5835 [=](MachineInstrBuilder &MIB) {
5836 MIB.addReg(
5837 copyToVGPRIfSrcFolded(Src, Mods, Root, MIB, /* ForceVGPR */ true));
5838 },
5839 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
5840 }};
5841}
5842
5844AMDGPUInstructionSelector::selectVINTERPModsHi(MachineOperand &Root) const {
5845 Register Src;
5846 unsigned Mods;
5847 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
5848 /*IsCanonicalizing=*/true,
5849 /*AllowAbs=*/false,
5850 /*OpSel=*/true);
5851
5852 return {{
5853 [=](MachineInstrBuilder &MIB) {
5854 MIB.addReg(
5855 copyToVGPRIfSrcFolded(Src, Mods, Root, MIB, /* ForceVGPR */ true));
5856 },
5857 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
5858 }};
5859}
5860
5861// Given \p Offset and load specified by the \p Root operand check if \p Offset
5862// is a multiple of the load byte size. If it is update \p Offset to a
5863// pre-scaled value and return true.
5864bool AMDGPUInstructionSelector::selectScaleOffset(MachineOperand &Root,
5866 bool IsSigned) const {
5867 if (!Subtarget->hasScaleOffset())
5868 return false;
5869
5870 const MachineInstr &MI = *Root.getParent();
5871 MachineMemOperand *MMO = *MI.memoperands_begin();
5872
5873 if (!MMO->getSize().hasValue())
5874 return false;
5875
5876 uint64_t Size = MMO->getSize().getValue();
5877
5878 Register OffsetReg = matchExtendFromS32OrS32(Offset, IsSigned);
5879 if (!OffsetReg)
5880 OffsetReg = Offset;
5881
5882 if (auto Def = getDefSrcRegIgnoringCopies(OffsetReg, *MRI))
5883 OffsetReg = Def->Reg;
5884
5885 Register Op0;
5886 MachineInstr *Mul;
5887 bool ScaleOffset =
5888 (isPowerOf2_64(Size) &&
5889 mi_match(OffsetReg, *MRI,
5890 m_GShl(m_Reg(Op0),
5893 mi_match(OffsetReg, *MRI,
5895 m_Copy(m_SpecificICst(Size))))) ||
5896 mi_match(
5897 OffsetReg, *MRI,
5898 m_BinOp(IsSigned ? AMDGPU::S_MUL_I64_I32_PSEUDO : AMDGPU::S_MUL_U64,
5899 m_Reg(Op0), m_SpecificICst(Size))) ||
5900 // Match G_AMDGPU_MAD_U64_U32 offset, c, 0
5901 (mi_match(OffsetReg, *MRI, m_MInstr(Mul)) &&
5902 (Mul->getOpcode() == (IsSigned ? AMDGPU::G_AMDGPU_MAD_I64_I32
5903 : AMDGPU::G_AMDGPU_MAD_U64_U32) ||
5904 (IsSigned && Mul->getOpcode() == AMDGPU::G_AMDGPU_MAD_U64_U32 &&
5905 VT->signBitIsZero(Mul->getOperand(2).getReg()))) &&
5906 mi_match(Mul->getOperand(4).getReg(), *MRI, m_ZeroInt()) &&
5907 mi_match(Mul->getOperand(3).getReg(), *MRI,
5909 m_Copy(m_SpecificICst(Size))))) &&
5910 mi_match(Mul->getOperand(2).getReg(), *MRI, m_Reg(Op0)));
5911
5912 if (ScaleOffset)
5913 Offset = Op0;
5914
5915 return ScaleOffset;
5916}
5917
5918bool AMDGPUInstructionSelector::selectSmrdOffset(MachineOperand &Root,
5919 Register &Base,
5920 Register *SOffset,
5921 int64_t *Offset,
5922 bool *ScaleOffset) const {
5923 MachineInstr *MI = Root.getParent();
5924 MachineBasicBlock *MBB = MI->getParent();
5925
5926 // FIXME: We should shrink the GEP if the offset is known to be <= 32-bits,
5927 // then we can select all ptr + 32-bit offsets.
5928 SmallVector<GEPInfo, 4> AddrInfo;
5929 getAddrModeInfo(*MI, *MRI, AddrInfo);
5930
5931 if (AddrInfo.empty())
5932 return false;
5933
5934 const GEPInfo &GEPI = AddrInfo[0];
5935 std::optional<int64_t> EncodedImm;
5936
5937 if (ScaleOffset)
5938 *ScaleOffset = false;
5939
5940 if (SOffset && Offset) {
5941 EncodedImm = AMDGPU::getSMRDEncodedOffset(STI, GEPI.Imm, /*IsBuffer=*/false,
5942 /*HasSOffset=*/true);
5943 if (GEPI.SgprParts.size() == 1 && GEPI.Imm != 0 && EncodedImm &&
5944 AddrInfo.size() > 1) {
5945 const GEPInfo &GEPI2 = AddrInfo[1];
5946 if (GEPI2.SgprParts.size() == 2 && GEPI2.Imm == 0) {
5947 Register OffsetReg = GEPI2.SgprParts[1];
5948 if (ScaleOffset)
5949 *ScaleOffset =
5950 selectScaleOffset(Root, OffsetReg, false /* IsSigned */);
5951 OffsetReg = matchZeroExtendFromS32OrS32(OffsetReg);
5952 if (OffsetReg) {
5953 Base = GEPI2.SgprParts[0];
5954 *SOffset = OffsetReg;
5955 *Offset = *EncodedImm;
5956 if (*Offset >= 0 || !AMDGPU::hasSMRDSignedImmOffset(STI))
5957 return true;
5958
5959 // For unbuffered smem loads, it is illegal for the Immediate Offset
5960 // to be negative if the resulting (Offset + (M0 or SOffset or zero)
5961 // is negative. Handle the case where the Immediate Offset + SOffset
5962 // is negative.
5963 auto SKnown = VT->getKnownBits(*SOffset);
5964 if (*Offset + SKnown.getMinValue().getSExtValue() < 0)
5965 return false;
5966
5967 return true;
5968 }
5969 }
5970 }
5971 return false;
5972 }
5973
5974 EncodedImm = AMDGPU::getSMRDEncodedOffset(STI, GEPI.Imm, /*IsBuffer=*/false,
5975 /*HasSOffset=*/false);
5976 if (Offset && GEPI.SgprParts.size() == 1 && EncodedImm) {
5977 Base = GEPI.SgprParts[0];
5978 *Offset = *EncodedImm;
5979 return true;
5980 }
5981
5982 // SGPR offset is unsigned.
5983 if (SOffset && GEPI.SgprParts.size() == 1 && isUInt<32>(GEPI.Imm) &&
5984 GEPI.Imm != 0) {
5985 // If we make it this far we have a load with an 32-bit immediate offset.
5986 // It is OK to select this using a sgpr offset, because we have already
5987 // failed trying to select this load into one of the _IMM variants since
5988 // the _IMM Patterns are considered before the _SGPR patterns.
5989 Base = GEPI.SgprParts[0];
5990 *SOffset = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
5991 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), *SOffset)
5992 .addImm(GEPI.Imm);
5993 return true;
5994 }
5995
5996 if (SOffset && GEPI.SgprParts.size() && GEPI.Imm == 0) {
5997 Register OffsetReg = GEPI.SgprParts[1];
5998 if (ScaleOffset)
5999 *ScaleOffset = selectScaleOffset(Root, OffsetReg, false /* IsSigned */);
6000 OffsetReg = matchZeroExtendFromS32OrS32(OffsetReg);
6001 if (OffsetReg) {
6002 Base = GEPI.SgprParts[0];
6003 *SOffset = OffsetReg;
6004 return true;
6005 }
6006 }
6007
6008 return false;
6009}
6010
6012AMDGPUInstructionSelector::selectSmrdImm(MachineOperand &Root) const {
6013 Register Base;
6014 int64_t Offset;
6015 if (!selectSmrdOffset(Root, Base, /* SOffset= */ nullptr, &Offset,
6016 /* ScaleOffset */ nullptr))
6017 return std::nullopt;
6018
6019 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
6020 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }}};
6021}
6022
6024AMDGPUInstructionSelector::selectSmrdImm32(MachineOperand &Root) const {
6025 SmallVector<GEPInfo, 4> AddrInfo;
6026 getAddrModeInfo(*Root.getParent(), *MRI, AddrInfo);
6027
6028 if (AddrInfo.empty() || AddrInfo[0].SgprParts.size() != 1)
6029 return std::nullopt;
6030
6031 const GEPInfo &GEPInfo = AddrInfo[0];
6032 Register PtrReg = GEPInfo.SgprParts[0];
6033 std::optional<int64_t> EncodedImm =
6034 AMDGPU::getSMRDEncodedLiteralOffset32(STI, GEPInfo.Imm);
6035 if (!EncodedImm)
6036 return std::nullopt;
6037
6038 return {{
6039 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrReg); },
6040 [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); }
6041 }};
6042}
6043
6045AMDGPUInstructionSelector::selectSmrdSgpr(MachineOperand &Root) const {
6046 Register Base, SOffset;
6047 bool ScaleOffset;
6048 if (!selectSmrdOffset(Root, Base, &SOffset, /* Offset= */ nullptr,
6049 &ScaleOffset))
6050 return std::nullopt;
6051
6052 unsigned CPol = ScaleOffset ? AMDGPU::CPol::SCAL : 0;
6053 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
6054 [=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
6055 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); }}};
6056}
6057
6059AMDGPUInstructionSelector::selectSmrdSgprImm(MachineOperand &Root) const {
6060 Register Base, SOffset;
6061 int64_t Offset;
6062 bool ScaleOffset;
6063 if (!selectSmrdOffset(Root, Base, &SOffset, &Offset, &ScaleOffset))
6064 return std::nullopt;
6065
6066 unsigned CPol = ScaleOffset ? AMDGPU::CPol::SCAL : 0;
6067 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
6068 [=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
6069 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); },
6070 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); }}};
6071}
6072
6073std::pair<Register, int> AMDGPUInstructionSelector::selectFlatOffsetImpl(
6074 MachineOperand &Root, AMDGPU::FlatAddrSpace FlatVariant) const {
6075 MachineInstr *MI = Root.getParent();
6076
6077 auto Default = std::pair(Root.getReg(), 0);
6078
6079 if (!STI.hasFlatInstOffsets())
6080 return Default;
6081
6082 Register PtrBase;
6083 int64_t ConstOffset;
6084 bool IsInBounds;
6085 std::tie(PtrBase, ConstOffset, IsInBounds) =
6086 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6087
6088 // Adding the offset to the base address with an immediate in a FLAT
6089 // instruction must not change the memory aperture in which the address falls.
6090 // Therefore we can only fold offsets from inbounds GEPs into FLAT
6091 // instructions.
6092 if (ConstOffset == 0 ||
6093 (FlatVariant == AMDGPU::FlatAddrSpace::FlatScratch &&
6094 !isFlatScratchBaseLegal(Root.getReg())) ||
6095 (FlatVariant == AMDGPU::FlatAddrSpace::FLAT && !IsInBounds))
6096 return Default;
6097
6098 unsigned AddrSpace = (*MI->memoperands_begin())->getAddrSpace();
6099 if (!TII.isLegalFLATOffset(ConstOffset, AddrSpace, FlatVariant))
6100 return Default;
6101
6102 return std::pair(PtrBase, ConstOffset);
6103}
6104
6106AMDGPUInstructionSelector::selectFlatOffset(MachineOperand &Root) const {
6107 auto PtrWithOffset = selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FLAT);
6108
6109 return {{
6110 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6111 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6112 }};
6113}
6114
6116AMDGPUInstructionSelector::selectGlobalOffset(MachineOperand &Root) const {
6117 auto PtrWithOffset =
6118 selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FlatGlobal);
6119
6120 return {{
6121 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6122 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6123 }};
6124}
6125
6127AMDGPUInstructionSelector::selectScratchOffset(MachineOperand &Root) const {
6128 auto PtrWithOffset =
6129 selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FlatScratch);
6130
6131 return {{
6132 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6133 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6134 }};
6135}
6136
6137// Match (64-bit SGPR base) + (zext vgpr offset) + sext(imm offset)
6139AMDGPUInstructionSelector::selectGlobalSAddr(MachineOperand &Root,
6140 unsigned CPolBits,
6141 bool NeedIOffset) const {
6142 Register Addr = Root.getReg();
6143 Register PtrBase;
6144 int64_t ConstOffset;
6145 int64_t ImmOffset = 0;
6146
6147 // Match the immediate offset first, which canonically is moved as low as
6148 // possible.
6149 std::tie(PtrBase, ConstOffset, std::ignore) =
6150 getPtrBaseWithConstantOffset(Addr, *MRI);
6151
6152 if (ConstOffset != 0) {
6153 if (NeedIOffset &&
6154 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::GLOBAL_ADDRESS,
6156 Addr = PtrBase;
6157 ImmOffset = ConstOffset;
6158 } else {
6159 auto PtrBaseDef = getDefSrcRegIgnoringCopies(PtrBase, *MRI);
6160 if (isSGPR(PtrBaseDef->Reg)) {
6161 if (ConstOffset > 0) {
6162 // Offset is too large.
6163 //
6164 // saddr + large_offset -> saddr +
6165 // (voffset = large_offset & ~MaxOffset) +
6166 // (large_offset & MaxOffset);
6167 int64_t SplitImmOffset = 0, RemainderOffset = ConstOffset;
6168 if (NeedIOffset) {
6169 std::tie(SplitImmOffset, RemainderOffset) =
6170 TII.splitFlatOffset(ConstOffset, AMDGPUAS::GLOBAL_ADDRESS,
6172 }
6173
6174 if (Subtarget->hasSignedGVSOffset() ? isInt<32>(RemainderOffset)
6175 : isUInt<32>(RemainderOffset)) {
6176 MachineInstr *MI = Root.getParent();
6177 MachineBasicBlock *MBB = MI->getParent();
6178 Register HighBits =
6179 MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6180
6181 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32),
6182 HighBits)
6183 .addImm(RemainderOffset);
6184
6185 if (NeedIOffset)
6186 return {{
6187 [=](MachineInstrBuilder &MIB) {
6188 MIB.addReg(PtrBase);
6189 }, // saddr
6190 [=](MachineInstrBuilder &MIB) {
6191 MIB.addReg(HighBits);
6192 }, // voffset
6193 [=](MachineInstrBuilder &MIB) { MIB.addImm(SplitImmOffset); },
6194 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); },
6195 }};
6196 return {{
6197 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrBase); }, // saddr
6198 [=](MachineInstrBuilder &MIB) {
6199 MIB.addReg(HighBits);
6200 }, // voffset
6201 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); },
6202 }};
6203 }
6204 }
6205
6206 // We are adding a 64 bit SGPR and a constant. If constant bus limit
6207 // is 1 we would need to perform 1 or 2 extra moves for each half of
6208 // the constant and it is better to do a scalar add and then issue a
6209 // single VALU instruction to materialize zero. Otherwise it is less
6210 // instructions to perform VALU adds with immediates or inline literals.
6211 unsigned NumLiterals =
6212 !TII.isInlineConstant(APInt(32, Lo_32(ConstOffset))) +
6213 !TII.isInlineConstant(APInt(32, Hi_32(ConstOffset)));
6214 if (STI.getConstantBusLimit(AMDGPU::V_ADD_U32_e64) > NumLiterals)
6215 return std::nullopt;
6216 }
6217 }
6218 }
6219
6220 // Match the variable offset.
6221 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6222 if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
6223 // Look through the SGPR->VGPR copy.
6224 Register SAddr =
6225 getSrcRegIgnoringCopies(AddrDef->MI->getOperand(1).getReg(), *MRI);
6226
6227 if (isSGPR(SAddr)) {
6228 Register PtrBaseOffset = AddrDef->MI->getOperand(2).getReg();
6229
6230 // It's possible voffset is an SGPR here, but the copy to VGPR will be
6231 // inserted later.
6232 bool ScaleOffset = selectScaleOffset(Root, PtrBaseOffset,
6233 Subtarget->hasSignedGVSOffset());
6234 if (Register VOffset = matchExtendFromS32OrS32(
6235 PtrBaseOffset, Subtarget->hasSignedGVSOffset())) {
6236 if (NeedIOffset)
6237 return {{[=](MachineInstrBuilder &MIB) { // saddr
6238 MIB.addReg(SAddr);
6239 },
6240 [=](MachineInstrBuilder &MIB) { // voffset
6241 MIB.addReg(VOffset);
6242 },
6243 [=](MachineInstrBuilder &MIB) { // offset
6244 MIB.addImm(ImmOffset);
6245 },
6246 [=](MachineInstrBuilder &MIB) { // cpol
6247 MIB.addImm(CPolBits |
6248 (ScaleOffset ? AMDGPU::CPol::SCAL : 0));
6249 }}};
6250 return {{[=](MachineInstrBuilder &MIB) { // saddr
6251 MIB.addReg(SAddr);
6252 },
6253 [=](MachineInstrBuilder &MIB) { // voffset
6254 MIB.addReg(VOffset);
6255 },
6256 [=](MachineInstrBuilder &MIB) { // cpol
6257 MIB.addImm(CPolBits |
6258 (ScaleOffset ? AMDGPU::CPol::SCAL : 0));
6259 }}};
6260 }
6261 }
6262 }
6263
6264 // FIXME: We should probably have folded COPY (G_IMPLICIT_DEF) earlier, and
6265 // drop this.
6266 if (AddrDef->MI->getOpcode() == AMDGPU::G_IMPLICIT_DEF ||
6267 AddrDef->MI->getOpcode() == AMDGPU::G_CONSTANT || !isSGPR(AddrDef->Reg))
6268 return std::nullopt;
6269
6270 // It's cheaper to materialize a single 32-bit zero for vaddr than the two
6271 // moves required to copy a 64-bit SGPR to VGPR.
6272 MachineInstr *MI = Root.getParent();
6273 MachineBasicBlock *MBB = MI->getParent();
6274 Register VOffset = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6275
6276 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32), VOffset)
6277 .addImm(0);
6278
6279 if (NeedIOffset)
6280 return {{
6281 [=](MachineInstrBuilder &MIB) { MIB.addReg(AddrDef->Reg); }, // saddr
6282 [=](MachineInstrBuilder &MIB) { MIB.addReg(VOffset); }, // voffset
6283 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6284 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); } // cpol
6285 }};
6286 return {{
6287 [=](MachineInstrBuilder &MIB) { MIB.addReg(AddrDef->Reg); }, // saddr
6288 [=](MachineInstrBuilder &MIB) { MIB.addReg(VOffset); }, // voffset
6289 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); } // cpol
6290 }};
6291}
6292
6294AMDGPUInstructionSelector::selectGlobalSAddr(MachineOperand &Root) const {
6295 return selectGlobalSAddr(Root, 0);
6296}
6297
6299AMDGPUInstructionSelector::selectGlobalSAddrCPol(MachineOperand &Root) const {
6300 const MachineInstr &I = *Root.getParent();
6301
6302 // We are assuming CPol is always the last operand of the intrinsic.
6303 auto PassedCPol =
6304 I.getOperand(I.getNumOperands() - 1).getImm() & ~AMDGPU::CPol::SCAL;
6305 return selectGlobalSAddr(Root, PassedCPol);
6306}
6307
6309AMDGPUInstructionSelector::selectGlobalSAddrCPolM0(MachineOperand &Root) const {
6310 const MachineInstr &I = *Root.getParent();
6311
6312 // We are assuming CPol is second from last operand of the intrinsic.
6313 auto PassedCPol =
6314 I.getOperand(I.getNumOperands() - 2).getImm() & ~AMDGPU::CPol::SCAL;
6315 return selectGlobalSAddr(Root, PassedCPol);
6316}
6317
6319AMDGPUInstructionSelector::selectGlobalSAddrGLC(MachineOperand &Root) const {
6320 return selectGlobalSAddr(Root, AMDGPU::CPol::GLC);
6321}
6322
6324AMDGPUInstructionSelector::selectGlobalSAddrNoIOffset(
6325 MachineOperand &Root) const {
6326 const MachineInstr &I = *Root.getParent();
6327
6328 // We are assuming CPol is always the last operand of the intrinsic.
6329 auto PassedCPol =
6330 I.getOperand(I.getNumOperands() - 1).getImm() & ~AMDGPU::CPol::SCAL;
6331 return selectGlobalSAddr(Root, PassedCPol, false);
6332}
6333
6335AMDGPUInstructionSelector::selectGlobalSAddrNoIOffsetM0(
6336 MachineOperand &Root) const {
6337 const MachineInstr &I = *Root.getParent();
6338
6339 // We are assuming CPol is second from last operand of the intrinsic.
6340 auto PassedCPol =
6341 I.getOperand(I.getNumOperands() - 2).getImm() & ~AMDGPU::CPol::SCAL;
6342 return selectGlobalSAddr(Root, PassedCPol, false);
6343}
6344
6346AMDGPUInstructionSelector::selectScratchSAddr(MachineOperand &Root) const {
6347 Register Addr = Root.getReg();
6348 Register PtrBase;
6349 int64_t ConstOffset;
6350 int64_t ImmOffset = 0;
6351
6352 // Match the immediate offset first, which canonically is moved as low as
6353 // possible.
6354 std::tie(PtrBase, ConstOffset, std::ignore) =
6355 getPtrBaseWithConstantOffset(Addr, *MRI);
6356
6357 if (ConstOffset != 0 && isFlatScratchBaseLegal(Addr) &&
6358 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::PRIVATE_ADDRESS,
6360 Addr = PtrBase;
6361 ImmOffset = ConstOffset;
6362 }
6363
6364 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6365 if (AddrDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX) {
6366 int FI = AddrDef->MI->getOperand(1).getIndex();
6367 return {{
6368 [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); }, // saddr
6369 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); } // offset
6370 }};
6371 }
6372
6373 Register SAddr = AddrDef->Reg;
6374
6375 if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
6376 Register LHS = AddrDef->MI->getOperand(1).getReg();
6377 Register RHS = AddrDef->MI->getOperand(2).getReg();
6378 auto LHSDef = getDefSrcRegIgnoringCopies(LHS, *MRI);
6379 auto RHSDef = getDefSrcRegIgnoringCopies(RHS, *MRI);
6380
6381 if (LHSDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX &&
6382 isSGPR(RHSDef->Reg)) {
6383 int FI = LHSDef->MI->getOperand(1).getIndex();
6384 MachineInstr &I = *Root.getParent();
6385 MachineBasicBlock *BB = I.getParent();
6386 const DebugLoc &DL = I.getDebugLoc();
6387 SAddr = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
6388
6389 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADD_I32), SAddr)
6390 .addFrameIndex(FI)
6391 .addReg(RHSDef->Reg)
6392 .setOperandDead(3); // Dead scc
6393 }
6394 }
6395
6396 if (!isSGPR(SAddr))
6397 return std::nullopt;
6398
6399 return {{
6400 [=](MachineInstrBuilder &MIB) { MIB.addReg(SAddr); }, // saddr
6401 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); } // offset
6402 }};
6403}
6404
6405// Check whether the flat scratch SVS swizzle bug affects this access.
6406bool AMDGPUInstructionSelector::checkFlatScratchSVSSwizzleBug(
6407 Register VAddr, Register SAddr, uint64_t ImmOffset) const {
6408 if (!Subtarget->hasFlatScratchSVSSwizzleBug())
6409 return false;
6410
6411 // The bug affects the swizzling of SVS accesses if there is any carry out
6412 // from the two low order bits (i.e. from bit 1 into bit 2) when adding
6413 // voffset to (soffset + inst_offset).
6414 auto VKnown = VT->getKnownBits(VAddr);
6415 auto SKnown = KnownBits::add(VT->getKnownBits(SAddr),
6416 KnownBits::makeConstant(APInt(32, ImmOffset)));
6417 uint64_t VMax = VKnown.getMaxValue().getZExtValue();
6418 uint64_t SMax = SKnown.getMaxValue().getZExtValue();
6419 return (VMax & 3) + (SMax & 3) >= 4;
6420}
6421
6423AMDGPUInstructionSelector::selectScratchSVAddr(MachineOperand &Root) const {
6424 Register Addr = Root.getReg();
6425 Register PtrBase;
6426 int64_t ConstOffset;
6427 int64_t ImmOffset = 0;
6428
6429 // Match the immediate offset first, which canonically is moved as low as
6430 // possible.
6431 std::tie(PtrBase, ConstOffset, std::ignore) =
6432 getPtrBaseWithConstantOffset(Addr, *MRI);
6433
6434 Register OrigAddr = Addr;
6435 if (ConstOffset != 0 &&
6436 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::PRIVATE_ADDRESS,
6438 Addr = PtrBase;
6439 ImmOffset = ConstOffset;
6440 }
6441
6442 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6443 if (AddrDef->MI->getOpcode() != AMDGPU::G_PTR_ADD)
6444 return std::nullopt;
6445
6446 Register RHS = AddrDef->MI->getOperand(2).getReg();
6447 if (RBI.getRegBank(RHS, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID)
6448 return std::nullopt;
6449
6450 Register LHS = AddrDef->MI->getOperand(1).getReg();
6451 auto LHSDef = getDefSrcRegIgnoringCopies(LHS, *MRI);
6452
6453 if (OrigAddr != Addr) {
6454 if (!isFlatScratchBaseLegalSVImm(OrigAddr))
6455 return std::nullopt;
6456 } else {
6457 if (!isFlatScratchBaseLegalSV(OrigAddr))
6458 return std::nullopt;
6459 }
6460
6461 if (checkFlatScratchSVSSwizzleBug(RHS, LHS, ImmOffset))
6462 return std::nullopt;
6463
6464 unsigned CPol = selectScaleOffset(Root, RHS, true /* IsSigned */)
6466 : 0;
6467
6468 if (LHSDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX) {
6469 int FI = LHSDef->MI->getOperand(1).getIndex();
6470 return {{
6471 [=](MachineInstrBuilder &MIB) { MIB.addReg(RHS); }, // vaddr
6472 [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); }, // saddr
6473 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6474 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); } // cpol
6475 }};
6476 }
6477
6478 if (!isSGPR(LHS))
6479 if (auto Def = getDefSrcRegIgnoringCopies(LHS, *MRI))
6480 LHS = Def->Reg;
6481
6482 if (!isSGPR(LHS))
6483 return std::nullopt;
6484
6485 return {{
6486 [=](MachineInstrBuilder &MIB) { MIB.addReg(RHS); }, // vaddr
6487 [=](MachineInstrBuilder &MIB) { MIB.addReg(LHS); }, // saddr
6488 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6489 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); } // cpol
6490 }};
6491}
6492
6494AMDGPUInstructionSelector::selectMUBUFScratchOffen(MachineOperand &Root) const {
6495 MachineInstr *MI = Root.getParent();
6496 MachineBasicBlock *MBB = MI->getParent();
6498 const SIMachineFunctionInfo *Info = MF->getInfo<SIMachineFunctionInfo>();
6499
6500 int64_t Offset = 0;
6501 if (mi_match(Root.getReg(), *MRI, m_ICst(Offset)) &&
6503 Register HighBits = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6504
6505 // TODO: Should this be inside the render function? The iterator seems to
6506 // move.
6507 const int64_t MaxOffset = SIInstrInfo::getMaxMUBUFImmOffset(*Subtarget);
6508 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32),
6509 HighBits)
6510 .addImm(Offset & ~MaxOffset);
6511
6512 return {{[=](MachineInstrBuilder &MIB) { // rsrc
6513 MIB.addReg(Info->getScratchRSrcReg());
6514 },
6515 [=](MachineInstrBuilder &MIB) { // vaddr
6516 MIB.addReg(HighBits);
6517 },
6518 [=](MachineInstrBuilder &MIB) { // soffset
6519 // Use constant zero for soffset and rely on eliminateFrameIndex
6520 // to choose the appropriate frame register if need be.
6521 MIB.addImm(0);
6522 },
6523 [=](MachineInstrBuilder &MIB) { // offset
6524 MIB.addImm(Offset & MaxOffset);
6525 }}};
6526 }
6527
6528 assert(Offset == 0 || Offset == -1);
6529
6530 // Try to fold a frame index directly into the MUBUF vaddr field, and any
6531 // offsets.
6532 std::optional<int> FI;
6533 Register VAddr = Root.getReg();
6534
6535 Register PtrBase;
6536 int64_t ConstOffset;
6537 std::tie(PtrBase, ConstOffset, std::ignore) =
6538 getPtrBaseWithConstantOffset(VAddr, *MRI);
6539 int MatchedFI;
6540 if (ConstOffset != 0) {
6541 if (TII.isLegalMUBUFImmOffset(ConstOffset) &&
6542 (!STI.privateMemoryResourceIsRangeChecked() ||
6543 VT->signBitIsZero(PtrBase))) {
6544 if (mi_match(PtrBase, *MRI, m_GFrameIndex(MatchedFI)))
6545 FI = MatchedFI;
6546 else
6547 VAddr = PtrBase;
6548 Offset = ConstOffset;
6549 }
6550 } else if (mi_match(Root.getReg(), *MRI, m_GFrameIndex(MatchedFI))) {
6551 FI = MatchedFI;
6552 }
6553
6554 return {{[=](MachineInstrBuilder &MIB) { // rsrc
6555 MIB.addReg(Info->getScratchRSrcReg());
6556 },
6557 [=](MachineInstrBuilder &MIB) { // vaddr
6558 if (FI)
6559 MIB.addFrameIndex(*FI);
6560 else
6561 MIB.addReg(VAddr);
6562 },
6563 [=](MachineInstrBuilder &MIB) { // soffset
6564 // Use constant zero for soffset and rely on eliminateFrameIndex
6565 // to choose the appropriate frame register if need be.
6566 MIB.addImm(0);
6567 },
6568 [=](MachineInstrBuilder &MIB) { // offset
6569 MIB.addImm(Offset);
6570 }}};
6571}
6572
6573bool AMDGPUInstructionSelector::isDSOffsetLegal(Register Base,
6574 int64_t Offset) const {
6575 if (!isUInt<16>(Offset))
6576 return false;
6577
6578 if (STI.hasUsableDSOffset() || STI.unsafeDSOffsetFoldingEnabled())
6579 return true;
6580
6581 // On Southern Islands instruction with a negative base value and an offset
6582 // don't seem to work.
6583 return VT->signBitIsZero(Base);
6584}
6585
6586bool AMDGPUInstructionSelector::isDSOffset2Legal(Register Base, int64_t Offset0,
6587 int64_t Offset1,
6588 unsigned Size) const {
6589 if (Offset0 % Size != 0 || Offset1 % Size != 0)
6590 return false;
6591 if (!isUInt<8>(Offset0 / Size) || !isUInt<8>(Offset1 / Size))
6592 return false;
6593
6594 if (STI.hasUsableDSOffset() || STI.unsafeDSOffsetFoldingEnabled())
6595 return true;
6596
6597 // On Southern Islands instruction with a negative base value and an offset
6598 // don't seem to work.
6599 return VT->signBitIsZero(Base);
6600}
6601
6602// Return whether the operation has NoUnsignedWrap property.
6603static bool isNoUnsignedWrap(MachineInstr *Addr) {
6604 return Addr->getOpcode() == TargetOpcode::G_OR ||
6605 (Addr->getOpcode() == TargetOpcode::G_PTR_ADD &&
6607}
6608
6609// Check that the base address of flat scratch load/store in the form of `base +
6610// offset` is legal to be put in SGPR/VGPR (i.e. unsigned per hardware
6611// requirement). We always treat the first operand as the base address here.
6612bool AMDGPUInstructionSelector::isFlatScratchBaseLegal(Register Addr) const {
6613 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6614
6615 if (isNoUnsignedWrap(AddrMI))
6616 return true;
6617
6618 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6619 // values.
6620 if (STI.hasSignedScratchOffsets())
6621 return true;
6622
6623 Register LHS = AddrMI->getOperand(1).getReg();
6624 Register RHS = AddrMI->getOperand(2).getReg();
6625
6626 if (AddrMI->getOpcode() == TargetOpcode::G_PTR_ADD) {
6627 std::optional<ValueAndVReg> RhsValReg =
6629 // If the immediate offset is negative and within certain range, the base
6630 // address cannot also be negative. If the base is also negative, the sum
6631 // would be either negative or much larger than the valid range of scratch
6632 // memory a thread can access.
6633 if (RhsValReg && RhsValReg->Value.getSExtValue() < 0 &&
6634 RhsValReg->Value.getSExtValue() > -0x40000000)
6635 return true;
6636 }
6637
6638 return VT->signBitIsZero(LHS);
6639}
6640
6641// Check address value in SGPR/VGPR are legal for flat scratch in the form
6642// of: SGPR + VGPR.
6643bool AMDGPUInstructionSelector::isFlatScratchBaseLegalSV(Register Addr) const {
6644 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6645
6646 if (isNoUnsignedWrap(AddrMI))
6647 return true;
6648
6649 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6650 // values.
6651 if (STI.hasSignedScratchOffsets())
6652 return true;
6653
6654 Register LHS = AddrMI->getOperand(1).getReg();
6655 Register RHS = AddrMI->getOperand(2).getReg();
6656 return VT->signBitIsZero(RHS) && VT->signBitIsZero(LHS);
6657}
6658
6659// Check address value in SGPR/VGPR are legal for flat scratch in the form
6660// of: SGPR + VGPR + Imm.
6661bool AMDGPUInstructionSelector::isFlatScratchBaseLegalSVImm(
6662 Register Addr) const {
6663 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6664 // values.
6665 if (STI.hasSignedScratchOffsets())
6666 return true;
6667
6668 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6669 Register Base = AddrMI->getOperand(1).getReg();
6670 std::optional<DefinitionAndSourceRegister> BaseDef =
6672 std::optional<ValueAndVReg> RHSOffset =
6674 assert(RHSOffset);
6675
6676 // If the immediate offset is negative and within certain range, the base
6677 // address cannot also be negative. If the base is also negative, the sum
6678 // would be either negative or much larger than the valid range of scratch
6679 // memory a thread can access.
6680 if (isNoUnsignedWrap(BaseDef->MI) &&
6681 (isNoUnsignedWrap(AddrMI) ||
6682 (RHSOffset->Value.getSExtValue() < 0 &&
6683 RHSOffset->Value.getSExtValue() > -0x40000000)))
6684 return true;
6685
6686 Register LHS = BaseDef->MI->getOperand(1).getReg();
6687 Register RHS = BaseDef->MI->getOperand(2).getReg();
6688 return VT->signBitIsZero(RHS) && VT->signBitIsZero(LHS);
6689}
6690
6691bool AMDGPUInstructionSelector::isUnneededShiftMask(const MachineInstr &MI,
6692 unsigned ShAmtBits) const {
6693 assert(MI.getOpcode() == TargetOpcode::G_AND);
6694
6695 std::optional<APInt> RHS =
6696 getIConstantVRegVal(MI.getOperand(2).getReg(), *MRI);
6697 if (!RHS)
6698 return false;
6699
6700 if (RHS->countr_one() >= ShAmtBits)
6701 return true;
6702
6703 const APInt &LHSKnownZeros = VT->getKnownZeroes(MI.getOperand(1).getReg());
6704 return (LHSKnownZeros | *RHS).countr_one() >= ShAmtBits;
6705}
6706
6708AMDGPUInstructionSelector::selectMUBUFScratchOffset(
6709 MachineOperand &Root) const {
6710 Register Reg = Root.getReg();
6711 const SIMachineFunctionInfo *Info = MF->getInfo<SIMachineFunctionInfo>();
6712
6713 std::optional<DefinitionAndSourceRegister> Def =
6715 assert(Def && "this shouldn't be an optional result");
6716 Reg = Def->Reg;
6717
6718 if (Register WaveBase = getWaveAddress(Def->MI)) {
6719 return {{
6720 [=](MachineInstrBuilder &MIB) { // rsrc
6721 MIB.addReg(Info->getScratchRSrcReg());
6722 },
6723 [=](MachineInstrBuilder &MIB) { // soffset
6724 MIB.addReg(WaveBase);
6725 },
6726 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // offset
6727 }};
6728 }
6729
6730 int64_t Offset = 0;
6731
6732 // FIXME: Copy check is a hack
6734 if (mi_match(Reg, *MRI,
6735 m_GPtrAdd(m_Reg(BasePtr),
6737 if (!TII.isLegalMUBUFImmOffset(Offset))
6738 return {};
6739 MachineInstr *BasePtrDef = getDefIgnoringCopies(BasePtr, *MRI);
6740 Register WaveBase = getWaveAddress(BasePtrDef);
6741 if (!WaveBase)
6742 return {};
6743
6744 return {{
6745 [=](MachineInstrBuilder &MIB) { // rsrc
6746 MIB.addReg(Info->getScratchRSrcReg());
6747 },
6748 [=](MachineInstrBuilder &MIB) { // soffset
6749 MIB.addReg(WaveBase);
6750 },
6751 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); } // offset
6752 }};
6753 }
6754
6755 if (!mi_match(Root.getReg(), *MRI, m_ICst(Offset)) ||
6756 !TII.isLegalMUBUFImmOffset(Offset))
6757 return {};
6758
6759 return {{
6760 [=](MachineInstrBuilder &MIB) { // rsrc
6761 MIB.addReg(Info->getScratchRSrcReg());
6762 },
6763 [=](MachineInstrBuilder &MIB) { // soffset
6764 MIB.addImm(0);
6765 },
6766 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); } // offset
6767 }};
6768}
6769
6770std::pair<Register, unsigned>
6771AMDGPUInstructionSelector::selectDS1Addr1OffsetImpl(
6772 MachineOperand &Root) const {
6773 int64_t ConstAddr = 0;
6774
6775 Register PtrBase;
6776 int64_t Offset;
6777 std::tie(PtrBase, Offset, std::ignore) =
6778 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6779
6780 if (Offset) {
6781 if (isDSOffsetLegal(PtrBase, Offset)) {
6782 // (add n0, c0)
6783 return std::pair(PtrBase, Offset);
6784 }
6785 } else if (mi_match(Root.getReg(), *MRI, m_GSub(m_Reg(), m_Reg()))) {
6786 // TODO
6787
6788 } else if (mi_match(Root.getReg(), *MRI, m_ICst(ConstAddr))) {
6789 // TODO
6790 }
6791
6792 return std::pair(Root.getReg(), 0);
6793}
6794
6796AMDGPUInstructionSelector::selectDS1Addr1Offset(MachineOperand &Root) const {
6797 Register Reg;
6798 unsigned Offset;
6799 std::tie(Reg, Offset) = selectDS1Addr1OffsetImpl(Root);
6800 return {{
6801 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
6802 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }
6803 }};
6804}
6805
6807AMDGPUInstructionSelector::selectDS64Bit4ByteAligned(MachineOperand &Root) const {
6808 return selectDSReadWrite2(Root, 4);
6809}
6810
6812AMDGPUInstructionSelector::selectDS128Bit8ByteAligned(MachineOperand &Root) const {
6813 return selectDSReadWrite2(Root, 8);
6814}
6815
6817AMDGPUInstructionSelector::selectDSReadWrite2(MachineOperand &Root,
6818 unsigned Size) const {
6819 Register Reg;
6820 unsigned Offset;
6821 std::tie(Reg, Offset) = selectDSReadWrite2Impl(Root, Size);
6822 return {{
6823 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
6824 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); },
6825 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset+1); }
6826 }};
6827}
6828
6829std::pair<Register, unsigned>
6830AMDGPUInstructionSelector::selectDSReadWrite2Impl(MachineOperand &Root,
6831 unsigned Size) const {
6832 int64_t ConstAddr = 0;
6833
6834 Register PtrBase;
6835 int64_t Offset;
6836 std::tie(PtrBase, Offset, std::ignore) =
6837 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6838
6839 if (Offset) {
6840 int64_t OffsetValue0 = Offset;
6841 int64_t OffsetValue1 = Offset + Size;
6842 if (isDSOffset2Legal(PtrBase, OffsetValue0, OffsetValue1, Size)) {
6843 // (add n0, c0)
6844 return std::pair(PtrBase, OffsetValue0 / Size);
6845 }
6846 } else if (mi_match(Root.getReg(), *MRI, m_GSub(m_Reg(), m_Reg()))) {
6847 // TODO
6848
6849 } else if (mi_match(Root.getReg(), *MRI, m_ICst(ConstAddr))) {
6850 // TODO
6851 }
6852
6853 return std::pair(Root.getReg(), 0);
6854}
6855
6856/// If \p Root is a G_PTR_ADD with a G_CONSTANT on the right hand side, return
6857/// the base value with the constant offset, and if the offset computation is
6858/// known to be inbounds. There may be intervening copies between \p Root and
6859/// the identified constant. Returns \p Root, 0, false if this does not match
6860/// the pattern.
6861std::tuple<Register, int64_t, bool>
6862AMDGPUInstructionSelector::getPtrBaseWithConstantOffset(
6863 Register Root, const MachineRegisterInfo &MRI) const {
6864 MachineInstr *RootI = getDefIgnoringCopies(Root, MRI);
6865 if (RootI->getOpcode() != TargetOpcode::G_PTR_ADD)
6866 return {Root, 0, false};
6867
6868 MachineOperand &RHS = RootI->getOperand(2);
6869 std::optional<ValueAndVReg> MaybeOffset =
6871 if (!MaybeOffset)
6872 return {Root, 0, false};
6873 bool IsInBounds = RootI->getFlag(MachineInstr::MIFlag::InBounds);
6874 return {RootI->getOperand(1).getReg(), MaybeOffset->Value.getSExtValue(),
6875 IsInBounds};
6876}
6877
6879 MIB.addImm(0);
6880}
6881
6882/// Return a resource descriptor for use with an arbitrary 64-bit pointer. If \p
6883/// BasePtr is not valid, a null base pointer will be used.
6885 uint32_t FormatLo, uint32_t FormatHi,
6886 Register BasePtr) {
6887 Register RSrc2 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6888 Register RSrc3 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6889 Register RSrcHi = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
6890 Register RSrc = MRI.createVirtualRegister(&AMDGPU::SGPR_128RegClass);
6891
6892 B.buildInstr(AMDGPU::S_MOV_B32)
6893 .addDef(RSrc2)
6894 .addImm(FormatLo);
6895 B.buildInstr(AMDGPU::S_MOV_B32)
6896 .addDef(RSrc3)
6897 .addImm(FormatHi);
6898
6899 // Build the half of the subregister with the constants before building the
6900 // full 128-bit register. If we are building multiple resource descriptors,
6901 // this will allow CSEing of the 2-component register.
6902 B.buildInstr(AMDGPU::REG_SEQUENCE)
6903 .addDef(RSrcHi)
6904 .addReg(RSrc2)
6905 .addImm(AMDGPU::sub0)
6906 .addReg(RSrc3)
6907 .addImm(AMDGPU::sub1);
6908
6909 Register RSrcLo = BasePtr;
6910 if (!BasePtr) {
6911 RSrcLo = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
6912 B.buildInstr(AMDGPU::S_MOV_B64)
6913 .addDef(RSrcLo)
6914 .addImm(0);
6915 }
6916
6917 B.buildInstr(AMDGPU::REG_SEQUENCE)
6918 .addDef(RSrc)
6919 .addReg(RSrcLo)
6920 .addImm(AMDGPU::sub0_sub1)
6921 .addReg(RSrcHi)
6922 .addImm(AMDGPU::sub2_sub3);
6923
6924 return RSrc;
6925}
6926
6928 const SIInstrInfo &TII, Register BasePtr) {
6929 uint64_t DefaultFormat = TII.getDefaultRsrcDataFormat();
6930
6931 // FIXME: Why are half the "default" bits ignored based on the addressing
6932 // mode?
6933 return buildRSRC(B, MRI, 0, Hi_32(DefaultFormat), BasePtr);
6934}
6935
6937 const SIInstrInfo &TII, Register BasePtr) {
6938 uint64_t DefaultFormat = TII.getDefaultRsrcDataFormat();
6939
6940 // FIXME: Why are half the "default" bits ignored based on the addressing
6941 // mode?
6942 return buildRSRC(B, MRI, -1, Hi_32(DefaultFormat), BasePtr);
6943}
6944
6945AMDGPUInstructionSelector::MUBUFAddressData
6946AMDGPUInstructionSelector::parseMUBUFAddress(Register Src) const {
6947 MUBUFAddressData Data;
6948 Data.N0 = Src;
6949
6950 Register PtrBase;
6951 int64_t Offset;
6952
6953 std::tie(PtrBase, Offset, std::ignore) =
6954 getPtrBaseWithConstantOffset(Src, *MRI);
6955 if (isUInt<32>(Offset)) {
6956 Data.N0 = PtrBase;
6957 Data.Offset = Offset;
6958 }
6959
6960 if (MachineInstr *InputAdd
6961 = getOpcodeDef(TargetOpcode::G_PTR_ADD, Data.N0, *MRI)) {
6962 Data.N2 = InputAdd->getOperand(1).getReg();
6963 Data.N3 = InputAdd->getOperand(2).getReg();
6964
6965 // FIXME: Need to fix extra SGPR->VGPRcopies inserted
6966 // FIXME: Don't know this was defined by operand 0
6967 //
6968 // TODO: Remove this when we have copy folding optimizations after
6969 // RegBankSelect.
6970 Data.N2 = getDefIgnoringCopies(Data.N2, *MRI)->getOperand(0).getReg();
6971 Data.N3 = getDefIgnoringCopies(Data.N3, *MRI)->getOperand(0).getReg();
6972 }
6973
6974 return Data;
6975}
6976
6977/// Return if the addr64 mubuf mode should be used for the given address.
6978bool AMDGPUInstructionSelector::shouldUseAddr64(MUBUFAddressData Addr) const {
6979 // (ptr_add N2, N3) -> addr64, or
6980 // (ptr_add (ptr_add N2, N3), C1) -> addr64
6981 if (Addr.N2)
6982 return true;
6983
6984 const RegisterBank *N0Bank = RBI.getRegBank(Addr.N0, *MRI, TRI);
6985 return N0Bank->getID() == AMDGPU::VGPRRegBankID;
6986}
6987
6988/// Split an immediate offset \p ImmOffset depending on whether it fits in the
6989/// immediate field. Modifies \p ImmOffset and sets \p SOffset to the variable
6990/// component.
6991void AMDGPUInstructionSelector::splitIllegalMUBUFOffset(
6992 MachineIRBuilder &B, Register &SOffset, int64_t &ImmOffset) const {
6993 if (TII.isLegalMUBUFImmOffset(ImmOffset))
6994 return;
6995
6996 // Illegal offset, store it in soffset.
6997 SOffset = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
6998 B.buildInstr(AMDGPU::S_MOV_B32)
6999 .addDef(SOffset)
7000 .addImm(ImmOffset);
7001 ImmOffset = 0;
7002}
7003
7004bool AMDGPUInstructionSelector::selectMUBUFAddr64Impl(
7005 MachineOperand &Root, Register &VAddr, Register &RSrcReg,
7006 Register &SOffset, int64_t &Offset) const {
7007 // FIXME: Predicates should stop this from reaching here.
7008 // addr64 bit was removed for volcanic islands.
7009 if (!STI.hasAddr64() || STI.useFlatForGlobal())
7010 return false;
7011
7012 MUBUFAddressData AddrData = parseMUBUFAddress(Root.getReg());
7013 if (!shouldUseAddr64(AddrData))
7014 return false;
7015
7016 Register N0 = AddrData.N0;
7017 Register N2 = AddrData.N2;
7018 Register N3 = AddrData.N3;
7019 Offset = AddrData.Offset;
7020
7021 // Base pointer for the SRD.
7022 Register SRDPtr;
7023
7024 if (N2) {
7025 if (RBI.getRegBank(N2, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
7026 assert(N3);
7027 if (RBI.getRegBank(N3, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
7028 // Both N2 and N3 are divergent. Use N0 (the result of the add) as the
7029 // addr64, and construct the default resource from a 0 address.
7030 VAddr = N0;
7031 } else {
7032 SRDPtr = N3;
7033 VAddr = N2;
7034 }
7035 } else {
7036 // N2 is not divergent.
7037 SRDPtr = N2;
7038 VAddr = N3;
7039 }
7040 } else if (RBI.getRegBank(N0, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
7041 // Use the default null pointer in the resource
7042 VAddr = N0;
7043 } else {
7044 // N0 -> offset, or
7045 // (N0 + C1) -> offset
7046 SRDPtr = N0;
7047 }
7048
7049 MachineIRBuilder B(*Root.getParent());
7050 RSrcReg = buildAddr64RSrc(B, *MRI, TII, SRDPtr);
7051 splitIllegalMUBUFOffset(B, SOffset, Offset);
7052 return true;
7053}
7054
7055bool AMDGPUInstructionSelector::selectMUBUFOffsetImpl(
7056 MachineOperand &Root, Register &RSrcReg, Register &SOffset,
7057 int64_t &Offset) const {
7058
7059 // FIXME: Pattern should not reach here.
7060 if (STI.useFlatForGlobal())
7061 return false;
7062
7063 MUBUFAddressData AddrData = parseMUBUFAddress(Root.getReg());
7064 if (shouldUseAddr64(AddrData))
7065 return false;
7066
7067 // N0 -> offset, or
7068 // (N0 + C1) -> offset
7069 Register SRDPtr = AddrData.N0;
7070 Offset = AddrData.Offset;
7071
7072 // TODO: Look through extensions for 32-bit soffset.
7073 MachineIRBuilder B(*Root.getParent());
7074
7075 RSrcReg = buildOffsetSrc(B, *MRI, TII, SRDPtr);
7076 splitIllegalMUBUFOffset(B, SOffset, Offset);
7077 return true;
7078}
7079
7081AMDGPUInstructionSelector::selectMUBUFAddr64(MachineOperand &Root) const {
7082 Register VAddr;
7083 Register RSrcReg;
7084 Register SOffset;
7085 int64_t Offset = 0;
7086
7087 if (!selectMUBUFAddr64Impl(Root, VAddr, RSrcReg, SOffset, Offset))
7088 return {};
7089
7090 // FIXME: Use defaulted operands for trailing 0s and remove from the complex
7091 // pattern.
7092 return {{
7093 [=](MachineInstrBuilder &MIB) { // rsrc
7094 MIB.addReg(RSrcReg);
7095 },
7096 [=](MachineInstrBuilder &MIB) { // vaddr
7097 MIB.addReg(VAddr);
7098 },
7099 [=](MachineInstrBuilder &MIB) { // soffset
7100 if (SOffset)
7101 MIB.addReg(SOffset);
7102 else if (STI.hasRestrictedSOffset())
7103 MIB.addReg(AMDGPU::SGPR_NULL);
7104 else
7105 MIB.addImm(0);
7106 },
7107 [=](MachineInstrBuilder &MIB) { // offset
7108 MIB.addImm(Offset);
7109 },
7110 addZeroImm, // cpol
7111 addZeroImm, // tfe
7112 addZeroImm // swz
7113 }};
7114}
7115
7117AMDGPUInstructionSelector::selectMUBUFOffset(MachineOperand &Root) const {
7118 Register RSrcReg;
7119 Register SOffset;
7120 int64_t Offset = 0;
7121
7122 if (!selectMUBUFOffsetImpl(Root, RSrcReg, SOffset, Offset))
7123 return {};
7124
7125 return {{
7126 [=](MachineInstrBuilder &MIB) { // rsrc
7127 MIB.addReg(RSrcReg);
7128 },
7129 [=](MachineInstrBuilder &MIB) { // soffset
7130 if (SOffset)
7131 MIB.addReg(SOffset);
7132 else if (STI.hasRestrictedSOffset())
7133 MIB.addReg(AMDGPU::SGPR_NULL);
7134 else
7135 MIB.addImm(0);
7136 },
7137 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }, // offset
7138 addZeroImm, // cpol
7139 addZeroImm, // tfe
7140 addZeroImm, // swz
7141 }};
7142}
7143
7145AMDGPUInstructionSelector::selectBUFSOffset(MachineOperand &Root) const {
7146
7147 Register SOffset = Root.getReg();
7148
7149 if (STI.hasRestrictedSOffset() && mi_match(SOffset, *MRI, m_ZeroInt()))
7150 SOffset = AMDGPU::SGPR_NULL;
7151
7152 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); }}};
7153}
7154
7155/// Get an immediate that must be 32-bits, and treated as zero extended.
7156static std::optional<uint64_t>
7158 // getIConstantVRegVal sexts any values, so see if that matters.
7159 std::optional<int64_t> OffsetVal = getIConstantVRegSExtVal(Reg, MRI);
7160 if (!OffsetVal || !isInt<32>(*OffsetVal))
7161 return std::nullopt;
7162 return Lo_32(*OffsetVal);
7163}
7164
7166AMDGPUInstructionSelector::selectSMRDBufferImm(MachineOperand &Root) const {
7167 std::optional<uint64_t> OffsetVal =
7168 Root.isImm() ? Root.getImm() : getConstantZext32Val(Root.getReg(), *MRI);
7169 if (!OffsetVal)
7170 return {};
7171
7172 std::optional<int64_t> EncodedImm =
7173 AMDGPU::getSMRDEncodedOffset(STI, *OffsetVal, true);
7174 if (!EncodedImm)
7175 return {};
7176
7177 return {{ [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); } }};
7178}
7179
7181AMDGPUInstructionSelector::selectSMRDBufferImm32(MachineOperand &Root) const {
7182 assert(STI.getGeneration() == AMDGPUSubtarget::SEA_ISLANDS);
7183
7184 std::optional<uint64_t> OffsetVal = getConstantZext32Val(Root.getReg(), *MRI);
7185 if (!OffsetVal)
7186 return {};
7187
7188 std::optional<int64_t> EncodedImm =
7190 if (!EncodedImm)
7191 return {};
7192
7193 return {{ [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); } }};
7194}
7195
7197AMDGPUInstructionSelector::selectSMRDBufferSgprImm(MachineOperand &Root) const {
7198 // Match the (soffset + offset) pair as a 32-bit register base and
7199 // an immediate offset.
7200 Register SOffset;
7201 unsigned Offset;
7202 std::tie(SOffset, Offset) = AMDGPU::getBaseWithConstantOffset(
7203 *MRI, Root.getReg(), VT, /*CheckNUW*/ true);
7204 if (!SOffset)
7205 return std::nullopt;
7206
7207 std::optional<int64_t> EncodedOffset =
7208 AMDGPU::getSMRDEncodedOffset(STI, Offset, /* IsBuffer */ true);
7209 if (!EncodedOffset)
7210 return std::nullopt;
7211
7212 assert(MRI->getType(SOffset).getSizeInBits() == 32);
7213 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
7214 [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedOffset); }}};
7215}
7216
7217std::pair<Register, unsigned>
7218AMDGPUInstructionSelector::selectVOP3PMadMixModsImpl(MachineOperand &Root,
7219 bool &Matched) const {
7220 Matched = false;
7221
7222 Register Src;
7223 unsigned Mods;
7224 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
7225
7226 if (mi_match(Src, *MRI, m_GFPExt(m_Reg(Src)))) {
7227 assert(MRI->getType(Src) == LLT::scalar(16));
7228
7229 // Only change Src if src modifier could be gained. In such cases new Src
7230 // could be sgpr but this does not violate constant bus restriction for
7231 // instruction that is being selected.
7232 Src = stripBitCast(Src, *MRI);
7233
7234 const auto CheckAbsNeg = [&]() {
7235 // Be careful about folding modifiers if we already have an abs. fneg is
7236 // applied last, so we don't want to apply an earlier fneg.
7237 if ((Mods & SISrcMods::ABS) == 0) {
7238 unsigned ModsTmp;
7239 std::tie(Src, ModsTmp) = selectVOP3ModsImpl(Src);
7240
7241 if ((ModsTmp & SISrcMods::NEG) != 0)
7242 Mods ^= SISrcMods::NEG;
7243
7244 if ((ModsTmp & SISrcMods::ABS) != 0)
7245 Mods |= SISrcMods::ABS;
7246 }
7247 };
7248
7249 CheckAbsNeg();
7250
7251 // op_sel/op_sel_hi decide the source type and source.
7252 // If the source's op_sel_hi is set, it indicates to do a conversion from
7253 // fp16. If the sources's op_sel is set, it picks the high half of the
7254 // source register.
7255
7256 Mods |= SISrcMods::OP_SEL_1;
7257
7258 // The instruction reads a 32-bit source and selects a half of it, so look
7259 // for the 32-bit register the 16-bit value is a half of.
7260 if (isExtractHiElt(*MRI, Src, Src)) {
7261 // Src is now the 32-bit source and op_sel picks its high half.
7262 Mods |= SISrcMods::OP_SEL_0;
7263 CheckAbsNeg();
7264 } else if (isExtractLoElt(*MRI, Src, Src)) {
7265 // op_sel already picks the low half, so the 32-bit source can be used
7266 // directly. Unlike the high half, only an fneg/fabs that acts on each
7267 // 16-bit element can be folded here: one that acts on the 32-bit value
7268 // touches bit 31 and leaves the low half alone.
7269 if (MRI->getType(Src).isFixedVector(2, 16))
7270 CheckAbsNeg();
7271 }
7272 // Otherwise Src is genuinely 16 bits wide and widenSrcIfVGPR16 widens it
7273 // when the operand is rendered.
7274
7275 Matched = true;
7276 }
7277
7278 return {Src, Mods};
7279}
7280
7282AMDGPUInstructionSelector::selectVOP3PMadMixModsExt(
7283 MachineOperand &Root) const {
7284 Register Src;
7285 unsigned Mods;
7286 bool Matched;
7287 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7288 if (!Matched)
7289 return {};
7290
7291 return {{
7292 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7293 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
7294 }};
7295}
7296
7298AMDGPUInstructionSelector::selectVOP3PMadMixMods(MachineOperand &Root) const {
7299 Register Src;
7300 unsigned Mods;
7301 bool Matched;
7302 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7303
7304 return {{
7305 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7306 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
7307 }};
7308}
7309
7311AMDGPUInstructionSelector::selectVOP3PMadMixModsExtNeg(
7312 MachineOperand &Root) const {
7313 Register Src;
7314 unsigned Mods;
7315 bool Matched;
7316 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7317 if (!Matched)
7318 return {};
7319
7320 return {{
7321 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7322 [=](MachineInstrBuilder &MIB) {
7323 MIB.addImm(Mods ^ SISrcMods::NEG);
7324 } // src_mods
7325 }};
7326}
7327
7329AMDGPUInstructionSelector::selectVOP3PMadMixModsNeg(
7330 MachineOperand &Root) const {
7331 Register Src;
7332 unsigned Mods;
7333 bool Matched;
7334 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7335
7336 return {{
7337 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7338 [=](MachineInstrBuilder &MIB) {
7339 MIB.addImm(Mods ^ SISrcMods::NEG);
7340 } // src_mods
7341 }};
7342}
7343
7344bool AMDGPUInstructionSelector::selectSBarrierSignalIsfirst(
7345 MachineInstr &I, Intrinsic::ID IntrID) const {
7346 MachineBasicBlock *MBB = I.getParent();
7347 const DebugLoc &DL = I.getDebugLoc();
7348 Register CCReg = I.getOperand(0).getReg();
7349
7350 // Set SCC to true, in case the barrier instruction gets converted to a NOP.
7351 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_CMP_EQ_U32)).addImm(0).addImm(0);
7352
7353 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM))
7354 .addImm(I.getOperand(2).getImm());
7355
7356 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), CCReg).addReg(AMDGPU::SCC);
7357
7358 I.eraseFromParent();
7359 return RBI.constrainGenericRegister(CCReg, AMDGPU::SReg_32_XM0_XEXECRegClass,
7360 *MRI);
7361}
7362
7363bool AMDGPUInstructionSelector::selectSGetBarrierState(
7364 MachineInstr &I, Intrinsic::ID IntrID) const {
7365 MachineBasicBlock *MBB = I.getParent();
7366 const DebugLoc &DL = I.getDebugLoc();
7367 const MachineOperand &BarOp = I.getOperand(2);
7368 std::optional<int64_t> BarValImm =
7369 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7370
7371 if (!BarValImm) {
7372 auto CopyMIB = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
7373 .addReg(BarOp.getReg());
7374 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7375 }
7376 MachineInstrBuilder MIB;
7377 unsigned Opc = BarValImm ? AMDGPU::S_GET_BARRIER_STATE_IMM
7378 : AMDGPU::S_GET_BARRIER_STATE_M0;
7379 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7380
7381 auto DstReg = I.getOperand(0).getReg();
7382 const TargetRegisterClass *DstRC =
7383 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
7384 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
7385 return false;
7386 MIB.addDef(DstReg);
7387 if (BarValImm) {
7388 MIB.addImm(*BarValImm);
7389 }
7390 I.eraseFromParent();
7391 return true;
7392}
7393
7394unsigned getNamedBarrierOp(bool HasInlineConst, Intrinsic::ID IntrID) {
7395 if (HasInlineConst) {
7396 switch (IntrID) {
7397 default:
7398 llvm_unreachable("not a named barrier op");
7399 case Intrinsic::amdgcn_s_barrier_join:
7400 return AMDGPU::S_BARRIER_JOIN_IMM;
7401 case Intrinsic::amdgcn_s_wakeup_barrier:
7402 return AMDGPU::S_WAKEUP_BARRIER_IMM;
7403 case Intrinsic::amdgcn_s_get_named_barrier_state:
7404 return AMDGPU::S_GET_BARRIER_STATE_IMM;
7405 };
7406 } else {
7407 switch (IntrID) {
7408 default:
7409 llvm_unreachable("not a named barrier op");
7410 case Intrinsic::amdgcn_s_barrier_join:
7411 return AMDGPU::S_BARRIER_JOIN_M0;
7412 case Intrinsic::amdgcn_s_wakeup_barrier:
7413 return AMDGPU::S_WAKEUP_BARRIER_M0;
7414 case Intrinsic::amdgcn_s_get_named_barrier_state:
7415 return AMDGPU::S_GET_BARRIER_STATE_M0;
7416 };
7417 }
7418}
7419
7420bool AMDGPUInstructionSelector::selectNamedBarrierInit(
7421 MachineInstr &I, Intrinsic::ID IntrID) const {
7422 MachineBasicBlock *MBB = I.getParent();
7423 const DebugLoc &DL = I.getDebugLoc();
7424 const MachineOperand &BarOp = I.getOperand(1);
7425 const MachineOperand &CntOp = I.getOperand(2);
7426
7427 // A member count of 0 means "keep existing member count". That plus a known
7428 // constant value for the barrier ID lets us use the immarg form.
7429 if (IntrID == Intrinsic::amdgcn_s_barrier_signal_var) {
7430 std::optional<int64_t> CntImm =
7431 getIConstantVRegSExtVal(CntOp.getReg(), *MRI);
7432 if (CntImm && *CntImm == 0) {
7433 std::optional<int64_t> BarValImm =
7434 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7435 if (BarValImm) {
7436 uint32_t BarID = *BarValImm & 0x3F;
7437 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_BARRIER_SIGNAL_IMM))
7438 .addImm(BarID);
7439 I.eraseFromParent();
7440 return true;
7441 }
7442 }
7443 }
7444
7445 // BarID = BarOp & 0x3F
7446 Register TmpReg1 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7447 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg1)
7448 .add(BarOp)
7449 .addImm(0x3F)
7450 .setOperandDead(3); // Dead scc
7451
7452 // MO = ((CntOp & 0x3F) << shAmt) | BarID
7453 Register TmpReg2 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7454 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg2)
7455 .add(CntOp)
7456 .addImm(0x3F)
7457 .setOperandDead(3); // Dead scc
7458
7459 Register TmpReg3 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7460 constexpr unsigned ShAmt = 16;
7461 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_LSHL_B32), TmpReg3)
7462 .addReg(TmpReg2)
7463 .addImm(ShAmt)
7464 .setOperandDead(3); // Dead scc
7465
7466 Register TmpReg4 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7467 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_OR_B32), TmpReg4)
7468 .addReg(TmpReg1)
7469 .addReg(TmpReg3)
7470 .setOperandDead(3); // Dead scc;
7471
7472 auto CopyMIB =
7473 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0).addReg(TmpReg4);
7474 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7475
7476 unsigned Opc = IntrID == Intrinsic::amdgcn_s_barrier_init
7477 ? AMDGPU::S_BARRIER_INIT_M0
7478 : AMDGPU::S_BARRIER_SIGNAL_M0;
7479 MachineInstrBuilder MIB;
7480 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7481
7482 I.eraseFromParent();
7483 return true;
7484}
7485
7486bool AMDGPUInstructionSelector::selectNamedBarrierInst(
7487 MachineInstr &I, Intrinsic::ID IntrID) const {
7488 MachineBasicBlock *MBB = I.getParent();
7489 const DebugLoc &DL = I.getDebugLoc();
7490 MachineOperand BarOp = IntrID == Intrinsic::amdgcn_s_get_named_barrier_state
7491 ? I.getOperand(2)
7492 : I.getOperand(1);
7493 std::optional<int64_t> BarValImm =
7494 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7495
7496 if (!BarValImm) {
7497 // BarID = BarOp & 0x3F
7498 Register TmpReg1 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7499 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg1)
7500 .addReg(BarOp.getReg())
7501 .addImm(0x3F)
7502 .setOperandDead(3); // Dead scc;
7503
7504 auto CopyMIB = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
7505 .addReg(TmpReg1);
7506 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7507 }
7508
7509 MachineInstrBuilder MIB;
7510 unsigned Opc = getNamedBarrierOp(BarValImm.has_value(), IntrID);
7511 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7512
7513 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state) {
7514 auto DstReg = I.getOperand(0).getReg();
7515 const TargetRegisterClass *DstRC =
7516 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
7517 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
7518 return false;
7519 MIB.addDef(DstReg);
7520 }
7521
7522 if (BarValImm) {
7523 uint32_t BarId = *BarValImm & 0x3F;
7524 MIB.addImm(BarId);
7525 }
7526
7527 I.eraseFromParent();
7528 return true;
7529}
7530
7531void AMDGPUInstructionSelector::renderNegateImm(MachineInstrBuilder &MIB,
7532 const MachineInstr &MI,
7533 int OpIdx) const {
7534 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
7535 "Expected G_CONSTANT");
7536 MIB.addImm(-MI.getOperand(1).getCImm()->getSExtValue());
7537}
7538
7539void AMDGPUInstructionSelector::renderBitcastFPImm(MachineInstrBuilder &MIB,
7540 const MachineInstr &MI,
7541 int OpIdx) const {
7542 const MachineOperand &Op = MI.getOperand(1);
7543 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1);
7544 MIB.addImm(Op.getFPImm()->getValueAPF().bitcastToAPInt().getZExtValue());
7545}
7546
7547void AMDGPUInstructionSelector::renderCountTrailingOnesImm(
7548 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7549 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
7550 "Expected G_CONSTANT");
7551 MIB.addImm(MI.getOperand(1).getCImm()->getValue().countTrailingOnes());
7552}
7553
7554/// This only really exists to satisfy DAG type checking machinery, so is a
7555/// no-op here.
7556void AMDGPUInstructionSelector::renderTruncTImm(MachineInstrBuilder &MIB,
7557 const MachineInstr &MI,
7558 int OpIdx) const {
7559 const MachineOperand &Op = MI.getOperand(OpIdx);
7560 int64_t Imm;
7561 if (Op.isReg() && mi_match(Op.getReg(), *MRI, m_ICst(Imm)))
7562 MIB.addImm(Imm);
7563 else
7564 MIB.addImm(Op.getImm());
7565}
7566
7567void AMDGPUInstructionSelector::renderZextBoolTImm(MachineInstrBuilder &MIB,
7568 const MachineInstr &MI,
7569 int OpIdx) const {
7570 MIB.addImm(MI.getOperand(OpIdx).getImm() != 0);
7571}
7572
7573void AMDGPUInstructionSelector::renderOpSelTImm(MachineInstrBuilder &MIB,
7574 const MachineInstr &MI,
7575 int OpIdx) const {
7576 assert(OpIdx >= 0 && "expected to match an immediate operand");
7577 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7578}
7579
7580void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_0_0(
7581 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7582 assert(OpIdx >= 0 && "expected to match an immediate operand");
7583 MIB.addImm(
7584 (MI.getOperand(OpIdx).getImm() & 0x1) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7585}
7586
7587void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_0_1(
7588 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7589 assert(OpIdx >= 0 && "expected to match an immediate operand");
7590 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x1)
7592 : (int64_t)SISrcMods::DST_OP_SEL);
7593}
7594
7595void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_1_0(
7596 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7597 assert(OpIdx >= 0 && "expected to match an immediate operand");
7598 MIB.addImm(
7599 (MI.getOperand(OpIdx).getImm() & 0x2) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7600}
7601
7602void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_1_1(
7603 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7604 assert(OpIdx >= 0 && "expected to match an immediate operand");
7605 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x2)
7606 ? (int64_t)(SISrcMods::OP_SEL_0)
7607 : 0);
7608}
7609
7610void AMDGPUInstructionSelector::renderDstSelToOpSelXForm(
7611 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7612 assert(OpIdx >= 0 && "expected to match an immediate operand");
7613 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)(SISrcMods::DST_OP_SEL)
7614 : 0);
7615}
7616
7617void AMDGPUInstructionSelector::renderSrcSelToOpSelXForm(
7618 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7619 assert(OpIdx >= 0 && "expected to match an immediate operand");
7620 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)(SISrcMods::OP_SEL_0)
7621 : 0);
7622}
7623
7624void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_2_0(
7625 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7626 assert(OpIdx >= 0 && "expected to match an immediate operand");
7627 MIB.addImm(
7628 (MI.getOperand(OpIdx).getImm() & 0x1) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7629}
7630
7631void AMDGPUInstructionSelector::renderDstSelToOpSel3XFormXForm(
7632 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7633 assert(OpIdx >= 0 && "expected to match an immediate operand");
7634 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x2)
7635 ? (int64_t)SISrcMods::DST_OP_SEL
7636 : 0);
7637}
7638
7639void AMDGPUInstructionSelector::renderExtractCPol(MachineInstrBuilder &MIB,
7640 const MachineInstr &MI,
7641 int OpIdx) const {
7642 assert(OpIdx >= 0 && "expected to match an immediate operand");
7643 MIB.addImm(MI.getOperand(OpIdx).getImm() &
7646}
7647
7648void AMDGPUInstructionSelector::renderExtractSWZ(MachineInstrBuilder &MIB,
7649 const MachineInstr &MI,
7650 int OpIdx) const {
7651 assert(OpIdx >= 0 && "expected to match an immediate operand");
7652 const bool Swizzle = MI.getOperand(OpIdx).getImm() &
7655 MIB.addImm(Swizzle);
7656}
7657
7658void AMDGPUInstructionSelector::renderExtractCpolSetGLC(
7659 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7660 assert(OpIdx >= 0 && "expected to match an immediate operand");
7661 const uint32_t Cpol = MI.getOperand(OpIdx).getImm() &
7664 MIB.addImm(Cpol | AMDGPU::CPol::GLC);
7665}
7666
7667void AMDGPUInstructionSelector::renderFPPow2ToExponent(MachineInstrBuilder &MIB,
7668 const MachineInstr &MI,
7669 int OpIdx) const {
7670 const APFloat &APF = MI.getOperand(1).getFPImm()->getValueAPF();
7671 int ExpVal = APF.getExactLog2Abs();
7672 assert(ExpVal != INT_MIN);
7673 MIB.addImm(ExpVal);
7674}
7675
7676void AMDGPUInstructionSelector::renderRoundMode(MachineInstrBuilder &MIB,
7677 const MachineInstr &MI,
7678 int OpIdx) const {
7679 // "round.towardzero" -> TowardZero 0 -> FP_ROUND_ROUND_TO_ZERO 3
7680 // "round.tonearest" -> NearestTiesToEven 1 -> FP_ROUND_ROUND_TO_NEAREST 0
7681 // "round.upward" -> TowardPositive 2 -> FP_ROUND_ROUND_TO_INF 1
7682 // "round.downward -> TowardNegative 3 -> FP_ROUND_ROUND_TO_NEGINF 2
7683 MIB.addImm((MI.getOperand(OpIdx).getImm() + 3) % 4);
7684}
7685
7686void AMDGPUInstructionSelector::renderVOP3PModsNeg(MachineInstrBuilder &MIB,
7687 const MachineInstr &MI,
7688 int OpIdx) const {
7689 unsigned Mods = SISrcMods::OP_SEL_1;
7690 if (MI.getOperand(OpIdx).getImm())
7691 Mods ^= SISrcMods::NEG;
7692 MIB.addImm((int64_t)Mods);
7693}
7694
7695void AMDGPUInstructionSelector::renderVOP3PModsNegs(MachineInstrBuilder &MIB,
7696 const MachineInstr &MI,
7697 int OpIdx) const {
7698 unsigned Mods = SISrcMods::OP_SEL_1;
7699 if (MI.getOperand(OpIdx).getImm())
7701 MIB.addImm((int64_t)Mods);
7702}
7703
7704void AMDGPUInstructionSelector::renderVOP3PModsNegAbs(MachineInstrBuilder &MIB,
7705 const MachineInstr &MI,
7706 int OpIdx) const {
7707 unsigned Val = MI.getOperand(OpIdx).getImm();
7708 unsigned Mods = SISrcMods::OP_SEL_1; // default: none
7709 if (Val == 1) // neg
7710 Mods ^= SISrcMods::NEG;
7711 if (Val == 2) // abs
7712 Mods ^= SISrcMods::ABS;
7713 if (Val == 3) // neg and abs
7714 Mods ^= (SISrcMods::NEG | SISrcMods::ABS);
7715 MIB.addImm((int64_t)Mods);
7716}
7717
7718void AMDGPUInstructionSelector::renderPrefetchLoc(MachineInstrBuilder &MIB,
7719 const MachineInstr &MI,
7720 int OpIdx) const {
7721 uint32_t V = MI.getOperand(2).getImm();
7724 if (!Subtarget->hasSafeCUPrefetch())
7725 V = std::max(V, (uint32_t)AMDGPU::CPol::SCOPE_SE); // CU scope is unsafe
7726 MIB.addImm(V);
7727}
7728
7729/// Convert from 2-bit value to enum values used for op_sel* source modifiers.
7730void AMDGPUInstructionSelector::renderScaledMAIIntrinsicOperand(
7731 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7732 unsigned Val = MI.getOperand(OpIdx).getImm();
7733 unsigned New = 0;
7734 if (Val & 0x1)
7736 if (Val & 0x2)
7738 MIB.addImm(New);
7739}
7740
7741bool AMDGPUInstructionSelector::isInlineImmediate(const APInt &Imm) const {
7742 return TII.isInlineConstant(Imm);
7743}
7744
7745bool AMDGPUInstructionSelector::isInlineImmediate(const APFloat &Imm) const {
7746 return TII.isInlineConstant(Imm);
7747}
MachineInstrBuilder MachineInstrBuilder & DefMI
static unsigned getIntrinsicID(const SDNode *N)
#define GET_GLOBALISEL_PREDICATES_INIT
#define GET_GLOBALISEL_TEMPORARIES_INIT
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isShlHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is shift left with half bits, such as reg0:2n =G_SHL reg1:2n, CONST(n)
static bool isNoUnsignedWrap(MachineInstr *Addr)
static Register buildOffsetSrc(MachineIRBuilder &B, MachineRegisterInfo &MRI, const SIInstrInfo &TII, Register BasePtr)
unsigned getNamedBarrierOp(bool HasInlineConst, Intrinsic::ID IntrID)
static Register getLegalRegBank(Register NewReg, Register RootReg, MachineInstr &Use, const AMDGPURegisterBankInfo &RBI, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const SIInstrInfo &TII)
static bool checkRB(Register Reg, unsigned int RBNo, const AMDGPURegisterBankInfo &RBI, const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI)
static unsigned updateMods(SrcStatus HiStat, SrcStatus LoStat, unsigned Mods)
static bool isTruncHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is truncating to half, such as reg0:n = G_TRUNC reg1:2n
static Register getWaveAddress(const MachineInstr *Def)
static bool isExtractHiElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static bool shouldUseAndMask(unsigned Size, unsigned &Mask)
static std::pair< unsigned, uint8_t > BitOp3_Op(Register R, SmallVectorImpl< Register > &Src, const MachineRegisterInfo &MRI)
static TypeClass isVectorOfTwoOrScalar(Register Reg, const MachineRegisterInfo &MRI)
static bool isLaneMaskFromSameBlock(Register Reg, MachineRegisterInfo &MRI, MachineBasicBlock *MBB)
static bool isExtractLoElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE, bool &IsTexFail)
static void addZeroImm(MachineInstrBuilder &MIB)
static unsigned gwsIntrinToOpcode(unsigned IntrID)
static bool isConstant(const MachineInstr &MI)
static bool isSameBitWidth(Register Reg1, Register Reg2, const MachineRegisterInfo &MRI)
static Register buildRegSequence(SmallVectorImpl< Register > &Elts, MachineInstr *InsertPt, MachineRegisterInfo &MRI)
static Register buildRSRC(MachineIRBuilder &B, MachineRegisterInfo &MRI, uint32_t FormatLo, uint32_t FormatHi, Register BasePtr)
Return a resource descriptor for use with an arbitrary 64-bit pointer.
static bool isAsyncLDSDMA(Intrinsic::ID Intr)
static std::pair< Register, unsigned > computeIndirectRegIndex(MachineRegisterInfo &MRI, const SIRegisterInfo &TRI, const TargetRegisterClass *SuperRC, Register IdxReg, unsigned EltSize, GISelValueTracking &ValueTracking)
Return the register to use for the index value, and the subregister to use for the indirectly accesse...
static unsigned getLogicalBitOpcode(unsigned Opc, bool Is64)
static std::pair< Register, SrcStatus > getLastSameOrNeg(Register Reg, const MachineRegisterInfo &MRI, SearchOptions SO, int MaxDepth=3)
static Register stripCopy(Register Reg, MachineRegisterInfo &MRI)
static std::optional< std::pair< Register, SrcStatus > > calcNextStatus(std::pair< Register, SrcStatus > Curr, const MachineRegisterInfo &MRI)
static Register stripBitCast(Register Reg, MachineRegisterInfo &MRI)
static std::optional< uint64_t > getConstantZext32Val(Register Reg, const MachineRegisterInfo &MRI)
Get an immediate that must be 32-bits, and treated as zero extended.
static bool isValidToPack(SrcStatus HiStat, SrcStatus LoStat, Register NewReg, Register RootReg, const SIInstrInfo &TII, const MachineRegisterInfo &MRI)
static int getV_CMPOpcode(CmpInst::Predicate P, unsigned Size, const GCNSubtarget &ST)
static SmallVector< std::pair< Register, SrcStatus > > getSrcStats(Register Reg, const MachineRegisterInfo &MRI, SearchOptions SO, int MaxDepth=3)
static bool isUnmergeHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test function, if the MI is reg0:n, reg1:n = G_UNMERGE_VALUES reg2:2n
static SrcStatus getNegStatus(Register Reg, SrcStatus S, const MachineRegisterInfo &MRI)
static bool isVCmpResult(Register Reg, MachineRegisterInfo &MRI)
static Register buildAddr64RSrc(MachineIRBuilder &B, MachineRegisterInfo &MRI, const SIInstrInfo &TII, Register BasePtr)
static bool isLshrHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is logic shift right with half bits, such as reg0:2n =G_LSHR reg1:2n,...
static void selectWMMAModsNegAbs(unsigned ModOpcode, unsigned &Mods, SmallVectorImpl< Register > &Elts, Register &Src, MachineInstr *InsertPt, MachineRegisterInfo &MRI)
This file declares the targeting of the InstructionSelector class for AMDGPU.
constexpr LLT S1
constexpr LLT S32
AMDGPU Register Bank Select
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool isAllZeros(StringRef Arr)
Return true if the array is empty or all zeros.
dxil translate DXIL Translate Metadata
Provides analysis for querying information about KnownBits during GISel passes.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineIRBuilder class.
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define P(N)
static std::vector< std::pair< int, unsigned > > Swizzle(std::vector< std::pair< int, unsigned > > Src, R600InstrInfo::BankSwizzle Swz)
#define LLVM_DEBUG(...)
Definition Debug.h:119
Value * RHS
Value * LHS
This is used to control valid status that current MI supports.
bool checkOptions(SrcStatus Stat) const
SearchOptions(Register Reg, const MachineRegisterInfo &MRI)
AMDGPUInstructionSelector(const GCNSubtarget &STI, const AMDGPURegisterBankInfo &RBI)
static const char * getName()
bool select(MachineInstr &I) override
Select the (possibly generic) instruction I to only use target-specific opcodes.
void setupMF(MachineFunction &MF, GISelValueTracking *VT, CodeGenCoverage *CoverageInfo, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) override
Setup per-MF executor state.
LLVM_READONLY int getExactLog2Abs() const
Definition APFloat.h:1639
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
DILocation * get() const
Get the underlying DILocation.
Definition DebugLoc.h:226
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
std::optional< SmallVector< std::function< void(MachineInstrBuilder &)>, 4 > > ComplexRendererFns
virtual void setupMF(MachineFunction &mf, GISelValueTracking *vt, CodeGenCoverage *covinfo=nullptr, ProfileSummaryInfo *psi=nullptr, BlockFrequencyInfo *bfi=nullptr)
Setup per-MF executor state.
Register getSourceReg(unsigned I) const
Returns the I'th source register.
unsigned getNumSources() const
Returns the number of source registers.
Represents a G_UNMERGE_VALUES.
unsigned getNumDefs() const
Returns the number of def registers.
Register getSourceReg() const
Get the unmerge source register.
constexpr bool isScalar() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isValid() const
constexpr bool isVector() const
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr unsigned getAddressSpace() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
bool hasValue() const
TypeSize getValue() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
unsigned getID() const
getID() - Return the register class ID number.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
LLVM_ABI StringRef getString() const
Definition Metadata.cpp:615
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void setReturnAddressIsTaken(bool s)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Helper class to build MachineInstr.
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool getFlag(MIFlag Flag) const
Return whether an MI flag is set.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
LocationSize getSize() const
Return the size in bytes of the memory reference.
unsigned getAddrSpace() const
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
const MachinePointerInfo & getPointerInfo() const
Flags getFlags() const
Return the raw flags of the source value,.
const Value * getValue() const
Return the base address of the memory access.
Align getBaseAlign() const
Return the minimum known alignment in bytes of the base address, without the offset.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
const ConstantInt * getCImm() const
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
static MachineOperand CreateImm(int64_t Val)
bool isEarlyClobber() const
Register getReg() const
getReg - Returns the register number.
bool isInternalRead() const
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
const RegisterBank * getRegBankOrNull(Register Reg) const
Return the register bank of Reg, or null if Reg has not been assigned a register bank or has been ass...
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
Definition Type.cpp:887
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Analysis providing profile information.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
Definition StringRef.h:138
static bool isGenericOpcode(unsigned Opc)
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
LLVM_READONLY const MIMGG16MappingInfo * getMIMGG16MappingInfo(unsigned G)
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool isGFX12Plus(const MCSubtargetInfo &STI)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords, bool IndexedRsrc, bool IndexedSamp)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
LLVM_READONLY int32_t getGlobalSaddrOp(uint32_t Opcode)
bool isGFX13Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX10Plus(const MCSubtargetInfo &STI)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfo(unsigned DimEnum)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
Intrinsic::ID getIntrinsicID(const MachineInstr &I)
Return the intrinsic ID for opcodes with the G_AMDGPU_INTRIN_ prefix.
std::pair< Register, unsigned > getBaseWithConstantOffset(MachineRegisterInfo &MRI, Register Reg, GISelValueTracking *ValueTracking=nullptr, bool CheckNUW=false)
Returns base register and constant offset.
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
IndexMode
ARM Index Modes.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
operand_type_match m_Reg()
SpecificConstantMatch m_SpecificICst(const APInt &RequestedValue)
Matches a constant equal to RequestedValue.
GInstrBind< GBuildVector > m_GBuildVector(GBuildVector *&Inst)
GCstAndRegMatch m_GCst(std::optional< ValueAndVReg > &ValReg)
UnaryOp_match< SrcTy, TargetOpcode::COPY > m_Copy(SrcTy &&Src)
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
BinaryOp_match< LHS, RHS, TargetOpcode::G_XOR, true > m_GXor(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_SEXT > m_GSExt(const SrcTy &Src)
UnaryOp_match< SrcTy, TargetOpcode::G_FPEXT > m_GFPExt(const SrcTy &Src)
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
ConstantMatch< APInt > m_ICst(APInt &Cst)
SpecificConstantMatch m_AllOnesInt()
BinaryOp_match< LHS, RHS, TargetOpcode::G_OR, true > m_GOr(const LHS &L, const RHS &R)
ICstOrSplatMatch< APInt > m_ICstOrSplat(APInt &Cst)
ImplicitDefMatch m_GImplicitDef()
GInstrBind< GConcatVectors > m_GConcatVectors(GConcatVectors *&Inst)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
GInstrBind< GUnmerge > m_GUnmerge(GUnmerge *&Inst)
Instruction binders for ops with no operand-form matcher (constant-immediate or variadic-source ops).
BinaryOp_match< LHS, RHS, TargetOpcode::G_SUB > m_GSub(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_ASHR, false > m_GAShr(const LHS &L, const RHS &R)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
SpecificRegisterMatch m_SpecificReg(Register RequestedReg)
Matches a register only if it is equal to RequestedReg.
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
GFrameIndexMatch m_GFrameIndex(int &FI)
Or< Preds... > m_any_of(Preds &&... preds)
BinaryOp_match< LHS, RHS, TargetOpcode::G_AND, true > m_GAnd(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_BITCAST > m_GBitcast(const SrcTy &Src)
bind_ty< MachineInstr * > m_MInstr(MachineInstr *&MI)
UnaryOp_match< SrcTy, TargetOpcode::G_FNEG > m_GFNeg(const SrcTy &Src)
GFCstOrSplatGFCstMatch m_GFCstOrSplat(std::optional< FPValueAndVReg > &FPValReg)
UnaryOp_match< SrcTy, TargetOpcode::G_FABS > m_GFabs(const SrcTy &Src)
BinaryOp_match< LHS, RHS, TargetOpcode::G_LSHR, false > m_GLShr(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_ANYEXT > m_GAnyExt(const SrcTy &Src)
ShuffleVectorMatch< Src1Ty, Src2Ty > m_GShuffleVector(const Src1Ty &Src1, const Src2Ty &Src2, ArrayRef< int > &Mask)
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, TargetOpcode::G_MUL, true > m_GMul(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_TRUNC > m_GTrunc(const SrcTy &Src)
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Register getFunctionLiveInPhysReg(MachineFunction &MF, const TargetInstrInfo &TII, MCRegister PhysReg, const TargetRegisterClass &RC, const DebugLoc &DL, LLT RegTy=LLT())
Return a virtual register corresponding to the incoming argument register PhysReg.
Definition Utils.cpp:848
@ Offset
Definition DWP.cpp:577
LLVM_ABI bool isBuildVectorAllZeros(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndef=false)
Return true if the specified instruction is a G_BUILD_VECTOR or G_BUILD_VECTOR_TRUNC where all of the...
Definition Utils.cpp:1434
LLVM_ABI Register constrainOperandRegClass(const MachineFunction &MF, const TargetRegisterInfo &TRI, MachineRegisterInfo &MRI, const TargetInstrInfo &TII, const RegisterBankInfo &RBI, MachineInstr &InsertPt, const TargetRegisterClass &RegClass, MachineOperand &RegMO)
Constrain the Register operand OpIdx, so that it is now constrained to the TargetRegisterClass passed...
Definition Utils.cpp:60
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
Definition Utils.cpp:656
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
LLVM_ABI const ConstantFP * getConstantFPVRegVal(Register VReg, const MachineRegisterInfo &MRI)
Definition Utils.cpp:464
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI std::optional< APInt > getIConstantVRegVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT, return the corresponding value.
Definition Utils.cpp:297
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI void constrainSelectedInstRegOperands(MachineInstr &I, const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
Mutate the newly-selected instruction I to constrain its (possibly generic) virtual register operands...
Definition Utils.cpp:159
@ Load
The value being inserted comes from a load (InsertElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
Definition Utils.cpp:497
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI std::optional< int64_t > getIConstantVRegSExtVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT fits in int64_t returns it.
Definition Utils.cpp:317
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI std::optional< ValueAndVReg > getAnyConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true, bool LookThroughAnyExt=false)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT or G_FCONST...
Definition Utils.cpp:442
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
LLVM_ABI std::optional< DefinitionAndSourceRegister > getDefSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, and underlying value Register folding away any copies.
Definition Utils.cpp:472
LLVM_ABI Register getSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the source register for Reg, folding away any trivial copies.
Definition Utils.cpp:504
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
constexpr RegState getUndefRegState(bool B)
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
Definition KnownBits.h:315
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
int64_t Offset
Offset - This is an offset from the base Value*.
PointerUnion< const Value *, const PseudoSourceValue * > V
This is the IR pointer value for the access, or it is null if unknown.