LLVM 24.0.0git
AMDGPUInstructionSelector.cpp
Go to the documentation of this file.
1//===- AMDGPUInstructionSelector.cpp ----------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the InstructionSelector class for
10/// AMDGPU.
11/// \todo This should be generated by TableGen.
12//===----------------------------------------------------------------------===//
13
15#include "AMDGPU.h"
17#include "AMDGPUInstrInfo.h"
28#include "llvm/IR/IntrinsicsAMDGPU.h"
29#include <optional>
30
31#define DEBUG_TYPE "amdgpu-isel"
32
33using namespace llvm;
34using namespace MIPatternMatch;
35
36#define GET_GLOBALISEL_IMPL
37#define AMDGPUSubtarget GCNSubtarget
38#include "AMDGPUGenGlobalISel.inc"
39#undef GET_GLOBALISEL_IMPL
40#undef AMDGPUSubtarget
41
43 const GCNSubtarget &STI, const AMDGPURegisterBankInfo &RBI)
44 : TII(*STI.getInstrInfo()), TRI(*STI.getRegisterInfo()), RBI(RBI), STI(STI),
46#include "AMDGPUGenGlobalISel.inc"
49#include "AMDGPUGenGlobalISel.inc"
51{
52}
53
54const char *AMDGPUInstructionSelector::getName() { return DEBUG_TYPE; }
55
66
67// Return the wave level SGPR base address if this is a wave address.
69 return Def->getOpcode() == AMDGPU::G_AMDGPU_WAVE_ADDRESS
70 ? Def->getOperand(1).getReg()
71 : Register();
72}
73
74bool AMDGPUInstructionSelector::isVCC(Register Reg,
75 const MachineRegisterInfo &MRI) const {
76 // The verifier is oblivious to s1 being a valid value for wavesize registers.
77 if (Reg.isPhysical())
78 return false;
79
80 auto &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
81 const TargetRegisterClass *RC =
83 if (RC) {
84 const LLT Ty = MRI.getType(Reg);
85 if (!Ty.isValid() || Ty.getSizeInBits() != 1)
86 return false;
87 // G_TRUNC s1 result is never vcc.
88 return !mi_match(Reg, MRI, m_GTrunc(m_Reg())) &&
89 RC->hasSuperClassEq(TRI.getBoolRC());
90 }
91
92 const RegisterBank *RB = cast<const RegisterBank *>(RegClassOrBank);
93 return RB->getID() == AMDGPU::VCCRegBankID;
94}
95
96bool AMDGPUInstructionSelector::constrainCopyLikeIntrin(MachineInstr &MI,
97 unsigned NewOpc) const {
98 MI.setDesc(TII.get(NewOpc));
99 MI.removeOperand(1); // Remove intrinsic ID.
100 MI.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
101
102 Register DstReg = MI.getOperand(0).getReg();
103 Register SrcReg = MI.getOperand(1).getReg();
104
105 // TODO: This should be legalized to s32 if needed
106 if (MRI->getType(DstReg) == LLT::scalar(1))
107 return false;
108
109 const TargetRegisterClass *DstRC =
110 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
111 const TargetRegisterClass *SrcRC =
112 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
113 if (!DstRC || DstRC != SrcRC)
114 return false;
115
116 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
117 !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
118 return false;
119 const MCInstrDesc &MCID = MI.getDesc();
120 if (MCID.getOperandConstraint(0, MCOI::EARLY_CLOBBER) != -1) {
121 MI.getOperand(0).setIsEarlyClobber(true);
122 }
123 return true;
124}
125
126bool AMDGPUInstructionSelector::selectCOPY(MachineInstr &I) const {
127 const DebugLoc &DL = I.getDebugLoc();
128 MachineBasicBlock *BB = I.getParent();
129 I.setDesc(TII.get(TargetOpcode::COPY));
130
131 Register DstReg = I.getOperand(0).getReg();
132 Register SrcReg = I.getOperand(1).getReg();
133
134 if (isVCC(DstReg, *MRI)) {
135 if (SrcReg == AMDGPU::SCC) {
136 const TargetRegisterClass *RC =
137 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
138 if (!RC)
139 return true;
140 return RBI.constrainGenericRegister(DstReg, *RC, *MRI);
141 }
142
143 if (!isVCC(SrcReg, *MRI)) {
144 // TODO: Should probably leave the copy and let copyPhysReg expand it.
145 if (!RBI.constrainGenericRegister(DstReg, *TRI.getBoolRC(), *MRI))
146 return false;
147
148 const TargetRegisterClass *SrcRC =
149 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
150
151 std::optional<ValueAndVReg> ConstVal =
152 getIConstantVRegValWithLookThrough(SrcReg, *MRI, true);
153 if (ConstVal) {
154 unsigned MovOpc =
155 STI.isWave64() ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
156 BuildMI(*BB, &I, DL, TII.get(MovOpc), DstReg)
157 .addImm(ConstVal->Value.getBoolValue() ? -1 : 0);
158 } else {
159 Register MaskedReg = MRI->createVirtualRegister(SrcRC);
160
161 // We can't trust the high bits at this point, so clear them.
162
163 // TODO: Skip masking high bits if def is known boolean.
164
165 if (AMDGPU::getRegBitWidth(SrcRC->getID()) == 16) {
166 assert(Subtarget->useRealTrue16Insts());
167 const int64_t NoMods = 0;
168 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_AND_B16_t16_e64), MaskedReg)
169 .addImm(NoMods)
170 .addImm(1)
171 .addImm(NoMods)
172 .addReg(SrcReg)
173 .addImm(NoMods);
174 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CMP_NE_U16_t16_e64), DstReg)
175 .addImm(NoMods)
176 .addImm(0)
177 .addImm(NoMods)
178 .addReg(MaskedReg)
179 .addImm(NoMods);
180 } else {
181 bool IsSGPR = TRI.isSGPRClass(SrcRC);
182 unsigned AndOpc = IsSGPR ? AMDGPU::S_AND_B32 : AMDGPU::V_AND_B32_e32;
183 auto And = BuildMI(*BB, &I, DL, TII.get(AndOpc), MaskedReg)
184 .addImm(1)
185 .addReg(SrcReg);
186 if (IsSGPR)
187 And.setOperandDead(3); // Dead scc
188
189 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CMP_NE_U32_e64), DstReg)
190 .addImm(0)
191 .addReg(MaskedReg);
192 }
193 }
194
195 if (!MRI->getRegClassOrNull(SrcReg))
196 MRI->setRegClass(SrcReg, SrcRC);
197 I.eraseFromParent();
198 return true;
199 }
200
201 const TargetRegisterClass *RC =
202 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
203 if (RC && !RBI.constrainGenericRegister(DstReg, *RC, *MRI))
204 return false;
205
206 return true;
207 }
208
209 for (const MachineOperand &MO : I.operands()) {
210 if (MO.getReg().isPhysical())
211 continue;
212
213 const TargetRegisterClass *RC =
214 TRI.getConstrainedRegClassForReg(MO.getReg(), *MRI);
215 if (!RC)
216 continue;
217 RBI.constrainGenericRegister(MO.getReg(), *RC, *MRI);
218 }
219 return true;
220}
221
222bool AMDGPUInstructionSelector::selectCOPY_SCC_VCC(MachineInstr &I) const {
223 const DebugLoc &DL = I.getDebugLoc();
224 MachineBasicBlock *BB = I.getParent();
225 Register VCCReg = I.getOperand(1).getReg();
226 MachineInstr *Cmp;
227
228 // Set SCC as a side effect with S_CMP or S_OR.
229 if (STI.hasScalarCompareEq64()) {
230 unsigned CmpOpc =
231 STI.isWave64() ? AMDGPU::S_CMP_LG_U64 : AMDGPU::S_CMP_LG_U32;
232 Cmp = BuildMI(*BB, &I, DL, TII.get(CmpOpc)).addReg(VCCReg).addImm(0);
233 } else {
234 Register DeadDst = MRI->createVirtualRegister(&AMDGPU::SReg_64RegClass);
235 Cmp = BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_OR_B64), DeadDst)
236 .addReg(VCCReg)
237 .addReg(VCCReg);
238 }
239
240 constrainSelectedInstRegOperands(*Cmp, TII, TRI, RBI);
241
242 Register DstReg = I.getOperand(0).getReg();
243 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), DstReg).addReg(AMDGPU::SCC);
244
245 I.eraseFromParent();
246 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
247}
248
249bool AMDGPUInstructionSelector::selectCOPY_VCC_SCC(MachineInstr &I) const {
250 const DebugLoc &DL = I.getDebugLoc();
251 MachineBasicBlock *BB = I.getParent();
252
253 Register DstReg = I.getOperand(0).getReg();
254 Register SrcReg = I.getOperand(1).getReg();
255 std::optional<ValueAndVReg> Arg =
256 getIConstantVRegValWithLookThrough(I.getOperand(1).getReg(), *MRI);
257
258 if (Arg) {
259 const int64_t Value = Arg->Value.getZExtValue();
260 if (Value == 0) {
261 unsigned Opcode = STI.isWave64() ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
262 BuildMI(*BB, &I, DL, TII.get(Opcode), DstReg).addImm(0);
263 } else {
264 assert(Value == 1);
265 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), DstReg).addReg(TRI.getExec());
266 }
267 I.eraseFromParent();
268 return RBI.constrainGenericRegister(DstReg, *TRI.getBoolRC(), *MRI);
269 }
270
271 // RegBankLegalize ensures that SrcReg is bool in reg (high bits are 0).
272 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC).addReg(SrcReg);
273
274 unsigned SelectOpcode =
275 STI.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
276 MachineInstr *Select = BuildMI(*BB, &I, DL, TII.get(SelectOpcode), DstReg)
277 .addReg(TRI.getExec())
278 .addImm(0);
279
280 I.eraseFromParent();
282 return true;
283}
284
285bool AMDGPUInstructionSelector::selectReadAnyLane(MachineInstr &I) const {
286 Register DstReg = I.getOperand(0).getReg();
287 Register SrcReg = I.getOperand(1).getReg();
288
289 const DebugLoc &DL = I.getDebugLoc();
290 MachineBasicBlock *BB = I.getParent();
291
292 auto RFL = BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
293 .addReg(SrcReg);
294
295 I.eraseFromParent();
296 constrainSelectedInstRegOperands(*RFL, TII, TRI, RBI);
297 return true;
298}
299
300bool AMDGPUInstructionSelector::selectPHI(MachineInstr &I) const {
301 const Register DefReg = I.getOperand(0).getReg();
302 const LLT DefTy = MRI->getType(DefReg);
303
304 // S1 G_PHIs should not be selected in instruction-select, instead:
305 // - divergent S1 G_PHI should go through lane mask merging algorithm
306 // and be fully inst-selected in AMDGPUGlobalISelDivergenceLowering
307 // - uniform S1 G_PHI should be lowered into S32 G_PHI in AMDGPURegBankSelect
308 if (DefTy == LLT::scalar(1))
309 return false;
310
311 // TODO: Verify this doesn't have insane operands (i.e. VGPR to SGPR copy)
312
313 const RegClassOrRegBank &RegClassOrBank =
314 MRI->getRegClassOrRegBank(DefReg);
315
316 const TargetRegisterClass *DefRC =
318 if (!DefRC) {
319 if (!DefTy.isValid()) {
320 LLVM_DEBUG(dbgs() << "PHI operand has no type, not a gvreg?\n");
321 return false;
322 }
323
324 const RegisterBank &RB = *cast<const RegisterBank *>(RegClassOrBank);
325 DefRC = TRI.getRegClassForTypeOnBank(DefTy, RB);
326 if (!DefRC) {
327 LLVM_DEBUG(dbgs() << "PHI operand has unexpected size/bank\n");
328 return false;
329 }
330 }
331
332 // If inputs have register bank, assign corresponding reg class.
333 // Note: registers don't need to have the same reg bank.
334 for (unsigned i = 1; i != I.getNumOperands(); i += 2) {
335 const Register SrcReg = I.getOperand(i).getReg();
336
337 const RegisterBank *RB = MRI->getRegBankOrNull(SrcReg);
338 if (RB) {
339 const LLT SrcTy = MRI->getType(SrcReg);
340 const TargetRegisterClass *SrcRC =
341 TRI.getRegClassForTypeOnBank(SrcTy, *RB);
342 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
343 return false;
344 }
345 }
346
347 I.setDesc(TII.get(TargetOpcode::PHI));
348 return RBI.constrainGenericRegister(DefReg, *DefRC, *MRI);
349}
350
352AMDGPUInstructionSelector::getSubOperand64(MachineOperand &MO,
353 const TargetRegisterClass &SubRC,
354 unsigned SubIdx) const {
355
356 MachineInstr *MI = MO.getParent();
357 MachineBasicBlock *BB = MO.getParent()->getParent();
358 Register DstReg = MRI->createVirtualRegister(&SubRC);
359
360 if (MO.isReg()) {
361 unsigned ComposedSubIdx = TRI.composeSubRegIndices(MO.getSubReg(), SubIdx);
362 Register Reg = MO.getReg();
363 BuildMI(*BB, MI, MI->getDebugLoc(), TII.get(AMDGPU::COPY), DstReg)
364 .addReg(Reg, {}, ComposedSubIdx);
365
366 return MachineOperand::CreateReg(DstReg, MO.isDef(), MO.isImplicit(),
367 MO.isKill(), MO.isDead(), MO.isUndef(),
368 MO.isEarlyClobber(), 0, MO.isDebug(),
369 MO.isInternalRead());
370 }
371
372 assert(MO.isImm());
373
374 APInt Imm(64, MO.getImm());
375
376 switch (SubIdx) {
377 default:
378 llvm_unreachable("do not know to split immediate with this sub index.");
379 case AMDGPU::sub0:
380 return MachineOperand::CreateImm(Imm.getLoBits(32).getSExtValue());
381 case AMDGPU::sub1:
382 return MachineOperand::CreateImm(Imm.getHiBits(32).getSExtValue());
383 }
384}
385
386static unsigned getLogicalBitOpcode(unsigned Opc, bool Is64) {
387 switch (Opc) {
388 case AMDGPU::G_AND:
389 return Is64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
390 case AMDGPU::G_OR:
391 return Is64 ? AMDGPU::S_OR_B64 : AMDGPU::S_OR_B32;
392 case AMDGPU::G_XOR:
393 return Is64 ? AMDGPU::S_XOR_B64 : AMDGPU::S_XOR_B32;
394 default:
395 llvm_unreachable("not a bit op");
396 }
397}
398
399bool AMDGPUInstructionSelector::selectG_AND_OR_XOR(MachineInstr &I) const {
400 Register DstReg = I.getOperand(0).getReg();
401 unsigned Size = RBI.getSizeInBits(DstReg, *MRI, TRI);
402
403 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
404 if (DstRB->getID() != AMDGPU::SGPRRegBankID &&
405 DstRB->getID() != AMDGPU::VCCRegBankID)
406 return false;
407
408 bool Is64 = Size > 32 || (DstRB->getID() == AMDGPU::VCCRegBankID &&
409 STI.isWave64());
410 I.setDesc(TII.get(getLogicalBitOpcode(I.getOpcode(), Is64)));
411
412 // Dead implicit-def of scc
413 I.addOperand(MachineOperand::CreateReg(AMDGPU::SCC, true, // isDef
414 true, // isImp
415 false, // isKill
416 true)); // isDead
417 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
418 return true;
419}
420
421bool AMDGPUInstructionSelector::selectG_ADD_SUB(MachineInstr &I) const {
422 MachineBasicBlock *BB = I.getParent();
424 Register DstReg = I.getOperand(0).getReg();
425 const DebugLoc &DL = I.getDebugLoc();
426 LLT Ty = MRI->getType(DstReg);
427 if (Ty.isVector())
428 return false;
429
430 unsigned Size = Ty.getSizeInBits();
431 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
432 const bool IsSALU = DstRB->getID() == AMDGPU::SGPRRegBankID;
433 const bool Sub = I.getOpcode() == TargetOpcode::G_SUB;
434
435 if (Size == 32) {
436 if (IsSALU) {
437 const unsigned Opc = Sub ? AMDGPU::S_SUB_U32 : AMDGPU::S_ADD_U32;
438 MachineInstr *Add =
439 BuildMI(*BB, &I, DL, TII.get(Opc), DstReg)
440 .add(I.getOperand(1))
441 .add(I.getOperand(2))
442 .setOperandDead(3); // Dead scc
443 I.eraseFromParent();
444 constrainSelectedInstRegOperands(*Add, TII, TRI, RBI);
445 return true;
446 }
447
448 if (STI.hasAddNoCarryInsts()) {
449 const unsigned Opc = Sub ? AMDGPU::V_SUB_U32_e64 : AMDGPU::V_ADD_U32_e64;
450 I.setDesc(TII.get(Opc));
451 I.addOperand(*MF, MachineOperand::CreateImm(0));
452 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
453 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
454 return true;
455 }
456
457 const unsigned Opc = Sub ? AMDGPU::V_SUB_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e64;
458
459 Register UnusedCarry = MRI->createVirtualRegister(TRI.getWaveMaskRegClass());
460 MachineInstr *Add
461 = BuildMI(*BB, &I, DL, TII.get(Opc), DstReg)
462 .addDef(UnusedCarry, RegState::Dead)
463 .add(I.getOperand(1))
464 .add(I.getOperand(2))
465 .addImm(0);
466 I.eraseFromParent();
467 constrainSelectedInstRegOperands(*Add, TII, TRI, RBI);
468 return true;
469 }
470
471 assert(!Sub && "illegal sub should not reach here");
472
473 const TargetRegisterClass &RC
474 = IsSALU ? AMDGPU::SReg_64_XEXECRegClass : AMDGPU::VReg_64RegClass;
475 const TargetRegisterClass &HalfRC
476 = IsSALU ? AMDGPU::SReg_32RegClass : AMDGPU::VGPR_32RegClass;
477
478 MachineOperand Lo1(getSubOperand64(I.getOperand(1), HalfRC, AMDGPU::sub0));
479 MachineOperand Lo2(getSubOperand64(I.getOperand(2), HalfRC, AMDGPU::sub0));
480 MachineOperand Hi1(getSubOperand64(I.getOperand(1), HalfRC, AMDGPU::sub1));
481 MachineOperand Hi2(getSubOperand64(I.getOperand(2), HalfRC, AMDGPU::sub1));
482
483 Register DstLo = MRI->createVirtualRegister(&HalfRC);
484 Register DstHi = MRI->createVirtualRegister(&HalfRC);
485
486 if (IsSALU) {
487 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADD_U32), DstLo)
488 .add(Lo1)
489 .add(Lo2);
490 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADDC_U32), DstHi)
491 .add(Hi1)
492 .add(Hi2)
493 .setOperandDead(3); // Dead scc
494 } else {
495 const TargetRegisterClass *CarryRC = TRI.getWaveMaskRegClass();
496 Register CarryReg = MRI->createVirtualRegister(CarryRC);
497 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_ADD_CO_U32_e64), DstLo)
498 .addDef(CarryReg)
499 .add(Lo1)
500 .add(Lo2)
501 .addImm(0);
502 MachineInstr *Addc = BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_ADDC_U32_e64), DstHi)
503 .addDef(MRI->createVirtualRegister(CarryRC), RegState::Dead)
504 .add(Hi1)
505 .add(Hi2)
506 .addReg(CarryReg, RegState::Kill)
507 .addImm(0);
508
509 constrainSelectedInstRegOperands(*Addc, TII, TRI, RBI);
510 }
511
512 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
513 .addReg(DstLo)
514 .addImm(AMDGPU::sub0)
515 .addReg(DstHi)
516 .addImm(AMDGPU::sub1);
517
518
519 if (!RBI.constrainGenericRegister(DstReg, RC, *MRI))
520 return false;
521
522 I.eraseFromParent();
523 return true;
524}
525
526bool AMDGPUInstructionSelector::selectG_UADDO_USUBO_UADDE_USUBE(
527 MachineInstr &I) const {
528 MachineBasicBlock *BB = I.getParent();
530 const DebugLoc &DL = I.getDebugLoc();
531 Register Dst0Reg = I.getOperand(0).getReg();
532 Register Dst1Reg = I.getOperand(1).getReg();
533 const bool IsAdd = I.getOpcode() == AMDGPU::G_UADDO ||
534 I.getOpcode() == AMDGPU::G_UADDE;
535 const bool HasCarryIn = I.getOpcode() == AMDGPU::G_UADDE ||
536 I.getOpcode() == AMDGPU::G_USUBE;
537
538 if (isVCC(Dst1Reg, *MRI)) {
539 unsigned NoCarryOpc =
540 IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
541 unsigned CarryOpc = IsAdd ? AMDGPU::V_ADDC_U32_e64 : AMDGPU::V_SUBB_U32_e64;
542 I.setDesc(TII.get(HasCarryIn ? CarryOpc : NoCarryOpc));
543 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
544 I.addOperand(*MF, MachineOperand::CreateImm(0));
545 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
546 return true;
547 }
548
549 Register Src0Reg = I.getOperand(2).getReg();
550 Register Src1Reg = I.getOperand(3).getReg();
551
552 if (HasCarryIn) {
553 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC)
554 .addReg(I.getOperand(4).getReg());
555 }
556
557 unsigned NoCarryOpc = IsAdd ? AMDGPU::S_ADD_U32 : AMDGPU::S_SUB_U32;
558 unsigned CarryOpc = IsAdd ? AMDGPU::S_ADDC_U32 : AMDGPU::S_SUBB_U32;
559
560 auto CarryInst = BuildMI(*BB, &I, DL, TII.get(HasCarryIn ? CarryOpc : NoCarryOpc), Dst0Reg)
561 .add(I.getOperand(2))
562 .add(I.getOperand(3));
563
564 if (MRI->use_nodbg_empty(Dst1Reg)) {
565 CarryInst.setOperandDead(3); // Dead scc
566 } else {
567 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst1Reg)
568 .addReg(AMDGPU::SCC);
569 if (!MRI->getRegClassOrNull(Dst1Reg))
570 MRI->setRegClass(Dst1Reg, &AMDGPU::SReg_32RegClass);
571 }
572
573 if (!RBI.constrainGenericRegister(Dst0Reg, AMDGPU::SReg_32RegClass, *MRI) ||
574 !RBI.constrainGenericRegister(Src0Reg, AMDGPU::SReg_32RegClass, *MRI) ||
575 !RBI.constrainGenericRegister(Src1Reg, AMDGPU::SReg_32RegClass, *MRI))
576 return false;
577
578 if (HasCarryIn &&
579 !RBI.constrainGenericRegister(I.getOperand(4).getReg(),
580 AMDGPU::SReg_32RegClass, *MRI))
581 return false;
582
583 I.eraseFromParent();
584 return true;
585}
586
587bool AMDGPUInstructionSelector::selectG_AMDGPU_MAD_64_32(
588 MachineInstr &I) const {
589 MachineBasicBlock *BB = I.getParent();
591 const bool IsUnsigned = I.getOpcode() == AMDGPU::G_AMDGPU_MAD_U64_U32;
592 bool UseNoCarry = Subtarget->hasMadNC64_32Insts() &&
593 MRI->use_nodbg_empty(I.getOperand(1).getReg());
594
595 unsigned Opc;
596 if (Subtarget->hasMADIntraFwdBug())
597 Opc = IsUnsigned ? AMDGPU::V_MAD_U64_U32_gfx11_e64
598 : AMDGPU::V_MAD_I64_I32_gfx11_e64;
599 else if (UseNoCarry)
600 Opc = IsUnsigned ? AMDGPU::V_MAD_NC_U64_U32_e64
601 : AMDGPU::V_MAD_NC_I64_I32_e64;
602 else
603 Opc = IsUnsigned ? AMDGPU::V_MAD_U64_U32_e64 : AMDGPU::V_MAD_I64_I32_e64;
604
605 if (UseNoCarry)
606 I.removeOperand(1);
607
608 I.setDesc(TII.get(Opc));
609 I.addOperand(*MF, MachineOperand::CreateImm(0));
610 I.addImplicitDefUseOperands(*MF);
611 I.getOperand(0).setIsEarlyClobber(true);
612 constrainSelectedInstRegOperands(I, TII, TRI, RBI);
613 return true;
614}
615
616// TODO: We should probably legalize these to only using 32-bit results.
617bool AMDGPUInstructionSelector::selectG_EXTRACT(MachineInstr &I) const {
618 MachineBasicBlock *BB = I.getParent();
619 Register DstReg = I.getOperand(0).getReg();
620 Register SrcReg = I.getOperand(1).getReg();
621 LLT DstTy = MRI->getType(DstReg);
622 LLT SrcTy = MRI->getType(SrcReg);
623 const unsigned SrcSize = SrcTy.getSizeInBits();
624 unsigned DstSize = DstTy.getSizeInBits();
625
626 // TODO: Should handle any multiple of 32 offset.
627 unsigned Offset = I.getOperand(2).getImm();
628 if (Offset % 32 != 0 || DstSize > 128)
629 return false;
630
631 // 16-bit operations really use 32-bit registers.
632 // FIXME: Probably should not allow 16-bit G_EXTRACT results.
633 if (DstSize == 16)
634 DstSize = 32;
635
636 const TargetRegisterClass *DstRC =
637 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
638 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
639 return false;
640
641 const RegisterBank *SrcBank = RBI.getRegBank(SrcReg, *MRI, TRI);
642 const TargetRegisterClass *SrcRC =
643 TRI.getRegClassForSizeOnBank(SrcSize, *SrcBank);
644 if (!SrcRC)
645 return false;
646 unsigned SubReg = SIRegisterInfo::getSubRegFromChannel(Offset / 32,
647 DstSize / 32);
648 SrcRC = TRI.getSubClassWithSubReg(SrcRC, SubReg);
649 if (!SrcRC)
650 return false;
651
652 SrcReg = constrainOperandRegClass(*MF, TRI, *MRI, TII, RBI, I,
653 *SrcRC, I.getOperand(1));
654 const DebugLoc &DL = I.getDebugLoc();
655 BuildMI(*BB, &I, DL, TII.get(TargetOpcode::COPY), DstReg)
656 .addReg(SrcReg, {}, SubReg);
657
658 I.eraseFromParent();
659 return true;
660}
661
662bool AMDGPUInstructionSelector::selectS16MergeToS32(MachineInstr &MI) const {
663 Register Dst = MI.getOperand(0).getReg();
664 Register Src0 = MI.getOperand(1).getReg();
665 Register Src1 = MI.getOperand(2).getReg();
666
667 LLT Src0Ty = MRI->getType(Src0);
668 LLT Src1Ty = MRI->getType(Src1);
669
670 const RegisterBank *DstBank = RBI.getRegBank(Dst, *MRI, TRI);
671 const RegisterBank *Src0Bank = RBI.getRegBank(Src0, *MRI, TRI);
672 const RegisterBank *Src1Bank = RBI.getRegBank(Src1, *MRI, TRI);
673 const bool IsVector = DstBank->getID() == AMDGPU::VGPRRegBankID;
674
675 Register ShiftSrc0;
676 Register ShiftSrc1;
677
678 const DebugLoc &DL = MI.getDebugLoc();
679 MachineBasicBlock *BB = MI.getParent();
680
681 // VGPR case
682 if (IsVector) {
683 // If source are both VGPR16, use REG_SEQUENCE with lo16/hi16 subregisters
684 if (Src0Bank->getID() == AMDGPU::VGPRRegBankID &&
685 Src1Bank->getID() == AMDGPU::VGPRRegBankID &&
686 Src0Ty == LLT::scalar(16) && Src1Ty == LLT::scalar(16)) {
687 BuildMI(*BB, MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), Dst)
688 .addReg(Src0)
689 .addImm(AMDGPU::lo16)
690 .addReg(Src1)
691 .addImm(AMDGPU::hi16);
692
693 if (!RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI))
694 return false;
695
696 MI.eraseFromParent();
697 return true;
698 }
699
700 // Otherwise, use V_LSHL_OR_B32_e64
701 Register TmpReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
702 auto MIB = BuildMI(*BB, MI, DL, TII.get(AMDGPU::V_AND_B32_e32), TmpReg)
703 .addImm(0xFFFF)
704 .addReg(Src0);
705 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
706
707 MIB = BuildMI(*BB, MI, DL, TII.get(AMDGPU::V_LSHL_OR_B32_e64), Dst)
708 .addReg(Src1)
709 .addImm(16)
710 .addReg(TmpReg);
711 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
712
713 MI.eraseFromParent();
714 return true;
715 }
716
717 // SGPR case -> S_PACK_*_B32_B16
718 // With multiple uses of the shift, this will duplicate the shift and
719 // increase register pressure.
720 //
721 // (merge (lshr_oneuse $src0, 16), (lshr_oneuse $src1, 16)
722 // => (S_PACK_HH_B32_B16 $src0, $src1)
723 // (merge (lshr_oneuse SReg_32:$src0, 16), $src1)
724 // => (S_PACK_HL_B32_B16 $src0, $src1)
725 // (merge $src0, (lshr_oneuse SReg_32:$src1, 16))
726 // => (S_PACK_LH_B32_B16 $src0, $src1)
727 // (merge $src0, $src1)
728 // => (S_PACK_LL_B32_B16 $src0, $src1)
729
730 bool Shift0 = mi_match(
731 Src0, *MRI, m_OneUse(m_GLShr(m_Reg(ShiftSrc0), m_SpecificICst(16))));
732
733 bool Shift1 = mi_match(
734 Src1, *MRI, m_OneUse(m_GLShr(m_Reg(ShiftSrc1), m_SpecificICst(16))));
735
736 unsigned Opc = AMDGPU::S_PACK_LL_B32_B16;
737 if (Shift0 && Shift1) {
738 Opc = AMDGPU::S_PACK_HH_B32_B16;
739 MI.getOperand(1).setReg(ShiftSrc0);
740 MI.getOperand(2).setReg(ShiftSrc1);
741 } else if (Shift1) {
742 Opc = AMDGPU::S_PACK_LH_B32_B16;
743 MI.getOperand(2).setReg(ShiftSrc1);
744 } else if (Shift0) {
745 auto ConstSrc1 =
746 getAnyConstantVRegValWithLookThrough(Src1, *MRI, true, true);
747 if (ConstSrc1 && ConstSrc1->Value == 0) {
748 // build_vector_trunc (lshr $src0, 16), 0 -> s_lshr_b32 $src0, 16
749 auto MIB = BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_LSHR_B32), Dst)
750 .addReg(ShiftSrc0)
751 .addImm(16)
752 .setOperandDead(3); // Dead scc
753
754 MI.eraseFromParent();
755 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
756 return true;
757 }
758 if (STI.hasSPackHL()) {
759 Opc = AMDGPU::S_PACK_HL_B32_B16;
760 MI.getOperand(1).setReg(ShiftSrc0);
761 }
762 }
763
764 MI.setDesc(TII.get(Opc));
765 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
766 return true;
767}
768
769// Pack each pair of s16 into an s32 with S_PACK_LL_B32_B16, then combine the
770// s32 pieces into the destination with a REG_SEQUENCE.
771bool AMDGPUInstructionSelector::selectS16MergeToWide(MachineInstr &MI) const {
772 MachineBasicBlock *BB = MI.getParent();
773 const DebugLoc &DL = MI.getDebugLoc();
774 Register DstReg = MI.getOperand(0).getReg();
775 const unsigned DstSize = MRI->getType(DstReg).getSizeInBits();
776 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
777 const unsigned NumSrc = MI.getNumOperands() - 1;
778
779 // Pack each pair of s16 sources into an s32.
781 for (unsigned I = 0; I != NumSrc; I += 2) {
782 Register S32 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
783 auto Pack = BuildMI(*BB, MI, DL, TII.get(AMDGPU::S_PACK_LL_B32_B16), S32)
784 .addReg(MI.getOperand(I + 1).getReg())
785 .addReg(MI.getOperand(I + 2).getReg());
786 constrainSelectedInstRegOperands(*Pack, TII, TRI, RBI);
787 S32Regs.push_back(S32);
788 }
789
790 // Combine the s32 pieces into the destination with a REG_SEQUENCE.
791 const TargetRegisterClass *DstRC =
792 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
793 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
794 return false;
795 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(DstRC, /*EltSize=*/4);
796 auto MIB = BuildMI(*BB, MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), DstReg);
797 for (unsigned I = 0, E = S32Regs.size(); I != E; ++I)
798 MIB.addReg(S32Regs[I]).addImm(SubRegs[I]);
799
800 MI.eraseFromParent();
801 return true;
802}
803
804bool AMDGPUInstructionSelector::selectG_MERGE_VALUES(MachineInstr &MI) const {
805 MachineBasicBlock *BB = MI.getParent();
806 Register DstReg = MI.getOperand(0).getReg();
807 LLT DstTy = MRI->getType(DstReg);
808 LLT SrcTy = MRI->getType(MI.getOperand(1).getReg());
809
810 const unsigned SrcSize = SrcTy.getSizeInBits();
811 if (SrcSize < 32) {
812 // Handle s32 <- G_MERGE_VALUES s16, s16
813 if (SrcSize == 16 && DstTy.getSizeInBits() == 32 &&
814 MI.getNumOperands() == 3) {
815 return selectS16MergeToS32(MI);
816 }
817 // With true16 a scalar s16 is a register type, so a scalar wider than 32
818 // bits can be built from s16 pieces.
819 bool IsWideS16Merge = SrcSize == 16 && DstTy.getSizeInBits() > 32 &&
820 DstTy.getSizeInBits() % 32 == 0;
821
822 // SGPRs have no 16-bit subregisters, so pack pairs of s16 with S_PACK.
823 if (IsWideS16Merge &&
824 RBI.getRegBank(DstReg, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID)
825 return selectS16MergeToWide(MI);
826
827 // A VGPR wide s16 merge falls through to the generic path below.
828 if (!IsWideS16Merge)
829 return selectImpl(MI, *CoverageInfo);
830 }
831
832 const DebugLoc &DL = MI.getDebugLoc();
833 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
834 const unsigned DstSize = DstTy.getSizeInBits();
835 const TargetRegisterClass *DstRC =
836 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
837 if (!DstRC)
838 return false;
839
840 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(DstRC, SrcSize / 8);
841 MachineInstrBuilder MIB =
842 BuildMI(*BB, &MI, DL, TII.get(TargetOpcode::REG_SEQUENCE), DstReg);
843 for (int I = 0, E = MI.getNumOperands() - 1; I != E; ++I) {
844 MachineOperand &Src = MI.getOperand(I + 1);
845 Register SrcReg = Src.getReg();
846 MIB.addReg(SrcReg, getUndefRegState(Src.isUndef()));
847 MIB.addImm(SubRegs[I]);
848
849 const TargetRegisterClass *SrcRC =
850 TRI.getConstrainedRegClassForReg(SrcReg, *MRI);
851 if (SrcRC && !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
852 return false;
853 }
854
855 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
856 return false;
857
858 MI.eraseFromParent();
859 return true;
860}
861
862bool AMDGPUInstructionSelector::selectG_UNMERGE_VALUES(MachineInstr &MI) const {
863 MachineBasicBlock *BB = MI.getParent();
864 const int NumDst = MI.getNumOperands() - 1;
865
866 MachineOperand &Src = MI.getOperand(NumDst);
867
868 Register SrcReg = Src.getReg();
869 Register DstReg0 = MI.getOperand(0).getReg();
870 LLT DstTy = MRI->getType(DstReg0);
871 LLT SrcTy = MRI->getType(SrcReg);
872
873 const unsigned DstSize = DstTy.getSizeInBits();
874 const unsigned SrcSize = SrcTy.getSizeInBits();
875 const DebugLoc &DL = MI.getDebugLoc();
876 const RegisterBank *SrcBank = RBI.getRegBank(SrcReg, *MRI, TRI);
877
878 const TargetRegisterClass *SrcRC =
879 TRI.getRegClassForSizeOnBank(SrcSize, *SrcBank);
880
881 // Note we could have mixed SGPR and VGPR destination banks for an SGPR
882 // source, and this relies on the fact that the same subregister indices are
883 // used for both.
884 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(SrcRC, DstSize / 8);
885 for (int I = 0, E = NumDst; I != E; ++I) {
886 Register DstReg = MI.getOperand(I).getReg();
887 // hi16:sreg_32 is not allowed so explicitly shift upper 16-bits.
888 if (SrcBank->getID() == AMDGPU::SGPRRegBankID &&
889 SubRegs[I] == AMDGPU::hi16) {
890 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_LSHR_B32), DstReg)
891 .addReg(SrcReg)
892 .addImm(16);
893 } else {
894 BuildMI(*BB, &MI, DL, TII.get(TargetOpcode::COPY), DstReg)
895 .addReg(SrcReg, {}, SubRegs[I]);
896
897 // Make sure the subregister index is valid for the source register.
898 SrcRC = TRI.getSubClassWithSubReg(SrcRC, SubRegs[I]);
899 }
900
901 if (!SrcRC || !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI))
902 return false;
903
904 const TargetRegisterClass *DstRC =
905 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
906 if (DstRC && !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
907 return false;
908 }
909
910 MI.eraseFromParent();
911 return true;
912}
913
914bool AMDGPUInstructionSelector::selectG_BUILD_VECTOR(MachineInstr &MI) const {
915 assert(MI.getOpcode() == AMDGPU::G_BUILD_VECTOR_TRUNC ||
916 MI.getOpcode() == AMDGPU::G_BUILD_VECTOR);
917
918 Register Src0 = MI.getOperand(1).getReg();
919 Register Src1 = MI.getOperand(2).getReg();
920 LLT SrcTy = MRI->getType(Src0);
921 const unsigned SrcSize = SrcTy.getSizeInBits();
922
923 // BUILD_VECTOR with >=32 bits source is handled by MERGE_VALUE.
924 if (MI.getOpcode() == AMDGPU::G_BUILD_VECTOR && SrcSize >= 32) {
925 return selectG_MERGE_VALUES(MI);
926 }
927
928 // Selection logic below is for V2S16 only.
929 // For G_BUILD_VECTOR_TRUNC, additionally check that the operands are s32.
930 Register Dst = MI.getOperand(0).getReg();
931 if (MRI->getType(Dst) != LLT::fixed_vector(2, 16) ||
932 (MI.getOpcode() == AMDGPU::G_BUILD_VECTOR_TRUNC &&
933 SrcTy != LLT::scalar(32)))
934 return selectImpl(MI, *CoverageInfo);
935
936 const RegisterBank *DstBank = RBI.getRegBank(Dst, *MRI, TRI);
937 if (DstBank->getID() == AMDGPU::AGPRRegBankID)
938 return false;
939
940 assert(DstBank->getID() == AMDGPU::SGPRRegBankID ||
941 DstBank->getID() == AMDGPU::VGPRRegBankID);
942 const bool IsVector = DstBank->getID() == AMDGPU::VGPRRegBankID;
943
944 const DebugLoc &DL = MI.getDebugLoc();
945 MachineBasicBlock *BB = MI.getParent();
946
947 // First, before trying TableGen patterns, check if both sources are
948 // constants. In those cases, we can trivially compute the final constant
949 // and emit a simple move.
950 auto ConstSrc1 = getAnyConstantVRegValWithLookThrough(Src1, *MRI, true, true);
951 if (ConstSrc1) {
952 auto ConstSrc0 =
953 getAnyConstantVRegValWithLookThrough(Src0, *MRI, true, true);
954 if (ConstSrc0) {
955 const int64_t K0 = ConstSrc0->Value.getSExtValue();
956 const int64_t K1 = ConstSrc1->Value.getSExtValue();
957 uint32_t Lo16 = static_cast<uint32_t>(K0) & 0xffff;
958 uint32_t Hi16 = static_cast<uint32_t>(K1) & 0xffff;
959 uint32_t Imm = Lo16 | (Hi16 << 16);
960
961 // VALU
962 if (IsVector) {
963 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::V_MOV_B32_e32), Dst).addImm(Imm);
964 MI.eraseFromParent();
965 return RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI);
966 }
967
968 // SALU
969 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), Dst).addImm(Imm);
970 MI.eraseFromParent();
971 return RBI.constrainGenericRegister(Dst, AMDGPU::SReg_32RegClass, *MRI);
972 }
973 }
974
975 // Now try TableGen patterns.
976 if (selectImpl(MI, *CoverageInfo))
977 return true;
978
979 // TODO: This should probably be a combine somewhere
980 // (build_vector $src0, undef) -> copy $src0
981 MachineInstr *Src1Def = getDefIgnoringCopies(Src1, *MRI);
982 if (Src1Def->getOpcode() == AMDGPU::G_IMPLICIT_DEF) {
983 MI.setDesc(TII.get(AMDGPU::COPY));
984 MI.removeOperand(2);
985 const auto &RC =
986 IsVector ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
987 return RBI.constrainGenericRegister(Dst, RC, *MRI) &&
988 RBI.constrainGenericRegister(Src0, RC, *MRI);
989 }
990
991 return selectS16MergeToS32(MI);
992}
993
994bool AMDGPUInstructionSelector::selectG_IMPLICIT_DEF(MachineInstr &I) const {
995 const MachineOperand &MO = I.getOperand(0);
996
997 // FIXME: Interface for getConstrainedRegClassForReg needs work. The
998 // regbank check here is to know why getConstrainedRegClassForReg failed.
999 const TargetRegisterClass *RC =
1000 TRI.getConstrainedRegClassForReg(MO.getReg(), *MRI);
1001 if ((!RC && !MRI->getRegBankOrNull(MO.getReg())) ||
1002 (RC && RBI.constrainGenericRegister(MO.getReg(), *RC, *MRI))) {
1003 I.setDesc(TII.get(TargetOpcode::IMPLICIT_DEF));
1004 return true;
1005 }
1006
1007 return false;
1008}
1009
1010bool AMDGPUInstructionSelector::selectG_INSERT(MachineInstr &I) const {
1011 MachineBasicBlock *BB = I.getParent();
1012
1013 Register DstReg = I.getOperand(0).getReg();
1014 Register Src0Reg = I.getOperand(1).getReg();
1015 Register Src1Reg = I.getOperand(2).getReg();
1016 LLT Src1Ty = MRI->getType(Src1Reg);
1017
1018 unsigned DstSize = MRI->getType(DstReg).getSizeInBits();
1019 unsigned InsSize = Src1Ty.getSizeInBits();
1020
1021 int64_t Offset = I.getOperand(3).getImm();
1022
1023 // FIXME: These cases should have been illegal and unnecessary to check here.
1024 if (Offset % 32 != 0 || InsSize % 32 != 0)
1025 return false;
1026
1027 // Currently not handled by getSubRegFromChannel.
1028 if (InsSize > 128)
1029 return false;
1030
1031 unsigned SubReg = TRI.getSubRegFromChannel(Offset / 32, InsSize / 32);
1032 if (SubReg == AMDGPU::NoSubRegister)
1033 return false;
1034
1035 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
1036 const TargetRegisterClass *DstRC =
1037 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
1038 if (!DstRC)
1039 return false;
1040
1041 const RegisterBank *Src0Bank = RBI.getRegBank(Src0Reg, *MRI, TRI);
1042 const RegisterBank *Src1Bank = RBI.getRegBank(Src1Reg, *MRI, TRI);
1043 const TargetRegisterClass *Src0RC =
1044 TRI.getRegClassForSizeOnBank(DstSize, *Src0Bank);
1045 const TargetRegisterClass *Src1RC =
1046 TRI.getRegClassForSizeOnBank(InsSize, *Src1Bank);
1047
1048 // Deal with weird cases where the class only partially supports the subreg
1049 // index.
1050 Src0RC = TRI.getSubClassWithSubReg(Src0RC, SubReg);
1051 if (!Src0RC || !Src1RC)
1052 return false;
1053
1054 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
1055 !RBI.constrainGenericRegister(Src0Reg, *Src0RC, *MRI) ||
1056 !RBI.constrainGenericRegister(Src1Reg, *Src1RC, *MRI))
1057 return false;
1058
1059 const DebugLoc &DL = I.getDebugLoc();
1060 BuildMI(*BB, &I, DL, TII.get(TargetOpcode::INSERT_SUBREG), DstReg)
1061 .addReg(Src0Reg)
1062 .addReg(Src1Reg)
1063 .addImm(SubReg);
1064
1065 I.eraseFromParent();
1066 return true;
1067}
1068
1069bool AMDGPUInstructionSelector::selectG_SBFX_UBFX(MachineInstr &MI) const {
1070 Register DstReg = MI.getOperand(0).getReg();
1071 Register SrcReg = MI.getOperand(1).getReg();
1072 Register OffsetReg = MI.getOperand(2).getReg();
1073 Register WidthReg = MI.getOperand(3).getReg();
1074
1075 assert(RBI.getRegBank(DstReg, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID &&
1076 "scalar BFX instructions are expanded in regbankselect");
1077 assert(MRI->getType(MI.getOperand(0).getReg()).getSizeInBits() == 32 &&
1078 "64-bit vector BFX instructions are expanded in regbankselect");
1079
1080 const DebugLoc &DL = MI.getDebugLoc();
1081 MachineBasicBlock *MBB = MI.getParent();
1082
1083 bool IsSigned = MI.getOpcode() == TargetOpcode::G_SBFX;
1084 unsigned Opc = IsSigned ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
1085 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), DstReg)
1086 .addReg(SrcReg)
1087 .addReg(OffsetReg)
1088 .addReg(WidthReg);
1089 MI.eraseFromParent();
1090 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1091 return true;
1092}
1093
1094bool AMDGPUInstructionSelector::selectInterpP1F16(MachineInstr &MI) const {
1095 if (STI.getLDSBankCount() != 16)
1096 return selectImpl(MI, *CoverageInfo);
1097
1098 Register Dst = MI.getOperand(0).getReg();
1099 Register Src0 = MI.getOperand(2).getReg();
1100 Register M0Val = MI.getOperand(6).getReg();
1101 if (!RBI.constrainGenericRegister(M0Val, AMDGPU::SReg_32RegClass, *MRI) ||
1102 !RBI.constrainGenericRegister(Dst, AMDGPU::VGPR_32RegClass, *MRI) ||
1103 !RBI.constrainGenericRegister(Src0, AMDGPU::VGPR_32RegClass, *MRI))
1104 return false;
1105
1106 // This requires 2 instructions. It is possible to write a pattern to support
1107 // this, but the generated isel emitter doesn't correctly deal with multiple
1108 // output instructions using the same physical register input. The copy to m0
1109 // is incorrectly placed before the second instruction.
1110 //
1111 // TODO: Match source modifiers.
1112
1113 Register InterpMov = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1114 const DebugLoc &DL = MI.getDebugLoc();
1115 MachineBasicBlock *MBB = MI.getParent();
1116
1117 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1118 .addReg(M0Val);
1119 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_INTERP_MOV_F32), InterpMov)
1120 .addImm(2)
1121 .addImm(MI.getOperand(4).getImm()) // $attr
1122 .addImm(MI.getOperand(3).getImm()); // $attrchan
1123
1124 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_INTERP_P1LV_F16), Dst)
1125 .addImm(0) // $src0_modifiers
1126 .addReg(Src0) // $src0
1127 .addImm(MI.getOperand(4).getImm()) // $attr
1128 .addImm(MI.getOperand(3).getImm()) // $attrchan
1129 .addImm(0) // $src2_modifiers
1130 .addReg(InterpMov) // $src2 - 2 f16 values selected by high
1131 .addImm(MI.getOperand(5).getImm()) // $high
1132 .addImm(0) // $clamp
1133 .addImm(0); // $omod
1134
1135 MI.eraseFromParent();
1136 return true;
1137}
1138
1139// Writelane is special in that it can use SGPR and M0 (which would normally
1140// count as using the constant bus twice - but in this case it is allowed since
1141// the lane selector doesn't count as a use of the constant bus). However, it is
1142// still required to abide by the 1 SGPR rule. Fix this up if we might have
1143// multiple SGPRs.
1144bool AMDGPUInstructionSelector::selectWritelane(MachineInstr &MI) const {
1145 // With a constant bus limit of at least 2, there's no issue.
1146 if (STI.getConstantBusLimit(AMDGPU::V_WRITELANE_B32) > 1)
1147 return selectImpl(MI, *CoverageInfo);
1148
1149 MachineBasicBlock *MBB = MI.getParent();
1150 const DebugLoc &DL = MI.getDebugLoc();
1151 Register VDst = MI.getOperand(0).getReg();
1152 Register Val = MI.getOperand(2).getReg();
1153 Register LaneSelect = MI.getOperand(3).getReg();
1154 Register VDstIn = MI.getOperand(4).getReg();
1155
1156 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_WRITELANE_B32), VDst);
1157
1158 std::optional<ValueAndVReg> ConstSelect =
1159 getIConstantVRegValWithLookThrough(LaneSelect, *MRI);
1160 if (ConstSelect) {
1161 // The selector has to be an inline immediate, so we can use whatever for
1162 // the other operands.
1163 MIB.addReg(Val);
1164 MIB.addImm(ConstSelect->Value.getSExtValue() &
1165 maskTrailingOnes<uint64_t>(STI.getWavefrontSizeLog2()));
1166 } else {
1167 std::optional<ValueAndVReg> ConstVal =
1169
1170 // If the value written is an inline immediate, we can get away without a
1171 // copy to m0.
1172 if (ConstVal && AMDGPU::isInlinableLiteral32(ConstVal->Value.getSExtValue(),
1173 STI.hasInv2PiInlineImm())) {
1174 MIB.addImm(ConstVal->Value.getSExtValue());
1175 MIB.addReg(LaneSelect);
1176 } else {
1177 MIB.addReg(Val);
1178
1179 // If the lane selector was originally in a VGPR and copied with
1180 // readfirstlane, there's a hazard to read the same SGPR from the
1181 // VALU. Constrain to a different SGPR to help avoid needing a nop later.
1182 RBI.constrainGenericRegister(LaneSelect, AMDGPU::SReg_32_XM0RegClass, *MRI);
1183
1184 BuildMI(*MBB, *MIB, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1185 .addReg(LaneSelect);
1186 MIB.addReg(AMDGPU::M0);
1187 }
1188 }
1189
1190 MIB.addReg(VDstIn);
1191
1192 MI.eraseFromParent();
1193 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1194 return true;
1195}
1196
1197// We need to handle this here because tablegen doesn't support matching
1198// instructions with multiple outputs.
1199bool AMDGPUInstructionSelector::selectDivScale(MachineInstr &MI) const {
1200 Register Dst0 = MI.getOperand(0).getReg();
1201 Register Dst1 = MI.getOperand(1).getReg();
1202
1203 LLT Ty = MRI->getType(Dst0);
1204 unsigned Opc;
1205 if (Ty == LLT::scalar(32))
1206 Opc = AMDGPU::V_DIV_SCALE_F32_e64;
1207 else if (Ty == LLT::scalar(64))
1208 Opc = AMDGPU::V_DIV_SCALE_F64_e64;
1209 else
1210 return false;
1211
1212 // TODO: Match source modifiers.
1213
1214 const DebugLoc &DL = MI.getDebugLoc();
1215 MachineBasicBlock *MBB = MI.getParent();
1216
1217 Register Numer = MI.getOperand(3).getReg();
1218 Register Denom = MI.getOperand(4).getReg();
1219 unsigned ChooseDenom = MI.getOperand(5).getImm();
1220
1221 Register Src0 = ChooseDenom != 0 ? Numer : Denom;
1222
1223 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), Dst0)
1224 .addDef(Dst1)
1225 .addImm(0) // $src0_modifiers
1226 .addUse(Src0) // $src0
1227 .addImm(0) // $src1_modifiers
1228 .addUse(Denom) // $src1
1229 .addImm(0) // $src2_modifiers
1230 .addUse(Numer) // $src2
1231 .addImm(0) // $clamp
1232 .addImm(0); // $omod
1233
1234 MI.eraseFromParent();
1235 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1236 return true;
1237}
1238
1239bool AMDGPUInstructionSelector::selectG_INTRINSIC(MachineInstr &I) const {
1240 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(I).getIntrinsicID();
1241 switch (IntrinsicID) {
1242 case Intrinsic::amdgcn_if_break: {
1243 MachineBasicBlock *BB = I.getParent();
1244
1245 // FIXME: Manually selecting to avoid dealing with the SReg_1 trick
1246 // SelectionDAG uses for wave32 vs wave64.
1247 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::SI_IF_BREAK))
1248 .add(I.getOperand(0))
1249 .add(I.getOperand(2))
1250 .add(I.getOperand(3))
1251 .setOperandDead(3); // implicit-def $scc
1252
1253 Register DstReg = I.getOperand(0).getReg();
1254 Register Src0Reg = I.getOperand(2).getReg();
1255 Register Src1Reg = I.getOperand(3).getReg();
1256
1257 I.eraseFromParent();
1258
1259 for (Register Reg : { DstReg, Src0Reg, Src1Reg })
1260 MRI->setRegClass(Reg, TRI.getWaveMaskRegClass());
1261
1262 return true;
1263 }
1264 case Intrinsic::amdgcn_interp_p1_f16:
1265 return selectInterpP1F16(I);
1266 case Intrinsic::amdgcn_wqm:
1267 return constrainCopyLikeIntrin(I, AMDGPU::WQM);
1268 case Intrinsic::amdgcn_softwqm:
1269 return constrainCopyLikeIntrin(I, AMDGPU::SOFT_WQM);
1270 case Intrinsic::amdgcn_strict_wwm:
1271 case Intrinsic::amdgcn_wwm:
1272 return constrainCopyLikeIntrin(I, AMDGPU::STRICT_WWM);
1273 case Intrinsic::amdgcn_strict_wqm:
1274 return constrainCopyLikeIntrin(I, AMDGPU::STRICT_WQM);
1275 case Intrinsic::amdgcn_writelane:
1276 return selectWritelane(I);
1277 case Intrinsic::amdgcn_div_scale:
1278 return selectDivScale(I);
1279 case Intrinsic::amdgcn_ballot:
1280 return selectBallot(I);
1281 case Intrinsic::amdgcn_reloc_constant:
1282 return selectRelocConstant(I);
1283 case Intrinsic::amdgcn_groupstaticsize:
1284 return selectGroupStaticSize(I);
1285 case Intrinsic::returnaddress:
1286 return selectReturnAddress(I);
1287 case Intrinsic::amdgcn_smfmac_f32_16x16x32_f16:
1288 case Intrinsic::amdgcn_smfmac_f32_32x32x16_f16:
1289 case Intrinsic::amdgcn_smfmac_f32_16x16x32_bf16:
1290 case Intrinsic::amdgcn_smfmac_f32_32x32x16_bf16:
1291 case Intrinsic::amdgcn_smfmac_i32_16x16x64_i8:
1292 case Intrinsic::amdgcn_smfmac_i32_32x32x32_i8:
1293 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_bf8:
1294 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_fp8:
1295 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_bf8:
1296 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_fp8:
1297 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_bf8:
1298 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_fp8:
1299 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_bf8:
1300 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_fp8:
1301 case Intrinsic::amdgcn_smfmac_f32_16x16x64_f16:
1302 case Intrinsic::amdgcn_smfmac_f32_32x32x32_f16:
1303 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf16:
1304 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf16:
1305 case Intrinsic::amdgcn_smfmac_i32_16x16x128_i8:
1306 case Intrinsic::amdgcn_smfmac_i32_32x32x64_i8:
1307 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_bf8:
1308 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_fp8:
1309 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_bf8:
1310 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_fp8:
1311 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_bf8:
1312 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_fp8:
1313 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_bf8:
1314 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_fp8:
1315 return selectSMFMACIntrin(I);
1316 case Intrinsic::amdgcn_permlane16_swap:
1317 case Intrinsic::amdgcn_permlane32_swap:
1318 return selectPermlaneSwapIntrin(I, IntrinsicID);
1319 case Intrinsic::amdgcn_wave_shuffle:
1320 return selectWaveShuffleIntrin(I);
1321 default:
1322 return selectImpl(I, *CoverageInfo);
1323 }
1324}
1325
1327 const GCNSubtarget &ST) {
1328 if (Size != 16 && Size != 32 && Size != 64)
1329 return -1;
1330
1331 if (Size == 16 && !ST.has16BitInsts())
1332 return -1;
1333
1334 const auto Select = [&](unsigned S16Opc, unsigned TrueS16Opc,
1335 unsigned FakeS16Opc, unsigned S32Opc,
1336 unsigned S64Opc) {
1337 if (Size == 16)
1338 return ST.hasTrue16BitInsts()
1339 ? ST.useRealTrue16Insts() ? TrueS16Opc : FakeS16Opc
1340 : S16Opc;
1341 if (Size == 32)
1342 return S32Opc;
1343 return S64Opc;
1344 };
1345
1346 switch (P) {
1347 default:
1348 llvm_unreachable("Unknown condition code!");
1349 case CmpInst::ICMP_NE:
1350 return Select(AMDGPU::V_CMP_NE_U16_e64, AMDGPU::V_CMP_NE_U16_t16_e64,
1351 AMDGPU::V_CMP_NE_U16_fake16_e64, AMDGPU::V_CMP_NE_U32_e64,
1352 AMDGPU::V_CMP_NE_U64_e64);
1353 case CmpInst::ICMP_EQ:
1354 return Select(AMDGPU::V_CMP_EQ_U16_e64, AMDGPU::V_CMP_EQ_U16_t16_e64,
1355 AMDGPU::V_CMP_EQ_U16_fake16_e64, AMDGPU::V_CMP_EQ_U32_e64,
1356 AMDGPU::V_CMP_EQ_U64_e64);
1357 case CmpInst::ICMP_SGT:
1358 return Select(AMDGPU::V_CMP_GT_I16_e64, AMDGPU::V_CMP_GT_I16_t16_e64,
1359 AMDGPU::V_CMP_GT_I16_fake16_e64, AMDGPU::V_CMP_GT_I32_e64,
1360 AMDGPU::V_CMP_GT_I64_e64);
1361 case CmpInst::ICMP_SGE:
1362 return Select(AMDGPU::V_CMP_GE_I16_e64, AMDGPU::V_CMP_GE_I16_t16_e64,
1363 AMDGPU::V_CMP_GE_I16_fake16_e64, AMDGPU::V_CMP_GE_I32_e64,
1364 AMDGPU::V_CMP_GE_I64_e64);
1365 case CmpInst::ICMP_SLT:
1366 return Select(AMDGPU::V_CMP_LT_I16_e64, AMDGPU::V_CMP_LT_I16_t16_e64,
1367 AMDGPU::V_CMP_LT_I16_fake16_e64, AMDGPU::V_CMP_LT_I32_e64,
1368 AMDGPU::V_CMP_LT_I64_e64);
1369 case CmpInst::ICMP_SLE:
1370 return Select(AMDGPU::V_CMP_LE_I16_e64, AMDGPU::V_CMP_LE_I16_t16_e64,
1371 AMDGPU::V_CMP_LE_I16_fake16_e64, AMDGPU::V_CMP_LE_I32_e64,
1372 AMDGPU::V_CMP_LE_I64_e64);
1373 case CmpInst::ICMP_UGT:
1374 return Select(AMDGPU::V_CMP_GT_U16_e64, AMDGPU::V_CMP_GT_U16_t16_e64,
1375 AMDGPU::V_CMP_GT_U16_fake16_e64, AMDGPU::V_CMP_GT_U32_e64,
1376 AMDGPU::V_CMP_GT_U64_e64);
1377 case CmpInst::ICMP_UGE:
1378 return Select(AMDGPU::V_CMP_GE_U16_e64, AMDGPU::V_CMP_GE_U16_t16_e64,
1379 AMDGPU::V_CMP_GE_U16_fake16_e64, AMDGPU::V_CMP_GE_U32_e64,
1380 AMDGPU::V_CMP_GE_U64_e64);
1381 case CmpInst::ICMP_ULT:
1382 return Select(AMDGPU::V_CMP_LT_U16_e64, AMDGPU::V_CMP_LT_U16_t16_e64,
1383 AMDGPU::V_CMP_LT_U16_fake16_e64, AMDGPU::V_CMP_LT_U32_e64,
1384 AMDGPU::V_CMP_LT_U64_e64);
1385 case CmpInst::ICMP_ULE:
1386 return Select(AMDGPU::V_CMP_LE_U16_e64, AMDGPU::V_CMP_LE_U16_t16_e64,
1387 AMDGPU::V_CMP_LE_U16_fake16_e64, AMDGPU::V_CMP_LE_U32_e64,
1388 AMDGPU::V_CMP_LE_U64_e64);
1389
1390 case CmpInst::FCMP_OEQ:
1391 return Select(AMDGPU::V_CMP_EQ_F16_e64, AMDGPU::V_CMP_EQ_F16_t16_e64,
1392 AMDGPU::V_CMP_EQ_F16_fake16_e64, AMDGPU::V_CMP_EQ_F32_e64,
1393 AMDGPU::V_CMP_EQ_F64_e64);
1394 case CmpInst::FCMP_OGT:
1395 return Select(AMDGPU::V_CMP_GT_F16_e64, AMDGPU::V_CMP_GT_F16_t16_e64,
1396 AMDGPU::V_CMP_GT_F16_fake16_e64, AMDGPU::V_CMP_GT_F32_e64,
1397 AMDGPU::V_CMP_GT_F64_e64);
1398 case CmpInst::FCMP_OGE:
1399 return Select(AMDGPU::V_CMP_GE_F16_e64, AMDGPU::V_CMP_GE_F16_t16_e64,
1400 AMDGPU::V_CMP_GE_F16_fake16_e64, AMDGPU::V_CMP_GE_F32_e64,
1401 AMDGPU::V_CMP_GE_F64_e64);
1402 case CmpInst::FCMP_OLT:
1403 return Select(AMDGPU::V_CMP_LT_F16_e64, AMDGPU::V_CMP_LT_F16_t16_e64,
1404 AMDGPU::V_CMP_LT_F16_fake16_e64, AMDGPU::V_CMP_LT_F32_e64,
1405 AMDGPU::V_CMP_LT_F64_e64);
1406 case CmpInst::FCMP_OLE:
1407 return Select(AMDGPU::V_CMP_LE_F16_e64, AMDGPU::V_CMP_LE_F16_t16_e64,
1408 AMDGPU::V_CMP_LE_F16_fake16_e64, AMDGPU::V_CMP_LE_F32_e64,
1409 AMDGPU::V_CMP_LE_F64_e64);
1410 case CmpInst::FCMP_ONE:
1411 return Select(AMDGPU::V_CMP_NEQ_F16_e64, AMDGPU::V_CMP_NEQ_F16_t16_e64,
1412 AMDGPU::V_CMP_NEQ_F16_fake16_e64, AMDGPU::V_CMP_NEQ_F32_e64,
1413 AMDGPU::V_CMP_NEQ_F64_e64);
1414 case CmpInst::FCMP_ORD:
1415 return Select(AMDGPU::V_CMP_O_F16_e64, AMDGPU::V_CMP_O_F16_t16_e64,
1416 AMDGPU::V_CMP_O_F16_fake16_e64, AMDGPU::V_CMP_O_F32_e64,
1417 AMDGPU::V_CMP_O_F64_e64);
1418 case CmpInst::FCMP_UNO:
1419 return Select(AMDGPU::V_CMP_U_F16_e64, AMDGPU::V_CMP_U_F16_t16_e64,
1420 AMDGPU::V_CMP_U_F16_fake16_e64, AMDGPU::V_CMP_U_F32_e64,
1421 AMDGPU::V_CMP_U_F64_e64);
1422 case CmpInst::FCMP_UEQ:
1423 return Select(AMDGPU::V_CMP_NLG_F16_e64, AMDGPU::V_CMP_NLG_F16_t16_e64,
1424 AMDGPU::V_CMP_NLG_F16_fake16_e64, AMDGPU::V_CMP_NLG_F32_e64,
1425 AMDGPU::V_CMP_NLG_F64_e64);
1426 case CmpInst::FCMP_UGT:
1427 return Select(AMDGPU::V_CMP_NLE_F16_e64, AMDGPU::V_CMP_NLE_F16_t16_e64,
1428 AMDGPU::V_CMP_NLE_F16_fake16_e64, AMDGPU::V_CMP_NLE_F32_e64,
1429 AMDGPU::V_CMP_NLE_F64_e64);
1430 case CmpInst::FCMP_UGE:
1431 return Select(AMDGPU::V_CMP_NLT_F16_e64, AMDGPU::V_CMP_NLT_F16_t16_e64,
1432 AMDGPU::V_CMP_NLT_F16_fake16_e64, AMDGPU::V_CMP_NLT_F32_e64,
1433 AMDGPU::V_CMP_NLT_F64_e64);
1434 case CmpInst::FCMP_ULT:
1435 return Select(AMDGPU::V_CMP_NGE_F16_e64, AMDGPU::V_CMP_NGE_F16_t16_e64,
1436 AMDGPU::V_CMP_NGE_F16_fake16_e64, AMDGPU::V_CMP_NGE_F32_e64,
1437 AMDGPU::V_CMP_NGE_F64_e64);
1438 case CmpInst::FCMP_ULE:
1439 return Select(AMDGPU::V_CMP_NGT_F16_e64, AMDGPU::V_CMP_NGT_F16_t16_e64,
1440 AMDGPU::V_CMP_NGT_F16_fake16_e64, AMDGPU::V_CMP_NGT_F32_e64,
1441 AMDGPU::V_CMP_NGT_F64_e64);
1442 case CmpInst::FCMP_UNE:
1443 return Select(AMDGPU::V_CMP_NEQ_F16_e64, AMDGPU::V_CMP_NEQ_F16_t16_e64,
1444 AMDGPU::V_CMP_NEQ_F16_fake16_e64, AMDGPU::V_CMP_NEQ_F32_e64,
1445 AMDGPU::V_CMP_NEQ_F64_e64);
1446 case CmpInst::FCMP_TRUE:
1447 return Select(AMDGPU::V_CMP_TRU_F16_e64, AMDGPU::V_CMP_TRU_F16_t16_e64,
1448 AMDGPU::V_CMP_TRU_F16_fake16_e64, AMDGPU::V_CMP_TRU_F32_e64,
1449 AMDGPU::V_CMP_TRU_F64_e64);
1451 return Select(AMDGPU::V_CMP_F_F16_e64, AMDGPU::V_CMP_F_F16_t16_e64,
1452 AMDGPU::V_CMP_F_F16_fake16_e64, AMDGPU::V_CMP_F_F32_e64,
1453 AMDGPU::V_CMP_F_F64_e64);
1454 }
1455}
1456
1457int AMDGPUInstructionSelector::getS_CMPOpcode(CmpInst::Predicate P,
1458 unsigned Size) const {
1459 if (Size == 64) {
1460 if (!STI.hasScalarCompareEq64())
1461 return -1;
1462
1463 switch (P) {
1464 case CmpInst::ICMP_NE:
1465 return AMDGPU::S_CMP_LG_U64;
1466 case CmpInst::ICMP_EQ:
1467 return AMDGPU::S_CMP_EQ_U64;
1468 default:
1469 return -1;
1470 }
1471 }
1472
1473 if (Size == 32) {
1474 switch (P) {
1475 case CmpInst::ICMP_NE:
1476 return AMDGPU::S_CMP_LG_U32;
1477 case CmpInst::ICMP_EQ:
1478 return AMDGPU::S_CMP_EQ_U32;
1479 case CmpInst::ICMP_SGT:
1480 return AMDGPU::S_CMP_GT_I32;
1481 case CmpInst::ICMP_SGE:
1482 return AMDGPU::S_CMP_GE_I32;
1483 case CmpInst::ICMP_SLT:
1484 return AMDGPU::S_CMP_LT_I32;
1485 case CmpInst::ICMP_SLE:
1486 return AMDGPU::S_CMP_LE_I32;
1487 case CmpInst::ICMP_UGT:
1488 return AMDGPU::S_CMP_GT_U32;
1489 case CmpInst::ICMP_UGE:
1490 return AMDGPU::S_CMP_GE_U32;
1491 case CmpInst::ICMP_ULT:
1492 return AMDGPU::S_CMP_LT_U32;
1493 case CmpInst::ICMP_ULE:
1494 return AMDGPU::S_CMP_LE_U32;
1495 case CmpInst::FCMP_OEQ:
1496 return AMDGPU::S_CMP_EQ_F32;
1497 case CmpInst::FCMP_OGT:
1498 return AMDGPU::S_CMP_GT_F32;
1499 case CmpInst::FCMP_OGE:
1500 return AMDGPU::S_CMP_GE_F32;
1501 case CmpInst::FCMP_OLT:
1502 return AMDGPU::S_CMP_LT_F32;
1503 case CmpInst::FCMP_OLE:
1504 return AMDGPU::S_CMP_LE_F32;
1505 case CmpInst::FCMP_ONE:
1506 return AMDGPU::S_CMP_LG_F32;
1507 case CmpInst::FCMP_ORD:
1508 return AMDGPU::S_CMP_O_F32;
1509 case CmpInst::FCMP_UNO:
1510 return AMDGPU::S_CMP_U_F32;
1511 case CmpInst::FCMP_UEQ:
1512 return AMDGPU::S_CMP_NLG_F32;
1513 case CmpInst::FCMP_UGT:
1514 return AMDGPU::S_CMP_NLE_F32;
1515 case CmpInst::FCMP_UGE:
1516 return AMDGPU::S_CMP_NLT_F32;
1517 case CmpInst::FCMP_ULT:
1518 return AMDGPU::S_CMP_NGE_F32;
1519 case CmpInst::FCMP_ULE:
1520 return AMDGPU::S_CMP_NGT_F32;
1521 case CmpInst::FCMP_UNE:
1522 return AMDGPU::S_CMP_NEQ_F32;
1523 default:
1524 llvm_unreachable("Unknown condition code!");
1525 }
1526 }
1527
1528 if (Size == 16) {
1529 if (!STI.hasSALUFloatInsts())
1530 return -1;
1531
1532 switch (P) {
1533 case CmpInst::FCMP_OEQ:
1534 return AMDGPU::S_CMP_EQ_F16;
1535 case CmpInst::FCMP_OGT:
1536 return AMDGPU::S_CMP_GT_F16;
1537 case CmpInst::FCMP_OGE:
1538 return AMDGPU::S_CMP_GE_F16;
1539 case CmpInst::FCMP_OLT:
1540 return AMDGPU::S_CMP_LT_F16;
1541 case CmpInst::FCMP_OLE:
1542 return AMDGPU::S_CMP_LE_F16;
1543 case CmpInst::FCMP_ONE:
1544 return AMDGPU::S_CMP_LG_F16;
1545 case CmpInst::FCMP_ORD:
1546 return AMDGPU::S_CMP_O_F16;
1547 case CmpInst::FCMP_UNO:
1548 return AMDGPU::S_CMP_U_F16;
1549 case CmpInst::FCMP_UEQ:
1550 return AMDGPU::S_CMP_NLG_F16;
1551 case CmpInst::FCMP_UGT:
1552 return AMDGPU::S_CMP_NLE_F16;
1553 case CmpInst::FCMP_UGE:
1554 return AMDGPU::S_CMP_NLT_F16;
1555 case CmpInst::FCMP_ULT:
1556 return AMDGPU::S_CMP_NGE_F16;
1557 case CmpInst::FCMP_ULE:
1558 return AMDGPU::S_CMP_NGT_F16;
1559 case CmpInst::FCMP_UNE:
1560 return AMDGPU::S_CMP_NEQ_F16;
1561 default:
1562 llvm_unreachable("Unknown condition code!");
1563 }
1564 }
1565
1566 return -1;
1567}
1568
1569bool AMDGPUInstructionSelector::selectG_ICMP_or_FCMP(MachineInstr &I) const {
1570
1571 MachineBasicBlock *BB = I.getParent();
1572 const DebugLoc &DL = I.getDebugLoc();
1573
1574 Register SrcReg = I.getOperand(2).getReg();
1575 unsigned Size = RBI.getSizeInBits(SrcReg, *MRI, TRI);
1576
1577 auto Pred = (CmpInst::Predicate)I.getOperand(1).getPredicate();
1578
1579 Register CCReg = I.getOperand(0).getReg();
1580 if (!isVCC(CCReg, *MRI)) {
1581 int Opcode = getS_CMPOpcode(Pred, Size);
1582 if (Opcode == -1)
1583 return false;
1584 MachineInstr *ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode))
1585 .add(I.getOperand(2))
1586 .add(I.getOperand(3));
1587 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), CCReg)
1588 .addReg(AMDGPU::SCC);
1589 constrainSelectedInstRegOperands(*ICmp, TII, TRI, RBI);
1590 bool Ret =
1591 RBI.constrainGenericRegister(CCReg, AMDGPU::SReg_32RegClass, *MRI);
1592 I.eraseFromParent();
1593 return Ret;
1594 }
1595
1596 if (I.getOpcode() == AMDGPU::G_FCMP)
1597 return false;
1598
1599 int Opcode = getV_CMPOpcode(Pred, Size, *Subtarget);
1600 if (Opcode == -1)
1601 return false;
1602
1603 MachineInstrBuilder ICmp;
1604 // t16 instructions
1605 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src0_modifiers)) {
1606 ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode), I.getOperand(0).getReg())
1607 .addImm(0)
1608 .add(I.getOperand(2))
1609 .addImm(0)
1610 .add(I.getOperand(3))
1611 .addImm(0); // op_sel
1612 } else {
1613 ICmp = BuildMI(*BB, &I, DL, TII.get(Opcode), I.getOperand(0).getReg())
1614 .add(I.getOperand(2))
1615 .add(I.getOperand(3));
1616 }
1617
1618 RBI.constrainGenericRegister(ICmp->getOperand(0).getReg(),
1619 *TRI.getBoolRC(), *MRI);
1620 constrainSelectedInstRegOperands(*ICmp, TII, TRI, RBI);
1621 I.eraseFromParent();
1622 return true;
1623}
1624
1625// Ballot has to zero bits in input lane-mask that are zero in current exec,
1626// Done as AND with exec. For inputs that are results of instruction that
1627// implicitly use same exec, for example compares in same basic block or SCC to
1628// VCC copy, use copy.
1631 MachineInstr *MI = MRI.getVRegDef(Reg);
1632 if (MI->getParent() != MBB)
1633 return false;
1634
1635 // Lane mask generated by SCC to VCC copy.
1636 if (MI->getOpcode() == AMDGPU::COPY) {
1637 auto DstRB = MRI.getRegBankOrNull(MI->getOperand(0).getReg());
1638 auto SrcRB = MRI.getRegBankOrNull(MI->getOperand(1).getReg());
1639 if (DstRB && SrcRB && DstRB->getID() == AMDGPU::VCCRegBankID &&
1640 SrcRB->getID() == AMDGPU::SGPRRegBankID)
1641 return true;
1642 }
1643
1644 // Lane mask generated by SCC to VCC copy
1645 if (MI->getOpcode() == AMDGPU::G_AMDGPU_COPY_VCC_SCC)
1646 return true;
1647
1648 // Lane mask generated using compare with same exec.
1649 if (isa<GAnyCmp>(MI))
1650 return true;
1651
1652 Register LHS, RHS;
1653 // Look through AND.
1654 if (mi_match(Reg, MRI, m_GAnd(m_Reg(LHS), m_Reg(RHS))))
1655 return isLaneMaskFromSameBlock(LHS, MRI, MBB) ||
1657
1658 return false;
1659}
1660
1661bool AMDGPUInstructionSelector::selectBallot(MachineInstr &I) const {
1662 MachineBasicBlock *BB = I.getParent();
1663 const DebugLoc &DL = I.getDebugLoc();
1664 Register DstReg = I.getOperand(0).getReg();
1665 Register SrcReg = I.getOperand(2).getReg();
1666 const unsigned BallotSize = MRI->getType(DstReg).getSizeInBits();
1667 const unsigned WaveSize = STI.getWavefrontSize();
1668
1669 // In the common case, the return type matches the wave size.
1670 // However we also support emitting i64 ballots in wave32 mode.
1671 if (BallotSize != WaveSize && (BallotSize != 64 || WaveSize != 32))
1672 return false;
1673
1674 std::optional<ValueAndVReg> Arg =
1676
1677 Register Dst = DstReg;
1678 // i64 ballot on Wave32: new Dst(i32) for WaveSize ballot.
1679 if (BallotSize != WaveSize) {
1680 Dst = MRI->createVirtualRegister(TRI.getBoolRC());
1681 }
1682
1683 if (Arg) {
1684 const int64_t Value = Arg->Value.getZExtValue();
1685 if (Value == 0) {
1686 // Dst = S_MOV 0
1687 unsigned Opcode = WaveSize == 64 ? AMDGPU::S_MOV_B64 : AMDGPU::S_MOV_B32;
1688 BuildMI(*BB, &I, DL, TII.get(Opcode), Dst).addImm(0);
1689 } else {
1690 // Dst = COPY EXEC
1691 assert(Value == 1);
1692 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst).addReg(TRI.getExec());
1693 }
1694 if (!RBI.constrainGenericRegister(Dst, *TRI.getBoolRC(), *MRI))
1695 return false;
1696 } else {
1697 if (isLaneMaskFromSameBlock(SrcReg, *MRI, BB)) {
1698 // Dst = COPY SrcReg
1699 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), Dst).addReg(SrcReg);
1700 if (!RBI.constrainGenericRegister(Dst, *TRI.getBoolRC(), *MRI))
1701 return false;
1702 } else {
1703 // Dst = S_AND SrcReg, EXEC
1704 unsigned AndOpc = WaveSize == 64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
1705 auto And = BuildMI(*BB, &I, DL, TII.get(AndOpc), Dst)
1706 .addReg(SrcReg)
1707 .addReg(TRI.getExec())
1708 .setOperandDead(3); // Dead scc
1709 constrainSelectedInstRegOperands(*And, TII, TRI, RBI);
1710 }
1711 }
1712
1713 // i64 ballot on Wave32: zero-extend i32 ballot to i64.
1714 if (BallotSize != WaveSize) {
1715 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1716 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_MOV_B32), HiReg).addImm(0);
1717 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
1718 .addReg(Dst)
1719 .addImm(AMDGPU::sub0)
1720 .addReg(HiReg)
1721 .addImm(AMDGPU::sub1);
1722 }
1723
1724 I.eraseFromParent();
1725 return true;
1726}
1727
1728bool AMDGPUInstructionSelector::selectRelocConstant(MachineInstr &I) const {
1729 Register DstReg = I.getOperand(0).getReg();
1730 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
1731 const TargetRegisterClass *DstRC = TRI.getRegClassForSizeOnBank(32, *DstBank);
1732 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
1733 return false;
1734
1735 const bool IsVALU = DstBank->getID() == AMDGPU::VGPRRegBankID;
1736
1737 Module *M = MF->getFunction().getParent();
1738 const MDNode *Metadata = I.getOperand(2).getMetadata();
1739 auto SymbolName = cast<MDString>(Metadata->getOperand(0))->getString();
1740 auto *RelocSymbol = cast<GlobalVariable>(
1741 M->getOrInsertGlobal(SymbolName, Type::getInt32Ty(M->getContext())));
1742
1743 MachineBasicBlock *BB = I.getParent();
1744 BuildMI(*BB, &I, I.getDebugLoc(),
1745 TII.get(IsVALU ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32), DstReg)
1747
1748 I.eraseFromParent();
1749 return true;
1750}
1751
1752bool AMDGPUInstructionSelector::selectGroupStaticSize(MachineInstr &I) const {
1753 Triple::OSType OS = MF->getTarget().getTargetTriple().getOS();
1754
1755 Register DstReg = I.getOperand(0).getReg();
1756 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
1757 unsigned Mov = DstRB->getID() == AMDGPU::SGPRRegBankID ?
1758 AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1759
1760 MachineBasicBlock *MBB = I.getParent();
1761 const DebugLoc &DL = I.getDebugLoc();
1762
1763 auto MIB = BuildMI(*MBB, &I, DL, TII.get(Mov), DstReg);
1764
1765 if (OS == Triple::AMDHSA || OS == Triple::AMDPAL) {
1766 const SIMachineFunctionInfo *MFI = MF->getInfo<SIMachineFunctionInfo>();
1767 MIB.addImm(MFI->getLDSSize());
1768 } else {
1769 Module *M = MF->getFunction().getParent();
1770 const GlobalValue *GV =
1771 Intrinsic::getOrInsertDeclaration(M, Intrinsic::amdgcn_groupstaticsize);
1773 }
1774
1775 I.eraseFromParent();
1776 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
1777 return true;
1778}
1779
1780bool AMDGPUInstructionSelector::selectReturnAddress(MachineInstr &I) const {
1781 MachineBasicBlock *MBB = I.getParent();
1783 const DebugLoc &DL = I.getDebugLoc();
1784
1785 Register DstReg = I.getOperand(0).getReg();
1786 unsigned Depth = I.getOperand(2).getImm();
1787
1788 const TargetRegisterClass *RC =
1789 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
1790 if (!RC->hasSubClassEq(&AMDGPU::SGPR_64RegClass) ||
1791 !RBI.constrainGenericRegister(DstReg, *RC, *MRI))
1792 return false;
1793
1794 // Check for kernel and shader functions
1795 if (Depth != 0 ||
1796 MF.getInfo<SIMachineFunctionInfo>()->isEntryFunction()) {
1797 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_MOV_B64), DstReg)
1798 .addImm(0);
1799 I.eraseFromParent();
1800 return true;
1801 }
1802
1803 MachineFrameInfo &MFI = MF.getFrameInfo();
1804 // There is a call to @llvm.returnaddress in this function
1805 MFI.setReturnAddressIsTaken(true);
1806
1807 // Get the return address reg and mark it as an implicit live-in
1808 Register ReturnAddrReg = TRI.getReturnAddressReg(MF);
1809 Register LiveIn = getFunctionLiveInPhysReg(MF, TII, ReturnAddrReg,
1810 AMDGPU::SReg_64RegClass, DL);
1811 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), DstReg)
1812 .addReg(LiveIn);
1813 I.eraseFromParent();
1814 return true;
1815}
1816
1817bool AMDGPUInstructionSelector::selectEndCfIntrinsic(MachineInstr &MI) const {
1818 // FIXME: Manually selecting to avoid dealing with the SReg_1 trick
1819 // SelectionDAG uses for wave32 vs wave64.
1820 MachineBasicBlock *BB = MI.getParent();
1821 BuildMI(*BB, &MI, MI.getDebugLoc(), TII.get(AMDGPU::SI_END_CF))
1822 .add(MI.getOperand(1))
1823 .setOperandDead(2); // implicit-def $scc
1824
1825 Register Reg = MI.getOperand(1).getReg();
1826 MI.eraseFromParent();
1827
1828 if (!MRI->getRegClassOrNull(Reg))
1829 MRI->setRegClass(Reg, TRI.getWaveMaskRegClass());
1830 return true;
1831}
1832
1833bool AMDGPUInstructionSelector::selectDSOrderedIntrinsic(
1834 MachineInstr &MI, Intrinsic::ID IntrID) const {
1835 MachineBasicBlock *MBB = MI.getParent();
1837 const DebugLoc &DL = MI.getDebugLoc();
1838
1839 unsigned IndexOperand = MI.getOperand(7).getImm();
1840 bool WaveRelease = MI.getOperand(8).getImm() != 0;
1841 bool WaveDone = MI.getOperand(9).getImm() != 0;
1842
1843 if (WaveDone && !WaveRelease) {
1844 // TODO: Move this to IR verifier
1845 const Function &Fn = MF->getFunction();
1846 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1847 Fn, "ds_ordered_count: wave_done requires wave_release", DL));
1848 }
1849
1850 unsigned OrderedCountIndex = IndexOperand & 0x3f;
1851 IndexOperand &= ~0x3f;
1852 unsigned CountDw = 0;
1853
1854 if (STI.getGeneration() >= AMDGPUSubtarget::GFX10) {
1855 CountDw = (IndexOperand >> 24) & 0xf;
1856 IndexOperand &= ~(0xf << 24);
1857
1858 if (CountDw < 1 || CountDw > 4) {
1859 const Function &Fn = MF->getFunction();
1860 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1861 Fn, "ds_ordered_count: dword count must be between 1 and 4", DL));
1862 CountDw = 1;
1863 }
1864 }
1865
1866 if (IndexOperand) {
1867 const Function &Fn = MF->getFunction();
1868 Fn.getContext().diagnose(DiagnosticInfoUnsupported(
1869 Fn, "ds_ordered_count: bad index operand", DL));
1870 }
1871
1872 unsigned Instruction = IntrID == Intrinsic::amdgcn_ds_ordered_add ? 0 : 1;
1873 unsigned ShaderType = SIInstrInfo::getDSShaderTypeValue(*MF);
1874
1875 unsigned Offset0 = OrderedCountIndex << 2;
1876 unsigned Offset1 = WaveRelease | (WaveDone << 1) | (Instruction << 4);
1877
1878 if (STI.getGeneration() >= AMDGPUSubtarget::GFX10)
1879 Offset1 |= (CountDw - 1) << 6;
1880
1881 if (STI.getGeneration() < AMDGPUSubtarget::GFX11)
1882 Offset1 |= ShaderType << 2;
1883
1884 unsigned Offset = Offset0 | (Offset1 << 8);
1885
1886 Register M0Val = MI.getOperand(2).getReg();
1887 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1888 .addReg(M0Val);
1889
1890 Register DstReg = MI.getOperand(0).getReg();
1891 Register ValReg = MI.getOperand(3).getReg();
1892 MachineInstrBuilder DS =
1893 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::DS_ORDERED_COUNT), DstReg)
1894 .addReg(ValReg)
1895 .addImm(Offset)
1896 .cloneMemRefs(MI);
1897
1898 if (!RBI.constrainGenericRegister(M0Val, AMDGPU::SReg_32RegClass, *MRI))
1899 return false;
1900
1901 constrainSelectedInstRegOperands(*DS, TII, TRI, RBI);
1902 MI.eraseFromParent();
1903 return true;
1904}
1905
1906static unsigned gwsIntrinToOpcode(unsigned IntrID) {
1907 switch (IntrID) {
1908 case Intrinsic::amdgcn_ds_gws_init:
1909 return AMDGPU::DS_GWS_INIT;
1910 case Intrinsic::amdgcn_ds_gws_barrier:
1911 return AMDGPU::DS_GWS_BARRIER;
1912 case Intrinsic::amdgcn_ds_gws_sema_v:
1913 return AMDGPU::DS_GWS_SEMA_V;
1914 case Intrinsic::amdgcn_ds_gws_sema_br:
1915 return AMDGPU::DS_GWS_SEMA_BR;
1916 case Intrinsic::amdgcn_ds_gws_sema_p:
1917 return AMDGPU::DS_GWS_SEMA_P;
1918 case Intrinsic::amdgcn_ds_gws_sema_release_all:
1919 return AMDGPU::DS_GWS_SEMA_RELEASE_ALL;
1920 default:
1921 llvm_unreachable("not a gws intrinsic");
1922 }
1923}
1924
1925bool AMDGPUInstructionSelector::selectDSGWSIntrinsic(MachineInstr &MI,
1926 Intrinsic::ID IID) const {
1927 if (!STI.hasGWS() || (IID == Intrinsic::amdgcn_ds_gws_sema_release_all &&
1928 !STI.hasGWSSemaReleaseAll()))
1929 return false;
1930
1931 // intrinsic ID, vsrc, offset
1932 const bool HasVSrc = MI.getNumOperands() == 3;
1933 assert(HasVSrc || MI.getNumOperands() == 2);
1934
1935 Register BaseOffset = MI.getOperand(HasVSrc ? 2 : 1).getReg();
1936 const RegisterBank *OffsetRB = RBI.getRegBank(BaseOffset, *MRI, TRI);
1937 if (OffsetRB->getID() != AMDGPU::SGPRRegBankID)
1938 return false;
1939
1940 MachineInstr *OffsetDef = getDefIgnoringCopies(BaseOffset, *MRI);
1941 unsigned ImmOffset;
1942
1943 MachineBasicBlock *MBB = MI.getParent();
1944 const DebugLoc &DL = MI.getDebugLoc();
1945
1946 MachineInstr *Readfirstlane = nullptr;
1947
1948 // If we legalized the VGPR input, strip out the readfirstlane to analyze the
1949 // incoming offset, in case there's an add of a constant. We'll have to put it
1950 // back later.
1951 if (OffsetDef->getOpcode() == AMDGPU::V_READFIRSTLANE_B32) {
1952 Readfirstlane = OffsetDef;
1953 BaseOffset = OffsetDef->getOperand(1).getReg();
1954 OffsetDef = getDefIgnoringCopies(BaseOffset, *MRI);
1955 }
1956
1957 if (OffsetDef->getOpcode() == AMDGPU::G_CONSTANT) {
1958 // If we have a constant offset, try to use the 0 in m0 as the base.
1959 // TODO: Look into changing the default m0 initialization value. If the
1960 // default -1 only set the low 16-bits, we could leave it as-is and add 1 to
1961 // the immediate offset.
1962
1963 ImmOffset = OffsetDef->getOperand(1).getCImm()->getZExtValue();
1964 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), AMDGPU::M0)
1965 .addImm(0);
1966 } else {
1967 std::tie(BaseOffset, ImmOffset) =
1968 AMDGPU::getBaseWithConstantOffset(*MRI, BaseOffset, VT);
1969
1970 if (Readfirstlane) {
1971 // We have the constant offset now, so put the readfirstlane back on the
1972 // variable component.
1973 if (!RBI.constrainGenericRegister(BaseOffset, AMDGPU::VGPR_32RegClass, *MRI))
1974 return false;
1975
1976 Readfirstlane->getOperand(1).setReg(BaseOffset);
1977 BaseOffset = Readfirstlane->getOperand(0).getReg();
1978 } else {
1979 if (!RBI.constrainGenericRegister(BaseOffset,
1980 AMDGPU::SReg_32RegClass, *MRI))
1981 return false;
1982 }
1983
1984 Register M0Base = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1985 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::S_LSHL_B32), M0Base)
1986 .addReg(BaseOffset)
1987 .addImm(16)
1988 .setOperandDead(3); // Dead scc
1989
1990 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
1991 .addReg(M0Base);
1992 }
1993
1994 // The resource id offset is computed as (<isa opaque base> + M0[21:16] +
1995 // offset field) % 64. Some versions of the programming guide omit the m0
1996 // part, or claim it's from offset 0.
1997
1998 unsigned Opc = gwsIntrinToOpcode(IID);
1999 const MCInstrDesc &InstrDesc = TII.get(Opc);
2000
2001 if (HasVSrc) {
2002 Register VSrc = MI.getOperand(1).getReg();
2003
2004 int Data0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
2005 const TargetRegisterClass *DataRC = TII.getRegClass(InstrDesc, Data0Idx);
2006 const TargetRegisterClass *SubRC =
2007 TRI.getSubRegisterClass(DataRC, AMDGPU::sub0);
2008
2009 if (!SubRC) {
2010 // 32-bit normal case.
2011 if (!RBI.constrainGenericRegister(VSrc, *DataRC, *MRI))
2012 return false;
2013
2014 BuildMI(*MBB, &MI, DL, InstrDesc)
2015 .addReg(VSrc)
2016 .addImm(ImmOffset)
2017 .cloneMemRefs(MI);
2018 } else {
2019 // Requires even register alignment, so create 64-bit value and pad the
2020 // top half with undef.
2021 Register DataReg = MRI->createVirtualRegister(DataRC);
2022 if (!RBI.constrainGenericRegister(VSrc, *SubRC, *MRI))
2023 return false;
2024
2025 Register UndefReg = MRI->createVirtualRegister(SubRC);
2026 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2027 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), DataReg)
2028 .addReg(VSrc)
2029 .addImm(AMDGPU::sub0)
2030 .addReg(UndefReg)
2031 .addImm(AMDGPU::sub1);
2032
2033 BuildMI(*MBB, &MI, DL, InstrDesc)
2034 .addReg(DataReg)
2035 .addImm(ImmOffset)
2036 .cloneMemRefs(MI);
2037 }
2038 } else {
2039 BuildMI(*MBB, &MI, DL, InstrDesc)
2040 .addImm(ImmOffset)
2041 .cloneMemRefs(MI);
2042 }
2043
2044 MI.eraseFromParent();
2045 return true;
2046}
2047
2048bool AMDGPUInstructionSelector::selectDSAppendConsume(MachineInstr &MI,
2049 bool IsAppend) const {
2050 Register PtrBase = MI.getOperand(2).getReg();
2051 LLT PtrTy = MRI->getType(PtrBase);
2052 bool IsGDS = PtrTy.getAddressSpace() == AMDGPUAS::REGION_ADDRESS;
2053
2054 unsigned Offset;
2055 std::tie(PtrBase, Offset) = selectDS1Addr1OffsetImpl(MI.getOperand(2));
2056
2057 // TODO: Should this try to look through readfirstlane like GWS?
2058 if (!isDSOffsetLegal(PtrBase, Offset)) {
2059 PtrBase = MI.getOperand(2).getReg();
2060 Offset = 0;
2061 }
2062
2063 MachineBasicBlock *MBB = MI.getParent();
2064 const DebugLoc &DL = MI.getDebugLoc();
2065 const unsigned Opc = IsAppend ? AMDGPU::DS_APPEND : AMDGPU::DS_CONSUME;
2066
2067 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
2068 .addReg(PtrBase);
2069 if (!RBI.constrainGenericRegister(PtrBase, AMDGPU::SReg_32RegClass, *MRI))
2070 return false;
2071
2072 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), MI.getOperand(0).getReg())
2073 .addImm(Offset)
2074 .addImm(IsGDS ? -1 : 0)
2075 .cloneMemRefs(MI);
2076 MI.eraseFromParent();
2077 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2078 return true;
2079}
2080
2081bool AMDGPUInstructionSelector::selectInitWholeWave(MachineInstr &MI) const {
2082 MachineFunction *MF = MI.getMF();
2083 SIMachineFunctionInfo *MFInfo = MF->getInfo<SIMachineFunctionInfo>();
2084
2085 MFInfo->setInitWholeWave();
2086 return selectImpl(MI, *CoverageInfo);
2087}
2088
2089static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE,
2090 bool &IsTexFail) {
2091 if (TexFailCtrl)
2092 IsTexFail = true;
2093
2094 TFE = TexFailCtrl & 0x1;
2095 TexFailCtrl &= ~(uint64_t)0x1;
2096 LWE = TexFailCtrl & 0x2;
2097 TexFailCtrl &= ~(uint64_t)0x2;
2098
2099 return TexFailCtrl == 0;
2100}
2101
2102bool AMDGPUInstructionSelector::selectImageIntrinsic(
2103 MachineInstr &MI, const AMDGPU::ImageDimIntrinsicInfo *Intr) const {
2104 MachineBasicBlock *MBB = MI.getParent();
2105 const DebugLoc &DL = MI.getDebugLoc();
2106 unsigned IntrOpcode = Intr->BaseOpcode;
2107
2108 // For image atomic: use no-return opcode if result is unused.
2109 if (Intr->AtomicNoRetBaseOpcode != Intr->BaseOpcode) {
2110 Register ResultDef = MI.getOperand(0).getReg();
2111 if (MRI->use_nodbg_empty(ResultDef))
2112 IntrOpcode = Intr->AtomicNoRetBaseOpcode;
2113 }
2114
2115 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
2117
2118 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfo(Intr->Dim);
2119 const bool IsGFX10Plus = AMDGPU::isGFX10Plus(STI);
2120 const bool IsGFX11Plus = AMDGPU::isGFX11Plus(STI);
2121 const bool IsGFX12Plus = AMDGPU::isGFX12Plus(STI);
2122 const bool IsGFX13Plus = AMDGPU::isGFX13Plus(STI);
2123
2124 const unsigned ArgOffset = MI.getNumExplicitDefs() + 1;
2125
2126 Register VDataIn = AMDGPU::NoRegister;
2127 Register VDataOut = AMDGPU::NoRegister;
2128 LLT VDataTy;
2129 int NumVDataDwords = -1;
2130 bool IsD16 = MI.getOpcode() == AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16 ||
2131 MI.getOpcode() == AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16;
2132
2133 bool Unorm;
2134 if (!BaseOpcode->Sampler)
2135 Unorm = true;
2136 else
2137 Unorm = MI.getOperand(ArgOffset + Intr->UnormIndex).getImm() != 0;
2138
2139 bool TFE;
2140 bool LWE;
2141 bool IsTexFail = false;
2142 if (!parseTexFail(MI.getOperand(ArgOffset + Intr->TexFailCtrlIndex).getImm(),
2143 TFE, LWE, IsTexFail))
2144 return false;
2145
2146 const int Flags = MI.getOperand(ArgOffset + Intr->NumArgs).getImm();
2147 const bool IsA16 = (Flags & 1) != 0;
2148 const bool IsG16 = (Flags & 2) != 0;
2149
2150 // A16 implies 16 bit gradients if subtarget doesn't support G16
2151 if (IsA16 && !STI.hasG16() && !IsG16)
2152 return false;
2153
2154 unsigned DMask = 0;
2155 unsigned DMaskLanes = 0;
2156
2157 if (BaseOpcode->Atomic) {
2158 if (!BaseOpcode->NoReturn)
2159 VDataOut = MI.getOperand(0).getReg();
2160 VDataIn = MI.getOperand(2).getReg();
2161 LLT Ty = MRI->getType(VDataIn);
2162
2163 // Be careful to allow atomic swap on 16-bit element vectors.
2164 const bool Is64Bit = BaseOpcode->AtomicX2 ?
2165 Ty.getSizeInBits() == 128 :
2166 Ty.getSizeInBits() == 64;
2167
2168 if (BaseOpcode->AtomicX2) {
2169 assert(MI.getOperand(3).getReg() == AMDGPU::NoRegister);
2170
2171 DMask = Is64Bit ? 0xf : 0x3;
2172 NumVDataDwords = Is64Bit ? 4 : 2;
2173 } else {
2174 DMask = Is64Bit ? 0x3 : 0x1;
2175 NumVDataDwords = Is64Bit ? 2 : 1;
2176 }
2177 } else {
2178 DMask = MI.getOperand(ArgOffset + Intr->DMaskIndex).getImm();
2179 DMaskLanes = BaseOpcode->Gather4 ? 4 : llvm::popcount(DMask);
2180
2181 if (BaseOpcode->Store) {
2182 VDataIn = MI.getOperand(1).getReg();
2183 VDataTy = MRI->getType(VDataIn);
2184 NumVDataDwords = (VDataTy.getSizeInBits() + 31) / 32;
2185 } else if (BaseOpcode->NoReturn) {
2186 NumVDataDwords = 0;
2187 } else {
2188 VDataOut = MI.getOperand(0).getReg();
2189 VDataTy = MRI->getType(VDataOut);
2190 NumVDataDwords = DMaskLanes;
2191
2192 if (IsD16 && !STI.hasUnpackedD16VMem())
2193 NumVDataDwords = (DMaskLanes + 1) / 2;
2194 }
2195 }
2196
2197 // Set G16 opcode
2198 if (Subtarget->hasG16() && IsG16) {
2199 const AMDGPU::MIMGG16MappingInfo *G16MappingInfo =
2201 assert(G16MappingInfo);
2202 IntrOpcode = G16MappingInfo->G16; // set opcode to variant with _g16
2203 }
2204
2205 // TODO: Check this in verifier.
2206 assert((!IsTexFail || DMaskLanes >= 1) && "should have legalized this");
2207
2208 unsigned CPol = MI.getOperand(ArgOffset + Intr->CachePolicyIndex).getImm();
2209 // Keep GLC only when the atomic's result is actually used.
2210 if (BaseOpcode->Atomic && !BaseOpcode->NoReturn)
2212 if (CPol & ~((IsGFX12Plus ? AMDGPU::CPol::ALL : AMDGPU::CPol::ALL_pregfx12) |
2214 return false;
2215
2216 int NumVAddrRegs = 0;
2217 int NumVAddrDwords = 0;
2218 for (unsigned I = Intr->VAddrStart; I < Intr->VAddrEnd; I++) {
2219 // Skip the $noregs and 0s inserted during legalization.
2220 MachineOperand &AddrOp = MI.getOperand(ArgOffset + I);
2221 if (!AddrOp.isReg())
2222 continue; // XXX - Break?
2223
2224 Register Addr = AddrOp.getReg();
2225 if (!Addr)
2226 break;
2227
2228 ++NumVAddrRegs;
2229 NumVAddrDwords += (MRI->getType(Addr).getSizeInBits() + 31) / 32;
2230 }
2231
2232 // The legalizer preprocessed the intrinsic arguments. If we aren't using
2233 // NSA, these should have been packed into a single value in the first
2234 // address register
2235 const bool UseNSA =
2236 NumVAddrRegs != 1 &&
2237 (STI.hasPartialNSAEncoding() ? NumVAddrDwords >= NumVAddrRegs
2238 : NumVAddrDwords == NumVAddrRegs);
2239 if (UseNSA && !STI.hasFeature(AMDGPU::FeatureNSAEncoding)) {
2240 LLVM_DEBUG(dbgs() << "Trying to use NSA on non-NSA target\n");
2241 return false;
2242 }
2243
2244 if (IsTexFail)
2245 ++NumVDataDwords;
2246
2247 int Opcode = -1;
2248 if (IsGFX13Plus) {
2249 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx13,
2250 NumVDataDwords, NumVAddrDwords);
2251 } else if (IsGFX12Plus) {
2252 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx12,
2253 NumVDataDwords, NumVAddrDwords);
2254 } else if (IsGFX11Plus) {
2255 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
2256 UseNSA ? AMDGPU::MIMGEncGfx11NSA
2257 : AMDGPU::MIMGEncGfx11Default,
2258 NumVDataDwords, NumVAddrDwords);
2259 } else if (IsGFX10Plus) {
2260 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
2261 UseNSA ? AMDGPU::MIMGEncGfx10NSA
2262 : AMDGPU::MIMGEncGfx10Default,
2263 NumVDataDwords, NumVAddrDwords);
2264 } else {
2265 if (Subtarget->hasGFX90AInsts()) {
2266 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx90a,
2267 NumVDataDwords, NumVAddrDwords);
2268 if (Opcode == -1) {
2269 LLVM_DEBUG(
2270 dbgs()
2271 << "requested image instruction is not supported on this GPU\n");
2272 return false;
2273 }
2274 }
2275 if (Opcode == -1 &&
2276 STI.getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
2277 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx8,
2278 NumVDataDwords, NumVAddrDwords);
2279 if (Opcode == -1)
2280 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx6,
2281 NumVDataDwords, NumVAddrDwords);
2282 }
2283 if (Opcode == -1)
2284 return false;
2285
2286 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opcode))
2287 .cloneMemRefs(MI);
2288
2289 if (VDataOut) {
2290 if (BaseOpcode->AtomicX2) {
2291 const bool Is64 = MRI->getType(VDataOut).getSizeInBits() == 64;
2292
2293 Register TmpReg = MRI->createVirtualRegister(
2294 Is64 ? &AMDGPU::VReg_128RegClass : &AMDGPU::VReg_64RegClass);
2295 unsigned SubReg = Is64 ? AMDGPU::sub0_sub1 : AMDGPU::sub0;
2296
2297 MIB.addDef(TmpReg);
2298 if (!MRI->use_empty(VDataOut)) {
2299 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), VDataOut)
2300 .addReg(TmpReg, RegState::Kill, SubReg);
2301 }
2302
2303 } else {
2304 MIB.addDef(VDataOut); // vdata output
2305 }
2306 }
2307
2308 if (VDataIn)
2309 MIB.addReg(VDataIn); // vdata input
2310
2311 for (int I = 0; I != NumVAddrRegs; ++I) {
2312 MachineOperand &SrcOp = MI.getOperand(ArgOffset + Intr->VAddrStart + I);
2313 if (SrcOp.isReg()) {
2314 assert(SrcOp.getReg() != 0);
2315 MIB.addReg(SrcOp.getReg());
2316 }
2317 }
2318
2319 MIB.addReg(MI.getOperand(ArgOffset + Intr->RsrcIndex).getReg());
2320 if (BaseOpcode->Sampler)
2321 MIB.addReg(MI.getOperand(ArgOffset + Intr->SampIndex).getReg());
2322
2323 MIB.addImm(DMask); // dmask
2324
2325 if (IsGFX10Plus)
2326 MIB.addImm(DimInfo->Encoding);
2327 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::unorm))
2328 MIB.addImm(Unorm);
2329
2330 MIB.addImm(CPol);
2331 MIB.addImm(IsA16 && // a16 or r128
2332 STI.hasFeature(AMDGPU::FeatureR128A16) ? -1 : 0);
2333 if (IsGFX10Plus)
2334 MIB.addImm(IsA16 ? -1 : 0);
2335
2336 if (!Subtarget->hasGFX90AInsts()) {
2337 MIB.addImm(TFE); // tfe
2338 } else if (TFE) {
2339 LLVM_DEBUG(dbgs() << "TFE is not supported on this GPU\n");
2340 return false;
2341 }
2342
2343 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::lwe))
2344 MIB.addImm(LWE); // lwe
2345 if (!IsGFX10Plus)
2346 MIB.addImm(DimInfo->DA ? -1 : 0);
2347 if (BaseOpcode->HasD16)
2348 MIB.addImm(IsD16 ? -1 : 0);
2349
2350 MI.eraseFromParent();
2351 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2352 TII.enforceOperandRCAlignment(*MIB, AMDGPU::OpName::vaddr);
2353 return true;
2354}
2355
2356// We need to handle this here because tablegen doesn't support matching
2357// instructions with multiple outputs.
2358bool AMDGPUInstructionSelector::selectDSBvhStackIntrinsic(
2359 MachineInstr &MI) const {
2360 Register Dst0 = MI.getOperand(0).getReg();
2361 Register Dst1 = MI.getOperand(1).getReg();
2362
2363 const DebugLoc &DL = MI.getDebugLoc();
2364 MachineBasicBlock *MBB = MI.getParent();
2365
2366 Register Addr = MI.getOperand(3).getReg();
2367 Register Data0 = MI.getOperand(4).getReg();
2368 Register Data1 = MI.getOperand(5).getReg();
2369 unsigned Offset = MI.getOperand(6).getImm();
2370
2371 unsigned Opc;
2372 switch (cast<GIntrinsic>(MI).getIntrinsicID()) {
2373 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
2374 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
2375 Opc = AMDGPU::DS_BVH_STACK_RTN_B32;
2376 break;
2377 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
2378 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP1_RTN_B32;
2379 break;
2380 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
2381 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP2_RTN_B64;
2382 break;
2383 }
2384
2385 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc), Dst0)
2386 .addDef(Dst1)
2387 .addUse(Addr)
2388 .addUse(Data0)
2389 .addUse(Data1)
2390 .addImm(Offset)
2391 .cloneMemRefs(MI);
2392
2393 MI.eraseFromParent();
2394 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
2395 return true;
2396}
2397
2398bool AMDGPUInstructionSelector::selectG_INTRINSIC_W_SIDE_EFFECTS(
2399 MachineInstr &I) const {
2400 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(I).getIntrinsicID();
2401 switch (IntrinsicID) {
2402 case Intrinsic::amdgcn_end_cf:
2403 return selectEndCfIntrinsic(I);
2404 case Intrinsic::amdgcn_ds_ordered_add:
2405 case Intrinsic::amdgcn_ds_ordered_swap:
2406 return selectDSOrderedIntrinsic(I, IntrinsicID);
2407 case Intrinsic::amdgcn_ds_gws_init:
2408 case Intrinsic::amdgcn_ds_gws_barrier:
2409 case Intrinsic::amdgcn_ds_gws_sema_v:
2410 case Intrinsic::amdgcn_ds_gws_sema_br:
2411 case Intrinsic::amdgcn_ds_gws_sema_p:
2412 case Intrinsic::amdgcn_ds_gws_sema_release_all:
2413 return selectDSGWSIntrinsic(I, IntrinsicID);
2414 case Intrinsic::amdgcn_ds_append:
2415 return selectDSAppendConsume(I, true);
2416 case Intrinsic::amdgcn_ds_consume:
2417 return selectDSAppendConsume(I, false);
2418 case Intrinsic::amdgcn_init_whole_wave:
2419 return selectInitWholeWave(I);
2420 case Intrinsic::amdgcn_raw_buffer_load_lds:
2421 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
2422 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
2423 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
2424 case Intrinsic::amdgcn_struct_buffer_load_lds:
2425 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
2426 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
2427 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
2428 return selectBufferLoadLds(I);
2429 // Until we can store both the address space of the global and the LDS
2430 // arguments by having tto MachineMemOperands on an intrinsic, we just trust
2431 // that the argument is a global pointer (buffer pointers have been handled by
2432 // a LLVM IR-level lowering).
2433 case Intrinsic::amdgcn_load_to_lds:
2434 case Intrinsic::amdgcn_load_async_to_lds:
2435 case Intrinsic::amdgcn_global_load_lds:
2436 case Intrinsic::amdgcn_global_load_async_lds:
2437 return selectGlobalLoadLds(I);
2438 case Intrinsic::amdgcn_tensor_load_to_lds:
2439 case Intrinsic::amdgcn_tensor_store_from_lds:
2440 return selectTensorLoadStore(I, IntrinsicID);
2441 case Intrinsic::amdgcn_asyncmark:
2442 case Intrinsic::amdgcn_wait_asyncmark:
2443 if (!Subtarget->hasAsyncMark())
2444 return false;
2445 break;
2446 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
2447 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
2448 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
2449 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
2450 return selectDSBvhStackIntrinsic(I);
2451 case Intrinsic::amdgcn_s_alloc_vgpr: {
2452 // S_ALLOC_VGPR doesn't have a destination register, it just implicitly sets
2453 // SCC. We then need to COPY it into the result vreg.
2454 MachineBasicBlock *MBB = I.getParent();
2455 const DebugLoc &DL = I.getDebugLoc();
2456
2457 Register ResReg = I.getOperand(0).getReg();
2458
2459 MachineInstr *AllocMI = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_ALLOC_VGPR))
2460 .add(I.getOperand(2));
2461 (void)BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), ResReg)
2462 .addReg(AMDGPU::SCC);
2463 I.eraseFromParent();
2464 constrainSelectedInstRegOperands(*AllocMI, TII, TRI, RBI);
2465 return RBI.constrainGenericRegister(ResReg, AMDGPU::SReg_32RegClass, *MRI);
2466 }
2467 case Intrinsic::amdgcn_s_barrier_init:
2468 case Intrinsic::amdgcn_s_barrier_signal_var:
2469 return selectNamedBarrierInit(I, IntrinsicID);
2470 case Intrinsic::amdgcn_s_wakeup_barrier:
2471 case Intrinsic::amdgcn_s_barrier_join:
2472 case Intrinsic::amdgcn_s_get_named_barrier_state:
2473 return selectNamedBarrierInst(I, IntrinsicID);
2474 case Intrinsic::amdgcn_s_get_barrier_state:
2475 return selectSGetBarrierState(I, IntrinsicID);
2476 case Intrinsic::amdgcn_s_barrier_signal_isfirst:
2477 return selectSBarrierSignalIsfirst(I, IntrinsicID);
2478 }
2479 return selectImpl(I, *CoverageInfo);
2480}
2481
2482bool AMDGPUInstructionSelector::selectG_SELECT(MachineInstr &I) const {
2483 if (selectImpl(I, *CoverageInfo))
2484 return true;
2485
2486 MachineBasicBlock *BB = I.getParent();
2487 const DebugLoc &DL = I.getDebugLoc();
2488
2489 Register DstReg = I.getOperand(0).getReg();
2490 unsigned Size = RBI.getSizeInBits(DstReg, *MRI, TRI);
2491 assert(Size <= 32 || Size == 64);
2492 const MachineOperand &CCOp = I.getOperand(1);
2493 Register CCReg = CCOp.getReg();
2494 if (!isVCC(CCReg, *MRI)) {
2495 unsigned SelectOpcode = Size == 64 ? AMDGPU::S_CSELECT_B64 :
2496 AMDGPU::S_CSELECT_B32;
2497 MachineInstr *CopySCC = BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::SCC)
2498 .addReg(CCReg);
2499
2500 // The generic constrainSelectedInstRegOperands doesn't work for the scc register
2501 // bank, because it does not cover the register class that we used to represent
2502 // for it. So we need to manually set the register class here.
2503 if (!MRI->getRegClassOrNull(CCReg))
2504 MRI->setRegClass(CCReg, TRI.getConstrainedRegClassForReg(CCReg, *MRI));
2505 MachineInstr *Select = BuildMI(*BB, &I, DL, TII.get(SelectOpcode), DstReg)
2506 .add(I.getOperand(2))
2507 .add(I.getOperand(3));
2508
2510 constrainSelectedInstRegOperands(*CopySCC, TII, TRI, RBI);
2511 I.eraseFromParent();
2512 return true;
2513 }
2514
2515 // Wide VGPR select should have been split in RegBankSelect.
2516 if (Size > 32)
2517 return false;
2518
2519 MachineInstr *Select =
2520 BuildMI(*BB, &I, DL, TII.get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
2521 .addImm(0)
2522 .add(I.getOperand(3))
2523 .addImm(0)
2524 .add(I.getOperand(2))
2525 .add(I.getOperand(1));
2526
2528 I.eraseFromParent();
2529 return true;
2530}
2531
2532bool AMDGPUInstructionSelector::selectG_TRUNC(MachineInstr &I) const {
2533 Register DstReg = I.getOperand(0).getReg();
2534 Register SrcReg = I.getOperand(1).getReg();
2535 const LLT DstTy = MRI->getType(DstReg);
2536 const LLT SrcTy = MRI->getType(SrcReg);
2537 const LLT S1 = LLT::scalar(1);
2538
2539 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
2540 const RegisterBank *DstRB;
2541 if (DstTy == S1) {
2542 // This is a special case. We don't treat s1 for legalization artifacts as
2543 // vcc booleans.
2544 DstRB = SrcRB;
2545 } else {
2546 DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
2547 if (SrcRB != DstRB)
2548 return false;
2549 }
2550
2551 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
2552
2553 unsigned DstSize = DstTy.getSizeInBits();
2554 unsigned SrcSize = SrcTy.getSizeInBits();
2555
2556 const TargetRegisterClass *SrcRC =
2557 TRI.getRegClassForSizeOnBank(SrcSize, *SrcRB);
2558 const TargetRegisterClass *DstRC =
2559 TRI.getRegClassForSizeOnBank(DstSize, *DstRB);
2560 if (!SrcRC || !DstRC)
2561 return false;
2562
2563 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
2564 !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI)) {
2565 LLVM_DEBUG(dbgs() << "Failed to constrain G_TRUNC\n");
2566 return false;
2567 }
2568
2569 if (DstRC == &AMDGPU::VGPR_16RegClass && SrcSize == 32) {
2570 assert(STI.useRealTrue16Insts());
2571 const DebugLoc &DL = I.getDebugLoc();
2572 MachineBasicBlock *MBB = I.getParent();
2573 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), DstReg)
2574 .addReg(SrcReg, {}, AMDGPU::lo16);
2575 I.eraseFromParent();
2576 return true;
2577 }
2578
2579 if (DstTy == LLT::fixed_vector(2, 16) && SrcTy == LLT::fixed_vector(2, 32)) {
2580 MachineBasicBlock *MBB = I.getParent();
2581 const DebugLoc &DL = I.getDebugLoc();
2582
2583 Register LoReg = MRI->createVirtualRegister(DstRC);
2584 Register HiReg = MRI->createVirtualRegister(DstRC);
2585 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), LoReg)
2586 .addReg(SrcReg, {}, AMDGPU::sub0);
2587 BuildMI(*MBB, I, DL, TII.get(AMDGPU::COPY), HiReg)
2588 .addReg(SrcReg, {}, AMDGPU::sub1);
2589
2590 if (IsVALU && STI.hasSDWA()) {
2591 // Write the low 16-bits of the high element into the high 16-bits of the
2592 // low element.
2593 MachineInstr *MovSDWA =
2594 BuildMI(*MBB, I, DL, TII.get(AMDGPU::V_MOV_B32_sdwa), DstReg)
2595 .addImm(0) // $src0_modifiers
2596 .addReg(HiReg) // $src0
2597 .addImm(0) // $clamp
2598 .addImm(AMDGPU::SDWA::WORD_1) // $dst_sel
2599 .addImm(AMDGPU::SDWA::UNUSED_PRESERVE) // $dst_unused
2600 .addImm(AMDGPU::SDWA::WORD_0) // $src0_sel
2601 .addReg(LoReg, RegState::Implicit);
2602 MovSDWA->tieOperands(0, MovSDWA->getNumOperands() - 1);
2603 } else {
2604 Register TmpReg0 = MRI->createVirtualRegister(DstRC);
2605 Register TmpReg1 = MRI->createVirtualRegister(DstRC);
2606 Register ImmReg = MRI->createVirtualRegister(DstRC);
2607 if (IsVALU) {
2608 BuildMI(*MBB, I, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), TmpReg0)
2609 .addImm(16)
2610 .addReg(HiReg);
2611 } else {
2612 BuildMI(*MBB, I, DL, TII.get(AMDGPU::S_LSHL_B32), TmpReg0)
2613 .addReg(HiReg)
2614 .addImm(16)
2615 .setOperandDead(3); // Dead scc
2616 }
2617
2618 unsigned MovOpc = IsVALU ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32;
2619 unsigned AndOpc = IsVALU ? AMDGPU::V_AND_B32_e64 : AMDGPU::S_AND_B32;
2620 unsigned OrOpc = IsVALU ? AMDGPU::V_OR_B32_e64 : AMDGPU::S_OR_B32;
2621
2622 BuildMI(*MBB, I, DL, TII.get(MovOpc), ImmReg)
2623 .addImm(0xffff);
2624 auto And = BuildMI(*MBB, I, DL, TII.get(AndOpc), TmpReg1)
2625 .addReg(LoReg)
2626 .addReg(ImmReg);
2627 auto Or = BuildMI(*MBB, I, DL, TII.get(OrOpc), DstReg)
2628 .addReg(TmpReg0)
2629 .addReg(TmpReg1);
2630
2631 if (!IsVALU) {
2632 And.setOperandDead(3); // Dead scc
2633 Or.setOperandDead(3); // Dead scc
2634 }
2635 }
2636
2637 I.eraseFromParent();
2638 return true;
2639 }
2640
2641 if (!DstTy.isScalar())
2642 return false;
2643
2644 if (SrcSize > 32) {
2645 unsigned SubRegIdx = DstSize < 32
2646 ? static_cast<unsigned>(AMDGPU::sub0)
2647 : TRI.getSubRegFromChannel(0, DstSize / 32);
2648 if (SubRegIdx == AMDGPU::NoSubRegister)
2649 return false;
2650
2651 // Deal with weird cases where the class only partially supports the subreg
2652 // index.
2653 const TargetRegisterClass *SrcWithSubRC
2654 = TRI.getSubClassWithSubReg(SrcRC, SubRegIdx);
2655 if (!SrcWithSubRC)
2656 return false;
2657
2658 if (SrcWithSubRC != SrcRC) {
2659 if (!RBI.constrainGenericRegister(SrcReg, *SrcWithSubRC, *MRI))
2660 return false;
2661 }
2662
2663 I.getOperand(1).setSubReg(SubRegIdx);
2664 }
2665
2666 I.setDesc(TII.get(TargetOpcode::COPY));
2667 return true;
2668}
2669
2670/// \returns true if a bitmask for \p Size bits will be an inline immediate.
2671static bool shouldUseAndMask(unsigned Size, unsigned &Mask) {
2673 int SignedMask = static_cast<int>(Mask);
2674 return SignedMask >= -16 && SignedMask <= 64;
2675}
2676
2677// Like RegisterBankInfo::getRegBank, but don't assume vcc for s1.
2678const RegisterBank *AMDGPUInstructionSelector::getArtifactRegBank(
2679 Register Reg, const MachineRegisterInfo &MRI,
2680 const TargetRegisterInfo &TRI) const {
2681 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
2682 if (auto *RB = dyn_cast<const RegisterBank *>(RegClassOrBank))
2683 return RB;
2684
2685 // Ignore the type, since we don't use vcc in artifacts.
2686 if (auto *RC = dyn_cast<const TargetRegisterClass *>(RegClassOrBank))
2687 return &RBI.getRegBankFromRegClass(*RC, LLT());
2688 return nullptr;
2689}
2690
2691bool AMDGPUInstructionSelector::selectG_SZA_EXT(MachineInstr &I) const {
2692 bool InReg = I.getOpcode() == AMDGPU::G_SEXT_INREG;
2693 bool Signed = I.getOpcode() == AMDGPU::G_SEXT || InReg;
2694 const DebugLoc &DL = I.getDebugLoc();
2695 MachineBasicBlock &MBB = *I.getParent();
2696 const Register DstReg = I.getOperand(0).getReg();
2697 const Register SrcReg = I.getOperand(1).getReg();
2698
2699 const LLT DstTy = MRI->getType(DstReg);
2700 const LLT SrcTy = MRI->getType(SrcReg);
2701 const unsigned SrcSize = I.getOpcode() == AMDGPU::G_SEXT_INREG ?
2702 I.getOperand(2).getImm() : SrcTy.getSizeInBits();
2703 const unsigned DstSize = DstTy.getSizeInBits();
2704 if (!DstTy.isScalar())
2705 return false;
2706
2707 // Artifact casts should never use vcc.
2708 const RegisterBank *SrcBank = getArtifactRegBank(SrcReg, *MRI, TRI);
2709
2710 // FIXME: This should probably be illegal and split earlier.
2711 if (I.getOpcode() == AMDGPU::G_ANYEXT) {
2712 if (DstSize <= 32)
2713 return selectCOPY(I);
2714
2715 const TargetRegisterClass *SrcRC =
2716 TRI.getRegClassForTypeOnBank(SrcTy, *SrcBank);
2717 const RegisterBank *DstBank = RBI.getRegBank(DstReg, *MRI, TRI);
2718 const TargetRegisterClass *DstRC =
2719 TRI.getRegClassForSizeOnBank(DstSize, *DstBank);
2720
2721 Register UndefReg = MRI->createVirtualRegister(SrcRC);
2722 BuildMI(MBB, I, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2723 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
2724 .addReg(SrcReg)
2725 .addImm(AMDGPU::sub0)
2726 .addReg(UndefReg)
2727 .addImm(AMDGPU::sub1);
2728 I.eraseFromParent();
2729
2730 return RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) &&
2731 RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI);
2732 }
2733
2734 if (SrcBank->getID() == AMDGPU::VGPRRegBankID && DstSize <= 32) {
2735 // 64-bit should have been split up in RegBankSelect
2736
2737 // Try to use an and with a mask if it will save code size.
2738 unsigned Mask;
2739 if (!Signed && shouldUseAndMask(SrcSize, Mask)) {
2740 MachineInstr *ExtI =
2741 BuildMI(MBB, I, DL, TII.get(AMDGPU::V_AND_B32_e32), DstReg)
2742 .addImm(Mask)
2743 .addReg(SrcReg);
2744 I.eraseFromParent();
2745 constrainSelectedInstRegOperands(*ExtI, TII, TRI, RBI);
2746 return true;
2747 }
2748
2749 const unsigned BFE = Signed ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
2750 MachineInstr *ExtI =
2751 BuildMI(MBB, I, DL, TII.get(BFE), DstReg)
2752 .addReg(SrcReg)
2753 .addImm(0) // Offset
2754 .addImm(SrcSize); // Width
2755 I.eraseFromParent();
2756 constrainSelectedInstRegOperands(*ExtI, TII, TRI, RBI);
2757 return true;
2758 }
2759
2760 if (SrcBank->getID() == AMDGPU::SGPRRegBankID && DstSize <= 64) {
2761 const TargetRegisterClass &SrcRC = InReg && DstSize > 32 ?
2762 AMDGPU::SReg_64RegClass : AMDGPU::SReg_32RegClass;
2763 if (!RBI.constrainGenericRegister(SrcReg, SrcRC, *MRI))
2764 return false;
2765
2766 if (Signed && DstSize == 32 && (SrcSize == 8 || SrcSize == 16)) {
2767 const unsigned SextOpc = SrcSize == 8 ?
2768 AMDGPU::S_SEXT_I32_I8 : AMDGPU::S_SEXT_I32_I16;
2769 BuildMI(MBB, I, DL, TII.get(SextOpc), DstReg)
2770 .addReg(SrcReg);
2771 I.eraseFromParent();
2772 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
2773 }
2774
2775 // Using a single 32-bit SALU to calculate the high half is smaller than
2776 // S_BFE with a literal constant operand.
2777 if (DstSize > 32 && SrcSize == 32) {
2778 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2779 unsigned SubReg = InReg ? AMDGPU::sub0 : AMDGPU::NoSubRegister;
2780 if (Signed) {
2781 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_ASHR_I32), HiReg)
2782 .addReg(SrcReg, {}, SubReg)
2783 .addImm(31)
2784 .setOperandDead(3); // Dead scc
2785 } else {
2786 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_MOV_B32), HiReg)
2787 .addImm(0);
2788 }
2789 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
2790 .addReg(SrcReg, {}, SubReg)
2791 .addImm(AMDGPU::sub0)
2792 .addReg(HiReg)
2793 .addImm(AMDGPU::sub1);
2794 I.eraseFromParent();
2795 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_64RegClass,
2796 *MRI);
2797 }
2798
2799 const unsigned BFE64 = Signed ? AMDGPU::S_BFE_I64 : AMDGPU::S_BFE_U64;
2800 const unsigned BFE32 = Signed ? AMDGPU::S_BFE_I32 : AMDGPU::S_BFE_U32;
2801
2802 // Scalar BFE is encoded as S1[5:0] = offset, S1[22:16]= width.
2803 if (DstSize > 32 && (SrcSize <= 32 || InReg)) {
2804 // We need a 64-bit register source, but the high bits don't matter.
2805 Register ExtReg = MRI->createVirtualRegister(&AMDGPU::SReg_64RegClass);
2806 Register UndefReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2807 unsigned SubReg = InReg ? AMDGPU::sub0 : AMDGPU::NoSubRegister;
2808
2809 BuildMI(MBB, I, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefReg);
2810 BuildMI(MBB, I, DL, TII.get(AMDGPU::REG_SEQUENCE), ExtReg)
2811 .addReg(SrcReg, {}, SubReg)
2812 .addImm(AMDGPU::sub0)
2813 .addReg(UndefReg)
2814 .addImm(AMDGPU::sub1);
2815
2816 BuildMI(MBB, I, DL, TII.get(BFE64), DstReg)
2817 .addReg(ExtReg)
2818 .addImm(SrcSize << 16);
2819
2820 I.eraseFromParent();
2821 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_64RegClass, *MRI);
2822 }
2823
2824 unsigned Mask;
2825 if (!Signed && shouldUseAndMask(SrcSize, Mask)) {
2826 BuildMI(MBB, I, DL, TII.get(AMDGPU::S_AND_B32), DstReg)
2827 .addReg(SrcReg)
2828 .addImm(Mask)
2829 .setOperandDead(3); // Dead scc
2830 } else {
2831 BuildMI(MBB, I, DL, TII.get(BFE32), DstReg)
2832 .addReg(SrcReg)
2833 .addImm(SrcSize << 16);
2834 }
2835
2836 I.eraseFromParent();
2837 return RBI.constrainGenericRegister(DstReg, AMDGPU::SReg_32RegClass, *MRI);
2838 }
2839
2840 return false;
2841}
2842
2846
2848 Register BitcastSrc;
2849 if (mi_match(Reg, MRI, m_GBitcast(m_Reg(BitcastSrc))))
2850 Reg = BitcastSrc;
2851 return Reg;
2852}
2853
2855 Register &Out) {
2856 // When unmerging a register that is composed of 2 x 16-bit values allow to
2857 // use an extract hi instruction for the upper 16 bits. We only need to check
2858 // the size of `In` as all defs are guaranteed to be the same type for
2859 // GUnmerge.
2860 GUnmerge *Unmerge;
2861 if (mi_match(In, MRI, m_GUnmerge(Unmerge))) {
2862 if (Unmerge->getNumDefs() == 2 && Unmerge->getOperand(1).getReg() == In &&
2863 MRI.getType(In).getSizeInBits() == 16) {
2864 Out = stripBitCast(Unmerge->getSourceReg(), MRI);
2865 return true;
2866 }
2867 }
2868
2869 Register Trunc;
2870 if (!mi_match(In, MRI, m_GTrunc(m_Reg(Trunc))))
2871 return false;
2872
2873 Register LShlSrc;
2874 Register Cst;
2875 if (mi_match(Trunc, MRI, m_GLShr(m_Reg(LShlSrc), m_Reg(Cst)))) {
2876 Cst = stripCopy(Cst, MRI);
2877 if (mi_match(Cst, MRI, m_SpecificICst(16))) {
2878 Out = stripBitCast(LShlSrc, MRI);
2879 return true;
2880 }
2881 }
2882
2883 ArrayRef<int> Mask;
2884 Register Src1;
2885 if (!mi_match(Trunc, MRI, m_GShuffleVector(m_Reg(Src1), m_Reg(), Mask)))
2886 return false;
2887
2888 assert(MRI.getType(Src1) == LLT::fixed_vector(2, 16));
2889 assert(Mask.size() == 2);
2890
2891 if (Mask[0] == 1 && Mask[1] <= 1) {
2892 Out = Trunc;
2893 return true;
2894 }
2895
2896 return false;
2897}
2898
2900 Register &Out) {
2901 // There could be a bitcast between the extraction and its use.
2902 In = stripBitCast(In, MRI);
2903
2904 // The first def of a 2 x 16-bit unmerge is the low half of its source.
2905 if (auto *Unmerge = dyn_cast<GUnmerge>(MRI.getVRegDef(In))) {
2906 if (Unmerge->getNumDefs() == 2 && Unmerge->getOperand(0).getReg() == In &&
2907 MRI.getType(In).getSizeInBits() == 16) {
2908 Out = stripBitCast(Unmerge->getSourceReg(), MRI);
2909 return true;
2910 }
2911 }
2912
2913 // A truncation from 32 to 16 bits keeps the low half in place.
2914 Register Trunc;
2915 if (mi_match(In, MRI, m_GTrunc(m_Reg(Trunc))) &&
2916 MRI.getType(Trunc).getSizeInBits() == 32) {
2917 Out = stripBitCast(Trunc, MRI);
2918 return true;
2919 }
2920
2921 return false;
2922}
2923
2924bool AMDGPUInstructionSelector::selectG_FPEXT(MachineInstr &I) const {
2925 if (!Subtarget->hasSALUFloatInsts())
2926 return false;
2927
2928 Register Dst = I.getOperand(0).getReg();
2929 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
2930 if (DstRB->getID() != AMDGPU::SGPRRegBankID)
2931 return false;
2932
2933 Register Src = I.getOperand(1).getReg();
2934
2935 if (MRI->getType(Dst) == LLT::scalar(32) &&
2936 MRI->getType(Src) == LLT::scalar(16)) {
2937 if (isExtractHiElt(*MRI, Src, Src)) {
2938 MachineBasicBlock *BB = I.getParent();
2939 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::S_CVT_HI_F32_F16), Dst)
2940 .addUse(Src);
2941 I.eraseFromParent();
2942 return RBI.constrainGenericRegister(Dst, AMDGPU::SReg_32RegClass, *MRI);
2943 }
2944 }
2945
2946 return false;
2947}
2948
2949bool AMDGPUInstructionSelector::selectG_FNEG(MachineInstr &MI) const {
2950 // Only manually handle the f64 SGPR case.
2951 //
2952 // FIXME: This is a workaround for 2.5 different tablegen problems. Because
2953 // the bit ops theoretically have a second result due to the implicit def of
2954 // SCC, the GlobalISelEmitter is overly conservative and rejects it. Fixing
2955 // that is easy by disabling the check. The result works, but uses a
2956 // nonsensical sreg32orlds_and_sreg_1 regclass.
2957 //
2958 // The DAG emitter is more problematic, and incorrectly adds both S_XOR_B32 to
2959 // the variadic REG_SEQUENCE operands.
2960
2961 Register Dst = MI.getOperand(0).getReg();
2962 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
2963 if (DstRB->getID() != AMDGPU::SGPRRegBankID ||
2964 MRI->getType(Dst) != LLT::scalar(64))
2965 return false;
2966
2967 Register Src = MI.getOperand(1).getReg();
2968 MachineInstr *Fabs = getOpcodeDef(TargetOpcode::G_FABS, Src, *MRI);
2969 if (Fabs)
2970 Src = Fabs->getOperand(1).getReg();
2971
2972 if (!RBI.constrainGenericRegister(Src, AMDGPU::SReg_64RegClass, *MRI) ||
2973 !RBI.constrainGenericRegister(Dst, AMDGPU::SReg_64RegClass, *MRI))
2974 return false;
2975
2976 MachineBasicBlock *BB = MI.getParent();
2977 const DebugLoc &DL = MI.getDebugLoc();
2978 Register LoReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2979 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2980 Register ConstReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2981 Register OpReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2982
2983 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), LoReg)
2984 .addReg(Src, {}, AMDGPU::sub0);
2985 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), HiReg)
2986 .addReg(Src, {}, AMDGPU::sub1);
2987 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), ConstReg)
2988 .addImm(0x80000000);
2989
2990 // Set or toggle sign bit.
2991 unsigned Opc = Fabs ? AMDGPU::S_OR_B32 : AMDGPU::S_XOR_B32;
2992 BuildMI(*BB, &MI, DL, TII.get(Opc), OpReg)
2993 .addReg(HiReg)
2994 .addReg(ConstReg)
2995 .setOperandDead(3); // Dead scc
2996 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), Dst)
2997 .addReg(LoReg)
2998 .addImm(AMDGPU::sub0)
2999 .addReg(OpReg)
3000 .addImm(AMDGPU::sub1);
3001 MI.eraseFromParent();
3002 return true;
3003}
3004
3005// FIXME: This is a workaround for the same tablegen problems as G_FNEG
3006bool AMDGPUInstructionSelector::selectG_FABS(MachineInstr &MI) const {
3007 Register Dst = MI.getOperand(0).getReg();
3008 const RegisterBank *DstRB = RBI.getRegBank(Dst, *MRI, TRI);
3009 if (DstRB->getID() != AMDGPU::SGPRRegBankID ||
3010 MRI->getType(Dst) != LLT::scalar(64))
3011 return false;
3012
3013 Register Src = MI.getOperand(1).getReg();
3014 MachineBasicBlock *BB = MI.getParent();
3015 const DebugLoc &DL = MI.getDebugLoc();
3016 Register LoReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3017 Register HiReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3018 Register ConstReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3019 Register OpReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
3020
3021 if (!RBI.constrainGenericRegister(Src, AMDGPU::SReg_64RegClass, *MRI) ||
3022 !RBI.constrainGenericRegister(Dst, AMDGPU::SReg_64RegClass, *MRI))
3023 return false;
3024
3025 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), LoReg)
3026 .addReg(Src, {}, AMDGPU::sub0);
3027 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), HiReg)
3028 .addReg(Src, {}, AMDGPU::sub1);
3029 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_MOV_B32), ConstReg)
3030 .addImm(0x7fffffff);
3031
3032 // Clear sign bit.
3033 // TODO: Should this used S_BITSET0_*?
3034 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::S_AND_B32), OpReg)
3035 .addReg(HiReg)
3036 .addReg(ConstReg)
3037 .setOperandDead(3); // Dead scc
3038 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::REG_SEQUENCE), Dst)
3039 .addReg(LoReg)
3040 .addImm(AMDGPU::sub0)
3041 .addReg(OpReg)
3042 .addImm(AMDGPU::sub1);
3043
3044 MI.eraseFromParent();
3045 return true;
3046}
3047
3048static bool isConstant(const MachineInstr &MI) {
3049 return MI.getOpcode() == TargetOpcode::G_CONSTANT;
3050}
3051
3052void AMDGPUInstructionSelector::getAddrModeInfo(const MachineInstr &Load,
3053 const MachineRegisterInfo &MRI, SmallVectorImpl<GEPInfo> &AddrInfo) const {
3054
3055 unsigned OpNo = Load.getOpcode() == AMDGPU::G_PREFETCH ? 0 : 1;
3056 const MachineInstr *PtrMI =
3057 MRI.getUniqueVRegDef(Load.getOperand(OpNo).getReg());
3058
3059 assert(PtrMI);
3060
3061 if (PtrMI->getOpcode() != TargetOpcode::G_PTR_ADD)
3062 return;
3063
3064 GEPInfo GEPInfo;
3065
3066 for (unsigned i = 1; i != 3; ++i) {
3067 const MachineOperand &GEPOp = PtrMI->getOperand(i);
3068 const MachineInstr *OpDef = MRI.getUniqueVRegDef(GEPOp.getReg());
3069 assert(OpDef);
3070 if (i == 2 && isConstant(*OpDef)) {
3071 // TODO: Could handle constant base + variable offset, but a combine
3072 // probably should have commuted it.
3073 assert(GEPInfo.Imm == 0);
3074 GEPInfo.Imm = OpDef->getOperand(1).getCImm()->getSExtValue();
3075 continue;
3076 }
3077 const RegisterBank *OpBank = RBI.getRegBank(GEPOp.getReg(), MRI, TRI);
3078 if (OpBank->getID() == AMDGPU::SGPRRegBankID)
3079 GEPInfo.SgprParts.push_back(GEPOp.getReg());
3080 else
3081 GEPInfo.VgprParts.push_back(GEPOp.getReg());
3082 }
3083
3084 AddrInfo.push_back(GEPInfo);
3085 getAddrModeInfo(*PtrMI, MRI, AddrInfo);
3086}
3087
3088bool AMDGPUInstructionSelector::isSGPR(Register Reg) const {
3089 return RBI.getRegBank(Reg, *MRI, TRI)->getID() == AMDGPU::SGPRRegBankID;
3090}
3091
3092bool AMDGPUInstructionSelector::isInstrUniform(const MachineInstr &MI) const {
3093 if (!MI.hasOneMemOperand())
3094 return false;
3095
3096 const MachineMemOperand *MMO = *MI.memoperands_begin();
3097 const Value *Ptr = MMO->getValue();
3098
3099 // UndefValue means this is a load of a kernel input. These are uniform.
3100 // Sometimes LDS instructions have constant pointers.
3101 // If Ptr is null, then that means this mem operand contains a
3102 // PseudoSourceValue like GOT.
3104 return true;
3105
3107 return true;
3108
3109 if (MI.getOpcode() == AMDGPU::G_PREFETCH)
3110 return RBI.getRegBank(MI.getOperand(0).getReg(), *MRI, TRI)->getID() ==
3111 AMDGPU::SGPRRegBankID;
3112
3113 const Instruction *I = dyn_cast<Instruction>(Ptr);
3114 return I && I->getMetadata("amdgpu.uniform");
3115}
3116
3117bool AMDGPUInstructionSelector::hasVgprParts(ArrayRef<GEPInfo> AddrInfo) const {
3118 for (const GEPInfo &GEPInfo : AddrInfo) {
3119 if (!GEPInfo.VgprParts.empty())
3120 return true;
3121 }
3122 return false;
3123}
3124
3125void AMDGPUInstructionSelector::initM0(MachineInstr &I) const {
3126 const LLT PtrTy = MRI->getType(I.getOperand(1).getReg());
3127 unsigned AS = PtrTy.getAddressSpace();
3129 STI.ldsRequiresM0Init()) {
3130 MachineBasicBlock *BB = I.getParent();
3131
3132 // If DS instructions require M0 initialization, insert it before selecting.
3133 BuildMI(*BB, &I, I.getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), AMDGPU::M0)
3134 .addImm(-1);
3135 }
3136}
3137
3138bool AMDGPUInstructionSelector::selectG_LOAD_STORE_ATOMICRMW(
3139 MachineInstr &I) const {
3140 initM0(I);
3141 return selectImpl(I, *CoverageInfo);
3142}
3143
3145 if (Reg.isPhysical())
3146 return false;
3147
3149 const unsigned Opcode = MI.getOpcode();
3150
3151 if (Opcode == AMDGPU::COPY)
3152 return isVCmpResult(MI.getOperand(1).getReg(), MRI);
3153
3154 if (Opcode == AMDGPU::G_AND || Opcode == AMDGPU::G_OR ||
3155 Opcode == AMDGPU::G_XOR)
3156 return isVCmpResult(MI.getOperand(1).getReg(), MRI) &&
3157 isVCmpResult(MI.getOperand(2).getReg(), MRI);
3158
3159 if (auto *GI = dyn_cast<GIntrinsic>(&MI))
3160 return GI->is(Intrinsic::amdgcn_class);
3161
3162 return Opcode == AMDGPU::G_ICMP || Opcode == AMDGPU::G_FCMP;
3163}
3164
3165bool AMDGPUInstructionSelector::selectG_BRCOND(MachineInstr &I) const {
3166 MachineBasicBlock *BB = I.getParent();
3167 MachineOperand &CondOp = I.getOperand(0);
3168 Register CondReg = CondOp.getReg();
3169 const DebugLoc &DL = I.getDebugLoc();
3170
3171 unsigned BrOpcode;
3172 Register CondPhysReg;
3173 const TargetRegisterClass *ConstrainRC;
3174
3175 // In SelectionDAG, we inspect the IR block for uniformity metadata to decide
3176 // whether the branch is uniform when selecting the instruction. In
3177 // GlobalISel, we should push that decision into RegBankSelect. Assume for now
3178 // RegBankSelect knows what it's doing if the branch condition is scc, even
3179 // though it currently does not.
3180 if (!isVCC(CondReg, *MRI)) {
3181 if (MRI->getType(CondReg) != LLT::scalar(32))
3182 return false;
3183
3184 CondPhysReg = AMDGPU::SCC;
3185 BrOpcode = AMDGPU::S_CBRANCH_SCC1;
3186 ConstrainRC = &AMDGPU::SReg_32RegClass;
3187 } else {
3188 // FIXME: Should scc->vcc copies and with exec?
3189
3190 // Unless the value of CondReg is a result of a V_CMP* instruction then we
3191 // need to insert an and with exec.
3192 if (!isVCmpResult(CondReg, *MRI)) {
3193 const bool Is64 = STI.isWave64();
3194 const unsigned Opcode = Is64 ? AMDGPU::S_AND_B64 : AMDGPU::S_AND_B32;
3195 const Register Exec = Is64 ? AMDGPU::EXEC : AMDGPU::EXEC_LO;
3196
3197 Register TmpReg = MRI->createVirtualRegister(TRI.getBoolRC());
3198 BuildMI(*BB, &I, DL, TII.get(Opcode), TmpReg)
3199 .addReg(CondReg)
3200 .addReg(Exec)
3201 .setOperandDead(3); // Dead scc
3202 CondReg = TmpReg;
3203 }
3204
3205 CondPhysReg = TRI.getVCC();
3206 BrOpcode = AMDGPU::S_CBRANCH_VCCNZ;
3207 ConstrainRC = TRI.getBoolRC();
3208 }
3209
3210 if (!MRI->getRegClassOrNull(CondReg))
3211 MRI->setRegClass(CondReg, ConstrainRC);
3212
3213 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), CondPhysReg)
3214 .addReg(CondReg);
3215 BuildMI(*BB, &I, DL, TII.get(BrOpcode))
3216 .addMBB(I.getOperand(1).getMBB());
3217
3218 I.eraseFromParent();
3219 return true;
3220}
3221
3222bool AMDGPUInstructionSelector::selectG_GLOBAL_VALUE(
3223 MachineInstr &I) const {
3224 Register DstReg = I.getOperand(0).getReg();
3225 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3226 const bool IsVGPR = DstRB->getID() == AMDGPU::VGPRRegBankID;
3227 I.setDesc(TII.get(IsVGPR ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32));
3228 if (IsVGPR)
3229 I.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
3230
3231 return RBI.constrainGenericRegister(
3232 DstReg, IsVGPR ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass, *MRI);
3233}
3234
3235bool AMDGPUInstructionSelector::selectG_PTRMASK(MachineInstr &I) const {
3236 Register DstReg = I.getOperand(0).getReg();
3237 Register SrcReg = I.getOperand(1).getReg();
3238 Register MaskReg = I.getOperand(2).getReg();
3239 LLT Ty = MRI->getType(DstReg);
3240 LLT MaskTy = MRI->getType(MaskReg);
3241 MachineBasicBlock *BB = I.getParent();
3242 const DebugLoc &DL = I.getDebugLoc();
3243
3244 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3245 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
3246 const RegisterBank *MaskRB = RBI.getRegBank(MaskReg, *MRI, TRI);
3247 const bool IsVGPR = DstRB->getID() == AMDGPU::VGPRRegBankID;
3248 if (DstRB != SrcRB) // Should only happen for hand written MIR.
3249 return false;
3250
3251 // Try to avoid emitting a bit operation when we only need to touch half of
3252 // the 64-bit pointer.
3253 APInt MaskOnes = VT->getKnownOnes(MaskReg).zext(64);
3254 const APInt MaskHi32 = APInt::getHighBitsSet(64, 32);
3255 const APInt MaskLo32 = APInt::getLowBitsSet(64, 32);
3256
3257 const bool CanCopyLow32 = (MaskOnes & MaskLo32) == MaskLo32;
3258 const bool CanCopyHi32 = (MaskOnes & MaskHi32) == MaskHi32;
3259
3260 if (!IsVGPR && Ty.getSizeInBits() == 64 &&
3261 !CanCopyLow32 && !CanCopyHi32) {
3262 auto MIB = BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_AND_B64), DstReg)
3263 .addReg(SrcReg)
3264 .addReg(MaskReg)
3265 .setOperandDead(3); // Dead scc
3266 I.eraseFromParent();
3267 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3268 return true;
3269 }
3270
3271 unsigned NewOpc = IsVGPR ? AMDGPU::V_AND_B32_e64 : AMDGPU::S_AND_B32;
3272 const TargetRegisterClass &RegRC
3273 = IsVGPR ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
3274
3275 const TargetRegisterClass *DstRC = TRI.getRegClassForTypeOnBank(Ty, *DstRB);
3276 const TargetRegisterClass *SrcRC = TRI.getRegClassForTypeOnBank(Ty, *SrcRB);
3277 const TargetRegisterClass *MaskRC =
3278 TRI.getRegClassForTypeOnBank(MaskTy, *MaskRB);
3279
3280 if (!RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
3281 !RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
3282 !RBI.constrainGenericRegister(MaskReg, *MaskRC, *MRI))
3283 return false;
3284
3285 if (Ty.getSizeInBits() == 32) {
3286 assert(MaskTy.getSizeInBits() == 32 &&
3287 "ptrmask should have been narrowed during legalize");
3288
3289 auto NewOp = BuildMI(*BB, &I, DL, TII.get(NewOpc), DstReg)
3290 .addReg(SrcReg)
3291 .addReg(MaskReg);
3292
3293 if (!IsVGPR)
3294 NewOp.setOperandDead(3); // Dead scc
3295 I.eraseFromParent();
3296 return true;
3297 }
3298
3299 Register HiReg = MRI->createVirtualRegister(&RegRC);
3300 Register LoReg = MRI->createVirtualRegister(&RegRC);
3301
3302 // Extract the subregisters from the source pointer.
3303 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), LoReg)
3304 .addReg(SrcReg, {}, AMDGPU::sub0);
3305 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), HiReg)
3306 .addReg(SrcReg, {}, AMDGPU::sub1);
3307
3308 Register MaskedLo, MaskedHi;
3309
3310 if (CanCopyLow32) {
3311 // If all the bits in the low half are 1, we only need a copy for it.
3312 MaskedLo = LoReg;
3313 } else {
3314 // Extract the mask subregister and apply the and.
3315 Register MaskLo = MRI->createVirtualRegister(&RegRC);
3316 MaskedLo = MRI->createVirtualRegister(&RegRC);
3317
3318 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), MaskLo)
3319 .addReg(MaskReg, {}, AMDGPU::sub0);
3320 BuildMI(*BB, &I, DL, TII.get(NewOpc), MaskedLo)
3321 .addReg(LoReg)
3322 .addReg(MaskLo);
3323 }
3324
3325 if (CanCopyHi32) {
3326 // If all the bits in the high half are 1, we only need a copy for it.
3327 MaskedHi = HiReg;
3328 } else {
3329 Register MaskHi = MRI->createVirtualRegister(&RegRC);
3330 MaskedHi = MRI->createVirtualRegister(&RegRC);
3331
3332 BuildMI(*BB, &I, DL, TII.get(AMDGPU::COPY), MaskHi)
3333 .addReg(MaskReg, {}, AMDGPU::sub1);
3334 BuildMI(*BB, &I, DL, TII.get(NewOpc), MaskedHi)
3335 .addReg(HiReg)
3336 .addReg(MaskHi);
3337 }
3338
3339 BuildMI(*BB, &I, DL, TII.get(AMDGPU::REG_SEQUENCE), DstReg)
3340 .addReg(MaskedLo)
3341 .addImm(AMDGPU::sub0)
3342 .addReg(MaskedHi)
3343 .addImm(AMDGPU::sub1);
3344 I.eraseFromParent();
3345 return true;
3346}
3347
3348/// Return the register to use for the index value, and the subregister to use
3349/// for the indirectly accessed register.
3350static std::pair<Register, unsigned>
3352 const TargetRegisterClass *SuperRC, Register IdxReg,
3353 unsigned EltSize, GISelValueTracking &ValueTracking) {
3354 Register IdxBaseReg;
3355 int Offset;
3356
3357 std::tie(IdxBaseReg, Offset) =
3358 AMDGPU::getBaseWithConstantOffset(MRI, IdxReg, &ValueTracking);
3359 if (IdxBaseReg == AMDGPU::NoRegister) {
3360 // This will happen if the index is a known constant. This should ordinarily
3361 // be legalized out, but handle it as a register just in case.
3362 assert(Offset == 0);
3363 IdxBaseReg = IdxReg;
3364 }
3365
3366 ArrayRef<int16_t> SubRegs = TRI.getRegSplitParts(SuperRC, EltSize);
3367
3368 // Skip out of bounds offsets, or else we would end up using an undefined
3369 // register.
3370 if (static_cast<unsigned>(Offset) >= SubRegs.size())
3371 return std::pair(IdxReg, SubRegs[0]);
3372 return std::pair(IdxBaseReg, SubRegs[Offset]);
3373}
3374
3375bool AMDGPUInstructionSelector::selectG_EXTRACT_VECTOR_ELT(
3376 MachineInstr &MI) const {
3377 Register DstReg = MI.getOperand(0).getReg();
3378 Register SrcReg = MI.getOperand(1).getReg();
3379 Register IdxReg = MI.getOperand(2).getReg();
3380
3381 LLT DstTy = MRI->getType(DstReg);
3382 LLT SrcTy = MRI->getType(SrcReg);
3383
3384 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
3385 const RegisterBank *SrcRB = RBI.getRegBank(SrcReg, *MRI, TRI);
3386 const RegisterBank *IdxRB = RBI.getRegBank(IdxReg, *MRI, TRI);
3387
3388 // The index must be scalar. If it wasn't RegBankSelect should have moved this
3389 // into a waterfall loop.
3390 if (IdxRB->getID() != AMDGPU::SGPRRegBankID)
3391 return false;
3392
3393 const TargetRegisterClass *SrcRC =
3394 TRI.getRegClassForTypeOnBank(SrcTy, *SrcRB);
3395 const TargetRegisterClass *DstRC =
3396 TRI.getRegClassForTypeOnBank(DstTy, *DstRB);
3397 if (!SrcRC || !DstRC)
3398 return false;
3399 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, *MRI) ||
3400 !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI) ||
3401 !RBI.constrainGenericRegister(IdxReg, AMDGPU::SReg_32RegClass, *MRI))
3402 return false;
3403
3404 MachineBasicBlock *BB = MI.getParent();
3405 const DebugLoc &DL = MI.getDebugLoc();
3406 const bool Is64 = DstTy.getSizeInBits() == 64;
3407
3408 unsigned SubReg;
3409 std::tie(IdxReg, SubReg) = computeIndirectRegIndex(
3410 *MRI, TRI, SrcRC, IdxReg, DstTy.getSizeInBits() / 8, *VT);
3411
3412 if (SrcRB->getID() == AMDGPU::SGPRRegBankID) {
3413 if (DstTy.getSizeInBits() != 32 && !Is64)
3414 return false;
3415
3416 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3417 .addReg(IdxReg);
3418
3419 unsigned Opc = Is64 ? AMDGPU::S_MOVRELS_B64 : AMDGPU::S_MOVRELS_B32;
3420 BuildMI(*BB, &MI, DL, TII.get(Opc), DstReg)
3421 .addReg(SrcReg, {}, SubReg)
3422 .addReg(SrcReg, RegState::Implicit);
3423 MI.eraseFromParent();
3424 return true;
3425 }
3426
3427 if (SrcRB->getID() != AMDGPU::VGPRRegBankID || DstTy.getSizeInBits() != 32)
3428 return false;
3429
3430 if (!STI.useVGPRIndexMode()) {
3431 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3432 .addReg(IdxReg);
3433 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::V_MOVRELS_B32_e32), DstReg)
3434 .addReg(SrcReg, {}, SubReg)
3435 .addReg(SrcReg, RegState::Implicit);
3436 MI.eraseFromParent();
3437 return true;
3438 }
3439
3440 const MCInstrDesc &GPRIDXDesc =
3441 TII.getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*SrcRC), true);
3442 BuildMI(*BB, MI, DL, GPRIDXDesc, DstReg)
3443 .addReg(SrcReg)
3444 .addReg(IdxReg)
3445 .addImm(SubReg);
3446
3447 MI.eraseFromParent();
3448 return true;
3449}
3450
3451// TODO: Fold insert_vector_elt (extract_vector_elt) into movrelsd
3452bool AMDGPUInstructionSelector::selectG_INSERT_VECTOR_ELT(
3453 MachineInstr &MI) const {
3454 Register DstReg = MI.getOperand(0).getReg();
3455 Register VecReg = MI.getOperand(1).getReg();
3456 Register ValReg = MI.getOperand(2).getReg();
3457 Register IdxReg = MI.getOperand(3).getReg();
3458
3459 LLT VecTy = MRI->getType(DstReg);
3460 LLT ValTy = MRI->getType(ValReg);
3461 unsigned VecSize = VecTy.getSizeInBits();
3462 unsigned ValSize = ValTy.getSizeInBits();
3463
3464 const RegisterBank *VecRB = RBI.getRegBank(VecReg, *MRI, TRI);
3465 const RegisterBank *ValRB = RBI.getRegBank(ValReg, *MRI, TRI);
3466 const RegisterBank *IdxRB = RBI.getRegBank(IdxReg, *MRI, TRI);
3467
3468 assert(VecTy.getElementType() == ValTy);
3469
3470 // The index must be scalar. If it wasn't RegBankSelect should have moved this
3471 // into a waterfall loop.
3472 if (IdxRB->getID() != AMDGPU::SGPRRegBankID)
3473 return false;
3474
3475 const TargetRegisterClass *VecRC =
3476 TRI.getRegClassForTypeOnBank(VecTy, *VecRB);
3477 const TargetRegisterClass *ValRC =
3478 TRI.getRegClassForTypeOnBank(ValTy, *ValRB);
3479
3480 if (!RBI.constrainGenericRegister(VecReg, *VecRC, *MRI) ||
3481 !RBI.constrainGenericRegister(DstReg, *VecRC, *MRI) ||
3482 !RBI.constrainGenericRegister(ValReg, *ValRC, *MRI) ||
3483 !RBI.constrainGenericRegister(IdxReg, AMDGPU::SReg_32RegClass, *MRI))
3484 return false;
3485
3486 if (VecRB->getID() == AMDGPU::VGPRRegBankID && ValSize != 32)
3487 return false;
3488
3489 unsigned SubReg;
3490 std::tie(IdxReg, SubReg) =
3491 computeIndirectRegIndex(*MRI, TRI, VecRC, IdxReg, ValSize / 8, *VT);
3492
3493 const bool IndexMode = VecRB->getID() == AMDGPU::VGPRRegBankID &&
3494 STI.useVGPRIndexMode();
3495
3496 MachineBasicBlock *BB = MI.getParent();
3497 const DebugLoc &DL = MI.getDebugLoc();
3498
3499 if (!IndexMode) {
3500 BuildMI(*BB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3501 .addReg(IdxReg);
3502
3503 const MCInstrDesc &RegWriteOp = TII.getIndirectRegWriteMovRelPseudo(
3504 VecSize, ValSize, VecRB->getID() == AMDGPU::SGPRRegBankID);
3505 BuildMI(*BB, MI, DL, RegWriteOp, DstReg)
3506 .addReg(VecReg)
3507 .addReg(ValReg)
3508 .addImm(SubReg);
3509 MI.eraseFromParent();
3510 return true;
3511 }
3512
3513 const MCInstrDesc &GPRIDXDesc =
3514 TII.getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), false);
3515 BuildMI(*BB, MI, DL, GPRIDXDesc, DstReg)
3516 .addReg(VecReg)
3517 .addReg(ValReg)
3518 .addReg(IdxReg)
3519 .addImm(SubReg);
3520
3521 MI.eraseFromParent();
3522 return true;
3523}
3524
3525static bool isAsyncLDSDMA(Intrinsic::ID Intr) {
3526 switch (Intr) {
3527 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
3528 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
3529 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
3530 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
3531 case Intrinsic::amdgcn_load_async_to_lds:
3532 case Intrinsic::amdgcn_global_load_async_lds:
3533 return true;
3534 }
3535 return false;
3536}
3537
3538bool AMDGPUInstructionSelector::selectBufferLoadLds(MachineInstr &MI) const {
3539 if (!Subtarget->hasVMemToLDSLoad())
3540 return false;
3541 unsigned Opc;
3542 unsigned Size = MI.getOperand(3).getImm();
3543 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
3544
3545 // The struct intrinsic variants add one additional operand over raw.
3546 const bool HasVIndex = MI.getNumOperands() == 9;
3547 Register VIndex;
3548 int OpOffset = 0;
3549 if (HasVIndex) {
3550 VIndex = MI.getOperand(4).getReg();
3551 OpOffset = 1;
3552 }
3553
3554 Register VOffset = MI.getOperand(4 + OpOffset).getReg();
3555 std::optional<ValueAndVReg> MaybeVOffset =
3557 const bool HasVOffset = !MaybeVOffset || MaybeVOffset->Value.getZExtValue();
3558
3559 switch (Size) {
3560 default:
3561 return false;
3562 case 1:
3563 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_BOTHEN
3564 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_IDXEN
3565 : HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFEN
3566 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFSET;
3567 break;
3568 case 2:
3569 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_BOTHEN
3570 : AMDGPU::BUFFER_LOAD_USHORT_LDS_IDXEN
3571 : HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFEN
3572 : AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFSET;
3573 break;
3574 case 4:
3575 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_BOTHEN
3576 : AMDGPU::BUFFER_LOAD_DWORD_LDS_IDXEN
3577 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFEN
3578 : AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFSET;
3579 break;
3580 case 12:
3581 if (!Subtarget->hasLDSLoadB96_B128())
3582 return false;
3583
3584 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_BOTHEN
3585 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_IDXEN
3586 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFEN
3587 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFSET;
3588 break;
3589 case 16:
3590 if (!Subtarget->hasLDSLoadB96_B128())
3591 return false;
3592
3593 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_BOTHEN
3594 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_IDXEN
3595 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFEN
3596 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFSET;
3597 break;
3598 }
3599
3600 MachineBasicBlock *MBB = MI.getParent();
3601 const DebugLoc &DL = MI.getDebugLoc();
3602 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3603 .add(MI.getOperand(2));
3604
3605 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc));
3606
3607 if (HasVIndex && HasVOffset) {
3608 Register IdxReg = MRI->createVirtualRegister(TRI.getVGPR64Class());
3609 BuildMI(*MBB, &*MIB, DL, TII.get(AMDGPU::REG_SEQUENCE), IdxReg)
3610 .addReg(VIndex)
3611 .addImm(AMDGPU::sub0)
3612 .addReg(VOffset)
3613 .addImm(AMDGPU::sub1);
3614
3615 MIB.addReg(IdxReg);
3616 } else if (HasVIndex) {
3617 MIB.addReg(VIndex);
3618 } else if (HasVOffset) {
3619 MIB.addReg(VOffset);
3620 }
3621
3622 MIB.add(MI.getOperand(1)); // rsrc
3623 MIB.add(MI.getOperand(5 + OpOffset)); // soffset
3624 MIB.add(MI.getOperand(6 + OpOffset)); // imm offset
3625 bool IsGFX12Plus = AMDGPU::isGFX12Plus(STI);
3626 unsigned Aux = MI.getOperand(7 + OpOffset).getImm();
3627 MIB.addImm(Aux & (IsGFX12Plus ? AMDGPU::CPol::ALL
3628 : AMDGPU::CPol::ALL_pregfx12)); // cpol
3629 MIB.addImm(
3630 Aux & (IsGFX12Plus ? AMDGPU::CPol::SWZ : AMDGPU::CPol::SWZ_pregfx12)
3631 ? 1
3632 : 0); // swz
3633 MIB.addImm(isAsyncLDSDMA(IntrinsicID));
3634
3635 MachineMemOperand *LoadMMO = *MI.memoperands_begin();
3636 // Don't set the offset value here because the pointer points to the base of
3637 // the buffer.
3638 MachinePointerInfo LoadPtrI = LoadMMO->getPointerInfo();
3639
3640 MachinePointerInfo StorePtrI = LoadPtrI;
3641 LoadPtrI.V = PoisonValue::get(PointerType::get(MF->getFunction().getContext(),
3645
3646 auto F = LoadMMO->getFlags() &
3648 LoadMMO = MF->getMachineMemOperand(LoadPtrI, F | MachineMemOperand::MOLoad,
3649 Size, LoadMMO->getBaseAlign());
3650
3651 MachineMemOperand *StoreMMO =
3652 MF->getMachineMemOperand(StorePtrI, F | MachineMemOperand::MOStore,
3653 sizeof(int32_t), LoadMMO->getBaseAlign());
3654
3655 MIB.setMemRefs({LoadMMO, StoreMMO});
3656
3657 MI.eraseFromParent();
3658 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3659 return true;
3660}
3661
3662/// Match a zero extend from a 32-bit value to 64-bits.
3663Register AMDGPUInstructionSelector::matchZeroExtendFromS32(Register Reg) const {
3664 Register ZExtSrc;
3665 if (mi_match(Reg, *MRI, m_GZExt(m_Reg(ZExtSrc))))
3666 return MRI->getType(ZExtSrc) == LLT::scalar(32) ? ZExtSrc : Register();
3667
3668 // Match legalized form %zext = G_MERGE_VALUES (s32 %x), (s32 0)
3669 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3670 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3671 return Register();
3672
3673 assert(Def->getNumOperands() == 3 &&
3674 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3675 if (mi_match(Def->getOperand(2).getReg(), *MRI, m_ZeroInt())) {
3676 return Def->getOperand(1).getReg();
3677 }
3678
3679 return Register();
3680}
3681
3682/// Match a sign extend from a 32-bit value to 64-bits.
3683Register AMDGPUInstructionSelector::matchSignExtendFromS32(Register Reg) const {
3684 Register SExtSrc;
3685 if (mi_match(Reg, *MRI, m_GSExt(m_Reg(SExtSrc))))
3686 return MRI->getType(SExtSrc) == LLT::scalar(32) ? SExtSrc : Register();
3687
3688 // Match legalized form %sext = G_MERGE_VALUES (s32 %x), G_ASHR((S32 %x, 31))
3689 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3690 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3691 return Register();
3692
3693 assert(Def->getNumOperands() == 3 &&
3694 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3695 if (mi_match(Def->getOperand(2).getReg(), *MRI,
3696 m_GAShr(m_SpecificReg(Def->getOperand(1).getReg()),
3697 m_SpecificICst(31))))
3698 return Def->getOperand(1).getReg();
3699
3700 Register ZextSrc = matchZeroExtendFromS32(Reg);
3701 if (ZextSrc && VT->signBitIsZero(ZextSrc))
3702 return ZextSrc;
3703
3704 return Register();
3705}
3706
3707/// Match a zero extend from a 32-bit value to 64-bits, or \p Reg itself if it
3708/// is 32-bit.
3710AMDGPUInstructionSelector::matchZeroExtendFromS32OrS32(Register Reg) const {
3711 return MRI->getType(Reg) == LLT::scalar(32) ? Reg
3712 : matchZeroExtendFromS32(Reg);
3713}
3714
3715/// Match a sign extend from a 32-bit value to 64-bits, or \p Reg itself if it
3716/// is 32-bit.
3718AMDGPUInstructionSelector::matchSignExtendFromS32OrS32(Register Reg) const {
3719 return MRI->getType(Reg) == LLT::scalar(32) ? Reg
3720 : matchSignExtendFromS32(Reg);
3721}
3722
3724AMDGPUInstructionSelector::matchExtendFromS32OrS32(Register Reg,
3725 bool IsSigned) const {
3726 if (IsSigned)
3727 return matchSignExtendFromS32OrS32(Reg);
3728
3729 return matchZeroExtendFromS32OrS32(Reg);
3730}
3731
3732Register AMDGPUInstructionSelector::matchAnyExtendFromS32(Register Reg) const {
3733 Register AnyExtSrc;
3734 if (mi_match(Reg, *MRI, m_GAnyExt(m_Reg(AnyExtSrc))))
3735 return MRI->getType(AnyExtSrc) == LLT::scalar(32) ? AnyExtSrc : Register();
3736
3737 // Match legalized form %zext = G_MERGE_VALUES (s32 %x), (s32 G_IMPLICIT_DEF)
3738 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
3739 if (Def->getOpcode() != AMDGPU::G_MERGE_VALUES)
3740 return Register();
3741
3742 assert(Def->getNumOperands() == 3 &&
3743 MRI->getType(Def->getOperand(0).getReg()) == LLT::scalar(64));
3744
3745 if (mi_match(Def->getOperand(2).getReg(), *MRI, m_GImplicitDef()))
3746 return Def->getOperand(1).getReg();
3747
3748 return Register();
3749}
3750
3751bool AMDGPUInstructionSelector::selectGlobalLoadLds(MachineInstr &MI) const{
3752 if (!Subtarget->hasVMemToLDSLoad())
3753 return false;
3754
3755 unsigned Opc;
3756 unsigned Size = MI.getOperand(3).getImm();
3757 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
3758
3759 switch (Size) {
3760 default:
3761 return false;
3762 case 1:
3763 Opc = AMDGPU::GLOBAL_LOAD_LDS_UBYTE;
3764 break;
3765 case 2:
3766 Opc = AMDGPU::GLOBAL_LOAD_LDS_USHORT;
3767 break;
3768 case 4:
3769 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORD;
3770 break;
3771 case 12:
3772 if (!Subtarget->hasLDSLoadB96_B128())
3773 return false;
3774 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX3;
3775 break;
3776 case 16:
3777 if (!Subtarget->hasLDSLoadB96_B128())
3778 return false;
3779 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX4;
3780 break;
3781 }
3782
3783 MachineBasicBlock *MBB = MI.getParent();
3784 const DebugLoc &DL = MI.getDebugLoc();
3785 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
3786 .add(MI.getOperand(2));
3787
3788 Register Addr = MI.getOperand(1).getReg();
3789 Register VOffset;
3790 // Try to split SAddr and VOffset. Global and LDS pointers share the same
3791 // immediate offset, so we cannot use a regular SelectGlobalSAddr().
3792 if (!isSGPR(Addr)) {
3793 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
3794 if (isSGPR(AddrDef->Reg)) {
3795 Addr = AddrDef->Reg;
3796 } else if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
3797 Register SAddr =
3798 getSrcRegIgnoringCopies(AddrDef->MI->getOperand(1).getReg(), *MRI);
3799 if (isSGPR(SAddr)) {
3800 Register PtrBaseOffset = AddrDef->MI->getOperand(2).getReg();
3801 if (Register Off = matchZeroExtendFromS32(PtrBaseOffset)) {
3802 Addr = SAddr;
3803 VOffset = Off;
3804 }
3805 }
3806 }
3807 }
3808
3809 if (isSGPR(Addr)) {
3811 if (!VOffset) {
3812 VOffset = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
3813 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::V_MOV_B32_e32), VOffset)
3814 .addImm(0);
3815 }
3816 }
3817
3818 auto MIB = BuildMI(*MBB, &MI, DL, TII.get(Opc))
3819 .addReg(Addr);
3820
3821 if (isSGPR(Addr))
3822 MIB.addReg(VOffset);
3823
3824 MIB.add(MI.getOperand(4)); // offset
3825
3826 unsigned Aux = MI.getOperand(5).getImm();
3827 MIB.addImm(Aux & ~AMDGPU::CPol::VIRTUAL_BITS); // cpol
3828 MIB.addImm(isAsyncLDSDMA(IntrinsicID));
3829
3830 MachineMemOperand *LoadMMO = *MI.memoperands_begin();
3831 MachinePointerInfo LoadPtrI = LoadMMO->getPointerInfo();
3832 LoadPtrI.Offset = MI.getOperand(4).getImm();
3833 MachinePointerInfo StorePtrI = LoadPtrI;
3834 LoadPtrI.V = PoisonValue::get(PointerType::get(MF->getFunction().getContext(),
3838 auto F = LoadMMO->getFlags() &
3840 LoadMMO = MF->getMachineMemOperand(LoadPtrI, F | MachineMemOperand::MOLoad,
3841 Size, LoadMMO->getBaseAlign());
3842 MachineMemOperand *StoreMMO =
3843 MF->getMachineMemOperand(StorePtrI, F | MachineMemOperand::MOStore,
3844 sizeof(int32_t), Align(4));
3845
3846 MIB.setMemRefs({LoadMMO, StoreMMO});
3847
3848 MI.eraseFromParent();
3849 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
3850 return true;
3851}
3852
3853bool AMDGPUInstructionSelector::selectTensorLoadStore(MachineInstr &MI,
3854 Intrinsic::ID IID) const {
3855 bool IsLoad = IID == Intrinsic::amdgcn_tensor_load_to_lds;
3856 unsigned Opc =
3857 IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d4 : AMDGPU::TENSOR_STORE_FROM_LDS_d4;
3858 int NumGroups = 4;
3859
3860 // A lamda function to check whether an operand is a vector of all 0s.
3861 const auto isAllZeros = [&](MachineOperand &Opnd) {
3862 const MachineInstr *DefMI = MRI->getVRegDef(Opnd.getReg());
3863 if (!DefMI)
3864 return false;
3865 return llvm::isBuildVectorAllZeros(*DefMI, *MRI, true);
3866 };
3867
3868 // Use _D2 version if both group 2 and 3 are zero-initialized.
3869 if (isAllZeros(MI.getOperand(3)) && isAllZeros(MI.getOperand(4))) {
3870 NumGroups = 2;
3871 Opc = IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d2
3872 : AMDGPU::TENSOR_STORE_FROM_LDS_d2;
3873 }
3874
3875 // TODO: Handle the fifth group: MI.getOpetand(5), which is silently ignored
3876 // for now because all existing targets only support up to 4 groups.
3877 MachineBasicBlock *MBB = MI.getParent();
3878 auto MIB = BuildMI(*MBB, &MI, MI.getDebugLoc(), TII.get(Opc))
3879 .add(MI.getOperand(1)) // D# group 0
3880 .add(MI.getOperand(2)); // D# group 1
3881
3882 if (NumGroups >= 4) { // Has at least 4 groups
3883 MIB.add(MI.getOperand(3)) // D# group 2
3884 .add(MI.getOperand(4)); // D# group 3
3885 }
3886
3887 MIB.addImm(0) // r128
3888 .add(MI.getOperand(6)); // cpol
3889
3890 MI.eraseFromParent();
3891 return true;
3892}
3893
3894bool AMDGPUInstructionSelector::selectBVHIntersectRayIntrinsic(
3895 MachineInstr &MI) const {
3896 unsigned OpcodeOpIdx =
3897 MI.getOpcode() == AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY ? 1 : 3;
3898 MI.setDesc(TII.get(MI.getOperand(OpcodeOpIdx).getImm()));
3899 MI.removeOperand(OpcodeOpIdx);
3900 MI.addImplicitDefUseOperands(*MI.getMF());
3901 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
3902 return true;
3903}
3904
3905// FIXME: This should be removed and let the patterns select. We just need the
3906// AGPR/VGPR combination versions.
3907bool AMDGPUInstructionSelector::selectSMFMACIntrin(MachineInstr &MI) const {
3908 unsigned Opc;
3909 switch (cast<GIntrinsic>(MI).getIntrinsicID()) {
3910 case Intrinsic::amdgcn_smfmac_f32_16x16x32_f16:
3911 Opc = AMDGPU::V_SMFMAC_F32_16X16X32_F16_e64;
3912 break;
3913 case Intrinsic::amdgcn_smfmac_f32_32x32x16_f16:
3914 Opc = AMDGPU::V_SMFMAC_F32_32X32X16_F16_e64;
3915 break;
3916 case Intrinsic::amdgcn_smfmac_f32_16x16x32_bf16:
3917 Opc = AMDGPU::V_SMFMAC_F32_16X16X32_BF16_e64;
3918 break;
3919 case Intrinsic::amdgcn_smfmac_f32_32x32x16_bf16:
3920 Opc = AMDGPU::V_SMFMAC_F32_32X32X16_BF16_e64;
3921 break;
3922 case Intrinsic::amdgcn_smfmac_i32_16x16x64_i8:
3923 Opc = AMDGPU::V_SMFMAC_I32_16X16X64_I8_e64;
3924 break;
3925 case Intrinsic::amdgcn_smfmac_i32_32x32x32_i8:
3926 Opc = AMDGPU::V_SMFMAC_I32_32X32X32_I8_e64;
3927 break;
3928 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_bf8:
3929 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF8_BF8_e64;
3930 break;
3931 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf8_fp8:
3932 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF8_FP8_e64;
3933 break;
3934 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_bf8:
3935 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_FP8_BF8_e64;
3936 break;
3937 case Intrinsic::amdgcn_smfmac_f32_16x16x64_fp8_fp8:
3938 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_FP8_FP8_e64;
3939 break;
3940 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_bf8:
3941 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF8_BF8_e64;
3942 break;
3943 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf8_fp8:
3944 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF8_FP8_e64;
3945 break;
3946 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_bf8:
3947 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_FP8_BF8_e64;
3948 break;
3949 case Intrinsic::amdgcn_smfmac_f32_32x32x32_fp8_fp8:
3950 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_FP8_FP8_e64;
3951 break;
3952 case Intrinsic::amdgcn_smfmac_f32_16x16x64_f16:
3953 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_F16_e64;
3954 break;
3955 case Intrinsic::amdgcn_smfmac_f32_32x32x32_f16:
3956 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_F16_e64;
3957 break;
3958 case Intrinsic::amdgcn_smfmac_f32_16x16x64_bf16:
3959 Opc = AMDGPU::V_SMFMAC_F32_16X16X64_BF16_e64;
3960 break;
3961 case Intrinsic::amdgcn_smfmac_f32_32x32x32_bf16:
3962 Opc = AMDGPU::V_SMFMAC_F32_32X32X32_BF16_e64;
3963 break;
3964 case Intrinsic::amdgcn_smfmac_i32_16x16x128_i8:
3965 Opc = AMDGPU::V_SMFMAC_I32_16X16X128_I8_e64;
3966 break;
3967 case Intrinsic::amdgcn_smfmac_i32_32x32x64_i8:
3968 Opc = AMDGPU::V_SMFMAC_I32_32X32X64_I8_e64;
3969 break;
3970 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_bf8:
3971 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_BF8_BF8_e64;
3972 break;
3973 case Intrinsic::amdgcn_smfmac_f32_16x16x128_bf8_fp8:
3974 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_BF8_FP8_e64;
3975 break;
3976 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_bf8:
3977 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_FP8_BF8_e64;
3978 break;
3979 case Intrinsic::amdgcn_smfmac_f32_16x16x128_fp8_fp8:
3980 Opc = AMDGPU::V_SMFMAC_F32_16X16X128_FP8_FP8_e64;
3981 break;
3982 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_bf8:
3983 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_BF8_BF8_e64;
3984 break;
3985 case Intrinsic::amdgcn_smfmac_f32_32x32x64_bf8_fp8:
3986 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_BF8_FP8_e64;
3987 break;
3988 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_bf8:
3989 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_FP8_BF8_e64;
3990 break;
3991 case Intrinsic::amdgcn_smfmac_f32_32x32x64_fp8_fp8:
3992 Opc = AMDGPU::V_SMFMAC_F32_32X32X64_FP8_FP8_e64;
3993 break;
3994 default:
3995 llvm_unreachable("unhandled smfmac intrinsic");
3996 }
3997
3998 auto VDst_In = MI.getOperand(4);
3999
4000 MI.setDesc(TII.get(Opc));
4001 MI.removeOperand(4); // VDst_In
4002 MI.removeOperand(1); // Intrinsic ID
4003 MI.addOperand(VDst_In); // Readd VDst_In to the end
4004 MI.addImplicitDefUseOperands(*MI.getMF());
4005 const MCInstrDesc &MCID = MI.getDesc();
4006 if (MCID.getOperandConstraint(0, MCOI::EARLY_CLOBBER) != -1) {
4007 MI.getOperand(0).setIsEarlyClobber(true);
4008 }
4009 return true;
4010}
4011
4012bool AMDGPUInstructionSelector::selectPermlaneSwapIntrin(
4013 MachineInstr &MI, Intrinsic::ID IntrID) const {
4014 if (IntrID == Intrinsic::amdgcn_permlane16_swap &&
4015 !Subtarget->hasPermlane16Swap())
4016 return false;
4017 if (IntrID == Intrinsic::amdgcn_permlane32_swap &&
4018 !Subtarget->hasPermlane32Swap())
4019 return false;
4020
4021 unsigned Opcode = IntrID == Intrinsic::amdgcn_permlane16_swap
4022 ? AMDGPU::V_PERMLANE16_SWAP_B32_e64
4023 : AMDGPU::V_PERMLANE32_SWAP_B32_e64;
4024
4025 MI.removeOperand(2);
4026 MI.setDesc(TII.get(Opcode));
4027 MI.addOperand(*MF, MachineOperand::CreateReg(AMDGPU::EXEC, false, true));
4028
4029 MachineOperand &FI = MI.getOperand(4);
4031
4032 constrainSelectedInstRegOperands(MI, TII, TRI, RBI);
4033 return true;
4034}
4035
4036bool AMDGPUInstructionSelector::selectWaveAddress(MachineInstr &MI) const {
4037 Register DstReg = MI.getOperand(0).getReg();
4038 Register SrcReg = MI.getOperand(1).getReg();
4039 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4040 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
4041 MachineBasicBlock *MBB = MI.getParent();
4042 const DebugLoc &DL = MI.getDebugLoc();
4043
4044 if (IsVALU) {
4045 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHRREV_B32_e64), DstReg)
4046 .addImm(Subtarget->getWavefrontSizeLog2())
4047 .addReg(SrcReg);
4048 } else {
4049 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::S_LSHR_B32), DstReg)
4050 .addReg(SrcReg)
4051 .addImm(Subtarget->getWavefrontSizeLog2())
4052 .setOperandDead(3); // Dead scc
4053 }
4054
4055 const TargetRegisterClass &RC =
4056 IsVALU ? AMDGPU::VGPR_32RegClass : AMDGPU::SReg_32RegClass;
4057 if (!RBI.constrainGenericRegister(DstReg, RC, *MRI))
4058 return false;
4059
4060 MI.eraseFromParent();
4061 return true;
4062}
4063
4064bool AMDGPUInstructionSelector::selectWaveShuffleIntrin(
4065 MachineInstr &MI) const {
4066 assert(MI.getNumOperands() == 4);
4067 MachineBasicBlock *MBB = MI.getParent();
4068 const DebugLoc &DL = MI.getDebugLoc();
4069
4070 Register DstReg = MI.getOperand(0).getReg();
4071 Register ValReg = MI.getOperand(2).getReg();
4072 Register IdxReg = MI.getOperand(3).getReg();
4073
4074 const LLT DstTy = MRI->getType(DstReg);
4075 unsigned DstSize = DstTy.getSizeInBits();
4076 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4077 const TargetRegisterClass *DstRC =
4078 TRI.getRegClassForSizeOnBank(DstSize, *DstRB);
4079
4080 if (DstTy != LLT::scalar(32))
4081 return false;
4082
4083 if (!Subtarget->supportsBPermute())
4084 return false;
4085
4086 // If we can bpermute across the whole wave, then just do that
4087 if (Subtarget->supportsWaveWideBPermute()) {
4088 Register ShiftIdxReg = MRI->createVirtualRegister(DstRC);
4089 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), ShiftIdxReg)
4090 .addImm(2)
4091 .addReg(IdxReg);
4092
4093 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), DstReg)
4094 .addReg(ShiftIdxReg)
4095 .addReg(ValReg)
4096 .addImm(0);
4097 } else {
4098 // Otherwise, we need to make use of whole wave mode
4099 assert(Subtarget->isWave64());
4100
4101 // Set inactive lanes to poison
4102 Register UndefValReg =
4103 MRI->createVirtualRegister(TRI.getRegClass(AMDGPU::SReg_32RegClassID));
4104 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefValReg);
4105
4106 Register UndefExecReg = MRI->createVirtualRegister(
4107 TRI.getRegClass(AMDGPU::SReg_64_XEXECRegClassID));
4108 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::IMPLICIT_DEF), UndefExecReg);
4109
4110 Register PoisonValReg = MRI->createVirtualRegister(DstRC);
4111 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32), PoisonValReg)
4112 .addImm(0)
4113 .addReg(ValReg)
4114 .addImm(0)
4115 .addReg(UndefValReg)
4116 .addReg(UndefExecReg);
4117
4118 // ds_bpermute requires index to be multiplied by 4
4119 Register ShiftIdxReg = MRI->createVirtualRegister(DstRC);
4120 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_LSHLREV_B32_e64), ShiftIdxReg)
4121 .addImm(2)
4122 .addReg(IdxReg);
4123
4124 Register PoisonIdxReg = MRI->createVirtualRegister(DstRC);
4125 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32), PoisonIdxReg)
4126 .addImm(0)
4127 .addReg(ShiftIdxReg)
4128 .addImm(0)
4129 .addReg(UndefValReg)
4130 .addReg(UndefExecReg);
4131
4132 Register PoisonUnshiftedIdxReg = MRI->createVirtualRegister(DstRC);
4133 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_SET_INACTIVE_B32),
4134 PoisonUnshiftedIdxReg)
4135 .addImm(0)
4136 .addReg(IdxReg)
4137 .addImm(0)
4138 .addReg(UndefValReg)
4139 .addReg(UndefExecReg);
4140
4141 // Get permutation of each half, then we'll select which one to use
4142 Register SameSidePermReg = MRI->createVirtualRegister(DstRC);
4143 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), SameSidePermReg)
4144 .addReg(PoisonIdxReg)
4145 .addReg(PoisonValReg)
4146 .addImm(0);
4147
4148 Register SwappedValReg = MRI->createVirtualRegister(DstRC);
4149 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_PERMLANE64_B32), SwappedValReg)
4150 .addReg(PoisonValReg);
4151
4152 Register OppSidePermReg = MRI->createVirtualRegister(DstRC);
4153 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::DS_BPERMUTE_B32), OppSidePermReg)
4154 .addReg(PoisonIdxReg)
4155 .addReg(SwappedValReg)
4156 .addImm(0);
4157
4158 Register WWMSwapPermReg = MRI->createVirtualRegister(DstRC);
4159 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::STRICT_WWM), WWMSwapPermReg)
4160 .addReg(OppSidePermReg);
4161
4162 // Select which side to take the permute from
4163 // We can get away with only using mbcnt_lo here since we're only
4164 // trying to detect which side of 32 each lane is on, and mbcnt_lo
4165 // returns 32 for lanes 32-63.
4166 Register ThreadIDReg = MRI->createVirtualRegister(DstRC);
4167 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_MBCNT_LO_U32_B32_e64), ThreadIDReg)
4168 .addImm(-1)
4169 .addImm(0);
4170
4171 Register XORReg = MRI->createVirtualRegister(DstRC);
4172 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_XOR_B32_e64), XORReg)
4173 .addReg(ThreadIDReg)
4174 .addReg(PoisonUnshiftedIdxReg);
4175
4176 Register ANDReg = MRI->createVirtualRegister(DstRC);
4177 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_AND_B32_e64), ANDReg)
4178 .addReg(XORReg)
4179 .addImm(32);
4180
4181 Register CompareReg = MRI->createVirtualRegister(
4182 TRI.getRegClass(AMDGPU::SReg_64_XEXECRegClassID));
4183 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_CMP_EQ_U32_e64), CompareReg)
4184 .addReg(ANDReg)
4185 .addImm(0);
4186
4187 // Finally do the selection
4188 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::V_CNDMASK_B32_e64), DstReg)
4189 .addImm(0)
4190 .addReg(WWMSwapPermReg)
4191 .addImm(0)
4192 .addReg(SameSidePermReg)
4193 .addReg(CompareReg);
4194 }
4195
4196 MI.eraseFromParent();
4197 return true;
4198}
4199
4200// Match BITOP3 operation and return a number of matched instructions plus
4201// truth table.
4202static std::pair<unsigned, uint8_t> BitOp3_Op(Register R,
4204 const MachineRegisterInfo &MRI) {
4205 unsigned NumOpcodes = 0;
4206 uint8_t LHSBits, RHSBits;
4207
4208 auto getOperandBits = [&Src, R, &MRI](Register Op, uint8_t &Bits) -> bool {
4209 // Define truth table given Src0, Src1, Src2 bits permutations:
4210 // 0 0 0
4211 // 0 0 1
4212 // 0 1 0
4213 // 0 1 1
4214 // 1 0 0
4215 // 1 0 1
4216 // 1 1 0
4217 // 1 1 1
4218 const uint8_t SrcBits[3] = { 0xf0, 0xcc, 0xaa };
4219
4220 if (mi_match(Op, MRI, m_AllOnesInt())) {
4221 Bits = 0xff;
4222 return true;
4223 }
4224 if (mi_match(Op, MRI, m_ZeroInt())) {
4225 Bits = 0;
4226 return true;
4227 }
4228
4229 for (unsigned I = 0; I < Src.size(); ++I) {
4230 // Try to find existing reused operand
4231 if (Src[I] == Op) {
4232 Bits = SrcBits[I];
4233 return true;
4234 }
4235 // Try to replace parent operator
4236 if (Src[I] == R) {
4237 Bits = SrcBits[I];
4238 Src[I] = Op;
4239 return true;
4240 }
4241 }
4242
4243 if (Src.size() == 3) {
4244 // No room left for operands. Try one last time, there can be a 'not' of
4245 // one of our source operands. In this case we can compute the bits
4246 // without growing Src vector.
4247 Register LHS;
4248 if (mi_match(Op, MRI, m_Not(m_Reg(LHS)))) {
4250 for (unsigned I = 0; I < Src.size(); ++I) {
4251 if (Src[I] == LHS) {
4252 Bits = ~SrcBits[I];
4253 return true;
4254 }
4255 }
4256 }
4257
4258 return false;
4259 }
4260
4261 Bits = SrcBits[Src.size()];
4262 Src.push_back(Op);
4263 return true;
4264 };
4265
4266 MachineInstr *MI = MRI.getVRegDef(R);
4267 switch (MI->getOpcode()) {
4268 case TargetOpcode::G_AND:
4269 case TargetOpcode::G_OR:
4270 case TargetOpcode::G_XOR: {
4271 Register LHS = getSrcRegIgnoringCopies(MI->getOperand(1).getReg(), MRI);
4272 Register RHS = getSrcRegIgnoringCopies(MI->getOperand(2).getReg(), MRI);
4273
4274 SmallVector<Register, 3> Backup(Src.begin(), Src.end());
4275 if (!getOperandBits(LHS, LHSBits) ||
4276 !getOperandBits(RHS, RHSBits)) {
4277 Src = std::move(Backup);
4278 return std::make_pair(0, 0);
4279 }
4280
4281 // Recursion is naturally limited by the size of the operand vector.
4282 //
4283 // When LHS and RHS share a common sub-expression, one side's recursion
4284 // may decompose that sub-expression and replace the Src slot the other
4285 // side occupies with sub-operands via the "replace parent" path in
4286 // getOperandBits. The other side's cached bit-pattern then refers to a
4287 // slot whose contents changed, producing a wrong truth table.
4288 //
4289 // We detect this in three ways:
4290 // (A) If LHS recursed, its truth table is valid against the Src state
4291 // when LHS recursion completed (SrcAfterLHS). If RHS recursion
4292 // then mutates a Src slot that LHSBits depends on, LHSBits is
4293 // stale.
4294 // (B) If RHS did not recurse, RHSBits came from getOperandBits and
4295 // refers to a specific Src slot. If that slot's contents changed
4296 // (by either recursion), RHSBits is stale.
4297 // (C) Symmetrically for LHS if it did not recurse.
4298 SmallVector<Register, 3> SrcBeforeRecurse(Src.begin(), Src.end());
4299 uint8_t LHSBitsOrig = LHSBits;
4300 uint8_t RHSBitsOrig = RHSBits;
4301
4302 auto LHSOp = BitOp3_Op(LHS, Src, MRI);
4303 if (LHSOp.first) {
4304 NumOpcodes += LHSOp.first;
4305 LHSBits = LHSOp.second;
4306 }
4307
4308 SmallVector<Register, 3> SrcAfterLHS(Src.begin(), Src.end());
4309
4310 auto RHSOp = BitOp3_Op(RHS, Src, MRI);
4311 if (RHSOp.first) {
4312 NumOpcodes += RHSOp.first;
4313 RHSBits = RHSOp.second;
4314 }
4315
4316 // dependsOnSlot: true iff the truth table TT varies with slot Slot.
4317 auto dependsOnSlot = [](uint8_t TT, int Slot) -> bool {
4318 if (Slot < 0 || Slot > 2)
4319 return false;
4320 const uint8_t Masks[3] = {0x0f, 0x33, 0x55};
4321 const int Shifts[3] = {4, 2, 1};
4322 return ((TT ^ (TT >> Shifts[Slot])) & Masks[Slot]) != 0;
4323 };
4324
4325 // findSlot: locate the Src slot a getOperandBits result depends on,
4326 // including negated (NOT) patterns that getOperandBits resolves via
4327 // the ~SrcBits[I] shortcut.
4328 const uint8_t SrcBitsConst[3] = {0xf0, 0xcc, 0xaa};
4329 auto findSlot = [&](uint8_t Bits, Register Op,
4330 const SmallVectorImpl<Register> &S) -> int {
4331 Register NegatedInner;
4332 bool IsNegationOp = mi_match(Op, MRI, m_Not(m_Reg(NegatedInner)));
4333 if (IsNegationOp)
4334 NegatedInner = getSrcRegIgnoringCopies(NegatedInner, MRI);
4335 for (int I = 0; I < (int)S.size(); I++) {
4336 if (Bits == SrcBitsConst[I] && S[I] == Op)
4337 return I;
4338 if (IsNegationOp && Bits == (uint8_t)~SrcBitsConst[I] &&
4339 S[I] == NegatedInner)
4340 return I;
4341 }
4342 return -1;
4343 };
4344
4345 bool Stale = false;
4346
4347 // (A) LHS recursed: its truth table is against SrcAfterLHS.
4348 // Check if RHS recursion mutated a slot that LHSBits uses.
4349 if (LHSOp.first) {
4350 for (int I = 0; I < (int)SrcAfterLHS.size() && I < 3; I++) {
4351 if (I < (int)Src.size() && Src[I] != SrcAfterLHS[I] &&
4352 dependsOnSlot(LHSBits, I)) {
4353 Stale = true;
4354 break;
4355 }
4356 }
4357 }
4358
4359 // (B) RHS did not recurse: RHSBits from getOperandBits is against
4360 // SrcBeforeRecurse. Check if that slot was mutated since then.
4361 if (!Stale && !RHSOp.first) {
4362 int Slot = findSlot(RHSBitsOrig, RHS, SrcBeforeRecurse);
4363 if (Slot >= 0 &&
4364 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4365 Stale = true;
4366 }
4367
4368 // (C) LHS did not recurse: LHSBits from getOperandBits is against
4369 // SrcBeforeRecurse. Check if that slot was mutated since then.
4370 if (!Stale && !LHSOp.first) {
4371 int Slot = findSlot(LHSBitsOrig, LHS, SrcBeforeRecurse);
4372 if (Slot >= 0 &&
4373 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4374 Stale = true;
4375 }
4376
4377 if (Stale) {
4378 Src = std::move(SrcBeforeRecurse);
4379 LHSBits = LHSBitsOrig;
4380 RHSBits = RHSBitsOrig;
4381 NumOpcodes = 0;
4382 }
4383 break;
4384 }
4385 default:
4386 return std::make_pair(0, 0);
4387 }
4388
4389 uint8_t TTbl;
4390 switch (MI->getOpcode()) {
4391 case TargetOpcode::G_AND:
4392 TTbl = LHSBits & RHSBits;
4393 break;
4394 case TargetOpcode::G_OR:
4395 TTbl = LHSBits | RHSBits;
4396 break;
4397 case TargetOpcode::G_XOR:
4398 TTbl = LHSBits ^ RHSBits;
4399 break;
4400 default:
4401 break;
4402 }
4403
4404 return std::make_pair(NumOpcodes + 1, TTbl);
4405}
4406
4407bool AMDGPUInstructionSelector::selectBITOP3(MachineInstr &MI) const {
4408 if (!Subtarget->hasBitOp3Insts())
4409 return false;
4410
4411 Register DstReg = MI.getOperand(0).getReg();
4412 const RegisterBank *DstRB = RBI.getRegBank(DstReg, *MRI, TRI);
4413 const bool IsVALU = DstRB->getID() == AMDGPU::VGPRRegBankID;
4414 if (!IsVALU)
4415 return false;
4416
4418 uint8_t TTbl;
4419 unsigned NumOpcodes;
4420
4421 std::tie(NumOpcodes, TTbl) = BitOp3_Op(DstReg, Src, *MRI);
4422
4423 // Src.empty() case can happen if all operands are all zero or all ones.
4424 // Normally it shall be optimized out before reaching this.
4425 if (NumOpcodes < 2 || Src.empty())
4426 return false;
4427
4428 // RegBankSelect splits wider VALU logic ops and widens 1-bit ones, so only
4429 // 16 and 32 bit types reach here. Note that <2 x i16> is 32 bits wide.
4430 unsigned Size = MRI->getType(DstReg).getSizeInBits();
4431 assert((Size == 16 || Size == 32) && "unexpected VALU logic op size");
4432 const bool IsB32 = Size == 32;
4433 if (NumOpcodes == 2 && IsB32) {
4434 // Avoid using BITOP3 for OR3, XOR3, AND_OR. This is not faster but makes
4435 // asm more readable. This cannot be modeled with AddedComplexity because
4436 // selector does not know how many operations did we match.
4437 if (mi_match(MI, *MRI, m_GXor(m_GXor(m_Reg(), m_Reg()), m_Reg())) ||
4438 mi_match(MI, *MRI, m_GOr(m_GOr(m_Reg(), m_Reg()), m_Reg())) ||
4439 mi_match(MI, *MRI, m_GOr(m_GAnd(m_Reg(), m_Reg()), m_Reg())))
4440 return false;
4441 } else if (NumOpcodes < 4) {
4442 // For a uniform case threshold should be higher to account for moves
4443 // between VGPRs and SGPRs. It needs one operand in a VGPR, rest two can be
4444 // in SGPRs and a readtfirstlane after.
4445 return false;
4446 }
4447
4448 unsigned Opc = IsB32 ? AMDGPU::V_BITOP3_B32_e64 : AMDGPU::V_BITOP3_B16_e64;
4449 if (!IsB32 && STI.hasTrue16BitInsts())
4450 Opc = STI.useRealTrue16Insts() ? AMDGPU::V_BITOP3_B16_gfx1250_t16_e64
4451 : AMDGPU::V_BITOP3_B16_gfx1250_fake16_e64;
4452 unsigned CBL = STI.getConstantBusLimit(Opc);
4453 MachineBasicBlock *MBB = MI.getParent();
4454 const DebugLoc &DL = MI.getDebugLoc();
4455
4456 for (unsigned I = 0; I < Src.size(); ++I) {
4457 const RegisterBank *RB = RBI.getRegBank(Src[I], *MRI, TRI);
4458 if (RB->getID() != AMDGPU::SGPRRegBankID)
4459 continue;
4460 if (CBL > 0) {
4461 --CBL;
4462 continue;
4463 }
4464 Register NewReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
4465 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::COPY), NewReg)
4466 .addReg(Src[I]);
4467 Src[I] = NewReg;
4468 }
4469
4470 // Last operand can be ignored, turning a ternary operation into a binary.
4471 // For example: (~a & b & c) | (~a & b & ~c) -> (~a & b). We can replace
4472 // 'c' with 'a' here without changing the answer. In some pathological
4473 // cases it should be possible to get an operation with a single operand
4474 // too if optimizer would not catch it.
4475 while (Src.size() < 3)
4476 Src.push_back(Src[0]);
4477
4478 auto MIB = BuildMI(*MBB, MI, DL, TII.get(Opc), DstReg);
4479 if (!IsB32)
4480 MIB.addImm(0); // src_mod0
4481 MIB.addReg(Src[0]);
4482 if (!IsB32)
4483 MIB.addImm(0); // src_mod1
4484 MIB.addReg(Src[1]);
4485 if (!IsB32)
4486 MIB.addImm(0); // src_mod2
4487 MIB.addReg(Src[2])
4488 .addImm(TTbl);
4489 if (!IsB32)
4490 MIB.addImm(0); // op_sel
4491
4492 constrainSelectedInstRegOperands(*MIB, TII, TRI, RBI);
4493 MI.eraseFromParent();
4494
4495 return true;
4496}
4497
4498bool AMDGPUInstructionSelector::selectStackRestore(MachineInstr &MI) const {
4499 Register SrcReg = MI.getOperand(0).getReg();
4500 if (!RBI.constrainGenericRegister(SrcReg, AMDGPU::SReg_32RegClass, *MRI))
4501 return false;
4502
4503 MachineInstr *DefMI = MRI->getVRegDef(SrcReg);
4504 Register SP =
4505 Subtarget->getTargetLowering()->getStackPointerRegisterToSaveRestore();
4506 Register WaveAddr = getWaveAddress(DefMI);
4507 MachineBasicBlock *MBB = MI.getParent();
4508 const DebugLoc &DL = MI.getDebugLoc();
4509
4510 if (!WaveAddr) {
4511 WaveAddr = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
4512 BuildMI(*MBB, MI, DL, TII.get(AMDGPU::S_LSHR_B32), WaveAddr)
4513 .addReg(SrcReg)
4514 .addImm(Subtarget->getWavefrontSizeLog2())
4515 .setOperandDead(3); // Dead scc
4516 }
4517
4518 BuildMI(*MBB, &MI, DL, TII.get(AMDGPU::COPY), SP)
4519 .addReg(WaveAddr);
4520
4521 MI.eraseFromParent();
4522 return true;
4523}
4524
4526
4527 if (!I.isPreISelOpcode()) {
4528 if (I.isCopy())
4529 return selectCOPY(I);
4530 return true;
4531 }
4532
4533 switch (I.getOpcode()) {
4534 case TargetOpcode::G_AND:
4535 case TargetOpcode::G_OR:
4536 case TargetOpcode::G_XOR:
4537 if (selectBITOP3(I))
4538 return true;
4539 if (selectImpl(I, *CoverageInfo))
4540 return true;
4541 return selectG_AND_OR_XOR(I);
4542 case TargetOpcode::G_ADD:
4543 case TargetOpcode::G_SUB:
4544 case TargetOpcode::G_PTR_ADD:
4545 if (selectImpl(I, *CoverageInfo))
4546 return true;
4547 return selectG_ADD_SUB(I);
4548 case TargetOpcode::G_UADDO:
4549 case TargetOpcode::G_USUBO:
4550 case TargetOpcode::G_UADDE:
4551 case TargetOpcode::G_USUBE:
4552 return selectG_UADDO_USUBO_UADDE_USUBE(I);
4553 case AMDGPU::G_AMDGPU_MAD_U64_U32:
4554 case AMDGPU::G_AMDGPU_MAD_I64_I32:
4555 return selectG_AMDGPU_MAD_64_32(I);
4556 case TargetOpcode::G_INTTOPTR:
4557 case TargetOpcode::G_BITCAST:
4558 case TargetOpcode::G_PTRTOINT:
4559 case TargetOpcode::G_FREEZE:
4560 return selectCOPY(I);
4561 case TargetOpcode::G_FNEG:
4562 if (selectImpl(I, *CoverageInfo))
4563 return true;
4564 return selectG_FNEG(I);
4565 case TargetOpcode::G_FABS:
4566 if (selectImpl(I, *CoverageInfo))
4567 return true;
4568 return selectG_FABS(I);
4569 case TargetOpcode::G_EXTRACT:
4570 return selectG_EXTRACT(I);
4571 case TargetOpcode::G_MERGE_VALUES:
4572 case TargetOpcode::G_CONCAT_VECTORS:
4573 return selectG_MERGE_VALUES(I);
4574 case TargetOpcode::G_UNMERGE_VALUES:
4575 return selectG_UNMERGE_VALUES(I);
4576 case TargetOpcode::G_BUILD_VECTOR:
4577 case TargetOpcode::G_BUILD_VECTOR_TRUNC:
4578 return selectG_BUILD_VECTOR(I);
4579 case TargetOpcode::G_IMPLICIT_DEF:
4580 return selectG_IMPLICIT_DEF(I);
4581 case TargetOpcode::G_INSERT:
4582 return selectG_INSERT(I);
4583 case TargetOpcode::G_INTRINSIC:
4584 case TargetOpcode::G_INTRINSIC_CONVERGENT:
4585 return selectG_INTRINSIC(I);
4586 case TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS:
4587 case TargetOpcode::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS:
4588 return selectG_INTRINSIC_W_SIDE_EFFECTS(I);
4589 case TargetOpcode::G_ICMP:
4590 case TargetOpcode::G_FCMP:
4591 if (selectG_ICMP_or_FCMP(I))
4592 return true;
4593 return selectImpl(I, *CoverageInfo);
4594 case TargetOpcode::G_LOAD:
4595 case TargetOpcode::G_ZEXTLOAD:
4596 case TargetOpcode::G_SEXTLOAD:
4597 case TargetOpcode::G_STORE:
4598 case TargetOpcode::G_ATOMIC_CMPXCHG:
4599 case TargetOpcode::G_ATOMICRMW_XCHG:
4600 case TargetOpcode::G_ATOMICRMW_ADD:
4601 case TargetOpcode::G_ATOMICRMW_SUB:
4602 case TargetOpcode::G_ATOMICRMW_AND:
4603 case TargetOpcode::G_ATOMICRMW_OR:
4604 case TargetOpcode::G_ATOMICRMW_XOR:
4605 case TargetOpcode::G_ATOMICRMW_MIN:
4606 case TargetOpcode::G_ATOMICRMW_MAX:
4607 case TargetOpcode::G_ATOMICRMW_UMIN:
4608 case TargetOpcode::G_ATOMICRMW_UMAX:
4609 case TargetOpcode::G_ATOMICRMW_UINC_WRAP:
4610 case TargetOpcode::G_ATOMICRMW_UDEC_WRAP:
4611 case TargetOpcode::G_ATOMICRMW_USUB_COND:
4612 case TargetOpcode::G_ATOMICRMW_USUB_SAT:
4613 case TargetOpcode::G_ATOMICRMW_FADD:
4614 case TargetOpcode::G_ATOMICRMW_FMIN:
4615 case TargetOpcode::G_ATOMICRMW_FMAX:
4616 return selectG_LOAD_STORE_ATOMICRMW(I);
4617 case TargetOpcode::G_SELECT:
4618 return selectG_SELECT(I);
4619 case TargetOpcode::G_TRUNC:
4620 return selectG_TRUNC(I);
4621 case TargetOpcode::G_SEXT:
4622 case TargetOpcode::G_ZEXT:
4623 case TargetOpcode::G_ANYEXT:
4624 case TargetOpcode::G_SEXT_INREG:
4625 // This is a workaround. For extension from type i1, `selectImpl()` uses
4626 // patterns from TD file and generates an illegal VGPR to SGPR COPY as type
4627 // i1 can only be hold in a SGPR class.
4628 if (MRI->getType(I.getOperand(1).getReg()) != LLT::scalar(1) &&
4629 selectImpl(I, *CoverageInfo))
4630 return true;
4631 return selectG_SZA_EXT(I);
4632 case TargetOpcode::G_FPEXT:
4633 if (selectG_FPEXT(I))
4634 return true;
4635 return selectImpl(I, *CoverageInfo);
4636 case TargetOpcode::G_BRCOND:
4637 return selectG_BRCOND(I);
4638 case TargetOpcode::G_GLOBAL_VALUE:
4639 return selectG_GLOBAL_VALUE(I);
4640 case TargetOpcode::G_PTRMASK:
4641 return selectG_PTRMASK(I);
4642 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
4643 return selectG_EXTRACT_VECTOR_ELT(I);
4644 case TargetOpcode::G_INSERT_VECTOR_ELT:
4645 return selectG_INSERT_VECTOR_ELT(I);
4646 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD:
4647 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_D16:
4648 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_LOAD_NORET:
4649 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE:
4650 case AMDGPU::G_AMDGPU_INTRIN_IMAGE_STORE_D16: {
4651 const AMDGPU::ImageDimIntrinsicInfo *Intr =
4653 assert(Intr && "not an image intrinsic with image pseudo");
4654 return selectImageIntrinsic(I, Intr);
4655 }
4656 case AMDGPU::G_AMDGPU_BVH_DUAL_INTERSECT_RAY:
4657 case AMDGPU::G_AMDGPU_BVH_INTERSECT_RAY:
4658 case AMDGPU::G_AMDGPU_BVH8_INTERSECT_RAY:
4659 return selectBVHIntersectRayIntrinsic(I);
4660 case AMDGPU::G_SBFX:
4661 case AMDGPU::G_UBFX:
4662 return selectG_SBFX_UBFX(I);
4663 case AMDGPU::G_SI_CALL:
4664 I.setDesc(TII.get(AMDGPU::SI_CALL));
4665 return true;
4666 case AMDGPU::G_AMDGPU_WAVE_ADDRESS:
4667 return selectWaveAddress(I);
4668 case AMDGPU::G_AMDGPU_WHOLE_WAVE_FUNC_RETURN: {
4669 I.setDesc(TII.get(AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN));
4670 return true;
4671 }
4672 case AMDGPU::G_STACKRESTORE:
4673 return selectStackRestore(I);
4674 case AMDGPU::G_PHI:
4675 return selectPHI(I);
4676 case AMDGPU::G_AMDGPU_COPY_SCC_VCC:
4677 return selectCOPY_SCC_VCC(I);
4678 case AMDGPU::G_AMDGPU_COPY_VCC_SCC:
4679 return selectCOPY_VCC_SCC(I);
4680 case AMDGPU::G_AMDGPU_READANYLANE:
4681 return selectReadAnyLane(I);
4682 case TargetOpcode::G_CONSTANT:
4683 case TargetOpcode::G_FCONSTANT:
4684 default:
4685 return selectImpl(I, *CoverageInfo);
4686 }
4687 return false;
4688}
4689
4691AMDGPUInstructionSelector::selectVCSRC(MachineOperand &Root) const {
4692 return {{
4693 [=](MachineInstrBuilder &MIB) { MIB.add(Root); }
4694 }};
4695
4696}
4697
4698std::pair<Register, unsigned> AMDGPUInstructionSelector::selectVOP3ModsImpl(
4699 Register Src, bool IsCanonicalizing, bool AllowAbs, bool OpSel) const {
4700 unsigned Mods = 0;
4701 MachineInstr *MI = getDefIgnoringCopies(Src, *MRI);
4702
4703 if (MI->getOpcode() == AMDGPU::G_FNEG) {
4704 Src = MI->getOperand(1).getReg();
4705 Mods |= SISrcMods::NEG;
4706 MI = getDefIgnoringCopies(Src, *MRI);
4707 } else if (MI->getOpcode() == AMDGPU::G_FSUB && IsCanonicalizing) {
4708 // Fold fsub [+-]0 into fneg. This may not have folded depending on the
4709 // denormal mode, but we're implicitly canonicalizing in a source operand.
4710 const ConstantFP *LHS =
4711 getConstantFPVRegVal(MI->getOperand(1).getReg(), *MRI);
4712 if (LHS && LHS->isZero()) {
4713 Mods |= SISrcMods::NEG;
4714 Src = MI->getOperand(2).getReg();
4715 }
4716 }
4717
4718 if (AllowAbs && MI->getOpcode() == AMDGPU::G_FABS) {
4719 Src = MI->getOperand(1).getReg();
4720 Mods |= SISrcMods::ABS;
4721 }
4722
4723 if (OpSel)
4724 Mods |= SISrcMods::OP_SEL_0;
4725
4726 return std::pair(Src, Mods);
4727}
4728
4729std::pair<Register, unsigned>
4730AMDGPUInstructionSelector::selectVOP3PModsF32Impl(Register Src) const {
4731 unsigned Mods;
4732 std::tie(Src, Mods) = selectVOP3ModsImpl(Src);
4733 Mods |= SISrcMods::OP_SEL_1;
4734 return std::pair(Src, Mods);
4735}
4736
4737Register AMDGPUInstructionSelector::copyToVGPRIfSrcFolded(
4738 Register Src, unsigned Mods, MachineOperand Root, MachineInstr *InsertPt,
4739 bool ForceVGPR) const {
4740 if ((Mods != 0 || ForceVGPR) &&
4741 RBI.getRegBank(Src, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID) {
4742
4743 // If we looked through copies to find source modifiers on an SGPR operand,
4744 // we now have an SGPR register source. To avoid potentially violating the
4745 // constant bus restriction, we need to insert a copy to a VGPR.
4746 Register VGPRSrc = MRI->cloneVirtualRegister(Root.getReg());
4747 BuildMI(*InsertPt->getParent(), InsertPt, InsertPt->getDebugLoc(),
4748 TII.get(AMDGPU::COPY), VGPRSrc)
4749 .addReg(Src);
4750 Src = VGPRSrc;
4751 }
4752
4753 return Src;
4754}
4755
4756/// Some instructions must have 32-bit sources. With real true16 instructions a
4757/// 16-bit VALU value lives in a VGPR_16, which they cannot read, so place it in
4758/// the low half of a new 32-bit VGPR.
4760AMDGPUInstructionSelector::widenSrcIfVGPR16(Register Src,
4761 MachineInstr *InsertPt) const {
4762 if (!Subtarget->useRealTrue16Insts() || MRI->getType(Src) != LLT::scalar(16))
4763 return Src;
4764
4765 const RegisterBank *SrcRB = RBI.getRegBank(Src, *MRI, TRI);
4766 if (!SrcRB || SrcRB->getID() != AMDGPU::VGPRRegBankID)
4767 return Src;
4768
4769 MachineIRBuilder B(*InsertPt);
4770
4771 Register ImpDefReg = MRI->createVirtualRegister(&AMDGPU::VGPR_16RegClass);
4772 B.buildInstr(TargetOpcode::IMPLICIT_DEF).addDef(ImpDefReg);
4773
4774 Register DstReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
4775 B.buildInstr(AMDGPU::REG_SEQUENCE)
4776 .addDef(DstReg)
4777 .addReg(Src)
4778 .addImm(AMDGPU::lo16)
4779 .addReg(ImpDefReg)
4780 .addImm(AMDGPU::hi16);
4781
4782 return DstReg;
4783}
4784
4785///
4786/// This will select either an SGPR or VGPR operand and will save us from
4787/// having to write an extra tablegen pattern.
4789AMDGPUInstructionSelector::selectVSRC0(MachineOperand &Root) const {
4790 return {{
4791 [=](MachineInstrBuilder &MIB) { MIB.add(Root); }
4792 }};
4793}
4794
4796AMDGPUInstructionSelector::selectVOP3Mods0(MachineOperand &Root) const {
4797 Register Src;
4798 unsigned Mods;
4799 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
4800
4801 return {{
4802 [=](MachineInstrBuilder &MIB) {
4803 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4804 },
4805 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
4806 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4807 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4808 }};
4809}
4810
4812AMDGPUInstructionSelector::selectVOP3BMods0(MachineOperand &Root) const {
4813 Register Src;
4814 unsigned Mods;
4815 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
4816 /*IsCanonicalizing=*/true,
4817 /*AllowAbs=*/false);
4818
4819 return {{
4820 [=](MachineInstrBuilder &MIB) {
4821 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4822 },
4823 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
4824 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4825 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4826 }};
4827}
4828
4830AMDGPUInstructionSelector::selectVOP3OMods(MachineOperand &Root) const {
4831 return {{
4832 [=](MachineInstrBuilder &MIB) { MIB.add(Root); },
4833 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); }, // clamp
4834 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // omod
4835 }};
4836}
4837
4839AMDGPUInstructionSelector::selectVOP3Mods(MachineOperand &Root) const {
4840 Register Src;
4841 unsigned Mods;
4842 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
4843
4844 return {{
4845 [=](MachineInstrBuilder &MIB) {
4846 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4847 },
4848 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4849 }};
4850}
4851
4853AMDGPUInstructionSelector::selectVOP3ModsNonCanonicalizing(
4854 MachineOperand &Root) const {
4855 Register Src;
4856 unsigned Mods;
4857 std::tie(Src, Mods) =
4858 selectVOP3ModsImpl(Root.getReg(), /*IsCanonicalizing=*/false);
4859
4860 return {{
4861 [=](MachineInstrBuilder &MIB) {
4862 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4863 },
4864 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4865 }};
4866}
4867
4869AMDGPUInstructionSelector::selectVOP3BMods(MachineOperand &Root) const {
4870 Register Src;
4871 unsigned Mods;
4872 std::tie(Src, Mods) =
4873 selectVOP3ModsImpl(Root.getReg(), /*IsCanonicalizing=*/true,
4874 /*AllowAbs=*/false);
4875
4876 return {{
4877 [=](MachineInstrBuilder &MIB) {
4878 MIB.addReg(copyToVGPRIfSrcFolded(Src, Mods, Root, MIB));
4879 },
4880 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
4881 }};
4882}
4883
4885AMDGPUInstructionSelector::selectVOP3NoMods(MachineOperand &Root) const {
4886 Register Reg = Root.getReg();
4887 const MachineInstr *Def = getDefIgnoringCopies(Reg, *MRI);
4888 if (Def->getOpcode() == AMDGPU::G_FNEG || Def->getOpcode() == AMDGPU::G_FABS)
4889 return {};
4890 return {{
4891 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
4892 }};
4893}
4894
4895enum class SrcStatus {
4900 // This means current op = [op_upper, op_lower] and src = -op_lower.
4903 // This means current op = [op_upper, op_lower] and src = [op_upper,
4904 // -op_lower].
4912};
4913/// Test if the MI is truncating to half, such as `%reg0:n = G_TRUNC %reg1:2n`
4914static bool isTruncHalf(const MachineInstr *MI,
4915 const MachineRegisterInfo &MRI) {
4916 if (MI->getOpcode() != AMDGPU::G_TRUNC)
4917 return false;
4918
4919 unsigned DstSize = MRI.getType(MI->getOperand(0).getReg()).getSizeInBits();
4920 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4921 return DstSize * 2 == SrcSize;
4922}
4923
4924/// Test if the MI is logic shift right with half bits,
4925/// such as `%reg0:2n =G_LSHR %reg1:2n, CONST(n)`
4926static bool isLshrHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI) {
4927 if (MI->getOpcode() != AMDGPU::G_LSHR)
4928 return false;
4929
4930 Register ShiftSrc;
4931 std::optional<ValueAndVReg> ShiftAmt;
4932 if (mi_match(MI->getOperand(0).getReg(), MRI,
4933 m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt)))) {
4934 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4935 unsigned Shift = ShiftAmt->Value.getZExtValue();
4936 return Shift * 2 == SrcSize;
4937 }
4938 return false;
4939}
4940
4941/// Test if the MI is shift left with half bits,
4942/// such as `%reg0:2n =G_SHL %reg1:2n, CONST(n)`
4943static bool isShlHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI) {
4944 if (MI->getOpcode() != AMDGPU::G_SHL)
4945 return false;
4946
4947 Register ShiftSrc;
4948 std::optional<ValueAndVReg> ShiftAmt;
4949 if (mi_match(MI->getOperand(0).getReg(), MRI,
4950 m_GShl(m_Reg(ShiftSrc), m_GCst(ShiftAmt)))) {
4951 unsigned SrcSize = MRI.getType(MI->getOperand(1).getReg()).getSizeInBits();
4952 unsigned Shift = ShiftAmt->Value.getZExtValue();
4953 return Shift * 2 == SrcSize;
4954 }
4955 return false;
4956}
4957
4958/// Test function, if the MI is `%reg0:n, %reg1:n = G_UNMERGE_VALUES %reg2:2n`
4959static bool isUnmergeHalf(const MachineInstr *MI,
4960 const MachineRegisterInfo &MRI) {
4961 if (MI->getOpcode() != AMDGPU::G_UNMERGE_VALUES)
4962 return false;
4963 return MI->getNumOperands() == 3 && MI->getOperand(0).isDef() &&
4964 MI->getOperand(1).isDef() && !MI->getOperand(2).isDef();
4965}
4966
4968
4970 const MachineRegisterInfo &MRI) {
4971 LLT OpTy = MRI.getType(Reg);
4972 if (OpTy.isScalar())
4973 return TypeClass::SCALAR;
4974 if (OpTy.isVector() && OpTy.getNumElements() == 2)
4977}
4978
4980 const MachineRegisterInfo &MRI) {
4981 TypeClass NegType = isVectorOfTwoOrScalar(Reg, MRI);
4982 if (NegType != TypeClass::VECTOR_OF_TWO && NegType != TypeClass::SCALAR)
4983 return SrcStatus::INVALID;
4984
4985 switch (S) {
4986 case SrcStatus::IS_SAME:
4987 if (NegType == TypeClass::VECTOR_OF_TWO) {
4988 // Vector of 2:
4989 // [SrcHi, SrcLo] = [CurrHi, CurrLo]
4990 // [CurrHi, CurrLo] = neg [OpHi, OpLo](2 x Type)
4991 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
4992 // [SrcHi, SrcLo] = [-OpHi, -OpLo]
4994 }
4995 if (NegType == TypeClass::SCALAR) {
4996 // Scalar:
4997 // [SrcHi, SrcLo] = [CurrHi, CurrLo]
4998 // [CurrHi, CurrLo] = neg [OpHi, OpLo](Type)
4999 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5000 // [SrcHi, SrcLo] = [-OpHi, OpLo]
5001 return SrcStatus::IS_HI_NEG;
5002 }
5003 break;
5005 if (NegType == TypeClass::VECTOR_OF_TWO) {
5006 // Vector of 2:
5007 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5008 // [CurrHi, CurrLo] = neg [OpHi, OpLo](2 x Type)
5009 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5010 // [SrcHi, SrcLo] = [-(-OpHi), -OpLo] = [OpHi, -OpLo]
5011 return SrcStatus::IS_LO_NEG;
5012 }
5013 if (NegType == TypeClass::SCALAR) {
5014 // Scalar:
5015 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5016 // [CurrHi, CurrLo] = neg [OpHi, OpLo](Type)
5017 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5018 // [SrcHi, SrcLo] = [-(-OpHi), OpLo] = [OpHi, OpLo]
5019 return SrcStatus::IS_SAME;
5020 }
5021 break;
5023 if (NegType == TypeClass::VECTOR_OF_TWO) {
5024 // Vector of 2:
5025 // [SrcHi, SrcLo] = [CurrHi, -CurrLo]
5026 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](2 x Type)
5027 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5028 // [SrcHi, SrcLo] = [-OpHi, -(-OpLo)] = [-OpHi, OpLo]
5029 return SrcStatus::IS_HI_NEG;
5030 }
5031 if (NegType == TypeClass::SCALAR) {
5032 // Scalar:
5033 // [SrcHi, SrcLo] = [CurrHi, -CurrLo]
5034 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](Type)
5035 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5036 // [SrcHi, SrcLo] = [-OpHi, -OpLo]
5038 }
5039 break;
5041 if (NegType == TypeClass::VECTOR_OF_TWO) {
5042 // Vector of 2:
5043 // [SrcHi, SrcLo] = [-CurrHi, -CurrLo]
5044 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](2 x Type)
5045 // [CurrHi, CurrLo] = [-OpHi, -OpLo](2 x Type)
5046 // [SrcHi, SrcLo] = [OpHi, OpLo]
5047 return SrcStatus::IS_SAME;
5048 }
5049 if (NegType == TypeClass::SCALAR) {
5050 // Scalar:
5051 // [SrcHi, SrcLo] = [-CurrHi, -CurrLo]
5052 // [CurrHi, CurrLo] = fneg [OpHi, OpLo](Type)
5053 // [CurrHi, CurrLo] = [-OpHi, OpLo](Type)
5054 // [SrcHi, SrcLo] = [OpHi, -OpLo]
5055 return SrcStatus::IS_LO_NEG;
5056 }
5057 break;
5059 // Vector of 2:
5060 // Src = CurrUpper
5061 // Curr = [CurrUpper, CurrLower]
5062 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5063 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5064 // Src = -OpUpper
5065 //
5066 // Scalar:
5067 // Src = CurrUpper
5068 // Curr = [CurrUpper, CurrLower]
5069 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5070 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5071 // Src = -OpUpper
5074 if (NegType == TypeClass::VECTOR_OF_TWO) {
5075 // Vector of 2:
5076 // Src = CurrLower
5077 // Curr = [CurrUpper, CurrLower]
5078 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5079 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5080 // Src = -OpLower
5082 }
5083 if (NegType == TypeClass::SCALAR) {
5084 // Scalar:
5085 // Src = CurrLower
5086 // Curr = [CurrUpper, CurrLower]
5087 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5088 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5089 // Src = OpLower
5091 }
5092 break;
5094 // Vector of 2:
5095 // Src = -CurrUpper
5096 // Curr = [CurrUpper, CurrLower]
5097 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5098 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5099 // Src = -(-OpUpper) = OpUpper
5100 //
5101 // Scalar:
5102 // Src = -CurrUpper
5103 // Curr = [CurrUpper, CurrLower]
5104 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5105 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5106 // Src = -(-OpUpper) = OpUpper
5109 if (NegType == TypeClass::VECTOR_OF_TWO) {
5110 // Vector of 2:
5111 // Src = -CurrLower
5112 // Curr = [CurrUpper, CurrLower]
5113 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](2 x Type)
5114 // [CurrUpper, CurrLower] = [-OpUpper, -OpLower](2 x Type)
5115 // Src = -(-OpLower) = OpLower
5117 }
5118 if (NegType == TypeClass::SCALAR) {
5119 // Scalar:
5120 // Src = -CurrLower
5121 // Curr = [CurrUpper, CurrLower]
5122 // [CurrUpper, CurrLower] = fneg [OpUpper, OpLower](Type)
5123 // [CurrUpper, CurrLower] = [-OpUpper, OpLower](Type)
5124 // Src = -OpLower
5126 }
5127 break;
5128 default:
5129 break;
5130 }
5131 llvm_unreachable("unexpected SrcStatus & NegType combination");
5132}
5133
5134static std::optional<std::pair<Register, SrcStatus>>
5135calcNextStatus(std::pair<Register, SrcStatus> Curr,
5136 const MachineRegisterInfo &MRI) {
5137 const MachineInstr *MI = MRI.getVRegDef(Curr.first);
5138
5139 unsigned Opc = MI->getOpcode();
5140
5141 // Handle general Opc cases.
5142 switch (Opc) {
5143 case AMDGPU::G_BITCAST:
5144 return std::optional<std::pair<Register, SrcStatus>>(
5145 {MI->getOperand(1).getReg(), Curr.second});
5146 case AMDGPU::COPY:
5147 if (MI->getOperand(1).getReg().isPhysical())
5148 return std::nullopt;
5149 return std::optional<std::pair<Register, SrcStatus>>(
5150 {MI->getOperand(1).getReg(), Curr.second});
5151 case AMDGPU::G_FNEG: {
5152 SrcStatus Stat = getNegStatus(Curr.first, Curr.second, MRI);
5153 if (Stat == SrcStatus::INVALID)
5154 return std::nullopt;
5155 return std::optional<std::pair<Register, SrcStatus>>(
5156 {MI->getOperand(1).getReg(), Stat});
5157 }
5158 default:
5159 break;
5160 }
5161
5162 // Calc next Stat from current Stat.
5163 switch (Curr.second) {
5164 case SrcStatus::IS_SAME:
5165 if (isTruncHalf(MI, MRI))
5166 return std::optional<std::pair<Register, SrcStatus>>(
5167 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF});
5168 else if (isUnmergeHalf(MI, MRI)) {
5169 if (Curr.first == MI->getOperand(0).getReg())
5170 return std::optional<std::pair<Register, SrcStatus>>(
5171 {MI->getOperand(2).getReg(), SrcStatus::IS_LOWER_HALF});
5172 return std::optional<std::pair<Register, SrcStatus>>(
5173 {MI->getOperand(2).getReg(), SrcStatus::IS_UPPER_HALF});
5174 }
5175 break;
5177 if (isTruncHalf(MI, MRI)) {
5178 // [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5179 // [CurrHi, CurrLo] = trunc [OpUpper, OpLower] = OpLower
5180 // = [OpLowerHi, OpLowerLo]
5181 // Src = [SrcHi, SrcLo] = [-CurrHi, CurrLo]
5182 // = [-OpLowerHi, OpLowerLo]
5183 // = -OpLower
5184 return std::optional<std::pair<Register, SrcStatus>>(
5185 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5186 }
5187 if (isUnmergeHalf(MI, MRI)) {
5188 if (Curr.first == MI->getOperand(0).getReg())
5189 return std::optional<std::pair<Register, SrcStatus>>(
5190 {MI->getOperand(2).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5191 return std::optional<std::pair<Register, SrcStatus>>(
5192 {MI->getOperand(2).getReg(), SrcStatus::IS_UPPER_HALF_NEG});
5193 }
5194 break;
5196 if (isShlHalf(MI, MRI))
5197 return std::optional<std::pair<Register, SrcStatus>>(
5198 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF});
5199 break;
5201 if (isLshrHalf(MI, MRI))
5202 return std::optional<std::pair<Register, SrcStatus>>(
5203 {MI->getOperand(1).getReg(), SrcStatus::IS_UPPER_HALF});
5204 break;
5206 if (isShlHalf(MI, MRI))
5207 return std::optional<std::pair<Register, SrcStatus>>(
5208 {MI->getOperand(1).getReg(), SrcStatus::IS_LOWER_HALF_NEG});
5209 break;
5211 if (isLshrHalf(MI, MRI))
5212 return std::optional<std::pair<Register, SrcStatus>>(
5213 {MI->getOperand(1).getReg(), SrcStatus::IS_UPPER_HALF_NEG});
5214 break;
5215 default:
5216 break;
5217 }
5218 return std::nullopt;
5219}
5220
5221/// This is used to control valid status that current MI supports. For example,
5222/// non floating point intrinsic such as @llvm.amdgcn.sdot2 does not support NEG
5223/// bit on VOP3P.
5224/// The class can be further extended to recognize support on SEL, NEG, ABS bit
5225/// for different MI on different arch
5227private:
5228 bool HasNeg = false;
5229 // Assume all complex pattern of VOP3P have opsel.
5230 bool HasOpsel = true;
5231
5232public:
5234 const MachineInstr *MI = MRI.getVRegDef(Reg);
5235 unsigned Opc = MI->getOpcode();
5236
5237 if (Opc == TargetOpcode::G_INTRINSIC) {
5238 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(*MI).getIntrinsicID();
5239 // Only float point intrinsic has neg & neg_hi bits.
5240 if (IntrinsicID == Intrinsic::amdgcn_fdot2)
5241 HasNeg = true;
5243 // Keep same for generic op.
5244 HasNeg = true;
5245 }
5246 }
5247 bool checkOptions(SrcStatus Stat) const {
5248 if (!HasNeg &&
5249 (Stat >= SrcStatus::NEG_START && Stat <= SrcStatus::NEG_END)) {
5250 return false;
5251 }
5252 if (!HasOpsel &&
5253 (Stat >= SrcStatus::HALF_START && Stat <= SrcStatus::HALF_END)) {
5254 return false;
5255 }
5256 return true;
5257 }
5258};
5259
5262 int MaxDepth = 3) {
5263 int Depth = 0;
5264 auto Curr = calcNextStatus({Reg, SrcStatus::IS_SAME}, MRI);
5266
5267 while (Depth <= MaxDepth && Curr.has_value()) {
5268 Depth++;
5269 if (SO.checkOptions(Curr.value().second))
5270 Statlist.push_back(Curr.value());
5271 Curr = calcNextStatus(Curr.value(), MRI);
5272 }
5273
5274 return Statlist;
5275}
5276
5277static std::pair<Register, SrcStatus>
5279 int MaxDepth = 3) {
5280 int Depth = 0;
5281 std::pair<Register, SrcStatus> LastSameOrNeg = {Reg, SrcStatus::IS_SAME};
5282 auto Curr = calcNextStatus(LastSameOrNeg, MRI);
5283
5284 while (Depth <= MaxDepth && Curr.has_value()) {
5285 Depth++;
5286 SrcStatus Stat = Curr.value().second;
5287 if (SO.checkOptions(Stat)) {
5288 if (Stat == SrcStatus::IS_SAME || Stat == SrcStatus::IS_HI_NEG ||
5290 LastSameOrNeg = Curr.value();
5291 }
5292 Curr = calcNextStatus(Curr.value(), MRI);
5293 }
5294
5295 return LastSameOrNeg;
5296}
5297
5298static bool isSameBitWidth(Register Reg1, Register Reg2,
5299 const MachineRegisterInfo &MRI) {
5300 unsigned Width1 = MRI.getType(Reg1).getSizeInBits();
5301 unsigned Width2 = MRI.getType(Reg2).getSizeInBits();
5302 return Width1 == Width2;
5303}
5304
5305static unsigned updateMods(SrcStatus HiStat, SrcStatus LoStat, unsigned Mods) {
5306 // SrcStatus::IS_LOWER_HALF remain 0.
5307 if (HiStat == SrcStatus::IS_UPPER_HALF_NEG) {
5308 Mods ^= SISrcMods::NEG_HI;
5309 Mods |= SISrcMods::OP_SEL_1;
5310 } else if (HiStat == SrcStatus::IS_UPPER_HALF)
5311 Mods |= SISrcMods::OP_SEL_1;
5312 else if (HiStat == SrcStatus::IS_LOWER_HALF_NEG)
5313 Mods ^= SISrcMods::NEG_HI;
5314 else if (HiStat == SrcStatus::IS_HI_NEG)
5315 Mods ^= SISrcMods::NEG_HI;
5316
5317 if (LoStat == SrcStatus::IS_UPPER_HALF_NEG) {
5318 Mods ^= SISrcMods::NEG;
5319 Mods |= SISrcMods::OP_SEL_0;
5320 } else if (LoStat == SrcStatus::IS_UPPER_HALF)
5321 Mods |= SISrcMods::OP_SEL_0;
5322 else if (LoStat == SrcStatus::IS_LOWER_HALF_NEG)
5323 Mods |= SISrcMods::NEG;
5324 else if (LoStat == SrcStatus::IS_HI_NEG)
5325 Mods ^= SISrcMods::NEG;
5326
5327 return Mods;
5328}
5329
5330static bool isValidToPack(SrcStatus HiStat, SrcStatus LoStat, Register NewReg,
5331 Register RootReg, const SIInstrInfo &TII,
5332 const MachineRegisterInfo &MRI) {
5333 auto IsHalfState = [](SrcStatus S) {
5336 };
5337 return isSameBitWidth(NewReg, RootReg, MRI) && IsHalfState(LoStat) &&
5338 IsHalfState(HiStat);
5339}
5340
5341std::pair<Register, unsigned> AMDGPUInstructionSelector::selectVOP3PModsImpl(
5342 Register RootReg, const MachineRegisterInfo &MRI, bool IsDOT) const {
5343 unsigned Mods = 0;
5344 // No modification if Root type is not form of <2 x Type>.
5345 if (isVectorOfTwoOrScalar(RootReg, MRI) != TypeClass::VECTOR_OF_TWO) {
5346 Mods |= SISrcMods::OP_SEL_1;
5347 return {RootReg, Mods};
5348 }
5349
5350 SearchOptions SO(RootReg, MRI);
5351
5352 std::pair<Register, SrcStatus> Stat = getLastSameOrNeg(RootReg, MRI, SO);
5353
5354 if (Stat.second == SrcStatus::IS_BOTH_NEG)
5356 else if (Stat.second == SrcStatus::IS_HI_NEG)
5357 Mods ^= SISrcMods::NEG_HI;
5358 else if (Stat.second == SrcStatus::IS_LO_NEG)
5359 Mods ^= SISrcMods::NEG;
5360
5361 // 64-bit VOP3P instructions do not have OPSEL or ABS. Bail on v2f64 or v2i64.
5362 // TODO: Select NEG_LO and NEG_HI modifiers from BUILD_VECTOR.
5363 if (MRI.getType(RootReg).getSizeInBits() == 128) {
5364 Mods |= SISrcMods::OP_SEL_1; // Just the default, OPSEL unsupported.
5365 return {Stat.first, Mods};
5366 }
5367
5368 GBuildVector *MI;
5369 if (!mi_match(Stat.first, MRI, m_GBuildVector(MI)) ||
5370 MI->getNumOperands() != 3 || (IsDOT && Subtarget->hasDOTOpSelHazard())) {
5371 Mods |= SISrcMods::OP_SEL_1;
5372 return {Stat.first, Mods};
5373 }
5374
5376 getSrcStats(MI->getOperand(2).getReg(), MRI, SO);
5377
5378 if (StatlistHi.empty()) {
5379 Mods |= SISrcMods::OP_SEL_1;
5380 return {Stat.first, Mods};
5381 }
5382
5384 getSrcStats(MI->getOperand(1).getReg(), MRI, SO);
5385
5386 if (StatlistLo.empty()) {
5387 Mods |= SISrcMods::OP_SEL_1;
5388 return {Stat.first, Mods};
5389 }
5390
5391 for (int I = StatlistHi.size() - 1; I >= 0; I--) {
5392 for (int J = StatlistLo.size() - 1; J >= 0; J--) {
5393 if (StatlistHi[I].first == StatlistLo[J].first &&
5394 isValidToPack(StatlistHi[I].second, StatlistLo[J].second,
5395 StatlistHi[I].first, RootReg, TII, MRI))
5396 return {StatlistHi[I].first,
5397 updateMods(StatlistHi[I].second, StatlistLo[J].second, Mods)};
5398 }
5399 }
5400 // Packed instructions do not have abs modifiers.
5401 Mods |= SISrcMods::OP_SEL_1;
5402
5403 return {Stat.first, Mods};
5404}
5405
5406// Removed unused function `getAllKindImm` to eliminate dead code.
5407
5408static bool checkRB(Register Reg, unsigned int RBNo,
5409 const AMDGPURegisterBankInfo &RBI,
5410 const MachineRegisterInfo &MRI,
5411 const TargetRegisterInfo &TRI) {
5412 const RegisterBank *RB = RBI.getRegBank(Reg, MRI, TRI);
5413 return RB->getID() == RBNo;
5414}
5415
5416// This function is used to get the correct register bank for returned reg.
5417// Assume:
5418// 1. VOP3P is always legal for VGPR.
5419// 2. RootOp's regbank is legal.
5420// Thus
5421// 1. If RootOp is SGPR, then NewOp can be SGPR or VGPR.
5422// 2. If RootOp is VGPR, then NewOp must be VGPR.
5423static Register
5426 const TargetRegisterInfo &TRI, const SIInstrInfo &TII) {
5427 // RootOp can only be VGPR or SGPR (some hand written cases such as.
5428 // inst-select-ashr.v2s16.mir::ashr_v2s16_vs).
5429 if (checkRB(RootReg, AMDGPU::SGPRRegBankID, RBI, MRI, TRI) ||
5430 checkRB(NewReg, AMDGPU::VGPRRegBankID, RBI, MRI, TRI))
5431 return NewReg;
5432
5433 if (mi_match(RootReg, MRI, m_Copy(m_SpecificReg(NewReg)))) {
5434 // RootOp is VGPR, NewOp is not VGPR, but RootOp = COPY NewOp.
5435 return RootReg;
5436 }
5437
5438 Register DstReg = MRI.cloneVirtualRegister(RootReg);
5439 MachineInstrBuilder MIB = BuildMI(*Use.getParent(), Use, Use.getDebugLoc(),
5440 TII.get(AMDGPU::COPY), DstReg)
5441 .addReg(NewReg);
5442
5443 // Only accept VGPR.
5444 return MIB->getOperand(0).getReg();
5445}
5446
5448AMDGPUInstructionSelector::selectVOP3PRetHelper(MachineOperand &Root,
5449 bool IsDOT) const {
5450 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
5451 Register Reg;
5452 unsigned Mods;
5453 std::tie(Reg, Mods) = selectVOP3PModsImpl(Root.getReg(), MRI, IsDOT);
5454
5455 Reg = getLegalRegBank(Reg, Root.getReg(), *Root.getParent(), RBI, MRI, TRI,
5456 TII);
5457 return {{
5458 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
5459 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5460 }};
5461}
5462
5464AMDGPUInstructionSelector::selectVOP3PMods(MachineOperand &Root) const {
5465
5466 return selectVOP3PRetHelper(Root);
5467}
5468
5470AMDGPUInstructionSelector::selectVOP3PModsDOT(MachineOperand &Root) const {
5471
5472 return selectVOP3PRetHelper(Root, true);
5473}
5474
5476AMDGPUInstructionSelector::selectVOP3PNoModsDOT(MachineOperand &Root) const {
5477 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
5478 Register Src;
5479 unsigned Mods;
5480 std::tie(Src, Mods) = selectVOP3PModsImpl(Root.getReg(), MRI, true /*IsDOT*/);
5481 if (Mods != SISrcMods::OP_SEL_1)
5482 return {};
5483
5484 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); }}};
5485}
5486
5488AMDGPUInstructionSelector::selectVOP3PModsF32(MachineOperand &Root) const {
5489 Register Src;
5490 unsigned Mods;
5491 std::tie(Src, Mods) = selectVOP3PModsF32Impl(Root.getReg());
5492
5493 return {{
5494 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5495 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5496 }};
5497}
5498
5500AMDGPUInstructionSelector::selectVOP3PNoModsF32(MachineOperand &Root) const {
5501 Register Src;
5502 unsigned Mods;
5503 std::tie(Src, Mods) = selectVOP3PModsF32Impl(Root.getReg());
5504 if (Mods != SISrcMods::OP_SEL_1)
5505 return {};
5506
5507 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); }}};
5508}
5509
5511AMDGPUInstructionSelector::selectWMMAOpSelVOP3PMods(
5512 MachineOperand &Root) const {
5513 assert((Root.isImm() && (Root.getImm() == -1 || Root.getImm() == 0)) &&
5514 "expected i1 value");
5515 unsigned Mods = SISrcMods::OP_SEL_1;
5516 if (Root.getImm() != 0)
5517 Mods |= SISrcMods::OP_SEL_0;
5518
5519 return {{
5520 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5521 }};
5522}
5523
5525 MachineInstr *InsertPt,
5526 MachineRegisterInfo &MRI) {
5527 const TargetRegisterClass *DstRegClass;
5528 switch (Elts.size()) {
5529 case 8:
5530 DstRegClass = &AMDGPU::VReg_256RegClass;
5531 break;
5532 case 4:
5533 DstRegClass = &AMDGPU::VReg_128RegClass;
5534 break;
5535 case 2:
5536 DstRegClass = &AMDGPU::VReg_64RegClass;
5537 break;
5538 default:
5539 llvm_unreachable("unhandled Reg sequence size");
5540 }
5541
5542 MachineIRBuilder B(*InsertPt);
5543 auto MIB = B.buildInstr(AMDGPU::REG_SEQUENCE)
5544 .addDef(MRI.createVirtualRegister(DstRegClass));
5545 for (unsigned i = 0; i < Elts.size(); ++i) {
5546 MIB.addReg(Elts[i]);
5548 }
5549 return MIB->getOperand(0).getReg();
5550}
5551
5552static void selectWMMAModsNegAbs(unsigned ModOpcode, unsigned &Mods,
5554 MachineInstr *InsertPt,
5555 MachineRegisterInfo &MRI) {
5556 if (ModOpcode == TargetOpcode::G_FNEG) {
5557 Mods |= SISrcMods::NEG;
5558 // Check if all elements also have abs modifier
5559 SmallVector<Register, 8> NegAbsElts;
5560 for (auto El : Elts) {
5561 Register FabsSrc;
5562 if (!mi_match(El, MRI, m_GFabs(m_Reg(FabsSrc))))
5563 break;
5564 NegAbsElts.push_back(FabsSrc);
5565 }
5566 if (Elts.size() != NegAbsElts.size()) {
5567 // Neg
5568 Src = buildRegSequence(Elts, InsertPt, MRI);
5569 } else {
5570 // Neg and Abs
5571 Mods |= SISrcMods::NEG_HI;
5572 Src = buildRegSequence(NegAbsElts, InsertPt, MRI);
5573 }
5574 } else {
5575 assert(ModOpcode == TargetOpcode::G_FABS);
5576 // Abs
5577 Mods |= SISrcMods::NEG_HI;
5578 Src = buildRegSequence(Elts, InsertPt, MRI);
5579 }
5580}
5581
5583AMDGPUInstructionSelector::selectWMMAModsF32NegAbs(MachineOperand &Root) const {
5584 Register Src = Root.getReg();
5585 unsigned Mods = SISrcMods::OP_SEL_1;
5587
5588 GBuildVector *BV;
5589 if (mi_match(Src, *MRI, m_GBuildVector(BV))) {
5590 assert(BV->getNumSources() > 0);
5591 // Based on first element decide which mod we match, neg or abs
5592 MachineInstr *ElF32 = MRI->getVRegDef(BV->getSourceReg(0));
5593 unsigned ModOpcode = (ElF32->getOpcode() == AMDGPU::G_FNEG)
5594 ? AMDGPU::G_FNEG
5595 : AMDGPU::G_FABS;
5596 for (unsigned i = 0; i < BV->getNumSources(); ++i) {
5597 ElF32 = MRI->getVRegDef(BV->getSourceReg(i));
5598 if (ElF32->getOpcode() != ModOpcode)
5599 break;
5600 EltsF32.push_back(ElF32->getOperand(1).getReg());
5601 }
5602
5603 // All elements had ModOpcode modifier
5604 if (BV->getNumSources() == EltsF32.size()) {
5605 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF32, Src, Root.getParent(),
5606 *MRI);
5607 }
5608 }
5609
5610 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5611 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5612}
5613
5615AMDGPUInstructionSelector::selectWMMAModsF16Neg(MachineOperand &Root) const {
5616 Register Src = Root.getReg();
5617 unsigned Mods = SISrcMods::OP_SEL_1;
5618 SmallVector<Register, 8> EltsV2F16;
5619
5620 GConcatVectors *CV;
5621 if (mi_match(Src, *MRI, m_GConcatVectors(CV))) {
5622 for (unsigned i = 0; i < CV->getNumSources(); ++i) {
5623 Register FNegSrc;
5624 if (!mi_match(CV->getSourceReg(i), *MRI, m_GFNeg(m_Reg(FNegSrc))))
5625 break;
5626 EltsV2F16.push_back(FNegSrc);
5627 }
5628
5629 // All elements had ModOpcode modifier
5630 if (CV->getNumSources() == EltsV2F16.size()) {
5631 Mods |= SISrcMods::NEG;
5632 Mods |= SISrcMods::NEG_HI;
5633 Src = buildRegSequence(EltsV2F16, Root.getParent(), *MRI);
5634 }
5635 }
5636
5637 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5638 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5639}
5640
5642AMDGPUInstructionSelector::selectWMMAModsF16NegAbs(MachineOperand &Root) const {
5643 Register Src = Root.getReg();
5644 unsigned Mods = SISrcMods::OP_SEL_1;
5645 SmallVector<Register, 8> EltsV2F16;
5646
5647 GConcatVectors *CV;
5648 if (mi_match(Src, *MRI, m_GConcatVectors(CV))) {
5649 assert(CV->getNumSources() > 0);
5650 MachineInstr *ElV2F16 = MRI->getVRegDef(CV->getSourceReg(0));
5651 // Based on first element decide which mod we match, neg or abs
5652 unsigned ModOpcode = (ElV2F16->getOpcode() == AMDGPU::G_FNEG)
5653 ? AMDGPU::G_FNEG
5654 : AMDGPU::G_FABS;
5655
5656 for (unsigned i = 0; i < CV->getNumSources(); ++i) {
5657 ElV2F16 = MRI->getVRegDef(CV->getSourceReg(i));
5658 if (ElV2F16->getOpcode() != ModOpcode)
5659 break;
5660 EltsV2F16.push_back(ElV2F16->getOperand(1).getReg());
5661 }
5662
5663 // All elements had ModOpcode modifier
5664 if (CV->getNumSources() == EltsV2F16.size()) {
5665 MachineIRBuilder B(*Root.getParent());
5666 selectWMMAModsNegAbs(ModOpcode, Mods, EltsV2F16, Src, Root.getParent(),
5667 *MRI);
5668 }
5669 }
5670
5671 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5672 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }}};
5673}
5674
5676AMDGPUInstructionSelector::selectWMMAVISrc(MachineOperand &Root) const {
5677 std::optional<FPValueAndVReg> FPValReg;
5678 if (mi_match(Root.getReg(), *MRI, m_GFCstOrSplat(FPValReg))) {
5679 if (TII.isInlineConstant(FPValReg->Value)) {
5680 return {{[=](MachineInstrBuilder &MIB) {
5681 MIB.addImm(FPValReg->Value.bitcastToAPInt().getSExtValue());
5682 }}};
5683 }
5684 // Non-inlineable splat floats should not fall-through for integer immediate
5685 // checks.
5686 return {};
5687 }
5688
5689 APInt ICst;
5690 if (mi_match(Root.getReg(), *MRI, m_ICstOrSplat(ICst))) {
5691 if (TII.isInlineConstant(ICst)) {
5692 return {
5693 {[=](MachineInstrBuilder &MIB) { MIB.addImm(ICst.getSExtValue()); }}};
5694 }
5695 }
5696
5697 return {};
5698}
5699
5701AMDGPUInstructionSelector::selectSWMMACIndex8(MachineOperand &Root) const {
5702 Register Src =
5703 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5704 unsigned Key = 0;
5705
5706 Register ShiftSrc;
5707 std::optional<ValueAndVReg> ShiftAmt;
5708 if (mi_match(Src, *MRI, m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt))) &&
5709 MRI->getType(ShiftSrc).getSizeInBits() == 32 &&
5710 ShiftAmt->Value.getZExtValue() % 8 == 0) {
5711 Key = ShiftAmt->Value.getZExtValue() / 8;
5712 Src = ShiftSrc;
5713 }
5714
5715 return {{
5716 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5717 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5718 }};
5719}
5720
5722AMDGPUInstructionSelector::selectSWMMACIndex16(MachineOperand &Root) const {
5723
5724 Register Src =
5725 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5726 unsigned Key = 0;
5727
5728 Register ShiftSrc;
5729 std::optional<ValueAndVReg> ShiftAmt;
5730 if (mi_match(Src, *MRI, m_GLShr(m_Reg(ShiftSrc), m_GCst(ShiftAmt))) &&
5731 MRI->getType(ShiftSrc).getSizeInBits() == 32 &&
5732 ShiftAmt->Value.getZExtValue() == 16) {
5733 Src = ShiftSrc;
5734 Key = 1;
5735 }
5736
5737 return {{
5738 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5739 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5740 }};
5741}
5742
5744AMDGPUInstructionSelector::selectSWMMACIndex32(MachineOperand &Root) const {
5745 Register Src =
5746 getDefIgnoringCopies(Root.getReg(), *MRI)->getOperand(0).getReg();
5747 unsigned Key = 0;
5748
5749 Register S32 = matchZeroExtendFromS32(Src);
5750 if (!S32)
5751 S32 = matchAnyExtendFromS32(Src);
5752
5753 if (S32) {
5754 const MachineInstr *Def = getDefIgnoringCopies(S32, *MRI);
5755 if (Def->getOpcode() == TargetOpcode::G_UNMERGE_VALUES) {
5756 assert(Def->getNumOperands() == 3);
5757 Register DstReg1 = Def->getOperand(1).getReg();
5758 if (mi_match(S32, *MRI,
5759 m_any_of(m_SpecificReg(DstReg1), m_Copy(m_Reg(DstReg1))))) {
5760 Src = Def->getOperand(2).getReg();
5761 Key = 1;
5762 }
5763 }
5764 }
5765
5766 return {{
5767 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5768 [=](MachineInstrBuilder &MIB) { MIB.addImm(Key); } // index_key
5769 }};
5770}
5771
5773AMDGPUInstructionSelector::selectVOP3OpSelMods(MachineOperand &Root) const {
5774 Register Src;
5775 unsigned Mods;
5776 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
5777
5778 // FIXME: Handle op_sel
5779 return {{
5780 [=](MachineInstrBuilder &MIB) { MIB.addReg(Src); },
5781 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
5782 }};
5783}
5784
5785// FIXME-TRUE16 remove when fake16 is removed
5787AMDGPUInstructionSelector::selectVINTERPMods(MachineOperand &Root) const {
5788 Register Src;
5789 unsigned Mods;
5790 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
5791 /*IsCanonicalizing=*/true,
5792 /*AllowAbs=*/false,
5793 /*OpSel=*/false);
5794
5795 return {{
5796 [=](MachineInstrBuilder &MIB) {
5797 MIB.addReg(
5798 copyToVGPRIfSrcFolded(Src, Mods, Root, MIB, /* ForceVGPR */ true));
5799 },
5800 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
5801 }};
5802}
5803
5805AMDGPUInstructionSelector::selectVINTERPModsHi(MachineOperand &Root) const {
5806 Register Src;
5807 unsigned Mods;
5808 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg(),
5809 /*IsCanonicalizing=*/true,
5810 /*AllowAbs=*/false,
5811 /*OpSel=*/true);
5812
5813 return {{
5814 [=](MachineInstrBuilder &MIB) {
5815 MIB.addReg(
5816 copyToVGPRIfSrcFolded(Src, Mods, Root, MIB, /* ForceVGPR */ true));
5817 },
5818 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); }, // src0_mods
5819 }};
5820}
5821
5822// Given \p Offset and load specified by the \p Root operand check if \p Offset
5823// is a multiple of the load byte size. If it is update \p Offset to a
5824// pre-scaled value and return true.
5825bool AMDGPUInstructionSelector::selectScaleOffset(MachineOperand &Root,
5827 bool IsSigned) const {
5828 if (!Subtarget->hasScaleOffset())
5829 return false;
5830
5831 const MachineInstr &MI = *Root.getParent();
5832 MachineMemOperand *MMO = *MI.memoperands_begin();
5833
5834 if (!MMO->getSize().hasValue())
5835 return false;
5836
5837 uint64_t Size = MMO->getSize().getValue();
5838
5839 Register OffsetReg = matchExtendFromS32OrS32(Offset, IsSigned);
5840 if (!OffsetReg)
5841 OffsetReg = Offset;
5842
5843 if (auto Def = getDefSrcRegIgnoringCopies(OffsetReg, *MRI))
5844 OffsetReg = Def->Reg;
5845
5846 Register Op0;
5847 MachineInstr *Mul;
5848 bool ScaleOffset =
5849 (isPowerOf2_64(Size) &&
5850 mi_match(OffsetReg, *MRI,
5851 m_GShl(m_Reg(Op0),
5854 mi_match(OffsetReg, *MRI,
5856 m_Copy(m_SpecificICst(Size))))) ||
5857 mi_match(
5858 OffsetReg, *MRI,
5859 m_BinOp(IsSigned ? AMDGPU::S_MUL_I64_I32_PSEUDO : AMDGPU::S_MUL_U64,
5860 m_Reg(Op0), m_SpecificICst(Size))) ||
5861 // Match G_AMDGPU_MAD_U64_U32 offset, c, 0
5862 (mi_match(OffsetReg, *MRI, m_MInstr(Mul)) &&
5863 (Mul->getOpcode() == (IsSigned ? AMDGPU::G_AMDGPU_MAD_I64_I32
5864 : AMDGPU::G_AMDGPU_MAD_U64_U32) ||
5865 (IsSigned && Mul->getOpcode() == AMDGPU::G_AMDGPU_MAD_U64_U32 &&
5866 VT->signBitIsZero(Mul->getOperand(2).getReg()))) &&
5867 mi_match(Mul->getOperand(4).getReg(), *MRI, m_ZeroInt()) &&
5868 mi_match(Mul->getOperand(3).getReg(), *MRI,
5870 m_Copy(m_SpecificICst(Size))))) &&
5871 mi_match(Mul->getOperand(2).getReg(), *MRI, m_Reg(Op0)));
5872
5873 if (ScaleOffset)
5874 Offset = Op0;
5875
5876 return ScaleOffset;
5877}
5878
5879bool AMDGPUInstructionSelector::selectSmrdOffset(MachineOperand &Root,
5880 Register &Base,
5881 Register *SOffset,
5882 int64_t *Offset,
5883 bool *ScaleOffset) const {
5884 MachineInstr *MI = Root.getParent();
5885 MachineBasicBlock *MBB = MI->getParent();
5886
5887 // FIXME: We should shrink the GEP if the offset is known to be <= 32-bits,
5888 // then we can select all ptr + 32-bit offsets.
5889 SmallVector<GEPInfo, 4> AddrInfo;
5890 getAddrModeInfo(*MI, *MRI, AddrInfo);
5891
5892 if (AddrInfo.empty())
5893 return false;
5894
5895 const GEPInfo &GEPI = AddrInfo[0];
5896 std::optional<int64_t> EncodedImm;
5897
5898 if (ScaleOffset)
5899 *ScaleOffset = false;
5900
5901 if (SOffset && Offset) {
5902 EncodedImm = AMDGPU::getSMRDEncodedOffset(STI, GEPI.Imm, /*IsBuffer=*/false,
5903 /*HasSOffset=*/true);
5904 if (GEPI.SgprParts.size() == 1 && GEPI.Imm != 0 && EncodedImm &&
5905 AddrInfo.size() > 1) {
5906 const GEPInfo &GEPI2 = AddrInfo[1];
5907 if (GEPI2.SgprParts.size() == 2 && GEPI2.Imm == 0) {
5908 Register OffsetReg = GEPI2.SgprParts[1];
5909 if (ScaleOffset)
5910 *ScaleOffset =
5911 selectScaleOffset(Root, OffsetReg, false /* IsSigned */);
5912 OffsetReg = matchZeroExtendFromS32OrS32(OffsetReg);
5913 if (OffsetReg) {
5914 Base = GEPI2.SgprParts[0];
5915 *SOffset = OffsetReg;
5916 *Offset = *EncodedImm;
5917 if (*Offset >= 0 || !AMDGPU::hasSMRDSignedImmOffset(STI))
5918 return true;
5919
5920 // For unbuffered smem loads, it is illegal for the Immediate Offset
5921 // to be negative if the resulting (Offset + (M0 or SOffset or zero)
5922 // is negative. Handle the case where the Immediate Offset + SOffset
5923 // is negative.
5924 auto SKnown = VT->getKnownBits(*SOffset);
5925 if (*Offset + SKnown.getMinValue().getSExtValue() < 0)
5926 return false;
5927
5928 return true;
5929 }
5930 }
5931 }
5932 return false;
5933 }
5934
5935 EncodedImm = AMDGPU::getSMRDEncodedOffset(STI, GEPI.Imm, /*IsBuffer=*/false,
5936 /*HasSOffset=*/false);
5937 if (Offset && GEPI.SgprParts.size() == 1 && EncodedImm) {
5938 Base = GEPI.SgprParts[0];
5939 *Offset = *EncodedImm;
5940 return true;
5941 }
5942
5943 // SGPR offset is unsigned.
5944 if (SOffset && GEPI.SgprParts.size() == 1 && isUInt<32>(GEPI.Imm) &&
5945 GEPI.Imm != 0) {
5946 // If we make it this far we have a load with an 32-bit immediate offset.
5947 // It is OK to select this using a sgpr offset, because we have already
5948 // failed trying to select this load into one of the _IMM variants since
5949 // the _IMM Patterns are considered before the _SGPR patterns.
5950 Base = GEPI.SgprParts[0];
5951 *SOffset = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
5952 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::S_MOV_B32), *SOffset)
5953 .addImm(GEPI.Imm);
5954 return true;
5955 }
5956
5957 if (SOffset && GEPI.SgprParts.size() && GEPI.Imm == 0) {
5958 Register OffsetReg = GEPI.SgprParts[1];
5959 if (ScaleOffset)
5960 *ScaleOffset = selectScaleOffset(Root, OffsetReg, false /* IsSigned */);
5961 OffsetReg = matchZeroExtendFromS32OrS32(OffsetReg);
5962 if (OffsetReg) {
5963 Base = GEPI.SgprParts[0];
5964 *SOffset = OffsetReg;
5965 return true;
5966 }
5967 }
5968
5969 return false;
5970}
5971
5973AMDGPUInstructionSelector::selectSmrdImm(MachineOperand &Root) const {
5974 Register Base;
5975 int64_t Offset;
5976 if (!selectSmrdOffset(Root, Base, /* SOffset= */ nullptr, &Offset,
5977 /* ScaleOffset */ nullptr))
5978 return std::nullopt;
5979
5980 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
5981 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }}};
5982}
5983
5985AMDGPUInstructionSelector::selectSmrdImm32(MachineOperand &Root) const {
5986 SmallVector<GEPInfo, 4> AddrInfo;
5987 getAddrModeInfo(*Root.getParent(), *MRI, AddrInfo);
5988
5989 if (AddrInfo.empty() || AddrInfo[0].SgprParts.size() != 1)
5990 return std::nullopt;
5991
5992 const GEPInfo &GEPInfo = AddrInfo[0];
5993 Register PtrReg = GEPInfo.SgprParts[0];
5994 std::optional<int64_t> EncodedImm =
5995 AMDGPU::getSMRDEncodedLiteralOffset32(STI, GEPInfo.Imm);
5996 if (!EncodedImm)
5997 return std::nullopt;
5998
5999 return {{
6000 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrReg); },
6001 [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); }
6002 }};
6003}
6004
6006AMDGPUInstructionSelector::selectSmrdSgpr(MachineOperand &Root) const {
6007 Register Base, SOffset;
6008 bool ScaleOffset;
6009 if (!selectSmrdOffset(Root, Base, &SOffset, /* Offset= */ nullptr,
6010 &ScaleOffset))
6011 return std::nullopt;
6012
6013 unsigned CPol = ScaleOffset ? AMDGPU::CPol::SCAL : 0;
6014 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
6015 [=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
6016 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); }}};
6017}
6018
6020AMDGPUInstructionSelector::selectSmrdSgprImm(MachineOperand &Root) const {
6021 Register Base, SOffset;
6022 int64_t Offset;
6023 bool ScaleOffset;
6024 if (!selectSmrdOffset(Root, Base, &SOffset, &Offset, &ScaleOffset))
6025 return std::nullopt;
6026
6027 unsigned CPol = ScaleOffset ? AMDGPU::CPol::SCAL : 0;
6028 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(Base); },
6029 [=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
6030 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); },
6031 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); }}};
6032}
6033
6034std::pair<Register, int> AMDGPUInstructionSelector::selectFlatOffsetImpl(
6035 MachineOperand &Root, AMDGPU::FlatAddrSpace FlatVariant) const {
6036 MachineInstr *MI = Root.getParent();
6037
6038 auto Default = std::pair(Root.getReg(), 0);
6039
6040 if (!STI.hasFlatInstOffsets())
6041 return Default;
6042
6043 Register PtrBase;
6044 int64_t ConstOffset;
6045 bool IsInBounds;
6046 std::tie(PtrBase, ConstOffset, IsInBounds) =
6047 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6048
6049 // Adding the offset to the base address with an immediate in a FLAT
6050 // instruction must not change the memory aperture in which the address falls.
6051 // Therefore we can only fold offsets from inbounds GEPs into FLAT
6052 // instructions.
6053 if (ConstOffset == 0 ||
6054 (FlatVariant == AMDGPU::FlatAddrSpace::FlatScratch &&
6055 !isFlatScratchBaseLegal(Root.getReg())) ||
6056 (FlatVariant == AMDGPU::FlatAddrSpace::FLAT && !IsInBounds))
6057 return Default;
6058
6059 unsigned AddrSpace = (*MI->memoperands_begin())->getAddrSpace();
6060 if (!TII.isLegalFLATOffset(ConstOffset, AddrSpace, FlatVariant))
6061 return Default;
6062
6063 return std::pair(PtrBase, ConstOffset);
6064}
6065
6067AMDGPUInstructionSelector::selectFlatOffset(MachineOperand &Root) const {
6068 auto PtrWithOffset = selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FLAT);
6069
6070 return {{
6071 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6072 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6073 }};
6074}
6075
6077AMDGPUInstructionSelector::selectGlobalOffset(MachineOperand &Root) const {
6078 auto PtrWithOffset =
6079 selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FlatGlobal);
6080
6081 return {{
6082 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6083 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6084 }};
6085}
6086
6088AMDGPUInstructionSelector::selectScratchOffset(MachineOperand &Root) const {
6089 auto PtrWithOffset =
6090 selectFlatOffsetImpl(Root, AMDGPU::FlatAddrSpace::FlatScratch);
6091
6092 return {{
6093 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrWithOffset.first); },
6094 [=](MachineInstrBuilder &MIB) { MIB.addImm(PtrWithOffset.second); },
6095 }};
6096}
6097
6098// Match (64-bit SGPR base) + (zext vgpr offset) + sext(imm offset)
6100AMDGPUInstructionSelector::selectGlobalSAddr(MachineOperand &Root,
6101 unsigned CPolBits,
6102 bool NeedIOffset) const {
6103 Register Addr = Root.getReg();
6104 Register PtrBase;
6105 int64_t ConstOffset;
6106 int64_t ImmOffset = 0;
6107
6108 // Match the immediate offset first, which canonically is moved as low as
6109 // possible.
6110 std::tie(PtrBase, ConstOffset, std::ignore) =
6111 getPtrBaseWithConstantOffset(Addr, *MRI);
6112
6113 if (ConstOffset != 0) {
6114 if (NeedIOffset &&
6115 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::GLOBAL_ADDRESS,
6117 Addr = PtrBase;
6118 ImmOffset = ConstOffset;
6119 } else {
6120 auto PtrBaseDef = getDefSrcRegIgnoringCopies(PtrBase, *MRI);
6121 if (isSGPR(PtrBaseDef->Reg)) {
6122 if (ConstOffset > 0) {
6123 // Offset is too large.
6124 //
6125 // saddr + large_offset -> saddr +
6126 // (voffset = large_offset & ~MaxOffset) +
6127 // (large_offset & MaxOffset);
6128 int64_t SplitImmOffset = 0, RemainderOffset = ConstOffset;
6129 if (NeedIOffset) {
6130 std::tie(SplitImmOffset, RemainderOffset) =
6131 TII.splitFlatOffset(ConstOffset, AMDGPUAS::GLOBAL_ADDRESS,
6133 }
6134
6135 if (Subtarget->hasSignedGVSOffset() ? isInt<32>(RemainderOffset)
6136 : isUInt<32>(RemainderOffset)) {
6137 MachineInstr *MI = Root.getParent();
6138 MachineBasicBlock *MBB = MI->getParent();
6139 Register HighBits =
6140 MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6141
6142 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32),
6143 HighBits)
6144 .addImm(RemainderOffset);
6145
6146 if (NeedIOffset)
6147 return {{
6148 [=](MachineInstrBuilder &MIB) {
6149 MIB.addReg(PtrBase);
6150 }, // saddr
6151 [=](MachineInstrBuilder &MIB) {
6152 MIB.addReg(HighBits);
6153 }, // voffset
6154 [=](MachineInstrBuilder &MIB) { MIB.addImm(SplitImmOffset); },
6155 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); },
6156 }};
6157 return {{
6158 [=](MachineInstrBuilder &MIB) { MIB.addReg(PtrBase); }, // saddr
6159 [=](MachineInstrBuilder &MIB) {
6160 MIB.addReg(HighBits);
6161 }, // voffset
6162 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); },
6163 }};
6164 }
6165 }
6166
6167 // We are adding a 64 bit SGPR and a constant. If constant bus limit
6168 // is 1 we would need to perform 1 or 2 extra moves for each half of
6169 // the constant and it is better to do a scalar add and then issue a
6170 // single VALU instruction to materialize zero. Otherwise it is less
6171 // instructions to perform VALU adds with immediates or inline literals.
6172 unsigned NumLiterals =
6173 !TII.isInlineConstant(APInt(32, Lo_32(ConstOffset))) +
6174 !TII.isInlineConstant(APInt(32, Hi_32(ConstOffset)));
6175 if (STI.getConstantBusLimit(AMDGPU::V_ADD_U32_e64) > NumLiterals)
6176 return std::nullopt;
6177 }
6178 }
6179 }
6180
6181 // Match the variable offset.
6182 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6183 if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
6184 // Look through the SGPR->VGPR copy.
6185 Register SAddr =
6186 getSrcRegIgnoringCopies(AddrDef->MI->getOperand(1).getReg(), *MRI);
6187
6188 if (isSGPR(SAddr)) {
6189 Register PtrBaseOffset = AddrDef->MI->getOperand(2).getReg();
6190
6191 // It's possible voffset is an SGPR here, but the copy to VGPR will be
6192 // inserted later.
6193 bool ScaleOffset = selectScaleOffset(Root, PtrBaseOffset,
6194 Subtarget->hasSignedGVSOffset());
6195 if (Register VOffset = matchExtendFromS32OrS32(
6196 PtrBaseOffset, Subtarget->hasSignedGVSOffset())) {
6197 if (NeedIOffset)
6198 return {{[=](MachineInstrBuilder &MIB) { // saddr
6199 MIB.addReg(SAddr);
6200 },
6201 [=](MachineInstrBuilder &MIB) { // voffset
6202 MIB.addReg(VOffset);
6203 },
6204 [=](MachineInstrBuilder &MIB) { // offset
6205 MIB.addImm(ImmOffset);
6206 },
6207 [=](MachineInstrBuilder &MIB) { // cpol
6208 MIB.addImm(CPolBits |
6209 (ScaleOffset ? AMDGPU::CPol::SCAL : 0));
6210 }}};
6211 return {{[=](MachineInstrBuilder &MIB) { // saddr
6212 MIB.addReg(SAddr);
6213 },
6214 [=](MachineInstrBuilder &MIB) { // voffset
6215 MIB.addReg(VOffset);
6216 },
6217 [=](MachineInstrBuilder &MIB) { // cpol
6218 MIB.addImm(CPolBits |
6219 (ScaleOffset ? AMDGPU::CPol::SCAL : 0));
6220 }}};
6221 }
6222 }
6223 }
6224
6225 // FIXME: We should probably have folded COPY (G_IMPLICIT_DEF) earlier, and
6226 // drop this.
6227 if (AddrDef->MI->getOpcode() == AMDGPU::G_IMPLICIT_DEF ||
6228 AddrDef->MI->getOpcode() == AMDGPU::G_CONSTANT || !isSGPR(AddrDef->Reg))
6229 return std::nullopt;
6230
6231 // It's cheaper to materialize a single 32-bit zero for vaddr than the two
6232 // moves required to copy a 64-bit SGPR to VGPR.
6233 MachineInstr *MI = Root.getParent();
6234 MachineBasicBlock *MBB = MI->getParent();
6235 Register VOffset = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6236
6237 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32), VOffset)
6238 .addImm(0);
6239
6240 if (NeedIOffset)
6241 return {{
6242 [=](MachineInstrBuilder &MIB) { MIB.addReg(AddrDef->Reg); }, // saddr
6243 [=](MachineInstrBuilder &MIB) { MIB.addReg(VOffset); }, // voffset
6244 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6245 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); } // cpol
6246 }};
6247 return {{
6248 [=](MachineInstrBuilder &MIB) { MIB.addReg(AddrDef->Reg); }, // saddr
6249 [=](MachineInstrBuilder &MIB) { MIB.addReg(VOffset); }, // voffset
6250 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPolBits); } // cpol
6251 }};
6252}
6253
6255AMDGPUInstructionSelector::selectGlobalSAddr(MachineOperand &Root) const {
6256 return selectGlobalSAddr(Root, 0);
6257}
6258
6260AMDGPUInstructionSelector::selectGlobalSAddrCPol(MachineOperand &Root) const {
6261 const MachineInstr &I = *Root.getParent();
6262
6263 // We are assuming CPol is always the last operand of the intrinsic.
6264 auto PassedCPol =
6265 I.getOperand(I.getNumOperands() - 1).getImm() & ~AMDGPU::CPol::SCAL;
6266 return selectGlobalSAddr(Root, PassedCPol);
6267}
6268
6270AMDGPUInstructionSelector::selectGlobalSAddrCPolM0(MachineOperand &Root) const {
6271 const MachineInstr &I = *Root.getParent();
6272
6273 // We are assuming CPol is second from last operand of the intrinsic.
6274 auto PassedCPol =
6275 I.getOperand(I.getNumOperands() - 2).getImm() & ~AMDGPU::CPol::SCAL;
6276 return selectGlobalSAddr(Root, PassedCPol);
6277}
6278
6280AMDGPUInstructionSelector::selectGlobalSAddrGLC(MachineOperand &Root) const {
6281 return selectGlobalSAddr(Root, AMDGPU::CPol::GLC);
6282}
6283
6285AMDGPUInstructionSelector::selectGlobalSAddrNoIOffset(
6286 MachineOperand &Root) const {
6287 const MachineInstr &I = *Root.getParent();
6288
6289 // We are assuming CPol is always the last operand of the intrinsic.
6290 auto PassedCPol =
6291 I.getOperand(I.getNumOperands() - 1).getImm() & ~AMDGPU::CPol::SCAL;
6292 return selectGlobalSAddr(Root, PassedCPol, false);
6293}
6294
6296AMDGPUInstructionSelector::selectGlobalSAddrNoIOffsetM0(
6297 MachineOperand &Root) const {
6298 const MachineInstr &I = *Root.getParent();
6299
6300 // We are assuming CPol is second from last operand of the intrinsic.
6301 auto PassedCPol =
6302 I.getOperand(I.getNumOperands() - 2).getImm() & ~AMDGPU::CPol::SCAL;
6303 return selectGlobalSAddr(Root, PassedCPol, false);
6304}
6305
6307AMDGPUInstructionSelector::selectScratchSAddr(MachineOperand &Root) const {
6308 Register Addr = Root.getReg();
6309 Register PtrBase;
6310 int64_t ConstOffset;
6311 int64_t ImmOffset = 0;
6312
6313 // Match the immediate offset first, which canonically is moved as low as
6314 // possible.
6315 std::tie(PtrBase, ConstOffset, std::ignore) =
6316 getPtrBaseWithConstantOffset(Addr, *MRI);
6317
6318 if (ConstOffset != 0 && isFlatScratchBaseLegal(Addr) &&
6319 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::PRIVATE_ADDRESS,
6321 Addr = PtrBase;
6322 ImmOffset = ConstOffset;
6323 }
6324
6325 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6326 if (AddrDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX) {
6327 int FI = AddrDef->MI->getOperand(1).getIndex();
6328 return {{
6329 [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); }, // saddr
6330 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); } // offset
6331 }};
6332 }
6333
6334 Register SAddr = AddrDef->Reg;
6335
6336 if (AddrDef->MI->getOpcode() == AMDGPU::G_PTR_ADD) {
6337 Register LHS = AddrDef->MI->getOperand(1).getReg();
6338 Register RHS = AddrDef->MI->getOperand(2).getReg();
6339 auto LHSDef = getDefSrcRegIgnoringCopies(LHS, *MRI);
6340 auto RHSDef = getDefSrcRegIgnoringCopies(RHS, *MRI);
6341
6342 if (LHSDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX &&
6343 isSGPR(RHSDef->Reg)) {
6344 int FI = LHSDef->MI->getOperand(1).getIndex();
6345 MachineInstr &I = *Root.getParent();
6346 MachineBasicBlock *BB = I.getParent();
6347 const DebugLoc &DL = I.getDebugLoc();
6348 SAddr = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
6349
6350 BuildMI(*BB, &I, DL, TII.get(AMDGPU::S_ADD_I32), SAddr)
6351 .addFrameIndex(FI)
6352 .addReg(RHSDef->Reg)
6353 .setOperandDead(3); // Dead scc
6354 }
6355 }
6356
6357 if (!isSGPR(SAddr))
6358 return std::nullopt;
6359
6360 return {{
6361 [=](MachineInstrBuilder &MIB) { MIB.addReg(SAddr); }, // saddr
6362 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); } // offset
6363 }};
6364}
6365
6366// Check whether the flat scratch SVS swizzle bug affects this access.
6367bool AMDGPUInstructionSelector::checkFlatScratchSVSSwizzleBug(
6368 Register VAddr, Register SAddr, uint64_t ImmOffset) const {
6369 if (!Subtarget->hasFlatScratchSVSSwizzleBug())
6370 return false;
6371
6372 // The bug affects the swizzling of SVS accesses if there is any carry out
6373 // from the two low order bits (i.e. from bit 1 into bit 2) when adding
6374 // voffset to (soffset + inst_offset).
6375 auto VKnown = VT->getKnownBits(VAddr);
6376 auto SKnown = KnownBits::add(VT->getKnownBits(SAddr),
6377 KnownBits::makeConstant(APInt(32, ImmOffset)));
6378 uint64_t VMax = VKnown.getMaxValue().getZExtValue();
6379 uint64_t SMax = SKnown.getMaxValue().getZExtValue();
6380 return (VMax & 3) + (SMax & 3) >= 4;
6381}
6382
6384AMDGPUInstructionSelector::selectScratchSVAddr(MachineOperand &Root) const {
6385 Register Addr = Root.getReg();
6386 Register PtrBase;
6387 int64_t ConstOffset;
6388 int64_t ImmOffset = 0;
6389
6390 // Match the immediate offset first, which canonically is moved as low as
6391 // possible.
6392 std::tie(PtrBase, ConstOffset, std::ignore) =
6393 getPtrBaseWithConstantOffset(Addr, *MRI);
6394
6395 Register OrigAddr = Addr;
6396 if (ConstOffset != 0 &&
6397 TII.isLegalFLATOffset(ConstOffset, AMDGPUAS::PRIVATE_ADDRESS,
6399 Addr = PtrBase;
6400 ImmOffset = ConstOffset;
6401 }
6402
6403 auto AddrDef = getDefSrcRegIgnoringCopies(Addr, *MRI);
6404 if (AddrDef->MI->getOpcode() != AMDGPU::G_PTR_ADD)
6405 return std::nullopt;
6406
6407 Register RHS = AddrDef->MI->getOperand(2).getReg();
6408 if (RBI.getRegBank(RHS, *MRI, TRI)->getID() != AMDGPU::VGPRRegBankID)
6409 return std::nullopt;
6410
6411 Register LHS = AddrDef->MI->getOperand(1).getReg();
6412 auto LHSDef = getDefSrcRegIgnoringCopies(LHS, *MRI);
6413
6414 if (OrigAddr != Addr) {
6415 if (!isFlatScratchBaseLegalSVImm(OrigAddr))
6416 return std::nullopt;
6417 } else {
6418 if (!isFlatScratchBaseLegalSV(OrigAddr))
6419 return std::nullopt;
6420 }
6421
6422 if (checkFlatScratchSVSSwizzleBug(RHS, LHS, ImmOffset))
6423 return std::nullopt;
6424
6425 unsigned CPol = selectScaleOffset(Root, RHS, true /* IsSigned */)
6427 : 0;
6428
6429 if (LHSDef->MI->getOpcode() == AMDGPU::G_FRAME_INDEX) {
6430 int FI = LHSDef->MI->getOperand(1).getIndex();
6431 return {{
6432 [=](MachineInstrBuilder &MIB) { MIB.addReg(RHS); }, // vaddr
6433 [=](MachineInstrBuilder &MIB) { MIB.addFrameIndex(FI); }, // saddr
6434 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6435 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); } // cpol
6436 }};
6437 }
6438
6439 if (!isSGPR(LHS))
6440 if (auto Def = getDefSrcRegIgnoringCopies(LHS, *MRI))
6441 LHS = Def->Reg;
6442
6443 if (!isSGPR(LHS))
6444 return std::nullopt;
6445
6446 return {{
6447 [=](MachineInstrBuilder &MIB) { MIB.addReg(RHS); }, // vaddr
6448 [=](MachineInstrBuilder &MIB) { MIB.addReg(LHS); }, // saddr
6449 [=](MachineInstrBuilder &MIB) { MIB.addImm(ImmOffset); }, // offset
6450 [=](MachineInstrBuilder &MIB) { MIB.addImm(CPol); } // cpol
6451 }};
6452}
6453
6455AMDGPUInstructionSelector::selectMUBUFScratchOffen(MachineOperand &Root) const {
6456 MachineInstr *MI = Root.getParent();
6457 MachineBasicBlock *MBB = MI->getParent();
6459 const SIMachineFunctionInfo *Info = MF->getInfo<SIMachineFunctionInfo>();
6460
6461 int64_t Offset = 0;
6462 if (mi_match(Root.getReg(), *MRI, m_ICst(Offset)) &&
6464 Register HighBits = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6465
6466 // TODO: Should this be inside the render function? The iterator seems to
6467 // move.
6468 const int64_t MaxOffset = SIInstrInfo::getMaxMUBUFImmOffset(*Subtarget);
6469 BuildMI(*MBB, MI, MI->getDebugLoc(), TII.get(AMDGPU::V_MOV_B32_e32),
6470 HighBits)
6471 .addImm(Offset & ~MaxOffset);
6472
6473 return {{[=](MachineInstrBuilder &MIB) { // rsrc
6474 MIB.addReg(Info->getScratchRSrcReg());
6475 },
6476 [=](MachineInstrBuilder &MIB) { // vaddr
6477 MIB.addReg(HighBits);
6478 },
6479 [=](MachineInstrBuilder &MIB) { // soffset
6480 // Use constant zero for soffset and rely on eliminateFrameIndex
6481 // to choose the appropriate frame register if need be.
6482 MIB.addImm(0);
6483 },
6484 [=](MachineInstrBuilder &MIB) { // offset
6485 MIB.addImm(Offset & MaxOffset);
6486 }}};
6487 }
6488
6489 assert(Offset == 0 || Offset == -1);
6490
6491 // Try to fold a frame index directly into the MUBUF vaddr field, and any
6492 // offsets.
6493 std::optional<int> FI;
6494 Register VAddr = Root.getReg();
6495
6496 Register PtrBase;
6497 int64_t ConstOffset;
6498 std::tie(PtrBase, ConstOffset, std::ignore) =
6499 getPtrBaseWithConstantOffset(VAddr, *MRI);
6500 int MatchedFI;
6501 if (ConstOffset != 0) {
6502 if (TII.isLegalMUBUFImmOffset(ConstOffset) &&
6503 (!STI.privateMemoryResourceIsRangeChecked() ||
6504 VT->signBitIsZero(PtrBase))) {
6505 if (mi_match(PtrBase, *MRI, m_GFrameIndex(MatchedFI)))
6506 FI = MatchedFI;
6507 else
6508 VAddr = PtrBase;
6509 Offset = ConstOffset;
6510 }
6511 } else if (mi_match(Root.getReg(), *MRI, m_GFrameIndex(MatchedFI))) {
6512 FI = MatchedFI;
6513 }
6514
6515 return {{[=](MachineInstrBuilder &MIB) { // rsrc
6516 MIB.addReg(Info->getScratchRSrcReg());
6517 },
6518 [=](MachineInstrBuilder &MIB) { // vaddr
6519 if (FI)
6520 MIB.addFrameIndex(*FI);
6521 else
6522 MIB.addReg(VAddr);
6523 },
6524 [=](MachineInstrBuilder &MIB) { // soffset
6525 // Use constant zero for soffset and rely on eliminateFrameIndex
6526 // to choose the appropriate frame register if need be.
6527 MIB.addImm(0);
6528 },
6529 [=](MachineInstrBuilder &MIB) { // offset
6530 MIB.addImm(Offset);
6531 }}};
6532}
6533
6534bool AMDGPUInstructionSelector::isDSOffsetLegal(Register Base,
6535 int64_t Offset) const {
6536 if (!isUInt<16>(Offset))
6537 return false;
6538
6539 if (STI.hasUsableDSOffset() || STI.unsafeDSOffsetFoldingEnabled())
6540 return true;
6541
6542 // On Southern Islands instruction with a negative base value and an offset
6543 // don't seem to work.
6544 return VT->signBitIsZero(Base);
6545}
6546
6547bool AMDGPUInstructionSelector::isDSOffset2Legal(Register Base, int64_t Offset0,
6548 int64_t Offset1,
6549 unsigned Size) const {
6550 if (Offset0 % Size != 0 || Offset1 % Size != 0)
6551 return false;
6552 if (!isUInt<8>(Offset0 / Size) || !isUInt<8>(Offset1 / Size))
6553 return false;
6554
6555 if (STI.hasUsableDSOffset() || STI.unsafeDSOffsetFoldingEnabled())
6556 return true;
6557
6558 // On Southern Islands instruction with a negative base value and an offset
6559 // don't seem to work.
6560 return VT->signBitIsZero(Base);
6561}
6562
6563// Return whether the operation has NoUnsignedWrap property.
6564static bool isNoUnsignedWrap(MachineInstr *Addr) {
6565 return Addr->getOpcode() == TargetOpcode::G_OR ||
6566 (Addr->getOpcode() == TargetOpcode::G_PTR_ADD &&
6568}
6569
6570// Check that the base address of flat scratch load/store in the form of `base +
6571// offset` is legal to be put in SGPR/VGPR (i.e. unsigned per hardware
6572// requirement). We always treat the first operand as the base address here.
6573bool AMDGPUInstructionSelector::isFlatScratchBaseLegal(Register Addr) const {
6574 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6575
6576 if (isNoUnsignedWrap(AddrMI))
6577 return true;
6578
6579 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6580 // values.
6581 if (STI.hasSignedScratchOffsets())
6582 return true;
6583
6584 Register LHS = AddrMI->getOperand(1).getReg();
6585 Register RHS = AddrMI->getOperand(2).getReg();
6586
6587 if (AddrMI->getOpcode() == TargetOpcode::G_PTR_ADD) {
6588 std::optional<ValueAndVReg> RhsValReg =
6590 // If the immediate offset is negative and within certain range, the base
6591 // address cannot also be negative. If the base is also negative, the sum
6592 // would be either negative or much larger than the valid range of scratch
6593 // memory a thread can access.
6594 if (RhsValReg && RhsValReg->Value.getSExtValue() < 0 &&
6595 RhsValReg->Value.getSExtValue() > -0x40000000)
6596 return true;
6597 }
6598
6599 return VT->signBitIsZero(LHS);
6600}
6601
6602// Check address value in SGPR/VGPR are legal for flat scratch in the form
6603// of: SGPR + VGPR.
6604bool AMDGPUInstructionSelector::isFlatScratchBaseLegalSV(Register Addr) const {
6605 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6606
6607 if (isNoUnsignedWrap(AddrMI))
6608 return true;
6609
6610 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6611 // values.
6612 if (STI.hasSignedScratchOffsets())
6613 return true;
6614
6615 Register LHS = AddrMI->getOperand(1).getReg();
6616 Register RHS = AddrMI->getOperand(2).getReg();
6617 return VT->signBitIsZero(RHS) && VT->signBitIsZero(LHS);
6618}
6619
6620// Check address value in SGPR/VGPR are legal for flat scratch in the form
6621// of: SGPR + VGPR + Imm.
6622bool AMDGPUInstructionSelector::isFlatScratchBaseLegalSVImm(
6623 Register Addr) const {
6624 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
6625 // values.
6626 if (STI.hasSignedScratchOffsets())
6627 return true;
6628
6629 MachineInstr *AddrMI = getDefIgnoringCopies(Addr, *MRI);
6630 Register Base = AddrMI->getOperand(1).getReg();
6631 std::optional<DefinitionAndSourceRegister> BaseDef =
6633 std::optional<ValueAndVReg> RHSOffset =
6635 assert(RHSOffset);
6636
6637 // If the immediate offset is negative and within certain range, the base
6638 // address cannot also be negative. If the base is also negative, the sum
6639 // would be either negative or much larger than the valid range of scratch
6640 // memory a thread can access.
6641 if (isNoUnsignedWrap(BaseDef->MI) &&
6642 (isNoUnsignedWrap(AddrMI) ||
6643 (RHSOffset->Value.getSExtValue() < 0 &&
6644 RHSOffset->Value.getSExtValue() > -0x40000000)))
6645 return true;
6646
6647 Register LHS = BaseDef->MI->getOperand(1).getReg();
6648 Register RHS = BaseDef->MI->getOperand(2).getReg();
6649 return VT->signBitIsZero(RHS) && VT->signBitIsZero(LHS);
6650}
6651
6652bool AMDGPUInstructionSelector::isUnneededShiftMask(const MachineInstr &MI,
6653 unsigned ShAmtBits) const {
6654 assert(MI.getOpcode() == TargetOpcode::G_AND);
6655
6656 std::optional<APInt> RHS =
6657 getIConstantVRegVal(MI.getOperand(2).getReg(), *MRI);
6658 if (!RHS)
6659 return false;
6660
6661 if (RHS->countr_one() >= ShAmtBits)
6662 return true;
6663
6664 const APInt &LHSKnownZeros = VT->getKnownZeroes(MI.getOperand(1).getReg());
6665 return (LHSKnownZeros | *RHS).countr_one() >= ShAmtBits;
6666}
6667
6669AMDGPUInstructionSelector::selectMUBUFScratchOffset(
6670 MachineOperand &Root) const {
6671 Register Reg = Root.getReg();
6672 const SIMachineFunctionInfo *Info = MF->getInfo<SIMachineFunctionInfo>();
6673
6674 std::optional<DefinitionAndSourceRegister> Def =
6676 assert(Def && "this shouldn't be an optional result");
6677 Reg = Def->Reg;
6678
6679 if (Register WaveBase = getWaveAddress(Def->MI)) {
6680 return {{
6681 [=](MachineInstrBuilder &MIB) { // rsrc
6682 MIB.addReg(Info->getScratchRSrcReg());
6683 },
6684 [=](MachineInstrBuilder &MIB) { // soffset
6685 MIB.addReg(WaveBase);
6686 },
6687 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); } // offset
6688 }};
6689 }
6690
6691 int64_t Offset = 0;
6692
6693 // FIXME: Copy check is a hack
6695 if (mi_match(Reg, *MRI,
6696 m_GPtrAdd(m_Reg(BasePtr),
6698 if (!TII.isLegalMUBUFImmOffset(Offset))
6699 return {};
6700 MachineInstr *BasePtrDef = getDefIgnoringCopies(BasePtr, *MRI);
6701 Register WaveBase = getWaveAddress(BasePtrDef);
6702 if (!WaveBase)
6703 return {};
6704
6705 return {{
6706 [=](MachineInstrBuilder &MIB) { // rsrc
6707 MIB.addReg(Info->getScratchRSrcReg());
6708 },
6709 [=](MachineInstrBuilder &MIB) { // soffset
6710 MIB.addReg(WaveBase);
6711 },
6712 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); } // offset
6713 }};
6714 }
6715
6716 if (!mi_match(Root.getReg(), *MRI, m_ICst(Offset)) ||
6717 !TII.isLegalMUBUFImmOffset(Offset))
6718 return {};
6719
6720 return {{
6721 [=](MachineInstrBuilder &MIB) { // rsrc
6722 MIB.addReg(Info->getScratchRSrcReg());
6723 },
6724 [=](MachineInstrBuilder &MIB) { // soffset
6725 MIB.addImm(0);
6726 },
6727 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); } // offset
6728 }};
6729}
6730
6731std::pair<Register, unsigned>
6732AMDGPUInstructionSelector::selectDS1Addr1OffsetImpl(
6733 MachineOperand &Root) const {
6734 int64_t ConstAddr = 0;
6735
6736 Register PtrBase;
6737 int64_t Offset;
6738 std::tie(PtrBase, Offset, std::ignore) =
6739 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6740
6741 if (Offset) {
6742 if (isDSOffsetLegal(PtrBase, Offset)) {
6743 // (add n0, c0)
6744 return std::pair(PtrBase, Offset);
6745 }
6746 } else if (mi_match(Root.getReg(), *MRI, m_GSub(m_Reg(), m_Reg()))) {
6747 // TODO
6748
6749 } else if (mi_match(Root.getReg(), *MRI, m_ICst(ConstAddr))) {
6750 // TODO
6751 }
6752
6753 return std::pair(Root.getReg(), 0);
6754}
6755
6757AMDGPUInstructionSelector::selectDS1Addr1Offset(MachineOperand &Root) const {
6758 Register Reg;
6759 unsigned Offset;
6760 std::tie(Reg, Offset) = selectDS1Addr1OffsetImpl(Root);
6761 return {{
6762 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
6763 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }
6764 }};
6765}
6766
6768AMDGPUInstructionSelector::selectDS64Bit4ByteAligned(MachineOperand &Root) const {
6769 return selectDSReadWrite2(Root, 4);
6770}
6771
6773AMDGPUInstructionSelector::selectDS128Bit8ByteAligned(MachineOperand &Root) const {
6774 return selectDSReadWrite2(Root, 8);
6775}
6776
6778AMDGPUInstructionSelector::selectDSReadWrite2(MachineOperand &Root,
6779 unsigned Size) const {
6780 Register Reg;
6781 unsigned Offset;
6782 std::tie(Reg, Offset) = selectDSReadWrite2Impl(Root, Size);
6783 return {{
6784 [=](MachineInstrBuilder &MIB) { MIB.addReg(Reg); },
6785 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); },
6786 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset+1); }
6787 }};
6788}
6789
6790std::pair<Register, unsigned>
6791AMDGPUInstructionSelector::selectDSReadWrite2Impl(MachineOperand &Root,
6792 unsigned Size) const {
6793 int64_t ConstAddr = 0;
6794
6795 Register PtrBase;
6796 int64_t Offset;
6797 std::tie(PtrBase, Offset, std::ignore) =
6798 getPtrBaseWithConstantOffset(Root.getReg(), *MRI);
6799
6800 if (Offset) {
6801 int64_t OffsetValue0 = Offset;
6802 int64_t OffsetValue1 = Offset + Size;
6803 if (isDSOffset2Legal(PtrBase, OffsetValue0, OffsetValue1, Size)) {
6804 // (add n0, c0)
6805 return std::pair(PtrBase, OffsetValue0 / Size);
6806 }
6807 } else if (mi_match(Root.getReg(), *MRI, m_GSub(m_Reg(), m_Reg()))) {
6808 // TODO
6809
6810 } else if (mi_match(Root.getReg(), *MRI, m_ICst(ConstAddr))) {
6811 // TODO
6812 }
6813
6814 return std::pair(Root.getReg(), 0);
6815}
6816
6817/// If \p Root is a G_PTR_ADD with a G_CONSTANT on the right hand side, return
6818/// the base value with the constant offset, and if the offset computation is
6819/// known to be inbounds. There may be intervening copies between \p Root and
6820/// the identified constant. Returns \p Root, 0, false if this does not match
6821/// the pattern.
6822std::tuple<Register, int64_t, bool>
6823AMDGPUInstructionSelector::getPtrBaseWithConstantOffset(
6824 Register Root, const MachineRegisterInfo &MRI) const {
6825 MachineInstr *RootI = getDefIgnoringCopies(Root, MRI);
6826 if (RootI->getOpcode() != TargetOpcode::G_PTR_ADD)
6827 return {Root, 0, false};
6828
6829 MachineOperand &RHS = RootI->getOperand(2);
6830 std::optional<ValueAndVReg> MaybeOffset =
6832 if (!MaybeOffset)
6833 return {Root, 0, false};
6834 bool IsInBounds = RootI->getFlag(MachineInstr::MIFlag::InBounds);
6835 return {RootI->getOperand(1).getReg(), MaybeOffset->Value.getSExtValue(),
6836 IsInBounds};
6837}
6838
6840 MIB.addImm(0);
6841}
6842
6843/// Return a resource descriptor for use with an arbitrary 64-bit pointer. If \p
6844/// BasePtr is not valid, a null base pointer will be used.
6846 uint32_t FormatLo, uint32_t FormatHi,
6847 Register BasePtr) {
6848 Register RSrc2 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6849 Register RSrc3 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6850 Register RSrcHi = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
6851 Register RSrc = MRI.createVirtualRegister(&AMDGPU::SGPR_128RegClass);
6852
6853 B.buildInstr(AMDGPU::S_MOV_B32)
6854 .addDef(RSrc2)
6855 .addImm(FormatLo);
6856 B.buildInstr(AMDGPU::S_MOV_B32)
6857 .addDef(RSrc3)
6858 .addImm(FormatHi);
6859
6860 // Build the half of the subregister with the constants before building the
6861 // full 128-bit register. If we are building multiple resource descriptors,
6862 // this will allow CSEing of the 2-component register.
6863 B.buildInstr(AMDGPU::REG_SEQUENCE)
6864 .addDef(RSrcHi)
6865 .addReg(RSrc2)
6866 .addImm(AMDGPU::sub0)
6867 .addReg(RSrc3)
6868 .addImm(AMDGPU::sub1);
6869
6870 Register RSrcLo = BasePtr;
6871 if (!BasePtr) {
6872 RSrcLo = MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
6873 B.buildInstr(AMDGPU::S_MOV_B64)
6874 .addDef(RSrcLo)
6875 .addImm(0);
6876 }
6877
6878 B.buildInstr(AMDGPU::REG_SEQUENCE)
6879 .addDef(RSrc)
6880 .addReg(RSrcLo)
6881 .addImm(AMDGPU::sub0_sub1)
6882 .addReg(RSrcHi)
6883 .addImm(AMDGPU::sub2_sub3);
6884
6885 return RSrc;
6886}
6887
6889 const SIInstrInfo &TII, Register BasePtr) {
6890 uint64_t DefaultFormat = TII.getDefaultRsrcDataFormat();
6891
6892 // FIXME: Why are half the "default" bits ignored based on the addressing
6893 // mode?
6894 return buildRSRC(B, MRI, 0, Hi_32(DefaultFormat), BasePtr);
6895}
6896
6898 const SIInstrInfo &TII, Register BasePtr) {
6899 uint64_t DefaultFormat = TII.getDefaultRsrcDataFormat();
6900
6901 // FIXME: Why are half the "default" bits ignored based on the addressing
6902 // mode?
6903 return buildRSRC(B, MRI, -1, Hi_32(DefaultFormat), BasePtr);
6904}
6905
6906AMDGPUInstructionSelector::MUBUFAddressData
6907AMDGPUInstructionSelector::parseMUBUFAddress(Register Src) const {
6908 MUBUFAddressData Data;
6909 Data.N0 = Src;
6910
6911 Register PtrBase;
6912 int64_t Offset;
6913
6914 std::tie(PtrBase, Offset, std::ignore) =
6915 getPtrBaseWithConstantOffset(Src, *MRI);
6916 if (isUInt<32>(Offset)) {
6917 Data.N0 = PtrBase;
6918 Data.Offset = Offset;
6919 }
6920
6921 if (MachineInstr *InputAdd
6922 = getOpcodeDef(TargetOpcode::G_PTR_ADD, Data.N0, *MRI)) {
6923 Data.N2 = InputAdd->getOperand(1).getReg();
6924 Data.N3 = InputAdd->getOperand(2).getReg();
6925
6926 // FIXME: Need to fix extra SGPR->VGPRcopies inserted
6927 // FIXME: Don't know this was defined by operand 0
6928 //
6929 // TODO: Remove this when we have copy folding optimizations after
6930 // RegBankSelect.
6931 Data.N2 = getDefIgnoringCopies(Data.N2, *MRI)->getOperand(0).getReg();
6932 Data.N3 = getDefIgnoringCopies(Data.N3, *MRI)->getOperand(0).getReg();
6933 }
6934
6935 return Data;
6936}
6937
6938/// Return if the addr64 mubuf mode should be used for the given address.
6939bool AMDGPUInstructionSelector::shouldUseAddr64(MUBUFAddressData Addr) const {
6940 // (ptr_add N2, N3) -> addr64, or
6941 // (ptr_add (ptr_add N2, N3), C1) -> addr64
6942 if (Addr.N2)
6943 return true;
6944
6945 const RegisterBank *N0Bank = RBI.getRegBank(Addr.N0, *MRI, TRI);
6946 return N0Bank->getID() == AMDGPU::VGPRRegBankID;
6947}
6948
6949/// Split an immediate offset \p ImmOffset depending on whether it fits in the
6950/// immediate field. Modifies \p ImmOffset and sets \p SOffset to the variable
6951/// component.
6952void AMDGPUInstructionSelector::splitIllegalMUBUFOffset(
6953 MachineIRBuilder &B, Register &SOffset, int64_t &ImmOffset) const {
6954 if (TII.isLegalMUBUFImmOffset(ImmOffset))
6955 return;
6956
6957 // Illegal offset, store it in soffset.
6958 SOffset = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
6959 B.buildInstr(AMDGPU::S_MOV_B32)
6960 .addDef(SOffset)
6961 .addImm(ImmOffset);
6962 ImmOffset = 0;
6963}
6964
6965bool AMDGPUInstructionSelector::selectMUBUFAddr64Impl(
6966 MachineOperand &Root, Register &VAddr, Register &RSrcReg,
6967 Register &SOffset, int64_t &Offset) const {
6968 // FIXME: Predicates should stop this from reaching here.
6969 // addr64 bit was removed for volcanic islands.
6970 if (!STI.hasAddr64() || STI.useFlatForGlobal())
6971 return false;
6972
6973 MUBUFAddressData AddrData = parseMUBUFAddress(Root.getReg());
6974 if (!shouldUseAddr64(AddrData))
6975 return false;
6976
6977 Register N0 = AddrData.N0;
6978 Register N2 = AddrData.N2;
6979 Register N3 = AddrData.N3;
6980 Offset = AddrData.Offset;
6981
6982 // Base pointer for the SRD.
6983 Register SRDPtr;
6984
6985 if (N2) {
6986 if (RBI.getRegBank(N2, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
6987 assert(N3);
6988 if (RBI.getRegBank(N3, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
6989 // Both N2 and N3 are divergent. Use N0 (the result of the add) as the
6990 // addr64, and construct the default resource from a 0 address.
6991 VAddr = N0;
6992 } else {
6993 SRDPtr = N3;
6994 VAddr = N2;
6995 }
6996 } else {
6997 // N2 is not divergent.
6998 SRDPtr = N2;
6999 VAddr = N3;
7000 }
7001 } else if (RBI.getRegBank(N0, *MRI, TRI)->getID() == AMDGPU::VGPRRegBankID) {
7002 // Use the default null pointer in the resource
7003 VAddr = N0;
7004 } else {
7005 // N0 -> offset, or
7006 // (N0 + C1) -> offset
7007 SRDPtr = N0;
7008 }
7009
7010 MachineIRBuilder B(*Root.getParent());
7011 RSrcReg = buildAddr64RSrc(B, *MRI, TII, SRDPtr);
7012 splitIllegalMUBUFOffset(B, SOffset, Offset);
7013 return true;
7014}
7015
7016bool AMDGPUInstructionSelector::selectMUBUFOffsetImpl(
7017 MachineOperand &Root, Register &RSrcReg, Register &SOffset,
7018 int64_t &Offset) const {
7019
7020 // FIXME: Pattern should not reach here.
7021 if (STI.useFlatForGlobal())
7022 return false;
7023
7024 MUBUFAddressData AddrData = parseMUBUFAddress(Root.getReg());
7025 if (shouldUseAddr64(AddrData))
7026 return false;
7027
7028 // N0 -> offset, or
7029 // (N0 + C1) -> offset
7030 Register SRDPtr = AddrData.N0;
7031 Offset = AddrData.Offset;
7032
7033 // TODO: Look through extensions for 32-bit soffset.
7034 MachineIRBuilder B(*Root.getParent());
7035
7036 RSrcReg = buildOffsetSrc(B, *MRI, TII, SRDPtr);
7037 splitIllegalMUBUFOffset(B, SOffset, Offset);
7038 return true;
7039}
7040
7042AMDGPUInstructionSelector::selectMUBUFAddr64(MachineOperand &Root) const {
7043 Register VAddr;
7044 Register RSrcReg;
7045 Register SOffset;
7046 int64_t Offset = 0;
7047
7048 if (!selectMUBUFAddr64Impl(Root, VAddr, RSrcReg, SOffset, Offset))
7049 return {};
7050
7051 // FIXME: Use defaulted operands for trailing 0s and remove from the complex
7052 // pattern.
7053 return {{
7054 [=](MachineInstrBuilder &MIB) { // rsrc
7055 MIB.addReg(RSrcReg);
7056 },
7057 [=](MachineInstrBuilder &MIB) { // vaddr
7058 MIB.addReg(VAddr);
7059 },
7060 [=](MachineInstrBuilder &MIB) { // soffset
7061 if (SOffset)
7062 MIB.addReg(SOffset);
7063 else if (STI.hasRestrictedSOffset())
7064 MIB.addReg(AMDGPU::SGPR_NULL);
7065 else
7066 MIB.addImm(0);
7067 },
7068 [=](MachineInstrBuilder &MIB) { // offset
7069 MIB.addImm(Offset);
7070 },
7071 addZeroImm, // cpol
7072 addZeroImm, // tfe
7073 addZeroImm // swz
7074 }};
7075}
7076
7078AMDGPUInstructionSelector::selectMUBUFOffset(MachineOperand &Root) const {
7079 Register RSrcReg;
7080 Register SOffset;
7081 int64_t Offset = 0;
7082
7083 if (!selectMUBUFOffsetImpl(Root, RSrcReg, SOffset, Offset))
7084 return {};
7085
7086 return {{
7087 [=](MachineInstrBuilder &MIB) { // rsrc
7088 MIB.addReg(RSrcReg);
7089 },
7090 [=](MachineInstrBuilder &MIB) { // soffset
7091 if (SOffset)
7092 MIB.addReg(SOffset);
7093 else if (STI.hasRestrictedSOffset())
7094 MIB.addReg(AMDGPU::SGPR_NULL);
7095 else
7096 MIB.addImm(0);
7097 },
7098 [=](MachineInstrBuilder &MIB) { MIB.addImm(Offset); }, // offset
7099 addZeroImm, // cpol
7100 addZeroImm, // tfe
7101 addZeroImm, // swz
7102 }};
7103}
7104
7106AMDGPUInstructionSelector::selectBUFSOffset(MachineOperand &Root) const {
7107
7108 Register SOffset = Root.getReg();
7109
7110 if (STI.hasRestrictedSOffset() && mi_match(SOffset, *MRI, m_ZeroInt()))
7111 SOffset = AMDGPU::SGPR_NULL;
7112
7113 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); }}};
7114}
7115
7116/// Get an immediate that must be 32-bits, and treated as zero extended.
7117static std::optional<uint64_t>
7119 // getIConstantVRegVal sexts any values, so see if that matters.
7120 std::optional<int64_t> OffsetVal = getIConstantVRegSExtVal(Reg, MRI);
7121 if (!OffsetVal || !isInt<32>(*OffsetVal))
7122 return std::nullopt;
7123 return Lo_32(*OffsetVal);
7124}
7125
7127AMDGPUInstructionSelector::selectSMRDBufferImm(MachineOperand &Root) const {
7128 std::optional<uint64_t> OffsetVal =
7129 Root.isImm() ? Root.getImm() : getConstantZext32Val(Root.getReg(), *MRI);
7130 if (!OffsetVal)
7131 return {};
7132
7133 std::optional<int64_t> EncodedImm =
7134 AMDGPU::getSMRDEncodedOffset(STI, *OffsetVal, true);
7135 if (!EncodedImm)
7136 return {};
7137
7138 return {{ [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); } }};
7139}
7140
7142AMDGPUInstructionSelector::selectSMRDBufferImm32(MachineOperand &Root) const {
7143 assert(STI.getGeneration() == AMDGPUSubtarget::SEA_ISLANDS);
7144
7145 std::optional<uint64_t> OffsetVal = getConstantZext32Val(Root.getReg(), *MRI);
7146 if (!OffsetVal)
7147 return {};
7148
7149 std::optional<int64_t> EncodedImm =
7151 if (!EncodedImm)
7152 return {};
7153
7154 return {{ [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedImm); } }};
7155}
7156
7158AMDGPUInstructionSelector::selectSMRDBufferSgprImm(MachineOperand &Root) const {
7159 // Match the (soffset + offset) pair as a 32-bit register base and
7160 // an immediate offset.
7161 Register SOffset;
7162 unsigned Offset;
7163 std::tie(SOffset, Offset) = AMDGPU::getBaseWithConstantOffset(
7164 *MRI, Root.getReg(), VT, /*CheckNUW*/ true);
7165 if (!SOffset)
7166 return std::nullopt;
7167
7168 std::optional<int64_t> EncodedOffset =
7169 AMDGPU::getSMRDEncodedOffset(STI, Offset, /* IsBuffer */ true);
7170 if (!EncodedOffset)
7171 return std::nullopt;
7172
7173 assert(MRI->getType(SOffset).getSizeInBits() == 32);
7174 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(SOffset); },
7175 [=](MachineInstrBuilder &MIB) { MIB.addImm(*EncodedOffset); }}};
7176}
7177
7178std::pair<Register, unsigned>
7179AMDGPUInstructionSelector::selectVOP3PMadMixModsImpl(MachineOperand &Root,
7180 bool &Matched) const {
7181 Matched = false;
7182
7183 Register Src;
7184 unsigned Mods;
7185 std::tie(Src, Mods) = selectVOP3ModsImpl(Root.getReg());
7186
7187 if (mi_match(Src, *MRI, m_GFPExt(m_Reg(Src)))) {
7188 assert(MRI->getType(Src) == LLT::scalar(16));
7189
7190 // Only change Src if src modifier could be gained. In such cases new Src
7191 // could be sgpr but this does not violate constant bus restriction for
7192 // instruction that is being selected.
7193 Src = stripBitCast(Src, *MRI);
7194
7195 const auto CheckAbsNeg = [&]() {
7196 // Be careful about folding modifiers if we already have an abs. fneg is
7197 // applied last, so we don't want to apply an earlier fneg.
7198 if ((Mods & SISrcMods::ABS) == 0) {
7199 unsigned ModsTmp;
7200 std::tie(Src, ModsTmp) = selectVOP3ModsImpl(Src);
7201
7202 if ((ModsTmp & SISrcMods::NEG) != 0)
7203 Mods ^= SISrcMods::NEG;
7204
7205 if ((ModsTmp & SISrcMods::ABS) != 0)
7206 Mods |= SISrcMods::ABS;
7207 }
7208 };
7209
7210 CheckAbsNeg();
7211
7212 // op_sel/op_sel_hi decide the source type and source.
7213 // If the source's op_sel_hi is set, it indicates to do a conversion from
7214 // fp16. If the sources's op_sel is set, it picks the high half of the
7215 // source register.
7216
7217 Mods |= SISrcMods::OP_SEL_1;
7218
7219 // The instruction reads a 32-bit source and selects a half of it, so look
7220 // for the 32-bit register the 16-bit value is a half of.
7221 if (isExtractHiElt(*MRI, Src, Src)) {
7222 // Src is now the 32-bit source and op_sel picks its high half.
7223 Mods |= SISrcMods::OP_SEL_0;
7224 CheckAbsNeg();
7225 } else {
7226 // op_sel already picks the low half, so use the 32-bit source directly if
7227 // the 16-bit value is the low half of one. Otherwise Src is genuinely 16
7228 // bits wide and widenSrcIfVGPR16 widens it when the operand is
7229 // rendered.
7230 isExtractLoElt(*MRI, Src, Src);
7231 }
7232
7233 Matched = true;
7234 }
7235
7236 return {Src, Mods};
7237}
7238
7240AMDGPUInstructionSelector::selectVOP3PMadMixModsExt(
7241 MachineOperand &Root) const {
7242 Register Src;
7243 unsigned Mods;
7244 bool Matched;
7245 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7246 if (!Matched)
7247 return {};
7248
7249 return {{
7250 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7251 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
7252 }};
7253}
7254
7256AMDGPUInstructionSelector::selectVOP3PMadMixMods(MachineOperand &Root) const {
7257 Register Src;
7258 unsigned Mods;
7259 bool Matched;
7260 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7261
7262 return {{
7263 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7264 [=](MachineInstrBuilder &MIB) { MIB.addImm(Mods); } // src_mods
7265 }};
7266}
7267
7269AMDGPUInstructionSelector::selectVOP3PMadMixModsExtNeg(
7270 MachineOperand &Root) const {
7271 Register Src;
7272 unsigned Mods;
7273 bool Matched;
7274 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7275 if (!Matched)
7276 return {};
7277
7278 return {{
7279 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7280 [=](MachineInstrBuilder &MIB) {
7281 MIB.addImm(Mods ^ SISrcMods::NEG);
7282 } // src_mods
7283 }};
7284}
7285
7287AMDGPUInstructionSelector::selectVOP3PMadMixModsNeg(
7288 MachineOperand &Root) const {
7289 Register Src;
7290 unsigned Mods;
7291 bool Matched;
7292 std::tie(Src, Mods) = selectVOP3PMadMixModsImpl(Root, Matched);
7293
7294 return {{
7295 [=](MachineInstrBuilder &MIB) { MIB.addReg(widenSrcIfVGPR16(Src, MIB)); },
7296 [=](MachineInstrBuilder &MIB) {
7297 MIB.addImm(Mods ^ SISrcMods::NEG);
7298 } // src_mods
7299 }};
7300}
7301
7302bool AMDGPUInstructionSelector::selectSBarrierSignalIsfirst(
7303 MachineInstr &I, Intrinsic::ID IntrID) const {
7304 MachineBasicBlock *MBB = I.getParent();
7305 const DebugLoc &DL = I.getDebugLoc();
7306 Register CCReg = I.getOperand(0).getReg();
7307
7308 // Set SCC to true, in case the barrier instruction gets converted to a NOP.
7309 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_CMP_EQ_U32)).addImm(0).addImm(0);
7310
7311 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM))
7312 .addImm(I.getOperand(2).getImm());
7313
7314 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), CCReg).addReg(AMDGPU::SCC);
7315
7316 I.eraseFromParent();
7317 return RBI.constrainGenericRegister(CCReg, AMDGPU::SReg_32_XM0_XEXECRegClass,
7318 *MRI);
7319}
7320
7321bool AMDGPUInstructionSelector::selectSGetBarrierState(
7322 MachineInstr &I, Intrinsic::ID IntrID) const {
7323 MachineBasicBlock *MBB = I.getParent();
7324 const DebugLoc &DL = I.getDebugLoc();
7325 const MachineOperand &BarOp = I.getOperand(2);
7326 std::optional<int64_t> BarValImm =
7327 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7328
7329 if (!BarValImm) {
7330 auto CopyMIB = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
7331 .addReg(BarOp.getReg());
7332 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7333 }
7334 MachineInstrBuilder MIB;
7335 unsigned Opc = BarValImm ? AMDGPU::S_GET_BARRIER_STATE_IMM
7336 : AMDGPU::S_GET_BARRIER_STATE_M0;
7337 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7338
7339 auto DstReg = I.getOperand(0).getReg();
7340 const TargetRegisterClass *DstRC =
7341 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
7342 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
7343 return false;
7344 MIB.addDef(DstReg);
7345 if (BarValImm) {
7346 MIB.addImm(*BarValImm);
7347 }
7348 I.eraseFromParent();
7349 return true;
7350}
7351
7352unsigned getNamedBarrierOp(bool HasInlineConst, Intrinsic::ID IntrID) {
7353 if (HasInlineConst) {
7354 switch (IntrID) {
7355 default:
7356 llvm_unreachable("not a named barrier op");
7357 case Intrinsic::amdgcn_s_barrier_join:
7358 return AMDGPU::S_BARRIER_JOIN_IMM;
7359 case Intrinsic::amdgcn_s_wakeup_barrier:
7360 return AMDGPU::S_WAKEUP_BARRIER_IMM;
7361 case Intrinsic::amdgcn_s_get_named_barrier_state:
7362 return AMDGPU::S_GET_BARRIER_STATE_IMM;
7363 };
7364 } else {
7365 switch (IntrID) {
7366 default:
7367 llvm_unreachable("not a named barrier op");
7368 case Intrinsic::amdgcn_s_barrier_join:
7369 return AMDGPU::S_BARRIER_JOIN_M0;
7370 case Intrinsic::amdgcn_s_wakeup_barrier:
7371 return AMDGPU::S_WAKEUP_BARRIER_M0;
7372 case Intrinsic::amdgcn_s_get_named_barrier_state:
7373 return AMDGPU::S_GET_BARRIER_STATE_M0;
7374 };
7375 }
7376}
7377
7378bool AMDGPUInstructionSelector::selectNamedBarrierInit(
7379 MachineInstr &I, Intrinsic::ID IntrID) const {
7380 MachineBasicBlock *MBB = I.getParent();
7381 const DebugLoc &DL = I.getDebugLoc();
7382 const MachineOperand &BarOp = I.getOperand(1);
7383 const MachineOperand &CntOp = I.getOperand(2);
7384
7385 // A member count of 0 means "keep existing member count". That plus a known
7386 // constant value for the barrier ID lets us use the immarg form.
7387 if (IntrID == Intrinsic::amdgcn_s_barrier_signal_var) {
7388 std::optional<int64_t> CntImm =
7389 getIConstantVRegSExtVal(CntOp.getReg(), *MRI);
7390 if (CntImm && *CntImm == 0) {
7391 std::optional<int64_t> BarValImm =
7392 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7393 if (BarValImm) {
7394 uint32_t BarID = *BarValImm & 0x3F;
7395 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_BARRIER_SIGNAL_IMM))
7396 .addImm(BarID);
7397 I.eraseFromParent();
7398 return true;
7399 }
7400 }
7401 }
7402
7403 // BarID = BarOp & 0x3F
7404 Register TmpReg1 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7405 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg1)
7406 .add(BarOp)
7407 .addImm(0x3F)
7408 .setOperandDead(3); // Dead scc
7409
7410 // MO = ((CntOp & 0x3F) << shAmt) | BarID
7411 Register TmpReg2 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7412 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg2)
7413 .add(CntOp)
7414 .addImm(0x3F)
7415 .setOperandDead(3); // Dead scc
7416
7417 Register TmpReg3 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7418 constexpr unsigned ShAmt = 16;
7419 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_LSHL_B32), TmpReg3)
7420 .addReg(TmpReg2)
7421 .addImm(ShAmt)
7422 .setOperandDead(3); // Dead scc
7423
7424 Register TmpReg4 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7425 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_OR_B32), TmpReg4)
7426 .addReg(TmpReg1)
7427 .addReg(TmpReg3)
7428 .setOperandDead(3); // Dead scc;
7429
7430 auto CopyMIB =
7431 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0).addReg(TmpReg4);
7432 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7433
7434 unsigned Opc = IntrID == Intrinsic::amdgcn_s_barrier_init
7435 ? AMDGPU::S_BARRIER_INIT_M0
7436 : AMDGPU::S_BARRIER_SIGNAL_M0;
7437 MachineInstrBuilder MIB;
7438 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7439
7440 I.eraseFromParent();
7441 return true;
7442}
7443
7444bool AMDGPUInstructionSelector::selectNamedBarrierInst(
7445 MachineInstr &I, Intrinsic::ID IntrID) const {
7446 MachineBasicBlock *MBB = I.getParent();
7447 const DebugLoc &DL = I.getDebugLoc();
7448 MachineOperand BarOp = IntrID == Intrinsic::amdgcn_s_get_named_barrier_state
7449 ? I.getOperand(2)
7450 : I.getOperand(1);
7451 std::optional<int64_t> BarValImm =
7452 getIConstantVRegSExtVal(BarOp.getReg(), *MRI);
7453
7454 if (!BarValImm) {
7455 // BarID = BarOp & 0x3F
7456 Register TmpReg1 = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
7457 BuildMI(*MBB, &I, DL, TII.get(AMDGPU::S_AND_B32), TmpReg1)
7458 .addReg(BarOp.getReg())
7459 .addImm(0x3F)
7460 .setOperandDead(3); // Dead scc;
7461
7462 auto CopyMIB = BuildMI(*MBB, &I, DL, TII.get(AMDGPU::COPY), AMDGPU::M0)
7463 .addReg(TmpReg1);
7464 constrainSelectedInstRegOperands(*CopyMIB, TII, TRI, RBI);
7465 }
7466
7467 MachineInstrBuilder MIB;
7468 unsigned Opc = getNamedBarrierOp(BarValImm.has_value(), IntrID);
7469 MIB = BuildMI(*MBB, &I, DL, TII.get(Opc));
7470
7471 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state) {
7472 auto DstReg = I.getOperand(0).getReg();
7473 const TargetRegisterClass *DstRC =
7474 TRI.getConstrainedRegClassForReg(DstReg, *MRI);
7475 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, *MRI))
7476 return false;
7477 MIB.addDef(DstReg);
7478 }
7479
7480 if (BarValImm) {
7481 uint32_t BarId = *BarValImm & 0x3F;
7482 MIB.addImm(BarId);
7483 }
7484
7485 I.eraseFromParent();
7486 return true;
7487}
7488
7489void AMDGPUInstructionSelector::renderTruncImm32(MachineInstrBuilder &MIB,
7490 const MachineInstr &MI,
7491 int OpIdx) const {
7492 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
7493 "Expected G_CONSTANT");
7494 MIB.addImm(MI.getOperand(1).getCImm()->getSExtValue());
7495}
7496
7497void AMDGPUInstructionSelector::renderNegateImm(MachineInstrBuilder &MIB,
7498 const MachineInstr &MI,
7499 int OpIdx) const {
7500 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
7501 "Expected G_CONSTANT");
7502 MIB.addImm(-MI.getOperand(1).getCImm()->getSExtValue());
7503}
7504
7505void AMDGPUInstructionSelector::renderBitcastFPImm(MachineInstrBuilder &MIB,
7506 const MachineInstr &MI,
7507 int OpIdx) const {
7508 const MachineOperand &Op = MI.getOperand(1);
7509 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1);
7510 MIB.addImm(Op.getFPImm()->getValueAPF().bitcastToAPInt().getZExtValue());
7511}
7512
7513void AMDGPUInstructionSelector::renderCountTrailingOnesImm(
7514 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7515 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
7516 "Expected G_CONSTANT");
7517 MIB.addImm(MI.getOperand(1).getCImm()->getValue().countTrailingOnes());
7518}
7519
7520/// This only really exists to satisfy DAG type checking machinery, so is a
7521/// no-op here.
7522void AMDGPUInstructionSelector::renderTruncTImm(MachineInstrBuilder &MIB,
7523 const MachineInstr &MI,
7524 int OpIdx) const {
7525 const MachineOperand &Op = MI.getOperand(OpIdx);
7526 int64_t Imm;
7527 if (Op.isReg() && mi_match(Op.getReg(), *MRI, m_ICst(Imm)))
7528 MIB.addImm(Imm);
7529 else
7530 MIB.addImm(Op.getImm());
7531}
7532
7533void AMDGPUInstructionSelector::renderZextBoolTImm(MachineInstrBuilder &MIB,
7534 const MachineInstr &MI,
7535 int OpIdx) const {
7536 MIB.addImm(MI.getOperand(OpIdx).getImm() != 0);
7537}
7538
7539void AMDGPUInstructionSelector::renderOpSelTImm(MachineInstrBuilder &MIB,
7540 const MachineInstr &MI,
7541 int OpIdx) const {
7542 assert(OpIdx >= 0 && "expected to match an immediate operand");
7543 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7544}
7545
7546void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_0_0(
7547 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7548 assert(OpIdx >= 0 && "expected to match an immediate operand");
7549 MIB.addImm(
7550 (MI.getOperand(OpIdx).getImm() & 0x1) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7551}
7552
7553void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_0_1(
7554 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7555 assert(OpIdx >= 0 && "expected to match an immediate operand");
7556 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x1)
7558 : (int64_t)SISrcMods::DST_OP_SEL);
7559}
7560
7561void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_1_0(
7562 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7563 assert(OpIdx >= 0 && "expected to match an immediate operand");
7564 MIB.addImm(
7565 (MI.getOperand(OpIdx).getImm() & 0x2) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7566}
7567
7568void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_1_1(
7569 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7570 assert(OpIdx >= 0 && "expected to match an immediate operand");
7571 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x2)
7572 ? (int64_t)(SISrcMods::OP_SEL_0)
7573 : 0);
7574}
7575
7576void AMDGPUInstructionSelector::renderDstSelToOpSelXForm(
7577 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7578 assert(OpIdx >= 0 && "expected to match an immediate operand");
7579 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)(SISrcMods::DST_OP_SEL)
7580 : 0);
7581}
7582
7583void AMDGPUInstructionSelector::renderSrcSelToOpSelXForm(
7584 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7585 assert(OpIdx >= 0 && "expected to match an immediate operand");
7586 MIB.addImm(MI.getOperand(OpIdx).getImm() ? (int64_t)(SISrcMods::OP_SEL_0)
7587 : 0);
7588}
7589
7590void AMDGPUInstructionSelector::renderSrcAndDstSelToOpSelXForm_2_0(
7591 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7592 assert(OpIdx >= 0 && "expected to match an immediate operand");
7593 MIB.addImm(
7594 (MI.getOperand(OpIdx).getImm() & 0x1) ? (int64_t)SISrcMods::OP_SEL_0 : 0);
7595}
7596
7597void AMDGPUInstructionSelector::renderDstSelToOpSel3XFormXForm(
7598 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7599 assert(OpIdx >= 0 && "expected to match an immediate operand");
7600 MIB.addImm((MI.getOperand(OpIdx).getImm() & 0x2)
7601 ? (int64_t)SISrcMods::DST_OP_SEL
7602 : 0);
7603}
7604
7605void AMDGPUInstructionSelector::renderExtractCPol(MachineInstrBuilder &MIB,
7606 const MachineInstr &MI,
7607 int OpIdx) const {
7608 assert(OpIdx >= 0 && "expected to match an immediate operand");
7609 MIB.addImm(MI.getOperand(OpIdx).getImm() &
7612}
7613
7614void AMDGPUInstructionSelector::renderExtractSWZ(MachineInstrBuilder &MIB,
7615 const MachineInstr &MI,
7616 int OpIdx) const {
7617 assert(OpIdx >= 0 && "expected to match an immediate operand");
7618 const bool Swizzle = MI.getOperand(OpIdx).getImm() &
7621 MIB.addImm(Swizzle);
7622}
7623
7624void AMDGPUInstructionSelector::renderExtractCpolSetGLC(
7625 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7626 assert(OpIdx >= 0 && "expected to match an immediate operand");
7627 const uint32_t Cpol = MI.getOperand(OpIdx).getImm() &
7630 MIB.addImm(Cpol | AMDGPU::CPol::GLC);
7631}
7632
7633void AMDGPUInstructionSelector::renderFPPow2ToExponent(MachineInstrBuilder &MIB,
7634 const MachineInstr &MI,
7635 int OpIdx) const {
7636 const APFloat &APF = MI.getOperand(1).getFPImm()->getValueAPF();
7637 int ExpVal = APF.getExactLog2Abs();
7638 assert(ExpVal != INT_MIN);
7639 MIB.addImm(ExpVal);
7640}
7641
7642void AMDGPUInstructionSelector::renderRoundMode(MachineInstrBuilder &MIB,
7643 const MachineInstr &MI,
7644 int OpIdx) const {
7645 // "round.towardzero" -> TowardZero 0 -> FP_ROUND_ROUND_TO_ZERO 3
7646 // "round.tonearest" -> NearestTiesToEven 1 -> FP_ROUND_ROUND_TO_NEAREST 0
7647 // "round.upward" -> TowardPositive 2 -> FP_ROUND_ROUND_TO_INF 1
7648 // "round.downward -> TowardNegative 3 -> FP_ROUND_ROUND_TO_NEGINF 2
7649 MIB.addImm((MI.getOperand(OpIdx).getImm() + 3) % 4);
7650}
7651
7652void AMDGPUInstructionSelector::renderVOP3PModsNeg(MachineInstrBuilder &MIB,
7653 const MachineInstr &MI,
7654 int OpIdx) const {
7655 unsigned Mods = SISrcMods::OP_SEL_1;
7656 if (MI.getOperand(OpIdx).getImm())
7657 Mods ^= SISrcMods::NEG;
7658 MIB.addImm((int64_t)Mods);
7659}
7660
7661void AMDGPUInstructionSelector::renderVOP3PModsNegs(MachineInstrBuilder &MIB,
7662 const MachineInstr &MI,
7663 int OpIdx) const {
7664 unsigned Mods = SISrcMods::OP_SEL_1;
7665 if (MI.getOperand(OpIdx).getImm())
7667 MIB.addImm((int64_t)Mods);
7668}
7669
7670void AMDGPUInstructionSelector::renderVOP3PModsNegAbs(MachineInstrBuilder &MIB,
7671 const MachineInstr &MI,
7672 int OpIdx) const {
7673 unsigned Val = MI.getOperand(OpIdx).getImm();
7674 unsigned Mods = SISrcMods::OP_SEL_1; // default: none
7675 if (Val == 1) // neg
7676 Mods ^= SISrcMods::NEG;
7677 if (Val == 2) // abs
7678 Mods ^= SISrcMods::ABS;
7679 if (Val == 3) // neg and abs
7680 Mods ^= (SISrcMods::NEG | SISrcMods::ABS);
7681 MIB.addImm((int64_t)Mods);
7682}
7683
7684void AMDGPUInstructionSelector::renderPrefetchLoc(MachineInstrBuilder &MIB,
7685 const MachineInstr &MI,
7686 int OpIdx) const {
7687 uint32_t V = MI.getOperand(2).getImm();
7690 if (!Subtarget->hasSafeCUPrefetch())
7691 V = std::max(V, (uint32_t)AMDGPU::CPol::SCOPE_SE); // CU scope is unsafe
7692 MIB.addImm(V);
7693}
7694
7695/// Convert from 2-bit value to enum values used for op_sel* source modifiers.
7696void AMDGPUInstructionSelector::renderScaledMAIIntrinsicOperand(
7697 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
7698 unsigned Val = MI.getOperand(OpIdx).getImm();
7699 unsigned New = 0;
7700 if (Val & 0x1)
7702 if (Val & 0x2)
7704 MIB.addImm(New);
7705}
7706
7707bool AMDGPUInstructionSelector::isInlineImmediate(const APInt &Imm) const {
7708 return TII.isInlineConstant(Imm);
7709}
7710
7711bool AMDGPUInstructionSelector::isInlineImmediate(const APFloat &Imm) const {
7712 return TII.isInlineConstant(Imm);
7713}
MachineInstrBuilder MachineInstrBuilder & DefMI
static unsigned getIntrinsicID(const SDNode *N)
#define GET_GLOBALISEL_PREDICATES_INIT
#define GET_GLOBALISEL_TEMPORARIES_INIT
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isShlHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is shift left with half bits, such as reg0:2n =G_SHL reg1:2n, CONST(n)
static bool isNoUnsignedWrap(MachineInstr *Addr)
static Register buildOffsetSrc(MachineIRBuilder &B, MachineRegisterInfo &MRI, const SIInstrInfo &TII, Register BasePtr)
unsigned getNamedBarrierOp(bool HasInlineConst, Intrinsic::ID IntrID)
static Register getLegalRegBank(Register NewReg, Register RootReg, MachineInstr &Use, const AMDGPURegisterBankInfo &RBI, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const SIInstrInfo &TII)
static bool checkRB(Register Reg, unsigned int RBNo, const AMDGPURegisterBankInfo &RBI, const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI)
static unsigned updateMods(SrcStatus HiStat, SrcStatus LoStat, unsigned Mods)
static bool isTruncHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is truncating to half, such as reg0:n = G_TRUNC reg1:2n
static Register getWaveAddress(const MachineInstr *Def)
static bool isExtractHiElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static bool shouldUseAndMask(unsigned Size, unsigned &Mask)
static std::pair< unsigned, uint8_t > BitOp3_Op(Register R, SmallVectorImpl< Register > &Src, const MachineRegisterInfo &MRI)
static TypeClass isVectorOfTwoOrScalar(Register Reg, const MachineRegisterInfo &MRI)
static bool isLaneMaskFromSameBlock(Register Reg, MachineRegisterInfo &MRI, MachineBasicBlock *MBB)
static bool isExtractLoElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE, bool &IsTexFail)
static void addZeroImm(MachineInstrBuilder &MIB)
static unsigned gwsIntrinToOpcode(unsigned IntrID)
static bool isConstant(const MachineInstr &MI)
static bool isSameBitWidth(Register Reg1, Register Reg2, const MachineRegisterInfo &MRI)
static Register buildRegSequence(SmallVectorImpl< Register > &Elts, MachineInstr *InsertPt, MachineRegisterInfo &MRI)
static Register buildRSRC(MachineIRBuilder &B, MachineRegisterInfo &MRI, uint32_t FormatLo, uint32_t FormatHi, Register BasePtr)
Return a resource descriptor for use with an arbitrary 64-bit pointer.
static bool isAsyncLDSDMA(Intrinsic::ID Intr)
static std::pair< Register, unsigned > computeIndirectRegIndex(MachineRegisterInfo &MRI, const SIRegisterInfo &TRI, const TargetRegisterClass *SuperRC, Register IdxReg, unsigned EltSize, GISelValueTracking &ValueTracking)
Return the register to use for the index value, and the subregister to use for the indirectly accesse...
static unsigned getLogicalBitOpcode(unsigned Opc, bool Is64)
static std::pair< Register, SrcStatus > getLastSameOrNeg(Register Reg, const MachineRegisterInfo &MRI, SearchOptions SO, int MaxDepth=3)
static Register stripCopy(Register Reg, MachineRegisterInfo &MRI)
static std::optional< std::pair< Register, SrcStatus > > calcNextStatus(std::pair< Register, SrcStatus > Curr, const MachineRegisterInfo &MRI)
static Register stripBitCast(Register Reg, MachineRegisterInfo &MRI)
static std::optional< uint64_t > getConstantZext32Val(Register Reg, const MachineRegisterInfo &MRI)
Get an immediate that must be 32-bits, and treated as zero extended.
static bool isValidToPack(SrcStatus HiStat, SrcStatus LoStat, Register NewReg, Register RootReg, const SIInstrInfo &TII, const MachineRegisterInfo &MRI)
static int getV_CMPOpcode(CmpInst::Predicate P, unsigned Size, const GCNSubtarget &ST)
static SmallVector< std::pair< Register, SrcStatus > > getSrcStats(Register Reg, const MachineRegisterInfo &MRI, SearchOptions SO, int MaxDepth=3)
static bool isUnmergeHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test function, if the MI is reg0:n, reg1:n = G_UNMERGE_VALUES reg2:2n
static SrcStatus getNegStatus(Register Reg, SrcStatus S, const MachineRegisterInfo &MRI)
static bool isVCmpResult(Register Reg, MachineRegisterInfo &MRI)
static Register buildAddr64RSrc(MachineIRBuilder &B, MachineRegisterInfo &MRI, const SIInstrInfo &TII, Register BasePtr)
static bool isLshrHalf(const MachineInstr *MI, const MachineRegisterInfo &MRI)
Test if the MI is logic shift right with half bits, such as reg0:2n =G_LSHR reg1:2n,...
static void selectWMMAModsNegAbs(unsigned ModOpcode, unsigned &Mods, SmallVectorImpl< Register > &Elts, Register &Src, MachineInstr *InsertPt, MachineRegisterInfo &MRI)
This file declares the targeting of the InstructionSelector class for AMDGPU.
constexpr LLT S1
constexpr LLT S32
AMDGPU Register Bank Select
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool isAllZeros(StringRef Arr)
Return true if the array is empty or all zeros.
dxil translate DXIL Translate Metadata
Provides analysis for querying information about KnownBits during GISel passes.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineIRBuilder class.
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define P(N)
static std::vector< std::pair< int, unsigned > > Swizzle(std::vector< std::pair< int, unsigned > > Src, R600InstrInfo::BankSwizzle Swz)
#define LLVM_DEBUG(...)
Definition Debug.h:119
Value * RHS
Value * LHS
This is used to control valid status that current MI supports.
bool checkOptions(SrcStatus Stat) const
SearchOptions(Register Reg, const MachineRegisterInfo &MRI)
AMDGPUInstructionSelector(const GCNSubtarget &STI, const AMDGPURegisterBankInfo &RBI)
static const char * getName()
bool select(MachineInstr &I) override
Select the (possibly generic) instruction I to only use target-specific opcodes.
void setupMF(MachineFunction &MF, GISelValueTracking *VT, CodeGenCoverage *CoverageInfo, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) override
Setup per-MF executor state.
LLVM_READONLY int getExactLog2Abs() const
Definition APFloat.h:1639
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
DILocation * get() const
Get the underlying DILocation.
Definition DebugLoc.h:220
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
std::optional< SmallVector< std::function< void(MachineInstrBuilder &)>, 4 > > ComplexRendererFns
virtual void setupMF(MachineFunction &mf, GISelValueTracking *vt, CodeGenCoverage *covinfo=nullptr, ProfileSummaryInfo *psi=nullptr, BlockFrequencyInfo *bfi=nullptr)
Setup per-MF executor state.
Register getSourceReg(unsigned I) const
Returns the I'th source register.
unsigned getNumSources() const
Returns the number of source registers.
Represents a G_UNMERGE_VALUES.
unsigned getNumDefs() const
Returns the number of def registers.
Register getSourceReg() const
Get the unmerge source register.
constexpr bool isScalar() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isValid() const
constexpr bool isVector() const
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr unsigned getAddressSpace() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
bool hasValue() const
TypeSize getValue() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
unsigned getID() const
getID() - Return the register class ID number.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void setReturnAddressIsTaken(bool s)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Helper class to build MachineInstr.
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool getFlag(MIFlag Flag) const
Return whether an MI flag is set.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
LocationSize getSize() const
Return the size in bytes of the memory reference.
unsigned getAddrSpace() const
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
const MachinePointerInfo & getPointerInfo() const
Flags getFlags() const
Return the raw flags of the source value,.
const Value * getValue() const
Return the base address of the memory access.
Align getBaseAlign() const
Return the minimum known alignment in bytes of the base address, without the offset.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
const ConstantInt * getCImm() const
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
static MachineOperand CreateImm(int64_t Val)
bool isEarlyClobber() const
Register getReg() const
getReg - Returns the register number.
bool isInternalRead() const
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
const RegisterBank * getRegBankOrNull(Register Reg) const
Return the register bank of Reg, or null if Reg has not been assigned a register bank or has been ass...
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
Definition Type.cpp:887
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Analysis providing profile information.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
static bool isGenericOpcode(unsigned Opc)
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
LLVM_READONLY const MIMGG16MappingInfo * getMIMGG16MappingInfo(unsigned G)
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool isGFX12Plus(const MCSubtargetInfo &STI)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords, bool IndexedRsrc, bool IndexedSamp)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
LLVM_READONLY int32_t getGlobalSaddrOp(uint32_t Opcode)
bool isGFX13Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX10Plus(const MCSubtargetInfo &STI)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfo(unsigned DimEnum)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
Intrinsic::ID getIntrinsicID(const MachineInstr &I)
Return the intrinsic ID for opcodes with the G_AMDGPU_INTRIN_ prefix.
std::pair< Register, unsigned > getBaseWithConstantOffset(MachineRegisterInfo &MRI, Register Reg, GISelValueTracking *ValueTracking=nullptr, bool CheckNUW=false)
Returns base register and constant offset.
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
IndexMode
ARM Index Modes.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
operand_type_match m_Reg()
SpecificConstantMatch m_SpecificICst(const APInt &RequestedValue)
Matches a constant equal to RequestedValue.
GInstrBind< GBuildVector > m_GBuildVector(GBuildVector *&Inst)
GCstAndRegMatch m_GCst(std::optional< ValueAndVReg > &ValReg)
UnaryOp_match< SrcTy, TargetOpcode::COPY > m_Copy(SrcTy &&Src)
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
BinaryOp_match< LHS, RHS, TargetOpcode::G_XOR, true > m_GXor(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_SEXT > m_GSExt(const SrcTy &Src)
UnaryOp_match< SrcTy, TargetOpcode::G_FPEXT > m_GFPExt(const SrcTy &Src)
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
ConstantMatch< APInt > m_ICst(APInt &Cst)
SpecificConstantMatch m_AllOnesInt()
BinaryOp_match< LHS, RHS, TargetOpcode::G_OR, true > m_GOr(const LHS &L, const RHS &R)
ICstOrSplatMatch< APInt > m_ICstOrSplat(APInt &Cst)
ImplicitDefMatch m_GImplicitDef()
GInstrBind< GConcatVectors > m_GConcatVectors(GConcatVectors *&Inst)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
GInstrBind< GUnmerge > m_GUnmerge(GUnmerge *&Inst)
Instruction binders for ops with no operand-form matcher (constant-immediate or variadic-source ops).
BinaryOp_match< LHS, RHS, TargetOpcode::G_SUB > m_GSub(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_ASHR, false > m_GAShr(const LHS &L, const RHS &R)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
SpecificRegisterMatch m_SpecificReg(Register RequestedReg)
Matches a register only if it is equal to RequestedReg.
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
GFrameIndexMatch m_GFrameIndex(int &FI)
Or< Preds... > m_any_of(Preds &&... preds)
BinaryOp_match< LHS, RHS, TargetOpcode::G_AND, true > m_GAnd(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_BITCAST > m_GBitcast(const SrcTy &Src)
bind_ty< MachineInstr * > m_MInstr(MachineInstr *&MI)
UnaryOp_match< SrcTy, TargetOpcode::G_FNEG > m_GFNeg(const SrcTy &Src)
GFCstOrSplatGFCstMatch m_GFCstOrSplat(std::optional< FPValueAndVReg > &FPValReg)
UnaryOp_match< SrcTy, TargetOpcode::G_FABS > m_GFabs(const SrcTy &Src)
BinaryOp_match< LHS, RHS, TargetOpcode::G_LSHR, false > m_GLShr(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_ANYEXT > m_GAnyExt(const SrcTy &Src)
ShuffleVectorMatch< Src1Ty, Src2Ty > m_GShuffleVector(const Src1Ty &Src1, const Src2Ty &Src2, ArrayRef< int > &Mask)
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, TargetOpcode::G_MUL, true > m_GMul(const LHS &L, const RHS &R)
UnaryOp_match< SrcTy, TargetOpcode::G_TRUNC > m_GTrunc(const SrcTy &Src)
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Register getFunctionLiveInPhysReg(MachineFunction &MF, const TargetInstrInfo &TII, MCRegister PhysReg, const TargetRegisterClass &RC, const DebugLoc &DL, LLT RegTy=LLT())
Return a virtual register corresponding to the incoming argument register PhysReg.
Definition Utils.cpp:848
@ Offset
Definition DWP.cpp:577
LLVM_ABI bool isBuildVectorAllZeros(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndef=false)
Return true if the specified instruction is a G_BUILD_VECTOR or G_BUILD_VECTOR_TRUNC where all of the...
Definition Utils.cpp:1434
LLVM_ABI Register constrainOperandRegClass(const MachineFunction &MF, const TargetRegisterInfo &TRI, MachineRegisterInfo &MRI, const TargetInstrInfo &TII, const RegisterBankInfo &RBI, MachineInstr &InsertPt, const TargetRegisterClass &RegClass, MachineOperand &RegMO)
Constrain the Register operand OpIdx, so that it is now constrained to the TargetRegisterClass passed...
Definition Utils.cpp:60
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
Definition Utils.cpp:656
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
LLVM_ABI const ConstantFP * getConstantFPVRegVal(Register VReg, const MachineRegisterInfo &MRI)
Definition Utils.cpp:464
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI std::optional< APInt > getIConstantVRegVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT, return the corresponding value.
Definition Utils.cpp:297
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI void constrainSelectedInstRegOperands(MachineInstr &I, const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
Mutate the newly-selected instruction I to constrain its (possibly generic) virtual register operands...
Definition Utils.cpp:159
@ Load
The value being inserted comes from a load (InsertElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
Definition Utils.cpp:497
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI std::optional< int64_t > getIConstantVRegSExtVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT fits in int64_t returns it.
Definition Utils.cpp:317
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI std::optional< ValueAndVReg > getAnyConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true, bool LookThroughAnyExt=false)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT or G_FCONST...
Definition Utils.cpp:442
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
LLVM_ABI std::optional< DefinitionAndSourceRegister > getDefSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, and underlying value Register folding away any copies.
Definition Utils.cpp:472
LLVM_ABI Register getSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the source register for Reg, folding away any trivial copies.
Definition Utils.cpp:504
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
constexpr RegState getUndefRegState(bool B)
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
Definition KnownBits.h:315
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
int64_t Offset
Offset - This is an offset from the base Value*.
PointerUnion< const Value *, const PseudoSourceValue * > V
This is the IR pointer value for the access, or it is null if unknown.