LLVM 24.0.0git
AMDGPUISelDAGToDAG.cpp
Go to the documentation of this file.
1//===-- AMDGPUISelDAGToDAG.cpp - A dag to dag inst selector for AMDGPU ----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8//
9/// \file
10/// Defines an instruction selector for the AMDGPU target.
11//
12//===----------------------------------------------------------------------===//
13
14#include "AMDGPUISelDAGToDAG.h"
15#include "AMDGPU.h"
16#include "AMDGPUInstrInfo.h"
17#include "AMDGPUSubtarget.h"
19#include "R600RegisterInfo.h"
20#include "SIISelLowering.h"
27#include "llvm/IR/IntrinsicsAMDGPU.h"
30
31#ifdef EXPENSIVE_CHECKS
33#include "llvm/IR/Dominators.h"
34#endif
35
36#define DEBUG_TYPE "amdgpu-isel"
37
38using namespace llvm;
39
40//===----------------------------------------------------------------------===//
41// Instruction Selector Implementation
42//===----------------------------------------------------------------------===//
43
44namespace {
45static SDValue stripBitcast(SDValue Val) {
46 return Val.getOpcode() == ISD::BITCAST ? Val.getOperand(0) : Val;
47}
48
49// Figure out if this is really an extract of the high 16-bits of a dword.
50static bool isExtractHiElt(SDValue In, SDValue &Out) {
51 In = stripBitcast(In);
52
53 if (In.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
54 if (ConstantSDNode *Idx = dyn_cast<ConstantSDNode>(In.getOperand(1))) {
55 if (!Idx->isOne())
56 return false;
57 Out = In.getOperand(0);
58 return true;
59 }
60 }
61
62 if (In.getOpcode() != ISD::TRUNCATE)
63 return false;
64
65 SDValue Srl = In.getOperand(0);
66 if (Srl.getOpcode() == ISD::SRL) {
67 if (ConstantSDNode *ShiftAmt = dyn_cast<ConstantSDNode>(Srl.getOperand(1))) {
68 if (ShiftAmt->getZExtValue() == 16) {
69 Out = stripBitcast(Srl.getOperand(0));
70 return true;
71 }
72 }
73 }
74
75 return false;
76}
77
78static SDValue createVOP3PSrc32FromLo16(SDValue Lo, SDValue Src,
79 llvm::SelectionDAG *CurDAG,
80 const GCNSubtarget *Subtarget) {
81 if (!Subtarget->useRealTrue16Insts()) {
82 return Lo;
83 }
84
85 SDValue NewSrc;
86 SDLoc SL(Lo);
87
88 if (Lo->isDivergent()) {
89 SDValue Undef = SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,
90 SL, Lo.getValueType()),
91 0);
92 const SDValue Ops[] = {
93 CurDAG->getTargetConstant(AMDGPU::VGPR_32RegClassID, SL, MVT::i32), Lo,
94 CurDAG->getTargetConstant(AMDGPU::lo16, SL, MVT::i16), Undef,
95 CurDAG->getTargetConstant(AMDGPU::hi16, SL, MVT::i16)};
96
97 NewSrc = SDValue(CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, SL,
98 Src.getValueType(), Ops),
99 0);
100 } else {
101 // the S_MOV is needed since the Lo could still be a VGPR16.
102 // With S_MOV, isel insert a "sgpr32 = copy vgpr16" and we reply on
103 // the fixvgpr2sgprcopy pass to legalize it
104 NewSrc = SDValue(
105 CurDAG->getMachineNode(AMDGPU::S_MOV_B32, SL, Src.getValueType(), Lo),
106 0);
107 }
108
109 return NewSrc;
110}
111
112// Look through operations that obscure just looking at the low 16-bits of the
113// same register.
114static SDValue stripExtractLoElt(SDValue In) {
115 if (In.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
116 SDValue Idx = In.getOperand(1);
117 if (isNullConstant(Idx) && In.getValueSizeInBits() <= 32)
118 return In.getOperand(0);
119 }
120
121 if (In.getOpcode() == ISD::TRUNCATE) {
122 SDValue Src = In.getOperand(0);
123 if (Src.getValueType().getSizeInBits() == 32)
124 return stripBitcast(Src);
125 }
126
127 return In;
128}
129
130static SDValue emitRegSequence(llvm::SelectionDAG &CurDAG, unsigned DstRegClass,
131 EVT DstTy, ArrayRef<SDValue> Elts,
132 ArrayRef<unsigned> SubRegClass,
133 const SDLoc &DL) {
134 assert(Elts.size() == SubRegClass.size() && "array size mismatch");
135 unsigned NumElts = Elts.size();
136 SmallVector<SDValue, 17> Ops(2 * NumElts + 1);
137 Ops[0] = (CurDAG.getTargetConstant(DstRegClass, DL, MVT::i32));
138 for (unsigned i = 0; i < NumElts; ++i) {
139 Ops[2 * i + 1] = Elts[i];
140 Ops[2 * i + 2] = CurDAG.getTargetConstant(SubRegClass[i], DL, MVT::i32);
141 }
142 return SDValue(
143 CurDAG.getMachineNode(TargetOpcode::REG_SEQUENCE, DL, DstTy, Ops), 0);
144}
145
146} // end anonymous namespace
147
149 "AMDGPU DAG->DAG Pattern Instruction Selection", false,
150 false)
151INITIALIZE_PASS_DEPENDENCY(AMDGPUPerfHintAnalysisLegacy)
153#ifdef EXPENSIVE_CHECKS
156#endif
158 "AMDGPU DAG->DAG Pattern Instruction Selection", false,
159 false)
160
161/// This pass converts a legalized DAG into a AMDGPU-specific
162// DAG, ready for instruction scheduling.
164 CodeGenOptLevel OptLevel) {
165 return new AMDGPUDAGToDAGISelLegacy(TM, OptLevel);
166}
167
171
173 Subtarget = &MF.getSubtarget<GCNSubtarget>();
174 Subtarget->checkSubtargetFeatures(MF.getFunction());
175 Mode = SIModeRegisterDefaults(MF.getFunction(), *Subtarget);
177}
178
179bool AMDGPUDAGToDAGISel::fp16SrcZerosHighBits(unsigned Opc) const {
180 // XXX - only need to list legal operations.
181 switch (Opc) {
182 case ISD::POISON:
183 return true;
184 case ISD::FADD:
185 case ISD::FSUB:
186 case ISD::FMUL:
187 case ISD::FDIV:
188 case ISD::FREM:
190 case ISD::UINT_TO_FP:
191 case ISD::SINT_TO_FP:
192 case ISD::FABS:
193 // Fabs is lowered to a bit operation, but it's an and which will clear the
194 // high bits anyway.
195 case ISD::FSQRT:
196 case ISD::FSIN:
197 case ISD::FCOS:
198 case ISD::FPOWI:
199 case ISD::FPOW:
200 case ISD::FLOG:
201 case ISD::FLOG2:
202 case ISD::FLOG10:
203 case ISD::FEXP:
204 case ISD::FEXP2:
205 case ISD::FCEIL:
206 case ISD::FTRUNC:
207 case ISD::FRINT:
208 case ISD::FNEARBYINT:
209 case ISD::FROUNDEVEN:
210 case ISD::FROUND:
211 case ISD::FFLOOR:
212 case ISD::FMINNUM:
213 case ISD::FMAXNUM:
214 case ISD::FLDEXP:
215 case AMDGPUISD::FRACT:
216 case AMDGPUISD::CLAMP:
217 case AMDGPUISD::COS_HW:
218 case AMDGPUISD::SIN_HW:
219 case AMDGPUISD::FMIN3:
220 case AMDGPUISD::FMAX3:
221 case AMDGPUISD::FMED3:
222 case AMDGPUISD::FMAD_FTZ:
223 case AMDGPUISD::RCP:
224 case AMDGPUISD::RSQ:
225 case AMDGPUISD::RCP_IFLAG:
226 // On gfx10, all 16-bit instructions preserve the high bits.
227 return Subtarget->getGeneration() <= AMDGPUSubtarget::GFX9;
228 case ISD::FP_ROUND:
229 // We may select fptrunc (fma/mad) to mad_mixlo, which does not zero the
230 // high bits on gfx9.
231 // TODO: If we had the source node we could see if the source was fma/mad
233 case ISD::FMA:
234 case ISD::FMAD:
235 case AMDGPUISD::DIV_FIXUP:
237 default:
238 // fcopysign, select and others may be lowered to 32-bit bit operations
239 // which don't zero the high bits.
240 return false;
241 }
242}
243
245#ifdef EXPENSIVE_CHECKS
247 LoopInfo *LI = &getAnalysis<LoopInfoWrapperPass>().getLoopInfo();
248 for (auto &L : LI->getLoopsInPreorder()) {
249 assert(L->isLCSSAForm(DT));
250 }
251#endif
253}
254
263
265 assert(Subtarget->d16PreservesUnusedBits());
266 MVT VT = N->getValueType(0).getSimpleVT();
267 if (VT != MVT::v2i16 && VT != MVT::v2f16)
268 return false;
269
270 SDValue Lo = N->getOperand(0);
271 SDValue Hi = N->getOperand(1);
272
273 LoadSDNode *LdHi = dyn_cast<LoadSDNode>(stripBitcast(Hi));
274
275 // build_vector lo, (load ptr) -> load_d16_hi ptr, lo
276 // build_vector lo, (zextload ptr from i8) -> load_d16_hi_u8 ptr, lo
277 // build_vector lo, (sextload ptr from i8) -> load_d16_hi_i8 ptr, lo
278
279 // Need to check for possible indirect dependencies on the other half of the
280 // vector to avoid introducing a cycle.
281 if (LdHi && Hi.hasOneUse() && !LdHi->isPredecessorOf(Lo.getNode())) {
282 SDVTList VTList = CurDAG->getVTList(VT, MVT::Other);
283
284 SDValue TiedIn = CurDAG->getNode(ISD::SCALAR_TO_VECTOR, SDLoc(N), VT, Lo);
285 SDValue Ops[] = {
286 LdHi->getChain(), LdHi->getBasePtr(), TiedIn
287 };
288
289 unsigned LoadOp = AMDGPUISD::LOAD_D16_HI;
290 if (LdHi->getMemoryVT() == MVT::i8) {
291 LoadOp = LdHi->getExtensionType() == ISD::SEXTLOAD ?
292 AMDGPUISD::LOAD_D16_HI_I8 : AMDGPUISD::LOAD_D16_HI_U8;
293 } else {
294 assert(LdHi->getMemoryVT() == MVT::i16);
295 }
296
297 SDValue NewLoadHi =
298 CurDAG->getMemIntrinsicNode(LoadOp, SDLoc(LdHi), VTList,
299 Ops, LdHi->getMemoryVT(),
300 LdHi->getMemOperand());
301
302 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), NewLoadHi);
303 CurDAG->ReplaceAllUsesOfValueWith(SDValue(LdHi, 1), NewLoadHi.getValue(1));
304 return true;
305 }
306
307 // build_vector (load ptr), hi -> load_d16_lo ptr, hi
308 // build_vector (zextload ptr from i8), hi -> load_d16_lo_u8 ptr, hi
309 // build_vector (sextload ptr from i8), hi -> load_d16_lo_i8 ptr, hi
310 LoadSDNode *LdLo = dyn_cast<LoadSDNode>(stripBitcast(Lo));
311 if (LdLo && Lo.hasOneUse()) {
312 SDValue TiedIn = getHi16Elt(Hi);
313 if (!TiedIn || LdLo->isPredecessorOf(TiedIn.getNode()))
314 return false;
315
316 SDVTList VTList = CurDAG->getVTList(VT, MVT::Other);
317 unsigned LoadOp = AMDGPUISD::LOAD_D16_LO;
318 if (LdLo->getMemoryVT() == MVT::i8) {
319 LoadOp = LdLo->getExtensionType() == ISD::SEXTLOAD ?
320 AMDGPUISD::LOAD_D16_LO_I8 : AMDGPUISD::LOAD_D16_LO_U8;
321 } else {
322 assert(LdLo->getMemoryVT() == MVT::i16);
323 }
324
325 TiedIn = CurDAG->getNode(ISD::BITCAST, SDLoc(N), VT, TiedIn);
326
327 SDValue Ops[] = {
328 LdLo->getChain(), LdLo->getBasePtr(), TiedIn
329 };
330
331 SDValue NewLoadLo =
332 CurDAG->getMemIntrinsicNode(LoadOp, SDLoc(LdLo), VTList,
333 Ops, LdLo->getMemoryVT(),
334 LdLo->getMemOperand());
335
336 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), NewLoadLo);
337 CurDAG->ReplaceAllUsesOfValueWith(SDValue(LdLo, 1), NewLoadLo.getValue(1));
338 return true;
339 }
340
341 return false;
342}
343
345 auto *Mem = cast<MemSDNode>(N);
346 EVT VT = N->getValueType(0);
347 if (Mem->getAddressSpace() != AMDGPUAS::REGION_ADDRESS || VT.isVector() ||
348 VT.getSizeInBits() != 16)
349 return false;
350
351 SDLoc SL(N);
352 auto *Ld = dyn_cast<LoadSDNode>(N);
353 ISD::LoadExtType ExtType =
354 Ld ? Ld->getExtensionType() : cast<AtomicSDNode>(N)->getExtensionType();
355 if (ExtType == ISD::NON_EXTLOAD)
356 ExtType = ISD::EXTLOAD;
357
358 SDValue NewLoad =
359 Ld ? CurDAG->getExtLoad(ExtType, SL, MVT::i32, Mem->getChain(),
360 Mem->getBasePtr(), Mem->getMemoryVT(),
361 Mem->getMemOperand())
362 : CurDAG->getAtomicLoad(ExtType, SL, Mem->getMemoryVT(), MVT::i32,
363 Mem->getChain(), Mem->getBasePtr(),
364 Mem->getMemOperand());
365
366 SDValue Trunc = CurDAG->getNode(ISD::TRUNCATE, SL, MVT::i16, NewLoad);
367 SDValue Ops[] = {CurDAG->getBitcast(VT, Trunc), NewLoad.getValue(1)};
368 CurDAG->ReplaceAllUsesWith(N, Ops);
369 return true;
370}
371
373 SelectionDAG::allnodes_iterator Position = CurDAG->allnodes_end();
374
375 bool MadeChange = false;
376 while (Position != CurDAG->allnodes_begin()) {
377 SDNode *N = &*--Position;
378 if (N->use_empty())
379 continue;
380
381 switch (N->getOpcode()) {
383 // TODO: Match load d16 from shl (extload:i16), 16
384 if (Subtarget->d16PreservesUnusedBits())
385 MadeChange |= matchLoadD16FromBuildVector(N);
386 break;
387 case ISD::LOAD:
388 case ISD::ATOMIC_LOAD:
389 if (Subtarget->useRealTrue16Insts())
390 MadeChange |= widenRegionLoad16(N);
391 break;
392 default:
393 break;
394 }
395 }
396
397 if (MadeChange) {
398 CurDAG->RemoveDeadNodes();
399 LLVM_DEBUG(dbgs() << "After PreProcess:\n";
400 CurDAG->dump(););
401 }
402}
403
404bool AMDGPUDAGToDAGISel::isInlineImmediate(const SDNode *N) const {
405 if (N->isUndef())
406 return true;
407
408 const SIInstrInfo *TII = Subtarget->getInstrInfo();
410 return TII->isInlineConstant(C->getAPIntValue());
411
413 return TII->isInlineConstant(C->getValueAPF());
414
415 return false;
416}
417
418/// Determine the register class for \p OpNo
419/// \returns The register class of the virtual register that will be used for
420/// the given operand number \OpNo or NULL if the register class cannot be
421/// determined.
422const TargetRegisterClass *AMDGPUDAGToDAGISel::getOperandRegClass(SDNode *N,
423 unsigned OpNo) const {
424 if (!N->isMachineOpcode()) {
425 if (N->getOpcode() == ISD::CopyToReg) {
426 Register Reg = cast<RegisterSDNode>(N->getOperand(1))->getReg();
427 if (Reg.isVirtual()) {
429 return MRI.getRegClass(Reg);
430 }
431
432 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
433 return TRI->getPhysRegBaseClass(Reg);
434 }
435
436 return nullptr;
437 }
438
439 switch (N->getMachineOpcode()) {
440 default: {
441 const SIInstrInfo *TII = Subtarget->getInstrInfo();
442 const MCInstrDesc &Desc = TII->get(N->getMachineOpcode());
443 unsigned OpIdx = Desc.getNumDefs() + OpNo;
444 if (OpIdx >= Desc.getNumOperands())
445 return nullptr;
446
447 int16_t RegClass = TII->getOpRegClassID(Desc.operands()[OpIdx]);
448 if (RegClass == -1)
449 return nullptr;
450
451 return Subtarget->getRegisterInfo()->getRegClass(RegClass);
452 }
453 case AMDGPU::REG_SEQUENCE: {
454 unsigned RCID = N->getConstantOperandVal(0);
455 const TargetRegisterClass *SuperRC =
456 Subtarget->getRegisterInfo()->getRegClass(RCID);
457
458 SDValue SubRegOp = N->getOperand(OpNo + 1);
459 unsigned SubRegIdx = SubRegOp->getAsZExtVal();
460 return Subtarget->getRegisterInfo()->getSubClassWithSubReg(SuperRC,
461 SubRegIdx);
462 }
463 }
464}
465
466SDNode *AMDGPUDAGToDAGISel::glueCopyToOp(SDNode *N, SDValue NewChain,
467 SDValue Glue) const {
469 Ops.push_back(NewChain); // Replace the chain.
470 for (unsigned i = 1, e = N->getNumOperands(); i != e; ++i)
471 Ops.push_back(N->getOperand(i));
472
473 Ops.push_back(Glue);
474 return CurDAG->MorphNodeTo(N, N->getOpcode(), N->getVTList(), Ops);
475}
476
477SDNode *AMDGPUDAGToDAGISel::glueCopyToM0(SDNode *N, SDValue Val) const {
478 const SITargetLowering& Lowering =
479 *static_cast<const SITargetLowering*>(getTargetLowering());
480
481 assert(N->getOperand(0).getValueType() == MVT::Other && "Expected chain");
482
483 SDValue M0 = Lowering.copyToM0(*CurDAG, N->getOperand(0), SDLoc(N), Val);
484 return glueCopyToOp(N, M0, M0.getValue(1));
485}
486
487SDNode *AMDGPUDAGToDAGISel::glueCopyToM0LDSInit(SDNode *N) const {
488 unsigned AS = cast<MemSDNode>(N)->getAddressSpace();
489 if (AS == AMDGPUAS::LOCAL_ADDRESS) {
490 if (Subtarget->ldsRequiresM0Init())
491 return glueCopyToM0(
492 N, CurDAG->getSignedTargetConstant(-1, SDLoc(N), MVT::i32));
493 } else if (AS == AMDGPUAS::REGION_ADDRESS) {
494 MachineFunction &MF = CurDAG->getMachineFunction();
495 unsigned Value = MF.getInfo<SIMachineFunctionInfo>()->getGDSSize();
496 return
497 glueCopyToM0(N, CurDAG->getTargetConstant(Value, SDLoc(N), MVT::i32));
498 }
499 return N;
500}
501
502MachineSDNode *AMDGPUDAGToDAGISel::buildSMovImm64(SDLoc &DL, uint64_t Imm,
503 EVT VT) const {
504 SDNode *Lo = CurDAG->getMachineNode(
505 AMDGPU::S_MOV_B32, DL, MVT::i32,
506 CurDAG->getTargetConstant(Lo_32(Imm), DL, MVT::i32));
507 SDNode *Hi = CurDAG->getMachineNode(
508 AMDGPU::S_MOV_B32, DL, MVT::i32,
509 CurDAG->getTargetConstant(Hi_32(Imm), DL, MVT::i32));
510 const SDValue Ops[] = {
511 CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID, DL, MVT::i32),
512 SDValue(Lo, 0), CurDAG->getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
513 SDValue(Hi, 0), CurDAG->getTargetConstant(AMDGPU::sub1, DL, MVT::i32)};
514
515 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, DL, VT, Ops);
516}
517
518SDNode *AMDGPUDAGToDAGISel::packConstantV2I16(const SDNode *N,
519 SelectionDAG &DAG) const {
520 // TODO: Handle undef as zero
521
522 assert(N->getOpcode() == ISD::BUILD_VECTOR && N->getNumOperands() == 2);
523 uint32_t LHSVal, RHSVal;
524 if (getConstantValue(N->getOperand(0), LHSVal) &&
525 getConstantValue(N->getOperand(1), RHSVal)) {
526 SDLoc SL(N);
527 uint32_t K = (LHSVal & 0xffff) | (RHSVal << 16);
528 return DAG.getMachineNode(
529 isVGPRImm(N) ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32, SL,
530 N->getValueType(0), DAG.getTargetConstant(K, SL, MVT::i32));
531 }
532
533 return nullptr;
534}
535
536void AMDGPUDAGToDAGISel::SelectBuildVector(SDNode *N, unsigned RegClassID) {
537 EVT VT = N->getValueType(0);
538 unsigned NumVectorElts = VT.getVectorNumElements();
539 EVT EltVT = VT.getVectorElementType();
540 SDLoc DL(N);
541 SDValue RegClass = CurDAG->getTargetConstant(RegClassID, DL, MVT::i32);
542
543 if (NumVectorElts == 1) {
544 CurDAG->SelectNodeTo(N, AMDGPU::COPY_TO_REGCLASS, EltVT, N->getOperand(0),
545 RegClass);
546 return;
547 }
548
549 bool IsGCN = CurDAG->getSubtarget().getTargetTriple().isAMDGCN();
550 if (IsGCN && Subtarget->has64BitLiterals() && VT.getSizeInBits() == 64 &&
551 CurDAG->isConstantValueOfAnyType(SDValue(N, 0))) {
552 uint64_t C = 0;
553 bool AllConst = true;
554 unsigned EltSize = EltVT.getSizeInBits();
555 for (unsigned I = 0; I < NumVectorElts; ++I) {
556 SDValue Op = N->getOperand(I);
557 if (Op.isUndef()) {
558 AllConst = false;
559 break;
560 }
561 uint64_t Val;
563 Val = CF->getValueAPF().bitcastToAPInt().getZExtValue();
564 } else
565 Val = cast<ConstantSDNode>(Op)->getZExtValue();
566 C |= Val << (EltSize * I);
567 }
568 if (AllConst) {
569 SDValue CV = CurDAG->getTargetConstant(C, DL, MVT::i64);
570 MachineSDNode *Copy =
571 CurDAG->getMachineNode(AMDGPU::S_MOV_B64_IMM_PSEUDO, DL, VT, CV);
572 CurDAG->SelectNodeTo(N, AMDGPU::COPY_TO_REGCLASS, VT, SDValue(Copy, 0),
573 RegClass);
574 return;
575 }
576 }
577
578 assert(NumVectorElts <= 32 && "Vectors with more than 32 elements not "
579 "supported yet");
580 // 32 = Max Num Vector Elements
581 // 2 = 2 REG_SEQUENCE operands per element (value, subreg index)
582 // 1 = Vector Register Class
583 SmallVector<SDValue, 32 * 2 + 1> RegSeqArgs(NumVectorElts * 2 + 1);
584
585 RegSeqArgs[0] = CurDAG->getTargetConstant(RegClassID, DL, MVT::i32);
586 bool IsRegSeq = true;
587 unsigned NOps = N->getNumOperands();
588 unsigned EltSizeInRegs = EltVT.getSizeInBits() / 32;
589 assert(IsGCN || EltSizeInRegs == 1);
590 for (unsigned i = 0; i < NOps; i++) {
591 // XXX: Why is this here?
592 if (isa<RegisterSDNode>(N->getOperand(i))) {
593 IsRegSeq = false;
594 break;
595 }
596 unsigned Sub = IsGCN ? SIRegisterInfo::getSubRegFromChannel(
597 i * EltSizeInRegs, EltSizeInRegs)
599 RegSeqArgs[1 + (2 * i)] = N->getOperand(i);
600 RegSeqArgs[1 + (2 * i) + 1] = CurDAG->getTargetConstant(Sub, DL, MVT::i32);
601 }
602 if (NOps != NumVectorElts) {
603 // Fill in the missing undef elements if this was a scalar_to_vector.
604 assert(N->getOpcode() == ISD::SCALAR_TO_VECTOR && NOps < NumVectorElts);
605 MachineSDNode *ImpDef = CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,
606 DL, EltVT);
607 for (unsigned i = NOps; i < NumVectorElts; ++i) {
608 unsigned Sub = IsGCN ? SIRegisterInfo::getSubRegFromChannel(
609 i * EltSizeInRegs, EltSizeInRegs)
611 RegSeqArgs[1 + (2 * i)] = SDValue(ImpDef, 0);
612 RegSeqArgs[1 + (2 * i) + 1] =
613 CurDAG->getTargetConstant(Sub, DL, MVT::i32);
614 }
615 }
616
617 if (!IsRegSeq)
618 SelectCode(N);
619 CurDAG->SelectNodeTo(N, AMDGPU::REG_SEQUENCE, N->getVTList(), RegSeqArgs);
620}
621
623 EVT VT = N->getValueType(0);
624 EVT EltVT = VT.getVectorElementType();
625
626 // TODO: Handle 16-bit element vectors with even aligned masks.
627 if (!Subtarget->hasPkMovB32() || !EltVT.bitsEq(MVT::i32) ||
628 VT.getVectorNumElements() != 2) {
629 SelectCode(N);
630 return;
631 }
632
633 auto *SVN = cast<ShuffleVectorSDNode>(N);
634
635 SDValue Src0 = SVN->getOperand(0);
636 SDValue Src1 = SVN->getOperand(1);
637 ArrayRef<int> Mask = SVN->getMask();
638 SDLoc DL(N);
639
640 assert(Src0.getValueType().getVectorNumElements() == 2 && Mask.size() == 2 &&
641 Mask[0] < 4 && Mask[1] < 4);
642
643 SDValue VSrc0 = Mask[0] < 2 ? Src0 : Src1;
644 SDValue VSrc1 = Mask[1] < 2 ? Src0 : Src1;
645 unsigned Src0SubReg = Mask[0] & 1 ? AMDGPU::sub1 : AMDGPU::sub0;
646 unsigned Src1SubReg = Mask[1] & 1 ? AMDGPU::sub1 : AMDGPU::sub0;
647
648 if (Mask[0] < 0) {
649 Src0SubReg = Src1SubReg;
650 MachineSDNode *ImpDef =
651 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, VT);
652 VSrc0 = SDValue(ImpDef, 0);
653 }
654
655 if (Mask[1] < 0) {
656 Src1SubReg = Src0SubReg;
657 MachineSDNode *ImpDef =
658 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, VT);
659 VSrc1 = SDValue(ImpDef, 0);
660 }
661
662 // SGPR case needs to lower to copies.
663 //
664 // Also use subregister extract when we can directly blend the registers with
665 // a simple subregister copy.
666 //
667 // TODO: Maybe we should fold this out earlier
668 if (N->isDivergent() && Src0SubReg == AMDGPU::sub1 &&
669 Src1SubReg == AMDGPU::sub0) {
670 // The low element of the result always comes from src0.
671 // The high element of the result always comes from src1.
672 // op_sel selects the high half of src0.
673 // op_sel_hi selects the high half of src1.
674
675 unsigned Src0OpSel =
676 Src0SubReg == AMDGPU::sub1 ? SISrcMods::OP_SEL_0 : SISrcMods::NONE;
677 unsigned Src1OpSel =
678 Src1SubReg == AMDGPU::sub1 ? SISrcMods::OP_SEL_0 : SISrcMods::NONE;
679
680 // Enable op_sel_hi to avoid printing it. This should have no effect on the
681 // result.
682 Src0OpSel |= SISrcMods::OP_SEL_1;
683 Src1OpSel |= SISrcMods::OP_SEL_1;
684
685 SDValue Src0OpSelVal = CurDAG->getTargetConstant(Src0OpSel, DL, MVT::i32);
686 SDValue Src1OpSelVal = CurDAG->getTargetConstant(Src1OpSel, DL, MVT::i32);
687 SDValue ZeroMods = CurDAG->getTargetConstant(0, DL, MVT::i32);
688
689 CurDAG->SelectNodeTo(N, AMDGPU::V_PK_MOV_B32, N->getVTList(),
690 {Src0OpSelVal, VSrc0, Src1OpSelVal, VSrc1,
691 ZeroMods, // clamp
692 ZeroMods, // op_sel
693 ZeroMods, // op_sel_hi
694 ZeroMods, // neg_lo
695 ZeroMods}); // neg_hi
696 return;
697 }
698
699 SDValue ResultElt0 =
700 CurDAG->getTargetExtractSubreg(Src0SubReg, DL, EltVT, VSrc0);
701 SDValue ResultElt1 =
702 CurDAG->getTargetExtractSubreg(Src1SubReg, DL, EltVT, VSrc1);
703
704 const SDValue Ops[] = {
705 CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID, DL, MVT::i32),
706 ResultElt0, CurDAG->getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
707 ResultElt1, CurDAG->getTargetConstant(AMDGPU::sub1, DL, MVT::i32)};
708 CurDAG->SelectNodeTo(N, TargetOpcode::REG_SEQUENCE, VT, Ops);
709}
710
712 unsigned int Opc = N->getOpcode();
713 if (N->isMachineOpcode()) {
714 N->setNodeId(-1);
715 return; // Already selected.
716 }
717
718 // isa<MemSDNode> almost works but is slightly too permissive for some DS
719 // intrinsics.
720 if (Opc == ISD::LOAD || Opc == ISD::STORE || isa<AtomicSDNode>(N)) {
721 N = glueCopyToM0LDSInit(N);
722 SelectCode(N);
723 return;
724 }
725
726 switch (Opc) {
727 default:
728 break;
729 case ISD::UADDO_CARRY:
730 case ISD::USUBO_CARRY:
731 if (N->getValueType(0) == MVT::i64) {
732 SelectAddcSubbI64(N);
733 return;
734 }
735
736 if (N->getValueType(0) != MVT::i32)
737 break;
738
739 SelectAddcSubb(N);
740 return;
741 case ISD::UADDO:
742 case ISD::USUBO: {
743 if (N->getValueType(0) == MVT::i64) {
744 SelectAddcSubbI64(N);
745 return;
746 }
747
748 SelectUADDO_USUBO(N);
749 return;
750 }
751 case AMDGPUISD::FMUL_W_CHAIN: {
752 SelectFMUL_W_CHAIN(N);
753 return;
754 }
755 case AMDGPUISD::FMA_W_CHAIN: {
756 SelectFMA_W_CHAIN(N);
757 return;
758 }
759
761 case ISD::BUILD_VECTOR: {
762 EVT VT = N->getValueType(0);
763 unsigned NumVectorElts = VT.getVectorNumElements();
764 if (VT.getScalarSizeInBits() == 16) {
765 if (Opc == ISD::BUILD_VECTOR && NumVectorElts == 2) {
766 if (SDNode *Packed = packConstantV2I16(N, *CurDAG)) {
767 ReplaceNode(N, Packed);
768 return;
769 }
770 }
771
772 break;
773 }
774
775 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
776 EVT EltTy = VT.getVectorElementType();
777 assert(EltTy.bitsEq(MVT::i32) || EltTy.bitsEq(MVT::i64));
778 unsigned VecInBits = NumVectorElts * EltTy.getScalarSizeInBits();
779 const TargetRegisterClass *RegClass =
780 N->isDivergent() ? TRI->getDefaultVectorSuperClassForBitWidth(VecInBits)
782
783 SelectBuildVector(N, RegClass->getID());
784 return;
785 }
788 return;
789 case ISD::BUILD_PAIR: {
790 SDValue RC, SubReg0, SubReg1;
791 SDLoc DL(N);
792 if (N->getValueType(0) == MVT::i128) {
793 RC = CurDAG->getTargetConstant(AMDGPU::SGPR_128RegClassID, DL, MVT::i32);
794 SubReg0 = CurDAG->getTargetConstant(AMDGPU::sub0_sub1, DL, MVT::i32);
795 SubReg1 = CurDAG->getTargetConstant(AMDGPU::sub2_sub3, DL, MVT::i32);
796 } else if (N->getValueType(0) == MVT::i64) {
797 RC = CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID, DL, MVT::i32);
798 SubReg0 = CurDAG->getTargetConstant(AMDGPU::sub0, DL, MVT::i32);
799 SubReg1 = CurDAG->getTargetConstant(AMDGPU::sub1, DL, MVT::i32);
800 } else {
801 llvm_unreachable("Unhandled value type for BUILD_PAIR");
802 }
803 const SDValue Ops[] = { RC, N->getOperand(0), SubReg0,
804 N->getOperand(1), SubReg1 };
805 ReplaceNode(N, CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, DL,
806 N->getValueType(0), Ops));
807 return;
808 }
809
810 case ISD::Constant:
811 case ISD::ConstantFP: {
812 if (N->getValueType(0).getSizeInBits() != 64 || isInlineImmediate(N) ||
813 Subtarget->has64BitLiterals())
814 break;
815
816 uint64_t Imm;
818 Imm = FP->getValueAPF().bitcastToAPInt().getZExtValue();
820 break;
821 } else {
823 Imm = C->getZExtValue();
825 break;
826 }
827
828 SDLoc DL(N);
829 ReplaceNode(N, buildSMovImm64(DL, Imm, N->getValueType(0)));
830 return;
831 }
832 case AMDGPUISD::BFE_I32:
833 case AMDGPUISD::BFE_U32: {
834 // There is a scalar version available, but unlike the vector version which
835 // has a separate operand for the offset and width, the scalar version packs
836 // the width and offset into a single operand. Try to move to the scalar
837 // version if the offsets are constant, so that we can try to keep extended
838 // loads of kernel arguments in SGPRs.
839
840 // TODO: Technically we could try to pattern match scalar bitshifts of
841 // dynamic values, but it's probably not useful.
843 if (!Offset)
844 break;
845
846 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(N->getOperand(2));
847 if (!Width)
848 break;
849
850 bool Signed = Opc == AMDGPUISD::BFE_I32;
851
852 uint32_t OffsetVal = Offset->getZExtValue();
853 uint32_t WidthVal = Width->getZExtValue();
854
855 ReplaceNode(N, getBFE32(Signed, SDLoc(N), N->getOperand(0), OffsetVal,
856 WidthVal));
857 return;
858 }
859 case AMDGPUISD::DIV_SCALE: {
860 SelectDIV_SCALE(N);
861 return;
862 }
865 SelectMAD_64_32(N);
866 return;
867 }
868 case ISD::SMUL_LOHI:
869 case ISD::UMUL_LOHI:
870 return SelectMUL_LOHI(N);
871 case ISD::CopyToReg: {
873 *static_cast<const SITargetLowering*>(getTargetLowering());
874 N = Lowering.legalizeTargetIndependentNode(N, *CurDAG);
875 break;
876 }
877 case ISD::AND:
878 case ISD::SRL:
879 case ISD::SRA:
881 if (N->getValueType(0) != MVT::i32)
882 break;
883
884 SelectS_BFE(N);
885 return;
886 case ISD::BRCOND:
887 SelectBRCOND(N);
888 return;
889 case ISD::FP_EXTEND:
890 SelectFP_EXTEND(N);
891 return;
892 case AMDGPUISD::CVT_PKRTZ_F16_F32:
893 case AMDGPUISD::CVT_PKNORM_I16_F32:
894 case AMDGPUISD::CVT_PKNORM_U16_F32:
895 case AMDGPUISD::CVT_PK_U16_U32:
896 case AMDGPUISD::CVT_PK_I16_I32: {
897 // Hack around using a legal type if f16 is illegal.
898 if (N->getValueType(0) == MVT::i32) {
899 MVT NewVT = Opc == AMDGPUISD::CVT_PKRTZ_F16_F32 ? MVT::v2f16 : MVT::v2i16;
900 N = CurDAG->MorphNodeTo(N, N->getOpcode(), CurDAG->getVTList(NewVT),
901 { N->getOperand(0), N->getOperand(1) });
902 SelectCode(N);
903 return;
904 }
905
906 break;
907 }
909 SelectINTRINSIC_W_CHAIN(N);
910 return;
911 }
913 SelectINTRINSIC_WO_CHAIN(N);
914 return;
915 }
916 case ISD::INTRINSIC_VOID: {
917 SelectINTRINSIC_VOID(N);
918 return;
919 }
921 SelectWAVE_ADDRESS(N);
922 return;
923 }
924 case ISD::STACKRESTORE: {
925 SelectSTACKRESTORE(N);
926 return;
927 }
928 }
929
930 SelectCode(N);
931}
932
934 if (!Subtarget->hasSDWA())
935 return false;
936
937 if (N->getOpcode() == ISD::SIGN_EXTEND_INREG) {
938 EVT VT = cast<VTSDNode>(N->getOperand(1))->getVT();
939 return VT.getScalarSizeInBits() == 8 || VT.getScalarSizeInBits() == 16;
940 }
941
942 if (N->getOpcode() == ISD::AND)
943 if (auto *RHS = dyn_cast<ConstantSDNode>(N->getOperand(1)))
944 return RHS->getZExtValue() == 0xFF || RHS->getZExtValue() == 0xFFFF;
945
946 if (N->getOpcode() == ISD::SRA || N->getOpcode() == ISD::SRL)
947 if (auto *RHS = dyn_cast<ConstantSDNode>(N->getOperand(1)))
948 return (RHS->getZExtValue() % 8) == 0;
949
950 return false;
951}
952
953bool AMDGPUDAGToDAGISel::isUniformBr(const SDNode *N) const {
954 const BasicBlock *BB = FuncInfo->MBB->getBasicBlock();
955 const Instruction *Term = BB->getTerminator();
956 return Term->getMetadata("amdgpu.uniform") ||
957 Term->getMetadata("structurizecfg.uniform");
958}
959
960bool AMDGPUDAGToDAGISel::isUnneededShiftMask(const SDNode *N,
961 unsigned ShAmtBits) const {
962 assert(N->getOpcode() == ISD::AND);
963
964 const APInt &RHS = N->getConstantOperandAPInt(1);
965 if (RHS.countr_one() >= ShAmtBits)
966 return true;
967
968 const APInt &LHSKnownZeros = CurDAG->computeKnownBits(N->getOperand(0)).Zero;
969 return (LHSKnownZeros | RHS).countr_one() >= ShAmtBits;
970}
971
973 SDValue &N0, SDValue &N1) {
974 if (Addr.getValueType() == MVT::i64 && Addr.getOpcode() == ISD::BITCAST &&
976 // As we split 64-bit `or` earlier, it's complicated pattern to match, i.e.
977 // (i64 (bitcast (v2i32 (build_vector
978 // (or (extract_vector_elt V, 0), OFFSET),
979 // (extract_vector_elt V, 1)))))
980 SDValue Lo = Addr.getOperand(0).getOperand(0);
981 if (Lo.getOpcode() == ISD::OR && DAG.isBaseWithConstantOffset(Lo)) {
982 SDValue BaseLo = Lo.getOperand(0);
983 SDValue BaseHi = Addr.getOperand(0).getOperand(1);
984 // Check that split base (Lo and Hi) are extracted from the same one.
985 if (BaseLo.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
987 BaseLo.getOperand(0) == BaseHi.getOperand(0) &&
988 // Lo is statically extracted from index 0.
989 isa<ConstantSDNode>(BaseLo.getOperand(1)) &&
990 BaseLo.getConstantOperandVal(1) == 0 &&
991 // Hi is statically extracted from index 0.
992 isa<ConstantSDNode>(BaseHi.getOperand(1)) &&
993 BaseHi.getConstantOperandVal(1) == 1) {
994 N0 = BaseLo.getOperand(0).getOperand(0);
995 N1 = Lo.getOperand(1);
996 return true;
997 }
998 }
999 }
1000 return false;
1001}
1002
1003bool AMDGPUDAGToDAGISel::isBaseWithConstantOffset64(SDValue Addr, SDValue &LHS,
1004 SDValue &RHS) const {
1005 if (CurDAG->isBaseWithConstantOffset(Addr)) {
1006 LHS = Addr.getOperand(0);
1007 RHS = Addr.getOperand(1);
1008 return true;
1009 }
1010
1013 return true;
1014 }
1015
1016 return false;
1017}
1018
1020 return "AMDGPU DAG->DAG Pattern Instruction Selection";
1021}
1022
1026
1031 .getManager();
1032 auto &F = MF.getFunction();
1033 // UniformityInfoAnalysis is optional in generic dag isel,
1034 // AMDGPUISelDAGToDAGPass requires it, calculate it explicitly.
1035 FAM.getResult<UniformityInfoAnalysis>(F);
1036#ifdef EXPENSIVE_CHECKS
1037 DominatorTree &DT = FAM.getResult<DominatorTreeAnalysis>(F);
1038 LoopInfo &LI = FAM.getResult<LoopAnalysis>(F);
1039 for (auto &L : LI.getLoopsInPreorder())
1040 assert(L->isLCSSAForm(DT) && "Loop is not in LCSSA form!");
1041#endif
1042 return SelectionDAGISelPass::run(MF, MFAM);
1043}
1044
1045//===----------------------------------------------------------------------===//
1046// Complex Patterns
1047//===----------------------------------------------------------------------===//
1048
1049bool AMDGPUDAGToDAGISel::SelectADDRVTX_READ(SDValue Addr, SDValue &Base,
1050 SDValue &Offset) {
1051 return false;
1052}
1053
1054bool AMDGPUDAGToDAGISel::SelectADDRIndirect(SDValue Addr, SDValue &Base,
1055 SDValue &Offset) {
1057 SDLoc DL(Addr);
1058
1059 if ((C = dyn_cast<ConstantSDNode>(Addr))) {
1060 Base = CurDAG->getRegister(R600::INDIRECT_BASE_ADDR, MVT::i32);
1061 Offset = CurDAG->getTargetConstant(C->getZExtValue(), DL, MVT::i32);
1062 } else if ((Addr.getOpcode() == AMDGPUISD::DWORDADDR) &&
1063 (C = dyn_cast<ConstantSDNode>(Addr.getOperand(0)))) {
1064 Base = CurDAG->getRegister(R600::INDIRECT_BASE_ADDR, MVT::i32);
1065 Offset = CurDAG->getTargetConstant(C->getZExtValue(), DL, MVT::i32);
1066 } else if ((Addr.getOpcode() == ISD::ADD || Addr.getOpcode() == ISD::OR) &&
1067 (C = dyn_cast<ConstantSDNode>(Addr.getOperand(1)))) {
1068 Base = Addr.getOperand(0);
1069 Offset = CurDAG->getTargetConstant(C->getZExtValue(), DL, MVT::i32);
1070 } else {
1071 Base = Addr;
1072 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1073 }
1074
1075 return true;
1076}
1077
1078SDValue AMDGPUDAGToDAGISel::getMaterializedScalarImm32(int64_t Val,
1079 const SDLoc &DL) const {
1080 SDNode *Mov = CurDAG->getMachineNode(
1081 AMDGPU::S_MOV_B32, DL, MVT::i32,
1082 CurDAG->getTargetConstant(Val, DL, MVT::i32));
1083 return SDValue(Mov, 0);
1084}
1085
1086void AMDGPUDAGToDAGISel::SelectAddcSubb(SDNode *N) {
1087 SDValue LHS = N->getOperand(0);
1088 SDValue RHS = N->getOperand(1);
1089 SDValue CI = N->getOperand(2);
1090
1091 if (N->isDivergent()) {
1092 unsigned Opc = N->getOpcode() == ISD::UADDO_CARRY ? AMDGPU::V_ADDC_U32_e64
1093 : AMDGPU::V_SUBB_U32_e64;
1094 CurDAG->SelectNodeTo(
1095 N, Opc, N->getVTList(),
1096 {LHS, RHS, CI,
1097 CurDAG->getTargetConstant(0, {}, MVT::i1) /*clamp bit*/});
1098 } else {
1099 unsigned Opc = N->getOpcode() == ISD::UADDO_CARRY ? AMDGPU::S_ADD_CO_PSEUDO
1100 : AMDGPU::S_SUB_CO_PSEUDO;
1101 CurDAG->SelectNodeTo(N, Opc, N->getVTList(), {LHS, RHS, CI});
1102 }
1103}
1104
1105void AMDGPUDAGToDAGISel::SelectAddcSubbI64(SDNode *N) {
1106 SDLoc DL(N);
1107 SDValue LHS = N->getOperand(0);
1108 SDValue RHS = N->getOperand(1);
1109
1110 unsigned Opcode = N->getOpcode();
1111 bool ConsumeCarry = Opcode == ISD::UADDO_CARRY || Opcode == ISD::USUBO_CARRY;
1112 bool IsAdd = Opcode == ISD::UADDO || Opcode == ISD::UADDO_CARRY;
1113
1114 SDValue Sub0 = CurDAG->getTargetConstant(AMDGPU::sub0, DL, MVT::i32);
1115 SDValue Sub1 = CurDAG->getTargetConstant(AMDGPU::sub1, DL, MVT::i32);
1116
1117 SDNode *Lo0 = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
1118 MVT::i32, LHS, Sub0);
1119 SDNode *Hi0 = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
1120 MVT::i32, LHS, Sub1);
1121
1122 SDNode *Lo1 = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
1123 MVT::i32, RHS, Sub0);
1124 SDNode *Hi1 = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
1125 MVT::i32, RHS, Sub1);
1126
1127 SDVTList VTList = CurDAG->getVTList(MVT::i32, N->getValueType(1));
1128
1129 static const unsigned NoCarryOpcMap[2][2] = {
1130 {AMDGPU::S_USUBO_PSEUDO, AMDGPU::S_UADDO_PSEUDO},
1131 {AMDGPU::V_SUB_CO_U32_e64, AMDGPU::V_ADD_CO_U32_e64}};
1132 static const unsigned CarryOpcMap[2][2] = {
1133 {AMDGPU::S_SUB_CO_PSEUDO, AMDGPU::S_ADD_CO_PSEUDO},
1134 {AMDGPU::V_SUBB_U32_e64, AMDGPU::V_ADDC_U32_e64}};
1135
1136 bool IsVALU = N->isDivergent();
1137
1138 unsigned NoCarryOpc = NoCarryOpcMap[IsVALU][IsAdd];
1139 unsigned CarryOpc = CarryOpcMap[IsVALU][IsAdd];
1140 SDValue Clamp = CurDAG->getTargetConstant(0, DL, MVT::i1);
1141
1142 SDNode *AddLo;
1143 if (!ConsumeCarry) {
1144 if (IsVALU) {
1145 SDValue Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0), Clamp};
1146 AddLo = CurDAG->getMachineNode(NoCarryOpc, DL, VTList, Args);
1147 } else {
1148 SDValue Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0)};
1149 AddLo = CurDAG->getMachineNode(NoCarryOpc, DL, VTList, Args);
1150 }
1151 } else {
1152 if (IsVALU) {
1153 SDValue Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0), N->getOperand(2),
1154 Clamp};
1155 AddLo = CurDAG->getMachineNode(CarryOpc, DL, VTList, Args);
1156 } else {
1157 SDValue Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0), N->getOperand(2)};
1158 AddLo = CurDAG->getMachineNode(CarryOpc, DL, VTList, Args);
1159 }
1160 }
1161
1162 SDNode *AddHi;
1163 if (IsVALU) {
1164 SDValue Args[] = {SDValue(Hi0, 0), SDValue(Hi1, 0), SDValue(AddLo, 1),
1165 Clamp};
1166 AddHi = CurDAG->getMachineNode(CarryOpc, DL, VTList, Args);
1167 } else {
1168 SDValue Args[] = {SDValue(Hi0, 0), SDValue(Hi1, 0), SDValue(AddLo, 1)};
1169 AddHi = CurDAG->getMachineNode(CarryOpc, DL, VTList, Args);
1170 }
1171
1172 unsigned RC = IsVALU ? AMDGPU::VReg_64RegClassID : AMDGPU::SReg_64RegClassID;
1173 SDValue RegSequenceArgs[] = {CurDAG->getTargetConstant(RC, DL, MVT::i32),
1174 SDValue(AddLo, 0), Sub0, SDValue(AddHi, 0),
1175 Sub1};
1176 SDNode *RegSequence = CurDAG->getMachineNode(AMDGPU::REG_SEQUENCE, DL,
1177 MVT::i64, RegSequenceArgs);
1178
1179 ReplaceUses(SDValue(N, 1), SDValue(AddHi, 1));
1180 ReplaceNode(N, RegSequence);
1181}
1182
1183void AMDGPUDAGToDAGISel::SelectUADDO_USUBO(SDNode *N) {
1184 // The name of the opcodes are misleading. v_add_i32/v_sub_i32 have unsigned
1185 // carry out despite the _i32 name. These were renamed in VI to _U32.
1186 // FIXME: We should probably rename the opcodes here.
1187 bool IsAdd = N->getOpcode() == ISD::UADDO;
1188 bool IsVALU = N->isDivergent();
1189
1190 for (SDNode::user_iterator UI = N->user_begin(), E = N->user_end(); UI != E;
1191 ++UI)
1192 if (UI.getUse().getResNo() == 1) {
1193 if (UI->isMachineOpcode()) {
1194 if (UI->getMachineOpcode() !=
1195 (IsAdd ? AMDGPU::S_ADD_CO_PSEUDO : AMDGPU::S_SUB_CO_PSEUDO)) {
1196 IsVALU = true;
1197 break;
1198 }
1199 } else {
1200 if (UI->getOpcode() != (IsAdd ? ISD::UADDO_CARRY : ISD::USUBO_CARRY)) {
1201 IsVALU = true;
1202 break;
1203 }
1204 }
1205 }
1206
1207 if (IsVALU) {
1208 unsigned Opc = IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
1209
1210 CurDAG->SelectNodeTo(
1211 N, Opc, N->getVTList(),
1212 {N->getOperand(0), N->getOperand(1),
1213 CurDAG->getTargetConstant(0, {}, MVT::i1) /*clamp bit*/});
1214 } else {
1215 unsigned Opc = IsAdd ? AMDGPU::S_UADDO_PSEUDO : AMDGPU::S_USUBO_PSEUDO;
1216
1217 CurDAG->SelectNodeTo(N, Opc, N->getVTList(),
1218 {N->getOperand(0), N->getOperand(1)});
1219 }
1220}
1221
1222void AMDGPUDAGToDAGISel::SelectFMA_W_CHAIN(SDNode *N) {
1223 // src0_modifiers, src0, src1_modifiers, src1, src2_modifiers, src2, clamp, omod
1224 SDValue Ops[10];
1225
1226 SelectVOP3Mods0(N->getOperand(1), Ops[1], Ops[0], Ops[6], Ops[7]);
1227 SelectVOP3Mods(N->getOperand(2), Ops[3], Ops[2]);
1228 SelectVOP3Mods(N->getOperand(3), Ops[5], Ops[4]);
1229 Ops[8] = N->getOperand(0);
1230 Ops[9] = N->getOperand(4);
1231
1232 // If there are no source modifiers, prefer fmac over fma because it can use
1233 // the smaller VOP2 encoding.
1234 bool UseFMAC = Subtarget->hasDLInsts() &&
1235 cast<ConstantSDNode>(Ops[0])->isZero() &&
1236 cast<ConstantSDNode>(Ops[2])->isZero() &&
1237 cast<ConstantSDNode>(Ops[4])->isZero();
1238 unsigned Opcode = UseFMAC ? AMDGPU::V_FMAC_F32_e64 : AMDGPU::V_FMA_F32_e64;
1239 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), Ops);
1240}
1241
1242void AMDGPUDAGToDAGISel::SelectFMUL_W_CHAIN(SDNode *N) {
1243 // src0_modifiers, src0, src1_modifiers, src1, clamp, omod
1244 SDValue Ops[8];
1245
1246 SelectVOP3Mods0(N->getOperand(1), Ops[1], Ops[0], Ops[4], Ops[5]);
1247 SelectVOP3Mods(N->getOperand(2), Ops[3], Ops[2]);
1248 Ops[6] = N->getOperand(0);
1249 Ops[7] = N->getOperand(3);
1250
1251 CurDAG->SelectNodeTo(N, AMDGPU::V_MUL_F32_e64, N->getVTList(), Ops);
1252}
1253
1254// We need to handle this here because tablegen doesn't support matching
1255// instructions with multiple outputs.
1256void AMDGPUDAGToDAGISel::SelectDIV_SCALE(SDNode *N) {
1257 EVT VT = N->getValueType(0);
1258
1259 assert(VT == MVT::f32 || VT == MVT::f64);
1260
1261 unsigned Opc
1262 = (VT == MVT::f64) ? AMDGPU::V_DIV_SCALE_F64_e64 : AMDGPU::V_DIV_SCALE_F32_e64;
1263
1264 // src0_modifiers, src0, src1_modifiers, src1, src2_modifiers, src2, clamp,
1265 // omod
1266 SDValue Ops[8];
1267 SelectVOP3BMods0(N->getOperand(0), Ops[1], Ops[0], Ops[6], Ops[7]);
1268 SelectVOP3BMods(N->getOperand(1), Ops[3], Ops[2]);
1269 SelectVOP3BMods(N->getOperand(2), Ops[5], Ops[4]);
1270 CurDAG->SelectNodeTo(N, Opc, N->getVTList(), Ops);
1271}
1272
1273// We need to handle this here because tablegen doesn't support matching
1274// instructions with multiple outputs.
1275void AMDGPUDAGToDAGISel::SelectMAD_64_32(SDNode *N) {
1276 SDLoc SL(N);
1277 bool Signed = N->getOpcode() == AMDGPUISD::MAD_I64_I32;
1278 unsigned Opc;
1279 bool UseNoCarry = Subtarget->hasMadNC64_32Insts() && !N->hasAnyUseOfValue(1);
1280 if (Subtarget->hasMADIntraFwdBug())
1281 Opc = Signed ? AMDGPU::V_MAD_I64_I32_gfx11_e64
1282 : AMDGPU::V_MAD_U64_U32_gfx11_e64;
1283 else if (UseNoCarry)
1284 Opc = Signed ? AMDGPU::V_MAD_NC_I64_I32_e64 : AMDGPU::V_MAD_NC_U64_U32_e64;
1285 else
1286 Opc = Signed ? AMDGPU::V_MAD_I64_I32_e64 : AMDGPU::V_MAD_U64_U32_e64;
1287
1288 SDValue Clamp = CurDAG->getTargetConstant(0, SL, MVT::i1);
1289 SDValue Ops[] = { N->getOperand(0), N->getOperand(1), N->getOperand(2),
1290 Clamp };
1291
1292 if (UseNoCarry) {
1293 MachineSDNode *Mad = CurDAG->getMachineNode(Opc, SL, MVT::i64, Ops);
1294 ReplaceUses(SDValue(N, 0), SDValue(Mad, 0));
1295 CurDAG->RemoveDeadNode(N);
1296 return;
1297 }
1298
1299 CurDAG->SelectNodeTo(N, Opc, N->getVTList(), Ops);
1300}
1301
1302// We need to handle this here because tablegen doesn't support matching
1303// instructions with multiple outputs.
1304void AMDGPUDAGToDAGISel::SelectMUL_LOHI(SDNode *N) {
1305 SDLoc SL(N);
1306 bool Signed = N->getOpcode() == ISD::SMUL_LOHI;
1307 SDVTList VTList;
1308 unsigned Opc;
1309 if (Subtarget->hasMadNC64_32Insts()) {
1310 VTList = CurDAG->getVTList(MVT::i64);
1311 Opc = Signed ? AMDGPU::V_MAD_NC_I64_I32_e64 : AMDGPU::V_MAD_NC_U64_U32_e64;
1312 } else {
1313 VTList = CurDAG->getVTList(MVT::i64, MVT::i1);
1314 if (Subtarget->hasMADIntraFwdBug()) {
1315 Opc = Signed ? AMDGPU::V_MAD_I64_I32_gfx11_e64
1316 : AMDGPU::V_MAD_U64_U32_gfx11_e64;
1317 } else {
1318 Opc = Signed ? AMDGPU::V_MAD_I64_I32_e64 : AMDGPU::V_MAD_U64_U32_e64;
1319 }
1320 }
1321
1322 SDValue Zero = CurDAG->getTargetConstant(0, SL, MVT::i64);
1323 SDValue Clamp = CurDAG->getTargetConstant(0, SL, MVT::i1);
1324 SDValue Ops[] = {N->getOperand(0), N->getOperand(1), Zero, Clamp};
1325 SDNode *Mad = CurDAG->getMachineNode(Opc, SL, VTList, Ops);
1326 if (!SDValue(N, 0).use_empty()) {
1327 SDValue Sub0 = CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32);
1328 SDNode *Lo = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, SL,
1329 MVT::i32, SDValue(Mad, 0), Sub0);
1330 ReplaceUses(SDValue(N, 0), SDValue(Lo, 0));
1331 }
1332 if (!SDValue(N, 1).use_empty()) {
1333 SDValue Sub1 = CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32);
1334 SDNode *Hi = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, SL,
1335 MVT::i32, SDValue(Mad, 0), Sub1);
1336 ReplaceUses(SDValue(N, 1), SDValue(Hi, 0));
1337 }
1338 CurDAG->RemoveDeadNode(N);
1339}
1340
1341bool AMDGPUDAGToDAGISel::isDSOffsetLegal(SDValue Base, unsigned Offset) const {
1342 if (!isUInt<16>(Offset))
1343 return false;
1344
1345 if (!Base || Subtarget->hasUsableDSOffset() ||
1346 Subtarget->unsafeDSOffsetFoldingEnabled())
1347 return true;
1348
1349 // On Southern Islands instruction with a negative base value and an offset
1350 // don't seem to work.
1351 return CurDAG->SignBitIsZero(Base);
1352}
1353
1354bool AMDGPUDAGToDAGISel::SelectDS1Addr1Offset(SDValue Addr, SDValue &Base,
1355 SDValue &Offset) const {
1356 SDLoc DL(Addr);
1357 if (CurDAG->isBaseWithConstantOffset(Addr)) {
1358 SDValue N0 = Addr.getOperand(0);
1359 SDValue N1 = Addr.getOperand(1);
1360 ConstantSDNode *C1 = cast<ConstantSDNode>(N1);
1361 if (isDSOffsetLegal(N0, C1->getSExtValue())) {
1362 // (add n0, c0)
1363 Base = N0;
1364 Offset = CurDAG->getTargetConstant(C1->getZExtValue(), DL, MVT::i16);
1365 return true;
1366 }
1367 } else if (Addr.getOpcode() == ISD::SUB) {
1368 // sub C, x -> add (sub 0, x), C
1369 if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Addr.getOperand(0))) {
1370 int64_t ByteOffset = C->getSExtValue();
1371 if (isDSOffsetLegal(SDValue(), ByteOffset)) {
1372 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i32);
1373
1374 // XXX - This is kind of hacky. Create a dummy sub node so we can check
1375 // the known bits in isDSOffsetLegal. We need to emit the selected node
1376 // here, so this is thrown away.
1377 SDValue Sub = CurDAG->getNode(ISD::SUB, DL, MVT::i32,
1378 Zero, Addr.getOperand(1));
1379
1380 if (isDSOffsetLegal(Sub, ByteOffset)) {
1382 Opnds.push_back(Zero);
1383 Opnds.push_back(Addr.getOperand(1));
1384
1385 // FIXME: Select to VOP3 version for with-carry.
1386 unsigned SubOp = AMDGPU::V_SUB_CO_U32_e32;
1387 if (Subtarget->hasAddNoCarryInsts()) {
1388 SubOp = AMDGPU::V_SUB_U32_e64;
1389 Opnds.push_back(
1390 CurDAG->getTargetConstant(0, {}, MVT::i1)); // clamp bit
1391 }
1392
1393 MachineSDNode *MachineSub =
1394 CurDAG->getMachineNode(SubOp, DL, MVT::i32, Opnds);
1395
1396 Base = SDValue(MachineSub, 0);
1397 Offset = CurDAG->getTargetConstant(ByteOffset, DL, MVT::i16);
1398 return true;
1399 }
1400 }
1401 }
1402 } else if (const ConstantSDNode *CAddr = dyn_cast<ConstantSDNode>(Addr)) {
1403 // If we have a constant address, prefer to put the constant into the
1404 // offset. This can save moves to load the constant address since multiple
1405 // operations can share the zero base address register, and enables merging
1406 // into read2 / write2 instructions.
1407
1408 SDLoc DL(Addr);
1409
1410 if (isDSOffsetLegal(SDValue(), CAddr->getZExtValue())) {
1411 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i32);
1412 MachineSDNode *MovZero = CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32,
1413 DL, MVT::i32, Zero);
1414 Base = SDValue(MovZero, 0);
1415 Offset = CurDAG->getTargetConstant(CAddr->getZExtValue(), DL, MVT::i16);
1416 return true;
1417 }
1418 }
1419
1420 // default case
1421 Base = Addr;
1422 Offset = CurDAG->getTargetConstant(0, SDLoc(Addr), MVT::i16);
1423 return true;
1424}
1425
1426bool AMDGPUDAGToDAGISel::isDSOffset2Legal(SDValue Base, unsigned Offset0,
1427 unsigned Offset1,
1428 unsigned Size) const {
1429 if (Offset0 % Size != 0 || Offset1 % Size != 0)
1430 return false;
1431 if (!isUInt<8>(Offset0 / Size) || !isUInt<8>(Offset1 / Size))
1432 return false;
1433
1434 if (!Base || Subtarget->hasUsableDSOffset() ||
1435 Subtarget->unsafeDSOffsetFoldingEnabled())
1436 return true;
1437
1438 // On Southern Islands instruction with a negative base value and an offset
1439 // don't seem to work.
1440 return CurDAG->SignBitIsZero(Base);
1441}
1442
1443// Return whether the operation has NoUnsignedWrap property.
1444static bool isNoUnsignedWrap(SDValue Addr) {
1445 return (Addr.getOpcode() == ISD::ADD &&
1446 Addr->getFlags().hasNoUnsignedWrap()) ||
1447 Addr->getOpcode() == ISD::OR;
1448}
1449
1450// Check that the base address of flat scratch load/store in the form of `base +
1451// offset` is legal to be put in SGPR/VGPR (i.e. unsigned per hardware
1452// requirement). We always treat the first operand as the base address here.
1453bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegal(SDValue Addr) const {
1454 if (isNoUnsignedWrap(Addr))
1455 return true;
1456
1457 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
1458 // values.
1459 if (Subtarget->hasSignedScratchOffsets())
1460 return true;
1461
1462 auto LHS = Addr.getOperand(0);
1463 auto RHS = Addr.getOperand(1);
1464
1465 // If the immediate offset is negative and within certain range, the base
1466 // address cannot also be negative. If the base is also negative, the sum
1467 // would be either negative or much larger than the valid range of scratch
1468 // memory a thread can access.
1469 ConstantSDNode *ImmOp = nullptr;
1470 if (Addr.getOpcode() == ISD::ADD && (ImmOp = dyn_cast<ConstantSDNode>(RHS))) {
1471 if (ImmOp->getSExtValue() < 0 && ImmOp->getSExtValue() > -0x40000000)
1472 return true;
1473 }
1474
1475 return CurDAG->SignBitIsZero(LHS);
1476}
1477
1478// Check address value in SGPR/VGPR are legal for flat scratch in the form
1479// of: SGPR + VGPR.
1480bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegalSV(SDValue Addr) const {
1481 if (isNoUnsignedWrap(Addr))
1482 return true;
1483
1484 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
1485 // values.
1486 if (Subtarget->hasSignedScratchOffsets())
1487 return true;
1488
1489 auto LHS = Addr.getOperand(0);
1490 auto RHS = Addr.getOperand(1);
1491 return CurDAG->SignBitIsZero(RHS) && CurDAG->SignBitIsZero(LHS);
1492}
1493
1494// Check address value in SGPR/VGPR are legal for flat scratch in the form
1495// of: SGPR + VGPR + Imm.
1496bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegalSVImm(SDValue Addr) const {
1497 // Starting with GFX12, VADDR and SADDR fields in VSCRATCH can use negative
1498 // values.
1499 if (AMDGPU::isGFX12Plus(*Subtarget))
1500 return true;
1501
1502 auto Base = Addr.getOperand(0);
1503 auto *RHSImm = cast<ConstantSDNode>(Addr.getOperand(1));
1504 // If the immediate offset is negative and within certain range, the base
1505 // address cannot also be negative. If the base is also negative, the sum
1506 // would be either negative or much larger than the valid range of scratch
1507 // memory a thread can access.
1508 if (isNoUnsignedWrap(Base) &&
1509 (isNoUnsignedWrap(Addr) ||
1510 (RHSImm->getSExtValue() < 0 && RHSImm->getSExtValue() > -0x40000000)))
1511 return true;
1512
1513 auto LHS = Base.getOperand(0);
1514 auto RHS = Base.getOperand(1);
1515 return CurDAG->SignBitIsZero(RHS) && CurDAG->SignBitIsZero(LHS);
1516}
1517
1518// TODO: If offset is too big, put low 16-bit into offset.
1519bool AMDGPUDAGToDAGISel::SelectDS64Bit4ByteAligned(SDValue Addr, SDValue &Base,
1520 SDValue &Offset0,
1521 SDValue &Offset1) const {
1522 return SelectDSReadWrite2(Addr, Base, Offset0, Offset1, 4);
1523}
1524
1525bool AMDGPUDAGToDAGISel::SelectDS128Bit8ByteAligned(SDValue Addr, SDValue &Base,
1526 SDValue &Offset0,
1527 SDValue &Offset1) const {
1528 return SelectDSReadWrite2(Addr, Base, Offset0, Offset1, 8);
1529}
1530
1531bool AMDGPUDAGToDAGISel::SelectDSReadWrite2(SDValue Addr, SDValue &Base,
1532 SDValue &Offset0, SDValue &Offset1,
1533 unsigned Size) const {
1534 SDLoc DL(Addr);
1535
1536 if (CurDAG->isBaseWithConstantOffset(Addr)) {
1537 SDValue N0 = Addr.getOperand(0);
1538 SDValue N1 = Addr.getOperand(1);
1539 ConstantSDNode *C1 = cast<ConstantSDNode>(N1);
1540 unsigned OffsetValue0 = C1->getZExtValue();
1541 unsigned OffsetValue1 = OffsetValue0 + Size;
1542
1543 // (add n0, c0)
1544 if (isDSOffset2Legal(N0, OffsetValue0, OffsetValue1, Size)) {
1545 Base = N0;
1546 Offset0 = CurDAG->getTargetConstant(OffsetValue0 / Size, DL, MVT::i32);
1547 Offset1 = CurDAG->getTargetConstant(OffsetValue1 / Size, DL, MVT::i32);
1548 return true;
1549 }
1550 } else if (Addr.getOpcode() == ISD::SUB) {
1551 // sub C, x -> add (sub 0, x), C
1552 if (const ConstantSDNode *C =
1554 unsigned OffsetValue0 = C->getZExtValue();
1555 unsigned OffsetValue1 = OffsetValue0 + Size;
1556
1557 if (isDSOffset2Legal(SDValue(), OffsetValue0, OffsetValue1, Size)) {
1558 SDLoc DL(Addr);
1559 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i32);
1560
1561 // XXX - This is kind of hacky. Create a dummy sub node so we can check
1562 // the known bits in isDSOffsetLegal. We need to emit the selected node
1563 // here, so this is thrown away.
1564 SDValue Sub =
1565 CurDAG->getNode(ISD::SUB, DL, MVT::i32, Zero, Addr.getOperand(1));
1566
1567 if (isDSOffset2Legal(Sub, OffsetValue0, OffsetValue1, Size)) {
1569 Opnds.push_back(Zero);
1570 Opnds.push_back(Addr.getOperand(1));
1571 unsigned SubOp = AMDGPU::V_SUB_CO_U32_e32;
1572 if (Subtarget->hasAddNoCarryInsts()) {
1573 SubOp = AMDGPU::V_SUB_U32_e64;
1574 Opnds.push_back(
1575 CurDAG->getTargetConstant(0, {}, MVT::i1)); // clamp bit
1576 }
1577
1578 MachineSDNode *MachineSub = CurDAG->getMachineNode(
1579 SubOp, DL, MVT::getIntegerVT(Size * 8), Opnds);
1580
1581 Base = SDValue(MachineSub, 0);
1582 Offset0 =
1583 CurDAG->getTargetConstant(OffsetValue0 / Size, DL, MVT::i32);
1584 Offset1 =
1585 CurDAG->getTargetConstant(OffsetValue1 / Size, DL, MVT::i32);
1586 return true;
1587 }
1588 }
1589 }
1590 } else if (const ConstantSDNode *CAddr = dyn_cast<ConstantSDNode>(Addr)) {
1591 unsigned OffsetValue0 = CAddr->getZExtValue();
1592 unsigned OffsetValue1 = OffsetValue0 + Size;
1593
1594 if (isDSOffset2Legal(SDValue(), OffsetValue0, OffsetValue1, Size)) {
1595 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i32);
1596 MachineSDNode *MovZero =
1597 CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32, DL, MVT::i32, Zero);
1598 Base = SDValue(MovZero, 0);
1599 Offset0 = CurDAG->getTargetConstant(OffsetValue0 / Size, DL, MVT::i32);
1600 Offset1 = CurDAG->getTargetConstant(OffsetValue1 / Size, DL, MVT::i32);
1601 return true;
1602 }
1603 }
1604
1605 // default case
1606
1607 Base = Addr;
1608 Offset0 = CurDAG->getTargetConstant(0, DL, MVT::i32);
1609 Offset1 = CurDAG->getTargetConstant(1, DL, MVT::i32);
1610 return true;
1611}
1612
1613bool AMDGPUDAGToDAGISel::SelectMUBUF(SDValue Addr, SDValue &Ptr, SDValue &VAddr,
1614 SDValue &SOffset, SDValue &Offset,
1615 SDValue &Offen, SDValue &Idxen,
1616 SDValue &Addr64) const {
1617 // Subtarget prefers to use flat instruction
1618 // FIXME: This should be a pattern predicate and not reach here
1619 if (Subtarget->useFlatForGlobal())
1620 return false;
1621
1622 SDLoc DL(Addr);
1623
1624 Idxen = CurDAG->getTargetConstant(0, DL, MVT::i1);
1625 Offen = CurDAG->getTargetConstant(0, DL, MVT::i1);
1626 Addr64 = CurDAG->getTargetConstant(0, DL, MVT::i1);
1627 SOffset = Subtarget->hasRestrictedSOffset()
1628 ? CurDAG->getRegister(AMDGPU::SGPR_NULL, MVT::i32)
1629 : CurDAG->getTargetConstant(0, DL, MVT::i32);
1630
1631 ConstantSDNode *C1 = nullptr;
1632 SDValue N0 = Addr;
1633 if (CurDAG->isBaseWithConstantOffset(Addr)) {
1634 C1 = cast<ConstantSDNode>(Addr.getOperand(1));
1635 if (isUInt<32>(C1->getZExtValue()))
1636 N0 = Addr.getOperand(0);
1637 else
1638 C1 = nullptr;
1639 }
1640
1641 if (N0->isAnyAdd()) {
1642 // (add N2, N3) -> addr64, or
1643 // (add (add N2, N3), C1) -> addr64
1644 SDValue N2 = N0.getOperand(0);
1645 SDValue N3 = N0.getOperand(1);
1646 Addr64 = CurDAG->getTargetConstant(1, DL, MVT::i1);
1647
1648 if (N2->isDivergent()) {
1649 if (N3->isDivergent()) {
1650 // Both N2 and N3 are divergent. Use N0 (the result of the add) as the
1651 // addr64, and construct the resource from a 0 address.
1652 Ptr = SDValue(buildSMovImm64(DL, 0, MVT::v2i32), 0);
1653 VAddr = N0;
1654 } else {
1655 // N2 is divergent, N3 is not.
1656 Ptr = N3;
1657 VAddr = N2;
1658 }
1659 } else {
1660 // N2 is not divergent.
1661 Ptr = N2;
1662 VAddr = N3;
1663 }
1664 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1665 } else if (N0->isDivergent()) {
1666 // N0 is divergent. Use it as the addr64, and construct the resource from a
1667 // 0 address.
1668 Ptr = SDValue(buildSMovImm64(DL, 0, MVT::v2i32), 0);
1669 VAddr = N0;
1670 Addr64 = CurDAG->getTargetConstant(1, DL, MVT::i1);
1671 } else {
1672 // N0 -> offset, or
1673 // (N0 + C1) -> offset
1674 VAddr = CurDAG->getTargetConstant(0, DL, MVT::i32);
1675 Ptr = N0;
1676 }
1677
1678 if (!C1) {
1679 // No offset.
1680 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1681 return true;
1682 }
1683
1684 const SIInstrInfo *TII = Subtarget->getInstrInfo();
1685 if (TII->isLegalMUBUFImmOffset(C1->getZExtValue())) {
1686 // Legal offset for instruction.
1687 Offset = CurDAG->getTargetConstant(C1->getZExtValue(), DL, MVT::i32);
1688 return true;
1689 }
1690
1691 // Illegal offset, store it in soffset.
1692 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1693 SOffset =
1694 SDValue(CurDAG->getMachineNode(
1695 AMDGPU::S_MOV_B32, DL, MVT::i32,
1696 CurDAG->getTargetConstant(C1->getZExtValue(), DL, MVT::i32)),
1697 0);
1698 return true;
1699}
1700
1701bool AMDGPUDAGToDAGISel::SelectMUBUFAddr64(SDValue Addr, SDValue &SRsrc,
1702 SDValue &VAddr, SDValue &SOffset,
1703 SDValue &Offset) const {
1704 SDValue Ptr, Offen, Idxen, Addr64;
1705
1706 // addr64 bit was removed for volcanic islands.
1707 // FIXME: This should be a pattern predicate and not reach here
1708 if (!Subtarget->hasAddr64())
1709 return false;
1710
1711 if (!SelectMUBUF(Addr, Ptr, VAddr, SOffset, Offset, Offen, Idxen, Addr64))
1712 return false;
1713
1714 ConstantSDNode *C = cast<ConstantSDNode>(Addr64);
1715 if (C->getSExtValue()) {
1716 SDLoc DL(Addr);
1717
1718 const SITargetLowering& Lowering =
1719 *static_cast<const SITargetLowering*>(getTargetLowering());
1720
1721 SRsrc = SDValue(Lowering.wrapAddr64Rsrc(*CurDAG, DL, Ptr), 0);
1722 return true;
1723 }
1724
1725 return false;
1726}
1727
1728std::pair<SDValue, SDValue> AMDGPUDAGToDAGISel::foldFrameIndex(SDValue N) const {
1729 SDLoc DL(N);
1730
1731 auto *FI = dyn_cast<FrameIndexSDNode>(N);
1732 SDValue TFI =
1733 FI ? CurDAG->getTargetFrameIndex(FI->getIndex(), FI->getValueType(0)) : N;
1734
1735 // We rebase the base address into an absolute stack address and hence
1736 // use constant 0 for soffset. This value must be retained until
1737 // frame elimination and eliminateFrameIndex will choose the appropriate
1738 // frame register if need be.
1739 return std::pair(TFI, CurDAG->getTargetConstant(0, DL, MVT::i32));
1740}
1741
1742bool AMDGPUDAGToDAGISel::SelectMUBUFScratchOffen(SDNode *Parent,
1743 SDValue Addr, SDValue &Rsrc,
1744 SDValue &VAddr, SDValue &SOffset,
1745 SDValue &ImmOffset) const {
1746
1747 SDLoc DL(Addr);
1748 MachineFunction &MF = CurDAG->getMachineFunction();
1749 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
1750
1751 Rsrc = CurDAG->getRegister(Info->getScratchRSrcReg(), MVT::v4i32);
1752
1753 if (ConstantSDNode *CAddr = dyn_cast<ConstantSDNode>(Addr)) {
1754 int64_t Imm = CAddr->getSExtValue();
1755 const int64_t NullPtr =
1757 // Don't fold null pointer.
1758 if (Imm != NullPtr) {
1759 const int64_t MaxOffset = SIInstrInfo::getMaxMUBUFImmOffset(*Subtarget);
1760 SDValue HighBits =
1761 CurDAG->getTargetConstant(Imm & ~MaxOffset, DL, MVT::i32);
1762 MachineSDNode *MovHighBits = CurDAG->getMachineNode(
1763 AMDGPU::V_MOV_B32_e32, DL, MVT::i32, HighBits);
1764 VAddr = SDValue(MovHighBits, 0);
1765
1766 SOffset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1767 ImmOffset = CurDAG->getTargetConstant(Imm & MaxOffset, DL, MVT::i32);
1768 return true;
1769 }
1770 }
1771
1772 if (CurDAG->isBaseWithConstantOffset(Addr)) {
1773 // (add n0, c1)
1774
1775 SDValue N0 = Addr.getOperand(0);
1776 uint64_t C1 = Addr.getConstantOperandVal(1);
1777
1778 // Offsets in vaddr must be positive if range checking is enabled.
1779 //
1780 // The total computation of vaddr + soffset + offset must not overflow. If
1781 // vaddr is negative, even if offset is 0 the sgpr offset add will end up
1782 // overflowing.
1783 //
1784 // Prior to gfx9, MUBUF instructions with the vaddr offset enabled would
1785 // always perform a range check. If a negative vaddr base index was used,
1786 // this would fail the range check. The overall address computation would
1787 // compute a valid address, but this doesn't happen due to the range
1788 // check. For out-of-bounds MUBUF loads, a 0 is returned.
1789 //
1790 // Therefore it should be safe to fold any VGPR offset on gfx9 into the
1791 // MUBUF vaddr, but not on older subtargets which can only do this if the
1792 // sign bit is known 0.
1793 const SIInstrInfo *TII = Subtarget->getInstrInfo();
1794 if (TII->isLegalMUBUFImmOffset(C1) &&
1795 (!Subtarget->privateMemoryResourceIsRangeChecked() ||
1796 CurDAG->SignBitIsZero(N0))) {
1797 std::tie(VAddr, SOffset) = foldFrameIndex(N0);
1798 ImmOffset = CurDAG->getTargetConstant(C1, DL, MVT::i32);
1799 return true;
1800 }
1801 }
1802
1803 // (node)
1804 std::tie(VAddr, SOffset) = foldFrameIndex(Addr);
1805 ImmOffset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1806 return true;
1807}
1808
1809static bool IsCopyFromSGPR(const SIRegisterInfo &TRI, SDValue Val) {
1810 if (Val.getOpcode() != ISD::CopyFromReg)
1811 return false;
1812 auto Reg = cast<RegisterSDNode>(Val.getOperand(1))->getReg();
1813 if (!Reg.isPhysical())
1814 return false;
1815 const auto *RC = TRI.getPhysRegBaseClass(Reg);
1816 return RC && TRI.isSGPRClass(RC);
1817}
1818
1819bool AMDGPUDAGToDAGISel::SelectMUBUFScratchOffset(SDNode *Parent,
1820 SDValue Addr,
1821 SDValue &SRsrc,
1822 SDValue &SOffset,
1823 SDValue &Offset) const {
1824 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
1825 const SIInstrInfo *TII = Subtarget->getInstrInfo();
1826 MachineFunction &MF = CurDAG->getMachineFunction();
1827 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
1828 SDLoc DL(Addr);
1829
1830 // CopyFromReg <sgpr>
1831 if (IsCopyFromSGPR(*TRI, Addr)) {
1832 SRsrc = CurDAG->getRegister(Info->getScratchRSrcReg(), MVT::v4i32);
1833 SOffset = Addr;
1834 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1835 return true;
1836 }
1837
1838 ConstantSDNode *CAddr;
1839 if (Addr.getOpcode() == ISD::ADD) {
1840 // Add (CopyFromReg <sgpr>) <constant>
1841 CAddr = dyn_cast<ConstantSDNode>(Addr.getOperand(1));
1842 if (!CAddr || !TII->isLegalMUBUFImmOffset(CAddr->getZExtValue()))
1843 return false;
1844 if (!IsCopyFromSGPR(*TRI, Addr.getOperand(0)))
1845 return false;
1846
1847 SOffset = Addr.getOperand(0);
1848 } else if ((CAddr = dyn_cast<ConstantSDNode>(Addr)) &&
1849 TII->isLegalMUBUFImmOffset(CAddr->getZExtValue())) {
1850 // <constant>
1851 SOffset = CurDAG->getTargetConstant(0, DL, MVT::i32);
1852 } else {
1853 return false;
1854 }
1855
1856 SRsrc = CurDAG->getRegister(Info->getScratchRSrcReg(), MVT::v4i32);
1857
1858 Offset = CurDAG->getTargetConstant(CAddr->getZExtValue(), DL, MVT::i32);
1859 return true;
1860}
1861
1862bool AMDGPUDAGToDAGISel::SelectMUBUFOffset(SDValue Addr, SDValue &SRsrc,
1863 SDValue &SOffset, SDValue &Offset
1864 ) const {
1865 SDValue Ptr, VAddr, Offen, Idxen, Addr64;
1866 const SIInstrInfo *TII = Subtarget->getInstrInfo();
1867
1868 if (!SelectMUBUF(Addr, Ptr, VAddr, SOffset, Offset, Offen, Idxen, Addr64))
1869 return false;
1870
1871 if (!cast<ConstantSDNode>(Offen)->getSExtValue() &&
1872 !cast<ConstantSDNode>(Idxen)->getSExtValue() &&
1873 !cast<ConstantSDNode>(Addr64)->getSExtValue()) {
1874 uint64_t Rsrc = TII->getDefaultRsrcDataFormat() |
1875 maskTrailingOnes<uint64_t>(32); // Size
1876 SDLoc DL(Addr);
1877
1878 const SITargetLowering& Lowering =
1879 *static_cast<const SITargetLowering*>(getTargetLowering());
1880
1881 SRsrc = SDValue(Lowering.buildRSRC(*CurDAG, DL, Ptr, 0, Rsrc), 0);
1882 return true;
1883 }
1884 return false;
1885}
1886
1887bool AMDGPUDAGToDAGISel::SelectBUFSOffset(SDValue ByteOffsetNode,
1888 SDValue &SOffset) const {
1889 if (Subtarget->hasRestrictedSOffset() && isNullConstant(ByteOffsetNode)) {
1890 SOffset = CurDAG->getRegister(AMDGPU::SGPR_NULL, MVT::i32);
1891 return true;
1892 }
1893
1894 SOffset = ByteOffsetNode;
1895 return true;
1896}
1897
1898// Find a load or store from corresponding pattern root.
1899// Roots may be build_vector, bitconvert or their combinations.
1902 if (MemSDNode *MN = dyn_cast<MemSDNode>(N))
1903 return MN;
1905 for (SDValue V : N->op_values())
1906 if (MemSDNode *MN =
1908 return MN;
1909 llvm_unreachable("cannot find MemSDNode in the pattern!");
1910}
1911
1912bool AMDGPUDAGToDAGISel::SelectFlatOffsetImpl(
1913 SDNode *N, SDValue Addr, SDValue &VAddr, SDValue &Offset,
1914 AMDGPU::FlatAddrSpace FlatVariant) const {
1916 int64_t OffsetVal = 0;
1917
1918 unsigned AS = findMemSDNode(N)->getAddressSpace();
1919
1920 bool CanHaveFlatSegmentOffsetBug =
1921 Subtarget->hasFlatSegmentOffsetBug() &&
1922 FlatVariant == FlatAddrSpace::FLAT &&
1924
1925 if (Subtarget->hasFlatInstOffsets() && !CanHaveFlatSegmentOffsetBug) {
1926 SDValue N0, N1;
1927 if (isBaseWithConstantOffset64(Addr, N0, N1) &&
1928 (FlatVariant != FlatAddrSpace::FlatScratch ||
1929 isFlatScratchBaseLegal(Addr))) {
1930 int64_t COffsetVal = cast<ConstantSDNode>(N1)->getSExtValue();
1931
1932 // Adding the offset to the base address in a FLAT instruction must not
1933 // change the memory aperture in which the address falls. Therefore we can
1934 // only fold offsets from inbounds GEPs into FLAT instructions.
1935 bool IsInBounds =
1936 Addr.getOpcode() == ISD::PTRADD && Addr->getFlags().hasInBounds();
1937 if (COffsetVal == 0 || FlatVariant != FlatAddrSpace::FLAT || IsInBounds) {
1938 const SIInstrInfo *TII = Subtarget->getInstrInfo();
1939 if (TII->isLegalFLATOffset(COffsetVal, AS, FlatVariant)) {
1940 Addr = N0;
1941 OffsetVal = COffsetVal;
1942 } else {
1943 // If the offset doesn't fit, put the low bits into the offset field
1944 // and add the rest.
1945 //
1946 // For a FLAT instruction the hardware decides whether to access
1947 // global/scratch/shared memory based on the high bits of vaddr,
1948 // ignoring the offset field, so we have to ensure that when we add
1949 // remainder to vaddr it still points into the same underlying object.
1950 // The easiest way to do that is to make sure that we split the offset
1951 // into two pieces that are both >= 0 or both <= 0.
1952
1953 SDLoc DL(N);
1954 uint64_t RemainderOffset;
1955
1956 std::tie(OffsetVal, RemainderOffset) =
1957 TII->splitFlatOffset(COffsetVal, AS, FlatVariant);
1958
1959 SDValue AddOffsetLo =
1960 getMaterializedScalarImm32(Lo_32(RemainderOffset), DL);
1961 SDValue Clamp = CurDAG->getTargetConstant(0, DL, MVT::i1);
1962
1963 if (Addr.getValueType().getSizeInBits() == 32) {
1965 Opnds.push_back(N0);
1966 Opnds.push_back(AddOffsetLo);
1967 unsigned AddOp = AMDGPU::V_ADD_CO_U32_e32;
1968 if (Subtarget->hasAddNoCarryInsts()) {
1969 AddOp = AMDGPU::V_ADD_U32_e64;
1970 Opnds.push_back(Clamp);
1971 }
1972 Addr =
1973 SDValue(CurDAG->getMachineNode(AddOp, DL, MVT::i32, Opnds), 0);
1974 } else {
1975 // TODO: Should this try to use a scalar add pseudo if the base
1976 // address is uniform and saddr is usable?
1977 SDValue Sub0 =
1978 CurDAG->getTargetConstant(AMDGPU::sub0, DL, MVT::i32);
1979 SDValue Sub1 =
1980 CurDAG->getTargetConstant(AMDGPU::sub1, DL, MVT::i32);
1981
1982 SDNode *N0Lo = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
1983 DL, MVT::i32, N0, Sub0);
1984 SDNode *N0Hi = CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
1985 DL, MVT::i32, N0, Sub1);
1986
1987 SDValue AddOffsetHi =
1988 getMaterializedScalarImm32(Hi_32(RemainderOffset), DL);
1989
1990 SDVTList VTs = CurDAG->getVTList(MVT::i32, MVT::i1);
1991
1992 SDNode *Add =
1993 CurDAG->getMachineNode(AMDGPU::V_ADD_CO_U32_e64, DL, VTs,
1994 {AddOffsetLo, SDValue(N0Lo, 0), Clamp});
1995
1996 SDNode *Addc = CurDAG->getMachineNode(
1997 AMDGPU::V_ADDC_U32_e64, DL, VTs,
1998 {AddOffsetHi, SDValue(N0Hi, 0), SDValue(Add, 1), Clamp});
1999
2000 SDValue RegSequenceArgs[] = {
2001 CurDAG->getTargetConstant(AMDGPU::VReg_64RegClassID, DL,
2002 MVT::i32),
2003 SDValue(Add, 0), Sub0, SDValue(Addc, 0), Sub1};
2004
2005 Addr = SDValue(CurDAG->getMachineNode(AMDGPU::REG_SEQUENCE, DL,
2006 MVT::i64, RegSequenceArgs),
2007 0);
2008 }
2009 }
2010 }
2011 }
2012 }
2013
2014 VAddr = Addr;
2015 Offset = CurDAG->getSignedTargetConstant(OffsetVal, SDLoc(), MVT::i32);
2016 return true;
2017}
2018
2019bool AMDGPUDAGToDAGISel::SelectFlatOffset(SDNode *N, SDValue Addr,
2020 SDValue &VAddr,
2021 SDValue &Offset) const {
2022 return SelectFlatOffsetImpl(N, Addr, VAddr, Offset,
2024}
2025
2026bool AMDGPUDAGToDAGISel::SelectGlobalOffset(SDNode *N, SDValue Addr,
2027 SDValue &VAddr,
2028 SDValue &Offset) const {
2029 return SelectFlatOffsetImpl(N, Addr, VAddr, Offset,
2031}
2032
2033bool AMDGPUDAGToDAGISel::SelectScratchOffset(SDNode *N, SDValue Addr,
2034 SDValue &VAddr,
2035 SDValue &Offset) const {
2036 return SelectFlatOffsetImpl(N, Addr, VAddr, Offset,
2038}
2039
2040// If this matches *_extend i32:x, return x
2041// Otherwise if the value is I32 returns x.
2043 const SelectionDAG *DAG) {
2044 if (Op.getValueType() == MVT::i32)
2045 return Op;
2046
2047 if (Op.getOpcode() != (IsSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND) &&
2048 Op.getOpcode() != ISD::ANY_EXTEND &&
2049 !(Op.getOpcode() == (IsSigned ? ISD::ZERO_EXTEND : ISD::SIGN_EXTEND) &&
2050 DAG->SignBitIsZero(Op.getOperand(0))))
2051 return SDValue();
2052
2053 SDValue ExtSrc = Op.getOperand(0);
2054 return (ExtSrc.getValueType() == MVT::i32) ? ExtSrc : SDValue();
2055}
2056
2057// Match (64-bit SGPR base) + (zext vgpr offset) + sext(imm offset)
2058// or (64-bit SGPR base) + (sext vgpr offset) + sext(imm offset)
2059bool AMDGPUDAGToDAGISel::SelectGlobalSAddr(SDNode *N, SDValue Addr,
2060 SDValue &SAddr, SDValue &VOffset,
2061 SDValue &Offset, bool &ScaleOffset,
2062 bool NeedIOffset) const {
2064 int64_t ImmOffset = 0;
2065 ScaleOffset = false;
2066
2067 // Match the immediate offset first, which canonically is moved as low as
2068 // possible.
2069
2070 SDValue LHS, RHS;
2071 if (isBaseWithConstantOffset64(Addr, LHS, RHS)) {
2072 int64_t COffsetVal = cast<ConstantSDNode>(RHS)->getSExtValue();
2073 const SIInstrInfo *TII = Subtarget->getInstrInfo();
2074
2075 if (NeedIOffset &&
2076 TII->isLegalFLATOffset(COffsetVal, AMDGPUAS::GLOBAL_ADDRESS,
2077 FlatAddrSpace::FlatGlobal)) {
2078 Addr = LHS;
2079 ImmOffset = COffsetVal;
2080 } else if (!LHS->isDivergent()) {
2081 if (COffsetVal > 0) {
2082 SDLoc SL(N);
2083 // saddr + large_offset -> saddr +
2084 // (voffset = large_offset & ~MaxOffset) +
2085 // (large_offset & MaxOffset);
2086 int64_t SplitImmOffset = 0, RemainderOffset = COffsetVal;
2087 if (NeedIOffset) {
2088 std::tie(SplitImmOffset, RemainderOffset) = TII->splitFlatOffset(
2089 COffsetVal, AMDGPUAS::GLOBAL_ADDRESS, FlatAddrSpace::FlatGlobal);
2090 }
2091
2092 if (Subtarget->hasSignedGVSOffset() ? isInt<32>(RemainderOffset)
2093 : isUInt<32>(RemainderOffset)) {
2094 SDNode *VMov = CurDAG->getMachineNode(
2095 AMDGPU::V_MOV_B32_e32, SL, MVT::i32,
2096 CurDAG->getTargetConstant(RemainderOffset, SDLoc(), MVT::i32));
2097 VOffset = SDValue(VMov, 0);
2098 SAddr = LHS;
2099 Offset = CurDAG->getTargetConstant(SplitImmOffset, SDLoc(), MVT::i32);
2100 return true;
2101 }
2102 }
2103
2104 // We are adding a 64 bit SGPR and a constant. If constant bus limit
2105 // is 1 we would need to perform 1 or 2 extra moves for each half of
2106 // the constant and it is better to do a scalar add and then issue a
2107 // single VALU instruction to materialize zero. Otherwise it is less
2108 // instructions to perform VALU adds with immediates or inline literals.
2109 unsigned NumLiterals =
2110 !TII->isInlineConstant(APInt(32, Lo_32(COffsetVal))) +
2111 !TII->isInlineConstant(APInt(32, Hi_32(COffsetVal)));
2112 if (Subtarget->getConstantBusLimit(AMDGPU::V_ADD_U32_e64) > NumLiterals)
2113 return false;
2114 }
2115 }
2116
2117 // Match the variable offset.
2118 if (Addr->isAnyAdd()) {
2119 LHS = Addr.getOperand(0);
2120
2121 if (!LHS->isDivergent()) {
2122 // add (i64 sgpr), (*_extend (i32 vgpr))
2123 RHS = Addr.getOperand(1);
2124 ScaleOffset = SelectScaleOffset(N, RHS, Subtarget->hasSignedGVSOffset());
2125 if (SDValue ExtRHS = matchExtFromI32orI32(
2126 RHS, Subtarget->hasSignedGVSOffset(), CurDAG)) {
2127 SAddr = LHS;
2128 VOffset = ExtRHS;
2129 }
2130 }
2131
2132 RHS = Addr.getOperand(1);
2133 if (!SAddr && !RHS->isDivergent()) {
2134 // add (*_extend (i32 vgpr)), (i64 sgpr)
2135 ScaleOffset = SelectScaleOffset(N, LHS, Subtarget->hasSignedGVSOffset());
2136 if (SDValue ExtLHS = matchExtFromI32orI32(
2137 LHS, Subtarget->hasSignedGVSOffset(), CurDAG)) {
2138 SAddr = RHS;
2139 VOffset = ExtLHS;
2140 }
2141 }
2142
2143 if (SAddr) {
2144 Offset = CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2145 return true;
2146 }
2147 }
2148
2149 if (Subtarget->hasScaleOffset() &&
2150 (Addr.getOpcode() == (Subtarget->hasSignedGVSOffset()
2153 (Addr.getOpcode() == AMDGPUISD::MAD_U64_U32 &&
2154 CurDAG->SignBitIsZero(Addr.getOperand(0)))) &&
2155 Addr.getOperand(0)->isDivergent() &&
2157 !Addr.getOperand(2)->isDivergent()) {
2158 // mad_u64_u32 (i32 vgpr), (i32 c), (i64 sgpr)
2159 unsigned Size =
2160 (unsigned)cast<MemSDNode>(N)->getMemoryVT().getFixedSizeInBits() / 8;
2161 ScaleOffset = Addr.getConstantOperandVal(1) == Size;
2162 if (ScaleOffset) {
2163 SAddr = Addr.getOperand(2);
2164 VOffset = Addr.getOperand(0);
2165 Offset = CurDAG->getTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2166 return true;
2167 }
2168 }
2169
2170 if (Addr->isDivergent() || Addr.isUndef() || isa<ConstantSDNode>(Addr))
2171 return false;
2172
2173 // It's cheaper to materialize a single 32-bit zero for vaddr than the two
2174 // moves required to copy a 64-bit SGPR to VGPR.
2175 SAddr = Addr;
2176 SDNode *VMov =
2177 CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32, SDLoc(Addr), MVT::i32,
2178 CurDAG->getTargetConstant(0, SDLoc(), MVT::i32));
2179 VOffset = SDValue(VMov, 0);
2180 Offset = CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2181 return true;
2182}
2183
2184bool AMDGPUDAGToDAGISel::SelectGlobalSAddr(SDNode *N, SDValue Addr,
2185 SDValue &SAddr, SDValue &VOffset,
2186 SDValue &Offset,
2187 SDValue &CPol) const {
2188 bool ScaleOffset;
2189 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, Offset, ScaleOffset))
2190 return false;
2191
2192 CPol = CurDAG->getTargetConstant(ScaleOffset ? AMDGPU::CPol::SCAL : 0,
2193 SDLoc(), MVT::i32);
2194 return true;
2195}
2196
2197bool AMDGPUDAGToDAGISel::SelectGlobalSAddrCPol(SDNode *N, SDValue Addr,
2198 SDValue &SAddr, SDValue &VOffset,
2199 SDValue &Offset,
2200 SDValue &CPol) const {
2201 bool ScaleOffset;
2202 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, Offset, ScaleOffset))
2203 return false;
2204
2205 // We are assuming CPol is always the last operand of the intrinsic.
2206 auto PassedCPol =
2207 N->getConstantOperandVal(N->getNumOperands() - 1) & ~AMDGPU::CPol::SCAL;
2208 CPol = CurDAG->getTargetConstant(
2209 (ScaleOffset ? AMDGPU::CPol::SCAL : 0) | PassedCPol, SDLoc(), MVT::i32);
2210 return true;
2211}
2212
2213bool AMDGPUDAGToDAGISel::SelectGlobalSAddrCPolM0(SDNode *N, SDValue Addr,
2214 SDValue &SAddr,
2215 SDValue &VOffset,
2216 SDValue &Offset,
2217 SDValue &CPol) const {
2218 bool ScaleOffset;
2219 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, Offset, ScaleOffset))
2220 return false;
2221
2222 // We are assuming CPol is second from last operand of the intrinsic.
2223 auto PassedCPol =
2224 N->getConstantOperandVal(N->getNumOperands() - 2) & ~AMDGPU::CPol::SCAL;
2225 CPol = CurDAG->getTargetConstant(
2226 (ScaleOffset ? AMDGPU::CPol::SCAL : 0) | PassedCPol, SDLoc(), MVT::i32);
2227 return true;
2228}
2229
2230bool AMDGPUDAGToDAGISel::SelectGlobalSAddrGLC(SDNode *N, SDValue Addr,
2231 SDValue &SAddr, SDValue &VOffset,
2232 SDValue &Offset,
2233 SDValue &CPol) const {
2234 bool ScaleOffset;
2235 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, Offset, ScaleOffset))
2236 return false;
2237
2238 unsigned CPolVal = (ScaleOffset ? AMDGPU::CPol::SCAL : 0) | AMDGPU::CPol::GLC;
2239 CPol = CurDAG->getTargetConstant(CPolVal, SDLoc(), MVT::i32);
2240 return true;
2241}
2242
2243bool AMDGPUDAGToDAGISel::SelectGlobalSAddrNoIOffset(SDNode *N, SDValue Addr,
2244 SDValue &SAddr,
2245 SDValue &VOffset,
2246 SDValue &CPol) const {
2247 bool ScaleOffset;
2248 SDValue DummyOffset;
2249 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, DummyOffset, ScaleOffset,
2250 false))
2251 return false;
2252
2253 // We are assuming CPol is always the last operand of the intrinsic.
2254 auto PassedCPol =
2255 N->getConstantOperandVal(N->getNumOperands() - 1) & ~AMDGPU::CPol::SCAL;
2256 CPol = CurDAG->getTargetConstant(
2257 (ScaleOffset ? AMDGPU::CPol::SCAL : 0) | PassedCPol, SDLoc(), MVT::i32);
2258 return true;
2259}
2260
2261bool AMDGPUDAGToDAGISel::SelectGlobalSAddrNoIOffsetM0(SDNode *N, SDValue Addr,
2262 SDValue &SAddr,
2263 SDValue &VOffset,
2264 SDValue &CPol) const {
2265 bool ScaleOffset;
2266 SDValue DummyOffset;
2267 if (!SelectGlobalSAddr(N, Addr, SAddr, VOffset, DummyOffset, ScaleOffset,
2268 false))
2269 return false;
2270
2271 // We are assuming CPol is second from last operand of the intrinsic.
2272 auto PassedCPol =
2273 N->getConstantOperandVal(N->getNumOperands() - 2) & ~AMDGPU::CPol::SCAL;
2274 CPol = CurDAG->getTargetConstant(
2275 (ScaleOffset ? AMDGPU::CPol::SCAL : 0) | PassedCPol, SDLoc(), MVT::i32);
2276 return true;
2277}
2278
2280 if (auto *FI = dyn_cast<FrameIndexSDNode>(SAddr)) {
2281 SAddr = CurDAG->getTargetFrameIndex(FI->getIndex(), FI->getValueType(0));
2282 } else if (SAddr.getOpcode() == ISD::ADD &&
2284 // Materialize this into a scalar move for scalar address to avoid
2285 // readfirstlane.
2286 auto *FI = cast<FrameIndexSDNode>(SAddr.getOperand(0));
2287 SDValue TFI = CurDAG->getTargetFrameIndex(FI->getIndex(),
2288 FI->getValueType(0));
2289 SAddr = SDValue(CurDAG->getMachineNode(AMDGPU::S_ADD_I32, SDLoc(SAddr),
2290 MVT::i32, TFI, SAddr.getOperand(1)),
2291 0);
2292 }
2293
2294 return SAddr;
2295}
2296
2297// Match (32-bit SGPR base) + sext(imm offset)
2298bool AMDGPUDAGToDAGISel::SelectScratchSAddr(SDNode *Parent, SDValue Addr,
2299 SDValue &SAddr,
2300 SDValue &Offset) const {
2302 if (Addr->isDivergent())
2303 return false;
2304
2305 SDLoc DL(Addr);
2306
2307 int64_t COffsetVal = 0;
2308
2309 if (CurDAG->isBaseWithConstantOffset(Addr) && isFlatScratchBaseLegal(Addr)) {
2310 COffsetVal = cast<ConstantSDNode>(Addr.getOperand(1))->getSExtValue();
2311 SAddr = Addr.getOperand(0);
2312 } else {
2313 SAddr = Addr;
2314 }
2315
2316 SAddr = SelectSAddrFI(CurDAG, SAddr);
2317
2318 const SIInstrInfo *TII = Subtarget->getInstrInfo();
2319
2320 if (!TII->isLegalFLATOffset(COffsetVal, AMDGPUAS::PRIVATE_ADDRESS,
2321 FlatAddrSpace::FlatScratch)) {
2322 int64_t SplitImmOffset, RemainderOffset;
2323 std::tie(SplitImmOffset, RemainderOffset) = TII->splitFlatOffset(
2324 COffsetVal, AMDGPUAS::PRIVATE_ADDRESS, FlatAddrSpace::FlatScratch);
2325
2326 COffsetVal = SplitImmOffset;
2327
2328 SDValue AddOffset =
2330 ? getMaterializedScalarImm32(Lo_32(RemainderOffset), DL)
2331 : CurDAG->getSignedTargetConstant(RemainderOffset, DL, MVT::i32);
2332 SAddr = SDValue(CurDAG->getMachineNode(AMDGPU::S_ADD_I32, DL, MVT::i32,
2333 SAddr, AddOffset),
2334 0);
2335 }
2336
2337 Offset = CurDAG->getSignedTargetConstant(COffsetVal, DL, MVT::i32);
2338
2339 return true;
2340}
2341
2342// Check whether the flat scratch SVS swizzle bug affects this access.
2343bool AMDGPUDAGToDAGISel::checkFlatScratchSVSSwizzleBug(
2344 SDValue VAddr, SDValue SAddr, uint64_t ImmOffset) const {
2345 if (!Subtarget->hasFlatScratchSVSSwizzleBug())
2346 return false;
2347
2348 // The bug affects the swizzling of SVS accesses if there is any carry out
2349 // from the two low order bits (i.e. from bit 1 into bit 2) when adding
2350 // voffset to (soffset + inst_offset).
2351 KnownBits VKnown = CurDAG->computeKnownBits(VAddr);
2352 KnownBits SKnown =
2353 KnownBits::add(CurDAG->computeKnownBits(SAddr),
2354 KnownBits::makeConstant(APInt(32, ImmOffset,
2355 /*isSigned=*/true)));
2356 uint64_t VMax = VKnown.getMaxValue().getZExtValue();
2358 return (VMax & 3) + (SMax & 3) >= 4;
2359}
2360
2361bool AMDGPUDAGToDAGISel::SelectScratchSVAddr(SDNode *N, SDValue Addr,
2362 SDValue &VAddr, SDValue &SAddr,
2363 SDValue &Offset,
2364 SDValue &CPol) const {
2365 int64_t ImmOffset = 0;
2366
2367 SDValue LHS, RHS;
2368 SDValue OrigAddr = Addr;
2369 if (isBaseWithConstantOffset64(Addr, LHS, RHS)) {
2370 int64_t COffsetVal = cast<ConstantSDNode>(RHS)->getSExtValue();
2371 const SIInstrInfo *TII = Subtarget->getInstrInfo();
2372
2373 if (TII->isLegalFLATOffset(COffsetVal, AMDGPUAS::PRIVATE_ADDRESS,
2375 Addr = LHS;
2376 ImmOffset = COffsetVal;
2377 } else if (!LHS->isDivergent() && COffsetVal > 0) {
2378 SDLoc SL(N);
2379 // saddr + large_offset -> saddr + (vaddr = large_offset & ~MaxOffset) +
2380 // (large_offset & MaxOffset);
2381 int64_t SplitImmOffset, RemainderOffset;
2382 std::tie(SplitImmOffset, RemainderOffset) =
2383 TII->splitFlatOffset(COffsetVal, AMDGPUAS::PRIVATE_ADDRESS,
2385
2386 if (isUInt<32>(RemainderOffset)) {
2387 SDNode *VMov = CurDAG->getMachineNode(
2388 AMDGPU::V_MOV_B32_e32, SL, MVT::i32,
2389 CurDAG->getTargetConstant(RemainderOffset, SDLoc(), MVT::i32));
2390 VAddr = SDValue(VMov, 0);
2391 SAddr = LHS;
2392 if (!isFlatScratchBaseLegal(Addr))
2393 return false;
2394 if (checkFlatScratchSVSSwizzleBug(VAddr, SAddr, SplitImmOffset))
2395 return false;
2396 Offset = CurDAG->getTargetConstant(SplitImmOffset, SDLoc(), MVT::i32);
2397 CPol = CurDAG->getTargetConstant(0, SDLoc(), MVT::i32);
2398 return true;
2399 }
2400 }
2401 }
2402
2403 if (Addr.getOpcode() != ISD::ADD)
2404 return false;
2405
2406 LHS = Addr.getOperand(0);
2407 RHS = Addr.getOperand(1);
2408
2409 if (!LHS->isDivergent() && RHS->isDivergent()) {
2410 SAddr = LHS;
2411 VAddr = RHS;
2412 } else if (!RHS->isDivergent() && LHS->isDivergent()) {
2413 SAddr = RHS;
2414 VAddr = LHS;
2415 } else {
2416 return false;
2417 }
2418
2419 if (OrigAddr != Addr) {
2420 if (!isFlatScratchBaseLegalSVImm(OrigAddr))
2421 return false;
2422 } else {
2423 if (!isFlatScratchBaseLegalSV(OrigAddr))
2424 return false;
2425 }
2426
2427 if (checkFlatScratchSVSSwizzleBug(VAddr, SAddr, ImmOffset))
2428 return false;
2429 SAddr = SelectSAddrFI(CurDAG, SAddr);
2430 Offset = CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2431
2432 bool ScaleOffset = SelectScaleOffset(N, VAddr, true /* IsSigned */);
2433 CPol = CurDAG->getTargetConstant(ScaleOffset ? AMDGPU::CPol::SCAL : 0,
2434 SDLoc(), MVT::i32);
2435 return true;
2436}
2437
2438// For unbuffered smem loads, it is illegal for the Immediate Offset to be
2439// negative if the resulting (Offset + (M0 or SOffset or zero) is negative.
2440// Handle the case where the Immediate Offset + SOffset is negative.
2441bool AMDGPUDAGToDAGISel::isSOffsetLegalWithImmOffset(SDValue *SOffset,
2442 bool Imm32Only,
2443 bool IsBuffer,
2444 int64_t ImmOffset) const {
2445 if (!IsBuffer && !Imm32Only && ImmOffset < 0 &&
2446 AMDGPU::hasSMRDSignedImmOffset(*Subtarget)) {
2447 KnownBits SKnown = CurDAG->computeKnownBits(*SOffset);
2448 if (ImmOffset + SKnown.getMinValue().getSExtValue() < 0)
2449 return false;
2450 }
2451
2452 return true;
2453}
2454
2455// Given \p Offset and load node \p N check if an \p Offset is a multiple of
2456// the load byte size. If it is update \p Offset to a pre-scaled value and
2457// return true.
2458bool AMDGPUDAGToDAGISel::SelectScaleOffset(SDNode *N, SDValue &Offset,
2459 bool IsSigned) const {
2460 bool ScaleOffset = false;
2461 if (!Subtarget->hasScaleOffset() || !Offset)
2462 return false;
2463
2464 unsigned Size =
2465 (unsigned)cast<MemSDNode>(N)->getMemoryVT().getFixedSizeInBits() / 8;
2466
2467 SDValue Off = Offset;
2468 if (SDValue Ext = matchExtFromI32orI32(Offset, IsSigned, CurDAG))
2469 Off = Ext;
2470
2471 if (isPowerOf2_32(Size) && Off.getOpcode() == ISD::SHL) {
2472 if (auto *C = dyn_cast<ConstantSDNode>(Off.getOperand(1)))
2473 ScaleOffset = C->getZExtValue() == Log2_32(Size);
2474 } else if (Offset.getOpcode() == ISD::MUL ||
2475 (IsSigned && Offset.getOpcode() == AMDGPUISD::MUL_I24) ||
2476 Offset.getOpcode() == AMDGPUISD::MUL_U24 ||
2477 (Offset.isMachineOpcode() &&
2478 Offset.getMachineOpcode() ==
2479 (IsSigned ? AMDGPU::S_MUL_I64_I32_PSEUDO
2480 : AMDGPU::S_MUL_U64_U32_PSEUDO))) {
2481 if (auto *C = dyn_cast<ConstantSDNode>(Offset.getOperand(1)))
2482 ScaleOffset = C->getZExtValue() == Size;
2483 }
2484
2485 if (ScaleOffset)
2486 Offset = Off.getOperand(0);
2487
2488 return ScaleOffset;
2489}
2490
2491// Match an immediate (if Offset is not null) or an SGPR (if SOffset is
2492// not null) offset. If Imm32Only is true, match only 32-bit immediate
2493// offsets available on CI.
2494bool AMDGPUDAGToDAGISel::SelectSMRDOffset(SDNode *N, SDValue ByteOffsetNode,
2495 SDValue *SOffset, SDValue *Offset,
2496 bool Imm32Only, bool IsBuffer,
2497 bool HasSOffset, int64_t ImmOffset,
2498 bool *ScaleOffset) const {
2499 assert((!SOffset || !Offset) &&
2500 "Cannot match both soffset and offset at the same time!");
2501
2502 if (ScaleOffset) {
2503 assert(N && SOffset);
2504
2505 *ScaleOffset = SelectScaleOffset(N, ByteOffsetNode, false /* IsSigned */);
2506 }
2507
2508 ConstantSDNode *C = dyn_cast<ConstantSDNode>(ByteOffsetNode);
2509 if (!C) {
2510 if (!SOffset)
2511 return false;
2512
2513 if (ByteOffsetNode.getValueType().isScalarInteger() &&
2514 ByteOffsetNode.getValueType().getSizeInBits() == 32) {
2515 *SOffset = ByteOffsetNode;
2516 return isSOffsetLegalWithImmOffset(SOffset, Imm32Only, IsBuffer,
2517 ImmOffset);
2518 }
2519 if (ByteOffsetNode.getOpcode() == ISD::ZERO_EXTEND) {
2520 if (ByteOffsetNode.getOperand(0).getValueType().getSizeInBits() == 32) {
2521 *SOffset = ByteOffsetNode.getOperand(0);
2522 return isSOffsetLegalWithImmOffset(SOffset, Imm32Only, IsBuffer,
2523 ImmOffset);
2524 }
2525 }
2526 return false;
2527 }
2528
2529 SDLoc SL(ByteOffsetNode);
2530
2531 // GFX9 and GFX10 have signed byte immediate offsets. The immediate
2532 // offset for S_BUFFER instructions is unsigned.
2533 int64_t ByteOffset = IsBuffer ? C->getZExtValue() : C->getSExtValue();
2534 std::optional<int64_t> EncodedOffset = AMDGPU::getSMRDEncodedOffset(
2535 *Subtarget, ByteOffset, IsBuffer, HasSOffset);
2536 if (EncodedOffset && Offset && !Imm32Only) {
2537 *Offset = CurDAG->getSignedTargetConstant(*EncodedOffset, SL, MVT::i32);
2538 return true;
2539 }
2540
2541 // SGPR and literal offsets are unsigned.
2542 if (ByteOffset < 0)
2543 return false;
2544
2545 EncodedOffset = AMDGPU::getSMRDEncodedLiteralOffset32(*Subtarget, ByteOffset);
2546 if (EncodedOffset && Offset && Imm32Only) {
2547 *Offset = CurDAG->getTargetConstant(*EncodedOffset, SL, MVT::i32);
2548 return true;
2549 }
2550
2551 if (!isUInt<32>(ByteOffset) && !isInt<32>(ByteOffset))
2552 return false;
2553
2554 if (SOffset) {
2555 SDValue C32Bit = CurDAG->getTargetConstant(ByteOffset, SL, MVT::i32);
2556 *SOffset = SDValue(
2557 CurDAG->getMachineNode(AMDGPU::S_MOV_B32, SL, MVT::i32, C32Bit), 0);
2558 return true;
2559 }
2560
2561 return false;
2562}
2563
2564SDValue AMDGPUDAGToDAGISel::Expand32BitAddress(SDValue Addr) const {
2565 if (Addr.getValueType() != MVT::i32)
2566 return Addr;
2567
2568 // Zero-extend a 32-bit address.
2569 SDLoc SL(Addr);
2570
2571 const MachineFunction &MF = CurDAG->getMachineFunction();
2572 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
2573 unsigned AddrHiVal = Info->get32BitAddressHighBits();
2574 SDValue AddrHi = CurDAG->getTargetConstant(AddrHiVal, SL, MVT::i32);
2575
2576 const SDValue Ops[] = {
2577 CurDAG->getTargetConstant(AMDGPU::SReg_64_XEXECRegClassID, SL, MVT::i32),
2578 Addr,
2579 CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32),
2580 SDValue(CurDAG->getMachineNode(AMDGPU::S_MOV_B32, SL, MVT::i32, AddrHi),
2581 0),
2582 CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32),
2583 };
2584
2585 return SDValue(CurDAG->getMachineNode(AMDGPU::REG_SEQUENCE, SL, MVT::i64,
2586 Ops), 0);
2587}
2588
2589// Match a base and an immediate (if Offset is not null) or an SGPR (if
2590// SOffset is not null) or an immediate+SGPR offset. If Imm32Only is
2591// true, match only 32-bit immediate offsets available on CI.
2592bool AMDGPUDAGToDAGISel::SelectSMRDBaseOffset(SDNode *N, SDValue Addr,
2593 SDValue &SBase, SDValue *SOffset,
2594 SDValue *Offset, bool Imm32Only,
2595 bool IsBuffer, bool HasSOffset,
2596 int64_t ImmOffset,
2597 bool *ScaleOffset) const {
2598 if (SOffset && Offset) {
2599 assert(!Imm32Only && !IsBuffer);
2600 SDValue B;
2601
2602 if (!SelectSMRDBaseOffset(N, Addr, B, nullptr, Offset, false, false, true))
2603 return false;
2604
2605 int64_t ImmOff = 0;
2606 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(*Offset))
2607 ImmOff = C->getSExtValue();
2608
2609 return SelectSMRDBaseOffset(N, B, SBase, SOffset, nullptr, false, false,
2610 true, ImmOff, ScaleOffset);
2611 }
2612
2613 // A 32-bit (address + offset) should not cause unsigned 32-bit integer
2614 // wraparound, because s_load instructions perform the addition in 64 bits.
2615 if (Addr.getValueType() == MVT::i32 && Addr.getOpcode() == ISD::ADD &&
2616 !Addr->getFlags().hasNoUnsignedWrap())
2617 return false;
2618
2619 SDValue N0, N1;
2620 // Extract the base and offset if possible.
2621 if (Addr->isAnyAdd() || CurDAG->isADDLike(Addr)) {
2622 N0 = Addr.getOperand(0);
2623 N1 = Addr.getOperand(1);
2624 } else if (getBaseWithOffsetUsingSplitOR(*CurDAG, Addr, N0, N1)) {
2625 assert(N0 && N1 && isa<ConstantSDNode>(N1));
2626 }
2627 if (!N0 || !N1)
2628 return false;
2629
2630 if (SelectSMRDOffset(N, N1, SOffset, Offset, Imm32Only, IsBuffer, HasSOffset,
2631 ImmOffset, ScaleOffset)) {
2632 SBase = N0;
2633 return true;
2634 }
2635 if (SelectSMRDOffset(N, N0, SOffset, Offset, Imm32Only, IsBuffer, HasSOffset,
2636 ImmOffset, ScaleOffset)) {
2637 SBase = N1;
2638 return true;
2639 }
2640 return false;
2641}
2642
2643bool AMDGPUDAGToDAGISel::SelectSMRD(SDNode *N, SDValue Addr, SDValue &SBase,
2644 SDValue *SOffset, SDValue *Offset,
2645 bool Imm32Only, bool *ScaleOffset) const {
2646 if (SelectSMRDBaseOffset(N, Addr, SBase, SOffset, Offset, Imm32Only,
2647 /* IsBuffer */ false, /* HasSOffset */ false,
2648 /* ImmOffset */ 0, ScaleOffset)) {
2649 SBase = Expand32BitAddress(SBase);
2650 return true;
2651 }
2652
2653 if (Addr.getValueType() == MVT::i32 && Offset && !SOffset) {
2654 SBase = Expand32BitAddress(Addr);
2655 *Offset = CurDAG->getTargetConstant(0, SDLoc(Addr), MVT::i32);
2656 return true;
2657 }
2658
2659 return false;
2660}
2661
2662bool AMDGPUDAGToDAGISel::SelectSMRDImm(SDValue Addr, SDValue &SBase,
2663 SDValue &Offset) const {
2664 return SelectSMRD(/* N */ nullptr, Addr, SBase, /* SOffset */ nullptr,
2665 &Offset);
2666}
2667
2668bool AMDGPUDAGToDAGISel::SelectSMRDImm32(SDValue Addr, SDValue &SBase,
2669 SDValue &Offset) const {
2670 assert(Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS);
2671 return SelectSMRD(/* N */ nullptr, Addr, SBase, /* SOffset */ nullptr,
2672 &Offset, /* Imm32Only */ true);
2673}
2674
2675bool AMDGPUDAGToDAGISel::SelectSMRDSgpr(SDNode *N, SDValue Addr, SDValue &SBase,
2676 SDValue &SOffset, SDValue &CPol) const {
2677 bool ScaleOffset;
2678 if (!SelectSMRD(N, Addr, SBase, &SOffset, /* Offset */ nullptr,
2679 /* Imm32Only */ false, &ScaleOffset))
2680 return false;
2681
2682 CPol = CurDAG->getTargetConstant(ScaleOffset ? AMDGPU::CPol::SCAL : 0,
2683 SDLoc(N), MVT::i32);
2684 return true;
2685}
2686
2687bool AMDGPUDAGToDAGISel::SelectSMRDSgprImm(SDNode *N, SDValue Addr,
2688 SDValue &SBase, SDValue &SOffset,
2689 SDValue &Offset,
2690 SDValue &CPol) const {
2691 bool ScaleOffset;
2692 if (!SelectSMRD(N, Addr, SBase, &SOffset, &Offset, false, &ScaleOffset))
2693 return false;
2694
2695 CPol = CurDAG->getTargetConstant(ScaleOffset ? AMDGPU::CPol::SCAL : 0,
2696 SDLoc(N), MVT::i32);
2697 return true;
2698}
2699
2700bool AMDGPUDAGToDAGISel::SelectSMRDBufferImm(SDValue N, SDValue &Offset) const {
2701 return SelectSMRDOffset(/* N */ nullptr, N, /* SOffset */ nullptr, &Offset,
2702 /* Imm32Only */ false, /* IsBuffer */ true);
2703}
2704
2705bool AMDGPUDAGToDAGISel::SelectSMRDBufferImm32(SDValue N,
2706 SDValue &Offset) const {
2707 assert(Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS);
2708 return SelectSMRDOffset(/* N */ nullptr, N, /* SOffset */ nullptr, &Offset,
2709 /* Imm32Only */ true, /* IsBuffer */ true);
2710}
2711
2712bool AMDGPUDAGToDAGISel::SelectSMRDBufferSgprImm(SDValue N, SDValue &SOffset,
2713 SDValue &Offset) const {
2714 // Match the (soffset + offset) pair as a 32-bit register base and
2715 // an immediate offset.
2716 return N.getValueType() == MVT::i32 &&
2717 SelectSMRDBaseOffset(/* N */ nullptr, N, /* SBase */ SOffset,
2718 /* SOffset*/ nullptr, &Offset,
2719 /* Imm32Only */ false, /* IsBuffer */ true);
2720}
2721
2722bool AMDGPUDAGToDAGISel::SelectMOVRELOffset(SDValue Index,
2723 SDValue &Base,
2724 SDValue &Offset) const {
2725 SDLoc DL(Index);
2726
2727 if (CurDAG->isBaseWithConstantOffset(Index)) {
2728 SDValue N0 = Index.getOperand(0);
2729 SDValue N1 = Index.getOperand(1);
2730 ConstantSDNode *C1 = cast<ConstantSDNode>(N1);
2731
2732 // (add n0, c0)
2733 // Don't peel off the offset (c0) if doing so could possibly lead
2734 // the base (n0) to be negative.
2735 // (or n0, |c0|) can never change a sign given isBaseWithConstantOffset.
2736 if (C1->getSExtValue() <= 0 || CurDAG->SignBitIsZero(N0) ||
2737 (Index->getOpcode() == ISD::OR && C1->getSExtValue() >= 0)) {
2738 Base = N0;
2739 Offset = CurDAG->getTargetConstant(C1->getZExtValue(), DL, MVT::i32);
2740 return true;
2741 }
2742 }
2743
2744 if (isa<ConstantSDNode>(Index))
2745 return false;
2746
2747 Base = Index;
2748 Offset = CurDAG->getTargetConstant(0, DL, MVT::i32);
2749 return true;
2750}
2751
2752SDNode *AMDGPUDAGToDAGISel::getBFE32(bool IsSigned, const SDLoc &DL,
2753 SDValue Val, uint32_t Offset,
2754 uint32_t Width) {
2755 if (Val->isDivergent()) {
2756 unsigned Opcode = IsSigned ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
2757 SDValue Off = CurDAG->getTargetConstant(Offset, DL, MVT::i32);
2758 SDValue W = CurDAG->getTargetConstant(Width, DL, MVT::i32);
2759
2760 return CurDAG->getMachineNode(Opcode, DL, MVT::i32, Val, Off, W);
2761 }
2762 unsigned Opcode = IsSigned ? AMDGPU::S_BFE_I32 : AMDGPU::S_BFE_U32;
2763 // Transformation function, pack the offset and width of a BFE into
2764 // the format expected by the S_BFE_I32 / S_BFE_U32. In the second
2765 // source, bits [5:0] contain the offset and bits [22:16] the width.
2766 uint32_t PackedVal = Offset | (Width << 16);
2767 SDValue PackedConst = CurDAG->getTargetConstant(PackedVal, DL, MVT::i32);
2768
2769 return CurDAG->getMachineNode(Opcode, DL, MVT::i32, Val, PackedConst);
2770}
2771
2772void AMDGPUDAGToDAGISel::SelectS_BFEFromShifts(SDNode *N) {
2773 // "(a << b) srl c)" ---> "BFE_U32 a, (c-b), (32-c)
2774 // "(a << b) sra c)" ---> "BFE_I32 a, (c-b), (32-c)
2775 // Predicate: 0 < b <= c < 32
2776
2777 const SDValue &Shl = N->getOperand(0);
2778 ConstantSDNode *B = dyn_cast<ConstantSDNode>(Shl->getOperand(1));
2779 ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1));
2780
2781 if (B && C) {
2782 uint32_t BVal = B->getZExtValue();
2783 uint32_t CVal = C->getZExtValue();
2784
2785 if (0 < BVal && BVal <= CVal && CVal < 32) {
2786 bool Signed = N->getOpcode() == ISD::SRA;
2787 ReplaceNode(N, getBFE32(Signed, SDLoc(N), Shl.getOperand(0), CVal - BVal,
2788 32 - CVal));
2789 return;
2790 }
2791 }
2792 SelectCode(N);
2793}
2794
2795void AMDGPUDAGToDAGISel::SelectS_BFE(SDNode *N) {
2796 switch (N->getOpcode()) {
2797 case ISD::AND:
2798 if (N->getOperand(0).getOpcode() == ISD::SRL) {
2799 // "(a srl b) & mask" ---> "BFE_U32 a, b, popcount(mask)"
2800 // Predicate: isMask(mask)
2801 const SDValue &Srl = N->getOperand(0);
2802 ConstantSDNode *Shift = dyn_cast<ConstantSDNode>(Srl.getOperand(1));
2803 ConstantSDNode *Mask = dyn_cast<ConstantSDNode>(N->getOperand(1));
2804
2805 if (Shift && Mask) {
2806 uint32_t ShiftVal = Shift->getZExtValue();
2807 uint32_t MaskVal = Mask->getZExtValue();
2808
2809 if (isMask_32(MaskVal)) {
2810 uint32_t WidthVal = llvm::popcount(MaskVal);
2811 ReplaceNode(N, getBFE32(false, SDLoc(N), Srl.getOperand(0), ShiftVal,
2812 WidthVal));
2813 return;
2814 }
2815 }
2816 }
2817 break;
2818 case ISD::SRL:
2819 if (N->getOperand(0).getOpcode() == ISD::AND) {
2820 // "(a & mask) srl b)" ---> "BFE_U32 a, b, popcount(mask >> b)"
2821 // Predicate: isMask(mask >> b)
2822 const SDValue &And = N->getOperand(0);
2823 ConstantSDNode *Shift = dyn_cast<ConstantSDNode>(N->getOperand(1));
2824 ConstantSDNode *Mask = dyn_cast<ConstantSDNode>(And->getOperand(1));
2825
2826 if (Shift && Mask) {
2827 uint32_t ShiftVal = Shift->getZExtValue();
2828 uint32_t MaskVal = Mask->getZExtValue() >> ShiftVal;
2829
2830 if (isMask_32(MaskVal)) {
2831 uint32_t WidthVal = llvm::popcount(MaskVal);
2832 ReplaceNode(N, getBFE32(false, SDLoc(N), And.getOperand(0), ShiftVal,
2833 WidthVal));
2834 return;
2835 }
2836 }
2837 } else if (N->getOperand(0).getOpcode() == ISD::SHL) {
2838 SelectS_BFEFromShifts(N);
2839 return;
2840 }
2841 break;
2842 case ISD::SRA:
2843 if (N->getOperand(0).getOpcode() == ISD::SHL) {
2844 SelectS_BFEFromShifts(N);
2845 return;
2846 }
2847 break;
2848
2850 // sext_inreg (srl x, 16), i8 -> bfe_i32 x, 16, 8
2851 SDValue Src = N->getOperand(0);
2852 if (Src.getOpcode() != ISD::SRL)
2853 break;
2854
2855 const ConstantSDNode *Amt = dyn_cast<ConstantSDNode>(Src.getOperand(1));
2856 if (!Amt)
2857 break;
2858
2859 unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits();
2860 ReplaceNode(N, getBFE32(true, SDLoc(N), Src.getOperand(0),
2861 Amt->getZExtValue(), Width));
2862 return;
2863 }
2864 }
2865
2866 SelectCode(N);
2867}
2868
2869bool AMDGPUDAGToDAGISel::isCBranchSCC(const SDNode *N) const {
2870 assert(N->getOpcode() == ISD::BRCOND);
2871 if (!N->hasOneUse())
2872 return false;
2873
2874 SDValue Cond = N->getOperand(1);
2875 if (Cond.getOpcode() == ISD::CopyToReg)
2876 Cond = Cond.getOperand(2);
2877
2878 if (Cond.getOpcode() != ISD::SETCC || !Cond.hasOneUse())
2879 return false;
2880
2881 MVT VT = Cond.getOperand(0).getSimpleValueType();
2882 if (VT == MVT::i32)
2883 return true;
2884
2885 if (VT == MVT::i64) {
2886 ISD::CondCode CC = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
2887 return (CC == ISD::SETEQ || CC == ISD::SETNE) &&
2888 Subtarget->hasScalarCompareEq64();
2889 }
2890
2891 if ((VT == MVT::f16 || VT == MVT::f32) && Subtarget->hasSALUFloatInsts())
2892 return true;
2893
2894 return false;
2895}
2896
2897static SDValue combineBallotPattern(SDValue VCMP, bool &Negate) {
2898 assert(VCMP->getOpcode() == AMDGPUISD::SETCC);
2899 // Special case for amdgcn.ballot:
2900 // %Cond = i1 (and/or combination of i1 ISD::SETCCs)
2901 // %VCMP = i(WaveSize) AMDGPUISD::SETCC (ext %Cond), 0, setne/seteq
2902 // =>
2903 // Use i1 %Cond value instead of i(WaveSize) %VCMP.
2904 // This is possible because divergent ISD::SETCC is selected as V_CMP and
2905 // Cond becomes a i(WaveSize) full mask value.
2906 // Note that ballot doesn't use SETEQ condition but its easy to support it
2907 // here for completeness, so in this case Negate is set true on return.
2908 auto VCMP_CC = cast<CondCodeSDNode>(VCMP.getOperand(2))->get();
2909 if ((VCMP_CC == ISD::SETEQ || VCMP_CC == ISD::SETNE) &&
2910 isNullConstant(VCMP.getOperand(1))) {
2911
2912 auto Cond = VCMP.getOperand(0);
2913 if (ISD::isExtOpcode(Cond->getOpcode())) // Skip extension.
2914 Cond = Cond.getOperand(0);
2915
2916 if (isBoolSGPR(Cond)) {
2917 Negate = VCMP_CC == ISD::SETEQ;
2918 return Cond;
2919 }
2920 }
2921 return SDValue();
2922}
2923
2924void AMDGPUDAGToDAGISel::SelectBRCOND(SDNode *N) {
2925 SDValue Cond = N->getOperand(1);
2926
2927 if (Cond.isUndef()) {
2928 CurDAG->SelectNodeTo(N, AMDGPU::SI_BR_UNDEF, MVT::Other,
2929 N->getOperand(2), N->getOperand(0));
2930 return;
2931 }
2932
2933 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
2934
2935 bool UseSCCBr = isCBranchSCC(N) && isUniformBr(N);
2936 bool AndExec = !UseSCCBr;
2937 bool Negate = false;
2938
2939 if (Cond.getOpcode() == ISD::SETCC &&
2940 Cond->getOperand(0)->getOpcode() == AMDGPUISD::SETCC) {
2941 SDValue VCMP = Cond->getOperand(0);
2942 auto CC = cast<CondCodeSDNode>(Cond->getOperand(2))->get();
2943 if ((CC == ISD::SETEQ || CC == ISD::SETNE) &&
2944 isNullConstant(Cond->getOperand(1)) &&
2945 // We may encounter ballot.i64 in wave32 mode on -O0.
2946 VCMP.getValueType().getSizeInBits() == Subtarget->getWavefrontSize()) {
2947 // %VCMP = i(WaveSize) AMDGPUISD::SETCC ...
2948 // %C = i1 ISD::SETCC %VCMP, 0, setne/seteq
2949 // BRCOND i1 %C, %BB
2950 // =>
2951 // %VCMP = i(WaveSize) AMDGPUISD::SETCC ...
2952 // VCC = COPY i(WaveSize) %VCMP
2953 // S_CBRANCH_VCCNZ/VCCZ %BB
2954 Negate = CC == ISD::SETEQ;
2955 bool NegatedBallot = false;
2956 if (auto BallotCond = combineBallotPattern(VCMP, NegatedBallot)) {
2957 Cond = BallotCond;
2958 UseSCCBr = !BallotCond->isDivergent();
2959 Negate = Negate ^ NegatedBallot;
2960 } else {
2961 // TODO: don't use SCC here assuming that AMDGPUISD::SETCC is always
2962 // selected as V_CMP, but this may change for uniform condition.
2963 Cond = VCMP;
2964 UseSCCBr = false;
2965 }
2966 }
2967 // Cond is either V_CMP resulted from AMDGPUISD::SETCC or a combination of
2968 // V_CMPs resulted from ballot or ballot has uniform condition and SCC is
2969 // used.
2970 AndExec = false;
2971 }
2972
2973 unsigned BrOp =
2974 UseSCCBr ? (Negate ? AMDGPU::S_CBRANCH_SCC0 : AMDGPU::S_CBRANCH_SCC1)
2975 : (Negate ? AMDGPU::S_CBRANCH_VCCZ : AMDGPU::S_CBRANCH_VCCNZ);
2976 Register CondReg = UseSCCBr ? AMDGPU::SCC : TRI->getVCC();
2977 SDLoc SL(N);
2978
2979 if (AndExec) {
2980 // This is the case that we are selecting to S_CBRANCH_VCCNZ. We have not
2981 // analyzed what generates the vcc value, so we do not know whether vcc
2982 // bits for disabled lanes are 0. Thus we need to mask out bits for
2983 // disabled lanes.
2984 //
2985 // For the case that we select S_CBRANCH_SCC1 and it gets
2986 // changed to S_CBRANCH_VCCNZ in SIFixSGPRCopies, SIFixSGPRCopies calls
2987 // SIInstrInfo::moveToVALU which inserts the S_AND).
2988 //
2989 // We could add an analysis of what generates the vcc value here and omit
2990 // the S_AND when is unnecessary. But it would be better to add a separate
2991 // pass after SIFixSGPRCopies to do the unnecessary S_AND removal, so it
2992 // catches both cases.
2993 Cond = SDValue(
2994 CurDAG->getMachineNode(
2995 Subtarget->isWave32() ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64, SL,
2996 MVT::i1,
2997 CurDAG->getRegister(Subtarget->isWave32() ? AMDGPU::EXEC_LO
2998 : AMDGPU::EXEC,
2999 MVT::i1),
3000 Cond),
3001 0);
3002 }
3003
3004 SDValue VCC = CurDAG->getCopyToReg(N->getOperand(0), SL, CondReg, Cond);
3005 CurDAG->SelectNodeTo(N, BrOp, MVT::Other,
3006 N->getOperand(2), // Basic Block
3007 VCC.getValue(0));
3008}
3009
3010void AMDGPUDAGToDAGISel::SelectFP_EXTEND(SDNode *N) {
3011 if (Subtarget->hasSALUFloatInsts() && N->getValueType(0) == MVT::f32 &&
3012 !N->isDivergent()) {
3013 SDValue Src = N->getOperand(0);
3014 if (Src.getValueType() == MVT::f16) {
3015 if (isExtractHiElt(Src, Src)) {
3016 CurDAG->SelectNodeTo(N, AMDGPU::S_CVT_HI_F32_F16, N->getVTList(),
3017 {Src});
3018 return;
3019 }
3020 }
3021 }
3022
3023 SelectCode(N);
3024}
3025
3026void AMDGPUDAGToDAGISel::SelectDSAppendConsume(SDNode *N, unsigned IntrID) {
3027 // The address is assumed to be uniform, so if it ends up in a VGPR, it will
3028 // be copied to an SGPR with readfirstlane.
3029 unsigned Opc = IntrID == Intrinsic::amdgcn_ds_append ?
3030 AMDGPU::DS_APPEND : AMDGPU::DS_CONSUME;
3031
3032 SDValue Chain = N->getOperand(0);
3033 SDValue Ptr = N->getOperand(2);
3034 MemIntrinsicSDNode *M = cast<MemIntrinsicSDNode>(N);
3035 MachineMemOperand *MMO = M->getMemOperand();
3036 bool IsGDS = M->getAddressSpace() == AMDGPUAS::REGION_ADDRESS;
3037
3038 SDValue Offset;
3039 if (CurDAG->isBaseWithConstantOffset(Ptr)) {
3040 SDValue PtrBase = Ptr.getOperand(0);
3041 SDValue PtrOffset = Ptr.getOperand(1);
3042
3043 const APInt &OffsetVal = PtrOffset->getAsAPIntVal();
3044 if (isDSOffsetLegal(PtrBase, OffsetVal.getZExtValue())) {
3045 N = glueCopyToM0(N, PtrBase);
3046 Offset = CurDAG->getTargetConstant(OffsetVal, SDLoc(), MVT::i32);
3047 }
3048 }
3049
3050 if (!Offset) {
3051 N = glueCopyToM0(N, Ptr);
3052 Offset = CurDAG->getTargetConstant(0, SDLoc(), MVT::i32);
3053 }
3054
3055 SDValue Ops[] = {
3056 Offset,
3057 CurDAG->getTargetConstant(IsGDS, SDLoc(), MVT::i32),
3058 Chain,
3059 N->getOperand(N->getNumOperands() - 1) // New glue
3060 };
3061
3062 SDNode *Selected = CurDAG->SelectNodeTo(N, Opc, N->getVTList(), Ops);
3063 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Selected), {MMO});
3064}
3065
3066// We need to handle this here because tablegen doesn't support matching
3067// instructions with multiple outputs.
3068void AMDGPUDAGToDAGISel::SelectDSBvhStackIntrinsic(SDNode *N, unsigned IntrID) {
3069 unsigned Opc;
3070 switch (IntrID) {
3071 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
3072 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
3073 Opc = AMDGPU::DS_BVH_STACK_RTN_B32;
3074 break;
3075 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
3076 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP1_RTN_B32;
3077 break;
3078 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
3079 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP2_RTN_B64;
3080 break;
3081 }
3082 SDValue Ops[] = {N->getOperand(2), N->getOperand(3), N->getOperand(4),
3083 N->getOperand(5), N->getOperand(0)};
3084
3085 MemIntrinsicSDNode *M = cast<MemIntrinsicSDNode>(N);
3086 MachineMemOperand *MMO = M->getMemOperand();
3087 SDNode *Selected = CurDAG->SelectNodeTo(N, Opc, N->getVTList(), Ops);
3088 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Selected), {MMO});
3089}
3090
3091void AMDGPUDAGToDAGISel::SelectTensorLoadStore(SDNode *N, unsigned IntrID) {
3092 bool IsLoad = IntrID == Intrinsic::amdgcn_tensor_load_to_lds;
3093 unsigned Opc =
3094 IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d4 : AMDGPU::TENSOR_STORE_FROM_LDS_d4;
3095
3096 SmallVector<SDValue, 7> TensorOps;
3097 // First two groups
3098 TensorOps.push_back(N->getOperand(2)); // D# group 0
3099 TensorOps.push_back(N->getOperand(3)); // D# group 1
3100
3101 // Use _D2 version if both group 2 and 3 are zero-initialized.
3102 SDValue Group2 = N->getOperand(4);
3103 SDValue Group3 = N->getOperand(5);
3104 if (ISD::isBuildVectorAllZeros(Group2.getNode()) &&
3106 Opc = IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d2
3107 : AMDGPU::TENSOR_STORE_FROM_LDS_d2;
3108 } else { // Has at least 4 groups
3109 TensorOps.push_back(Group2); // D# group 2
3110 TensorOps.push_back(Group3); // D# group 3
3111 }
3112
3113 // TODO: Handle the fifth group: N->getOperand(6), which is silently ignored
3114 // for now because all existing targets only support up to 4 groups.
3115 TensorOps.push_back(CurDAG->getTargetConstant(0, SDLoc(N), MVT::i1)); // r128
3116 TensorOps.push_back(N->getOperand(7)); // cache policy
3117 TensorOps.push_back(N->getOperand(0)); // chain
3118
3119 (void)CurDAG->SelectNodeTo(N, Opc, MVT::Other, TensorOps);
3120}
3121
3122static unsigned gwsIntrinToOpcode(unsigned IntrID) {
3123 switch (IntrID) {
3124 case Intrinsic::amdgcn_ds_gws_init:
3125 return AMDGPU::DS_GWS_INIT;
3126 case Intrinsic::amdgcn_ds_gws_barrier:
3127 return AMDGPU::DS_GWS_BARRIER;
3128 case Intrinsic::amdgcn_ds_gws_sema_v:
3129 return AMDGPU::DS_GWS_SEMA_V;
3130 case Intrinsic::amdgcn_ds_gws_sema_br:
3131 return AMDGPU::DS_GWS_SEMA_BR;
3132 case Intrinsic::amdgcn_ds_gws_sema_p:
3133 return AMDGPU::DS_GWS_SEMA_P;
3134 case Intrinsic::amdgcn_ds_gws_sema_release_all:
3135 return AMDGPU::DS_GWS_SEMA_RELEASE_ALL;
3136 default:
3137 llvm_unreachable("not a gws intrinsic");
3138 }
3139}
3140
3141void AMDGPUDAGToDAGISel::SelectDS_GWS(SDNode *N, unsigned IntrID) {
3142 if (!Subtarget->hasGWS() ||
3143 (IntrID == Intrinsic::amdgcn_ds_gws_sema_release_all &&
3144 !Subtarget->hasGWSSemaReleaseAll())) {
3145 // Let this error.
3146 SelectCode(N);
3147 return;
3148 }
3149
3150 // Chain, intrinsic ID, vsrc, offset
3151 const bool HasVSrc = N->getNumOperands() == 4;
3152 assert(HasVSrc || N->getNumOperands() == 3);
3153
3154 SDLoc SL(N);
3155 SDValue BaseOffset = N->getOperand(HasVSrc ? 3 : 2);
3156 int ImmOffset = 0;
3157 MemIntrinsicSDNode *M = cast<MemIntrinsicSDNode>(N);
3158 MachineMemOperand *MMO = M->getMemOperand();
3159
3160 // Don't worry if the offset ends up in a VGPR. Only one lane will have
3161 // effect, so SIFixSGPRCopies will validly insert readfirstlane.
3162
3163 // The resource id offset is computed as (<isa opaque base> + M0[21:16] +
3164 // offset field) % 64. Some versions of the programming guide omit the m0
3165 // part, or claim it's from offset 0.
3166 if (ConstantSDNode *ConstOffset = dyn_cast<ConstantSDNode>(BaseOffset)) {
3167 // If we have a constant offset, try to use the 0 in m0 as the base.
3168 // TODO: Look into changing the default m0 initialization value. If the
3169 // default -1 only set the low 16-bits, we could leave it as-is and add 1 to
3170 // the immediate offset.
3171 glueCopyToM0(N, CurDAG->getTargetConstant(0, SL, MVT::i32));
3172 ImmOffset = ConstOffset->getZExtValue();
3173 } else {
3174 if (CurDAG->isBaseWithConstantOffset(BaseOffset)) {
3175 ImmOffset = BaseOffset.getConstantOperandVal(1);
3176 BaseOffset = BaseOffset.getOperand(0);
3177 }
3178
3179 // Prefer to do the shift in an SGPR since it should be possible to use m0
3180 // as the result directly. If it's already an SGPR, it will be eliminated
3181 // later.
3182 SDNode *SGPROffset
3183 = CurDAG->getMachineNode(AMDGPU::V_READFIRSTLANE_B32, SL, MVT::i32,
3184 BaseOffset);
3185 // Shift to offset in m0
3186 SDNode *M0Base
3187 = CurDAG->getMachineNode(AMDGPU::S_LSHL_B32, SL, MVT::i32,
3188 SDValue(SGPROffset, 0),
3189 CurDAG->getTargetConstant(16, SL, MVT::i32));
3190 glueCopyToM0(N, SDValue(M0Base, 0));
3191 }
3192
3193 SDValue Chain = N->getOperand(0);
3194 SDValue OffsetField = CurDAG->getTargetConstant(ImmOffset, SL, MVT::i32);
3195
3196 const unsigned Opc = gwsIntrinToOpcode(IntrID);
3197
3198 const MCInstrDesc &InstrDesc = TII->get(Opc);
3199 int Data0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::data0);
3200
3201 const TargetRegisterClass *DataRC = TII->getRegClass(InstrDesc, Data0Idx);
3202
3204 if (HasVSrc) {
3205 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
3206
3207 SDValue Data = N->getOperand(2);
3208 MVT DataVT = Data.getValueType().getSimpleVT();
3209 if (TRI->isTypeLegalForClass(*DataRC, DataVT)) {
3210 // Normal 32-bit case.
3211 Ops.push_back(N->getOperand(2));
3212 } else {
3213 // Operand is really 32-bits, but requires 64-bit alignment, so use the
3214 // even aligned 64-bit register class.
3215 const SDValue RegSeqOps[] = {
3216 CurDAG->getTargetConstant(DataRC->getID(), SL, MVT::i32), Data,
3217 CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32),
3218 SDValue(
3219 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, SL, MVT::i32),
3220 0),
3221 CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32)};
3222
3223 Ops.push_back(SDValue(CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE,
3224 SL, MVT::v2i32, RegSeqOps),
3225 0));
3226 }
3227 }
3228
3229 Ops.push_back(OffsetField);
3230 Ops.push_back(Chain);
3231
3232 SDNode *Selected = CurDAG->SelectNodeTo(N, Opc, N->getVTList(), Ops);
3233 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Selected), {MMO});
3234}
3235
3236void AMDGPUDAGToDAGISel::SelectInterpP1F16(SDNode *N) {
3237 if (Subtarget->getLDSBankCount() != 16) {
3238 // This is a single instruction with a pattern.
3239 SelectCode(N);
3240 return;
3241 }
3242
3243 SDLoc DL(N);
3244
3245 // This requires 2 instructions. It is possible to write a pattern to support
3246 // this, but the generated isel emitter doesn't correctly deal with multiple
3247 // output instructions using the same physical register input. The copy to m0
3248 // is incorrectly placed before the second instruction.
3249 //
3250 // TODO: Match source modifiers.
3251 //
3252 // def : Pat <
3253 // (int_amdgcn_interp_p1_f16
3254 // (VOP3Mods f32:$src0, i32:$src0_modifiers),
3255 // (i32 timm:$attrchan), (i32 timm:$attr),
3256 // (i1 timm:$high), M0),
3257 // (V_INTERP_P1LV_F16 $src0_modifiers, VGPR_32:$src0, timm:$attr,
3258 // timm:$attrchan, 0,
3259 // (V_INTERP_MOV_F32 2, timm:$attr, timm:$attrchan), timm:$high)> {
3260 // let Predicates = [has16BankLDS];
3261 // }
3262
3263 // 16 bank LDS
3264 SDValue ToM0 = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL, AMDGPU::M0,
3265 N->getOperand(5), SDValue());
3266
3267 SDVTList VTs = CurDAG->getVTList(MVT::f32, MVT::Other);
3268
3269 SDNode *InterpMov =
3270 CurDAG->getMachineNode(AMDGPU::V_INTERP_MOV_F32, DL, VTs, {
3271 CurDAG->getTargetConstant(2, DL, MVT::i32), // P0
3272 N->getOperand(3), // Attr
3273 N->getOperand(2), // Attrchan
3274 ToM0.getValue(1) // In glue
3275 });
3276
3277 SDNode *InterpP1LV =
3278 CurDAG->getMachineNode(AMDGPU::V_INTERP_P1LV_F16, DL, MVT::f32, {
3279 CurDAG->getTargetConstant(0, DL, MVT::i32), // $src0_modifiers
3280 N->getOperand(1), // Src0
3281 N->getOperand(3), // Attr
3282 N->getOperand(2), // Attrchan
3283 CurDAG->getTargetConstant(0, DL, MVT::i32), // $src2_modifiers
3284 SDValue(InterpMov, 0), // Src2 - holds two f16 values selected by high
3285 N->getOperand(4), // high
3286 CurDAG->getTargetConstant(0, DL, MVT::i1), // $clamp
3287 CurDAG->getTargetConstant(0, DL, MVT::i32), // $omod
3288 SDValue(InterpMov, 1)
3289 });
3290
3291 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), SDValue(InterpP1LV, 0));
3292}
3293
3294void AMDGPUDAGToDAGISel::SelectINTRINSIC_W_CHAIN(SDNode *N) {
3295 unsigned IntrID = N->getConstantOperandVal(1);
3296 switch (IntrID) {
3297 case Intrinsic::amdgcn_ds_append:
3298 case Intrinsic::amdgcn_ds_consume: {
3299 if (N->getValueType(0) != MVT::i32)
3300 break;
3301 SelectDSAppendConsume(N, IntrID);
3302 return;
3303 }
3304 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
3305 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
3306 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
3307 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
3308 SelectDSBvhStackIntrinsic(N, IntrID);
3309 return;
3310 case Intrinsic::amdgcn_init_whole_wave:
3311 CurDAG->getMachineFunction()
3312 .getInfo<SIMachineFunctionInfo>()
3313 ->setInitWholeWave();
3314 break;
3315 }
3316
3317 SelectCode(N);
3318}
3319
3320void AMDGPUDAGToDAGISel::SelectINTRINSIC_WO_CHAIN(SDNode *N) {
3321 unsigned IntrID = N->getConstantOperandVal(0);
3322 unsigned Opcode = AMDGPU::INSTRUCTION_LIST_END;
3323 SDNode *ConvGlueNode = N->getGluedNode();
3324 if (ConvGlueNode) {
3325 // FIXME: Possibly iterate over multiple glue nodes?
3326 assert(ConvGlueNode->getOpcode() == ISD::CONVERGENCECTRL_GLUE);
3327 ConvGlueNode = ConvGlueNode->getOperand(0).getNode();
3328 ConvGlueNode =
3329 CurDAG->getMachineNode(TargetOpcode::CONVERGENCECTRL_GLUE, {},
3330 MVT::Glue, SDValue(ConvGlueNode, 0));
3331 } else {
3332 ConvGlueNode = nullptr;
3333 }
3334 switch (IntrID) {
3335 case Intrinsic::amdgcn_wqm:
3336 Opcode = AMDGPU::WQM;
3337 break;
3338 case Intrinsic::amdgcn_softwqm:
3339 Opcode = AMDGPU::SOFT_WQM;
3340 break;
3341 case Intrinsic::amdgcn_wwm:
3342 case Intrinsic::amdgcn_strict_wwm:
3343 Opcode = AMDGPU::STRICT_WWM;
3344 break;
3345 case Intrinsic::amdgcn_strict_wqm:
3346 Opcode = AMDGPU::STRICT_WQM;
3347 break;
3348 case Intrinsic::amdgcn_interp_p1_f16:
3349 SelectInterpP1F16(N);
3350 return;
3351 case Intrinsic::amdgcn_permlane16_swap:
3352 case Intrinsic::amdgcn_permlane32_swap: {
3353 if ((IntrID == Intrinsic::amdgcn_permlane16_swap &&
3354 !Subtarget->hasPermlane16Swap()) ||
3355 (IntrID == Intrinsic::amdgcn_permlane32_swap &&
3356 !Subtarget->hasPermlane32Swap())) {
3357 SelectCode(N); // Hit the default error
3358 return;
3359 }
3360
3361 Opcode = IntrID == Intrinsic::amdgcn_permlane16_swap
3362 ? AMDGPU::V_PERMLANE16_SWAP_B32_e64
3363 : AMDGPU::V_PERMLANE32_SWAP_B32_e64;
3364
3365 SmallVector<SDValue, 4> NewOps(N->op_begin() + 1, N->op_end());
3366 if (ConvGlueNode)
3367 NewOps.push_back(SDValue(ConvGlueNode, 0));
3368
3369 bool FI = N->getConstantOperandVal(3);
3370 NewOps[2] = CurDAG->getTargetConstant(
3371 FI ? AMDGPU::DPP::DPP_FI_1 : AMDGPU::DPP::DPP_FI_0, SDLoc(), MVT::i32);
3372
3373 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), NewOps);
3374 return;
3375 }
3376 default:
3377 SelectCode(N);
3378 break;
3379 }
3380
3381 if (Opcode != AMDGPU::INSTRUCTION_LIST_END) {
3382 SDValue Src = N->getOperand(1);
3383 CurDAG->SelectNodeTo(N, Opcode, N->getVTList(), {Src});
3384 }
3385
3386 if (ConvGlueNode) {
3387 SmallVector<SDValue, 4> NewOps(N->ops());
3388 NewOps.push_back(SDValue(ConvGlueNode, 0));
3389 CurDAG->MorphNodeTo(N, N->getOpcode(), N->getVTList(), NewOps);
3390 }
3391}
3392
3393void AMDGPUDAGToDAGISel::SelectINTRINSIC_VOID(SDNode *N) {
3394 unsigned IntrID = N->getConstantOperandVal(1);
3395 switch (IntrID) {
3396 case Intrinsic::amdgcn_ds_gws_init:
3397 case Intrinsic::amdgcn_ds_gws_barrier:
3398 case Intrinsic::amdgcn_ds_gws_sema_v:
3399 case Intrinsic::amdgcn_ds_gws_sema_br:
3400 case Intrinsic::amdgcn_ds_gws_sema_p:
3401 case Intrinsic::amdgcn_ds_gws_sema_release_all:
3402 SelectDS_GWS(N, IntrID);
3403 return;
3404 case Intrinsic::amdgcn_tensor_load_to_lds:
3405 case Intrinsic::amdgcn_tensor_store_from_lds:
3406 SelectTensorLoadStore(N, IntrID);
3407 return;
3408 default:
3409 break;
3410 }
3411
3412 SelectCode(N);
3413}
3414
3415void AMDGPUDAGToDAGISel::SelectWAVE_ADDRESS(SDNode *N) {
3416 SDValue Log2WaveSize =
3417 CurDAG->getTargetConstant(Subtarget->getWavefrontSizeLog2(), SDLoc(N), MVT::i32);
3418 CurDAG->SelectNodeTo(N, AMDGPU::S_LSHR_B32, N->getVTList(),
3419 {N->getOperand(0), Log2WaveSize});
3420}
3421
3422void AMDGPUDAGToDAGISel::SelectSTACKRESTORE(SDNode *N) {
3423 SDValue SrcVal = N->getOperand(1);
3424 if (SrcVal.getValueType() != MVT::i32) {
3425 SelectCode(N); // Emit default error
3426 return;
3427 }
3428
3429 SDValue CopyVal;
3430 Register SP = TLI->getStackPointerRegisterToSaveRestore();
3431 SDLoc SL(N);
3432
3433 if (SrcVal.getOpcode() == AMDGPUISD::WAVE_ADDRESS) {
3434 CopyVal = SrcVal.getOperand(0);
3435 } else {
3436 SDValue Log2WaveSize = CurDAG->getTargetConstant(
3437 Subtarget->getWavefrontSizeLog2(), SL, MVT::i32);
3438
3439 if (N->isDivergent()) {
3440 SrcVal = SDValue(CurDAG->getMachineNode(AMDGPU::V_READFIRSTLANE_B32, SL,
3441 MVT::i32, SrcVal),
3442 0);
3443 }
3444
3445 CopyVal = SDValue(CurDAG->getMachineNode(AMDGPU::S_LSHL_B32, SL, MVT::i32,
3446 {SrcVal, Log2WaveSize}),
3447 0);
3448 }
3449
3450 SDValue CopyToSP = CurDAG->getCopyToReg(N->getOperand(0), SL, SP, CopyVal);
3451 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), CopyToSP);
3452}
3453
3454bool AMDGPUDAGToDAGISel::SelectVOP3ModsImpl(SDValue In, SDValue &Src,
3455 unsigned &Mods,
3456 bool IsCanonicalizing,
3457 bool AllowAbs) const {
3458 Mods = SISrcMods::NONE;
3459 Src = In;
3460
3461 if (Src.getOpcode() == ISD::FNEG) {
3462 Mods |= SISrcMods::NEG;
3463 Src = Src.getOperand(0);
3464 } else if (Src.getOpcode() == ISD::FSUB && IsCanonicalizing) {
3465 // Fold fsub [+-]0 into fneg. This may not have folded depending on the
3466 // denormal mode, but we're implicitly canonicalizing in a source operand.
3467 auto *LHS = dyn_cast<ConstantFPSDNode>(Src.getOperand(0));
3468 if (LHS && LHS->isZero()) {
3469 Mods |= SISrcMods::NEG;
3470 Src = Src.getOperand(1);
3471 }
3472 }
3473
3474 if (AllowAbs && Src.getOpcode() == ISD::FABS) {
3475 Mods |= SISrcMods::ABS;
3476 Src = Src.getOperand(0);
3477 }
3478
3479 if (Mods != SISrcMods::NONE)
3480 return true;
3481
3482 // Convert various sign-bit masks on integers to src mods. Currently disabled
3483 // for 16-bit types as the codegen replaces the operand without adding a
3484 // srcmod. This is intentionally finding the cases where we are performing
3485 // float neg and abs on int types, the goal is not to obtain two's complement
3486 // neg or abs. Limit converison to select operands via the nonCanonalizing
3487 // pattern.
3488 // TODO: Add 16-bit support.
3489 if (IsCanonicalizing)
3490 return true;
3491
3492 // v2i32 xor/or/and are legal. A vselect using these instructions as operands
3493 // is scalarised into two selects with EXTRACT_VECTOR_ELT operands. Peek
3494 // through the extract to the bitwise op.
3495 SDValue PeekSrc =
3496 Src->getOpcode() == ISD::EXTRACT_VECTOR_ELT ? Src->getOperand(0) : Src;
3497 // Convert various sign-bit masks to src mods. Currently disabled for 16-bit
3498 // types as the codegen replaces the operand without adding a srcmod.
3499 // This is intentionally finding the cases where we are performing float neg
3500 // and abs on int types, the goal is not to obtain two's complement neg or
3501 // abs.
3502 // TODO: Add 16-bit support.
3503 unsigned Opc = PeekSrc.getOpcode();
3504 EVT VT = Src.getValueType();
3505 if ((Opc != ISD::AND && Opc != ISD::OR && Opc != ISD::XOR) ||
3506 (VT != MVT::i32 && VT != MVT::v2i32 && VT != MVT::i64))
3507 return true;
3508
3509 ConstantSDNode *CRHS = isConstOrConstSplat(PeekSrc->getOperand(1));
3510 if (!CRHS)
3511 return true;
3512
3513 auto ReplaceSrc = [&]() -> SDValue {
3514 if (Src->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
3515 return Src.getOperand(0);
3516
3517 SDValue LHS = PeekSrc->getOperand(0);
3518 SDValue Index = Src->getOperand(1);
3519 return CurDAG->getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(Src),
3520 Src.getValueType(), LHS, Index);
3521 };
3522
3523 // Recognise Srcmods:
3524 // (xor a, 0x80000000) or v2i32 (xor a, {0x80000000,0x80000000}) as NEG.
3525 // (and a, 0x7fffffff) or v2i32 (and a, {0x7fffffff,0x7fffffff}) as ABS.
3526 // (or a, 0x80000000) or v2i32 (or a, {0x80000000,0x80000000}) as NEG+ABS
3527 // SrcModifiers.
3528 if (Opc == ISD::XOR && CRHS->getAPIntValue().isSignMask()) {
3529 Mods |= SISrcMods::NEG;
3530 Src = ReplaceSrc();
3531 } else if (Opc == ISD::AND && AllowAbs &&
3532 CRHS->getAPIntValue().isMaxSignedValue()) {
3533 Mods |= SISrcMods::ABS;
3534 Src = ReplaceSrc();
3535 } else if (Opc == ISD::OR && AllowAbs && CRHS->getAPIntValue().isSignMask()) {
3537 Src = ReplaceSrc();
3538 }
3539
3540 return true;
3541}
3542
3543bool AMDGPUDAGToDAGISel::SelectVOP3Mods(SDValue In, SDValue &Src,
3544 SDValue &SrcMods) const {
3545 unsigned Mods;
3546 if (SelectVOP3ModsImpl(In, Src, Mods, /*IsCanonicalizing=*/true,
3547 /*AllowAbs=*/true)) {
3548 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3549 return true;
3550 }
3551
3552 return false;
3553}
3554
3555bool AMDGPUDAGToDAGISel::SelectVOP3ModsNonCanonicalizing(
3556 SDValue In, SDValue &Src, SDValue &SrcMods) const {
3557 unsigned Mods;
3558 if (SelectVOP3ModsImpl(In, Src, Mods, /*IsCanonicalizing=*/false,
3559 /*AllowAbs=*/true)) {
3560 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3561 return true;
3562 }
3563
3564 return false;
3565}
3566
3567bool AMDGPUDAGToDAGISel::SelectVOP3BMods(SDValue In, SDValue &Src,
3568 SDValue &SrcMods) const {
3569 unsigned Mods;
3570 if (SelectVOP3ModsImpl(In, Src, Mods,
3571 /*IsCanonicalizing=*/true,
3572 /*AllowAbs=*/false)) {
3573 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3574 return true;
3575 }
3576
3577 return false;
3578}
3579
3580bool AMDGPUDAGToDAGISel::SelectVOP3NoMods(SDValue In, SDValue &Src) const {
3581 if (In.getOpcode() == ISD::FABS || In.getOpcode() == ISD::FNEG)
3582 return false;
3583
3584 Src = In;
3585 return true;
3586}
3587
3588bool AMDGPUDAGToDAGISel::SelectVINTERPModsImpl(SDValue In, SDValue &Src,
3589 SDValue &SrcMods,
3590 bool OpSel) const {
3591 unsigned Mods;
3592 if (SelectVOP3ModsImpl(In, Src, Mods,
3593 /*IsCanonicalizing=*/true,
3594 /*AllowAbs=*/false)) {
3595 if (OpSel)
3596 Mods |= SISrcMods::OP_SEL_0;
3597 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3598 return true;
3599 }
3600
3601 return false;
3602}
3603
3604bool AMDGPUDAGToDAGISel::SelectVINTERPMods(SDValue In, SDValue &Src,
3605 SDValue &SrcMods) const {
3606 return SelectVINTERPModsImpl(In, Src, SrcMods, /* OpSel */ false);
3607}
3608
3609bool AMDGPUDAGToDAGISel::SelectVINTERPModsHi(SDValue In, SDValue &Src,
3610 SDValue &SrcMods) const {
3611 return SelectVINTERPModsImpl(In, Src, SrcMods, /* OpSel */ true);
3612}
3613
3614bool AMDGPUDAGToDAGISel::SelectVOP3Mods0(SDValue In, SDValue &Src,
3615 SDValue &SrcMods, SDValue &Clamp,
3616 SDValue &Omod) const {
3617 SDLoc DL(In);
3618 Clamp = CurDAG->getTargetConstant(0, DL, MVT::i1);
3619 Omod = CurDAG->getTargetConstant(0, DL, MVT::i1);
3620
3621 return SelectVOP3Mods(In, Src, SrcMods);
3622}
3623
3624bool AMDGPUDAGToDAGISel::SelectVOP3BMods0(SDValue In, SDValue &Src,
3625 SDValue &SrcMods, SDValue &Clamp,
3626 SDValue &Omod) const {
3627 SDLoc DL(In);
3628 Clamp = CurDAG->getTargetConstant(0, DL, MVT::i1);
3629 Omod = CurDAG->getTargetConstant(0, DL, MVT::i1);
3630
3631 return SelectVOP3BMods(In, Src, SrcMods);
3632}
3633
3634bool AMDGPUDAGToDAGISel::SelectVOP3OMods(SDValue In, SDValue &Src,
3635 SDValue &Clamp, SDValue &Omod) const {
3636 Src = In;
3637
3638 SDLoc DL(In);
3639 Clamp = CurDAG->getTargetConstant(0, DL, MVT::i1);
3640 Omod = CurDAG->getTargetConstant(0, DL, MVT::i1);
3641
3642 return true;
3643}
3644
3645bool AMDGPUDAGToDAGISel::SelectVOP3PMods(SDValue In, SDValue &Src,
3646 SDValue &SrcMods, bool IsDOT) const {
3647 unsigned Mods = SISrcMods::NONE;
3648 Src = In;
3649
3650 // TODO: Handle G_FSUB 0 as fneg
3651 if (Src.getOpcode() == ISD::FNEG) {
3653 Src = Src.getOperand(0);
3654 }
3655
3656 // 64-bit VOP3P instructions do not have OPSEL or ABS.
3657 bool HasOpSel = Src.getValueSizeInBits() != 128;
3658
3659 if (Src.getOpcode() == ISD::BUILD_VECTOR && Src.getNumOperands() == 2 &&
3660 (!IsDOT || !Subtarget->hasDOTOpSelHazard())) {
3661 unsigned VecMods = Mods;
3662
3663 SDValue Lo = stripBitcast(Src.getOperand(0));
3664 SDValue Hi = stripBitcast(Src.getOperand(1));
3665
3666 if (Lo.getOpcode() == ISD::FNEG) {
3667 Lo = stripBitcast(Lo.getOperand(0));
3668 Mods ^= SISrcMods::NEG;
3669 }
3670
3671 if (Hi.getOpcode() == ISD::FNEG) {
3672 Hi = stripBitcast(Hi.getOperand(0));
3673 Mods ^= SISrcMods::NEG_HI;
3674 }
3675
3676 if (HasOpSel) {
3677 if (isExtractHiElt(Lo, Lo))
3678 Mods |= SISrcMods::OP_SEL_0;
3679
3680 if (isExtractHiElt(Hi, Hi))
3681 Mods |= SISrcMods::OP_SEL_1;
3682 }
3683
3684 unsigned VecSize = Src.getValueSizeInBits();
3685 Lo = stripExtractLoElt(Lo);
3686 Hi = stripExtractLoElt(Hi);
3687
3688 if (Lo.getValueSizeInBits() > VecSize) {
3689 Lo = CurDAG->getTargetExtractSubreg(
3690 (VecSize > 32) ? AMDGPU::sub0_sub1 : AMDGPU::sub0, SDLoc(In),
3691 MVT::getIntegerVT(VecSize), Lo);
3692 }
3693
3694 if (Hi.getValueSizeInBits() > VecSize) {
3695 Hi = CurDAG->getTargetExtractSubreg(
3696 (VecSize > 32) ? AMDGPU::sub0_sub1 : AMDGPU::sub0, SDLoc(In),
3697 MVT::getIntegerVT(VecSize), Hi);
3698 }
3699
3700 assert(Lo.getValueSizeInBits() <= VecSize &&
3701 Hi.getValueSizeInBits() <= VecSize);
3702
3703 if (Lo == Hi && !isInlineImmediate(Lo.getNode())) {
3704 // Really a scalar input. Just select from the low half of the register to
3705 // avoid packing.
3706
3707 if (VecSize == Lo.getValueSizeInBits()) {
3708 Src = Lo;
3709 } else if (VecSize == 32) {
3710 Src = createVOP3PSrc32FromLo16(Lo, Src, CurDAG, Subtarget);
3711 } else {
3712 assert((Lo.getValueSizeInBits() == 32 && VecSize == 64) ||
3713 (Lo.getValueSizeInBits() == 64 && VecSize == 128));
3714
3715 SDLoc SL(In);
3716 SDValue Undef = SDValue(
3717 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, SL,
3718 Lo.getValueType()), 0);
3719 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
3720 // <2 x 64> instructions do not have OPSEL and also replicate low 64
3721 // bits of a scalar input into high 64 bits. Use VGPRs in this case.
3722 // TODO: This fact can be exploited but we need to set proper OPSEL for
3723 // codegen folding purposes. It will not affect a final instruction.
3724 auto RC = Lo->isDivergent() ? TRI->getVGPRClassForBitWidth(VecSize)
3725 : TRI->getSGPRClassForBitWidth(VecSize);
3726 unsigned NumRegs = Lo.getValueSizeInBits() == 32 ? 1 : 2;
3727 const SDValue Ops[] = {
3728 CurDAG->getTargetConstant(RC->getID(), SL, MVT::i32), Lo,
3729 CurDAG->getTargetConstant(TRI->getSubRegFromChannel(0, NumRegs), SL,
3730 MVT::i32),
3731 // For packed 64-bit ops without OPSEL support, a later pass will
3732 // optimize the splat sgpr patterns to save registers.
3733 HasOpSel ? Undef : Lo,
3734 CurDAG->getTargetConstant(
3735 TRI->getSubRegFromChannel(NumRegs, NumRegs), SL, MVT::i32)};
3736
3737 Src = SDValue(CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, SL,
3738 Src.getValueType(), Ops), 0);
3739 // Check that both op_sel_0 and op_sel_1 are zero.
3741 }
3742 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3743 return true;
3744 }
3745
3746 if (VecSize == 64 && Lo == Hi && isa<ConstantFPSDNode>(Lo)) {
3747 uint64_t Lit = cast<ConstantFPSDNode>(Lo)->getValueAPF()
3748 .bitcastToAPInt().getZExtValue();
3749 if (AMDGPU::isInlinableLiteral32(Lit, Subtarget->hasInv2PiInlineImm())) {
3750 Src = CurDAG->getTargetConstant(Lit, SDLoc(In), MVT::i64);
3751 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3752 return true;
3753 }
3754 }
3755
3756 Mods = VecMods;
3757 } else if (Src.getOpcode() == ISD::VECTOR_SHUFFLE &&
3758 Src.getNumOperands() == 2) {
3759
3760 // TODO: We should repeat the build_vector source check above for the
3761 // vector_shuffle for negates and casts of individual elements.
3762
3763 assert(Src.getValueSizeInBits() != 128 &&
3764 "<2 x 64> VECTOR_SHUFFLE should not be legal.");
3765
3766 auto *SVN = cast<ShuffleVectorSDNode>(Src);
3767 ArrayRef<int> Mask = SVN->getMask();
3768
3769 if (Mask[0] < 2 && Mask[1] < 2) {
3770 // src1 should be undef.
3771 SDValue ShuffleSrc = SVN->getOperand(0);
3772
3773 if (ShuffleSrc.getOpcode() == ISD::FNEG) {
3774 ShuffleSrc = ShuffleSrc.getOperand(0);
3776 }
3777
3778 if (Mask[0] == 1)
3779 Mods |= SISrcMods::OP_SEL_0;
3780 if (Mask[1] == 1)
3781 Mods |= SISrcMods::OP_SEL_1;
3782
3783 Src = ShuffleSrc;
3784 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3785 return true;
3786 }
3787 }
3788
3789 // Packed instructions do not have abs modifiers.
3790 Mods |= SISrcMods::OP_SEL_1;
3791
3792 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3793 return true;
3794}
3795
3796bool AMDGPUDAGToDAGISel::SelectVOP3PModsDOT(SDValue In, SDValue &Src,
3797 SDValue &SrcMods) const {
3798 return SelectVOP3PMods(In, Src, SrcMods, true);
3799}
3800
3801bool AMDGPUDAGToDAGISel::SelectVOP3PNoModsDOT(SDValue In, SDValue &Src) const {
3802 SDValue SrcTmp, SrcModsTmp;
3803 SelectVOP3PMods(In, SrcTmp, SrcModsTmp, true);
3804 if (cast<ConstantSDNode>(SrcModsTmp)->getZExtValue() == SISrcMods::OP_SEL_1) {
3805 Src = SrcTmp;
3806 return true;
3807 }
3808
3809 return false;
3810}
3811
3812bool AMDGPUDAGToDAGISel::SelectVOP3PModsF32(SDValue In, SDValue &Src,
3813 SDValue &SrcMods) const {
3814 SelectVOP3Mods(In, Src, SrcMods);
3815 unsigned Mods = SISrcMods::OP_SEL_1;
3816 Mods |= cast<ConstantSDNode>(SrcMods)->getZExtValue();
3817 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3818 return true;
3819}
3820
3821bool AMDGPUDAGToDAGISel::SelectVOP3PNoModsF32(SDValue In, SDValue &Src) const {
3822 SDValue SrcTmp, SrcModsTmp;
3823 SelectVOP3PModsF32(In, SrcTmp, SrcModsTmp);
3824 if (cast<ConstantSDNode>(SrcModsTmp)->getZExtValue() == SISrcMods::OP_SEL_1) {
3825 Src = SrcTmp;
3826 return true;
3827 }
3828
3829 return false;
3830}
3831
3832bool AMDGPUDAGToDAGISel::SelectWMMAOpSelVOP3PMods(SDValue In,
3833 SDValue &Src) const {
3834 const ConstantSDNode *C = cast<ConstantSDNode>(In);
3835 assert(C->getAPIntValue().getBitWidth() == 1 && "expected i1 value");
3836
3837 unsigned Mods = SISrcMods::OP_SEL_1;
3838 unsigned SrcVal = C->getZExtValue();
3839 if (SrcVal == 1)
3840 Mods |= SISrcMods::OP_SEL_0;
3841
3842 Src = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3843 return true;
3844}
3845
3847AMDGPUDAGToDAGISel::buildRegSequence32(SmallVectorImpl<SDValue> &Elts,
3848 const SDLoc &DL) const {
3849 unsigned DstRegClass;
3850 EVT DstTy;
3851 switch (Elts.size()) {
3852 case 8:
3853 DstRegClass = AMDGPU::VReg_256RegClassID;
3854 DstTy = MVT::v8i32;
3855 break;
3856 case 4:
3857 DstRegClass = AMDGPU::VReg_128RegClassID;
3858 DstTy = MVT::v4i32;
3859 break;
3860 case 2:
3861 DstRegClass = AMDGPU::VReg_64RegClassID;
3862 DstTy = MVT::v2i32;
3863 break;
3864 default:
3865 llvm_unreachable("unhandled Reg sequence size");
3866 }
3867
3869 Ops.push_back(CurDAG->getTargetConstant(DstRegClass, DL, MVT::i32));
3870 for (unsigned i = 0; i < Elts.size(); ++i) {
3871 Ops.push_back(Elts[i]);
3872 Ops.push_back(CurDAG->getTargetConstant(
3874 }
3875 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, DL, DstTy, Ops);
3876}
3877
3879AMDGPUDAGToDAGISel::buildRegSequence16(SmallVectorImpl<SDValue> &Elts,
3880 const SDLoc &DL) const {
3881 SmallVector<SDValue, 8> PackedElts;
3882 assert("unhandled Reg sequence size" &&
3883 (Elts.size() == 8 || Elts.size() == 16));
3884
3885 // Pack 16-bit elements in pairs into 32-bit register. If both elements are
3886 // unpacked from 32-bit source use it, otherwise pack them using v_perm.
3887 for (unsigned i = 0; i < Elts.size(); i += 2) {
3888 SDValue LoSrc = stripExtractLoElt(stripBitcast(Elts[i]));
3889 SDValue HiSrc;
3890 if (isExtractHiElt(Elts[i + 1], HiSrc) && LoSrc == HiSrc) {
3891 PackedElts.push_back(HiSrc);
3892 } else {
3893 if (Subtarget->useRealTrue16Insts()) {
3894 // FIXME-TRUE16. For now pack VGPR_32 for 16-bit source before
3895 // passing to v_perm_b32. Eventually we should use replace v_perm_b32
3896 // by reg_sequence.
3897 SDValue Undef = SDValue(
3898 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i16),
3899 0);
3900 Elts[i] =
3901 emitRegSequence(*CurDAG, AMDGPU::VGPR_32RegClassID, MVT::i32,
3902 {Elts[i], Undef}, {AMDGPU::lo16, AMDGPU::hi16}, DL);
3903 Elts[i + 1] = emitRegSequence(*CurDAG, AMDGPU::VGPR_32RegClassID,
3904 MVT::i32, {Elts[i + 1], Undef},
3905 {AMDGPU::lo16, AMDGPU::hi16}, DL);
3906 }
3907 SDValue PackLoLo = CurDAG->getTargetConstant(0x05040100, DL, MVT::i32);
3908 MachineSDNode *Packed =
3909 CurDAG->getMachineNode(AMDGPU::V_PERM_B32_e64, DL, MVT::i32,
3910 {Elts[i + 1], Elts[i], PackLoLo});
3911 PackedElts.push_back(SDValue(Packed, 0));
3912 }
3913 }
3914 return buildRegSequence32(PackedElts, DL);
3915}
3916
3918AMDGPUDAGToDAGISel::buildRegSequence(SmallVectorImpl<SDValue> &Elts,
3919 const SDLoc &DL,
3920 unsigned ElementSize) const {
3921 if (ElementSize == 16)
3922 return buildRegSequence16(Elts, DL);
3923 if (ElementSize == 32)
3924 return buildRegSequence32(Elts, DL);
3925 llvm_unreachable("Unhandled element size");
3926}
3927
3928void AMDGPUDAGToDAGISel::selectWMMAModsNegAbs(unsigned ModOpcode,
3929 unsigned &Mods,
3931 SDValue &Src, const SDLoc &DL,
3932 unsigned ElementSize) const {
3933 if (ModOpcode == ISD::FNEG) {
3934 Mods |= SISrcMods::NEG;
3935 // Check if all elements also have abs modifier
3936 SmallVector<SDValue, 8> NegAbsElts;
3937 for (auto El : Elts) {
3938 if (El.getOpcode() != ISD::FABS)
3939 break;
3940 NegAbsElts.push_back(El->getOperand(0));
3941 }
3942 if (Elts.size() != NegAbsElts.size()) {
3943 // Neg
3944 Src = SDValue(buildRegSequence(Elts, DL, ElementSize), 0);
3945 } else {
3946 // Neg and Abs
3947 Mods |= SISrcMods::NEG_HI;
3948 Src = SDValue(buildRegSequence(NegAbsElts, DL, ElementSize), 0);
3949 }
3950 } else {
3951 assert(ModOpcode == ISD::FABS);
3952 // Abs
3953 Mods |= SISrcMods::NEG_HI;
3954 Src = SDValue(buildRegSequence(Elts, DL, ElementSize), 0);
3955 }
3956}
3957
3958// Check all f16 elements for modifiers while looking through b32 and v2b16
3959// build vector, stop if element does not satisfy ModifierCheck.
3960static void
3962 std::function<bool(SDValue)> ModifierCheck) {
3963 for (unsigned i = 0; i < BV->getNumOperands(); ++i) {
3964 if (auto *F16Pair =
3965 dyn_cast<BuildVectorSDNode>(stripBitcast(BV->getOperand(i)))) {
3966 for (unsigned i = 0; i < F16Pair->getNumOperands(); ++i) {
3967 SDValue ElF16 = stripBitcast(F16Pair->getOperand(i));
3968 if (!ModifierCheck(ElF16))
3969 break;
3970 }
3971 }
3972 }
3973}
3974
3975bool AMDGPUDAGToDAGISel::SelectWMMAModsF16Neg(SDValue In, SDValue &Src,
3976 SDValue &SrcMods) const {
3977 Src = In;
3978 unsigned Mods = SISrcMods::OP_SEL_1;
3979
3980 // mods are on f16 elements
3981 if (auto *BV = dyn_cast<BuildVectorSDNode>(stripBitcast(In))) {
3983
3984 checkWMMAElementsModifiersF16(BV, [&](SDValue Element) -> bool {
3985 if (Element.getOpcode() != ISD::FNEG)
3986 return false;
3987 EltsF16.push_back(Element.getOperand(0));
3988 return true;
3989 });
3990
3991 // All elements have neg modifier
3992 if (BV->getNumOperands() * 2 == EltsF16.size()) {
3993 Src = SDValue(buildRegSequence16(EltsF16, SDLoc(In)), 0);
3994 Mods |= SISrcMods::NEG;
3995 Mods |= SISrcMods::NEG_HI;
3996 }
3997 }
3998
3999 // mods are on v2f16 elements
4000 if (auto *BV = dyn_cast<BuildVectorSDNode>(stripBitcast(In))) {
4001 SmallVector<SDValue, 8> EltsV2F16;
4002 for (unsigned i = 0; i < BV->getNumOperands(); ++i) {
4003 SDValue ElV2f16 = stripBitcast(BV->getOperand(i));
4004 // Based on first element decide which mod we match, neg or abs
4005 if (ElV2f16.getOpcode() != ISD::FNEG)
4006 break;
4007 EltsV2F16.push_back(ElV2f16.getOperand(0));
4008 }
4009
4010 // All pairs of elements have neg modifier
4011 if (BV->getNumOperands() == EltsV2F16.size()) {
4012 Src = SDValue(buildRegSequence32(EltsV2F16, SDLoc(In)), 0);
4013 Mods |= SISrcMods::NEG;
4014 Mods |= SISrcMods::NEG_HI;
4015 }
4016 }
4017
4018 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4019 return true;
4020}
4021
4022bool AMDGPUDAGToDAGISel::SelectWMMAModsF16NegAbs(SDValue In, SDValue &Src,
4023 SDValue &SrcMods) const {
4024 Src = In;
4025 unsigned Mods = SISrcMods::OP_SEL_1;
4026 unsigned ModOpcode;
4027
4028 // mods are on f16 elements
4029 if (auto *BV = dyn_cast<BuildVectorSDNode>(stripBitcast(In))) {
4031 checkWMMAElementsModifiersF16(BV, [&](SDValue ElF16) -> bool {
4032 // Based on first element decide which mod we match, neg or abs
4033 if (EltsF16.empty())
4034 ModOpcode = (ElF16.getOpcode() == ISD::FNEG) ? ISD::FNEG : ISD::FABS;
4035 if (ElF16.getOpcode() != ModOpcode)
4036 return false;
4037 EltsF16.push_back(ElF16.getOperand(0));
4038 return true;
4039 });
4040
4041 // All elements have ModOpcode modifier
4042 if (BV->getNumOperands() * 2 == EltsF16.size())
4043 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF16, Src, SDLoc(In), 16);
4044 }
4045
4046 // mods are on v2f16 elements
4047 if (auto *BV = dyn_cast<BuildVectorSDNode>(stripBitcast(In))) {
4048 SmallVector<SDValue, 8> EltsV2F16;
4049
4050 for (unsigned i = 0; i < BV->getNumOperands(); ++i) {
4051 SDValue ElV2f16 = stripBitcast(BV->getOperand(i));
4052 // Based on first element decide which mod we match, neg or abs
4053 if (EltsV2F16.empty())
4054 ModOpcode = (ElV2f16.getOpcode() == ISD::FNEG) ? ISD::FNEG : ISD::FABS;
4055 if (ElV2f16->getOpcode() != ModOpcode)
4056 break;
4057 EltsV2F16.push_back(ElV2f16->getOperand(0));
4058 }
4059
4060 // All elements have ModOpcode modifier
4061 if (BV->getNumOperands() == EltsV2F16.size())
4062 selectWMMAModsNegAbs(ModOpcode, Mods, EltsV2F16, Src, SDLoc(In), 32);
4063 }
4064
4065 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4066 return true;
4067}
4068
4069bool AMDGPUDAGToDAGISel::SelectWMMAModsF32NegAbs(SDValue In, SDValue &Src,
4070 SDValue &SrcMods) const {
4071 Src = In;
4072 unsigned Mods = SISrcMods::OP_SEL_1;
4074
4075 if (auto *BV = dyn_cast<BuildVectorSDNode>(stripBitcast(In))) {
4076 assert(BV->getNumOperands() > 0);
4077 // Based on first element decide which mod we match, neg or abs
4078 SDValue ElF32 = stripBitcast(BV->getOperand(0));
4079 unsigned ModOpcode =
4080 (ElF32.getOpcode() == ISD::FNEG) ? ISD::FNEG : ISD::FABS;
4081 for (unsigned i = 0; i < BV->getNumOperands(); ++i) {
4082 SDValue ElF32 = stripBitcast(BV->getOperand(i));
4083 if (ElF32.getOpcode() != ModOpcode)
4084 break;
4085 EltsF32.push_back(ElF32.getOperand(0));
4086 }
4087
4088 // All elements had ModOpcode modifier
4089 if (BV->getNumOperands() == EltsF32.size())
4090 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF32, Src, SDLoc(In), 32);
4091 }
4092
4093 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4094 return true;
4095}
4096
4097bool AMDGPUDAGToDAGISel::SelectWMMAVISrc(SDValue In, SDValue &Src) const {
4098 if (auto *BV = dyn_cast<BuildVectorSDNode>(In)) {
4099 BitVector UndefElements;
4100 if (SDValue Splat = BV->getSplatValue(&UndefElements))
4101 if (isInlineImmediate(Splat.getNode())) {
4102 if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Splat)) {
4103 unsigned Imm = C->getAPIntValue().getSExtValue();
4104 Src = CurDAG->getTargetConstant(Imm, SDLoc(In), MVT::i32);
4105 return true;
4106 }
4107 if (const ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Splat)) {
4108 unsigned Imm = C->getValueAPF().bitcastToAPInt().getSExtValue();
4109 Src = CurDAG->getTargetConstant(Imm, SDLoc(In), MVT::i32);
4110 return true;
4111 }
4112 llvm_unreachable("unhandled Constant node");
4113 }
4114 }
4115
4116 // 16 bit splat
4117 SDValue SplatSrc32 = stripBitcast(In);
4118 if (auto *SplatSrc32BV = dyn_cast<BuildVectorSDNode>(SplatSrc32))
4119 if (SDValue Splat32 = SplatSrc32BV->getSplatValue()) {
4120 SDValue SplatSrc16 = stripBitcast(Splat32);
4121 if (auto *SplatSrc16BV = dyn_cast<BuildVectorSDNode>(SplatSrc16))
4122 if (SDValue Splat = SplatSrc16BV->getSplatValue()) {
4123 const SIInstrInfo *TII = Subtarget->getInstrInfo();
4124 std::optional<APInt> RawValue;
4125 if (const ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(Splat))
4126 RawValue = C->getValueAPF().bitcastToAPInt();
4127 else if (const ConstantSDNode *C = dyn_cast<ConstantSDNode>(Splat))
4128 RawValue = C->getAPIntValue();
4129
4130 if (RawValue.has_value()) {
4131 EVT VT = In.getValueType().getScalarType();
4132 if (VT.getSimpleVT() == MVT::f16 || VT.getSimpleVT() == MVT::bf16) {
4133 APFloat FloatVal(VT.getSimpleVT() == MVT::f16
4136 RawValue.value());
4137 if (TII->isInlineConstant(FloatVal)) {
4138 Src = CurDAG->getTargetConstant(RawValue.value(), SDLoc(In),
4139 MVT::i16);
4140 return true;
4141 }
4142 } else if (VT.getSimpleVT() == MVT::i16) {
4143 if (TII->isInlineConstant(RawValue.value())) {
4144 Src = CurDAG->getTargetConstant(RawValue.value(), SDLoc(In),
4145 MVT::i16);
4146 return true;
4147 }
4148 } else
4149 llvm_unreachable("unknown 16-bit type");
4150 }
4151 }
4152 }
4153
4154 // Currently f64 immediate vectors are represented as vectors of v2i32, with
4155 // different lo and hi 32-bit values even though double values are splated.
4156 // So we have to manually compare to determine whether it is splated.
4157 if (CurDAG->isConstantIntBuildVectorOrConstantInt(SplatSrc32)) {
4158 int64_t Imm64 = 0;
4159 for (unsigned i = 0; i < SplatSrc32->getNumOperands(); i += 2) {
4160 auto Lo32 = cast<ConstantSDNode>(SplatSrc32->getOperand(i));
4161 auto Hi32 = cast<ConstantSDNode>(SplatSrc32->getOperand(i + 1));
4162 int64_t LoImm = Lo32->getAPIntValue().getSExtValue();
4163 int64_t HiImm = Hi32->getAPIntValue().getSExtValue();
4164 int64_t Imm64I = (HiImm << 32) + LoImm;
4165 if (i == 0) {
4166 if (!isInlineImmediate(APInt(64, Imm64I)))
4167 return false;
4168 Imm64 = Imm64I;
4169 } else if (Imm64I != Imm64)
4170 return false;
4171 } // end for
4172
4173 Src = CurDAG->getTargetConstant(Imm64, SDLoc(In), MVT::i64);
4174 return true;
4175 }
4176
4177 return false;
4178}
4179
4180bool AMDGPUDAGToDAGISel::SelectSWMMACIndex8(SDValue In, SDValue &Src,
4181 SDValue &IndexKey) const {
4182 unsigned Key = 0;
4183 Src = In;
4184
4185 if (In.getOpcode() == ISD::SRL) {
4186 const llvm::SDValue &ShiftSrc = In.getOperand(0);
4187 ConstantSDNode *ShiftAmt = dyn_cast<ConstantSDNode>(In.getOperand(1));
4188 if (ShiftSrc.getValueType().getSizeInBits() == 32 && ShiftAmt &&
4189 ShiftAmt->getZExtValue() % 8 == 0) {
4190 Key = ShiftAmt->getZExtValue() / 8;
4191 Src = ShiftSrc;
4192 }
4193 }
4194
4195 IndexKey = CurDAG->getTargetConstant(Key, SDLoc(In), MVT::i32);
4196 return true;
4197}
4198
4199bool AMDGPUDAGToDAGISel::SelectSWMMACIndex16(SDValue In, SDValue &Src,
4200 SDValue &IndexKey) const {
4201 unsigned Key = 0;
4202 Src = In;
4203
4204 if (In.getOpcode() == ISD::SRL) {
4205 const llvm::SDValue &ShiftSrc = In.getOperand(0);
4206 ConstantSDNode *ShiftAmt = dyn_cast<ConstantSDNode>(In.getOperand(1));
4207 if (ShiftSrc.getValueType().getSizeInBits() == 32 && ShiftAmt &&
4208 ShiftAmt->getZExtValue() == 16) {
4209 Key = 1;
4210 Src = ShiftSrc;
4211 }
4212 }
4213
4214 IndexKey = CurDAG->getTargetConstant(Key, SDLoc(In), MVT::i32);
4215 return true;
4216}
4217
4218bool AMDGPUDAGToDAGISel::SelectSWMMACIndex32(SDValue In, SDValue &Src,
4219 SDValue &IndexKey) const {
4220 unsigned Key = 0;
4221 Src = In;
4222
4223 SDValue InI32;
4224
4225 if (In.getOpcode() == ISD::ANY_EXTEND || In.getOpcode() == ISD::ZERO_EXTEND) {
4226 const SDValue &ExtendSrc = In.getOperand(0);
4227 if (ExtendSrc.getValueSizeInBits() == 32)
4228 InI32 = ExtendSrc;
4229 } else if (In->getOpcode() == ISD::BITCAST) {
4230 const SDValue &CastSrc = In.getOperand(0);
4231 if (CastSrc.getOpcode() == ISD::BUILD_VECTOR &&
4232 CastSrc.getOperand(0).getValueSizeInBits() == 32) {
4233 ConstantSDNode *Zero = dyn_cast<ConstantSDNode>(CastSrc.getOperand(1));
4234 if (Zero && Zero->getZExtValue() == 0)
4235 InI32 = CastSrc.getOperand(0);
4236 }
4237 }
4238
4239 if (InI32 && InI32.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
4240 const SDValue &ExtractVecEltSrc = InI32.getOperand(0);
4241 ConstantSDNode *EltIdx = dyn_cast<ConstantSDNode>(InI32.getOperand(1));
4242 if (ExtractVecEltSrc.getValueSizeInBits() == 64 && EltIdx &&
4243 EltIdx->getZExtValue() == 1) {
4244 Key = 1;
4245 Src = ExtractVecEltSrc;
4246 }
4247 }
4248
4249 IndexKey = CurDAG->getTargetConstant(Key, SDLoc(In), MVT::i32);
4250 return true;
4251}
4252
4253bool AMDGPUDAGToDAGISel::SelectVOP3OpSel(SDValue In, SDValue &Src,
4254 SDValue &SrcMods) const {
4255 Src = In;
4256 // FIXME: Handle op_sel
4257 SrcMods = CurDAG->getTargetConstant(0, SDLoc(In), MVT::i32);
4258 return true;
4259}
4260
4261bool AMDGPUDAGToDAGISel::SelectVOP3OpSelMods(SDValue In, SDValue &Src,
4262 SDValue &SrcMods) const {
4263 // FIXME: Handle op_sel
4264 return SelectVOP3Mods(In, Src, SrcMods);
4265}
4266
4267// Match lowered fpext from bf16 to f32. This is a bit operation extending
4268// a 16-bit value with 16-bit of zeroes at LSB:
4269//
4270// 1. (f32 (bitcast (build_vector (i16 0), (i16 (bitcast bf16:val)))))
4271// 2. (f32 (bitcast (and i32:val, 0xffff0000))) -> IsExtractHigh = true
4272// 3. (f32 (bitcast (shl i32:va, 16) -> IsExtractHigh = false
4273static SDValue matchBF16FPExtendLike(SDValue Op, bool &IsExtractHigh) {
4274 if (Op.getValueType() != MVT::f32 || Op.getOpcode() != ISD::BITCAST)
4275 return SDValue();
4276 Op = Op.getOperand(0);
4277
4278 IsExtractHigh = false;
4279 if (Op.getValueType() == MVT::v2i16 && Op.getOpcode() == ISD::BUILD_VECTOR) {
4280 auto Low16 = dyn_cast<ConstantSDNode>(Op.getOperand(0));
4281 if (!Low16 || !Low16->isZero())
4282 return SDValue();
4283 Op = stripBitcast(Op.getOperand(1));
4284 if (Op.getValueType() != MVT::bf16)
4285 return SDValue();
4286 return Op;
4287 }
4288
4289 if (Op.getValueType() != MVT::i32)
4290 return SDValue();
4291
4292 if (Op.getOpcode() == ISD::AND) {
4293 if (auto Mask = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
4294 if (Mask->getZExtValue() == 0xffff0000) {
4295 IsExtractHigh = true;
4296 return Op.getOperand(0);
4297 }
4298 }
4299 return SDValue();
4300 }
4301
4302 if (Op.getOpcode() == ISD::SHL) {
4303 if (auto Amt = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
4304 if (Amt->getZExtValue() == 16)
4305 return Op.getOperand(0);
4306 }
4307 }
4308
4309 return SDValue();
4310}
4311
4312// The return value is not whether the match is possible (which it always is),
4313// but whether or not it a conversion is really used.
4314bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsImpl(SDValue In, SDValue &Src,
4315 unsigned &Mods,
4316 MVT VT) const {
4317 Mods = 0;
4318 SelectVOP3ModsImpl(In, Src, Mods);
4319
4320 bool IsExtractHigh = false;
4321 if (Src.getOpcode() == ISD::FP_EXTEND &&
4322 Src.getOperand(0).getValueType() == VT) {
4323 Src = Src.getOperand(0);
4324 } else if (VT == MVT::bf16) {
4325 SDValue B16 = matchBF16FPExtendLike(Src, IsExtractHigh);
4326 if (!B16)
4327 return false;
4328 Src = B16;
4329 } else
4330 return false;
4331
4332 if (Src.getValueType() != VT &&
4333 (VT != MVT::bf16 || Src.getValueType() != MVT::i32))
4334 return false;
4335
4336 Src = stripBitcast(Src);
4337
4338 // Be careful about folding modifiers if we already have an abs. fneg is
4339 // applied last, so we don't want to apply an earlier fneg.
4340 if ((Mods & SISrcMods::ABS) == 0) {
4341 unsigned ModsTmp;
4342 SelectVOP3ModsImpl(Src, Src, ModsTmp);
4343
4344 if ((ModsTmp & SISrcMods::NEG) != 0)
4345 Mods ^= SISrcMods::NEG;
4346
4347 if ((ModsTmp & SISrcMods::ABS) != 0)
4348 Mods |= SISrcMods::ABS;
4349 }
4350
4351 // op_sel/op_sel_hi decide the source type and source.
4352 // If the source's op_sel_hi is set, it indicates to do a conversion from
4353 // fp16. If the sources's op_sel is set, it picks the high half of the source
4354 // register.
4355
4356 Mods |= SISrcMods::OP_SEL_1;
4357 if (Src.getValueSizeInBits() == 16) {
4358 if (isExtractHiElt(Src, Src)) {
4359 Mods |= SISrcMods::OP_SEL_0;
4360
4361 // TODO: Should we try to look for neg/abs here?
4362 return true;
4363 }
4364
4365 if (Src.getOpcode() == ISD::TRUNCATE &&
4366 Src.getOperand(0).getValueType() == MVT::i32) {
4367 Src = Src.getOperand(0);
4368 return true;
4369 }
4370
4371 if (Subtarget->useRealTrue16Insts())
4372 // In true16 mode, pack src to a 32bit
4373 Src = createVOP3PSrc32FromLo16(Src, In, CurDAG, Subtarget);
4374 } else if (IsExtractHigh)
4375 Mods |= SISrcMods::OP_SEL_0;
4376
4377 return true;
4378}
4379
4380bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsExt(SDValue In, SDValue &Src,
4381 SDValue &SrcMods) const {
4382 unsigned Mods = 0;
4383 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16))
4384 return false;
4385 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4386 return true;
4387}
4388
4389bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixMods(SDValue In, SDValue &Src,
4390 SDValue &SrcMods) const {
4391 unsigned Mods = 0;
4392 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16);
4393 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4394 return true;
4395}
4396
4397bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsExtNeg(SDValue In, SDValue &Src,
4398 SDValue &SrcMods) const {
4399 unsigned Mods = 0;
4400 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16))
4401 return false;
4402 SrcMods =
4403 CurDAG->getTargetConstant(Mods ^ SISrcMods::NEG, SDLoc(In), MVT::i32);
4404 return true;
4405}
4406
4407bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsNeg(SDValue In, SDValue &Src,
4408 SDValue &SrcMods) const {
4409 unsigned Mods = 0;
4410 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16);
4411 SrcMods =
4412 CurDAG->getTargetConstant(Mods ^ SISrcMods::NEG, SDLoc(In), MVT::i32);
4413 return true;
4414}
4415
4416bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsExt(SDValue In, SDValue &Src,
4417 SDValue &SrcMods) const {
4418 unsigned Mods = 0;
4419 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16))
4420 return false;
4421 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4422 return true;
4423}
4424
4425bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16Mods(SDValue In, SDValue &Src,
4426 SDValue &SrcMods) const {
4427 unsigned Mods = 0;
4428 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16);
4429 SrcMods = CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4430 return true;
4431}
4432
4433bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsExtNeg(
4434 SDValue In, SDValue &Src, SDValue &SrcMods) const {
4435 unsigned Mods = 0;
4436 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16))
4437 return false;
4438 SrcMods =
4439 CurDAG->getTargetConstant(Mods ^ SISrcMods::NEG, SDLoc(In), MVT::i32);
4440 return true;
4441}
4442
4443bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsNeg(SDValue In, SDValue &Src,
4444 SDValue &SrcMods) const {
4445 unsigned Mods = 0;
4446 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16);
4447 SrcMods =
4448 CurDAG->getTargetConstant(Mods ^ SISrcMods::NEG, SDLoc(In), MVT::i32);
4449 return true;
4450}
4451
4452// Match BITOP3 operation and return a number of matched instructions plus
4453// truth table.
4454static std::pair<unsigned, uint8_t> BitOp3_Op(SDValue In,
4456 unsigned NumOpcodes = 0;
4457 uint8_t LHSBits, RHSBits;
4458
4459 auto getOperandBits = [&Src, In](SDValue Op, uint8_t &Bits) -> bool {
4460 // Define truth table given Src0, Src1, Src2 bits permutations:
4461 // 0 0 0
4462 // 0 0 1
4463 // 0 1 0
4464 // 0 1 1
4465 // 1 0 0
4466 // 1 0 1
4467 // 1 1 0
4468 // 1 1 1
4469 const uint8_t SrcBits[3] = { 0xf0, 0xcc, 0xaa };
4470
4471 if (auto *C = dyn_cast<ConstantSDNode>(Op)) {
4472 if (C->isAllOnes()) {
4473 Bits = 0xff;
4474 return true;
4475 }
4476 if (C->isZero()) {
4477 Bits = 0;
4478 return true;
4479 }
4480 }
4481
4482 for (unsigned I = 0; I < Src.size(); ++I) {
4483 // Try to find existing reused operand
4484 if (Src[I] == Op) {
4485 Bits = SrcBits[I];
4486 return true;
4487 }
4488 // Try to replace parent operator
4489 if (Src[I] == In) {
4490 Bits = SrcBits[I];
4491 Src[I] = Op;
4492 return true;
4493 }
4494 }
4495
4496 if (Src.size() == 3) {
4497 // No room left for operands. Try one last time, there can be a 'not' of
4498 // one of our source operands. In this case we can compute the bits
4499 // without growing Src vector.
4500 if (Op.getOpcode() == ISD::XOR) {
4501 if (auto *C = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
4502 if (C->isAllOnes()) {
4503 SDValue LHS = Op.getOperand(0);
4504 for (unsigned I = 0; I < Src.size(); ++I) {
4505 if (Src[I] == LHS) {
4506 Bits = ~SrcBits[I];
4507 return true;
4508 }
4509 }
4510 }
4511 }
4512 }
4513
4514 return false;
4515 }
4516
4517 Bits = SrcBits[Src.size()];
4518 Src.push_back(Op);
4519 return true;
4520 };
4521
4522 switch (In.getOpcode()) {
4523 case ISD::AND:
4524 case ISD::OR:
4525 case ISD::XOR: {
4526 SDValue LHS = In.getOperand(0);
4527 SDValue RHS = In.getOperand(1);
4528
4529 SmallVector<SDValue, 3> Backup(Src.begin(), Src.end());
4530 if (!getOperandBits(LHS, LHSBits) ||
4531 !getOperandBits(RHS, RHSBits)) {
4532 Src = std::move(Backup);
4533 return std::make_pair(0, 0);
4534 }
4535
4536 // Recursion is naturally limited by the size of the operand vector.
4537 //
4538 // When LHS and RHS share a common sub-expression, one side's recursion
4539 // may decompose that sub-expression and replace the Src slot the other
4540 // side occupies with sub-operands via the "replace parent" path in
4541 // getOperandBits. The other side's cached bit-pattern then refers to a
4542 // slot whose contents changed, producing a wrong truth table.
4543 //
4544 // We detect this in three ways:
4545 // (A) If LHS recursed, its truth table is valid against the Src state
4546 // when LHS recursion completed (SrcAfterLHS). If RHS recursion
4547 // then mutates a Src slot that LHSBits depends on, LHSBits is
4548 // stale.
4549 // (B) If RHS did not recurse, RHSBits came from getOperandBits and
4550 // refers to a specific Src slot. If that slot's contents changed
4551 // (by either recursion), RHSBits is stale.
4552 // (C) Symmetrically for LHS if it did not recurse.
4553 SmallVector<SDValue, 3> SrcBeforeRecurse(Src.begin(), Src.end());
4554 uint8_t LHSBitsOrig = LHSBits;
4555 uint8_t RHSBitsOrig = RHSBits;
4556
4557 auto LHSOp = BitOp3_Op(LHS, Src);
4558 if (LHSOp.first) {
4559 NumOpcodes += LHSOp.first;
4560 LHSBits = LHSOp.second;
4561 }
4562
4563 SmallVector<SDValue, 3> SrcAfterLHS(Src.begin(), Src.end());
4564
4565 auto RHSOp = BitOp3_Op(RHS, Src);
4566 if (RHSOp.first) {
4567 NumOpcodes += RHSOp.first;
4568 RHSBits = RHSOp.second;
4569 }
4570
4571 // dependsOnSlot: true iff the truth table TT varies with slot Slot.
4572 auto dependsOnSlot = [](uint8_t TT, int Slot) -> bool {
4573 if (Slot < 0 || Slot > 2)
4574 return false;
4575 const uint8_t Masks[3] = {0x0f, 0x33, 0x55};
4576 const int Shifts[3] = {4, 2, 1};
4577 return ((TT ^ (TT >> Shifts[Slot])) & Masks[Slot]) != 0;
4578 };
4579
4580 // findSlot: locate the Src slot a getOperandBits result depends on,
4581 // including negated (XOR with -1) patterns that getOperandBits
4582 // resolves via the NOT shortcut (~SrcBits[I]).
4583 const uint8_t SrcBitsConst[3] = {0xf0, 0xcc, 0xaa};
4584 auto findSlot = [&](uint8_t Bits, SDValue Op,
4585 const SmallVectorImpl<SDValue> &S) -> int {
4586 SDValue NegatedInner;
4587 bool IsNegationOp =
4588 Op.getOpcode() == ISD::XOR && isAllOnesConstant(Op.getOperand(1));
4589 if (IsNegationOp)
4590 NegatedInner = Op.getOperand(0);
4591 for (int I = 0; I < (int)S.size(); I++) {
4592 if (Bits == SrcBitsConst[I] && S[I] == Op)
4593 return I;
4594 if (IsNegationOp && Bits == (uint8_t)~SrcBitsConst[I] &&
4595 S[I] == NegatedInner)
4596 return I;
4597 }
4598 return -1;
4599 };
4600
4601 bool Stale = false;
4602
4603 // (A) LHS recursed: its truth table is against SrcAfterLHS.
4604 // Check if RHS recursion mutated a slot that LHSBits uses.
4605 if (LHSOp.first) {
4606 for (int I = 0; I < (int)SrcAfterLHS.size() && I < 3; I++) {
4607 if (I < (int)Src.size() && Src[I] != SrcAfterLHS[I] &&
4608 dependsOnSlot(LHSBits, I)) {
4609 Stale = true;
4610 break;
4611 }
4612 }
4613 }
4614
4615 // (B) RHS did not recurse: RHSBits from getOperandBits is against
4616 // SrcBeforeRecurse. Check if that slot was mutated since then.
4617 if (!Stale && !RHSOp.first) {
4618 int Slot = findSlot(RHSBitsOrig, RHS, SrcBeforeRecurse);
4619 if (Slot >= 0 &&
4620 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4621 Stale = true;
4622 }
4623
4624 // (C) LHS did not recurse: LHSBits from getOperandBits is against
4625 // SrcBeforeRecurse. Check if that slot was mutated since then.
4626 if (!Stale && !LHSOp.first) {
4627 int Slot = findSlot(LHSBitsOrig, LHS, SrcBeforeRecurse);
4628 if (Slot >= 0 &&
4629 (Slot >= (int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4630 Stale = true;
4631 }
4632
4633 if (Stale) {
4634 Src = std::move(SrcBeforeRecurse);
4635 LHSBits = LHSBitsOrig;
4636 RHSBits = RHSBitsOrig;
4637 NumOpcodes = 0;
4638 }
4639 break;
4640 }
4641 default:
4642 return std::make_pair(0, 0);
4643 }
4644
4645 uint8_t TTbl;
4646 switch (In.getOpcode()) {
4647 case ISD::AND:
4648 TTbl = LHSBits & RHSBits;
4649 break;
4650 case ISD::OR:
4651 TTbl = LHSBits | RHSBits;
4652 break;
4653 case ISD::XOR:
4654 TTbl = LHSBits ^ RHSBits;
4655 break;
4656 default:
4657 break;
4658 }
4659
4660 return std::make_pair(NumOpcodes + 1, TTbl);
4661}
4662
4663bool AMDGPUDAGToDAGISel::SelectBITOP3(SDValue In, SDValue &Src0, SDValue &Src1,
4664 SDValue &Src2, SDValue &Tbl) const {
4666 uint8_t TTbl;
4667 unsigned NumOpcodes;
4668
4669 std::tie(NumOpcodes, TTbl) = BitOp3_Op(In, Src);
4670
4671 // Src.empty() case can happen if all operands are all zero or all ones.
4672 // Normally it shall be optimized out before reaching this.
4673 if (NumOpcodes < 2 || Src.empty())
4674 return false;
4675
4676 // For a uniform case threshold should be higher to account for moves between
4677 // VGPRs and SGPRs. It needs one operand in a VGPR, rest two can be in SGPRs
4678 // and a readtfirstlane after.
4679 if (NumOpcodes < 4 && !In->isDivergent())
4680 return false;
4681
4682 if (NumOpcodes == 2 && In.getValueType() == MVT::i32) {
4683 // Avoid using BITOP3 for OR3, XOR3, AND_OR. This is not faster but makes
4684 // asm more readable. This cannot be modeled with AddedComplexity because
4685 // selector does not know how many operations did we match.
4686 if ((In.getOpcode() == ISD::XOR || In.getOpcode() == ISD::OR) &&
4687 (In.getOperand(0).getOpcode() == In.getOpcode() ||
4688 In.getOperand(1).getOpcode() == In.getOpcode()))
4689 return false;
4690
4691 if (In.getOpcode() == ISD::OR &&
4692 (In.getOperand(0).getOpcode() == ISD::AND ||
4693 In.getOperand(1).getOpcode() == ISD::AND))
4694 return false;
4695 }
4696
4697 // Last operand can be ignored, turning a ternary operation into a binary.
4698 // For example: (~a & b & c) | (~a & b & ~c) -> (~a & b). We can replace
4699 // 'c' with 'a' here without changing the answer. In some pathological
4700 // cases it should be possible to get an operation with a single operand
4701 // too if optimizer would not catch it.
4702 while (Src.size() < 3)
4703 Src.push_back(Src[0]);
4704
4705 Src0 = Src[0];
4706 Src1 = Src[1];
4707 Src2 = Src[2];
4708
4709 Tbl = CurDAG->getTargetConstant(TTbl, SDLoc(In), MVT::i32);
4710 return true;
4711}
4712
4713SDValue AMDGPUDAGToDAGISel::getHi16Elt(SDValue In) const {
4714 if (In.getOpcode() == ISD::POISON)
4715 return CurDAG->getPOISON(MVT::i32);
4716
4717 if (In.getOpcode() == ISD::UNDEF)
4718 return CurDAG->getUNDEF(MVT::i32);
4719
4720 if (ConstantSDNode *C = dyn_cast<ConstantSDNode>(In)) {
4721 SDLoc SL(In);
4722 return CurDAG->getConstant(C->getZExtValue() << 16, SL, MVT::i32);
4723 }
4724
4725 if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(In)) {
4726 SDLoc SL(In);
4727 return CurDAG->getConstant(
4728 C->getValueAPF().bitcastToAPInt().getZExtValue() << 16, SL, MVT::i32);
4729 }
4730
4731 SDValue Src;
4732 if (isExtractHiElt(In, Src))
4733 return Src;
4734
4735 return SDValue();
4736}
4737
4738bool AMDGPUDAGToDAGISel::isVGPRImm(const SDNode * N) const {
4739 assert(CurDAG->getTarget().getTargetTriple().isAMDGCN());
4740
4741 const SIRegisterInfo *SIRI = Subtarget->getRegisterInfo();
4742 const SIInstrInfo *SII = Subtarget->getInstrInfo();
4743
4744 unsigned Limit = 0;
4745 bool AllUsesAcceptSReg = true;
4746 for (SDNode::use_iterator U = N->use_begin(), E = SDNode::use_end();
4747 Limit < 10 && U != E; ++U, ++Limit) {
4748 const TargetRegisterClass *RC =
4749 getOperandRegClass(U->getUser(), U->getOperandNo());
4750
4751 // If the register class is unknown, it could be an unknown
4752 // register class that needs to be an SGPR, e.g. an inline asm
4753 // constraint
4754 if (!RC || SIRI->isSGPRClass(RC))
4755 return false;
4756
4757 if (RC != &AMDGPU::VS_32RegClass && RC != &AMDGPU::VS_64RegClass &&
4758 RC != &AMDGPU::VS_64_Align2RegClass) {
4759 AllUsesAcceptSReg = false;
4760 SDNode *User = U->getUser();
4761 if (User->isMachineOpcode()) {
4762 unsigned Opc = User->getMachineOpcode();
4763 const MCInstrDesc &Desc = SII->get(Opc);
4764 if (Desc.isCommutable()) {
4765 unsigned OpIdx = Desc.getNumDefs() + U->getOperandNo();
4766 unsigned CommuteIdx1 = TargetInstrInfo::CommuteAnyOperandIndex;
4767 if (SII->findCommutedOpIndices(Desc, OpIdx, CommuteIdx1)) {
4768 unsigned CommutedOpNo = CommuteIdx1 - Desc.getNumDefs();
4769 const TargetRegisterClass *CommutedRC =
4770 getOperandRegClass(U->getUser(), CommutedOpNo);
4771 if (CommutedRC == &AMDGPU::VS_32RegClass ||
4772 CommutedRC == &AMDGPU::VS_64RegClass ||
4773 CommutedRC == &AMDGPU::VS_64_Align2RegClass)
4774 AllUsesAcceptSReg = true;
4775 }
4776 }
4777 }
4778 // If "AllUsesAcceptSReg == false" so far we haven't succeeded
4779 // commuting current user. This means have at least one use
4780 // that strictly require VGPR. Thus, we will not attempt to commute
4781 // other user instructions.
4782 if (!AllUsesAcceptSReg)
4783 break;
4784 }
4785 }
4786 return !AllUsesAcceptSReg && (Limit < 10);
4787}
4788
4791 *static_cast<const AMDGPUTargetLowering*>(getTargetLowering());
4792 bool IsModified = false;
4793 do {
4794 IsModified = false;
4795
4796 // Go over all selected nodes and try to fold them a bit more
4797 SelectionDAG::allnodes_iterator Position = CurDAG->allnodes_begin();
4798 while (Position != CurDAG->allnodes_end()) {
4799 SDNode *Node = &*Position++;
4801 if (!MachineNode)
4802 continue;
4803
4804 SDNode *ResNode = Lowering.PostISelFolding(MachineNode, *CurDAG);
4805 if (ResNode != Node) {
4806 if (ResNode)
4807 ReplaceUses(Node, ResNode);
4808 IsModified = true;
4809 }
4810 }
4811 CurDAG->RemoveDeadNodes();
4812 } while (IsModified);
4813}
4814
4819
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
static bool getBaseWithOffsetUsingSplitOR(SelectionDAG &DAG, SDValue Addr, SDValue &N0, SDValue &N1)
static SDValue SelectSAddrFI(SelectionDAG *CurDAG, SDValue SAddr)
static SDValue matchExtFromI32orI32(SDValue Op, bool IsSigned, const SelectionDAG *DAG)
static MemSDNode * findMemSDNode(SDNode *N)
static bool IsCopyFromSGPR(const SIRegisterInfo &TRI, SDValue Val)
static SDValue combineBallotPattern(SDValue VCMP, bool &Negate)
static SDValue matchBF16FPExtendLike(SDValue Op, bool &IsExtractHigh)
static void checkWMMAElementsModifiersF16(BuildVectorSDNode *BV, std::function< bool(SDValue)> ModifierCheck)
Defines an instruction selector for the AMDGPU target.
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isNoUnsignedWrap(MachineInstr *Addr)
static bool isExtractHiElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static std::pair< unsigned, uint8_t > BitOp3_Op(Register R, SmallVectorImpl< Register > &Src, const MachineRegisterInfo &MRI)
static unsigned gwsIntrinToOpcode(unsigned IntrID)
Base class for AMDGPU specific classes of TargetSubtarget.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
const HexagonInstrInfo * TII
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
Provides R600 specific target descriptions.
Interface definition for R600RegisterInfo.
const SmallVectorImpl< MachineOperand > & Cond
SI DAG Lowering interface definition.
#define LLVM_DEBUG(...)
Definition Debug.h:119
LLVM IR instance of the generic uniformity analysis.
Value * RHS
Value * LHS
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
AMDGPUDAGToDAGISelLegacy(TargetMachine &TM, CodeGenOptLevel OptLevel)
bool runOnMachineFunction(MachineFunction &MF) override
runOnMachineFunction - This method must be overloaded to perform the desired machine code transformat...
StringRef getPassName() const override
getPassName - Return a nice clean name for a pass.
AMDGPU specific code to select AMDGPU machine instructions for SelectionDAG operations.
bool isSDWAOperand(const SDNode *N) const
void SelectBuildVector(SDNode *N, unsigned RegClassID)
void Select(SDNode *N) override
Main hook for targets to transform nodes into machine nodes.
bool widenRegionLoad16(SDNode *N) const
bool runOnMachineFunction(MachineFunction &MF) override
void PreprocessISelDAG() override
PreprocessISelDAG - This hook allows targets to hack on the graph before instruction selection starts...
void PostprocessISelDAG() override
PostprocessISelDAG() - This hook allows the target to hack on the graph right after selection.
bool matchLoadD16FromBuildVector(SDNode *N) const
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static SDValue stripBitcast(SDValue Val)
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
bool isSignMask() const
Check if the APInt's value is returned by getSignMask.
Definition APInt.h:462
bool isMaxSignedValue() const
Determine if this is the largest signed value.
Definition APInt.h:401
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
unsigned countr_one() const
Count the number of trailing one bits.
Definition APInt.h:1676
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
A "pseudo-class" with methods for operating on BUILD_VECTORs.
LLVM_ABI SDValue getSplatValue(const APInt &DemandedElts, BitVector *UndefElements=nullptr) const
Returns the demanded splatted value or a null value if this is not a splat.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
int64_t getSExtValue() const
Analysis pass which computes a DominatorTree.
Definition Dominators.h:241
Legacy analysis pass which computes a DominatorTree.
Definition Dominators.h:277
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
const SIInstrInfo * getInstrInfo() const override
bool useRealTrue16Insts() const
Return true if real (non-fake) variants of True16 instructions using 16-bit registers should be code-...
Generation getGeneration() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
ISD::LoadExtType getExtensionType() const
Return whether this is a plain node, or one of the varieties of value-extending loads.
Analysis pass that exposes the LoopInfo for a function.
Definition LoopInfo.h:594
SmallVector< LoopT *, 4 > getLoopsInPreorder() const
Return all of the loops in the function in preorder across the loop nests, with siblings in forward p...
The legacy pass manager's analysis pass to compute loop information.
Definition LoopInfo.h:619
unsigned getID() const
getID() - Return the register class ID number.
Machine Value Type.
static MVT getIntegerVT(unsigned BitWidth)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
An SDNode that represents everything that will be needed to construct a MachineInstr.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
AnalysisType & getAnalysis() const
getAnalysis<AnalysisType>() - This function is used by subclasses to get to the analysis information ...
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
const APInt & getAsAPIntVal() const
Helper method returns the APInt value of a ConstantSDNode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool isDivergent() const
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumOperands() const
Return the number of values used by this operation.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
bool isPredecessorOf(const SDNode *N) const
Return true if this node is a predecessor of N.
bool isAnyAdd() const
Returns true if the node type is ADD or PTRADD.
static use_iterator use_end()
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
static bool isSGPRClass(const TargetRegisterClass *RC)
bool runOnMachineFunction(MachineFunction &MF) override
runOnMachineFunction - This method must be overloaded to perform the desired machine code transformat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
SelectionDAGISelLegacy(char &ID, std::unique_ptr< SelectionDAGISel > S)
SelectionDAGISelPass(std::unique_ptr< SelectionDAGISel > Selector)
LLVM_ABI PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
std::unique_ptr< FunctionLoweringInfo > FuncInfo
const TargetLowering * TLI
const TargetInstrInfo * TII
void ReplaceUses(SDValue F, SDValue T)
ReplaceUses - replace all uses of the old node F with the use of the new node T.
void ReplaceNode(SDNode *F, SDNode *T)
Replace all uses of F with T, then remove F from the DAG.
SelectionDAGISel(TargetMachine &tm, CodeGenOptLevel OL=CodeGenOptLevel::Default)
virtual bool runOnMachineFunction(MachineFunction &mf)
const TargetLowering * getTargetLowering() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI bool isBaseWithConstantOffset(SDValue Op) const
Return true if the specified operand is an ISD::ADD with a ConstantSDNode on the right-hand side,...
MachineFunction & getMachineFunction() const
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
ilist< SDNode >::iterator allnodes_iterator
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static const unsigned CommuteAnyOperandIndex
Primary interface to the complete machine description for the target machine.
Analysis pass which computes UniformityInfo.
Legacy analysis pass which computes a CycleInfo.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool isGFX12Plus(const MCSubtargetInfo &STI)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ PTRADD
PTRADD represents pointer arithmetic semantics, for targets that opt in using shouldPreservePtrArith(...
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:527
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ CONVERGENCECTRL_GLUE
This does not correspond to any convergence control intrinsic.
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:546
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ TargetFrameIndex
Definition ISDOpcodes.h:189
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ BRCOND
BRCOND - Conditional branch.
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
bool isExtOpcode(unsigned Opcode)
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
constexpr bool isMask_32(uint32_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:256
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
bool isBoolSGPR(SDValue V)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
static bool getConstantValue(SDValue N, uint32_t &Out)
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
FunctionPass * createAMDGPUISelDag(TargetMachine &TM, CodeGenOptLevel OptLevel)
This pass converts a legalized DAG into a AMDGPU-specific.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
Definition VE.h:376
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
#define N
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
bool bitsEq(EVT VT) const
Return true if this has the same number of bits as VT.
Definition ValueTypes.h:279
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
Definition KnownBits.h:315
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
Definition KnownBits.h:146
APInt getMinValue() const
Return the minimal unsigned value possible given these KnownBits.
Definition KnownBits.h:130
static unsigned getSubRegFromChannel(unsigned Channel)
bool hasNoUnsignedWrap() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.