LLVM 24.0.0git
R600ISelLowering.cpp
Go to the documentation of this file.
1//===-- R600ISelLowering.cpp - R600 DAG Lowering Implementation -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Custom DAG lowering for R600
11//
12//===----------------------------------------------------------------------===//
13
14#include "R600ISelLowering.h"
15#include "AMDGPU.h"
18#include "R600Defines.h"
20#include "R600Subtarget.h"
22#include "llvm/IR/IntrinsicsAMDGPU.h"
23#include "llvm/IR/IntrinsicsR600.h"
24
25using namespace llvm;
26
27#define GET_CALLING_CONV_IMPL
28#include "R600GenCallingConv.inc"
29
31 const R600Subtarget &STI)
32 : AMDGPUTargetLowering(TM, STI, STI), Subtarget(&STI),
33 Gen(STI.getGeneration()) {
34 addRegisterClass(MVT::f32, &R600::R600_Reg32RegClass);
35 addRegisterClass(MVT::i32, &R600::R600_Reg32RegClass);
36 addRegisterClass(MVT::v2f32, &R600::R600_Reg64RegClass);
37 addRegisterClass(MVT::v2i32, &R600::R600_Reg64RegClass);
38 addRegisterClass(MVT::v4f32, &R600::R600_Reg128RegClass);
39 addRegisterClass(MVT::v4i32, &R600::R600_Reg128RegClass);
40
43
44 computeRegisterProperties(Subtarget->getRegisterInfo());
45
46 // Legalize loads and stores to the private address space.
47 setOperationAction(ISD::LOAD, {MVT::i32, MVT::v2i32, MVT::v4i32}, Custom);
48
49 // EXTLOAD should be the same as ZEXTLOAD. It is legal for some address
50 // spaces, so it is custom lowered to handle those where it isn't.
52 for (MVT VT : MVT::integer_valuetypes()) {
53 setLoadExtAction(Op, VT, MVT::i1, Promote);
54 setLoadExtAction(Op, VT, MVT::i8, Custom);
55 setLoadExtAction(Op, VT, MVT::i16, Custom);
56 }
57
58 // Workaround for LegalizeDAG asserting on expansion of i1 vector loads.
60 MVT::v2i1, Expand);
61
63 MVT::v4i1, Expand);
64
65 setOperationAction(ISD::STORE, {MVT::i8, MVT::i32, MVT::v2i32, MVT::v4i32},
66 Custom);
67
68 setTruncStoreAction(MVT::i32, MVT::i8, Custom);
69 setTruncStoreAction(MVT::i32, MVT::i16, Custom);
70 // We need to include these since trunc STORES to PRIVATE need
71 // special handling to accommodate RMW
72 setTruncStoreAction(MVT::v2i32, MVT::v2i16, Custom);
73 setTruncStoreAction(MVT::v4i32, MVT::v4i16, Custom);
74 setTruncStoreAction(MVT::v8i32, MVT::v8i16, Custom);
75 setTruncStoreAction(MVT::v16i32, MVT::v16i16, Custom);
76 setTruncStoreAction(MVT::v32i32, MVT::v32i16, Custom);
77 setTruncStoreAction(MVT::v2i32, MVT::v2i8, Custom);
78 setTruncStoreAction(MVT::v4i32, MVT::v4i8, Custom);
79 setTruncStoreAction(MVT::v8i32, MVT::v8i8, Custom);
80 setTruncStoreAction(MVT::v16i32, MVT::v16i8, Custom);
81 setTruncStoreAction(MVT::v32i32, MVT::v32i8, Custom);
82
83 // Workaround for LegalizeDAG asserting on expansion of i1 vector stores.
84 setTruncStoreAction(MVT::v2i32, MVT::v2i1, Expand);
85 setTruncStoreAction(MVT::v4i32, MVT::v4i1, Expand);
86
87 // Set condition code actions
91 MVT::f32, Expand);
92
94 MVT::i32, Expand);
95
97
98 setOperationAction(ISD::SETCC, {MVT::v4i32, MVT::v2i32}, Expand);
99
100 setOperationAction(ISD::BR_CC, {MVT::i32, MVT::f32}, Expand);
102
104
106 {MVT::f32, MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
107 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v16f32},
108 Expand);
109
111 MVT::f64, Custom);
112
113 setOperationAction(ISD::SELECT_CC, {MVT::f32, MVT::i32}, Custom);
114
115 setOperationAction(ISD::SETCC, {MVT::i32, MVT::f32}, Expand);
116 setOperationAction({ISD::FP_TO_UINT, ISD::FP_TO_SINT}, {MVT::i1, MVT::i64},
117 Custom);
118
119 setOperationAction(ISD::SELECT, {MVT::i32, MVT::f32, MVT::v2i32, MVT::v4i32},
120 Expand);
121
122 // ADD, SUB overflow.
123 // TODO: turn these into Legal?
124 if (Subtarget->hasCARRY())
126
127 if (Subtarget->hasBORROW())
129
130 // Expand sign extension of vectors
131 if (!Subtarget->hasBFE())
133
134 setOperationAction(ISD::SIGN_EXTEND_INREG, {MVT::v2i1, MVT::v4i1}, Expand);
135
136 if (!Subtarget->hasBFE())
138 setOperationAction(ISD::SIGN_EXTEND_INREG, {MVT::v2i8, MVT::v4i8}, Expand);
139
140 if (!Subtarget->hasBFE())
142 setOperationAction(ISD::SIGN_EXTEND_INREG, {MVT::v2i16, MVT::v4i16}, Expand);
143
145 setOperationAction(ISD::SIGN_EXTEND_INREG, {MVT::v2i32, MVT::v4i32}, Expand);
146
148
150
152 {MVT::v2i32, MVT::v2f32, MVT::v4i32, MVT::v4f32}, Custom);
153
155 {MVT::v2i32, MVT::v2f32, MVT::v4i32, MVT::v4f32}, Custom);
156
157 // We don't have 64-bit shifts. Thus we need either SHX i64 or SHX_PARTS i32
158 // to be Legal/Custom in order to avoid library calls.
160 Custom);
161
162 if (!Subtarget->hasFMA())
163 setOperationAction(ISD::FMA, {MVT::f32, MVT::f64}, Expand);
164
165 // FIXME: May need no denormals check
167
168 if (!Subtarget->hasBFI())
169 // fcopysign can be done in a single instruction with BFI.
170 setOperationAction(ISD::FCOPYSIGN, {MVT::f32, MVT::f64}, Expand);
171
172 if (!Subtarget->hasBCNT(32))
174
175 if (!Subtarget->hasBCNT(64))
177
178 if (Subtarget->hasFFBH())
180
181 if (Subtarget->hasFFBL())
183
184 // FIXME: This was moved from AMDGPUTargetLowering, I'm not sure if we
185 // need it for R600.
186 if (Subtarget->hasBFE())
188
191
192 // LLVM will expand these to atomic_cmp_swap(0)
193 // and atomic_swap, respectively.
195
196 // We need to custom lower some of the intrinsics
198 Custom);
199
201
204}
205
207 if (std::next(I) == I->getParent()->end())
208 return false;
209 return std::next(I)->getOpcode() == R600::RETURN;
210}
211
214 MachineBasicBlock *BB) const {
215 MachineFunction *MF = BB->getParent();
216 MachineRegisterInfo &MRI = MF->getRegInfo();
218 const R600InstrInfo *TII = Subtarget->getInstrInfo();
219
220 switch (MI.getOpcode()) {
221 default:
222 // Replace LDS_*_RET instruction that don't have any uses with the
223 // equivalent LDS_*_NORET instruction.
224 if (TII->isLDSRetInstr(MI.getOpcode())) {
225 int DstIdx = TII->getOperandIdx(MI.getOpcode(), R600::OpName::dst);
226 assert(DstIdx != -1);
228 // FIXME: getLDSNoRetOp method only handles LDS_1A1D LDS ops. Add
229 // LDS_1A2D support and remove this special case.
230 if (!MRI.use_empty(MI.getOperand(DstIdx).getReg()) ||
231 MI.getOpcode() == R600::LDS_CMPST_RET)
232 return BB;
233
234 NewMI = BuildMI(*BB, I, BB->findDebugLoc(I),
235 TII->get(R600::getLDSNoRetOp(MI.getOpcode())));
236 for (const MachineOperand &MO : llvm::drop_begin(MI.operands()))
237 NewMI.add(MO);
238 } else {
240 }
241 break;
242
243 case R600::FABS_R600: {
244 MachineInstr *NewMI = TII->buildDefaultInstruction(
245 *BB, I, R600::MOV, MI.getOperand(0).getReg(),
246 MI.getOperand(1).getReg());
247 TII->addFlag(*NewMI, 0, MO_FLAG_ABS);
248 break;
249 }
250
251 case R600::FNEG_R600: {
252 MachineInstr *NewMI = TII->buildDefaultInstruction(
253 *BB, I, R600::MOV, MI.getOperand(0).getReg(),
254 MI.getOperand(1).getReg());
255 TII->addFlag(*NewMI, 0, MO_FLAG_NEG);
256 break;
257 }
258
259 case R600::MASK_WRITE: {
260 Register maskedRegister = MI.getOperand(0).getReg();
261 assert(maskedRegister.isVirtual());
262 MachineInstr * defInstr = MRI.getVRegDef(maskedRegister);
263 TII->addFlag(*defInstr, 0, MO_FLAG_MASK);
264 break;
265 }
266
267 case R600::MOV_IMM_F32:
268 TII->buildMovImm(*BB, I, MI.getOperand(0).getReg(), MI.getOperand(1)
269 .getFPImm()
270 ->getValueAPF()
271 .bitcastToAPInt()
272 .getZExtValue());
273 break;
274
275 case R600::MOV_IMM_I32:
276 TII->buildMovImm(*BB, I, MI.getOperand(0).getReg(),
277 MI.getOperand(1).getImm());
278 break;
279
280 case R600::MOV_IMM_GLOBAL_ADDR: {
281 //TODO: Perhaps combine this instruction with the next if possible
282 auto MIB = TII->buildDefaultInstruction(
283 *BB, MI, R600::MOV, MI.getOperand(0).getReg(), R600::ALU_LITERAL_X);
284 int Idx = TII->getOperandIdx(*MIB, R600::OpName::literal);
285 //TODO: Ugh this is rather ugly
286 const MachineOperand &MO = MI.getOperand(1);
287 MIB->getOperand(Idx).ChangeToGA(MO.getGlobal(), MO.getOffset(),
288 MO.getTargetFlags());
289 break;
290 }
291
292 case R600::CONST_COPY: {
293 MachineInstr *NewMI = TII->buildDefaultInstruction(
294 *BB, MI, R600::MOV, MI.getOperand(0).getReg(), R600::ALU_CONST);
295 TII->setImmOperand(*NewMI, R600::OpName::src0_sel,
296 MI.getOperand(1).getImm());
297 break;
298 }
299
300 case R600::RAT_WRITE_CACHELESS_32_eg:
301 case R600::RAT_WRITE_CACHELESS_64_eg:
302 case R600::RAT_WRITE_CACHELESS_128_eg:
303 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(MI.getOpcode()))
304 .add(MI.getOperand(0))
305 .add(MI.getOperand(1))
306 .addImm(isEOP(I)); // Set End of program bit
307 break;
308
309 case R600::RAT_STORE_TYPED_eg:
310 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(MI.getOpcode()))
311 .add(MI.getOperand(0))
312 .add(MI.getOperand(1))
313 .add(MI.getOperand(2))
314 .addImm(isEOP(I)); // Set End of program bit
315 break;
316
317 case R600::BRANCH:
318 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(R600::JUMP))
319 .add(MI.getOperand(0));
320 break;
321
322 case R600::BRANCH_COND_f32: {
323 MachineInstr *NewMI =
324 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(R600::PRED_X),
325 R600::PREDICATE_BIT)
326 .add(MI.getOperand(1))
327 .addImm(R600::PRED_SETNE)
328 .addImm(0); // Flags
329 TII->addFlag(*NewMI, 0, MO_FLAG_PUSH);
330 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(R600::JUMP_COND))
331 .add(MI.getOperand(0))
332 .addReg(R600::PREDICATE_BIT, RegState::Kill);
333 break;
334 }
335
336 case R600::BRANCH_COND_i32: {
337 MachineInstr *NewMI =
338 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(R600::PRED_X),
339 R600::PREDICATE_BIT)
340 .add(MI.getOperand(1))
341 .addImm(R600::PRED_SETNE_INT)
342 .addImm(0); // Flags
343 TII->addFlag(*NewMI, 0, MO_FLAG_PUSH);
344 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(R600::JUMP_COND))
345 .add(MI.getOperand(0))
346 .addReg(R600::PREDICATE_BIT, RegState::Kill);
347 break;
348 }
349
350 case R600::EG_ExportSwz:
351 case R600::R600_ExportSwz: {
352 // Instruction is left unmodified if its not the last one of its type
353 bool isLastInstructionOfItsType = true;
354 unsigned InstExportType = MI.getOperand(1).getImm();
355 for (MachineBasicBlock::iterator NextExportInst = std::next(I),
356 EndBlock = BB->end(); NextExportInst != EndBlock;
357 NextExportInst = std::next(NextExportInst)) {
358 if (NextExportInst->getOpcode() == R600::EG_ExportSwz ||
359 NextExportInst->getOpcode() == R600::R600_ExportSwz) {
360 unsigned CurrentInstExportType = NextExportInst->getOperand(1)
361 .getImm();
362 if (CurrentInstExportType == InstExportType) {
363 isLastInstructionOfItsType = false;
364 break;
365 }
366 }
367 }
368 bool EOP = isEOP(I);
369 if (!EOP && !isLastInstructionOfItsType)
370 return BB;
371 unsigned CfInst = (MI.getOpcode() == R600::EG_ExportSwz) ? 84 : 40;
372 BuildMI(*BB, I, BB->findDebugLoc(I), TII->get(MI.getOpcode()))
373 .add(MI.getOperand(0))
374 .add(MI.getOperand(1))
375 .add(MI.getOperand(2))
376 .add(MI.getOperand(3))
377 .add(MI.getOperand(4))
378 .add(MI.getOperand(5))
379 .add(MI.getOperand(6))
380 .addImm(CfInst)
381 .addImm(EOP);
382 break;
383 }
384 case R600::RETURN: {
385 return BB;
386 }
387 }
388
389 MI.eraseFromParent();
390 return BB;
391}
392
393//===----------------------------------------------------------------------===//
394// Custom DAG Lowering Operations
395//===----------------------------------------------------------------------===//
396
400 switch (Op.getOpcode()) {
401 default: return AMDGPUTargetLowering::LowerOperation(Op, DAG);
402 case ISD::EXTRACT_VECTOR_ELT: return LowerEXTRACT_VECTOR_ELT(Op, DAG);
403 case ISD::INSERT_VECTOR_ELT: return LowerINSERT_VECTOR_ELT(Op, DAG);
404 case ISD::SHL_PARTS:
405 case ISD::SRA_PARTS:
406 case ISD::SRL_PARTS: return LowerShiftParts(Op, DAG);
407 case ISD::UADDO: return LowerUADDSUBO(Op, DAG, ISD::ADD, AMDGPUISD::CARRY);
408 case ISD::USUBO: return LowerUADDSUBO(Op, DAG, ISD::SUB, AMDGPUISD::BORROW);
409 case ISD::FCOS:
410 case ISD::FSIN: return LowerTrig(Op, DAG);
411 case ISD::SELECT_CC: return LowerSELECT_CC(Op, DAG);
412 case ISD::STORE: return LowerSTORE(Op, DAG);
413 case ISD::LOAD: {
414 SDValue Result = LowerLOAD(Op, DAG);
415 assert((!Result.getNode() ||
416 Result.getNode()->getNumValues() == 2) &&
417 "Load should return a value and a chain");
418 return Result;
419 }
420
421 case ISD::BRCOND: return LowerBRCOND(Op, DAG);
422 case ISD::GlobalAddress: return LowerGlobalAddress(MFI, Op, DAG);
423 case ISD::FrameIndex: return lowerFrameIndex(Op, DAG);
425 return lowerADDRSPACECAST(Op, DAG);
426 case ISD::INTRINSIC_VOID: {
427 SDValue Chain = Op.getOperand(0);
428 unsigned IntrinsicID = Op.getConstantOperandVal(1);
429 switch (IntrinsicID) {
430 case Intrinsic::r600_store_swizzle: {
431 SDLoc DL(Op);
432 const SDValue Args[8] = {
433 Chain,
434 Op.getOperand(2), // Export Value
435 Op.getOperand(3), // ArrayBase
436 Op.getOperand(4), // Type
437 DAG.getConstant(0, DL, MVT::i32), // SWZ_X
438 DAG.getConstant(1, DL, MVT::i32), // SWZ_Y
439 DAG.getConstant(2, DL, MVT::i32), // SWZ_Z
440 DAG.getConstant(3, DL, MVT::i32) // SWZ_W
441 };
442 return DAG.getNode(AMDGPUISD::R600_EXPORT, DL, Op.getValueType(), Args);
443 }
444
445 // default for switch(IntrinsicID)
446 default: break;
447 }
448 // break out of case ISD::INTRINSIC_VOID in switch(Op.getOpcode())
449 break;
450 }
452 unsigned IntrinsicID = Op.getConstantOperandVal(0);
453 EVT VT = Op.getValueType();
454 SDLoc DL(Op);
455 switch (IntrinsicID) {
456 case Intrinsic::r600_tex:
457 case Intrinsic::r600_texc: {
458 unsigned TextureOp;
459 switch (IntrinsicID) {
460 case Intrinsic::r600_tex:
461 TextureOp = 0;
462 break;
463 case Intrinsic::r600_texc:
464 TextureOp = 1;
465 break;
466 default:
467 llvm_unreachable("unhandled texture operation");
468 }
469
470 SDValue TexArgs[19] = {
471 DAG.getConstant(TextureOp, DL, MVT::i32),
472 Op.getOperand(1),
473 DAG.getConstant(0, DL, MVT::i32),
474 DAG.getConstant(1, DL, MVT::i32),
475 DAG.getConstant(2, DL, MVT::i32),
476 DAG.getConstant(3, DL, MVT::i32),
477 Op.getOperand(2),
478 Op.getOperand(3),
479 Op.getOperand(4),
480 DAG.getConstant(0, DL, MVT::i32),
481 DAG.getConstant(1, DL, MVT::i32),
482 DAG.getConstant(2, DL, MVT::i32),
483 DAG.getConstant(3, DL, MVT::i32),
484 Op.getOperand(5),
485 Op.getOperand(6),
486 Op.getOperand(7),
487 Op.getOperand(8),
488 Op.getOperand(9),
489 Op.getOperand(10)
490 };
491 return DAG.getNode(AMDGPUISD::TEXTURE_FETCH, DL, MVT::v4f32, TexArgs);
492 }
493 case Intrinsic::r600_dot4: {
494 SDValue Args[8] = {
495 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(1),
496 DAG.getConstant(0, DL, MVT::i32)),
497 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(2),
498 DAG.getConstant(0, DL, MVT::i32)),
499 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(1),
500 DAG.getConstant(1, DL, MVT::i32)),
501 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(2),
502 DAG.getConstant(1, DL, MVT::i32)),
503 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(1),
504 DAG.getConstant(2, DL, MVT::i32)),
505 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(2),
506 DAG.getConstant(2, DL, MVT::i32)),
507 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(1),
508 DAG.getConstant(3, DL, MVT::i32)),
509 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, Op.getOperand(2),
510 DAG.getConstant(3, DL, MVT::i32))
511 };
512 return DAG.getNode(AMDGPUISD::DOT4, DL, MVT::f32, Args);
513 }
514
515 case Intrinsic::r600_implicitarg_ptr: {
518 return DAG.getConstant(ByteOffset, DL, PtrVT);
519 }
520 case Intrinsic::r600_read_ngroups_x:
521 return LowerImplicitParameter(DAG, VT, DL, 0);
522 case Intrinsic::r600_read_ngroups_y:
523 return LowerImplicitParameter(DAG, VT, DL, 1);
524 case Intrinsic::r600_read_ngroups_z:
525 return LowerImplicitParameter(DAG, VT, DL, 2);
526 case Intrinsic::r600_read_global_size_x:
527 return LowerImplicitParameter(DAG, VT, DL, 3);
528 case Intrinsic::r600_read_global_size_y:
529 return LowerImplicitParameter(DAG, VT, DL, 4);
530 case Intrinsic::r600_read_global_size_z:
531 return LowerImplicitParameter(DAG, VT, DL, 5);
532 case Intrinsic::r600_read_local_size_x:
533 return LowerImplicitParameter(DAG, VT, DL, 6);
534 case Intrinsic::r600_read_local_size_y:
535 return LowerImplicitParameter(DAG, VT, DL, 7);
536 case Intrinsic::r600_read_local_size_z:
537 return LowerImplicitParameter(DAG, VT, DL, 8);
538
539 case Intrinsic::r600_read_tgid_x:
540 case Intrinsic::amdgcn_workgroup_id_x:
541 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
542 R600::T1_X, VT);
543 case Intrinsic::r600_read_tgid_y:
544 case Intrinsic::amdgcn_workgroup_id_y:
545 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
546 R600::T1_Y, VT);
547 case Intrinsic::r600_read_tgid_z:
548 case Intrinsic::amdgcn_workgroup_id_z:
549 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
550 R600::T1_Z, VT);
551 case Intrinsic::r600_read_tidig_x:
552 case Intrinsic::amdgcn_workitem_id_x:
553 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
554 R600::T0_X, VT);
555 case Intrinsic::r600_read_tidig_y:
556 case Intrinsic::amdgcn_workitem_id_y:
557 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
558 R600::T0_Y, VT);
559 case Intrinsic::r600_read_tidig_z:
560 case Intrinsic::amdgcn_workitem_id_z:
561 return CreateLiveInRegisterRaw(DAG, &R600::R600_TReg32RegClass,
562 R600::T0_Z, VT);
563
564 case Intrinsic::r600_recipsqrt_ieee:
565 return DAG.getNode(AMDGPUISD::RSQ, DL, VT, Op.getOperand(1));
566
567 case Intrinsic::r600_recipsqrt_clamped:
568 return DAG.getNode(AMDGPUISD::RSQ_CLAMP, DL, VT, Op.getOperand(1));
569 default:
570 return Op;
571 }
572
573 // break out of case ISD::INTRINSIC_WO_CHAIN in switch(Op.getOpcode())
574 break;
575 }
576 } // end switch(Op.getOpcode())
577 return SDValue();
578}
579
582 SelectionDAG &DAG) const {
583 switch (N->getOpcode()) {
584 default:
586 return;
587 case ISD::FP_TO_UINT:
588 if (N->getValueType(0) == MVT::i1) {
589 Results.push_back(lowerFP_TO_UINT(N->getOperand(0), DAG));
590 return;
591 }
592 // Since we don't care about out of bounds values we can use FP_TO_SINT for
593 // uints too. The DAGLegalizer code for uint considers some extra cases
594 // which are not necessary here.
595 [[fallthrough]];
596 case ISD::FP_TO_SINT: {
597 if (N->getValueType(0) == MVT::i1) {
598 Results.push_back(lowerFP_TO_SINT(N->getOperand(0), DAG));
599 return;
600 }
601
602 SDValue Result;
603 if (expandFP_TO_SINT(N, Result, DAG))
604 Results.push_back(Result);
605 return;
606 }
607 case ISD::SDIVREM: {
608 SDValue Op = SDValue(N, 1);
609 SDValue RES = LowerSDIVREM(Op, DAG);
610 Results.push_back(RES);
611 Results.push_back(RES.getValue(1));
612 break;
613 }
614 case ISD::UDIVREM: {
615 SDValue Op = SDValue(N, 0);
617 break;
618 }
619 }
620}
621
622SDValue R600TargetLowering::vectorToVerticalVector(SelectionDAG &DAG,
623 SDValue Vector) const {
624 SDLoc DL(Vector);
625 EVT VecVT = Vector.getValueType();
626 EVT EltVT = VecVT.getVectorElementType();
628
629 for (unsigned i = 0, e = VecVT.getVectorNumElements(); i != e; ++i) {
630 Args.push_back(DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, Vector,
631 DAG.getVectorIdxConstant(i, DL)));
632 }
633
634 return DAG.getNode(AMDGPUISD::BUILD_VERTICAL_VECTOR, DL, VecVT, Args);
635}
636
637SDValue R600TargetLowering::LowerEXTRACT_VECTOR_ELT(SDValue Op,
638 SelectionDAG &DAG) const {
639 SDLoc DL(Op);
640 SDValue Vector = Op.getOperand(0);
641 SDValue Index = Op.getOperand(1);
642
643 if (isa<ConstantSDNode>(Index) ||
645 return Op;
646
647 Vector = vectorToVerticalVector(DAG, Vector);
648 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, Op.getValueType(),
649 Vector, Index);
650}
651
652SDValue R600TargetLowering::LowerINSERT_VECTOR_ELT(SDValue Op,
653 SelectionDAG &DAG) const {
654 SDLoc DL(Op);
655 SDValue Vector = Op.getOperand(0);
656 SDValue Value = Op.getOperand(1);
657 SDValue Index = Op.getOperand(2);
658
659 if (isa<ConstantSDNode>(Index) ||
661 return Op;
662
663 Vector = vectorToVerticalVector(DAG, Vector);
664 SDValue Insert = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, Op.getValueType(),
665 Vector, Value, Index);
666 return vectorToVerticalVector(DAG, Insert);
667}
668
669SDValue R600TargetLowering::LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI,
670 SDValue Op,
671 SelectionDAG &DAG) const {
672 GlobalAddressSDNode *GSD = cast<GlobalAddressSDNode>(Op);
675
676 const DataLayout &DL = DAG.getDataLayout();
677 const GlobalValue *GV = GSD->getGlobal();
678 MVT ConstPtrVT = getPointerTy(DL, AMDGPUAS::CONSTANT_ADDRESS);
679
680 SDValue GA = DAG.getTargetGlobalAddress(GV, SDLoc(GSD), ConstPtrVT);
681 return DAG.getNode(AMDGPUISD::CONST_DATA_PTR, SDLoc(GSD), ConstPtrVT, GA);
682}
683
684SDValue R600TargetLowering::LowerTrig(SDValue Op, SelectionDAG &DAG) const {
685 // On hw >= R700, COS/SIN input must be between -1. and 1.
686 // Thus we lower them to TRIG ( FRACT ( x / 2Pi + 0.5) - 0.5)
687 EVT VT = Op.getValueType();
688 SDValue Arg = Op.getOperand(0);
689 SDLoc DL(Op);
690
691 // TODO: Should this propagate fast-math-flags?
692 SDValue FractPart = DAG.getNode(AMDGPUISD::FRACT, DL, VT,
693 DAG.getNode(ISD::FADD, DL, VT,
694 DAG.getNode(ISD::FMUL, DL, VT, Arg,
695 DAG.getConstantFP(0.15915494309, DL, MVT::f32)),
696 DAG.getConstantFP(0.5, DL, MVT::f32)));
697 unsigned TrigNode;
698 switch (Op.getOpcode()) {
699 case ISD::FCOS:
700 TrigNode = AMDGPUISD::COS_HW;
701 break;
702 case ISD::FSIN:
703 TrigNode = AMDGPUISD::SIN_HW;
704 break;
705 default:
706 llvm_unreachable("Wrong trig opcode");
707 }
708 SDValue TrigVal = DAG.getNode(TrigNode, DL, VT,
709 DAG.getNode(ISD::FADD, DL, VT, FractPart,
710 DAG.getConstantFP(-0.5, DL, MVT::f32)));
711 if (Gen >= AMDGPUSubtarget::R700)
712 return TrigVal;
713 // On R600 hw, COS/SIN input must be between -Pi and Pi.
714 return DAG.getNode(ISD::FMUL, DL, VT, TrigVal,
715 DAG.getConstantFP(numbers::pif, DL, MVT::f32));
716}
717
718SDValue R600TargetLowering::LowerShiftParts(SDValue Op,
719 SelectionDAG &DAG) const {
720 SDValue Lo, Hi;
721 expandShiftParts(Op.getNode(), Lo, Hi, DAG);
722 return DAG.getMergeValues({Lo, Hi}, SDLoc(Op));
723}
724
725SDValue R600TargetLowering::LowerUADDSUBO(SDValue Op, SelectionDAG &DAG,
726 unsigned mainop, unsigned ovf) const {
727 SDLoc DL(Op);
728 EVT VT = Op.getValueType();
729
730 SDValue Lo = Op.getOperand(0);
731 SDValue Hi = Op.getOperand(1);
732
733 SDValue OVF = DAG.getNode(ovf, DL, VT, Lo, Hi);
734 // Extend sign.
735 OVF = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, OVF,
736 DAG.getValueType(MVT::i1));
737
738 SDValue Res = DAG.getNode(mainop, DL, VT, Lo, Hi);
739
740 return DAG.getNode(ISD::MERGE_VALUES, DL, DAG.getVTList(VT, VT), Res, OVF);
741}
742
743SDValue R600TargetLowering::lowerFP_TO_UINT(SDValue Op, SelectionDAG &DAG) const {
744 SDLoc DL(Op);
745 return DAG.getNode(
747 DL,
748 MVT::i1,
749 Op, DAG.getConstantFP(1.0f, DL, MVT::f32),
751}
752
753SDValue R600TargetLowering::lowerFP_TO_SINT(SDValue Op, SelectionDAG &DAG) const {
754 SDLoc DL(Op);
755 return DAG.getNode(
757 DL,
758 MVT::i1,
759 Op, DAG.getConstantFP(-1.0f, DL, MVT::f32),
761}
762
763SDValue R600TargetLowering::LowerImplicitParameter(SelectionDAG &DAG, EVT VT,
764 const SDLoc &DL,
765 unsigned DwordOffset) const {
766 unsigned ByteOffset = DwordOffset * 4;
767 PointerType *PtrType =
769
770 // We shouldn't be using an offset wider than 16-bits for implicit parameters.
771 assert(isInt<16>(ByteOffset));
772
773 return DAG.getLoad(VT, DL, DAG.getEntryNode(),
774 DAG.getConstant(ByteOffset, DL, MVT::i32), // PTR
775 MachinePointerInfo(ConstantPointerNull::get(PtrType)));
776}
777
778bool R600TargetLowering::isZero(SDValue Op) const {
779 if (ConstantSDNode *Cst = dyn_cast<ConstantSDNode>(Op))
780 return Cst->isZero();
781 if (ConstantFPSDNode *CstFP = dyn_cast<ConstantFPSDNode>(Op))
782 return CstFP->isZero();
783 return false;
784}
785
786bool R600TargetLowering::isHWTrueValue(SDValue Op) const {
787 if (ConstantFPSDNode * CFP = dyn_cast<ConstantFPSDNode>(Op)) {
788 return CFP->isOne();
789 }
790 return isAllOnesConstant(Op);
791}
792
793bool R600TargetLowering::isHWFalseValue(SDValue Op) const {
794 if (ConstantFPSDNode * CFP = dyn_cast<ConstantFPSDNode>(Op)) {
795 return CFP->getValueAPF().isZero();
796 }
797 return isNullConstant(Op);
798}
799
800SDValue R600TargetLowering::LowerSELECT_CC(SDValue Op, SelectionDAG &DAG) const {
801 SDLoc DL(Op);
802 EVT VT = Op.getValueType();
803
804 SDValue LHS = Op.getOperand(0);
805 SDValue RHS = Op.getOperand(1);
806 SDValue True = Op.getOperand(2);
807 SDValue False = Op.getOperand(3);
808 SDValue CC = Op.getOperand(4);
809 SDValue Temp;
810
811 if (VT == MVT::f32) {
812 DAGCombinerInfo DCI(DAG, AfterLegalizeVectorOps, true, nullptr);
813 SDValue MinMax = combineFMinMaxLegacy(DL, VT, LHS, RHS, True, False, CC, DCI);
814 if (MinMax)
815 return MinMax;
816 }
817
818 // LHS and RHS are guaranteed to be the same value type
819 EVT CompareVT = LHS.getValueType();
820
821 // Check if we can lower this to a native operation.
822
823 // Try to lower to a SET* instruction:
824 //
825 // SET* can match the following patterns:
826 //
827 // select_cc f32, f32, -1, 0, cc_supported
828 // select_cc f32, f32, 1.0f, 0.0f, cc_supported
829 // select_cc i32, i32, -1, 0, cc_supported
830 //
831
832 // Move hardware True/False values to the correct operand.
833 if (isHWTrueValue(False) && isHWFalseValue(True)) {
834 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
835 ISD::CondCode InverseCC = ISD::getSetCCInverse(CCOpcode, CompareVT);
836 if (isCondCodeLegal(InverseCC, CompareVT.getSimpleVT())) {
837 std::swap(False, True);
838 CC = DAG.getCondCode(InverseCC);
839 } else {
840 ISD::CondCode SwapInvCC = ISD::getSetCCSwappedOperands(InverseCC);
841 if (isCondCodeLegal(SwapInvCC, CompareVT.getSimpleVT())) {
842 std::swap(False, True);
843 std::swap(LHS, RHS);
844 CC = DAG.getCondCode(SwapInvCC);
845 }
846 }
847 }
848
849 if (isHWTrueValue(True) && isHWFalseValue(False) &&
850 (CompareVT == VT || VT == MVT::i32)) {
851 // This can be matched by a SET* instruction.
852 return DAG.getNode(ISD::SELECT_CC, DL, VT, LHS, RHS, True, False, CC);
853 }
854
855 // Try to lower to a CND* instruction:
856 //
857 // CND* can match the following patterns:
858 //
859 // select_cc f32, 0.0, f32, f32, cc_supported
860 // select_cc f32, 0.0, i32, i32, cc_supported
861 // select_cc i32, 0, f32, f32, cc_supported
862 // select_cc i32, 0, i32, i32, cc_supported
863 //
864
865 // Try to move the zero value to the RHS
866 if (isZero(LHS)) {
867 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
868 // Try swapping the operands
869 ISD::CondCode CCSwapped = ISD::getSetCCSwappedOperands(CCOpcode);
870 if (isCondCodeLegal(CCSwapped, CompareVT.getSimpleVT())) {
871 std::swap(LHS, RHS);
872 CC = DAG.getCondCode(CCSwapped);
873 } else {
874 // Try inverting the condition and then swapping the operands
875 ISD::CondCode CCInv = ISD::getSetCCInverse(CCOpcode, CompareVT);
876 CCSwapped = ISD::getSetCCSwappedOperands(CCInv);
877 if (isCondCodeLegal(CCSwapped, CompareVT.getSimpleVT())) {
878 std::swap(True, False);
879 std::swap(LHS, RHS);
880 CC = DAG.getCondCode(CCSwapped);
881 }
882 }
883 }
884 if (isZero(RHS)) {
885 SDValue Cond = LHS;
886 SDValue Zero = RHS;
887 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
888 if (CompareVT != VT) {
889 // Bitcast True / False to the correct types. This will end up being
890 // a nop, but it allows us to define only a single pattern in the
891 // .TD files for each CND* instruction rather than having to have
892 // one pattern for integer True/False and one for fp True/False
893 True = DAG.getNode(ISD::BITCAST, DL, CompareVT, True);
894 False = DAG.getNode(ISD::BITCAST, DL, CompareVT, False);
895 }
896
897 switch (CCOpcode) {
898 case ISD::SETONE:
899 case ISD::SETUNE:
900 case ISD::SETNE:
901 CCOpcode = ISD::getSetCCInverse(CCOpcode, CompareVT);
902 Temp = True;
903 True = False;
904 False = Temp;
905 break;
906 default:
907 break;
908 }
909 SDValue SelectNode = DAG.getNode(ISD::SELECT_CC, DL, CompareVT,
910 Cond, Zero,
911 True, False,
912 DAG.getCondCode(CCOpcode));
913 return DAG.getNode(ISD::BITCAST, DL, VT, SelectNode);
914 }
915
916 // If we make it this for it means we have no native instructions to handle
917 // this SELECT_CC, so we must lower it.
918 SDValue HWTrue, HWFalse;
919
920 if (CompareVT == MVT::f32) {
921 HWTrue = DAG.getConstantFP(1.0f, DL, CompareVT);
922 HWFalse = DAG.getConstantFP(0.0f, DL, CompareVT);
923 } else if (CompareVT == MVT::i32) {
924 HWTrue = DAG.getAllOnesConstant(DL, CompareVT);
925 HWFalse = DAG.getConstant(0, DL, CompareVT);
926 }
927 else {
928 llvm_unreachable("Unhandled value type in LowerSELECT_CC");
929 }
930
931 // Lower this unsupported SELECT_CC into a combination of two supported
932 // SELECT_CC operations.
933 SDValue Cond = DAG.getNode(ISD::SELECT_CC, DL, CompareVT, LHS, RHS, HWTrue, HWFalse, CC);
934
935 return DAG.getNode(ISD::SELECT_CC, DL, VT,
936 Cond, HWFalse,
937 True, False,
939}
940
941SDValue R600TargetLowering::lowerADDRSPACECAST(SDValue Op,
942 SelectionDAG &DAG) const {
943 SDLoc SL(Op);
944 EVT VT = Op.getValueType();
945
946 const AddrSpaceCastSDNode *ASC = cast<AddrSpaceCastSDNode>(Op);
947 unsigned SrcAS = ASC->getSrcAddressSpace();
948 unsigned DestAS = ASC->getDestAddressSpace();
949
950 if (isNullConstant(Op.getOperand(0)) && SrcAS == AMDGPUAS::FLAT_ADDRESS)
951 return DAG.getSignedConstant(AMDGPU::getNullPointerValue(DestAS), SL, VT);
952
953 return Op;
954}
955
956/// LLVM generates byte-addressed pointers. For indirect addressing, we need to
957/// convert these pointers to a register index. Each register holds
958/// 16 bytes, (4 x 32bit sub-register), but we need to take into account the
959/// \p StackWidth, which tells us how many of the 4 sub-registers will be used
960/// for indirect addressing.
961SDValue R600TargetLowering::stackPtrToRegIndex(SDValue Ptr,
962 unsigned StackWidth,
963 SelectionDAG &DAG) const {
964 unsigned SRLPad;
965 switch(StackWidth) {
966 case 1:
967 SRLPad = 2;
968 break;
969 case 2:
970 SRLPad = 3;
971 break;
972 case 4:
973 SRLPad = 4;
974 break;
975 default: llvm_unreachable("Invalid stack width");
976 }
977
978 SDLoc DL(Ptr);
979 return DAG.getNode(ISD::SRL, DL, Ptr.getValueType(), Ptr,
980 DAG.getConstant(SRLPad, DL, MVT::i32));
981}
982
983void R600TargetLowering::getStackAddress(unsigned StackWidth,
984 unsigned ElemIdx,
985 unsigned &Channel,
986 unsigned &PtrIncr) const {
987 switch (StackWidth) {
988 default:
989 case 1:
990 Channel = 0;
991 if (ElemIdx > 0) {
992 PtrIncr = 1;
993 } else {
994 PtrIncr = 0;
995 }
996 break;
997 case 2:
998 Channel = ElemIdx % 2;
999 if (ElemIdx == 2) {
1000 PtrIncr = 1;
1001 } else {
1002 PtrIncr = 0;
1003 }
1004 break;
1005 case 4:
1006 Channel = ElemIdx;
1007 PtrIncr = 0;
1008 break;
1009 }
1010}
1011
1012SDValue R600TargetLowering::lowerPrivateTruncStore(StoreSDNode *Store,
1013 SelectionDAG &DAG) const {
1014 SDLoc DL(Store);
1015 //TODO: Who creates the i8 stores?
1016 assert(Store->isTruncatingStore()
1017 || Store->getValue().getValueType() == MVT::i8);
1018 assert(Store->getAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS);
1019
1020 SDValue Mask;
1021 if (Store->getMemoryVT() == MVT::i8) {
1022 assert(Store->getAlign() >= 1);
1023 Mask = DAG.getConstant(0xff, DL, MVT::i32);
1024 } else if (Store->getMemoryVT() == MVT::i16) {
1025 assert(Store->getAlign() >= 2);
1026 Mask = DAG.getConstant(0xffff, DL, MVT::i32);
1027 } else {
1028 llvm_unreachable("Unsupported private trunc store");
1029 }
1030
1031 SDValue OldChain = Store->getChain();
1032 bool VectorTrunc = (OldChain.getOpcode() == AMDGPUISD::DUMMY_CHAIN);
1033 // Skip dummy
1034 SDValue Chain = VectorTrunc ? OldChain->getOperand(0) : OldChain;
1035 SDValue BasePtr = Store->getBasePtr();
1036 SDValue Offset = Store->getOffset();
1037 EVT MemVT = Store->getMemoryVT();
1038
1039 SDValue LoadPtr = BasePtr;
1040 if (!Offset.isUndef()) {
1041 LoadPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr, Offset);
1042 }
1043
1044 // Get dword location
1045 // TODO: this should be eliminated by the future SHR ptr, 2
1046 SDValue Ptr = DAG.getNode(ISD::AND, DL, MVT::i32, LoadPtr,
1047 DAG.getConstant(0xfffffffc, DL, MVT::i32));
1048
1049 // Load dword
1050 // TODO: can we be smarter about machine pointer info?
1051 MachinePointerInfo PtrInfo(AMDGPUAS::PRIVATE_ADDRESS);
1052 SDValue Dst = DAG.getLoad(MVT::i32, DL, Chain, Ptr, PtrInfo);
1053
1054 Chain = Dst.getValue(1);
1055
1056 // Get offset in dword
1057 SDValue ByteIdx = DAG.getNode(ISD::AND, DL, MVT::i32, LoadPtr,
1058 DAG.getConstant(0x3, DL, MVT::i32));
1059
1060 // Convert byte offset to bit shift
1061 SDValue ShiftAmt = DAG.getNode(ISD::SHL, DL, MVT::i32, ByteIdx,
1062 DAG.getConstant(3, DL, MVT::i32));
1063
1064 // TODO: Contrary to the name of the function,
1065 // it also handles sub i32 non-truncating stores (like i1)
1066 SDValue SExtValue = DAG.getNode(ISD::SIGN_EXTEND, DL, MVT::i32,
1067 Store->getValue());
1068
1069 // Mask the value to the right type
1070 SDValue MaskedValue = DAG.getZeroExtendInReg(SExtValue, DL, MemVT);
1071
1072 // Shift the value in place
1073 SDValue ShiftedValue = DAG.getNode(ISD::SHL, DL, MVT::i32,
1074 MaskedValue, ShiftAmt);
1075
1076 // Shift the mask in place
1077 SDValue DstMask = DAG.getNode(ISD::SHL, DL, MVT::i32, Mask, ShiftAmt);
1078
1079 // Invert the mask. NOTE: if we had native ROL instructions we could
1080 // use inverted mask
1081 DstMask = DAG.getNOT(DL, DstMask, MVT::i32);
1082
1083 // Cleanup the target bits
1084 Dst = DAG.getNode(ISD::AND, DL, MVT::i32, Dst, DstMask);
1085
1086 // Add the new bits
1087 SDValue Value = DAG.getNode(ISD::OR, DL, MVT::i32, Dst, ShiftedValue);
1088
1089 // Store dword
1090 // TODO: Can we be smarter about MachinePointerInfo?
1091 SDValue NewStore = DAG.getStore(Chain, DL, Value, Ptr, PtrInfo);
1092
1093 // If we are part of expanded vector, make our neighbors depend on this store
1094 if (VectorTrunc) {
1095 // Make all other vector elements depend on this store
1096 Chain = DAG.getNode(AMDGPUISD::DUMMY_CHAIN, DL, MVT::Other, NewStore);
1097 DAG.ReplaceAllUsesOfValueWith(OldChain, Chain);
1098 }
1099 return NewStore;
1100}
1101
1102SDValue R600TargetLowering::LowerSTORE(SDValue Op, SelectionDAG &DAG) const {
1103 StoreSDNode *StoreNode = cast<StoreSDNode>(Op);
1104 unsigned AS = StoreNode->getAddressSpace();
1105
1106 SDValue Chain = StoreNode->getChain();
1107 SDValue Ptr = StoreNode->getBasePtr();
1108 SDValue Value = StoreNode->getValue();
1109
1110 EVT VT = Value.getValueType();
1111 EVT MemVT = StoreNode->getMemoryVT();
1112 EVT PtrVT = Ptr.getValueType();
1113
1114 SDLoc DL(Op);
1115
1116 const bool TruncatingStore = StoreNode->isTruncatingStore();
1117
1118 // Neither LOCAL nor PRIVATE can do vectors at the moment
1120 TruncatingStore) &&
1121 VT.isVector()) {
1122 if ((AS == AMDGPUAS::PRIVATE_ADDRESS) && TruncatingStore) {
1123 // Add an extra level of chain to isolate this vector
1124 SDValue NewChain = DAG.getNode(AMDGPUISD::DUMMY_CHAIN, DL, MVT::Other, Chain);
1125 SmallVector<SDValue, 4> NewOps(StoreNode->ops());
1126 NewOps[0] = NewChain;
1127 StoreNode = cast<StoreSDNode>(DAG.UpdateNodeOperands(StoreNode, NewOps));
1128 }
1129
1130 return scalarizeVectorStore(StoreNode, DAG);
1131 }
1132
1133 Align Alignment = StoreNode->getAlign();
1134 if (Alignment < MemVT.getStoreSize() &&
1135 !allowsMisalignedMemoryAccesses(MemVT, AS, Alignment,
1136 StoreNode->getMemOperand()->getFlags(),
1137 nullptr)) {
1138 return expandUnalignedStore(StoreNode, DAG);
1139 }
1140
1141 SDValue DWordAddr = DAG.getNode(ISD::SRL, DL, PtrVT, Ptr,
1142 DAG.getConstant(2, DL, PtrVT));
1143
1144 if (AS == AMDGPUAS::GLOBAL_ADDRESS) {
1145 // It is beneficial to create MSKOR here instead of combiner to avoid
1146 // artificial dependencies introduced by RMW
1147 if (TruncatingStore) {
1148 assert(VT.bitsLE(MVT::i32));
1149 SDValue MaskConstant;
1150 if (MemVT == MVT::i8) {
1151 MaskConstant = DAG.getConstant(0xFF, DL, MVT::i32);
1152 } else {
1153 assert(MemVT == MVT::i16);
1154 assert(StoreNode->getAlign() >= 2);
1155 MaskConstant = DAG.getConstant(0xFFFF, DL, MVT::i32);
1156 }
1157
1158 SDValue ByteIndex = DAG.getNode(ISD::AND, DL, PtrVT, Ptr,
1159 DAG.getConstant(0x00000003, DL, PtrVT));
1160 SDValue BitShift = DAG.getNode(ISD::SHL, DL, VT, ByteIndex,
1161 DAG.getConstant(3, DL, VT));
1162
1163 // Put the mask in correct place
1164 SDValue Mask = DAG.getNode(ISD::SHL, DL, VT, MaskConstant, BitShift);
1165
1166 // Put the value bits in correct place
1167 SDValue TruncValue = DAG.getNode(ISD::AND, DL, VT, Value, MaskConstant);
1168 SDValue ShiftedValue = DAG.getNode(ISD::SHL, DL, VT, TruncValue, BitShift);
1169
1170 // XXX: If we add a 64-bit ZW register class, then we could use a 2 x i32
1171 // vector instead.
1172 SDValue Src[4] = {
1173 ShiftedValue,
1174 DAG.getConstant(0, DL, MVT::i32),
1175 DAG.getConstant(0, DL, MVT::i32),
1176 Mask
1177 };
1178 SDValue Input = DAG.getBuildVector(MVT::v4i32, DL, Src);
1179 SDValue Args[3] = { Chain, Input, DWordAddr };
1180 return DAG.getMemIntrinsicNode(AMDGPUISD::STORE_MSKOR, DL,
1181 Op->getVTList(), Args, MemVT,
1182 StoreNode->getMemOperand());
1183 }
1184 if (Ptr->getOpcode() != AMDGPUISD::DWORDADDR && VT.bitsGE(MVT::i32)) {
1185 // Convert pointer from byte address to dword address.
1186 Ptr = DAG.getNode(AMDGPUISD::DWORDADDR, DL, PtrVT, DWordAddr);
1187
1188 if (StoreNode->isIndexed()) {
1189 llvm_unreachable("Indexed stores not supported yet");
1190 } else {
1191 Chain = DAG.getStore(Chain, DL, Value, Ptr, StoreNode->getMemOperand());
1192 }
1193 return Chain;
1194 }
1195 }
1196
1197 // GLOBAL_ADDRESS has been handled above, LOCAL_ADDRESS allows all sizes
1198 if (AS != AMDGPUAS::PRIVATE_ADDRESS)
1199 return SDValue();
1200
1201 if (MemVT.bitsLT(MVT::i32))
1202 return lowerPrivateTruncStore(StoreNode, DAG);
1203
1204 // Standard i32+ store, tag it with DWORDADDR to note that the address
1205 // has been shifted
1206 if (Ptr.getOpcode() != AMDGPUISD::DWORDADDR) {
1207 Ptr = DAG.getNode(AMDGPUISD::DWORDADDR, DL, PtrVT, DWordAddr);
1208 return DAG.getStore(Chain, DL, Value, Ptr, StoreNode->getMemOperand());
1209 }
1210
1211 // Tagged i32+ stores will be matched by patterns
1212 return SDValue();
1213}
1214
1215// return (512 + (kc_bank << 12)
1216static int
1218 switch (AddressSpace) {
1220 return 512;
1222 return 512 + 4096;
1224 return 512 + 4096 * 2;
1226 return 512 + 4096 * 3;
1228 return 512 + 4096 * 4;
1230 return 512 + 4096 * 5;
1232 return 512 + 4096 * 6;
1234 return 512 + 4096 * 7;
1236 return 512 + 4096 * 8;
1238 return 512 + 4096 * 9;
1240 return 512 + 4096 * 10;
1242 return 512 + 4096 * 11;
1244 return 512 + 4096 * 12;
1246 return 512 + 4096 * 13;
1248 return 512 + 4096 * 14;
1250 return 512 + 4096 * 15;
1251 default:
1252 return -1;
1253 }
1254}
1255
1256SDValue R600TargetLowering::lowerPrivateExtLoad(SDValue Op,
1257 SelectionDAG &DAG) const {
1258 SDLoc DL(Op);
1259 LoadSDNode *Load = cast<LoadSDNode>(Op);
1260 ISD::LoadExtType ExtType = Load->getExtensionType();
1261 EVT MemVT = Load->getMemoryVT();
1262 assert(Load->getAlign() >= MemVT.getStoreSize());
1263
1264 SDValue BasePtr = Load->getBasePtr();
1265 SDValue Chain = Load->getChain();
1266 SDValue Offset = Load->getOffset();
1267
1268 SDValue LoadPtr = BasePtr;
1269 if (!Offset.isUndef()) {
1270 LoadPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr, Offset);
1271 }
1272
1273 // Get dword location
1274 // NOTE: this should be eliminated by the future SHR ptr, 2
1275 SDValue Ptr = DAG.getNode(ISD::AND, DL, MVT::i32, LoadPtr,
1276 DAG.getConstant(0xfffffffc, DL, MVT::i32));
1277
1278 // Load dword
1279 // TODO: can we be smarter about machine pointer info?
1280 MachinePointerInfo PtrInfo(AMDGPUAS::PRIVATE_ADDRESS);
1281 SDValue Read = DAG.getLoad(MVT::i32, DL, Chain, Ptr, PtrInfo);
1282
1283 // Get offset within the register.
1284 SDValue ByteIdx = DAG.getNode(ISD::AND, DL, MVT::i32,
1285 LoadPtr, DAG.getConstant(0x3, DL, MVT::i32));
1286
1287 // Bit offset of target byte (byteIdx * 8).
1288 SDValue ShiftAmt = DAG.getNode(ISD::SHL, DL, MVT::i32, ByteIdx,
1289 DAG.getConstant(3, DL, MVT::i32));
1290
1291 // Shift to the right.
1292 SDValue Ret = DAG.getNode(ISD::SRL, DL, MVT::i32, Read, ShiftAmt);
1293
1294 // Eliminate the upper bits by setting them to ...
1295 EVT MemEltVT = MemVT.getScalarType();
1296
1297 if (ExtType == ISD::SEXTLOAD) { // ... ones.
1298 SDValue MemEltVTNode = DAG.getValueType(MemEltVT);
1299 Ret = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32, Ret, MemEltVTNode);
1300 } else { // ... or zeros.
1301 Ret = DAG.getZeroExtendInReg(Ret, DL, MemEltVT);
1302 }
1303
1304 SDValue Ops[] = {
1305 Ret,
1306 Read.getValue(1) // This should be our output chain
1307 };
1308
1309 return DAG.getMergeValues(Ops, DL);
1310}
1311
1312SDValue R600TargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
1313 LoadSDNode *LoadNode = cast<LoadSDNode>(Op);
1314 unsigned AS = LoadNode->getAddressSpace();
1315 EVT MemVT = LoadNode->getMemoryVT();
1316 ISD::LoadExtType ExtType = LoadNode->getExtensionType();
1317
1318 if (AS == AMDGPUAS::PRIVATE_ADDRESS &&
1319 ExtType != ISD::NON_EXTLOAD && MemVT.bitsLT(MVT::i32)) {
1320 return lowerPrivateExtLoad(Op, DAG);
1321 }
1322
1323 SDLoc DL(Op);
1324 EVT VT = Op.getValueType();
1325 SDValue Chain = LoadNode->getChain();
1326 SDValue Ptr = LoadNode->getBasePtr();
1327
1328 if ((LoadNode->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
1330 VT.isVector()) {
1331 SDValue Ops[2];
1332 std::tie(Ops[0], Ops[1]) = scalarizeVectorLoad(LoadNode, DAG);
1333 return DAG.getMergeValues(Ops, DL);
1334 }
1335
1336 // This is still used for explicit load from addrspace(8)
1337 int ConstantBlock = ConstantAddressBlock(LoadNode->getAddressSpace());
1338 if (ConstantBlock > -1 &&
1339 ((LoadNode->getExtensionType() == ISD::NON_EXTLOAD) ||
1340 (LoadNode->getExtensionType() == ISD::ZEXTLOAD))) {
1342 if (isa<Constant>(LoadNode->getMemOperand()->getValue()) ||
1343 isa<ConstantSDNode>(Ptr)) {
1344 return constBufferLoad(LoadNode, LoadNode->getAddressSpace(), DAG);
1345 }
1346 // TODO: Does this even work?
1347 // non-constant ptr can't be folded, keeps it as a v4f32 load
1348 Result = DAG.getNode(AMDGPUISD::CONST_ADDRESS, DL, MVT::v4i32,
1349 DAG.getNode(ISD::SRL, DL, MVT::i32, Ptr,
1350 DAG.getConstant(4, DL, MVT::i32)),
1351 DAG.getConstant(LoadNode->getAddressSpace() -
1353 DL, MVT::i32));
1354
1355 if (!VT.isVector()) {
1356 Result = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, Result,
1357 DAG.getConstant(0, DL, MVT::i32));
1358 }
1359
1360 SDValue MergedValues[2] = {
1361 Result,
1362 Chain
1363 };
1364 return DAG.getMergeValues(MergedValues, DL);
1365 }
1366
1367 // For most operations returning SDValue() will result in the node being
1368 // expanded by the DAG Legalizer. This is not the case for ISD::LOAD, so we
1369 // need to manually expand loads that may be legal in some address spaces and
1370 // illegal in others. SEXT loads from CONSTANT_BUFFER_0 are supported for
1371 // compute shaders, since the data is sign extended when it is uploaded to the
1372 // buffer. However SEXT loads from other address spaces are not supported, so
1373 // we need to expand them here.
1374 if (LoadNode->getExtensionType() == ISD::SEXTLOAD) {
1375 assert(!MemVT.isVector() && (MemVT == MVT::i16 || MemVT == MVT::i8));
1376 SDValue NewLoad = DAG.getExtLoad(
1377 ISD::EXTLOAD, DL, VT, Chain, Ptr, LoadNode->getPointerInfo(), MemVT,
1378 LoadNode->getAlign(), LoadNode->getMemOperand()->getFlags());
1379 SDValue Res = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, NewLoad,
1380 DAG.getValueType(MemVT));
1381
1382 SDValue MergedValues[2] = { Res, Chain };
1383 return DAG.getMergeValues(MergedValues, DL);
1384 }
1385
1386 if (LoadNode->getAddressSpace() != AMDGPUAS::PRIVATE_ADDRESS) {
1387 return SDValue();
1388 }
1389
1390 // DWORDADDR ISD marks already shifted address
1391 if (Ptr.getOpcode() != AMDGPUISD::DWORDADDR) {
1392 assert(VT == MVT::i32);
1393 Ptr = DAG.getNode(ISD::SRL, DL, MVT::i32, Ptr, DAG.getConstant(2, DL, MVT::i32));
1394 Ptr = DAG.getNode(AMDGPUISD::DWORDADDR, DL, MVT::i32, Ptr);
1395 return DAG.getLoad(MVT::i32, DL, Chain, Ptr, LoadNode->getMemOperand());
1396 }
1397 return SDValue();
1398}
1399
1400SDValue R600TargetLowering::LowerBRCOND(SDValue Op, SelectionDAG &DAG) const {
1401 SDValue Chain = Op.getOperand(0);
1402 SDValue Cond = Op.getOperand(1);
1403 SDValue Jump = Op.getOperand(2);
1404
1405 return DAG.getNode(AMDGPUISD::BRANCH_COND, SDLoc(Op), Op.getValueType(),
1406 Chain, Jump, Cond);
1407}
1408
1409SDValue R600TargetLowering::lowerFrameIndex(SDValue Op,
1410 SelectionDAG &DAG) const {
1412 const R600FrameLowering *TFL = Subtarget->getFrameLowering();
1413
1414 FrameIndexSDNode *FIN = cast<FrameIndexSDNode>(Op);
1415
1416 unsigned FrameIndex = FIN->getIndex();
1417 Register IgnoredFrameReg;
1418 StackOffset Offset =
1419 TFL->getFrameIndexReference(MF, FrameIndex, IgnoredFrameReg);
1420 return DAG.getConstant(Offset.getFixed() * 4 * TFL->getStackWidth(MF),
1421 SDLoc(Op), Op.getValueType());
1422}
1423
1425 bool IsVarArg) const {
1426 switch (CC) {
1429 case CallingConv::C:
1430 case CallingConv::Fast:
1431 case CallingConv::Cold:
1432 llvm_unreachable("kernels should not be handled here");
1440 return CC_R600;
1441 default:
1442 reportFatalUsageError("unsupported calling convention");
1443 }
1444}
1445
1446/// XXX Only kernel functions are supported, so we can assume for now that
1447/// every function is a kernel function, but in the future we should use
1448/// separate calling conventions for kernel and non-kernel functions.
1450 SDValue Chain, CallingConv::ID CallConv, bool isVarArg,
1451 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
1452 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
1454 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs,
1455 *DAG.getContext());
1457
1458 if (AMDGPU::isShader(CallConv)) {
1459 CCInfo.AnalyzeFormalArguments(Ins, CCAssignFnForCall(CallConv, isVarArg));
1460 } else {
1461 analyzeFormalArgumentsCompute(CCInfo, Ins);
1462 }
1463
1464 for (unsigned i = 0, e = Ins.size(); i < e; ++i) {
1465 CCValAssign &VA = ArgLocs[i];
1466 const ISD::InputArg &In = Ins[i];
1467 EVT VT = In.VT;
1468 EVT MemVT = VA.getLocVT();
1469 if (!VT.isVector() && MemVT.isVector()) {
1470 // Get load source type if scalarized.
1471 MemVT = MemVT.getVectorElementType();
1472 }
1473
1474 if (VT.isInteger() && !MemVT.isInteger())
1475 MemVT = MemVT.changeTypeToInteger();
1476
1477 if (AMDGPU::isShader(CallConv)) {
1478 Register Reg = MF.addLiveIn(VA.getLocReg(), &R600::R600_Reg128RegClass);
1479 SDValue Register = DAG.getCopyFromReg(Chain, DL, Reg, VT);
1480 InVals.push_back(Register);
1481 continue;
1482 }
1483
1484 // i64 isn't a legal type, so the register type used ends up as i32, which
1485 // isn't expected here. It attempts to create this sextload, but it ends up
1486 // being invalid. Somehow this seems to work with i64 arguments, but breaks
1487 // for <1 x i64>.
1488
1489 // The first 36 bytes of the input buffer contains information about
1490 // thread group and global sizes.
1492 if (MemVT.getScalarSizeInBits() != VT.getScalarSizeInBits()) {
1493 if (VT.isFloatingPoint()) {
1494 Ext = ISD::EXTLOAD;
1495 } else {
1496 // FIXME: This should really check the extload type, but the handling of
1497 // extload vector parameters seems to be broken.
1498
1499 // Ext = In.Flags.isSExt() ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
1500 Ext = ISD::SEXTLOAD;
1501 }
1502 }
1503
1504 // Compute the offset from the value.
1505 // XXX - I think PartOffset should give you this, but it seems to give the
1506 // size of the register which isn't useful.
1507
1508 unsigned PartOffset = VA.getLocMemOffset();
1509 Align Alignment = commonAlignment(Align(VT.getStoreSize()), PartOffset);
1510
1512 SDValue Arg =
1513 DAG.getLoad(ISD::UNINDEXED, Ext, VT, DL, Chain,
1514 DAG.getConstant(PartOffset, DL, MVT::i32),
1515 DAG.getPOISON(MVT::i32), PtrInfo, MemVT, Alignment,
1519
1520 InVals.push_back(Arg);
1521 }
1522 return Chain;
1523}
1524
1526 EVT VT) const {
1527 if (!VT.isVector())
1528 return MVT::i32;
1530}
1531
1533 const MachineFunction &MF) const {
1534 // Local and Private addresses do not handle vectors. Limit to i32
1536 return (MemVT.getSizeInBits() <= 32);
1537 }
1538 return true;
1539}
1540
1542 EVT VT, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags,
1543 unsigned *IsFast) const {
1544 if (IsFast)
1545 *IsFast = 0;
1546
1547 if (!VT.isSimple() || VT == MVT::Other)
1548 return false;
1549
1550 if (VT.bitsLT(MVT::i32))
1551 return false;
1552
1553 // TODO: This is a rough estimate.
1554 if (IsFast)
1555 *IsFast = 1;
1556
1557 return VT.bitsGT(MVT::i32) && Alignment >= Align(4);
1558}
1559
1561 SelectionDAG &DAG, SDValue VectorEntry,
1562 DenseMap<unsigned, unsigned> &RemapSwizzle) {
1563 assert(RemapSwizzle.empty());
1564
1565 SDLoc DL(VectorEntry);
1566 EVT EltTy = VectorEntry.getValueType().getVectorElementType();
1567
1568 SDValue NewBldVec[4];
1569 for (unsigned i = 0; i < 4; i++)
1570 NewBldVec[i] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltTy, VectorEntry,
1571 DAG.getIntPtrConstant(i, DL));
1572
1573 for (unsigned i = 0; i < 4; i++) {
1574 if (NewBldVec[i].isUndef())
1575 // We mask write here to teach later passes that the ith element of this
1576 // vector is undef. Thus we can use it to reduce 128 bits reg usage,
1577 // break false dependencies and additionally make assembly easier to read.
1578 RemapSwizzle[i] = 7; // SEL_MASK_WRITE
1579 if (ConstantFPSDNode *C = dyn_cast<ConstantFPSDNode>(NewBldVec[i])) {
1580 if (C->isZero()) {
1581 RemapSwizzle[i] = 4; // SEL_0
1582 NewBldVec[i] = DAG.getUNDEF(MVT::f32);
1583 } else if (C->isOne()) {
1584 RemapSwizzle[i] = 5; // SEL_1
1585 NewBldVec[i] = DAG.getUNDEF(MVT::f32);
1586 }
1587 }
1588
1589 if (NewBldVec[i].isUndef())
1590 continue;
1591
1592 for (unsigned j = 0; j < i; j++) {
1593 if (NewBldVec[i] == NewBldVec[j]) {
1594 NewBldVec[i] = DAG.getUNDEF(NewBldVec[i].getValueType());
1595 RemapSwizzle[i] = j;
1596 break;
1597 }
1598 }
1599 }
1600
1601 return DAG.getBuildVector(VectorEntry.getValueType(), SDLoc(VectorEntry),
1602 NewBldVec);
1603}
1604
1606 DenseMap<unsigned, unsigned> &RemapSwizzle) {
1607 assert(RemapSwizzle.empty());
1608
1609 SDLoc DL(VectorEntry);
1610 EVT EltTy = VectorEntry.getValueType().getVectorElementType();
1611
1612 SDValue NewBldVec[4];
1613 bool isUnmovable[4] = {false, false, false, false};
1614 for (unsigned i = 0; i < 4; i++)
1615 NewBldVec[i] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltTy, VectorEntry,
1616 DAG.getIntPtrConstant(i, DL));
1617
1618 for (unsigned i = 0; i < 4; i++) {
1619 RemapSwizzle[i] = i;
1620 if (NewBldVec[i].getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
1621 unsigned Idx = NewBldVec[i].getConstantOperandVal(1);
1622 if (i == Idx)
1623 isUnmovable[Idx] = true;
1624 }
1625 }
1626
1627 for (unsigned i = 0; i < 4; i++) {
1628 if (NewBldVec[i].getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
1629 unsigned Idx = NewBldVec[i].getConstantOperandVal(1);
1630 if (isUnmovable[Idx])
1631 continue;
1632 // Swap i and Idx
1633 std::swap(NewBldVec[Idx], NewBldVec[i]);
1634 std::swap(RemapSwizzle[i], RemapSwizzle[Idx]);
1635 break;
1636 }
1637 }
1638
1639 return DAG.getBuildVector(VectorEntry.getValueType(), SDLoc(VectorEntry),
1640 NewBldVec);
1641}
1642
1643SDValue R600TargetLowering::OptimizeSwizzle(SDValue BuildVector, SDValue Swz[],
1644 SelectionDAG &DAG,
1645 const SDLoc &DL) const {
1646 // Old -> New swizzle values
1647 DenseMap<unsigned, unsigned> SwizzleRemap;
1648
1649 BuildVector = CompactSwizzlableVector(DAG, BuildVector, SwizzleRemap);
1650 for (unsigned i = 0; i < 4; i++) {
1651 unsigned Idx = Swz[i]->getAsZExtVal();
1652 auto It = SwizzleRemap.find(Idx);
1653 if (It != SwizzleRemap.end())
1654 Swz[i] = DAG.getConstant(It->second, DL, MVT::i32);
1655 }
1656
1657 SwizzleRemap.clear();
1658 BuildVector = ReorganizeVector(DAG, BuildVector, SwizzleRemap);
1659 for (unsigned i = 0; i < 4; i++) {
1660 unsigned Idx = Swz[i]->getAsZExtVal();
1661 auto It = SwizzleRemap.find(Idx);
1662 if (It != SwizzleRemap.end())
1663 Swz[i] = DAG.getConstant(It->second, DL, MVT::i32);
1664 }
1665
1666 return BuildVector;
1667}
1668
1669SDValue R600TargetLowering::constBufferLoad(LoadSDNode *LoadNode, int Block,
1670 SelectionDAG &DAG) const {
1671 SDLoc DL(LoadNode);
1672 EVT VT = LoadNode->getValueType(0);
1673 SDValue Chain = LoadNode->getChain();
1674 SDValue Ptr = LoadNode->getBasePtr();
1676
1677 //TODO: Support smaller loads
1678 if (LoadNode->getMemoryVT().getScalarType() != MVT::i32 || !ISD::isNON_EXTLoad(LoadNode))
1679 return SDValue();
1680
1681 if (LoadNode->getAlign() < Align(4))
1682 return SDValue();
1683
1684 int ConstantBlock = ConstantAddressBlock(Block);
1685
1686 SDValue Slots[4];
1687 for (unsigned i = 0; i < 4; i++) {
1688 // We want Const position encoded with the following formula :
1689 // (((512 + (kc_bank << 12) + const_index) << 2) + chan)
1690 // const_index is Ptr computed by llvm using an alignment of 16.
1691 // Thus we add (((512 + (kc_bank << 12)) + chan ) * 4 here and
1692 // then div by 4 at the ISel step
1693 SDValue NewPtr = DAG.getNode(ISD::ADD, DL, Ptr.getValueType(), Ptr,
1694 DAG.getConstant(4 * i + ConstantBlock * 16, DL, MVT::i32));
1695 Slots[i] = DAG.getNode(AMDGPUISD::CONST_ADDRESS, DL, MVT::i32, NewPtr);
1696 }
1697 EVT NewVT = MVT::v4i32;
1698 unsigned NumElements = 4;
1699 if (VT.isVector()) {
1700 NewVT = VT;
1701 NumElements = VT.getVectorNumElements();
1702 }
1703 SDValue Result = DAG.getBuildVector(NewVT, DL, ArrayRef(Slots, NumElements));
1704 if (!VT.isVector()) {
1705 Result = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, Result,
1706 DAG.getConstant(0, DL, MVT::i32));
1707 }
1708 SDValue MergedValues[2] = {
1709 Result,
1710 Chain
1711 };
1712 return DAG.getMergeValues(MergedValues, DL);
1713}
1714
1715//===----------------------------------------------------------------------===//
1716// Custom DAG Optimizations
1717//===----------------------------------------------------------------------===//
1718
1720 DAGCombinerInfo &DCI) const {
1721 SelectionDAG &DAG = DCI.DAG;
1722 SDLoc DL(N);
1723
1724 switch (N->getOpcode()) {
1725 // (f32 fp_round (f64 uint_to_fp a)) -> (f32 uint_to_fp a)
1726 case ISD::FP_ROUND: {
1727 SDValue Arg = N->getOperand(0);
1728 if (Arg.getOpcode() == ISD::UINT_TO_FP && Arg.getValueType() == MVT::f64) {
1729 return DAG.getNode(ISD::UINT_TO_FP, DL, N->getValueType(0),
1730 Arg.getOperand(0));
1731 }
1732 break;
1733 }
1734
1735 // (i32 fp_to_sint (fneg (select_cc f32, f32, 1.0, 0.0 cc))) ->
1736 // (i32 select_cc f32, f32, -1, 0 cc)
1737 //
1738 // Mesa's GLSL frontend generates the above pattern a lot and we can lower
1739 // this to one of the SET*_DX10 instructions.
1740 case ISD::FP_TO_SINT: {
1741 SDValue FNeg = N->getOperand(0);
1742 if (FNeg.getOpcode() != ISD::FNEG) {
1743 return SDValue();
1744 }
1745 SDValue SelectCC = FNeg.getOperand(0);
1746 if (SelectCC.getOpcode() != ISD::SELECT_CC ||
1747 SelectCC.getOperand(0).getValueType() != MVT::f32 || // LHS
1748 SelectCC.getOperand(2).getValueType() != MVT::f32 || // True
1749 !isHWTrueValue(SelectCC.getOperand(2)) ||
1750 !isHWFalseValue(SelectCC.getOperand(3))) {
1751 return SDValue();
1752 }
1753
1754 return DAG.getNode(ISD::SELECT_CC, DL, N->getValueType(0),
1755 SelectCC.getOperand(0), // LHS
1756 SelectCC.getOperand(1), // RHS
1757 DAG.getAllOnesConstant(DL, MVT::i32), // True
1758 DAG.getConstant(0, DL, MVT::i32), // False
1759 SelectCC.getOperand(4)); // CC
1760 }
1761
1762 // insert_vector_elt (build_vector elt0, ... , eltN), NewEltIdx, idx
1763 // => build_vector elt0, ... , NewEltIdx, ... , eltN
1765 SDValue InVec = N->getOperand(0);
1766 SDValue InVal = N->getOperand(1);
1767 SDValue EltNo = N->getOperand(2);
1768
1769 // If the inserted element is an UNDEF, just use the input vector.
1770 if (InVal.isUndef())
1771 return InVec;
1772
1773 EVT VT = InVec.getValueType();
1774
1775 // If we can't generate a legal BUILD_VECTOR, exit
1777 return SDValue();
1778
1779 // Check that we know which element is being inserted
1780 if (!isa<ConstantSDNode>(EltNo))
1781 return SDValue();
1782 unsigned Elt = EltNo->getAsZExtVal();
1783
1784 // Check that the operand is a BUILD_VECTOR (or UNDEF, which can essentially
1785 // be converted to a BUILD_VECTOR). Fill in the Ops vector with the
1786 // vector elements.
1788 if (InVec.getOpcode() == ISD::BUILD_VECTOR) {
1789 Ops.append(InVec.getNode()->op_begin(),
1790 InVec.getNode()->op_end());
1791 } else if (InVec.isUndef()) {
1792 unsigned NElts = VT.getVectorNumElements();
1793 Ops.append(NElts, DAG.getUNDEF(InVal.getValueType()));
1794 } else {
1795 return SDValue();
1796 }
1797
1798 // Insert the element
1799 if (Elt < Ops.size()) {
1800 // All the operands of BUILD_VECTOR must have the same type;
1801 // we enforce that here.
1802 EVT OpVT = Ops[0].getValueType();
1803 if (InVal.getValueType() != OpVT)
1804 InVal = OpVT.bitsGT(InVal.getValueType()) ?
1805 DAG.getNode(ISD::ANY_EXTEND, DL, OpVT, InVal) :
1806 DAG.getNode(ISD::TRUNCATE, DL, OpVT, InVal);
1807 Ops[Elt] = InVal;
1808 }
1809
1810 // Return the new vector
1811 return DAG.getBuildVector(VT, DL, Ops);
1812 }
1813
1814 // Extract_vec (Build_vector) generated by custom lowering
1815 // also needs to be customly combined
1817 SDValue Arg = N->getOperand(0);
1818 if (Arg.getOpcode() == ISD::BUILD_VECTOR) {
1819 if (ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N->getOperand(1))) {
1820 unsigned Element = Const->getZExtValue();
1821 return Arg->getOperand(Element);
1822 }
1823 }
1824 if (Arg.getOpcode() == ISD::BITCAST &&
1828 if (ConstantSDNode *Const = dyn_cast<ConstantSDNode>(N->getOperand(1))) {
1829 unsigned Element = Const->getZExtValue();
1830 return DAG.getNode(ISD::BITCAST, DL, N->getVTList(),
1831 Arg->getOperand(0).getOperand(Element));
1832 }
1833 }
1834 break;
1835 }
1836
1837 case ISD::SELECT_CC: {
1838 // Try common optimizations
1840 return Ret;
1841
1842 // fold selectcc (selectcc x, y, a, b, cc), b, a, b, seteq ->
1843 // selectcc x, y, a, b, inv(cc)
1844 //
1845 // fold selectcc (selectcc x, y, a, b, cc), b, a, b, setne ->
1846 // selectcc x, y, a, b, cc
1847 SDValue LHS = N->getOperand(0);
1848 if (LHS.getOpcode() != ISD::SELECT_CC) {
1849 return SDValue();
1850 }
1851
1852 SDValue RHS = N->getOperand(1);
1853 SDValue True = N->getOperand(2);
1854 SDValue False = N->getOperand(3);
1855 ISD::CondCode NCC = cast<CondCodeSDNode>(N->getOperand(4))->get();
1856
1857 if (LHS.getOperand(2).getNode() != True.getNode() ||
1858 LHS.getOperand(3).getNode() != False.getNode() ||
1859 RHS.getNode() != False.getNode()) {
1860 return SDValue();
1861 }
1862
1863 switch (NCC) {
1864 default: return SDValue();
1865 case ISD::SETNE: return LHS;
1866 case ISD::SETEQ: {
1867 ISD::CondCode LHSCC = cast<CondCodeSDNode>(LHS.getOperand(4))->get();
1868 LHSCC = ISD::getSetCCInverse(LHSCC, LHS.getOperand(0).getValueType());
1869 if (DCI.isBeforeLegalizeOps() ||
1870 isCondCodeLegal(LHSCC, LHS.getOperand(0).getSimpleValueType()))
1871 return DAG.getSelectCC(DL,
1872 LHS.getOperand(0),
1873 LHS.getOperand(1),
1874 LHS.getOperand(2),
1875 LHS.getOperand(3),
1876 LHSCC);
1877 break;
1878 }
1879 }
1880 return SDValue();
1881 }
1882
1884 SDValue Arg = N->getOperand(1);
1885 if (Arg.getOpcode() != ISD::BUILD_VECTOR)
1886 break;
1887
1888 SDValue NewArgs[8] = {
1889 N->getOperand(0), // Chain
1890 SDValue(),
1891 N->getOperand(2), // ArrayBase
1892 N->getOperand(3), // Type
1893 N->getOperand(4), // SWZ_X
1894 N->getOperand(5), // SWZ_Y
1895 N->getOperand(6), // SWZ_Z
1896 N->getOperand(7) // SWZ_W
1897 };
1898 NewArgs[1] = OptimizeSwizzle(N->getOperand(1), &NewArgs[4], DAG, DL);
1899 return DAG.getNode(AMDGPUISD::R600_EXPORT, DL, N->getVTList(), NewArgs);
1900 }
1902 SDValue Arg = N->getOperand(1);
1903 if (Arg.getOpcode() != ISD::BUILD_VECTOR)
1904 break;
1905
1906 SDValue NewArgs[19] = {
1907 N->getOperand(0),
1908 N->getOperand(1),
1909 N->getOperand(2),
1910 N->getOperand(3),
1911 N->getOperand(4),
1912 N->getOperand(5),
1913 N->getOperand(6),
1914 N->getOperand(7),
1915 N->getOperand(8),
1916 N->getOperand(9),
1917 N->getOperand(10),
1918 N->getOperand(11),
1919 N->getOperand(12),
1920 N->getOperand(13),
1921 N->getOperand(14),
1922 N->getOperand(15),
1923 N->getOperand(16),
1924 N->getOperand(17),
1925 N->getOperand(18),
1926 };
1927 NewArgs[1] = OptimizeSwizzle(N->getOperand(1), &NewArgs[2], DAG, DL);
1928 return DAG.getNode(AMDGPUISD::TEXTURE_FETCH, DL, N->getVTList(), NewArgs);
1929 }
1930
1931 case ISD::LOAD: {
1932 LoadSDNode *LoadNode = cast<LoadSDNode>(N);
1933 SDValue Ptr = LoadNode->getBasePtr();
1934 if (LoadNode->getAddressSpace() == AMDGPUAS::PARAM_I_ADDRESS &&
1936 return constBufferLoad(LoadNode, AMDGPUAS::CONSTANT_BUFFER_0, DAG);
1937 break;
1938 }
1939
1940 default: break;
1941 }
1942
1944}
1945
1946bool R600TargetLowering::FoldOperand(SDNode *ParentNode, unsigned SrcIdx,
1947 SDValue &Src, SDValue &Neg, SDValue &Abs,
1948 SDValue &Sel, SDValue &Imm,
1949 SelectionDAG &DAG) const {
1950 const R600InstrInfo *TII = Subtarget->getInstrInfo();
1951 if (!Src.isMachineOpcode())
1952 return false;
1953
1954 switch (Src.getMachineOpcode()) {
1955 case R600::FNEG_R600:
1956 if (!Neg.getNode())
1957 return false;
1958 Src = Src.getOperand(0);
1959 Neg = DAG.getTargetConstant(1, SDLoc(ParentNode), MVT::i32);
1960 return true;
1961 case R600::FABS_R600:
1962 if (!Abs.getNode())
1963 return false;
1964 Src = Src.getOperand(0);
1965 Abs = DAG.getTargetConstant(1, SDLoc(ParentNode), MVT::i32);
1966 return true;
1967 case R600::CONST_COPY: {
1968 unsigned Opcode = ParentNode->getMachineOpcode();
1969 bool HasDst = TII->getOperandIdx(Opcode, R600::OpName::dst) > -1;
1970
1971 if (!Sel.getNode())
1972 return false;
1973
1974 SDValue CstOffset = Src.getOperand(0);
1975 if (ParentNode->getValueType(0).isVector())
1976 return false;
1977
1978 // Gather constants values
1979 int SrcIndices[] = {
1980 TII->getOperandIdx(Opcode, R600::OpName::src0),
1981 TII->getOperandIdx(Opcode, R600::OpName::src1),
1982 TII->getOperandIdx(Opcode, R600::OpName::src2),
1983 TII->getOperandIdx(Opcode, R600::OpName::src0_X),
1984 TII->getOperandIdx(Opcode, R600::OpName::src0_Y),
1985 TII->getOperandIdx(Opcode, R600::OpName::src0_Z),
1986 TII->getOperandIdx(Opcode, R600::OpName::src0_W),
1987 TII->getOperandIdx(Opcode, R600::OpName::src1_X),
1988 TII->getOperandIdx(Opcode, R600::OpName::src1_Y),
1989 TII->getOperandIdx(Opcode, R600::OpName::src1_Z),
1990 TII->getOperandIdx(Opcode, R600::OpName::src1_W)
1991 };
1992 std::vector<unsigned> Consts;
1993 for (int OtherSrcIdx : SrcIndices) {
1994 int OtherSelIdx = TII->getSelIdx(Opcode, OtherSrcIdx);
1995 if (OtherSrcIdx < 0 || OtherSelIdx < 0)
1996 continue;
1997 if (HasDst) {
1998 OtherSrcIdx--;
1999 OtherSelIdx--;
2000 }
2001 if (RegisterSDNode *Reg =
2002 dyn_cast<RegisterSDNode>(ParentNode->getOperand(OtherSrcIdx))) {
2003 if (Reg->getReg() == R600::ALU_CONST) {
2004 Consts.push_back(ParentNode->getConstantOperandVal(OtherSelIdx));
2005 }
2006 }
2007 }
2008
2009 ConstantSDNode *Cst = cast<ConstantSDNode>(CstOffset);
2010 Consts.push_back(Cst->getZExtValue());
2011 if (!TII->fitsConstReadLimitations(Consts)) {
2012 return false;
2013 }
2014
2015 Sel = CstOffset;
2016 Src = DAG.getRegister(R600::ALU_CONST, MVT::f32);
2017 return true;
2018 }
2019 case R600::MOV_IMM_GLOBAL_ADDR:
2020 // Check if the Imm slot is used. Taken from below.
2021 if (Imm->getAsZExtVal())
2022 return false;
2023 Imm = Src.getOperand(0);
2024 Src = DAG.getRegister(R600::ALU_LITERAL_X, MVT::i32);
2025 return true;
2026 case R600::MOV_IMM_I32:
2027 case R600::MOV_IMM_F32: {
2028 unsigned ImmReg = R600::ALU_LITERAL_X;
2029 uint64_t ImmValue = 0;
2030
2031 if (Src.getMachineOpcode() == R600::MOV_IMM_F32) {
2032 ConstantFPSDNode *FPC = cast<ConstantFPSDNode>(Src.getOperand(0));
2033 float FloatValue = FPC->getValueAPF().convertToFloat();
2034 if (FloatValue == 0.0) {
2035 ImmReg = R600::ZERO;
2036 } else if (FloatValue == 0.5) {
2037 ImmReg = R600::HALF;
2038 } else if (FloatValue == 1.0) {
2039 ImmReg = R600::ONE;
2040 } else {
2041 ImmValue = FPC->getValueAPF().bitcastToAPInt().getZExtValue();
2042 }
2043 } else {
2044 uint64_t Value = Src.getConstantOperandVal(0);
2045 if (Value == 0) {
2046 ImmReg = R600::ZERO;
2047 } else if (Value == 1) {
2048 ImmReg = R600::ONE_INT;
2049 } else {
2050 ImmValue = Value;
2051 }
2052 }
2053
2054 // Check that we aren't already using an immediate.
2055 // XXX: It's possible for an instruction to have more than one
2056 // immediate operand, but this is not supported yet.
2057 if (ImmReg == R600::ALU_LITERAL_X) {
2058 if (!Imm.getNode())
2059 return false;
2060 ConstantSDNode *C = cast<ConstantSDNode>(Imm);
2061 if (C->getZExtValue())
2062 return false;
2063 Imm = DAG.getTargetConstant(ImmValue, SDLoc(ParentNode), MVT::i32);
2064 }
2065 Src = DAG.getRegister(ImmReg, MVT::i32);
2066 return true;
2067 }
2068 default:
2069 return false;
2070 }
2071}
2072
2073/// Fold the instructions after selecting them
2074SDNode *R600TargetLowering::PostISelFolding(MachineSDNode *Node,
2075 SelectionDAG &DAG) const {
2076 const R600InstrInfo *TII = Subtarget->getInstrInfo();
2077 if (!Node->isMachineOpcode())
2078 return Node;
2079
2080 unsigned Opcode = Node->getMachineOpcode();
2081 SDValue FakeOp;
2082
2083 std::vector<SDValue> Ops(Node->op_begin(), Node->op_end());
2084
2085 if (Opcode == R600::DOT_4) {
2086 int OperandIdx[] = {
2087 TII->getOperandIdx(Opcode, R600::OpName::src0_X),
2088 TII->getOperandIdx(Opcode, R600::OpName::src0_Y),
2089 TII->getOperandIdx(Opcode, R600::OpName::src0_Z),
2090 TII->getOperandIdx(Opcode, R600::OpName::src0_W),
2091 TII->getOperandIdx(Opcode, R600::OpName::src1_X),
2092 TII->getOperandIdx(Opcode, R600::OpName::src1_Y),
2093 TII->getOperandIdx(Opcode, R600::OpName::src1_Z),
2094 TII->getOperandIdx(Opcode, R600::OpName::src1_W)
2095 };
2096 int NegIdx[] = {
2097 TII->getOperandIdx(Opcode, R600::OpName::src0_neg_X),
2098 TII->getOperandIdx(Opcode, R600::OpName::src0_neg_Y),
2099 TII->getOperandIdx(Opcode, R600::OpName::src0_neg_Z),
2100 TII->getOperandIdx(Opcode, R600::OpName::src0_neg_W),
2101 TII->getOperandIdx(Opcode, R600::OpName::src1_neg_X),
2102 TII->getOperandIdx(Opcode, R600::OpName::src1_neg_Y),
2103 TII->getOperandIdx(Opcode, R600::OpName::src1_neg_Z),
2104 TII->getOperandIdx(Opcode, R600::OpName::src1_neg_W)
2105 };
2106 int AbsIdx[] = {
2107 TII->getOperandIdx(Opcode, R600::OpName::src0_abs_X),
2108 TII->getOperandIdx(Opcode, R600::OpName::src0_abs_Y),
2109 TII->getOperandIdx(Opcode, R600::OpName::src0_abs_Z),
2110 TII->getOperandIdx(Opcode, R600::OpName::src0_abs_W),
2111 TII->getOperandIdx(Opcode, R600::OpName::src1_abs_X),
2112 TII->getOperandIdx(Opcode, R600::OpName::src1_abs_Y),
2113 TII->getOperandIdx(Opcode, R600::OpName::src1_abs_Z),
2114 TII->getOperandIdx(Opcode, R600::OpName::src1_abs_W)
2115 };
2116 for (unsigned i = 0; i < 8; i++) {
2117 if (OperandIdx[i] < 0)
2118 return Node;
2119 SDValue &Src = Ops[OperandIdx[i] - 1];
2120 SDValue &Neg = Ops[NegIdx[i] - 1];
2121 SDValue &Abs = Ops[AbsIdx[i] - 1];
2122 bool HasDst = TII->getOperandIdx(Opcode, R600::OpName::dst) > -1;
2123 int SelIdx = TII->getSelIdx(Opcode, OperandIdx[i]);
2124 if (HasDst)
2125 SelIdx--;
2126 SDValue &Sel = (SelIdx > -1) ? Ops[SelIdx] : FakeOp;
2127 if (FoldOperand(Node, i, Src, Neg, Abs, Sel, FakeOp, DAG))
2128 return DAG.getMachineNode(Opcode, SDLoc(Node), Node->getVTList(), Ops);
2129 }
2130 } else if (Opcode == R600::REG_SEQUENCE) {
2131 for (unsigned i = 1, e = Node->getNumOperands(); i < e; i += 2) {
2132 SDValue &Src = Ops[i];
2133 if (FoldOperand(Node, i, Src, FakeOp, FakeOp, FakeOp, FakeOp, DAG))
2134 return DAG.getMachineNode(Opcode, SDLoc(Node), Node->getVTList(), Ops);
2135 }
2136 } else {
2137 if (!TII->hasInstrModifiers(Opcode))
2138 return Node;
2139 int OperandIdx[] = {
2140 TII->getOperandIdx(Opcode, R600::OpName::src0),
2141 TII->getOperandIdx(Opcode, R600::OpName::src1),
2142 TII->getOperandIdx(Opcode, R600::OpName::src2)
2143 };
2144 int NegIdx[] = {
2145 TII->getOperandIdx(Opcode, R600::OpName::src0_neg),
2146 TII->getOperandIdx(Opcode, R600::OpName::src1_neg),
2147 TII->getOperandIdx(Opcode, R600::OpName::src2_neg)
2148 };
2149 int AbsIdx[] = {
2150 TII->getOperandIdx(Opcode, R600::OpName::src0_abs),
2151 TII->getOperandIdx(Opcode, R600::OpName::src1_abs),
2152 -1
2153 };
2154 for (unsigned i = 0; i < 3; i++) {
2155 if (OperandIdx[i] < 0)
2156 return Node;
2157 SDValue &Src = Ops[OperandIdx[i] - 1];
2158 SDValue &Neg = Ops[NegIdx[i] - 1];
2159 SDValue FakeAbs;
2160 SDValue &Abs = (AbsIdx[i] > -1) ? Ops[AbsIdx[i] - 1] : FakeAbs;
2161 bool HasDst = TII->getOperandIdx(Opcode, R600::OpName::dst) > -1;
2162 int SelIdx = TII->getSelIdx(Opcode, OperandIdx[i]);
2163 int ImmIdx = TII->getOperandIdx(Opcode, R600::OpName::literal);
2164 if (HasDst) {
2165 SelIdx--;
2166 ImmIdx--;
2167 }
2168 SDValue &Sel = (SelIdx > -1) ? Ops[SelIdx] : FakeOp;
2169 SDValue &Imm = Ops[ImmIdx];
2170 if (FoldOperand(Node, i, Src, Neg, Abs, Sel, Imm, DAG))
2171 return DAG.getMachineNode(Opcode, SDLoc(Node), Node->getVTList(), Ops);
2172 }
2173 }
2174
2175 return Node;
2176}
2177
2179R600TargetLowering::shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const {
2180 switch (RMW->getOperation()) {
2191 // FIXME: Cayman at least appears to have instructions for this, but the
2192 // instruction definitions appear to be missing.
2194 case AtomicRMWInst::Xchg: {
2195 const DataLayout &DL = RMW->getFunction()->getDataLayout();
2196 unsigned ValSize = DL.getTypeSizeInBits(RMW->getType());
2197 if (ValSize == 32 || ValSize == 64)
2200 }
2201 default:
2202 if (auto *IntTy = dyn_cast<IntegerType>(RMW->getType())) {
2203 unsigned Size = IntTy->getBitWidth();
2204 if (Size == 32 || Size == 64)
2206 }
2207
2209 }
2210
2211 llvm_unreachable("covered atomicrmw op switch");
2212}
return SDValue()
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define I(x, y, z)
Definition MD5.cpp:57
static bool isUndef(const MachineInstr &MI)
Register Reg
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define MO_FLAG_NEG
Definition R600Defines.h:15
#define MO_FLAG_ABS
Definition R600Defines.h:16
#define MO_FLAG_MASK
Definition R600Defines.h:17
#define MO_FLAG_PUSH
Definition R600Defines.h:18
static bool isEOP(MachineBasicBlock::iterator I)
static SDValue ReorganizeVector(SelectionDAG &DAG, SDValue VectorEntry, DenseMap< unsigned, unsigned > &RemapSwizzle)
static int ConstantAddressBlock(unsigned AddressSpace)
static SDValue CompactSwizzlableVector(SelectionDAG &DAG, SDValue VectorEntry, DenseMap< unsigned, unsigned > &RemapSwizzle)
R600 DAG Lowering interface definition.
Provides R600 specific target descriptions.
AMDGPU R600 specific subclass of TargetSubtarget.
const SmallVectorImpl< MachineOperand > & Cond
Value * RHS
Value * LHS
unsigned getStackWidth(const MachineFunction &MF) const
SDValue combineFMinMaxLegacy(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, DAGCombinerInfo &DCI) const
Generate Min/Max node.
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue LowerSDIVREM(SDValue Op, SelectionDAG &DAG) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue CreateLiveInRegisterRaw(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT) const
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
void LowerUDIVREM64(SDValue Op, SelectionDAG &DAG, SmallVectorImpl< SDValue > &Results) const
LLVM_ABI float convertToFloat() const
Converts this APFloat to host float value.
Definition APFloat.cpp:6097
APInt bitcastToAPInt() const
Definition APFloat.h:1475
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
unsigned getSrcAddressSpace() const
unsigned getDestAddressSpace() const
an instruction that atomically reads a memory location, combines it with another value,...
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FSub
*p = old - v
@ UIncWrap
Increment one up to a maximum value.
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ Nand
*p = ~(old & v)
BinOp getOperation() const
CCState - This class holds information needed while lowering arguments and return values.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
int64_t getLocMemOffset() const
const APFloat & getValueAPF() const
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
uint64_t getZExtValue() const
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:223
bool empty() const
Definition DenseMap.h:171
iterator end()
Definition DenseMap.h:141
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Definition Function.cpp:360
LLVM_ABI unsigned getAddressSpace() const
const GlobalValue * getGlobal() const
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
bool isIndexed() const
Return true if this is a pre/post inc/dec load/store.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
ISD::LoadExtType getExtensionType() const
Return whether this is a plain node, or one of the varieties of value-extending loads.
Machine Value Type.
static auto integer_valuetypes()
LLVM_ABI DebugLoc findDebugLoc(instr_iterator MBBI)
Find the next valid DebugLoc starting at MBBI, skipping any debug instructions.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
Representation of each machine instruction.
Flags
Flags values. These may be or'd together.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MONonTemporal
The memory access is non-temporal.
@ MOInvariant
The memory access always returns the same value (or traps).
Flags getFlags() const
Return the raw flags of the source value,.
const Value * getValue() const
Return the base address of the memory access.
MachineOperand class - Representation of each machine instruction operand.
const GlobalValue * getGlobal() const
unsigned getTargetFlags() const
int64_t getOffset() const
Return the offset from the symbol in this operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_empty(Register RegNo) const
use_empty - Return true if there are no instructions using the specified register.
An SDNode that represents everything that will be needed to construct a MachineInstr.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
Align getAlign() const
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const MachinePointerInfo & getPointerInfo() const
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
Definition Type.cpp:911
StackOffset getFrameIndexReference(const MachineFunction &MF, int FI, Register &FrameReg) const override
const R600InstrInfo * getInstrInfo() const override
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
bool canMergeStoresTo(unsigned AS, EVT MemVT, const MachineFunction &MF) const override
Returns if it's reasonable to merge stores to MemVT size.
R600TargetLowering(const TargetMachine &TM, const R600Subtarget &STI)
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &, EVT VT) const override
Return the ValueType of the result of SETCC operations.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AS, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *IsFast=nullptr) const override
Determine if the target supports unaligned memory accesses.
CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg) const
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
XXX Only kernel functions are supported, so we can assume for now that every function is a kernel fun...
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
op_iterator op_end() const
op_iterator op_begin() const
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
LLVM_ABI SDValue getCondCode(ISD::CondCode Cond)
LLVMContext * getContext() const
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getValue() const
bool isTruncatingStore() const
Return true if the op does a truncation before store.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
void setHasExtractBitsInsn(bool hasExtractInsn=true)
Tells the code generator that the target has BitExtract instructions.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isCondCodeLegal(ISD::CondCode CC, MVT VT) const
Return true if the specified condition code is legal for a comparison of the specified types on this ...
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setCondCodeAction(ArrayRef< ISD::CondCode > CCs, MVT VT, LegalizeAction Action)
Indicate that the specified condition code is or isn't supported on the target and indicate what to d...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
bool expandFP_TO_SINT(SDNode *N, SDValue &Result, SelectionDAG &DAG) const
Expand float(f32) to SINT(i64) conversion.
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
virtual MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
void expandShiftParts(SDNode *N, SDValue &Lo, SDValue &Hi, SelectionDAG &DAG) const
Expand shift-by-parts.
Primary interface to the complete machine description for the target machine.
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
@ PARAM_I_ADDRESS
Address space for indirect addressable parameter memory (VTX1).
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUILD_VERTICAL_VECTOR
This node is for VLIW targets and it is used to represent a vector that is stored in consecutive regi...
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
LLVM_READNONE constexpr bool isShader(CallingConv::ID CC)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:829
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
Definition ISDOpcodes.h:261
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:524
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:863
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:220
@ GlobalAddress
Definition ISDOpcodes.h:88
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:798
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:550
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:806
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:821
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:898
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:205
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:567
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:797
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:843
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:558
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
int32_t getLDSNoRetOp(uint32_t Opcode)
constexpr float pif
Definition MathExtras.h:54
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
@ Offset
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Kill
The last use of a register.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ AfterLegalizeVectorOps
Definition DAGCombine.h:18
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
@ Custom
The result value requires a custom uniformity check.
Definition Uniformity.h:31
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool bitsLE(EVT VT) const
Return true if this has no more bits than VT.
Definition ValueTypes.h:331
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
This class contains a discriminated union of information about pointers in memory operands,...