46#include "llvm/IR/IntrinsicsAMDGPU.h"
47#include "llvm/IR/IntrinsicsR600.h"
60#define DEBUG_TYPE "si-lower"
66 cl::desc(
"Do not align and prefetch loops"),
70 "amdgpu-use-divergent-register-indexing",
cl::Hidden,
71 cl::desc(
"Use indirect register addressing for divergent indexes"),
89 unsigned NumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
90 for (
unsigned Reg = 0;
Reg < NumSGPRs; ++
Reg) {
92 return AMDGPU::SGPR0 +
Reg;
108 TRI->getDefaultVectorSuperClassForBitWidth(32);
114 TRI->getDefaultVectorSuperClassForBitWidth(64);
152 TRI->getDefaultVectorSuperClassForBitWidth(320));
156 TRI->getDefaultVectorSuperClassForBitWidth(352));
160 TRI->getDefaultVectorSuperClassForBitWidth(384));
164 TRI->getDefaultVectorSuperClassForBitWidth(512));
171 TRI->getDefaultVectorSuperClassForBitWidth(1024));
173 if (Subtarget->has16BitInsts()) {
174 if (Subtarget->useRealTrue16Insts()) {
204 TRI->getDefaultVectorSuperClassForBitWidth(1024));
220 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
221 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
222 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32,
223 MVT::i1, MVT::v32i32},
227 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
228 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
229 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32,
230 MVT::i1, MVT::v32i32},
248 if (Subtarget->hasBF16PackedInsts()) {
312 {MVT::f32, MVT::i32, MVT::i64, MVT::f64, MVT::i1},
Expand);
319 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
320 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
321 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32},
324 {MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
325 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v9f32,
326 MVT::v10f32, MVT::v11f32, MVT::v12f32, MVT::v16f32},
330 {MVT::v2i1, MVT::v4i1, MVT::v2i8, MVT::v4i8, MVT::v2i16,
331 MVT::v3i16, MVT::v4i16, MVT::Other},
336 {MVT::i1, MVT::i32, MVT::i64, MVT::f32, MVT::f64},
Expand);
352 {MVT::v8i32, MVT::v8f32, MVT::v9i32, MVT::v9f32, MVT::v10i32,
353 MVT::v10f32, MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32,
354 MVT::v16i32, MVT::v16f32, MVT::v2i64, MVT::v2f64, MVT::v4i16,
355 MVT::v4f16, MVT::v4bf16, MVT::v3i64, MVT::v3f64, MVT::v6i32,
356 MVT::v6f32, MVT::v4i64, MVT::v4f64, MVT::v8i64, MVT::v8f64,
357 MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16, MVT::v16f16,
358 MVT::v16bf16, MVT::v16i64, MVT::v16f64, MVT::v32i32, MVT::v32f32,
359 MVT::v32i16, MVT::v32f16, MVT::v32bf16}) {
394 for (
MVT Vec64 : {MVT::v2i64, MVT::v2f64}) {
408 for (
MVT Vec64 : {MVT::v3i64, MVT::v3f64}) {
422 for (
MVT Vec64 : {MVT::v4i64, MVT::v4f64}) {
436 for (
MVT Vec64 : {MVT::v8i64, MVT::v8f64}) {
450 for (
MVT Vec64 : {MVT::v16i64, MVT::v16f64}) {
465 {MVT::v4i32, MVT::v4f32, MVT::v8i32, MVT::v8f32,
466 MVT::v16i32, MVT::v16f32, MVT::v32i32, MVT::v32f32},
469 if (Subtarget->hasPkMovB32()) {
490 {MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::v2i8, MVT::v4i8,
491 MVT::v8i8, MVT::v4i16, MVT::v4f16, MVT::v4bf16},
496 {MVT::v3i32, MVT::v3f32, MVT::v4i32, MVT::v4f32},
Custom);
500 {MVT::v5i32, MVT::v5f32, MVT::v6i32, MVT::v6f32,
501 MVT::v7i32, MVT::v7f32, MVT::v8i32, MVT::v8f32,
502 MVT::v9i32, MVT::v9f32, MVT::v10i32, MVT::v10f32,
503 MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32},
527 if (Subtarget->hasSMemRealTime() ||
532 if (Subtarget->has16BitInsts()) {
542 if (Subtarget->hasMadMacF32Insts())
560 if (Subtarget->hasIntClamp())
563 if (Subtarget->hasAddNoCarryInsts())
569 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
572 {MVT::f64, MVT::f32},
Legal);
576 {MVT::f64, MVT::f32},
Custom);
581 {MVT::f64, MVT::f32},
Legal);
584 if (Subtarget->haveRoundOpsF64())
614 if (Subtarget->has16BitInsts()) {
668 if (Subtarget->hasBF16TransInsts())
684 {MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::v4i16, MVT::v4f16,
685 MVT::v4bf16, MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16,
686 MVT::v16f16, MVT::v16bf16, MVT::v32i16, MVT::v32f16}) {
721 if (Subtarget->hasVCvtPkIU16F32())
724 {MVT::v2i16, MVT::v4i16, MVT::v8i16, MVT::v16i16, MVT::v32i16},
731 {MVT::v2i16, MVT::v2f16, MVT::v2bf16},
Legal);
831 {MVT::v2f16, MVT::v2bf16, MVT::v4f16, MVT::v4bf16,
832 MVT::v8f16, MVT::v8bf16, MVT::v16f16, MVT::v16bf16,
833 MVT::v32f16, MVT::v32bf16},
835 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
842 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
Custom);
853 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
857 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
862 {MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16, MVT::v16f16,
863 MVT::v16bf16, MVT::v32i16, MVT::v32f16, MVT::v32bf16}) {
871 if (Subtarget->hasVOP3PInsts()) {
882 {MVT::v2i16, MVT::v2f16, MVT::v2bf16},
Custom);
885 {MVT::v4f16, MVT::v4i16, MVT::v4bf16, MVT::v8f16,
886 MVT::v8i16, MVT::v8bf16, MVT::v16f16, MVT::v16i16,
887 MVT::v16bf16, MVT::v32f16, MVT::v32i16, MVT::v32bf16},
890 for (
MVT VT : {MVT::v4i16, MVT::v8i16, MVT::v16i16, MVT::v32i16})
898 for (
MVT VT : {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16})
904 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
914 {MVT::v2f16, MVT::v4f16},
Custom);
920 if (Subtarget->hasBF16PackedInsts()) {
926 for (
MVT VT : {MVT::v4bf16, MVT::v8bf16, MVT::v16bf16, MVT::v32bf16})
934 if (Subtarget->hasAnyPackedFP32Ops()) {
938 {MVT::v4f32, MVT::v8f32, MVT::v16f32, MVT::v32f32},
941 if (Subtarget->hasAnyPackedFP64Ops()) {
947 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64},
Custom);
949 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
956 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64},
Custom);
965 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64},
970 if (Subtarget->hasAnyPackedU64Ops()) {
974 {MVT::v4i64, MVT::v8i64, MVT::v16i64, MVT::v32i64},
981 if (Subtarget->has16BitInsts()) {
996 {MVT::v4i16, MVT::v4f16, MVT::v4bf16, MVT::v2i8, MVT::v4i8,
997 MVT::v8i8, MVT::v8i16, MVT::v8f16, MVT::v8bf16,
998 MVT::v16i16, MVT::v16f16, MVT::v16bf16, MVT::v32i16,
999 MVT::v32f16, MVT::v32bf16},
1004 if (Subtarget->useVMulU64Inst())
1006 else if (Subtarget->hasScalarSMulU64())
1009 if (Subtarget->hasMad64_32())
1012 if (Subtarget->hasSafeSmemPrefetch() || Subtarget->hasVmemPrefInsts())
1015 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
1017 {MVT::f16, MVT::f32, MVT::f64, MVT::v2f16},
Legal);
1020 if (Subtarget->hasMinimum3Maximum3F32())
1023 if (Subtarget->hasMinimum3Maximum3PKF16()) {
1027 if (!Subtarget->hasMinimum3Maximum3F16())
1033 if (Subtarget->hasVOP3PInsts()) {
1036 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
1040 if (Subtarget->useMinMaxI64Insts())
1045 {MVT::Other, MVT::f32, MVT::v4f32, MVT::i16, MVT::f16,
1046 MVT::bf16, MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::i128,
1051 {MVT::v2f16, MVT::v2i16, MVT::v2bf16, MVT::v3f16,
1052 MVT::v3i16, MVT::v4f16, MVT::v4i16, MVT::v4bf16,
1053 MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::Other, MVT::f16,
1054 MVT::i16, MVT::bf16, MVT::i8, MVT::i128},
1069 SBufferLoadDiagnosticVTs.set(VT.SimpleTy);
1074 {MVT::Other, MVT::v2i16, MVT::v2f16, MVT::v2bf16,
1075 MVT::v3i16, MVT::v3f16, MVT::v4f16, MVT::v4i16,
1076 MVT::v4bf16, MVT::v8i16, MVT::v8f16, MVT::v8bf16,
1077 MVT::f16, MVT::i16, MVT::bf16, MVT::i8, MVT::i128},
1092 if (Subtarget->hasBF16ConversionInsts()) {
1094 {MVT::bf16, MVT::v2bf16},
Custom);
1098 if (Subtarget->hasBF16TransInsts()) {
1102 const bool HasE5M3ConversionInsts =
1103 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
1104 if (Subtarget->hasOCPFP8ConversionInsts() || HasE5M3ConversionInsts) {
1115 if (Subtarget->hasFP8F16ConversionInsts()) {
1120 if (Subtarget->hasCvtPkF16F32Inst()) {
1122 {MVT::v2f16, MVT::v4f16, MVT::v8f16, MVT::v16f16},
1174 if (Subtarget->has16BitInsts() && !Subtarget->hasMed3_16())
1215 static const MCPhysReg RCRegs[] = {AMDGPU::MODE};
1228 EVT DestVT,
EVT SrcVT)
const {
1230 ((((Opcode ==
ISD::FMAD && Subtarget->hasMadMixInsts()) ||
1231 (Opcode ==
ISD::FMA && Subtarget->hasFmaMixInsts())) &&
1233 (Opcode ==
ISD::FMA && Subtarget->hasFmaMixBF16Insts() &&
1240 LLT DestTy,
LLT SrcTy)
const {
1241 return ((Opcode == TargetOpcode::G_FMAD && Subtarget->hasMadMixInsts()) ||
1242 (Opcode == TargetOpcode::G_FMA && Subtarget->hasFmaMixInsts())) &&
1244 SrcTy.getScalarSizeInBits() == 16 &&
1265 return Subtarget->has16BitInsts()
1271 return Subtarget->has16BitInsts() ? MVT::i16 : MVT::i32;
1275 if (!Subtarget->has16BitInsts() && VT.
getSizeInBits() == 16)
1297 return (NumElts + 1) / 2;
1303 return NumElts * ((
Size + 31) / 32);
1312 unsigned &NumIntermediates,
MVT &RegisterVT)
const {
1321 MVT SimpleIntermediateVT =
1323 IntermediateVT = SimpleIntermediateVT;
1324 RegisterVT = Subtarget->has16BitInsts() ? SimpleIntermediateVT : MVT::i32;
1325 NumIntermediates = (NumElts + 1) / 2;
1326 return (NumElts + 1) / 2;
1331 IntermediateVT = RegisterVT;
1332 NumIntermediates = NumElts;
1333 return NumIntermediates;
1338 RegisterVT = MVT::i16;
1339 IntermediateVT = ScalarVT;
1340 NumIntermediates = NumElts;
1341 return NumIntermediates;
1345 RegisterVT = MVT::i32;
1346 IntermediateVT = ScalarVT;
1347 NumIntermediates = NumElts;
1348 return NumIntermediates;
1352 RegisterVT = MVT::i32;
1353 IntermediateVT = RegisterVT;
1354 NumIntermediates = NumElts * ((
Size + 31) / 32);
1355 return NumIntermediates;
1360 Context, CC, VT, IntermediateVT, NumIntermediates, RegisterVT);
1365 unsigned MaxNumLanes) {
1366 assert(MaxNumLanes != 0);
1370 unsigned NumElts = std::min(MaxNumLanes, VT->getNumElements());
1381 unsigned MaxNumLanes) {
1387 assert(ST->getNumContainedTypes() == 2 &&
1388 ST->getContainedType(1)->isIntegerTy(32));
1402 return MVT::amdgpuBufferFatPointer;
1404 DL.getPointerSizeInBits(AS) == 192)
1405 return MVT::amdgpuBufferStridedPointer;
1414 DL.getPointerSizeInBits(AS) == 160) ||
1416 DL.getPointerSizeInBits(AS) == 192))
1423 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1424 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1425 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1427 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1428 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1429 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1430 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
1431 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
1432 case Intrinsic::amdgcn_flat_load_monitor_b32:
1433 case Intrinsic::amdgcn_global_load_monitor_b32:
1435 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1436 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1437 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1438 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
1439 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
1440 case Intrinsic::amdgcn_flat_load_monitor_b64:
1441 case Intrinsic::amdgcn_global_load_monitor_b64:
1443 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1444 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128:
1445 case Intrinsic::amdgcn_global_store_async_from_lds_b128:
1446 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B:
1447 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B:
1448 case Intrinsic::amdgcn_flat_load_monitor_b128:
1449 case Intrinsic::amdgcn_global_load_monitor_b128:
1485 unsigned IntrID)
const {
1487 if (CI.
hasMetadata(LLVMContext::MD_invariant_load))
1501 bool IsSPrefetch = IntrID == Intrinsic::amdgcn_s_buffer_prefetch_data;
1515 if (RsrcIntr->IsImage) {
1530 Info.ptrVal = RsrcArg;
1534 if (RsrcIntr->IsImage) {
1535 unsigned MaxNumLanes = 4;
1550 std::numeric_limits<unsigned>::max());
1560 if (RsrcIntr->IsImage) {
1580 if ((RsrcIntr->IsImage && BaseOpcode->
NoReturn) || IsSPrefetch) {
1582 Info.memVT = MVT::i32;
1589 case Intrinsic::amdgcn_raw_buffer_load_lds:
1590 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
1591 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
1592 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
1593 case Intrinsic::amdgcn_struct_buffer_load_lds:
1594 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
1595 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
1596 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
1610 CI.
getContext(), Width * 8 * Subtarget->getWavefrontSize());
1619 case Intrinsic::amdgcn_raw_atomic_buffer_load:
1620 case Intrinsic::amdgcn_raw_ptr_atomic_buffer_load:
1621 case Intrinsic::amdgcn_struct_atomic_buffer_load:
1622 case Intrinsic::amdgcn_struct_ptr_atomic_buffer_load: {
1625 std::numeric_limits<unsigned>::max());
1638 case Intrinsic::amdgcn_ds_ordered_add:
1639 case Intrinsic::amdgcn_ds_ordered_swap: {
1653 case Intrinsic::amdgcn_ds_add_gs_reg_rtn:
1654 case Intrinsic::amdgcn_ds_sub_gs_reg_rtn: {
1657 Info.ptrVal =
nullptr;
1663 case Intrinsic::amdgcn_ds_append:
1664 case Intrinsic::amdgcn_ds_consume: {
1678 case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
1679 case Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64: {
1680 Info.opc = (IntrID == Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64)
1685 Info.memVT = MVT::i64;
1693 case Intrinsic::amdgcn_image_bvh_dual_intersect_ray:
1694 case Intrinsic::amdgcn_image_bvh_intersect_ray:
1695 case Intrinsic::amdgcn_image_bvh8_intersect_ray: {
1698 MVT::getVT(IntrID == Intrinsic::amdgcn_image_bvh_intersect_ray
1701 ->getElementType(0));
1710 case Intrinsic::amdgcn_global_atomic_fmin_num:
1711 case Intrinsic::amdgcn_global_atomic_fmax_num:
1712 case Intrinsic::amdgcn_global_atomic_ordered_add_b64:
1713 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1714 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
1725 case Intrinsic::amdgcn_cluster_load_b32:
1726 case Intrinsic::amdgcn_cluster_load_b64:
1727 case Intrinsic::amdgcn_cluster_load_b128:
1728 case Intrinsic::amdgcn_ds_load_tr6_b96:
1729 case Intrinsic::amdgcn_ds_load_tr4_b64:
1730 case Intrinsic::amdgcn_ds_load_tr8_b64:
1731 case Intrinsic::amdgcn_ds_load_tr16_b128:
1732 case Intrinsic::amdgcn_global_load_tr6_b96:
1733 case Intrinsic::amdgcn_global_load_tr4_b64:
1734 case Intrinsic::amdgcn_global_load_tr_b64:
1735 case Intrinsic::amdgcn_global_load_tr_b128:
1736 case Intrinsic::amdgcn_ds_read_tr4_b64:
1737 case Intrinsic::amdgcn_ds_read_tr6_b96:
1738 case Intrinsic::amdgcn_ds_read_tr8_b64:
1739 case Intrinsic::amdgcn_ds_read_tr16_b64: {
1748 case Intrinsic::amdgcn_flat_load_monitor_b32:
1749 case Intrinsic::amdgcn_flat_load_monitor_b64:
1750 case Intrinsic::amdgcn_flat_load_monitor_b128:
1751 case Intrinsic::amdgcn_global_load_monitor_b32:
1752 case Intrinsic::amdgcn_global_load_monitor_b64:
1753 case Intrinsic::amdgcn_global_load_monitor_b128: {
1764 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
1765 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
1766 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B: {
1777 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
1778 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
1779 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B: {
1790 case Intrinsic::amdgcn_ds_gws_init:
1791 case Intrinsic::amdgcn_ds_gws_barrier:
1792 case Intrinsic::amdgcn_ds_gws_sema_v:
1793 case Intrinsic::amdgcn_ds_gws_sema_br:
1794 case Intrinsic::amdgcn_ds_gws_sema_p:
1795 case Intrinsic::amdgcn_ds_gws_sema_release_all: {
1805 Info.memVT = MVT::i32;
1807 Info.align =
Align(4);
1809 if (IntrID == Intrinsic::amdgcn_ds_gws_barrier)
1816 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1817 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1818 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1819 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1820 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1821 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1822 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1823 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
1838 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1839 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1840 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1841 case Intrinsic::amdgcn_global_store_async_from_lds_b128: {
1856 case Intrinsic::amdgcn_av_load_b128:
1857 case Intrinsic::amdgcn_av_store_b128: {
1858 bool IsStore = IntrID == Intrinsic::amdgcn_av_store_b128;
1860 Info.memVT = MVT::v4i32;
1862 Info.align =
Align(16);
1870 unsigned ScopeIdx = CI.
arg_size() - 1;
1874 Info.ssid = Ctx.getOrInsertSyncScopeID(Scope);
1878 case Intrinsic::amdgcn_load_to_lds:
1879 case Intrinsic::amdgcn_load_async_to_lds:
1880 case Intrinsic::amdgcn_global_load_lds:
1881 case Intrinsic::amdgcn_global_load_async_lds: {
1900 Width * 8 * Subtarget->getWavefrontSize());
1906 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
1907 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
1908 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
1909 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn: {
1919 Info.memVT = MVT::i32;
1921 Info.align =
Align(4);
1927 case Intrinsic::amdgcn_s_prefetch_data:
1928 case Intrinsic::amdgcn_s_prefetch_inst:
1929 case Intrinsic::amdgcn_flat_prefetch:
1930 case Intrinsic::amdgcn_global_prefetch: {
1945 Type *&AccessTy)
const {
1946 Value *Ptr =
nullptr;
1947 switch (
II->getIntrinsicID()) {
1948 case Intrinsic::amdgcn_cluster_load_b128:
1949 case Intrinsic::amdgcn_cluster_load_b64:
1950 case Intrinsic::amdgcn_cluster_load_b32:
1951 case Intrinsic::amdgcn_ds_append:
1952 case Intrinsic::amdgcn_ds_consume:
1953 case Intrinsic::amdgcn_ds_load_tr8_b64:
1954 case Intrinsic::amdgcn_ds_load_tr16_b128:
1955 case Intrinsic::amdgcn_ds_load_tr4_b64:
1956 case Intrinsic::amdgcn_ds_load_tr6_b96:
1957 case Intrinsic::amdgcn_ds_read_tr4_b64:
1958 case Intrinsic::amdgcn_ds_read_tr6_b96:
1959 case Intrinsic::amdgcn_ds_read_tr8_b64:
1960 case Intrinsic::amdgcn_ds_read_tr16_b64:
1961 case Intrinsic::amdgcn_ds_ordered_add:
1962 case Intrinsic::amdgcn_ds_ordered_swap:
1963 case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
1964 case Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64:
1965 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1966 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1967 case Intrinsic::amdgcn_global_atomic_fmax_num:
1968 case Intrinsic::amdgcn_global_atomic_fmin_num:
1969 case Intrinsic::amdgcn_global_atomic_ordered_add_b64:
1970 case Intrinsic::amdgcn_global_load_tr_b64:
1971 case Intrinsic::amdgcn_global_load_tr_b128:
1972 case Intrinsic::amdgcn_global_load_tr4_b64:
1973 case Intrinsic::amdgcn_global_load_tr6_b96:
1974 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1975 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1976 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1977 case Intrinsic::amdgcn_global_store_async_from_lds_b128:
1978 case Intrinsic::amdgcn_av_load_b128:
1979 case Intrinsic::amdgcn_av_store_b128:
1980 Ptr =
II->getArgOperand(0);
1982 case Intrinsic::amdgcn_load_to_lds:
1983 case Intrinsic::amdgcn_load_async_to_lds:
1984 case Intrinsic::amdgcn_global_load_lds:
1985 case Intrinsic::amdgcn_global_load_async_lds:
1986 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1987 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1988 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1989 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1990 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1991 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1992 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1993 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128:
1994 Ptr =
II->getArgOperand(1);
1999 AccessTy =
II->getType();
2005 unsigned AddrSpace)
const {
2006 if (!Subtarget->hasFlatInstOffsets()) {
2013 FlatAddrSpace FlatVariant =
2016 : FlatAddrSpace::FLAT;
2018 return AM.
Scale == 0 &&
2019 (AM.
BaseOffs == 0 || Subtarget->getInstrInfo()->isLegalFLATOffset(
2020 AM.
BaseOffs, AddrSpace, FlatVariant));
2024 if (Subtarget->hasFlatGlobalInsts())
2027 if (!Subtarget->hasAddr64() || Subtarget->useFlatForGlobal()) {
2040 return isLegalMUBUFAddressingMode(AM);
2043bool SITargetLowering::isLegalMUBUFAddressingMode(
const AddrMode &AM)
const {
2054 if (!
TII->isLegalMUBUFImmOffset(AM.BaseOffs))
2066 if (AM.HasBaseReg) {
2098 return isLegalMUBUFAddressingMode(AM);
2100 if (!Subtarget->hasScalarSubwordLoads()) {
2105 if (Ty->isSized() &&
DL.getTypeStoreSize(Ty) < 4)
2153 return Subtarget->hasFlatScratchEnabled()
2155 : isLegalMUBUFAddressingMode(AM);
2202 unsigned Size,
unsigned AddrSpace,
Align Alignment,
2211 if (!Subtarget->hasUnalignedDSAccessEnabled() && Alignment <
Align(4))
2214 Align RequiredAlignment(
2216 if (Subtarget->hasLDSMisalignedBugInWGPMode() &&
Size > 32 &&
2217 Alignment < RequiredAlignment)
2232 if (!Subtarget->hasUsableDSOffset() && Alignment <
Align(8))
2238 RequiredAlignment =
Align(4);
2240 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2256 *IsFast = (Alignment >= RequiredAlignment) ? 64
2257 : (Alignment <
Align(4)) ? 32
2264 if (!Subtarget->hasDS96AndDS128())
2270 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2279 *IsFast = (Alignment >= RequiredAlignment) ? 96
2280 : (Alignment <
Align(4)) ? 32
2287 if (!Subtarget->hasDS96AndDS128() || !Subtarget->useDS128())
2293 RequiredAlignment =
Align(8);
2295 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2304 *IsFast = (Alignment >= RequiredAlignment) ? 128
2305 : (Alignment <
Align(4)) ? 32
2322 *IsFast = (Alignment >= RequiredAlignment) ?
Size : 0;
2324 return Alignment >= RequiredAlignment ||
2325 Subtarget->hasUnalignedDSAccessEnabled();
2333 bool AlignedBy4 = Alignment >=
Align(4);
2334 if (Subtarget->hasUnalignedScratchAccessEnabled()) {
2336 *IsFast = AlignedBy4 ?
Size : 1;
2341 *IsFast = AlignedBy4;
2352 return Alignment >=
Align(4) ||
2353 Subtarget->hasUnalignedBufferAccessEnabled();
2366 if (!Subtarget->hasRelaxedBufferOOBMode() &&
2381 return Size >= 32 && Alignment >=
Align(4);
2386 unsigned *IsFast)
const {
2388 Alignment, Flags, IsFast);
2399 if (
Op.size() >= 16 &&
2403 if (
Op.size() >= 8 &&
Op.isDstAligned(
Align(4)))
2421 unsigned DestAS)
const {
2424 Subtarget->hasGloballyAddressableScratch()) {
2455 unsigned Index)
const {
2469 unsigned MinAlign = Subtarget->useRealTrue16Insts() ? 16 : 32;
2474 if (Subtarget->has16BitInsts() && VT == MVT::i16) {
2497 if (Subtarget->hasScalarSubwordLoads() &&
N->getOpcode() ==
ISD::LOAD &&
2548 auto [InputPtrReg, RC, ArgTy] =
2564 const SDLoc &SL)
const {
2571 const SDLoc &SL)
const {
2574 std::optional<uint32_t> KnownSize =
2576 if (KnownSize.has_value())
2603 Val = getFPExtOrFPRound(DAG, Val, SL, VT);
2618SDValue SITargetLowering::lowerKernargMemParameter(
2623 MachinePointerInfo PtrInfo =
2632 int64_t OffsetDiff =
Offset - AlignDownOffset;
2638 SDValue Ptr = lowerKernArgParameterPtr(DAG, SL, Chain, AlignDownOffset);
2639 SDValue
Load = DAG.
getLoad(MVT::i32, SL, Chain, Ptr,
2644 SDValue ShiftAmt = DAG.
getConstant(OffsetDiff * 8, SL, MVT::i32);
2649 ArgVal = convertArgType(DAG, VT, MemVT, SL, ArgVal,
Signed, Arg);
2654 SDValue Ptr = lowerKernArgParameterPtr(DAG, SL, Chain,
Offset);
2659 SDValue Val = convertArgType(DAG, VT, MemVT, SL,
Load,
Signed, Arg);
2668 const SDLoc &SL)
const {
2737 ExtType, SL, VA.
getLocVT(), Chain, FIN,
2740 SDValue ConvertedVal = convertABITypeToValueType(DAG, ArgValue, VA, SL);
2741 if (ConvertedVal == ArgValue)
2742 return ConvertedVal;
2747SDValue SITargetLowering::lowerWorkGroupId(
2752 if (!Subtarget->hasClusters())
2753 return getPreloadedValue(DAG, MFI, VT, WorkGroupIdPV);
2761 SDValue ClusterIdXYZ = getPreloadedValue(DAG, MFI, VT, WorkGroupIdPV);
2762 SDLoc SL(ClusterIdXYZ);
2763 SDValue ClusterMaxIdXYZ = getPreloadedValue(DAG, MFI, VT, ClusterMaxIdPV);
2765 SDValue ClusterSizeXYZ = DAG.
getNode(
ISD::ADD, SL, VT, ClusterMaxIdXYZ, One);
2766 SDValue ClusterWorkGroupIdXYZ =
2767 getPreloadedValue(DAG, MFI, VT, ClusterWorkGroupIdPV);
2768 SDValue GlobalIdXYZ =
2777 return ClusterIdXYZ;
2779 using namespace AMDGPU::Hwreg;
2780 SDValue ClusterIdField =
2783 DAG.
getMachineNode(AMDGPU::S_GETREG_B32_const, SL, VT, ClusterIdField);
2784 SDValue ClusterId(GetReg, 0);
2794SDValue SITargetLowering::getPreloadedValue(
2797 const ArgDescriptor *
Reg =
nullptr;
2802 const ArgDescriptor WorkGroupIDX =
2810 const ArgDescriptor WorkGroupIDZ =
2812 const ArgDescriptor ClusterWorkGroupIDX =
2814 const ArgDescriptor ClusterWorkGroupIDY =
2816 const ArgDescriptor ClusterWorkGroupIDZ =
2818 const ArgDescriptor ClusterWorkGroupMaxIDX =
2820 const ArgDescriptor ClusterWorkGroupMaxIDY =
2822 const ArgDescriptor ClusterWorkGroupMaxIDZ =
2824 const ArgDescriptor ClusterWorkGroupMaxFlatID =
2827 auto LoadConstant = [&](
unsigned N) {
2831 if (Subtarget->hasArchitectedSGPRs() &&
2838 Reg = &WorkGroupIDX;
2839 RC = &AMDGPU::SReg_32RegClass;
2843 Reg = &WorkGroupIDY;
2844 RC = &AMDGPU::SReg_32RegClass;
2848 Reg = &WorkGroupIDZ;
2849 RC = &AMDGPU::SReg_32RegClass;
2853 if (HasFixedDims && ClusterDims.
getDims()[0] == 1)
2854 return LoadConstant(0);
2855 Reg = &ClusterWorkGroupIDX;
2856 RC = &AMDGPU::SReg_32RegClass;
2860 if (HasFixedDims && ClusterDims.
getDims()[1] == 1)
2861 return LoadConstant(0);
2862 Reg = &ClusterWorkGroupIDY;
2863 RC = &AMDGPU::SReg_32RegClass;
2867 if (HasFixedDims && ClusterDims.
getDims()[2] == 1)
2868 return LoadConstant(0);
2869 Reg = &ClusterWorkGroupIDZ;
2870 RC = &AMDGPU::SReg_32RegClass;
2875 return LoadConstant(ClusterDims.
getDims()[0] - 1);
2876 Reg = &ClusterWorkGroupMaxIDX;
2877 RC = &AMDGPU::SReg_32RegClass;
2882 return LoadConstant(ClusterDims.
getDims()[1] - 1);
2883 Reg = &ClusterWorkGroupMaxIDY;
2884 RC = &AMDGPU::SReg_32RegClass;
2889 return LoadConstant(ClusterDims.
getDims()[2] - 1);
2890 Reg = &ClusterWorkGroupMaxIDZ;
2891 RC = &AMDGPU::SReg_32RegClass;
2895 Reg = &ClusterWorkGroupMaxFlatID;
2896 RC = &AMDGPU::SReg_32RegClass;
2927 for (
unsigned I = 0,
E = Ins.
size(), PSInputNum = 0;
I !=
E; ++
I) {
2931 "vector type argument should have been split");
2936 bool SkipArg = !Arg->
Used && !Info->isPSInputAllocated(PSInputNum);
2944 "unexpected vector split in ps argument type");
2958 Info->markPSInputAllocated(PSInputNum);
2960 Info->markPSInputEnabled(PSInputNum);
2976 if (Info.hasWorkItemIDX()) {
2982 (Subtarget->hasPackedTID() && Info.hasWorkItemIDY()) ? 0x3ff : ~0u;
2986 if (Info.hasWorkItemIDY()) {
2987 assert(Info.hasWorkItemIDX());
2988 if (Subtarget->hasPackedTID()) {
2989 Info.setWorkItemIDY(
2992 unsigned Reg = AMDGPU::VGPR1;
3000 if (Info.hasWorkItemIDZ()) {
3001 assert(Info.hasWorkItemIDX() && Info.hasWorkItemIDY());
3002 if (Subtarget->hasPackedTID()) {
3003 Info.setWorkItemIDZ(
3006 unsigned Reg = AMDGPU::VGPR2;
3017 unsigned NumArgRegs) {
3020 if (RegIdx == ArgSGPRs.
size())
3023 unsigned Reg = ArgSGPRs[RegIdx];
3068 const unsigned Mask = 0x3ff;
3077 auto &
ArgInfo = Info.getArgInfo();
3089 if (Info.hasImplicitArgPtr())
3097 if (Info.hasWorkGroupIDX())
3100 if (Info.hasWorkGroupIDY())
3103 if (Info.hasWorkGroupIDZ())
3106 if (Info.hasLDSKernelId())
3117 Register ImplicitBufferPtrReg = Info.addImplicitBufferPtr(
TRI);
3118 MF.
addLiveIn(ImplicitBufferPtrReg, &AMDGPU::SGPR_64RegClass);
3124 Register PrivateSegmentBufferReg = Info.addPrivateSegmentBuffer(
TRI);
3125 MF.
addLiveIn(PrivateSegmentBufferReg, &AMDGPU::SGPR_128RegClass);
3130 Register DispatchPtrReg = Info.addDispatchPtr(
TRI);
3131 MF.
addLiveIn(DispatchPtrReg, &AMDGPU::SGPR_64RegClass);
3137 MF.
addLiveIn(QueuePtrReg, &AMDGPU::SGPR_64RegClass);
3143 Register InputPtrReg = Info.addKernargSegmentPtr(
TRI);
3152 MF.
addLiveIn(DispatchIDReg, &AMDGPU::SGPR_64RegClass);
3157 Register FlatScratchInitReg = Info.addFlatScratchInit(
TRI);
3158 MF.
addLiveIn(FlatScratchInitReg, &AMDGPU::SGPR_64RegClass);
3163 Register PrivateSegmentSizeReg = Info.addPrivateSegmentSize(
TRI);
3164 MF.
addLiveIn(PrivateSegmentSizeReg, &AMDGPU::SGPR_32RegClass);
3179 unsigned LastExplicitArgOffset = Subtarget->getExplicitKernelArgOffset();
3181 bool InPreloadSequence =
true;
3183 bool AlignedForImplictArgs =
false;
3184 unsigned ImplicitArgOffset = 0;
3185 for (
auto &Arg :
F.args()) {
3186 if (!InPreloadSequence || !Arg.hasInRegAttr())
3189 unsigned ArgIdx = Arg.getArgNo();
3192 if (InIdx < Ins.
size() &&
3193 (!Ins[InIdx].isOrigArg() || Ins[InIdx].getOrigArgIndex() != ArgIdx))
3196 for (; InIdx < Ins.
size() && Ins[InIdx].isOrigArg() &&
3197 Ins[InIdx].getOrigArgIndex() == ArgIdx;
3199 assert(ArgLocs[ArgIdx].isMemLoc());
3200 auto &ArgLoc = ArgLocs[InIdx];
3202 unsigned ArgOffset = ArgLoc.getLocMemOffset();
3204 unsigned NumAllocSGPRs =
3205 alignTo(ArgLoc.getLocVT().getFixedSizeInBits(), 32) / 32;
3208 if (Arg.hasAttribute(
"amdgpu-hidden-argument")) {
3209 if (!AlignedForImplictArgs) {
3211 alignTo(LastExplicitArgOffset,
3212 Subtarget->getAlignmentForImplicitArgPtr()) -
3213 LastExplicitArgOffset;
3214 AlignedForImplictArgs =
true;
3216 ArgOffset += ImplicitArgOffset;
3220 if (ArgLoc.getLocVT().getStoreSize() < 4 && Alignment < 4) {
3221 assert(InIdx >= 1 &&
"No previous SGPR");
3222 Info.getArgInfo().PreloadKernArgs[InIdx].Regs.push_back(
3223 Info.getArgInfo().PreloadKernArgs[InIdx - 1].Regs[0]);
3227 unsigned Padding = ArgOffset - LastExplicitArgOffset;
3228 unsigned PaddingSGPRs =
alignTo(Padding, 4) / 4;
3231 InPreloadSequence =
false;
3237 TRI.getSGPRClassForBitWidth(NumAllocSGPRs * 32);
3239 Info.addPreloadedKernArg(
TRI, RC, NumAllocSGPRs, InIdx, PaddingSGPRs);
3241 if (PreloadRegs->
size() > 1)
3242 RC = &AMDGPU::SGPR_32RegClass;
3243 for (
auto &Reg : *PreloadRegs) {
3249 LastExplicitArgOffset = NumAllocSGPRs * 4 + ArgOffset;
3258 if (Info.hasLDSKernelId()) {
3259 Register Reg = Info.addLDSKernelId();
3260 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3269 bool IsShader)
const {
3270 bool HasArchitectedSGPRs = Subtarget->hasArchitectedSGPRs();
3271 if (Subtarget->hasUserSGPRInit16BugInWave32() && !IsShader) {
3277 assert(!HasArchitectedSGPRs &&
"Unhandled feature for the subtarget");
3279 unsigned CurrentUserSGPRs = Info.getNumUserSGPRs();
3283 unsigned NumRequiredSystemSGPRs =
3284 Info.hasWorkGroupIDX() + Info.hasWorkGroupIDY() +
3285 Info.hasWorkGroupIDZ() + Info.hasWorkGroupInfo();
3286 for (
unsigned i = NumRequiredSystemSGPRs + CurrentUserSGPRs; i < 16; ++i) {
3287 Register Reg = Info.addReservedUserSGPR();
3288 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3293 if (!HasArchitectedSGPRs) {
3294 if (Info.hasWorkGroupIDX()) {
3295 Register Reg = Info.addWorkGroupIDX();
3296 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3300 if (Info.hasWorkGroupIDY()) {
3301 Register Reg = Info.addWorkGroupIDY();
3302 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3306 if (Info.hasWorkGroupIDZ()) {
3307 Register Reg = Info.addWorkGroupIDZ();
3308 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3313 if (Info.hasWorkGroupInfo()) {
3314 Register Reg = Info.addWorkGroupInfo();
3315 MF.
addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3319 if (Info.hasPrivateSegmentWaveByteOffset()) {
3321 unsigned PrivateSegmentWaveByteOffsetReg;
3324 PrivateSegmentWaveByteOffsetReg =
3325 Info.getPrivateSegmentWaveByteOffsetSystemSGPR();
3329 if (PrivateSegmentWaveByteOffsetReg == AMDGPU::NoRegister) {
3331 Info.setPrivateSegmentWaveByteOffset(PrivateSegmentWaveByteOffsetReg);
3334 PrivateSegmentWaveByteOffsetReg = Info.addPrivateSegmentWaveByteOffset();
3336 MF.
addLiveIn(PrivateSegmentWaveByteOffsetReg, &AMDGPU::SGPR_32RegClass);
3337 CCInfo.
AllocateReg(PrivateSegmentWaveByteOffsetReg);
3340 assert(!Subtarget->hasUserSGPRInit16BugInWave32() || IsShader ||
3341 Info.getNumPreloadedSGPRs() >= 16);
3356 if (HasStackObjects)
3357 Info.setHasNonSpillStackObjects(
true);
3362 HasStackObjects =
true;
3366 bool RequiresStackAccess = HasStackObjects || MFI.
hasCalls();
3368 if (!ST.hasFlatScratchEnabled()) {
3369 if (RequiresStackAccess && ST.isAmdHsaOrMesa(MF.
getFunction())) {
3376 Info.setScratchRSrcReg(PrivateSegmentBufferReg);
3378 unsigned ReservedBufferReg =
TRI.reservedPrivateSegmentBufferReg(MF);
3388 Info.setScratchRSrcReg(ReservedBufferReg);
3407 if (!MRI.
isLiveIn(AMDGPU::SGPR32)) {
3408 Info.setStackPtrOffsetReg(AMDGPU::SGPR32);
3415 for (
unsigned Reg : AMDGPU::SGPR_32RegClass) {
3417 Info.setStackPtrOffsetReg(
Reg);
3422 if (Info.getStackPtrOffsetReg() == AMDGPU::SP_REG)
3429 if (ST.getFrameLowering()->hasFP(MF)) {
3430 Info.setFrameOffsetReg(AMDGPU::SGPR33);
3446 const MCPhysReg *IStart =
TRI->getCalleeSavedRegsViaCopy(Entry->getParent());
3455 if (AMDGPU::SReg_64RegClass.
contains(*
I))
3456 RC = &AMDGPU::SGPR_64RegClass;
3457 else if (AMDGPU::SReg_32RegClass.
contains(*
I))
3458 RC = &AMDGPU::SGPR_32RegClass;
3464 Entry->addLiveIn(*
I);
3469 for (
auto *Exit : Exits)
3471 TII->get(TargetOpcode::COPY), *
I)
3486 bool IsError =
false;
3490 Fn,
"unsupported non-compute shaders with HSA",
DL.getDebugLoc()));
3508 !Info->hasLDSKernelId() && !Info->hasWorkItemIDX() &&
3509 !Info->hasWorkItemIDY() && !Info->hasWorkItemIDZ());
3511 if (!Subtarget->hasFlatScratchEnabled())
3516 !Subtarget->hasArchitectedSGPRs())
3517 assert(!Info->hasWorkGroupIDX() && !Info->hasWorkGroupIDY() &&
3518 !Info->hasWorkGroupIDZ());
3521 bool IsWholeWaveFunc = Info->isWholeWaveFunction();
3539 if ((Info->getPSInputAddr() & 0x7F) == 0 ||
3540 ((Info->getPSInputAddr() & 0xF) == 0 && Info->isPSInputAllocated(11))) {
3543 Info->markPSInputAllocated(0);
3544 Info->markPSInputEnabled(0);
3546 if (Subtarget->isAmdPalOS()) {
3555 unsigned PsInputBits = Info->getPSInputAddr() & Info->getPSInputEnable();
3556 if ((PsInputBits & 0x7F) == 0 ||
3557 ((PsInputBits & 0xF) == 0 && (PsInputBits >> 11 & 1)))
3560 }
else if (IsKernel) {
3561 assert(Info->hasWorkGroupIDX() && Info->hasWorkItemIDX());
3573 if (IsKernel && Subtarget->hasKernargPreload())
3577 }
else if (!IsGraphics) {
3582 if (!Subtarget->hasFlatScratchEnabled())
3594 Info->setNumWaveDispatchSGPRs(
3596 Info->setNumWaveDispatchVGPRs(
3598 }
else if (Info->getNumKernargPreloadedSGPRs()) {
3599 Info->setNumWaveDispatchSGPRs(Info->getNumUserSGPRs());
3604 if (IsWholeWaveFunc) {
3606 {MVT::i1, MVT::Other}, Chain);
3618 for (
unsigned i = IsWholeWaveFunc ? 1 : 0, e = Ins.
size(), ArgIdx = 0; i != e;
3629 if (IsEntryFunc && VA.
isMemLoc()) {
3653 if (Arg.
isOrigArg() && Info->getArgInfo().PreloadKernArgs.count(i)) {
3657 int64_t OffsetDiff =
Offset - AlignDownOffset;
3664 Info->getArgInfo().PreloadKernArgs.find(i)->getSecond().Regs[0];
3667 Register VReg = MRI.getLiveInVirtReg(Reg);
3675 NewArg = convertArgType(DAG, VT, MemVT,
DL, ArgVal,
3676 Ins[i].Flags.isSExt(), &Ins[i]);
3684 Info->getArgInfo().PreloadKernArgs.find(i)->getSecond().Regs;
3687 if (PreloadRegs.
size() == 1) {
3688 Register VReg = MRI.getLiveInVirtReg(PreloadRegs[0]);
3693 TRI->getRegSizeInBits(*RC)));
3701 for (
auto Reg : PreloadRegs) {
3702 Register VReg = MRI.getLiveInVirtReg(Reg);
3708 PreloadRegs.size()),
3725 NewArg = convertArgType(DAG, VT, MemVT,
DL, NewArg,
3726 Ins[i].Flags.isSExt(), &Ins[i]);
3738 "hidden argument in kernel signature was not preloaded",
3744 lowerKernargMemParameter(DAG, VT, MemVT,
DL, Chain,
Offset,
3745 Alignment, Ins[i].Flags.isSExt(), &Ins[i]);
3765 if (!IsEntryFunc && VA.
isMemLoc()) {
3766 SDValue Val = lowerStackParameter(DAG, VA,
DL, Chain, Arg);
3777 if (AMDGPU::VGPR_32RegClass.
contains(Reg))
3778 RC = &AMDGPU::VGPR_32RegClass;
3779 else if (AMDGPU::SGPR_32RegClass.
contains(Reg))
3780 RC = &AMDGPU::SGPR_32RegClass;
3786 if (Arg.
Flags.
isInReg() && RC == &AMDGPU::VGPR_32RegClass) {
3792 ReadFirstLane, Val);
3801 Val = convertABITypeToValueType(DAG, Val, VA,
DL);
3810 Info->setBytesInStackArgArea(StackArgSize);
3812 return Chains.
empty() ? Chain
3821 const Type *RetTy)
const {
3829 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
3834 unsigned MaxNumVGPRs = Subtarget->getMaxNumVGPRs(MF);
3835 unsigned TotalNumVGPRs = Subtarget->getAddressableNumArchVGPRs();
3836 for (
unsigned i = MaxNumVGPRs; i < TotalNumVGPRs; ++i)
3837 if (CCInfo.
isAllocated(AMDGPU::VGPR_32RegClass.getRegister(i)))
3860 Info->setIfReturnsVoid(Outs.
empty());
3861 bool IsWaveEnd = Info->returnsVoid() && IsShader;
3880 for (
unsigned I = 0, RealRVLocIdx = 0, E = RVLocs.
size();
I != E;
3881 ++
I, ++RealRVLocIdx) {
3885 SDValue Arg = OutVals[RealRVLocIdx];
3908 ReadFirstLane, Arg);
3915 if (!Info->isEntryFunction()) {
3921 if (AMDGPU::SReg_64RegClass.
contains(*
I))
3923 else if (AMDGPU::SReg_32RegClass.
contains(*
I))
3936 unsigned Opc = AMDGPUISD::ENDPGM;
3938 Opc = Info->isWholeWaveFunction() ? AMDGPUISD::WHOLE_WAVE_RETURN
3939 : IsShader ? AMDGPUISD::RETURN_TO_EPILOG
3940 : AMDGPUISD::RET_GLUE;
4045 const auto [OutgoingArg, ArgRC, ArgTy] =
4050 const auto [IncomingArg, IncomingArgRC, Ty] =
4052 assert(IncomingArgRC == ArgRC);
4055 EVT ArgVT =
TRI->getSpillSize(*ArgRC) == 8 ? MVT::i64 : MVT::i32;
4063 InputReg = getImplicitArgPtr(DAG,
DL);
4065 std::optional<uint32_t> Id =
4067 if (Id.has_value()) {
4078 if (OutgoingArg->isRegister()) {
4079 RegsToPass.emplace_back(OutgoingArg->getRegister(), InputReg);
4080 if (!CCInfo.
AllocateReg(OutgoingArg->getRegister()))
4083 unsigned SpecialArgOffset =
4094 auto [OutgoingArg, ArgRC, Ty] =
4097 std::tie(OutgoingArg, ArgRC, Ty) =
4100 std::tie(OutgoingArg, ArgRC, Ty) =
4115 const bool NeedWorkItemIDX = !CLI.
CB->
hasFnAttr(
"amdgpu-no-workitem-id-x");
4116 const bool NeedWorkItemIDY = !CLI.
CB->
hasFnAttr(
"amdgpu-no-workitem-id-y");
4117 const bool NeedWorkItemIDZ = !CLI.
CB->
hasFnAttr(
"amdgpu-no-workitem-id-z");
4122 if (Subtarget->getMaxWorkitemID(
F, 0) != 0) {
4130 NeedWorkItemIDY && Subtarget->getMaxWorkitemID(
F, 1) != 0) {
4140 NeedWorkItemIDZ && Subtarget->getMaxWorkitemID(
F, 2) != 0) {
4149 if (!InputReg && (NeedWorkItemIDX || NeedWorkItemIDY || NeedWorkItemIDZ)) {
4150 if (!IncomingArgX && !IncomingArgY && !IncomingArgZ) {
4161 : IncomingArgY ? *IncomingArgY
4168 if (OutgoingArg->isRegister()) {
4170 RegsToPass.emplace_back(OutgoingArg->getRegister(), InputReg);
4196 if (Callee->isDivergent())
4203 const uint32_t *CallerPreserved =
TRI->getCallPreservedMask(MF, CallerCC);
4207 if (!CallerPreserved)
4210 bool CCMatch = CallerCC == CalleeCC;
4223 if (Arg.hasByValAttr())
4237 const uint32_t *CalleePreserved =
TRI->getCallPreservedMask(MF, CalleeCC);
4238 if (!
TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved))
4247 CCState CCInfo(CalleeCC, IsVarArg, MF, ArgLocs, Ctx);
4260 for (
const auto &[CCVA, ArgVal] :
zip_equal(ArgLocs, OutVals)) {
4262 if (!CCVA.isRegLoc())
4267 if (ArgVal->
isDivergent() &&
TRI->isSGPRPhysReg(CCVA.getLocReg())) {
4269 dbgs() <<
"Cannot tail call due to divergent outgoing argument in "
4293enum ChainCallArgIdx {
4315 bool UsesDynamicVGPRs =
false;
4316 if (IsChainCallConv) {
4321 auto RequestedExecIt =
4323 return Arg.OrigArgIndex == 2;
4325 assert(RequestedExecIt != CLI.
Outs.end() &&
"No node for EXEC");
4327 size_t SpecialArgsBeginIdx = RequestedExecIt - CLI.
Outs.begin();
4330 CLI.
Outs.erase(RequestedExecIt, CLI.
Outs.end());
4333 "Haven't popped all the special args");
4336 CLI.
Args[ChainCallArgIdx::Exec];
4337 if (!RequestedExecArg.
Ty->
isIntegerTy(Subtarget->getWavefrontSize()))
4345 ArgNode->getAPIntValue(),
DL, ArgNode->getValueType(0)));
4347 ChainCallSpecialArgs.
push_back(Arg.Node);
4350 PushNodeOrTargetConstant(RequestedExecArg);
4356 if (FlagsValue.
isZero()) {
4357 if (CLI.
Args.size() > ChainCallArgIdx::Flags + 1)
4359 "no additional args allowed if flags == 0");
4361 if (CLI.
Args.size() != ChainCallArgIdx::FallbackCallee + 1) {
4365 if (!Subtarget->isWave32()) {
4367 CLI, InVals,
"dynamic VGPR mode is only supported for wave32");
4370 UsesDynamicVGPRs =
true;
4371 std::for_each(CLI.
Args.begin() + ChainCallArgIdx::NumVGPRs,
4372 CLI.
Args.end(), PushNodeOrTargetConstant);
4381 bool IsSibCall =
false;
4395 "unsupported call to variadic function ");
4403 "unsupported required tail call to function ");
4408 Outs, OutVals, Ins, DAG);
4412 "site marked musttail or on llvm.amdgcn.cs.chain");
4419 if (!TailCallOpt && IsTailCall)
4443 if (!Subtarget->hasFlatScratchEnabled())
4464 auto *
TRI = Subtarget->getRegisterInfo();
4471 if (!IsSibCall || IsChainCallConv) {
4472 if (!Subtarget->hasFlatScratchEnabled()) {
4478 RegsToPass.emplace_back(IsChainCallConv
4479 ? AMDGPU::SGPR48_SGPR49_SGPR50_SGPR51
4480 : AMDGPU::SGPR0_SGPR1_SGPR2_SGPR3,
4487 const unsigned NumSpecialInputs = RegsToPass.size();
4489 MVT PtrVT = MVT::i32;
4492 for (
unsigned i = 0, e = ArgLocs.
size(); i != e; ++i) {
4520 RegsToPass.push_back(std::pair(VA.
getLocReg(), Arg));
4528 int32_t
Offset = LocMemOffset;
4535 unsigned OpSize = Flags.isByVal() ? Flags.getByValSize()
4541 ? Flags.getNonZeroByValAlign()
4568 if (Outs[i].Flags.isByVal()) {
4570 DAG.
getConstant(Outs[i].Flags.getByValSize(),
DL, MVT::i32);
4573 Outs[i].Flags.getNonZeroByValAlign(),
4574 Outs[i].Flags.getNonZeroByValAlign(),
4576 nullptr, std::nullopt, DstInfo,
4582 DAG.
getStore(Chain,
DL, Arg, DstAddr, DstInfo, Alignment);
4588 if (!MemOpChains.
empty())
4604 unsigned ArgIdx = 0;
4605 for (
auto [Reg, Val] : RegsToPass) {
4606 if (ArgIdx++ >= NumSpecialInputs &&
4607 (IsChainCallConv || !Val->
isDivergent()) &&
TRI->isSGPRPhysReg(Reg)) {
4633 if (IsTailCall && !IsSibCall) {
4638 std::vector<SDValue>
Ops({Chain});
4644 Ops.push_back(Callee);
4661 Ops.push_back(Callee);
4672 if (IsChainCallConv)
4677 for (
auto &[Reg, Val] : RegsToPass)
4681 const uint32_t *Mask =
TRI->getCallPreservedMask(MF, CallConv);
4682 assert(Mask &&
"Missing call preserved mask for calling convention");
4692 MVT::Glue, GlueOps),
4697 Ops.push_back(InGlue);
4703 unsigned OPC = AMDGPUISD::TC_RETURN;
4706 OPC = AMDGPUISD::TC_RETURN_GFX;
4710 OPC = UsesDynamicVGPRs ? AMDGPUISD::TC_RETURN_CHAIN_DVGPR
4711 : AMDGPUISD::TC_RETURN_CHAIN;
4717 if (Info->isWholeWaveFunction())
4718 OPC = AMDGPUISD::TC_RETURN_GFX_WholeWave;
4728 Chain =
Call.getValue(0);
4729 InGlue =
Call.getValue(1);
4731 uint64_t CalleePopBytes = NumBytes;
4752 EVT VT =
Op.getValueType();
4766 "Stack grows upwards for AMDGPU");
4768 Chain = BaseAddr.getValue(1);
4770 const bool HasFlatScratch = Subtarget->hasFlatScratchEnabled();
4771 const unsigned WavefrontSizeLog2 = Subtarget->getWavefrontSizeLog2();
4774 if (Alignment > StackAlign) {
4775 uint64_t ScaledAlignment = Alignment.value()
4776 << (HasFlatScratch ? 0 : WavefrontSizeLog2);
4777 uint64_t StackAlignMask = ScaledAlignment - 1;
4784 assert(
Size.getValueType() == MVT::i32 &&
"Size must be 32-bit");
4793 DAG.
getConstant(WavefrontSizeLog2, dl, MVT::i32));
4804 if (!HasFlatScratch) {
4807 DAG.
getConstant(WavefrontSizeLog2, dl, MVT::i32));
4824 if (
Op.getValueType() != MVT::i32)
4843 assert(
Op.getValueType() == MVT::i32);
4852 Op.getOperand(0), IntrinID, GetRoundBothImm);
4886 SDValue RoundModeTimesNumBits =
4906 TableEntry, EnumOffset);
4922 static_cast<uint32_t>(ConstMode->getZExtValue()),
4934 if (UseReducedTable) {
4940 SDValue RoundModeTimesNumBits =
4960 SDValue RoundModeTimesNumBits =
4969 NewMode = TruncTable;
4978 ReadFirstLaneID, NewMode);
4991 IntrinID, RoundBothImm, NewMode);
4997 if (
Op->isDivergent() &&
4998 (!Subtarget->hasVmemPrefInsts() || !
Op.getConstantOperandVal(4)))
5008 if (Subtarget->hasSafeSmemPrefetch())
5016 if (!Subtarget->hasSafeSmemPrefetch() && !
Op.getConstantOperandVal(4))
5025 SDValue Src =
Op.getOperand(IsStrict ? 1 : 0);
5026 EVT SrcVT = Src.getValueType();
5035 EVT DstVT =
Op.getValueType();
5044 if (
Op.getValueType() != MVT::i64)
5058 Op.getOperand(0), IntrinID, ModeHwRegImm);
5060 Op.getOperand(0), IntrinID, TrapHwRegImm);
5074 if (
Op.getOperand(1).getValueType() != MVT::i64)
5086 ReadFirstLaneID, NewModeReg);
5088 ReadFirstLaneID, NewTrapReg);
5090 unsigned ModeHwReg =
5093 unsigned TrapHwReg =
5101 IntrinID, ModeHwRegImm, NewModeReg);
5104 IntrinID, TrapHwRegImm, NewTrapReg);
5112 .
Case(
"m0", AMDGPU::M0)
5113 .
Case(
"exec", AMDGPU::EXEC)
5114 .
Case(
"exec_lo", AMDGPU::EXEC_LO)
5115 .
Case(
"exec_hi", AMDGPU::EXEC_HI)
5116 .
Case(
"flat_scratch", AMDGPU::FLAT_SCR)
5117 .
Case(
"flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
5118 .
Case(
"flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
5119 .
Case(
"src_flat_scratch_base", AMDGPU::SRC_FLAT_SCRATCH_BASE)
5120 .
Case(
"src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
5121 .
Case(
"src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
5127 if (!Subtarget->hasFlatScrRegister() &&
5128 TRI->regsOverlap(Reg, AMDGPU::FLAT_SCR))
5131 if (!Subtarget->hasGloballyAddressableScratch() &&
5132 TRI->regsOverlap(Reg, AMDGPU::SRC_FLAT_SCRATCH_BASE))
5137 case AMDGPU::EXEC_LO:
5138 case AMDGPU::EXEC_HI:
5139 case AMDGPU::FLAT_SCR_LO:
5140 case AMDGPU::FLAT_SCR_HI:
5141 case AMDGPU::SRC_FLAT_SCRATCH_BASE_LO:
5142 case AMDGPU::SRC_FLAT_SCRATCH_BASE_HI:
5147 case AMDGPU::FLAT_SCR:
5148 case AMDGPU::SRC_FLAT_SCRATCH_BASE:
5167 MI.setDesc(
TII->getKillTerminatorFromPseudo(
MI.getOpcode()));
5176static std::pair<MachineBasicBlock *, MachineBasicBlock *>
5198 auto Next = std::next(
I);
5209 MBB.addSuccessor(LoopBB);
5211 return std::pair(LoopBB, RemainderBB);
5218 auto I =
MI.getIterator();
5219 auto E = std::next(
I);
5241 Src->setIsKill(
false);
5251 BuildMI(*LoopBB, LoopBB->begin(),
DL,
TII->get(AMDGPU::S_SETREG_IMM32_B32))
5260 BuildMI(*LoopBB,
I,
DL,
TII->get(AMDGPU::S_GETREG_B32), Reg)
5284 unsigned InitReg,
unsigned ResultReg,
unsigned PhiReg,
5285 unsigned InitSaveExecReg,
int Offset,
bool UseGPRIdxMode,
5307 BuildMI(LoopBB,
I,
DL,
TII->get(TargetOpcode::PHI), PhiExec)
5314 BuildMI(LoopBB,
I,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), CurrentIdxReg)
5318 BuildMI(LoopBB,
I,
DL,
TII->get(AMDGPU::V_CMP_EQ_U32_e64), CondReg)
5320 .
addReg(Idx.getReg(), {}, Idx.getSubReg());
5329 if (UseGPRIdxMode) {
5331 SGPRIdxReg = CurrentIdxReg;
5334 BuildMI(LoopBB,
I,
DL,
TII->get(AMDGPU::S_ADD_I32), SGPRIdxReg)
5345 BuildMI(LoopBB,
I,
DL,
TII->get(AMDGPU::S_ADD_I32), AMDGPU::M0)
5378 unsigned InitResultReg,
unsigned PhiReg,
int Offset,
5379 bool UseGPRIdxMode,
Register &SGPRIdxReg) {
5387 const auto *BoolXExecRC =
TRI->getWaveMaskRegClass();
5406 InitResultReg, DstReg, PhiReg, TmpExec,
5407 Offset, UseGPRIdxMode, SGPRIdxReg);
5413 LoopBB->removeSuccessor(RemainderBB);
5415 LoopBB->addSuccessor(LandingPad);
5426static std::pair<unsigned, int>
5430 int NumElts =
TRI.getRegSizeInBits(*SuperRC) / 32;
5435 return std::pair(AMDGPU::sub0,
Offset);
5449 assert(Idx->getReg() != AMDGPU::NoRegister);
5474 return Idx->getReg();
5494 Register SrcReg =
TII->getNamedOperand(
MI, AMDGPU::OpName::src)->getReg();
5495 int Offset =
TII->getNamedOperand(
MI, AMDGPU::OpName::offset)->getImm();
5501 std::tie(SubReg,
Offset) =
5504 const bool UseGPRIdxMode = ST.useVGPRIndexMode();
5507 if (
TII->getRegisterInfo().isSGPRClass(IdxRC)) {
5511 if (UseGPRIdxMode) {
5518 TII->getIndirectGPRIDXPseudo(
TRI.getRegSizeInBits(*VecRC),
true);
5527 .
addReg(SrcReg, {}, SubReg)
5531 MI.eraseFromParent();
5547 UseGPRIdxMode, SGPRIdxReg);
5551 if (UseGPRIdxMode) {
5553 TII->getIndirectGPRIDXPseudo(
TRI.getRegSizeInBits(*VecRC),
true);
5555 BuildMI(*LoopBB, InsPt,
DL, GPRIDXDesc, Dst)
5560 BuildMI(*LoopBB, InsPt,
DL,
TII->get(AMDGPU::V_MOVRELS_B32_e32), Dst)
5561 .
addReg(SrcReg, {}, SubReg)
5565 MI.eraseFromParent();
5582 int Offset =
TII->getNamedOperand(
MI, AMDGPU::OpName::offset)->getImm();
5590 std::tie(SubReg,
Offset) =
5592 const bool UseGPRIdxMode = ST.useVGPRIndexMode();
5594 if (Idx->getReg() == AMDGPU::NoRegister) {
5605 MI.eraseFromParent();
5610 if (
TII->getRegisterInfo().isSGPRClass(IdxRC)) {
5614 if (UseGPRIdxMode) {
5618 TII->getIndirectGPRIDXPseudo(
TRI.getRegSizeInBits(*VecRC),
false);
5627 const MCInstrDesc &MovRelDesc =
TII->getIndirectRegWriteMovRelPseudo(
5628 TRI.getRegSizeInBits(*VecRC), 32,
false);
5634 MI.eraseFromParent();
5648 UseGPRIdxMode, SGPRIdxReg);
5651 if (UseGPRIdxMode) {
5653 TII->getIndirectGPRIDXPseudo(
TRI.getRegSizeInBits(*VecRC),
false);
5655 BuildMI(*LoopBB, InsPt,
DL, GPRIDXDesc, Dst)
5661 const MCInstrDesc &MovRelDesc =
TII->getIndirectRegWriteMovRelPseudo(
5662 TRI.getRegSizeInBits(*VecRC), 32,
false);
5663 BuildMI(*LoopBB, InsPt,
DL, MovRelDesc, Dst)
5669 MI.eraseFromParent();
5685 bool IsAdd = (
MI.getOpcode() == AMDGPU::S_ADD_U64_PSEUDO);
5686 if (ST.hasScalarAddSub64()) {
5688 unsigned Opc = IsAdd ? AMDGPU::S_ADD_U64 : AMDGPU::S_SUB_U64;
5698 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
5699 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
5702 MI, MRI, Src0, BoolRC, AMDGPU::sub0, &AMDGPU::SReg_32RegClass);
5704 MI, MRI, Src0, BoolRC, AMDGPU::sub1, &AMDGPU::SReg_32RegClass);
5707 MI, MRI, Src1, BoolRC, AMDGPU::sub0, &AMDGPU::SReg_32RegClass);
5709 MI, MRI, Src1, BoolRC, AMDGPU::sub1, &AMDGPU::SReg_32RegClass);
5714 unsigned LoOpc = IsAdd ? AMDGPU::S_ADD_U32 : AMDGPU::S_SUB_U32;
5715 unsigned HiOpc = IsAdd ? AMDGPU::S_ADDC_U32 : AMDGPU::S_SUBB_U32;
5721 Hi.setOperandDead(3);
5728 MI.eraseFromParent();
5742 Register SrcCond =
MI.getOperand(3).getReg();
5750 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src0);
5752 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src1);
5754 TRI->getAllocatableClass(
TII->getRegClass(
MI.getDesc(), Src0Idx));
5756 TRI->getAllocatableClass(
TII->getRegClass(
MI.getDesc(), Src1Idx));
5759 TRI->getSubRegisterClass(Src0RC, AMDGPU::sub0);
5761 TRI->getSubRegisterClass(Src1RC, AMDGPU::sub1);
5764 MI, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
5766 MI, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
5769 MI, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
5771 MI, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
5793 MI.eraseFromParent();
5798 case AMDGPU::S_MIN_U32:
5799 return std::numeric_limits<uint32_t>::max();
5800 case AMDGPU::S_MIN_I32:
5801 return std::numeric_limits<int32_t>::max();
5802 case AMDGPU::S_MAX_U32:
5803 return std::numeric_limits<uint32_t>::min();
5804 case AMDGPU::S_MAX_I32:
5805 return std::numeric_limits<int32_t>::min();
5806 case AMDGPU::V_ADD_F32_e64:
5808 case AMDGPU::V_SUB_F32_e64:
5810 case AMDGPU::S_ADD_I32:
5811 case AMDGPU::S_SUB_I32:
5812 case AMDGPU::S_OR_B32:
5813 case AMDGPU::S_XOR_B32:
5814 return std::numeric_limits<uint32_t>::min();
5815 case AMDGPU::S_AND_B32:
5816 return std::numeric_limits<uint32_t>::max();
5817 case AMDGPU::V_MIN_F32_e64:
5818 case AMDGPU::V_MAX_F32_e64:
5820 case AMDGPU::V_CMP_LT_U64_e64:
5821 return std::numeric_limits<uint64_t>::max();
5822 case AMDGPU::V_CMP_LT_I64_e64:
5823 return std::numeric_limits<int64_t>::max();
5824 case AMDGPU::V_CMP_GT_U64_e64:
5825 return std::numeric_limits<uint64_t>::min();
5826 case AMDGPU::V_CMP_GT_I64_e64:
5827 return std::numeric_limits<int64_t>::min();
5828 case AMDGPU::V_MIN_F64_e64:
5829 case AMDGPU::V_MAX_F64_e64:
5830 case AMDGPU::V_MIN_NUM_F64_e64:
5831 case AMDGPU::V_MAX_NUM_F64_e64:
5832 return 0x7FF8000000000000;
5833 case AMDGPU::S_ADD_U64_PSEUDO:
5834 case AMDGPU::S_SUB_U64_PSEUDO:
5835 case AMDGPU::S_OR_B64:
5836 case AMDGPU::S_XOR_B64:
5837 return std::numeric_limits<uint64_t>::min();
5838 case AMDGPU::S_AND_B64:
5839 return std::numeric_limits<uint64_t>::max();
5840 case AMDGPU::V_ADD_F64_e64:
5841 case AMDGPU::V_ADD_F64_pseudo_e64:
5842 return 0x8000000000000000;
5849 return Opc == AMDGPU::S_MIN_U32 ||
Opc == AMDGPU::S_MIN_I32 ||
5850 Opc == AMDGPU::S_MAX_U32 ||
Opc == AMDGPU::S_MAX_I32 ||
5851 Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_SUB_I32 ||
5852 Opc == AMDGPU::S_AND_B32 ||
Opc == AMDGPU::S_OR_B32 ||
5853 Opc == AMDGPU::S_XOR_B32 ||
Opc == AMDGPU::V_MIN_F32_e64 ||
5854 Opc == AMDGPU::V_MAX_F32_e64 ||
Opc == AMDGPU::V_ADD_F32_e64 ||
5855 Opc == AMDGPU::V_SUB_F32_e64;
5859 return Opc == AMDGPU::V_MIN_F32_e64 ||
Opc == AMDGPU::V_MAX_F32_e64 ||
5860 Opc == AMDGPU::V_ADD_F32_e64 ||
Opc == AMDGPU::V_SUB_F32_e64 ||
5861 Opc == AMDGPU::V_MIN_F64_e64 ||
Opc == AMDGPU::V_MAX_F64_e64 ||
5862 Opc == AMDGPU::V_MIN_NUM_F64_e64 ||
Opc == AMDGPU::V_MAX_NUM_F64_e64 ||
5863 Opc == AMDGPU::V_ADD_F64_e64 ||
Opc == AMDGPU::V_ADD_F64_pseudo_e64;
5866static std::tuple<unsigned, unsigned>
5870 case AMDGPU::S_MIN_U32:
5871 DPPOpc = AMDGPU::V_MIN_U32_dpp;
5873 case AMDGPU::S_MIN_I32:
5874 DPPOpc = AMDGPU::V_MIN_I32_dpp;
5876 case AMDGPU::S_MAX_U32:
5877 DPPOpc = AMDGPU::V_MAX_U32_dpp;
5879 case AMDGPU::S_MAX_I32:
5880 DPPOpc = AMDGPU::V_MAX_I32_dpp;
5882 case AMDGPU::S_ADD_I32:
5883 case AMDGPU::S_SUB_I32:
5884 DPPOpc = ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_dpp
5885 : AMDGPU::V_ADD_CO_U32_dpp;
5887 case AMDGPU::S_AND_B32:
5888 DPPOpc = AMDGPU::V_AND_B32_dpp;
5890 case AMDGPU::S_OR_B32:
5891 DPPOpc = AMDGPU::V_OR_B32_dpp;
5893 case AMDGPU::S_XOR_B32:
5894 DPPOpc = AMDGPU::V_XOR_B32_dpp;
5896 case AMDGPU::V_ADD_F32_e64:
5897 case AMDGPU::V_SUB_F32_e64:
5898 DPPOpc = AMDGPU::V_ADD_F32_dpp;
5900 case AMDGPU::V_MIN_F32_e64:
5901 DPPOpc = AMDGPU::V_MIN_F32_dpp;
5903 case AMDGPU::V_MAX_F32_e64:
5904 DPPOpc = AMDGPU::V_MAX_F32_dpp;
5906 case AMDGPU::V_CMP_LT_U64_e64:
5907 case AMDGPU::V_CMP_LT_I64_e64:
5908 case AMDGPU::V_CMP_GT_U64_e64:
5909 case AMDGPU::V_CMP_GT_I64_e64:
5910 case AMDGPU::S_ADD_U64_PSEUDO:
5911 case AMDGPU::S_SUB_U64_PSEUDO:
5912 case AMDGPU::S_AND_B64:
5913 case AMDGPU::S_OR_B64:
5914 case AMDGPU::S_XOR_B64:
5915 case AMDGPU::V_MIN_NUM_F64_e64:
5916 case AMDGPU::V_MIN_F64_e64:
5917 case AMDGPU::V_MAX_NUM_F64_e64:
5918 case AMDGPU::V_MAX_F64_e64:
5919 case AMDGPU::V_ADD_F64_pseudo_e64:
5920 case AMDGPU::V_ADD_F64_e64:
5921 DPPOpc = AMDGPU::V_MOV_B64_DPP_PSEUDO;
5926 unsigned ClampOpc =
Opc;
5927 if (!ST.getInstrInfo()->isVALU(
Opc,
true)) {
5928 if (
Opc == AMDGPU::S_SUB_I32)
5929 ClampOpc = AMDGPU::S_ADD_I32;
5930 if (
Opc == AMDGPU::S_ADD_U64_PSEUDO ||
Opc == AMDGPU::S_SUB_U64_PSEUDO)
5931 ClampOpc = AMDGPU::V_ADD_CO_U32_e64;
5932 else if (
Opc == AMDGPU::S_AND_B64)
5933 ClampOpc = AMDGPU::V_AND_B32_e64;
5934 else if (
Opc == AMDGPU::S_OR_B64)
5935 ClampOpc = AMDGPU::V_OR_B32_e64;
5936 else if (
Opc == AMDGPU::S_XOR_B64)
5937 ClampOpc = AMDGPU::V_XOR_B32_e64;
5939 ClampOpc = ST.getInstrInfo()->getVALUOp(ClampOpc);
5941 return {DPPOpc, ClampOpc};
5944static std::pair<Register, Register>
5951 TRI->getSubRegisterClass(SrcRC, AMDGPU::sub0);
5953 TII->buildExtractSubReg(
MI, MRI,
Op, SrcRC, AMDGPU::sub0, SrcSubRC);
5955 TII->buildExtractSubReg(
MI, MRI,
Op, SrcRC, AMDGPU::sub1, SrcSubRC);
5956 return {Op1L, Op1H};
5972 unsigned Stratergy =
static_cast<unsigned>(
MI.getOperand(2).
getImm());
5973 enum WAVE_REDUCE_STRATEGY :
unsigned {
DEFAULT = 0, ITERATIVE = 1,
DPP = 2 };
5975 unsigned MIOpc =
MI.getOpcode();
5989 case AMDGPU::S_MIN_U32:
5990 case AMDGPU::S_MIN_I32:
5991 case AMDGPU::V_MIN_F32_e64:
5992 case AMDGPU::S_MAX_U32:
5993 case AMDGPU::S_MAX_I32:
5994 case AMDGPU::V_MAX_F32_e64:
5995 case AMDGPU::S_AND_B32:
5996 case AMDGPU::S_OR_B32: {
6002 case AMDGPU::V_CMP_LT_U64_e64:
6003 case AMDGPU::V_CMP_LT_I64_e64:
6004 case AMDGPU::V_CMP_GT_U64_e64:
6005 case AMDGPU::V_CMP_GT_I64_e64:
6006 case AMDGPU::V_MIN_F64_e64:
6007 case AMDGPU::V_MIN_NUM_F64_e64:
6008 case AMDGPU::V_MAX_F64_e64:
6009 case AMDGPU::V_MAX_NUM_F64_e64:
6010 case AMDGPU::S_AND_B64:
6011 case AMDGPU::S_OR_B64: {
6017 case AMDGPU::S_XOR_B32:
6018 case AMDGPU::S_XOR_B64:
6019 case AMDGPU::S_ADD_I32:
6020 case AMDGPU::S_ADD_U64_PSEUDO:
6021 case AMDGPU::V_ADD_F32_e64:
6022 case AMDGPU::V_ADD_F64_e64:
6023 case AMDGPU::V_ADD_F64_pseudo_e64:
6024 case AMDGPU::S_SUB_I32:
6025 case AMDGPU::S_SUB_U64_PSEUDO:
6026 case AMDGPU::V_SUB_F32_e64: {
6033 bool IsWave32 = ST.isWave32();
6034 unsigned MovOpc = IsWave32 ? AMDGPU::S_MOV_B32 : AMDGPU::S_MOV_B64;
6035 MCRegister ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
6036 unsigned BitCountOpc =
6037 IsWave32 ? AMDGPU::S_BCNT1_I32_B32 : AMDGPU::S_BCNT1_I32_B64;
6041 auto NewAccumulator =
6047 case AMDGPU::S_XOR_B32:
6048 case AMDGPU::S_XOR_B64: {
6056 .
addReg(NewAccumulator->getOperand(0).getReg())
6064 if (
Opc == AMDGPU::S_XOR_B32) {
6071 BuildRegSequence(BB,
MI, DstReg, ParityRegister, DstHi);
6076 if (
Opc == AMDGPU::S_XOR_B32) {
6093 BuildRegSequence(BB,
MI, DstReg, DestSub0, DestSub1);
6097 case AMDGPU::S_SUB_I32: {
6106 .
addReg(NewAccumulator->getOperand(0).getReg());
6109 case AMDGPU::S_ADD_I32: {
6116 .
addReg(NewAccumulator->getOperand(0).getReg());
6122 .
addReg(NewAccumulator->getOperand(0).getReg());
6125 case AMDGPU::S_ADD_U64_PSEUDO:
6126 case AMDGPU::S_SUB_U64_PSEUDO: {
6146 if (
Imm == 1 &&
Opc == AMDGPU::S_ADD_U64_PSEUDO) {
6151 BuildRegSequence(BB,
MI, DstReg,
6152 NewAccumulator->getOperand(0).getReg(), DstHi);
6156 if (
Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6159 .
addReg(NewAccumulator->getOperand(0).getReg())
6169 Register LowOpcode =
Opc == AMDGPU::S_SUB_U64_PSEUDO
6171 : NewAccumulator->getOperand(0).getReg();
6175 if (ST.hasScalarMulHiInsts()) {
6186 BuildMI(BB,
MI,
DL,
TII->get(AMDGPU::V_MUL_HI_U32_e64), VCarryReg)
6189 BuildMI(BB,
MI,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), CarryReg)
6196 Register HiVal =
Opc == AMDGPU::S_SUB_U64_PSEUDO ? AddReg : DestSub1;
6202 if (
Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6208 BuildRegSequence(BB,
MI, DstReg, DestSub0, DestSub1);
6211 case AMDGPU::V_ADD_F32_e64:
6212 case AMDGPU::V_ADD_F64_e64:
6213 case AMDGPU::V_ADD_F64_pseudo_e64:
6214 case AMDGPU::V_SUB_F32_e64: {
6221 TII->get(is32BitOpc ? AMDGPU::V_CVT_F32_I32_e64
6222 : AMDGPU::V_CVT_F64_I32_e64),
6224 .
addReg(NewAccumulator->getOperand(0).getReg())
6229 unsigned srcMod = (MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32 ||
6230 MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64)
6233 unsigned MulOpc = is32BitOpc ? AMDGPU::V_MUL_F32_e64
6235 ? AMDGPU::V_MUL_F64_pseudo_e64
6236 : AMDGPU::V_MUL_F64_e64;
6246 BuildMI(BB,
MI,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
6263 BuildRegSequence(BB,
MI, DstReg, LaneValueLoReg, LaneValueHiReg);
6275 bool NeedsMovDPP = !is32BitOpc;
6280 bool IsWave32 = ST.isWave32();
6281 unsigned MovOpcForExec = IsWave32 ? AMDGPU::S_MOV_B32 : AMDGPU::S_MOV_B64;
6282 unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
6283 if (Stratergy == WAVE_REDUCE_STRATEGY::ITERATIVE ||
6309 MI.getOpcode() == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64
6313 TII->get(is32BitOpc ? AMDGPU::S_MOV_B32
6314 : AMDGPU::S_MOV_B64_IMM_PSEUDO),
6323 I = ComputeLoop->begin();
6325 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::PHI), AccumulatorReg)
6329 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::PHI), ActiveBitsReg)
6333 I = ComputeLoop->end();
6337 IsWave32 ? AMDGPU::S_FF1_I32_B32 : AMDGPU::S_FF1_I32_B64;
6342 bool hasSrc0Modifier = AMDGPU::getNamedOperandIdx(
6343 Opc, AMDGPU::OpName::src0_modifiers) != -1;
6344 bool hasSrc1Modifier = AMDGPU::getNamedOperandIdx(
6345 Opc, AMDGPU::OpName::src1_modifiers) != -1;
6347 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::clamp) != -1;
6349 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel) != -1;
6351 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::omod) != -1;
6352 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::V_READLANE_B32),
6356 if (ST.getInstrInfo()->isVALU(
Opc,
true)) {
6360 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::COPY), LaneValVgpr)
6362 OpDstReg = VgprResultReg;
6363 LaneValueReg = LaneValVgpr;
6366 if (hasSrc0Modifier)
6368 OpInstr.addReg(AccumulatorReg);
6369 if (hasSrc1Modifier)
6371 OpInstr.addReg(LaneValueReg);
6379 OpInstr.setOperandDead(3);
6380 if (ST.getInstrInfo()->isVALU(
Opc,
true)) {
6381 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32),
6395 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::V_READLANE_B32),
6399 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::V_READLANE_B32),
6403 auto LaneValue = BuildRegSequence(*ComputeLoop,
I, LaneValReg,
6404 LaneValueLoReg, LaneValueHiReg);
6406 case AMDGPU::S_OR_B64:
6407 case AMDGPU::S_AND_B64:
6408 case AMDGPU::S_XOR_B64: {
6411 .
addReg(LaneValue->getOperand(0).getReg())
6415 case AMDGPU::V_CMP_GT_I64_e64:
6416 case AMDGPU::V_CMP_GT_U64_e64:
6417 case AMDGPU::V_CMP_LT_I64_e64:
6418 case AMDGPU::V_CMP_LT_U64_e64: {
6423 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src);
6425 TRI->getAllocatableClass(
TII->getRegClass(
MI.getDesc(), SrcIdx));
6429 BuildRegSequence(*ComputeLoop,
I, AccumulatorVReg, SrcReg0Sub0,
6432 .
addReg(LaneValue->getOperand(0).getReg())
6433 .
addReg(AccumulatorVReg);
6435 unsigned AndOpc = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
6436 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AndOpc), ComparisonResultReg)
6440 NewAccumulator =
BuildMI(*ComputeLoop,
I,
DL,
6441 TII->get(AMDGPU::S_CSELECT_B64), DstReg)
6442 .
addReg(LaneValue->getOperand(0).getReg())
6446 case AMDGPU::V_MIN_F64_e64:
6447 case AMDGPU::V_MIN_NUM_F64_e64:
6448 case AMDGPU::V_MAX_F64_e64:
6449 case AMDGPU::V_MAX_NUM_F64_e64:
6450 case AMDGPU::V_ADD_F64_e64:
6451 case AMDGPU::V_ADD_F64_pseudo_e64: {
6453 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::src);
6455 TRI->getAllocatableClass(
TII->getRegClass(
MI.getDesc(), SrcIdx));
6462 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::COPY), AccumulatorVReg)
6465 MI.getOpcode() == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64
6471 .
addReg(LaneValue->getOperand(0).getReg())
6478 TII->get(AMDGPU::V_READFIRSTLANE_B32), LaneValLo);
6481 TII->get(AMDGPU::V_READFIRSTLANE_B32), LaneValHi);
6483 auto [Op1L, Op1H] =
ExtractSubRegs(*Iters, DstVregInst->getOperand(0),
6485 ReadLaneLo.addReg(Op1L);
6486 ReadLaneHi.addReg(Op1H);
6488 BuildRegSequence(*ComputeLoop,
I, DstReg, LaneValLo, LaneValHi);
6491 case AMDGPU::S_ADD_U64_PSEUDO:
6492 case AMDGPU::S_SUB_U64_PSEUDO: {
6495 .
addReg(LaneValue->getOperand(0).getReg())
6504 unsigned BITSETOpc =
6505 IsWave32 ? AMDGPU::S_BITSET0_B32 : AMDGPU::S_BITSET0_B64;
6506 BuildMI(*ComputeLoop,
I,
DL,
TII->get(BITSETOpc), NewActiveBitsReg)
6512 ActiveBits.addReg(NewActiveBitsReg).addMBB(ComputeLoop);
6516 if (!ST.hasScalarCompareEq64()) {
6519 unsigned CMPOpc = IsWave32 ? AMDGPU::S_OR_B32 : AMDGPU::S_OR_B64;
6521 BuildMI(*ComputeLoop,
I,
DL,
TII->get(CMPOpc), LaneMaskReg);
6524 IsWave32 ? AMDGPU::S_CMP_LG_U32 : AMDGPU::S_CMP_LG_U64;
6525 SetSCCInstr =
BuildMI(*ComputeLoop,
I,
DL,
TII->get(CMPOpc));
6527 SetSCCInstr.
addReg(NewActiveBitsReg);
6528 if (ST.hasScalarCompareEq64())
6531 SetSCCInstr.
addReg(NewActiveBitsReg);
6532 BuildMI(*ComputeLoop,
I,
DL,
TII->get(AMDGPU::S_CBRANCH_SCC1))
6537 assert(ST.hasDPP() &&
"Sub Target does not support DPP Operations");
6554 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::IMPLICIT_DEF), UndefExec);
6558 TII->get(is32BitOpc ? AMDGPU::S_MOV_B32
6559 : AMDGPU::S_MOV_B64_IMM_PSEUDO),
6562 auto IdentityCopyInstr =
6566 unsigned DPPOpc = std::get<0>(DPPClampOpcPair);
6567 unsigned ClampOpc = std::get<1>(DPPClampOpcPair);
6582 if (isFPOp && !NeedsMovDPP)
6585 if (isFPOp && !NeedsMovDPP)
6589 if (AMDGPU::getNamedOperandIdx(DPPOpc, AMDGPU::OpName::clamp) >= 0)
6598 bool isAddSub =
false,
6599 bool needsCarryIn =
false,
6601 unsigned InstrOpc = ClampOpc;
6604 InstrOpc = AMDGPU::V_ADDC_U32_e64;
6605 auto ClampInstr =
BuildMI(*CurrBB,
MI,
DL,
TII->get(InstrOpc), Dst);
6610 ClampInstr.addReg(CarryOutReg,
6616 ClampInstr.addReg(Src0);
6619 ClampInstr.addReg(Src1);
6622 if (AMDGPU::getNamedOperandIdx(InstrOpc, AMDGPU::OpName::clamp) >= 0)
6623 ClampInstr.addImm(0);
6625 ClampInstr.addImm(0);
6626 LastBcastInstr = ClampInstr;
6631 Opc == AMDGPU::S_ADD_U64_PSEUDO ||
Opc == AMDGPU::S_SUB_U64_PSEUDO;
6632 bool isBitWiseOpc =
Opc == AMDGPU::S_AND_B64 ||
6633 Opc == AMDGPU::S_OR_B64 ||
Opc == AMDGPU::S_XOR_B64;
6635 if (isAddSubOpc || isBitWiseOpc) {
6642 auto [Src0Lo, Src0Hi] =
6644 auto [Src1Lo, Src1Hi] =
6646 Register CarryReg = BuildClampInstr(
6647 ResLo, Src0Lo, Src1Lo, isAddSubOpc,
false);
6648 BuildClampInstr(ResHi, Src0Hi, Src1Hi, isAddSubOpc,
6649 isAddSubOpc, CarryReg);
6650 BuildRegSequence(*CurrBB,
MI, ReturnReg, ResLo, ResHi);
6679 SrcWithIdentityInstr =
6680 BuildSetInactiveInstr(SrcWithIdentity, SrcReg, IdentityVGPR);
6687 MI, IdentityCopyInstr->getOperand(0), SrcRegClass, ST, MRI);
6688 auto [SrcReg0Sub0, SrcReg0Sub1] =
6691 BuildSetInactiveInstr(SrcWithIdentitylo, SrcReg0Sub0, Reg0Sub0);
6693 BuildSetInactiveInstr(SrcWithIdentityhi, SrcReg0Sub1, Reg0Sub1);
6694 SrcWithIdentityInstr =
6695 BuildRegSequence(*CurrBB,
MI, SrcWithIdentity,
6702 BuildDPPMachineInstr(DPPRowShr1, SrcWithIdentityReg,
6705 DPPRowShr1 = BuildPostDPPInstr(SrcWithIdentityReg, DPPRowShr1);
6707 BuildDPPMachineInstr(DPPRowShr2, DPPRowShr1,
6710 DPPRowShr2 = BuildPostDPPInstr(DPPRowShr1, DPPRowShr2);
6712 BuildDPPMachineInstr(DPPRowShr4, DPPRowShr2,
6715 DPPRowShr4 = BuildPostDPPInstr(DPPRowShr2, DPPRowShr4);
6717 BuildDPPMachineInstr(DPPRowShr8, DPPRowShr4,
6720 DPPRowShr8 = BuildPostDPPInstr(DPPRowShr4, DPPRowShr8);
6722 if (ST.hasDPPBroadcasts()) {
6725 RowBcast15 = BuildPostDPPInstr(DPPRowShr8, RowBcast15);
6740 BuildClampInstr(RowBcast15, DPPRowShr8, SwizzledValue);
6761 BuildRegSequence(*CurrBB,
MI, SwizzledValue64, SwizzledValuelo,
6764 RowBcast15 = BuildPostDPPInstr(DPPRowShr8, SwizzledValue64);
6766 BuildClampInstr(RowBcast15, DPPRowShr8, SwizzledValue64);
6769 FinalDPPResult = RowBcast15;
6771 if (ST.hasDPPBroadcasts()) {
6774 RowBcast31 = BuildPostDPPInstr(RowBcast15, RowBcast31);
6790 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::V_MBCNT_LO_U32_B32_e64),
6794 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::V_MBCNT_HI_U32_B32_e64),
6800 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::S_MOV_B32), Lane32Offset)
6808 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::S_MOV_B32), WordSizeConst)
6813 .
addReg(ShiftedThreadID);
6818 .
addReg(PermuteByteOffset)
6828 auto [RowBcast15Lo, RowBcast15Hi] =
6832 .
addReg(PermuteByteOffset)
6837 .
addReg(PermuteByteOffset)
6840 BuildRegSequence(*CurrBB,
MI, PermutedValue, PermutedValuelo,
6844 RowBcast31 = BuildPostDPPInstr(RowBcast15, PermutedValue);
6846 BuildClampInstr(RowBcast31, RowBcast15, PermutedValue);
6848 FinalDPPResult = RowBcast31;
6850 if (MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32 ||
6851 MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64) {
6861 .
addReg(IsWave32 ? RowBcast15 : RowBcast31)
6864 FinalDPPResult = NegatedValVGPR;
6871 .
addImm(ST.getWavefrontSize() - 1);
6886 .
addImm(ST.getWavefrontSize() - 1);
6890 .
addImm(ST.getWavefrontSize() - 1);
6891 BuildRegSequence(*CurrBB,
MI, ReducedValSGPR, LaneValueLoReg,
6894 if (
Opc == AMDGPU::S_SUB_I32) {
6895 BuildMI(*CurrBB,
MI,
DL,
TII->get(AMDGPU::S_SUB_I32), NegatedReducedVal)
6899 }
else if (
Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6900 auto NegatedValInstr =
6909 .
addReg(
Opc == AMDGPU::S_SUB_I32 ||
Opc == AMDGPU::S_SUB_U64_PSEUDO
6915 MI.eraseFromParent();
6930 switch (
MI.getOpcode()) {
6931 case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U32:
6933 case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U64:
6935 case AMDGPU::WAVE_REDUCE_MIN_PSEUDO_I32:
6937 case AMDGPU::WAVE_REDUCE_MIN_PSEUDO_I64:
6939 case AMDGPU::WAVE_REDUCE_FMIN_PSEUDO_F32:
6941 case AMDGPU::WAVE_REDUCE_FMIN_PSEUDO_F64:
6944 ? AMDGPU::V_MIN_NUM_F64_e64
6945 : AMDGPU::V_MIN_F64_e64);
6946 case AMDGPU::WAVE_REDUCE_UMAX_PSEUDO_U32:
6948 case AMDGPU::WAVE_REDUCE_UMAX_PSEUDO_U64:
6950 case AMDGPU::WAVE_REDUCE_MAX_PSEUDO_I32:
6952 case AMDGPU::WAVE_REDUCE_MAX_PSEUDO_I64:
6954 case AMDGPU::WAVE_REDUCE_FMAX_PSEUDO_F32:
6956 case AMDGPU::WAVE_REDUCE_FMAX_PSEUDO_F64:
6959 ? AMDGPU::V_MAX_NUM_F64_e64
6960 : AMDGPU::V_MAX_F64_e64);
6961 case AMDGPU::WAVE_REDUCE_ADD_PSEUDO_I32:
6963 case AMDGPU::WAVE_REDUCE_ADD_PSEUDO_U64:
6965 case AMDGPU::WAVE_REDUCE_FADD_PSEUDO_F32:
6967 case AMDGPU::WAVE_REDUCE_FADD_PSEUDO_F64:
6970 ? AMDGPU::V_ADD_F64_pseudo_e64
6971 : AMDGPU::V_ADD_F64_e64);
6972 case AMDGPU::WAVE_REDUCE_SUB_PSEUDO_I32:
6974 case AMDGPU::WAVE_REDUCE_SUB_PSEUDO_U64:
6976 case AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32:
6978 case AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64:
6983 ? AMDGPU::V_ADD_F64_pseudo_e64
6984 : AMDGPU::V_ADD_F64_e64);
6985 case AMDGPU::WAVE_REDUCE_AND_PSEUDO_B32:
6987 case AMDGPU::WAVE_REDUCE_AND_PSEUDO_B64:
6989 case AMDGPU::WAVE_REDUCE_OR_PSEUDO_B32:
6991 case AMDGPU::WAVE_REDUCE_OR_PSEUDO_B64:
6993 case AMDGPU::WAVE_REDUCE_XOR_PSEUDO_B32:
6995 case AMDGPU::WAVE_REDUCE_XOR_PSEUDO_B64:
6997 case AMDGPU::S_UADDO_PSEUDO:
6998 case AMDGPU::S_USUBO_PSEUDO: {
7004 unsigned Opc = (
MI.getOpcode() == AMDGPU::S_UADDO_PSEUDO)
7006 : AMDGPU::S_SUB_U32;
7014 Subtarget->isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
7017 MI.eraseFromParent();
7020 case AMDGPU::S_ADD_U64_PSEUDO:
7021 case AMDGPU::S_SUB_U64_PSEUDO: {
7024 case AMDGPU::V_ADD_U64_PSEUDO:
7025 case AMDGPU::V_SUB_U64_PSEUDO: {
7026 bool IsAdd = (
MI.getOpcode() == AMDGPU::V_ADD_U64_PSEUDO);
7032 if (ST.hasAddSubU64Insts()) {
7034 TII->get(IsAdd ? AMDGPU::V_ADD_U64_e64
7035 : AMDGPU::V_SUB_U64_e64),
7040 TII->legalizeOperands(*
I);
7041 MI.eraseFromParent();
7045 if (IsAdd && ST.hasLshlAddU64Inst()) {
7051 TII->legalizeOperands(*
Add);
7052 MI.eraseFromParent();
7056 const auto *CarryRC =
TRI->getWaveMaskRegClass();
7066 : &AMDGPU::VReg_64RegClass;
7069 : &AMDGPU::VReg_64RegClass;
7072 TRI->getSubRegisterClass(Src0RC, AMDGPU::sub0);
7074 TRI->getSubRegisterClass(Src1RC, AMDGPU::sub1);
7077 MI, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
7079 MI, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
7082 MI, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
7084 MI, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
7087 IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
7094 unsigned HiOpc = IsAdd ? AMDGPU::V_ADDC_U32_e64 : AMDGPU::V_SUBB_U32_e64;
7108 TII->legalizeOperands(*LoHalf);
7109 TII->legalizeOperands(*HiHalf);
7110 MI.eraseFromParent();
7113 case AMDGPU::S_ADD_CO_PSEUDO:
7114 case AMDGPU::S_SUB_CO_PSEUDO: {
7124 if (Src0.isReg() &&
TRI->isVectorRegister(MRI, Src0.getReg())) {
7126 BuildMI(*BB, MII,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp0)
7128 Src0.setReg(RegOp0);
7130 if (Src1.isReg() &&
TRI->isVectorRegister(MRI, Src1.getReg())) {
7132 BuildMI(*BB, MII,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp1)
7134 Src1.setReg(RegOp1);
7137 if (
TRI->isVectorRegister(MRI, Src2.getReg())) {
7138 BuildMI(*BB, MII,
DL,
TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp2)
7140 Src2.setReg(RegOp2);
7143 if (ST.isWave64()) {
7144 if (ST.hasScalarCompareEq64()) {
7151 TRI->getSubRegisterClass(Src2RC, AMDGPU::sub0);
7153 MII, MRI, Src2, Src2RC, AMDGPU::sub0, SubRC);
7155 MII, MRI, Src2, Src2RC, AMDGPU::sub1, SubRC);
7158 BuildMI(*BB, MII,
DL,
TII->get(AMDGPU::S_OR_B32), Src2_32)
7172 unsigned Opc =
MI.getOpcode() == AMDGPU::S_ADD_CO_PSEUDO
7173 ? AMDGPU::S_ADDC_U32
7174 : AMDGPU::S_SUBB_U32;
7179 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
7185 MI.eraseFromParent();
7188 case AMDGPU::SI_INIT_M0: {
7191 TII->get(M0Init.
isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32),
7194 MI.eraseFromParent();
7197 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM: {
7200 TII->get(AMDGPU::S_CMP_EQ_U32))
7205 case AMDGPU::GET_GROUPSTATICSIZE: {
7209 .
add(
MI.getOperand(0))
7211 MI.eraseFromParent();
7214 case AMDGPU::GET_SHADERCYCLESHILO: {
7229 .
addImm(HwregEncoding::encode(ID_SHADER_CYCLES_HI, 0, 32));
7232 .
addImm(HwregEncoding::encode(ID_SHADER_CYCLES, 0, 32));
7235 .
addImm(HwregEncoding::encode(ID_SHADER_CYCLES_HI, 0, 32));
7244 .
add(
MI.getOperand(0))
7249 MI.eraseFromParent();
7252 case AMDGPU::SI_INDIRECT_SRC_V1:
7253 case AMDGPU::SI_INDIRECT_SRC_V2:
7254 case AMDGPU::SI_INDIRECT_SRC_V3:
7255 case AMDGPU::SI_INDIRECT_SRC_V4:
7256 case AMDGPU::SI_INDIRECT_SRC_V5:
7257 case AMDGPU::SI_INDIRECT_SRC_V6:
7258 case AMDGPU::SI_INDIRECT_SRC_V7:
7259 case AMDGPU::SI_INDIRECT_SRC_V8:
7260 case AMDGPU::SI_INDIRECT_SRC_V9:
7261 case AMDGPU::SI_INDIRECT_SRC_V10:
7262 case AMDGPU::SI_INDIRECT_SRC_V11:
7263 case AMDGPU::SI_INDIRECT_SRC_V12:
7264 case AMDGPU::SI_INDIRECT_SRC_V16:
7265 case AMDGPU::SI_INDIRECT_SRC_V32:
7267 case AMDGPU::SI_INDIRECT_DST_V1:
7268 case AMDGPU::SI_INDIRECT_DST_V2:
7269 case AMDGPU::SI_INDIRECT_DST_V3:
7270 case AMDGPU::SI_INDIRECT_DST_V4:
7271 case AMDGPU::SI_INDIRECT_DST_V5:
7272 case AMDGPU::SI_INDIRECT_DST_V6:
7273 case AMDGPU::SI_INDIRECT_DST_V7:
7274 case AMDGPU::SI_INDIRECT_DST_V8:
7275 case AMDGPU::SI_INDIRECT_DST_V9:
7276 case AMDGPU::SI_INDIRECT_DST_V10:
7277 case AMDGPU::SI_INDIRECT_DST_V11:
7278 case AMDGPU::SI_INDIRECT_DST_V12:
7279 case AMDGPU::SI_INDIRECT_DST_V16:
7280 case AMDGPU::SI_INDIRECT_DST_V32:
7282 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
7283 case AMDGPU::SI_KILL_I1_PSEUDO:
7285 case AMDGPU::V_CNDMASK_B64_PSEUDO: {
7289 case AMDGPU::SI_BR_UNDEF: {
7291 .
add(
MI.getOperand(0));
7293 MI.eraseFromParent();
7296 case AMDGPU::ADJCALLSTACKUP:
7297 case AMDGPU::ADJCALLSTACKDOWN: {
7304 case AMDGPU::SI_CALL_ISEL: {
7305 unsigned ReturnAddrReg =
TII->getRegisterInfo().getReturnAddressReg(*MF);
7316 MI.eraseFromParent();
7319 case AMDGPU::V_ADDC_U32_e32:
7320 case AMDGPU::V_SUBB_U32_e32:
7321 case AMDGPU::V_SUBBREV_U32_e32:
7324 TII->legalizeOperands(
MI);
7326 case AMDGPU::DS_GWS_INIT:
7327 case AMDGPU::DS_GWS_SEMA_BR:
7328 case AMDGPU::DS_GWS_BARRIER:
7329 case AMDGPU::DS_GWS_SEMA_V:
7330 case AMDGPU::DS_GWS_SEMA_P:
7331 case AMDGPU::DS_GWS_SEMA_RELEASE_ALL:
7339 case AMDGPU::S_SETREG_B32: {
7349 auto [ID,
Offset, Width] =
7355 const unsigned SetMask = WidthMask <<
Offset;
7358 unsigned SetDenormOp = 0;
7359 unsigned SetRoundOp = 0;
7367 SetRoundOp = AMDGPU::S_ROUND_MODE;
7368 SetDenormOp = AMDGPU::S_DENORM_MODE;
7370 SetRoundOp = AMDGPU::S_ROUND_MODE;
7372 SetDenormOp = AMDGPU::S_DENORM_MODE;
7375 if (SetRoundOp || SetDenormOp) {
7377 if (Def && Def->isMoveImmediate() && Def->getOperand(1).isImm()) {
7378 unsigned ImmVal = Def->getOperand(1).getImm();
7392 MI.eraseFromParent();
7401 MI.setDesc(
TII->get(AMDGPU::S_SETREG_B32_mode));
7405 case AMDGPU::S_INVERSE_BALLOT_U32:
7406 case AMDGPU::S_INVERSE_BALLOT_U64:
7409 MI.setDesc(
TII->get(AMDGPU::COPY));
7411 case AMDGPU::ENDPGM_TRAP: {
7413 MI.setDesc(
TII->get(AMDGPU::S_ENDPGM));
7433 MI.eraseFromParent();
7436 case AMDGPU::SIMULATED_TRAP: {
7437 assert(Subtarget->hasPrivEnabledTrap2NopBug());
7439 TII->insertSimulatedTrap(MRI, *BB,
MI,
MI.getDebugLoc());
7440 MI.eraseFromParent();
7443 case AMDGPU::SI_TCRETURN_GFX_WholeWave:
7444 case AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN: {
7450 assert(Setup &&
"Couldn't find SI_SETUP_WHOLE_WAVE_FUNC");
7451 Register OriginalExec = Setup->getOperand(0).getReg();
7453 MI.getOperand(0).setReg(OriginalExec);
7456 case AMDGPU::V_DOT2_F32_F16:
7457 case AMDGPU::V_DOT2_F32_BF16: {
7464 case AMDGPU::SCHED_BARRIER:
7465 case AMDGPU::SCHED_GROUP_BARRIER:
7466 MI.getOperand(0).setImm(
MI.getOperand(0).getImm() &
7503 return (VT == MVT::i16) ? MVT::i16 : MVT::i32;
7507 return (Ty.getScalarSizeInBits() <= 16 && Subtarget->has16BitInsts())
7536 if (!Subtarget->hasMadMacF32Insts())
7537 return Subtarget->hasFastFMAF32();
7543 return Subtarget->hasFastFMAF32() || Subtarget->hasDLInsts();
7546 return Subtarget->hasFastFMAF32() && Subtarget->hasDLInsts();
7552 return Subtarget->has16BitInsts() &&
7570 F.getDenormalFPEnv());
7575 switch (Ty.getScalarSizeInBits()) {
7593 return Subtarget->hasMadMacF32Insts() &&
7596 return Subtarget->hasMadF16() &&
7607 if (Ty.getScalarSizeInBits() == 16)
7609 if (Ty.getScalarSizeInBits() == 32)
7624 F.getDenormalFPEnv());
7635 unsigned Opc =
Op.getOpcode();
7636 EVT VT =
Op.getValueType();
7648 LoOps.
append(TrailingOps.begin(), TrailingOps.end());
7649 HiOps.
append(TrailingOps.begin(), TrailingOps.end());
7662 [[maybe_unused]]
EVT VT =
Op.getValueType();
7664 assert((VT == MVT::v2i32 || VT == MVT::v4i32 || VT == MVT::v8i32 ||
7665 VT == MVT::v16i32) &&
7666 "Unexpected ValueType.");
7675 unsigned Opc =
Op.getOpcode();
7676 EVT VT =
Op.getValueType();
7685 DAG.
getNode(
Opc, SL, Lo0.getValueType(), Lo0, Lo1,
Op->getFlags());
7687 DAG.
getNode(
Opc, SL, Hi0.getValueType(), Hi0, Hi1,
Op->getFlags());
7694 unsigned Opc =
Op.getOpcode();
7695 EVT VT =
Op.getValueType();
7712 DAG.
getNode(
Opc, SL, ResVT.first, Lo0, Lo1, Lo2,
Op->getFlags());
7714 DAG.
getNode(
Opc, SL, ResVT.second, Hi0, Hi1, Hi2,
Op->getFlags());
7720 switch (
Op.getOpcode()) {
7724 return LowerBRCOND(
Op, DAG);
7726 return LowerRETURNADDR(
Op, DAG);
7728 return LowerSPONENTRY(
Op, DAG);
7731 assert((!Result.getNode() || Result.getNode()->getNumValues() == 2) &&
7732 "Load should return a value and a chain");
7736 EVT VT =
Op.getValueType();
7738 return lowerFSQRTF32(
Op, DAG);
7740 return lowerFSQRTF64(
Op, DAG);
7745 return LowerTrig(
Op, DAG);
7747 return LowerSELECT(
Op, DAG);
7749 return LowerFDIV(
Op, DAG);
7751 return LowerFFREXP(
Op, DAG);
7753 return LowerATOMIC_CMP_SWAP(
Op, DAG);
7755 return LowerSTORE(
Op, DAG);
7759 return LowerGlobalAddress(MFI,
Op, DAG);
7764 return LowerExternalSymbol(
Op, DAG);
7766 return LowerINTRINSIC_WO_CHAIN(
Op, DAG);
7768 return LowerCONVERT_FROM_ARBITRARY_FP(
Op, DAG);
7770 return LowerCONVERT_TO_ARBITRARY_FP(
Op, DAG);
7772 return LowerINTRINSIC_W_CHAIN(
Op, DAG);
7774 return LowerINTRINSIC_VOID(
Op, DAG);
7776 return lowerADDRSPACECAST(
Op, DAG);
7778 return lowerINSERT_SUBVECTOR(
Op, DAG);
7780 return lowerINSERT_VECTOR_ELT(
Op, DAG);
7782 return lowerEXTRACT_VECTOR_ELT(
Op, DAG);
7784 return lowerVECTOR_SHUFFLE(
Op, DAG);
7786 return lowerSCALAR_TO_VECTOR(
Op, DAG);
7788 return lowerBUILD_VECTOR(
Op, DAG);
7791 return lowerFP_ROUND(
Op, DAG);
7793 return lowerTRAP(
Op, DAG);
7795 return lowerDEBUGTRAP(
Op, DAG);
7804 if (
Op.getValueType().isVector() &&
Op.getValueType() != MVT::v2i16 &&
7805 Op.getOperand(0).getValueType().getScalarType() == MVT::f32)
7809 if (
Op.getValueType() == MVT::bf16) {
7839 return lowerFMINNUM_FMAXNUM(
Op, DAG);
7842 return lowerFMINIMUMNUM_FMAXIMUMNUM(
Op, DAG);
7845 return lowerFLDEXP(
Op, DAG);
7850 if (Subtarget->hasVCvtPkIU16F32() &&
Op.getValueType() == MVT::i16 &&
7851 Op.getOperand(0).getValueType() == MVT::f32) {
7877 return lowerFCOPYSIGN(
Op, DAG);
7879 return lowerMUL(
Op, DAG);
7882 return lowerXMULO(
Op, DAG);
7885 return lowerXMUL_LOHI(
Op, DAG);
7906 return LowerINLINEASM(
Op, DAG);
7912static std::pair<SDValue, SDValue>
7942 EVT FittingLoadVT = LoadVT;
7974SDValue SITargetLowering::adjustLoadValueType(
unsigned Opcode,
MemSDNode *M,
7977 bool IsIntrinsic)
const {
7980 bool IsTFE =
M->getNumValues() == 3;
7981 bool Unpacked = Subtarget->hasUnpackedD16VMem();
7982 EVT LoadVT =
M->getValueType(0);
7984 EVT EquivLoadVT = LoadVT;
8001 SDVTList VTList = DAG.
getVTList(LoadDWordsVT, MVT::Other);
8003 Opcode,
DL, VTList,
Ops,
M->getMemoryVT(),
M->getMemOperand());
8011 SDVTList VTList = DAG.
getVTList(EquivLoadVT, MVT::Other);
8015 M->getMemoryVT(),
M->getMemOperand());
8026 EVT LoadVT =
M->getValueType(0);
8035 "unsupported sub-dword format buffer load",
DL.getDebugLoc()));
8039 assert(
M->getNumValues() == 2 ||
M->getNumValues() == 3);
8040 bool IsTFE =
M->getNumValues() == 3;
8042 if (IsD16 && IsTFE && !Subtarget->hasBufferTFEFormatD16()) {
8045 "TFE D16 format buffer load is not supported on this GPU",
8048 M->getOperand(0),
DL);
8051 unsigned Opc = IsD16 ? (IsTFE ? AMDGPUISD::BUFFER_LOAD_FORMAT_D16_TFE
8052 : AMDGPUISD::BUFFER_LOAD_FORMAT_D16)
8053 : IsFormat ? (IsTFE ? AMDGPUISD::BUFFER_LOAD_FORMAT_TFE
8054 : AMDGPUISD::BUFFER_LOAD_FORMAT)
8055 : IsTFE ? AMDGPUISD::BUFFER_LOAD_TFE
8056 : AMDGPUISD::BUFFER_LOAD;
8059 return adjustLoadValueType(
Opc, M, DAG,
Ops);
8063 return handleByteShortBufferLoads(DAG, LoadVT,
DL,
Ops,
M->getMemOperand(),
8067 return getMemIntrinsicNode(
Opc,
DL,
M->getVTList(),
Ops, IntVT,
8068 M->getMemOperand(), DAG);
8072 SDVTList VTList = IsTFE ? DAG.
getVTList(CastVT, MVT::i32, MVT::Other)
8074 SDValue MemNode = getMemIntrinsicNode(
Opc,
DL, VTList,
Ops, CastVT,
8075 M->getMemOperand(), DAG);
8085 EVT VT =
N->getValueType(0);
8109 Exec = AMDGPU::EXEC_LO;
8111 Exec = AMDGPU::EXEC;
8128 bool Signed = IntrinsicID == Intrinsic::amdgcn_sbfe;
8130 EVT VT =
Op.getValueType();
8135 if (VT != MVT::i32) {
8143 return DAG.
getNode(
Signed ? AMDGPUISD::BFE_I32 : AMDGPUISD::BFE_U32,
DL, VT,
8152 EVT VT =
N->getValueType(0);
8154 unsigned IID =
N->getConstantOperandVal(0);
8155 bool IsPermLane16 = IID == Intrinsic::amdgcn_permlane16 ||
8156 IID == Intrinsic::amdgcn_permlanex16;
8157 bool IsSetInactive = IID == Intrinsic::amdgcn_set_inactive ||
8158 IID == Intrinsic::amdgcn_set_inactive_chain_arg;
8159 bool IsPermlaneShuffle = IID == Intrinsic::amdgcn_permlane_bcast ||
8160 IID == Intrinsic::amdgcn_permlane_up ||
8161 IID == Intrinsic::amdgcn_permlane_down ||
8162 IID == Intrinsic::amdgcn_permlane_xor;
8167 unsigned SplitSize = 32;
8168 if (IID == Intrinsic::amdgcn_update_dpp && (ValSize % 64 == 0) &&
8169 ST->hasDPALU_DPP() &&
8177 case Intrinsic::amdgcn_permlane16:
8178 case Intrinsic::amdgcn_permlanex16:
8179 case Intrinsic::amdgcn_update_dpp:
8184 case Intrinsic::amdgcn_writelane:
8185 case Intrinsic::amdgcn_permlane_bcast:
8186 case Intrinsic::amdgcn_permlane_up:
8187 case Intrinsic::amdgcn_permlane_down:
8188 case Intrinsic::amdgcn_permlane_xor:
8191 case Intrinsic::amdgcn_readlane:
8192 case Intrinsic::amdgcn_set_inactive:
8193 case Intrinsic::amdgcn_set_inactive_chain_arg:
8194 case Intrinsic::amdgcn_mov_dpp8:
8197 case Intrinsic::amdgcn_readfirstlane:
8198 case Intrinsic::amdgcn_permlane64:
8208 if (
SDNode *GL =
N->getGluedNode()) {
8210 GL = GL->getOperand(0).getNode();
8220 if (IID == Intrinsic::amdgcn_readlane || IID == Intrinsic::amdgcn_writelane ||
8221 IID == Intrinsic::amdgcn_mov_dpp8 ||
8222 IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16 ||
8223 IsPermlaneShuffle) {
8224 Src1 =
N->getOperand(2);
8225 if (IID == Intrinsic::amdgcn_writelane ||
8226 IID == Intrinsic::amdgcn_update_dpp || IsPermLane16 ||
8228 Src2 =
N->getOperand(3);
8231 if (ValSize == SplitSize) {
8241 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16) {
8246 if (IID == Intrinsic::amdgcn_writelane) {
8251 SDValue LaneOp = createLaneOp(Src0, Src1, Src2, MVT::i32);
8253 return IsFloat ? DAG.
getBitcast(VT, Trunc) : Trunc;
8256 if (ValSize % SplitSize != 0)
8260 EVT VT =
N->getValueType(0);
8264 unsigned NumOperands =
N->getNumOperands();
8266 SDNode *GL =
N->getGluedNode();
8271 for (
unsigned i = 0; i != NE; ++i) {
8272 for (
unsigned j = 0, e = GL ? NumOperands - 1 : NumOperands; j != e;
8274 SDValue Operand =
N->getOperand(j);
8304 if (SplitSize == 32) {
8306 return unrollLaneOp(LaneOp.
getNode());
8312 unsigned SubVecNumElt =
8316 SDValue Src0SubVec, Src1SubVec, Src2SubVec;
8317 for (
unsigned i = 0, EltIdx = 0; i < ValSize / SplitSize; i++) {
8321 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive ||
8327 createLaneOp(Src0SubVec, Src1SubVec, Src2, SubVecVT));
8328 }
else if (IID == Intrinsic::amdgcn_writelane) {
8332 createLaneOp(Src0SubVec, Src1, Src2SubVec, SubVecVT));
8334 Pieces.
push_back(createLaneOp(Src0SubVec, Src1, Src2, SubVecVT));
8337 EltIdx += SubVecNumElt;
8351 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16)
8354 if (IID == Intrinsic::amdgcn_writelane)
8357 SDValue LaneOp = createLaneOp(Src0, Src1, Src2, VecVT);
8364 EVT VT =
N->getValueType(0);
8382 auto MakeIntrinsic = [&DAG, &SL](
unsigned IID,
MVT RetVT,
8392 SDValue BPermute = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute, MVT::i32,
8393 {ShiftedIndex, ValueI32});
8403 SDValue WWMValue = MakeIntrinsic(Intrinsic::amdgcn_set_inactive, MVT::i32,
8404 {ValueI32, PoisonVal});
8405 SDValue WWMIndex = MakeIntrinsic(Intrinsic::amdgcn_set_inactive, MVT::i32,
8406 {ShiftedIndex, PoisonVal});
8409 MakeIntrinsic(Intrinsic::amdgcn_permlane64, MVT::i32, {WWMValue});
8412 SDValue BPermSameHalf = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute, MVT::i32,
8413 {WWMIndex, WWMValue});
8414 SDValue BPermOtherHalf = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute,
8415 MVT::i32, {WWMIndex, Swapped});
8417 MakeIntrinsic(Intrinsic::amdgcn_wwm, MVT::i32, {BPermOtherHalf});
8425 MakeIntrinsic(Intrinsic::amdgcn_mbcnt_lo, MVT::i32,
8433 DAG.
getSetCC(SL, MVT::i1, SameOrOtherHalf,
8443 switch (
N->getOpcode()) {
8460 unsigned IID =
N->getConstantOperandVal(0);
8462 case Intrinsic::amdgcn_wave_reduce_min:
8463 case Intrinsic::amdgcn_wave_reduce_umin:
8464 case Intrinsic::amdgcn_wave_reduce_max:
8465 case Intrinsic::amdgcn_wave_reduce_umax:
8466 case Intrinsic::amdgcn_wave_reduce_add:
8467 case Intrinsic::amdgcn_wave_reduce_sub:
8468 case Intrinsic::amdgcn_wave_reduce_and:
8469 case Intrinsic::amdgcn_wave_reduce_or:
8470 case Intrinsic::amdgcn_wave_reduce_xor: {
8471 EVT VT =
N->getValueType(0);
8475 bool NeedsSignExt = IID == Intrinsic::amdgcn_wave_reduce_min ||
8476 IID == Intrinsic::amdgcn_wave_reduce_max ||
8477 IID == Intrinsic::amdgcn_wave_reduce_add ||
8478 IID == Intrinsic::amdgcn_wave_reduce_sub;
8482 N->getOperand(0), ExtSrc,
N->getOperand(2));
8486 case Intrinsic::amdgcn_make_buffer_rsrc:
8487 Results.push_back(lowerPointerAsRsrcIntrin(
N, DAG));
8489 case Intrinsic::amdgcn_cvt_pkrtz: {
8494 DAG.
getNode(AMDGPUISD::CVT_PKRTZ_F16_F32, SL, MVT::i32, Src0, Src1);
8498 case Intrinsic::amdgcn_cvt_pknorm_i16:
8499 case Intrinsic::amdgcn_cvt_pknorm_u16:
8500 case Intrinsic::amdgcn_cvt_pk_i16:
8501 case Intrinsic::amdgcn_cvt_pk_u16: {
8507 if (IID == Intrinsic::amdgcn_cvt_pknorm_i16)
8508 Opcode = AMDGPUISD::CVT_PKNORM_I16_F32;
8509 else if (IID == Intrinsic::amdgcn_cvt_pknorm_u16)
8510 Opcode = AMDGPUISD::CVT_PKNORM_U16_F32;
8511 else if (IID == Intrinsic::amdgcn_cvt_pk_i16)
8512 Opcode = AMDGPUISD::CVT_PK_I16_I32;
8514 Opcode = AMDGPUISD::CVT_PK_U16_U32;
8516 EVT VT =
N->getValueType(0);
8525 case Intrinsic::amdgcn_s_buffer_load: {
8527 EVT VT =
Op.getValueType();
8529 Op.getOperand(1),
Op.getOperand(2),
8530 Op.getOperand(3), DAG));
8533 case Intrinsic::amdgcn_dead: {
8534 for (
unsigned I = 0, E =
N->getNumValues();
I < E; ++
I)
8542 if (
N->getConstantOperandVal(1) != Intrinsic::amdgcn_ptr_s_buffer_load &&
8543 N->getValueType(0).isSimple() &&
8544 SBufferLoadDiagnosticVTs[
N->getSimpleValueType(0).SimpleTy])
8549 for (
unsigned I = 0;
I < Res.getNumOperands();
I++) {
8550 Results.push_back(Res.getOperand(
I));
8553 for (
unsigned I = 0;
I <
N->getNumValues(); ++
I)
8554 Results.push_back(Res.getValue(
I));
8563 EVT VT =
N->getValueType(0);
8568 EVT SelectVT = NewVT;
8569 if (NewVT.
bitsLT(MVT::i32)) {
8572 SelectVT = MVT::i32;
8578 if (NewVT != SelectVT)
8584 if (
N->getValueType(0) != MVT::v2f16)
8596 if (
N->getValueType(0) != MVT::v2f16)
8608 if (
N->getValueType(0) != MVT::f16)
8623 if (U.get() !=
Value)
8626 if (U.getUser()->getOpcode() == Opcode)
8632unsigned SITargetLowering::isCFIntrinsic(
const SDNode *Intr)
const {
8635 case Intrinsic::amdgcn_if:
8636 return AMDGPUISD::IF;
8637 case Intrinsic::amdgcn_else:
8638 return AMDGPUISD::ELSE;
8639 case Intrinsic::amdgcn_loop:
8640 return AMDGPUISD::LOOP;
8641 case Intrinsic::amdgcn_end_cf:
8661 if (Subtarget->isAmdPalOS() || Subtarget->isMesa3DOS())
8685 assert(GVar->isDeclaration() &&
8686 "AS 3 & 13 GVs should be declaration here "
8687 "when object linking is enabled");
8702 SDNode *Intr = BRCOND.getOperand(1).getNode();
8719 Intr =
LHS.getNode();
8727 assert(BR &&
"brcond missing unconditional branch user");
8732 unsigned CFNode = isCFIntrinsic(Intr);
8752 Ops.push_back(Target);
8760 SDValue
Ops[] = {SDValue(Result, 0),
BRCOND.getOperand(0)};
8767 SDValue
Ops[] = {
BR->getOperand(0),
BRCOND.getOperand(2)};
8772 SDValue Chain = SDValue(Result,
Result->getNumValues() - 1);
8775 for (
unsigned i = 1, e = Intr->
getNumValues() - 1; i != e; ++i) {
8781 SDValue(Result, i - 1), SDValue());
8794 MVT VT =
Op.getSimpleValueType();
8797 if (
Op.getConstantOperandVal(0) != 0)
8801 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
8803 if (
Info->isEntryFunction())
8820 SIMachineFunctionInfo *MFI = MF.
getInfo<SIMachineFunctionInfo>();
8834 return Op.getValueType().bitsLE(VT)
8842 EVT DstVT =
Op.getValueType();
8849 unsigned Opc =
Op.getOpcode();
8850 SDValue
Flags =
Op.getOperand(1);
8860 bool IsStrict =
Op->isStrictFPOpcode();
8861 SDValue Src =
Op.getOperand(IsStrict ? 1 : 0);
8862 EVT SrcVT = Src.getValueType();
8863 EVT DstVT =
Op.getValueType();
8866 assert(Subtarget->hasCvtPkF16F32Inst() &&
"support v_cvt_pk_f16_f32");
8869 return SrcVT == MVT::v2f32 ?
Op : splitFP_ROUNDVectorOp(
Op, DAG);
8876 if (DstVT == MVT::f16) {
8881 if (!Subtarget->has16BitInsts()) {
8886 if (
Op->getFlags().hasApproximateFuncs()) {
8887 SDValue
Flags =
Op.getOperand(1);
8897 "custom lower FP_ROUND for f16 or bf16");
8898 assert(Subtarget->hasBF16ConversionInsts() &&
"f32 -> bf16 is legal");
8915 EVT VT =
Op.getValueType();
8917 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
8918 bool IsIEEEMode =
Info->getMode().IEEE;
8924 if (IsIEEEMode && !Subtarget->hasIEEEMinimumMaximumInsts())
8927 if (VT == MVT::v4f16 || VT == MVT::v8f16 || VT == MVT::v16f16 ||
8928 VT == MVT::v32f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16 ||
8929 VT == MVT::v16bf16 || VT == MVT::v32bf16 || VT == MVT::v4f64 ||
8930 VT == MVT::v8f64 || VT == MVT::v16f64 || VT == MVT::v32f64)
8936SITargetLowering::lowerFMINIMUMNUM_FMAXIMUMNUM(
SDValue Op,
8938 EVT VT =
Op.getValueType();
8940 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
8941 bool IsIEEEMode =
Info->getMode().IEEE;
8943 if (IsIEEEMode && !Subtarget->hasIEEEMinimumMaximumInsts())
8946 if (VT == MVT::v4f16 || VT == MVT::v8f16 || VT == MVT::v16f16 ||
8947 VT == MVT::v32f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16 ||
8948 VT == MVT::v16bf16 || VT == MVT::v32bf16 || VT == MVT::v4f64 ||
8949 VT == MVT::v8f64 || VT == MVT::v16f64 || VT == MVT::v32f64)
8956 EVT VT =
Op.getValueType();
8959 SDValue
Exp =
Op.getOperand(IsStrict ? 2 : 1);
8960 EVT ExpVT =
Exp.getValueType();
8961 if (ExpVT == MVT::i16)
8982 {
Op.getOperand(0),
Op.getOperand(1), TruncExp});
8989 switch (
Op->getOpcode()) {
9022SITargetLowering::promoteUniformUnaryOpToI32(
SDValue Op,
9023 DAGCombinerInfo &DCI)
const {
9024 EVT OpTy =
Op.getValueType();
9025 SelectionDAG &DAG = DCI.DAG;
9032 SDValue Input =
Op.getOperand(0);
9034 Input = DAG.
getNode(ExtOp,
DL, ExtTy, Input);
9036 SDValue NewVal = DAG.
getNode(
Op.getOpcode(),
DL, ExtTy, Input);
9042 DAGCombinerInfo &DCI)
const {
9043 const unsigned Opc =
Op.getOpcode();
9052 :
Op->getOperand(0).getValueType();
9053 auto &DAG = DCI.DAG;
9056 if (DCI.isBeforeLegalizeOps() ||
9064 LHS =
Op->getOperand(1);
9065 RHS =
Op->getOperand(2);
9067 LHS =
Op->getOperand(0);
9068 RHS =
Op->getOperand(1);
9103 SDValue Mag =
Op.getOperand(0);
9109 SDValue Sign =
Op.getOperand(1);
9112 if (MagVT == SignVT)
9122 SDValue SignShifted =
9133 EVT VT =
Op.getValueType();
9139 assert(VT == MVT::i64 &&
"The following code is a special for s_mul_u64");
9166 if (
Op->isDivergent())
9169 SDValue Op0 =
Op.getOperand(0);
9170 SDValue Op1 =
Op.getOperand(1);
9179 if (Op0LeadingZeros >= 32 && Op1LeadingZeros >= 32)
9181 DAG.
getMachineNode(AMDGPU::S_MUL_U64_U32_PSEUDO, SL, VT, Op0, Op1), 0);
9184 if (Op0SignBits >= 33 && Op1SignBits >= 33)
9186 DAG.
getMachineNode(AMDGPU::S_MUL_I64_I32_PSEUDO, SL, VT, Op0, Op1), 0);
9192 EVT VT =
Op.getValueType();
9194 SDValue
LHS =
Op.getOperand(0);
9195 SDValue
RHS =
Op.getOperand(1);
9199 const APInt &
C = RHSC->getAPIntValue();
9201 if (
C.isPowerOf2()) {
9203 bool UseArithShift =
isSigned && !
C.isMinSignedValue();
9204 SDValue ShiftAmt = DAG.
getConstant(
C.logBase2(), SL, MVT::i32);
9230 if (
Op->isDivergent()) {
9234 if (Subtarget->hasSMulHi()) {
9245 if (!Subtarget->hasTrapHandler() ||
9247 return lowerTrapEndpgm(
Op, DAG);
9249 return Subtarget->supportsGetDoorbellID() ? lowerTrapHsa(
Op, DAG)
9250 : lowerTrapHsaQueuePtr(
Op, DAG);
9255 SDValue Chain =
Op.getOperand(0);
9256 return DAG.
getNode(AMDGPUISD::ENDPGM_TRAP, SL, MVT::Other, Chain);
9260SITargetLowering::loadImplicitKernelArgument(
SelectionDAG &DAG,
MVT VT,
9262 ImplicitParameter Param)
const {
9266 MachinePointerInfo PtrInfo =
9276 SDValue Chain =
Op.getOperand(0);
9283 loadImplicitKernelArgument(DAG, MVT::i64, SL,
Align(8),
QUEUE_PTR);
9286 SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
9289 if (UserSGPR == AMDGPU::NoRegister) {
9300 SDValue SGPR01 = DAG.
getRegister(AMDGPU::SGPR0_SGPR1, MVT::i64);
9301 SDValue ToReg = DAG.
getCopyToReg(Chain, SL, SGPR01, QueuePtr, SDValue());
9306 return DAG.
getNode(AMDGPUISD::TRAP, SL, MVT::Other,
Ops);
9311 SDValue Chain =
Op.getOperand(0);
9315 if (Subtarget->hasPrivEnabledTrap2NopBug())
9316 return DAG.
getNode(AMDGPUISD::SIMULATED_TRAP, SL, MVT::Other, Chain);
9320 return DAG.
getNode(AMDGPUISD::TRAP, SL, MVT::Other,
Ops);
9325 SDValue Chain =
Op.getOperand(0);
9328 if (!Subtarget->hasTrapHandler() ||
9332 "debugtrap handler not supported",
9340 return DAG.
getNode(AMDGPUISD::TRAP, SL, MVT::Other,
Ops);
9350 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
9351 SmallSet<Register, 8> SGPRInputRegs;
9353 unsigned NumVals = 0;
9356 const InlineAsm::Flag
Flags(
Op.getConstantOperandVal(
I));
9357 NumVals =
Flags.getNumOperandRegisters();
9361 NumVals > 0 &&
Flags.hasRegClassConstraint(RCID) &&
9362 TRI->isSGPRClass(
TRI->getRegClass(RCID));
9364 for (
unsigned J = 0; J < NumVals; ++J) {
9365 SDValue Val =
Op.getOperand(
I + 1 + J);
9366 if (
const RegisterSDNode *RegNode =
9375 if (SGPRInputRegs.
empty())
9380 SDNode *
N =
Op.getOperand(
NumOps - 1).getNode();
9384 SDValue SrcVal =
N->getOperand(2);
9388 SDValue ReadFirstLaneID =
9390 SDValue ReadFirstLane =
9392 ReadFirstLaneID, SrcVal);
9396 if (
N->getNumOperands() > 3)
9397 Ops.push_back(
N->getOperand(3));
9403 SDNode *
Next =
nullptr;
9404 for (
unsigned I = 0,
E =
N->getNumOperands();
I !=
E; ++
I) {
9405 if (
N->getOperand(
I).getValueType() == MVT::Glue) {
9406 Next =
N->getOperand(
I).getNode();
9416SDValue SITargetLowering::getSegmentAperture(
unsigned AS,
const SDLoc &
DL,
9418 unsigned BaseAS = AS;
9423 SDValue Aperture = getBaseSegmentAperture(BaseAS,
DL, DAG);
9433SDValue SITargetLowering::getBaseSegmentAperture(
unsigned AS,
const SDLoc &
DL,
9437 if (Subtarget->hasApertureRegs()) {
9438 const unsigned ApertureRegNo =
9439 IsLDS ? AMDGPU::SRC_SHARED_BASE : AMDGPU::SRC_PRIVATE_BASE;
9440 assert((ApertureRegNo != AMDGPU::SRC_PRIVATE_BASE ||
9441 !Subtarget->hasGloballyAddressableScratch()) &&
9442 "Cannot use src_private_base with globally addressable scratch!");
9462 return loadImplicitKernelArgument(DAG, MVT::i32,
DL,
Align(4), Param);
9466 SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
9468 if (UserSGPR == AMDGPU::NoRegister) {
9479 uint32_t StructOffset = IsLDS ? 0x40 : 0x44;
9513 const AMDGPUTargetMachine &TM =
9517 unsigned SrcAS = ASC->getSrcAddressSpace();
9518 SDValue Src = ASC->getOperand(0);
9519 unsigned DestAS = ASC->getDestAddressSpace();
9520 bool IsNonNull = ASC->getFlags().hasNonNull();
9522 SDValue FlatNullPtr = DAG.
getConstant(0, SL, MVT::i64);
9531 Subtarget->hasGloballyAddressableScratch()) {
9534 SDValue FlatScratchBaseLo(
9536 AMDGPU::S_MOV_B32, SL, MVT::i32,
9537 DAG.
getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_LO, MVT::i32)),
9546 SDValue SegmentNullPtr = DAG.
getConstant(NullVal, SL, MVT::i32);
9560 Subtarget->hasGloballyAddressableScratch()) {
9564 SDValue ThreadID = DAG.
getConstant(0, SL, MVT::i32);
9569 if (Subtarget->isWave64())
9575 57 - 32 - Subtarget->getWavefrontSizeLog2(), MVT::i32, SL);
9581 SDValue FlatScratchBase = {
9583 AMDGPU::S_MOV_B64, SL, MVT::i64,
9584 DAG.
getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE, MVT::i64)),
9586 CvtPtr = DAG.
getNode(
ISD::ADD, SL, MVT::i64, CvtPtr, FlatScratchBase);
9588 SDValue Aperture = getSegmentAperture(SrcAS, SL, DAG);
9598 SDValue SegmentNullPtr = DAG.
getConstant(NullVal, SL, MVT::i32);
9609 Op.getValueType() == MVT::i64) {
9610 const SIMachineFunctionInfo *
Info =
9612 if (
Info->get32BitAddressHighBits() == 0)
9621 Src.getValueType() == MVT::i64)
9637 SDValue Vec =
Op.getOperand(0);
9638 SDValue Ins =
Op.getOperand(1);
9639 SDValue
Idx =
Op.getOperand(2);
9644 unsigned IdxVal =
Idx->getAsZExtVal();
9649 assert(InsNumElts % 2 == 0 &&
"expect legal vector types");
9654 EVT NewInsVT = InsNumElts == 2 ? MVT::i32
9656 MVT::i32, InsNumElts / 2);
9661 for (
unsigned I = 0;
I != InsNumElts / 2; ++
I) {
9663 if (InsNumElts == 2) {
9676 for (
unsigned I = 0;
I != InsNumElts; ++
I) {
9687 SDValue Vec =
Op.getOperand(0);
9688 SDValue InsVal =
Op.getOperand(1);
9689 SDValue
Idx =
Op.getOperand(2);
9699 if (NumElts == 4 && EltSize == 16 && KIdx) {
9710 unsigned Idx = KIdx->getZExtValue();
9711 bool InsertLo =
Idx < 2;
9712 SDValue InsHalf = DAG.
getNode(
9715 DAG.
getConstant(InsertLo ? Idx : (Idx - 2), SL, MVT::i32));
9721 : DAG.getBuildVector(MVT::v2i32, SL, {LoHalf, InsHalf});
9734 assert(VecSize <= 64 &&
"Expected target vector size to be <= 64 bits");
9742 SDValue ScaledIdx = DAG.
getNode(
ISD::SHL, SL, MVT::i32, Idx, ScaleFactor);
9769 EVT ResultVT =
Op.getValueType();
9770 SDValue Vec =
Op.getOperand(0);
9771 SDValue
Idx =
Op.getOperand(1);
9782 if (SDValue Combined = performExtractVectorEltCombine(
Op.getNode(), DCI))
9785 if (VecSize == 128 || VecSize == 256 || VecSize == 512) {
9789 if (VecSize == 128) {
9790 SDValue V2 = DAG.
getBitcast(MVT::v2i64, Vec);
9797 }
else if (VecSize == 256) {
9798 SDValue V2 = DAG.
getBitcast(MVT::v4i64, Vec);
9800 for (
unsigned P = 0;
P < 4; ++
P) {
9806 Parts[0], Parts[1]));
9808 Parts[2], Parts[3]));
9812 SDValue V2 = DAG.
getBitcast(MVT::v8i64, Vec);
9814 for (
unsigned P = 0;
P < 8; ++
P) {
9821 Parts[0], Parts[1], Parts[2], Parts[3]));
9824 Parts[4], Parts[5], Parts[6], Parts[7]));
9827 EVT IdxVT =
Idx.getValueType();
9830 SDValue IdxMask = DAG.
getConstant(NElem / 2 - 1, SL, IdxVT);
9844 Src = DAG.
getBitcast(Src.getValueType().changeTypeToInteger(), Src);
9854 SDValue ScaledIdx = DAG.
getNode(
ISD::SHL, SL, MVT::i32, Idx, ScaleFactor);
9859 if (ResultVT == MVT::f16 || ResultVT == MVT::bf16) {
9869 return Mask[Elt + 1] == Mask[Elt] + 1 && (Mask[Elt] % 2 == 0);
9874 return Mask[Elt] >= 0 && Mask[Elt + 1] >= 0 && (Mask[Elt] & 1) &&
9875 !(Mask[Elt + 1] & 1);
9881 EVT ResultVT =
Op.getValueType();
9884 const int NewSrcNumElts = 2;
9886 int SrcNumElts =
Op.getOperand(0).getValueType().getVectorNumElements();
9902 const bool ShouldUseConsecutiveExtract = EltVT.
getSizeInBits() == 16;
9924 if (ShouldUseConsecutiveExtract &&
9927 int VecIdx =
Idx < SrcNumElts ? 0 : 1;
9928 int EltIdx =
Idx < SrcNumElts ?
Idx :
Idx - SrcNumElts;
9939 SDValue SrcOp1 = SrcOp0;
9940 if (Idx0 >= SrcNumElts) {
9945 if (Idx1 >= SrcNumElts) {
9950 int AlignedIdx0 = Idx0 & ~(NewSrcNumElts - 1);
9951 int AlignedIdx1 = Idx1 & ~(NewSrcNumElts - 1);
9959 int NewMaskIdx0 = Idx0 - AlignedIdx0;
9960 int NewMaskIdx1 = Idx1 - AlignedIdx1;
9962 SDValue Result0 = SubVec0;
9963 SDValue Result1 = SubVec0;
9965 if (SubVec0 != SubVec1) {
9966 NewMaskIdx1 += NewSrcNumElts;
9973 {NewMaskIdx0, NewMaskIdx1});
9978 int VecIdx0 = Idx0 < SrcNumElts ? 0 : 1;
9979 int VecIdx1 = Idx1 < SrcNumElts ? 0 : 1;
9980 int EltIdx0 = Idx0 < SrcNumElts ? Idx0 : Idx0 - SrcNumElts;
9981 int EltIdx1 = Idx1 < SrcNumElts ? Idx1 : Idx1 - SrcNumElts;
9999 SDValue SVal =
Op.getOperand(0);
10000 EVT ResultVT =
Op.getValueType();
10002 SDValue UndefVal = DAG.
getPOISON(SValVT);
10016 EVT VT =
Op.getValueType();
10018 if (VT == MVT::v2f16 || VT == MVT::v2i16 || VT == MVT::v2bf16) {
10019 assert(!Subtarget->hasVOP3PInsts() &&
"this should be legal");
10021 SDValue
Lo =
Op.getOperand(0);
10022 SDValue
Hi =
Op.getOperand(1);
10025 if (
Hi.isUndef()) {
10053 for (
unsigned P = 0;
P < NumParts; ++
P) {
10055 PartVT, SL, {
Op.getOperand(
P * 2),
Op.getOperand(
P * 2 + 1)});
10081 if (!Subtarget->isAmdHsaOS())
10124 return DAG.
getNode(AMDGPUISD::PC_ADD_REL_OFFSET64,
DL, PtrVT, Ptr);
10133 return DAG.
getNode(AMDGPUISD::PC_ADD_REL_OFFSET,
DL, PtrVT, PtrLo, PtrHi);
10141 EVT PtrVT =
Op.getValueType();
10143 const GlobalValue *GV = GSD->
getGlobal();
10156 assert(PtrVT == MVT::i32 &&
"32-bit pointer is expected.");
10171 return SDValue(DAG.
getMachineNode(AMDGPU::S_MOV_B32,
DL, MVT::i32, GA), 0);
10177 return DAG.
getNode(AMDGPUISD::LDS,
DL, MVT::i32, GA);
10180 if (Subtarget->isAmdPalOS() || Subtarget->isMesa3DOS()) {
10181 if (Subtarget->has64BitLiterals()) {
10190 AddrLo = {DAG.
getMachineNode(AMDGPU::S_MOV_B32,
DL, MVT::i32, AddrLo), 0};
10194 AddrHi = {DAG.
getMachineNode(AMDGPU::S_MOV_B32,
DL, MVT::i32, AddrHi), 0};
10212 MachinePointerInfo PtrInfo =
10225 Fn,
"unsupported external symbol",
Op.getDebugLoc()));
10247 unsigned Offset)
const {
10249 SDValue Param = lowerKernargMemParameter(
10260 "non-hsa intrinsic with hsa target",
DL.getDebugLoc()));
10268 "intrinsic not supported on subtarget",
DL.getDebugLoc()));
10276 unsigned NumElts = Elts.
size();
10278 if (NumElts <= 12) {
10282 Type = MVT::v16f32;
10287 for (
unsigned i = 0; i < Elts.
size(); ++i) {
10293 for (
unsigned i = Elts.
size(); i < NumElts; ++i)
10302 SDValue Src,
int ExtraElts) {
10303 EVT SrcVT = Src.getValueType();
10313 while (ExtraElts--)
10324 bool Unpacked,
bool IsD16,
int DMaskPop,
10325 int NumVDataDwords,
bool IsAtomicPacked16Bit,
10329 EVT ReqRetVT = ResultTypes[0];
10331 int NumDataDwords = ((IsD16 && !Unpacked) || IsAtomicPacked16Bit)
10332 ? (ReqRetNumElts + 1) / 2
10335 int MaskPopDwords = (!IsD16 || Unpacked) ? DMaskPop : (DMaskPop + 1) / 2;
10338 NumDataDwords == 1 ? MVT::i32 :
MVT::getVectorVT(MVT::i32, NumDataDwords);
10341 MaskPopDwords == 1 ? MVT::i32 :
MVT::getVectorVT(MVT::i32, MaskPopDwords);
10346 if (DMaskPop > 0 &&
Data.getValueType() != MaskPopVT) {
10350 SDValue(Result, 0), ZeroIdx);
10353 SDValue(Result, 0), ZeroIdx);
10357 if (DataDwordVT.
isVector() && !IsAtomicPacked16Bit)
10359 NumDataDwords - MaskPopDwords);
10364 EVT LegalReqRetVT = ReqRetVT;
10366 if (!
Data.getValueType().isInteger())
10368 Data.getValueType().changeTypeToInteger(),
Data);
10389 if (Result->getNumValues() == 1)
10396 SDValue *LWE,
bool &IsTexFail) {
10416 unsigned DimIdx,
unsigned EndIdx,
10417 unsigned NumGradients) {
10419 for (
unsigned I = DimIdx;
I < EndIdx;
I++) {
10427 if (((
I + 1) >= EndIdx) ||
10428 ((NumGradients / 2) % 2 == 1 && (
I == DimIdx + (NumGradients / 2) - 1 ||
10429 I == DimIdx + NumGradients - 1))) {
10461 !
Op.getNode()->hasAnyUseOfValue(0))
10463 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
10474 ResultTypes.erase(&ResultTypes[0]);
10476 bool IsD16 =
false;
10477 bool IsG16 =
false;
10478 bool IsA16 =
false;
10480 int NumVDataDwords = 0;
10481 bool AdjustRetType =
false;
10482 bool IsAtomicPacked16Bit =
false;
10485 const unsigned ArgOffset = WithChain ? 2 : 1;
10488 unsigned DMaskLanes = 0;
10490 if (BaseOpcode->
Atomic) {
10491 VData =
Op.getOperand(2);
10493 IsAtomicPacked16Bit =
10494 (IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_F16 ||
10495 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_F16_NORTN ||
10496 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_BF16 ||
10497 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_BF16_NORTN);
10502 "unsupported image atomic data type");
10507 SDValue VData2 =
Op.getOperand(3);
10514 ResultTypes[0] = Is64Bit ? MVT::v2i64 : MVT::v2i32;
10516 DMask = Is64Bit ? 0xf : 0x3;
10517 NumVDataDwords = Is64Bit ? 4 : 2;
10519 DMask = Is64Bit ? 0x3 : 0x1;
10520 NumVDataDwords = Is64Bit ? 2 : 1;
10523 DMask =
Op->getConstantOperandVal(ArgOffset + Intr->
DMaskIndex);
10526 if (BaseOpcode->
Store) {
10527 VData =
Op.getOperand(2);
10531 if (StoreScalarVT != MVT::f16 && StoreScalarVT.
getSizeInBits() != 32 &&
10534 "unsupported image store data type");
10536 if (StoreScalarVT == MVT::f16) {
10537 if (!Subtarget->hasD16Images() || !BaseOpcode->
HasD16)
10541 VData = handleD16VData(VData, DAG,
true);
10544 NumVDataDwords = (VData.
getValueType().getSizeInBits() + 31) / 32;
10545 }
else if (!BaseOpcode->
NoReturn) {
10550 if (LoadScalarVT != MVT::f16 && LoadScalarVT.
getSizeInBits() != 32 &&
10553 "unsupported image load data type");
10555 if (LoadScalarVT == MVT::f16) {
10556 if (!Subtarget->hasD16Images() || !BaseOpcode->
HasD16)
10564 (!LoadVT.
isVector() && DMaskLanes > 1))
10570 if (IsD16 && !Subtarget->hasUnpackedD16VMem() &&
10571 !(BaseOpcode->
Gather4 && Subtarget->hasImageGather4D16Bug()))
10572 NumVDataDwords = (DMaskLanes + 1) / 2;
10574 NumVDataDwords = DMaskLanes;
10576 AdjustRetType =
true;
10580 unsigned VAddrEnd = ArgOffset + Intr->
VAddrEnd;
10587 MVT GradPackVectorVT = VAddrScalarVT == MVT::f16 ? MVT::v2f16 : MVT::v2i16;
10588 IsG16 = VAddrScalarVT == MVT::f16 || VAddrScalarVT == MVT::i16;
10590 VAddrVT =
Op.getOperand(ArgOffset + Intr->
CoordStart).getSimpleValueType();
10592 MVT AddrPackVectorVT = VAddrScalarVT == MVT::f16 ? MVT::v2f16 : MVT::v2i16;
10593 IsA16 = VAddrScalarVT == MVT::f16 || VAddrScalarVT == MVT::i16;
10597 if (IsA16 && (
Op.getOperand(ArgOffset +
I).getValueType() == MVT::f16)) {
10603 {
Op.getOperand(ArgOffset +
I), DAG.
getPOISON(MVT::f16)});
10607 "Bias needs to be converted to 16 bit in A16 mode");
10612 if (BaseOpcode->
Gradients && !
ST->hasG16() && (IsA16 != IsG16)) {
10616 dbgs() <<
"Failed to lower image intrinsic: 16 bit addresses "
10617 "require 16 bit args for both gradients and addresses");
10622 if (!
ST->hasA16()) {
10623 LLVM_DEBUG(
dbgs() <<
"Failed to lower image intrinsic: Target does not "
10624 "support 16 bit addresses\n");
10634 if (BaseOpcode->
Gradients && IsG16 &&
ST->hasG16()) {
10636 const AMDGPU::MIMGG16MappingInfo *G16MappingInfo =
10638 IntrOpcode = G16MappingInfo->
G16;
10661 for (
unsigned I = ArgOffset + Intr->
CoordStart;
I < VAddrEnd;
I++)
10679 const unsigned NSAMaxSize =
ST->getNSAMaxSize(BaseOpcode->
Sampler);
10680 const bool HasPartialNSAEncoding =
ST->hasPartialNSAEncoding();
10681 const bool UseNSA =
ST->hasNSAEncoding() &&
10682 VAddrs.
size() >=
ST->getNSAThreshold(MF) &&
10683 (VAddrs.
size() <= NSAMaxSize || HasPartialNSAEncoding);
10684 const bool UsePartialNSA =
10685 UseNSA && HasPartialNSAEncoding && VAddrs.
size() > NSAMaxSize;
10688 if (UsePartialNSA) {
10690 ArrayRef(VAddrs).drop_front(NSAMaxSize - 1));
10691 }
else if (!UseNSA) {
10702 Op.getConstantOperandVal(ArgOffset + Intr->
UnormIndex);
10710 bool IsTexFail =
false;
10711 if (!
parseTexFail(TexFail, DAG, &TFE, &LWE, IsTexFail))
10720 NumVDataDwords = 1;
10722 NumVDataDwords += 1;
10723 AdjustRetType =
true;
10728 if (AdjustRetType) {
10731 if (DMaskLanes == 0 && !BaseOpcode->
Store) {
10740 MVT::i32, NumVDataDwords)
10743 ResultTypes[0] = NewVT;
10744 if (ResultTypes.size() == 3) {
10748 ResultTypes.erase(&ResultTypes[1]);
10762 Ops.push_back(VData);
10763 if (UsePartialNSA) {
10765 Ops.push_back(VAddr);
10769 Ops.push_back(VAddr);
10770 SDValue Rsrc =
Op.getOperand(ArgOffset + Intr->
RsrcIndex);
10772 if (RsrcVT != MVT::v4i32 && RsrcVT != MVT::v8i32)
10774 Ops.push_back(Rsrc);
10776 SDValue Samp =
Op.getOperand(ArgOffset + Intr->
SampIndex);
10779 Ops.push_back(Samp);
10784 if (!IsGFX12Plus || BaseOpcode->
Sampler || BaseOpcode->
MSAA)
10785 Ops.push_back(Unorm);
10787 Ops.push_back(IsA16 &&
10788 ST->hasFeature(AMDGPU::FeatureR128A16)
10794 if (!Subtarget->hasGFX90AInsts())
10795 Ops.push_back(TFE);
10799 "TFE is not supported on this GPU",
DL.getDebugLoc()));
10802 if (!IsGFX12Plus || BaseOpcode->
Sampler || BaseOpcode->
MSAA)
10803 Ops.push_back(LWE);
10809 Ops.push_back(
Op.getOperand(0));
10811 int NumVAddrDwords =
10817 NumVDataDwords, NumVAddrDwords);
10818 }
else if (IsGFX12Plus) {
10820 NumVDataDwords, NumVAddrDwords);
10821 }
else if (IsGFX11Plus) {
10823 UseNSA ? AMDGPU::MIMGEncGfx11NSA
10824 : AMDGPU::MIMGEncGfx11Default,
10825 NumVDataDwords, NumVAddrDwords);
10826 }
else if (IsGFX10Plus) {
10828 UseNSA ? AMDGPU::MIMGEncGfx10NSA
10829 : AMDGPU::MIMGEncGfx10Default,
10830 NumVDataDwords, NumVAddrDwords);
10832 if (Subtarget->hasGFX90AInsts()) {
10834 NumVDataDwords, NumVAddrDwords);
10835 if (Opcode == -1) {
10837 DAG,
Op, OrigResultTypes,
DL,
10838 "requested image instruction is not supported on this GPU");
10841 if (Opcode == -1 &&
10844 NumVDataDwords, NumVAddrDwords);
10847 NumVDataDwords, NumVAddrDwords);
10854 MachineMemOperand *MemRef = MemOp->getMemOperand();
10861 {DAG.
getPOISON(OrigResultTypes[0]), SDValue(NewNode, 0)},
DL);
10863 return SDValue(NewNode, 0);
10873 Subtarget->hasUnpackedD16VMem(), IsD16, DMaskLanes,
10874 NumVDataDwords, IsAtomicPacked16Bit,
DL);
10883 bool HasChainResult = MMO !=
nullptr;
10887 bool IsSubwordLoad = (MemVT == MVT::i8 || MemVT == MVT::i16) &&
10888 Subtarget->hasScalarSubwordLoads();
10891 MF.
getFunction(),
"unsupported s_buffer_load result type",
10892 DL.getDebugLoc()));
10893 EVT ResultTypes[] = {VT, MVT::Other};
10895 ArrayRef(ResultTypes, HasChainResult ? 2 : 1), Chain,
DL);
10898 if (!HasChainResult) {
10910 if (!
Offset->isDivergent()) {
10911 SDValue
Ops[] = {Chain, Rsrc,
Offset, CachePolicy};
10918 auto HandleScalarSubwordLoads = [&](
unsigned Opcode) -> SDValue {
10920 Opcode,
DL, DAG.
getVTList(MVT::i32, MVT::Other),
Ops, MemVT, MMO);
10923 if (HasChainResult)
10927 if (MemVT == MVT::i8 && Subtarget->hasScalarSubwordLoads())
10928 return HandleScalarSubwordLoads(AMDGPUISD::SBUFFER_LOAD_UBYTE);
10930 if (MemVT == MVT::i16 && Subtarget->hasScalarSubwordLoads())
10931 return HandleScalarSubwordLoads(AMDGPUISD::SBUFFER_LOAD_USHORT);
10936 !Subtarget->hasScalarDwordx3Loads()) {
10940 AMDGPUISD::SBUFFER_LOAD,
DL, DAG.
getVTList(WidenedVT, MVT::Other),
10945 if (HasChainResult)
10967 if ((MemVT == MVT::i8 || MemVT == MVT::i16) &&
10968 Subtarget->hasScalarSubwordLoads()) {
10970 SDValue
Load = handleByteShortBufferLoads(DAG, MemVT,
DL,
Ops, MMO);
10972 if (HasChainResult)
10978 unsigned NumLoads = 1;
10984 if (NumElts == 8 || NumElts == 16) {
10985 NumLoads = NumElts / 4;
10989 SDVTList VTList = DAG.
getVTList({LoadVT, MVT::Other});
10994 NumLoads > 1 ?
Align(16 * NumLoads) :
Align(4));
10998 for (
unsigned i = 0; i < NumLoads; ++i) {
11001 Loads.
push_back(getMemIntrinsicNode(AMDGPUISD::BUFFER_LOAD,
DL, VTList,
Ops,
11002 LoadVT, LoadMMO, DAG));
11005 if (NumElts == 8 || NumElts == 16) {
11007 if (HasChainResult) {
11009 for (SDValue
Load : Loads)
11022 if (!Subtarget->hasArchitectedSGPRs())
11027 return DAG.
getNode(AMDGPUISD::BFE_U32, SL, VT, TTMP8,
11034 unsigned Width)
const {
11036 using namespace AMDGPU::Hwreg;
11038 AMDGPU::S_GETREG_B32_const, SL, MVT::i32,
11058 SDValue Val =
loadInputValue(DAG, &AMDGPU::VGPR_32RegClass, MVT::i32,
11077 SDValue Src =
Op.getOperand(0);
11078 EVT DstVT =
Op.getValueType();
11080 assert((!IsF16 || Subtarget->hasFP8F16ConversionInsts()) &&
11081 "fp8/bf8 -> f16 conversion requires FP8F16ConversionInsts");
11085 Opc = IsBF8 ? AMDGPUISD::CVT_PK_F16_BF8 : AMDGPUISD::CVT_PK_F16_FP8;
11087 Opc = IsBF8 ? AMDGPUISD::CVT_PK_F32_BF8 : AMDGPUISD::CVT_PK_F32_FP8;
11100SITargetLowering::LowerCONVERT_FROM_ARBITRARY_FP(
SDValue Op,
11109 const bool HasE5M3ConversionInsts =
11110 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
11111 const bool IsSupported = IsFP8 || IsBF8 || (IsE5M3 && HasE5M3ConversionInsts);
11115 EVT DstVT =
Op.getValueType();
11119 !Subtarget->hasFP8F16ConversionInsts())
11127 SDValue Src =
Op.getOperand(0);
11129 "only the v2f32 vector result is custom lowered");
11135 auto ConvertByte = [&](
unsigned ByteSel) {
11136 return DAG.
getNode(AMDGPUISD::CVT_F32_FP8_E5M3, SL, MVT::f32, Src,
11141 return ConvertByte(0);
11142 return DAG.
getBuildVector(DstVT, SL, {ConvertByte(0), ConvertByte(1)});
11146 SDValue Src =
Op.getOperand(0);
11147 if (Src.getValueType() != MVT::i32) {
11157 if (EltVT == MVT::f16 || EltVT == MVT::f32)
11158 return lowerFromFP8(
Op, IsBF8, DAG);
11165 SDValue Src =
Op.getOperand(0);
11166 EVT ResVT =
Op.getValueType();
11167 bool IsF16 = Src.getValueType().getScalarType() == MVT::f16;
11168 assert((!IsF16 || Subtarget->hasF16FP8ConversionInsts()) &&
11169 "f16 -> fp8/bf8 conversion requires F16FP8ConversionInsts");
11171 "only the v2i8 vector result is custom lowered");
11175 IsBF8 ? AMDGPUISD::CVT_PK_BF8_F16 : AMDGPUISD::CVT_PK_FP8_F16;
11176 SDValue Bytes = DAG.
getNode(
Opc, SL, MVT::i16, Src);
11180 unsigned Opc = IsBF8 ? AMDGPUISD::CVT_PK_BF8_F32
11181 : IsE5M3 ? AMDGPUISD::CVT_PK_FP8_F32_E5M3
11182 : AMDGPUISD::CVT_PK_FP8_F32;
11183 SDValue PoisonI32 = DAG.
getPOISON(MVT::i32);
11190 DAG.
getNode(
Opc, SL, MVT::i32, Src, Src, PoisonI32, WordSel);
11202SITargetLowering::LowerCONVERT_TO_ARBITRARY_FP(
SDValue Op,
11211 const bool HasE5M3ConversionInsts =
11212 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
11213 const bool IsSupported = IsFP8 || IsBF8 || (IsE5M3 && HasE5M3ConversionInsts);
11223 if (!IsE5M3 &&
Op.getConstantOperandVal(3) != 0)
11226 EVT SrcEltVT =
Op.getOperand(0).getValueType().getScalarType();
11229 if (SrcEltVT == MVT::f32)
11230 return lowerToFP8(
Op, IsBF8, IsE5M3, DAG);
11231 if (!IsE5M3 && SrcEltVT == MVT::f16 &&
11232 Subtarget->hasF16FP8ConversionInsts()) {
11235 if (!
Op.getValueType().isVector())
11237 return lowerToFP8(
Op, IsBF8,
false, DAG);
11245 auto *MFI = MF.
getInfo<SIMachineFunctionInfo>();
11247 EVT VT =
Op.getValueType();
11249 unsigned IntrinsicID =
Op.getConstantOperandVal(0);
11253 switch (IntrinsicID) {
11254 case Intrinsic::amdgcn_wave_reduce_min:
11255 case Intrinsic::amdgcn_wave_reduce_umin:
11256 case Intrinsic::amdgcn_wave_reduce_fmin:
11257 case Intrinsic::amdgcn_wave_reduce_max:
11258 case Intrinsic::amdgcn_wave_reduce_umax:
11259 case Intrinsic::amdgcn_wave_reduce_fmax:
11260 case Intrinsic::amdgcn_wave_reduce_add:
11261 case Intrinsic::amdgcn_wave_reduce_fadd:
11262 case Intrinsic::amdgcn_wave_reduce_sub:
11263 case Intrinsic::amdgcn_wave_reduce_fsub:
11264 case Intrinsic::amdgcn_wave_reduce_and:
11265 case Intrinsic::amdgcn_wave_reduce_or:
11266 case Intrinsic::amdgcn_wave_reduce_xor: {
11267 EVT SrcVT =
Op.getOperand(1).getValueType();
11270 bool NeedsSignExt = IntrinsicID == Intrinsic::amdgcn_wave_reduce_min ||
11271 IntrinsicID == Intrinsic::amdgcn_wave_reduce_max ||
11272 IntrinsicID == Intrinsic::amdgcn_wave_reduce_add ||
11273 IntrinsicID == Intrinsic::amdgcn_wave_reduce_sub;
11277 auto SrcType = IsFPOp ? MVT::f16 : MVT::i16;
11278 auto ExtType = IsFPOp ? MVT::f32 : MVT::i32;
11279 SDValue ExtendedSrc = DAG.
getNode(ExtOpc,
DL, ExtType,
Op.getOperand(1));
11280 SDValue Strategy =
Op.getOperand(2);
11282 Op.getOperand(0), ExtendedSrc, Strategy);
11291 case Intrinsic::amdgcn_implicit_buffer_ptr: {
11294 return getPreloadedValue(DAG, *MFI, VT,
11297 case Intrinsic::amdgcn_dispatch_ptr:
11298 case Intrinsic::amdgcn_queue_ptr: {
11299 if (!Subtarget->isAmdHsaOrMesa(MF.
getFunction())) {
11301 MF.
getFunction(),
"unsupported hsa intrinsic without hsa target",
11302 DL.getDebugLoc()));
11306 auto RegID = IntrinsicID == Intrinsic::amdgcn_dispatch_ptr
11309 return getPreloadedValue(DAG, *MFI, VT, RegID);
11311 case Intrinsic::amdgcn_implicitarg_ptr: {
11313 return getImplicitArgPtr(DAG,
DL);
11314 return getPreloadedValue(DAG, *MFI, VT,
11317 case Intrinsic::amdgcn_kernarg_segment_ptr: {
11323 return getPreloadedValue(DAG, *MFI, VT,
11326 case Intrinsic::amdgcn_dispatch_id: {
11329 case Intrinsic::amdgcn_rcp:
11330 return DAG.
getNode(AMDGPUISD::RCP,
DL, VT,
Op.getOperand(1));
11331 case Intrinsic::amdgcn_rsq:
11332 return DAG.
getNode(AMDGPUISD::RSQ,
DL, VT,
Op.getOperand(1));
11333 case Intrinsic::amdgcn_rsq_legacy:
11337 case Intrinsic::amdgcn_rcp_legacy:
11340 return DAG.
getNode(AMDGPUISD::RCP_LEGACY,
DL, VT,
Op.getOperand(1));
11341 case Intrinsic::amdgcn_fma_legacy:
11342 case Intrinsic::amdgcn_sudot4:
11343 case Intrinsic::amdgcn_sudot8:
11344 case Intrinsic::amdgcn_tanh:
11346 case Intrinsic::amdgcn_rsq_clamp: {
11348 return DAG.
getNode(AMDGPUISD::RSQ_CLAMP,
DL, VT,
Op.getOperand(1));
11354 SDValue Rsq = DAG.
getNode(AMDGPUISD::RSQ,
DL, VT,
Op.getOperand(1));
11360 case Intrinsic::r600_read_ngroups_x:
11361 if (Subtarget->isAmdHsaOS())
11364 return lowerKernargMemParameter(DAG, VT, VT,
DL, DAG.
getEntryNode(),
11367 case Intrinsic::r600_read_ngroups_y:
11368 if (Subtarget->isAmdHsaOS())
11371 return lowerKernargMemParameter(DAG, VT, VT,
DL, DAG.
getEntryNode(),
11374 case Intrinsic::r600_read_ngroups_z:
11375 if (Subtarget->isAmdHsaOS())
11378 return lowerKernargMemParameter(DAG, VT, VT,
DL, DAG.
getEntryNode(),
11381 case Intrinsic::r600_read_local_size_x:
11382 if (Subtarget->isAmdHsaOS())
11385 return lowerImplicitZextParam(DAG,
Op, MVT::i16,
11387 case Intrinsic::r600_read_local_size_y:
11388 if (Subtarget->isAmdHsaOS())
11391 return lowerImplicitZextParam(DAG,
Op, MVT::i16,
11393 case Intrinsic::r600_read_local_size_z:
11394 if (Subtarget->isAmdHsaOS())
11397 return lowerImplicitZextParam(DAG,
Op, MVT::i16,
11399 case Intrinsic::amdgcn_workgroup_id_x:
11400 return lowerWorkGroupId(DAG, *MFI, VT,
11404 case Intrinsic::amdgcn_workgroup_id_y:
11405 return lowerWorkGroupId(DAG, *MFI, VT,
11409 case Intrinsic::amdgcn_workgroup_id_z:
11410 return lowerWorkGroupId(DAG, *MFI, VT,
11414 case Intrinsic::amdgcn_cluster_id_x:
11415 return Subtarget->hasClusters()
11416 ? getPreloadedValue(DAG, *MFI, VT,
11418 : DAG.getPOISON(VT);
11419 case Intrinsic::amdgcn_cluster_id_y:
11420 return Subtarget->hasClusters()
11421 ? getPreloadedValue(DAG, *MFI, VT,
11424 case Intrinsic::amdgcn_cluster_id_z:
11425 return Subtarget->hasClusters()
11426 ? getPreloadedValue(DAG, *MFI, VT,
11429 case Intrinsic::amdgcn_cluster_workgroup_id_x:
11430 return Subtarget->hasClusters()
11431 ? getPreloadedValue(
11435 case Intrinsic::amdgcn_cluster_workgroup_id_y:
11436 return Subtarget->hasClusters()
11437 ? getPreloadedValue(
11441 case Intrinsic::amdgcn_cluster_workgroup_id_z:
11442 return Subtarget->hasClusters()
11443 ? getPreloadedValue(
11447 case Intrinsic::amdgcn_cluster_workgroup_flat_id:
11448 return Subtarget->hasClusters()
11451 case Intrinsic::amdgcn_cluster_workgroup_max_id_x:
11452 return Subtarget->hasClusters()
11453 ? getPreloadedValue(
11457 case Intrinsic::amdgcn_cluster_workgroup_max_id_y:
11458 return Subtarget->hasClusters()
11459 ? getPreloadedValue(
11463 case Intrinsic::amdgcn_cluster_workgroup_max_id_z:
11464 return Subtarget->hasClusters()
11465 ? getPreloadedValue(
11469 case Intrinsic::amdgcn_cluster_workgroup_max_flat_id:
11470 return Subtarget->hasClusters()
11471 ? getPreloadedValue(
11475 case Intrinsic::amdgcn_wave_id:
11476 return lowerWaveID(DAG,
Op);
11477 case Intrinsic::amdgcn_lds_kernel_id: {
11479 return getLDSKernelId(DAG,
DL);
11480 return getPreloadedValue(DAG, *MFI, VT,
11483 case Intrinsic::amdgcn_workitem_id_x:
11484 return lowerWorkitemID(DAG,
Op, 0, MFI->getArgInfo().WorkItemIDX);
11485 case Intrinsic::amdgcn_workitem_id_y:
11486 return lowerWorkitemID(DAG,
Op, 1, MFI->getArgInfo().WorkItemIDY);
11487 case Intrinsic::amdgcn_workitem_id_z:
11488 return lowerWorkitemID(DAG,
Op, 2, MFI->getArgInfo().WorkItemIDZ);
11489 case Intrinsic::amdgcn_wavefrontsize:
11491 SDLoc(
Op), MVT::i32);
11492 case Intrinsic::amdgcn_s_buffer_load: {
11493 unsigned CPol =
Op.getConstantOperandVal(3);
11501 Op.getOperand(2),
Op.getOperand(3), DAG);
11503 case Intrinsic::amdgcn_fdiv_fast:
11504 return lowerFDIV_FAST(
Op, DAG);
11505 case Intrinsic::amdgcn_sin:
11506 return DAG.
getNode(AMDGPUISD::SIN_HW,
DL, VT,
Op.getOperand(1));
11508 case Intrinsic::amdgcn_cos:
11509 return DAG.
getNode(AMDGPUISD::COS_HW,
DL, VT,
Op.getOperand(1));
11511 case Intrinsic::amdgcn_mul_u24:
11512 return DAG.
getNode(AMDGPUISD::MUL_U24,
DL, VT,
Op.getOperand(1),
11514 case Intrinsic::amdgcn_mul_i24:
11515 return DAG.
getNode(AMDGPUISD::MUL_I24,
DL, VT,
Op.getOperand(1),
11518 case Intrinsic::amdgcn_log_clamp: {
11524 case Intrinsic::amdgcn_fract:
11525 return DAG.
getNode(AMDGPUISD::FRACT,
DL, VT,
Op.getOperand(1));
11527 case Intrinsic::amdgcn_class: {
11528 SDValue Src =
Op.getOperand(1);
11529 EVT SrcVT = Src.getValueType();
11530 bool IsLegal = SrcVT == MVT::f32 || SrcVT == MVT::f64 ||
11531 (SrcVT == MVT::f16 && Subtarget->has16BitInsts());
11535 "llvm.amdgcn.class only supports f16, f32, and f64",
11536 DL.getDebugLoc()));
11539 return DAG.
getNode(AMDGPUISD::FP_CLASS,
DL, VT, Src,
Op.getOperand(2));
11541 case Intrinsic::amdgcn_div_fmas:
11542 return DAG.
getNode(AMDGPUISD::DIV_FMAS,
DL, VT,
Op.getOperand(1),
11543 Op.getOperand(2),
Op.getOperand(3),
Op.getOperand(4));
11545 case Intrinsic::amdgcn_div_fixup:
11546 return DAG.
getNode(AMDGPUISD::DIV_FIXUP,
DL, VT,
Op.getOperand(1),
11547 Op.getOperand(2),
Op.getOperand(3));
11549 case Intrinsic::amdgcn_div_scale: {
11554 SDValue Numerator =
Op.getOperand(1);
11555 SDValue Denominator =
Op.getOperand(2);
11562 SDValue
Src0 =
Param->isAllOnes() ? Numerator : Denominator;
11564 return DAG.
getNode(AMDGPUISD::DIV_SCALE,
DL,
Op->getVTList(), Src0,
11565 Denominator, Numerator);
11567 case Intrinsic::amdgcn_ballot:
11569 case Intrinsic::amdgcn_fmed3:
11570 return DAG.
getNode(AMDGPUISD::FMED3,
DL, VT,
Op.getOperand(1),
11571 Op.getOperand(2),
Op.getOperand(3),
Op->getFlags());
11572 case Intrinsic::amdgcn_fdot2:
11573 return DAG.
getNode(AMDGPUISD::FDOT2,
DL, VT,
Op.getOperand(1),
11574 Op.getOperand(2),
Op.getOperand(3),
Op.getOperand(4));
11575 case Intrinsic::amdgcn_fmul_legacy:
11576 return DAG.
getNode(AMDGPUISD::FMUL_LEGACY,
DL, VT,
Op.getOperand(1),
11578 case Intrinsic::amdgcn_sbfe:
11579 case Intrinsic::amdgcn_ubfe:
11581 case Intrinsic::amdgcn_cvt_pkrtz:
11582 case Intrinsic::amdgcn_cvt_pknorm_i16:
11583 case Intrinsic::amdgcn_cvt_pknorm_u16:
11584 case Intrinsic::amdgcn_cvt_pk_i16:
11585 case Intrinsic::amdgcn_cvt_pk_u16: {
11587 EVT VT =
Op.getValueType();
11590 if (IntrinsicID == Intrinsic::amdgcn_cvt_pkrtz)
11591 Opcode = AMDGPUISD::CVT_PKRTZ_F16_F32;
11592 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pknorm_i16)
11593 Opcode = AMDGPUISD::CVT_PKNORM_I16_F32;
11594 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pknorm_u16)
11595 Opcode = AMDGPUISD::CVT_PKNORM_U16_F32;
11596 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pk_i16)
11597 Opcode = AMDGPUISD::CVT_PK_I16_I32;
11599 Opcode = AMDGPUISD::CVT_PK_U16_U32;
11602 return DAG.
getNode(Opcode,
DL, VT,
Op.getOperand(1),
Op.getOperand(2));
11605 DAG.
getNode(Opcode,
DL, MVT::i32,
Op.getOperand(1),
Op.getOperand(2));
11608 case Intrinsic::amdgcn_fmad_ftz:
11609 return DAG.
getNode(AMDGPUISD::FMAD_FTZ,
DL, VT,
Op.getOperand(1),
11610 Op.getOperand(2),
Op.getOperand(3));
11612 case Intrinsic::amdgcn_if_break:
11614 Op->getOperand(1),
Op->getOperand(2)),
11617 case Intrinsic::amdgcn_groupstaticsize: {
11623 const GlobalValue *GV =
11629 case Intrinsic::amdgcn_is_shared:
11630 case Intrinsic::amdgcn_is_private: {
11637 unsigned AS = (IntrinsicID == Intrinsic::amdgcn_is_shared)
11641 Subtarget->hasGloballyAddressableScratch()) {
11642 SDValue FlatScratchBaseHi(
11644 AMDGPU::S_MOV_B32,
DL, MVT::i32,
11645 DAG.
getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI, MVT::i32)),
11654 SDValue Aperture = getSegmentAperture(AS, SL, DAG);
11657 case Intrinsic::amdgcn_perm:
11658 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
Op.getOperand(1),
11659 Op.getOperand(2),
Op.getOperand(3));
11660 case Intrinsic::amdgcn_reloc_constant: {
11670 case Intrinsic::amdgcn_swmmac_f16_16x16x32_f16:
11671 case Intrinsic::amdgcn_swmmac_bf16_16x16x32_bf16:
11672 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf16:
11673 case Intrinsic::amdgcn_swmmac_f32_16x16x32_f16:
11674 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_fp8:
11675 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_bf8:
11676 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_fp8:
11677 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_bf8: {
11678 if (
Op.getOperand(4).getValueType() == MVT::i32)
11684 Op.getOperand(0),
Op.getOperand(1),
Op.getOperand(2),
11685 Op.getOperand(3), IndexKeyi32);
11687 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_fp8:
11688 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_bf8:
11689 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_fp8:
11690 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_bf8:
11691 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_fp8:
11692 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_bf8:
11693 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_fp8:
11694 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_bf8: {
11695 if (
Op.getOperand(4).getValueType() == MVT::i64)
11700 Op.getOperand(4).getValueType() == MVT::v2i32
11704 {Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11705 Op.getOperand(3), IndexKeyi64, Op.getOperand(5),
11706 Op.getOperand(6)});
11708 case Intrinsic::amdgcn_swmmac_f16_16x16x64_f16:
11709 case Intrinsic::amdgcn_swmmac_bf16_16x16x64_bf16:
11710 case Intrinsic::amdgcn_swmmac_f32_16x16x64_bf16:
11711 case Intrinsic::amdgcn_swmmac_bf16f32_16x16x64_bf16:
11712 case Intrinsic::amdgcn_swmmac_f32_16x16x64_f16:
11713 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8: {
11714 EVT IndexKeyTy = IntrinsicID == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8
11717 if (
Op.getOperand(6).getValueType() == IndexKeyTy)
11722 Op.getOperand(6).getValueType().isVector()
11726 Op.getOperand(0),
Op.getOperand(1),
Op.getOperand(2),
11727 Op.getOperand(3),
Op.getOperand(4),
Op.getOperand(5),
11728 IndexKey,
Op.getOperand(7),
Op.getOperand(8)};
11729 if (IntrinsicID == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8)
11730 Args.push_back(
Op.getOperand(9));
11733 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu4:
11734 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu8:
11735 case Intrinsic::amdgcn_swmmac_i32_16x16x64_iu4: {
11736 if (
Op.getOperand(6).getValueType() == MVT::i32)
11742 {Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11743 Op.getOperand(3), Op.getOperand(4), Op.getOperand(5),
11744 IndexKeyi32, Op.getOperand(7)});
11746 case Intrinsic::amdgcn_wmma_scale_f32_16x16x128_f8f6f4:
11747 case Intrinsic::amdgcn_wmma_scale16_f32_16x16x128_f8f6f4: {
11748 unsigned AFmt = (unsigned)
Op.getConstantOperandVal(1);
11749 unsigned BFmt = (unsigned)
Op.getConstantOperandVal(3);
11750 unsigned AScaleFmt = (unsigned)
Op.getConstantOperandVal(8);
11751 unsigned BScaleFmt = (unsigned)
Op.getConstantOperandVal(11);
11755 "invalid matrix and scale format combination in wmma call");
11761 case Intrinsic::amdgcn_readlane:
11762 case Intrinsic::amdgcn_readfirstlane:
11763 case Intrinsic::amdgcn_writelane:
11764 case Intrinsic::amdgcn_permlane16:
11765 case Intrinsic::amdgcn_permlanex16:
11766 case Intrinsic::amdgcn_permlane64:
11767 case Intrinsic::amdgcn_set_inactive:
11768 case Intrinsic::amdgcn_set_inactive_chain_arg:
11769 case Intrinsic::amdgcn_mov_dpp8:
11770 case Intrinsic::amdgcn_update_dpp:
11771 case Intrinsic::amdgcn_permlane_bcast:
11772 case Intrinsic::amdgcn_permlane_up:
11773 case Intrinsic::amdgcn_permlane_down:
11774 case Intrinsic::amdgcn_permlane_xor:
11776 case Intrinsic::amdgcn_dead: {
11778 for (
const EVT ValTy :
Op.getNode()->values())
11782 case Intrinsic::amdgcn_wave_shuffle:
11785 if (
const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
11787 return lowerImage(
Op, ImageDimIntr, DAG,
false);
11797 if (Subtarget->hasRestrictedSOffset() &&
isNullConstant(SOffset))
11798 return DAG.
getRegister(AMDGPU::SGPR_NULL, MVT::i32);
11804 unsigned NewOpcode)
const {
11807 SDValue VData =
Op.getOperand(2);
11811 "unsupported buffer atomic data type");
11813 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
11814 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(4), DAG);
11832 M->getMemOperand());
11837 unsigned NewOpcode)
const {
11840 SDValue VData =
Op.getOperand(2);
11844 "unsupported buffer atomic data type");
11846 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
11847 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(5), DAG);
11865 M->getMemOperand());
11872 unsigned NumOperands =
N->getNumOperands();
11873 if (
N->getOperand(NumOperands - 1) == Zero)
11876 Ops[NumOperands - 1] = Zero;
11882 unsigned IntrID =
Op.getConstantOperandVal(1);
11886 case Intrinsic::amdgcn_cluster_load_b32:
11887 case Intrinsic::amdgcn_cluster_load_b64:
11888 case Intrinsic::amdgcn_cluster_load_b128: {
11889 if (Subtarget->hasGFX1250_STRICT())
11893 case Intrinsic::amdgcn_ds_ordered_add:
11894 case Intrinsic::amdgcn_ds_ordered_swap: {
11896 SDValue Chain =
M->getOperand(0);
11897 SDValue
M0 =
M->getOperand(2);
11898 SDValue
Value =
M->getOperand(3);
11899 unsigned IndexOperand =
M->getConstantOperandVal(7);
11900 unsigned WaveRelease =
M->getConstantOperandVal(8);
11901 unsigned WaveDone =
M->getConstantOperandVal(9);
11903 unsigned OrderedCountIndex = IndexOperand & 0x3f;
11904 IndexOperand &= ~0x3f;
11905 unsigned CountDw = 0;
11908 CountDw = (IndexOperand >> 24) & 0xf;
11909 IndexOperand &= ~(0xf << 24);
11911 if (CountDw < 1 || CountDw > 4) {
11914 Fn,
"ds_ordered_count: dword count must be between 1 and 4",
11915 DL.getDebugLoc()));
11920 if (IndexOperand) {
11923 Fn,
"ds_ordered_count: bad index operand",
DL.getDebugLoc()));
11926 if (WaveDone && !WaveRelease) {
11930 Fn,
"ds_ordered_count: wave_done requires wave_release",
11931 DL.getDebugLoc()));
11934 unsigned Instruction = IntrID == Intrinsic::amdgcn_ds_ordered_add ? 0 : 1;
11935 unsigned ShaderType =
11937 unsigned Offset0 = OrderedCountIndex << 2;
11938 unsigned Offset1 = WaveRelease | (WaveDone << 1) | (Instruction << 4);
11941 Offset1 |= (CountDw - 1) << 6;
11944 Offset1 |= ShaderType << 2;
11946 unsigned Offset = Offset0 | (Offset1 << 8);
11953 M->getVTList(),
Ops,
M->getMemoryVT(),
11954 M->getMemOperand());
11956 case Intrinsic::amdgcn_ptr_s_buffer_load: {
11957 unsigned CPol =
Op.getConstantOperandVal(4);
11964 return lowerSBuffer(
11965 Op.getValueType(),
M->getMemoryVT(),
DL,
Op.getOperand(0),
11966 bufferRsrcPtrToVector(
Op.getOperand(2), DAG),
Op.getOperand(3),
11967 Op.getOperand(4), DAG,
M->getMemOperand());
11969 case Intrinsic::amdgcn_raw_buffer_load:
11970 case Intrinsic::amdgcn_raw_ptr_buffer_load:
11971 case Intrinsic::amdgcn_raw_atomic_buffer_load:
11972 case Intrinsic::amdgcn_raw_ptr_atomic_buffer_load:
11973 case Intrinsic::amdgcn_raw_buffer_load_format:
11974 case Intrinsic::amdgcn_raw_ptr_buffer_load_format: {
11975 const bool IsFormat =
11976 IntrID == Intrinsic::amdgcn_raw_buffer_load_format ||
11977 IntrID == Intrinsic::amdgcn_raw_ptr_buffer_load_format;
11979 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(2), DAG);
11980 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(3), DAG);
11994 return lowerIntrinsicLoad(M, IsFormat, DAG,
Ops);
11996 case Intrinsic::amdgcn_struct_buffer_load:
11997 case Intrinsic::amdgcn_struct_ptr_buffer_load:
11998 case Intrinsic::amdgcn_struct_buffer_load_format:
11999 case Intrinsic::amdgcn_struct_ptr_buffer_load_format:
12000 case Intrinsic::amdgcn_struct_atomic_buffer_load:
12001 case Intrinsic::amdgcn_struct_ptr_atomic_buffer_load: {
12002 const bool IsFormat =
12003 IntrID == Intrinsic::amdgcn_struct_buffer_load_format ||
12004 IntrID == Intrinsic::amdgcn_struct_ptr_buffer_load_format;
12006 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(2), DAG);
12007 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(4), DAG);
12022 case Intrinsic::amdgcn_raw_tbuffer_load:
12023 case Intrinsic::amdgcn_raw_ptr_tbuffer_load: {
12025 EVT LoadVT =
Op.getValueType();
12026 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(2), DAG);
12027 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(3), DAG);
12043 return adjustLoadValueType(AMDGPUISD::TBUFFER_LOAD_FORMAT_D16, M, DAG,
12045 return getMemIntrinsicNode(AMDGPUISD::TBUFFER_LOAD_FORMAT,
DL,
12046 Op->getVTList(),
Ops, LoadVT,
M->getMemOperand(),
12049 case Intrinsic::amdgcn_struct_tbuffer_load:
12050 case Intrinsic::amdgcn_struct_ptr_tbuffer_load: {
12052 EVT LoadVT =
Op.getValueType();
12053 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(2), DAG);
12054 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(4), DAG);
12070 return adjustLoadValueType(AMDGPUISD::TBUFFER_LOAD_FORMAT_D16, M, DAG,
12072 return getMemIntrinsicNode(AMDGPUISD::TBUFFER_LOAD_FORMAT,
DL,
12073 Op->getVTList(),
Ops, LoadVT,
M->getMemOperand(),
12076 case Intrinsic::amdgcn_raw_buffer_atomic_fadd:
12077 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fadd:
12078 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FADD);
12079 case Intrinsic::amdgcn_struct_buffer_atomic_fadd:
12080 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fadd:
12081 return lowerStructBufferAtomicIntrin(
Op, DAG,
12082 AMDGPUISD::BUFFER_ATOMIC_FADD);
12083 case Intrinsic::amdgcn_raw_buffer_atomic_fmin:
12084 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmin:
12085 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FMIN);
12086 case Intrinsic::amdgcn_struct_buffer_atomic_fmin:
12087 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fmin:
12088 return lowerStructBufferAtomicIntrin(
Op, DAG,
12089 AMDGPUISD::BUFFER_ATOMIC_FMIN);
12090 case Intrinsic::amdgcn_raw_buffer_atomic_fmax:
12091 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmax:
12092 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FMAX);
12093 case Intrinsic::amdgcn_struct_buffer_atomic_fmax:
12094 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fmax:
12095 return lowerStructBufferAtomicIntrin(
Op, DAG,
12096 AMDGPUISD::BUFFER_ATOMIC_FMAX);
12097 case Intrinsic::amdgcn_raw_buffer_atomic_swap:
12098 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_swap:
12099 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SWAP);
12100 case Intrinsic::amdgcn_raw_buffer_atomic_add:
12101 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_add:
12102 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_ADD);
12103 case Intrinsic::amdgcn_raw_buffer_atomic_sub:
12104 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub:
12105 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SUB);
12106 case Intrinsic::amdgcn_raw_buffer_atomic_smin:
12107 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smin:
12108 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SMIN);
12109 case Intrinsic::amdgcn_raw_buffer_atomic_umin:
12110 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umin:
12111 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_UMIN);
12112 case Intrinsic::amdgcn_raw_buffer_atomic_smax:
12113 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smax:
12114 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SMAX);
12115 case Intrinsic::amdgcn_raw_buffer_atomic_umax:
12116 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umax:
12117 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_UMAX);
12118 case Intrinsic::amdgcn_raw_buffer_atomic_and:
12119 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_and:
12120 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_AND);
12121 case Intrinsic::amdgcn_raw_buffer_atomic_or:
12122 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_or:
12123 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_OR);
12124 case Intrinsic::amdgcn_raw_buffer_atomic_xor:
12125 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_xor:
12126 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_XOR);
12127 case Intrinsic::amdgcn_raw_buffer_atomic_inc:
12128 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_inc:
12129 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_INC);
12130 case Intrinsic::amdgcn_raw_buffer_atomic_dec:
12131 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_dec:
12132 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_DEC);
12133 case Intrinsic::amdgcn_struct_buffer_atomic_swap:
12134 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_swap:
12135 return lowerStructBufferAtomicIntrin(
Op, DAG,
12136 AMDGPUISD::BUFFER_ATOMIC_SWAP);
12137 case Intrinsic::amdgcn_struct_buffer_atomic_add:
12138 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_add:
12139 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_ADD);
12140 case Intrinsic::amdgcn_struct_buffer_atomic_sub:
12141 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_sub:
12142 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SUB);
12143 case Intrinsic::amdgcn_struct_buffer_atomic_smin:
12144 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smin:
12145 return lowerStructBufferAtomicIntrin(
Op, DAG,
12146 AMDGPUISD::BUFFER_ATOMIC_SMIN);
12147 case Intrinsic::amdgcn_struct_buffer_atomic_umin:
12148 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umin:
12149 return lowerStructBufferAtomicIntrin(
Op, DAG,
12150 AMDGPUISD::BUFFER_ATOMIC_UMIN);
12151 case Intrinsic::amdgcn_struct_buffer_atomic_smax:
12152 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smax:
12153 return lowerStructBufferAtomicIntrin(
Op, DAG,
12154 AMDGPUISD::BUFFER_ATOMIC_SMAX);
12155 case Intrinsic::amdgcn_struct_buffer_atomic_umax:
12156 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umax:
12157 return lowerStructBufferAtomicIntrin(
Op, DAG,
12158 AMDGPUISD::BUFFER_ATOMIC_UMAX);
12159 case Intrinsic::amdgcn_struct_buffer_atomic_and:
12160 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_and:
12161 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_AND);
12162 case Intrinsic::amdgcn_struct_buffer_atomic_or:
12163 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_or:
12164 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_OR);
12165 case Intrinsic::amdgcn_struct_buffer_atomic_xor:
12166 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_xor:
12167 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_XOR);
12168 case Intrinsic::amdgcn_struct_buffer_atomic_inc:
12169 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_inc:
12170 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_INC);
12171 case Intrinsic::amdgcn_struct_buffer_atomic_dec:
12172 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_dec:
12173 return lowerStructBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_DEC);
12174 case Intrinsic::amdgcn_raw_buffer_atomic_sub_clamp_u32:
12175 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub_clamp_u32:
12176 return lowerRawBufferAtomicIntrin(
Op, DAG, AMDGPUISD::BUFFER_ATOMIC_CSUB);
12177 case Intrinsic::amdgcn_struct_buffer_atomic_sub_clamp_u32:
12178 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_sub_clamp_u32:
12179 return lowerStructBufferAtomicIntrin(
Op, DAG,
12180 AMDGPUISD::BUFFER_ATOMIC_CSUB);
12181 case Intrinsic::amdgcn_raw_buffer_atomic_cond_sub_u32:
12182 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_cond_sub_u32:
12183 return lowerRawBufferAtomicIntrin(
Op, DAG,
12184 AMDGPUISD::BUFFER_ATOMIC_COND_SUB_U32);
12185 case Intrinsic::amdgcn_struct_buffer_atomic_cond_sub_u32:
12186 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_cond_sub_u32:
12187 return lowerStructBufferAtomicIntrin(
Op, DAG,
12188 AMDGPUISD::BUFFER_ATOMIC_COND_SUB_U32);
12189 case Intrinsic::amdgcn_raw_buffer_atomic_cmpswap:
12190 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_cmpswap: {
12191 SDValue Src =
Op.getOperand(2);
12192 if (Src.getValueSizeInBits() != 32 && Src.getValueSizeInBits() != 64) {
12195 "unsupported buffer atomic data type");
12197 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(4), DAG);
12198 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(5), DAG);
12212 EVT VT =
Op.getValueType();
12216 Op->getVTList(),
Ops, VT,
12217 M->getMemOperand());
12219 case Intrinsic::amdgcn_struct_buffer_atomic_cmpswap:
12220 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_cmpswap: {
12221 SDValue Src =
Op.getOperand(2);
12222 if (Src.getValueSizeInBits() != 32 && Src.getValueSizeInBits() != 64) {
12225 "unsupported buffer atomic data type");
12227 SDValue Rsrc = bufferRsrcPtrToVector(
Op->getOperand(4), DAG);
12228 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(6), DAG);
12242 EVT VT =
Op.getValueType();
12246 Op->getVTList(),
Ops, VT,
12247 M->getMemOperand());
12249 case Intrinsic::amdgcn_image_bvh_dual_intersect_ray:
12250 case Intrinsic::amdgcn_image_bvh8_intersect_ray: {
12252 SDValue NodePtr =
M->getOperand(2);
12253 SDValue RayExtent =
M->getOperand(3);
12254 SDValue InstanceMask =
M->getOperand(4);
12255 SDValue RayOrigin =
M->getOperand(5);
12256 SDValue RayDir =
M->getOperand(6);
12257 SDValue
Offsets =
M->getOperand(7);
12258 SDValue TDescr =
M->getOperand(8);
12263 bool IsBVH8 = IntrID == Intrinsic::amdgcn_image_bvh8_intersect_ray;
12264 const unsigned NumVDataDwords = 10;
12265 const unsigned NumVAddrDwords = IsBVH8 ? 11 : 12;
12267 IsBVH8 ? AMDGPU::IMAGE_BVH8_INTERSECT_RAY
12268 : AMDGPU::IMAGE_BVH_DUAL_INTERSECT_RAY,
12269 AMDGPU::MIMGEncGfx12, NumVDataDwords, NumVAddrDwords);
12273 Ops.push_back(NodePtr);
12276 {DAG.getBitcast(MVT::i32, RayExtent),
12277 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, InstanceMask)}));
12278 Ops.push_back(RayOrigin);
12279 Ops.push_back(RayDir);
12280 Ops.push_back(Offsets);
12281 Ops.push_back(TDescr);
12282 Ops.push_back(
M->getChain());
12285 MachineMemOperand *MemRef =
M->getMemOperand();
12287 return SDValue(NewNode, 0);
12289 case Intrinsic::amdgcn_image_bvh_intersect_ray: {
12291 SDValue NodePtr =
M->getOperand(2);
12292 SDValue RayExtent =
M->getOperand(3);
12293 SDValue RayOrigin =
M->getOperand(4);
12294 SDValue RayDir =
M->getOperand(5);
12295 SDValue RayInvDir =
M->getOperand(6);
12296 SDValue TDescr =
M->getOperand(7);
12308 const unsigned NumVDataDwords = 4;
12309 const unsigned NumVAddrDwords = IsA16 ? (Is64 ? 9 : 8) : (Is64 ? 12 : 11);
12310 const unsigned NumVAddrs = IsGFX11Plus ? (IsA16 ? 4 : 5) : NumVAddrDwords;
12311 const bool UseNSA = (Subtarget->hasNSAEncoding() &&
12314 const unsigned BaseOpcodes[2][2] = {
12315 {AMDGPU::IMAGE_BVH_INTERSECT_RAY, AMDGPU::IMAGE_BVH_INTERSECT_RAY_a16},
12316 {AMDGPU::IMAGE_BVH64_INTERSECT_RAY,
12317 AMDGPU::IMAGE_BVH64_INTERSECT_RAY_a16}};
12321 IsGFX12Plus ? AMDGPU::MIMGEncGfx12
12322 : IsGFX11 ? AMDGPU::MIMGEncGfx11NSA
12323 : AMDGPU::MIMGEncGfx10NSA,
12324 NumVDataDwords, NumVAddrDwords);
12328 IsGFX11 ? AMDGPU::MIMGEncGfx11Default
12329 : AMDGPU::MIMGEncGfx10Default,
12330 NumVDataDwords, NumVAddrDwords);
12336 auto packLanes = [&DAG, &
Ops, &
DL](SDValue
Op,
bool IsAligned) {
12339 if (Lanes[0].getValueSizeInBits() == 32) {
12340 for (
unsigned I = 0;
I < 3; ++
I)
12347 Ops.push_back(Lanes[2]);
12349 SDValue Elt0 =
Ops.pop_back_val();
12359 if (UseNSA && IsGFX11Plus) {
12360 Ops.push_back(NodePtr);
12362 Ops.push_back(RayOrigin);
12367 for (
unsigned I = 0;
I < 3; ++
I) {
12370 {DirLanes[I], InvDirLanes[I]})));
12374 Ops.push_back(RayDir);
12375 Ops.push_back(RayInvDir);
12382 Ops.push_back(NodePtr);
12385 packLanes(RayOrigin,
true);
12386 packLanes(RayDir,
true);
12387 packLanes(RayInvDir,
false);
12392 if (NumVAddrDwords > 12) {
12397 SDValue MergedOps =
12400 Ops.push_back(MergedOps);
12403 Ops.push_back(TDescr);
12405 Ops.push_back(
M->getChain());
12408 MachineMemOperand *MemRef =
M->getMemOperand();
12410 return SDValue(NewNode, 0);
12412 case Intrinsic::amdgcn_global_atomic_fmin_num:
12413 case Intrinsic::amdgcn_global_atomic_fmax_num:
12414 case Intrinsic::amdgcn_flat_atomic_fmin_num:
12415 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
12422 unsigned Opcode = 0;
12424 case Intrinsic::amdgcn_global_atomic_fmin_num:
12425 case Intrinsic::amdgcn_flat_atomic_fmin_num: {
12429 case Intrinsic::amdgcn_global_atomic_fmax_num:
12430 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
12437 return DAG.
getAtomic(Opcode, SDLoc(
Op),
M->getMemoryVT(),
M->getVTList(),
12438 Ops,
M->getMemOperand());
12440 case Intrinsic::amdgcn_s_alloc_vgpr: {
12445 SDValue ReadFirstLaneID =
12448 ReadFirstLaneID, NumVGPRs);
12451 Op.getOperand(0),
Op.getOperand(1), NumVGPRs);
12453 case Intrinsic::amdgcn_s_get_barrier_state:
12454 case Intrinsic::amdgcn_s_get_named_barrier_state: {
12455 SDValue Chain =
Op->getOperand(0);
12461 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state)
12462 BarID = BarID & 0x3F;
12463 Opc = AMDGPU::S_GET_BARRIER_STATE_IMM;
12466 Ops.push_back(Chain);
12468 Opc = AMDGPU::S_GET_BARRIER_STATE_M0;
12469 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state) {
12478 return SDValue(NewMI, 0);
12480 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
12481 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
12482 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B: {
12484 SDValue Chain =
Op->getOperand(0);
12485 SDValue Ptr =
Op->getOperand(2);
12486 EVT VT =
Op->getValueType(0);
12490 case Intrinsic::amdgcn_av_load_b128: {
12492 SDValue Chain =
Op->getOperand(0);
12493 SDValue Ptr =
Op->getOperand(2);
12494 EVT VT =
Op->getValueType(0);
12501 case Intrinsic::amdgcn_flat_load_monitor_b32:
12502 case Intrinsic::amdgcn_flat_load_monitor_b64:
12503 case Intrinsic::amdgcn_flat_load_monitor_b128: {
12505 SDValue Chain =
Op->getOperand(0);
12506 SDValue Ptr =
Op->getOperand(2);
12508 Op->getVTList(), {Chain, Ptr},
12511 case Intrinsic::amdgcn_global_load_monitor_b32:
12512 case Intrinsic::amdgcn_global_load_monitor_b64:
12513 case Intrinsic::amdgcn_global_load_monitor_b128: {
12515 SDValue Chain =
Op->getOperand(0);
12516 SDValue Ptr =
Op->getOperand(2);
12518 Op->getVTList(), {Chain, Ptr},
12523 if (
const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
12525 return lowerImage(
Op, ImageDimIntr, DAG,
true);
12533SDValue SITargetLowering::getMemIntrinsicNode(
unsigned Opcode,
const SDLoc &
DL,
12540 EVT VT = VTList.
VTs[0];
12543 bool IsTFE = VTList.
NumVTs == 3;
12546 unsigned NumOpDWords = NumValueDWords + 1;
12548 SDVTList OpDWordsVTList = DAG.
getVTList(OpDWordsVT, VTList.
VTs[2]);
12549 MachineMemOperand *OpDWordsMMO =
12551 SDValue
Op = getMemIntrinsicNode(Opcode,
DL, OpDWordsVTList,
Ops,
12552 OpDWordsVT, OpDWordsMMO, DAG);
12557 if (!Subtarget->hasDwordx3LoadStores() &&
12558 (VT == MVT::v3i32 || VT == MVT::v3f32)) {
12562 SDVTList WidenedVTList = DAG.
getVTList(WidenedVT, VTList.
VTs[1]);
12564 WidenedMemVT, WidenedMMO);
12574 bool ImageStore)
const {
12584 if (Subtarget->hasUnpackedD16VMem()) {
12598 if (ImageStore && Subtarget->hasImageStoreD16Bug()) {
12609 for (
unsigned I = 0;
I < Elts.
size() / 2;
I += 1) {
12615 if ((NumElements % 2) == 1) {
12617 unsigned I = Elts.
size() / 2;
12633 if (NumElements == 3) {
12652 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
12653 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
12654 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
12655 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
12656 case Intrinsic::amdgcn_load_async_to_lds:
12657 case Intrinsic::amdgcn_global_load_async_lds:
12666 SDValue Chain =
Op.getOperand(0);
12667 unsigned IntrinsicID =
Op.getConstantOperandVal(1);
12669 switch (IntrinsicID) {
12670 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
12671 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
12672 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
12673 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
12674 if (Subtarget->hasGFX1250_STRICT())
12678 case Intrinsic::amdgcn_exp_compr: {
12679 SDValue
Src0 =
Op.getOperand(4);
12680 SDValue
Src1 =
Op.getOperand(5);
12687 const SDValue
Ops[] = {
12699 unsigned Opc =
Done->isZero() ? AMDGPU::EXP : AMDGPU::EXP_DONE;
12703 case Intrinsic::amdgcn_struct_tbuffer_store:
12704 case Intrinsic::amdgcn_struct_ptr_tbuffer_store: {
12705 SDValue VData =
Op.getOperand(2);
12708 VData = handleD16VData(VData, DAG);
12709 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
12710 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(5), DAG);
12724 unsigned Opc = IsD16 ? AMDGPUISD::TBUFFER_STORE_FORMAT_D16
12725 : AMDGPUISD::TBUFFER_STORE_FORMAT;
12728 M->getMemoryVT(),
M->getMemOperand());
12731 case Intrinsic::amdgcn_raw_tbuffer_store:
12732 case Intrinsic::amdgcn_raw_ptr_tbuffer_store: {
12733 SDValue VData =
Op.getOperand(2);
12736 VData = handleD16VData(VData, DAG);
12737 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
12738 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(4), DAG);
12752 unsigned Opc = IsD16 ? AMDGPUISD::TBUFFER_STORE_FORMAT_D16
12753 : AMDGPUISD::TBUFFER_STORE_FORMAT;
12756 M->getMemoryVT(),
M->getMemOperand());
12759 case Intrinsic::amdgcn_raw_buffer_store:
12760 case Intrinsic::amdgcn_raw_ptr_buffer_store:
12761 case Intrinsic::amdgcn_raw_buffer_store_format:
12762 case Intrinsic::amdgcn_raw_ptr_buffer_store_format: {
12763 const bool IsFormat =
12764 IntrinsicID == Intrinsic::amdgcn_raw_buffer_store_format ||
12765 IntrinsicID == Intrinsic::amdgcn_raw_ptr_buffer_store_format;
12767 SDValue VData =
Op.getOperand(2);
12775 "unsupported sub-dword format buffer store",
DL.getDebugLoc()));
12780 VData = handleD16VData(VData, DAG);
12790 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
12791 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(4), DAG);
12805 IsFormat ? AMDGPUISD::BUFFER_STORE_FORMAT : AMDGPUISD::BUFFER_STORE;
12806 Opc = IsD16 ? AMDGPUISD::BUFFER_STORE_FORMAT_D16 :
Opc;
12811 return handleByteShortBufferStores(DAG, VDataVT,
DL,
Ops, M);
12814 M->getMemoryVT(),
M->getMemOperand());
12817 case Intrinsic::amdgcn_struct_buffer_store:
12818 case Intrinsic::amdgcn_struct_ptr_buffer_store:
12819 case Intrinsic::amdgcn_struct_buffer_store_format:
12820 case Intrinsic::amdgcn_struct_ptr_buffer_store_format: {
12821 const bool IsFormat =
12822 IntrinsicID == Intrinsic::amdgcn_struct_buffer_store_format ||
12823 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_store_format;
12825 SDValue VData =
Op.getOperand(2);
12833 "unsupported sub-dword format buffer store",
DL.getDebugLoc()));
12838 VData = handleD16VData(VData, DAG);
12848 auto Rsrc = bufferRsrcPtrToVector(
Op.getOperand(3), DAG);
12849 auto [VOffset,
Offset] = splitBufferOffsets(
Op.getOperand(5), DAG);
12863 !IsFormat ? AMDGPUISD::BUFFER_STORE : AMDGPUISD::BUFFER_STORE_FORMAT;
12864 Opc = IsD16 ? AMDGPUISD::BUFFER_STORE_FORMAT_D16 :
Opc;
12868 EVT VDataType = VData.getValueType().getScalarType();
12870 return handleByteShortBufferStores(DAG, VDataType,
DL,
Ops, M);
12873 M->getMemoryVT(),
M->getMemOperand());
12875 case Intrinsic::amdgcn_raw_buffer_load_lds:
12876 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
12877 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
12878 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
12879 case Intrinsic::amdgcn_struct_buffer_load_lds:
12880 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
12881 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
12882 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
12885 IntrinsicID == Intrinsic::amdgcn_struct_buffer_load_lds ||
12886 IntrinsicID == Intrinsic::amdgcn_struct_buffer_load_async_lds ||
12887 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_load_lds ||
12888 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds;
12889 unsigned OpOffset = HasVIndex ? 1 : 0;
12890 SDValue VOffset =
Op.getOperand(5 + OpOffset);
12892 unsigned Size =
Op->getConstantOperandVal(4);
12898 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_BOTHEN
12899 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_IDXEN
12900 : HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFEN
12901 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFSET;
12904 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_BOTHEN
12905 : AMDGPU::BUFFER_LOAD_USHORT_LDS_IDXEN
12906 : HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFEN
12907 : AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFSET;
12910 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_BOTHEN
12911 : AMDGPU::BUFFER_LOAD_DWORD_LDS_IDXEN
12912 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFEN
12913 : AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFSET;
12916 if (!Subtarget->hasLDSLoadB96_B128())
12918 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_BOTHEN
12919 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_IDXEN
12920 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFEN
12921 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFSET;
12924 if (!Subtarget->hasLDSLoadB96_B128())
12926 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_BOTHEN
12927 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_IDXEN
12928 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFEN
12929 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFSET;
12933 SDValue M0Val =
copyToM0(DAG, Chain,
DL,
Op.getOperand(3));
12937 if (HasVIndex && HasVOffset)
12941 else if (HasVIndex)
12942 Ops.push_back(
Op.getOperand(5));
12943 else if (HasVOffset)
12944 Ops.push_back(VOffset);
12946 SDValue Rsrc = bufferRsrcPtrToVector(
Op.getOperand(2), DAG);
12947 Ops.push_back(Rsrc);
12948 Ops.push_back(
Op.getOperand(6 + OpOffset));
12949 Ops.push_back(
Op.getOperand(7 + OpOffset));
12951 unsigned Aux =
Op.getConstantOperandVal(8 + OpOffset);
12969 return SDValue(
Load, 0);
12974 case Intrinsic::amdgcn_load_to_lds:
12975 case Intrinsic::amdgcn_load_async_to_lds:
12976 case Intrinsic::amdgcn_global_load_lds:
12977 case Intrinsic::amdgcn_global_load_async_lds: {
12978 if (!Subtarget->hasVMemToLDSLoad())
12982 unsigned Size =
Op->getConstantOperandVal(4);
12987 Opc = AMDGPU::GLOBAL_LOAD_LDS_UBYTE;
12990 Opc = AMDGPU::GLOBAL_LOAD_LDS_USHORT;
12993 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORD;
12996 if (!Subtarget->hasLDSLoadB96_B128())
12998 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX3;
13001 if (!Subtarget->hasLDSLoadB96_B128())
13003 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX4;
13007 SDValue M0Val =
copyToM0(DAG, Chain,
DL,
Op.getOperand(3));
13011 SDValue Addr =
Op.getOperand(2);
13019 if (
LHS->isDivergent())
13023 RHS.getOperand(0).getValueType() == MVT::i32) {
13026 VOffset =
RHS.getOperand(0);
13030 Ops.push_back(Addr);
13038 Ops.push_back(VOffset);
13041 Ops.push_back(
Op.getOperand(5));
13043 unsigned Aux =
Op.getConstantOperandVal(6);
13056 return SDValue(
Load, 0);
13058 case Intrinsic::amdgcn_end_cf:
13060 Op->getOperand(2), Chain),
13062 case Intrinsic::amdgcn_s_barrier_signal_var: {
13067 SDValue CntOp =
Op->getOperand(3);
13069 if (CntC && CntC->isZero()) {
13070 SDValue Chain =
Op->getOperand(0);
13071 SDValue BarOp =
Op->getOperand(2);
13074 std::optional<uint64_t> BarVal;
13076 BarVal =
C->getZExtValue();
13080 BarVal = *Addr + GA->getOffset();
13083 unsigned BarID = *BarVal & 0x3F;
13085 Ops.push_back(Chain);
13087 Op->getVTList(),
Ops);
13088 return SDValue(NewMI, 0);
13093 case Intrinsic::amdgcn_s_barrier_init: {
13095 SDValue Chain =
Op->getOperand(0);
13097 SDValue BarOp =
Op->getOperand(2);
13098 SDValue CntOp =
Op->getOperand(3);
13100 unsigned Opc = IntrinsicID == Intrinsic::amdgcn_s_barrier_init
13101 ? AMDGPU::S_BARRIER_INIT_M0
13102 : AMDGPU::S_BARRIER_SIGNAL_M0;
13110 constexpr unsigned ShAmt = 16;
13119 return SDValue(NewMI, 0);
13121 case Intrinsic::amdgcn_s_wakeup_barrier: {
13122 if (!Subtarget->hasSWakeupBarrier())
13126 case Intrinsic::amdgcn_s_barrier_join: {
13128 SDValue Chain =
Op->getOperand(0);
13130 SDValue BarOp =
Op->getOperand(2);
13135 switch (IntrinsicID) {
13138 case Intrinsic::amdgcn_s_barrier_join:
13139 Opc = AMDGPU::S_BARRIER_JOIN_IMM;
13141 case Intrinsic::amdgcn_s_wakeup_barrier:
13142 Opc = AMDGPU::S_WAKEUP_BARRIER_IMM;
13146 unsigned BarID = BarVal & 0x3F;
13149 Ops.push_back(Chain);
13151 switch (IntrinsicID) {
13154 case Intrinsic::amdgcn_s_barrier_join:
13155 Opc = AMDGPU::S_BARRIER_JOIN_M0;
13157 case Intrinsic::amdgcn_s_wakeup_barrier:
13158 Opc = AMDGPU::S_WAKEUP_BARRIER_M0;
13168 return SDValue(NewMI, 0);
13170 case Intrinsic::amdgcn_s_prefetch_data:
13171 case Intrinsic::amdgcn_s_prefetch_inst: {
13174 return Op.getOperand(0);
13177 case Intrinsic::amdgcn_s_buffer_prefetch_data: {
13179 Chain, bufferRsrcPtrToVector(
Op.getOperand(2), DAG),
13186 Op->getVTList(),
Ops,
M->getMemoryVT(),
13187 M->getMemOperand());
13189 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
13190 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
13191 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B: {
13193 SDValue Chain =
Op->getOperand(0);
13194 SDValue Ptr =
Op->getOperand(2);
13195 SDValue Val =
Op->getOperand(3);
13199 case Intrinsic::amdgcn_av_store_b128: {
13201 SDValue Chain =
Op->getOperand(0);
13202 SDValue Ptr =
Op->getOperand(2);
13203 SDValue Val =
Op->getOperand(3);
13207 if (
const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
13209 return lowerImage(
Op, ImageDimIntr, DAG,
true);
13225 return PtrVT == MVT::i64;
13239std::pair<SDValue, SDValue>
13252 bool CheckNUW = Subtarget->hasGFX1250Insts();
13269 unsigned Overflow = ImmOffset & ~MaxImm;
13270 ImmOffset -= Overflow;
13271 if ((int32_t)Overflow < 0) {
13272 Overflow += ImmOffset;
13277 auto OverflowVal = DAG.
getConstant(Overflow,
DL, MVT::i32);
13281 SDValue
Ops[] = {N0, OverflowVal};
13290 return {N0, SDValue(C1, 0)};
13296void SITargetLowering::setBufferOffsets(
SDValue CombinedOffset,
13298 Align Alignment)
const {
13300 SDLoc
DL(CombinedOffset);
13302 uint32_t
Imm =
C->getZExtValue();
13303 uint32_t SOffset, ImmOffset;
13304 if (
TII->splitMUBUFOffset(
Imm, SOffset, ImmOffset, Alignment)) {
13315 bool CheckNUW = Subtarget->hasGFX1250Insts();
13318 uint32_t SOffset, ImmOffset;
13321 TII->splitMUBUFOffset(
Offset, SOffset, ImmOffset, Alignment)) {
13329 SDValue SOffsetZero = Subtarget->hasRestrictedSOffset()
13338SDValue SITargetLowering::bufferRsrcPtrToVector(
SDValue MaybePointer,
13341 return MaybePointer;
13343 SDValue Rsrc = DAG.
getBitcast(MVT::v4i32, MaybePointer);
13354 SDValue Stride =
Op->getOperand(2);
13355 SDValue NumRecords =
Op->getOperand(3);
13356 SDValue
Flags =
Op->getOperand(4);
13361 if (Subtarget->getBufferResourceNumRecordsWidth() == 45) {
13364 DAG.
getConstant((1ULL << 45) - 1, Loc, MVT::i64));
13369 SDValue NumRecordsLHS =
13377 SDValue NumRecordsRHS =
13380 SDValue ShiftedStride =
13383 SDValue ExtShiftedStrideVec =
13385 SDValue ExtShiftedStride =
13387 SDValue ShiftedFlags =
13390 SDValue ExtShiftedFlagsVec =
13392 SDValue ExtShiftedFlags =
13394 SDValue CombinedFields =
13395 DAG.
getNode(
ISD::OR, Loc, MVT::i64, NumRecordsRHS, ExtShiftedStride);
13397 DAG.
getNode(
ISD::OR, Loc, MVT::i64, CombinedFields, ExtShiftedFlags);
13402 auto [LowHalf, HighHalf] =
13403 DAG.
SplitScalar(Pointer, Loc, MVT::i32, MVT::i32);
13406 SDValue ShiftedStride =
13409 SDValue NewHighHalf =
13413 NumRecords, Flags);
13425 bool IsTFE)
const {
13430 ? AMDGPUISD::BUFFER_LOAD_UBYTE_TFE
13431 : AMDGPUISD::BUFFER_LOAD_USHORT_TFE;
13434 SDVTList VTs = DAG.
getVTList(MVT::v2i32, MVT::Other);
13435 SDValue
Op = getMemIntrinsicNode(
Opc,
DL, VTs,
Ops, MVT::v2i32, OpMMO, DAG);
13446 ? AMDGPUISD::BUFFER_LOAD_UBYTE
13447 : AMDGPUISD::BUFFER_LOAD_USHORT;
13449 SDVTList ResList = DAG.
getVTList(MVT::i32, MVT::Other);
13450 SDValue BufferLoad =
13463 if (VDataType == MVT::f16 || VDataType == MVT::bf16)
13467 Ops[1] = BufferStoreExt;
13468 unsigned Opc = (VDataType == MVT::i8) ? AMDGPUISD::BUFFER_STORE_BYTE
13469 : AMDGPUISD::BUFFER_STORE_SHORT;
13472 M->getMemOperand());
13497 DAGCombinerInfo &DCI)
const {
13498 SelectionDAG &DAG = DCI.DAG;
13499 if (Ld->getAlign() <
Align(4) || Ld->isDivergent())
13503 unsigned AS = Ld->getAddressSpace();
13512 EVT MemVT = Ld->getMemoryVT();
13513 if ((MemVT.
isSimple() && !DCI.isAfterLegalizeDAG()) ||
13520 "unexpected vector extload");
13523 SDValue Ptr = Ld->getBasePtr();
13524 SDValue NewLoad = DAG.
getLoad(
13526 Ld->getOffset(), Ld->getPointerInfo(), MVT::i32, Ld->getAlign(),
13527 Ld->getMemOperand()->getFlags(), Ld->getAAInfo());
13532 "unexpected fp extload");
13536 SDValue Cvt = NewLoad;
13547 EVT VT = Ld->getValueType(0);
13550 DCI.AddToWorklist(Cvt.
getNode());
13555 DCI.AddToWorklist(Cvt.
getNode());
13566 if (Info.isEntryFunction())
13567 return Info.getUserSGPRInfo().hasFlatScratchInit();
13575 EVT MemVT =
Load->getMemoryVT();
13576 MachineMemOperand *MMO =
Load->getMemOperand();
13583 isTypeLegal(MemVT) && Subtarget->hasScalarSubwordLoads() &&
13585 SDValue Chain =
Load->getChain();
13590 BasePtr, MVT::i16, MMO);
13596 SDValue
Result = (MemVT == MVT::i16)
13610 SDValue Chain =
Load->getChain();
13613 EVT RealMemVT = (MemVT == MVT::i1) ? MVT::i8 : MVT::i16;
13641 assert(
Op.getValueType().getVectorElementType() == MVT::i32 &&
13642 "Custom lowering for non-i32 vectors hasn't been implemented.");
13645 unsigned AS =
Load->getAddressSpace();
13646 if (Subtarget->hasLDSMisalignedBugInWGPMode() &&
13653 SIMachineFunctionInfo *MFI = MF.
getInfo<SIMachineFunctionInfo>();
13657 !Subtarget->hasMultiDwordFlatScratchAddressing())
13669 Alignment >=
Align(4) && NumElements < 32) {
13671 (Subtarget->hasScalarDwordx3Loads() && NumElements == 3))
13683 if (NumElements > 4)
13686 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
13696 switch (Subtarget->getMaxPrivateElementSize()) {
13702 if (NumElements > 2)
13707 if (NumElements > 4)
13710 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
13719 auto Flags =
Load->getMemOperand()->getFlags();
13721 Load->getAlign(), Flags, &
Fast) &&
13730 MemVT, *
Load->getMemOperand())) {
13739 EVT VT =
Op.getValueType();
13774 SDValue
LHS =
Op.getOperand(0);
13775 SDValue
RHS =
Op.getOperand(1);
13776 EVT VT =
Op.getValueType();
13777 const SDNodeFlags
Flags =
Op->getFlags();
13779 bool AllowInaccurateRcp =
Flags.hasApproximateFuncs();
13785 if (!AllowInaccurateRcp && VT != MVT::f16 && VT != MVT::bf16)
13788 if (CLHS->isOne()) {
13801 return DAG.
getNode(AMDGPUISD::RCP, SL, VT,
RHS);
13805 if (CLHS->isMinusOne()) {
13808 return DAG.
getNode(AMDGPUISD::RCP, SL, VT, FNegRHS);
13814 if (!AllowInaccurateRcp &&
13815 ((VT != MVT::f16 && VT != MVT::bf16) || !
Flags.hasAllowReciprocal()))
13820 SDValue Recip = DAG.
getNode(AMDGPUISD::RCP, SL, VT,
RHS);
13827 SDValue
X =
Op.getOperand(0);
13828 SDValue
Y =
Op.getOperand(1);
13829 EVT VT =
Op.getValueType();
13830 const SDNodeFlags
Flags =
Op->getFlags();
13832 bool AllowInaccurateDiv =
Flags.hasApproximateFuncs();
13833 if (!AllowInaccurateDiv)
13846 SDValue
R = DAG.
getNode(AMDGPUISD::RCP, SL, VT,
Y);
13857 if (IsNegRcp || (CLHS && CLHS->
isOne()))
13869 return DAG.
getNode(Opcode, SL, VT,
A,
B, Flags);
13879 Opcode = AMDGPUISD::FMUL_W_CHAIN;
13883 return DAG.
getNode(Opcode, SL, VTList,
13892 return DAG.
getNode(Opcode, SL, VT, {
A,
B,
C}, Flags);
13902 Opcode = AMDGPUISD::FMA_W_CHAIN;
13906 return DAG.
getNode(Opcode, SL, VTList,
13912 if (SDValue FastLowered = lowerFastUnsafeFDIV(
Op, DAG))
13913 return FastLowered;
13916 EVT VT =
Op.getValueType();
13917 SDValue
LHS =
Op.getOperand(0);
13918 SDValue
RHS =
Op.getOperand(1);
13923 if (VT == MVT::bf16) {
13946 unsigned FMADOpCode =
13950 DAG.
getNode(AMDGPUISD::RCP, SL, MVT::f32, RHSExt,
Op->getFlags());
13953 SDValue Err = DAG.
getNode(FMADOpCode, SL, MVT::f32, NegRHSExt, Quot, LHSExt,
13955 Quot = DAG.
getNode(FMADOpCode, SL, MVT::f32, Err, Rcp, Quot,
Op->getFlags());
13956 Err = DAG.
getNode(FMADOpCode, SL, MVT::f32, NegRHSExt, Quot, LHSExt,
13966 return DAG.
getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f16, RDst,
RHS,
LHS,
13972 SDNodeFlags
Flags =
Op->getFlags();
13974 SDValue
LHS =
Op.getOperand(1);
13975 SDValue
RHS =
Op.getOperand(2);
13982 const APFloat K0Val(0x1p+96f);
13985 const APFloat K1Val(0x1p-32f);
14000 SDValue
r0 = DAG.
getNode(AMDGPUISD::RCP, SL, MVT::f32,
r1, Flags);
14012 assert(ST->hasDenormModeInst() &&
"Requires S_DENORM_MODE");
14013 uint32_t DPDenormModeDefault = Info->getMode().fpDenormModeDPValue();
14014 uint32_t Mode = SPDenormMode | (DPDenormModeDefault << 2);
14019 if (SDValue FastLowered = lowerFastUnsafeFDIV(
Op, DAG))
14020 return FastLowered;
14026 SDNodeFlags
Flags =
Op->getFlags();
14027 Flags.setNoFPExcept(
true);
14030 SDValue
LHS =
Op.getOperand(0);
14031 SDValue
RHS =
Op.getOperand(1);
14035 SDVTList ScaleVT = DAG.
getVTList(MVT::f32, MVT::i1);
14037 SDValue DenominatorScaled =
14039 SDValue NumeratorScaled =
14043 SDValue ApproxRcp =
14044 DAG.
getNode(AMDGPUISD::RCP, SL, MVT::f32, DenominatorScaled, Flags);
14045 SDValue NegDivScale0 =
14048 using namespace AMDGPU::Hwreg;
14049 const unsigned Denorm32Reg = HwregEncoding::encode(ID_MODE, 4, 2);
14053 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
14054 const DenormalMode DenormMode =
Info->getMode().FP32Denormals;
14057 const bool HasDynamicDenormals =
14061 SDValue SavedDenormMode;
14063 if (!PreservesDenormals) {
14068 SDVTList BindParamVTs = DAG.
getVTList(MVT::Other, MVT::Glue);
14071 if (HasDynamicDenormals) {
14075 SavedDenormMode = SDValue(GetReg, 0);
14078 {DAG.
getEntryNode(), SDValue(GetReg, 0), SDValue(GetReg, 1)}, SL);
14081 SDNode *EnableDenorm;
14082 if (Subtarget->hasDenormModeInst()) {
14083 const SDValue EnableDenormValue =
14086 EnableDenorm = DAG.
getNode(AMDGPUISD::DENORM_MODE, SL, BindParamVTs, Glue,
14090 const SDValue EnableDenormValue =
14092 EnableDenorm = DAG.
getMachineNode(AMDGPU::S_SETREG_B32, SL, BindParamVTs,
14093 {EnableDenormValue,
BitField, Glue});
14096 SDValue
Ops[3] = {NegDivScale0, SDValue(EnableDenorm, 0),
14097 SDValue(EnableDenorm, 1)};
14103 ApproxRcp, One, NegDivScale0, Flags);
14106 ApproxRcp, Fma0, Flags);
14112 NumeratorScaled,
Mul, Flags);
14118 NumeratorScaled, Fma3, Flags);
14120 if (!PreservesDenormals) {
14121 SDNode *DisableDenorm;
14122 if (!HasDynamicDenormals && Subtarget->hasDenormModeInst()) {
14126 SDVTList BindParamVTs = DAG.
getVTList(MVT::Other, MVT::Glue);
14128 DAG.
getNode(AMDGPUISD::DENORM_MODE, SL, BindParamVTs,
14132 assert(HasDynamicDenormals == (
bool)SavedDenormMode);
14133 const SDValue DisableDenormValue =
14134 HasDynamicDenormals
14139 AMDGPU::S_SETREG_B32, SL, MVT::Other,
14144 SDValue(DisableDenorm, 0), DAG.
getRoot());
14148 SDValue Scale = NumeratorScaled.
getValue(1);
14149 SDValue Fmas = DAG.
getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f32,
14150 {Fma4, Fma1, Fma3, Scale},
Flags);
14152 return DAG.
getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f32, Fmas,
RHS,
LHS, Flags);
14156 if (SDValue FastLowered = lowerFastUnsafeFDIV64(
Op, DAG))
14157 return FastLowered;
14160 SDValue
X =
Op.getOperand(0);
14161 SDValue
Y =
Op.getOperand(1);
14165 SDVTList ScaleVT = DAG.
getVTList(MVT::f64, MVT::i1);
14167 SDValue DivScale0 = DAG.
getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT,
Y,
Y,
X);
14171 SDValue Rcp = DAG.
getNode(AMDGPUISD::RCP, SL, MVT::f64, DivScale0);
14173 SDValue Fma0 = DAG.
getNode(
ISD::FMA, SL, MVT::f64, NegDivScale0, Rcp, One);
14177 SDValue Fma2 = DAG.
getNode(
ISD::FMA, SL, MVT::f64, NegDivScale0, Fma1, One);
14179 SDValue DivScale1 = DAG.
getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT,
X,
Y,
X);
14181 SDValue Fma3 = DAG.
getNode(
ISD::FMA, SL, MVT::f64, Fma1, Fma2, Fma1);
14189 if (!Subtarget->hasUsableDivScaleConditionOutput()) {
14219 DAG.
getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f64, Fma4, Fma3,
Mul, Scale);
14221 return DAG.
getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f64, Fmas,
Y,
X);
14225 EVT VT =
Op.getValueType();
14227 if (VT == MVT::f32)
14228 return LowerFDIV32(
Op, DAG);
14230 if (VT == MVT::f64)
14231 return LowerFDIV64(
Op, DAG);
14233 if (VT == MVT::f16 || VT == MVT::bf16)
14234 return LowerFDIV16(
Op, DAG);
14241 SDValue Val =
Op.getOperand(0);
14243 EVT ResultExpVT =
Op->getValueType(1);
14244 EVT InstrExpVT = VT == MVT::f16 ? MVT::i16 : MVT::i32;
14254 if (Subtarget->hasFractBug()) {
14272 EVT VT =
Store->getMemoryVT();
14274 if (VT == MVT::i1) {
14278 Store->getBasePtr(), MVT::i1,
Store->getMemOperand());
14282 Store->getValue().getValueType().getScalarType() == MVT::i32);
14284 unsigned AS =
Store->getAddressSpace();
14285 if (Subtarget->hasLDSMisalignedBugInWGPMode() &&
14293 SIMachineFunctionInfo *MFI = MF.
getInfo<SIMachineFunctionInfo>();
14297 !Subtarget->hasMultiDwordFlatScratchAddressing())
14304 if (NumElements > 4)
14307 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
14311 VT, *
Store->getMemOperand()))
14317 switch (Subtarget->getMaxPrivateElementSize()) {
14321 if (NumElements > 2)
14325 if (NumElements > 4 ||
14326 (NumElements == 3 && !Subtarget->hasFlatScratchEnabled()))
14334 auto Flags =
Store->getMemOperand()->getFlags();
14353 assert(!Subtarget->has16BitInsts());
14354 SDNodeFlags
Flags =
Op->getFlags();
14368 SDNodeFlags
Flags =
Op->getFlags();
14369 MVT VT =
Op.getValueType().getSimpleVT();
14370 const SDValue
X =
Op.getOperand(0);
14396 SDValue SqrtSNextDownInt =
14401 SDValue NegSqrtSNextDown =
14425 SDValue SqrtR = DAG.
getNode(AMDGPUISD::RSQ,
DL, VT, SqrtX, Flags);
14445 SDValue ScaledDown =
14449 SDValue IsZeroOrInf =
14477 SDNodeFlags
Flags =
Op->getFlags();
14481 SDValue
X =
Op.getOperand(0);
14486 if (!
Flags.hasApproximateFuncs()) {
14491 SDValue ScaleUpFactor = DAG.
getConstant(256,
DL, MVT::i32);
14497 SDValue SqrtY = DAG.
getNode(AMDGPUISD::RSQ,
DL, MVT::f64, SqrtX);
14517 SDValue SqrtRet = SqrtS2;
14518 if (!
Flags.hasApproximateFuncs()) {
14527 ScaleDownFactor, ZeroInt);
14533 SDValue IsZeroOrInf;
14534 if (
Flags.hasNoInfs()) {
14550 EVT VT =
Op.getValueType();
14551 SDValue Arg =
Op.getOperand(0);
14560 auto UnrollIfVec = [&DAG](SDValue
V) -> SDValue {
14561 if (!
V.getValueType().isVector())
14569 if (Subtarget->hasTrigReducedRange()) {
14571 TrigVal = UnrollIfVec(DAG.
getNode(AMDGPUISD::FRACT,
DL, VT, MulVal, Flags));
14576 switch (
Op.getOpcode()) {
14578 TrigVal = DAG.
getNode(AMDGPUISD::COS_HW, SDLoc(
Op), VT, TrigVal, Flags);
14581 TrigVal = DAG.
getNode(AMDGPUISD::SIN_HW, SDLoc(
Op), VT, TrigVal, Flags);
14587 return UnrollIfVec(TrigVal);
14603 SDValue ChainIn =
Op.getOperand(0);
14604 SDValue Addr =
Op.getOperand(1);
14605 SDValue Old =
Op.getOperand(2);
14606 SDValue
New =
Op.getOperand(3);
14607 EVT VT =
Op.getValueType();
14612 SDValue
Ops[] = {ChainIn, Addr, NewOld};
14615 Op->getVTList(),
Ops, VT,
14624SITargetLowering::performUCharToFloatCombine(
SDNode *
N,
14625 DAGCombinerInfo &DCI)
const {
14626 EVT VT =
N->getValueType(0);
14628 if (ScalarVT != MVT::f32 && ScalarVT != MVT::f16)
14631 SelectionDAG &DAG = DCI.DAG;
14634 SDValue Src =
N->getOperand(0);
14635 EVT SrcVT = Src.getValueType();
14641 if (DCI.isAfterLegalizeDAG() && SrcVT == MVT::i32) {
14643 SDValue Cvt = DAG.
getNode(AMDGPUISD::CVT_F32_UBYTE0,
DL, MVT::f32, Src);
14644 DCI.AddToWorklist(Cvt.
getNode());
14647 if (ScalarVT != MVT::f32) {
14659 DAGCombinerInfo &DCI)
const {
14670 SelectionDAG &DAG = DCI.DAG;
14689 for (
unsigned I = 0;
I != NumElts; ++
I) {
14697 SDValue SignOpElt =
14713 if (NewElts.
size() == 1)
14735 for (
unsigned I = 0;
I != NumElts; ++
I) {
14737 SDValue SignAsF32 =
14770SDValue SITargetLowering::performSHLPtrCombine(
SDNode *
N,
unsigned AddrSpace,
14772 DAGCombinerInfo &DCI)
const {
14789 SelectionDAG &DAG = DCI.DAG;
14802 AM.BaseOffs =
Offset.getSExtValue();
14807 EVT VT =
N->getValueType(0);
14813 Flags.setNoUnsignedWrap(
14814 N->getFlags().hasNoUnsignedWrap() &&
14826 switch (
N->getOpcode()) {
14837 DAGCombinerInfo &DCI)
const {
14838 SelectionDAG &DAG = DCI.DAG;
14845 SDValue NewPtr = performSHLPtrCombine(Ptr.
getNode(),
N->getAddressSpace(),
14846 N->getMemoryVT(), DCI);
14850 NewOps[PtrIdx] = NewPtr;
14859 return (
Opc ==
ISD::AND && (Val == 0 || Val == 0xffffffff)) ||
14860 (
Opc ==
ISD::OR && (Val == 0xffffffff || Val == 0)) ||
14869SDValue SITargetLowering::splitBinaryBitConstantOp(
14873 uint32_t ValLo =
Lo_32(Val);
14874 uint32_t ValHi =
Hi_32(Val);
14881 if (Subtarget->has64BitLiterals() && CRHS->
hasOneUse() &&
14895 if (V.getValueType() != MVT::i1)
14897 switch (V.getOpcode()) {
14902 case AMDGPUISD::FP_CLASS:
14914 return V.getResNo() == 1;
14916 unsigned IntrinsicID = V.getConstantOperandVal(0);
14917 switch (IntrinsicID) {
14918 case Intrinsic::amdgcn_is_shared:
14919 case Intrinsic::amdgcn_is_private:
14936 if (!(
C & 0x000000ff))
14937 ZeroByteMask |= 0x000000ff;
14938 if (!(
C & 0x0000ff00))
14939 ZeroByteMask |= 0x0000ff00;
14940 if (!(
C & 0x00ff0000))
14941 ZeroByteMask |= 0x00ff0000;
14942 if (!(
C & 0xff000000))
14943 ZeroByteMask |= 0xff000000;
14944 uint32_t NonZeroByteMask = ~ZeroByteMask;
14945 if ((NonZeroByteMask &
C) != NonZeroByteMask)
14958 assert(V.getValueSizeInBits() == 32);
14960 if (V.getNumOperands() != 2)
14969 switch (V.getOpcode()) {
14974 return (0x03020100 & ConstMask) | (0x0c0c0c0c & ~ConstMask);
14979 return (0x03020100 & ~ConstMask) | ConstMask;
14986 return uint32_t((0x030201000c0c0c0cull <<
C) >> 32);
14992 return uint32_t(0x0c0c0c0c03020100ull >>
C);
14999 DAGCombinerInfo &DCI)
const {
15000 if (DCI.isBeforeLegalize())
15003 SelectionDAG &DAG = DCI.DAG;
15004 EVT VT =
N->getValueType(0);
15005 SDValue
LHS =
N->getOperand(0);
15006 SDValue
RHS =
N->getOperand(1);
15009 if (VT == MVT::i64 && CRHS) {
15010 if (SDValue Split =
15011 splitBinaryBitConstantOp(DCI, SDLoc(
N),
ISD::AND,
LHS, CRHS))
15015 if (CRHS && VT == MVT::i32) {
15025 unsigned Shift = CShift->getZExtValue();
15027 unsigned Offset = NB + Shift;
15028 if ((
Offset & (Bits - 1)) == 0) {
15031 DAG.
getNode(AMDGPUISD::BFE_U32, SL, MVT::i32,
LHS->getOperand(0),
15052 Sel = (
LHS.getConstantOperandVal(2) & Sel) | (~Sel & 0x0c0c0c0c);
15054 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
LHS.getOperand(0),
15065 SDValue
X =
LHS.getOperand(0);
15066 SDValue
Y =
RHS.getOperand(0);
15067 if (
Y.getOpcode() !=
ISD::FABS ||
Y.getOperand(0) !=
X ||
15072 if (
X !=
LHS.getOperand(1))
15076 const ConstantFPSDNode *C1 =
15093 return DAG.
getNode(AMDGPUISD::FP_CLASS,
DL, MVT::i1,
X,
15099 if (
RHS.getOpcode() ==
ISD::SETCC &&
LHS.getOpcode() == AMDGPUISD::FP_CLASS)
15102 if (
LHS.getOpcode() ==
ISD::SETCC &&
RHS.getOpcode() == AMDGPUISD::FP_CLASS &&
15110 (
RHS.getOperand(0) ==
LHS.getOperand(0) &&
15111 LHS.getOperand(0) ==
LHS.getOperand(1))) {
15113 unsigned NewMask = LCC ==
ISD::SETO ?
Mask->getZExtValue() & ~OrdMask
15114 :
Mask->getZExtValue() & OrdMask;
15117 return DAG.
getNode(AMDGPUISD::FP_CLASS,
DL, MVT::i1,
RHS.getOperand(0),
15135 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
15138 if (LHSMask != ~0u && RHSMask != ~0u) {
15141 if (LHSMask > RHSMask) {
15148 uint32_t LHSUsedLanes = ~(LHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15149 uint32_t RHSUsedLanes = ~(RHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15152 if (!(LHSUsedLanes & RHSUsedLanes) &&
15155 !(LHSUsedLanes == 0x0c0c0000 && RHSUsedLanes == 0x00000c0c)) {
15161 uint32_t
Mask = LHSMask & RHSMask;
15162 for (
unsigned I = 0;
I < 32;
I += 8) {
15163 uint32_t ByteSel = 0xff <<
I;
15164 if ((LHSMask & ByteSel) == 0x0c || (RHSMask & ByteSel) == 0x0c)
15165 Mask &= (0x0c <<
I) & 0xffffffff;
15170 uint32_t Sel =
Mask | (LHSUsedLanes & 0x04040404);
15173 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
LHS.getOperand(0),
15223static const std::optional<ByteProvider<SDValue>>
15225 unsigned Depth = 0) {
15228 return std::nullopt;
15230 if (
Op.getValueSizeInBits() < 8)
15231 return std::nullopt;
15233 if (
Op.getValueType().isVector())
15236 switch (
Op->getOpcode()) {
15249 NarrowVT = VTSign->getVT();
15252 return std::nullopt;
15255 if (SrcIndex >= NarrowByteWidth)
15256 return std::nullopt;
15264 return std::nullopt;
15266 uint64_t BitShift = ShiftOp->getZExtValue();
15268 if (BitShift % 8 != 0)
15269 return std::nullopt;
15271 uint64_t NewSrcIndex = SrcIndex + BitShift / 8;
15272 if (NewSrcIndex >=
Op.getScalarValueSizeInBits() / 8)
15273 return std::nullopt;
15292static const std::optional<ByteProvider<SDValue>>
15294 unsigned StartingIndex = 0) {
15298 return std::nullopt;
15300 unsigned BitWidth =
Op.getScalarValueSizeInBits();
15302 return std::nullopt;
15304 return std::nullopt;
15306 bool IsVec =
Op.getValueType().isVector();
15307 switch (
Op.getOpcode()) {
15310 return std::nullopt;
15315 return std::nullopt;
15319 return std::nullopt;
15322 if (!
LHS->isConstantZero() && !
RHS->isConstantZero())
15323 return std::nullopt;
15324 if (!
LHS ||
LHS->isConstantZero())
15326 if (!
RHS ||
RHS->isConstantZero())
15328 return std::nullopt;
15333 return std::nullopt;
15337 return std::nullopt;
15339 uint32_t BitMask = BitMaskOp->getZExtValue();
15341 uint32_t IndexMask = 0xFF << (Index * 8);
15343 if ((IndexMask & BitMask) != IndexMask) {
15346 if (IndexMask & BitMask)
15347 return std::nullopt;
15356 return std::nullopt;
15360 if (!ShiftOp ||
Op.getValueType().isVector())
15361 return std::nullopt;
15363 uint64_t BitsProvided =
Op.getValueSizeInBits();
15364 if (BitsProvided % 8 != 0)
15365 return std::nullopt;
15367 uint64_t BitShift = ShiftOp->getAPIntValue().urem(BitsProvided);
15369 return std::nullopt;
15371 uint64_t ConcatSizeInBytes = BitsProvided / 4;
15372 uint64_t ByteShift = BitShift / 8;
15374 uint64_t NewIndex = (Index + ByteShift) % ConcatSizeInBytes;
15375 uint64_t BytesProvided = BitsProvided / 8;
15376 SDValue NextOp =
Op.getOperand(NewIndex >= BytesProvided ? 0 : 1);
15377 NewIndex %= BytesProvided;
15384 return std::nullopt;
15388 return std::nullopt;
15390 uint64_t BitShift = ShiftOp->getZExtValue();
15392 return std::nullopt;
15394 auto BitsProvided =
Op.getScalarValueSizeInBits();
15395 if (BitsProvided % 8 != 0)
15396 return std::nullopt;
15398 uint64_t BytesProvided = BitsProvided / 8;
15399 uint64_t ByteShift = BitShift / 8;
15400 if (Index + ByteShift < BytesProvided)
15402 Index + ByteShift);
15405 return std::nullopt;
15411 return std::nullopt;
15415 return std::nullopt;
15417 uint64_t BitShift = ShiftOp->getZExtValue();
15418 if (BitShift % 8 != 0)
15419 return std::nullopt;
15420 uint64_t ByteShift = BitShift / 8;
15426 return Index < ByteShift
15429 Depth + 1, StartingIndex);
15438 return std::nullopt;
15446 NarrowBitWidth = VTSign->getVT().getSizeInBits();
15448 if (NarrowBitWidth % 8 != 0)
15449 return std::nullopt;
15450 uint64_t NarrowByteWidth = NarrowBitWidth / 8;
15452 if (Index >= NarrowByteWidth)
15454 ? std::optional<ByteProvider<SDValue>>(
15462 return std::nullopt;
15466 if (NarrowByteWidth >= Index) {
15471 return std::nullopt;
15478 return std::nullopt;
15484 unsigned NarrowBitWidth = L->getMemoryVT().getSizeInBits();
15485 if (NarrowBitWidth % 8 != 0)
15486 return std::nullopt;
15487 uint64_t NarrowByteWidth = NarrowBitWidth / 8;
15492 if (Index >= NarrowByteWidth) {
15494 ? std::optional<ByteProvider<SDValue>>(
15499 if (NarrowByteWidth > Index) {
15503 return std::nullopt;
15508 return std::nullopt;
15511 Depth + 1, StartingIndex);
15517 return std::nullopt;
15518 auto VecIdx = IdxOp->getZExtValue();
15519 auto ScalarSize =
Op.getScalarValueSizeInBits();
15520 if (ScalarSize < 32)
15521 Index = ScalarSize == 8 ? VecIdx : VecIdx * 2 + Index;
15523 StartingIndex, Index);
15526 case AMDGPUISD::PERM: {
15528 return std::nullopt;
15532 return std::nullopt;
15535 (PermMask->getZExtValue() & (0xFF << (Index * 8))) >> (Index * 8);
15536 if (IdxMask > 0x07 && IdxMask != 0x0c)
15537 return std::nullopt;
15539 auto NextOp =
Op.getOperand(IdxMask > 0x03 ? 0 : 1);
15540 auto NextIndex = IdxMask > 0x03 ? IdxMask % 4 : IdxMask;
15542 return IdxMask != 0x0c ?
calculateSrcByte(NextOp, StartingIndex, NextIndex)
15548 return std::nullopt;
15563 return !OpVT.
isVector() && OpVT.getSizeInBits() == 16;
15570 auto MemVT = L->getMemoryVT();
15573 return L->getMemoryVT().getSizeInBits() == 16;
15583 int Low8 = Mask & 0xff;
15584 int Hi8 = (Mask & 0xff00) >> 8;
15586 assert(Low8 < 8 && Hi8 < 8);
15588 bool IsConsecutive = (Hi8 - Low8 == 1);
15593 bool Is16Aligned = !(Low8 % 2);
15595 return IsConsecutive && Is16Aligned;
15603 int Low16 = PermMask & 0xffff;
15604 int Hi16 = (PermMask & 0xffff0000) >> 16;
15614 auto OtherOpIs16Bit = TempOtherOp.getValueSizeInBits() == 16 ||
15616 if (!OtherOpIs16Bit)
15624 unsigned DWordOffset) {
15629 assert(Src.getValueSizeInBits().isKnownMultipleOf(8));
15634 if (Src.getValueType().isVector()) {
15635 auto ScalarTySize = Src.getScalarValueSizeInBits();
15636 auto ScalarTy = Src.getValueType().getScalarType();
15637 if (ScalarTySize == 32) {
15641 if (ScalarTySize > 32) {
15644 DAG.
getConstant(DWordOffset / (ScalarTySize / 32), SL, MVT::i32));
15645 auto ShiftVal = 32 * (DWordOffset % (ScalarTySize / 32));
15652 assert(ScalarTySize < 32);
15661 auto NumElements =
TypeSize / ScalarTySize;
15662 auto Trunc32Elements = (ScalarTySize * NumElements) / 32;
15663 auto NormalizedTrunc = Trunc32Elements * 32 / ScalarTySize;
15664 auto NumElementsIn32 = 32 / ScalarTySize;
15665 auto NumAvailElements = DWordOffset < Trunc32Elements
15667 : NumElements - NormalizedTrunc;
15680 auto ShiftVal = 32 * DWordOffset;
15688 [[maybe_unused]]
EVT VT =
N->getValueType(0);
15693 for (
int i = 0; i < 4; i++) {
15695 std::optional<ByteProvider<SDValue>>
P =
15698 if (!
P ||
P->isConstantZero())
15703 if (PermNodes.
size() != 4)
15706 std::pair<unsigned, unsigned> FirstSrc(0, PermNodes[0].SrcOffset / 4);
15707 std::optional<std::pair<unsigned, unsigned>> SecondSrc;
15709 for (
size_t i = 0; i < PermNodes.
size(); i++) {
15710 auto PermOp = PermNodes[i];
15713 int SrcByteAdjust = 4;
15717 if (!PermOp.hasSameSrc(PermNodes[FirstSrc.first]) ||
15718 ((PermOp.SrcOffset / 4) != FirstSrc.second)) {
15720 if (!PermOp.hasSameSrc(PermNodes[SecondSrc->first]) ||
15721 ((PermOp.SrcOffset / 4) != SecondSrc->second))
15725 SecondSrc = {i, PermNodes[i].SrcOffset / 4};
15726 assert(!(PermNodes[SecondSrc->first].Src->getValueSizeInBits() % 8));
15729 assert((PermOp.SrcOffset % 4) + SrcByteAdjust < 8);
15731 PermMask |= ((PermOp.SrcOffset % 4) + SrcByteAdjust) << (i * 8);
15734 SDValue Op = *PermNodes[FirstSrc.first].Src;
15736 assert(
Op.getValueSizeInBits() == 32);
15740 int Low16 = PermMask & 0xffff;
15741 int Hi16 = (PermMask & 0xffff0000) >> 16;
15743 bool WellFormedLow = (Low16 == 0x0504) || (Low16 == 0x0100);
15744 bool WellFormedHi = (Hi16 == 0x0706) || (Hi16 == 0x0302);
15747 if (WellFormedLow && WellFormedHi)
15751 SDValue OtherOp = SecondSrc ? *PermNodes[SecondSrc->first].Src :
Op;
15760 (
N->getOperand(0) ==
Op ||
N->getOperand(0) == OtherOp) &&
15761 (
N->getOperand(1) ==
Op ||
N->getOperand(1) == OtherOp))
15766 assert(
Op.getValueType().isByteSized() &&
15777 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
Op, OtherOp,
15784 DAGCombinerInfo &DCI)
const {
15785 SelectionDAG &DAG = DCI.DAG;
15786 SDValue
LHS =
N->getOperand(0);
15787 SDValue
RHS =
N->getOperand(1);
15789 EVT VT =
N->getValueType(0);
15790 if (VT == MVT::i1) {
15792 if (
LHS.getOpcode() == AMDGPUISD::FP_CLASS &&
15793 RHS.getOpcode() == AMDGPUISD::FP_CLASS) {
15794 SDValue Src =
LHS.getOperand(0);
15795 if (Src !=
RHS.getOperand(0))
15800 if (!CLHS || !CRHS)
15804 static const uint32_t MaxMask = 0x3ff;
15809 return DAG.
getNode(AMDGPUISD::FP_CLASS,
DL, MVT::i1, Src,
15818 LHS.getOpcode() == AMDGPUISD::PERM &&
15824 Sel |=
LHS.getConstantOperandVal(2);
15826 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
LHS.getOperand(0),
15833 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
15837 auto usesCombinedOperand = [](SDNode *OrUse) {
15840 !OrUse->getValueType(0).isVector())
15844 for (
auto *VUser : OrUse->users()) {
15845 if (!VUser->getValueType(0).isVector())
15852 if (VUser->getOpcode() == VectorwiseOp)
15858 if (!
any_of(
N->users(), usesCombinedOperand))
15864 if (LHSMask != ~0u && RHSMask != ~0u) {
15867 if (LHSMask > RHSMask) {
15874 uint32_t LHSUsedLanes = ~(LHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15875 uint32_t RHSUsedLanes = ~(RHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15878 if (!(LHSUsedLanes & RHSUsedLanes) &&
15881 !(LHSUsedLanes == 0x0c0c0000 && RHSUsedLanes == 0x00000c0c)) {
15883 LHSMask &= ~RHSUsedLanes;
15884 RHSMask &= ~LHSUsedLanes;
15886 LHSMask |= LHSUsedLanes & 0x04040404;
15888 uint32_t Sel = LHSMask | RHSMask;
15891 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32,
LHS.getOperand(0),
15896 if (LHSMask == ~0u || RHSMask == ~0u) {
15927 SDValue LEVE =
LHS->getOperand(0);
15928 SDValue REVE =
RHS->getOperand(1);
15937 return IdentitySrc;
15943 if (VT != MVT::i64 || DCI.isBeforeLegalizeOps())
15956 SDValue ExtSrc =
RHS.getOperand(0);
15958 if (SrcVT == MVT::i32) {
15961 SDValue LowOr = DAG.
getNode(
ISD::OR, SL, MVT::i32, LowLHS, ExtSrc);
15963 DCI.AddToWorklist(LowOr.
getNode());
15964 DCI.AddToWorklist(HiBits.getNode());
15974 if (SDValue Split = splitBinaryBitConstantOp(DCI, SDLoc(
N),
ISD::OR,
15975 N->getOperand(0), CRHS))
15983 DAGCombinerInfo &DCI)
const {
15984 if (SDValue RV = reassociateScalarOps(
N, DCI.DAG))
15987 SDValue
LHS =
N->getOperand(0);
15988 SDValue
RHS =
N->getOperand(1);
15991 SelectionDAG &DAG = DCI.DAG;
15993 EVT VT =
N->getValueType(0);
15994 if (CRHS && VT == MVT::i64) {
15995 if (SDValue Split =
15996 splitBinaryBitConstantOp(DCI, SDLoc(
N),
ISD::XOR,
LHS, CRHS))
16003 unsigned Opc =
LHS.getOpcode();
16007 SDValue CC =
LHS->getOperand(0);
16008 SDValue TRUE =
LHS->getOperand(1);
16009 SDValue FALSE =
LHS->getOperand(2);
16033 LHS->getOperand(0), FNegLHS, FNegRHS);
16042SITargetLowering::performZeroOrAnyExtendCombine(
SDNode *
N,
16043 DAGCombinerInfo &DCI)
const {
16044 if (!Subtarget->has16BitInsts() ||
16048 EVT VT =
N->getValueType(0);
16049 if (VT != MVT::i32)
16052 SDValue Src =
N->getOperand(0);
16053 if (Src.getValueType() != MVT::i16)
16056 if (!Src->hasOneUse())
16063 std::optional<ByteProvider<SDValue>> BP0 =
16065 if (!BP0 || BP0->SrcOffset >= 4 || !BP0->Src)
16067 SDValue
V0 = *BP0->Src;
16069 std::optional<ByteProvider<SDValue>> BP1 =
16071 if (!BP1 || BP1->SrcOffset >= 4 || !BP1->Src)
16074 SDValue
V1 = *BP1->Src;
16079 SelectionDAG &DAG = DCI.DAG;
16081 uint32_t PermMask = 0x0c0c0c0c;
16084 PermMask = (PermMask & ~0xFF) | (BP0->SrcOffset + 4);
16089 PermMask = (PermMask & ~(0xFF << 8)) | (BP1->SrcOffset << 8);
16092 return DAG.
getNode(AMDGPUISD::PERM,
DL, MVT::i32, V0,
V1,
16097SITargetLowering::performSignExtendInRegCombine(
SDNode *
N,
16098 DAGCombinerInfo &DCI)
const {
16099 SDValue Src =
N->getOperand(0);
16104 if (((Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_UBYTE &&
16105 VTSign->getVT() == MVT::i8) ||
16106 (Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_USHORT &&
16107 VTSign->getVT() == MVT::i16))) {
16108 assert(Subtarget->hasScalarSubwordLoads() &&
16109 "s_buffer_load_{u8, i8} are supported "
16110 "in GFX12 (or newer) architectures.");
16111 unsigned Opc = (Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_UBYTE)
16112 ? AMDGPUISD::SBUFFER_LOAD_BYTE
16113 : AMDGPUISD::SBUFFER_LOAD_SHORT;
16116 DCI.DAG.getVTList(MVT::i32, Src.getOperand(0).getValueType());
16124 SDValue BufferLoad = DCI.DAG.getMemIntrinsicNode(
16125 Opc,
DL, ResList,
Ops,
M->getMemoryVT(),
M->getMemOperand());
16126 return DCI.DAG.getMergeValues({BufferLoad, BufferLoad.
getValue(1)},
DL);
16128 if (((Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_UBYTE &&
16129 VTSign->getVT() == MVT::i8) ||
16130 (Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_USHORT &&
16131 VTSign->getVT() == MVT::i16)) &&
16134 SDValue
Ops[] = {Src.getOperand(0),
16140 Src.getOperand(6), Src.getOperand(7)};
16143 DCI.DAG.getVTList(MVT::i32, Src.getOperand(0).getValueType());
16144 unsigned Opc = (Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_UBYTE)
16145 ? AMDGPUISD::BUFFER_LOAD_BYTE
16146 : AMDGPUISD::BUFFER_LOAD_SHORT;
16147 SDValue BufferLoadSignExt = DCI.DAG.getMemIntrinsicNode(
16148 Opc, SDLoc(
N), ResList,
Ops,
M->getMemoryVT(),
M->getMemOperand());
16149 return DCI.DAG.getMergeValues(
16150 {BufferLoadSignExt, BufferLoadSignExt.
getValue(1)}, SDLoc(
N));
16156 DAGCombinerInfo &DCI)
const {
16157 SelectionDAG &DAG = DCI.DAG;
16158 SDValue
Mask =
N->getOperand(1);
16164 if (
N->getOperand(0).isUndef())
16171 DAGCombinerInfo &DCI)
const {
16172 EVT VT =
N->getValueType(0);
16183 return DCI.DAG.getNode(AMDGPUISD::RSQ, SDLoc(
N), VT, N0.
getOperand(0),
16192 unsigned MaxDepth)
const {
16193 EVT VT =
Op.getValueType();
16195 "expected a floating-point value to query canonicality of");
16201 unsigned MaxDepth)
const {
16203 "QueryVT must be a floating-point scalar type");
16204 EVT VT =
Op.getValueType();
16208 unsigned Opcode =
Op.getOpcode();
16213 const auto &
F = CFP->getValueAPF();
16214 if (
F.isNaN() &&
F.isSignaling())
16216 if (!
F.isDenormal())
16248 case AMDGPUISD::FMUL_LEGACY:
16249 case AMDGPUISD::FMAD_FTZ:
16250 case AMDGPUISD::RCP:
16251 case AMDGPUISD::RSQ:
16252 case AMDGPUISD::RSQ_CLAMP:
16253 case AMDGPUISD::RCP_LEGACY:
16254 case AMDGPUISD::RCP_IFLAG:
16255 case AMDGPUISD::LOG:
16256 case AMDGPUISD::EXP:
16257 case AMDGPUISD::DIV_SCALE:
16258 case AMDGPUISD::DIV_FMAS:
16259 case AMDGPUISD::DIV_FIXUP:
16260 case AMDGPUISD::FRACT:
16261 case AMDGPUISD::CVT_PKRTZ_F16_F32:
16262 case AMDGPUISD::CVT_F32_UBYTE0:
16263 case AMDGPUISD::CVT_F32_UBYTE1:
16264 case AMDGPUISD::CVT_F32_UBYTE2:
16265 case AMDGPUISD::CVT_F32_UBYTE3:
16266 case AMDGPUISD::FP_TO_FP16:
16267 case AMDGPUISD::SIN_HW:
16268 case AMDGPUISD::COS_HW:
16280 if (
Op.getValueType() == MVT::i32) {
16286 if (RHS->getZExtValue() == 0xffff0000) {
16297 return Op.getValueType().getScalarType() != MVT::f16;
16307 case AMDGPUISD::CLAMP:
16308 case AMDGPUISD::FMED3:
16309 case AMDGPUISD::FMAX3:
16310 case AMDGPUISD::FMIN3:
16311 case AMDGPUISD::FMAXIMUM3:
16312 case AMDGPUISD::FMINIMUM3: {
16318 if (Subtarget->supportsMinMaxDenormModes() ||
16328 for (
unsigned I = 0, E =
Op.getNumOperands();
I != E; ++
I) {
16343 for (
unsigned i = 0, e =
Op.getNumOperands(); i != e; ++i) {
16377 if (
Op.getValueType() == MVT::i16) {
16389 unsigned IntrinsicID =
Op.getConstantOperandVal(0);
16391 switch (IntrinsicID) {
16392 case Intrinsic::amdgcn_cvt_pkrtz:
16393 case Intrinsic::amdgcn_cubeid:
16394 case Intrinsic::amdgcn_frexp_mant:
16395 case Intrinsic::amdgcn_fdot2:
16396 case Intrinsic::amdgcn_rcp:
16397 case Intrinsic::amdgcn_rsq:
16398 case Intrinsic::amdgcn_rsq_clamp:
16399 case Intrinsic::amdgcn_rcp_legacy:
16400 case Intrinsic::amdgcn_rsq_legacy:
16401 case Intrinsic::amdgcn_trig_preop:
16402 case Intrinsic::amdgcn_tanh:
16403 case Intrinsic::amdgcn_log:
16404 case Intrinsic::amdgcn_exp2:
16405 case Intrinsic::amdgcn_sqrt:
16423 unsigned MaxDepth)
const {
16426 unsigned Opcode =
MI->getOpcode();
16428 if (Opcode == AMDGPU::G_FCANONICALIZE)
16431 std::optional<FPValueAndVReg> FCR;
16434 if (FCR->Value.isSignaling())
16436 if (!FCR->Value.isDenormal())
16447 case AMDGPU::G_FADD:
16448 case AMDGPU::G_FSUB:
16449 case AMDGPU::G_FMUL:
16450 case AMDGPU::G_FCEIL:
16451 case AMDGPU::G_FFLOOR:
16452 case AMDGPU::G_FRINT:
16453 case AMDGPU::G_FNEARBYINT:
16454 case AMDGPU::G_INTRINSIC_FPTRUNC_ROUND:
16455 case AMDGPU::G_INTRINSIC_TRUNC:
16456 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
16457 case AMDGPU::G_FMA:
16458 case AMDGPU::G_FMAD:
16459 case AMDGPU::G_FSQRT:
16460 case AMDGPU::G_FDIV:
16461 case AMDGPU::G_FREM:
16462 case AMDGPU::G_FPOW:
16463 case AMDGPU::G_FPEXT:
16464 case AMDGPU::G_FLOG:
16465 case AMDGPU::G_FLOG2:
16466 case AMDGPU::G_FLOG10:
16467 case AMDGPU::G_FPTRUNC:
16468 case AMDGPU::G_AMDGPU_RCP_IFLAG:
16469 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE0:
16470 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE1:
16471 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE2:
16472 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE3:
16474 case AMDGPU::G_FNEG:
16475 case AMDGPU::G_FABS:
16476 case AMDGPU::G_FCOPYSIGN:
16478 case AMDGPU::G_FMINNUM:
16479 case AMDGPU::G_FMAXNUM:
16480 case AMDGPU::G_FMINNUM_IEEE:
16481 case AMDGPU::G_FMAXNUM_IEEE:
16482 case AMDGPU::G_FMINIMUM:
16483 case AMDGPU::G_FMAXIMUM:
16484 case AMDGPU::G_FMINIMUMNUM:
16485 case AMDGPU::G_FMAXIMUMNUM: {
16486 if (Subtarget->supportsMinMaxDenormModes() ||
16493 case AMDGPU::G_BUILD_VECTOR:
16498 case AMDGPU::G_INTRINSIC:
16499 case AMDGPU::G_INTRINSIC_CONVERGENT:
16501 case Intrinsic::amdgcn_fmul_legacy:
16502 case Intrinsic::amdgcn_fmad_ftz:
16503 case Intrinsic::amdgcn_sqrt:
16504 case Intrinsic::amdgcn_fmed3:
16505 case Intrinsic::amdgcn_sin:
16506 case Intrinsic::amdgcn_cos:
16507 case Intrinsic::amdgcn_log:
16508 case Intrinsic::amdgcn_exp2:
16509 case Intrinsic::amdgcn_log_clamp:
16510 case Intrinsic::amdgcn_rcp:
16511 case Intrinsic::amdgcn_rcp_legacy:
16512 case Intrinsic::amdgcn_rsq:
16513 case Intrinsic::amdgcn_rsq_clamp:
16514 case Intrinsic::amdgcn_rsq_legacy:
16515 case Intrinsic::amdgcn_div_scale:
16516 case Intrinsic::amdgcn_div_fmas:
16517 case Intrinsic::amdgcn_div_fixup:
16518 case Intrinsic::amdgcn_fract:
16519 case Intrinsic::amdgcn_cvt_pkrtz:
16520 case Intrinsic::amdgcn_cubeid:
16521 case Intrinsic::amdgcn_cubema:
16522 case Intrinsic::amdgcn_cubesc:
16523 case Intrinsic::amdgcn_cubetc:
16524 case Intrinsic::amdgcn_frexp_mant:
16525 case Intrinsic::amdgcn_fdot2:
16526 case Intrinsic::amdgcn_trig_preop:
16527 case Intrinsic::amdgcn_tanh:
16546 if (
C.isDenormal()) {
16559 if (
C.isSignaling()) {
16574SITargetLowering::performFCanonicalizeCombine(
SDNode *
N,
16575 DAGCombinerInfo &DCI)
const {
16576 SelectionDAG &DAG = DCI.DAG;
16578 EVT VT =
N->getValueType(0);
16587 return getCanonicalConstantFP(DAG, SDLoc(
N), VT, CFP->getValueAPF());
16599 SDValue NewElts[2];
16602 EVT EltVT =
Lo.getValueType();
16611 for (
unsigned I = 0;
I != 2; ++
I) {
16615 getCanonicalConstantFP(DAG, SL, EltVT, CFP->getValueAPF());
16616 }
else if (
Op.isUndef()) {
16651 return AMDGPUISD::FMAX3;
16653 return AMDGPUISD::FMAXIMUM3;
16655 return AMDGPUISD::SMAX3;
16657 return AMDGPUISD::UMAX3;
16661 return AMDGPUISD::FMIN3;
16663 return AMDGPUISD::FMINIMUM3;
16665 return AMDGPUISD::SMIN3;
16667 return AMDGPUISD::UMIN3;
16688 if (!MinK || !MaxK)
16700 unsigned Med3Opc =
Signed ? AMDGPUISD::SMED3 : AMDGPUISD::UMED3;
16701 if (VT == MVT::i32 || (VT == MVT::i16 && Subtarget->hasMed3_16()))
16702 return DAG.
getNode(Med3Opc, SL, VT, Src, MaxVal, MinVal);
16726 bool IsKnownNoNaNs)
const {
16762 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
16768 if (
Info->getMode().DX10Clamp) {
16777 if (VT == MVT::f32 || (VT == MVT::f16 && Subtarget->hasMed3_16())) {
16791 SDValue(K0, 0), SDValue(K1, 0));
16809 case AMDGPUISD::FMIN_LEGACY:
16810 case AMDGPUISD::FMAX_LEGACY:
16811 return (VT == MVT::f32) || (VT == MVT::f16 && Subtarget.
hasMin3Max3_16()) ||
16812 (VT == MVT::v2f16 && Subtarget.hasMin3Max3PKF16());
16815 return (VT == MVT::f32 && Subtarget.hasMinimum3Maximum3F32()) ||
16816 (VT == MVT::f16 && Subtarget.hasMinimum3Maximum3F16()) ||
16817 (VT == MVT::v2f16 && Subtarget.hasMinimum3Maximum3PKF16());
16822 return (VT == MVT::i32) || (VT == MVT::i16 && Subtarget.
hasMin3Max3_16());
16831 DAGCombinerInfo &DCI)
const {
16832 SelectionDAG &DAG = DCI.DAG;
16843 auto IsTreeWithCombinableChildren = [
Opc](SDValue
Op) {
16844 return (
Op.getOperand(0).getOpcode() ==
Opc &&
16845 Op.getOperand(0).hasOneUse()) ||
16847 Op.getOperand(1).hasOneUse());
16852 bool HasCombinableTreeChild =
16853 CanTreeCombineApply && (IsTreeWithCombinableChildren(Op0) ||
16854 IsTreeWithCombinableChildren(Op1));
16863 if (CanTreeCombineApply && !HasCombinableTreeChild) {
16901 if (
Known.isNonZero() &&
Known.Zero.getBoolValue())
16909 if (SDValue Med3 = performIntMed3ImmCombine(
16914 if (SDValue Med3 = performIntMed3ImmCombine(
16920 if (SDValue Med3 = performIntMed3ImmCombine(
16925 if (SDValue Med3 = performIntMed3ImmCombine(
16938 (
Opc == AMDGPUISD::FMIN_LEGACY &&
16939 Op0.
getOpcode() == AMDGPUISD::FMAX_LEGACY)) &&
16940 (VT == MVT::f32 || VT == MVT::f64 ||
16941 (VT == MVT::f16 && Subtarget->has16BitInsts()) ||
16942 (VT == MVT::bf16 && Subtarget->hasBF16PackedInsts()) ||
16943 (VT == MVT::v2bf16 && Subtarget->hasBF16PackedInsts()) ||
16944 (VT == MVT::v2f16 && Subtarget->hasVOP3PInsts())) &&
16946 if (SDValue Res = performFPMed3ImmCombine(DAG, SDLoc(
N), Op0, Op1,
16947 N->getFlags().hasNoNaNs()))
16954 const SDNodeFlags
Flags =
N->getFlags();
16956 !Subtarget->hasIEEEMinimumMaximumInsts() &&
16960 return DAG.
getNode(NewOpc, SDLoc(
N), VT, Op0, Op1, Flags);
16970 return (CA->isPosZero() && CB->isOne()) ||
16971 (CA->isOne() && CB->isPosZero());
16980 DAGCombinerInfo &DCI)
const {
16981 EVT VT =
N->getValueType(0);
16985 SelectionDAG &DAG = DCI.DAG;
16988 SDValue
Src0 =
N->getOperand(0);
16989 SDValue
Src1 =
N->getOperand(1);
16990 SDValue
Src2 =
N->getOperand(2);
16996 return DAG.
getNode(AMDGPUISD::CLAMP, SL, VT, Src2);
17000 const SIMachineFunctionInfo *
Info = MF.
getInfo<SIMachineFunctionInfo>();
17004 if (
Info->getMode().DX10Clamp) {
17017 return DAG.
getNode(AMDGPUISD::CLAMP, SL, VT, Src0);
17024 DAGCombinerInfo &DCI)
const {
17025 SDValue
Src0 =
N->getOperand(0);
17026 SDValue
Src1 =
N->getOperand(1);
17027 if (
Src0.isUndef() &&
Src1.isUndef())
17028 return DCI.DAG.getUNDEF(
N->getValueType(0));
17036 bool IsDivergentIdx,
17041 unsigned VecSize = EltSize * NumElem;
17044 if (VecSize <= 64 && EltSize < 32)
17053 if (IsDivergentIdx)
17057 unsigned NumInsts = NumElem +
17058 ((EltSize + 31) / 32) * NumElem ;
17062 if (Subtarget->useVGPRIndexMode())
17063 return NumInsts <= 16;
17067 if (Subtarget->hasMovrel())
17068 return NumInsts <= 15;
17074 SDValue Idx =
N->getOperand(
N->getNumOperands() - 1);
17085 EltSize, NumElem, Idx->isDivergent(),
getSubtarget());
17089SITargetLowering::performExtractVectorEltCombine(
SDNode *
N,
17090 DAGCombinerInfo &DCI)
const {
17096 EVT ResVT =
N->getValueType(0);
17120 if (!
C ||
C->getZExtValue() != 0x1f)
17136 if (Vec.
hasOneUse() && DCI.isBeforeLegalize() && VecEltVT == ResVT) {
17138 SDValue
Idx =
N->getOperand(1);
17164 DCI.AddToWorklist(Elt0.
getNode());
17165 DCI.AddToWorklist(Elt1.
getNode());
17174 SDValue
Idx =
N->getOperand(1);
17196 if (KImm && KImm->getValueType(0).getSizeInBits() == 64) {
17197 uint64_t KImmValue = KImm->getZExtValue();
17199 (KImmValue >> (32 *
Idx->getZExtValue())) & 0xffffffff, SL, MVT::i32);
17202 if (KFPImm && KFPImm->getValueType(0).getSizeInBits() == 64) {
17204 KFPImm->getValueAPF().bitcastToAPInt().getZExtValue();
17205 return DAG.
getConstant((KFPImmValue >> (32 *
Idx->getZExtValue())) &
17211 if (!DCI.isBeforeLegalize())
17218 VecSize > 32 && VecSize % 32 == 0 && Idx) {
17221 unsigned BitIndex =
Idx->getZExtValue() * VecEltSize;
17222 unsigned EltIdx = BitIndex / 32;
17223 unsigned LeftoverBitIdx = BitIndex % 32;
17227 DCI.AddToWorklist(Cast.
getNode());
17231 DCI.AddToWorklist(Elt.
getNode());
17234 DCI.AddToWorklist(Srl.
getNode());
17238 DCI.AddToWorklist(Trunc.
getNode());
17240 if (VecEltVT == ResVT) {
17252SITargetLowering::performInsertVectorEltCombine(
SDNode *
N,
17253 DAGCombinerInfo &DCI)
const {
17255 SDValue
Idx =
N->getOperand(2);
17264 SelectionDAG &DAG = DCI.DAG;
17267 EVT IdxVT =
Idx.getValueType();
17284 Src.getOperand(0).getValueType() == MVT::f16) {
17285 return Src.getOperand(0);
17289 APFloat Val = CFP->getValueAPF();
17290 bool LosesInfo =
true;
17300 DAGCombinerInfo &DCI)
const {
17301 assert(Subtarget->has16BitInsts() && !Subtarget->hasMed3_16() &&
17302 "combine only useful on gfx8");
17304 SDValue TruncSrc =
N->getOperand(0);
17305 EVT VT =
N->getValueType(0);
17306 if (VT != MVT::f16)
17309 if (TruncSrc.
getOpcode() != AMDGPUISD::FMED3 ||
17313 SelectionDAG &DAG = DCI.DAG;
17344unsigned SITargetLowering::getFusedOpcode(
const SelectionDAG &DAG,
17346 const SDNode *N1)
const {
17351 if (((VT == MVT::f32 &&
17353 (VT == MVT::f16 && Subtarget->hasMadF16() &&
17370 EVT VT =
N->getValueType(0);
17371 if (VT != MVT::i32 && VT != MVT::i64)
17377 unsigned Opc =
N->getOpcode();
17399 SDValue Add1 = DAG.
getNode(
Opc, SL, VT, Op0, Op1);
17432 if (!Const ||
Hi_32(Const->getZExtValue()) !=
uint32_t(-1))
17451 DAGCombinerInfo &DCI)
const {
17454 SelectionDAG &DAG = DCI.DAG;
17455 EVT VT =
N->getValueType(0);
17457 SDValue
LHS =
N->getOperand(0);
17458 SDValue
RHS =
N->getOperand(1);
17465 if (!
N->isDivergent() && Subtarget->hasSMulHi())
17469 if (NumBits <= 32 || NumBits > 64)
17480 if (!Subtarget->hasFullRate64Ops()) {
17481 unsigned NumUsers = 0;
17482 for (SDNode *User :
LHS->
users()) {
17485 if (!
User->isAnyAdd())
17496 SDValue MulLHS =
LHS.getOperand(0);
17497 SDValue MulRHS =
LHS.getOperand(1);
17498 SDValue AddRHS =
RHS;
17509 bool MulSignedLo =
false;
17510 if (!MulLHSUnsigned32 || !MulRHSUnsigned32) {
17519 if (VT != MVT::i64) {
17542 getMad64_32(DAG, SL, MVT::i64, MulLHSLo, MulRHSLo, AddRHS, MulSignedLo);
17544 if (!MulSignedLo && (!MulLHSUnsigned32 || !MulRHSUnsigned32)) {
17545 auto [AccumLo, AccumHi] = DAG.
SplitScalar(Accum, SL, MVT::i32, MVT::i32);
17547 if (!MulLHSUnsigned32) {
17550 SDValue MulHi = DAG.
getNode(
ISD::MUL, SL, MVT::i32, MulLHSHi, MulRHSLo);
17554 if (!MulRHSUnsigned32) {
17557 SDValue MulHi = DAG.
getNode(
ISD::MUL, SL, MVT::i32, MulLHSLo, MulRHSHi);
17565 if (VT != MVT::i64)
17571SITargetLowering::foldAddSub64WithZeroLowBitsTo32(
SDNode *
N,
17572 DAGCombinerInfo &DCI)
const {
17573 SDValue
RHS =
N->getOperand(1);
17582 SelectionDAG &DAG = DCI.DAG;
17584 SDValue
LHS =
N->getOperand(0);
17597 unsigned Opcode =
N->getOpcode();
17601 DAG.
getNode(Opcode, SL, MVT::i32,
Hi, ConstHi32,
N->getFlags());
17612static std::optional<ByteProvider<SDValue>>
17615 if (!Byte0 || Byte0->isConstantZero()) {
17616 return std::nullopt;
17619 if (Byte1 && !Byte1->isConstantZero()) {
17620 return std::nullopt;
17626 unsigned FirstCs =
First & 0x0c0c0c0c;
17627 unsigned SecondCs = Second & 0x0c0c0c0c;
17628 unsigned FirstNoCs =
First & ~0x0c0c0c0c;
17629 unsigned SecondNoCs = Second & ~0x0c0c0c0c;
17631 assert((FirstCs & 0xFF) | (SecondCs & 0xFF));
17632 assert((FirstCs & 0xFF00) | (SecondCs & 0xFF00));
17633 assert((FirstCs & 0xFF0000) | (SecondCs & 0xFF0000));
17634 assert((FirstCs & 0xFF000000) | (SecondCs & 0xFF000000));
17636 return (FirstNoCs | SecondNoCs) | (FirstCs & SecondCs);
17650 assert(Src0.Src.has_value() && Src1.Src.has_value());
17653 Src0s.
push_back({*Src0.Src, ((Src0.SrcOffset % 4) << 24) + 0x0c0c0c,
17654 Src0.SrcOffset / 4});
17655 Src1s.
push_back({*Src1.Src, ((Src1.SrcOffset % 4) << 24) + 0x0c0c0c,
17656 Src1.SrcOffset / 4});
17660 for (
int BPI = 0; BPI < 2; BPI++) {
17663 BPP = {Src1, Src0};
17665 unsigned ZeroMask = 0x0c0c0c0c;
17666 unsigned FMask = 0xFF << (8 * (3 - Step));
17668 unsigned FirstMask =
17669 (BPP.first.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask);
17670 unsigned SecondMask =
17671 (BPP.second.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask);
17675 int FirstGroup = -1;
17676 for (
int I = 0;
I < 2;
I++) {
17678 auto MatchesFirst = [&BPP](
DotSrc &IterElt) {
17679 return IterElt.SrcOp == *BPP.first.Src &&
17680 (IterElt.DWordOffset == (BPP.first.SrcOffset / 4));
17684 if (Match != Srcs.
end()) {
17685 Match->PermMask =
addPermMasks(FirstMask, Match->PermMask);
17690 if (FirstGroup != -1) {
17692 auto MatchesSecond = [&BPP](
DotSrc &IterElt) {
17693 return IterElt.SrcOp == *BPP.second.Src &&
17694 (IterElt.DWordOffset == (BPP.second.SrcOffset / 4));
17697 if (Match != Srcs.
end()) {
17698 Match->PermMask =
addPermMasks(SecondMask, Match->PermMask);
17700 Srcs.
push_back({*BPP.second.Src, SecondMask, BPP.second.SrcOffset / 4});
17708 unsigned ZeroMask = 0x0c0c0c0c;
17709 unsigned FMask = 0xFF << (8 * (3 - Step));
17713 ((Src0.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask)),
17714 Src0.SrcOffset / 4});
17717 ((Src1.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask)),
17718 Src1.SrcOffset / 4});
17726 if (Srcs.
size() == 1) {
17727 auto *Elt = Srcs.
begin();
17731 if (Elt->PermMask == 0x3020100)
17734 return DAG.
getNode(AMDGPUISD::PERM, SL, MVT::i32, EltOp, EltOp,
17738 auto *FirstElt = Srcs.
begin();
17739 auto *SecondElt = std::next(FirstElt);
17746 auto FirstMask = FirstElt->PermMask;
17747 auto SecondMask = SecondElt->PermMask;
17749 unsigned FirstCs = FirstMask & 0x0c0c0c0c;
17750 unsigned FirstPlusFour = FirstMask | 0x04040404;
17753 FirstMask = (FirstPlusFour & 0x0F0F0F0F) | FirstCs;
17765 FirstElt = std::next(SecondElt);
17766 if (FirstElt == Srcs.
end())
17769 SecondElt = std::next(FirstElt);
17772 if (SecondElt == Srcs.
end()) {
17777 DAG.
getNode(AMDGPUISD::PERM, SL, MVT::i32, EltOp, EltOp,
17778 DAG.
getConstant(FirstElt->PermMask, SL, MVT::i32)));
17784 return Perms.
size() == 2
17790 for (
auto &[EntryVal, EntryMask, EntryOffset] : Srcs) {
17791 EntryMask = EntryMask >> ((4 - ChainLength) * 8);
17792 auto ZeroMask = ChainLength == 2 ? 0x0c0c0000 : 0x0c000000;
17793 EntryMask += ZeroMask;
17798 auto Opcode =
Op.getOpcode();
17800 return (Opcode ==
ISD::MUL || Opcode == AMDGPUISD::MUL_U24 ||
17801 Opcode == AMDGPUISD::MUL_I24);
17804static std::optional<bool>
17815 bool S0IsSigned = Known0.countMinLeadingOnes() > 0;
17818 bool S1IsSigned = Known1.countMinLeadingOnes() > 0;
17820 assert(!(S0IsUnsigned && S0IsSigned));
17821 assert(!(S1IsUnsigned && S1IsSigned));
17829 if ((S0IsUnsigned && S1IsUnsigned) || (S0IsSigned && S1IsSigned))
17835 if ((S0IsUnsigned && S1IsSigned) || (S0IsSigned && S1IsUnsigned))
17836 return std::nullopt;
17848 if ((S0IsSigned && !(S1IsSigned || S1IsUnsigned)) ||
17849 ((S1IsSigned && !(S0IsSigned || S0IsUnsigned))))
17854 if ((!(S1IsSigned || S1IsUnsigned) && !(S0IsSigned || S0IsUnsigned)))
17860 if ((S0IsUnsigned && !(S1IsSigned || S1IsUnsigned)) ||
17861 ((S1IsUnsigned && !(S0IsSigned || S0IsUnsigned))))
17862 return std::nullopt;
17868 DAGCombinerInfo &DCI)
const {
17869 SelectionDAG &DAG = DCI.DAG;
17870 EVT VT =
N->getValueType(0);
17872 SDValue
LHS =
N->getOperand(0);
17873 SDValue
RHS =
N->getOperand(1);
17876 if (Subtarget->hasMad64_32()) {
17877 if (SDValue Folded = tryFoldToMad64_32(
N, DCI))
17882 if (SDValue V = reassociateScalarOps(
N, DAG)) {
17886 if (VT == MVT::i64) {
17887 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(
N, DCI))
17894 (Subtarget->hasDot1Insts() || Subtarget->hasDot8Insts())) {
17895 SDValue TempNode(
N, 0);
17896 std::optional<bool> IsSigned;
17902 int ChainLength = 0;
17903 for (
int I = 0;
I < 4;
I++) {
17915 TempNode->getOperand(MulIdx), *Src0, *Src1,
17916 TempNode->getOperand(MulIdx)->getOperand(0),
17917 TempNode->getOperand(MulIdx)->getOperand(1), DAG);
17921 IsSigned = *IterIsSigned;
17922 if (*IterIsSigned != *IsSigned)
17925 auto AddIdx = 1 - MulIdx;
17928 if (
I == 2 &&
isMul(TempNode->getOperand(AddIdx))) {
17929 Src2s.
push_back(TempNode->getOperand(AddIdx));
17939 TempNode->getOperand(AddIdx), *Src0, *Src1,
17940 TempNode->getOperand(AddIdx)->getOperand(0),
17941 TempNode->getOperand(AddIdx)->getOperand(1), DAG);
17945 if (*IterIsSigned != *IsSigned)
17949 ChainLength =
I + 2;
17953 TempNode = TempNode->getOperand(AddIdx);
17955 ChainLength =
I + 1;
17957 if (TempNode.getOpcode() !=
ISD::ADD)
17959 LHS = TempNode->getOperand(0);
17960 RHS = TempNode->getOperand(1);
17963 if (ChainLength < 2)
17969 if (ChainLength < 4) {
17979 bool UseOriginalSrc =
false;
17980 if (ChainLength == 4 && Src0s.
size() == 1 && Src1s.
size() == 1 &&
17981 Src0s.
begin()->PermMask == Src1s.
begin()->PermMask &&
17982 Src0s.
begin()->SrcOp.getValueSizeInBits() >= 32 &&
17983 Src1s.
begin()->SrcOp.getValueSizeInBits() >= 32) {
17984 SmallVector<unsigned, 4> SrcBytes;
17985 auto Src0Mask = Src0s.
begin()->PermMask;
17986 SrcBytes.
push_back(Src0Mask & 0xFF000000);
17987 bool UniqueEntries =
true;
17988 for (
auto I = 1;
I < 4;
I++) {
17989 auto NextByte = Src0Mask & (0xFF << ((3 -
I) * 8));
17992 UniqueEntries =
false;
17998 if (UniqueEntries) {
17999 UseOriginalSrc =
true;
18001 auto *FirstElt = Src0s.
begin();
18005 auto *SecondElt = Src1s.
begin();
18007 SecondElt->DWordOffset);
18016 if (!UseOriginalSrc) {
18023 DAG.
getExtOrTrunc(*IsSigned, Src2s[ChainLength - 1], SL, MVT::i32);
18026 : Intrinsic::amdgcn_udot4,
18036 if (VT != MVT::i32 || !DCI.isAfterLegalizeDAG())
18041 unsigned Opc =
LHS.getOpcode();
18053 auto Cond =
RHS.getOperand(0);
18058 SDVTList VTList = DAG.
getVTList(MVT::i32, MVT::i1);
18067 SDValue
Args[] = {
LHS,
RHS.getOperand(0),
RHS.getOperand(2)};
18075 DAGCombinerInfo &DCI)
const {
18076 SelectionDAG &DAG = DCI.DAG;
18078 EVT VT =
N->getValueType(0);
18091 SDNodeFlags ShlFlags = N1->
getFlags();
18095 SDNodeFlags NewShlFlags =
18100 DCI.AddToWorklist(Inner.
getNode());
18107 if (Subtarget->hasMad64_32()) {
18108 if (SDValue Folded = tryFoldToMad64_32(
N, DCI))
18117 if (VT == MVT::i64) {
18118 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(
N, DCI))
18131 if (!YIsConstant && !ZIsConstant && !
X->isDivergent() &&
18132 Y->isDivergent() !=
Z->isDivergent()) {
18141 if (
Y->isDivergent())
18144 SDNodeFlags ReassocFlags =
18147 DCI.AddToWorklist(UniformInner.
getNode());
18159 DAGCombinerInfo &DCI)
const {
18160 SelectionDAG &DAG = DCI.DAG;
18161 EVT VT =
N->getValueType(0);
18163 if (VT == MVT::i64) {
18164 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(
N, DCI))
18168 if (VT != MVT::i32)
18172 SDValue
LHS =
N->getOperand(0);
18173 SDValue
RHS =
N->getOperand(1);
18177 unsigned Opc =
RHS.getOpcode();
18184 auto Cond =
RHS.getOperand(0);
18189 SDVTList VTList = DAG.
getVTList(MVT::i32, MVT::i1);
18200 SDValue
Args[] = {
LHS.getOperand(0),
RHS,
LHS.getOperand(2)};
18206 SDValue CtlzSrc =
LHS.getOperand(0);
18215 ConstantSDNode *ShiftAmt =
18217 unsigned BitWidth =
X.getValueType().getScalarSizeInBits();
18228 DAGCombinerInfo &DCI)
const {
18232 SelectionDAG &DAG = DCI.DAG;
18233 EVT VT =
N->getValueType(0);
18236 SDValue
LHS =
N->getOperand(0);
18237 SDValue
RHS =
N->getOperand(1);
18244 SDValue
A =
LHS.getOperand(0);
18245 if (
A ==
LHS.getOperand(1)) {
18246 unsigned FusedOp = getFusedOpcode(DAG,
N,
LHS.getNode());
18247 if (FusedOp != 0) {
18249 return DAG.
getNode(FusedOp, SL, VT,
A, Two,
RHS);
18256 SDValue
A =
RHS.getOperand(0);
18257 if (
A ==
RHS.getOperand(1)) {
18258 unsigned FusedOp = getFusedOpcode(DAG,
N,
RHS.getNode());
18259 if (FusedOp != 0) {
18261 return DAG.
getNode(FusedOp, SL, VT,
A, Two,
LHS);
18270 DAGCombinerInfo &DCI)
const {
18274 SelectionDAG &DAG = DCI.DAG;
18276 EVT VT =
N->getValueType(0);
18284 SDValue
LHS =
N->getOperand(0);
18285 SDValue
RHS =
N->getOperand(1);
18288 SDValue
A =
LHS.getOperand(0);
18289 if (
A ==
LHS.getOperand(1)) {
18290 unsigned FusedOp = getFusedOpcode(DAG,
N,
LHS.getNode());
18291 if (FusedOp != 0) {
18295 return DAG.
getNode(FusedOp, SL, VT,
A, Two, NegRHS);
18303 SDValue
A =
RHS.getOperand(0);
18304 if (
A ==
RHS.getOperand(1)) {
18305 unsigned FusedOp = getFusedOpcode(DAG,
N,
RHS.getNode());
18306 if (FusedOp != 0) {
18308 return DAG.
getNode(FusedOp, SL, VT,
A, NegTwo,
LHS);
18317 DAGCombinerInfo &DCI)
const {
18318 SelectionDAG &DAG = DCI.DAG;
18320 EVT VT =
N->getValueType(0);
18322 if (VT != MVT::f16 && VT != MVT::bf16)
18325 SDValue
LHS =
N->getOperand(0);
18326 SDValue
RHS =
N->getOperand(1);
18328 SDNodeFlags
Flags =
N->getFlags();
18329 SDNodeFlags RHSFlags =
RHS->getFlags();
18335 bool IsNegative =
false;
18336 if (CLHS->
isOne() || (IsNegative = CLHS->isMinusOne())) {
18341 SDValue SqrtOp =
RHS.getOperand(0);
18345 Rsq = DAG.
getNode(AMDGPUISD::RSQ, SL, VT, SqrtOp, Flags);
18346 }
else if (VT == MVT::f16) {
18355 DAG.
getNode(AMDGPUISD::RSQ, SL, MVT::f32, Ext, Flags);
18372 DAGCombinerInfo &DCI)
const {
18373 SelectionDAG &DAG = DCI.DAG;
18374 EVT VT =
N->getValueType(0);
18378 if (!
N->isDivergent() &&
getSubtarget()->hasSALUFloatInsts() &&
18379 (ScalarVT == MVT::f32 || ScalarVT == MVT::f16)) {
18384 SDValue
LHS =
N->getOperand(0);
18385 SDValue
RHS =
N->getOperand(1);
18394 if ((ScalarVT == MVT::f64 || ScalarVT == MVT::f32 || ScalarVT == MVT::f16) &&
18399 const ConstantFPSDNode *FalseNode =
18409 if (ScalarVT == MVT::f32 &&
18415 if (TrueNodeExpVal == INT_MIN)
18418 if (FalseNodeExpVal == INT_MIN)
18422 SDValue SelectNode =
18438 DAGCombinerInfo &DCI)
const {
18439 SelectionDAG &DAG = DCI.DAG;
18440 EVT VT =
N->getValueType(0);
18443 if (!Subtarget->hasDot10Insts() || VT != MVT::f32)
18450 SDValue
FMA =
N->getOperand(2);
18471 bool AllowInaccuracy =
N->getFlags().hasApproximateFuncs() &&
18472 FMA->getFlags().hasApproximateFuncs();
18473 if (!AllowInaccuracy) {
18476 if (Subtarget->dot2UnconditionalFlush()) {
18488 if (
N->getFlags().hasAllowContract() &&
FMA->getFlags().hasAllowContract()) {
18499 SDValue FMAOp1 =
FMA.getOperand(0);
18500 SDValue FMAOp2 =
FMA.getOperand(1);
18501 SDValue FMAAcc =
FMA.getOperand(2);
18524 if (Vec1 == Vec2 || Vec3 == Vec4)
18530 if ((Vec1 == Vec3 && Vec2 == Vec4) || (Vec1 == Vec4 && Vec2 == Vec3)) {
18531 return DAG.
getNode(AMDGPUISD::FDOT2, SL, MVT::f32, Vec1, Vec2, FMAAcc,
18574 EVT VT =
LHS.getValueType();
18575 assert(VT == MVT::f64 &&
"Incorrect operand type!");
18607 if (CC ==
ISD::SETOEQ && LHSMaybeNaN && RHSMaybeNaN)
18611 if (CC ==
ISD::SETUEQ && (LHSMaybeNaN || RHSMaybeNaN))
18615 if (CC ==
ISD::SETONE && (LHSMaybeNaN || RHSMaybeNaN))
18619 if (CC ==
ISD::SETUNE && LHSMaybeNaN && RHSMaybeNaN)
18622 const std::optional<bool> KnownEq =
18651 if (CC ==
ISD::SETULT && (LHSMaybeNaN || RHSMaybeNaN))
18655 if (CC ==
ISD::SETOGE && (LHSMaybeNaN || RHSMaybeNaN))
18663 const std::optional<bool> KnownUge =
18688 if (CC ==
ISD::SETOLE && (LHSMaybeNaN || RHSMaybeNaN))
18702 if (CC ==
ISD::SETUGT && (LHSMaybeNaN || RHSMaybeNaN))
18705 const std::optional<bool> KnownUle =
18728 DAGCombinerInfo &DCI)
const {
18729 SelectionDAG &DAG = DCI.DAG;
18732 SDValue
LHS =
N->getOperand(0);
18733 SDValue
RHS =
N->getOperand(1);
18734 EVT VT =
LHS.getValueType();
18763 return LHS.getOperand(0);
18777 const APInt &CT =
LHS.getConstantOperandAPInt(1);
18778 const APInt &CF =
LHS.getConstantOperandAPInt(2);
18783 return DAG.
getNOT(SL,
LHS.getOperand(0), MVT::i1);
18786 return LHS.getOperand(0);
18807 if (VT == MVT::i64) {
18819 const std::optional<bool> KnownEq =
18827 const std::optional<bool> KnownEq =
18838 const std::optional<bool> KnownUge =
18858 const std::optional<bool> KnownUle =
18898 SDValue Op0 =
LHS.getOperand(0);
18899 SDValue Op1 =
LHS.getOperand(1);
18909 DAG.
getVTList(MVT::i32, MVT::i1), {Op0Lo, Op1Lo});
18911 SDValue CarryInHi = NodeLo.
getValue(1);
18914 {Op0Hi, Op1Hi, CarryInHi});
18916 SDValue ResultLo = NodeLo.
getValue(0);
18917 SDValue ResultHi = NodeHi.
getValue(0);
18919 SDValue JoinedResult =
18923 SDValue Overflow = NodeHi.
getValue(1);
18924 DCI.CombineTo(
LHS.getNode(), Result);
18928 if (VT != MVT::f32 && VT != MVT::f64 &&
18929 (!Subtarget->has16BitInsts() || VT != MVT::f16))
18944 const unsigned IsInfMask =
18946 const unsigned IsFiniteMask =
18951 return DAG.
getNode(AMDGPUISD::FP_CLASS, SL, MVT::i1,
LHS.getOperand(0),
18956 if (VT == MVT::f64) {
18967SITargetLowering::performCvtF32UByteNCombine(
SDNode *
N,
18968 DAGCombinerInfo &DCI)
const {
18969 SelectionDAG &DAG = DCI.DAG;
18971 unsigned Offset =
N->getOpcode() - AMDGPUISD::CVT_F32_UBYTE0;
18973 SDValue Src =
N->getOperand(0);
18974 SDValue Shift =
N->getOperand(0);
18990 unsigned ShiftOffset = 8 *
Offset;
18992 ShiftOffset -=
C->getZExtValue();
18994 ShiftOffset +=
C->getZExtValue();
18996 if (ShiftOffset < 32 && (ShiftOffset % 8) == 0) {
18997 return DAG.
getNode(AMDGPUISD::CVT_F32_UBYTE0 + ShiftOffset / 8, SL,
18998 MVT::f32, Shifted);
19009 DCI.AddToWorklist(
N);
19010 return SDValue(
N, 0);
19014 if (SDValue DemandedSrc =
19016 return DAG.
getNode(
N->getOpcode(), SL, MVT::f32, DemandedSrc);
19022 DAGCombinerInfo &DCI)
const {
19031 (
F.isNaN() && MF.
getInfo<SIMachineFunctionInfo>()->getMode().DX10Clamp)) {
19032 return DCI.DAG.getConstantFP(Zero, SDLoc(
N),
N->getValueType(0));
19037 return DCI.DAG.getConstantFP(One, SDLoc(
N),
N->getValueType(0));
19039 return getCanonicalConstantFP(DCI.DAG, SDLoc(
N),
N->getValueType(0),
F);
19048 if (V.getOpcode() ==
ISD::FFREXP && V.getResNo() == 1) {
19059SITargetLowering::performFrexpSelectCombine(
SDNode *
N,
19060 DAGCombinerInfo &DCI)
const {
19062 if (Subtarget->hasFractBug())
19065 SDValue
Cond =
N->getOperand(0);
19066 SDValue
TrueVal =
N->getOperand(1);
19074 bool CondSelectsZero;
19078 SDValue FrexpInput;
19082 CondSelectsZero =
true;
19083 }
else if (
isFrexpExp(TrueVal, FrexpInput)) {
19086 CondSelectsZero =
false;
19098 bool IsNonFiniteTest =
false;
19104 SDValue CondLHS =
Cond.getOperand(0);
19105 SDValue CondRHS =
Cond.getOperand(1);
19110 bool LHSMatchesFrexp =
19111 (CondLHS == FrexpInput) ||
19112 (LHSIsFabs &&
peekFPSignOps(FAbsInput) == FrexpInputStripped) ||
19114 bool RHSMatchesFrexp = (CondRHS == FrexpInput) ||
19122 SelectionDAG &DAG = DCI.DAG;
19123 if (LHSMatchesFrexp &&
19125 IsNonFiniteTest = CondSelectsZero;
19127 IsNonFiniteTest = CondSelectsZero;
19134 IsNonFiniteTest = CondSelectsZero;
19141 IsNonFiniteTest = !CondSelectsZero;
19147 SelectionDAG &DAG = DCI.DAG;
19148 if (LHSMatchesFrexp &&
19150 IsNonFiniteTest = !CondSelectsZero;
19152 IsNonFiniteTest = !CondSelectsZero;
19156 if (!IsNonFiniteTest)
19164 DAGCombinerInfo &DCI)
const {
19173 SDValue
Cond =
N->getOperand(0);
19174 SDValue
TrueVal =
N->getOperand(1);
19181 SDValue
LHS =
Cond.getOperand(0);
19182 SDValue
RHS =
Cond.getOperand(1);
19185 bool isFloatingPoint =
LHS.getValueType().isFloatingPoint();
19186 bool isInteger =
LHS.getValueType().isInteger();
19189 if (!isFloatingPoint && !isInteger)
19194 bool isNonEquality =
19196 if (!isEquality && !isNonEquality)
19199 SDValue ArgVal, ConstVal;
19213 if (isFloatingPoint) {
19215 if (!Val.
isNormal() || Subtarget->getInstrInfo()->isInlineConstant(Val))
19218 const std::optional<int64_t> Val =
19227 if (!(isEquality && TrueVal == ConstVal) &&
19228 !(isNonEquality && FalseVal == ConstVal))
19232 if (isFloatingPoint && isNonEquality && FalseVal == ConstVal &&
19233 !
Cond->getFlags().hasNoNaNs() && !DCI.DAG.isKnownNeverNaN(ArgVal))
19236 SDValue SelectLHS = (isEquality &&
TrueVal == ConstVal) ? ArgVal :
TrueVal;
19237 SDValue SelectRHS =
19240 SelectLHS, SelectRHS);
19245 switch (
N->getOpcode()) {
19267 if (
auto Res = promoteUniformOpToI32(
SDValue(
N, 0), DCI))
19277 switch (
N->getOpcode()) {
19279 return performAddCombine(
N, DCI);
19281 return performPtrAddCombine(
N, DCI);
19283 return performSubCombine(
N, DCI);
19285 return performFAddCombine(
N, DCI);
19287 return performFSubCombine(
N, DCI);
19289 return performFDivCombine(
N, DCI);
19291 return performFMulCombine(
N, DCI);
19293 return performSetCCCombine(
N, DCI);
19295 if (
auto Res = performFrexpSelectCombine(
N, DCI))
19297 if (
auto Res = performSelectCombine(
N, DCI))
19312 case AMDGPUISD::FMIN_LEGACY:
19313 case AMDGPUISD::FMAX_LEGACY:
19314 return performMinMaxCombine(
N, DCI);
19316 return performFMACombine(
N, DCI);
19318 return performAndCombine(
N, DCI);
19320 return performOrCombine(
N, DCI);
19323 if (
N->getValueType(0) == MVT::i32 &&
N->isDivergent() &&
19324 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
19330 return performXorCombine(
N, DCI);
19333 return performZeroOrAnyExtendCombine(
N, DCI);
19335 return performSignExtendInRegCombine(
N, DCI);
19336 case AMDGPUISD::FP_CLASS:
19337 return performClassCombine(
N, DCI);
19339 return performFCanonicalizeCombine(
N, DCI);
19340 case AMDGPUISD::RCP:
19341 return performRcpCombine(
N, DCI);
19343 case AMDGPUISD::FRACT:
19344 case AMDGPUISD::RSQ:
19345 case AMDGPUISD::RCP_LEGACY:
19346 case AMDGPUISD::RCP_IFLAG:
19347 case AMDGPUISD::RSQ_CLAMP: {
19356 return performUCharToFloatCombine(
N, DCI);
19358 return performFCopySignCombine(
N, DCI);
19359 case AMDGPUISD::CVT_F32_UBYTE0:
19360 case AMDGPUISD::CVT_F32_UBYTE1:
19361 case AMDGPUISD::CVT_F32_UBYTE2:
19362 case AMDGPUISD::CVT_F32_UBYTE3:
19363 return performCvtF32UByteNCombine(
N, DCI);
19364 case AMDGPUISD::FMED3:
19365 return performFMed3Combine(
N, DCI);
19366 case AMDGPUISD::CVT_PKRTZ_F16_F32:
19367 return performCvtPkRTZCombine(
N, DCI);
19368 case AMDGPUISD::CLAMP:
19369 return performClampCombine(
N, DCI);
19372 EVT VT =
N->getValueType(0);
19377 if (VT == MVT::v2bf16 && Subtarget->hasBF16InlineConstFromUpperFP32()) {
19380 C->getValueAPF().bitcastToAPInt().getSExtValue(),
19381 Subtarget->hasInv2PiInlineImm()))
19383 {
N->getOperand(0),
N->getOperand(0)});
19387 if (VT == MVT::v2i16 || VT == MVT::v2f16 || VT == MVT::v2bf16) {
19390 EVT EltVT = Src.getValueType();
19391 if (EltVT != MVT::i16)
19401 return performExtractVectorEltCombine(
N, DCI);
19403 return performInsertVectorEltCombine(
N, DCI);
19405 return performFPRoundCombine(
N, DCI);
19414 return performMemSDNodeCombine(MemNode, DCI);
19445 unsigned Opcode =
Node->getMachineOpcode();
19448 int D16Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::d16) - 1;
19449 if (D16Idx >= 0 &&
Node->getConstantOperandVal(D16Idx))
19452 SDNode *
Users[5] = {
nullptr};
19454 unsigned DmaskIdx =
19455 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::dmask) - 1;
19456 unsigned OldDmask =
Node->getConstantOperandVal(DmaskIdx);
19457 unsigned NewDmask = 0;
19458 unsigned TFEIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::tfe) - 1;
19459 unsigned LWEIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::lwe) - 1;
19460 bool UsesTFC = (int(TFEIdx) >= 0 &&
Node->getConstantOperandVal(TFEIdx)) ||
19461 (
int(LWEIdx) >= 0 &&
Node->getConstantOperandVal(LWEIdx));
19462 unsigned TFCLane = 0;
19463 bool HasChain =
Node->getNumValues() > 1;
19465 if (OldDmask == 0) {
19473 TFCLane = OldBitsSet;
19477 for (SDUse &Use :
Node->uses()) {
19480 if (
Use.getResNo() != 0)
19483 SDNode *
User =
Use.getUser();
19486 if (!
User->isMachineOpcode() ||
19487 User->getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG)
19499 if (UsesTFC && Lane == TFCLane) {
19504 for (
unsigned i = 0, Dmask = OldDmask; (i <= Lane) && (Dmask != 0); i++) {
19506 Dmask &= ~(1 << Comp);
19514 NewDmask |= 1 << Comp;
19519 bool NoChannels = !NewDmask;
19526 if (OldBitsSet == 1)
19532 if (NewDmask == OldDmask)
19541 unsigned NewChannels = BitsSet + UsesTFC;
19545 assert(NewOpcode != -1 &&
19546 NewOpcode !=
static_cast<int>(
Node->getMachineOpcode()) &&
19547 "failed to find equivalent MIMG op");
19555 MVT SVT =
Node->getValueType(0).getVectorElementType().getSimpleVT();
19557 MVT ResultVT = NewChannels == 1
19560 : NewChannels == 5 ? 8
19562 SDVTList NewVTList =
19565 MachineSDNode *NewNode =
19574 if (NewChannels == 1) {
19584 for (
unsigned i = 0, Idx = AMDGPU::sub0; i < 5; ++i) {
19589 if (i || !NoChannels)
19594 if (NewUser != User) {
19604 Idx = AMDGPU::sub1;
19607 Idx = AMDGPU::sub2;
19610 Idx = AMDGPU::sub3;
19613 Idx = AMDGPU::sub4;
19624 Op =
Op.getOperand(0);
19649 Node->getOperand(0), SL, VReg, SrcVal,
19655 return ToResultReg.
getNode();
19660 for (
unsigned i = 0; i <
Node->getNumOperands(); ++i) {
19662 Ops.push_back(
Node->getOperand(i));
19668 Node->getOperand(i).getValueType(),
19669 Node->getOperand(i)),
19681 unsigned Opcode =
Node->getMachineOpcode();
19683 if (
TII->isImage(Opcode) && !
TII->get(Opcode).mayStore() &&
19684 !
TII->isGather4(Opcode) &&
19686 return adjustWritemask(
Node, DAG);
19689 if (Opcode == AMDGPU::INSERT_SUBREG || Opcode == AMDGPU::REG_SEQUENCE) {
19695 case AMDGPU::V_DIV_SCALE_F32_e64:
19696 case AMDGPU::V_DIV_SCALE_F64_e64: {
19704 if ((Src0.isMachineOpcode() &&
19705 Src0.getMachineOpcode() != AMDGPU::IMPLICIT_DEF) &&
19706 (Src0 == Src1 || Src0 == Src2))
19709 MVT VT = Src0.getValueType().getSimpleVT();
19721 if (Src0.isMachineOpcode() &&
19722 Src0.getMachineOpcode() == AMDGPU::IMPLICIT_DEF) {
19723 if (Src1.isMachineOpcode() &&
19724 Src1.getMachineOpcode() != AMDGPU::IMPLICIT_DEF)
19726 else if (Src2.isMachineOpcode() &&
19727 Src2.getMachineOpcode() != AMDGPU::IMPLICIT_DEF)
19730 assert(Src1.getMachineOpcode() == AMDGPU::IMPLICIT_DEF);
19762 AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::vdata);
19763 unsigned InitIdx = 0;
19765 if (
TII->isImage(
MI)) {
19773 unsigned TFEVal = TFE ? TFE->
getImm() : 0;
19774 unsigned LWEVal = LWE ? LWE->
getImm() : 0;
19775 unsigned D16Val = D16 ? D16->getImm() : 0;
19777 if (!TFEVal && !LWEVal)
19788 assert(MO_Dmask &&
"Expected dmask operand in instruction");
19790 unsigned dmask = MO_Dmask->
getImm();
19795 bool Packed = !Subtarget->hasUnpackedD16VMem();
19797 InitIdx = D16Val && Packed ? ((ActiveLanes + 1) >> 1) + 1 : ActiveLanes + 1;
19804 uint32_t DstSize =
TRI.getRegSizeInBits(*DstRC) / 32;
19805 if (DstSize < InitIdx)
19809 InitIdx =
TRI.getRegSizeInBits(*DstRC) / 32;
19818 unsigned NewDst = 0;
19823 unsigned SizeLeft = Subtarget->usePRTStrictNull() ? InitIdx : 1;
19824 unsigned CurrIdx = Subtarget->usePRTStrictNull() ? 0 : (InitIdx - 1);
19827 for (; SizeLeft; SizeLeft--, CurrIdx++) {
19848 MI.tieOperands(DstIdx,
MI.getNumOperands() - 1);
19860 if (
TII->isVOP3(
MI.getOpcode())) {
19862 TII->legalizeOperandsVOP3(MRI,
MI);
19864 if (
TII->isMAI(
MI)) {
19869 int Src0Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
19870 AMDGPU::OpName::scale_src0);
19871 if (Src0Idx != -1) {
19872 int Src1Idx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
19873 AMDGPU::OpName::scale_src1);
19874 if (
TII->usesConstantBus(MRI,
MI, Src0Idx) &&
19875 TII->usesConstantBus(MRI,
MI, Src1Idx))
19876 TII->legalizeOpWithMove(
MI, Src1Idx);
19883 if (
TII->isImage(
MI))
19884 TII->enforceOperandRCAlignment(
MI, AMDGPU::OpName::vaddr);
19926 uint64_t RsrcDword2And3)
const {
19956std::pair<unsigned, const TargetRegisterClass *>
19963 if (Constraint.
size() == 1) {
19967 if (VT == MVT::Other)
19970 switch (Constraint[0]) {
19977 RC = &AMDGPU::SReg_32RegClass;
19980 RC = &AMDGPU::SGPR_64RegClass;
19985 return std::pair(0U,
nullptr);
19992 return std::pair(0U,
nullptr);
19994 RC = Subtarget->useRealTrue16Insts() ? &AMDGPU::VGPR_16RegClass
19995 : &AMDGPU::VGPR_32_Lo256RegClass;
19998 RC = Subtarget->has1024AddressableVGPRs()
19999 ?
TRI->getAlignedLo256VGPRClassForBitWidth(
BitWidth)
20002 return std::pair(0U,
nullptr);
20007 if (!Subtarget->hasMAIInsts())
20011 return std::pair(0U,
nullptr);
20013 RC = &AMDGPU::AGPR_32RegClass;
20018 return std::pair(0U,
nullptr);
20023 }
else if (Constraint ==
"VA" && Subtarget->hasGFX90AInsts()) {
20027 RC = &AMDGPU::AV_32RegClass;
20030 RC =
TRI->getVectorSuperClassForBitWidth(
BitWidth);
20032 return std::pair(0U,
nullptr);
20041 return std::pair(0U, RC);
20044 if (Kind !=
'\0') {
20046 RC = &AMDGPU::VGPR_32_Lo256RegClass;
20047 }
else if (Kind ==
's') {
20048 RC = &AMDGPU::SGPR_32RegClass;
20049 }
else if (Kind ==
'a') {
20050 RC = &AMDGPU::AGPR_32RegClass;
20056 return std::pair(0U,
nullptr);
20062 return std::pair(0U,
nullptr);
20066 RC =
TRI->getVGPRClassForBitWidth(Width);
20068 RC =
TRI->getSGPRClassForBitWidth(Width);
20070 RC =
TRI->getAGPRClassForBitWidth(Width);
20072 Reg =
TRI->getMatchingSuperReg(Reg, AMDGPU::sub0, RC);
20077 return std::pair(0U,
nullptr);
20079 return std::pair(Reg, RC);
20088 return std::pair(0U,
nullptr);
20089 if (RC && Idx < RC->getNumRegs())
20091 return std::pair(0U,
nullptr);
20097 Ret.second =
TRI->getPhysRegBaseClass(Ret.first);
20103 if (Constraint.
size() == 1) {
20104 switch (Constraint[0]) {
20114 }
else if (Constraint ==
"DA" || Constraint ==
"DB") {
20122 if (Constraint.
size() == 1) {
20123 switch (Constraint[0]) {
20131 }
else if (Constraint.
size() == 2) {
20132 if (Constraint ==
"VA")
20150 std::vector<SDValue> &
Ops,
20165 unsigned Size =
Op.getScalarValueSizeInBits();
20169 if (
Size == 16 && !Subtarget->has16BitInsts())
20173 Val =
C->getSExtValue();
20177 Val =
C->getValueAPF().bitcastToAPInt().getSExtValue();
20181 if (
Size != 16 ||
Op.getNumOperands() != 2)
20183 if (
Op.getOperand(0).isUndef() ||
Op.getOperand(1).isUndef())
20186 Val =
C->getSExtValue();
20190 Val =
C->getValueAPF().bitcastToAPInt().getSExtValue();
20199 uint64_t Val)
const {
20200 if (Constraint.
size() == 1) {
20201 switch (Constraint[0]) {
20216 }
else if (Constraint.
size() == 2) {
20217 if (Constraint ==
"DA") {
20218 int64_t HiBits =
static_cast<int32_t
>(Val >> 32);
20219 int64_t LoBits =
static_cast<int32_t
>(Val);
20223 if (Constraint ==
"DB") {
20231 unsigned MaxSize)
const {
20232 unsigned Size = std::min<unsigned>(
Op.getScalarValueSizeInBits(), MaxSize);
20233 bool HasInv2Pi = Subtarget->hasInv2PiInlineImm();
20235 MVT VT =
Op.getSimpleValueType();
20260 switch (UnalignedClassID) {
20261 case AMDGPU::VReg_64RegClassID:
20262 return AMDGPU::VReg_64_Align2RegClassID;
20263 case AMDGPU::VReg_96RegClassID:
20264 return AMDGPU::VReg_96_Align2RegClassID;
20265 case AMDGPU::VReg_128RegClassID:
20266 return AMDGPU::VReg_128_Align2RegClassID;
20267 case AMDGPU::VReg_160RegClassID:
20268 return AMDGPU::VReg_160_Align2RegClassID;
20269 case AMDGPU::VReg_192RegClassID:
20270 return AMDGPU::VReg_192_Align2RegClassID;
20271 case AMDGPU::VReg_224RegClassID:
20272 return AMDGPU::VReg_224_Align2RegClassID;
20273 case AMDGPU::VReg_256RegClassID:
20274 return AMDGPU::VReg_256_Align2RegClassID;
20275 case AMDGPU::VReg_288RegClassID:
20276 return AMDGPU::VReg_288_Align2RegClassID;
20277 case AMDGPU::VReg_320RegClassID:
20278 return AMDGPU::VReg_320_Align2RegClassID;
20279 case AMDGPU::VReg_352RegClassID:
20280 return AMDGPU::VReg_352_Align2RegClassID;
20281 case AMDGPU::VReg_384RegClassID:
20282 return AMDGPU::VReg_384_Align2RegClassID;
20283 case AMDGPU::VReg_512RegClassID:
20284 return AMDGPU::VReg_512_Align2RegClassID;
20285 case AMDGPU::VReg_1024RegClassID:
20286 return AMDGPU::VReg_1024_Align2RegClassID;
20287 case AMDGPU::AReg_64RegClassID:
20288 return AMDGPU::AReg_64_Align2RegClassID;
20289 case AMDGPU::AReg_96RegClassID:
20290 return AMDGPU::AReg_96_Align2RegClassID;
20291 case AMDGPU::AReg_128RegClassID:
20292 return AMDGPU::AReg_128_Align2RegClassID;
20293 case AMDGPU::AReg_160RegClassID:
20294 return AMDGPU::AReg_160_Align2RegClassID;
20295 case AMDGPU::AReg_192RegClassID:
20296 return AMDGPU::AReg_192_Align2RegClassID;
20297 case AMDGPU::AReg_256RegClassID:
20298 return AMDGPU::AReg_256_Align2RegClassID;
20299 case AMDGPU::AReg_512RegClassID:
20300 return AMDGPU::AReg_512_Align2RegClassID;
20301 case AMDGPU::AReg_1024RegClassID:
20302 return AMDGPU::AReg_1024_Align2RegClassID;
20318 if (Info->isEntryFunction()) {
20325 unsigned MaxNumSGPRs = ST.getMaxNumSGPRs(MF);
20327 ? AMDGPU::SGPR_32RegClass.getRegister(MaxNumSGPRs - 1)
20328 :
TRI->getAlignedHighSGPRForRC(MF, 2,
20329 &AMDGPU::SGPR_64RegClass);
20330 Info->setSGPRForEXECCopy(SReg);
20332 assert(!
TRI->isSubRegister(Info->getScratchRSrcReg(),
20333 Info->getStackPtrOffsetReg()));
20334 if (Info->getStackPtrOffsetReg() != AMDGPU::SP_REG)
20335 MRI.
replaceRegWith(AMDGPU::SP_REG, Info->getStackPtrOffsetReg());
20339 if (Info->getScratchRSrcReg() != AMDGPU::PRIVATE_RSRC_REG)
20340 MRI.
replaceRegWith(AMDGPU::PRIVATE_RSRC_REG, Info->getScratchRSrcReg());
20342 if (Info->getFrameOffsetReg() != AMDGPU::FP_REG)
20345 Info->limitOccupancy(MF);
20347 if (ST.isWave32() && !MF.
empty()) {
20348 for (
auto &
MBB : MF) {
20349 for (
auto &
MI :
MBB) {
20350 TII->fixImplicitOperands(
MI);
20360 if (ST.needsAlignedVGPRs()) {
20367 if (NewClassID != -1)
20377 const APInt &DemandedElts,
20379 unsigned Depth)
const {
20381 unsigned Opc =
Op.getOpcode();
20384 unsigned IID =
Op.getConstantOperandVal(0);
20386 case Intrinsic::amdgcn_mbcnt_lo:
20387 case Intrinsic::amdgcn_mbcnt_hi: {
20392 Known.Zero.setBitsFrom(
20393 IID == Intrinsic::amdgcn_mbcnt_lo ? ST.getWavefrontSizeLog2() : 5);
20419 unsigned MaxValue =
20426 unsigned BFEWidth,
bool SExt,
unsigned Depth) {
20430 unsigned Src1Cst = 0;
20431 if (Src1.isImm()) {
20432 Src1Cst = Src1.getImm();
20433 }
else if (Src1.isReg()) {
20437 Src1Cst = Cst->Value.getZExtValue();
20448 if (Width >= BFEWidth)
20465 unsigned Depth)
const {
20468 switch (
MI->getOpcode()) {
20469 case AMDGPU::S_BFE_I32:
20472 case AMDGPU::S_BFE_U32:
20475 case AMDGPU::S_BFE_I64:
20478 case AMDGPU::S_BFE_U64:
20481 case AMDGPU::G_INTRINSIC:
20482 case AMDGPU::G_INTRINSIC_CONVERGENT: {
20485 case Intrinsic::amdgcn_workitem_id_x:
20488 case Intrinsic::amdgcn_workitem_id_y:
20491 case Intrinsic::amdgcn_workitem_id_z:
20494 case Intrinsic::amdgcn_mbcnt_lo:
20495 case Intrinsic::amdgcn_mbcnt_hi: {
20498 Known.Zero.setBitsFrom(IID == Intrinsic::amdgcn_mbcnt_lo
20507 case Intrinsic::amdgcn_groupstaticsize: {
20511 Known.Zero.setHighBits(
20515 case Intrinsic::amdgcn_readfirstlane:
20516 case Intrinsic::amdgcn_readlane: {
20525 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
20526 Known.Zero.setHighBits(24);
20528 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
20529 Known.Zero.setHighBits(16);
20531 case AMDGPU::G_AMDGPU_COPY_SCC_VCC:
20534 Known.Zero.setHighBits(
Known.getBitWidth() - 1);
20536 case AMDGPU::G_AMDGPU_SMED3:
20537 case AMDGPU::G_AMDGPU_UMED3: {
20538 auto [Dst, Src0, Src1, Src2] =
MI->getFirst4Regs();
20565 unsigned Depth)
const {
20574 if (
MaybeAlign RetAlign = Attrs.getRetAlignment())
20595 BlockToAlign =
ML->getHeader();
20599 if (needsFetchWindowAlignment(*BlockToAlign))
20620 if (Header->getAlignment() != PrefAlign)
20621 return Header->getAlignment();
20623 unsigned LoopSize = 0;
20628 LoopSize +=
MBB->getAlignment().value() / 2;
20631 LoopSize +=
TII->getInstSizeInBytes(
MI);
20632 if (LoopSize > 192)
20637 if (LoopSize <= 64)
20640 if (LoopSize <= 128)
20641 return CacheLineAlign;
20647 auto I = Exit->getFirstNonDebugInstr();
20648 if (
I != Exit->end() &&
I->getOpcode() == AMDGPU::S_INST_PREFETCH)
20649 return CacheLineAlign;
20658 if (PreTerm == Pre->
begin() ||
20659 std::prev(PreTerm)->getOpcode() != AMDGPU::S_INST_PREFETCH)
20663 auto ExitHead = Exit->getFirstNonDebugInstr();
20664 if (ExitHead == Exit->end() ||
20665 ExitHead->getOpcode() != AMDGPU::S_INST_PREFETCH)
20670 return CacheLineAlign;
20678 if (needsFetchWindowAlignment(*
MBB))
20683bool SITargetLowering::needsFetchWindowAlignment(
20685 if (!
getSubtarget()->hasLoopHeadInstSplitSensitivity())
20689 if (
MI.isMetaInstruction())
20692 return TII->getInstSizeInBytes(
MI) > 4;
20702 N =
N->getOperand(0).getNode();
20712 switch (
N->getOpcode()) {
20720 if (Reg.isPhysical() || MRI.
isLiveIn(Reg))
20721 return !
TRI->isSGPRReg(MRI, Reg);
20727 return !
TRI->isSGPRReg(MRI, Reg);
20731 unsigned AS = L->getAddressSpace();
20741 case AMDGPUISD::ATOMIC_CMP_SWAP:
20742 case AMDGPUISD::BUFFER_ATOMIC_SWAP:
20743 case AMDGPUISD::BUFFER_ATOMIC_ADD:
20744 case AMDGPUISD::BUFFER_ATOMIC_SUB:
20745 case AMDGPUISD::BUFFER_ATOMIC_SMIN:
20746 case AMDGPUISD::BUFFER_ATOMIC_UMIN:
20747 case AMDGPUISD::BUFFER_ATOMIC_SMAX:
20748 case AMDGPUISD::BUFFER_ATOMIC_UMAX:
20749 case AMDGPUISD::BUFFER_ATOMIC_AND:
20750 case AMDGPUISD::BUFFER_ATOMIC_OR:
20751 case AMDGPUISD::BUFFER_ATOMIC_XOR:
20752 case AMDGPUISD::BUFFER_ATOMIC_INC:
20753 case AMDGPUISD::BUFFER_ATOMIC_DEC:
20754 case AMDGPUISD::BUFFER_ATOMIC_CMPSWAP:
20755 case AMDGPUISD::BUFFER_ATOMIC_FADD:
20756 case AMDGPUISD::BUFFER_ATOMIC_FMIN:
20757 case AMDGPUISD::BUFFER_ATOMIC_FMAX:
20763 return A->readMem() &&
A->writeMem();
20784 switch (Ty.getScalarSizeInBits()) {
20796 const APInt &DemandedElts,
20799 unsigned Depth)
const {
20800 if (
Op.getOpcode() == AMDGPUISD::CLAMP) {
20804 if (Info->getMode().DX10Clamp)
20817enum class AtomicFlushDenormalReason {
20820 IgnoreDenormalMode,
20821 FunctionFlushesDenormals
20826enum class GlobalFPAtomicLegality {
20828 AgentScopeFineGrainedRemoteMemory,
20829 EmulatedSystemScope,
20831 NoFineGrainedMemory
20838static AtomicFlushDenormalReason
20840 if (RMW->
hasMetadata(LLVMContext::MD_atomic_ignore_denormal_mode))
20841 return AtomicFlushDenormalReason::IgnoreDenormalMode;
20846 ? AtomicFlushDenormalReason::FunctionFlushesDenormals
20847 : AtomicFlushDenormalReason::IEEE;
20852 GlobalFPAtomicLegality MemLegality,
20853 AtomicFlushDenormalReason DenormReason) {
20856 if (MemScope.empty())
20857 MemScope =
"system";
20860 R <<
"hardware instruction generated for atomic "
20862 <<
" at " <<
ore::NV(
"SyncScope", MemScope) <<
" scope since ";
20864 switch (MemLegality) {
20865 case GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory:
20866 R <<
"fine-grained remote memory atomics work below system scope";
20868 case GlobalFPAtomicLegality::EmulatedSystemScope:
20869 R <<
"system scope atomics are emulated in hardware";
20871 case GlobalFPAtomicLegality::NoRemoteMemory:
20872 R <<
"memory is not remote (!amdgpu.no.remote.memory)";
20874 case GlobalFPAtomicLegality::NoFineGrainedMemory:
20875 R <<
"memory is not fine-grained (!amdgpu.no.fine.grained.memory)";
20877 case GlobalFPAtomicLegality::Illegal:
20881 switch (DenormReason) {
20882 case AtomicFlushDenormalReason::Native:
20884 case AtomicFlushDenormalReason::IgnoreDenormalMode:
20885 R <<
", and denormals may be flushed (!atomic.ignore.denormal.mode)";
20887 case AtomicFlushDenormalReason::FunctionFlushesDenormals:
20888 R <<
", and the floating-point environment flushes denormals";
20890 case AtomicFlushDenormalReason::IEEE:
20899 Type *EltTy = VT->getElementType();
20900 return VT->getNumElements() == 2 &&
20920 unsigned BW =
IT->getBitWidth();
20921 return BW == 32 || BW == 64;
20935 unsigned BW =
DL.getPointerSizeInBits(PT->getAddressSpace());
20936 return BW == 32 || BW == 64;
20939 if (Ty->isFloatTy() || Ty->isDoubleTy())
20943 return VT->getNumElements() == 2 &&
20944 VT->getElementType()->getPrimitiveSizeInBits() == 16;
20952static GlobalFPAtomicLegality
20961 if (HasSystemScope) {
20962 if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics() &&
20964 return GlobalFPAtomicLegality::NoRemoteMemory;
20965 if (Subtarget.hasEmulatedSystemScopeAtomics())
20966 return GlobalFPAtomicLegality::EmulatedSystemScope;
20967 }
else if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics())
20968 return GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory;
20970 return RMW->
hasMetadata(
"amdgpu.no.fine.grained.memory")
20971 ? GlobalFPAtomicLegality::NoFineGrainedMemory
20972 : GlobalFPAtomicLegality::Illegal;
20985 const MDNode *MD =
I->getMetadata(LLVMContext::MD_noalias_addrspace);
20993 return STI.hasGloballyAddressableScratch()
21011 DL.getTypeSizeInBits(RMW->
getType()) == 64 &&
21015 GlobalFPAtomicLegality MemLegality = GlobalFPAtomicLegality::Illegal;
21016 AtomicFlushDenormalReason DenormReason = AtomicFlushDenormalReason::Native;
21026 bool HasSystemScope =
21058 if (!Subtarget->hasSubClampInsts() ||
21060 !Subtarget->hasAtomicDsCondSubClampInsts()) ||
21062 !Subtarget->hasAtomicCondSubClampFlatInsts()))
21067 if (!
IT ||
IT->getBitWidth() != 32)
21073 if (Subtarget->hasEmulatedSystemScopeAtomics())
21089 if (!HasSystemScope &&
21090 Subtarget->hasAgentScopeFineGrainedRemoteMemoryAtomics())
21102 if (RMW->
hasMetadata(
"amdgpu.no.fine.grained.memory"))
21111 ConstVal && ConstVal->isNullValue() &&
21150 if (Ty->isFloatTy()) {
21155 if (Ty->isDoubleTy()) {
21176 if (Ty->isFloatTy() &&
21177 !Subtarget->hasMemoryAtomicFaddF32DenormalSupport()) {
21179 if (DenormReason == AtomicFlushDenormalReason::IEEE)
21185 if (MemLegality != GlobalFPAtomicLegality::Illegal) {
21192 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() &&
isV2F16(Ty))
21196 if (Subtarget->hasAtomicGlobalPkAddBF16Inst() &&
isV2BF16(Ty))
21200 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() &&
isV2F16(Ty))
21205 if (Subtarget->hasAtomicBufferPkAddBF16Inst() &&
isV2BF16(Ty))
21210 if (Subtarget->hasFlatBufferGlobalAtomicFaddF64Inst() && Ty->isDoubleTy())
21214 if (Ty->isFloatTy()) {
21217 if (RMW->
use_empty() && Subtarget->hasAtomicFaddNoRtnInsts())
21220 if (!RMW->
use_empty() && Subtarget->hasAtomicFaddRtnInsts())
21225 Subtarget->hasAtomicBufferGlobalPkAddF16NoRtnInsts() &&
21233 if (Subtarget->hasFlatAtomicFaddF32Inst())
21242 if (Subtarget->hasLDSFPAtomicAddF32()) {
21243 if (RMW->
use_empty() && Subtarget->hasAtomicFaddNoRtnInsts())
21245 if (!RMW->
use_empty() && Subtarget->hasAtomicFaddRtnInsts())
21265 if (MemLegality != GlobalFPAtomicLegality::Illegal) {
21275 if (Subtarget->hasAtomicFMinFMaxF32FlatInsts() && Ty->isFloatTy())
21277 if (Subtarget->hasAtomicFMinFMaxF64FlatInsts() && Ty->isDoubleTy())
21281 if (Subtarget->hasAtomicFMinFMaxF32GlobalInsts() && Ty->isFloatTy())
21283 if (Subtarget->hasAtomicFMinFMaxF64GlobalInsts() && Ty->isDoubleTy())
21337 if (RC == &AMDGPU::VReg_1RegClass && !isDivergent)
21338 return Subtarget->isWave64() ? &AMDGPU::SReg_64RegClass
21339 : &AMDGPU::SReg_32RegClass;
21340 if (!
TRI->isSGPRClass(RC) && !isDivergent)
21341 return TRI->getEquivalentSGPRClass(RC);
21342 if (
TRI->isSGPRClass(RC) && isDivergent) {
21343 if (Subtarget->hasGFX90AInsts())
21344 return TRI->getEquivalentAVClass(RC);
21345 return TRI->getEquivalentVGPRClass(RC);
21358 unsigned WaveSize) {
21363 if (!
IT ||
IT->getBitWidth() != WaveSize)
21368 if (!Visited.
insert(V).second)
21370 bool Result =
false;
21371 for (
const auto *U : V->users()) {
21373 if (V == U->getOperand(1)) {
21378 case Intrinsic::amdgcn_if_break:
21379 case Intrinsic::amdgcn_if:
21380 case Intrinsic::amdgcn_else:
21385 if (V == U->getOperand(0)) {
21390 case Intrinsic::amdgcn_end_cf:
21391 case Intrinsic::amdgcn_loop:
21397 Result =
hasCFUser(U, Visited, WaveSize);
21406 const Value *V)
const {
21408 if (CI->isInlineAsm()) {
21417 for (
auto &TC : TargetConstraints) {
21431 return hasCFUser(V, Visited, Subtarget->getWavefrontSize());
21466 if (
I.getMetadata(
"amdgpu.noclobber"))
21468 if (
I.getMetadata(
"amdgpu.last.use"))
21532 Alignment = RMW->getAlign();
21545 bool FullFlatEmulation =
21547 ((Subtarget->hasAtomicFaddInsts() && RMW->getType()->isFloatTy()) ||
21548 (Subtarget->hasFlatBufferGlobalAtomicFaddF64Inst() &&
21549 RMW->getType()->isDoubleTy()));
21552 bool ReturnValueIsUsed = !AI->
use_empty();
21561 if (FullFlatEmulation) {
21572 std::prev(BB->
end())->eraseFromParent();
21573 Builder.SetInsertPoint(BB);
21575 Value *LoadedShared =
nullptr;
21576 if (FullFlatEmulation) {
21577 Value *IsShared = Builder.CreateIntrinsic(Intrinsic::amdgcn_is_shared,
21578 {Addr},
nullptr,
"is.shared");
21579 Builder.CreateCondBr(IsShared, SharedBB, CheckPrivateBB);
21580 Builder.SetInsertPoint(SharedBB);
21581 Value *CastToLocal = Builder.CreateAddrSpaceCast(
21587 LoadedShared = Clone;
21589 Builder.CreateBr(PhiBB);
21590 Builder.SetInsertPoint(CheckPrivateBB);
21593 Value *IsPrivate = Builder.CreateIntrinsic(Intrinsic::amdgcn_is_private,
21594 {Addr},
nullptr,
"is.private");
21595 Builder.CreateCondBr(IsPrivate, PrivateBB, GlobalBB);
21597 Builder.SetInsertPoint(PrivateBB);
21599 Value *CastToPrivate = Builder.CreateAddrSpaceCast(
21602 Value *LoadedPrivate;
21604 LoadedPrivate = Builder.CreateAlignedLoad(
21605 RMW->getType(), CastToPrivate, RMW->getAlign(), RMW->isVolatile(),
21609 LoadedPrivate, RMW->getValOperand());
21611 Builder.CreateAlignedStore(NewVal, CastToPrivate, RMW->getAlign(),
21612 RMW->isVolatile());
21620 LoadedPrivate = Builder.CreateInsertValue(Insert, Equal, 1);
21623 Builder.CreateBr(PhiBB);
21625 Builder.SetInsertPoint(GlobalBB);
21629 if (FullFlatEmulation) {
21630 Value *CastToGlobal = Builder.CreateAddrSpaceCast(
21639 if (!FullFlatEmulation) {
21644 MDNode *RangeNotPrivate =
21647 LoadedGlobal->
setMetadata(LLVMContext::MD_noalias_addrspace,
21651 Builder.CreateBr(PhiBB);
21653 Builder.SetInsertPoint(PhiBB);
21655 if (ReturnValueIsUsed) {
21658 if (FullFlatEmulation)
21659 Loaded->addIncoming(LoadedShared, SharedBB);
21660 Loaded->addIncoming(LoadedPrivate, PrivateBB);
21661 Loaded->addIncoming(LoadedGlobal, GlobalBB);
21662 Loaded->takeName(AI);
21665 Builder.CreateBr(ExitBB);
21669 unsigned PtrOpIdx) {
21670 Value *PtrOp =
I->getOperand(PtrOpIdx);
21677 I->setOperand(PtrOpIdx, ASCast);
21689 ConstVal && ConstVal->isNullValue() &&
21720 "Expand Atomic Load only handles SCRATCH -> FLAT conversion");
21728 "Expand Atomic Store only handles SCRATCH -> FLAT conversion");
static bool isMul(MachineInstr *MI)
static unsigned getIntrinsicID(const SDNode *N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
static constexpr std::pair< ImplicitArgumentMask, StringLiteral > ImplicitAttrs[]
static bool allUsesHaveSourceMods(MachineInstr &MI, MachineRegisterInfo &MRI, unsigned CostThreshold=4)
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
static bool isCtlzOpc(unsigned Opc)
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isNoUnsignedWrap(MachineInstr *Addr)
static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE, bool &IsTexFail)
static bool isAsyncLDSDMA(Intrinsic::ID Intr)
static void packImage16bitOpsToDwords(MachineIRBuilder &B, MachineInstr &MI, SmallVectorImpl< Register > &PackedAddrs, unsigned ArgOffset, const AMDGPU::ImageDimIntrinsicInfo *Intr, bool IsA16, bool IsG16)
Turn a set of f16 typed registers in AddrRegs into a dword sized vector with f16 typed elements.
static bool isKnownNonNull(Register Val, MachineRegisterInfo &MRI, const AMDGPUTargetMachine &TM, unsigned AddrSpace)
Return true if the value is a known valid address, such that a null check is not necessary.
Provides AMDGPU specific target descriptions.
The AMDGPU TargetMachine interface definition for hw codegen targets.
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Function Alias Analysis Results
@ DEFAULT
Default weight is used in cases when there is no dedicated execution weight set.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static std::optional< SDByteProvider > calculateByteProvider(SDValue Op, unsigned Index, unsigned Depth, std::optional< uint64_t > VectorIndex, unsigned StartingIndex=0, MutableArrayRef< uint8_t > ByteMask={})
static bool isSigned(unsigned Opcode)
Utilities for dealing with flags related to floating point properties and mode controls.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
iv Induction Variable Users
static constexpr Value * getValue(Ty &ValueOrUse)
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Contains matchers for matching SSA Machine Instructions.
static bool isUndef(const MachineInstr &MI)
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static unsigned getAddressSpace(const Value *V, unsigned MaxLookup)
uint64_t IntrinsicInst * II
static constexpr MCPhysReg SPReg
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
Contains matchers for matching SelectionDAG nodes and values.
static void r0(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
static void r3(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
static void r2(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
static void r1(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
#define FP_DENORM_FLUSH_NONE
#define FP_DENORM_FLUSH_IN_FLUSH_OUT
static void reservePrivateMemoryRegs(const TargetMachine &TM, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info)
static SDValue adjustLoadValueTypeImpl(SDValue Result, EVT LoadVT, const SDLoc &DL, SelectionDAG &DAG, bool Unpacked)
static MachineBasicBlock * emitIndirectSrc(MachineInstr &MI, MachineBasicBlock &MBB, const GCNSubtarget &ST)
static bool denormalModeIsFlushAllF64F16(const MachineFunction &MF)
static bool isAtomicRMWLegalIntTy(Type *Ty)
static void knownBitsForWorkitemID(const GCNSubtarget &ST, GISelValueTracking &VT, KnownBits &Known, unsigned Dim)
static bool flatInstrMayAccessPrivate(const Instruction *I)
Return if a flat address space atomicrmw can access private memory.
static std::pair< unsigned, int > computeIndirectRegAndOffset(const SIRegisterInfo &TRI, const TargetRegisterClass *SuperRC, unsigned VecReg, int Offset)
static bool denormalModeIsFlushAllF32(const MachineFunction &MF)
static bool addresses16Bits(int Mask)
static MachineBasicBlock * expand64BitScalarArithmetic(MachineInstr &MI, MachineBasicBlock *BB)
static bool isClampZeroToOne(SDValue A, SDValue B)
static bool supportsMin3Max3(const GCNSubtarget &Subtarget, unsigned Opc, EVT VT)
static AtomicFlushDenormalReason getAtomicFlushDenormalReason(const AtomicRMWInst *RMW)
static unsigned findFirstFreeSGPR(CCState &CCInfo)
static uint32_t getPermuteMask(SDValue V)
static SDValue lowerLaneOp(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static int getAlignedAGPRClassID(unsigned UnalignedClassID)
static void processPSInputArgs(SmallVectorImpl< ISD::InputArg > &Splits, CallingConv::ID CallConv, ArrayRef< ISD::InputArg > Ins, BitVector &Skipped, FunctionType *FType, SIMachineFunctionInfo *Info)
static SDValue selectSOffset(SDValue SOffset, SelectionDAG &DAG, const GCNSubtarget *Subtarget)
static SDValue getLoadExtOrTrunc(SelectionDAG &DAG, ISD::LoadExtType ExtType, SDValue Op, const SDLoc &SL, EVT VT)
static std::tuple< unsigned, unsigned > getDPPOpcForWaveReduction(unsigned Opc, const GCNSubtarget &ST)
static void fixMasks(SmallVectorImpl< DotSrc > &Srcs, unsigned ChainLength)
static bool is32bitWaveReduceOperation(unsigned Opc)
static TargetLowering::AtomicExpansionKind atomicSupportedIfLegalIntType(const AtomicRMWInst *RMW)
static SDValue strictFPExtFromF16(SelectionDAG &DAG, SDValue Src)
Return the source of an fp_extend from f16 to f32, or a converted FP constant.
static bool isAtomicRMWLegalXChgTy(const AtomicRMWInst *RMW)
static bool bitOpWithConstantIsReducible(unsigned Opc, uint32_t Val)
static void convertScratchAtomicToFlatAtomic(Instruction *I, unsigned PtrOpIdx)
static bool isCopyFromRegOfInlineAsm(const SDNode *N)
static bool elementPairIsOddToEven(ArrayRef< int > Mask, int Elt)
static SDValue lowerBFEIntrinsic(SDValue Op, SelectionDAG &DAG, Intrinsic::ID IntrinsicID)
static cl::opt< bool > DisableLoopAlignment("amdgpu-disable-loop-alignment", cl::desc("Do not align and prefetch loops"), cl::init(false))
static SDValue getDWordFromOffset(SelectionDAG &DAG, SDLoc SL, SDValue Src, unsigned DWordOffset)
static MachineBasicBlock::iterator loadM0FromVGPR(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineInstr &MI, unsigned InitResultReg, unsigned PhiReg, int Offset, bool UseGPRIdxMode, Register &SGPRIdxReg)
static bool isFloatingPointWaveReduceOperation(unsigned Opc)
static bool isImmConstraint(StringRef Constraint)
static SDValue padEltsToUndef(SelectionDAG &DAG, const SDLoc &DL, EVT CastVT, SDValue Src, int ExtraElts)
static bool hasCFUser(const Value *V, SmallPtrSet< const Value *, 16 > &Visited, unsigned WaveSize)
static std::pair< Register, Register > ExtractSubRegs(MachineInstr &MI, MachineOperand &Op, const TargetRegisterClass *SrcRC, const GCNSubtarget &ST, MachineRegisterInfo &MRI)
static unsigned SubIdx2Lane(unsigned Idx)
Helper function for adjustWritemask.
static TargetLowering::AtomicExpansionKind getPrivateAtomicExpansionKind(const GCNSubtarget &STI)
static bool addressMayBeAccessedAsPrivate(const MachineMemOperand *MMO, const SIMachineFunctionInfo &Info)
static MachineBasicBlock * lowerWaveReduce(MachineInstr &MI, MachineBasicBlock &BB, const GCNSubtarget &ST, unsigned Opc)
static bool elementPairIsContiguous(ArrayRef< int > Mask, int Elt)
static bool isV2BF16(Type *Ty)
static bool isFrexpExp(SDValue V, SDValue &FrexpInput)
static ArgDescriptor allocateSGPR32InputImpl(CCState &CCInfo, const TargetRegisterClass *RC, unsigned NumArgRegs)
static SDValue getMad64_32(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue N0, SDValue N1, SDValue N2, bool Signed)
static SDValue resolveSources(SelectionDAG &DAG, SDLoc SL, SmallVectorImpl< DotSrc > &Srcs, bool IsSigned, bool IsAny)
static bool hasNon16BitAccesses(uint64_t PermMask, SDValue &Op, SDValue &OtherOp)
static SDValue lowerWaveShuffle(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static SDValue diagnoseUnsupportedImage(SelectionDAG &DAG, SDValue Op, ArrayRef< EVT > ResultTypes, const SDLoc &DL, const Twine &Msg)
Emit a DiagnosticInfoUnsupported for an unsupported image intrinsic and return poison values of Resul...
static void placeSources(ByteProvider< SDValue > &Src0, ByteProvider< SDValue > &Src1, SmallVectorImpl< DotSrc > &Src0s, SmallVectorImpl< DotSrc > &Src1s, int Step)
static unsigned parseSyncscopeMDArg(const CallBase &CI, unsigned ArgIdx)
static EVT memVTFromLoadIntrReturn(const SITargetLowering &TLI, const DataLayout &DL, Type *Ty, unsigned MaxNumLanes)
static MachineBasicBlock::iterator emitLoadM0FromVGPRLoop(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineBasicBlock &OrigBB, MachineBasicBlock &LoopBB, const DebugLoc &DL, const MachineOperand &Idx, unsigned InitReg, unsigned ResultReg, unsigned PhiReg, unsigned InitSaveExecReg, int Offset, bool UseGPRIdxMode, Register &SGPRIdxReg)
static SDValue matchPERM(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool isFrameIndexOp(SDValue Op)
static ConstantFPSDNode * getSplatConstantFP(SDValue Op)
static void allocateSGPR32Input(CCState &CCInfo, ArgDescriptor &Arg)
static void knownBitsForSBFE(const MachineInstr &MI, GISelValueTracking &VT, KnownBits &Known, const APInt &DemandedElts, unsigned BFEWidth, bool SExt, unsigned Depth)
static bool isExtendedFrom16Bits(SDValue &Operand)
static std::optional< bool > checkDot4MulSignedness(const SDValue &N, ByteProvider< SDValue > &Src0, ByteProvider< SDValue > &Src1, const SDValue &S0Op, const SDValue &S1Op, const SelectionDAG &DAG)
static bool vectorEltWillFoldAway(SDValue Op)
static SDValue getSPDenormModeValue(uint32_t SPDenormMode, SelectionDAG &DAG, const SIMachineFunctionInfo *Info, const GCNSubtarget *ST)
static uint32_t getConstantPermuteMask(uint32_t C)
static AtomicOrdering parseAtomicOrderingCABIArg(const CallBase &CI, unsigned ArgIdx)
static MachineBasicBlock * emitIndirectDst(MachineInstr &MI, MachineBasicBlock &MBB, const GCNSubtarget &ST)
static void setM0ToIndexFromSGPR(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineInstr &MI, int Offset)
static DenormalFPEnv getDenormalFPEnv(const MachineFunction &MF)
static std::pair< MachineBasicBlock *, MachineBasicBlock * > splitBlockForLoop(MachineInstr &MI, MachineBasicBlock &MBB, bool InstInLoop)
static unsigned getBasePtrIndex(const MemSDNode *N)
MemSDNode::getBasePtr() does not work for intrinsics, which needs to offset by the chain and intrinsi...
static void allocateFixedSGPRInputImpl(CCState &CCInfo, const TargetRegisterClass *RC, MCRegister Reg)
static SDValue constructRetValue(SelectionDAG &DAG, MachineSDNode *Result, ArrayRef< EVT > ResultTypes, bool IsTexFail, bool Unpacked, bool IsD16, int DMaskPop, int NumVDataDwords, bool IsAtomicPacked16Bit, const SDLoc &DL)
static std::pair< SDValue, SDValue > splitTFEValueAndStatus(SDValue Op, EVT VT, const SDLoc &DL, SelectionDAG &DAG)
static std::optional< ByteProvider< SDValue > > handleMulOperand(const SDValue &MulOperand)
static ISD::CondCode tryReduceF64CompareToHiHalf(const ISD::CondCode CC, const SDValue LHS, const SDValue RHS, const SelectionDAG &DAG)
static Register getIndirectSGPRIdx(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineInstr &MI, int Offset)
static SDValue emitNonHSAIntrinsicError(SelectionDAG &DAG, const SDLoc &DL, EVT VT)
static EVT memVTFromLoadIntrData(const SITargetLowering &TLI, const DataLayout &DL, Type *Ty, unsigned MaxNumLanes)
static unsigned minMaxOpcToMin3Max3Opc(unsigned Opc)
static unsigned getExtOpcodeForPromotedOp(SDValue Op)
static void expand64BitV_CNDMASK(MachineInstr &MI, MachineBasicBlock *BB)
static GlobalFPAtomicLegality getGlobalMemoryFPAtomicLegality(const GCNSubtarget &Subtarget, const AtomicRMWInst *RMW, bool HasSystemScope)
static SDValue lowerBALLOTIntrinsic(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static SDValue buildSMovImm32(SelectionDAG &DAG, const SDLoc &DL, uint64_t Val)
static SDValue tryFoldMADwithSRL(SelectionDAG &DAG, const SDLoc &SL, SDValue MulLHS, SDValue MulRHS, SDValue AddRHS)
static unsigned getIntrMemWidth(unsigned IntrID)
static SDValue getBuildDwordsVector(SelectionDAG &DAG, SDLoc DL, ArrayRef< SDValue > Elts)
static SDNode * findUser(SDValue Value, unsigned Opcode)
Helper function for LowerBRCOND.
static unsigned addPermMasks(unsigned First, unsigned Second)
static uint64_t clearUnusedBits(uint64_t Val, unsigned Size)
static SDValue getFPTernOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue C, SDValue GlueChain, SDNodeFlags Flags)
static bool isV2F16OrV2BF16(Type *Ty)
static OptimizationRemark emitAtomicRMWLegalRemark(const AtomicRMWInst *RMW, GlobalFPAtomicLegality MemLegality, AtomicFlushDenormalReason DenormReason)
static SDValue emitRemovedIntrinsicError(SelectionDAG &DAG, const SDLoc &DL, EVT VT)
static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue GlueChain, SDNodeFlags Flags)
static SDValue buildPCRelGlobalAddress(SelectionDAG &DAG, const GlobalValue *GV, const SDLoc &DL, int64_t Offset, EVT PtrVT, unsigned GAFlags=SIInstrInfo::MO_NONE)
static cl::opt< bool > UseDivergentRegisterIndexing("amdgpu-use-divergent-register-indexing", cl::Hidden, cl::desc("Use indirect register addressing for divergent indexes"), cl::init(false))
static const std::optional< ByteProvider< SDValue > > calculateSrcByte(const SDValue Op, uint64_t DestByte, uint64_t SrcIndex=0, unsigned Depth=0)
static bool isV2F16(Type *Ty)
static void initializeM0ToZeroForClusterLoad(SDValue Op, SelectionDAG &DAG, SDLoc DL)
static void allocateSGPR64Input(CCState &CCInfo, ArgDescriptor &Arg)
static uint64_t getIdentityValueForWaveReduction(unsigned Opc)
SI DAG Lowering interface definition.
Interface definition for SIRegisterInfo.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static constexpr int Concat[]
static std::optional< uint32_t > getLDSKernelIdMetadata(const Function &F)
void setDynLDSAlign(const Function &F, const GlobalVariable &GV)
static std::optional< uint32_t > get32BitAbsoluteAddress(const GlobalValue &GV, unsigned AS)
void setUsesDynamicLDS(bool DynLDS)
bool isBottomOfStack() const
uint32_t getLDSSize() const
bool isEntryFunction() const
static unsigned numBitsSigned(SDValue Op, SelectionDAG &DAG)
SDValue SplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Split a vector load into 2 loads of half the vector.
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
SDValue LowerF64ToF16Safe(SDValue Src, const SDLoc &DL, SelectionDAG &DAG) const
SDValue storeStackInputValue(SelectionDAG &DAG, const SDLoc &SL, SDValue Chain, SDValue ArgVal, int64_t Offset) const
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
SDValue splitBinaryBitConstantOpImpl(DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS, uint32_t ValLo, uint32_t ValHi) const
Split the 64-bit value LHS into two 32-bit components, and perform the binary operation Opc to it wit...
SDValue lowerUnhandledCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals, StringRef Reason) const
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue addTokenForArgument(SDValue Chain, SelectionDAG &DAG, MachineFrameInfo &MFI, int ClobberedFI) const
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
static bool needsDenormHandlingF32(const SelectionDAG &DAG, SDValue Src, SDNodeFlags Flags)
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const
SDValue loadInputValue(SelectionDAG &DAG, const TargetRegisterClass *RC, EVT VT, const SDLoc &SL, const ArgDescriptor &Arg) const
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
static EVT getEquivalentMemType(LLVMContext &Context, EVT VT)
SDValue LowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
SDValue CreateLiveInRegister(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT, const SDLoc &SL, bool RawReg=false) const
Helper function that adds Reg to the LiveIn list of the DAG's MachineFunction.
SDValue SplitVectorStore(SDValue Op, SelectionDAG &DAG) const
Split a vector store into 2 stores of half the vector.
std::pair< SDValue, SDValue > split64BitValue(SDValue Op, SelectionDAG &DAG) const
Return 64-bit value Op as two 32-bit integers.
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
Selects the correct CCAssignFn for a given CallingConvention value.
static unsigned numBitsUnsigned(SDValue Op, SelectionDAG &DAG)
static bool allowApproxFunc(const SelectionDAG &DAG, SDNodeFlags Flags)
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue performRcpCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const
static bool shouldFoldFNegIntoSrc(SDNode *FNeg, SDValue FNegSrc)
bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const override
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
SDValue WidenOrSplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Widen a suitably aligned v3 load.
SDValue getHiHalf64(SDValue Op, SelectionDAG &DAG) const
bool isNoopAddrSpaceCast(const DataLayout &DL, unsigned SrcAS, unsigned DestAS) const override
Returns true if a cast between SrcAS and DestAS is a noop.
static bool EnableObjectLinking
const std::array< unsigned, 3 > & getDims() const
static const LaneMaskConstants & get(const GCNSubtarget &ST)
const unsigned XorTermOpc
const unsigned AndSaveExecOpc
static const fltSemantics & IEEEsingle()
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
LLVM_READONLY int getExactLog2Abs() const
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
static APFloat getInf(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Infinity.
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
bool isSignMask() const
Check if the APInt's value is returned by getSignMask.
unsigned countr_zero() const
Count the number of trailing zero bits.
bool isOneBitSet(unsigned BitNo) const
Determine if this APInt Value only has the specified bit set.
bool isSignBitSet() const
Determine if sign bit of this APInt is set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
bool sge(const APInt &RHS) const
Signed greater or equal comparison.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
This class represents an incoming formal argument to a Function.
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Check if an argument has a given attribute.
const Function * getParent() const
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
An instruction that atomically checks whether a specified value is in a memory location,...
Value * getNewValOperand()
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
Value * getCompareOperand()
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
static unsigned getPointerOperandIndex()
an instruction that atomically reads a memory location, combines it with another value,...
static unsigned getPointerOperandIndex()
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ USubCond
Subtract only if no unsigned overflow.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
void setOperation(BinOp Operation)
BinOp getOperation() const
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
static LLVM_ABI StringRef getOperationName(BinOp Op)
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
bool isCompareAndSwap() const
Returns true if this SDNode represents cmpxchg atomic operation, false otherwise.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI MemoryEffects getMemoryEffects() const
LLVM Basic Block Representation.
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
A "pseudo-class" with methods for operating on BUILD_VECTORs.
Represents known origin of an individual byte in combine pattern.
static ByteProvider getConstantZero()
static ByteProvider getSrc(std::optional< ISelOp > Val, int64_t ByteOffset, int64_t VectorOffset)
CCState - This class holds information needed while lowering arguments and return values.
MachineFunction & getMachineFunction() const
unsigned getFirstUnallocated(ArrayRef< MCPhysReg > Regs) const
getFirstUnallocated - Return the index of the first unallocated register in the set,...
static LLVM_ABI bool resultsCompatible(CallingConv::ID CalleeCC, CallingConv::ID CallerCC, MachineFunction &MF, LLVMContext &C, const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn CalleeFn, CCAssignFn CallerFn)
Returns true if the results of the two calling conventions are compatible.
LLVM_ABI void AnalyzeCallResult(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeCallResult - Analyze the return values of a call, incorporating info about the passed values i...
MCRegister AllocateReg(MCPhysReg Reg)
AllocateReg - Attempt to allocate one register.
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
int64_t AllocateStack(unsigned Size, Align Alignment)
AllocateStack - Allocate a chunk of stack space with the specified size and alignment.
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
bool isAllocated(MCRegister Reg) const
isAllocated - Return true if the specified register (or an alias) is allocated.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
int64_t getLocMemOffset() const
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
LLVM_ABI bool isMustTailCall() const
Tests if this call site must be tail call optimized.
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
static LLVM_ABI CastInst * CreatePointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, AddrSpaceCast or a PtrToInt cast instruction.
const APFloat & getValueAPF() const
bool isPosZero() const
Return true if the value is positive zero.
bool isOne() const
Returns true if this value is exactly +1.0.
bool isMinusOne() const
Returns true if this value is exactly -1.0.
bool isNegative() const
Return true if the value is negative.
bool isInfinity() const
Return true if the value is an infinity.
This is the shared class of boolean and integer constants.
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
This is an important base class in LLVM.
uint64_t getNumOperands() const
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI Align getABITypeAlign(Type *Ty) const
Returns the minimum ABI-required alignment for the specified type.
Diagnostic information for unsupported feature in backend.
static constexpr ElementCount getFixed(ScalarTy MinVal)
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
Register DemoteRegister
DemoteRegister - if CanLowerReturn is false, DemoteRegister is a vreg allocated to hold a pointer to ...
LLVM_ABI const Value * getValueFromVirtualReg(Register Vreg)
This method is called from TargetLowerinInfo::isSDNodeSourceOfDivergence to get the Value correspondi...
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
FunctionType * getFunctionType() const
Returns the FunctionType for me.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
iterator_range< arg_iterator > args()
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Argument * getArg(unsigned i) const
const SIInstrInfo * getInstrInfo() const override
unsigned getInstCacheLineSize() const
Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
const SIRegisterInfo * getRegisterInfo() const override
bool hasMin3Max3_16() const
bool supportsWaveWideBPermute() const
unsigned getMaxPrivateElementSize(bool ForBufferRSrc=false) const
bool hasKernargSegmentPtr() const
bool hasDispatchID() const
bool hasPrivateSegmentBuffer() const
unsigned getNumFreeUserSGPRs()
bool hasImplicitBufferPtr() const
bool hasPrivateSegmentSize() const
bool hasDispatchPtr() const
bool hasFlatScratchInit() const
const MachineFunction & getMachineFunction() const
void computeKnownBitsImpl(Register R, KnownBits &Known, const APInt &DemandedElts, unsigned Depth=0)
Evaluate a known-bits query with an explicit worklist instead of recursive descent.
int64_t getOffset() const
LLVM_ABI unsigned getAddressSpace() const
const GlobalValue * getGlobal() const
bool hasExternalLinkage() const
unsigned getAddressSpace() const
Module * getParent()
Get the module that this global value is contained inside of...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Type * getValueType() const
LLVM_ABI uint64_t getGlobalSize(const DataLayout &DL) const
Get the size of this global variable in bytes.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
bool hasMetadata() const
Return true if this instruction has any metadata attached to it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
LLVM_ABI InstListType::iterator insertInto(BasicBlock *ParentBB, InstListType::iterator It)
Inserts an unlinked instruction into ParentBB at position It and returns the iterator of the inserted...
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
constexpr unsigned getScalarSizeInBits() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
static LLT integer(unsigned SizeInBits)
LLT changeElementSize(unsigned NewEltSize) const
If this type is a vector, return a vector with the same number of elements but the new element size.
This is an important class for using LLVM in a threaded context.
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
LLVM_ABI SyncScope::ID getOrInsertSyncScopeID(StringRef SSN)
getOrInsertSyncScopeID - Maps synchronization scope name to synchronization scope ID.
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
static unsigned getPointerOperandIndex()
This class is used to represent ISD::LOAD nodes.
TypeSize getValue() const
Describe properties that are true of each instruction in the target description file.
unsigned getID() const
getID() - Return the register class ID number.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
iterator begin() const
begin/end - Return all of the registers in this class.
Wrapper class representing physical registers. Should be passed by value.
LLVM_ABI MDNode * createRange(const APInt &Lo, const APInt &Hi)
Return metadata describing the range [Lo, Hi).
const MDOperand & getOperand(unsigned I) const
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
uint64_t getScalarSizeInBits() const
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static LLVM_ABI MVT getVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
static auto all_valuetypes()
SimpleValueType Iteration.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
static MVT getVectorVT(MVT VT, unsigned NumElements)
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
Align getAlignment() const
Return alignment of the basic block.
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
bool hasCalls() const
Return true if the current function has any function calls.
void setHasTailCall(bool V=true)
void setReturnAddressIsTaken(bool s)
bool hasStackObjects() const
Return true if there are any stack objects in this function.
PseudoSourceValueManager & getPSVManager() const
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
void push_back(MachineBasicBlock *MBB)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
LocationSize getSize() const
Return the size in bytes of the memory reference.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MONonTemporal
The memory access is non-temporal.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
Flags getFlags() const
Return the raw flags of the source value,.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
static MachineOperand CreateImm(int64_t Val)
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI bool isLiveIn(Register Reg) const
LLVM_ABI void setType(Register VReg, LLT Ty)
Set the low-level type of VReg to Ty.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register getLiveInVirtReg(MCRegister PReg) const
getLiveInVirtReg - If PReg is a live-in physical register, return the corresponding live-in virtual r...
const TargetRegisterClass * getRegClassOrNull(Register Reg) const
Return the register class of Reg, or null if Reg has not been assigned a register class yet.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
unsigned getNumVirtRegs() const
getNumVirtRegs - Return the number of virtual registers created.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
An SDNode that represents everything that will be needed to construct a MachineInstr.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
EVT getMemoryVT() const
Return the type of the in-memory value.
bool onlyWritesMemory() const
Whether this function only (at most) writes memory.
bool doesNotAccessMemory() const
Whether this function accesses no memory.
bool onlyReadsMemory() const
Whether this function only (at most) reads memory.
const Triple & getTargetTriple() const
Get the target triple which is a string describing the target host.
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
LLVM_ABI const PseudoSourceValue * getConstantPool()
Return a pseudo source value referencing the constant pool.
Wrapper class representing virtual and physical registers.
static Register index2VirtReg(unsigned Index)
Convert a 0-based index to a virtual register number.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
value_iterator value_end() const
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumValues() const
Return the number of values defined/returned by this operator.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
user_iterator user_begin() const
Provide iteration support to walk over all users of an SDNode.
op_iterator op_end() const
bool isAnyAdd() const
Returns true if the node type is ADD or PTRADD.
value_iterator value_begin() const
op_iterator op_begin() const
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
unsigned getNumOperands() const
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
bool isWholeWaveFunction() const
bool hasWorkGroupIDZ() const
AMDGPU::ClusterDimsAttr getClusterDims() const
SIModeRegisterDefaults getMode() const
std::tuple< const ArgDescriptor *, const TargetRegisterClass *, LLT > getPreloadedValue(AMDGPUFunctionArgInfo::PreloadedValue Value) const
unsigned getBytesInStackArgArea() const
const AMDGPUGWSResourcePseudoSourceValue * getGWSPSV(const AMDGPUTargetMachine &TM)
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
static bool isVGPRClass(const TargetRegisterClass *RC)
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override
Return true if folding a constant offset with the given GlobalAddress is legal.
bool isTypeDesirableForOp(unsigned Op, EVT VT) const override
Return true if the target has native support for the specified value type and it is 'desirable' to us...
SDNode * PostISelFolding(MachineSDNode *N, SelectionDAG &DAG) const override
Fold the instructions after selecting them.
SDValue splitTernaryVectorOp(SDValue Op, SelectionDAG &DAG) const
MachineSDNode * wrapAddr64Rsrc(SelectionDAG &DAG, const SDLoc &DL, SDValue Ptr) const
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
SDValue lowerGET_ROUNDING(SDValue Op, SelectionDAG &DAG) const
AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
bool requiresUniformRegister(MachineFunction &MF, const Value *V) const override
Allows target to decide about the register class of the specific value that is live outside the defin...
bool isFMADLegal(const SelectionDAG &DAG, const SDNode *N) const override
Returns true if be combined with to form an ISD::FMAD.
AtomicExpansionKind shouldExpandAtomicStoreInIR(StoreInst *SI) const override
Returns how the given (atomic) store should be expanded by the IR-level AtomicExpand pass into.
void bundleInstWithWaitcnt(MachineInstr &MI) const
Insert MI into a BUNDLE with an S_WAITCNT 0 immediately following it.
SDValue lowerROTR(SDValue Op, SelectionDAG &DAG) const
MVT getScalarShiftAmountTy(const DataLayout &, EVT) const override
Return the type to use for a scalar shift opcode, given the shifted amount type.
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
MVT getPointerTy(const DataLayout &DL, unsigned AS) const override
Map address space 7 to MVT::amdgpuBufferFatPointer because that's its in-memory representation.
bool denormalsEnabledForType(const SelectionDAG &DAG, EVT VT) const
void insertCopiesSplitCSR(MachineBasicBlock *Entry, const SmallVectorImpl< MachineBasicBlock * > &Exits) const override
Insert explicit copies in entry and exit blocks.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
Return the ValueType of the result of SETCC operations.
SDNode * legalizeTargetIndependentNode(SDNode *Node, SelectionDAG &DAG) const
Legalize target independent instructions (e.g.
bool allowsMisalignedMemoryAccessesImpl(unsigned Size, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *IsFast=nullptr) const
TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const override
Return the preferred vector type legalization action.
SDValue lowerFP_EXTEND(SDValue Op, SelectionDAG &DAG) const
const GCNSubtarget * getSubtarget() const
bool enableAggressiveFMAFusion(EVT VT) const override
Return true if target always benefits from combining into FMA for a given value type.
bool shouldEmitGOTReloc(const GlobalValue *GV) const
SDValue splitUnaryVectorOp(SDValue Op, SelectionDAG &DAG) const
SDValue lowerGET_FPENV(SDValue Op, SelectionDAG &DAG) const
bool isCanonicalized(SelectionDAG &DAG, SDValue Op, SDNodeFlags UserFlags={}, unsigned MaxDepth=5) const
void allocateSpecialInputSGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
void allocateLDSKernelId(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
SDValue LowerSTACKSAVE(SDValue Op, SelectionDAG &DAG) const
bool isReassocProfitable(SelectionDAG &DAG, SDValue N0, SDValue N1) const override
void allocateHSAUserSGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
ArrayRef< MCPhysReg > getRoundingControlRegisters() const override
Returns a 0 terminated array of rounding control registers that can be attached into strict FP call.
ConstraintType getConstraintType(StringRef Constraint) const override
Given a constraint, return the type of constraint it is for this target.
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent) const override
Return the register class that should be used for the specified value type.
void AddMemOpInit(MachineInstr &MI) const
MachineMemOperand::Flags getTargetMMOFlags(const Instruction &I) const override
This callback is used to inspect load/store instructions and add target-specific MachineMemOperand fl...
bool isLegalGlobalAddressingMode(const AddrMode &AM) const
bool shouldConvertConstantLoadToIntImm(const APInt &Imm, Type *Ty) const override
Return true if it is beneficial to convert a load of a constant to just the constant itself.
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
void emitExpandAtomicStore(StoreInst *SI) const override
Perform a atomic store using a target-specific way.
AtomicExpansionKind shouldExpandAtomicLoadInIR(LoadInst *LI) const override
Returns how the given (atomic) load should be expanded by the IR-level AtomicExpand pass.
Align computeKnownAlignForTargetInstr(GISelValueTracking &Analysis, Register R, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
Determine the known alignment for the pointer value R.
bool getAsmOperandConstVal(SDValue Op, uint64_t &Val) const
bool isShuffleMaskLegal(ArrayRef< int >, EVT) const override
Targets can use this to indicate that they only support some VECTOR_SHUFFLE operations,...
void emitExpandAtomicLoad(LoadInst *LI) const override
Perform a atomic load using a target-specific way.
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
Returns the target specific optimal type for load and store operations as a result of memset,...
void computeKnownBitsForStackObjectPointer(KnownBits &Known, const MachineFunction &MF, Align Alignment) const override
Determine known bits of a pointer to a known valid stack object.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const override
Lower the specified operand into the Ops vector.
LLT getPreferredShiftAmountTy(LLT Ty) const override
Return the preferred type to use for a shift opcode, given the shifted amount type is ShiftValueTy.
ExtractSubvectorCost getExtractSubvectorCost(EVT ResVT, EVT SrcVT, unsigned Index) const override
Return the cost of extracting a subvector of type ResVT from a vector of type SrcVT,...
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
Align getPrefLoopAlignment(MachineLoop *ML, const MachineBasicBlock *BlockToAlign) const override
Return the preferred loop alignment.
SDValue lowerSET_FPENV(SDValue Op, SelectionDAG &DAG) const
bool shouldPreservePtrArith(const Function &F, EVT PtrVT) const override
True if target has some particular form of dealing with pointer arithmetic semantics for pointers wit...
void getTgtMemIntrinsic(SmallVectorImpl< IntrinsicInfo > &, const CallBase &, MachineFunction &MF, unsigned IntrinsicID) const override
Given an intrinsic, checks if on the target the intrinsic will need to map to a MemIntrinsicNode (tou...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
SDValue lowerSET_ROUNDING(SDValue Op, SelectionDAG &DAG) const
void allocateSpecialInputVGPRsFixed(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
Allocate implicit function VGPR arguments in fixed registers.
MachineBasicBlock * emitGWSMemViolTestLoop(MachineInstr &MI, MachineBasicBlock *BB) const
bool getAddrModeArguments(const IntrinsicInst *I, SmallVectorImpl< Value * > &Ops, Type *&AccessTy) const override
CodeGenPrepare sinks address calculations into the same BB as Load/Store instructions reading the add...
bool checkAsmConstraintValA(SDValue Op, uint64_t Val, unsigned MaxSize=64) const
bool shouldEmitFixup(const GlobalValue *GV) const
MachineBasicBlock * splitKillBlock(MachineInstr &MI, MachineBasicBlock *BB) const
void emitExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) const override
Perform a cmpxchg expansion using a target-specific method.
bool canTransformPtrArithOutOfBounds(const Function &F, EVT PtrVT) const override
True if the target allows transformations of in-bounds pointer arithmetic that cause out-of-bounds in...
bool hasMemSDNodeUser(SDNode *N) const
bool isSDNodeSourceOfDivergence(const SDNode *N, FunctionLoweringInfo *FLI, UniformityInfo *UA) const override
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
bool isEligibleForTailCallOptimization(SDValue Callee, CallingConv::ID CalleeCC, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SmallVectorImpl< ISD::InputArg > &Ins, SelectionDAG &DAG) const
bool isMemOpHasNoClobberedMemOperand(const SDNode *N) const
bool isLegalFlatAddressingMode(const AddrMode &AM, unsigned AddrSpace) const
SDValue LowerCallResult(SDValue Chain, SDValue InGlue, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals, bool isThisReturn, SDValue ThisVal) const
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
bool isFPExtFoldable(const SelectionDAG &DAG, unsigned Opcode, EVT DestVT, EVT SrcVT) const override
Return true if an fpext operation input to an Opcode operation is free (for instance,...
void AdjustInstrPostInstrSelection(MachineInstr &MI, SDNode *Node) const override
Assign the register class depending on the number of bits set in the writemask.
MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void finalizeLowering(MachineFunction &MF) const override
Execute target specific actions to finalize target lowering.
static bool isNonGlobalAddrSpace(unsigned AS)
void emitExpandAtomicAddrSpacePredicate(Instruction *AI) const
MachineSDNode * buildRSRC(SelectionDAG &DAG, const SDLoc &DL, SDValue Ptr, uint32_t RsrcDword1, uint64_t RsrcDword2And3) const
Return a resource descriptor with the 'Add TID' bit enabled The TID (Thread ID) is multiplied by the ...
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
bool mayBeEmittedAsTailCall(const CallInst *) const override
Return true if the target may be able emit the call instruction as a tail call.
void passSpecialInputs(CallLoweringInfo &CLI, CCState &CCInfo, const SIMachineFunctionInfo &Info, SmallVectorImpl< std::pair< unsigned, SDValue > > &RegsToPass, SmallVectorImpl< SDValue > &MemOpChains, SDValue Chain) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
bool checkAsmConstraintVal(SDValue Op, StringRef Constraint, uint64_t Val) const
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
void emitExpandAtomicRMW(AtomicRMWInst *AI) const override
Perform a atomicrmw expansion using a target-specific way.
static bool shouldExpandVectorDynExt(unsigned EltSize, unsigned NumElem, bool IsDivergentIdx, const GCNSubtarget *Subtarget)
Check if EXTRACT_VECTOR_ELT/INSERT_VECTOR_ELT (<n x e>, var-idx) should be expanded into a set of cmp...
bool shouldUseLDSConstAddress(const GlobalValue *GV) const
bool supportSplitCSR(MachineFunction *MF) const override
Return true if the target supports that a subset of CSRs for the given machine function is handled ex...
bool isExtractVecEltCheap(EVT VT, unsigned Index) const override
Return true if extraction of a scalar element from the given vector type at the given index is cheap.
SDValue LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
bool allowsMisalignedMemoryAccesses(LLT Ty, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *IsFast=nullptr) const override
LLT handling variant.
bool canMergeStoresTo(unsigned AS, EVT MemVT, const MachineFunction &MF) const override
Returns if it's reasonable to merge stores to MemVT size.
SDValue lowerPREFETCH(SDValue Op, SelectionDAG &DAG) const
SITargetLowering(const TargetMachine &tm, const GCNSubtarget &STI)
void computeKnownBitsForTargetInstr(GISelValueTracking &Analysis, Register R, KnownBits &Known, const APInt &DemandedElts, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool isFreeAddrSpaceCast(const DataLayout &DL, unsigned SrcAS, unsigned DestAS) const override
Returns true if a cast from SrcAS to DestAS is "cheap", such that e.g.
bool shouldEmitPCReloc(const GlobalValue *GV) const
bool isUniformLoad(const LoadSDNode *Load) const
AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *AI) const override
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
void initializeSplitCSR(MachineBasicBlock *Entry) const override
Perform necessary initialization to handle a subset of CSRs explicitly via copies.
void allocateSpecialEntryInputVGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
void allocatePreloadKernArgSGPRs(CCState &CCInfo, SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< ISD::InputArg > &Ins, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
SDValue copyToM0(SelectionDAG &DAG, SDValue Chain, const SDLoc &DL, SDValue V) const
SDValue splitBinaryVectorOp(SDValue Op, SelectionDAG &DAG) const
MachinePointerInfo getKernargSegmentPtrInfo(MachineFunction &MF) const
unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
MVT getPointerMemTy(const DataLayout &DL, unsigned AS) const override
Similarly, the in-memory representation of a p7 is {p8, i32}, aka v8i32 when padding is added.
void allocateSystemSGPRs(CCState &CCInfo, MachineFunction &MF, SIMachineFunctionInfo &Info, CallingConv::ID CallConv, bool IsShader) const
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
unsigned getMaxPermittedBytesForAlignment(MachineBasicBlock *MBB) const override
Return the maximum amount of bytes allowed to be emitted when padding for alignment.
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
SDValue getExtractVectorElt(const SDLoc &DL, EVT VT, SDValue Vec, unsigned Idx)
Extract element at Idx from Vec.
const SDValue & getRoot() const
Return the root tag of the SelectionDAG.
bool isKnownNeverSNaN(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
const TargetSubtargetInfo & getSubtarget() const
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue getAtomicLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT MemVT, EVT VT, SDValue Chain, SDValue Ptr, MachineMemOperand *MMO)
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
LLVM_ABI bool isConstantIntBuildVectorOrConstantInt(SDValue N, bool AllowOpaques=true) const
Test whether the given value is a constant int or similar node.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI bool haveNoCommonBitsSet(SDValue A, SDValue B) const
Return true if A and B have no common bits set.
LLVM_ABI SDValue getAddrSpaceCast(const SDLoc &dl, EVT VT, SDValue Ptr, unsigned SrcAS, unsigned DestAS, const SDNodeFlags Flags=SDNodeFlags())
Return an AddrSpaceCastSDNode.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI bool SignBitIsZeroFP(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero, for a floating-point value.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getAtomic(unsigned Opcode, const SDLoc &dl, EVT MemVT, SDValue Chain, SDValue Ptr, SDValue Val, MachineMemOperand *MMO)
Gets a node for an atomic op, produces result (if relevant) and chain and takes 2 operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
std::pair< SDValue, SDValue > SplitVectorOperand(const SDNode *N, unsigned OpNo)
Split the node's operand with EXTRACT_SUBVECTOR and return the low/high part.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
LLVM_ABI std::pair< EVT, EVT > GetSplitDestVTs(const EVT &VT) const
Compute the VTs needed for the low/hi parts of a type which is split (or expanded) into two not neces...
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getBitcastedAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by first bitcasting (from potentia...
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI void setNodeMemRefs(MachineSDNode *N, ArrayRef< MachineMemOperand * > NewMemRefs)
Mutate the specified machine node's memory references to the provided list.
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getTokenFactor(const SDLoc &DL, SmallVectorImpl< SDValue > &Vals)
Creates a new TokenFactor containing Vals.
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI SDValue getMemBasePlusOffset(SDValue Base, TypeSize Offset, const SDLoc &DL, const SDNodeFlags Flags=SDNodeFlags())
Returns sum of the base pointer and offset.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI void RemoveDeadNode(SDNode *N)
Remove the specified node from the system.
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either any-extending or truncat...
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
LLVM_ABI bool isBaseWithConstantOffset(SDValue Op) const
Return true if the specified operand is an ISD::ADD with a ConstantSDNode on the right-hand side,...
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getErrorMergeValues(ArrayRef< EVT > ResultTypes, SDValue Chain, const SDLoc &dl)
Return poison values for each of ResultTypes, substituting Chain for any result of type MVT::Other,...
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI SDValue getCondCode(ISD::CondCode Cond)
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
const SDValue & setRoot(SDValue N)
Set the current root tag of the SelectionDAG.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
int getMaskElt(unsigned Idx) const
ArrayRef< int > getMask() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
constexpr bool empty() const
Check if the string is empty.
constexpr size_t size() const
Get the string size.
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Information about stack frame layout on the target.
Align getStackAlign() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
StackDirection getStackGrowthDirection() const
getStackGrowthDirection - Return the direction the stack grows
TargetInstrInfo - Interface to description of machine instruction set.
Type * Ty
Same as OrigTy, or partially legalized for soft float libcalls.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
virtual void finalizeLowering(MachineFunction &MF) const
Execute target specific actions to finalize target lowering.
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
virtual Align getPrefLoopAlignment(MachineLoop *ML=nullptr, const MachineBasicBlock *BlockToAlign=nullptr) const
Return the preferred loop alignment.
virtual unsigned getMaxPermittedBytesForAlignment(MachineBasicBlock *MBB) const
Return the maximum amount of bytes allowed to be emitted when padding for alignment.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void setOperationPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
Convenience method to set an operation to Promote and specify the type in a single call.
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
void setHasExtractBitsInsn(bool hasExtractInsn=true)
Tells the code generator that the target has BitExtract instructions.
virtual TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const
Return the preferred vector type legalization action.
virtual unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
Register getStackPointerRegisterToSaveRestore() const
If a physical register, this specifies the register that llvm.savestack/llvm.restorestack should save...
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
ExtractSubvectorCost
Enum that specifies how expensive lowering an EXTRACT_SUBVECTOR is.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
@ ZeroOrOneBooleanContent
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
virtual bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
bool allowsMemoryAccessForAlignment(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
This function returns true if the memory access is aligned or if the target allows this specific unal...
virtual MVT getPointerMemTy(const DataLayout &DL, uint32_t AS=0) const
Return the in-memory pointer type for the given address space, defaults to the pointer type from the ...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
LegalizeAction getOperationAction(unsigned Op, EVT VT) const
Return how this operation should be treated: either it is legal, needs to be promoted to a larger siz...
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
std::vector< AsmOperandInfo > AsmOperandInfoVector
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
bool parametersInCSRMatch(const MachineRegisterInfo &MRI, const uint32_t *CallerPreservedMask, const SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< SDValue > &OutVals) const
Check whether parameters to a call that are passed in callee saved registers are the same as from the...
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
SDValue expandFMINIMUMNUM_FMAXIMUMNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimumnum/fmaximumnum into multiple comparison with selects.
virtual bool isTypeDesirableForOp(unsigned, EVT VT) const
Return true if the target has native support for the specified value type and it is 'desirable' to us...
virtual void computeKnownBitsForStackObjectPointer(KnownBits &Known, const MachineFunction &MF, Align Alignment) const
Determine known bits of a pointer to a known valid stack object.
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
TargetLowering(const TargetLowering &)=delete
virtual MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
virtual AsmOperandInfoVector ParseConstraints(const DataLayout &DL, const TargetRegisterInfo *TRI, const CallBase &Call) const
Split up the constraint string from the inline assembly value into the specific constraints and their...
SDValue expandRoundInexactToOdd(EVT ResultVT, SDValue Op, const SDLoc &DL, SelectionDAG &DAG) const
Truncate Op to ResultVT.
virtual void ComputeConstraintToUse(AsmOperandInfo &OpInfo, SDValue Op, SelectionDAG *DAG=nullptr) const
Determines the constraint code and constraint type to use for the specific AsmOperandInfo,...
SDValue annotateStackObjectPointer(SDValue Ptr, SelectionDAG &DAG, const SDLoc &DL, Align Alignment) const
Annotate a stack object pointer with known-bits assertions.
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
SDValue expandFMINNUM_FMAXNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminnum/fmaxnum into fminnum_ieee/fmaxnum_ieee with quieted inputs.
Primary interface to the complete machine description for the target machine.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
const Triple & getTargetTriple() const
bool shouldAssumeDSOLocal(const GlobalValue *GV) const
unsigned GuaranteedTailCallOpt
GuaranteedTailCallOpt - This flag is enabled when -tailcallopt is specified on the commandline.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
OSType getOS() const
Get the parsed operating system type of this triple.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
static constexpr TypeSize getFixed(ScalarTy ExactSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
bool isFunctionTy() const
True if this is an instance of FunctionType.
bool isIntegerTy() const
True if this is an instance of IntegerType.
LLVM_ABI const fltSemantics & getFltSemantics() const
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
LLVM_ABI void set(Value *Val)
User * getUser() const
Returns the User that contains this Use.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
LLVMContext & getContext() const
All values hold a context through their type.
iterator_range< user_iterator > users()
iterator_range< use_iterator > uses()
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ BUFFER_STRIDED_POINTER
Address space for 192-bit fat buffer pointers with an additional index.
@ BARRIER
Address space for modeling barrier IDs as addresses.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ STREAMOUT_REGISTER
Internal address spaces. Can be freely renumbered.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ BUFFER_FAT_POINTER
Address space for 160-bit buffer fat pointers.
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumVGPRs[]
Key for Kernel::CodeProps::Metadata::mNumVGPRs.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
LLVM_READONLY const MIMGG16MappingInfo * getMIMGG16MappingInfo(unsigned G)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isShader(CallingConv::ID CC)
bool shouldEmitConstantsToTextSection(const Triple &TT)
bool isFlatGlobalAddrSpace(unsigned AS)
const uint64_t FltRoundToHWConversionTable
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
bool isGFX11(const MCSubtargetInfo &STI)
bool isGFX13(const MCSubtargetInfo &STI)
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val)
Checks if Val is inside MD, a !range-like metadata.
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READNONE constexpr bool mayTailCallThisCC(CallingConv::ID CC)
Return true if we might ever do TCO for calls with this calling convention.
unsigned getAMDHSACodeObjectVersion(const Module &M)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords, bool IndexedRsrc, bool IndexedSamp)
LLVM_READNONE constexpr bool isKernel(CallingConv::ID CC)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isCompute(CallingConv::ID CC)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
bool getMUBUFTfe(unsigned Opc)
TargetExtType * isNamedBarrier(const GlobalVariable &GV)
LLVM_READONLY int32_t getGlobalSaddrOp(uint32_t Opcode)
bool isGFX11Plus(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV2F16(uint32_t Literal)
std::tuple< char, unsigned, unsigned > parseAsmConstraintPhysReg(StringRef Constraint)
Returns a valid charcode or 0 in the first entry if this is a valid physical register constraint.
bool isGFX10Plus(const MCSubtargetInfo &STI)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ TowardZeroF32_TowardNegativeF64
bool isUniformMMO(const MachineMemOperand *MMO)
std::optional< unsigned > getInlineEncodingV2I16(uint32_t Literal)
uint32_t decodeFltRoundToHWConversionTable(uint32_t FltRounds)
Read the hardware rounding mode equivalent of a AMDGPUFltRounds value.
bool isExtendedGlobalAddrSpace(unsigned AS)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfo(unsigned DimEnum)
std::optional< unsigned > getInlineEncodingV2BF16(uint32_t Literal)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
unsigned getSyntheticApertureNumber(unsigned AS)
LLVM_READNONE constexpr bool isChainCC(CallingConv::ID CC)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool canGuaranteeTCO(CallingConv::ID CC)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const RsrcIntrinsic * lookupRsrcIntrinsic(unsigned Intr)
const uint64_t FltRoundConversionTable
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ MaxID
The highest possible ID. Must be some 2^k - 1.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ PTRADD
PTRADD represents pointer arithmetic semantics, for targets that opt in using shouldPreservePtrArith(...
@ DELETED_NODE
DELETED_NODE - This is an illegal value that is used to catch errors.
@ POISON
POISON - A poison node.
@ SET_FPENV
Sets the current floating-point environment.
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
@ BSWAP
Byte Swap and Counting operators.
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
@ ADD
Simple integer binary arithmetic operators.
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
@ ATOMIC_CMP_SWAP_WITH_SUCCESS
Val, Success, OUTCHAIN = ATOMIC_CMP_SWAP_WITH_SUCCESS(INCHAIN, ptr, cmp, swap) N.b.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
@ FADD
Simple binary floating point operators.
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ CONVERT_FROM_ARBITRARY_FP
CONVERT_FROM_ARBITRARY_FP - This operator converts from an arbitrary floating-point represented as an...
@ SET_ROUNDING
Set rounding mode.
@ CONVERGENCECTRL_GLUE
This does not correspond to any convergence control intrinsic.
@ SIGN_EXTEND
Conversion operators.
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
@ READSTEADYCOUNTER
READSTEADYCOUNTER - This corresponds to the readfixedcounter intrinsic.
@ BR
Control flow instructions. These all have token chains.
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ SSUBO
Same for subtraction.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
@ CTLS
Count leading redundant sign bits.
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ GET_FPMODE
Reads the current dynamic floating-point control modes.
@ GET_FPENV
Gets the current floating-point environment.
@ SHL
Shift and rotation operations.
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
@ INLINEASM_BR
INLINEASM_BR - Branching version of inline asm. Used by asm-goto.
@ BF16_TO_FP
BF16_TO_FP, FP_TO_BF16 - These operators are used to perform promotions and truncation for bfloat16.
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ READCYCLECOUNTER
READCYCLECOUNTER - This corresponds to the readcyclecounter intrinsic.
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ SPONENTRY
SPONENTRY - Represents the llvm.sponentry intrinsic.
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ INLINEASM
INLINEASM - Represents an inline asm block.
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
@ BRCOND
BRCOND - Conditional branch.
@ CONVERT_TO_ARBITRARY_FP
CONVERT_TO_ARBITRARY_FP - Converts a native FP value to an arbitrary floating-point format,...
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
bool isSignedIntSetCC(CondCode Code)
Return true if this is a setcc instruction that performs a signed comparison when used with integer o...
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getDeclarationIfExists(const Module *M, ID id)
Look up the Function declaration of the intrinsic id in the Module M and return it if it exists.
LLVM_ABI AttributeSet getFnAttributes(LLVMContext &C, ID id)
Return the function attributes for an intrinsic.
LLVM_ABI StringRef getBaseName(ID id)
Return the LLVM name for an intrinsic, without encoded types for overloading, such as "llvm....
LLVM_ABI AttributeList getAttributes(LLVMContext &C, ID id, FunctionType *FT)
Return the attributes for an intrinsic.
LLVM_ABI FunctionType * getType(LLVMContext &Context, ID id, ArrayRef< Type * > OverloadTys={})
Return the function type for an intrinsic.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
GFCstOrSplatGFCstMatch m_GFCstOrSplat(std::optional< FPValueAndVReg > &FPValReg)
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_FAbs(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_IntrinsicWOChain(const OpndPreds &...Opnds)
bool sd_match(SDValue N, Pattern &&P)
ConstantInt_match m_ConstInt()
Match any integer constants or splat of an integer constant.
@ System
Synchronized with respect to all concurrently executing threads.
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< UseNode * > Use
NodeAddr< NodeBase * > Node
friend class Instruction
Iterator for Instructions in a `BasicBlock.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
GenericUniformityInfo< SSAContext > UniformityInfo
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
int bit_width(T Value)
Returns the number of bits needed to represent Value if Value is nonzero.
SDValue peekFPSignOps(SDValue Val)
Strip fabs/fneg/fcopysign from a value to get the underlying source.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
MemoryEffectsBase< IRMemLocation > MemoryEffects
Summary of how a function affects memory in the program.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
AtomicOrderingCABI
Atomic ordering for C11 / C++11's memory models.
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
bool isBoolSGPR(SDValue V)
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI Value * buildAtomicRMWValue(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *Val)
Emit IR to implement the given atomicrmw operation on values in registers, returning the new value.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ NearestTiesToEven
roundTiesToEven.
unsigned M0(unsigned Val)
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::pair< Value *, Value * > buildCmpXchgValue(IRBuilderBase &Builder, Value *Ptr, Value *Cmp, Value *Val, Align Alignment, bool IsVolatile=false)
Emit IR to implement the given cmpxchg operation on values in registers, returning the new value.
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
std::optional< StringRef > getAtomicScopeIRString(const Triple &T, AtomicScope S, bool IsSingleAddressSpace=false)
Returns the LLVM IR syncscope string that T uses to spell S.
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
constexpr RegState getUndefRegState(bool B)
@ Custom
The result value requires a custom uniformity check.
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ CLUSTER_WORKGROUP_MAX_ID_X
@ CLUSTER_WORKGROUP_MAX_ID_Z
@ CLUSTER_WORKGROUP_MAX_FLAT_ID
@ CLUSTER_WORKGROUP_MAX_ID_Y
ArgDescriptor WorkItemIDZ
ArgDescriptor WorkItemIDY
std::tuple< const ArgDescriptor *, const TargetRegisterClass *, LLT > getPreloadedValue(PreloadedValue Value) const
ArgDescriptor WorkItemIDX
static const AMDGPUFunctionArgInfo FixedABIFunctionInfo
static constexpr uint64_t encode(Fields... Values)
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
unsigned AtomicNoRetBaseOpcode
This struct is a compact representation of a valid (non-zero power of two) alignment.
MCRegister getRegister() const
static ArgDescriptor createArg(const ArgDescriptor &Arg, unsigned Mask)
static ArgDescriptor createRegister(Register Reg, unsigned Mask=~0u)
Helper struct shared between Function Specialization and SCCP Solver.
Represents the full denormal controls for a function, including the default mode and the f32 specific...
Represent subnormal handling kind for floating point instruction inputs and outputs.
@ Dynamic
Denormals have unknown treatment.
static constexpr DenormalMode getPreserveSign()
static constexpr DenormalMode getIEEE()
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
ElementCount getVectorElementCount() const
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
bool isByteSized() const
Return true if the bit size is a multiple of 8.
uint64_t getScalarSizeInBits() const
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool isVector() const
Return true if this is a vector value type.
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
bool isVectorOf(EVT EltVT) const
Return true if this is a vector with matching element type.
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
LLVM_ABI const fltSemantics & getFltSemantics() const
Returns an APFloat semantics tag appropriate for the value type.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
unsigned getPointerAddrSpace() const
unsigned getByValSize() const
Align getNonZeroMemAlign() const
OutputArg - This struct carries flags and a value for a single outgoing (actual) argument or outgoing...
static LLVM_ABI std::optional< bool > eq(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_EQ result.
bool isUnknown() const
Returns true if we don't know any bits.
KnownBits trunc(unsigned BitWidth) const
Return known bits for a truncation of the value we're tracking.
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
static LLVM_ABI std::optional< bool > ule(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_ULE result.
static LLVM_ABI std::optional< bool > uge(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_UGE result.
bool isKnownNeverNaN() const
Return true if it's known this can never be a nan.
static LLVM_ABI KnownFPClass bitcast(const fltSemantics &FltSemantics, const KnownBits &Bits)
Report known values for a bitcast into a float with provided semantics.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
These are IR-level optimization flags that may be propagated to SDNodes.
bool hasNoUnsignedWrap() const
bool hasAllowContract() const
bool hasNoSignedWrap() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
std::optional< unsigned > fallbackAddressSpace
This structure contains all information that is necessary for lowering calls.
SDValue ConvergenceControlToken
SmallVector< ISD::InputArg, 32 > Ins
SmallVector< ISD::OutputArg, 32 > Outs
SmallVector< SDValue, 32 > OutVals
bool isBeforeLegalize() const