27#include "llvm/IR/IntrinsicsAMDGPU.h"
31#ifdef EXPENSIVE_CHECKS
36#define DEBUG_TYPE "amdgpu-isel"
51 In = stripBitcast(In);
57 Out = In.getOperand(0);
68 if (ShiftAmt->getZExtValue() == 16) {
88 if (
Lo->isDivergent()) {
90 SL,
Lo.getValueType()),
98 Src.getValueType(),
Ops),
116 SDValue Idx = In.getOperand(1);
118 return In.getOperand(0);
122 SDValue Src = In.getOperand(0);
123 if (Src.getValueType().getSizeInBits() == 32)
124 return stripBitcast(Src);
134 assert(Elts.
size() == SubRegClass.
size() &&
"array size mismatch");
135 unsigned NumElts = Elts.
size();
138 for (
unsigned i = 0; i < NumElts; ++i) {
139 Ops[2 * i + 1] = Elts[i];
149 "AMDGPU DAG->DAG Pattern Instruction Selection",
false,
153#ifdef EXPENSIVE_CHECKS
158 "AMDGPU DAG->DAG Pattern Instruction Selection",
false,
179bool AMDGPUDAGToDAGISel::fp16SrcZerosHighBits(
unsigned Opc)
const {
215 case AMDGPUISD::FRACT:
216 case AMDGPUISD::CLAMP:
217 case AMDGPUISD::COS_HW:
218 case AMDGPUISD::SIN_HW:
219 case AMDGPUISD::FMIN3:
220 case AMDGPUISD::FMAX3:
221 case AMDGPUISD::FMED3:
222 case AMDGPUISD::FMAD_FTZ:
225 case AMDGPUISD::RCP_IFLAG:
235 case AMDGPUISD::DIV_FIXUP:
245#ifdef EXPENSIVE_CHECKS
249 assert(L->isLCSSAForm(DT));
257#ifdef EXPENSIVE_CHECKS
265 assert(Subtarget->d16PreservesUnusedBits());
266 MVT VT =
N->getValueType(0).getSimpleVT();
267 if (VT != MVT::v2i16 && VT != MVT::v2f16)
289 unsigned LoadOp = AMDGPUISD::LOAD_D16_HI;
292 AMDGPUISD::LOAD_D16_HI_I8 : AMDGPUISD::LOAD_D16_HI_U8;
298 CurDAG->getMemIntrinsicNode(LoadOp,
SDLoc(LdHi), VTList,
311 if (LdLo &&
Lo.hasOneUse()) {
317 unsigned LoadOp = AMDGPUISD::LOAD_D16_LO;
320 AMDGPUISD::LOAD_D16_LO_I8 : AMDGPUISD::LOAD_D16_LO_U8;
332 CurDAG->getMemIntrinsicNode(LoadOp,
SDLoc(LdLo), VTList,
346 EVT VT =
N->getValueType(0);
359 Ld ?
CurDAG->getExtLoad(ExtType, SL, MVT::i32, Mem->getChain(),
360 Mem->getBasePtr(), Mem->getMemoryVT(),
361 Mem->getMemOperand())
362 :
CurDAG->getAtomicLoad(ExtType, SL, Mem->getMemoryVT(), MVT::i32,
363 Mem->getChain(), Mem->getBasePtr(),
364 Mem->getMemOperand());
375 bool MadeChange =
false;
376 while (Position !=
CurDAG->allnodes_begin()) {
381 switch (
N->getOpcode()) {
384 if (Subtarget->d16PreservesUnusedBits())
389 if (Subtarget->useRealTrue16Insts())
398 CurDAG->RemoveDeadNodes();
404bool AMDGPUDAGToDAGISel::isInlineImmediate(
const SDNode *
N)
const {
410 return TII->isInlineConstant(
C->getAPIntValue());
413 return TII->isInlineConstant(
C->getValueAPF());
423 unsigned OpNo)
const {
424 if (!
N->isMachineOpcode()) {
427 if (
Reg.isVirtual()) {
432 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
433 return TRI->getPhysRegBaseClass(
Reg);
439 switch (
N->getMachineOpcode()) {
441 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
442 const MCInstrDesc &
Desc =
TII->get(
N->getMachineOpcode());
443 unsigned OpIdx =
Desc.getNumDefs() + OpNo;
444 if (OpIdx >=
Desc.getNumOperands())
447 int16_t RegClass =
TII->getOpRegClassID(
Desc.operands()[OpIdx]);
451 return Subtarget->getRegisterInfo()->getRegClass(RegClass);
453 case AMDGPU::REG_SEQUENCE: {
454 unsigned RCID =
N->getConstantOperandVal(0);
456 Subtarget->getRegisterInfo()->getRegClass(RCID);
458 SDValue SubRegOp =
N->getOperand(OpNo + 1);
460 return Subtarget->getRegisterInfo()->getSubClassWithSubReg(SuperRC,
469 Ops.push_back(NewChain);
470 for (
unsigned i = 1, e =
N->getNumOperands(); i != e; ++i)
471 Ops.push_back(
N->getOperand(i));
474 return CurDAG->MorphNodeTo(
N,
N->getOpcode(),
N->getVTList(),
Ops);
481 assert(
N->getOperand(0).getValueType() == MVT::Other &&
"Expected chain");
484 return glueCopyToOp(
N,
M0,
M0.getValue(1));
487SDNode *AMDGPUDAGToDAGISel::glueCopyToM0LDSInit(
SDNode *
N)
const {
490 if (Subtarget->ldsRequiresM0Init())
492 N,
CurDAG->getSignedTargetConstant(-1, SDLoc(
N), MVT::i32));
495 unsigned Value =
MF.getInfo<SIMachineFunctionInfo>()->getGDSSize();
497 glueCopyToM0(
N,
CurDAG->getTargetConstant(
Value, SDLoc(
N), MVT::i32));
504 SDNode *
Lo =
CurDAG->getMachineNode(
505 AMDGPU::S_MOV_B32,
DL, MVT::i32,
507 SDNode *
Hi =
CurDAG->getMachineNode(
508 AMDGPU::S_MOV_B32,
DL, MVT::i32,
510 const SDValue
Ops[] = {
511 CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID,
DL, MVT::i32),
512 SDValue(
Lo, 0),
CurDAG->getTargetConstant(AMDGPU::sub0,
DL, MVT::i32),
513 SDValue(
Hi, 0),
CurDAG->getTargetConstant(AMDGPU::sub1,
DL, MVT::i32)};
515 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE,
DL, VT,
Ops);
518SDNode *AMDGPUDAGToDAGISel::packConstantV2I16(
const SDNode *
N,
523 uint32_t LHSVal, RHSVal;
527 uint32_t
K = (LHSVal & 0xffff) | (RHSVal << 16);
529 isVGPRImm(
N) ? AMDGPU::V_MOV_B32_e32 : AMDGPU::S_MOV_B32, SL,
537 EVT VT =
N->getValueType(0);
541 SDValue RegClass =
CurDAG->getTargetConstant(RegClassID,
DL, MVT::i32);
543 if (NumVectorElts == 1) {
544 CurDAG->SelectNodeTo(
N, AMDGPU::COPY_TO_REGCLASS, EltVT,
N->getOperand(0),
549 bool IsGCN =
CurDAG->getSubtarget().getTargetTriple().isAMDGCN();
550 if (IsGCN && Subtarget->has64BitLiterals() && VT.
getSizeInBits() == 64 &&
553 bool AllConst =
true;
555 for (
unsigned I = 0;
I < NumVectorElts; ++
I) {
563 Val = CF->getValueAPF().bitcastToAPInt().getZExtValue();
566 C |= Val << (EltSize *
I);
571 CurDAG->getMachineNode(AMDGPU::S_MOV_B64_IMM_PSEUDO,
DL, VT, CV);
572 CurDAG->SelectNodeTo(
N, AMDGPU::COPY_TO_REGCLASS, VT,
SDValue(Copy, 0),
578 assert(NumVectorElts <= 32 &&
"Vectors with more than 32 elements not "
585 RegSeqArgs[0] =
CurDAG->getTargetConstant(RegClassID,
DL, MVT::i32);
586 bool IsRegSeq =
true;
587 unsigned NOps =
N->getNumOperands();
589 assert(IsGCN || EltSizeInRegs == 1);
590 for (
unsigned i = 0; i < NOps; i++) {
597 i * EltSizeInRegs, EltSizeInRegs)
599 RegSeqArgs[1 + (2 * i)] =
N->getOperand(i);
600 RegSeqArgs[1 + (2 * i) + 1] =
CurDAG->getTargetConstant(
Sub,
DL, MVT::i32);
602 if (NOps != NumVectorElts) {
607 for (
unsigned i = NOps; i < NumVectorElts; ++i) {
609 i * EltSizeInRegs, EltSizeInRegs)
611 RegSeqArgs[1 + (2 * i)] =
SDValue(ImpDef, 0);
612 RegSeqArgs[1 + (2 * i) + 1] =
619 CurDAG->SelectNodeTo(
N, AMDGPU::REG_SEQUENCE,
N->getVTList(), RegSeqArgs);
623 EVT VT =
N->getValueType(0);
627 if (!Subtarget->hasPkMovB32() || !EltVT.
bitsEq(MVT::i32) ||
641 Mask[0] < 4 && Mask[1] < 4);
643 SDValue VSrc0 = Mask[0] < 2 ? Src0 : Src1;
644 SDValue VSrc1 = Mask[1] < 2 ? Src0 : Src1;
645 unsigned Src0SubReg = Mask[0] & 1 ? AMDGPU::sub1 : AMDGPU::sub0;
646 unsigned Src1SubReg = Mask[1] & 1 ? AMDGPU::sub1 : AMDGPU::sub0;
649 Src0SubReg = Src1SubReg;
651 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,
DL, VT);
656 Src1SubReg = Src0SubReg;
658 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,
DL, VT);
668 if (
N->isDivergent() && Src0SubReg == AMDGPU::sub1 &&
669 Src1SubReg == AMDGPU::sub0) {
685 SDValue Src0OpSelVal =
CurDAG->getTargetConstant(Src0OpSel,
DL, MVT::i32);
686 SDValue Src1OpSelVal =
CurDAG->getTargetConstant(Src1OpSel,
DL, MVT::i32);
689 CurDAG->SelectNodeTo(
N, AMDGPU::V_PK_MOV_B32,
N->getVTList(),
690 {Src0OpSelVal, VSrc0, Src1OpSelVal, VSrc1,
700 CurDAG->getTargetExtractSubreg(Src0SubReg,
DL, EltVT, VSrc0);
702 CurDAG->getTargetExtractSubreg(Src1SubReg,
DL, EltVT, VSrc1);
705 CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID,
DL, MVT::i32),
706 ResultElt0,
CurDAG->getTargetConstant(AMDGPU::sub0,
DL, MVT::i32),
707 ResultElt1,
CurDAG->getTargetConstant(AMDGPU::sub1,
DL, MVT::i32)};
708 CurDAG->SelectNodeTo(
N, TargetOpcode::REG_SEQUENCE, VT,
Ops);
712 unsigned int Opc =
N->getOpcode();
713 if (
N->isMachineOpcode()) {
721 N = glueCopyToM0LDSInit(
N);
731 if (
N->getValueType(0) == MVT::i64) {
732 SelectAddcSubbI64(
N);
736 if (
N->getValueType(0) != MVT::i32)
743 if (
N->getValueType(0) == MVT::i64) {
744 SelectAddcSubbI64(
N);
748 SelectUADDO_USUBO(
N);
751 case AMDGPUISD::FMUL_W_CHAIN: {
752 SelectFMUL_W_CHAIN(
N);
755 case AMDGPUISD::FMA_W_CHAIN: {
756 SelectFMA_W_CHAIN(
N);
762 EVT VT =
N->getValueType(0);
780 N->isDivergent() ?
TRI->getDefaultVectorSuperClassForBitWidth(VecInBits)
792 if (
N->getValueType(0) == MVT::i128) {
793 RC =
CurDAG->getTargetConstant(AMDGPU::SGPR_128RegClassID,
DL, MVT::i32);
794 SubReg0 =
CurDAG->getTargetConstant(AMDGPU::sub0_sub1,
DL, MVT::i32);
795 SubReg1 =
CurDAG->getTargetConstant(AMDGPU::sub2_sub3,
DL, MVT::i32);
796 }
else if (
N->getValueType(0) == MVT::i64) {
797 RC =
CurDAG->getTargetConstant(AMDGPU::SReg_64RegClassID,
DL, MVT::i32);
798 SubReg0 =
CurDAG->getTargetConstant(AMDGPU::sub0,
DL, MVT::i32);
799 SubReg1 =
CurDAG->getTargetConstant(AMDGPU::sub1,
DL, MVT::i32);
803 const SDValue Ops[] = { RC,
N->getOperand(0), SubReg0,
804 N->getOperand(1), SubReg1 };
806 N->getValueType(0),
Ops));
812 if (
N->getValueType(0).getSizeInBits() != 64 || isInlineImmediate(
N) ||
813 Subtarget->has64BitLiterals())
818 Imm =
FP->getValueAPF().bitcastToAPInt().getZExtValue();
823 Imm =
C->getZExtValue();
832 case AMDGPUISD::BFE_I32:
833 case AMDGPUISD::BFE_U32: {
859 case AMDGPUISD::DIV_SCALE: {
870 return SelectMUL_LOHI(
N);
881 if (
N->getValueType(0) != MVT::i32)
892 case AMDGPUISD::CVT_PKRTZ_F16_F32:
893 case AMDGPUISD::CVT_PKNORM_I16_F32:
894 case AMDGPUISD::CVT_PKNORM_U16_F32:
895 case AMDGPUISD::CVT_PK_U16_U32:
896 case AMDGPUISD::CVT_PK_I16_I32: {
898 if (
N->getValueType(0) == MVT::i32) {
899 MVT NewVT =
Opc == AMDGPUISD::CVT_PKRTZ_F16_F32 ? MVT::v2f16 : MVT::v2i16;
901 { N->getOperand(0), N->getOperand(1) });
909 SelectINTRINSIC_W_CHAIN(
N);
913 SelectINTRINSIC_WO_CHAIN(
N);
917 SelectINTRINSIC_VOID(
N);
921 SelectWAVE_ADDRESS(
N);
925 SelectSTACKRESTORE(
N);
934 if (!Subtarget->hasSDWA())
944 return RHS->getZExtValue() == 0xFF || RHS->getZExtValue() == 0xFFFF;
948 return (RHS->getZExtValue() % 8) == 0;
953bool AMDGPUDAGToDAGISel::isUniformBr(
const SDNode *
N)
const {
956 return Term->getMetadata(
"amdgpu.uniform") ||
957 Term->getMetadata(
"structurizecfg.uniform");
960bool AMDGPUDAGToDAGISel::isUnneededShiftMask(
const SDNode *
N,
961 unsigned ShAmtBits)
const {
964 const APInt &
RHS =
N->getConstantOperandAPInt(1);
965 if (
RHS.countr_one() >= ShAmtBits)
995 N1 =
Lo.getOperand(1);
1005 if (
CurDAG->isBaseWithConstantOffset(Addr)) {
1020 return "AMDGPU DAG->DAG Pattern Instruction Selection";
1036#ifdef EXPENSIVE_CHECKS
1039 for (
auto &L : LI.getLoopsInPreorder())
1040 assert(L->isLCSSAForm(DT) &&
"Loop is not in LCSSA form!");
1062 }
else if ((Addr.
getOpcode() == AMDGPUISD::DWORDADDR) &&
1064 Base =
CurDAG->getRegister(R600::INDIRECT_BASE_ADDR, MVT::i32);
1078SDValue AMDGPUDAGToDAGISel::getMaterializedScalarImm32(int64_t Val,
1080 SDNode *Mov =
CurDAG->getMachineNode(
1081 AMDGPU::S_MOV_B32,
DL, MVT::i32,
1082 CurDAG->getTargetConstant(Val,
DL, MVT::i32));
1083 return SDValue(Mov, 0);
1086void AMDGPUDAGToDAGISel::SelectAddcSubb(
SDNode *
N) {
1087 SDValue
LHS =
N->getOperand(0);
1088 SDValue
RHS =
N->getOperand(1);
1089 SDValue CI =
N->getOperand(2);
1091 if (
N->isDivergent()) {
1093 : AMDGPU::V_SUBB_U32_e64;
1095 N,
Opc,
N->getVTList(),
1097 CurDAG->getTargetConstant(0, {}, MVT::i1) });
1100 : AMDGPU::S_SUB_CO_PSEUDO;
1101 CurDAG->SelectNodeTo(
N,
Opc,
N->getVTList(), {LHS, RHS, CI});
1105void AMDGPUDAGToDAGISel::SelectAddcSubbI64(
SDNode *
N) {
1107 SDValue
LHS =
N->getOperand(0);
1108 SDValue
RHS =
N->getOperand(1);
1110 unsigned Opcode =
N->getOpcode();
1114 SDValue Sub0 =
CurDAG->getTargetConstant(AMDGPU::sub0,
DL, MVT::i32);
1115 SDValue Sub1 =
CurDAG->getTargetConstant(AMDGPU::sub1,
DL, MVT::i32);
1117 SDNode *Lo0 =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
DL,
1118 MVT::i32,
LHS, Sub0);
1119 SDNode *Hi0 =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
DL,
1120 MVT::i32,
LHS, Sub1);
1122 SDNode *Lo1 =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
DL,
1123 MVT::i32,
RHS, Sub0);
1124 SDNode *Hi1 =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
DL,
1125 MVT::i32,
RHS, Sub1);
1127 SDVTList VTList =
CurDAG->getVTList(MVT::i32,
N->getValueType(1));
1129 static const unsigned NoCarryOpcMap[2][2] = {
1130 {AMDGPU::S_USUBO_PSEUDO, AMDGPU::S_UADDO_PSEUDO},
1131 {AMDGPU::V_SUB_CO_U32_e64, AMDGPU::V_ADD_CO_U32_e64}};
1132 static const unsigned CarryOpcMap[2][2] = {
1133 {AMDGPU::S_SUB_CO_PSEUDO, AMDGPU::S_ADD_CO_PSEUDO},
1134 {AMDGPU::V_SUBB_U32_e64, AMDGPU::V_ADDC_U32_e64}};
1136 bool IsVALU =
N->isDivergent();
1138 unsigned NoCarryOpc = NoCarryOpcMap[IsVALU][IsAdd];
1139 unsigned CarryOpc = CarryOpcMap[IsVALU][IsAdd];
1140 SDValue Clamp =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
1143 if (!ConsumeCarry) {
1145 SDValue
Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0), Clamp};
1146 AddLo =
CurDAG->getMachineNode(NoCarryOpc,
DL, VTList, Args);
1148 SDValue
Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0)};
1149 AddLo =
CurDAG->getMachineNode(NoCarryOpc,
DL, VTList, Args);
1153 SDValue
Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0),
N->getOperand(2),
1155 AddLo =
CurDAG->getMachineNode(CarryOpc,
DL, VTList, Args);
1157 SDValue
Args[] = {SDValue(Lo0, 0), SDValue(Lo1, 0),
N->getOperand(2)};
1158 AddLo =
CurDAG->getMachineNode(CarryOpc,
DL, VTList, Args);
1164 SDValue
Args[] = {SDValue(Hi0, 0), SDValue(Hi1, 0), SDValue(AddLo, 1),
1166 AddHi =
CurDAG->getMachineNode(CarryOpc,
DL, VTList, Args);
1168 SDValue
Args[] = {SDValue(Hi0, 0), SDValue(Hi1, 0), SDValue(AddLo, 1)};
1169 AddHi =
CurDAG->getMachineNode(CarryOpc,
DL, VTList, Args);
1172 unsigned RC = IsVALU ? AMDGPU::VReg_64RegClassID : AMDGPU::SReg_64RegClassID;
1173 SDValue RegSequenceArgs[] = {
CurDAG->getTargetConstant(RC,
DL, MVT::i32),
1174 SDValue(AddLo, 0), Sub0, SDValue(AddHi, 0),
1177 MVT::i64, RegSequenceArgs);
1183void AMDGPUDAGToDAGISel::SelectUADDO_USUBO(
SDNode *
N) {
1188 bool IsVALU =
N->isDivergent();
1190 for (SDNode::user_iterator UI =
N->user_begin(),
E =
N->user_end(); UI !=
E;
1192 if (UI.getUse().getResNo() == 1) {
1193 if (UI->isMachineOpcode()) {
1194 if (UI->getMachineOpcode() !=
1195 (IsAdd ? AMDGPU::S_ADD_CO_PSEUDO : AMDGPU::S_SUB_CO_PSEUDO)) {
1208 unsigned Opc = IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
1211 N,
Opc,
N->getVTList(),
1212 {N->getOperand(0), N->getOperand(1),
1213 CurDAG->getTargetConstant(0, {}, MVT::i1) });
1215 unsigned Opc = IsAdd ? AMDGPU::S_UADDO_PSEUDO : AMDGPU::S_USUBO_PSEUDO;
1217 CurDAG->SelectNodeTo(
N,
Opc,
N->getVTList(),
1218 {N->getOperand(0), N->getOperand(1)});
1222void AMDGPUDAGToDAGISel::SelectFMA_W_CHAIN(
SDNode *
N) {
1226 SelectVOP3Mods0(
N->getOperand(1),
Ops[1],
Ops[0],
Ops[6],
Ops[7]);
1227 SelectVOP3Mods(
N->getOperand(2),
Ops[3],
Ops[2]);
1228 SelectVOP3Mods(
N->getOperand(3),
Ops[5],
Ops[4]);
1229 Ops[8] =
N->getOperand(0);
1230 Ops[9] =
N->getOperand(4);
1234 bool UseFMAC = Subtarget->hasDLInsts() &&
1238 unsigned Opcode = UseFMAC ? AMDGPU::V_FMAC_F32_e64 : AMDGPU::V_FMA_F32_e64;
1239 CurDAG->SelectNodeTo(
N, Opcode,
N->getVTList(),
Ops);
1242void AMDGPUDAGToDAGISel::SelectFMUL_W_CHAIN(
SDNode *
N) {
1246 SelectVOP3Mods0(
N->getOperand(1),
Ops[1],
Ops[0],
Ops[4],
Ops[5]);
1247 SelectVOP3Mods(
N->getOperand(2),
Ops[3],
Ops[2]);
1248 Ops[6] =
N->getOperand(0);
1249 Ops[7] =
N->getOperand(3);
1251 CurDAG->SelectNodeTo(
N, AMDGPU::V_MUL_F32_e64,
N->getVTList(),
Ops);
1256void AMDGPUDAGToDAGISel::SelectDIV_SCALE(
SDNode *
N) {
1257 EVT VT =
N->getValueType(0);
1259 assert(VT == MVT::f32 || VT == MVT::f64);
1262 = (VT == MVT::f64) ? AMDGPU::V_DIV_SCALE_F64_e64 : AMDGPU::V_DIV_SCALE_F32_e64;
1267 SelectVOP3BMods0(
N->getOperand(0),
Ops[1],
Ops[0],
Ops[6],
Ops[7]);
1268 SelectVOP3BMods(
N->getOperand(1),
Ops[3],
Ops[2]);
1269 SelectVOP3BMods(
N->getOperand(2),
Ops[5],
Ops[4]);
1275void AMDGPUDAGToDAGISel::SelectMAD_64_32(
SDNode *
N) {
1279 bool UseNoCarry = Subtarget->hasMadNC64_32Insts() && !
N->hasAnyUseOfValue(1);
1280 if (Subtarget->hasMADIntraFwdBug())
1281 Opc =
Signed ? AMDGPU::V_MAD_I64_I32_gfx11_e64
1282 : AMDGPU::V_MAD_U64_U32_gfx11_e64;
1283 else if (UseNoCarry)
1284 Opc =
Signed ? AMDGPU::V_MAD_NC_I64_I32_e64 : AMDGPU::V_MAD_NC_U64_U32_e64;
1286 Opc =
Signed ? AMDGPU::V_MAD_I64_I32_e64 : AMDGPU::V_MAD_U64_U32_e64;
1288 SDValue Clamp =
CurDAG->getTargetConstant(0, SL, MVT::i1);
1289 SDValue
Ops[] = {
N->getOperand(0),
N->getOperand(1),
N->getOperand(2),
1293 MachineSDNode *Mad =
CurDAG->getMachineNode(
Opc, SL, MVT::i64,
Ops);
1304void AMDGPUDAGToDAGISel::SelectMUL_LOHI(
SDNode *
N) {
1309 if (Subtarget->hasMadNC64_32Insts()) {
1310 VTList =
CurDAG->getVTList(MVT::i64);
1311 Opc =
Signed ? AMDGPU::V_MAD_NC_I64_I32_e64 : AMDGPU::V_MAD_NC_U64_U32_e64;
1313 VTList =
CurDAG->getVTList(MVT::i64, MVT::i1);
1314 if (Subtarget->hasMADIntraFwdBug()) {
1315 Opc =
Signed ? AMDGPU::V_MAD_I64_I32_gfx11_e64
1316 : AMDGPU::V_MAD_U64_U32_gfx11_e64;
1318 Opc =
Signed ? AMDGPU::V_MAD_I64_I32_e64 : AMDGPU::V_MAD_U64_U32_e64;
1322 SDValue
Zero =
CurDAG->getTargetConstant(0, SL, MVT::i64);
1323 SDValue Clamp =
CurDAG->getTargetConstant(0, SL, MVT::i1);
1324 SDValue
Ops[] = {
N->getOperand(0),
N->getOperand(1),
Zero, Clamp};
1325 SDNode *Mad =
CurDAG->getMachineNode(
Opc, SL, VTList,
Ops);
1326 if (!SDValue(
N, 0).use_empty()) {
1327 SDValue Sub0 =
CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32);
1328 SDNode *
Lo =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, SL,
1329 MVT::i32, SDValue(Mad, 0), Sub0);
1332 if (!SDValue(
N, 1).use_empty()) {
1333 SDValue Sub1 =
CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32);
1334 SDNode *
Hi =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG, SL,
1335 MVT::i32, SDValue(Mad, 0), Sub1);
1345 if (!
Base || Subtarget->hasUsableDSOffset() ||
1346 Subtarget->unsafeDSOffsetFoldingEnabled())
1357 if (
CurDAG->isBaseWithConstantOffset(Addr)) {
1370 int64_t ByteOffset =
C->getSExtValue();
1371 if (isDSOffsetLegal(SDValue(), ByteOffset)) {
1372 SDValue
Zero =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1380 if (isDSOffsetLegal(
Sub, ByteOffset)) {
1386 unsigned SubOp = AMDGPU::V_SUB_CO_U32_e32;
1387 if (Subtarget->hasAddNoCarryInsts()) {
1388 SubOp = AMDGPU::V_SUB_U32_e64;
1390 CurDAG->getTargetConstant(0, {}, MVT::i1));
1393 MachineSDNode *MachineSub =
1394 CurDAG->getMachineNode(SubOp,
DL, MVT::i32, Opnds);
1396 Base = SDValue(MachineSub, 0);
1410 if (isDSOffsetLegal(SDValue(), CAddr->getZExtValue())) {
1411 SDValue
Zero =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1412 MachineSDNode *MovZero =
CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32,
1413 DL, MVT::i32, Zero);
1414 Base = SDValue(MovZero, 0);
1415 Offset =
CurDAG->getTargetConstant(CAddr->getZExtValue(),
DL, MVT::i16);
1422 Offset =
CurDAG->getTargetConstant(0, SDLoc(Addr), MVT::i16);
1426bool AMDGPUDAGToDAGISel::isDSOffset2Legal(
SDValue Base,
unsigned Offset0,
1428 unsigned Size)
const {
1429 if (Offset0 %
Size != 0 || Offset1 %
Size != 0)
1434 if (!
Base || Subtarget->hasUsableDSOffset() ||
1435 Subtarget->unsafeDSOffsetFoldingEnabled())
1453bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegal(
SDValue Addr)
const {
1459 if (Subtarget->hasSignedScratchOffsets())
1469 ConstantSDNode *ImmOp =
nullptr;
1480bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegalSV(
SDValue Addr)
const {
1486 if (Subtarget->hasSignedScratchOffsets())
1496bool AMDGPUDAGToDAGISel::isFlatScratchBaseLegalSVImm(
SDValue Addr)
const {
1510 (RHSImm->getSExtValue() < 0 && RHSImm->getSExtValue() > -0x40000000)))
1513 auto LHS =
Base.getOperand(0);
1514 auto RHS =
Base.getOperand(1);
1522 return SelectDSReadWrite2(Addr,
Base, Offset0, Offset1, 4);
1528 return SelectDSReadWrite2(Addr,
Base, Offset0, Offset1, 8);
1533 unsigned Size)
const {
1536 if (
CurDAG->isBaseWithConstantOffset(Addr)) {
1541 unsigned OffsetValue1 = OffsetValue0 +
Size;
1544 if (isDSOffset2Legal(N0, OffsetValue0, OffsetValue1,
Size)) {
1546 Offset0 =
CurDAG->getTargetConstant(OffsetValue0 /
Size,
DL, MVT::i32);
1547 Offset1 =
CurDAG->getTargetConstant(OffsetValue1 /
Size,
DL, MVT::i32);
1552 if (
const ConstantSDNode *
C =
1554 unsigned OffsetValue0 =
C->getZExtValue();
1555 unsigned OffsetValue1 = OffsetValue0 +
Size;
1557 if (isDSOffset2Legal(SDValue(), OffsetValue0, OffsetValue1,
Size)) {
1559 SDValue
Zero =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1567 if (isDSOffset2Legal(
Sub, OffsetValue0, OffsetValue1,
Size)) {
1571 unsigned SubOp = AMDGPU::V_SUB_CO_U32_e32;
1572 if (Subtarget->hasAddNoCarryInsts()) {
1573 SubOp = AMDGPU::V_SUB_U32_e64;
1575 CurDAG->getTargetConstant(0, {}, MVT::i1));
1578 MachineSDNode *MachineSub =
CurDAG->getMachineNode(
1581 Base = SDValue(MachineSub, 0);
1583 CurDAG->getTargetConstant(OffsetValue0 /
Size,
DL, MVT::i32);
1585 CurDAG->getTargetConstant(OffsetValue1 /
Size,
DL, MVT::i32);
1591 unsigned OffsetValue0 = CAddr->getZExtValue();
1592 unsigned OffsetValue1 = OffsetValue0 +
Size;
1594 if (isDSOffset2Legal(SDValue(), OffsetValue0, OffsetValue1,
Size)) {
1595 SDValue
Zero =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1596 MachineSDNode *MovZero =
1597 CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32,
DL, MVT::i32, Zero);
1598 Base = SDValue(MovZero, 0);
1599 Offset0 =
CurDAG->getTargetConstant(OffsetValue0 /
Size,
DL, MVT::i32);
1600 Offset1 =
CurDAG->getTargetConstant(OffsetValue1 /
Size,
DL, MVT::i32);
1608 Offset0 =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1609 Offset1 =
CurDAG->getTargetConstant(1,
DL, MVT::i32);
1619 if (Subtarget->useFlatForGlobal())
1624 Idxen =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
1625 Offen =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
1626 Addr64 =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
1627 SOffset = Subtarget->hasRestrictedSOffset()
1628 ?
CurDAG->getRegister(AMDGPU::SGPR_NULL, MVT::i32)
1629 :
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1631 ConstantSDNode *C1 =
nullptr;
1633 if (
CurDAG->isBaseWithConstantOffset(Addr)) {
1646 Addr64 =
CurDAG->getTargetConstant(1,
DL, MVT::i1);
1652 Ptr = SDValue(buildSMovImm64(
DL, 0, MVT::v2i32), 0);
1668 Ptr = SDValue(buildSMovImm64(
DL, 0, MVT::v2i32), 0);
1670 Addr64 =
CurDAG->getTargetConstant(1,
DL, MVT::i1);
1674 VAddr =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1684 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
1694 SDValue(
CurDAG->getMachineNode(
1695 AMDGPU::S_MOV_B32,
DL, MVT::i32,
1701bool AMDGPUDAGToDAGISel::SelectMUBUFAddr64(
SDValue Addr,
SDValue &SRsrc,
1704 SDValue Ptr, Offen, Idxen, Addr64;
1708 if (!Subtarget->hasAddr64())
1711 if (!SelectMUBUF(Addr, Ptr, VAddr, SOffset,
Offset, Offen, Idxen, Addr64))
1715 if (
C->getSExtValue()) {
1728std::pair<SDValue, SDValue> AMDGPUDAGToDAGISel::foldFrameIndex(
SDValue N)
const {
1733 FI ?
CurDAG->getTargetFrameIndex(FI->getIndex(), FI->getValueType(0)) :
N;
1739 return std::pair(TFI,
CurDAG->getTargetConstant(0,
DL, MVT::i32));
1742bool AMDGPUDAGToDAGISel::SelectMUBUFScratchOffen(
SDNode *Parent,
1749 const SIMachineFunctionInfo *
Info =
MF.getInfo<SIMachineFunctionInfo>();
1751 Rsrc =
CurDAG->getRegister(
Info->getScratchRSrcReg(), MVT::v4i32);
1754 int64_t
Imm = CAddr->getSExtValue();
1755 const int64_t NullPtr =
1758 if (
Imm != NullPtr) {
1761 CurDAG->getTargetConstant(
Imm & ~MaxOffset,
DL, MVT::i32);
1762 MachineSDNode *MovHighBits =
CurDAG->getMachineNode(
1763 AMDGPU::V_MOV_B32_e32,
DL, MVT::i32, HighBits);
1764 VAddr = SDValue(MovHighBits, 0);
1766 SOffset =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1767 ImmOffset =
CurDAG->getTargetConstant(
Imm & MaxOffset,
DL, MVT::i32);
1772 if (
CurDAG->isBaseWithConstantOffset(Addr)) {
1793 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
1794 if (
TII->isLegalMUBUFImmOffset(C1) &&
1795 (!Subtarget->privateMemoryResourceIsRangeChecked() ||
1796 CurDAG->SignBitIsZero(N0))) {
1797 std::tie(VAddr, SOffset) = foldFrameIndex(N0);
1798 ImmOffset =
CurDAG->getTargetConstant(C1,
DL, MVT::i32);
1804 std::tie(VAddr, SOffset) = foldFrameIndex(Addr);
1805 ImmOffset =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1813 if (!
Reg.isPhysical())
1815 const auto *RC =
TRI.getPhysRegBaseClass(
Reg);
1816 return RC &&
TRI.isSGPRClass(RC);
1819bool AMDGPUDAGToDAGISel::SelectMUBUFScratchOffset(
SDNode *Parent,
1824 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
1825 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
1827 const SIMachineFunctionInfo *
Info =
MF.getInfo<SIMachineFunctionInfo>();
1832 SRsrc =
CurDAG->getRegister(
Info->getScratchRSrcReg(), MVT::v4i32);
1838 ConstantSDNode *CAddr;
1851 SOffset =
CurDAG->getTargetConstant(0,
DL, MVT::i32);
1856 SRsrc =
CurDAG->getRegister(
Info->getScratchRSrcReg(), MVT::v4i32);
1862bool AMDGPUDAGToDAGISel::SelectMUBUFOffset(
SDValue Addr,
SDValue &SRsrc,
1865 SDValue Ptr, VAddr, Offen, Idxen, Addr64;
1866 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
1868 if (!SelectMUBUF(Addr, Ptr, VAddr, SOffset,
Offset, Offen, Idxen, Addr64))
1887bool AMDGPUDAGToDAGISel::SelectBUFSOffset(
SDValue ByteOffsetNode,
1889 if (Subtarget->hasRestrictedSOffset() &&
isNullConstant(ByteOffsetNode)) {
1890 SOffset =
CurDAG->getRegister(AMDGPU::SGPR_NULL, MVT::i32);
1894 SOffset = ByteOffsetNode;
1912bool AMDGPUDAGToDAGISel::SelectFlatOffsetImpl(
1916 int64_t OffsetVal = 0;
1920 bool CanHaveFlatSegmentOffsetBug =
1921 Subtarget->hasFlatSegmentOffsetBug() &&
1922 FlatVariant == FlatAddrSpace::FLAT &&
1925 if (Subtarget->hasFlatInstOffsets() && !CanHaveFlatSegmentOffsetBug) {
1927 if (isBaseWithConstantOffset64(Addr, N0, N1) &&
1928 (FlatVariant != FlatAddrSpace::FlatScratch ||
1929 isFlatScratchBaseLegal(Addr))) {
1937 if (COffsetVal == 0 || FlatVariant != FlatAddrSpace::FLAT || IsInBounds) {
1938 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
1939 if (
TII->isLegalFLATOffset(COffsetVal, AS, FlatVariant)) {
1941 OffsetVal = COffsetVal;
1956 std::tie(OffsetVal, RemainderOffset) =
1957 TII->splitFlatOffset(COffsetVal, AS, FlatVariant);
1959 SDValue AddOffsetLo =
1960 getMaterializedScalarImm32(
Lo_32(RemainderOffset),
DL);
1961 SDValue Clamp =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
1967 unsigned AddOp = AMDGPU::V_ADD_CO_U32_e32;
1968 if (Subtarget->hasAddNoCarryInsts()) {
1969 AddOp = AMDGPU::V_ADD_U32_e64;
1973 SDValue(
CurDAG->getMachineNode(AddOp,
DL, MVT::i32, Opnds), 0);
1978 CurDAG->getTargetConstant(AMDGPU::sub0,
DL, MVT::i32);
1980 CurDAG->getTargetConstant(AMDGPU::sub1,
DL, MVT::i32);
1982 SDNode *N0Lo =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
1983 DL, MVT::i32, N0, Sub0);
1984 SDNode *N0Hi =
CurDAG->getMachineNode(TargetOpcode::EXTRACT_SUBREG,
1985 DL, MVT::i32, N0, Sub1);
1987 SDValue AddOffsetHi =
1988 getMaterializedScalarImm32(
Hi_32(RemainderOffset),
DL);
1990 SDVTList VTs =
CurDAG->getVTList(MVT::i32, MVT::i1);
1993 CurDAG->getMachineNode(AMDGPU::V_ADD_CO_U32_e64,
DL, VTs,
1994 {AddOffsetLo, SDValue(N0Lo, 0), Clamp});
1996 SDNode *Addc =
CurDAG->getMachineNode(
1997 AMDGPU::V_ADDC_U32_e64,
DL, VTs,
1998 {AddOffsetHi, SDValue(N0Hi, 0), SDValue(
Add, 1), Clamp});
2000 SDValue RegSequenceArgs[] = {
2001 CurDAG->getTargetConstant(AMDGPU::VReg_64RegClassID,
DL,
2003 SDValue(
Add, 0), Sub0, SDValue(Addc, 0), Sub1};
2005 Addr = SDValue(
CurDAG->getMachineNode(AMDGPU::REG_SEQUENCE,
DL,
2006 MVT::i64, RegSequenceArgs),
2015 Offset =
CurDAG->getSignedTargetConstant(OffsetVal, SDLoc(), MVT::i32);
2019bool AMDGPUDAGToDAGISel::SelectFlatOffset(
SDNode *
N,
SDValue Addr,
2022 return SelectFlatOffsetImpl(
N, Addr, VAddr,
Offset,
2026bool AMDGPUDAGToDAGISel::SelectGlobalOffset(
SDNode *
N,
SDValue Addr,
2029 return SelectFlatOffsetImpl(
N, Addr, VAddr,
Offset,
2033bool AMDGPUDAGToDAGISel::SelectScratchOffset(
SDNode *
N,
SDValue Addr,
2036 return SelectFlatOffsetImpl(
N, Addr, VAddr,
Offset,
2044 if (
Op.getValueType() == MVT::i32)
2059bool AMDGPUDAGToDAGISel::SelectGlobalSAddr(
SDNode *
N,
SDValue Addr,
2062 bool NeedIOffset)
const {
2064 int64_t ImmOffset = 0;
2065 ScaleOffset =
false;
2071 if (isBaseWithConstantOffset64(Addr,
LHS,
RHS)) {
2073 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
2077 FlatAddrSpace::FlatGlobal)) {
2079 ImmOffset = COffsetVal;
2080 }
else if (!
LHS->isDivergent()) {
2081 if (COffsetVal > 0) {
2086 int64_t SplitImmOffset = 0, RemainderOffset = COffsetVal;
2088 std::tie(SplitImmOffset, RemainderOffset) =
TII->splitFlatOffset(
2092 if (Subtarget->hasSignedGVSOffset() ?
isInt<32>(RemainderOffset)
2094 SDNode *VMov =
CurDAG->getMachineNode(
2095 AMDGPU::V_MOV_B32_e32, SL, MVT::i32,
2096 CurDAG->getTargetConstant(RemainderOffset, SDLoc(), MVT::i32));
2097 VOffset = SDValue(VMov, 0);
2099 Offset =
CurDAG->getTargetConstant(SplitImmOffset, SDLoc(), MVT::i32);
2109 unsigned NumLiterals =
2110 !
TII->isInlineConstant(APInt(32,
Lo_32(COffsetVal))) +
2111 !
TII->isInlineConstant(APInt(32,
Hi_32(COffsetVal)));
2112 if (Subtarget->getConstantBusLimit(AMDGPU::V_ADD_U32_e64) > NumLiterals)
2121 if (!
LHS->isDivergent()) {
2124 ScaleOffset = SelectScaleOffset(
N,
RHS, Subtarget->hasSignedGVSOffset());
2126 RHS, Subtarget->hasSignedGVSOffset(),
CurDAG)) {
2133 if (!SAddr && !
RHS->isDivergent()) {
2135 ScaleOffset = SelectScaleOffset(
N,
LHS, Subtarget->hasSignedGVSOffset());
2137 LHS, Subtarget->hasSignedGVSOffset(),
CurDAG)) {
2144 Offset =
CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2149 if (Subtarget->hasScaleOffset() &&
2150 (Addr.
getOpcode() == (Subtarget->hasSignedGVSOffset()
2165 Offset =
CurDAG->getTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2177 CurDAG->getMachineNode(AMDGPU::V_MOV_B32_e32, SDLoc(Addr), MVT::i32,
2178 CurDAG->getTargetConstant(0, SDLoc(), MVT::i32));
2179 VOffset = SDValue(VMov, 0);
2180 Offset =
CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2184bool AMDGPUDAGToDAGISel::SelectGlobalSAddr(
SDNode *
N,
SDValue Addr,
2189 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset,
Offset, ScaleOffset))
2197bool AMDGPUDAGToDAGISel::SelectGlobalSAddrCPol(
SDNode *
N,
SDValue Addr,
2202 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset,
Offset, ScaleOffset))
2207 N->getConstantOperandVal(
N->getNumOperands() - 1) & ~AMDGPU::CPol::SCAL;
2213bool AMDGPUDAGToDAGISel::SelectGlobalSAddrCPolM0(
SDNode *
N,
SDValue Addr,
2219 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset,
Offset, ScaleOffset))
2224 N->getConstantOperandVal(
N->getNumOperands() - 2) & ~AMDGPU::CPol::SCAL;
2230bool AMDGPUDAGToDAGISel::SelectGlobalSAddrGLC(
SDNode *
N,
SDValue Addr,
2235 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset,
Offset, ScaleOffset))
2239 CPol =
CurDAG->getTargetConstant(CPolVal, SDLoc(), MVT::i32);
2243bool AMDGPUDAGToDAGISel::SelectGlobalSAddrNoIOffset(
SDNode *
N,
SDValue Addr,
2248 SDValue DummyOffset;
2249 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset, DummyOffset, ScaleOffset,
2255 N->getConstantOperandVal(
N->getNumOperands() - 1) & ~AMDGPU::CPol::SCAL;
2261bool AMDGPUDAGToDAGISel::SelectGlobalSAddrNoIOffsetM0(
SDNode *
N,
SDValue Addr,
2266 SDValue DummyOffset;
2267 if (!SelectGlobalSAddr(
N, Addr, SAddr, VOffset, DummyOffset, ScaleOffset,
2288 FI->getValueType(0));
2298bool AMDGPUDAGToDAGISel::SelectScratchSAddr(
SDNode *Parent,
SDValue Addr,
2307 int64_t COffsetVal = 0;
2309 if (
CurDAG->isBaseWithConstantOffset(Addr) && isFlatScratchBaseLegal(Addr)) {
2318 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
2321 FlatAddrSpace::FlatScratch)) {
2322 int64_t SplitImmOffset, RemainderOffset;
2323 std::tie(SplitImmOffset, RemainderOffset) =
TII->splitFlatOffset(
2326 COffsetVal = SplitImmOffset;
2330 ? getMaterializedScalarImm32(
Lo_32(RemainderOffset),
DL)
2331 :
CurDAG->getSignedTargetConstant(RemainderOffset,
DL, MVT::i32);
2332 SAddr = SDValue(
CurDAG->getMachineNode(AMDGPU::S_ADD_I32,
DL, MVT::i32,
2337 Offset =
CurDAG->getSignedTargetConstant(COffsetVal,
DL, MVT::i32);
2343bool AMDGPUDAGToDAGISel::checkFlatScratchSVSSwizzleBug(
2345 if (!Subtarget->hasFlatScratchSVSSwizzleBug())
2351 KnownBits VKnown =
CurDAG->computeKnownBits(VAddr);
2358 return (VMax & 3) + (
SMax & 3) >= 4;
2361bool AMDGPUDAGToDAGISel::SelectScratchSVAddr(
SDNode *
N,
SDValue Addr,
2365 int64_t ImmOffset = 0;
2368 SDValue OrigAddr = Addr;
2369 if (isBaseWithConstantOffset64(Addr,
LHS,
RHS)) {
2371 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
2376 ImmOffset = COffsetVal;
2377 }
else if (!
LHS->isDivergent() && COffsetVal > 0) {
2381 int64_t SplitImmOffset, RemainderOffset;
2382 std::tie(SplitImmOffset, RemainderOffset) =
2387 SDNode *VMov =
CurDAG->getMachineNode(
2388 AMDGPU::V_MOV_B32_e32, SL, MVT::i32,
2389 CurDAG->getTargetConstant(RemainderOffset, SDLoc(), MVT::i32));
2390 VAddr = SDValue(VMov, 0);
2392 if (!isFlatScratchBaseLegal(Addr))
2394 if (checkFlatScratchSVSSwizzleBug(VAddr, SAddr, SplitImmOffset))
2396 Offset =
CurDAG->getTargetConstant(SplitImmOffset, SDLoc(), MVT::i32);
2397 CPol =
CurDAG->getTargetConstant(0, SDLoc(), MVT::i32);
2409 if (!
LHS->isDivergent() &&
RHS->isDivergent()) {
2412 }
else if (!
RHS->isDivergent() &&
LHS->isDivergent()) {
2419 if (OrigAddr != Addr) {
2420 if (!isFlatScratchBaseLegalSVImm(OrigAddr))
2423 if (!isFlatScratchBaseLegalSV(OrigAddr))
2427 if (checkFlatScratchSVSSwizzleBug(VAddr, SAddr, ImmOffset))
2430 Offset =
CurDAG->getSignedTargetConstant(ImmOffset, SDLoc(), MVT::i32);
2432 bool ScaleOffset = SelectScaleOffset(
N, VAddr,
true );
2441bool AMDGPUDAGToDAGISel::isSOffsetLegalWithImmOffset(
SDValue *SOffset,
2444 int64_t ImmOffset)
const {
2445 if (!IsBuffer && !Imm32Only && ImmOffset < 0 &&
2447 KnownBits SKnown =
CurDAG->computeKnownBits(*SOffset);
2459 bool IsSigned)
const {
2460 bool ScaleOffset =
false;
2461 if (!Subtarget->hasScaleOffset() || !
Offset)
2475 (IsSigned &&
Offset.getOpcode() == AMDGPUISD::MUL_I24) ||
2476 Offset.getOpcode() == AMDGPUISD::MUL_U24 ||
2477 (
Offset.isMachineOpcode() &&
2478 Offset.getMachineOpcode() ==
2479 (IsSigned ? AMDGPU::S_MUL_I64_I32_PSEUDO
2480 : AMDGPU::S_MUL_U64_U32_PSEUDO))) {
2482 ScaleOffset =
C->getZExtValue() ==
Size;
2494bool AMDGPUDAGToDAGISel::SelectSMRDOffset(
SDNode *
N,
SDValue ByteOffsetNode,
2496 bool Imm32Only,
bool IsBuffer,
2497 bool HasSOffset, int64_t ImmOffset,
2498 bool *ScaleOffset)
const {
2500 "Cannot match both soffset and offset at the same time!");
2505 *ScaleOffset = SelectScaleOffset(
N, ByteOffsetNode,
false );
2515 *SOffset = ByteOffsetNode;
2516 return isSOffsetLegalWithImmOffset(SOffset, Imm32Only, IsBuffer,
2522 return isSOffsetLegalWithImmOffset(SOffset, Imm32Only, IsBuffer,
2529 SDLoc SL(ByteOffsetNode);
2533 int64_t ByteOffset = IsBuffer ?
C->getZExtValue() :
C->getSExtValue();
2535 *Subtarget, ByteOffset, IsBuffer, HasSOffset);
2536 if (EncodedOffset &&
Offset && !Imm32Only) {
2537 *
Offset =
CurDAG->getSignedTargetConstant(*EncodedOffset, SL, MVT::i32);
2546 if (EncodedOffset &&
Offset && Imm32Only) {
2547 *
Offset =
CurDAG->getTargetConstant(*EncodedOffset, SL, MVT::i32);
2555 SDValue C32Bit =
CurDAG->getTargetConstant(ByteOffset, SL, MVT::i32);
2557 CurDAG->getMachineNode(AMDGPU::S_MOV_B32, SL, MVT::i32, C32Bit), 0);
2564SDValue AMDGPUDAGToDAGISel::Expand32BitAddress(
SDValue Addr)
const {
2572 const SIMachineFunctionInfo *
Info =
MF.getInfo<SIMachineFunctionInfo>();
2573 unsigned AddrHiVal =
Info->get32BitAddressHighBits();
2574 SDValue AddrHi =
CurDAG->getTargetConstant(AddrHiVal, SL, MVT::i32);
2576 const SDValue
Ops[] = {
2577 CurDAG->getTargetConstant(AMDGPU::SReg_64_XEXECRegClassID, SL, MVT::i32),
2579 CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32),
2580 SDValue(
CurDAG->getMachineNode(AMDGPU::S_MOV_B32, SL, MVT::i32, AddrHi),
2582 CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32),
2585 return SDValue(
CurDAG->getMachineNode(AMDGPU::REG_SEQUENCE, SL, MVT::i64,
2592bool AMDGPUDAGToDAGISel::SelectSMRDBaseOffset(
SDNode *
N,
SDValue Addr,
2595 bool IsBuffer,
bool HasSOffset,
2597 bool *ScaleOffset)
const {
2599 assert(!Imm32Only && !IsBuffer);
2602 if (!SelectSMRDBaseOffset(
N, Addr,
B,
nullptr,
Offset,
false,
false,
true))
2607 ImmOff =
C->getSExtValue();
2609 return SelectSMRDBaseOffset(
N,
B, SBase, SOffset,
nullptr,
false,
false,
2610 true, ImmOff, ScaleOffset);
2630 if (SelectSMRDOffset(
N, N1, SOffset,
Offset, Imm32Only, IsBuffer, HasSOffset,
2631 ImmOffset, ScaleOffset)) {
2635 if (SelectSMRDOffset(
N, N0, SOffset,
Offset, Imm32Only, IsBuffer, HasSOffset,
2636 ImmOffset, ScaleOffset)) {
2645 bool Imm32Only,
bool *ScaleOffset)
const {
2646 if (SelectSMRDBaseOffset(
N, Addr, SBase, SOffset,
Offset, Imm32Only,
2649 SBase = Expand32BitAddress(SBase);
2654 SBase = Expand32BitAddress(Addr);
2655 *
Offset =
CurDAG->getTargetConstant(0, SDLoc(Addr), MVT::i32);
2662bool AMDGPUDAGToDAGISel::SelectSMRDImm(
SDValue Addr,
SDValue &SBase,
2664 return SelectSMRD(
nullptr, Addr, SBase,
nullptr,
2668bool AMDGPUDAGToDAGISel::SelectSMRDImm32(
SDValue Addr,
SDValue &SBase,
2671 return SelectSMRD(
nullptr, Addr, SBase,
nullptr,
2678 if (!SelectSMRD(
N, Addr, SBase, &SOffset,
nullptr,
2679 false, &ScaleOffset))
2683 SDLoc(
N), MVT::i32);
2687bool AMDGPUDAGToDAGISel::SelectSMRDSgprImm(
SDNode *
N,
SDValue Addr,
2692 if (!SelectSMRD(
N, Addr, SBase, &SOffset, &
Offset,
false, &ScaleOffset))
2696 SDLoc(
N), MVT::i32);
2701 return SelectSMRDOffset(
nullptr,
N,
nullptr, &
Offset,
2705bool AMDGPUDAGToDAGISel::SelectSMRDBufferImm32(
SDValue N,
2708 return SelectSMRDOffset(
nullptr,
N,
nullptr, &
Offset,
2712bool AMDGPUDAGToDAGISel::SelectSMRDBufferSgprImm(
SDValue N,
SDValue &SOffset,
2716 return N.getValueType() == MVT::i32 &&
2717 SelectSMRDBaseOffset(
nullptr,
N, SOffset,
2722bool AMDGPUDAGToDAGISel::SelectMOVRELOffset(
SDValue Index,
2727 if (
CurDAG->isBaseWithConstantOffset(Index)) {
2728 SDValue N0 =
Index.getOperand(0);
2729 SDValue N1 =
Index.getOperand(1);
2752SDNode *AMDGPUDAGToDAGISel::getBFE32(
bool IsSigned,
const SDLoc &
DL,
2756 unsigned Opcode = IsSigned ? AMDGPU::V_BFE_I32_e64 : AMDGPU::V_BFE_U32_e64;
2758 SDValue
W =
CurDAG->getTargetConstant(Width,
DL, MVT::i32);
2760 return CurDAG->getMachineNode(Opcode,
DL, MVT::i32, Val,
Off, W);
2762 unsigned Opcode = IsSigned ? AMDGPU::S_BFE_I32 : AMDGPU::S_BFE_U32;
2766 uint32_t PackedVal =
Offset | (Width << 16);
2767 SDValue PackedConst =
CurDAG->getTargetConstant(PackedVal,
DL, MVT::i32);
2769 return CurDAG->getMachineNode(Opcode,
DL, MVT::i32, Val, PackedConst);
2772void AMDGPUDAGToDAGISel::SelectS_BFEFromShifts(
SDNode *
N) {
2777 const SDValue &Shl =
N->getOperand(0);
2782 uint32_t BVal =
B->getZExtValue();
2783 uint32_t CVal =
C->getZExtValue();
2785 if (0 < BVal && BVal <= CVal && CVal < 32) {
2795void AMDGPUDAGToDAGISel::SelectS_BFE(
SDNode *
N) {
2796 switch (
N->getOpcode()) {
2798 if (
N->getOperand(0).getOpcode() ==
ISD::SRL) {
2801 const SDValue &Srl =
N->getOperand(0);
2805 if (Shift && Mask) {
2807 uint32_t MaskVal =
Mask->getZExtValue();
2819 if (
N->getOperand(0).getOpcode() ==
ISD::AND) {
2822 const SDValue &
And =
N->getOperand(0);
2826 if (Shift && Mask) {
2828 uint32_t MaskVal =
Mask->getZExtValue() >> ShiftVal;
2837 }
else if (
N->getOperand(0).getOpcode() ==
ISD::SHL) {
2838 SelectS_BFEFromShifts(
N);
2843 if (
N->getOperand(0).getOpcode() ==
ISD::SHL) {
2844 SelectS_BFEFromShifts(
N);
2851 SDValue Src =
N->getOperand(0);
2859 unsigned Width =
cast<VTSDNode>(
N->getOperand(1))->getVT().getSizeInBits();
2869bool AMDGPUDAGToDAGISel::isCBranchSCC(
const SDNode *
N)
const {
2871 if (!
N->hasOneUse())
2874 SDValue
Cond =
N->getOperand(1);
2881 MVT VT =
Cond.getOperand(0).getSimpleValueType();
2885 if (VT == MVT::i64) {
2888 Subtarget->hasScalarCompareEq64();
2891 if ((VT == MVT::f16 || VT == MVT::f32) && Subtarget->hasSALUFloatInsts())
2924void AMDGPUDAGToDAGISel::SelectBRCOND(
SDNode *
N) {
2925 SDValue
Cond =
N->getOperand(1);
2927 if (
Cond.isUndef()) {
2928 CurDAG->SelectNodeTo(
N, AMDGPU::SI_BR_UNDEF, MVT::Other,
2929 N->getOperand(2),
N->getOperand(0));
2933 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
2935 bool UseSCCBr = isCBranchSCC(
N) && isUniformBr(
N);
2936 bool AndExec = !UseSCCBr;
2937 bool Negate =
false;
2940 Cond->getOperand(0)->getOpcode() == AMDGPUISD::SETCC) {
2941 SDValue VCMP =
Cond->getOperand(0);
2955 bool NegatedBallot =
false;
2958 UseSCCBr = !BallotCond->isDivergent();
2959 Negate = Negate ^ NegatedBallot;
2974 UseSCCBr ? (Negate ? AMDGPU::S_CBRANCH_SCC0 : AMDGPU::S_CBRANCH_SCC1)
2975 : (Negate ? AMDGPU::S_CBRANCH_VCCZ : AMDGPU::S_CBRANCH_VCCNZ);
2976 Register CondReg = UseSCCBr ? AMDGPU::SCC :
TRI->getVCC();
2995 Subtarget->isWave32() ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64, SL,
2997 CurDAG->getRegister(Subtarget->isWave32() ? AMDGPU::EXEC_LO
3004 SDValue VCC =
CurDAG->getCopyToReg(
N->getOperand(0), SL, CondReg,
Cond);
3005 CurDAG->SelectNodeTo(
N, BrOp, MVT::Other,
3010void AMDGPUDAGToDAGISel::SelectFP_EXTEND(
SDNode *
N) {
3011 if (Subtarget->hasSALUFloatInsts() &&
N->getValueType(0) == MVT::f32 &&
3012 !
N->isDivergent()) {
3013 SDValue Src =
N->getOperand(0);
3014 if (Src.getValueType() == MVT::f16) {
3016 CurDAG->SelectNodeTo(
N, AMDGPU::S_CVT_HI_F32_F16,
N->getVTList(),
3026void AMDGPUDAGToDAGISel::SelectDSAppendConsume(
SDNode *
N,
unsigned IntrID) {
3029 unsigned Opc = IntrID == Intrinsic::amdgcn_ds_append ?
3030 AMDGPU::DS_APPEND : AMDGPU::DS_CONSUME;
3032 SDValue Chain =
N->getOperand(0);
3035 MachineMemOperand *MMO =
M->getMemOperand();
3039 if (
CurDAG->isBaseWithConstantOffset(Ptr)) {
3044 if (isDSOffsetLegal(PtrBase, OffsetVal.
getZExtValue())) {
3045 N = glueCopyToM0(
N, PtrBase);
3046 Offset =
CurDAG->getTargetConstant(OffsetVal, SDLoc(), MVT::i32);
3051 N = glueCopyToM0(
N, Ptr);
3052 Offset =
CurDAG->getTargetConstant(0, SDLoc(), MVT::i32);
3057 CurDAG->getTargetConstant(IsGDS, SDLoc(), MVT::i32),
3062 SDNode *Selected =
CurDAG->SelectNodeTo(
N,
Opc,
N->getVTList(),
Ops);
3068void AMDGPUDAGToDAGISel::SelectDSBvhStackIntrinsic(
SDNode *
N,
unsigned IntrID) {
3071 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
3072 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
3073 Opc = AMDGPU::DS_BVH_STACK_RTN_B32;
3075 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
3076 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP1_RTN_B32;
3078 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
3079 Opc = AMDGPU::DS_BVH_STACK_PUSH8_POP2_RTN_B64;
3082 SDValue
Ops[] = {
N->getOperand(2),
N->getOperand(3),
N->getOperand(4),
3083 N->getOperand(5),
N->getOperand(0)};
3086 MachineMemOperand *MMO =
M->getMemOperand();
3087 SDNode *Selected =
CurDAG->SelectNodeTo(
N,
Opc,
N->getVTList(),
Ops);
3091void AMDGPUDAGToDAGISel::SelectTensorLoadStore(
SDNode *
N,
unsigned IntrID) {
3092 bool IsLoad = IntrID == Intrinsic::amdgcn_tensor_load_to_lds;
3094 IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d4 : AMDGPU::TENSOR_STORE_FROM_LDS_d4;
3102 SDValue Group2 =
N->getOperand(4);
3103 SDValue Group3 =
N->getOperand(5);
3106 Opc = IsLoad ? AMDGPU::TENSOR_LOAD_TO_LDS_d2
3107 : AMDGPU::TENSOR_STORE_FROM_LDS_d2;
3119 (void)
CurDAG->SelectNodeTo(
N,
Opc, MVT::Other, TensorOps);
3124 case Intrinsic::amdgcn_ds_gws_init:
3125 return AMDGPU::DS_GWS_INIT;
3126 case Intrinsic::amdgcn_ds_gws_barrier:
3127 return AMDGPU::DS_GWS_BARRIER;
3128 case Intrinsic::amdgcn_ds_gws_sema_v:
3129 return AMDGPU::DS_GWS_SEMA_V;
3130 case Intrinsic::amdgcn_ds_gws_sema_br:
3131 return AMDGPU::DS_GWS_SEMA_BR;
3132 case Intrinsic::amdgcn_ds_gws_sema_p:
3133 return AMDGPU::DS_GWS_SEMA_P;
3134 case Intrinsic::amdgcn_ds_gws_sema_release_all:
3135 return AMDGPU::DS_GWS_SEMA_RELEASE_ALL;
3141void AMDGPUDAGToDAGISel::SelectDS_GWS(
SDNode *
N,
unsigned IntrID) {
3142 if (!Subtarget->hasGWS() ||
3143 (IntrID == Intrinsic::amdgcn_ds_gws_sema_release_all &&
3144 !Subtarget->hasGWSSemaReleaseAll())) {
3151 const bool HasVSrc =
N->getNumOperands() == 4;
3152 assert(HasVSrc ||
N->getNumOperands() == 3);
3155 SDValue BaseOffset =
N->getOperand(HasVSrc ? 3 : 2);
3158 MachineMemOperand *MMO =
M->getMemOperand();
3171 glueCopyToM0(
N,
CurDAG->getTargetConstant(0, SL, MVT::i32));
3172 ImmOffset = ConstOffset->getZExtValue();
3174 if (
CurDAG->isBaseWithConstantOffset(BaseOffset)) {
3183 =
CurDAG->getMachineNode(AMDGPU::V_READFIRSTLANE_B32, SL, MVT::i32,
3187 =
CurDAG->getMachineNode(AMDGPU::S_LSHL_B32, SL, MVT::i32,
3188 SDValue(SGPROffset, 0),
3189 CurDAG->getTargetConstant(16, SL, MVT::i32));
3190 glueCopyToM0(
N, SDValue(M0Base, 0));
3194 SDValue OffsetField =
CurDAG->getTargetConstant(ImmOffset, SL, MVT::i32);
3198 const MCInstrDesc &InstrDesc =
TII->get(
Opc);
3199 int Data0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::data0);
3205 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
3207 SDValue
Data =
N->getOperand(2);
3208 MVT DataVT =
Data.getValueType().getSimpleVT();
3209 if (
TRI->isTypeLegalForClass(*DataRC, DataVT)) {
3211 Ops.push_back(
N->getOperand(2));
3215 const SDValue RegSeqOps[] = {
3217 CurDAG->getTargetConstant(AMDGPU::sub0, SL, MVT::i32),
3219 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, SL, MVT::i32),
3221 CurDAG->getTargetConstant(AMDGPU::sub1, SL, MVT::i32)};
3223 Ops.push_back(SDValue(
CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE,
3224 SL, MVT::v2i32, RegSeqOps),
3229 Ops.push_back(OffsetField);
3230 Ops.push_back(Chain);
3232 SDNode *Selected =
CurDAG->SelectNodeTo(
N,
Opc,
N->getVTList(),
Ops);
3236void AMDGPUDAGToDAGISel::SelectInterpP1F16(
SDNode *
N) {
3237 if (Subtarget->getLDSBankCount() != 16) {
3264 SDValue ToM0 =
CurDAG->getCopyToReg(
CurDAG->getEntryNode(),
DL, AMDGPU::M0,
3265 N->getOperand(5), SDValue());
3267 SDVTList VTs =
CurDAG->getVTList(MVT::f32, MVT::Other);
3270 CurDAG->getMachineNode(AMDGPU::V_INTERP_MOV_F32,
DL, VTs, {
3271 CurDAG->getTargetConstant(2,
DL, MVT::i32),
3277 SDNode *InterpP1LV =
3278 CurDAG->getMachineNode(AMDGPU::V_INTERP_P1LV_F16,
DL, MVT::f32, {
3279 CurDAG->getTargetConstant(0,
DL, MVT::i32),
3283 CurDAG->getTargetConstant(0,
DL, MVT::i32),
3284 SDValue(InterpMov, 0),
3286 CurDAG->getTargetConstant(0,
DL, MVT::i1),
3287 CurDAG->getTargetConstant(0,
DL, MVT::i32),
3288 SDValue(InterpMov, 1)
3291 CurDAG->ReplaceAllUsesOfValueWith(SDValue(
N, 0), SDValue(InterpP1LV, 0));
3294void AMDGPUDAGToDAGISel::SelectINTRINSIC_W_CHAIN(
SDNode *
N) {
3295 unsigned IntrID =
N->getConstantOperandVal(1);
3297 case Intrinsic::amdgcn_ds_append:
3298 case Intrinsic::amdgcn_ds_consume: {
3299 if (
N->getValueType(0) != MVT::i32)
3301 SelectDSAppendConsume(
N, IntrID);
3304 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
3305 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
3306 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
3307 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn:
3308 SelectDSBvhStackIntrinsic(
N, IntrID);
3310 case Intrinsic::amdgcn_init_whole_wave:
3311 CurDAG->getMachineFunction()
3312 .getInfo<SIMachineFunctionInfo>()
3313 ->setInitWholeWave();
3320void AMDGPUDAGToDAGISel::SelectINTRINSIC_WO_CHAIN(
SDNode *
N) {
3321 unsigned IntrID =
N->getConstantOperandVal(0);
3322 unsigned Opcode = AMDGPU::INSTRUCTION_LIST_END;
3323 SDNode *ConvGlueNode =
N->getGluedNode();
3329 CurDAG->getMachineNode(TargetOpcode::CONVERGENCECTRL_GLUE, {},
3330 MVT::Glue, SDValue(ConvGlueNode, 0));
3332 ConvGlueNode =
nullptr;
3335 case Intrinsic::amdgcn_wqm:
3336 Opcode = AMDGPU::WQM;
3338 case Intrinsic::amdgcn_softwqm:
3339 Opcode = AMDGPU::SOFT_WQM;
3341 case Intrinsic::amdgcn_wwm:
3342 case Intrinsic::amdgcn_strict_wwm:
3343 Opcode = AMDGPU::STRICT_WWM;
3345 case Intrinsic::amdgcn_strict_wqm:
3346 Opcode = AMDGPU::STRICT_WQM;
3348 case Intrinsic::amdgcn_interp_p1_f16:
3349 SelectInterpP1F16(
N);
3351 case Intrinsic::amdgcn_permlane16_swap:
3352 case Intrinsic::amdgcn_permlane32_swap: {
3353 if ((IntrID == Intrinsic::amdgcn_permlane16_swap &&
3354 !Subtarget->hasPermlane16Swap()) ||
3355 (IntrID == Intrinsic::amdgcn_permlane32_swap &&
3356 !Subtarget->hasPermlane32Swap())) {
3361 Opcode = IntrID == Intrinsic::amdgcn_permlane16_swap
3362 ? AMDGPU::V_PERMLANE16_SWAP_B32_e64
3363 : AMDGPU::V_PERMLANE32_SWAP_B32_e64;
3367 NewOps.push_back(SDValue(ConvGlueNode, 0));
3369 bool FI =
N->getConstantOperandVal(3);
3370 NewOps[2] =
CurDAG->getTargetConstant(
3373 CurDAG->SelectNodeTo(
N, Opcode,
N->getVTList(), NewOps);
3381 if (Opcode != AMDGPU::INSTRUCTION_LIST_END) {
3382 SDValue Src =
N->getOperand(1);
3383 CurDAG->SelectNodeTo(
N, Opcode,
N->getVTList(), {Src});
3388 NewOps.push_back(SDValue(ConvGlueNode, 0));
3389 CurDAG->MorphNodeTo(
N,
N->getOpcode(),
N->getVTList(), NewOps);
3393void AMDGPUDAGToDAGISel::SelectINTRINSIC_VOID(
SDNode *
N) {
3394 unsigned IntrID =
N->getConstantOperandVal(1);
3396 case Intrinsic::amdgcn_ds_gws_init:
3397 case Intrinsic::amdgcn_ds_gws_barrier:
3398 case Intrinsic::amdgcn_ds_gws_sema_v:
3399 case Intrinsic::amdgcn_ds_gws_sema_br:
3400 case Intrinsic::amdgcn_ds_gws_sema_p:
3401 case Intrinsic::amdgcn_ds_gws_sema_release_all:
3402 SelectDS_GWS(
N, IntrID);
3404 case Intrinsic::amdgcn_tensor_load_to_lds:
3405 case Intrinsic::amdgcn_tensor_store_from_lds:
3406 SelectTensorLoadStore(
N, IntrID);
3415void AMDGPUDAGToDAGISel::SelectWAVE_ADDRESS(
SDNode *
N) {
3416 SDValue Log2WaveSize =
3417 CurDAG->getTargetConstant(Subtarget->getWavefrontSizeLog2(), SDLoc(
N), MVT::i32);
3418 CurDAG->SelectNodeTo(
N, AMDGPU::S_LSHR_B32,
N->getVTList(),
3419 {N->getOperand(0), Log2WaveSize});
3422void AMDGPUDAGToDAGISel::SelectSTACKRESTORE(
SDNode *
N) {
3423 SDValue SrcVal =
N->getOperand(1);
3436 SDValue Log2WaveSize =
CurDAG->getTargetConstant(
3437 Subtarget->getWavefrontSizeLog2(), SL, MVT::i32);
3439 if (
N->isDivergent()) {
3440 SrcVal = SDValue(
CurDAG->getMachineNode(AMDGPU::V_READFIRSTLANE_B32, SL,
3445 CopyVal = SDValue(
CurDAG->getMachineNode(AMDGPU::S_LSHL_B32, SL, MVT::i32,
3446 {SrcVal, Log2WaveSize}),
3450 SDValue CopyToSP =
CurDAG->getCopyToReg(
N->getOperand(0), SL,
SP, CopyVal);
3451 CurDAG->ReplaceAllUsesOfValueWith(SDValue(
N, 0), CopyToSP);
3454bool AMDGPUDAGToDAGISel::SelectVOP3ModsImpl(
SDValue In,
SDValue &Src,
3456 bool IsCanonicalizing,
3457 bool AllowAbs)
const {
3463 Src = Src.getOperand(0);
3464 }
else if (Src.getOpcode() ==
ISD::FSUB && IsCanonicalizing) {
3468 if (
LHS &&
LHS->isZero()) {
3470 Src = Src.getOperand(1);
3474 if (AllowAbs && Src.getOpcode() ==
ISD::FABS) {
3476 Src = Src.getOperand(0);
3489 if (IsCanonicalizing)
3504 EVT VT = Src.getValueType();
3506 (VT != MVT::i32 && VT != MVT::v2i32 && VT != MVT::i64))
3513 auto ReplaceSrc = [&]() -> SDValue {
3515 return Src.getOperand(0);
3518 SDValue
Index = Src->getOperand(1);
3520 Src.getValueType(),
LHS, Index);
3546 if (SelectVOP3ModsImpl(In, Src, Mods,
true,
3548 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3555bool AMDGPUDAGToDAGISel::SelectVOP3ModsNonCanonicalizing(
3558 if (SelectVOP3ModsImpl(In, Src, Mods,
false,
3560 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3567bool AMDGPUDAGToDAGISel::SelectVOP3BMods(
SDValue In,
SDValue &Src,
3570 if (SelectVOP3ModsImpl(In, Src, Mods,
3573 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3580bool AMDGPUDAGToDAGISel::SelectVOP3NoMods(
SDValue In,
SDValue &Src)
const {
3588bool AMDGPUDAGToDAGISel::SelectVINTERPModsImpl(
SDValue In,
SDValue &Src,
3592 if (SelectVOP3ModsImpl(In, Src, Mods,
3597 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3604bool AMDGPUDAGToDAGISel::SelectVINTERPMods(
SDValue In,
SDValue &Src,
3606 return SelectVINTERPModsImpl(In, Src, SrcMods,
false);
3609bool AMDGPUDAGToDAGISel::SelectVINTERPModsHi(
SDValue In,
SDValue &Src,
3611 return SelectVINTERPModsImpl(In, Src, SrcMods,
true);
3614bool AMDGPUDAGToDAGISel::SelectVOP3Mods0(
SDValue In,
SDValue &Src,
3618 Clamp =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3619 Omod =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3621 return SelectVOP3Mods(In, Src, SrcMods);
3624bool AMDGPUDAGToDAGISel::SelectVOP3BMods0(
SDValue In,
SDValue &Src,
3628 Clamp =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3629 Omod =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3631 return SelectVOP3BMods(In, Src, SrcMods);
3634bool AMDGPUDAGToDAGISel::SelectVOP3OMods(
SDValue In,
SDValue &Src,
3639 Clamp =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3640 Omod =
CurDAG->getTargetConstant(0,
DL, MVT::i1);
3645bool AMDGPUDAGToDAGISel::SelectVOP3PMods(
SDValue In,
SDValue &Src,
3646 SDValue &SrcMods,
bool IsDOT)
const {
3653 Src = Src.getOperand(0);
3657 bool HasOpSel = Src.getValueSizeInBits() != 128;
3660 (!IsDOT || !Subtarget->hasDOTOpSelHazard())) {
3661 unsigned VecMods = Mods;
3663 SDValue
Lo = stripBitcast(Src.getOperand(0));
3664 SDValue
Hi = stripBitcast(Src.getOperand(1));
3667 Lo = stripBitcast(
Lo.getOperand(0));
3672 Hi = stripBitcast(
Hi.getOperand(0));
3684 unsigned VecSize = Src.getValueSizeInBits();
3685 Lo = stripExtractLoElt(
Lo);
3686 Hi = stripExtractLoElt(
Hi);
3688 if (
Lo.getValueSizeInBits() > VecSize) {
3689 Lo =
CurDAG->getTargetExtractSubreg(
3690 (VecSize > 32) ? AMDGPU::sub0_sub1 : AMDGPU::sub0, SDLoc(In),
3694 if (
Hi.getValueSizeInBits() > VecSize) {
3695 Hi =
CurDAG->getTargetExtractSubreg(
3696 (VecSize > 32) ? AMDGPU::sub0_sub1 : AMDGPU::sub0, SDLoc(In),
3700 assert(
Lo.getValueSizeInBits() <= VecSize &&
3701 Hi.getValueSizeInBits() <= VecSize);
3703 if (
Lo ==
Hi && !isInlineImmediate(
Lo.getNode())) {
3707 if (VecSize ==
Lo.getValueSizeInBits()) {
3709 }
else if (VecSize == 32) {
3710 Src = createVOP3PSrc32FromLo16(
Lo, Src,
CurDAG, Subtarget);
3712 assert((
Lo.getValueSizeInBits() == 32 && VecSize == 64) ||
3713 (
Lo.getValueSizeInBits() == 64 && VecSize == 128));
3716 SDValue
Undef = SDValue(
3717 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, SL,
3718 Lo.getValueType()), 0);
3719 const SIRegisterInfo *
TRI = Subtarget->getRegisterInfo();
3724 auto RC =
Lo->isDivergent() ?
TRI->getVGPRClassForBitWidth(VecSize)
3725 :
TRI->getSGPRClassForBitWidth(VecSize);
3726 unsigned NumRegs =
Lo.getValueSizeInBits() == 32 ? 1 : 2;
3727 const SDValue
Ops[] = {
3728 CurDAG->getTargetConstant(RC->getID(), SL, MVT::i32),
Lo,
3729 CurDAG->getTargetConstant(
TRI->getSubRegFromChannel(0, NumRegs), SL,
3734 CurDAG->getTargetConstant(
3735 TRI->getSubRegFromChannel(NumRegs, NumRegs), SL, MVT::i32)};
3737 Src = SDValue(
CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, SL,
3738 Src.getValueType(),
Ops), 0);
3742 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3748 .bitcastToAPInt().getZExtValue();
3750 Src =
CurDAG->getTargetConstant(
Lit, SDLoc(In), MVT::i64);
3751 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3758 Src.getNumOperands() == 2) {
3763 assert(Src.getValueSizeInBits() != 128 &&
3764 "<2 x 64> VECTOR_SHUFFLE should not be legal.");
3767 ArrayRef<int>
Mask = SVN->getMask();
3769 if (Mask[0] < 2 && Mask[1] < 2) {
3771 SDValue ShuffleSrc = SVN->getOperand(0);
3784 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3792 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3796bool AMDGPUDAGToDAGISel::SelectVOP3PModsDOT(
SDValue In,
SDValue &Src,
3798 return SelectVOP3PMods(In, Src, SrcMods,
true);
3801bool AMDGPUDAGToDAGISel::SelectVOP3PNoModsDOT(
SDValue In,
SDValue &Src)
const {
3802 SDValue SrcTmp, SrcModsTmp;
3803 SelectVOP3PMods(In, SrcTmp, SrcModsTmp,
true);
3812bool AMDGPUDAGToDAGISel::SelectVOP3PModsF32(
SDValue In,
SDValue &Src,
3814 SelectVOP3Mods(In, Src, SrcMods);
3817 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3821bool AMDGPUDAGToDAGISel::SelectVOP3PNoModsF32(
SDValue In,
SDValue &Src)
const {
3822 SDValue SrcTmp, SrcModsTmp;
3823 SelectVOP3PModsF32(In, SrcTmp, SrcModsTmp);
3832bool AMDGPUDAGToDAGISel::SelectWMMAOpSelVOP3PMods(
SDValue In,
3835 assert(
C->getAPIntValue().getBitWidth() == 1 &&
"expected i1 value");
3838 unsigned SrcVal =
C->getZExtValue();
3842 Src =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
3849 unsigned DstRegClass;
3851 switch (Elts.
size()) {
3853 DstRegClass = AMDGPU::VReg_256RegClassID;
3857 DstRegClass = AMDGPU::VReg_128RegClassID;
3861 DstRegClass = AMDGPU::VReg_64RegClassID;
3869 Ops.push_back(
CurDAG->getTargetConstant(DstRegClass,
DL, MVT::i32));
3870 for (
unsigned i = 0; i < Elts.
size(); ++i) {
3871 Ops.push_back(Elts[i]);
3872 Ops.push_back(
CurDAG->getTargetConstant(
3875 return CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE,
DL, DstTy,
Ops);
3882 assert(
"unhandled Reg sequence size" &&
3883 (Elts.
size() == 8 || Elts.
size() == 16));
3887 for (
unsigned i = 0; i < Elts.
size(); i += 2) {
3888 SDValue LoSrc = stripExtractLoElt(stripBitcast(Elts[i]));
3893 if (Subtarget->useRealTrue16Insts()) {
3897 SDValue
Undef = SDValue(
3898 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF,
DL, MVT::i16),
3901 emitRegSequence(*
CurDAG, AMDGPU::VGPR_32RegClassID, MVT::i32,
3902 {Elts[i],
Undef}, {AMDGPU::lo16, AMDGPU::hi16},
DL);
3903 Elts[i + 1] = emitRegSequence(*
CurDAG, AMDGPU::VGPR_32RegClassID,
3904 MVT::i32, {Elts[i + 1],
Undef},
3905 {AMDGPU::lo16, AMDGPU::hi16},
DL);
3907 SDValue PackLoLo =
CurDAG->getTargetConstant(0x05040100,
DL, MVT::i32);
3909 CurDAG->getMachineNode(AMDGPU::V_PERM_B32_e64,
DL, MVT::i32,
3910 {Elts[i + 1], Elts[i], PackLoLo});
3911 PackedElts.
push_back(SDValue(Packed, 0));
3914 return buildRegSequence32(PackedElts,
DL);
3920 unsigned ElementSize)
const {
3921 if (ElementSize == 16)
3922 return buildRegSequence16(Elts,
DL);
3923 if (ElementSize == 32)
3924 return buildRegSequence32(Elts,
DL);
3928void AMDGPUDAGToDAGISel::selectWMMAModsNegAbs(
unsigned ModOpcode,
3932 unsigned ElementSize)
const {
3937 for (
auto El : Elts) {
3940 NegAbsElts.
push_back(El->getOperand(0));
3942 if (Elts.size() != NegAbsElts.
size()) {
3944 Src = SDValue(buildRegSequence(Elts,
DL, ElementSize), 0);
3948 Src = SDValue(buildRegSequence(NegAbsElts,
DL, ElementSize), 0);
3954 Src = SDValue(buildRegSequence(Elts,
DL, ElementSize), 0);
3962 std::function<
bool(
SDValue)> ModifierCheck) {
3966 for (
unsigned i = 0; i < F16Pair->getNumOperands(); ++i) {
3967 SDValue ElF16 = stripBitcast(F16Pair->getOperand(i));
3968 if (!ModifierCheck(ElF16))
3975bool AMDGPUDAGToDAGISel::SelectWMMAModsF16Neg(
SDValue In,
SDValue &Src,
3993 Src = SDValue(buildRegSequence16(EltsF16, SDLoc(In)), 0);
4003 SDValue ElV2f16 = stripBitcast(BV->
getOperand(i));
4012 Src = SDValue(buildRegSequence32(EltsV2F16, SDLoc(In)), 0);
4018 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4022bool AMDGPUDAGToDAGISel::SelectWMMAModsF16NegAbs(
SDValue In,
SDValue &Src,
4033 if (EltsF16.
empty())
4043 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF16, Src, SDLoc(In), 16);
4051 SDValue ElV2f16 = stripBitcast(BV->
getOperand(i));
4053 if (EltsV2F16.
empty())
4062 selectWMMAModsNegAbs(ModOpcode, Mods, EltsV2F16, Src, SDLoc(In), 32);
4065 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4069bool AMDGPUDAGToDAGISel::SelectWMMAModsF32NegAbs(
SDValue In,
SDValue &Src,
4078 SDValue ElF32 = stripBitcast(BV->
getOperand(0));
4079 unsigned ModOpcode =
4082 SDValue ElF32 = stripBitcast(BV->
getOperand(i));
4090 selectWMMAModsNegAbs(ModOpcode, Mods, EltsF32, Src, SDLoc(In), 32);
4093 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4097bool AMDGPUDAGToDAGISel::SelectWMMAVISrc(
SDValue In,
SDValue &Src)
const {
4099 BitVector UndefElements;
4101 if (isInlineImmediate(
Splat.getNode())) {
4103 unsigned Imm =
C->getAPIntValue().getSExtValue();
4104 Src =
CurDAG->getTargetConstant(
Imm, SDLoc(In), MVT::i32);
4108 unsigned Imm =
C->getValueAPF().bitcastToAPInt().getSExtValue();
4109 Src =
CurDAG->getTargetConstant(
Imm, SDLoc(In), MVT::i32);
4117 SDValue SplatSrc32 = stripBitcast(In);
4119 if (SDValue Splat32 = SplatSrc32BV->getSplatValue()) {
4120 SDValue SplatSrc16 = stripBitcast(Splat32);
4122 if (SDValue
Splat = SplatSrc16BV->getSplatValue()) {
4123 const SIInstrInfo *
TII = Subtarget->getInstrInfo();
4124 std::optional<APInt> RawValue;
4126 RawValue =
C->getValueAPF().bitcastToAPInt();
4128 RawValue =
C->getAPIntValue();
4130 if (RawValue.has_value()) {
4131 EVT VT =
In.getValueType().getScalarType();
4137 if (
TII->isInlineConstant(FloatVal)) {
4138 Src =
CurDAG->getTargetConstant(RawValue.value(), SDLoc(In),
4143 if (
TII->isInlineConstant(RawValue.value())) {
4144 Src =
CurDAG->getTargetConstant(RawValue.value(), SDLoc(In),
4157 if (
CurDAG->isConstantIntBuildVectorOrConstantInt(SplatSrc32)) {
4162 int64_t LoImm = Lo32->getAPIntValue().getSExtValue();
4163 int64_t HiImm = Hi32->getAPIntValue().getSExtValue();
4164 int64_t Imm64I = (HiImm << 32) + LoImm;
4166 if (!isInlineImmediate(APInt(64, Imm64I)))
4169 }
else if (Imm64I != Imm64)
4173 Src =
CurDAG->getTargetConstant(Imm64, SDLoc(In), MVT::i64);
4180bool AMDGPUDAGToDAGISel::SelectSWMMACIndex8(
SDValue In,
SDValue &Src,
4186 const llvm::SDValue &ShiftSrc =
In.getOperand(0);
4195 IndexKey =
CurDAG->getTargetConstant(
Key, SDLoc(In), MVT::i32);
4199bool AMDGPUDAGToDAGISel::SelectSWMMACIndex16(
SDValue In,
SDValue &Src,
4205 const llvm::SDValue &ShiftSrc =
In.getOperand(0);
4214 IndexKey =
CurDAG->getTargetConstant(
Key, SDLoc(In), MVT::i32);
4218bool AMDGPUDAGToDAGISel::SelectSWMMACIndex32(
SDValue In,
SDValue &Src,
4226 const SDValue &ExtendSrc =
In.getOperand(0);
4230 const SDValue &CastSrc =
In.getOperand(0);
4234 if (Zero &&
Zero->getZExtValue() == 0)
4240 const SDValue &ExtractVecEltSrc = InI32.
getOperand(0);
4245 Src = ExtractVecEltSrc;
4249 IndexKey =
CurDAG->getTargetConstant(
Key, SDLoc(In), MVT::i32);
4253bool AMDGPUDAGToDAGISel::SelectVOP3OpSel(
SDValue In,
SDValue &Src,
4257 SrcMods =
CurDAG->getTargetConstant(0, SDLoc(In), MVT::i32);
4261bool AMDGPUDAGToDAGISel::SelectVOP3OpSelMods(
SDValue In,
SDValue &Src,
4264 return SelectVOP3Mods(In, Src, SrcMods);
4276 Op =
Op.getOperand(0);
4278 IsExtractHigh =
false;
4281 if (!Low16 || !Low16->isZero())
4283 Op = stripBitcast(
Op.getOperand(1));
4284 if (
Op.getValueType() != MVT::bf16)
4289 if (
Op.getValueType() != MVT::i32)
4294 if (Mask->getZExtValue() == 0xffff0000) {
4295 IsExtractHigh =
true;
4296 return Op.getOperand(0);
4305 return Op.getOperand(0);
4314bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsImpl(
SDValue In,
SDValue &Src,
4318 SelectVOP3ModsImpl(In, Src, Mods);
4320 bool IsExtractHigh =
false;
4322 Src.getOperand(0).getValueType() == VT) {
4323 Src = Src.getOperand(0);
4324 }
else if (VT == MVT::bf16) {
4332 if (Src.getValueType() != VT &&
4333 (VT != MVT::bf16 || Src.getValueType() != MVT::i32))
4336 Src = stripBitcast(Src);
4342 SelectVOP3ModsImpl(Src, Src, ModsTmp);
4357 if (Src.getValueSizeInBits() == 16) {
4366 Src.getOperand(0).getValueType() == MVT::i32) {
4367 Src = Src.getOperand(0);
4371 if (Subtarget->useRealTrue16Insts())
4373 Src = createVOP3PSrc32FromLo16(Src, In,
CurDAG, Subtarget);
4374 }
else if (IsExtractHigh)
4380bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsExt(
SDValue In,
SDValue &Src,
4383 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16))
4385 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4389bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixMods(
SDValue In,
SDValue &Src,
4392 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16);
4393 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4397bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsExtNeg(
SDValue In,
SDValue &Src,
4400 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16))
4407bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixModsNeg(
SDValue In,
SDValue &Src,
4410 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::f16);
4416bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsExt(
SDValue In,
SDValue &Src,
4419 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16))
4421 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4425bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16Mods(
SDValue In,
SDValue &Src,
4428 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16);
4429 SrcMods =
CurDAG->getTargetConstant(Mods, SDLoc(In), MVT::i32);
4433bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsExtNeg(
4436 if (!SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16))
4443bool AMDGPUDAGToDAGISel::SelectVOP3PMadMixBF16ModsNeg(
SDValue In,
SDValue &Src,
4446 SelectVOP3PMadMixModsImpl(In, Src, Mods, MVT::bf16);
4456 unsigned NumOpcodes = 0;
4469 const uint8_t SrcBits[3] = { 0xf0, 0xcc, 0xaa };
4472 if (
C->isAllOnes()) {
4482 for (
unsigned I = 0;
I < Src.size(); ++
I) {
4496 if (Src.size() == 3) {
4502 if (
C->isAllOnes()) {
4504 for (
unsigned I = 0;
I < Src.size(); ++
I) {
4505 if (Src[
I] ==
LHS) {
4517 Bits = SrcBits[Src.size()];
4522 switch (In.getOpcode()) {
4530 if (!getOperandBits(
LHS, LHSBits) ||
4531 !getOperandBits(
RHS, RHSBits)) {
4532 Src = std::move(Backup);
4533 return std::make_pair(0, 0);
4554 uint8_t LHSBitsOrig = LHSBits;
4555 uint8_t RHSBitsOrig = RHSBits;
4559 NumOpcodes += LHSOp.first;
4560 LHSBits = LHSOp.second;
4567 NumOpcodes += RHSOp.first;
4568 RHSBits = RHSOp.second;
4572 auto dependsOnSlot = [](
uint8_t TT,
int Slot) ->
bool {
4573 if (Slot < 0 || Slot > 2)
4575 const uint8_t Masks[3] = {0x0f, 0x33, 0x55};
4576 const int Shifts[3] = {4, 2, 1};
4577 return ((TT ^ (TT >> Shifts[Slot])) & Masks[Slot]) != 0;
4583 const uint8_t SrcBitsConst[3] = {0xf0, 0xcc, 0xaa};
4590 NegatedInner =
Op.getOperand(0);
4591 for (
int I = 0;
I < (int)S.size();
I++) {
4592 if (Bits == SrcBitsConst[
I] && S[
I] ==
Op)
4594 if (IsNegationOp && Bits == (
uint8_t)~SrcBitsConst[
I] &&
4595 S[
I] == NegatedInner)
4606 for (
int I = 0;
I < (int)SrcAfterLHS.
size() &&
I < 3;
I++) {
4607 if (
I < (
int)Src.size() && Src[
I] != SrcAfterLHS[
I] &&
4608 dependsOnSlot(LHSBits,
I)) {
4617 if (!Stale && !RHSOp.first) {
4618 int Slot = findSlot(RHSBitsOrig,
RHS, SrcBeforeRecurse);
4620 (Slot >= (
int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4626 if (!Stale && !LHSOp.first) {
4627 int Slot = findSlot(LHSBitsOrig,
LHS, SrcBeforeRecurse);
4629 (Slot >= (
int)Src.size() || Src[Slot] != SrcBeforeRecurse[Slot]))
4634 Src = std::move(SrcBeforeRecurse);
4635 LHSBits = LHSBitsOrig;
4636 RHSBits = RHSBitsOrig;
4642 return std::make_pair(0, 0);
4646 switch (In.getOpcode()) {
4648 TTbl = LHSBits & RHSBits;
4651 TTbl = LHSBits | RHSBits;
4654 TTbl = LHSBits ^ RHSBits;
4660 return std::make_pair(NumOpcodes + 1, TTbl);
4667 unsigned NumOpcodes;
4669 std::tie(NumOpcodes, TTbl) =
BitOp3_Op(In, Src);
4673 if (NumOpcodes < 2 || Src.empty())
4679 if (NumOpcodes < 4 && !In->isDivergent())
4682 if (NumOpcodes == 2 &&
In.getValueType() == MVT::i32) {
4687 (
In.getOperand(0).getOpcode() ==
In.getOpcode() ||
4688 In.getOperand(1).getOpcode() ==
In.getOpcode()))
4702 while (Src.size() < 3)
4703 Src.push_back(Src[0]);
4709 Tbl =
CurDAG->getTargetConstant(TTbl, SDLoc(In), MVT::i32);
4715 return CurDAG->getPOISON(MVT::i32);
4718 return CurDAG->getUNDEF(MVT::i32);
4722 return CurDAG->getConstant(
C->getZExtValue() << 16, SL, MVT::i32);
4727 return CurDAG->getConstant(
4728 C->getValueAPF().bitcastToAPInt().getZExtValue() << 16, SL, MVT::i32);
4738bool AMDGPUDAGToDAGISel::isVGPRImm(
const SDNode *
N)
const {
4739 assert(
CurDAG->getTarget().getTargetTriple().isAMDGCN());
4741 const SIRegisterInfo *SIRI = Subtarget->getRegisterInfo();
4742 const SIInstrInfo *SII = Subtarget->getInstrInfo();
4745 bool AllUsesAcceptSReg =
true;
4747 Limit < 10 && U !=
E; ++U, ++Limit) {
4749 getOperandRegClass(
U->getUser(),
U->getOperandNo());
4757 if (RC != &AMDGPU::VS_32RegClass && RC != &AMDGPU::VS_64RegClass &&
4758 RC != &AMDGPU::VS_64_Align2RegClass) {
4759 AllUsesAcceptSReg =
false;
4760 SDNode *
User =
U->getUser();
4761 if (
User->isMachineOpcode()) {
4762 unsigned Opc =
User->getMachineOpcode();
4763 const MCInstrDesc &
Desc = SII->get(
Opc);
4764 if (
Desc.isCommutable()) {
4765 unsigned OpIdx =
Desc.getNumDefs() +
U->getOperandNo();
4768 unsigned CommutedOpNo = CommuteIdx1 -
Desc.getNumDefs();
4770 getOperandRegClass(
U->getUser(), CommutedOpNo);
4771 if (CommutedRC == &AMDGPU::VS_32RegClass ||
4772 CommutedRC == &AMDGPU::VS_64RegClass ||
4773 CommutedRC == &AMDGPU::VS_64_Align2RegClass)
4774 AllUsesAcceptSReg =
true;
4782 if (!AllUsesAcceptSReg)
4786 return !AllUsesAcceptSReg && (Limit < 10);
4792 bool IsModified =
false;
4798 while (Position !=
CurDAG->allnodes_end()) {
4805 if (ResNode !=
Node) {
4811 CurDAG->RemoveDeadNodes();
4812 }
while (IsModified);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static bool getBaseWithOffsetUsingSplitOR(SelectionDAG &DAG, SDValue Addr, SDValue &N0, SDValue &N1)
static SDValue SelectSAddrFI(SelectionDAG *CurDAG, SDValue SAddr)
static SDValue matchExtFromI32orI32(SDValue Op, bool IsSigned, const SelectionDAG *DAG)
static MemSDNode * findMemSDNode(SDNode *N)
static bool IsCopyFromSGPR(const SIRegisterInfo &TRI, SDValue Val)
static SDValue combineBallotPattern(SDValue VCMP, bool &Negate)
static SDValue matchBF16FPExtendLike(SDValue Op, bool &IsExtractHigh)
static void checkWMMAElementsModifiersF16(BuildVectorSDNode *BV, std::function< bool(SDValue)> ModifierCheck)
Defines an instruction selector for the AMDGPU target.
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isNoUnsignedWrap(MachineInstr *Addr)
static bool isExtractHiElt(MachineRegisterInfo &MRI, Register In, Register &Out)
static std::pair< unsigned, uint8_t > BitOp3_Op(Register R, SmallVectorImpl< Register > &Src, const MachineRegisterInfo &MRI)
static unsigned gwsIntrinToOpcode(unsigned IntrID)
Base class for AMDGPU specific classes of TargetSubtarget.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
const HexagonInstrInfo * TII
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Register const TargetRegisterInfo * TRI
Promote Memory to Register
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Provides R600 specific target descriptions.
Interface definition for R600RegisterInfo.
const SmallVectorImpl< MachineOperand > & Cond
SI DAG Lowering interface definition.
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
AMDGPUDAGToDAGISelLegacy(TargetMachine &TM, CodeGenOptLevel OptLevel)
bool runOnMachineFunction(MachineFunction &MF) override
runOnMachineFunction - This method must be overloaded to perform the desired machine code transformat...
StringRef getPassName() const override
getPassName - Return a nice clean name for a pass.
AMDGPU specific code to select AMDGPU machine instructions for SelectionDAG operations.
bool isSDWAOperand(const SDNode *N) const
void SelectBuildVector(SDNode *N, unsigned RegClassID)
void Select(SDNode *N) override
Main hook for targets to transform nodes into machine nodes.
bool widenRegionLoad16(SDNode *N) const
bool runOnMachineFunction(MachineFunction &MF) override
void SelectVectorShuffle(SDNode *N)
void PreprocessISelDAG() override
PreprocessISelDAG - This hook allows targets to hack on the graph before instruction selection starts...
AMDGPUDAGToDAGISel()=delete
void PostprocessISelDAG() override
PostprocessISelDAG() - This hook allows the target to hack on the graph right after selection.
bool matchLoadD16FromBuildVector(SDNode *N) const
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
AMDGPUISelDAGToDAGPass(TargetMachine &TM)
static SDValue stripBitcast(SDValue Val)
static const fltSemantics & BFloat()
static const fltSemantics & IEEEhalf()
Class for arbitrary precision integers.
uint64_t getZExtValue() const
Get zero extended value.
bool isSignMask() const
Check if the APInt's value is returned by getSignMask.
bool isMaxSignedValue() const
Determine if this is the largest signed value.
int64_t getSExtValue() const
Get sign extended value.
unsigned countr_one() const
Count the number of trailing one bits.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
A "pseudo-class" with methods for operating on BUILD_VECTORs.
LLVM_ABI SDValue getSplatValue(const APInt &DemandedElts, BitVector *UndefElements=nullptr) const
Returns the demanded splatted value or a null value if this is not a splat.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
int64_t getSExtValue() const
Analysis pass which computes a DominatorTree.
Legacy analysis pass which computes a DominatorTree.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
FunctionPass class - This class is used to implement most global optimizations.
const SIInstrInfo * getInstrInfo() const override
bool useRealTrue16Insts() const
Return true if real (non-fake) variants of True16 instructions using 16-bit registers should be code-...
Generation getGeneration() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
ISD::LoadExtType getExtensionType() const
Return whether this is a plain node, or one of the varieties of value-extending loads.
Analysis pass that exposes the LoopInfo for a function.
SmallVector< LoopT *, 4 > getLoopsInPreorder() const
Return all of the loops in the function in preorder across the loop nests, with siblings in forward p...
The legacy pass manager's analysis pass to compute loop information.
unsigned getID() const
getID() - Return the register class ID number.
static MVT getIntegerVT(unsigned BitWidth)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
An SDNode that represents everything that will be needed to construct a MachineInstr.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
AnalysisType & getAnalysis() const
getAnalysis<AnalysisType>() - This function is used by subclasses to get to the analysis information ...
A set of analyses that are preserved following a run of a transformation pass.
Wrapper class representing virtual and physical registers.
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
const APInt & getAsAPIntVal() const
Helper method returns the APInt value of a ConstantSDNode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumOperands() const
Return the number of values used by this operation.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
bool isPredecessorOf(const SDNode *N) const
Return true if this node is a predecessor of N.
bool isAnyAdd() const
Returns true if the node type is ADD or PTRADD.
static use_iterator use_end()
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
bool findCommutedOpIndices(const MachineInstr &MI, unsigned &SrcOpIdx0, unsigned &SrcOpIdx1) const override
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
static bool isSGPRClass(const TargetRegisterClass *RC)
bool runOnMachineFunction(MachineFunction &MF) override
runOnMachineFunction - This method must be overloaded to perform the desired machine code transformat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
SelectionDAGISelLegacy(char &ID, std::unique_ptr< SelectionDAGISel > S)
SelectionDAGISelPass(std::unique_ptr< SelectionDAGISel > Selector)
LLVM_ABI PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
std::unique_ptr< FunctionLoweringInfo > FuncInfo
const TargetLowering * TLI
const TargetInstrInfo * TII
void ReplaceUses(SDValue F, SDValue T)
ReplaceUses - replace all uses of the old node F with the use of the new node T.
void ReplaceNode(SDNode *F, SDNode *T)
Replace all uses of F with T, then remove F from the DAG.
SelectionDAGISel(TargetMachine &tm, CodeGenOptLevel OL=CodeGenOptLevel::Default)
virtual bool runOnMachineFunction(MachineFunction &mf)
const TargetLowering * getTargetLowering() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI bool isBaseWithConstantOffset(SDValue Op) const
Return true if the specified operand is an ISD::ADD with a ConstantSDNode on the right-hand side,...
MachineFunction & getMachineFunction() const
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
ilist< SDNode >::iterator allnodes_iterator
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
static const unsigned CommuteAnyOperandIndex
Primary interface to the complete machine description for the target machine.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ PRIVATE_ADDRESS
Address space for private memory.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool isGFX12Plus(const MCSubtargetInfo &STI)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ PTRADD
PTRADD represents pointer arithmetic semantics, for targets that opt in using shouldPreservePtrArith(...
@ POISON
POISON - A poison node.
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
@ ADD
Simple integer binary arithmetic operators.
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ CONVERGENCECTRL_GLUE
This does not correspond to any convergence control intrinsic.
@ SIGN_EXTEND
Conversion operators.
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
@ SHL
Shift and rotation operations.
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
@ BRCOND
BRCOND - Conditional branch.
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
bool isExtOpcode(unsigned Opcode)
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
@ User
could "use" a pointer
This is an optimization pass for GlobalISel generic memory operations.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
constexpr bool isMask_32(uint32_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
bool isBoolSGPR(SDValue V)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
static bool getConstantValue(SDValue N, uint32_t &Out)
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
CodeGenOptLevel
Code generation optimization level.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
FunctionPass * createAMDGPUISelDag(TargetMachine &TM, CodeGenOptLevel OptLevel)
This pass converts a legalized DAG into a AMDGPU-specific.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Implement std::hash so that hash_code can be used in STL containers.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
uint64_t getScalarSizeInBits() const
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isVector() const
Return true if this is a vector value type.
bool bitsEq(EVT VT) const
Return true if this has the same number of bits as VT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
APInt getMinValue() const
Return the minimal unsigned value possible given these KnownBits.
static unsigned getSubRegFromChannel(unsigned Channel)
bool hasNoUnsignedWrap() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.