26#define DEBUG_TYPE "si-fold-operands"
47 unsigned DefSubReg = AMDGPU::NoSubRegister;
52 FoldableDef() =
delete;
54 unsigned DefSubReg = AMDGPU::NoSubRegister)
55 : DefRC(DefRC), DefSubReg(DefSubReg), Kind(FoldOp.
getType()) {
58 ImmToFold = FoldOp.
getImm();
59 }
else if (FoldOp.
isFI()) {
60 FrameIndexToFold = FoldOp.
getIndex();
70 unsigned DefSubReg = AMDGPU::NoSubRegister)
71 : ImmToFold(FoldImm), DefRC(DefRC), DefSubReg(DefSubReg),
76 FoldableDef Copy(*
this);
77 Copy.DefSubReg =
TRI.composeSubRegIndices(DefSubReg, SubReg);
85 return OpToFold->getReg();
88 unsigned getSubReg()
const {
90 return OpToFold->getSubReg();
101 return FrameIndexToFold;
109 std::optional<int64_t> getEffectiveImmVal()
const {
117 unsigned OpIdx)
const {
120 std::optional<int64_t> ImmToFold = getEffectiveImmVal();
127 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
130 if (DefSubReg != AMDGPU::NoSubRegister)
133 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
138 if (DefSubReg != AMDGPU::NoSubRegister)
140 return TII.isOperandLegal(
MI, OpIdx, OpToFold);
147struct FoldCandidate {
155 bool Commuted =
false,
int ShrinkOp = -1)
156 :
UseMI(
MI), Def(Def), ShrinkOpcode(ShrinkOp), UseOpNo(OpNo),
157 Commuted(Commuted) {}
159 bool isFI()
const {
return Def.isFI(); }
163 return Def.FrameIndexToFold;
166 bool isImm()
const {
return Def.isImm(); }
168 bool isReg()
const {
return Def.isReg(); }
172 bool isGlobal()
const {
return Def.isGlobal(); }
174 bool needsShrink()
const {
return ShrinkOpcode != -1; }
177class SIFoldOperandsImpl {
188 const FoldableDef &OpToFold)
const;
191 unsigned convertToVALUOp(
unsigned Opc,
bool UseVOP3 =
false)
const {
193 case AMDGPU::S_ADD_I32: {
194 if (ST->hasAddNoCarryInsts())
195 return UseVOP3 ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_U32_e32;
196 return UseVOP3 ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
198 case AMDGPU::S_OR_B32:
199 return UseVOP3 ? AMDGPU::V_OR_B32_e64 : AMDGPU::V_OR_B32_e32;
200 case AMDGPU::S_AND_B32:
201 return UseVOP3 ? AMDGPU::V_AND_B32_e64 : AMDGPU::V_AND_B32_e32;
202 case AMDGPU::S_MUL_I32:
203 return AMDGPU::V_MUL_LO_U32_e64;
205 return AMDGPU::INSTRUCTION_LIST_END;
209 bool foldCopyToVGPROfScalarAddOfFrameIndex(
Register DstReg,
Register SrcReg,
215 int64_t ImmVal)
const;
219 int64_t ImmVal)
const;
223 const FoldableDef &OpToFold)
const;
226 bool isTemporallyDivergentUse(
const FoldableDef &OpToFold,
234 getRegSeqInit(
SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
237 std::pair<int64_t, const TargetRegisterClass *>
254 bool foldInstOperand(
MachineInstr &
MI,
const FoldableDef &OpToFold)
const;
256 bool foldCopyToAGPRRegSequence(
MachineInstr *CopyMI)
const;
263 std::pair<const MachineOperand *, int> isOMod(
const MachineInstr &
MI)
const;
272 SIFoldOperandsImpl() =
default;
287 &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
288 return SIFoldOperandsImpl().run(MF, MLI);
291 StringRef getPassName()
const override {
return "SI Fold Operands"; }
313char SIFoldOperandsLegacy::ID = 0;
322 TRI.getSubRegisterClass(RC, MO.getSubReg()))
330 case AMDGPU::V_MAC_F32_e64:
331 return AMDGPU::V_MAD_F32_e64;
332 case AMDGPU::V_MAC_F16_e64:
333 return AMDGPU::V_MAD_F16_e64;
334 case AMDGPU::V_FMAC_F32_e64:
335 return AMDGPU::V_FMA_F32_e64;
336 case AMDGPU::V_FMAC_F16_e64:
337 return AMDGPU::V_FMA_F16_gfx9_e64;
338 case AMDGPU::V_FMAC_F16_t16_e64:
339 return AMDGPU::V_FMA_F16_gfx9_t16_e64;
340 case AMDGPU::V_FMAC_F16_fake16_e64:
341 return AMDGPU::V_FMA_F16_gfx9_fake16_e64;
342 case AMDGPU::V_FMAC_LEGACY_F32_e64:
343 return AMDGPU::V_FMA_LEGACY_F32_e64;
344 case AMDGPU::V_FMAC_F64_e64:
345 return AMDGPU::V_FMA_F64_e64;
347 return AMDGPU::INSTRUCTION_LIST_END;
353 const FoldableDef &OpToFold)
const {
354 if (!OpToFold.isFI())
357 const unsigned Opc =
UseMI.getOpcode();
359 case AMDGPU::S_ADD_I32:
360 case AMDGPU::S_ADD_U32:
361 case AMDGPU::V_ADD_U32_e32:
362 case AMDGPU::V_ADD_CO_U32_e32:
366 return UseMI.getOperand(OpNo == 1 ? 2 : 1).isImm() &&
368 case AMDGPU::V_ADD_U32_e64:
369 case AMDGPU::V_ADD_CO_U32_e64:
370 return UseMI.getOperand(OpNo == 2 ? 3 : 2).isImm() &&
377 return OpNo == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
381 int SIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
385 int VIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
386 return OpNo == VIdx && SIdx == -1;
392bool SIFoldOperandsImpl::foldCopyToVGPROfScalarAddOfFrameIndex(
394 if (
TRI->isVGPR(*MRI, DstReg) &&
TRI->isSGPRReg(*MRI, SrcReg) &&
397 if (!Def ||
Def->getNumOperands() != 4)
400 MachineOperand *Src0 = &
Def->getOperand(1);
401 MachineOperand *Src1 = &
Def->getOperand(2);
412 const bool UseVOP3 = !Src0->
isImm() ||
TII->isInlineConstant(*Src0);
413 unsigned NewOp = convertToVALUOp(
Def->getOpcode(), UseVOP3);
414 if (NewOp == AMDGPU::INSTRUCTION_LIST_END ||
415 !
Def->getOperand(3).isDead())
418 MachineBasicBlock *
MBB =
Def->getParent();
420 if (NewOp != AMDGPU::V_ADD_CO_U32_e32) {
421 MachineInstrBuilder
Add =
424 if (
Add->getDesc().getNumDefs() == 2) {
426 Add.addDef(CarryOutReg, RegState::Dead);
430 Add.add(*Src0).add(*Src1).setMIFlags(
Def->getFlags());
434 Def->eraseFromParent();
435 MI.eraseFromParent();
439 assert(NewOp == AMDGPU::V_ADD_CO_U32_e32);
450 Def->eraseFromParent();
451 MI.eraseFromParent();
460 return new SIFoldOperandsLegacy();
463bool SIFoldOperandsImpl::canUseImmWithOpSel(
const MachineInstr *
MI,
465 int64_t ImmVal)
const {
472 int OpNo =
MI->getOperandNo(&Old);
474 unsigned Opcode =
MI->getOpcode();
475 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
497bool SIFoldOperandsImpl::tryFoldImmWithOpSel(MachineInstr *
MI,
unsigned UseOpNo,
498 int64_t ImmVal)
const {
499 MachineOperand &Old =
MI->getOperand(UseOpNo);
500 unsigned Opcode =
MI->getOpcode();
501 int OpNo =
MI->getOperandNo(&Old);
502 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
514 AMDGPU::OpName ModName = AMDGPU::OpName::NUM_OPERAND_NAMES;
515 unsigned SrcIdx = ~0;
516 if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0)) {
517 ModName = AMDGPU::OpName::src0_modifiers;
519 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1)) {
520 ModName = AMDGPU::OpName::src1_modifiers;
522 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2)) {
523 ModName = AMDGPU::OpName::src2_modifiers;
526 assert(ModName != AMDGPU::OpName::NUM_OPERAND_NAMES);
527 int ModIdx = AMDGPU::getNamedOperandIdx(Opcode, ModName);
528 MachineOperand &
Mod =
MI->getOperand(ModIdx);
529 unsigned ModVal =
Mod.getImm();
535 uint32_t
Imm = (
static_cast<uint32_t
>(ImmHi) << 16) | ImmLo;
540 auto tryFoldToInline = [&](uint32_t
Imm) ->
bool {
549 uint16_t
Lo =
static_cast<uint16_t
>(
Imm);
550 uint16_t
Hi =
static_cast<uint16_t
>(
Imm >> 16);
556 if (ST->hasBF16InlineConstFromUpperFP32() &&
560 Mod.setImm(NewModVal);
565 if (
static_cast<int16_t
>(
Lo) < 0) {
566 int32_t SExt =
static_cast<int16_t
>(
Lo);
568 Mod.setImm(NewModVal);
583 uint32_t Swapped = (
static_cast<uint32_t
>(
Lo) << 16) |
Hi;
594 if (tryFoldToInline(Imm))
603 bool IsUAdd = Opcode == AMDGPU::V_PK_ADD_U16;
604 bool IsUSub = Opcode == AMDGPU::V_PK_SUB_U16;
605 if (SrcIdx == 1 && (IsUAdd || IsUSub)) {
607 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::clamp);
608 bool Clamp =
MI->getOperand(ClampIdx).getImm() != 0;
611 uint16_t NegLo = -
static_cast<uint16_t
>(
Imm);
612 uint16_t NegHi = -
static_cast<uint16_t
>(
Imm >> 16);
613 uint32_t NegImm = (
static_cast<uint32_t
>(NegHi) << 16) | NegLo;
615 if (tryFoldToInline(NegImm)) {
617 IsUAdd ? AMDGPU::V_PK_SUB_U16 : AMDGPU::V_PK_ADD_U16;
618 MI->setDesc(
TII->get(NegOpcode));
627bool SIFoldOperandsImpl::updateOperand(FoldCandidate &Fold)
const {
628 MachineInstr *
MI = Fold.UseMI;
629 MachineOperand &Old =
MI->getOperand(Fold.UseOpNo);
632 std::optional<int64_t> ImmVal;
634 ImmVal = Fold.Def.getEffectiveImmVal();
636 if (ImmVal && canUseImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal)) {
637 if (tryFoldImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal))
643 int OpNo =
MI->getOperandNo(&Old);
644 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
650 if ((Fold.isImm() || Fold.isFI() || Fold.isGlobal()) && Fold.needsShrink()) {
651 MachineBasicBlock *
MBB =
MI->getParent();
658 int Op32 = Fold.ShrinkOpcode;
659 MachineOperand &Dst0 =
MI->getOperand(0);
660 MachineOperand &Dst1 =
MI->getOperand(1);
668 MachineInstr *Inst32 =
TII->buildShrunkInst(*
MI, Op32);
670 if (HaveNonDbgCarryUse) {
673 .
addReg(AMDGPU::VCC, RegState::Kill);
683 for (
unsigned I =
MI->getNumOperands() - 1;
I > 0; --
I)
684 MI->removeOperand(
I);
685 MI->setDesc(
TII->get(AMDGPU::IMPLICIT_DEF));
688 TII->commuteInstruction(*Inst32,
false);
692 assert(!Fold.needsShrink() &&
"not handled");
697 if (NewMFMAOpc == -1)
699 MI->setDesc(
TII->get(NewMFMAOpc));
700 MI->untieRegOperand(0);
701 const MCInstrDesc &MCID =
MI->getDesc();
702 for (
unsigned I = 0;
I <
MI->getNumDefs(); ++
I)
704 MI->getOperand(
I).setIsEarlyClobber(
true);
709 int OpNo =
MI->getOperandNo(&Old);
710 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
713 if (ST->hasBF16InlineConstFromUpperFP32() &&
715 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::src0)) {
716 unsigned Opcode =
MI->getOpcode();
717 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
720 TII->isInlineConstant(*ImmVal, OpType)) {
723 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0_modifiers);
726 MachineOperand &ModOp =
MI->getOperand(Mod0);
737 if (Fold.isGlobal()) {
738 Old.
ChangeToGA(Fold.Def.OpToFold->getGlobal(),
739 Fold.Def.OpToFold->getOffset(),
740 Fold.Def.OpToFold->getTargetFlags());
749 MachineOperand *
New = Fold.Def.OpToFold;
753 TII->getRegClass(
MI->getDesc(), Fold.UseOpNo)) {
755 TRI->getRegClassForReg(*MRI,
New->getReg());
758 if (
New->getSubReg()) {
760 TRI->getMatchingSuperRegClass(NewRC, OpRC,
New->getSubReg());
766 if (
New->getReg().isVirtual() &&
769 <<
TRI->getRegClassName(ConstrainRC) <<
'\n');
776 if (Old.
getSubReg() == AMDGPU::lo16 &&
TRI->isSGPRReg(*MRI,
New->getReg()))
778 if (
New->getReg().isPhysical()) {
786 if (
MI->isBundledWithPred()) {
788 for (MachineOperand &MO : Header.operands()) {
789 if (MO.getReg() == OldReg) {
790 MO.setReg(
New->getReg());
791 MO.setSubReg(
New->getSubReg());
800 FoldCandidate &&Entry) {
802 for (FoldCandidate &Fold : FoldList)
803 if (Fold.UseMI == Entry.UseMI && Fold.UseOpNo == Entry.UseOpNo)
805 LLVM_DEBUG(
dbgs() <<
"Append " << (Entry.Commuted ?
"commuted" :
"normal")
806 <<
" operand " << Entry.UseOpNo <<
"\n " << *Entry.UseMI);
812 const FoldableDef &FoldOp,
813 bool Commuted =
false,
int ShrinkOp = -1) {
815 FoldCandidate(
MI, OpNo, FoldOp, Commuted, ShrinkOp));
823 if (!ST->hasPKF32InstsReplicatingLower32BitsOfScalarInput())
833 const FoldableDef &OpToFold) {
834 assert(OpToFold.isImm() &&
"Expected immediate operand");
835 uint64_t ImmVal = OpToFold.getEffectiveImmVal().value();
841bool SIFoldOperandsImpl::tryAddToFoldList(
842 SmallVectorImpl<FoldCandidate> &FoldList, MachineInstr *
MI,
unsigned OpNo,
843 const FoldableDef &OpToFold)
const {
844 const unsigned Opc =
MI->getOpcode();
846 auto tryToFoldAsFMAAKorMK = [&]() {
847 if (!OpToFold.isImm())
850 const bool TryAK = OpNo == 3;
851 const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
852 MI->setDesc(
TII->get(NewOpc));
855 bool FoldAsFMAAKorMK =
856 tryAddToFoldList(FoldList,
MI, TryAK ? 3 : 2, OpToFold);
857 if (FoldAsFMAAKorMK) {
859 MI->untieRegOperand(3);
862 MachineOperand &Op1 =
MI->getOperand(1);
863 MachineOperand &Op2 =
MI->getOperand(2);
880 bool IsLegal = OpToFold.isOperandLegal(*
TII, *
MI, OpNo);
881 if (!IsLegal && OpToFold.isImm()) {
882 if (std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal())
883 IsLegal = canUseImmWithOpSel(
MI, OpNo, *ImmVal);
889 if (NewOpc != AMDGPU::INSTRUCTION_LIST_END) {
892 MI->setDesc(
TII->get(NewOpc));
897 bool FoldAsMAD = tryAddToFoldList(FoldList,
MI, OpNo, OpToFold);
899 MI->untieRegOperand(OpNo);
903 MI->removeOperand(
MI->getNumExplicitOperands() - 1);
909 if (
Opc == AMDGPU::S_FMAC_F32 && OpNo == 3) {
910 if (tryToFoldAsFMAAKorMK())
915 if (OpToFold.isImm()) {
917 if (
Opc == AMDGPU::S_SETREG_B32)
918 ImmOpc = AMDGPU::S_SETREG_IMM32_B32;
919 else if (
Opc == AMDGPU::S_SETREG_B32_mode)
920 ImmOpc = AMDGPU::S_SETREG_IMM32_B32_mode;
922 MI->setDesc(
TII->get(ImmOpc));
931 bool CanCommute =
TII->findCommutedOpIndices(*
MI, OpNo, CommuteOpNo);
935 MachineOperand &
Op =
MI->getOperand(OpNo);
936 MachineOperand &CommutedOp =
MI->getOperand(CommuteOpNo);
942 if (!
Op.isReg() || !CommutedOp.
isReg())
947 if (
Op.isReg() && CommutedOp.
isReg() &&
948 (
Op.getReg() == CommutedOp.
getReg() &&
952 if (!
TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo))
956 if (!OpToFold.isOperandLegal(*
TII, *
MI, CommuteOpNo)) {
957 if ((
Opc != AMDGPU::V_ADD_CO_U32_e64 &&
Opc != AMDGPU::V_SUB_CO_U32_e64 &&
958 Opc != AMDGPU::V_SUBREV_CO_U32_e64) ||
959 (!OpToFold.isImm() && !OpToFold.isFI() && !OpToFold.isGlobal())) {
960 TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo);
966 MachineOperand &OtherOp =
MI->getOperand(OpNo);
967 if (!OtherOp.
isReg() ||
974 unsigned MaybeCommutedOpc =
MI->getOpcode();
988 if (
Opc == AMDGPU::S_FMAC_F32 &&
989 (OpNo != 1 || !
MI->getOperand(1).isIdenticalTo(
MI->getOperand(2)))) {
990 if (tryToFoldAsFMAAKorMK())
996 if (OpToFold.isImm() &&
1005bool SIFoldOperandsImpl::isUseSafeToFold(
const MachineInstr &
MI,
1006 const MachineOperand &UseMO)
const {
1008 return !
TII->isSDWA(
MI);
1015 if (
MI.modifiesRegister(
TRI.getExec(), &
TRI))
1023bool SIFoldOperandsImpl::isTemporallyDivergentUse(
1024 const FoldableDef &OpToFold,
const MachineInstr &
UseMI)
const {
1025 if (!OpToFold.isReg())
1027 const MachineInstr *
DefMI = OpToFold.DefMI;
1030 !
TRI->isSGPRReg(*MRI, OpToFold.getReg()))
1042 SubDef &&
TII.isFoldableCopy(*SubDef);
1044 unsigned SrcIdx =
TII.getFoldableCopySrcIdx(*SubDef);
1053 if (
SrcOp.getSubReg())
1061 MachineInstr &RegSeq,
1062 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs)
const {
1078 else if (!
TRI->getCommonSubClass(RC, OpRC))
1083 Defs.emplace_back(&SrcOp, SubRegIdx);
1088 if (DefSrc && (DefSrc->
isReg() || DefSrc->
isImm())) {
1089 Defs.emplace_back(DefSrc, SubRegIdx);
1093 Defs.emplace_back(&SrcOp, SubRegIdx);
1103 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
1106 if (!Def || !
Def->isRegSequence())
1109 return getRegSeqInit(*Def, Defs);
1112std::pair<int64_t, const TargetRegisterClass *>
1113SIFoldOperandsImpl::isRegSeqSplat(MachineInstr &RegSeq)
const {
1119 bool TryToMatchSplat64 =
false;
1121 std::optional<int64_t>
Imm;
1122 for (
unsigned I = 0,
E = Defs.
size();
I !=
E; ++
I) {
1123 const MachineOperand *
Op = Defs[
I].first;
1127 if (!Def ||
Def->isImplicitDef())
1133 int64_t SubImm =
Op->getImm();
1139 if (Imm != SubImm) {
1140 if (
I == 1 && (
E & 1) == 0) {
1143 TryToMatchSplat64 =
true;
1151 if (!TryToMatchSplat64) {
1153 return {*
Imm, SrcRC};
1160 for (
unsigned I = 0,
E = Defs.
size();
I !=
E;
I += 2) {
1161 const MachineOperand *Op0 = Defs[
I].first;
1162 const MachineOperand *Op1 = Defs[
I + 1].first;
1167 unsigned SubReg0 = Defs[
I].second;
1168 unsigned SubReg1 = Defs[
I + 1].second;
1172 if (
TRI->getChannelFromSubReg(SubReg0) + 1 !=
1173 TRI->getChannelFromSubReg(SubReg1))
1176 if (
TRI->getSubRegIdxSize(SubReg0) != 32)
1181 SplatVal64 = MergedVal;
1182 else if (SplatVal64 != MergedVal)
1189 return {SplatVal64, RC64};
1192bool SIFoldOperandsImpl::tryFoldRegSeqSplat(
1193 MachineInstr *
UseMI,
unsigned UseOpIdx, int64_t SplatVal,
1196 if (UseOpIdx >=
Desc.getNumOperands())
1203 int16_t RCID =
TII->getOpRegClassID(
Desc.operands()[UseOpIdx]);
1212 if (SplatVal != 0 && SplatVal != -1) {
1216 uint8_t OpTy =
Desc.operands()[UseOpIdx].OperandType;
1223 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0);
1230 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0_sub1);
1236 if (!
TRI->getCommonSubClass(OpRC, SplatRC))
1241 if (!
TII->isOperandLegal(*
UseMI, UseOpIdx, &TmpOp))
1247bool SIFoldOperandsImpl::tryToFoldACImm(
1248 const FoldableDef &OpToFold, MachineInstr *
UseMI,
unsigned UseOpIdx,
1249 SmallVectorImpl<FoldCandidate> &FoldList)
const {
1251 if (UseOpIdx >=
Desc.getNumOperands())
1258 if (OpToFold.isImm() && OpToFold.isOperandLegal(*
TII, *
UseMI, UseOpIdx)) {
1269bool SIFoldOperandsImpl::foldOperand(
1270 FoldableDef OpToFold, MachineInstr *
UseMI,
int UseOpIdx,
1271 SmallVectorImpl<FoldCandidate> &FoldList,
1272 SmallVectorImpl<MachineInstr *> &CopiesToReplace)
const {
1276 if (!isUseSafeToFold(*
UseMI, *UseOp))
1279 if (isTemporallyDivergentUse(OpToFold, *
UseMI))
1283 if (UseOp->
isReg() && OpToFold.isReg()) {
1287 if (UseOp->
getSubReg() != AMDGPU::NoSubRegister &&
1289 !
TRI->isSGPRReg(*MRI, OpToFold.getReg())))
1302 std::tie(SplatVal, SplatRC) = isRegSeqSplat(*
UseMI);
1307 for (
unsigned I = 0;
I != UsesToProcess.size(); ++
I) {
1308 MachineOperand *RSUse = UsesToProcess[
I];
1309 MachineInstr *RSUseMI = RSUse->
getParent();
1319 if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
1320 FoldableDef SplatDef(SplatVal, SplatRC);
1328 if (RSUse->
getSubReg() != RegSeqDstSubReg)
1334 FoldList, CopiesToReplace);
1340 if (tryToFoldACImm(OpToFold,
UseMI, UseOpIdx, FoldList))
1343 if (frameIndexMayFold(*
UseMI, UseOpIdx, OpToFold)) {
1348 if (
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::srsrc)->getReg() !=
1354 MachineOperand &SOff =
1355 *
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::soffset);
1366 TII->getNamedOperand(*
UseMI, AMDGPU::OpName::cpol)->getImm();
1381 bool FoldingImmLike =
1382 OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
1401 for (
unsigned MovOp :
1402 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
1403 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_MOV_B16_t16_e64,
1404 AMDGPU::V_ACCVGPR_WRITE_B32_e64, AMDGPU::AV_MOV_B32_IMM_PSEUDO,
1405 AMDGPU::AV_MOV_B64_IMM_PSEUDO}) {
1406 const MCInstrDesc &MovDesc =
TII->get(MovOp);
1416 const int SrcIdx = MovOp == AMDGPU::V_MOV_B16_t16_e64 ? 2 : 1;
1418 int16_t RegClassID =
TII->getOpRegClassID(MovDesc.
operands()[SrcIdx]);
1419 if (RegClassID != -1) {
1423 MovSrcRC =
TRI->getMatchingSuperRegClass(SrcRC, MovSrcRC, UseSubReg);
1427 if (MovOp == AMDGPU::AV_MOV_B32_IMM_PSEUDO &&
1428 (!OpToFold.isImm() ||
1429 !
TII->isImmOperandLegal(MovDesc, SrcIdx,
1430 *OpToFold.getEffectiveImmVal())))
1443 if (!OpToFold.isImm() ||
1444 !
TII->isImmOperandLegal(MovDesc, 1, *OpToFold.getEffectiveImmVal()))
1450 while (ImpOpI != ImpOpE) {
1457 if (MovOp == AMDGPU::V_MOV_B16_t16_e64) {
1459 MachineOperand NewSrcOp(SrcOp);
1481 LLVM_DEBUG(
dbgs() <<
"Folding " << *OpToFold.OpToFold <<
"\n into "
1486 unsigned SubRegIdx = OpToFold.getSubReg();
1500 static_assert(AMDGPU::sub1_hi16 == 12,
"Subregister layout has changed");
1505 if (SubRegIdx > AMDGPU::sub1) {
1506 LaneBitmask
M =
TRI->getSubRegIndexLaneMask(SubRegIdx);
1507 M |=
M.getLane(
M.getHighestLane() - 1);
1508 SmallVector<unsigned, 4> Indexes;
1509 TRI->getCoveringSubRegIndexes(
TRI->getRegClassForReg(*MRI,
UseReg), M,
1511 assert(Indexes.
size() == 1 &&
"Expected one 32-bit subreg to cover");
1512 SubRegIdx = Indexes[0];
1514 }
else if (
TII->getOpSize(*
UseMI, 1) == 4)
1517 SubRegIdx = AMDGPU::sub0;
1522 OpToFold.OpToFold->setIsKill(
false);
1527 if (foldCopyToAGPRRegSequence(
UseMI))
1532 if (UseOpc == AMDGPU::V_READFIRSTLANE_B32 ||
1533 (UseOpc == AMDGPU::V_READLANE_B32 &&
1535 AMDGPU::getNamedOperandIdx(UseOpc, AMDGPU::OpName::src0))) {
1540 if (FoldingImmLike) {
1543 *OpToFold.DefMI, *
UseMI))
1549 if (OpToFold.isImm()) {
1551 *OpToFold.getEffectiveImmVal());
1552 }
else if (OpToFold.isFI())
1555 assert(OpToFold.isGlobal());
1557 OpToFold.OpToFold->getOffset(),
1558 OpToFold.OpToFold->getTargetFlags());
1564 if (OpToFold.isReg() &&
TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
1567 *OpToFold.DefMI, *
UseMI))
1589 UseDesc.
operands()[UseOpIdx].RegClass == -1)
1597 Changed |= tryAddToFoldList(FoldList,
UseMI, UseOpIdx, OpToFold);
1604 case AMDGPU::S_ADD_I32:
1605 case AMDGPU::S_ADD_U32:
1608 case AMDGPU::S_SUB_I32:
1609 case AMDGPU::S_SUB_U32:
1612 case AMDGPU::V_AND_B32_e64:
1613 case AMDGPU::V_AND_B32_e32:
1614 case AMDGPU::S_AND_B32:
1617 case AMDGPU::V_OR_B32_e64:
1618 case AMDGPU::V_OR_B32_e32:
1619 case AMDGPU::S_OR_B32:
1622 case AMDGPU::V_XOR_B32_e64:
1623 case AMDGPU::V_XOR_B32_e32:
1624 case AMDGPU::S_XOR_B32:
1627 case AMDGPU::S_XNOR_B32:
1630 case AMDGPU::S_NAND_B32:
1633 case AMDGPU::S_NOR_B32:
1636 case AMDGPU::S_ANDN2_B32:
1639 case AMDGPU::S_ORN2_B32:
1642 case AMDGPU::V_LSHL_B32_e64:
1643 case AMDGPU::V_LSHL_B32_e32:
1644 case AMDGPU::S_LSHL_B32:
1646 Result =
LHS << (
RHS & 31);
1648 case AMDGPU::V_LSHLREV_B32_e64:
1649 case AMDGPU::V_LSHLREV_B32_e32:
1650 Result =
RHS << (
LHS & 31);
1652 case AMDGPU::V_LSHR_B32_e64:
1653 case AMDGPU::V_LSHR_B32_e32:
1654 case AMDGPU::S_LSHR_B32:
1655 Result =
LHS >> (
RHS & 31);
1657 case AMDGPU::V_LSHRREV_B32_e64:
1658 case AMDGPU::V_LSHRREV_B32_e32:
1659 Result =
RHS >> (
LHS & 31);
1661 case AMDGPU::V_ASHR_I32_e64:
1662 case AMDGPU::V_ASHR_I32_e32:
1663 case AMDGPU::S_ASHR_I32:
1664 Result =
static_cast<int32_t
>(
LHS) >> (
RHS & 31);
1666 case AMDGPU::V_ASHRREV_I32_e64:
1667 case AMDGPU::V_ASHRREV_I32_e32:
1668 Result =
static_cast<int32_t
>(
RHS) >> (
LHS & 31);
1676 return IsScalar ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1682bool SIFoldOperandsImpl::tryConstantFoldOp(MachineInstr *
MI)
const {
1683 if (!
MI->allImplicitDefsAreDead())
1686 unsigned Opc =
MI->getOpcode();
1688 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
1692 MachineOperand *Src0 = &
MI->getOperand(Src0Idx);
1693 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*Src0);
1695 if ((
Opc == AMDGPU::V_NOT_B32_e64 ||
Opc == AMDGPU::V_NOT_B32_e32 ||
1696 Opc == AMDGPU::S_NOT_B32) &&
1698 MI->getOperand(1).ChangeToImmediate(~*Src0Imm);
1699 TII->mutateAndCleanupImplicit(
1704 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
1708 MachineOperand *Src1 = &
MI->getOperand(Src1Idx);
1709 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*Src1);
1711 if (!Src0Imm && !Src1Imm)
1717 if (Src0Imm && Src1Imm) {
1722 bool IsSGPR =
TRI->isSGPRReg(*MRI,
MI->getOperand(0).getReg());
1726 MI->getOperand(Src0Idx).ChangeToImmediate(NewImm);
1727 MI->removeOperand(Src1Idx);
1734 if (
Opc == AMDGPU::S_SUB_I32 ||
Opc == AMDGPU::S_SUB_U32) {
1735 if (Src1Imm &&
static_cast<int32_t
>(*Src1Imm) == 0) {
1737 MI->removeOperand(Src1Idx);
1738 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1744 if (!
MI->isCommutable())
1747 if (Src0Imm && !Src1Imm) {
1753 int32_t Src1Val =
static_cast<int32_t
>(*Src1Imm);
1754 if (
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_ADD_U32) {
1757 MI->removeOperand(Src1Idx);
1758 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1764 if (
Opc == AMDGPU::V_OR_B32_e64 ||
1765 Opc == AMDGPU::V_OR_B32_e32 ||
1766 Opc == AMDGPU::S_OR_B32) {
1769 MI->removeOperand(Src1Idx);
1770 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1771 }
else if (Src1Val == -1) {
1773 MI->removeOperand(Src0Idx);
1774 TII->mutateAndCleanupImplicit(
1782 if (
Opc == AMDGPU::V_AND_B32_e64 ||
Opc == AMDGPU::V_AND_B32_e32 ||
1783 Opc == AMDGPU::S_AND_B32) {
1786 MI->removeOperand(Src0Idx);
1787 TII->mutateAndCleanupImplicit(
1789 }
else if (Src1Val == -1) {
1791 MI->removeOperand(Src1Idx);
1792 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1799 if (
Opc == AMDGPU::V_XOR_B32_e64 ||
Opc == AMDGPU::V_XOR_B32_e32 ||
1800 Opc == AMDGPU::S_XOR_B32) {
1803 MI->removeOperand(Src1Idx);
1804 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1813bool SIFoldOperandsImpl::tryFoldCndMask(MachineInstr &
MI)
const {
1814 unsigned Opc =
MI.getOpcode();
1815 if (
Opc != AMDGPU::V_CNDMASK_B32_e32 &&
Opc != AMDGPU::V_CNDMASK_B32_e64 &&
1816 Opc != AMDGPU::V_CNDMASK_B64_PSEUDO)
1819 MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1820 MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1822 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*Src1);
1826 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*Src0);
1827 if (!Src0Imm || *Src0Imm != *Src1Imm)
1832 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1_modifiers);
1834 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
1835 if ((Src1ModIdx != -1 &&
MI.getOperand(Src1ModIdx).getImm() != 0) ||
1836 (Src0ModIdx != -1 &&
MI.getOperand(Src0ModIdx).getImm() != 0))
1842 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
1844 MI.removeOperand(Src2Idx);
1845 MI.removeOperand(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1));
1846 if (Src1ModIdx != -1)
1847 MI.removeOperand(Src1ModIdx);
1848 if (Src0ModIdx != -1)
1849 MI.removeOperand(Src0ModIdx);
1850 TII->mutateAndCleanupImplicit(
MI, NewDesc);
1855bool SIFoldOperandsImpl::tryFoldZeroHighBits(MachineInstr &
MI)
const {
1856 if (
MI.getOpcode() != AMDGPU::V_AND_B32_e64 &&
1857 MI.getOpcode() != AMDGPU::V_AND_B32_e32)
1860 std::optional<int64_t> Src0Imm =
1861 TII->getImmOrMaterializedImm(
MI.getOperand(1));
1862 if (!Src0Imm || *Src0Imm != 0xffff || !
MI.getOperand(2).isReg())
1866 MachineInstr *SrcDef = MRI->
getVRegDef(Src1);
1872 if (!
MI.getOperand(2).isKill())
1874 MI.eraseFromParent();
1878bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &
MI,
1879 const FoldableDef &OpToFold)
const {
1883 SmallVector<MachineInstr *, 4> CopiesToReplace;
1885 MachineOperand &Dst =
MI.getOperand(0);
1890 for (
auto *U : UsesToProcess) {
1891 MachineInstr *
UseMI =
U->getParent();
1893 FoldableDef SubOpToFold = OpToFold.getWithSubReg(*
TRI,
U->getSubReg());
1898 if (CopiesToReplace.
empty() && FoldList.
empty())
1902 for (MachineInstr *Copy : CopiesToReplace)
1903 Copy->addImplicitDefUseOperands(*MF);
1905 SetVector<MachineInstr *> ConstantFoldCandidates;
1906 for (FoldCandidate &Fold : FoldList) {
1907 assert(!Fold.isReg() || Fold.Def.OpToFold);
1908 if (Fold.isReg() && Fold.getReg().isVirtual()) {
1910 const MachineInstr *
DefMI = Fold.Def.DefMI;
1918 assert(Fold.Def.OpToFold && Fold.isReg());
1925 <<
static_cast<int>(Fold.UseOpNo) <<
" of "
1929 ConstantFoldCandidates.
insert(Fold.UseMI);
1931 }
else if (Fold.Commuted) {
1933 TII->commuteInstruction(*Fold.UseMI,
false);
1937 for (MachineInstr *
MI : ConstantFoldCandidates) {
1938 if (tryConstantFoldOp(
MI)) {
1948bool SIFoldOperandsImpl::foldCopyToAGPRRegSequence(MachineInstr *CopyMI)
const {
1955 if (!
TRI->isAGPRClass(DefRC))
1967 DenseMap<TargetInstrInfo::RegSubRegPair, Register> VGPRCopies;
1976 unsigned NumFoldable = 0;
1978 for (
unsigned I = 1;
I != NumRegSeqOperands;
I += 2) {
1995 DefRC, &AMDGPU::AGPR_32RegClass, SubRegIdx);
2015 TRI->getMatchingSuperRegClass(DefRC, InputRC, SubRegIdx);
2026 if (NumFoldable == 0)
2029 CopyMI->
setDesc(
TII->get(AMDGPU::REG_SEQUENCE));
2033 for (
auto [Def, DestSubIdx] : NewDefs) {
2034 if (!
Def->isReg()) {
2038 BuildMI(
MBB, CopyMI,
DL,
TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), Tmp)
2043 Def->setIsKill(
false);
2045 Register &VGPRCopy = VGPRCopies[Src];
2048 TRI->getSubRegisterClass(UseRC, DestSubIdx);
2073 B.addImm(DestSubIdx);
2080bool SIFoldOperandsImpl::tryFoldFoldableCopy(
2081 MachineInstr &
MI, MachineOperand *&CurrentKnownM0Val)
const {
2085 if (DstReg == AMDGPU::M0) {
2086 MachineOperand &NewM0Val =
MI.getOperand(1);
2087 if (CurrentKnownM0Val && CurrentKnownM0Val->
isIdenticalTo(NewM0Val)) {
2088 MI.eraseFromParent();
2099 MachineOperand *OpToFoldPtr;
2100 if (
MI.getOpcode() == AMDGPU::V_MOV_B16_t16_e64) {
2102 if (
TII->hasAnyModifiersSet(
MI))
2104 OpToFoldPtr = &
MI.getOperand(2);
2106 OpToFoldPtr = &
MI.getOperand(1);
2107 MachineOperand &OpToFold = *OpToFoldPtr;
2111 if (!FoldingImm && !OpToFold.
isReg())
2116 !
TRI->isConstantPhysReg(OpToFold.
getReg()))
2145 if (
MI.getOpcode() == AMDGPU::COPY && OpToFold.
isReg() &&
2147 if (DstRC == &AMDGPU::SReg_32RegClass &&
2156 if (OpToFold.
isReg() &&
MI.isCopy() && !
MI.getOperand(1).getSubReg()) {
2157 if (foldCopyToAGPRRegSequence(&
MI))
2161 FoldableDef
Def(OpToFold, DstRC);
2162 bool Changed = foldInstOperand(
MI, Def);
2169 auto *InstToErase = &
MI;
2171 auto &SrcOp = InstToErase->getOperand(1);
2173 InstToErase->eraseFromParent();
2175 InstToErase =
nullptr;
2179 if (!InstToErase || !
TII->isFoldableCopy(*InstToErase))
2183 if (InstToErase && InstToErase->isRegSequence() &&
2185 InstToErase->eraseFromParent();
2195 return OpToFold.
isReg() &&
2196 foldCopyToVGPROfScalarAddOfFrameIndex(DstReg, OpToFold.
getReg(),
MI);
2201const MachineOperand *
2202SIFoldOperandsImpl::isClamp(
const MachineInstr &
MI)
const {
2203 unsigned Op =
MI.getOpcode();
2205 case AMDGPU::V_MAX_F32_e64:
2206 case AMDGPU::V_MAX_F16_e64:
2207 case AMDGPU::V_MAX_F16_t16_e64:
2208 case AMDGPU::V_MAX_F16_fake16_e64:
2209 case AMDGPU::V_MAX_F64_e64:
2210 case AMDGPU::V_MAX_NUM_F64_e64:
2211 case AMDGPU::V_PK_MAX_F16:
2212 case AMDGPU::V_MAX_BF16_PSEUDO_e64:
2213 case AMDGPU::V_PK_MAX_NUM_BF16: {
2214 if (
MI.mayRaiseFPException())
2217 if (!
TII->getNamedOperand(
MI, AMDGPU::OpName::clamp)->getImm())
2221 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2222 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2226 Src0->
getSubReg() != AMDGPU::NoSubRegister)
2230 if (
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2234 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers)->getImm();
2236 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers)->getImm();
2240 unsigned UnsetMods =
2241 (
Op == AMDGPU::V_PK_MAX_F16 ||
Op == AMDGPU::V_PK_MAX_NUM_BF16)
2244 if (Src0Mods != UnsetMods && Src1Mods != UnsetMods)
2254bool SIFoldOperandsImpl::tryFoldClamp(MachineInstr &
MI) {
2255 const MachineOperand *ClampSrc = isClamp(
MI);
2271 if (
Def->mayRaiseFPException())
2274 MachineOperand *DefClamp =
TII->getNamedOperand(*Def, AMDGPU::OpName::clamp);
2278 LLVM_DEBUG(
dbgs() <<
"Folding clamp " << *DefClamp <<
" into " << *Def);
2284 Register MIDstReg =
MI.getOperand(0).getReg();
2285 if (
TRI->isSGPRReg(*MRI, DefReg)) {
2294 MI.eraseFromParent();
2299 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2300 Def->eraseFromParent();
2307 case AMDGPU::V_MUL_F64_e64:
2308 case AMDGPU::V_MUL_F64_pseudo_e64: {
2310 case 0x3fe0000000000000:
2312 case 0x4000000000000000:
2314 case 0x4010000000000000:
2320 case AMDGPU::V_MUL_F32_e64: {
2321 switch (
static_cast<uint32_t>(Val)) {
2332 case AMDGPU::V_MUL_F16_e64:
2333 case AMDGPU::V_MUL_F16_t16_e64:
2334 case AMDGPU::V_MUL_F16_fake16_e64: {
2335 switch (
static_cast<uint16_t>(Val)) {
2354std::pair<const MachineOperand *, int>
2355SIFoldOperandsImpl::isOMod(
const MachineInstr &
MI)
const {
2356 unsigned Op =
MI.getOpcode();
2358 case AMDGPU::V_MUL_F64_e64:
2359 case AMDGPU::V_MUL_F64_pseudo_e64:
2360 case AMDGPU::V_MUL_F32_e64:
2361 case AMDGPU::V_MUL_F16_t16_e64:
2362 case AMDGPU::V_MUL_F16_fake16_e64:
2363 case AMDGPU::V_MUL_F16_e64: {
2365 if ((
Op == AMDGPU::V_MUL_F32_e64 &&
2367 ((
Op == AMDGPU::V_MUL_F64_e64 ||
Op == AMDGPU::V_MUL_F64_pseudo_e64 ||
2368 Op == AMDGPU::V_MUL_F16_e64 ||
Op == AMDGPU::V_MUL_F16_t16_e64 ||
2369 Op == AMDGPU::V_MUL_F16_fake16_e64) &&
2372 MI.mayRaiseFPException())
2375 const MachineOperand *RegOp =
nullptr;
2376 const MachineOperand *ImmOp =
nullptr;
2377 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2378 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2379 if (Src0->
isImm()) {
2382 }
else if (Src1->
isImm()) {
2390 TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) ||
2391 TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) ||
2392 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2393 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2396 return std::pair(RegOp, OMod);
2398 case AMDGPU::V_ADD_F64_e64:
2399 case AMDGPU::V_ADD_F64_pseudo_e64:
2400 case AMDGPU::V_ADD_F32_e64:
2401 case AMDGPU::V_ADD_F16_e64:
2402 case AMDGPU::V_ADD_F16_t16_e64:
2403 case AMDGPU::V_ADD_F16_fake16_e64: {
2405 if ((
Op == AMDGPU::V_ADD_F32_e64 &&
2407 ((
Op == AMDGPU::V_ADD_F64_e64 ||
Op == AMDGPU::V_ADD_F64_pseudo_e64 ||
2408 Op == AMDGPU::V_ADD_F16_e64 ||
Op == AMDGPU::V_ADD_F16_t16_e64 ||
2409 Op == AMDGPU::V_ADD_F16_fake16_e64) &&
2414 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2415 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2419 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) &&
2420 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) &&
2421 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp) &&
2422 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2433bool SIFoldOperandsImpl::tryFoldOMod(MachineInstr &
MI) {
2434 const MachineOperand *RegOp;
2436 std::tie(RegOp, OMod) = isOMod(
MI);
2438 RegOp->
getSubReg() != AMDGPU::NoSubRegister ||
2443 MachineOperand *DefOMod =
TII->getNamedOperand(*Def, AMDGPU::OpName::omod);
2447 if (
Def->mayRaiseFPException())
2452 if (
TII->hasModifiersSet(*Def, AMDGPU::OpName::clamp))
2462 MI.eraseFromParent();
2467 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2468 Def->eraseFromParent();
2475bool SIFoldOperandsImpl::tryFoldRegSequence(MachineInstr &
MI) {
2477 auto Reg =
MI.getOperand(0).getReg();
2479 if (!ST->hasGFX90AInsts() || !
TRI->isVGPR(*MRI,
Reg) ||
2484 if (!getRegSeqInit(Defs,
Reg))
2487 for (
auto &[
Op, SubIdx] : Defs) {
2490 if (
TRI->isAGPR(*MRI,
Op->getReg()))
2493 const MachineInstr *SubDef = MRI->
getVRegDef(
Op->getReg());
2501 MachineInstr *
UseMI =
Op->getParent();
2510 if (
Op->getSubReg())
2516 if (!OpRC || !
TRI->isVectorSuperClass(OpRC))
2522 TII->get(AMDGPU::REG_SEQUENCE), Dst);
2524 for (
auto &[Def, SubIdx] : Defs) {
2525 Def->setIsKill(
false);
2526 if (
TRI->isAGPR(*MRI,
Def->getReg())) {
2537 if (!
TII->isOperandLegal(*
UseMI, OpIdx,
Op)) {
2539 RS->eraseFromParent();
2548 MI.eraseFromParent();
2556 Register &OutReg,
unsigned &OutSubReg) {
2566 if (
TRI.isAGPR(MRI, CopySrcReg)) {
2567 OutReg = CopySrcReg;
2576 if (!CopySrcDef || !CopySrcDef->
isCopy())
2583 OtherCopySrc.
getSubReg() != AMDGPU::NoSubRegister ||
2584 !
TRI.isAGPR(MRI, OtherCopySrcReg))
2587 OutReg = OtherCopySrcReg;
2621bool SIFoldOperandsImpl::tryFoldPhiAGPR(MachineInstr &
PHI) {
2625 if (!
TRI->isVGPR(*MRI, PhiOut))
2631 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2632 MachineOperand &MO =
PHI.getOperand(K);
2634 if (!Copy || !
Copy->isCopy())
2638 unsigned AGPRRegMask = AMDGPU::NoSubRegister;
2643 if (
const auto *SubRC =
TRI->getSubRegisterClass(CopyInRC, AGPRRegMask))
2654 bool IsAGPR32 = (ARC == &AMDGPU::AGPR_32RegClass);
2658 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2659 MachineOperand &MO =
PHI.getOperand(K);
2663 MachineBasicBlock *InsertMBB =
nullptr;
2666 unsigned CopyOpc = AMDGPU::COPY;
2671 if (
Def->isCopy()) {
2673 unsigned AGPRSubReg = AMDGPU::NoSubRegister;
2686 MachineOperand &CopyIn =
Def->getOperand(1);
2689 CopyOpc = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
2692 InsertMBB =
Def->getParent();
2700 MachineInstr *
MI =
BuildMI(*InsertMBB, InsertPt,
PHI.getDebugLoc(),
2701 TII->get(CopyOpc), NewReg)
2711 PHI.getOperand(0).setReg(NewReg);
2717 TII->get(AMDGPU::COPY), PhiOut)
2725bool SIFoldOperandsImpl::tryFoldLoad(MachineInstr &
MI) {
2727 if (!ST->hasGFX90AInsts() ||
MI.getNumExplicitDefs() != 1)
2730 MachineOperand &
Def =
MI.getOperand(0);
2747 while (!
Users.empty()) {
2748 const MachineInstr *
I =
Users.pop_back_val();
2749 if (!
I->isCopy() && !
I->isRegSequence())
2751 Register DstReg =
I->getOperand(0).getReg();
2755 if (
TRI->isAGPR(*MRI, DstReg))
2759 Users.push_back(&U);
2764 if (!
TII->isOperandLegal(
MI, 0, &Def)) {
2769 while (!MoveRegs.
empty()) {
2811bool SIFoldOperandsImpl::tryOptimizeAGPRPhis(MachineBasicBlock &
MBB) {
2814 if (ST->hasGFX90AInsts())
2818 DenseMap<std::pair<Register, unsigned>, std::vector<MachineOperand *>>
2821 for (
auto &
MI :
MBB) {
2825 if (!
TRI->isAGPR(*MRI,
MI.getOperand(0).getReg()))
2828 for (
unsigned K = 1;
K <
MI.getNumOperands();
K += 2) {
2829 MachineOperand &PhiMO =
MI.getOperand(K);
2839 for (
const auto &[Entry, MOs] : RegToMO) {
2840 if (MOs.size() == 1)
2845 MachineBasicBlock *DefMBB =
Def->getParent();
2852 MachineInstr *VGPRCopy =
2854 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64), TempVGPR)
2860 TII->get(AMDGPU::COPY), TempAGPR)
2864 for (MachineOperand *MO : MOs) {
2876bool SIFoldOperandsImpl::run(
MachineFunction &MF,
const MachineLoopInfo *MLI) {
2882 MFI = MF.
getInfo<SIMachineFunctionInfo>();
2893 MachineOperand *CurrentKnownM0Val =
nullptr;
2901 if (tryConstantFoldOp(&
MI)) {
2906 if (tryFoldZeroHighBits(
MI)) {
2911 if (
MI.isRegSequence() && tryFoldRegSequence(
MI)) {
2916 if (
MI.isPHI() && tryFoldPhiAGPR(
MI)) {
2921 if (
MI.mayLoad() && tryFoldLoad(
MI)) {
2926 if (
TII->isFoldableCopy(
MI)) {
2927 Changed |= tryFoldFoldableCopy(
MI, CurrentKnownM0Val);
2932 if (CurrentKnownM0Val &&
MI.modifiesRegister(AMDGPU::M0,
TRI))
2933 CurrentKnownM0Val =
nullptr;
2953 bool Changed = SIFoldOperandsImpl().run(MF, MLI);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
Provides AMDGPU specific target descriptions.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool updateOperand(Instruction *Inst, unsigned Idx, Instruction *Mat)
Updates the operand at Idx in instruction Inst with the result of instruction Mat.
This file builds on the ADT/GraphTraits.h file to build generic depth first graph iterator.
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI)
static unsigned macToMad(unsigned Opc)
static bool isAGPRCopy(const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI, const MachineInstr &Copy, Register &OutReg, unsigned &OutSubReg)
Checks whether Copy is a AGPR -> VGPR copy.
static void appendFoldCandidate(SmallVectorImpl< FoldCandidate > &FoldList, FoldCandidate &&Entry)
static const TargetRegisterClass * getRegOpRC(const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const MachineOperand &MO)
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result, uint32_t LHS, uint32_t RHS)
static int getOModValue(unsigned Opc, int64_t Val)
static unsigned getMovOpc(bool IsScalar)
static MachineOperand * lookUpCopyChain(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, Register SrcReg)
static bool checkImmOpForPKF32InstrReplicatesLower32BitsOfScalarOperand(const FoldableDef &OpToFold)
static bool isPKF32InstrReplicatesLower32BitsOfScalarOperand(const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo)
Interface definition for SIInstrInfo.
Interface definition for SIRegisterInfo.
static int Lookup(ArrayRef< TableEntry > Table, unsigned Opcode)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
FunctionPass class - This class is used to implement most global optimizations.
const SIInstrInfo * getInstrInfo() const override
bool hasDOTOpSelHazard() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
const HexagonRegisterInfo & getRegisterInfo() const
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
ArrayRef< MCOperandInfo > operands() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool isVariadic() const
Return true if this instruction can have a variable number of operands.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI iterator SkipPHIsLabelsAndDebug(iterator I, Register Reg=Register(), bool SkipPseudoOp=true)
Return the first instruction in MBB after I that is not a PHI, label or debug.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
unsigned getOperandNo(const_mop_iterator I) const
Returns the number of the operand iterator I points to.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
void clearFlag(MIFlag Flag)
clearFlag - Clear a MI flag.
bool isRegSequence() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
LLVM_ABI void substVirtReg(Register Reg, unsigned SubIdx, const TargetRegisterInfo &)
substVirtReg - Substitute the current register with the virtual subregister Reg:SubReg.
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
LLVM_ABI void substPhysReg(MCRegister Reg, const TargetRegisterInfo &)
substPhysReg - Substitute the current register with the physical register Reg, taking any existing Su...
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_GlobalAddress
Address of a global value.
@ MO_FrameIndex
Abstract Stack Frame Index.
@ MO_Register
Register operand.
static MachineOperand CreateFI(int Idx)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
use_nodbg_iterator use_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool hasOneNonDBGUser(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug instruction using the specified regis...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
SIModeRegisterDefaults getMode() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
Represent a constant reference to a string, i.e.
static const unsigned CommuteAnyOperandIndex
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
constexpr bool isVOP3(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isPacked(const T &...O)
NodeAddr< DefNode * > Def
This is an optimization pass for GlobalISel generic memory operations.
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
MachineBasicBlock::instr_iterator getBundleStart(MachineBasicBlock::instr_iterator I)
Returns an iterator to the first instruction in the bundle containing I.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createSIFoldOperandsLegacyPass()
char & SIFoldOperandsLegacyID
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
iterator_range< df_iterator< T > > depth_first(const T &G)
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.