31#define DEBUG_TYPE "si-peephole-sdwa"
33STATISTIC(NumSDWAPatternsFound,
"Number of SDWA patterns found.");
35 "Number of instruction converted to SDWA.");
54 SDWAOperandsMap PotentialMatches;
61 std::optional<std::pair<MachineOperand *, AMDGPU::SDWA::SdwaSel>>
73 bool convertToSDWA(
MachineInstr &
MI,
const SDWAOperandsVector &SDWAOperands);
85 SIPeepholeSDWALegacy() : MachineFunctionPass(ID) {}
87 StringRef getPassName()
const override {
return "SI Peephole SDWA"; }
91 void getAnalysisUsage(AnalysisUsage &AU)
const override {
101 MachineOperand *Target;
102 MachineOperand *Replaced;
106 virtual bool canCombineSelections(
const MachineInstr &
MI,
107 const SIInstrInfo *
TII) = 0;
110 SDWAOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp)
111 : Target(TargetOp), Replaced(ReplacedOp) {
113 assert(Replaced->isReg());
116 virtual ~SDWAOperand() =
default;
118 virtual MachineInstr *potentialToConvert(
const SIInstrInfo *
TII,
119 const GCNSubtarget &ST,
120 SDWAOperandsMap *PotentialMatches =
nullptr) = 0;
121 virtual bool convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII) = 0;
123 MachineOperand *getTargetOperand()
const {
return Target; }
124 MachineOperand *getReplacedOperand()
const {
return Replaced; }
125 MachineInstr *getParentInst()
const {
return Target->getParent(); }
127 MachineRegisterInfo *getMRI()
const {
128 return &getParentInst()->getMF()->getRegInfo();
131#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
132 virtual void print(raw_ostream& OS)
const = 0;
137class SDWASrcOperand :
public SDWAOperand {
145 SDWASrcOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
146 SdwaSel SrcSel_ =
DWORD,
bool Abs_ =
false,
bool Neg_ =
false,
148 : SDWAOperand(TargetOp, ReplacedOp), SrcSel(SrcSel_), Abs(Abs_),
149 Neg(Neg_), Sext(Sext_) {}
151 MachineInstr *potentialToConvert(
const SIInstrInfo *
TII,
152 const GCNSubtarget &ST,
153 SDWAOperandsMap *PotentialMatches =
nullptr)
override;
154 bool convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII)
override;
155 bool canCombineSelections(
const MachineInstr &
MI,
156 const SIInstrInfo *
TII)
override;
158 SdwaSel getSrcSel()
const {
return SrcSel; }
159 bool getAbs()
const {
return Abs; }
160 bool getNeg()
const {
return Neg; }
161 bool getSext()
const {
return Sext; }
164 const MachineOperand *SrcOp)
const;
166#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
167 void print(raw_ostream& OS)
const override;
171class SDWADstOperand :
public SDWAOperand {
177 SDWADstOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
179 : SDWAOperand(TargetOp, ReplacedOp), DstSel(DstSel_), DstUn(DstUn_) {}
181 MachineInstr *potentialToConvert(
const SIInstrInfo *
TII,
182 const GCNSubtarget &ST,
183 SDWAOperandsMap *PotentialMatches =
nullptr)
override;
184 bool convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII)
override;
185 bool canCombineSelections(
const MachineInstr &
MI,
186 const SIInstrInfo *
TII)
override;
188 SdwaSel getDstSel()
const {
return DstSel; }
189 DstUnused getDstUnused()
const {
return DstUn; }
191#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
192 void print(raw_ostream& OS)
const override;
196class SDWADstPreserveOperand :
public SDWADstOperand {
198 MachineOperand *Preserve;
201 SDWADstPreserveOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
204 Preserve(PreserveOp) {}
206 bool convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII)
override;
207 bool canCombineSelections(
const MachineInstr &
MI,
208 const SIInstrInfo *
TII)
override;
210 MachineOperand *getPreservedOperand()
const {
return Preserve; }
212#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
213 void print(raw_ostream& OS)
const override;
222char SIPeepholeSDWALegacy::ID = 0;
226#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
229 case BYTE_0: OS <<
"BYTE_0";
break;
230 case BYTE_1: OS <<
"BYTE_1";
break;
231 case BYTE_2: OS <<
"BYTE_2";
break;
232 case BYTE_3: OS <<
"BYTE_3";
break;
233 case WORD_0: OS <<
"WORD_0";
break;
234 case WORD_1: OS <<
"WORD_1";
break;
235 case DWORD: OS <<
"DWORD";
break;
251 OS <<
"SDWA src: " << *getTargetOperand()
252 <<
" src_sel:" << getSrcSel()
253 <<
" abs:" << getAbs() <<
" neg:" << getNeg()
254 <<
" sext:" << getSext() <<
'\n';
258void SDWADstOperand::print(raw_ostream& OS)
const {
259 OS <<
"SDWA dst: " << *getTargetOperand()
260 <<
" dst_sel:" << getDstSel()
261 <<
" dst_unused:" << getDstUnused() <<
'\n';
265void SDWADstPreserveOperand::print(raw_ostream& OS)
const {
266 OS <<
"SDWA preserve dst: " << *getTargetOperand()
267 <<
" dst_sel:" << getDstSel()
268 <<
" preserve:" << *getPreservedOperand() <<
'\n';
286 return LHS.isReg() &&
288 LHS.getReg() ==
RHS.getReg() &&
289 LHS.getSubReg() ==
RHS.getSubReg();
294 if (!
Reg->isReg() || !
Reg->isDef())
315 if (Sel == SdwaSel::DWORD)
318 if (Sel == OperandSel || OperandSel == SdwaSel::DWORD)
321 if (Sel == SdwaSel::WORD_1 || Sel == SdwaSel::BYTE_2 ||
322 Sel == SdwaSel::BYTE_3)
325 if (OperandSel == SdwaSel::WORD_0)
328 if (OperandSel == SdwaSel::WORD_1) {
329 if (Sel == SdwaSel::BYTE_0)
330 return SdwaSel::BYTE_2;
331 if (Sel == SdwaSel::BYTE_1)
332 return SdwaSel::BYTE_3;
333 if (Sel == SdwaSel::WORD_0)
334 return SdwaSel::WORD_1;
340uint64_t SDWASrcOperand::getSrcMods(
const SIInstrInfo *
TII,
341 const MachineOperand *SrcOp)
const {
344 if (
TII->getNamedOperand(*
MI, AMDGPU::OpName::src0) == SrcOp) {
345 if (
auto *
Mod =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src0_modifiers)) {
346 Mods =
Mod->getImm();
348 }
else if (
TII->getNamedOperand(*
MI, AMDGPU::OpName::src1) == SrcOp) {
349 if (
auto *
Mod =
TII->getNamedOperand(*
MI, AMDGPU::OpName::src1_modifiers)) {
350 Mods =
Mod->getImm();
355 "Float and integer src modifiers can't be set simultaneously");
365MachineInstr *SDWASrcOperand::potentialToConvert(
const SIInstrInfo *
TII,
366 const GCNSubtarget &ST,
367 SDWAOperandsMap *PotentialMatches) {
368 if (PotentialMatches !=
nullptr) {
370 MachineOperand *
Reg = getReplacedOperand();
371 if (!
Reg->isReg() || !
Reg->isDef())
374 for (MachineInstr &
UseMI : getMRI()->use_nodbg_instructions(
Reg->getReg()))
376 if (!isConvertibleToSDWA(
UseMI, ST,
TII) ||
382 for (MachineOperand &UseMO : getMRI()->use_nodbg_operands(
Reg->getReg())) {
386 SDWAOperandsMap &potentialMatchesMap = *PotentialMatches;
387 MachineInstr *
UseMI = UseMO.getParent();
388 potentialMatchesMap[
UseMI].push_back(
this);
395 MachineOperand *PotentialMO =
findSingleRegUse(getReplacedOperand(), getMRI());
399 MachineInstr *Parent = PotentialMO->
getParent();
401 return canCombineSelections(*Parent,
TII) ? Parent :
nullptr;
404bool SDWASrcOperand::convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII) {
405 assert((!Sext || !
TII->getSubtarget().zeroesHigh16BitsOfDest(
407 "Cannot use sign-extension with instruction that zeroes high bits");
408 switch (
MI.getOpcode()) {
409 case AMDGPU::V_CVT_F32_FP8_sdwa:
410 case AMDGPU::V_CVT_F32_BF8_sdwa:
411 case AMDGPU::V_CVT_PK_F32_FP8_sdwa:
412 case AMDGPU::V_CVT_PK_F32_BF8_sdwa:
415 case AMDGPU::V_CNDMASK_B32_sdwa:
434 bool IsPreserveSrc =
false;
435 MachineOperand *Src =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
436 MachineOperand *SrcSel =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_sel);
437 MachineOperand *SrcMods =
438 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
439 assert(Src && (Src->isReg() || Src->isImm()));
440 if (!
isSameReg(*Src, *getReplacedOperand())) {
442 Src =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
443 SrcSel =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_sel);
444 SrcMods =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
447 !
isSameReg(*Src, *getReplacedOperand())) {
454 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
456 TII->getNamedOperand(
MI, AMDGPU::OpName::dst_unused);
459 DstUnused->getImm() == AMDGPU::SDWA::DstUnused::UNUSED_PRESERVE) {
465 TII->getNamedImmOperand(
MI, AMDGPU::OpName::dst_sel));
466 if (DstSel == AMDGPU::SDWA::SdwaSel::WORD_1 &&
467 getSrcSel() == AMDGPU::SDWA::SdwaSel::WORD_0) {
468 IsPreserveSrc =
true;
469 auto DstIdx = AMDGPU::getNamedOperandIdx(
MI.getOpcode(),
470 AMDGPU::OpName::vdst);
471 auto TiedIdx =
MI.findTiedOperandIdx(DstIdx);
472 Src = &
MI.getOperand(TiedIdx);
481 assert(Src && Src->isReg());
483 if ((
MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
484 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
485 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
486 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
487 !
isSameReg(*Src, *getReplacedOperand())) {
494 (IsPreserveSrc || (SrcSel && SrcMods)));
497 if (!IsPreserveSrc) {
502 getTargetOperand()->setIsKill(
false);
509 AMDGPU::OpName SrcSelOpName,
SdwaSel OpSel) {
522 AMDGPU::OpName SrcOpName,
534bool SDWASrcOperand::canCombineSelections(
const MachineInstr &
MI,
535 const SIInstrInfo *
TII) {
536 if (!
TII->isSDWA(
MI.getOpcode()))
539 using namespace AMDGPU;
542 getReplacedOperand(), getSrcSel()) &&
544 getReplacedOperand(), getSrcSel());
547MachineInstr *SDWADstOperand::potentialToConvert(
const SIInstrInfo *
TII,
548 const GCNSubtarget &ST,
549 SDWAOperandsMap *PotentialMatches) {
552 MachineRegisterInfo *MRI = getMRI();
553 MachineInstr *ParentMI = getParentInst();
561 if (&UseInst != ParentMI)
565 MachineInstr *Parent = PotentialMO->
getParent();
566 return canCombineSelections(*Parent,
TII) ? Parent :
nullptr;
569bool SDWADstOperand::convertToSDWA(MachineInstr &
MI,
const SIInstrInfo *
TII) {
572 if ((
MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
573 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
574 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
575 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
581 MachineOperand *Operand =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
584 isSameReg(*Operand, *getReplacedOperand()));
586 MachineOperand *DstSel=
TII->getNamedOperand(
MI, AMDGPU::OpName::dst_sel);
592 MachineOperand *
DstUnused=
TII->getNamedOperand(
MI, AMDGPU::OpName::dst_unused);
598 getParentInst()->eraseFromParent();
602bool SDWADstOperand::canCombineSelections(
const MachineInstr &
MI,
603 const SIInstrInfo *
TII) {
604 if (!
TII->isSDWA(
MI.getOpcode()))
610bool SDWADstPreserveOperand::convertToSDWA(MachineInstr &
MI,
611 const SIInstrInfo *
TII) {
615 for (MachineOperand &MO :
MI.uses()) {
618 getMRI()->clearKillFlags(MO.getReg());
622 MI.getParent()->remove(&
MI);
623 getParentInst()->getParent()->insert(getParentInst(), &
MI);
626 MachineInstrBuilder MIB(*
MI.getMF(),
MI);
627 MIB.addReg(getPreservedOperand()->
getReg(),
628 RegState::ImplicitKill,
629 getPreservedOperand()->getSubReg());
632 MI.tieOperands(AMDGPU::getNamedOperandIdx(
MI.getOpcode(), AMDGPU::OpName::vdst),
633 MI.getNumOperands() - 1);
636 return SDWADstOperand::convertToSDWA(
MI,
TII);
639bool SDWADstPreserveOperand::canCombineSelections(
const MachineInstr &
MI,
640 const SIInstrInfo *
TII) {
641 return SDWADstOperand::canCombineSelections(
MI,
TII);
644std::optional<int64_t>
645SIPeepholeSDWA::foldToImm(
const MachineOperand &
Op)
const {
653 for (
const MachineOperand &Def : MRI->
def_operands(
Op.getReg())) {
657 const MachineInstr *DefInst =
Def.getParent();
658 if (!
TII->isFoldableCopy(*DefInst))
661 const MachineOperand &Copied = DefInst->
getOperand(1);
672std::optional<std::pair<MachineOperand *, SdwaSel>>
673SIPeepholeSDWA::matchAndMask(MachineInstr &
MI)
const {
674 if (
MI.getOpcode() != AMDGPU::V_AND_B32_e32 &&
675 MI.getOpcode() != AMDGPU::V_AND_B32_e64)
678 MachineOperand *
Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
679 MachineOperand *
Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
680 MachineOperand *ValSrc =
Src1;
681 std::optional<int64_t>
Imm = foldToImm(*Src0);
683 Imm = foldToImm(*Src1);
686 if (!
Imm || (*
Imm != 0x0000ffff && *
Imm != 0x000000ff))
692bool SIPeepholeSDWA::isSDWAWithDstSel(
const MachineInstr &Inst)
const {
693 return TII->isSDWA(Inst) &&
697std::unique_ptr<SDWAOperand>
698SIPeepholeSDWA::matchSDWAOperand(MachineInstr &
MI) {
699 unsigned Opcode =
MI.getOpcode();
701 case AMDGPU::V_LSHRREV_B32_e32:
702 case AMDGPU::V_ASHRREV_I32_e32:
703 case AMDGPU::V_LSHLREV_B32_e32:
704 case AMDGPU::V_LSHRREV_B32_e64:
705 case AMDGPU::V_ASHRREV_I32_e64:
706 case AMDGPU::V_LSHLREV_B32_e64: {
715 MachineOperand *
Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
716 auto Imm = foldToImm(*Src0);
720 if (*
Imm != 16 && *
Imm != 24)
723 MachineOperand *
Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
724 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
725 if (!
Src1->isReg() ||
Src1->getReg().isPhysical() ||
726 Dst->getReg().isPhysical())
729 if (Opcode == AMDGPU::V_LSHLREV_B32_e32 ||
730 Opcode == AMDGPU::V_LSHLREV_B32_e64) {
731 return std::make_unique<SDWADstOperand>(
734 return std::make_unique<SDWASrcOperand>(
736 Opcode != AMDGPU::V_LSHRREV_B32_e32 &&
737 Opcode != AMDGPU::V_LSHRREV_B32_e64);
741 case AMDGPU::V_LSHRREV_B16_e32:
742 case AMDGPU::V_LSHLREV_B16_e32:
743 case AMDGPU::V_LSHRREV_B16_e64:
744 case AMDGPU::V_LSHRREV_B16_opsel_e64:
745 case AMDGPU::V_LSHLREV_B16_opsel_e64:
746 case AMDGPU::V_LSHLREV_B16_e64: {
755 MachineOperand *
Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
756 auto Imm = foldToImm(*Src0);
760 MachineOperand *
Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
761 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
763 if (!
Src1->isReg() ||
Src1->getReg().isPhysical() ||
764 Dst->getReg().isPhysical())
767 if (Opcode == AMDGPU::V_LSHLREV_B16_e32 ||
768 Opcode == AMDGPU::V_LSHLREV_B16_opsel_e64 ||
769 Opcode == AMDGPU::V_LSHLREV_B16_e64)
771 return std::make_unique<SDWASrcOperand>(Src1, Dst,
BYTE_1,
false,
false,
776 case AMDGPU::V_BFE_I32_e64:
777 case AMDGPU::V_BFE_U32_e64: {
792 MachineOperand *
Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
793 auto Offset = foldToImm(*Src1);
797 MachineOperand *
Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
798 auto Width = foldToImm(*Src2);
804 if (*
Offset == 0 && *Width == 8)
806 else if (*
Offset == 0 && *Width == 16)
808 else if (*
Offset == 0 && *Width == 32)
810 else if (*
Offset == 8 && *Width == 8)
812 else if (*
Offset == 16 && *Width == 8)
814 else if (*
Offset == 16 && *Width == 16)
816 else if (*
Offset == 24 && *Width == 8)
821 MachineOperand *
Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
822 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
824 if (!
Src0->isReg() ||
Src0->getReg().isPhysical() ||
825 Dst->getReg().isPhysical())
828 return std::make_unique<SDWASrcOperand>(
829 Src0, Dst, SrcSel,
false,
false, Opcode != AMDGPU::V_BFE_U32_e64);
832 case AMDGPU::V_AND_B32_e32:
833 case AMDGPU::V_AND_B32_e64: {
837 auto Mask = matchAndMask(
MI);
840 MachineOperand *ValSrc =
Mask->first;
842 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
845 Dst->getReg().isPhysical())
848 return std::make_unique<SDWASrcOperand>(ValSrc, Dst,
Mask->second);
851 case AMDGPU::V_OR_B32_e32:
852 case AMDGPU::V_OR_B32_e64: {
863 std::optional<std::pair<MachineOperand *, MachineOperand *>>;
864 auto CheckOROperandsForSDWA =
865 [&](
const MachineOperand *Op1,
const MachineOperand *Op2) -> CheckRetType {
866 if (!Op1 || !Op1->
isReg() || !Op2 || !Op2->isReg())
867 return CheckRetType(std::nullopt);
871 return CheckRetType(std::nullopt);
873 MachineInstr *Op1Inst = Op1Def->
getParent();
874 if (!isSDWAWithDstSel(*Op1Inst))
875 return CheckRetType(std::nullopt);
879 return CheckRetType(std::nullopt);
881 return CheckRetType(std::pair(Op1Def, Op2Def));
884 MachineOperand *OrSDWA =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
885 MachineOperand *OrOther =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
886 assert(OrSDWA && OrOther);
887 auto Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
889 OrSDWA =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
890 OrOther =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
891 assert(OrSDWA && OrOther);
892 Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
897 MachineOperand *OrSDWADef = Res->first;
898 MachineOperand *OrOtherDef = Res->second;
899 assert(OrSDWADef && OrOtherDef);
901 MachineInstr *SDWAInst = OrSDWADef->
getParent();
902 MachineInstr *OtherInst = OrOtherDef->
getParent();
924 if (!isSDWAWithDstSel(*OtherInst))
928 TII->getNamedImmOperand(*SDWAInst, AMDGPU::OpName::dst_sel));
930 TII->getNamedImmOperand(*OtherInst, AMDGPU::OpName::dst_sel));
932 bool DstSelAgree =
false;
935 (OtherDstSel ==
BYTE_3) ||
939 (OtherDstSel ==
BYTE_1) ||
943 (OtherDstSel ==
BYTE_2) ||
944 (OtherDstSel ==
BYTE_3) ||
948 (OtherDstSel ==
BYTE_2) ||
949 (OtherDstSel ==
BYTE_3) ||
953 (OtherDstSel ==
BYTE_1) ||
954 (OtherDstSel ==
BYTE_3) ||
958 (OtherDstSel ==
BYTE_1) ||
959 (OtherDstSel ==
BYTE_2) ||
962 default: DstSelAgree =
false;
970 TII->getNamedImmOperand(*OtherInst, AMDGPU::OpName::dst_unused));
971 if (OtherDstUnused != DstUnused::UNUSED_PAD)
975 MachineOperand *OrDst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
978 return std::make_unique<SDWADstPreserveOperand>(
979 OrDst, OrSDWADef, OrOtherDef, DstSel);
984 return std::unique_ptr<SDWAOperand>(
nullptr);
994void SIPeepholeSDWA::matchSDWAOperands(MachineBasicBlock &
MBB) {
995 for (MachineInstr &
MI :
MBB) {
996 if (
auto Operand = matchSDWAOperand(
MI)) {
998 SDWAOperands[&
MI] = std::move(Operand);
999 ++NumSDWAPatternsFound;
1022void SIPeepholeSDWA::pseudoOpConvertToVOP2(MachineInstr &
MI,
1023 const GCNSubtarget &ST)
const {
1024 int Opc =
MI.getOpcode();
1025 assert((
Opc == AMDGPU::V_ADD_CO_U32_e64 ||
Opc == AMDGPU::V_SUB_CO_U32_e64) &&
1026 "Currently only handles V_ADD_CO_U32_e64 or V_SUB_CO_U32_e64");
1029 if (!
TII->canShrink(
MI, *MRI))
1033 const MachineOperand *Sdst =
TII->getNamedOperand(
MI, AMDGPU::OpName::sdst);
1039 MachineInstr &MISucc = *NextOp->
getParent();
1042 MachineOperand *CarryIn =
TII->getNamedOperand(MISucc, AMDGPU::OpName::src2);
1045 MachineOperand *CarryOut =
TII->getNamedOperand(MISucc, AMDGPU::OpName::sdst);
1062 if (
I->modifiesRegister(AMDGPU::VCC,
TRI))
1068 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst))
1069 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::src0))
1070 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::src1))
1073 MI.eraseFromParent();
1085void SIPeepholeSDWA::convertVcndmaskToVOP2(MachineInstr &
MI,
1086 const GCNSubtarget &ST)
const {
1087 assert(
MI.getOpcode() == AMDGPU::V_CNDMASK_B32_e64);
1090 if (!
TII->canShrink(
MI, *MRI)) {
1095 const MachineOperand &CarryIn =
1096 *
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
1098 MachineInstr *CarryDef = MRI->
getVRegDef(CarryReg);
1105 MCRegister
Vcc =
TRI->getVCC();
1110 LLVM_DEBUG(
dbgs() <<
"VCC not known to be dead before instruction\n");
1118 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst))
1119 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::src0))
1120 .
add(*
TII->getNamedOperand(
MI, AMDGPU::OpName::src1))
1122 TII->fixImplicitOperands(*Converted);
1125 MI.eraseFromParent();
1129bool isConvertibleToSDWA(MachineInstr &
MI,
1130 const GCNSubtarget &ST,
1131 const SIInstrInfo*
TII) {
1133 unsigned Opc =
MI.getOpcode();
1139 if (
Opc == AMDGPU::V_CNDMASK_B32_e64)
1149 if (!
ST.hasSDWAOmod() &&
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
1153 if (!
ST.hasSDWASdst()) {
1154 const MachineOperand *SDst =
TII->getNamedOperand(
MI, AMDGPU::OpName::sdst);
1155 if (SDst && (SDst->
getReg() != AMDGPU::VCC &&
1156 SDst->
getReg() != AMDGPU::VCC_LO))
1160 if (!
ST.hasSDWAOutModsVOPC() &&
1161 (
TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp) ||
1162 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod)))
1165 }
else if (
TII->getNamedOperand(
MI, AMDGPU::OpName::sdst) ||
1166 !
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst)) {
1170 if (!
ST.hasSDWAMac() && (
Opc == AMDGPU::V_FMAC_F16_e32 ||
1171 Opc == AMDGPU::V_FMAC_F32_e32 ||
1172 Opc == AMDGPU::V_MAC_F16_e32 ||
1173 Opc == AMDGPU::V_MAC_F32_e32))
1177 if (
TII->pseudoToMCOpcode(
Opc) == -1 ||
1181 if (MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0)) {
1182 if (!
Src0->isReg() && !
Src0->isImm())
1186 if (MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1)) {
1187 if (!
Src1->isReg() && !
Src1->isImm())
1195MachineInstr *SIPeepholeSDWA::createSDWAVersion(MachineInstr &
MI) {
1196 unsigned Opcode =
MI.getOpcode();
1200 if (SDWAOpcode == -1)
1202 assert(SDWAOpcode != -1);
1204 const MCInstrDesc &SDWADesc =
TII->get(SDWAOpcode);
1207 MachineInstrBuilder SDWAInst =
1212 MachineOperand *Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::vdst);
1216 }
else if ((Dst =
TII->getNamedOperand(
MI, AMDGPU::OpName::sdst))) {
1221 SDWAInst.
addReg(
TRI->getVCC(), RegState::Define);
1226 MachineOperand *
Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1229 if (
auto *
Mod =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers))
1233 SDWAInst.
add(*Src0);
1236 MachineOperand *
Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1240 if (
auto *
Mod =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers))
1244 SDWAInst.
add(*Src1);
1247 if (SDWAOpcode == AMDGPU::V_FMAC_F16_sdwa ||
1248 SDWAOpcode == AMDGPU::V_FMAC_F32_sdwa ||
1249 SDWAOpcode == AMDGPU::V_MAC_F16_sdwa ||
1250 SDWAOpcode == AMDGPU::V_MAC_F32_sdwa) {
1252 MachineOperand *
Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
1254 SDWAInst.
add(*Src2);
1259 MachineOperand *Clamp =
TII->getNamedOperand(
MI, AMDGPU::OpName::clamp);
1261 SDWAInst.
add(*Clamp);
1268 MachineOperand *OMod =
TII->getNamedOperand(
MI, AMDGPU::OpName::omod);
1270 SDWAInst.
add(*OMod);
1278 SDWAInst.
addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1281 SDWAInst.
addImm(AMDGPU::SDWA::DstUnused::UNUSED_PAD);
1284 SDWAInst.
addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1288 SDWAInst.
addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1292 MachineInstr *Ret = SDWAInst.
getInstr();
1293 TII->fixImplicitOperands(*Ret);
1297bool SIPeepholeSDWA::convertToSDWA(MachineInstr &
MI,
1298 const SDWAOperandsVector &SDWAOperands) {
1301 MachineInstr *SDWAInst;
1302 if (
TII->isSDWA(
MI.getOpcode())) {
1306 SDWAInst =
MI.
getMF()->CloneMachineInstr(&
MI);
1307 MI.getParent()->
insert(
MI.getIterator(), SDWAInst);
1309 SDWAInst = createSDWAVersion(
MI);
1313 bool Converted =
false;
1314 for (
auto &Operand : SDWAOperands) {
1326 if (PotentialMatches.count(Operand->getParentInst()) == 0)
1327 Converted |= Operand->convertToSDWA(*SDWAInst,
TII);
1335 ConvertedInstructions.
push_back(SDWAInst);
1336 for (MachineOperand &MO : SDWAInst->
uses()) {
1343 ++NumSDWAInstructionsPeepholed;
1345 MI.eraseFromParent();
1351void SIPeepholeSDWA::legalizeScalarOperands(MachineInstr &
MI,
1352 const GCNSubtarget &ST)
const {
1353 const MCInstrDesc &
Desc =
TII->get(
MI.getOpcode());
1354 unsigned ConstantBusCount = 0;
1355 for (MachineOperand &
Op :
MI.explicit_uses()) {
1357 if (
TRI->isVGPR(*MRI,
Op.getReg()))
1360 if (
ST.hasSDWAScalar() && ConstantBusCount == 0) {
1364 }
else if (!
Op.isImm())
1367 unsigned I =
Op.getOperandNo();
1369 if (!OpRC || !
TRI->isVSSuperClass(OpRC))
1374 TII->get(AMDGPU::V_MOV_B32_e32), VGPR);
1376 Copy.addImm(
Op.getImm());
1377 else if (
Op.isReg())
1379 Op.ChangeToRegister(VGPR,
false);
1385bool SIPeepholeSDWA::splitLshlOrForSDWA(MachineBasicBlock &
MBB) {
1387 MachineInstr *LshlOr;
1388 MachineInstr *AndMI;
1390 MachineOperand *ValSrc;
1394 for (MachineInstr &
MI :
MBB) {
1395 if (
MI.getOpcode() != AMDGPU::V_LSHL_OR_B32_e64)
1398 MachineOperand *Shift =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1399 std::optional<int64_t> ShiftImm = foldToImm(*Shift);
1400 if (!ShiftImm || *ShiftImm != 16)
1403 MachineOperand *
Hi =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1404 MachineOperand *
Src2 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src2);
1406 if (!
Hi->isReg() || !
Src2->isReg() || !
Src2->getReg().isVirtual())
1415 std::optional<std::pair<MachineOperand *, SdwaSel>>
Mask =
1416 matchAndMask(*AndMI);
1419 MachineOperand *ValSrc =
Mask->first;
1426 for (
const Candidate &
C : Candidates) {
1427 MachineOperand *Dst =
TII->getNamedOperand(*
C.LshlOr, AMDGPU::OpName::vdst);
1430 BuildMI(*
C.LshlOr->getParent(), *
C.LshlOr,
C.LshlOr->getDebugLoc(),
1431 TII->get(AMDGPU::V_LSHLREV_B32_e64), ShiftReg)
1437 BuildMI(*
C.LshlOr->getParent(), *
C.LshlOr,
C.LshlOr->getDebugLoc(),
1438 TII->get(AMDGPU::V_OR_B32_sdwa))
1451 C.LshlOr->eraseFromParent();
1452 C.AndMI->eraseFromParent();
1455 return !Candidates.empty();
1462 return SIPeepholeSDWA().run(MF);
1472 TRI =
ST.getRegisterInfo();
1473 TII =
ST.getInstrInfo();
1477 for (MachineBasicBlock &
MBB : MF) {
1480 Ret |= splitLshlOrForSDWA(
MBB);
1486 matchSDWAOperands(
MBB);
1487 for (
const auto &OperandPair : SDWAOperands) {
1488 const auto &Operand = OperandPair.second;
1489 MachineInstr *PotentialMI = Operand->potentialToConvert(
TII, ST);
1494 case AMDGPU::V_ADD_CO_U32_e64:
1495 case AMDGPU::V_SUB_CO_U32_e64:
1496 pseudoOpConvertToVOP2(*PotentialMI, ST);
1498 case AMDGPU::V_CNDMASK_B32_e64:
1499 convertVcndmaskToVOP2(*PotentialMI, ST);
1503 SDWAOperands.clear();
1506 matchSDWAOperands(
MBB);
1508 for (
const auto &OperandPair : SDWAOperands) {
1509 const auto &Operand = OperandPair.second;
1510 MachineInstr *PotentialMI =
1511 Operand->potentialToConvert(
TII, ST, &PotentialMatches);
1513 if (PotentialMI && isConvertibleToSDWA(*PotentialMI, ST,
TII))
1514 PotentialMatches[PotentialMI].push_back(Operand.get());
1517 for (
auto &PotentialPair : PotentialMatches) {
1518 MachineInstr &PotentialMI = *PotentialPair.first;
1519 convertToSDWA(PotentialMI, PotentialPair.second);
1522 PotentialMatches.clear();
1523 SDWAOperands.clear();
1529 while (!ConvertedInstructions.
empty())
1530 legalizeScalarOperands(*ConvertedInstructions.
pop_back_val(), ST);
MachineInstrBuilder & UseMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
static MachineOperand * findSingleRegDef(const MachineOperand *Reg, const MachineRegisterInfo *MRI)
static void copyRegOperand(MachineOperand &To, const MachineOperand &From)
static MachineOperand * findSingleRegUse(const MachineOperand *Reg, const MachineRegisterInfo *MRI)
static std::optional< SdwaSel > combineSdwaSel(SdwaSel Sel, SdwaSel OperandSel)
Combine an SDWA instruction's existing SDWA selection Sel with the SDWA selection OperandSel of its o...
static bool isSameReg(const MachineOperand &LHS, const MachineOperand &RHS)
static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII, AMDGPU::OpName SrcSelOpName, SdwaSel OpSel)
Verify that the SDWA selection operand SrcSelOpName of the SDWA instruction MI can be combined with t...
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
bool hasOptNone() const
Do not optimize this function (-O0).
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
LLVM_ABI void substituteRegister(Register FromReg, Register ToReg, unsigned SubIdx, const TargetRegisterInfo &RegInfo)
Replace all occurrences of FromReg with ToReg:SubIdx, properly composing subreg indices where necessa...
mop_range uses()
Returns all operands which may be register uses.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI MachineOperand * getOneNonDBGUse(Register RegNo) const
If the register has a single non-Debug use, returns it; otherwise returns nullptr.
MachineOperand * getOneDef(Register Reg) const
Returns the defining operand if there is exactly one operand defining the specified register,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
iterator_range< def_iterator > def_operands(Register Reg) const
This class implements a map that also provides access to all stored values in a deterministic order.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
self_iterator getIterator()
This class implements an extremely fast bulk output stream that can only output to a stream.
@ VGPR
Address space for VGPRs.
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
LLVM_READONLY int32_t getSDWAOp(uint32_t Opcode)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
NodeAddr< DefNode * > Def
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr RegState getKillRegState(bool B)
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
char & SIPeepholeSDWALegacyID
MCRegisterClass TargetRegisterClass