45#include "llvm/IR/IntrinsicsAArch64.h"
52#define DEBUG_TYPE "aarch64-isel"
65#define GET_GLOBALISEL_PREDICATE_BITSET
66#include "AArch64GenGlobalISel.inc"
67#undef GET_GLOBALISEL_PREDICATE_BITSET
87 ProduceNonFlagSettingCondBr =
135 bool tryOptAndIntoCompareBranch(
MachineInstr &AndInst,
bool Invert,
213 bool selectVectorLoadIntrinsic(
unsigned Opc,
unsigned NumVecs,
215 bool selectVectorLoadLaneIntrinsic(
unsigned Opc,
unsigned NumVecs,
217 void selectVectorStoreIntrinsic(
MachineInstr &
I,
unsigned NumVecs,
219 bool selectVectorStoreLaneIntrinsic(
MachineInstr &
I,
unsigned NumVecs,
236 unsigned Opc1,
unsigned Opc2,
bool isExt);
242 unsigned emitConstantPoolEntry(
const Constant *CPVal,
261 std::optional<CmpInst::Predicate> = std::nullopt)
const;
264 emitInstr(
unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
265 std::initializer_list<llvm::SrcOp> SrcOps,
267 const ComplexRendererFns &RenderFns = std::nullopt)
const;
302 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
325 MachineInstr *emitExtractVectorElt(std::optional<Register> DstReg,
347 std::pair<MachineInstr *, AArch64CC::CondCode>
382 ComplexRendererFns selectShiftA_32(
const MachineOperand &Root)
const;
383 ComplexRendererFns selectShiftB_32(
const MachineOperand &Root)
const;
384 ComplexRendererFns selectShiftA_64(
const MachineOperand &Root)
const;
385 ComplexRendererFns selectShiftB_64(
const MachineOperand &Root)
const;
387 template <
unsigned ShiftW
idth>
389 ComplexRendererFns select12BitValueWithLeftShift(
uint64_t Immed)
const;
391 ComplexRendererFns selectNegArithImmed(
MachineOperand &Root)
const;
394 unsigned Size)
const;
396 ComplexRendererFns selectAddrModeUnscaled8(
MachineOperand &Root)
const {
397 return selectAddrModeUnscaled(Root, 1);
399 ComplexRendererFns selectAddrModeUnscaled16(
MachineOperand &Root)
const {
400 return selectAddrModeUnscaled(Root, 2);
402 ComplexRendererFns selectAddrModeUnscaled32(
MachineOperand &Root)
const {
403 return selectAddrModeUnscaled(Root, 4);
405 ComplexRendererFns selectAddrModeUnscaled64(
MachineOperand &Root)
const {
406 return selectAddrModeUnscaled(Root, 8);
408 ComplexRendererFns selectAddrModeUnscaled128(
MachineOperand &Root)
const {
409 return selectAddrModeUnscaled(Root, 16);
414 ComplexRendererFns tryFoldAddLowIntoImm(
MachineInstr &RootDef,
unsigned Size,
418 unsigned Size)
const;
420 ComplexRendererFns selectAddrModeIndexed(
MachineOperand &Root)
const {
421 return selectAddrModeIndexed(Root, Width / 8);
430 bool IsAddrOperand)
const;
433 unsigned SizeInBytes)
const;
441 bool WantsExt)
const;
442 ComplexRendererFns selectAddrModeRegisterOffset(
MachineOperand &Root)
const;
444 unsigned SizeInBytes)
const;
446 ComplexRendererFns selectAddrModeXRO(
MachineOperand &Root)
const {
447 return selectAddrModeXRO(Root, Width / 8);
451 unsigned SizeInBytes)
const;
453 ComplexRendererFns selectAddrModeWRO(
MachineOperand &Root)
const {
454 return selectAddrModeWRO(Root, Width / 8);
458 bool AllowROR =
false)
const;
460 ComplexRendererFns selectArithShiftedRegister(
MachineOperand &Root)
const {
461 return selectShiftedRegister(Root);
464 ComplexRendererFns selectLogicalShiftedRegister(
MachineOperand &Root)
const {
465 return selectShiftedRegister(Root,
true);
475 bool IsLoadStore =
false)
const;
486 ComplexRendererFns selectArithExtendedRegister(
MachineOperand &Root)
const;
489 template <
unsigned W
idth>
490 ComplexRendererFns selectCVTFixedPoint(
MachineOperand &Root)
const;
491 template <
unsigned W
idth>
492 ComplexRendererFns selectCVTFixedPosRecipOperand(
MachineOperand &Root)
const;
493 ComplexRendererFns selectCVTFixedPointBase(
const MachineOperand &Root,
495 bool isReciprocal =
false)
const;
496 ComplexRendererFns selectCVTFixedPointVec(
MachineOperand &Root)
const;
501 unsigned getFixedPointWidthFromOperand(
const MachineOperand &Root)
const;
503 int OpIdx = -1)
const;
507 unsigned Width,
bool isReciprocal)
const;
509 int OpIdx = -1)
const;
511 int OpIdx = -1)
const;
513 int OpIdx = -1)
const;
517 int OpIdx = -1)
const;
519 int OpIdx = -1)
const;
521 int OpIdx = -1)
const;
524 int OpIdx = -1)
const;
530 bool tryOptSelect(
GSelect &Sel);
537 bool isLoadStoreOfNumBytes(
const MachineInstr &
MI,
unsigned NumBytes)
const;
550 bool ProduceNonFlagSettingCondBr =
false;
559#define GET_GLOBALISEL_PREDICATES_DECL
560#include "AArch64GenGlobalISel.inc"
561#undef GET_GLOBALISEL_PREDICATES_DECL
565#define GET_GLOBALISEL_TEMPORARIES_DECL
566#include "AArch64GenGlobalISel.inc"
567#undef GET_GLOBALISEL_TEMPORARIES_DECL
572#define GET_GLOBALISEL_IMPL
573#include "AArch64GenGlobalISel.inc"
574#undef GET_GLOBALISEL_IMPL
576AArch64InstructionSelector::AArch64InstructionSelector(
579 : TM(TM), STI(STI),
TII(*STI.getInstrInfo()),
TRI(*STI.getRegisterInfo()),
582#include
"AArch64GenGlobalISel.inc"
585#include
"AArch64GenGlobalISel.inc"
597 bool GetAllRegSet =
false) {
598 if (RB.
getID() == AArch64::GPRRegBankID) {
599 if (Ty.getSizeInBits() <= 32)
600 return GetAllRegSet ? &AArch64::GPR32allRegClass
601 : &AArch64::GPR32RegClass;
602 if (Ty.getSizeInBits() == 64)
603 return GetAllRegSet ? &AArch64::GPR64allRegClass
604 : &AArch64::GPR64RegClass;
605 if (Ty.getSizeInBits() == 128)
606 return &AArch64::XSeqPairsClassRegClass;
610 if (RB.
getID() == AArch64::FPRRegBankID) {
611 switch (Ty.getSizeInBits()) {
613 return &AArch64::FPR8RegClass;
615 return &AArch64::FPR16RegClass;
617 return &AArch64::FPR32RegClass;
619 return &AArch64::FPR64RegClass;
621 return &AArch64::FPR128RegClass;
633 bool GetAllRegSet =
false) {
636 "Expected FPR regbank for scalable type size");
637 return &AArch64::ZPRRegClass;
640 unsigned RegBankID = RB.
getID();
642 if (RegBankID == AArch64::GPRRegBankID) {
644 if (SizeInBits <= 32)
645 return GetAllRegSet ? &AArch64::GPR32allRegClass
646 : &AArch64::GPR32RegClass;
647 if (SizeInBits == 64)
648 return GetAllRegSet ? &AArch64::GPR64allRegClass
649 : &AArch64::GPR64RegClass;
650 if (SizeInBits == 128)
651 return &AArch64::XSeqPairsClassRegClass;
654 if (RegBankID == AArch64::FPRRegBankID) {
657 "Unexpected scalable register size");
658 return &AArch64::ZPRRegClass;
661 switch (SizeInBits) {
665 return &AArch64::FPR8RegClass;
667 return &AArch64::FPR16RegClass;
669 return &AArch64::FPR32RegClass;
671 return &AArch64::FPR64RegClass;
673 return &AArch64::FPR128RegClass;
683 switch (
TRI.getRegSizeInBits(*RC)) {
685 SubReg = AArch64::bsub;
688 SubReg = AArch64::hsub;
691 if (RC != &AArch64::FPR32RegClass)
692 SubReg = AArch64::sub_32;
694 SubReg = AArch64::ssub;
697 SubReg = AArch64::dsub;
701 dbgs() <<
"Couldn't find appropriate subregister for register class.");
710 switch (RB.
getID()) {
711 case AArch64::GPRRegBankID:
713 case AArch64::FPRRegBankID:
736 const unsigned RegClassIDs[],
738 unsigned NumRegs = Regs.
size();
741 assert(NumRegs >= 2 && NumRegs <= 4 &&
742 "Only support between two and 4 registers in a tuple!");
744 auto *DesiredClass =
TRI->getRegClass(RegClassIDs[NumRegs - 2]);
746 MIB.
buildInstr(TargetOpcode::REG_SEQUENCE, {DesiredClass}, {});
747 for (
unsigned I = 0,
E = Regs.
size();
I <
E; ++
I) {
748 RegSequence.addUse(Regs[
I]);
749 RegSequence.addImm(SubRegs[
I]);
751 return RegSequence.getReg(0);
756 static const unsigned RegClassIDs[] = {
757 AArch64::DDRegClassID, AArch64::DDDRegClassID, AArch64::DDDDRegClassID};
758 static const unsigned SubRegs[] = {AArch64::dsub0, AArch64::dsub1,
759 AArch64::dsub2, AArch64::dsub3};
760 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
765 static const unsigned RegClassIDs[] = {
766 AArch64::QQRegClassID, AArch64::QQQRegClassID, AArch64::QQQQRegClassID};
767 static const unsigned SubRegs[] = {AArch64::qsub0, AArch64::qsub1,
768 AArch64::qsub2, AArch64::qsub3};
769 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
774 auto &
MBB = *
MI.getParent();
775 auto &MF = *
MBB.getParent();
776 auto &MRI = MF.getRegInfo();
782 else if (Root.
isReg()) {
787 Immed = ValAndVReg->Value.getSExtValue();
798 if (RegBankID == AArch64::GPRRegBankID) {
800 switch (GenericOpc) {
801 case TargetOpcode::G_SHL:
802 return AArch64::LSLVWr;
803 case TargetOpcode::G_LSHR:
804 return AArch64::LSRVWr;
805 case TargetOpcode::G_ASHR:
806 return AArch64::ASRVWr;
810 }
else if (OpSize == 64) {
811 switch (GenericOpc) {
812 case TargetOpcode::G_SHL:
813 return AArch64::LSLVXr;
814 case TargetOpcode::G_LSHR:
815 return AArch64::LSRVXr;
816 case TargetOpcode::G_ASHR:
817 return AArch64::ASRVXr;
833 const bool isStore = GenericOpc == TargetOpcode::G_STORE;
835 case AArch64::GPRRegBankID:
838 return isStore ? AArch64::STRBBui : AArch64::LDRBBui;
840 return isStore ? AArch64::STRHHui : AArch64::LDRHHui;
842 return isStore ? AArch64::STRWui : AArch64::LDRWui;
844 return isStore ? AArch64::STRXui : AArch64::LDRXui;
847 case AArch64::FPRRegBankID:
850 return isStore ? AArch64::STRBui : AArch64::LDRBui;
852 return isStore ? AArch64::STRHui : AArch64::LDRHui;
854 return isStore ? AArch64::STRSui : AArch64::LDRSui;
856 return isStore ? AArch64::STRDui : AArch64::LDRDui;
858 return isStore ? AArch64::STRQui : AArch64::LDRQui;
872 assert(SrcReg.
isValid() &&
"Expected a valid source register?");
873 assert(To &&
"Destination register class cannot be null");
874 assert(SubReg &&
"Expected a valid subregister");
878 MIB.
buildInstr(TargetOpcode::COPY, {To}, {}).addReg(SrcReg, {}, SubReg);
880 RegOp.
setReg(SubRegCopy.getReg(0));
884 if (!
I.getOperand(0).getReg().isPhysical())
901 if (
Reg.isPhysical())
909 RC = getRegClassForTypeOnBank(Ty, RB);
912 dbgs() <<
"Warning: DBG_VALUE operand has unexpected size/bank\n");
925 Register DstReg =
I.getOperand(0).getReg();
926 Register SrcReg =
I.getOperand(1).getReg();
957 if (
I.getOpcode() == TargetOpcode::G_BITCAST &&
959 if (DstRegBank.
getID() == AArch64::FPRRegBankID &&
960 SrcRegBank.
getID() == AArch64::GPRRegBankID) {
969 BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::FMOVWSr))
972 I.setDesc(
TII.get(TargetOpcode::COPY));
973 I.getOperand(1).setReg(FPR32);
974 I.getOperand(1).setSubReg(AArch64::hsub);
978 if (DstRegBank.
getID() == AArch64::GPRRegBankID &&
979 SrcRegBank.
getID() == AArch64::FPRRegBankID) {
989 TII.get(TargetOpcode::SUBREG_TO_REG))
993 I.setDesc(
TII.get(AArch64::FMOVSWr));
994 I.getOperand(1).setReg(FPR32);
1003 LLVM_DEBUG(
dbgs() <<
"Couldn't determine source register class\n");
1007 const TypeSize SrcSize =
TRI.getRegSizeInBits(*SrcRC);
1008 const TypeSize DstSize =
TRI.getRegSizeInBits(*DstRC);
1009 unsigned SrcSubReg =
I.getOperand(1).getSubReg();
1023 auto Copy = MIB.
buildCopy({DstTempRC}, {SrcReg});
1024 copySubReg(
I, MRI, RBI, Copy.getReg(0), DstRC, SubReg);
1025 }
else if (SrcSize > DstSize) {
1032 }
else if (DstSize > SrcSize) {
1041 TII.get(AArch64::SUBREG_TO_REG), PromoteReg)
1045 RegOp.
setReg(PromoteReg);
1064 if (
I.getOpcode() == TargetOpcode::G_ZEXT) {
1065 I.setDesc(
TII.get(AArch64::COPY));
1066 assert(SrcRegBank.
getID() == AArch64::GPRRegBankID);
1070 I.setDesc(
TII.get(AArch64::COPY));
1078 MachineRegisterInfo &MRI = *MIB.
getMRI();
1081 "Expected both select operands to have the same regbank?");
1087 "Expected 32 bit or 64 bit select only?");
1088 const bool Is32Bit =
Size == 32;
1090 unsigned Opc = Is32Bit ? AArch64::FCSELSrrr : AArch64::FCSELDrrr;
1097 unsigned Opc = Is32Bit ? AArch64::CSELWr : AArch64::CSELXr;
1099 auto TryFoldBinOpIntoSelect = [&
Opc, Is32Bit, &CC, &MRI,
1114 Opc = Is32Bit ? AArch64::CSNEGWr : AArch64::CSNEGXr;
1131 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1150 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1166 auto TryOptSelectCst = [&
Opc, &
True, &
False, &CC, Is32Bit, &MRI,
1172 if (!TrueCst && !FalseCst)
1175 Register ZReg = Is32Bit ? AArch64::WZR : AArch64::XZR;
1176 if (TrueCst && FalseCst) {
1177 int64_t
T = TrueCst->Value.getSExtValue();
1178 int64_t
F = FalseCst->Value.getSExtValue();
1180 if (
T == 0 &&
F == 1) {
1182 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1188 if (
T == 0 &&
F == -1) {
1190 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1198 int64_t
T = TrueCst->Value.getSExtValue();
1201 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1210 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1219 int64_t
F = FalseCst->Value.getSExtValue();
1222 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1229 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1242 return &*SelectInst;
1247 MachineRegisterInfo *MRI =
nullptr) {
1260 if (ValAndVReg && ValAndVReg->Value == 0)
1267 if (ValAndVReg && ValAndVReg->Value == 0)
1371 assert(
Reg.isValid() &&
"Expected valid register!");
1372 bool HasZext =
false;
1374 unsigned Opc =
MI->getOpcode();
1376 if (!
MI->getOperand(0).isReg() ||
1385 if (
Opc == TargetOpcode::G_ANYEXT ||
Opc == TargetOpcode::G_ZEXT ||
1386 Opc == TargetOpcode::G_TRUNC) {
1387 if (
Opc == TargetOpcode::G_ZEXT)
1390 Register NextReg =
MI->getOperand(1).getReg();
1404 std::optional<uint64_t>
C;
1409 case TargetOpcode::G_AND:
1410 case TargetOpcode::G_XOR: {
1411 TestReg =
MI->getOperand(1).getReg();
1412 Register ConstantReg =
MI->getOperand(2).getReg();
1423 C = VRegAndVal->Value.getZExtValue();
1425 C = VRegAndVal->Value.getSExtValue();
1429 case TargetOpcode::G_ASHR:
1430 case TargetOpcode::G_LSHR:
1431 case TargetOpcode::G_SHL: {
1432 TestReg =
MI->getOperand(1).getReg();
1436 C = VRegAndVal->Value.getSExtValue();
1452 case TargetOpcode::G_AND:
1454 if ((*
C >> Bit) & 1)
1457 case TargetOpcode::G_SHL:
1460 if (*
C <= Bit && (Bit - *
C) < TestRegSize) {
1465 case TargetOpcode::G_ASHR:
1470 if (Bit >= TestRegSize)
1471 Bit = TestRegSize - 1;
1473 case TargetOpcode::G_LSHR:
1475 if ((Bit + *
C) < TestRegSize) {
1480 case TargetOpcode::G_XOR:
1489 if ((*
C >> Bit) & 1)
1504MachineInstr *AArch64InstructionSelector::emitTestBit(
1505 Register TestReg,
uint64_t Bit,
bool IsNegative, MachineBasicBlock *DstMBB,
1506 MachineIRBuilder &MIB)
const {
1508 assert(ProduceNonFlagSettingCondBr &&
1509 "Cannot emit TB(N)Z with speculation tracking!");
1510 MachineRegisterInfo &MRI = *MIB.
getMRI();
1514 LLT Ty = MRI.
getType(TestReg);
1517 assert(Bit < 64 &&
"Bit is too large!");
1521 bool UseWReg =
Bit < 32;
1522 unsigned NecessarySize = UseWReg ? 32 : 64;
1523 if (
Size != NecessarySize)
1524 TestReg = moveScalarRegClass(
1525 TestReg, UseWReg ? AArch64::GPR32RegClass : AArch64::GPR64RegClass,
1528 static const unsigned OpcTable[2][2] = {{AArch64::TBZX, AArch64::TBNZX},
1529 {AArch64::TBZW, AArch64::TBNZW}};
1530 unsigned Opc = OpcTable[UseWReg][IsNegative];
1537bool AArch64InstructionSelector::tryOptAndIntoCompareBranch(
1538 MachineInstr &AndInst,
bool Invert, MachineBasicBlock *DstMBB,
1539 MachineIRBuilder &MIB)
const {
1540 assert(AndInst.
getOpcode() == TargetOpcode::G_AND &&
"Expected G_AND only?");
1567 int32_t
Bit = MaybeBit->Value.exactLogBase2();
1574 emitTestBit(TestReg, Bit, Invert, DstMBB, MIB);
1578MachineInstr *AArch64InstructionSelector::emitCBZ(
Register CompareReg,
1580 MachineBasicBlock *DestMBB,
1581 MachineIRBuilder &MIB)
const {
1582 assert(ProduceNonFlagSettingCondBr &&
"CBZ does not set flags!");
1583 MachineRegisterInfo &MRI = *MIB.
getMRI();
1585 AArch64::GPRRegBankID &&
1586 "Expected GPRs only?");
1587 auto Ty = MRI.
getType(CompareReg);
1590 assert(Width <= 64 &&
"Expected width to be at most 64?");
1591 static const unsigned OpcTable[2][2] = {{AArch64::CBZW, AArch64::CBZX},
1592 {AArch64::CBNZW, AArch64::CBNZX}};
1593 unsigned Opc = OpcTable[IsNegative][Width == 64];
1594 auto BranchMI = MIB.
buildInstr(
Opc, {}, {CompareReg}).addMBB(DestMBB);
1599bool AArch64InstructionSelector::selectCompareBranchFedByFCmp(
1600 MachineInstr &
I, MachineInstr &FCmp, MachineIRBuilder &MIB)
const {
1602 assert(
I.getOpcode() == TargetOpcode::G_BRCOND);
1610 MachineBasicBlock *DestMBB =
I.getOperand(1).getMBB();
1614 I.eraseFromParent();
1618bool AArch64InstructionSelector::tryOptCompareBranchFedByICmp(
1619 MachineInstr &
I, MachineInstr &ICmp, MachineIRBuilder &MIB)
const {
1621 assert(
I.getOpcode() == TargetOpcode::G_BRCOND);
1627 if (!ProduceNonFlagSettingCondBr)
1630 MachineRegisterInfo &MRI = *MIB.
getMRI();
1631 MachineBasicBlock *DestMBB =
I.getOperand(1).getMBB();
1646 if (VRegAndVal && !AndInst) {
1647 int64_t
C = VRegAndVal->Value.getSExtValue();
1653 emitTestBit(
LHS, Bit,
false, DestMBB, MIB);
1654 I.eraseFromParent();
1662 emitTestBit(
LHS, Bit,
true, DestMBB, MIB);
1663 I.eraseFromParent();
1671 emitTestBit(
LHS, Bit,
false, DestMBB, MIB);
1672 I.eraseFromParent();
1686 if (VRegAndVal && VRegAndVal->Value == 0) {
1694 tryOptAndIntoCompareBranch(
1696 I.eraseFromParent();
1702 if (!LHSTy.isVector() && LHSTy.getSizeInBits() <= 64) {
1704 I.eraseFromParent();
1713bool AArch64InstructionSelector::selectCompareBranchFedByICmp(
1714 MachineInstr &
I, MachineInstr &ICmp, MachineIRBuilder &MIB)
const {
1716 assert(
I.getOpcode() == TargetOpcode::G_BRCOND);
1717 if (tryOptCompareBranchFedByICmp(
I, ICmp, MIB))
1721 MachineBasicBlock *DestMBB =
I.getOperand(1).getMBB();
1728 I.eraseFromParent();
1732bool AArch64InstructionSelector::selectCompareBranch(
1734 Register CondReg =
I.getOperand(0).getReg();
1735 MachineInstr *CCMI = MRI.
getVRegDef(CondReg);
1739 if (CCMIOpc == TargetOpcode::G_FCMP)
1740 return selectCompareBranchFedByFCmp(
I, *CCMI, MIB);
1741 if (CCMIOpc == TargetOpcode::G_ICMP)
1742 return selectCompareBranchFedByICmp(
I, *CCMI, MIB);
1747 if (ProduceNonFlagSettingCondBr) {
1748 emitTestBit(CondReg, 0,
true,
1749 I.getOperand(1).getMBB(), MIB);
1750 I.eraseFromParent();
1760 .
addMBB(
I.getOperand(1).getMBB());
1761 I.eraseFromParent();
1781 return std::nullopt;
1783 int64_t
Imm = *ShiftImm;
1785 return std::nullopt;
1786 switch (SrcTy.getElementType().getSizeInBits()) {
1789 return std::nullopt;
1792 return std::nullopt;
1796 return std::nullopt;
1800 return std::nullopt;
1804 return std::nullopt;
1810bool AArch64InstructionSelector::selectVectorSHL(MachineInstr &
I,
1811 MachineRegisterInfo &MRI) {
1812 assert(
I.getOpcode() == TargetOpcode::G_SHL);
1813 Register DstReg =
I.getOperand(0).getReg();
1814 const LLT Ty = MRI.
getType(DstReg);
1815 Register Src1Reg =
I.getOperand(1).getReg();
1816 Register Src2Reg =
I.getOperand(2).getReg();
1827 Opc = ImmVal ? AArch64::SHLv2i64_shift : AArch64::USHLv2i64;
1829 Opc = ImmVal ? AArch64::SHLv4i32_shift : AArch64::USHLv4i32;
1831 Opc = ImmVal ? AArch64::SHLv2i32_shift : AArch64::USHLv2i32;
1833 Opc = ImmVal ? AArch64::SHLv4i16_shift : AArch64::USHLv4i16;
1835 Opc = ImmVal ? AArch64::SHLv8i16_shift : AArch64::USHLv8i16;
1837 Opc = ImmVal ? AArch64::SHLv16i8_shift : AArch64::USHLv16i8;
1839 Opc = ImmVal ? AArch64::SHLv8i8_shift : AArch64::USHLv8i8;
1851 I.eraseFromParent();
1855bool AArch64InstructionSelector::selectVectorAshrLshr(
1856 MachineInstr &
I, MachineRegisterInfo &MRI) {
1857 assert(
I.getOpcode() == TargetOpcode::G_ASHR ||
1858 I.getOpcode() == TargetOpcode::G_LSHR);
1859 Register DstReg =
I.getOperand(0).getReg();
1860 const LLT Ty = MRI.
getType(DstReg);
1861 Register Src1Reg =
I.getOperand(1).getReg();
1862 Register Src2Reg =
I.getOperand(2).getReg();
1867 bool IsASHR =
I.getOpcode() == TargetOpcode::G_ASHR;
1877 unsigned NegOpc = 0;
1879 getRegClassForTypeOnBank(Ty, RBI.
getRegBank(AArch64::FPRRegBankID));
1881 Opc = IsASHR ? AArch64::SSHLv2i64 : AArch64::USHLv2i64;
1882 NegOpc = AArch64::NEGv2i64;
1884 Opc = IsASHR ? AArch64::SSHLv4i32 : AArch64::USHLv4i32;
1885 NegOpc = AArch64::NEGv4i32;
1887 Opc = IsASHR ? AArch64::SSHLv2i32 : AArch64::USHLv2i32;
1888 NegOpc = AArch64::NEGv2i32;
1890 Opc = IsASHR ? AArch64::SSHLv4i16 : AArch64::USHLv4i16;
1891 NegOpc = AArch64::NEGv4i16;
1893 Opc = IsASHR ? AArch64::SSHLv8i16 : AArch64::USHLv8i16;
1894 NegOpc = AArch64::NEGv8i16;
1896 Opc = IsASHR ? AArch64::SSHLv16i8 : AArch64::USHLv16i8;
1897 NegOpc = AArch64::NEGv16i8;
1899 Opc = IsASHR ? AArch64::SSHLv8i8 : AArch64::USHLv8i8;
1900 NegOpc = AArch64::NEGv8i8;
1906 auto Neg = MIB.
buildInstr(NegOpc, {RC}, {Src2Reg});
1910 I.eraseFromParent();
1914bool AArch64InstructionSelector::selectVaStartAAPCS(
1924 const AArch64FunctionInfo *FuncInfo = MF.
getInfo<AArch64FunctionInfo>();
1926 const auto *PtrRegClass =
1927 STI.
isTargetILP32() ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
1929 const MCInstrDesc &MCIDAddAddr =
1931 const MCInstrDesc &MCIDStoreAddr =
1943 const auto VAList =
I.getOperand(0).getReg();
1946 unsigned OffsetBytes = 0;
1950 const auto PushAddress = [&](
const int FrameIndex,
const int64_t
Imm) {
1952 auto MIB =
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(), MCIDAddAddr)
1959 const auto *MMO = *
I.memoperands_begin();
1960 MIB =
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(), MCIDStoreAddr)
1963 .
addImm(OffsetBytes / PtrSize)
1965 MMO->getPointerInfo().getWithOffset(OffsetBytes),
1969 OffsetBytes += PtrSize;
1985 const auto PushIntConstant = [&](
const int32_t
Value) {
1986 constexpr int IntSize = 4;
1989 BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::MOVi32imm))
1994 const auto *MMO = *
I.memoperands_begin();
1995 MIB =
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::STRWui))
1998 .
addImm(OffsetBytes / IntSize)
2000 MMO->getPointerInfo().getWithOffset(OffsetBytes),
2003 OffsetBytes += IntSize;
2007 PushIntConstant(-
static_cast<int32_t
>(GPRSize));
2010 PushIntConstant(-
static_cast<int32_t
>(FPRSize));
2014 I.eraseFromParent();
2018bool AArch64InstructionSelector::selectVaStartDarwin(
2020 AArch64FunctionInfo *FuncInfo = MF.
getInfo<AArch64FunctionInfo>();
2021 Register ListReg =
I.getOperand(0).getReg();
2026 if (MF.
getSubtarget<AArch64Subtarget>().isCallingConvWin64(
2034 BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::ADDXri))
2042 MIB =
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::STRXui))
2049 I.eraseFromParent();
2053void AArch64InstructionSelector::materializeLargeCMVal(
2054 MachineInstr &
I,
const Value *V,
unsigned OpFlags) {
2059 auto MovZ = MIB.
buildInstr(AArch64::MOVZXi, {&AArch64::GPR64RegClass}, {});
2074 GV, MovZ->getOperand(1).getOffset(), Flags));
2078 MovZ->getOperand(1).getOffset(), Flags));
2084 Register DstReg = BuildMovK(MovZ.getReg(0),
2090bool AArch64InstructionSelector::preISelLower(MachineInstr &
I) {
2095 switch (
I.getOpcode()) {
2096 case TargetOpcode::G_CONSTANT: {
2097 Register DefReg =
I.getOperand(0).getReg();
2098 const LLT DefTy = MRI.
getType(DefReg);
2104 APInt Val =
I.getOperand(1).getCImm()->getValue().zext(32);
2105 I.getOperand(1).setCImm(
2110 I.getOperand(0).setReg(WideReg);
2119 if (PtrSize != 32 && PtrSize != 64)
2125 case TargetOpcode::G_STORE: {
2126 bool Changed = contractCrossBankCopyIntoStore(
I, MRI);
2127 MachineOperand &SrcOp =
I.getOperand(0);
2140 case TargetOpcode::G_PTR_ADD: {
2144 if (TL->shouldPreservePtrArith(MF.
getFunction(), EVT()))
2146 return convertPtrAddToAdd(
I, MRI);
2148 case TargetOpcode::G_LOAD: {
2153 Register DstReg =
I.getOperand(0).getReg();
2154 const LLT DstTy = MRI.
getType(DstReg);
2160 case TargetOpcode::G_VECREDUCE_ADD:
2161 case TargetOpcode::G_VECREDUCE_SMAX:
2162 case TargetOpcode::G_VECREDUCE_SMIN:
2163 case TargetOpcode::G_VECREDUCE_UMAX:
2164 case TargetOpcode::G_VECREDUCE_UMIN: {
2167 Register DstReg =
I.getOperand(0).getReg();
2168 const RegisterBank &DstRB = *RBI.
getRegBank(DstReg, MRI,
TRI);
2169 if (DstRB.
getID() != AArch64::GPRRegBankID)
2172 LLT DstTy = MRI.
getType(DstReg);
2174 getRegClassForTypeOnBank(DstTy, DstRB,
true);
2180 I.getOperand(0).setReg(FPRDst);
2182 BuildMI(
MBB, std::next(
I.getIterator()), MIMetadata(
I),
2183 TII.get(TargetOpcode::COPY), DstReg)
2187 case AArch64::G_DUP: {
2189 LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
2193 MRI.
setType(
I.getOperand(0).getReg(),
2195 MRI.
setRegClass(NewSrc.getReg(0), &AArch64::GPR64RegClass);
2196 I.getOperand(1).setReg(NewSrc.getReg(0));
2199 case AArch64::G_INSERT_VECTOR_ELT: {
2200 LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
2201 LLT SrcVecTy = MRI.
getType(
I.getOperand(1).getReg());
2205 MRI.
setType(
I.getOperand(1).getReg(),
2207 MRI.
setType(
I.getOperand(0).getReg(),
2209 MRI.
setRegClass(NewSrc.getReg(0), &AArch64::GPR64RegClass);
2210 I.getOperand(2).setReg(NewSrc.getReg(0));
2214 Register EltReg =
I.getOperand(2).getReg();
2215 LLT EltTy = MRI.
getType(EltReg);
2221 MRI.
setRegClass(NewElt.getReg(0), &AArch64::GPR32RegClass);
2222 I.getOperand(2).setReg(NewElt.getReg(0));
2227 case TargetOpcode::G_UITOFP:
2228 case TargetOpcode::G_SITOFP: {
2233 Register SrcReg =
I.getOperand(1).getReg();
2234 LLT SrcTy = MRI.
getType(SrcReg);
2235 LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
2244 I.getOperand(1).setReg(
Copy.getReg(0));
2246 getRegClassForTypeOnBank(
2247 SrcTy, RBI.
getRegBank(AArch64::FPRRegBankID)));
2249 if (
I.getOpcode() == TargetOpcode::G_SITOFP)
2250 I.setDesc(
TII.get(AArch64::G_SITOF));
2252 I.setDesc(
TII.get(AArch64::G_UITOF));
2270bool AArch64InstructionSelector::convertPtrAddToAdd(
2271 MachineInstr &
I, MachineRegisterInfo &MRI) {
2272 assert(
I.getOpcode() == TargetOpcode::G_PTR_ADD &&
"Expected G_PTR_ADD");
2273 Register DstReg =
I.getOperand(0).getReg();
2274 Register AddOp1Reg =
I.getOperand(1).getReg();
2275 const LLT PtrTy = MRI.
getType(DstReg);
2279 const LLT CastPtrTy = PtrTy.
isVector()
2291 I.setDesc(
TII.get(TargetOpcode::G_ADD));
2292 MRI.
setType(DstReg, CastPtrTy);
2293 I.getOperand(1).setReg(PtrToInt.getReg(0));
2294 if (!select(*PtrToInt)) {
2295 LLVM_DEBUG(
dbgs() <<
"Failed to select G_PTRTOINT in convertPtrAddToAdd");
2304 I.getOperand(2).setReg(NegatedReg);
2305 I.setDesc(
TII.get(TargetOpcode::G_SUB));
2309bool AArch64InstructionSelector::earlySelectSHL(MachineInstr &
I,
2310 MachineRegisterInfo &MRI) {
2314 assert(
I.getOpcode() == TargetOpcode::G_SHL &&
"unexpected op");
2315 const auto &MO =
I.getOperand(2);
2320 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
2324 auto Imm1Fn = Is64Bit ? selectShiftA_64(MO) : selectShiftA_32(MO);
2325 auto Imm2Fn = Is64Bit ? selectShiftB_64(MO) : selectShiftB_32(MO);
2327 if (!Imm1Fn || !Imm2Fn)
2331 MIB.
buildInstr(Is64Bit ? AArch64::UBFMXri : AArch64::UBFMWri,
2332 {
I.getOperand(0).getReg()}, {
I.getOperand(1).getReg()});
2334 for (
auto &RenderFn : *Imm1Fn)
2336 for (
auto &RenderFn : *Imm2Fn)
2339 I.eraseFromParent();
2344bool AArch64InstructionSelector::contractCrossBankCopyIntoStore(
2345 MachineInstr &
I, MachineRegisterInfo &MRI) {
2346 assert(
I.getOpcode() == TargetOpcode::G_STORE &&
"Expected G_STORE");
2364 LLT DefDstTy = MRI.
getType(DefDstReg);
2365 Register StoreSrcReg =
I.getOperand(0).getReg();
2366 LLT StoreSrcTy = MRI.
getType(StoreSrcReg);
2382 I.getOperand(0).setReg(DefDstReg);
2386bool AArch64InstructionSelector::earlySelect(MachineInstr &
I) {
2387 assert(
I.getParent() &&
"Instruction should be in a basic block!");
2388 assert(
I.getParent()->getParent() &&
"Instruction should be in a function!");
2394 switch (
I.getOpcode()) {
2395 case AArch64::G_DUP: {
2398 Register Src =
I.getOperand(1).getReg();
2400 Src, MRI,
true,
true);
2404 Register Dst =
I.getOperand(0).getReg();
2410 if (!emitConstantVector(Dst, CV, MIB, MRI))
2412 I.eraseFromParent();
2415 case TargetOpcode::G_SEXT:
2418 if (selectUSMovFromExtend(
I, MRI))
2421 case TargetOpcode::G_BR:
2423 case TargetOpcode::G_SHL:
2424 return earlySelectSHL(
I, MRI);
2425 case TargetOpcode::G_CONSTANT: {
2426 bool IsZero =
false;
2427 if (
I.getOperand(1).isCImm())
2428 IsZero =
I.getOperand(1).getCImm()->isZero();
2429 else if (
I.getOperand(1).isImm())
2430 IsZero =
I.getOperand(1).getImm() == 0;
2435 Register DefReg =
I.getOperand(0).getReg();
2438 I.getOperand(1).ChangeToRegister(AArch64::XZR,
false);
2441 I.getOperand(1).ChangeToRegister(AArch64::WZR,
false);
2446 I.setDesc(
TII.get(TargetOpcode::COPY));
2450 case TargetOpcode::G_ADD: {
2459 Register AddDst =
I.getOperand(0).getReg();
2460 Register AddLHS =
I.getOperand(1).getReg();
2461 Register AddRHS =
I.getOperand(2).getReg();
2471 auto MatchCmp = [&](
Register Reg) -> MachineInstr * {
2492 MachineInstr *
Cmp = MatchCmp(AddRHS);
2496 Cmp = MatchCmp(AddRHS);
2500 auto &PredOp =
Cmp->getOperand(1);
2502 emitIntegerCompare(
Cmp->getOperand(2),
2503 Cmp->getOperand(3), PredOp, MIB);
2507 emitCSINC(AddDst, AddLHS, AddLHS, InvCC, MIB);
2508 I.eraseFromParent();
2511 case TargetOpcode::G_OR: {
2515 Register Dst =
I.getOperand(0).getReg();
2535 if (ShiftImm >
Size || ((1ULL << ShiftImm) - 1ULL) !=
uint64_t(MaskImm))
2538 int64_t Immr =
Size - ShiftImm;
2539 int64_t Imms =
Size - ShiftImm - 1;
2540 unsigned Opc =
Size == 32 ? AArch64::BFMWri : AArch64::BFMXri;
2541 emitInstr(
Opc, {Dst}, {MaskSrc, ShiftSrc, Immr, Imms}, MIB);
2542 I.eraseFromParent();
2545 case TargetOpcode::G_FENCE: {
2546 if (
I.getOperand(1).getImm() == 0)
2550 .
addImm(
I.getOperand(0).getImm() == 4 ? 0x9 : 0xb);
2551 I.eraseFromParent();
2554 case TargetOpcode::G_BRINDIRECT: {
2556 if (std::optional<uint16_t> BADisc =
2558 auto MI = MIB.
buildInstr(AArch64::BRA, {}, {
I.getOperand(0).getReg()});
2562 I.eraseFromParent();
2575bool AArch64InstructionSelector::select(MachineInstr &
I) {
2576 assert(
I.getParent() &&
"Instruction should be in a basic block!");
2577 assert(
I.getParent()->getParent() &&
"Instruction should be in a function!");
2583 const AArch64Subtarget *Subtarget = &MF.
getSubtarget<AArch64Subtarget>();
2584 if (Subtarget->requiresStrictAlign()) {
2586 LLVM_DEBUG(
dbgs() <<
"AArch64 GISel does not support strict-align yet\n");
2592 unsigned Opcode =
I.getOpcode();
2594 if (!
I.isPreISelOpcode() || Opcode == TargetOpcode::G_PHI) {
2597 if (Opcode == TargetOpcode::LOAD_STACK_GUARD) {
2602 if (Opcode == TargetOpcode::PHI || Opcode == TargetOpcode::G_PHI) {
2603 const Register DefReg =
I.getOperand(0).getReg();
2604 const LLT DefTy = MRI.
getType(DefReg);
2617 DefRC = getRegClassForTypeOnBank(DefTy, RB);
2624 I.setDesc(
TII.get(TargetOpcode::PHI));
2632 if (
I.isDebugInstr())
2639 if (
I.getNumOperands() !=
I.getNumExplicitOperands()) {
2641 dbgs() <<
"Generic instruction has unexpected implicit operands\n");
2648 if (preISelLower(
I)) {
2649 Opcode =
I.getOpcode();
2660 if (selectImpl(
I, *CoverageInfo))
2664 I.getOperand(0).isReg() ? MRI.
getType(
I.getOperand(0).getReg()) : LLT{};
2667 case TargetOpcode::G_SBFX:
2668 case TargetOpcode::G_UBFX: {
2669 static const unsigned OpcTable[2][2] = {
2670 {AArch64::UBFMWri, AArch64::UBFMXri},
2671 {AArch64::SBFMWri, AArch64::SBFMXri}};
2672 bool IsSigned = Opcode == TargetOpcode::G_SBFX;
2674 unsigned Opc = OpcTable[IsSigned][
Size == 64];
2677 assert(Cst1 &&
"Should have gotten a constant for src 1?");
2680 assert(Cst2 &&
"Should have gotten a constant for src 2?");
2681 auto LSB = Cst1->Value.getZExtValue();
2682 auto Width = Cst2->Value.getZExtValue();
2686 .
addImm(LSB + Width - 1);
2687 I.eraseFromParent();
2691 case TargetOpcode::G_BRCOND:
2692 return selectCompareBranch(
I, MF, MRI);
2694 case TargetOpcode::G_BRJT:
2695 return selectBrJT(
I, MRI);
2697 case AArch64::G_ADD_LOW: {
2702 MachineInstr *BaseMI = MRI.
getVRegDef(
I.getOperand(1).getReg());
2703 if (BaseMI->
getOpcode() != AArch64::ADRP) {
2704 I.setDesc(
TII.get(AArch64::ADDXri));
2710 "Expected small code model");
2712 auto Op2 =
I.getOperand(2);
2713 auto MovAddr = MIB.
buildInstr(AArch64::MOVaddr, {
I.getOperand(0)}, {})
2714 .addGlobalAddress(Op1.getGlobal(), Op1.getOffset(),
2715 Op1.getTargetFlags())
2717 Op2.getTargetFlags());
2718 I.eraseFromParent();
2723 case TargetOpcode::G_FCONSTANT: {
2724 const Register DefReg =
I.getOperand(0).getReg();
2725 const LLT DefTy = MRI.
getType(DefReg);
2736 bool OptForSize = shouldOptForSize(&MF);
2740 if (TLI->isFPImmLegal(
I.getOperand(1).getFPImm()->getValueAPF(),
2747 auto *FPImm =
I.getOperand(1).getFPImm();
2750 LLVM_DEBUG(
dbgs() <<
"Failed to load double constant pool entry\n");
2753 MIB.
buildCopy({DefReg}, {LoadMI->getOperand(0).getReg()});
2754 I.eraseFromParent();
2759 assert((DefSize == 32 || DefSize == 64) &&
"Unexpected const def size");
2762 DefSize == 32 ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass);
2763 MachineOperand &RegOp =
I.getOperand(0);
2769 LLVM_DEBUG(
dbgs() <<
"Failed to constrain G_FCONSTANT def operand\n");
2773 MachineOperand &ImmOp =
I.getOperand(1);
2777 const unsigned MovOpc =
2778 DefSize == 64 ? AArch64::MOVi64imm : AArch64::MOVi32imm;
2779 I.setDesc(
TII.get(MovOpc));
2783 case TargetOpcode::G_EXTRACT: {
2784 Register DstReg =
I.getOperand(0).getReg();
2785 Register SrcReg =
I.getOperand(1).getReg();
2786 LLT SrcTy = MRI.
getType(SrcReg);
2787 LLT DstTy = MRI.
getType(DstReg);
2799 unsigned Offset =
I.getOperand(2).getImm();
2804 const RegisterBank &SrcRB = *RBI.
getRegBank(SrcReg, MRI,
TRI);
2805 const RegisterBank &DstRB = *RBI.
getRegBank(DstReg, MRI,
TRI);
2808 if (SrcRB.
getID() == AArch64::GPRRegBankID) {
2810 MIB.
buildInstr(TargetOpcode::COPY, {DstReg}, {})
2812 Offset == 0 ? AArch64::sube64 : AArch64::subo64);
2814 AArch64::GPR64RegClass, NewI->getOperand(0));
2815 I.eraseFromParent();
2821 unsigned LaneIdx =
Offset / 64;
2822 MachineInstr *Extract = emitExtractVectorElt(
2823 DstReg, DstRB,
LLT::scalar(64), SrcReg, LaneIdx, MIB);
2826 I.eraseFromParent();
2830 I.setDesc(
TII.get(SrcSize == 64 ? AArch64::UBFMXri : AArch64::UBFMWri));
2831 MachineInstrBuilder(MF,
I).addImm(
I.getOperand(2).getImm() +
2836 "unexpected G_EXTRACT types");
2843 MIB.
buildInstr(TargetOpcode::COPY, {
I.getOperand(0).getReg()}, {})
2844 .addReg(DstReg, {}, AArch64::sub_32);
2846 AArch64::GPR32RegClass, MRI);
2847 I.getOperand(0).setReg(DstReg);
2853 case TargetOpcode::G_INSERT: {
2854 LLT SrcTy = MRI.
getType(
I.getOperand(2).getReg());
2855 LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
2862 I.setDesc(
TII.get(DstSize == 64 ? AArch64::BFMXri : AArch64::BFMWri));
2863 unsigned LSB =
I.getOperand(3).getImm();
2865 I.getOperand(3).setImm((DstSize - LSB) % DstSize);
2866 MachineInstrBuilder(MF,
I).addImm(Width - 1);
2870 "unexpected G_INSERT types");
2877 TII.get(AArch64::SUBREG_TO_REG))
2879 .
addUse(
I.getOperand(2).getReg())
2880 .
addImm(AArch64::sub_32);
2882 AArch64::GPR32RegClass, MRI);
2883 I.getOperand(2).setReg(SrcReg);
2888 case TargetOpcode::G_FRAME_INDEX: {
2895 I.setDesc(
TII.get(AArch64::ADDXri));
2905 case TargetOpcode::G_GLOBAL_VALUE: {
2906 const GlobalValue *GV =
nullptr;
2908 if (
I.getOperand(1).isSymbol()) {
2909 OpFlags =
I.getOperand(1).getTargetFlags();
2915 return selectTLSGlobalValue(
I, MRI);
2921 bool IsGOTSigned = MF.
getInfo<AArch64FunctionInfo>()->hasELFSignedGOT();
2922 I.setDesc(
TII.get(IsGOTSigned ? AArch64::LOADgotAUTH : AArch64::LOADgot));
2923 I.getOperand(1).setTargetFlags(OpFlags);
2924 I.addImplicitDefUseOperands(MF);
2925 I.setImplicitPhysRegDefsDead();
2929 materializeLargeCMVal(
I, GV, OpFlags);
2930 I.eraseFromParent();
2933 I.setDesc(
TII.get(AArch64::ADR));
2934 I.getOperand(1).setTargetFlags(OpFlags);
2936 I.setDesc(
TII.get(AArch64::MOVaddr));
2938 MachineInstrBuilder MIB(MF,
I);
2939 MIB.addGlobalAddress(GV,
I.getOperand(1).getOffset(),
2946 case TargetOpcode::G_PTRAUTH_GLOBAL_VALUE:
2947 return selectPtrAuthGlobalValue(
I, MRI);
2949 case TargetOpcode::G_ZEXTLOAD:
2950 case TargetOpcode::G_LOAD:
2951 case TargetOpcode::G_STORE: {
2953 bool IsZExtLoad =
I.getOpcode() == TargetOpcode::G_ZEXTLOAD;
2968 assert(MemSizeInBytes <= 8 &&
2969 "128-bit atomics should already be custom-legalized");
2972 static constexpr unsigned LDAPROpcodes[] = {
2973 AArch64::LDAPRB, AArch64::LDAPRH, AArch64::LDAPRW, AArch64::LDAPRX};
2974 static constexpr unsigned LDAROpcodes[] = {
2975 AArch64::LDARB, AArch64::LDARH, AArch64::LDARW, AArch64::LDARX};
2976 ArrayRef<unsigned> Opcodes =
2977 STI.hasRCPC() && Order != AtomicOrdering::SequentiallyConsistent
2980 I.setDesc(
TII.get(Opcodes[
Log2_32(MemSizeInBytes)]));
2982 static constexpr unsigned Opcodes[] = {AArch64::STLRB, AArch64::STLRH,
2983 AArch64::STLRW, AArch64::STLRX};
2988 MIB.
buildInstr(TargetOpcode::COPY, {NewVal}, {})
2989 .addReg(
I.getOperand(0).getReg(), {}, AArch64::sub_32);
2990 I.getOperand(0).setReg(NewVal);
2992 I.setDesc(
TII.get(Opcodes[
Log2_32(MemSizeInBytes)]));
3000 const RegisterBank &PtrRB = *RBI.
getRegBank(PtrReg, MRI,
TRI);
3003 "Load/Store pointer operand isn't a GPR");
3005 "Load/Store pointer operand isn't a pointer");
3010 LLT ValTy = MRI.
getType(ValReg);
3015 RB.
getID() == AArch64::FPRRegBankID) {
3018 auto *RC = getRegClassForTypeOnBank(MemTy, RB);
3024 .addReg(ValReg, {}, SubReg)
3031 if (RB.
getID() == AArch64::FPRRegBankID) {
3034 auto *RC = getRegClassForTypeOnBank(MemTy, RB);
3044 MIB.
buildInstr(AArch64::SUBREG_TO_REG, {OldDst}, {})
3047 auto SubRegRC = getRegClassForTypeOnBank(MRI.
getType(OldDst), RB);
3056 auto SelectLoadStoreAddressingMode = [&]() -> MachineInstr * {
3058 const unsigned NewOpc =
3060 if (NewOpc ==
I.getOpcode())
3064 selectAddrModeIndexed(
I.getOperand(1), MemSizeInBytes);
3067 I.setDesc(
TII.get(NewOpc));
3073 auto NewInst = MIB.
buildInstr(NewOpc, {}, {},
I.getFlags());
3074 Register CurValReg =
I.getOperand(0).getReg();
3075 IsStore ? NewInst.addUse(CurValReg) : NewInst.addDef(CurValReg);
3076 NewInst.cloneMemRefs(
I);
3077 for (
auto &Fn : *AddrModeFns)
3079 I.eraseFromParent();
3083 MachineInstr *
LoadStore = SelectLoadStoreAddressingMode();
3088 if (Opcode == TargetOpcode::G_STORE) {
3090 LoadStore->getOperand(0).getReg(), MRI);
3091 if (CVal && CVal->Value == 0) {
3093 case AArch64::STRWui:
3094 case AArch64::STRHHui:
3095 case AArch64::STRBBui:
3096 LoadStore->getOperand(0).setReg(AArch64::WZR);
3098 case AArch64::STRXui:
3099 LoadStore->getOperand(0).setReg(AArch64::XZR);
3105 if (IsZExtLoad || (Opcode == TargetOpcode::G_LOAD &&
3106 ValTy ==
LLT::scalar(64) && MemSizeInBits == 32)) {
3118 MIB.
buildInstr(AArch64::SUBREG_TO_REG, {DstReg}, {})
3120 .
addImm(AArch64::sub_32);
3129 case TargetOpcode::G_INDEXED_ZEXTLOAD:
3130 case TargetOpcode::G_INDEXED_SEXTLOAD:
3131 return selectIndexedExtLoad(
I, MRI);
3132 case TargetOpcode::G_INDEXED_LOAD:
3133 return selectIndexedLoad(
I, MRI);
3134 case TargetOpcode::G_INDEXED_STORE:
3137 case TargetOpcode::G_LSHR:
3138 case TargetOpcode::G_ASHR:
3140 return selectVectorAshrLshr(
I, MRI);
3142 case TargetOpcode::G_SHL: {
3143 if (Opcode == TargetOpcode::G_SHL &&
3145 return selectVectorSHL(
I, MRI);
3152 Register SrcReg =
I.getOperand(1).getReg();
3153 Register ShiftReg =
I.getOperand(2).getReg();
3154 const LLT ShiftTy = MRI.
getType(ShiftReg);
3155 const LLT SrcTy = MRI.
getType(SrcReg);
3160 auto Trunc = MIB.
buildInstr(TargetOpcode::COPY, {SrcTy}, {})
3161 .addReg(ShiftReg, {}, AArch64::sub_32);
3163 I.getOperand(2).setReg(Trunc.getReg(0));
3168 const Register DefReg =
I.getOperand(0).getReg();
3172 if (NewOpc ==
I.getOpcode())
3175 I.setDesc(
TII.get(NewOpc));
3183 case TargetOpcode::G_PTR_ADD: {
3184 emitADD(
I.getOperand(0).getReg(),
I.getOperand(1),
I.getOperand(2), MIB);
3185 I.eraseFromParent();
3189 case TargetOpcode::G_SADDE:
3190 case TargetOpcode::G_UADDE:
3191 case TargetOpcode::G_SSUBE:
3192 case TargetOpcode::G_USUBE:
3193 case TargetOpcode::G_SADDO:
3194 case TargetOpcode::G_UADDO:
3195 case TargetOpcode::G_SSUBO:
3196 case TargetOpcode::G_USUBO:
3197 return selectOverflowOp(
I, MRI);
3199 case TargetOpcode::G_PTRMASK: {
3200 Register MaskReg =
I.getOperand(2).getReg();
3207 I.setDesc(
TII.get(AArch64::ANDXri));
3208 I.getOperand(2).ChangeToImmediate(
3214 case TargetOpcode::G_PTRTOINT:
3215 case TargetOpcode::G_TRUNC: {
3216 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
3217 const LLT SrcTy = MRI.
getType(
I.getOperand(1).getReg());
3219 const Register DstReg =
I.getOperand(0).getReg();
3220 const Register SrcReg =
I.getOperand(1).getReg();
3222 const RegisterBank &DstRB = *RBI.
getRegBank(DstReg, MRI,
TRI);
3223 const RegisterBank &SrcRB = *RBI.
getRegBank(SrcReg, MRI,
TRI);
3227 dbgs() <<
"G_TRUNC/G_PTRTOINT input/output on different banks\n");
3231 if (DstRB.
getID() == AArch64::GPRRegBankID) {
3242 LLVM_DEBUG(
dbgs() <<
"Failed to constrain G_TRUNC/G_PTRTOINT\n");
3246 if (DstRC == SrcRC) {
3248 }
else if (Opcode == TargetOpcode::G_TRUNC && DstTy ==
LLT::scalar(32) &&
3252 }
else if (DstRC == &AArch64::GPR32RegClass &&
3253 SrcRC == &AArch64::GPR64RegClass) {
3254 I.getOperand(1).setSubReg(AArch64::sub_32);
3257 dbgs() <<
"Unhandled mismatched classes in G_TRUNC/G_PTRTOINT\n");
3261 I.setDesc(
TII.get(TargetOpcode::COPY));
3263 }
else if (DstRB.
getID() == AArch64::FPRRegBankID) {
3266 I.setDesc(
TII.get(AArch64::XTNv4i16));
3272 MachineInstr *Extract = emitExtractVectorElt(
3276 I.eraseFromParent();
3281 if (Opcode == TargetOpcode::G_PTRTOINT) {
3282 assert(DstTy.
isVector() &&
"Expected an FPR ptrtoint to be a vector");
3283 I.setDesc(
TII.get(TargetOpcode::COPY));
3291 case TargetOpcode::G_ANYEXT: {
3292 if (selectUSMovFromExtend(
I, MRI))
3295 const Register DstReg =
I.getOperand(0).getReg();
3296 const Register SrcReg =
I.getOperand(1).getReg();
3298 const RegisterBank &RBDst = *RBI.
getRegBank(DstReg, MRI,
TRI);
3299 if (RBDst.
getID() != AArch64::GPRRegBankID) {
3301 <<
", expected: GPR\n");
3305 const RegisterBank &RBSrc = *RBI.
getRegBank(SrcReg, MRI,
TRI);
3306 if (RBSrc.
getID() != AArch64::GPRRegBankID) {
3308 <<
", expected: GPR\n");
3315 LLVM_DEBUG(
dbgs() <<
"G_ANYEXT operand has no size, not a gvreg?\n");
3319 if (DstSize != 64 && DstSize > 32) {
3321 <<
", expected: 32 or 64\n");
3331 .
addImm(AArch64::sub_32);
3332 I.getOperand(1).setReg(ExtSrc);
3337 case TargetOpcode::G_ZEXT:
3338 case TargetOpcode::G_SEXT_INREG:
3339 case TargetOpcode::G_SEXT: {
3340 if (selectUSMovFromExtend(
I, MRI))
3343 unsigned Opcode =
I.getOpcode();
3344 const bool IsSigned = Opcode != TargetOpcode::G_ZEXT;
3345 const Register DefReg =
I.getOperand(0).getReg();
3346 Register SrcReg =
I.getOperand(1).getReg();
3347 const LLT DstTy = MRI.
getType(DefReg);
3348 const LLT SrcTy = MRI.
getType(SrcReg);
3354 if (Opcode == TargetOpcode::G_SEXT_INREG)
3355 SrcSize =
I.getOperand(2).getImm();
3361 AArch64::GPRRegBankID &&
3362 "Unexpected ext regbank");
3373 auto *LoadMI =
getOpcodeDef(TargetOpcode::G_LOAD, SrcReg, MRI);
3376 if (LoadMI && IsGPR) {
3377 const MachineMemOperand *MemOp = *LoadMI->memoperands_begin();
3378 unsigned BytesLoaded = MemOp->getSize().getValue();
3385 if (IsGPR && SrcSize == 32 && DstSize == 64) {
3388 const Register ZReg = AArch64::WZR;
3389 MIB.
buildInstr(AArch64::ORRWrs, {SubregToRegSrc}, {ZReg, SrcReg})
3392 MIB.
buildInstr(AArch64::SUBREG_TO_REG, {DefReg}, {})
3393 .addUse(SubregToRegSrc)
3394 .
addImm(AArch64::sub_32);
3398 LLVM_DEBUG(
dbgs() <<
"Failed to constrain G_ZEXT destination\n");
3408 I.eraseFromParent();
3413 if (DstSize == 64) {
3414 if (Opcode != TargetOpcode::G_SEXT_INREG) {
3422 SrcReg = MIB.
buildInstr(AArch64::SUBREG_TO_REG,
3423 {&AArch64::GPR64RegClass}, {})
3429 ExtI = MIB.
buildInstr(IsSigned ? AArch64::SBFMXri : AArch64::UBFMXri,
3433 }
else if (DstSize <= 32) {
3434 ExtI = MIB.
buildInstr(IsSigned ? AArch64::SBFMWri : AArch64::UBFMWri,
3443 I.eraseFromParent();
3447 case TargetOpcode::G_FREEZE:
3450 case TargetOpcode::G_INTTOPTR:
3455 case TargetOpcode::G_BITCAST:
3463 case TargetOpcode::G_SELECT: {
3465 const Register CondReg = Sel.getCondReg();
3467 const Register FReg = Sel.getFalseReg();
3469 if (tryOptSelect(Sel))
3475 auto TstMI = MIB.
buildInstr(AArch64::ANDSWri, {DeadVReg}, {CondReg})
3480 Sel.eraseFromParent();
3483 case TargetOpcode::G_ICMP: {
3493 auto &PredOp =
I.getOperand(1);
3494 emitIntegerCompare(
I.getOperand(2),
I.getOperand(3), PredOp, MIB);
3498 emitCSINC(
I.getOperand(0).getReg(), AArch64::WZR,
3499 AArch64::WZR, InvCC, MIB);
3500 I.eraseFromParent();
3504 case TargetOpcode::G_FCMP: {
3507 if (!emitFPCompare(
I.getOperand(2).getReg(),
I.getOperand(3).getReg(), MIB,
3509 !emitCSetForFCmp(
I.getOperand(0).getReg(), Pred, MIB))
3511 I.eraseFromParent();
3514 case TargetOpcode::G_VASTART:
3516 : selectVaStartAAPCS(
I, MF, MRI);
3517 case TargetOpcode::G_INTRINSIC:
3518 return selectIntrinsic(
I, MRI);
3519 case TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS:
3520 return selectIntrinsicWithSideEffects(
I, MRI);
3521 case TargetOpcode::G_IMPLICIT_DEF: {
3522 I.setDesc(
TII.get(TargetOpcode::IMPLICIT_DEF));
3523 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
3524 const Register DstReg =
I.getOperand(0).getReg();
3525 const RegisterBank &DstRB = *RBI.
getRegBank(DstReg, MRI,
TRI);
3530 case TargetOpcode::G_BLOCK_ADDR: {
3531 Function *BAFn =
I.getOperand(1).getBlockAddress()->getFunction();
3532 if (std::optional<uint16_t> BADisc =
3542 AArch64::GPR64RegClass, MRI);
3543 I.eraseFromParent();
3547 materializeLargeCMVal(
I,
I.getOperand(1).getBlockAddress(), 0);
3548 I.eraseFromParent();
3551 I.setDesc(
TII.get(AArch64::MOVaddrBA));
3552 auto MovMI =
BuildMI(
MBB,
I,
I.getDebugLoc(),
TII.get(AArch64::MOVaddrBA),
3553 I.getOperand(0).getReg())
3557 I.getOperand(1).getBlockAddress(), 0,
3559 I.eraseFromParent();
3564 case AArch64::G_DUP: {
3571 AArch64::GPRRegBankID)
3573 LLT VecTy = MRI.
getType(
I.getOperand(0).getReg());
3575 I.setDesc(
TII.get(AArch64::DUPv8i8gpr));
3577 I.setDesc(
TII.get(AArch64::DUPv16i8gpr));
3579 I.setDesc(
TII.get(AArch64::DUPv4i16gpr));
3581 I.setDesc(
TII.get(AArch64::DUPv8i16gpr));
3587 case TargetOpcode::G_BUILD_VECTOR:
3588 return selectBuildVector(
I, MRI);
3589 case TargetOpcode::G_MERGE_VALUES:
3591 case TargetOpcode::G_UNMERGE_VALUES:
3593 case TargetOpcode::G_SHUFFLE_VECTOR:
3594 return selectShuffleVector(
I, MRI);
3595 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
3596 return selectExtractElt(
I, MRI);
3597 case TargetOpcode::G_CONCAT_VECTORS:
3598 return selectConcatVectors(
I, MRI);
3599 case TargetOpcode::G_JUMP_TABLE:
3600 return selectJumpTable(
I, MRI);
3601 case TargetOpcode::G_MEMCPY:
3602 case TargetOpcode::G_MEMCPY_INLINE:
3603 case TargetOpcode::G_MEMMOVE:
3604 case TargetOpcode::G_MEMSET:
3605 case TargetOpcode::G_MEMSET_INLINE:
3606 assert(STI.hasMOPS() &&
"Shouldn't get here without +mops feature");
3607 return selectMOPS(
I, MRI);
3613bool AArch64InstructionSelector::selectAndRestoreState(MachineInstr &
I) {
3614 MachineIRBuilderState OldMIBState = MIB.
getState();
3620bool AArch64InstructionSelector::selectMOPS(MachineInstr &GI,
3621 MachineRegisterInfo &MRI) {
3624 case TargetOpcode::G_MEMCPY:
3625 case TargetOpcode::G_MEMCPY_INLINE:
3626 Mopcode = AArch64::MOPSMemoryCopyPseudo;
3628 case TargetOpcode::G_MEMMOVE:
3629 Mopcode = AArch64::MOPSMemoryMovePseudo;
3631 case TargetOpcode::G_MEMSET:
3632 case TargetOpcode::G_MEMSET_INLINE:
3634 Mopcode = AArch64::MOPSMemorySetPseudo;
3647 const bool IsSet = Mopcode == AArch64::MOPSMemorySetPseudo;
3648 const auto &SrcValRegClass =
3649 IsSet ? AArch64::GPR64RegClass : AArch64::GPR64commonRegClass;
3667 MIB.
buildInstr(Mopcode, {DefDstPtr, DefSize},
3668 {DstPtrCopy, SizeCopy, SrcValCopy})
3672 MIB.
buildInstr(Mopcode, {DefDstPtr, DefSrcPtr, DefSize},
3673 {DstPtrCopy, SrcValCopy, SizeCopy})
3681bool AArch64InstructionSelector::selectBrJT(MachineInstr &
I,
3682 MachineRegisterInfo &MRI) {
3683 assert(
I.getOpcode() == TargetOpcode::G_BRJT &&
"Expected G_BRJT");
3684 Register JTAddr =
I.getOperand(0).getReg();
3685 unsigned JTI =
I.getOperand(1).getIndex();
3688 MF->
getInfo<AArch64FunctionInfo>()->setJumpTableEntryInfo(JTI, 4,
nullptr);
3700 "jump table hardening only supported on MachO/ELF");
3708 I.eraseFromParent();
3715 auto JumpTableInst = MIB.
buildInstr(AArch64::JumpTableDest32,
3716 {TargetReg, ScratchReg}, {JTAddr,
Index})
3717 .addJumpTableIndex(JTI);
3719 MIB.
buildInstr(TargetOpcode::JUMP_TABLE_DEBUG_INFO, {},
3720 {
static_cast<int64_t
>(JTI)});
3722 MIB.
buildInstr(AArch64::BR, {}, {TargetReg});
3723 I.eraseFromParent();
3728bool AArch64InstructionSelector::selectJumpTable(MachineInstr &
I,
3729 MachineRegisterInfo &MRI) {
3730 assert(
I.getOpcode() == TargetOpcode::G_JUMP_TABLE &&
"Expected jump table");
3731 assert(
I.getOperand(1).isJTI() &&
"Jump table op should have a JTI!");
3733 Register DstReg =
I.getOperand(0).getReg();
3734 unsigned JTI =
I.getOperand(1).getIndex();
3737 MIB.
buildInstr(AArch64::MOVaddrJT, {DstReg}, {})
3740 I.eraseFromParent();
3745bool AArch64InstructionSelector::selectTLSLocalExecELF(
3746 const GlobalValue *GV, MachineInstr &
I, MachineRegisterInfo &MRI) {
3747 auto ConstrainRegOps = [&](MachineInstrBuilder MIB) {
3751 ConstrainRegOps(MIB.
buildInstr(AArch64::MOVbaseTLS, {ThreadBase}, {}));
3759 MIB.
buildInstr(AArch64::ADDXri, {I.getOperand(0).getReg()},
3770 MIB.
buildInstr(AArch64::ADDXri, {Addr}, {ThreadBase})
3774 MIB.
buildInstr(AArch64::ADDXri, {I.getOperand(0).getReg()}, {Addr})
3791 ConstrainRegOps(MIB.
buildInstr(AArch64::MOVKXi, {Addr2}, {Addr})
3796 ConstrainRegOps(MIB.
buildInstr(AArch64::ADDXrr, {I.getOperand(0).getReg()},
3797 {ThreadBase, Addr2}));
3811 ConstrainRegOps(MIB.
buildInstr(AArch64::MOVKXi, {Addr2}, {Addr})
3817 ConstrainRegOps(MIB.
buildInstr(AArch64::MOVKXi, {Addr3}, {Addr2})
3822 ConstrainRegOps(MIB.
buildInstr(AArch64::ADDXrr, {I.getOperand(0).getReg()},
3823 {ThreadBase, Addr3}));
3827 I.eraseFromParent();
3833bool AArch64InstructionSelector::selectTLSGlobalValueELF(
3834 MachineInstr &
I, MachineRegisterInfo &MRI) {
3835 const GlobalValue *GV =
I.
getOperand(1).getGlobal();
3836 auto *FuncInfo = MF->
getInfo<AArch64FunctionInfo>();
3843 return selectTLSLocalExecELF(GV,
I, MRI);
3845 MIB.
buildInstr(AArch64::LOADgot, {TPOff}, {})
3851 SMEAttrs
Attrs = MF->
getInfo<AArch64FunctionInfo>()->getSMEFnAttrs();
3853 !
Attrs.hasStreamingCompatibleInterface() &&
3854 "unsupported SME features reached GlobalISel TLS lowering");
3857 ? AArch64::TLSDESC_AUTH_CALLSEQ
3858 : AArch64::TLSDESC_CALLSEQ;
3877 MIB.
buildInstr(AArch64::ADDXri, {TPOff}, {Add1.getReg(0)})
3878 .addGlobalAddress(GV, 0,
3887 MIB.
buildInstr(AArch64::MOVbaseTLS, {ThreadBase}, {});
3888 auto Add = MIB.
buildInstr(AArch64::ADDXrr, {
I.getOperand(0).getReg()},
3889 {ThreadBase, TPOff});
3892 I.eraseFromParent();
3896bool AArch64InstructionSelector::selectTLSGlobalValueMachO(
3897 MachineInstr &
I, MachineRegisterInfo &MRI) {
3898 const auto &GlobalOp =
I.getOperand(1);
3899 assert(GlobalOp.getOffset() == 0 &&
3900 "Shouldn't have an offset on TLS globals!");
3902 const GlobalValue &GV = *GlobalOp.getGlobal();
3905 MIB.
buildInstr(AArch64::LOADgot, {&AArch64::GPR64commonRegClass}, {})
3908 auto Load = MIB.
buildInstr(AArch64::LDRXui, {&AArch64::GPR64commonRegClass},
3909 {LoadGOT.getReg(0)})
3920 assert(Opcode == AArch64::BLR);
3921 Opcode = AArch64::BLRAAZ;
3926 .
addUse(AArch64::X0, RegState::Implicit)
3927 .
addDef(AArch64::X0, RegState::Implicit)
3933 I.eraseFromParent();
3937bool AArch64InstructionSelector::selectTLSGlobalValue(
3938 MachineInstr &
I, MachineRegisterInfo &MRI) {
3944 return selectTLSGlobalValueELF(
I, MRI);
3947 return selectTLSGlobalValueMachO(
I, MRI);
3952MachineInstr *AArch64InstructionSelector::emitScalarToVector(
3954 MachineIRBuilder &MIRBuilder)
const {
3955 auto Undef = MIRBuilder.
buildInstr(TargetOpcode::IMPLICIT_DEF, {DstRC}, {});
3957 auto BuildFn = [&](
unsigned SubregIndex) {
3961 .addImm(SubregIndex);
3969 return BuildFn(AArch64::bsub);
3971 return BuildFn(AArch64::hsub);
3973 return BuildFn(AArch64::ssub);
3975 return BuildFn(AArch64::dsub);
3982AArch64InstructionSelector::emitNarrowVector(
Register DstReg,
Register SrcReg,
3983 MachineIRBuilder &MIB,
3984 MachineRegisterInfo &MRI)
const {
3985 LLT DstTy = MRI.
getType(DstReg);
3987 getRegClassForTypeOnBank(DstTy, *RBI.
getRegBank(SrcReg, MRI,
TRI));
3988 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
3992 unsigned SubReg = 0;
3995 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
4001 .addReg(SrcReg, {}, SubReg);
4006bool AArch64InstructionSelector::selectMergeValues(
4007 MachineInstr &
I, MachineRegisterInfo &MRI) {
4008 assert(
I.getOpcode() == TargetOpcode::G_MERGE_VALUES &&
"unexpected opcode");
4009 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
4010 const LLT SrcTy = MRI.
getType(
I.getOperand(1).getReg());
4012 const RegisterBank &RB = *RBI.
getRegBank(
I.getOperand(1).getReg(), MRI,
TRI);
4014 if (
I.getNumOperands() != 3)
4021 Register DstReg =
I.getOperand(0).getReg();
4022 Register Src1Reg =
I.getOperand(1).getReg();
4023 Register Src2Reg =
I.getOperand(2).getReg();
4024 auto Tmp = MIB.
buildInstr(TargetOpcode::IMPLICIT_DEF, {DstTy}, {});
4025 MachineInstr *InsMI = emitLaneInsert(std::nullopt, Tmp.getReg(0), Src1Reg,
4029 MachineInstr *Ins2MI = emitLaneInsert(DstReg, InsMI->
getOperand(0).
getReg(),
4030 Src2Reg, 1, RB, MIB);
4035 I.eraseFromParent();
4039 if (RB.
getID() != AArch64::GPRRegBankID)
4045 auto *DstRC = &AArch64::GPR64RegClass;
4047 MachineInstr &SubRegMI = *
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
4048 TII.get(TargetOpcode::SUBREG_TO_REG))
4050 .
addUse(
I.getOperand(1).getReg())
4051 .
addImm(AArch64::sub_32);
4054 MachineInstr &SubRegMI2 = *
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
4055 TII.get(TargetOpcode::SUBREG_TO_REG))
4057 .
addUse(
I.getOperand(2).getReg())
4058 .
addImm(AArch64::sub_32);
4060 *
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
TII.get(AArch64::BFMXri))
4061 .
addDef(
I.getOperand(0).getReg())
4069 I.eraseFromParent();
4074 const unsigned EltSize) {
4079 CopyOpc = AArch64::DUPi8;
4080 ExtractSubReg = AArch64::bsub;
4083 CopyOpc = AArch64::DUPi16;
4084 ExtractSubReg = AArch64::hsub;
4087 CopyOpc = AArch64::DUPi32;
4088 ExtractSubReg = AArch64::ssub;
4091 CopyOpc = AArch64::DUPi64;
4092 ExtractSubReg = AArch64::dsub;
4096 LLVM_DEBUG(
dbgs() <<
"Elt size '" << EltSize <<
"' unsupported.\n");
4102MachineInstr *AArch64InstructionSelector::emitExtractVectorElt(
4103 std::optional<Register> DstReg,
const RegisterBank &DstRB, LLT ScalarTy,
4104 Register VecReg,
unsigned LaneIdx, MachineIRBuilder &MIRBuilder)
const {
4105 MachineRegisterInfo &MRI = *MIRBuilder.
getMRI();
4106 unsigned CopyOpc = 0;
4107 unsigned ExtractSubReg = 0;
4110 dbgs() <<
"Couldn't determine lane copy opcode for instruction.\n");
4115 getRegClassForTypeOnBank(ScalarTy, DstRB,
true);
4117 LLVM_DEBUG(
dbgs() <<
"Could not determine destination register class.\n");
4121 const RegisterBank &VecRB = *RBI.
getRegBank(VecReg, MRI,
TRI);
4122 const LLT &VecTy = MRI.
getType(VecReg);
4124 getRegClassForTypeOnBank(VecTy, VecRB,
true);
4126 LLVM_DEBUG(
dbgs() <<
"Could not determine source register class.\n");
4136 auto Copy = MIRBuilder.
buildInstr(TargetOpcode::COPY, {*DstReg}, {})
4137 .addReg(VecReg, {}, ExtractSubReg);
4146 MachineInstr *ScalarToVector = emitScalarToVector(
4147 VecTy.
getSizeInBits(), &AArch64::FPR128RegClass, VecReg, MIRBuilder);
4148 if (!ScalarToVector)
4153 MachineInstr *LaneCopyMI =
4154 MIRBuilder.
buildInstr(CopyOpc, {*DstReg}, {InsertReg}).addImm(LaneIdx);
4162bool AArch64InstructionSelector::selectExtractElt(
4163 MachineInstr &
I, MachineRegisterInfo &MRI) {
4164 assert(
I.getOpcode() == TargetOpcode::G_EXTRACT_VECTOR_ELT &&
4165 "unexpected opcode!");
4166 Register DstReg =
I.getOperand(0).getReg();
4167 const LLT NarrowTy = MRI.
getType(DstReg);
4168 const Register SrcReg =
I.getOperand(1).getReg();
4169 const LLT WideTy = MRI.
getType(SrcReg);
4171 "source register size too small!");
4172 assert(!NarrowTy.
isVector() &&
"cannot extract vector into vector!");
4175 MachineOperand &LaneIdxOp =
I.getOperand(2);
4176 assert(LaneIdxOp.
isReg() &&
"Lane index operand was not a register?");
4182 unsigned LaneIdx = VRegAndVal->Value.getSExtValue();
4184 const RegisterBank &DstRB = *RBI.
getRegBank(DstReg, MRI,
TRI);
4185 if (DstRB.
getID() == AArch64::GPRRegBankID) {
4189 Opcode = AArch64::UMOVvi8;
4192 Opcode = AArch64::UMOVvi16;
4195 Opcode = AArch64::UMOVvi32;
4202 MachineInstr *ScalarToVector = emitScalarToVector(
4203 WideTy.
getSizeInBits(), &AArch64::FPR128RegClass, SrcReg, MIB);
4204 assert(ScalarToVector &&
"Didn't expect emitScalarToVector to fail!");
4208 I.setDesc(
TII.get(Opcode));
4209 I.getOperand(2).ChangeToImmediate(LaneIdx);
4214 MachineInstr *Extract = emitExtractVectorElt(DstReg, DstRB, NarrowTy, SrcReg,
4219 I.eraseFromParent();
4223bool AArch64InstructionSelector::selectSplitVectorUnmerge(
4224 MachineInstr &
I, MachineRegisterInfo &MRI) {
4225 unsigned NumElts =
I.getNumOperands() - 1;
4226 Register SrcReg =
I.getOperand(NumElts).getReg();
4227 const LLT NarrowTy = MRI.
getType(
I.getOperand(0).getReg());
4228 const LLT SrcTy = MRI.
getType(SrcReg);
4230 assert(NarrowTy.
isVector() &&
"Expected an unmerge into vectors");
4232 LLVM_DEBUG(
dbgs() <<
"Unexpected vector type for vec split unmerge");
4238 const RegisterBank &DstRB =
4240 for (
unsigned OpIdx = 0; OpIdx < NumElts; ++OpIdx) {
4241 Register Dst =
I.getOperand(OpIdx).getReg();
4242 MachineInstr *Extract =
4243 emitExtractVectorElt(Dst, DstRB, NarrowTy, SrcReg, OpIdx, MIB);
4247 I.eraseFromParent();
4251bool AArch64InstructionSelector::selectUnmergeValues(MachineInstr &
I,
4252 MachineRegisterInfo &MRI) {
4253 assert(
I.getOpcode() == TargetOpcode::G_UNMERGE_VALUES &&
4254 "unexpected opcode");
4258 unsigned NumElts =
I.getNumOperands() - 1;
4259 Register SrcReg =
I.getOperand(NumElts).getReg();
4260 Register LoReg =
I.getOperand(0).getReg();
4261 Register HiReg =
I.getOperand(1).getReg();
4262 const LLT NarrowTy = MRI.
getType(LoReg);
4263 const LLT WideTy = MRI.
getType(SrcReg);
4264 const RegisterBank &LoRB = *RBI.
getRegBank(LoReg, MRI,
TRI);
4265 const RegisterBank &HiRB = *RBI.
getRegBank(HiReg, MRI,
TRI);
4266 const RegisterBank &SrcRB = *RBI.
getRegBank(SrcReg, MRI,
TRI);
4270 LoRB.
getID() == AArch64::GPRRegBankID &&
4271 HiRB.
getID() == AArch64::GPRRegBankID &&
4272 SrcRB.
getID() == AArch64::FPRRegBankID) {
4273 MachineInstr &
Lo = *
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
4274 TII.get(AArch64::UMOVvi64), LoReg)
4277 MachineInstr &
Hi = *
BuildMI(*
I.getParent(),
I,
I.getDebugLoc(),
4278 TII.get(AArch64::UMOVvi64), HiReg)
4283 I.eraseFromParent();
4288 if (LoRB.
getID() != AArch64::FPRRegBankID ||
4289 HiRB.
getID() != AArch64::FPRRegBankID) {
4290 LLVM_DEBUG(
dbgs() <<
"Unmerging vector-to-gpr and scalar-to-scalar "
4291 "currently unsupported.\n");
4296 "source register size too small!");
4299 return selectSplitVectorUnmerge(
I, MRI);
4303 unsigned CopyOpc = 0;
4304 unsigned ExtractSubReg = 0;
4315 unsigned NumInsertRegs = NumElts - 1;
4321 InsertRegs.
assign(NumInsertRegs, SrcReg);
4330 unsigned SubReg = 0;
4333 assert(Found &&
"expected to find last operand's subeg idx");
4334 for (
unsigned Idx = 0; Idx < NumInsertRegs; ++Idx) {
4336 MachineInstr &ImpDefMI =
4337 *
BuildMI(
MBB,
I,
I.getDebugLoc(),
TII.get(TargetOpcode::IMPLICIT_DEF),
4342 MachineInstr &InsMI =
4344 TII.get(TargetOpcode::INSERT_SUBREG), InsertReg)
4361 Register CopyTo =
I.getOperand(0).getReg();
4362 auto FirstCopy = MIB.
buildInstr(TargetOpcode::COPY, {CopyTo}, {})
4363 .addReg(InsertRegs[0], {}, ExtractSubReg);
4367 unsigned LaneIdx = 1;
4368 for (
Register InsReg : InsertRegs) {
4369 Register CopyTo =
I.getOperand(LaneIdx).getReg();
4370 MachineInstr &CopyInst =
4389 I.eraseFromParent();
4393bool AArch64InstructionSelector::selectConcatVectors(
4394 MachineInstr &
I, MachineRegisterInfo &MRI) {
4395 assert(
I.getOpcode() == TargetOpcode::G_CONCAT_VECTORS &&
4396 "Unexpected opcode");
4397 Register Dst =
I.getOperand(0).getReg();
4398 Register Op1 =
I.getOperand(1).getReg();
4399 Register Op2 =
I.getOperand(2).getReg();
4400 MachineInstr *ConcatMI = emitVectorConcat(Dst, Op1, Op2, MIB);
4403 I.eraseFromParent();
4408AArch64InstructionSelector::emitConstantPoolEntry(
const Constant *CPVal,
4417MachineInstr *AArch64InstructionSelector::emitLoadFromConstantPool(
4418 const Constant *CPVal, MachineIRBuilder &MIRBuilder)
const {
4425 RC = &AArch64::FPR128RegClass;
4426 Opc = IsTiny ? AArch64::LDRQl : AArch64::LDRQui;
4429 RC = &AArch64::FPR64RegClass;
4430 Opc = IsTiny ? AArch64::LDRDl : AArch64::LDRDui;
4433 RC = &AArch64::FPR32RegClass;
4434 Opc = IsTiny ? AArch64::LDRSl : AArch64::LDRSui;
4437 RC = &AArch64::FPR16RegClass;
4438 Opc = AArch64::LDRHui;
4441 LLVM_DEBUG(
dbgs() <<
"Could not load from constant pool of type "
4446 MachineInstr *LoadMI =
nullptr;
4447 auto &MF = MIRBuilder.
getMF();
4448 unsigned CPIdx = emitConstantPoolEntry(CPVal, MF);
4449 if (IsTiny && (
Size == 16 ||
Size == 8 ||
Size == 4)) {
4451 LoadMI = &*MIRBuilder.
buildInstr(
Opc, {RC}, {}).addConstantPoolIndex(CPIdx);
4454 MIRBuilder.
buildInstr(AArch64::ADRP, {&AArch64::GPR64RegClass}, {})
4458 .addConstantPoolIndex(
4474static std::pair<unsigned, unsigned>
4476 unsigned Opc, SubregIdx;
4477 if (RB.
getID() == AArch64::GPRRegBankID) {
4479 Opc = AArch64::INSvi8gpr;
4480 SubregIdx = AArch64::bsub;
4481 }
else if (EltSize == 16) {
4482 Opc = AArch64::INSvi16gpr;
4483 SubregIdx = AArch64::ssub;
4484 }
else if (EltSize == 32) {
4485 Opc = AArch64::INSvi32gpr;
4486 SubregIdx = AArch64::ssub;
4487 }
else if (EltSize == 64) {
4488 Opc = AArch64::INSvi64gpr;
4489 SubregIdx = AArch64::dsub;
4495 Opc = AArch64::INSvi8lane;
4496 SubregIdx = AArch64::bsub;
4497 }
else if (EltSize == 16) {
4498 Opc = AArch64::INSvi16lane;
4499 SubregIdx = AArch64::hsub;
4500 }
else if (EltSize == 32) {
4501 Opc = AArch64::INSvi32lane;
4502 SubregIdx = AArch64::ssub;
4503 }
else if (EltSize == 64) {
4504 Opc = AArch64::INSvi64lane;
4505 SubregIdx = AArch64::dsub;
4510 return std::make_pair(
Opc, SubregIdx);
4513MachineInstr *AArch64InstructionSelector::emitInstr(
4514 unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
4515 std::initializer_list<llvm::SrcOp> SrcOps, MachineIRBuilder &MIRBuilder,
4516 const ComplexRendererFns &RenderFns)
const {
4517 assert(Opcode &&
"Expected an opcode?");
4519 "Function should only be used to produce selected instructions!");
4520 auto MI = MIRBuilder.
buildInstr(Opcode, DstOps, SrcOps);
4522 for (
auto &Fn : *RenderFns)
4528MachineInstr *AArch64InstructionSelector::emitAddSub(
4529 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
4531 MachineIRBuilder &MIRBuilder)
const {
4533 assert(
LHS.isReg() &&
RHS.isReg() &&
"Expected register operands?");
4537 assert((
Size == 32 ||
Size == 64) &&
"Expected a 32-bit or 64-bit type only");
4538 bool Is32Bit =
Size == 32;
4541 if (
auto Fns = selectArithImmed(
RHS))
4542 return emitInstr(AddrModeAndSizeToOpcode[0][Is32Bit], {Dst}, {
LHS},
4546 if (
auto Fns = selectNegArithImmed(
RHS))
4547 return emitInstr(AddrModeAndSizeToOpcode[3][Is32Bit], {Dst}, {
LHS},
4551 if (
auto Fns = selectArithExtendedRegister(
RHS))
4552 return emitInstr(AddrModeAndSizeToOpcode[4][Is32Bit], {Dst}, {
LHS},
4556 if (
auto Fns = selectShiftedRegister(
RHS))
4557 return emitInstr(AddrModeAndSizeToOpcode[1][Is32Bit], {Dst}, {
LHS},
4559 return emitInstr(AddrModeAndSizeToOpcode[2][Is32Bit], {Dst}, {
LHS,
RHS},
4564AArch64InstructionSelector::emitADD(
Register DefReg, MachineOperand &
LHS,
4565 MachineOperand &
RHS,
4566 MachineIRBuilder &MIRBuilder)
const {
4567 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4568 {{AArch64::ADDXri, AArch64::ADDWri},
4569 {AArch64::ADDXrs, AArch64::ADDWrs},
4570 {AArch64::ADDXrr, AArch64::ADDWrr},
4571 {AArch64::SUBXri, AArch64::SUBWri},
4572 {AArch64::ADDXrx, AArch64::ADDWrx}}};
4573 return emitAddSub(OpcTable, DefReg,
LHS,
RHS, MIRBuilder);
4577AArch64InstructionSelector::emitADDS(
Register Dst, MachineOperand &
LHS,
4578 MachineOperand &
RHS,
4579 MachineIRBuilder &MIRBuilder)
const {
4580 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4581 {{AArch64::ADDSXri, AArch64::ADDSWri},
4582 {AArch64::ADDSXrs, AArch64::ADDSWrs},
4583 {AArch64::ADDSXrr, AArch64::ADDSWrr},
4584 {AArch64::SUBSXri, AArch64::SUBSWri},
4585 {AArch64::ADDSXrx, AArch64::ADDSWrx}}};
4586 return emitAddSub(OpcTable, Dst,
LHS,
RHS, MIRBuilder);
4590AArch64InstructionSelector::emitSUBS(
Register Dst, MachineOperand &
LHS,
4591 MachineOperand &
RHS,
4592 MachineIRBuilder &MIRBuilder)
const {
4593 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4594 {{AArch64::SUBSXri, AArch64::SUBSWri},
4595 {AArch64::SUBSXrs, AArch64::SUBSWrs},
4596 {AArch64::SUBSXrr, AArch64::SUBSWrr},
4597 {AArch64::ADDSXri, AArch64::ADDSWri},
4598 {AArch64::SUBSXrx, AArch64::SUBSWrx}}};
4599 return emitAddSub(OpcTable, Dst,
LHS,
RHS, MIRBuilder);
4603AArch64InstructionSelector::emitADCS(
Register Dst, MachineOperand &
LHS,
4604 MachineOperand &
RHS,
4605 MachineIRBuilder &MIRBuilder)
const {
4606 assert(
LHS.isReg() &&
RHS.isReg() &&
"Expected register operands?");
4607 MachineRegisterInfo *MRI = MIRBuilder.
getMRI();
4609 static const unsigned OpcTable[2] = {AArch64::ADCSXr, AArch64::ADCSWr};
4610 return emitInstr(OpcTable[Is32Bit], {Dst}, {
LHS,
RHS}, MIRBuilder);
4614AArch64InstructionSelector::emitSBCS(
Register Dst, MachineOperand &
LHS,
4615 MachineOperand &
RHS,
4616 MachineIRBuilder &MIRBuilder)
const {
4617 assert(
LHS.isReg() &&
RHS.isReg() &&
"Expected register operands?");
4618 MachineRegisterInfo *MRI = MIRBuilder.
getMRI();
4620 static const unsigned OpcTable[2] = {AArch64::SBCSXr, AArch64::SBCSWr};
4621 return emitInstr(OpcTable[Is32Bit], {Dst}, {
LHS,
RHS}, MIRBuilder);
4625AArch64InstructionSelector::emitCMP(MachineOperand &
LHS, MachineOperand &
RHS,
4626 MachineIRBuilder &MIRBuilder)
const {
4629 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4634AArch64InstructionSelector::emitCMN(MachineOperand &
LHS, MachineOperand &
RHS,
4635 MachineIRBuilder &MIRBuilder)
const {
4638 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4643AArch64InstructionSelector::emitTST(MachineOperand &
LHS, MachineOperand &
RHS,
4644 MachineIRBuilder &MIRBuilder)
const {
4645 assert(
LHS.isReg() &&
RHS.isReg() &&
"Expected register operands?");
4649 bool Is32Bit = (
RegSize == 32);
4650 const unsigned OpcTable[3][2] = {{AArch64::ANDSXri, AArch64::ANDSWri},
4651 {AArch64::ANDSXrs, AArch64::ANDSWrs},
4652 {AArch64::ANDSXrr, AArch64::ANDSWrr}};
4656 int64_t
Imm = ValAndVReg->Value.getSExtValue();
4659 auto TstMI = MIRBuilder.
buildInstr(OpcTable[0][Is32Bit], {Ty}, {
LHS});
4666 if (
auto Fns = selectLogicalShiftedRegister(
RHS))
4667 return emitInstr(OpcTable[1][Is32Bit], {Ty}, {
LHS}, MIRBuilder, Fns);
4668 return emitInstr(OpcTable[2][Is32Bit], {Ty}, {
LHS,
RHS}, MIRBuilder);
4671MachineInstr *AArch64InstructionSelector::emitIntegerCompare(
4672 MachineOperand &
LHS, MachineOperand &
RHS, MachineOperand &Predicate,
4673 MachineIRBuilder &MIRBuilder)
const {
4674 assert(
LHS.isReg() &&
RHS.isReg() &&
"Expected LHS and RHS to be registers!");
4681 assert((
Size == 32 ||
Size == 64) &&
"Expected a 32-bit or 64-bit LHS/RHS?");
4683 if (
auto FoldCmp = tryFoldIntegerCompare(
LHS,
RHS, Predicate, MIRBuilder))
4685 return emitCMP(
LHS,
RHS, MIRBuilder);
4688MachineInstr *AArch64InstructionSelector::emitCSetForFCmp(
4690 MachineRegisterInfo &MRI = *MIRBuilder.
getMRI();
4694 "Expected a 32-bit scalar register?");
4696 const Register ZReg = AArch64::WZR;
4701 return emitCSINC(Dst, ZReg, ZReg, InvCC1,
4707 emitCSINC(Def1Reg, ZReg, ZReg, InvCC1, MIRBuilder);
4708 emitCSINC(Def2Reg, ZReg, ZReg, InvCC2, MIRBuilder);
4709 auto OrMI = MIRBuilder.
buildInstr(AArch64::ORRWrr, {Dst}, {Def1Reg, Def2Reg});
4714MachineInstr *AArch64InstructionSelector::emitFPCompare(
4716 std::optional<CmpInst::Predicate> Pred)
const {
4717 MachineRegisterInfo &MRI = *MIRBuilder.
getMRI();
4722 assert(OpSize == 16 || OpSize == 32 || OpSize == 64);
4732 if (!ShouldUseImm && Pred && IsEqualityPred(*Pred)) {
4735 ShouldUseImm =
true;
4739 unsigned CmpOpcTbl[2][3] = {
4740 {AArch64::FCMPHrr, AArch64::FCMPSrr, AArch64::FCMPDrr},
4741 {AArch64::FCMPHri, AArch64::FCMPSri, AArch64::FCMPDri}};
4743 CmpOpcTbl[ShouldUseImm][OpSize == 16 ? 0 : (OpSize == 32 ? 1 : 2)];
4755MachineInstr *AArch64InstructionSelector::emitVectorConcat(
4757 MachineIRBuilder &MIRBuilder)
const {
4764 const LLT Op1Ty = MRI.
getType(Op1);
4765 const LLT Op2Ty = MRI.
getType(Op2);
4767 if (Op1Ty != Op2Ty) {
4768 LLVM_DEBUG(
dbgs() <<
"Could not do vector concat of differing vector tys");
4771 assert(Op1Ty.
isVector() &&
"Expected a vector for vector concat");
4774 LLVM_DEBUG(
dbgs() <<
"Vector concat not supported for full size vectors");
4785 const RegisterBank &FPRBank = *RBI.
getRegBank(Op1, MRI,
TRI);
4789 MachineInstr *WidenedOp1 =
4790 emitScalarToVector(ScalarTy.
getSizeInBits(), DstRC, Op1, MIRBuilder);
4791 MachineInstr *WidenedOp2 =
4792 emitScalarToVector(ScalarTy.
getSizeInBits(), DstRC, Op2, MIRBuilder);
4793 if (!WidenedOp1 || !WidenedOp2) {
4794 LLVM_DEBUG(
dbgs() <<
"Could not emit a vector from scalar value");
4799 unsigned InsertOpc, InsSubRegIdx;
4800 std::tie(InsertOpc, InsSubRegIdx) =
4818 MachineIRBuilder &MIRBuilder)
const {
4819 auto &MRI = *MIRBuilder.
getMRI();
4825 Size =
TRI.getRegSizeInBits(*RC);
4829 assert(
Size <= 64 &&
"Expected 64 bits or less only!");
4830 static const unsigned OpcTable[2] = {AArch64::CSINCWr, AArch64::CSINCXr};
4831 unsigned Opc = OpcTable[
Size == 64];
4832 auto CSINC = MIRBuilder.
buildInstr(
Opc, {Dst}, {Src1, Src2}).addImm(Pred);
4837MachineInstr *AArch64InstructionSelector::emitCarryIn(MachineInstr &
I,
4839 MachineRegisterInfo *MRI = MIB.
getMRI();
4840 unsigned Opcode =
I.getOpcode();
4844 bool NeedsNegatedCarry =
4845 (Opcode == TargetOpcode::G_USUBE || Opcode == TargetOpcode::G_SSUBE);
4854 MachineInstr *SrcMI = MRI->
getVRegDef(CarryReg);
4855 if (SrcMI ==
I.getPrevNode()) {
4857 bool ProducesNegatedCarry = CarrySrcMI->isSub();
4858 if (NeedsNegatedCarry == ProducesNegatedCarry &&
4859 CarrySrcMI->isUnsigned() &&
4860 CarrySrcMI->getCarryOutReg() == CarryReg &&
4861 selectAndRestoreState(*SrcMI))
4868 if (NeedsNegatedCarry) {
4871 return emitInstr(AArch64::SUBSWrr, {DeadReg}, {ZReg, CarryReg}, MIB);
4875 auto Fns = select12BitValueWithLeftShift(1);
4876 return emitInstr(AArch64::SUBSWri, {DeadReg}, {CarryReg}, MIB, Fns);
4879bool AArch64InstructionSelector::selectOverflowOp(MachineInstr &
I,
4880 MachineRegisterInfo &MRI) {
4885 emitCarryIn(
I, CarryInMI->getCarryInReg());
4889 auto OpAndCC = emitOverflowOp(
I.getOpcode(), CarryMI.getDstReg(),
4890 CarryMI.getLHS(), CarryMI.getRHS(), MIB);
4892 Register CarryOutReg = CarryMI.getCarryOutReg();
4896 OpAndCC.first->addRegisterDead(AArch64::NZCV, &
TRI);
4903 emitCSINC(CarryOutReg, ZReg, ZReg,
4904 getInvertedCondCode(OpAndCC.second), MIB);
4907 I.eraseFromParent();
4911std::pair<MachineInstr *, AArch64CC::CondCode>
4912AArch64InstructionSelector::emitOverflowOp(
unsigned Opcode,
Register Dst,
4913 MachineOperand &
LHS,
4914 MachineOperand &
RHS,
4915 MachineIRBuilder &MIRBuilder)
const {
4919 case TargetOpcode::G_SADDO:
4921 case TargetOpcode::G_UADDO:
4923 case TargetOpcode::G_SSUBO:
4925 case TargetOpcode::G_USUBO:
4927 case TargetOpcode::G_SADDE:
4929 case TargetOpcode::G_UADDE:
4931 case TargetOpcode::G_SSUBE:
4933 case TargetOpcode::G_USUBE:
4954 unsigned Depth = 0) {
4961 MustBeFirst =
false;
4967 if (Opcode == TargetOpcode::G_AND || Opcode == TargetOpcode::G_OR) {
4968 bool IsOR = Opcode == TargetOpcode::G_OR;
4980 if (MustBeFirstL && MustBeFirstR)
4986 if (!CanNegateL && !CanNegateR)
4990 CanNegate = WillNegate && CanNegateL && CanNegateR;
4993 MustBeFirst = !CanNegate;
4995 assert(Opcode == TargetOpcode::G_AND &&
"Must be G_AND");
4998 MustBeFirst = MustBeFirstL || MustBeFirstR;
5005MachineInstr *AArch64InstructionSelector::emitConditionalComparison(
5008 MachineIRBuilder &MIB)
const {
5009 auto &MRI = *MIB.
getMRI();
5012 std::optional<ValueAndVReg>
C;
5016 if (!
C ||
C->Value.sgt(31) ||
C->Value.slt(-31))
5017 CCmpOpc = OpTy.
getSizeInBits() == 32 ? AArch64::CCMPWr : AArch64::CCMPXr;
5018 else if (
C->Value.ule(31))
5019 CCmpOpc = OpTy.
getSizeInBits() == 32 ? AArch64::CCMPWi : AArch64::CCMPXi;
5021 CCmpOpc = OpTy.
getSizeInBits() == 32 ? AArch64::CCMNWi : AArch64::CCMNXi;
5027 assert(STI.hasFullFP16() &&
"Expected Full FP16 for fp16 comparisons");
5028 CCmpOpc = AArch64::FCCMPHrr;
5031 CCmpOpc = AArch64::FCCMPSrr;
5034 CCmpOpc = AArch64::FCCMPDrr;
5044 if (CCmpOpc == AArch64::CCMPWi || CCmpOpc == AArch64::CCMPXi)
5045 CCmp.
addImm(
C->Value.getZExtValue());
5046 else if (CCmpOpc == AArch64::CCMNWi || CCmpOpc == AArch64::CCMNXi)
5047 CCmp.
addImm(
C->Value.abs().getZExtValue());
5055MachineInstr *AArch64InstructionSelector::emitConjunctionRec(
5059 auto &MRI = *MIB.
getMRI();
5077 MachineInstr *ExtraCmp;
5079 ExtraCmp = emitFPCompare(
LHS,
RHS, MIB, CC);
5091 return emitCMP(
Cmp->getOperand(2),
Cmp->getOperand(3), MIB);
5092 return emitFPCompare(
Cmp->getOperand(2).getReg(),
5093 Cmp->getOperand(3).getReg(), MIB);
5100 bool IsOR = Opcode == TargetOpcode::G_OR;
5106 assert(ValidL &&
"Valid conjunction/disjunction tree");
5113 assert(ValidR &&
"Valid conjunction/disjunction tree");
5118 assert(!MustBeFirstR &&
"Valid conjunction/disjunction tree");
5127 bool NegateAfterAll;
5128 if (Opcode == TargetOpcode::G_OR) {
5131 assert(CanNegateR &&
"at least one side must be negatable");
5132 assert(!MustBeFirstR &&
"invalid conjunction/disjunction tree");
5136 NegateAfterR =
true;
5139 NegateR = CanNegateR;
5140 NegateAfterR = !CanNegateR;
5143 NegateAfterAll = !Negate;
5145 assert(Opcode == TargetOpcode::G_AND &&
5146 "Valid conjunction/disjunction tree");
5147 assert(!Negate &&
"Valid conjunction/disjunction tree");
5151 NegateAfterR =
false;
5152 NegateAfterAll =
false;
5157 MachineInstr *CmpR =
5168MachineInstr *AArch64InstructionSelector::emitConjunction(
5170 bool DummyCanNegate;
5171 bool DummyMustBeFirst;
5178bool AArch64InstructionSelector::tryOptSelectConjunction(GSelect &SelI,
5179 MachineInstr &CondMI) {
5190bool AArch64InstructionSelector::tryOptSelect(GSelect &
I) {
5191 MachineRegisterInfo &MRI = *MIB.
getMRI();
5210 MachineInstr *CondDef = MRI.
getVRegDef(
I.getOperand(1).getReg());
5219 if (UI.getOpcode() != TargetOpcode::G_SELECT)
5225 unsigned CondOpc = CondDef->
getOpcode();
5226 if (CondOpc != TargetOpcode::G_ICMP && CondOpc != TargetOpcode::G_FCMP) {
5227 if (tryOptSelectConjunction(
I, *CondDef))
5233 if (CondOpc == TargetOpcode::G_ICMP) {
5262 emitSelect(
I.getOperand(0).getReg(),
I.getOperand(2).getReg(),
5263 I.getOperand(3).getReg(), CondCode, MIB);
5264 I.eraseFromParent();
5268MachineInstr *AArch64InstructionSelector::tryFoldIntegerCompare(
5269 MachineOperand &
LHS, MachineOperand &
RHS, MachineOperand &Predicate,
5270 MachineIRBuilder &MIRBuilder)
const {
5272 "Unexpected MachineOperand");
5273 MachineRegisterInfo &MRI = *MIRBuilder.
getMRI();
5296 if (
isCMN(RHSDef,
P, MRI))
5311 if (
isCMN(LHSDef,
P, MRI)) {
5328 LHSDef->
getOpcode() == TargetOpcode::G_AND) {
5331 if (!ValAndVReg || ValAndVReg->Value != 0)
5341bool AArch64InstructionSelector::selectShuffleVector(
5342 MachineInstr &
I, MachineRegisterInfo &MRI) {
5343 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
5344 Register Src1Reg =
I.getOperand(1).getReg();
5345 Register Src2Reg =
I.getOperand(2).getReg();
5346 ArrayRef<int>
Mask =
I.getOperand(3).getShuffleMask();
5348 "Expected equal shuffle types during selection");
5357 SmallVector<int> NewMask;
5358 bool FirstUsed =
false;
5359 bool SecondUsed =
false;
5360 for (
int M : Mask) {
5362 if (M < 0 || VT->getKnownBits(M < NumElts ? Src1Reg : Src2Reg,
5365 for (
unsigned Byte = 0;
Byte < BytesPerElt; ++
Byte)
5370 FirstUsed |=
M < NumElts;
5371 SecondUsed |=
M >= NumElts;
5372 for (
unsigned Byte = 0;
Byte < BytesPerElt; ++
Byte) {
5381 for (
int &M : NewMask) {
5383 assert(M >= ByteLanes && M < 2 * ByteLanes);
5393 transform(NewMask, std::back_inserter(CstIdxs), [&Ctx](
int M) {
5394 return ConstantInt::get(Type::getInt8Ty(Ctx), M);
5407 emitVectorConcat(std::nullopt, Src1Reg, Src2Reg, MIB);
5414 IndexLoad = emitScalarToVector(64, &AArch64::FPR128RegClass,
5418 AArch64::TBLv16i8One, {&AArch64::FPR128RegClass},
5423 MIB.
buildInstr(TargetOpcode::COPY, {
I.getOperand(0).getReg()}, {})
5424 .addReg(TBL1.getReg(0), {}, AArch64::dsub);
5426 I.eraseFromParent();
5431 auto TBL1 = MIB.
buildInstr(AArch64::TBLv16i8One, {
I.getOperand(0)},
5434 I.eraseFromParent();
5442 auto TBL2 = MIB.
buildInstr(AArch64::TBLv16i8Two, {
I.getOperand(0)},
5445 I.eraseFromParent();
5449MachineInstr *AArch64InstructionSelector::emitLaneInsert(
5451 unsigned LaneIdx,
const RegisterBank &RB,
5452 MachineIRBuilder &MIRBuilder)
const {
5453 MachineInstr *InsElt =
nullptr;
5455 MachineRegisterInfo &MRI = *MIRBuilder.
getMRI();
5464 if (RB.
getID() == AArch64::FPRRegBankID) {
5465 auto InsSub = emitScalarToVector(EltSize, DstRC, EltReg, MIRBuilder);
5468 .
addUse(InsSub->getOperand(0).getReg())
5480bool AArch64InstructionSelector::selectUSMovFromExtend(
5481 MachineInstr &
MI, MachineRegisterInfo &MRI) {
5482 if (
MI.getOpcode() != TargetOpcode::G_SEXT &&
5483 MI.getOpcode() != TargetOpcode::G_ZEXT &&
5484 MI.getOpcode() != TargetOpcode::G_ANYEXT)
5486 bool IsSigned =
MI.getOpcode() == TargetOpcode::G_SEXT;
5487 const Register DefReg =
MI.getOperand(0).getReg();
5488 const LLT DstTy = MRI.
getType(DefReg);
5491 if (DstSize != 32 && DstSize != 64)
5494 MachineInstr *Extract =
getOpcodeDef(TargetOpcode::G_EXTRACT_VECTOR_ELT,
5495 MI.getOperand(1).getReg(), MRI);
5501 const LLT VecTy = MRI.
getType(Src0);
5506 const MachineInstr *ScalarToVector = emitScalarToVector(
5507 VecTy.
getSizeInBits(), &AArch64::FPR128RegClass, Src0, MIB);
5508 assert(ScalarToVector &&
"Didn't expect emitScalarToVector to fail!");
5514 Opcode = IsSigned ? AArch64::SMOVvi32to64 : AArch64::UMOVvi32;
5516 Opcode = IsSigned ? AArch64::SMOVvi16to64 : AArch64::UMOVvi16;
5518 Opcode = IsSigned ? AArch64::SMOVvi8to64 : AArch64::UMOVvi8;
5520 Opcode = IsSigned ? AArch64::SMOVvi16to32 : AArch64::UMOVvi16;
5522 Opcode = IsSigned ? AArch64::SMOVvi8to32 : AArch64::UMOVvi8;
5530 MachineInstr *ExtI =
nullptr;
5531 if (DstSize == 64 && !IsSigned) {
5533 MIB.
buildInstr(Opcode, {NewReg}, {Src0}).addImm(Lane);
5534 ExtI = MIB.
buildInstr(AArch64::SUBREG_TO_REG, {DefReg}, {})
5536 .
addImm(AArch64::sub_32);
5539 ExtI = MIB.
buildInstr(Opcode, {DefReg}, {Src0}).addImm(Lane);
5542 MI.eraseFromParent();
5546MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm8(
5547 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5549 if (DstSize == 128) {
5550 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5552 Op = AArch64::MOVIv16b_ns;
5554 Op = AArch64::MOVIv8b_ns;
5561 auto Mov = Builder.
buildInstr(
Op, {Dst}, {}).addImm(Val);
5568MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm16(
5569 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5573 if (DstSize == 128) {
5574 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5576 Op = Inv ? AArch64::MVNIv8i16 : AArch64::MOVIv8i16;
5578 Op = Inv ? AArch64::MVNIv4i16 : AArch64::MOVIv4i16;
5598MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm32(
5599 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5603 if (DstSize == 128) {
5604 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5606 Op = Inv ? AArch64::MVNIv4i32 : AArch64::MOVIv4i32;
5608 Op = Inv ? AArch64::MVNIv2i32 : AArch64::MOVIv2i32;
5634MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm64(
5635 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5638 if (DstSize == 128) {
5639 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5641 Op = AArch64::MOVIv2d_ns;
5643 Op = AArch64::MOVID;
5649 auto Mov = Builder.
buildInstr(
Op, {Dst}, {}).addImm(Val);
5656MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm321s(
5657 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5661 if (DstSize == 128) {
5662 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5664 Op = Inv ? AArch64::MVNIv4s_msl : AArch64::MOVIv4s_msl;
5666 Op = Inv ? AArch64::MVNIv2s_msl : AArch64::MOVIv2s_msl;
5686MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImmFP(
5687 Register Dst,
unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5690 bool IsWide =
false;
5691 if (DstSize == 128) {
5692 if (
Bits.getHiBits(64) !=
Bits.getLoBits(64))
5694 Op = AArch64::FMOVv4f32_ns;
5697 Op = AArch64::FMOVv2f32_ns;
5706 Op = AArch64::FMOVv2f64_ns;
5710 auto Mov = Builder.
buildInstr(
Op, {Dst}, {}).addImm(Val);
5715bool AArch64InstructionSelector::selectIndexedExtLoad(
5716 MachineInstr &
MI, MachineRegisterInfo &MRI) {
5719 Register WriteBack = ExtLd.getWritebackReg();
5724 unsigned MemSizeBits = ExtLd.getMMO().getMemoryType().getSizeInBits();
5725 bool IsPre = ExtLd.isPre();
5727 unsigned InsertIntoSubReg = 0;
5733 if ((IsSExt && IsFPR) || Ty.
isVector())
5741 if (MemSizeBits == 8) {
5744 Opc = IsPre ? AArch64::LDRSBXpre : AArch64::LDRSBXpost;
5746 Opc = IsPre ? AArch64::LDRSBWpre : AArch64::LDRSBWpost;
5747 NewLdDstTy = IsDst64 ? s64 : s32;
5749 Opc = IsPre ? AArch64::LDRBpre : AArch64::LDRBpost;
5750 InsertIntoSubReg = AArch64::bsub;
5753 Opc = IsPre ? AArch64::LDRBBpre : AArch64::LDRBBpost;
5754 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5757 }
else if (MemSizeBits == 16) {
5760 Opc = IsPre ? AArch64::LDRSHXpre : AArch64::LDRSHXpost;
5762 Opc = IsPre ? AArch64::LDRSHWpre : AArch64::LDRSHWpost;
5763 NewLdDstTy = IsDst64 ? s64 : s32;
5765 Opc = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
5766 InsertIntoSubReg = AArch64::hsub;
5769 Opc = IsPre ? AArch64::LDRHHpre : AArch64::LDRHHpost;
5770 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5773 }
else if (MemSizeBits == 32) {
5775 Opc = IsPre ? AArch64::LDRSWpre : AArch64::LDRSWpost;
5778 Opc = IsPre ? AArch64::LDRSpre : AArch64::LDRSpost;
5779 InsertIntoSubReg = AArch64::ssub;
5782 Opc = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
5783 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5795 .addImm(Cst->getSExtValue());
5800 if (InsertIntoSubReg) {
5802 auto SubToReg = MIB.
buildInstr(TargetOpcode::SUBREG_TO_REG, {Dst}, {})
5803 .addUse(LdMI.getReg(1))
5804 .
addImm(InsertIntoSubReg);
5807 *getRegClassForTypeOnBank(MRI.
getType(Dst),
5814 MI.eraseFromParent();
5819bool AArch64InstructionSelector::selectIndexedLoad(MachineInstr &
MI,
5820 MachineRegisterInfo &MRI) {
5823 Register WriteBack = Ld.getWritebackReg();
5827 "Unexpected type for indexed load");
5828 unsigned MemSize = Ld.getMMO().getMemoryType().getSizeInBytes();
5831 return selectIndexedExtLoad(
MI, MRI);
5835 static constexpr unsigned GPROpcodes[] = {
5836 AArch64::LDRBBpre, AArch64::LDRHHpre, AArch64::LDRWpre,
5838 static constexpr unsigned FPROpcodes[] = {
5839 AArch64::LDRBpre, AArch64::LDRHpre, AArch64::LDRSpre, AArch64::LDRDpre,
5842 ? FPROpcodes[
Log2_32(MemSize)]
5843 : GPROpcodes[
Log2_32(MemSize)];
5846 static constexpr unsigned GPROpcodes[] = {
5847 AArch64::LDRBBpost, AArch64::LDRHHpost, AArch64::LDRWpost,
5849 static constexpr unsigned FPROpcodes[] = {
5850 AArch64::LDRBpost, AArch64::LDRHpost, AArch64::LDRSpost,
5851 AArch64::LDRDpost, AArch64::LDRQpost};
5853 ? FPROpcodes[
Log2_32(MemSize)]
5854 : GPROpcodes[
Log2_32(MemSize)];
5864 MI.eraseFromParent();
5868bool AArch64InstructionSelector::selectIndexedStore(GIndexedStore &
I,
5869 MachineRegisterInfo &MRI) {
5875 "Unexpected type for indexed store");
5877 LocationSize MemSize =
I.getMMO().getSize();
5878 unsigned MemSizeInBytes = MemSize.
getValue();
5880 assert(MemSizeInBytes && MemSizeInBytes <= 16 &&
5881 "Unexpected indexed store size");
5882 unsigned MemSizeLog2 =
Log2_32(MemSizeInBytes);
5886 static constexpr unsigned GPROpcodes[] = {
5887 AArch64::STRBBpre, AArch64::STRHHpre, AArch64::STRWpre,
5889 static constexpr unsigned FPROpcodes[] = {
5890 AArch64::STRBpre, AArch64::STRHpre, AArch64::STRSpre, AArch64::STRDpre,
5894 Opc = FPROpcodes[MemSizeLog2];
5896 Opc = GPROpcodes[MemSizeLog2];
5898 static constexpr unsigned GPROpcodes[] = {
5899 AArch64::STRBBpost, AArch64::STRHHpost, AArch64::STRWpost,
5901 static constexpr unsigned FPROpcodes[] = {
5902 AArch64::STRBpost, AArch64::STRHpost, AArch64::STRSpost,
5903 AArch64::STRDpost, AArch64::STRQpost};
5906 Opc = FPROpcodes[MemSizeLog2];
5908 Opc = GPROpcodes[MemSizeLog2];
5916 Str.cloneMemRefs(
I);
5918 I.eraseFromParent();
5923AArch64InstructionSelector::emitConstantVector(
Register Dst, Constant *CV,
5924 MachineIRBuilder &MIRBuilder,
5925 MachineRegisterInfo &MRI) {
5928 assert((DstSize == 64 || DstSize == 128) &&
5929 "Unexpected vector constant size");
5932 if (DstSize == 128) {
5934 MIRBuilder.
buildInstr(AArch64::MOVIv2d_ns, {Dst}, {}).addImm(0);
5939 if (DstSize == 64) {
5942 .
buildInstr(AArch64::MOVIv2d_ns, {&AArch64::FPR128RegClass}, {})
5945 .addReg(Mov.getReg(0), {}, AArch64::dsub);
5952 APInt SplatValueAsInt =
5955 : SplatValue->getUniqueInteger();
5958 auto TryMOVIWithBits = [&](APInt DefBits) -> MachineInstr * {
5959 MachineInstr *NewOp;
5983 if (
auto *NewOp = TryMOVIWithBits(DefBits))
5987 auto TryWithFNeg = [&](APInt DefBits,
int NumBits,
5988 unsigned NegOpc) -> MachineInstr * {
5991 APInt NegBits(DstSize, 0);
5992 unsigned NumElts = DstSize / NumBits;
5993 for (
unsigned i = 0; i < NumElts; i++)
5994 NegBits |= Neg << (NumBits * i);
5995 NegBits = DefBits ^ NegBits;
5999 if (
auto *NewOp = TryMOVIWithBits(NegBits)) {
6001 DstSize == 64 ? &AArch64::FPR64RegClass : &AArch64::FPR128RegClass);
6003 return MIRBuilder.
buildInstr(NegOpc, {Dst}, {NewDst});
6008 if ((R = TryWithFNeg(DefBits, 32,
6009 DstSize == 64 ? AArch64::FNEGv2f32
6010 : AArch64::FNEGv4f32)) ||
6011 (R = TryWithFNeg(DefBits, 64,
6012 DstSize == 64 ? AArch64::FNEGDr
6013 : AArch64::FNEGv2f64)) ||
6014 (STI.hasFullFP16() &&
6015 (R = TryWithFNeg(DefBits, 16,
6016 DstSize == 64 ? AArch64::FNEGv4f16
6017 : AArch64::FNEGv8f16))))
6023 LLVM_DEBUG(
dbgs() <<
"Could not generate cp load for constant vector!");
6027 auto Copy = MIRBuilder.
buildCopy(Dst, CPLoad->getOperand(0));
6029 Dst, *MRI.
getRegClass(CPLoad->getOperand(0).getReg()), MRI);
6033bool AArch64InstructionSelector::tryOptConstantBuildVec(
6034 MachineInstr &
I, LLT DstTy, MachineRegisterInfo &MRI) {
6035 assert(
I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6037 assert(DstSize <= 128 &&
"Unexpected build_vec type!");
6043 for (
unsigned Idx = 1; Idx <
I.getNumOperands(); ++Idx) {
6044 Register OpReg =
I.getOperand(Idx).getReg();
6053 std::move(AnyConst->Value)));
6066 if (!emitConstantVector(
I.getOperand(0).getReg(), CV, MIB, MRI))
6068 I.eraseFromParent();
6072bool AArch64InstructionSelector::tryOptBuildVecToSubregToReg(
6073 MachineInstr &
I, MachineRegisterInfo &MRI) {
6078 Register Dst =
I.getOperand(0).getReg();
6079 Register EltReg =
I.getOperand(1).getReg();
6080 LLT EltTy = MRI.
getType(EltReg);
6083 const RegisterBank &EltRB = *RBI.
getRegBank(EltReg, MRI,
TRI);
6088 return !getOpcodeDef(TargetOpcode::G_IMPLICIT_DEF, Op.getReg(), MRI);
6096 getRegClassForTypeOnBank(MRI.
getType(Dst), DstRB);
6101 auto SubregToReg = MIB.
buildInstr(AArch64::SUBREG_TO_REG, {Dst}, {})
6104 I.eraseFromParent();
6109bool AArch64InstructionSelector::selectBuildVector(MachineInstr &
I,
6110 MachineRegisterInfo &MRI) {
6111 assert(
I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6114 const LLT DstTy = MRI.
getType(
I.getOperand(0).getReg());
6115 const LLT EltTy = MRI.
getType(
I.getOperand(1).getReg());
6118 if (tryOptConstantBuildVec(
I, DstTy, MRI))
6120 if (tryOptBuildVecToSubregToReg(
I, MRI))
6123 if (EltSize != 8 && EltSize != 16 && EltSize != 32 && EltSize != 64)
6125 const RegisterBank &RB = *RBI.
getRegBank(
I.getOperand(1).getReg(), MRI,
TRI);
6128 MachineInstr *ScalarToVec =
6130 I.getOperand(1).getReg(), MIB);
6139 MachineInstr *PrevMI = ScalarToVec;
6140 for (
unsigned i = 2, e = DstSize / EltSize + 1; i <
e; ++i) {
6143 Register OpReg =
I.getOperand(i).getReg();
6146 PrevMI = &*emitLaneInsert(std::nullopt, DstVec, OpReg, i - 1, RB, MIB);
6153 if (DstSize < 128) {
6156 getRegClassForTypeOnBank(DstTy, *RBI.
getRegBank(DstVec, MRI,
TRI));
6159 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
6164 unsigned SubReg = 0;
6167 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
6168 LLVM_DEBUG(
dbgs() <<
"Unsupported destination size! (" << DstSize
6174 Register DstReg =
I.getOperand(0).getReg();
6176 MIB.
buildInstr(TargetOpcode::COPY, {DstReg}, {}).addReg(DstVec, {}, SubReg);
6177 MachineOperand &RegOp =
I.getOperand(1);
6197 if (PrevMI == ScalarToVec && DstReg.
isVirtual()) {
6199 getRegClassForTypeOnBank(DstTy, *RBI.
getRegBank(DstVec, MRI,
TRI));
6208bool AArch64InstructionSelector::selectVectorLoadIntrinsic(
unsigned Opc,
6211 assert(
I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6213 assert(NumVecs > 1 && NumVecs < 5 &&
"Only support 2, 3, or 4 vectors");
6214 auto &MRI = *MIB.
getMRI();
6215 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6218 "Destination must be 64 bits or 128 bits?");
6219 unsigned SubReg =
Size == 64 ? AArch64::dsub0 : AArch64::qsub0;
6220 auto Ptr =
I.getOperand(
I.getNumOperands() - 1).getReg();
6225 Register SelectedLoadDst =
Load->getOperand(0).getReg();
6226 for (
unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6227 auto Vec = MIB.
buildInstr(TargetOpcode::COPY, {
I.getOperand(Idx)}, {})
6228 .addReg(SelectedLoadDst, {}, SubReg + Idx);
6237bool AArch64InstructionSelector::selectVectorLoadLaneIntrinsic(
6238 unsigned Opc,
unsigned NumVecs, MachineInstr &
I) {
6239 assert(
I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6241 assert(NumVecs > 1 && NumVecs < 5 &&
"Only support 2, 3, or 4 vectors");
6242 auto &MRI = *MIB.
getMRI();
6243 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6246 auto FirstSrcRegIt =
I.operands_begin() + NumVecs + 1;
6248 std::transform(FirstSrcRegIt, FirstSrcRegIt + NumVecs, Regs.
begin(),
6249 [](
auto MO) { return MO.getReg(); });
6253 return emitScalarToVector(64, &AArch64::FPR128RegClass, Reg, MIB)
6268 .
addImm(LaneNo->getZExtValue())
6272 Register SelectedLoadDst =
Load->getOperand(0).getReg();
6273 unsigned SubReg = AArch64::qsub0;
6274 for (
unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6275 auto Vec = MIB.
buildInstr(TargetOpcode::COPY,
6276 {Narrow ? DstOp(&AArch64::FPR128RegClass)
6277 : DstOp(
I.getOperand(Idx).
getReg())},
6279 .addReg(SelectedLoadDst, {}, SubReg + Idx);
6284 !emitNarrowVector(
I.getOperand(Idx).getReg(), WideReg, MIB, MRI))
6290void AArch64InstructionSelector::selectVectorStoreIntrinsic(MachineInstr &
I,
6293 MachineRegisterInfo &MRI =
I.getParent()->getParent()->getRegInfo();
6294 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6295 Register Ptr =
I.getOperand(1 + NumVecs).getReg();
6298 std::transform(
I.operands_begin() + 1,
I.operands_begin() + 1 + NumVecs,
6299 Regs.
begin(), [](
auto MO) { return MO.getReg(); });
6308bool AArch64InstructionSelector::selectVectorStoreLaneIntrinsic(
6309 MachineInstr &
I,
unsigned NumVecs,
unsigned Opc) {
6310 MachineRegisterInfo &MRI =
I.getParent()->getParent()->getRegInfo();
6311 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6315 std::transform(
I.operands_begin() + 1,
I.operands_begin() + 1 + NumVecs,
6316 Regs.
begin(), [](
auto MO) { return MO.getReg(); });
6320 return emitScalarToVector(64, &AArch64::FPR128RegClass, Reg, MIB)
6330 Register Ptr =
I.getOperand(1 + NumVecs + 1).getReg();
6333 .
addImm(LaneNo->getZExtValue())
6340bool AArch64InstructionSelector::selectIntrinsicWithSideEffects(
6341 MachineInstr &
I, MachineRegisterInfo &MRI) {
6354 case Intrinsic::aarch64_ldxp:
6355 case Intrinsic::aarch64_ldaxp: {
6357 IntrinID == Intrinsic::aarch64_ldxp ? AArch64::LDXPX : AArch64::LDAXPX,
6358 {
I.getOperand(0).getReg(),
I.getOperand(1).getReg()},
6364 case Intrinsic::aarch64_neon_ld1x2: {
6365 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6368 Opc = AArch64::LD1Twov8b;
6370 Opc = AArch64::LD1Twov16b;
6372 Opc = AArch64::LD1Twov4h;
6374 Opc = AArch64::LD1Twov8h;
6376 Opc = AArch64::LD1Twov2s;
6378 Opc = AArch64::LD1Twov4s;
6380 Opc = AArch64::LD1Twov2d;
6381 else if (Ty ==
S64 || Ty == P0)
6382 Opc = AArch64::LD1Twov1d;
6385 selectVectorLoadIntrinsic(
Opc, 2,
I);
6388 case Intrinsic::aarch64_neon_ld1x3: {
6389 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6392 Opc = AArch64::LD1Threev8b;
6394 Opc = AArch64::LD1Threev16b;
6396 Opc = AArch64::LD1Threev4h;
6398 Opc = AArch64::LD1Threev8h;
6400 Opc = AArch64::LD1Threev2s;
6402 Opc = AArch64::LD1Threev4s;
6404 Opc = AArch64::LD1Threev2d;
6405 else if (Ty ==
S64 || Ty == P0)
6406 Opc = AArch64::LD1Threev1d;
6409 selectVectorLoadIntrinsic(
Opc, 3,
I);
6412 case Intrinsic::aarch64_neon_ld1x4: {
6413 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6416 Opc = AArch64::LD1Fourv8b;
6418 Opc = AArch64::LD1Fourv16b;
6420 Opc = AArch64::LD1Fourv4h;
6422 Opc = AArch64::LD1Fourv8h;
6424 Opc = AArch64::LD1Fourv2s;
6426 Opc = AArch64::LD1Fourv4s;
6428 Opc = AArch64::LD1Fourv2d;
6429 else if (Ty ==
S64 || Ty == P0)
6430 Opc = AArch64::LD1Fourv1d;
6433 selectVectorLoadIntrinsic(
Opc, 4,
I);
6436 case Intrinsic::aarch64_neon_ld2: {
6437 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6440 Opc = AArch64::LD2Twov8b;
6442 Opc = AArch64::LD2Twov16b;
6444 Opc = AArch64::LD2Twov4h;
6446 Opc = AArch64::LD2Twov8h;
6448 Opc = AArch64::LD2Twov2s;
6450 Opc = AArch64::LD2Twov4s;
6452 Opc = AArch64::LD2Twov2d;
6453 else if (Ty ==
S64 || Ty == P0)
6454 Opc = AArch64::LD1Twov1d;
6457 selectVectorLoadIntrinsic(
Opc, 2,
I);
6460 case Intrinsic::aarch64_neon_ld2lane: {
6461 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6464 Opc = AArch64::LD2i8;
6466 Opc = AArch64::LD2i16;
6468 Opc = AArch64::LD2i32;
6471 Opc = AArch64::LD2i64;
6474 if (!selectVectorLoadLaneIntrinsic(
Opc, 2,
I))
6478 case Intrinsic::aarch64_neon_ld2r: {
6479 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6482 Opc = AArch64::LD2Rv8b;
6484 Opc = AArch64::LD2Rv16b;
6486 Opc = AArch64::LD2Rv4h;
6488 Opc = AArch64::LD2Rv8h;
6490 Opc = AArch64::LD2Rv2s;
6492 Opc = AArch64::LD2Rv4s;
6494 Opc = AArch64::LD2Rv2d;
6495 else if (Ty ==
S64 || Ty == P0)
6496 Opc = AArch64::LD2Rv1d;
6499 selectVectorLoadIntrinsic(
Opc, 2,
I);
6502 case Intrinsic::aarch64_neon_ld3: {
6503 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6506 Opc = AArch64::LD3Threev8b;
6508 Opc = AArch64::LD3Threev16b;
6510 Opc = AArch64::LD3Threev4h;
6512 Opc = AArch64::LD3Threev8h;
6514 Opc = AArch64::LD3Threev2s;
6516 Opc = AArch64::LD3Threev4s;
6518 Opc = AArch64::LD3Threev2d;
6519 else if (Ty ==
S64 || Ty == P0)
6520 Opc = AArch64::LD1Threev1d;
6523 selectVectorLoadIntrinsic(
Opc, 3,
I);
6526 case Intrinsic::aarch64_neon_ld3lane: {
6527 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6530 Opc = AArch64::LD3i8;
6532 Opc = AArch64::LD3i16;
6534 Opc = AArch64::LD3i32;
6537 Opc = AArch64::LD3i64;
6540 if (!selectVectorLoadLaneIntrinsic(
Opc, 3,
I))
6544 case Intrinsic::aarch64_neon_ld3r: {
6545 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6548 Opc = AArch64::LD3Rv8b;
6550 Opc = AArch64::LD3Rv16b;
6552 Opc = AArch64::LD3Rv4h;
6554 Opc = AArch64::LD3Rv8h;
6556 Opc = AArch64::LD3Rv2s;
6558 Opc = AArch64::LD3Rv4s;
6560 Opc = AArch64::LD3Rv2d;
6561 else if (Ty ==
S64 || Ty == P0)
6562 Opc = AArch64::LD3Rv1d;
6565 selectVectorLoadIntrinsic(
Opc, 3,
I);
6568 case Intrinsic::aarch64_neon_ld4: {
6569 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6572 Opc = AArch64::LD4Fourv8b;
6574 Opc = AArch64::LD4Fourv16b;
6576 Opc = AArch64::LD4Fourv4h;
6578 Opc = AArch64::LD4Fourv8h;
6580 Opc = AArch64::LD4Fourv2s;
6582 Opc = AArch64::LD4Fourv4s;
6584 Opc = AArch64::LD4Fourv2d;
6585 else if (Ty ==
S64 || Ty == P0)
6586 Opc = AArch64::LD1Fourv1d;
6589 selectVectorLoadIntrinsic(
Opc, 4,
I);
6592 case Intrinsic::aarch64_neon_ld4lane: {
6593 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6596 Opc = AArch64::LD4i8;
6598 Opc = AArch64::LD4i16;
6600 Opc = AArch64::LD4i32;
6603 Opc = AArch64::LD4i64;
6606 if (!selectVectorLoadLaneIntrinsic(
Opc, 4,
I))
6610 case Intrinsic::aarch64_neon_ld4r: {
6611 LLT Ty = MRI.
getType(
I.getOperand(0).getReg());
6614 Opc = AArch64::LD4Rv8b;
6616 Opc = AArch64::LD4Rv16b;
6618 Opc = AArch64::LD4Rv4h;
6620 Opc = AArch64::LD4Rv8h;
6622 Opc = AArch64::LD4Rv2s;
6624 Opc = AArch64::LD4Rv4s;
6626 Opc = AArch64::LD4Rv2d;
6627 else if (Ty ==
S64 || Ty == P0)
6628 Opc = AArch64::LD4Rv1d;
6631 selectVectorLoadIntrinsic(
Opc, 4,
I);
6634 case Intrinsic::aarch64_neon_st1x2: {
6635 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6638 Opc = AArch64::ST1Twov8b;
6640 Opc = AArch64::ST1Twov16b;
6642 Opc = AArch64::ST1Twov4h;
6644 Opc = AArch64::ST1Twov8h;
6646 Opc = AArch64::ST1Twov2s;
6648 Opc = AArch64::ST1Twov4s;
6650 Opc = AArch64::ST1Twov2d;
6651 else if (Ty ==
S64 || Ty == P0)
6652 Opc = AArch64::ST1Twov1d;
6655 selectVectorStoreIntrinsic(
I, 2,
Opc);
6658 case Intrinsic::aarch64_neon_st1x3: {
6659 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6662 Opc = AArch64::ST1Threev8b;
6664 Opc = AArch64::ST1Threev16b;
6666 Opc = AArch64::ST1Threev4h;
6668 Opc = AArch64::ST1Threev8h;
6670 Opc = AArch64::ST1Threev2s;
6672 Opc = AArch64::ST1Threev4s;
6674 Opc = AArch64::ST1Threev2d;
6675 else if (Ty ==
S64 || Ty == P0)
6676 Opc = AArch64::ST1Threev1d;
6679 selectVectorStoreIntrinsic(
I, 3,
Opc);
6682 case Intrinsic::aarch64_neon_st1x4: {
6683 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6686 Opc = AArch64::ST1Fourv8b;
6688 Opc = AArch64::ST1Fourv16b;
6690 Opc = AArch64::ST1Fourv4h;
6692 Opc = AArch64::ST1Fourv8h;
6694 Opc = AArch64::ST1Fourv2s;
6696 Opc = AArch64::ST1Fourv4s;
6698 Opc = AArch64::ST1Fourv2d;
6699 else if (Ty ==
S64 || Ty == P0)
6700 Opc = AArch64::ST1Fourv1d;
6703 selectVectorStoreIntrinsic(
I, 4,
Opc);
6706 case Intrinsic::aarch64_neon_st2: {
6707 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6710 Opc = AArch64::ST2Twov8b;
6712 Opc = AArch64::ST2Twov16b;
6714 Opc = AArch64::ST2Twov4h;
6716 Opc = AArch64::ST2Twov8h;
6718 Opc = AArch64::ST2Twov2s;
6720 Opc = AArch64::ST2Twov4s;
6722 Opc = AArch64::ST2Twov2d;
6723 else if (Ty ==
S64 || Ty == P0)
6724 Opc = AArch64::ST1Twov1d;
6727 selectVectorStoreIntrinsic(
I, 2,
Opc);
6730 case Intrinsic::aarch64_neon_st3: {
6731 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6734 Opc = AArch64::ST3Threev8b;
6736 Opc = AArch64::ST3Threev16b;
6738 Opc = AArch64::ST3Threev4h;
6740 Opc = AArch64::ST3Threev8h;
6742 Opc = AArch64::ST3Threev2s;
6744 Opc = AArch64::ST3Threev4s;
6746 Opc = AArch64::ST3Threev2d;
6747 else if (Ty ==
S64 || Ty == P0)
6748 Opc = AArch64::ST1Threev1d;
6751 selectVectorStoreIntrinsic(
I, 3,
Opc);
6754 case Intrinsic::aarch64_neon_st4: {
6755 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6758 Opc = AArch64::ST4Fourv8b;
6760 Opc = AArch64::ST4Fourv16b;
6762 Opc = AArch64::ST4Fourv4h;
6764 Opc = AArch64::ST4Fourv8h;
6766 Opc = AArch64::ST4Fourv2s;
6768 Opc = AArch64::ST4Fourv4s;
6770 Opc = AArch64::ST4Fourv2d;
6771 else if (Ty ==
S64 || Ty == P0)
6772 Opc = AArch64::ST1Fourv1d;
6775 selectVectorStoreIntrinsic(
I, 4,
Opc);
6778 case Intrinsic::aarch64_neon_st2lane: {
6779 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6782 Opc = AArch64::ST2i8;
6784 Opc = AArch64::ST2i16;
6786 Opc = AArch64::ST2i32;
6789 Opc = AArch64::ST2i64;
6792 if (!selectVectorStoreLaneIntrinsic(
I, 2,
Opc))
6796 case Intrinsic::aarch64_neon_st3lane: {
6797 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6800 Opc = AArch64::ST3i8;
6802 Opc = AArch64::ST3i16;
6804 Opc = AArch64::ST3i32;
6807 Opc = AArch64::ST3i64;
6810 if (!selectVectorStoreLaneIntrinsic(
I, 3,
Opc))
6814 case Intrinsic::aarch64_neon_st4lane: {
6815 LLT Ty = MRI.
getType(
I.getOperand(1).getReg());
6818 Opc = AArch64::ST4i8;
6820 Opc = AArch64::ST4i16;
6822 Opc = AArch64::ST4i32;
6825 Opc = AArch64::ST4i64;
6828 if (!selectVectorStoreLaneIntrinsic(
I, 4,
Opc))
6832 case Intrinsic::aarch64_mops_memset_tag: {
6845 Register DstDef =
I.getOperand(0).getReg();
6847 Register DstUse =
I.getOperand(2).getReg();
6848 Register ValUse =
I.getOperand(3).getReg();
6849 Register SizeUse =
I.getOperand(4).getReg();
6856 auto Memset = MIB.
buildInstr(AArch64::MOPSMemorySetTaggingPseudo,
6857 {DstDef, SizeDef}, {DstUse, SizeUse, ValUse});
6863 case Intrinsic::ptrauth_resign_load_relative: {
6864 Register DstReg =
I.getOperand(0).getReg();
6865 Register ValReg =
I.getOperand(2).getReg();
6866 uint64_t AUTKey =
I.getOperand(3).getImm();
6867 Register AUTDisc =
I.getOperand(4).getReg();
6868 uint64_t PACKey =
I.getOperand(5).getImm();
6869 Register PACDisc =
I.getOperand(6).getReg();
6870 int64_t Addend =
I.getOperand(7).getImm();
6873 uint16_t AUTConstDiscC = 0;
6874 std::tie(AUTConstDiscC, AUTAddrDisc) =
6878 uint16_t PACConstDiscC = 0;
6879 std::tie(PACConstDiscC, PACAddrDisc) =
6882 MIB.
buildCopy({AArch64::X16}, {ValReg});
6898 I.eraseFromParent();
6903 I.eraseFromParent();
6907bool AArch64InstructionSelector::selectIntrinsic(MachineInstr &
I,
6908 MachineRegisterInfo &MRI) {
6914 case Intrinsic::ptrauth_resign: {
6915 Register DstReg =
I.getOperand(0).getReg();
6916 Register ValReg =
I.getOperand(2).getReg();
6917 uint64_t AUTKey =
I.getOperand(3).getImm();
6918 Register AUTDisc =
I.getOperand(4).getReg();
6919 uint64_t PACKey =
I.getOperand(5).getImm();
6920 Register PACDisc =
I.getOperand(6).getReg();
6923 uint16_t AUTConstDiscC = 0;
6924 std::tie(AUTConstDiscC, AUTAddrDisc) =
6928 uint16_t PACConstDiscC = 0;
6929 std::tie(PACConstDiscC, PACAddrDisc) =
6932 MIB.
buildCopy({AArch64::X16}, {ValReg});
6946 I.eraseFromParent();
6949 case Intrinsic::ptrauth_auth_with_pc_and_resign: {
6950 Register DstReg =
I.getOperand(0).getReg();
6951 Register ValReg =
I.getOperand(2).getReg();
6952 uint64_t AUTKey =
I.getOperand(3).getImm();
6953 Register AUTDisc =
I.getOperand(4).getReg();
6954 Register AUTPC =
I.getOperand(5).getReg();
6955 uint64_t PACKey =
I.getOperand(6).getImm();
6956 Register PACDisc =
I.getOperand(7).getReg();
6959 "auth_with_pc_and_resign only supports IA and IB keys");
6961 uint16_t PACConstDiscC = 0;
6963 std::tie(PACConstDiscC, PACAddrDisc) =
6967 PACAddrDisc = AArch64::XZR;
6969 MIB.
buildCopy({AArch64::X17}, {ValReg});
6970 MIB.
buildCopy({AArch64::X16}, {AUTDisc});
6985 I.eraseFromParent();
6988 case Intrinsic::ptrauth_auth: {
6989 Register DstReg =
I.getOperand(0).getReg();
6990 Register ValReg =
I.getOperand(2).getReg();
6991 uint64_t AUTKey =
I.getOperand(3).getImm();
6992 Register AUTDisc =
I.getOperand(4).getReg();
6995 uint16_t AUTConstDiscC = 0;
6996 std::tie(AUTConstDiscC, AUTAddrDisc) =
7000 MIB.
buildCopy({AArch64::X16}, {ValReg});
7020 Auth.constrainAllUses(
TII,
TRI, RBI);
7024 I.eraseFromParent();
7027 case Intrinsic::frameaddress:
7028 case Intrinsic::returnaddress: {
7032 unsigned Depth =
I.getOperand(2).getImm();
7033 Register DstReg =
I.getOperand(0).getReg();
7036 if (
Depth == 0 && IntrinID == Intrinsic::returnaddress) {
7037 if (!MFReturnAddr) {
7042 MF,
TII, AArch64::LR, AArch64::GPR64RegClass,
I.getDebugLoc());
7045 if (STI.hasPAuth()) {
7046 MIB.
buildInstr(AArch64::XPACI, {DstReg}, {MFReturnAddr});
7053 I.eraseFromParent();
7062 MIB.
buildInstr(AArch64::LDRXui, {NextFrame}, {FrameAddr}).addImm(0);
7064 FrameAddr = NextFrame;
7067 if (IntrinID == Intrinsic::frameaddress)
7072 if (STI.hasPAuth()) {
7074 MIB.
buildInstr(AArch64::LDRXui, {TmpReg}, {FrameAddr}).addImm(1);
7075 MIB.
buildInstr(AArch64::XPACI, {DstReg}, {TmpReg});
7084 I.eraseFromParent();
7087 case Intrinsic::aarch64_neon_tbl2:
7088 SelectTable(
I, MRI, 2, AArch64::TBLv8i8Two, AArch64::TBLv16i8Two,
false);
7090 case Intrinsic::aarch64_neon_tbl3:
7091 SelectTable(
I, MRI, 3, AArch64::TBLv8i8Three, AArch64::TBLv16i8Three,
7094 case Intrinsic::aarch64_neon_tbl4:
7095 SelectTable(
I, MRI, 4, AArch64::TBLv8i8Four, AArch64::TBLv16i8Four,
false);
7097 case Intrinsic::aarch64_neon_tbx2:
7098 SelectTable(
I, MRI, 2, AArch64::TBXv8i8Two, AArch64::TBXv16i8Two,
true);
7100 case Intrinsic::aarch64_neon_tbx3:
7101 SelectTable(
I, MRI, 3, AArch64::TBXv8i8Three, AArch64::TBXv16i8Three,
true);
7103 case Intrinsic::aarch64_neon_tbx4:
7104 SelectTable(
I, MRI, 4, AArch64::TBXv8i8Four, AArch64::TBXv16i8Four,
true);
7106 case Intrinsic::swift_async_context_addr:
7107 auto Sub = MIB.
buildInstr(AArch64::SUBXri, {
I.getOperand(0).getReg()},
7114 MF->
getInfo<AArch64FunctionInfo>()->setHasSwiftAsyncContext(
true);
7115 I.eraseFromParent();
7150bool AArch64InstructionSelector::selectPtrAuthGlobalValue(
7151 MachineInstr &
I, MachineRegisterInfo &MRI)
const {
7152 Register DefReg =
I.getOperand(0).getReg();
7153 Register Addr =
I.getOperand(1).getReg();
7155 Register AddrDisc =
I.getOperand(3).getReg();
7156 uint64_t Disc =
I.getOperand(4).getImm();
7166 "constant discriminator in ptrauth global out of range [0, 0xffff]");
7182 if (OffsetMI.
getOpcode() != TargetOpcode::G_CONSTANT)
7194 const GlobalValue *GV;
7205 MachineIRBuilder MIB(
I);
7211 "unsupported non-GOT op flags on ptrauth global reference");
7213 "unsupported non-GOT reference to weak ptrauth global");
7216 bool HasAddrDisc = !AddrDiscVal || *AddrDiscVal != 0;
7223 MIB.
buildInstr(NeedsGOTLoad ? AArch64::LOADgotPAC : AArch64::MOVaddrPAC)
7226 .
addReg(HasAddrDisc ? AddrDisc : AArch64::XZR)
7231 I.eraseFromParent();
7243 "unsupported non-zero offset in weak ptrauth global reference");
7248 MIB.
buildInstr(AArch64::LOADauthptrstatic, {DefReg}, {})
7249 .addGlobalAddress(GV,
Offset)
7254 I.eraseFromParent();
7258void AArch64InstructionSelector::SelectTable(MachineInstr &
I,
7259 MachineRegisterInfo &MRI,
7260 unsigned NumVec,
unsigned Opc1,
7261 unsigned Opc2,
bool isExt) {
7262 Register DstReg =
I.getOperand(0).getReg();
7267 for (
unsigned i = 0; i < NumVec; i++)
7268 Regs.
push_back(
I.getOperand(i + 2 + isExt).getReg());
7271 Register IdxReg =
I.getOperand(2 + NumVec + isExt).getReg();
7272 MachineInstrBuilder
Instr;
7279 I.eraseFromParent();
7282InstructionSelector::ComplexRendererFns
7283AArch64InstructionSelector::selectShiftA_32(
const MachineOperand &Root)
const {
7285 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7286 return std::nullopt;
7287 uint64_t Enc = (32 - *MaybeImmed) & 0x1f;
7288 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7291InstructionSelector::ComplexRendererFns
7292AArch64InstructionSelector::selectShiftB_32(
const MachineOperand &Root)
const {
7294 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7295 return std::nullopt;
7297 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7300InstructionSelector::ComplexRendererFns
7301AArch64InstructionSelector::selectShiftA_64(
const MachineOperand &Root)
const {
7303 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7304 return std::nullopt;
7305 uint64_t Enc = (64 - *MaybeImmed) & 0x3f;
7306 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7309InstructionSelector::ComplexRendererFns
7310AArch64InstructionSelector::selectShiftB_64(
const MachineOperand &Root)
const {
7312 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7313 return std::nullopt;
7315 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7318template <
unsigned ShiftW
idth>
7319InstructionSelector::ComplexRendererFns
7320AArch64InstructionSelector::selectShiftMask(MachineOperand &Root)
const {
7322 return std::nullopt;
7324 MachineRegisterInfo &MRI =
7331 if (ShiftWidth == 32) {
7334 ShAmtReg = ZExtSrcReg;
7343 ShAmtReg = AndSrcReg;
7355 (AddImm % ShiftWidth == 0)) {
7356 ShAmtReg = AddSrcReg;
7357 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(ShAmtReg); }}};
7366 SubImm != 0 && (SubImm % ShiftWidth == 0)) {
7367 return {{[=](MachineInstrBuilder &MIB) {
7368 MachineInstr *
I = MIB.getInstr();
7371 ShiftWidth == 32 ? AArch64::GPR32RegClass : AArch64::GPR64RegClass;
7372 unsigned SubOpc = ShiftWidth == 32 ? AArch64::SUBWrr : AArch64::SUBXrr;
7373 Register ZeroReg = ShiftWidth == 32 ? AArch64::WZR : AArch64::XZR;
7375 auto NegMI =
BuildMI(*
I->getParent(), *
I,
I->getDebugLoc(),
7376 TII.get(SubOpc), NegReg)
7390 (NotImm % ShiftWidth == ShiftWidth - 1)) {
7391 return {{[=](MachineInstrBuilder &MIB) {
7392 MachineInstr *
I = MIB.getInstr();
7395 ShiftWidth == 32 ? AArch64::GPR32RegClass : AArch64::GPR64RegClass;
7396 unsigned NotOpc = ShiftWidth == 32 ? AArch64::ORNWrr : AArch64::ORNXrr;
7397 Register ZeroReg = ShiftWidth == 32 ? AArch64::WZR : AArch64::XZR;
7399 auto NotMI =
BuildMI(*
I->getParent(), *
I,
I->getDebugLoc(),
7400 TII.get(NotOpc), NotReg)
7410 if (ShAmtReg == Root.
getReg())
7411 return std::nullopt;
7413 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(ShAmtReg); }}};
7421InstructionSelector::ComplexRendererFns
7422AArch64InstructionSelector::select12BitValueWithLeftShift(
7425 if (Immed >> 12 == 0) {
7427 }
else if ((Immed & 0xfff) == 0 && Immed >> 24 == 0) {
7429 Immed = Immed >> 12;
7431 return std::nullopt;
7435 [=](MachineInstrBuilder &MIB) { MIB.addImm(Immed); },
7436 [=](MachineInstrBuilder &MIB) { MIB.addImm(ShVal); },
7443InstructionSelector::ComplexRendererFns
7444AArch64InstructionSelector::selectArithImmed(MachineOperand &Root)
const {
7451 if (MaybeImmed == std::nullopt)
7452 return std::nullopt;
7453 return select12BitValueWithLeftShift(*MaybeImmed);
7458InstructionSelector::ComplexRendererFns
7459AArch64InstructionSelector::selectNegArithImmed(MachineOperand &Root)
const {
7463 return std::nullopt;
7465 if (MaybeImmed == std::nullopt)
7466 return std::nullopt;
7473 return std::nullopt;
7479 Immed = ~((uint32_t)Immed) + 1;
7481 Immed = ~Immed + 1ULL;
7483 if (Immed & 0xFFFFFFFFFF000000ULL)
7484 return std::nullopt;
7486 Immed &= 0xFFFFFFULL;
7487 return select12BitValueWithLeftShift(Immed);
7504std::optional<bool> AArch64InstructionSelector::isWorthFoldingIntoAddrMode(
7505 const MachineInstr &
MI,
const MachineRegisterInfo &MRI)
const {
7506 if (
MI.getOpcode() == AArch64::G_SHL) {
7510 MI.getOperand(2).getReg(), MRI)) {
7511 const APInt ShiftVal = ValAndVeg->Value;
7514 return !(STI.hasAddrLSLSlow14() && (ShiftVal == 1 || ShiftVal == 4));
7517 return std::nullopt;
7525bool AArch64InstructionSelector::isWorthFoldingIntoExtendedReg(
7526 const MachineInstr &
MI,
const MachineRegisterInfo &MRI,
7527 bool IsAddrOperand)
const {
7532 MI.getParent()->getParent()->getFunction().hasOptSize())
7535 if (IsAddrOperand) {
7537 if (
const auto Worth = isWorthFoldingIntoAddrMode(
MI, MRI))
7541 if (
MI.getOpcode() == AArch64::G_PTR_ADD) {
7542 MachineInstr *OffsetInst =
7548 if (
const auto Worth = isWorthFoldingIntoAddrMode(*OffsetInst, MRI))
7559 [](MachineInstr &Use) { return Use.mayLoadOrStore(); });
7562InstructionSelector::ComplexRendererFns
7563AArch64InstructionSelector::selectExtendedSHL(
7564 MachineOperand &Root, MachineOperand &
Base, MachineOperand &
Offset,
7565 unsigned SizeInBytes,
bool WantsExt)
const {
7566 assert(
Base.isReg() &&
"Expected base to be a register operand");
7567 assert(
Offset.isReg() &&
"Expected offset to be a register operand");
7572 unsigned OffsetOpc = OffsetInst->
getOpcode();
7573 bool LookedThroughZExt =
false;
7574 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL) {
7576 if (OffsetOpc != TargetOpcode::G_ZEXT || !WantsExt)
7577 return std::nullopt;
7581 LookedThroughZExt =
true;
7583 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL)
7584 return std::nullopt;
7587 int64_t LegalShiftVal =
Log2_32(SizeInBytes);
7588 if (LegalShiftVal == 0)
7589 return std::nullopt;
7590 if (!isWorthFoldingIntoExtendedReg(*OffsetInst, MRI,
true))
7591 return std::nullopt;
7602 if (OffsetOpc == TargetOpcode::G_SHL)
7603 return std::nullopt;
7609 return std::nullopt;
7614 int64_t ImmVal = ValAndVReg->Value.getSExtValue();
7618 if (OffsetOpc == TargetOpcode::G_MUL) {
7620 return std::nullopt;
7626 if ((ImmVal & 0x7) != ImmVal)
7627 return std::nullopt;
7631 if (ImmVal != LegalShiftVal)
7632 return std::nullopt;
7634 unsigned SignExtend = 0;
7638 if (!LookedThroughZExt) {
7640 auto Ext = getExtendTypeForInst(*ExtInst, MRI,
true);
7642 return std::nullopt;
7647 return std::nullopt;
7653 OffsetReg = moveScalarRegClass(OffsetReg, AArch64::GPR32RegClass, MIB);
7658 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(
Base.getReg()); },
7659 [=](MachineInstrBuilder &MIB) { MIB.addUse(OffsetReg); },
7660 [=](MachineInstrBuilder &MIB) {
7663 MIB.addImm(SignExtend);
7676InstructionSelector::ComplexRendererFns
7677AArch64InstructionSelector::selectAddrModeShiftedExtendXReg(
7678 MachineOperand &Root,
unsigned SizeInBytes)
const {
7680 return std::nullopt;
7695 MachineInstr *PtrAdd =
7697 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(*PtrAdd, MRI,
true))
7698 return std::nullopt;
7702 MachineInstr *OffsetInst =
7704 return selectExtendedSHL(Root, PtrAdd->
getOperand(1),
7717InstructionSelector::ComplexRendererFns
7718AArch64InstructionSelector::selectAddrModeRegisterOffset(
7719 MachineOperand &Root)
const {
7725 return std::nullopt;
7731 return std::nullopt;
7734 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(
Base); },
7735 [=](MachineInstrBuilder &MIB) { MIB.addUse(
Offset); },
7736 [=](MachineInstrBuilder &MIB) {
7746InstructionSelector::ComplexRendererFns
7747AArch64InstructionSelector::selectAddrModeXRO(MachineOperand &Root,
7748 unsigned SizeInBytes)
const {
7751 return std::nullopt;
7752 MachineInstr *PtrAdd =
7755 return std::nullopt;
7773 unsigned Scale =
Log2_32(SizeInBytes);
7774 int64_t ImmOff = ValAndVReg->Value.getSExtValue();
7778 if (ImmOff % SizeInBytes == 0 && ImmOff >= 0 &&
7779 ImmOff < (0x1000 << Scale))
7780 return std::nullopt;
7785 if ((ImmOff & 0xfffffffffffff000LL) == 0x0LL)
7789 if ((ImmOff & 0xffffffffff000fffLL) != 0x0LL)
7795 return (ImmOff & 0xffffffffff00ffffLL) != 0x0LL &&
7796 (ImmOff & 0xffffffffffff0fffLL) != 0x0LL;
7801 return std::nullopt;
7805 auto AddrModeFns = selectAddrModeShiftedExtendXReg(Root, SizeInBytes);
7811 return selectAddrModeRegisterOffset(Root);
7820InstructionSelector::ComplexRendererFns
7821AArch64InstructionSelector::selectAddrModeWRO(MachineOperand &Root,
7822 unsigned SizeInBytes)
const {
7825 MachineInstr *PtrAdd =
7827 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(*PtrAdd, MRI,
true))
7828 return std::nullopt;
7849 auto ExtendedShl = selectExtendedSHL(Root,
LHS, OffsetInst->
getOperand(0),
7858 if (!isWorthFoldingIntoExtendedReg(*OffsetInst, MRI,
true))
7859 return std::nullopt;
7863 getExtendTypeForInst(*OffsetInst, MRI,
true);
7865 return std::nullopt;
7868 MachineIRBuilder MIB(*PtrAdd);
7870 AArch64::GPR32RegClass, MIB);
7874 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(
LHS.getReg()); },
7875 [=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); },
7876 [=](MachineInstrBuilder &MIB) {
7877 MIB.addImm(SignExtend);
7887InstructionSelector::ComplexRendererFns
7888AArch64InstructionSelector::selectAddrModeUnscaled(MachineOperand &Root,
7889 unsigned Size)
const {
7890 MachineRegisterInfo &MRI =
7894 return std::nullopt;
7896 if (!isBaseWithConstantOffset(Root, MRI))
7897 return std::nullopt;
7901 MachineOperand &OffImm = RootDef->
getOperand(2);
7902 if (!OffImm.
isReg())
7903 return std::nullopt;
7905 if (
RHS->getOpcode() != TargetOpcode::G_CONSTANT)
7906 return std::nullopt;
7908 MachineOperand &RHSOp1 =
RHS->getOperand(1);
7910 return std::nullopt;
7913 if (RHSC >= -256 && RHSC < 256) {
7916 [=](MachineInstrBuilder &MIB) { MIB.add(
Base); },
7917 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
7920 return std::nullopt;
7923InstructionSelector::ComplexRendererFns
7924AArch64InstructionSelector::tryFoldAddLowIntoImm(MachineInstr &RootDef,
7926 MachineRegisterInfo &MRI)
const {
7927 if (RootDef.
getOpcode() != AArch64::G_ADD_LOW)
7928 return std::nullopt;
7931 return std::nullopt;
7936 return std::nullopt;
7940 return std::nullopt;
7944 return std::nullopt;
7947 MachineIRBuilder MIRBuilder(RootDef);
7949 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(AdrpReg); },
7950 [=](MachineInstrBuilder &MIB) {
7951 MIB.addGlobalAddress(GV,
Offset,
7960InstructionSelector::ComplexRendererFns
7961AArch64InstructionSelector::selectAddrModeIndexed(MachineOperand &Root,
7962 unsigned Size)
const {
7967 return std::nullopt;
7970 if (RootDef->
getOpcode() == TargetOpcode::G_FRAME_INDEX) {
7972 [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->
getOperand(1)); },
7973 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
7981 MachineInstr *RootParent = Root.
getParent();
7983 !(RootParent->
getOpcode() == AArch64::G_AARCH64_PREFETCH &&
7985 auto OpFns = tryFoldAddLowIntoImm(*RootDef,
Size, MRI);
7990 if (isBaseWithConstantOffset(Root, MRI)) {
7998 if ((RHSC & (
Size - 1)) == 0 && RHSC >= 0 && RHSC < (0x1000 << Scale)) {
7999 if (LHSDef->
getOpcode() == TargetOpcode::G_FRAME_INDEX)
8001 [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->
getOperand(1)); },
8002 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC >> Scale); },
8006 [=](MachineInstrBuilder &MIB) { MIB.add(
LHS); },
8007 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC >> Scale); },
8014 if (selectAddrModeUnscaled(Root,
Size))
8015 return std::nullopt;
8018 [=](MachineInstrBuilder &MIB) { MIB.add(Root); },
8019 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
8026 switch (
MI.getOpcode()) {
8029 case TargetOpcode::G_SHL:
8031 case TargetOpcode::G_LSHR:
8033 case TargetOpcode::G_ASHR:
8035 case TargetOpcode::G_ROTR:
8042InstructionSelector::ComplexRendererFns
8043AArch64InstructionSelector::selectShiftedRegister(MachineOperand &Root,
8044 bool AllowROR)
const {
8046 return std::nullopt;
8047 MachineRegisterInfo &MRI =
8055 return std::nullopt;
8057 return std::nullopt;
8058 if (!isWorthFoldingIntoExtendedReg(*ShiftInst, MRI,
false))
8059 return std::nullopt;
8062 MachineOperand &ShiftRHS = ShiftInst->
getOperand(2);
8065 return std::nullopt;
8069 MachineOperand &ShiftLHS = ShiftInst->
getOperand(1);
8073 unsigned Val = *Immed & (NumBits - 1);
8076 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ShiftReg); },
8077 [=](MachineInstrBuilder &MIB) { MIB.addImm(ShiftVal); }}};
8081 MachineInstr &
MI, MachineRegisterInfo &MRI,
bool IsLoadStore)
const {
8082 unsigned Opc =
MI.getOpcode();
8085 if (
Opc == TargetOpcode::G_SEXT ||
Opc == TargetOpcode::G_SEXT_INREG) {
8087 if (
Opc == TargetOpcode::G_SEXT)
8090 Size =
MI.getOperand(2).getImm();
8091 assert(
Size != 64 &&
"Extend from 64 bits?");
8104 if (
Opc == TargetOpcode::G_ZEXT ||
Opc == TargetOpcode::G_ANYEXT) {
8106 assert(
Size != 64 &&
"Extend from 64 bits?");
8121 if (
Opc != TargetOpcode::G_AND)
8140Register AArch64InstructionSelector::moveScalarRegClass(
8142 MachineRegisterInfo &MRI = *MIB.
getMRI();
8152 return Copy.getReg(0);
8157InstructionSelector::ComplexRendererFns
8158AArch64InstructionSelector::selectArithExtendedRegister(
8159 MachineOperand &Root)
const {
8161 return std::nullopt;
8162 MachineRegisterInfo &MRI =
8170 return std::nullopt;
8172 if (!isWorthFoldingIntoExtendedReg(*RootDef, MRI,
false))
8173 return std::nullopt;
8176 if (RootDef->
getOpcode() == TargetOpcode::G_SHL) {
8181 return std::nullopt;
8182 ShiftVal = *MaybeShiftVal;
8184 return std::nullopt;
8189 return std::nullopt;
8190 Ext = getExtendTypeForInst(*ExtDef, MRI);
8192 return std::nullopt;
8196 Ext = getExtendTypeForInst(*RootDef, MRI);
8198 return std::nullopt;
8206 MachineInstr *ExtInst = MRI.
getVRegDef(ExtReg);
8207 if (isDef32(*ExtInst))
8208 return std::nullopt;
8214 MachineIRBuilder MIB(*RootDef);
8215 ExtReg = moveScalarRegClass(ExtReg, AArch64::GPR32RegClass, MIB);
8217 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); },
8218 [=](MachineInstrBuilder &MIB) {
8219 MIB.addImm(getArithExtendImm(Ext, ShiftVal));
8223InstructionSelector::ComplexRendererFns
8224AArch64InstructionSelector::selectExtractHigh(MachineOperand &Root)
const {
8226 return std::nullopt;
8227 MachineRegisterInfo &MRI =
8231 while (Extract && Extract->MI->
getOpcode() == TargetOpcode::G_BITCAST &&
8236 return std::nullopt;
8239 if (Unmerge->getNumDefs() == 2 &&
8241 Register ExtReg = Unmerge->getSourceReg();
8242 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8246 LLT SrcTy = MRI.
getType(ExtElt->getVectorReg());
8250 LaneIdx->Value.getSExtValue() == 1) {
8251 Register ExtReg = ExtElt->getVectorReg();
8252 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8256 LLT SrcTy = MRI.
getType(Subvec->getSrcVec());
8257 auto LaneIdx = Subvec->getIndexImm();
8259 Register ExtReg = Subvec->getSrcVec();
8260 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8264 return std::nullopt;
8267InstructionSelector::ComplexRendererFns
8268AArch64InstructionSelector::selectCVTFixedPointBase(
const MachineOperand &Root,
8269 unsigned DstElemWidth,
8270 bool isReciprocal)
const {
8272 return std::nullopt;
8273 const MachineRegisterInfo &MRI =
8279 if (Dup && Dup->
getOpcode() == AArch64::G_DUP)
8282 std::optional<ValueAndVReg> CstVal =
8286 return std::nullopt;
8290 switch (CstElemWidth) {
8292 FVal =
APFloat(APFloat::IEEEhalf(), CstVal->Value);
8295 FVal =
APFloat(APFloat::IEEEsingle(), CstVal->Value);
8298 FVal =
APFloat(APFloat::IEEEdouble(), CstVal->Value);
8301 return std::nullopt;
8303 if (
unsigned FBits =
8305 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(FBits); }}};
8307 return std::nullopt;
8310unsigned AArch64InstructionSelector::getFixedPointWidthFromOperand(
8311 const MachineOperand &Root)
const {
8319template <
unsigned W
idth>
8320InstructionSelector::ComplexRendererFns
8321AArch64InstructionSelector::selectCVTFixedPoint(MachineOperand &Root)
const {
8322 return selectCVTFixedPointBase(Root, Width,
false);
8325template <
unsigned W
idth>
8326InstructionSelector::ComplexRendererFns
8327AArch64InstructionSelector::selectCVTFixedPosRecipOperand(
8328 MachineOperand &Root)
const {
8329 return selectCVTFixedPointBase(Root, Width,
true);
8332InstructionSelector::ComplexRendererFns
8333AArch64InstructionSelector::selectCVTFixedPointVec(MachineOperand &Root)
const {
8334 return selectCVTFixedPointBase(Root, getFixedPointWidthFromOperand(Root),
8338InstructionSelector::ComplexRendererFns
8339AArch64InstructionSelector::selectCVTFixedPosRecipOperandVec(
8340 MachineOperand &Root)
const {
8341 return selectCVTFixedPointBase(Root, getFixedPointWidthFromOperand(Root),
8345void AArch64InstructionSelector::renderFixedPointScalarXForm(
8346 MachineInstrBuilder &MIB,
const MachineInstr &
MI,
int OpIdx)
const {
8347 assert(OpIdx == 3 &&
MI.getOperand(OpIdx).isImm() &&
8348 "Expected vecshift immediate operand");
8349 MIB.
addImm(
MI.getOperand(OpIdx).getImm());
8352void AArch64InstructionSelector::renderFixedPointImm(MachineInstrBuilder &MIB,
8353 const MachineOperand &Root,
8355 bool isReciprocal)
const {
8359 InstructionSelector::ComplexRendererFns Renderer =
8360 selectCVTFixedPointBase(Root, Width, isReciprocal);
8361 assert((Renderer && Renderer->size() == 1) &&
8362 "Expected selectCVTFixedPointBase to provide a function\n");
8363 (Renderer->front())(MIB);
8366void AArch64InstructionSelector::renderFixedPointXForm(MachineInstrBuilder &MIB,
8367 const MachineInstr &
MI,
8369 const MachineOperand &Root =
MI.getOperand(OpIdx);
8370 renderFixedPointImm(MIB, Root, getFixedPointWidthFromOperand(Root),
8374void AArch64InstructionSelector::renderFixedPointRecipXForm(
8375 MachineInstrBuilder &MIB,
const MachineInstr &
MI,
int OpIdx)
const {
8376 const MachineOperand &Root =
MI.getOperand(OpIdx);
8377 renderFixedPointImm(MIB, Root, getFixedPointWidthFromOperand(Root),
8381void AArch64InstructionSelector::renderTruncImm(MachineInstrBuilder &MIB,
8382 const MachineInstr &
MI,
8384 const MachineRegisterInfo &MRI =
MI.getParent()->getParent()->getRegInfo();
8385 assert(
MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8386 "Expected G_CONSTANT");
8387 std::optional<int64_t> CstVal =
8389 assert(CstVal &&
"Expected constant value");
8393void AArch64InstructionSelector::renderLogicalImm32(
8394 MachineInstrBuilder &MIB,
const MachineInstr &
I,
int OpIdx)
const {
8395 assert(
I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8396 "Expected G_CONSTANT");
8397 uint64_t CstVal =
I.getOperand(1).getCImm()->getZExtValue();
8402void AArch64InstructionSelector::renderLogicalImm64(
8403 MachineInstrBuilder &MIB,
const MachineInstr &
I,
int OpIdx)
const {
8404 assert(
I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8405 "Expected G_CONSTANT");
8406 uint64_t CstVal =
I.getOperand(1).getCImm()->getZExtValue();
8411void AArch64InstructionSelector::renderUbsanTrap(MachineInstrBuilder &MIB,
8412 const MachineInstr &
MI,
8414 assert(
MI.getOpcode() == TargetOpcode::G_UBSANTRAP && OpIdx == 0 &&
8415 "Expected G_UBSANTRAP");
8416 MIB.
addImm(
MI.getOperand(0).getImm() | (
'U' << 8));
8419void AArch64InstructionSelector::renderFPImm16(MachineInstrBuilder &MIB,
8420 const MachineInstr &
MI,
8422 assert(
MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8423 "Expected G_FCONSTANT");
8428void AArch64InstructionSelector::renderFPImm32(MachineInstrBuilder &MIB,
8429 const MachineInstr &
MI,
8431 assert(
MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8432 "Expected G_FCONSTANT");
8437void AArch64InstructionSelector::renderFPImm64(MachineInstrBuilder &MIB,
8438 const MachineInstr &
MI,
8440 assert(
MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8441 "Expected G_FCONSTANT");
8446void AArch64InstructionSelector::renderFPImm32SIMDModImmType4(
8447 MachineInstrBuilder &MIB,
const MachineInstr &
MI,
int OpIdx)
const {
8448 assert(
MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8449 "Expected G_FCONSTANT");
8457bool AArch64InstructionSelector::isLoadStoreOfNumBytes(
8458 const MachineInstr &
MI,
unsigned NumBytes)
const {
8459 if (!
MI.mayLoadOrStore())
8462 "Expected load/store to have only one mem op!");
8463 return (*
MI.memoperands_begin())->getSize() == NumBytes;
8466bool AArch64InstructionSelector::isDef32(
const MachineInstr &
MI)
const {
8467 const MachineRegisterInfo &MRI =
MI.getParent()->getParent()->getRegInfo();
8475 switch (
MI.getOpcode()) {
8478 case TargetOpcode::COPY:
8479 case TargetOpcode::G_BITCAST:
8480 case TargetOpcode::G_TRUNC:
8481 case TargetOpcode::G_PHI:
8491 assert(
MI.getOpcode() == TargetOpcode::G_PHI &&
"Expected a G_PHI");
8494 assert(DstRB &&
"Expected PHI dst to have regbank assigned");
8512 if (InsertPt != OpDefBB.
end() && InsertPt->isPHI())
8517 MO.setReg(Copy.getReg(0));
8526 for (
auto &BB : MF) {
8527 for (
auto &
MI : BB) {
8528 if (
MI.getOpcode() == TargetOpcode::G_PHI)
8533 for (
auto *
MI : Phis) {
8555 bool HasGPROp =
false, HasFPROp =
false;
8559 const LLT &Ty = MRI.
getType(MO.getReg());
8569 if (RB->
getID() == AArch64::GPRRegBankID)
8575 if (HasGPROp && HasFPROp)
8581InstructionSelector *
8585 return new AArch64InstructionSelector(TM, Subtarget, RBI);
MachineInstrBuilder MachineInstrBuilder & DefMI
static std::tuple< SDValue, SDValue > extractPtrauthBlendDiscriminators(SDValue Disc, SelectionDAG *DAG)
static bool isPreferredADD(int64_t ImmOff)
static SDValue emitConditionalComparison(SDValue LHS, SDValue RHS, ISD::CondCode CC, SDValue CCOp, AArch64CC::CondCode Predicate, AArch64CC::CondCode OutCC, const SDLoc &DL, SelectionDAG &DAG)
can be transformed to: not (and (not (and (setCC (cmp C)) (setCD (cmp D)))) (and (not (setCA (cmp A))...
static SDValue tryAdvSIMDModImm16(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits, const SDValue *LHS=nullptr)
static SDValue tryAdvSIMDModImmFP(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static SDValue tryAdvSIMDModImm64(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static bool isCMN(SDValue Op, ISD::CondCode CC, SelectionDAG &DAG)
static SDValue tryAdvSIMDModImm8(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static SDValue emitConjunctionRec(SelectionDAG &DAG, SDValue Val, AArch64CC::CondCode &OutCC, bool Negate, SDValue CCOp, AArch64CC::CondCode Predicate)
Emit conjunction or disjunction tree with the CMP/FCMP followed by a chain of CCMP/CFCMP ops.
static SDValue tryAdvSIMDModImm321s(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static void changeFPCCToANDAArch64CC(ISD::CondCode CC, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
Convert a DAG fp condition code to an AArch64 CC.
static bool canEmitConjunction(SelectionDAG &DAG, const SDValue Val, bool &CanNegate, bool &MustBeFirst, bool &PreferFirst, bool WillNegate, unsigned Depth=0)
Returns true if Val is a tree of AND/OR/SETCC operations that can be expressed as a conjunction.
static SDValue tryAdvSIMDModImm32(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits, const SDValue *LHS=nullptr)
static SDValue emitConjunction(SelectionDAG &DAG, SDValue Val, AArch64CC::CondCode &OutCC)
Emit expression as a conjunction (a series of CCMP/CFCMP ops).
#define GET_GLOBALISEL_PREDICATES_INIT
#define GET_GLOBALISEL_TEMPORARIES_INIT
static Register getTestBitReg(Register Reg, uint64_t &Bit, bool &Invert, MachineRegisterInfo &MRI)
Return a register which can be used as a bit to test in a TB(N)Z.
static unsigned getMinSizeForRegBank(const RegisterBank &RB)
Returns the minimum size the given register bank can hold.
static std::optional< int64_t > getVectorShiftImm(Register Reg, MachineRegisterInfo &MRI)
Returns the element immediate value of a vector shift operand if found.
static unsigned selectLoadStoreUIOp(unsigned GenericOpc, unsigned RegBankID, unsigned OpSize)
Select the AArch64 opcode for the G_LOAD or G_STORE operation GenericOpc, appropriate for the (value)...
static const TargetRegisterClass * getMinClassForRegBank(const RegisterBank &RB, TypeSize SizeInBits, bool GetAllRegSet=false)
Given a register bank, and size in bits, return the smallest register class that can represent that c...
static unsigned selectBinaryOp(unsigned GenericOpc, unsigned RegBankID, unsigned OpSize)
Select the AArch64 opcode for the basic binary operation GenericOpc, appropriate for the register ban...
static bool getSubRegForClass(const TargetRegisterClass *RC, const TargetRegisterInfo &TRI, unsigned &SubReg)
Returns the correct subregister to use for a given register class.
static bool selectCopy(MachineInstr &I, const TargetInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
static bool copySubReg(MachineInstr &I, MachineRegisterInfo &MRI, const RegisterBankInfo &RBI, Register SrcReg, const TargetRegisterClass *To, unsigned SubReg)
Helper function for selectCopy.
static AArch64CC::CondCode changeICMPPredToAArch64CC(CmpInst::Predicate P, Register RHS={}, MachineRegisterInfo *MRI=nullptr)
static Register createDTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of D-registers using the registers in Regs.
static void fixupPHIOpBanks(MachineInstr &MI, MachineRegisterInfo &MRI, const AArch64RegisterBankInfo &RBI)
static bool selectDebugInstr(MachineInstr &I, MachineRegisterInfo &MRI, const RegisterBankInfo &RBI)
static AArch64_AM::ShiftExtendType getShiftTypeForInst(MachineInstr &MI)
Given a shift instruction, return the correct shift type for that instruction.
static bool getLaneCopyOpcode(unsigned &CopyOpc, unsigned &ExtractSubReg, const unsigned EltSize)
static Register createQTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of Q-registers using the registers in Regs.
static std::optional< uint64_t > getImmedFromMO(const MachineOperand &Root)
static std::pair< unsigned, unsigned > getInsertVecEltOpInfo(const RegisterBank &RB, unsigned EltSize)
Return an <Opcode, SubregIndex> pair to do an vector elt insert of a given size and RB.
static Register createTuple(ArrayRef< Register > Regs, const unsigned RegClassIDs[], const unsigned SubRegs[], MachineIRBuilder &MIB)
Create a REG_SEQUENCE instruction using the registers in Regs.
static std::optional< int64_t > getVectorSHLImm(LLT SrcTy, Register Reg, MachineRegisterInfo &MRI)
Matches and returns the shift immediate value for a SHL instruction given a shift operand.
static void changeFPCCToORAArch64CC(CmpInst::Predicate CC, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
changeFPCCToORAArch64CC - Convert an IR fp condition code to an AArch64 CC.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file declares the targeting of the RegisterBankInfo class for AArch64.
static bool isStore(int Opcode)
static bool selectMergeValues(MachineInstrBuilder &MIB, const ARMBaseInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
static bool selectUnmergeValues(MachineInstrBuilder &MIB, const ARMBaseInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file contains constants used for implementing Dwarf debug support.
Provides analysis for querying information about KnownBits during GISel passes.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
static void emitLoadFromConstantPool(Register DstReg, const Constant *ConstVal, MachineIRBuilder &MIRBuilder)
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
This file declares the MachineIRBuilder class.
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static MachineBasicBlock * emitSelect(MachineInstr &MI, MachineBasicBlock *BB, const TargetInstrInfo *TII, const PPCSubtarget &Subtarget)
Emit SELECT instruction, using ISEL if available, otherwise use branch-based control flow.
static StringRef getName(Value *V)
static constexpr int Concat[]
unsigned getVarArgsFPRSize() const
bool hasELFSignedGOT() const
int getVarArgsFPRIndex() const
void incNumLocalDynamicTLSAccesses()
int getVarArgsStackIndex() const
int getVarArgsGPRIndex() const
unsigned getVarArgsGPRSize() const
This class provides the information for the target register banks.
bool isTargetDarwin() const
bool isTargetILP32() const
std::optional< uint16_t > getPtrAuthBlockAddressDiscriminatorIfEnabled(const Function &ParentFn) const
Compute the integer discriminator for a given BlockAddress constant, if blockaddress signing is enabl...
const AArch64TargetLowering * getTargetLowering() const override
bool isTargetMachO() const
unsigned ClassifyGlobalReference(const GlobalValue *GV, const TargetMachine &TM) const
ClassifyGlobalReference - Find the target operand flags that describe how a global value should be re...
bool isLittleEndian() const
bool isX16X17Safer() const
Returns whether the operating system makes it safer to store sensitive values in x16 and x17 as oppos...
bool isCallingConvWin64(CallingConv::ID CC, bool IsVarArg) const
APInt bitcastToAPInt() const
Class for arbitrary precision integers.
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
uint64_t getZExtValue() const
Get zero extended value.
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
unsigned countr_one() const
Count the number of trailing one bits.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
bool isEquality() const
Determine if this is an equals/not equals predicate.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
bool isIntPredicate() const
static LLVM_ABI Constant * getSplat(unsigned NumElts, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
const APFloat & getValueAPF() const
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
unsigned getBitWidth() const
getBitWidth - Return the scalar bitwidth of this constant.
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
LLVM_ABI Align getPrefTypeAlign(Type *Ty) const
Returns the preferred stack/global alignment for the specified type.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
bool isVarArg() const
isVarArg - Return true if this function takes a variable number of arguments.
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
virtual void setupMF(MachineFunction &mf, GISelValueTracking *vt, CodeGenCoverage *covinfo=nullptr, ProfileSummaryInfo *psi=nullptr, BlockFrequencyInfo *bfi=nullptr)
Setup per-MF executor state.
Represents indexed stores.
Register getPointerReg() const
Get the source register of the pointer value.
MachineMemOperand & getMMO() const
Get the MachineMemOperand on this instruction.
LocationSize getMemSize() const
Returns the size in bytes of the memory access.
LocationSize getMemSizeInBits() const
Returns the size in bits of the memory access.
Register getCondReg() const
Register getFalseReg() const
Register getTrueReg() const
Register getReg(unsigned Idx) const
Access the Idx'th operand as a register and return it.
bool isThreadLocal() const
If the value is "Thread Local", its value isn't shared by the threads.
bool hasExternalWeakLinkage() const
bool isEquality() const
Return true if this predicate is either EQ or NE.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr bool isScalar() const
LLT multiplyElements(int Factor) const
Produce a vector type that is Factor times bigger, preserving the element type.
constexpr LLT changeElementType(LLT NewEltTy) const
If this type is a vector, return a vector with the same number of elements but the new element type.
LLT getScalarType() const
constexpr bool isPointerVector() const
constexpr bool isInteger() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isValid() const
constexpr uint16_t getNumElements() const
Returns the number of elements in a vector LLT.
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr bool isPointer() const
constexpr unsigned getAddressSpace() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
static LLT integer(unsigned SizeInBits)
constexpr TypeSize getSizeInBytes() const
Returns the total size of the type in bytes, i.e.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
TypeSize getValue() const
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LLVM_ABI unsigned getConstantPoolIndex(const Constant *C, Align Alignment)
getConstantPoolIndex - Create a new entry in the constant pool or return an existing one.
void setAdjustsStack(bool V)
void setFrameAddressIsTaken(bool T)
void setReturnAddressIsTaken(bool s)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineConstantPool * getConstantPool()
getConstantPool - Return the constant pool object for the current function.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Helper class to build MachineInstr.
void setInsertPt(MachineBasicBlock &MBB, MachineBasicBlock::iterator II)
Set the insertion point before the specified position.
void setInstr(MachineInstr &MI)
Set the insertion point to before MI.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
MachineFunction & getMF()
Getter for the function we currently build.
void setInstrAndDebugLoc(MachineInstr &MI)
Set the insertion point to before MI, and set the debug loc to MI's loc.
const MachineBasicBlock & getMBB() const
Getter for the basic block we currently build.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineIRBuilderState & getState()
Getter for the State.
MachineInstrBuilder buildCopy(const DstOp &Res, const SrcOp &Op)
Build and insert Res = COPY Op.
const DataLayout & getDataLayout() const
void setState(const MachineIRBuilderState &NewState)
Setter for the State.
MachineInstrBuilder buildPtrToInt(const DstOp &Dst, const SrcOp &Src)
Build and insert a G_PTRTOINT instruction.
Register getReg(unsigned Idx) const
Get the register for the operand index.
void constrainAllUses(const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addBlockAddress(const BlockAddress *BA, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addJumpTableIndex(unsigned Idx, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
void setImplicitPhysRegDefsDead()
Mark the implicit physreg defs named by the instruction description as dead.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
LLVM_ABI void addMemOperand(MachineFunction &MF, MachineMemOperand *MO)
Add a MachineMemOperand to the machine instruction.
LLT getMemoryType() const
Return the memory type of the memory reference.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
AtomicOrdering getSuccessOrdering() const
Return the atomic ordering requirements for this memory operation.
MachineOperand class - Representation of each machine instruction operand.
const GlobalValue * getGlobal() const
const ConstantInt * getCImm() const
bool isCImm() const
isCImm - Test if this is a MO_CImmediate operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
static MachineOperand CreatePredicate(unsigned Pred)
static MachineOperand CreateImm(int64_t Val)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
static MachineOperand CreateBA(const BlockAddress *BA, int64_t Offset, unsigned TargetFlags=0)
const ConstantFP * getFPImm() const
unsigned getPredicate() const
int64_t getOffset() const
Return the offset from the symbol in this operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
const RegClassOrRegBank & getRegClassOrRegBank(Register Reg) const
Return the register bank or register class of Reg.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
def_instr_iterator def_instr_begin(Register RegNo) const
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool hasOneUse(Register RegNo) const
hasOneUse - Return true if there is exactly one instruction using the specified register.
const RegisterBank * getRegBankOrNull(Register Reg) const
Return the register bank of Reg, or null if Reg has not been assigned a register bank or has been ass...
LLVM_ABI void setRegBank(Register Reg, const RegisterBank &RegBank)
Set the register bank to RegBank for Reg.
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
LLVM_ABI void setType(Register VReg, LLT Ty)
Set the low-level type of VReg to Ty.
const MachineFunction & getMF() const
bool hasOneDef(Register RegNo) const
Return true if there is exactly one operand defining the specified register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register createGenericVirtualRegister(LLT Ty, StringRef Name="")
Create and return a new generic virtual register with low-level type Ty.
const TargetRegisterClass * getRegClassOrNull(Register Reg) const
Return the register class of Reg, or null if Reg has not been assigned a register class yet.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
Analysis providing profile information.
Holds all the information related to register banks.
static const TargetRegisterClass * constrainGenericRegister(Register Reg, const TargetRegisterClass &RC, MachineRegisterInfo &MRI)
Constrain the (possibly generic) virtual register Reg to RC.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
TypeSize getSizeInBits(Register Reg, const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI) const
Get the size in bits of Reg.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
constexpr bool isValid() const
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
TargetInstrInfo - Interface to description of machine instruction set.
bool isPositionIndependent() const
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
CodeModel::Model getCodeModel() const
Returns the code model.
unsigned TLSSize
Bit size of immediate TLS offsets (0 == use the default).
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
virtual const TargetLowering * getTargetLowering() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static CondCode getInvertedCondCode(CondCode Code)
static unsigned getNZCVToSatisfyCondCode(CondCode Code)
Given a condition code, return NZCV flags that would satisfy that condition.
void changeFCMPPredToAArch64CC(const CmpInst::Predicate P, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
Find the AArch64 condition codes necessary to represent P for a scalar floating point comparison.
std::optional< int64_t > getAArch64VectorSplatScalar(const MachineInstr &MI, const MachineRegisterInfo &MRI)
@ MO_NC
MO_NC - Indicates whether the linker is expected to check the symbol reference for overflow.
@ MO_G1
MO_G1 - A symbol operand with this flag (granule 1) represents the bits 16-31 of a 64-bit address,...
@ MO_PAGEOFF
MO_PAGEOFF - A symbol operand with this flag represents the offset of that symbol within a 4K page.
@ MO_GOT
MO_GOT - This flag indicates that a symbol operand represents the address of the GOT entry for the sy...
@ MO_G0
MO_G0 - A symbol operand with this flag (granule 0) represents the bits 0-15 of a 64-bit address,...
@ MO_PAGE
MO_PAGE - A symbol operand with this flag represents the pc-relative offset of the 4K page containing...
@ MO_HI12
MO_HI12 - This flag indicates that a symbol operand represents the bits 13-24 of a 64-bit address,...
@ MO_TLS
MO_TLS - Indicates that the operand being accessed is some kind of thread-local symbol.
@ MO_G2
MO_G2 - A symbol operand with this flag (granule 2) represents the bits 32-47 of a 64-bit address,...
@ MO_G3
MO_G3 - A symbol operand with this flag (granule 3) represents the high 16-bits of a 64-bit address,...
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
static uint8_t encodeAdvSIMDModImmType2(uint64_t Imm)
static bool isAdvSIMDModImmType9(uint64_t Imm)
static bool isAdvSIMDModImmType4(uint64_t Imm)
static bool isAdvSIMDModImmType5(uint64_t Imm)
static int getFP32Imm(const APInt &Imm)
getFP32Imm - Return an 8-bit floating-point version of the 32-bit floating-point value.
static uint8_t encodeAdvSIMDModImmType7(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType12(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType10(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType9(uint64_t Imm)
static uint64_t encodeLogicalImmediate(uint64_t imm, unsigned regSize)
encodeLogicalImmediate - Return the encoded immediate value for a logical immediate instruction of th...
static bool isAdvSIMDModImmType7(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType5(uint64_t Imm)
static int getFP64Imm(const APInt &Imm)
getFP64Imm - Return an 8-bit floating-point version of the 64-bit floating-point value.
static bool isAdvSIMDModImmType10(uint64_t Imm)
static int getFP16Imm(const APInt &Imm)
getFP16Imm - Return an 8-bit floating-point version of the 16-bit floating-point value.
static uint8_t encodeAdvSIMDModImmType8(uint64_t Imm)
static bool isAdvSIMDModImmType12(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType11(uint64_t Imm)
static bool isAdvSIMDModImmType11(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType6(uint64_t Imm)
static bool isAdvSIMDModImmType8(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType4(uint64_t Imm)
static unsigned getShifterImm(AArch64_AM::ShiftExtendType ST, unsigned Imm)
getShifterImm - Encode the shift type and amount: imm: 6-bit shift amount shifter: 000 ==> lsl 001 ==...
static bool isAdvSIMDModImmType6(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType1(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType3(uint64_t Imm)
static bool isAdvSIMDModImmType2(uint64_t Imm)
static bool isAdvSIMDModImmType3(uint64_t Imm)
static bool isSignExtendShiftType(AArch64_AM::ShiftExtendType Type)
isSignExtendShiftType - Returns true if Type is sign extending.
static bool isAdvSIMDModImmType1(uint64_t Imm)
TLSModel::Model getELFTLSModel(const GlobalValue *GV, const AArch64TargetMachine &TM, bool HasELFSignedGOT)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
operand_type_match m_Reg()
SpecificConstantMatch m_SpecificICst(const APInt &RequestedValue)
Matches a constant equal to RequestedValue.
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
BinaryOp_match< LHS, RHS, TargetOpcode::G_ADD, true > m_GAdd(const LHS &L, const RHS &R)
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< LHS, RHS, TargetOpcode::G_OR, true > m_GOr(const LHS &L, const RHS &R)
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
ICstOrSplatMatch< APInt > m_ICstOrSplat(APInt &Cst)
OneNonDBGUse_match< SubPat > m_OneNonDBGUse(const SubPat &SP)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
BinaryOp_match< LHS, RHS, TargetOpcode::G_SUB > m_GSub(const LHS &L, const RHS &R)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
Or< Preds... > m_any_of(Preds &&... preds)
BinaryOp_match< LHS, RHS, TargetOpcode::G_AND, true > m_GAnd(const LHS &L, const RHS &R)
Predicate
Predicate - These are "(BI << 5) | BO" for various predicates.
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
NodeAddr< InstrNode * > Instr
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Register getFunctionLiveInPhysReg(MachineFunction &MF, const TargetInstrInfo &TII, MCRegister PhysReg, const TargetRegisterClass &RC, const DebugLoc &DL, LLT RegTy=LLT())
Return a virtual register corresponding to the incoming argument register PhysReg.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI Register constrainOperandRegClass(const MachineFunction &MF, const TargetRegisterInfo &TRI, MachineRegisterInfo &MRI, const TargetInstrInfo &TII, const RegisterBankInfo &RBI, MachineInstr &InsertPt, const TargetRegisterClass &RegClass, MachineOperand &RegMO)
Constrain the Register operand OpIdx, so that it is now constrained to the TargetRegisterClass passed...
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI std::optional< APInt > getIConstantVRegVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT, return the corresponding value.
unsigned CheckFixedPointOperandConstant(APFloat &FVal, unsigned RegWidth, bool isReciprocal)
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
bool isStrongerThanMonotonic(AtomicOrdering AO)
LLVM_ABI void constrainSelectedInstRegOperands(MachineInstr &I, const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
Mutate the newly-selected instruction I to constrain its (possibly generic) virtual register operands...
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool isPreISelGenericOpcode(unsigned Opcode)
Check whether the given Opcode is a generic opcode that is not supposed to appear after ISel.
unsigned getBLRCallOpcode(const MachineFunction &MF)
Return opcode to be used for indirect calls.
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI std::optional< int64_t > getIConstantVRegSExtVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT fits in int64_t returns it.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
InstructionSelector * createAArch64InstructionSelector(const AArch64TargetMachine &, const AArch64Subtarget &, const AArch64RegisterBankInfo &)
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
constexpr bool has_single_bit(T Value) noexcept
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI std::optional< ValueAndVReg > getAnyConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true, bool LookThroughAnyExt=false)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT or G_FCONST...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
LLVM_ABI std::optional< DefinitionAndSourceRegister > getDefSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, and underlying value Register folding away any copies.
LLVM_ABI Register getSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the source register for Reg, folding away any trivial copies.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
static EVT getFloatingPointVT(unsigned BitWidth)
Returns the EVT that represents a floating-point type with the given number of bits.
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.