LLVM 24.0.0git
AArch64InstrInfo.cpp
Go to the documentation of this file.
1//===- AArch64InstrInfo.cpp - AArch64 Instruction Information -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the AArch64 implementation of the TargetInstrInfo class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AArch64InstrInfo.h"
14#include "AArch64ExpandImm.h"
16#include "AArch64PointerAuth.h"
17#include "AArch64Subtarget.h"
22#include "llvm/ADT/ArrayRef.h"
23#include "llvm/ADT/STLExtras.h"
24#include "llvm/ADT/SmallSet.h"
26#include "llvm/ADT/Statistic.h"
45#include "llvm/IR/DebugLoc.h"
46#include "llvm/IR/GlobalValue.h"
47#include "llvm/IR/Module.h"
48#include "llvm/MC/MCAsmInfo.h"
49#include "llvm/MC/MCInst.h"
51#include "llvm/MC/MCInstrDesc.h"
56#include "llvm/Support/LEB128.h"
60#include <cassert>
61#include <cstdint>
62#include <iterator>
63#include <utility>
64
65using namespace llvm;
66
67#define GET_INSTRINFO_CTOR_DTOR
68#include "AArch64GenInstrInfo.inc"
69
70#define DEBUG_TYPE "AArch64InstrInfo"
71
72STATISTIC(NumCopyInstrs, "Number of COPY instructions expanded");
73STATISTIC(NumZCRegMoveInstrsGPR, "Number of zero-cycle GPR register move "
74 "instructions expanded from canonical COPY");
75STATISTIC(NumZCRegMoveInstrsFPR, "Number of zero-cycle FPR register move "
76 "instructions expanded from canonical COPY");
77STATISTIC(NumZCZeroingInstrsGPR, "Number of zero-cycle GPR zeroing "
78 "instructions expanded from canonical COPY");
79// NumZCZeroingInstrsFPR is counted at AArch64AsmPrinter
80
82 CBDisplacementBits("aarch64-cb-offset-bits", cl::Hidden, cl::init(9),
83 cl::desc("Restrict range of CB instructions (DEBUG)"));
84
86 "aarch64-tbz-offset-bits", cl::Hidden, cl::init(14),
87 cl::desc("Restrict range of TB[N]Z instructions (DEBUG)"));
88
90 "aarch64-cbz-offset-bits", cl::Hidden, cl::init(19),
91 cl::desc("Restrict range of CB[N]Z instructions (DEBUG)"));
92
94 BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19),
95 cl::desc("Restrict range of Bcc instructions (DEBUG)"));
96
98 BDisplacementBits("aarch64-b-offset-bits", cl::Hidden, cl::init(26),
99 cl::desc("Restrict range of B instructions (DEBUG)"));
100
102 "aarch64-search-limit", cl::Hidden, cl::init(2048),
103 cl::desc("Restrict range of instructions to search for the "
104 "machine-combiner gather pattern optimization"));
105
107 "aarch64-outliner-compact-unwind-frame", cl::Hidden, cl::init(true),
108 cl::desc("Use a frame record for Mach-O non-leaf outlined functions"));
109
111 : AArch64GenInstrInfo(STI, RI, AArch64::ADJCALLSTACKDOWN,
112 AArch64::ADJCALLSTACKUP, AArch64::CATCHRET),
113 RI(STI.getTargetTriple(), STI.getHwMode()), Subtarget(STI) {}
114
115/// Return the maximum number of bytes of code the specified instruction may be
116/// after LFI rewriting. If the instruction is not rewritten, std::nullopt is
117/// returned (use default sizing).
118///
119/// NOTE: the size estimates here must be kept in sync with the rewrites in
120/// AArch64MCLFIRewriter.cpp. Sizes may be overestimates of the rewritten
121/// instruction sequences.
122static std::optional<unsigned> getLFIInstSizeInBytes(const MachineInstr &MI) {
123 switch (MI.getOpcode()) {
124 case AArch64::SVC:
125 // SVC expands to 4 instructions.
126 return 16;
127 case AArch64::BR:
128 case AArch64::BLR:
129 // Indirect branches/calls expand to 2 instructions (guard + br/blr).
130 return 8;
131 case AArch64::RET:
132 // RET through another register expands to 2 instructions (guard + ret).
133 // RET through LR may also expand to 2 instructions if a deferred LR guard
134 // is flushed before the return.
135 return 8;
136 case AArch64::RETAA:
137 case AArch64::RETAB:
138 // Authenticated returns expand to 3 instructions (authenticate + guard +
139 // ret).
140 return 12;
141 case AArch64::BRAA:
142 case AArch64::BRAAZ:
143 case AArch64::BRAB:
144 case AArch64::BRABZ:
145 case AArch64::BLRAA:
146 case AArch64::BLRAAZ:
147 case AArch64::BLRAB:
148 case AArch64::BLRABZ:
149 // Authenticated branches/calls expand to 3 instructions (authenticate +
150 // guard + branch).
151 return 12;
152 case AArch64::AUTIASP:
153 case AArch64::AUTIBSP:
154 case AArch64::AUTIAZ:
155 case AArch64::AUTIBZ:
156 case AArch64::XPACLRI:
157 // Authenticating LR expands to the instruction plus a deferred LR guard.
158 return 8;
159 case AArch64::SYSxt:
160 // VA-based DC/IC ops (op1=3, Cn=7, op2=1) expand to 2 instructions.
161 if (MI.getOperand(0).getImm() == 3 && MI.getOperand(1).getImm() == 7 &&
162 MI.getOperand(3).getImm() == 1)
163 return 8;
164 return std::nullopt;
165 default:
166 break;
167 }
168
169 // Detect instructions that explicitly define SP or LR.
170 bool ModifiesLR = false;
171 bool ModifiesSP = false;
172 for (const MachineOperand &MO : MI.defs()) {
173 if (!MO.isReg())
174 continue;
175 if (MO.getReg() == AArch64::LR)
176 ModifiesLR = true;
177 else if (MO.getReg() == AArch64::SP)
178 ModifiesSP = true;
179 }
180
181 // Memory accesses expand to a base-register guard plus the rewritten access
182 // (8 bytes), with an extra base-register update for pre/post-index forms (12
183 // bytes total). If the access also defines LR, an LR mask is appended (+4
184 // bytes). Depending on additional optimizations that the rewriter performs,
185 // this may be an overestimate.
186 if (MI.mayLoadOrStore()) {
187 unsigned Size = isLFIPrePostMemAccess(MI.getOpcode()) ? 12 : 8;
188 if (ModifiesLR)
189 Size += 4;
190 return Size;
191 }
192
193 // Non memory operations that modify LR or SP expand to 2 instructions.
194 if (ModifiesSP || ModifiesLR)
195 return 8;
196
197 // Default case: instructions that don't cause expansion.
198 // - TP accesses in LFI are a single load/store, so no expansion.
199 // - All remaining instructions are not rewritten.
200 return std::nullopt;
201}
202
203/// GetInstSize - Return the number of bytes of code the specified
204/// instruction may be. This returns the maximum number of bytes.
206 const MCInstrDesc &Desc = MI.getDesc();
207 if (!Desc.isPseudo() && !Subtarget.isLFI()) {
208 assert(Desc.getSize() == 4 && "Unexpected instruction size");
209 return 4;
210 }
211
212 const MachineBasicBlock &MBB = *MI.getParent();
213 const MachineFunction *MF = MBB.getParent();
214 const Function &F = MF->getFunction();
215 const MCAsmInfo &MAI = MF->getTarget().getMCAsmInfo();
216
217 {
218 auto Op = MI.getOpcode();
219 if (Op == AArch64::INLINEASM || Op == AArch64::INLINEASM_BR)
220 return getInlineAsmLength(MI.getOperand(0).getSymbolName(), MAI);
221 }
222
223 // Meta-instructions emit no code.
224 if (MI.isMetaInstruction())
225 return 0;
226
227 // FIXME: We currently only handle pseudoinstructions that don't get expanded
228 // before the assembly printer.
229 unsigned NumBytes = 0;
230
231 // LFI rewriter expansions that supersede normal sizing.
232 const auto &STI = MF->getSubtarget<AArch64Subtarget>();
233 if (STI.isLFI())
234 if (auto Size = getLFIInstSizeInBytes(MI))
235 return *Size;
236
237 if (!MI.isBundle() && isTailCallReturnInst(MI)) {
238 NumBytes = Desc.getSize() ? Desc.getSize() : 4;
239
240 const auto *MFI = MF->getInfo<AArch64FunctionInfo>();
241 if (!MFI->shouldSignReturnAddress(*MF))
242 return NumBytes;
243
244 auto Method = STI.getAuthenticatedLRCheckMethod(*MF);
245 NumBytes += AArch64PAuth::getCheckerSizeInBytes(Method);
246 return NumBytes;
247 }
248
249 // Size should be preferably set in
250 // llvm/lib/Target/AArch64/AArch64InstrInfo.td (default case).
251 // Specific cases handle instructions of variable sizes
252 switch (Desc.getOpcode()) {
253 default:
254 if (Desc.getSize())
255 return Desc.getSize();
256
257 // Anything not explicitly designated otherwise (i.e. pseudo-instructions
258 // with fixed constant size but not specified in .td file) is a normal
259 // 4-byte insn.
260 NumBytes = 4;
261 break;
262 case TargetOpcode::STACKMAP:
263 // The upper bound for a stackmap intrinsic is the full length of its shadow
264 NumBytes = StackMapOpers(&MI).getNumPatchBytes();
265 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
266 break;
267 case TargetOpcode::PATCHPOINT:
268 // The size of the patchpoint intrinsic is the number of bytes requested
269 NumBytes = PatchPointOpers(&MI).getNumPatchBytes();
270 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
271 break;
272 case TargetOpcode::STATEPOINT:
273 NumBytes = StatepointOpers(&MI).getNumPatchBytes();
274 assert(NumBytes % 4 == 0 && "Invalid number of NOP bytes requested!");
275 // No patch bytes means a normal call inst is emitted
276 if (NumBytes == 0)
277 NumBytes = 4;
278 break;
279 case TargetOpcode::PATCHABLE_FUNCTION_ENTER:
280 // If `patchable-function-entry` is set, PATCHABLE_FUNCTION_ENTER
281 // instructions are expanded to the specified number of NOPs. Otherwise,
282 // they are expanded to 36-byte XRay sleds.
283 NumBytes =
284 F.getFnAttributeAsParsedInteger("patchable-function-entry", 9) * 4;
285 break;
286 case TargetOpcode::PATCHABLE_FUNCTION_EXIT:
287 case TargetOpcode::PATCHABLE_TAIL_CALL:
288 case TargetOpcode::PATCHABLE_TYPED_EVENT_CALL:
289 // An XRay sled can be 4 bytes of alignment plus a 32-byte block.
290 NumBytes = 36;
291 break;
292 case TargetOpcode::PATCHABLE_EVENT_CALL:
293 // EVENT_CALL XRay sleds are exactly 6 instructions long (no alignment).
294 NumBytes = 24;
295 break;
296
297 case AArch64::SPACE:
298 NumBytes = MI.getOperand(1).getImm();
299 break;
300 case AArch64::MOVaddr:
301 case AArch64::MOVaddrJT:
302 case AArch64::MOVaddrCP:
303 case AArch64::MOVaddrBA:
304 case AArch64::MOVaddrTLS:
305 case AArch64::MOVaddrEXT: {
306 // Use the same logic as the pseudo expansion to count instructions.
309 MI.getOperand(1).getTargetFlags(),
310 Subtarget.isTargetMachO(), Insn);
311 NumBytes = Insn.size() * 4;
312 break;
313 }
314
315 case AArch64::MOVi32imm:
316 case AArch64::MOVi64imm: {
317 // Use the same logic as the pseudo expansion to count instructions.
318 unsigned BitSize = Desc.getOpcode() == AArch64::MOVi32imm ? 32 : 64;
320 AArch64_IMM::expandMOVImm(MI.getOperand(1).getImm(), BitSize, Insn);
321 NumBytes = Insn.size() * 4;
322 break;
323 }
324
325 case TargetOpcode::BUNDLE:
326 NumBytes = getInstBundleSize(MI);
327 break;
328 }
329
330 return NumBytes;
331}
332
335 // Block ends with fall-through condbranch.
336 switch (LastInst->getOpcode()) {
337 default:
338 llvm_unreachable("Unknown branch instruction?");
339 case AArch64::Bcc:
340 Target = LastInst->getOperand(1).getMBB();
341 Cond.push_back(LastInst->getOperand(0));
342 break;
343 case AArch64::CBZW:
344 case AArch64::CBZX:
345 case AArch64::CBNZW:
346 case AArch64::CBNZX:
347 Target = LastInst->getOperand(1).getMBB();
348 Cond.push_back(MachineOperand::CreateImm(-1));
349 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
350 Cond.push_back(LastInst->getOperand(0));
351 break;
352 case AArch64::TBZW:
353 case AArch64::TBZX:
354 case AArch64::TBNZW:
355 case AArch64::TBNZX:
356 Target = LastInst->getOperand(2).getMBB();
357 Cond.push_back(MachineOperand::CreateImm(-1));
358 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
359 Cond.push_back(LastInst->getOperand(0));
360 Cond.push_back(LastInst->getOperand(1));
361 break;
362 case AArch64::CBWPri:
363 case AArch64::CBXPri:
364 case AArch64::CBWPrr:
365 case AArch64::CBXPrr:
366 Target = LastInst->getOperand(3).getMBB();
367 Cond.push_back(MachineOperand::CreateImm(-1));
368 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode()));
369 Cond.push_back(LastInst->getOperand(0));
370 Cond.push_back(LastInst->getOperand(1));
371 Cond.push_back(LastInst->getOperand(2));
372 break;
373 case AArch64::CBBAssertExt:
374 case AArch64::CBHAssertExt:
375 Target = LastInst->getOperand(3).getMBB();
376 Cond.push_back(MachineOperand::CreateImm(-1)); // -1
377 Cond.push_back(MachineOperand::CreateImm(LastInst->getOpcode())); // Opc
378 Cond.push_back(LastInst->getOperand(0)); // Cond
379 Cond.push_back(LastInst->getOperand(1)); // Op0
380 Cond.push_back(LastInst->getOperand(2)); // Op1
381 Cond.push_back(LastInst->getOperand(4)); // Ext0
382 Cond.push_back(LastInst->getOperand(5)); // Ext1
383 break;
384 }
385}
386
387static unsigned getBranchDisplacementBits(unsigned Opc) {
388 switch (Opc) {
389 default:
390 llvm_unreachable("unexpected opcode!");
391 case AArch64::B:
392 return BDisplacementBits;
393 case AArch64::TBNZW:
394 case AArch64::TBZW:
395 case AArch64::TBNZX:
396 case AArch64::TBZX:
397 return TBZDisplacementBits;
398 case AArch64::CBNZW:
399 case AArch64::CBZW:
400 case AArch64::CBNZX:
401 case AArch64::CBZX:
402 return CBZDisplacementBits;
403 case AArch64::Bcc:
404 return BCCDisplacementBits;
405 case AArch64::CBWPri:
406 case AArch64::CBXPri:
407 case AArch64::CBBAssertExt:
408 case AArch64::CBHAssertExt:
409 case AArch64::CBWPrr:
410 case AArch64::CBXPrr:
411 return CBDisplacementBits;
412 }
413}
414
416 int64_t BrOffset) const {
417 unsigned Bits = getBranchDisplacementBits(BranchOp);
418 assert(Bits >= 3 && "max branch displacement must be enough to jump"
419 "over conditional branch expansion");
420 return isIntN(Bits, BrOffset / 4);
421}
422
425 switch (MI.getOpcode()) {
426 default:
427 llvm_unreachable("unexpected opcode!");
428 case AArch64::B:
429 return MI.getOperand(0).getMBB();
430 case AArch64::TBZW:
431 case AArch64::TBNZW:
432 case AArch64::TBZX:
433 case AArch64::TBNZX:
434 return MI.getOperand(2).getMBB();
435 case AArch64::CBZW:
436 case AArch64::CBNZW:
437 case AArch64::CBZX:
438 case AArch64::CBNZX:
439 case AArch64::Bcc:
440 return MI.getOperand(1).getMBB();
441 case AArch64::CBWPri:
442 case AArch64::CBXPri:
443 case AArch64::CBBAssertExt:
444 case AArch64::CBHAssertExt:
445 case AArch64::CBWPrr:
446 case AArch64::CBXPrr:
447 return MI.getOperand(3).getMBB();
448 }
449}
450
452 MachineBasicBlock &NewDestBB,
453 MachineBasicBlock &RestoreBB,
454 const DebugLoc &DL,
455 int64_t BrOffset,
456 RegScavenger *RS) const {
457 assert(RS && "RegScavenger required for long branching");
458 assert(MBB.empty() &&
459 "new block should be inserted for expanding unconditional branch");
460 assert(MBB.pred_size() == 1);
461 assert(RestoreBB.empty() &&
462 "restore block should be inserted for restoring clobbered registers");
463
464 auto buildIndirectBranch = [&](Register Reg, MachineBasicBlock &DestBB) {
465 // Offsets outside of the signed 33-bit range are not supported for ADRP +
466 // ADD.
467 if (!isInt<33>(BrOffset))
469 "Branch offsets outside of the signed 33-bit range not supported");
470
471 BuildMI(MBB, MBB.end(), DL, get(AArch64::ADRP), Reg)
472 .addSym(DestBB.getSymbol(), AArch64II::MO_PAGE);
473 BuildMI(MBB, MBB.end(), DL, get(AArch64::ADDXri), Reg)
474 .addReg(Reg)
475 .addSym(DestBB.getSymbol(), AArch64II::MO_PAGEOFF | AArch64II::MO_NC)
476 .addImm(0);
477 BuildMI(MBB, MBB.end(), DL, get(AArch64::BR)).addReg(Reg);
478 };
479
480 RS->enterBasicBlockEnd(MBB);
481 // If X16 is unused, we can rely on the linker to insert a range extension
482 // thunk if NewDestBB is out of range of a single B instruction.
483 constexpr Register Reg = AArch64::X16;
484 if (!RS->isRegUsed(Reg)) {
485 insertUnconditionalBranch(MBB, &NewDestBB, DL);
486 RS->setRegUsed(Reg);
487 return;
488 }
489
490 // In a cold block without BTI, insert the indirect branch if a register is
491 // free. Skip this if BTI is enabled to avoid inserting a BTI at the target,
492 // prioritizing a dynamic cost in cold code over a static cost in hot code.
493 AArch64FunctionInfo *AFI = MBB.getParent()->getInfo<AArch64FunctionInfo>();
494 bool HasBTI = AFI && AFI->branchTargetEnforcement();
495 if (MBB.getSectionID() == MBBSectionID::ColdSectionID && !HasBTI) {
496 Register Scavenged = RS->FindUnusedReg(&AArch64::GPR64RegClass);
497 if (Scavenged.isValid()) {
498 buildIndirectBranch(Scavenged, NewDestBB);
499 RS->setRegUsed(Scavenged);
500 return;
501 }
502 }
503
504 // Note: Spilling X16 briefly moves the stack pointer, making it incompatible
505 // with red zones.
506 if (!AFI || AFI->hasRedZone().value_or(true))
508 "Unable to insert indirect branch inside function that has red zone");
509
510 // Otherwise, spill X16 and defer range extension to the linker.
511 BuildMI(MBB, MBB.end(), DL, get(AArch64::STRXpre))
512 .addReg(AArch64::SP, RegState::Define)
513 .addReg(Reg)
514 .addReg(AArch64::SP)
515 .addImm(-16);
516
517 BuildMI(MBB, MBB.end(), DL, get(AArch64::B)).addMBB(&RestoreBB);
518
519 BuildMI(RestoreBB, RestoreBB.end(), DL, get(AArch64::LDRXpost))
520 .addReg(AArch64::SP, RegState::Define)
522 .addReg(AArch64::SP)
523 .addImm(16);
524}
525
526// Branch analysis.
529 MachineBasicBlock *&FBB,
531 bool AllowModify) const {
532 // If the block has no terminators, it just falls into the block after it.
533 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
534 if (I == MBB.end())
535 return false;
536
537 // Skip over SpeculationBarrierEndBB terminators
538 if (I->getOpcode() == AArch64::SpeculationBarrierISBDSBEndBB ||
539 I->getOpcode() == AArch64::SpeculationBarrierSBEndBB) {
540 --I;
541 }
542
543 if (!isUnpredicatedTerminator(*I))
544 return false;
545
546 // Get the last instruction in the block.
547 MachineInstr *LastInst = &*I;
548
549 // If there is only one terminator instruction, process it.
550 unsigned LastOpc = LastInst->getOpcode();
551 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
552 if (isUncondBranchOpcode(LastOpc)) {
553 TBB = LastInst->getOperand(0).getMBB();
554 return false;
555 }
556 if (isCondBranchOpcode(LastOpc)) {
557 // Block ends with fall-through condbranch.
558 parseCondBranch(LastInst, TBB, Cond);
559 return false;
560 }
561 return true; // Can't handle indirect branch.
562 }
563
564 // Get the instruction before it if it is a terminator.
565 MachineInstr *SecondLastInst = &*I;
566 unsigned SecondLastOpc = SecondLastInst->getOpcode();
567
568 // If AllowModify is true and the block ends with two or more unconditional
569 // branches, delete all but the first unconditional branch.
570 if (AllowModify && isUncondBranchOpcode(LastOpc)) {
571 while (isUncondBranchOpcode(SecondLastOpc)) {
572 LastInst->eraseFromParent();
573 LastInst = SecondLastInst;
574 LastOpc = LastInst->getOpcode();
575 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
576 // Return now the only terminator is an unconditional branch.
577 TBB = LastInst->getOperand(0).getMBB();
578 return false;
579 }
580 SecondLastInst = &*I;
581 SecondLastOpc = SecondLastInst->getOpcode();
582 }
583 }
584
585 // If we're allowed to modify and the block ends in a unconditional branch
586 // which could simply fallthrough, remove the branch. (Note: This case only
587 // matters when we can't understand the whole sequence, otherwise it's also
588 // handled by BranchFolding.cpp.)
589 if (AllowModify && isUncondBranchOpcode(LastOpc) &&
590 MBB.isLayoutSuccessor(getBranchDestBlock(*LastInst))) {
591 LastInst->eraseFromParent();
592 LastInst = SecondLastInst;
593 LastOpc = LastInst->getOpcode();
594 if (I == MBB.begin() || !isUnpredicatedTerminator(*--I)) {
595 assert(!isUncondBranchOpcode(LastOpc) &&
596 "unreachable unconditional branches removed above");
597
598 if (isCondBranchOpcode(LastOpc)) {
599 // Block ends with fall-through condbranch.
600 parseCondBranch(LastInst, TBB, Cond);
601 return false;
602 }
603 return true; // Can't handle indirect branch.
604 }
605 SecondLastInst = &*I;
606 SecondLastOpc = SecondLastInst->getOpcode();
607 }
608
609 // If there are three terminators, we don't know what sort of block this is.
610 if (SecondLastInst && I != MBB.begin() && isUnpredicatedTerminator(*--I))
611 return true;
612
613 // If the block ends with a B and a Bcc, handle it.
614 if (isCondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
615 parseCondBranch(SecondLastInst, TBB, Cond);
616 FBB = LastInst->getOperand(0).getMBB();
617 return false;
618 }
619
620 // If the block ends with two unconditional branches, handle it. The second
621 // one is not executed, so remove it.
622 if (isUncondBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
623 TBB = SecondLastInst->getOperand(0).getMBB();
624 I = LastInst;
625 if (AllowModify)
626 I->eraseFromParent();
627 return false;
628 }
629
630 // ...likewise if it ends with an indirect branch followed by an unconditional
631 // branch.
632 if (isIndirectBranchOpcode(SecondLastOpc) && isUncondBranchOpcode(LastOpc)) {
633 I = LastInst;
634 if (AllowModify)
635 I->eraseFromParent();
636 return true;
637 }
638
639 // Otherwise, can't handle this.
640 return true;
641}
642
644 MachineBranchPredicate &MBP,
645 bool AllowModify) const {
646 // Use analyzeBranch to validate the branch pattern.
647 MachineBasicBlock *TBB = nullptr, *FBB = nullptr;
649 if (analyzeBranch(MBB, TBB, FBB, Cond, AllowModify))
650 return true;
651
652 // analyzeBranch returns success with empty Cond for unconditional branches.
653 if (Cond.empty())
654 return true;
655
656 MBP.TrueDest = TBB;
657 assert(MBP.TrueDest && "expected!");
658 MBP.FalseDest = FBB ? FBB : MBB.getNextNode();
659
660 MBP.ConditionDef = nullptr;
661 MBP.SingleUseCondition = false;
662
663 // Find the conditional branch. After analyzeBranch succeeds with non-empty
664 // Cond, there's exactly one conditional branch - either last (fallthrough)
665 // or second-to-last (followed by unconditional B).
666 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
667 if (I == MBB.end())
668 return true;
669
670 if (isUncondBranchOpcode(I->getOpcode())) {
671 if (I == MBB.begin())
672 return true;
673 --I;
674 }
675
676 MachineInstr *CondBranch = &*I;
677 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
678
679 switch (CondBranch->getOpcode()) {
680 default:
681 return true;
682
683 case AArch64::Bcc:
684 // Bcc takes the NZCV flag as the operand to branch on, walk up the
685 // instruction stream to find the last instruction to define NZCV.
687 if (MI.modifiesRegister(AArch64::NZCV, /*TRI=*/nullptr)) {
688 MBP.ConditionDef = &MI;
689 break;
690 }
691 }
692 return false;
693
694 case AArch64::CBZW:
695 case AArch64::CBZX:
696 case AArch64::CBNZW:
697 case AArch64::CBNZX: {
698 MBP.LHS = CondBranch->getOperand(0);
699 MBP.RHS = MachineOperand::CreateImm(0);
700 unsigned Opc = CondBranch->getOpcode();
701 MBP.Predicate = (Opc == AArch64::CBNZX || Opc == AArch64::CBNZW)
702 ? MachineBranchPredicate::PRED_NE
703 : MachineBranchPredicate::PRED_EQ;
704 Register CondReg = MBP.LHS.getReg();
705 if (CondReg.isVirtual())
706 MBP.ConditionDef = MRI.getVRegDef(CondReg);
707 return false;
708 }
709
710 case AArch64::TBZW:
711 case AArch64::TBZX:
712 case AArch64::TBNZW:
713 case AArch64::TBNZX: {
714 Register CondReg = CondBranch->getOperand(0).getReg();
715 if (CondReg.isVirtual())
716 MBP.ConditionDef = MRI.getVRegDef(CondReg);
717 return false;
718 }
719 }
720}
721
724 if (Cond[0].getImm() != -1) {
725 // Regular Bcc
726 AArch64CC::CondCode CC = (AArch64CC::CondCode)(int)Cond[0].getImm();
728 } else {
729 // Folded compare-and-branch
730 switch (Cond[1].getImm()) {
731 default:
732 llvm_unreachable("Unknown conditional branch!");
733 case AArch64::CBZW:
734 Cond[1].setImm(AArch64::CBNZW);
735 break;
736 case AArch64::CBNZW:
737 Cond[1].setImm(AArch64::CBZW);
738 break;
739 case AArch64::CBZX:
740 Cond[1].setImm(AArch64::CBNZX);
741 break;
742 case AArch64::CBNZX:
743 Cond[1].setImm(AArch64::CBZX);
744 break;
745 case AArch64::TBZW:
746 Cond[1].setImm(AArch64::TBNZW);
747 break;
748 case AArch64::TBNZW:
749 Cond[1].setImm(AArch64::TBZW);
750 break;
751 case AArch64::TBZX:
752 Cond[1].setImm(AArch64::TBNZX);
753 break;
754 case AArch64::TBNZX:
755 Cond[1].setImm(AArch64::TBZX);
756 break;
757
758 // Cond is { -1, Opcode, CC, Op0, Op1, ... }
759 case AArch64::CBWPri:
760 case AArch64::CBXPri:
761 case AArch64::CBBAssertExt:
762 case AArch64::CBHAssertExt:
763 case AArch64::CBWPrr:
764 case AArch64::CBXPrr: {
765 // Pseudos using standard 4bit Arm condition codes
767 static_cast<AArch64CC::CondCode>(Cond[2].getImm());
769 }
770 }
771 }
772
773 return false;
774}
775
777 int *BytesRemoved) const {
778 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
779 if (I == MBB.end())
780 return 0;
781
782 if (!isUncondBranchOpcode(I->getOpcode()) &&
783 !isCondBranchOpcode(I->getOpcode()))
784 return 0;
785
786 // Remove the branch.
787 I->eraseFromParent();
788
789 I = MBB.end();
790
791 if (I == MBB.begin()) {
792 if (BytesRemoved)
793 *BytesRemoved = 4;
794 return 1;
795 }
796 --I;
797 if (!isCondBranchOpcode(I->getOpcode())) {
798 if (BytesRemoved)
799 *BytesRemoved = 4;
800 return 1;
801 }
802
803 // Remove the branch.
804 I->eraseFromParent();
805 if (BytesRemoved)
806 *BytesRemoved = 8;
807
808 return 2;
809}
810
811void AArch64InstrInfo::instantiateCondBranch(
814 if (Cond[0].getImm() != -1) {
815 // Regular Bcc
816 BuildMI(&MBB, DL, get(AArch64::Bcc)).addImm(Cond[0].getImm()).addMBB(TBB);
817 } else {
818 // Folded compare-and-branch
819 // Note that we use addOperand instead of addReg to keep the flags.
820
821 // cbz, cbnz
822 const MachineInstrBuilder MIB =
823 BuildMI(&MBB, DL, get(Cond[1].getImm())).add(Cond[2]);
824
825 // tbz/tbnz
826 if (Cond.size() > 3)
827 MIB.add(Cond[3]);
828
829 // cb
830 if (Cond.size() > 4)
831 MIB.add(Cond[4]);
832
833 MIB.addMBB(TBB);
834
835 // cb[b,h]
836 if (Cond.size() > 5) {
837 MIB.addImm(Cond[5].getImm());
838 MIB.addImm(Cond[6].getImm());
839 }
840 }
841}
842
845 ArrayRef<MachineOperand> Cond, const DebugLoc &DL, int *BytesAdded) const {
846 // Shouldn't be a fall through.
847 assert(TBB && "insertBranch must not be told to insert a fallthrough");
848
849 if (!FBB) {
850 if (Cond.empty()) // Unconditional branch?
851 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(TBB);
852 else
853 instantiateCondBranch(MBB, DL, TBB, Cond);
854
855 if (BytesAdded)
856 *BytesAdded = 4;
857
858 return 1;
859 }
860
861 // Two-way conditional branch.
862 instantiateCondBranch(MBB, DL, TBB, Cond);
863 BuildMI(&MBB, DL, get(AArch64::B)).addMBB(FBB);
864
865 if (BytesAdded)
866 *BytesAdded = 8;
867
868 return 2;
869}
870
871#ifndef NDEBUG
873 switch (Ext) {
874 default:
875 return false;
876 case AArch64_AM::UXTB:
877 case AArch64_AM::SXTB:
878 return Opc == AArch64::CBBAssertExt;
879 case AArch64_AM::UXTH:
880 case AArch64_AM::SXTH:
881 return Opc == AArch64::CBHAssertExt;
882 }
883}
884#endif
885
889 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
890
891 // Parse the condition code, see parseCondBranch() above.
893 switch (Cond.size()) {
894 default:
895 llvm_unreachable("Unknown condition opcode in Cond");
896 case 1: // b.cc
898 break;
899 case 3: { // cbz/cbnz
900 // We must insert a compare against 0.
901 bool Is64Bit;
902 switch (Cond[1].getImm()) {
903 default:
904 llvm_unreachable("Unknown branch opcode in Cond");
905 case AArch64::CBZW:
906 Is64Bit = false;
907 CC = AArch64CC::EQ;
908 break;
909 case AArch64::CBZX:
910 Is64Bit = true;
911 CC = AArch64CC::EQ;
912 break;
913 case AArch64::CBNZW:
914 Is64Bit = false;
915 CC = AArch64CC::NE;
916 break;
917 case AArch64::CBNZX:
918 Is64Bit = true;
919 CC = AArch64CC::NE;
920 break;
921 }
922 Register SrcReg = Cond[2].getReg();
923 if (Is64Bit) {
924 // cmp reg, #0 is actually subs xzr, reg, #0.
925 MRI.constrainRegClass(SrcReg, &AArch64::GPR64spRegClass);
926 BuildMI(MBB, MI, DL, get(AArch64::SUBSXri), AArch64::XZR)
927 .addReg(SrcReg)
928 .addImm(0)
929 .addImm(0);
930 } else {
931 MRI.constrainRegClass(SrcReg, &AArch64::GPR32spRegClass);
932 BuildMI(MBB, MI, DL, get(AArch64::SUBSWri), AArch64::WZR)
933 .addReg(SrcReg)
934 .addImm(0)
935 .addImm(0);
936 }
937 } break;
938 case 4: { // tbz/tbnz
939 // We must insert a tst instruction.
940 switch (Cond[1].getImm()) {
941 default:
942 llvm_unreachable("Unknown branch opcode in Cond");
943 case AArch64::TBZW:
944 case AArch64::TBZX:
945 CC = AArch64CC::EQ;
946 break;
947 case AArch64::TBNZW:
948 case AArch64::TBNZX:
949 CC = AArch64CC::NE;
950 break;
951 }
952 // cmp reg, #foo is actually ands xzr, reg, #1<<foo.
953 if (Cond[1].getImm() == AArch64::TBZW || Cond[1].getImm() == AArch64::TBNZW)
954 BuildMI(MBB, MI, DL, get(AArch64::ANDSWri), AArch64::WZR)
955 .addReg(Cond[2].getReg())
956 .addImm(
958 else
959 BuildMI(MBB, MI, DL, get(AArch64::ANDSXri), AArch64::XZR)
960 .addReg(Cond[2].getReg())
961 .addImm(
963 } break;
964 case 5: { // cb
965 // We must insert a cmp, that is a subs
966 // 0 1 2 3 4
967 // Cond is { -1, Opcode, CC, Op0, Op1 }
968 unsigned SubsOpc, SubsDestReg;
969 bool IsImm = false;
970 CC = static_cast<AArch64CC::CondCode>(Cond[2].getImm());
971 switch (Cond[1].getImm()) {
972 default:
973 llvm_unreachable("Unknown branch opcode in Cond");
974 case AArch64::CBWPri:
975 SubsOpc = AArch64::SUBSWri;
976 SubsDestReg = AArch64::WZR;
977 IsImm = true;
978 break;
979 case AArch64::CBXPri:
980 SubsOpc = AArch64::SUBSXri;
981 SubsDestReg = AArch64::XZR;
982 IsImm = true;
983 break;
984 case AArch64::CBWPrr:
985 SubsOpc = AArch64::SUBSWrr;
986 SubsDestReg = AArch64::WZR;
987 IsImm = false;
988 break;
989 case AArch64::CBXPrr:
990 SubsOpc = AArch64::SUBSXrr;
991 SubsDestReg = AArch64::XZR;
992 IsImm = false;
993 break;
994 }
995
996 if (IsImm) {
997 MRI.constrainRegClass(Cond[3].getReg(), getRegClass(get(SubsOpc), 1));
998 BuildMI(MBB, MI, DL, get(SubsOpc), SubsDestReg)
999 .addReg(Cond[3].getReg())
1000 .addImm(Cond[4].getImm())
1001 .addImm(0);
1002 } else {
1003 MRI.constrainRegClass(Cond[3].getReg(), getRegClass(get(SubsOpc), 1));
1004 MRI.constrainRegClass(Cond[4].getReg(), getRegClass(get(SubsOpc), 2));
1005 BuildMI(MBB, MI, DL, get(SubsOpc), SubsDestReg)
1006 .addReg(Cond[3].getReg())
1007 .addReg(Cond[4].getReg());
1008 }
1009 } break;
1010 case 7: { // cb[b,h]
1011 // We must insert a cmp, that is a subs, but also zero- or sign-extensions
1012 // that have been folded. For the first operand we codegen an explicit
1013 // extension, for the second operand we fold the extension into cmp.
1014 // 0 1 2 3 4 5 6
1015 // Cond is { -1, Opcode, CC, Op0, Op1, Ext0, Ext1 }
1016
1017 // We need a new register for the now explicitly extended register
1018 Register Reg = Cond[3].getReg();
1020 unsigned ExtOpc;
1021 unsigned ExtBits;
1022 AArch64_AM::ShiftExtendType ExtendType =
1024 assert(isValidCBExtend(Cond[1].getImm(), ExtendType) &&
1025 "Unexpected compare-and-branch instruction for extend type");
1026 switch (ExtendType) {
1027 default:
1028 llvm_unreachable("Unknown shift-extend for CB instruction");
1029 case AArch64_AM::SXTB:
1030 ExtOpc = AArch64::SBFMWri;
1031 ExtBits = AArch64_AM::encodeLogicalImmediate(0xff, 32);
1032 break;
1033 case AArch64_AM::SXTH:
1034 ExtOpc = AArch64::SBFMWri;
1035 ExtBits = AArch64_AM::encodeLogicalImmediate(0xffff, 32);
1036 break;
1037 case AArch64_AM::UXTB:
1038 ExtOpc = AArch64::ANDWri;
1039 ExtBits = AArch64_AM::encodeLogicalImmediate(0xff, 32);
1040 break;
1041 case AArch64_AM::UXTH:
1042 ExtOpc = AArch64::ANDWri;
1043 ExtBits = AArch64_AM::encodeLogicalImmediate(0xffff, 32);
1044 break;
1045 }
1046
1047 // Build the explicit extension of the first operand
1048 Reg = MRI.createVirtualRegister(&AArch64::GPR32commonRegClass);
1050 BuildMI(MBB, MI, DL, get(ExtOpc), Reg).addReg(Cond[3].getReg());
1051 if (ExtOpc != AArch64::ANDWri)
1052 MBBI.addImm(0);
1053 MBBI.addImm(ExtBits);
1054 }
1055
1056 // Now, subs with an extended second operand
1058 MRI.constrainRegClass(Reg, &AArch64::GPR32commonRegClass);
1059 AArch64_AM::ShiftExtendType ExtendType =
1061 assert(isValidCBExtend(Cond[1].getImm(), ExtendType) &&
1062 "Unexpected compare-and-branch instruction for extend type");
1063 BuildMI(MBB, MI, DL, get(AArch64::SUBSWrx), AArch64::WZR)
1064 .addReg(Reg)
1065 .addReg(Cond[4].getReg())
1066 .addImm(AArch64_AM::getArithExtendImm(ExtendType, 0));
1067 } // If no extension is needed, just a regular subs
1068 else {
1069 BuildMI(MBB, MI, DL, get(AArch64::SUBSWrr), AArch64::WZR)
1070 .addReg(Reg)
1071 .addReg(Cond[4].getReg());
1072 }
1073
1074 CC = static_cast<AArch64CC::CondCode>(Cond[2].getImm());
1075 } break;
1076 }
1077 return CC;
1078}
1079
1081 const TargetInstrInfo &TII) {
1082 for (MachineInstr &MI : MBB->terminators()) {
1083 unsigned Opc = MI.getOpcode();
1084 switch (Opc) {
1085 case AArch64::CBZW:
1086 case AArch64::CBZX:
1087 case AArch64::TBZW:
1088 case AArch64::TBZX:
1089 // CBZ/TBZ with WZR/XZR -> unconditional B
1090 if (MI.getOperand(0).getReg() == AArch64::WZR ||
1091 MI.getOperand(0).getReg() == AArch64::XZR) {
1092 DEBUG_WITH_TYPE("optimizeTerminators",
1093 dbgs() << "Removing always taken branch: " << MI);
1094 MachineBasicBlock *Target = TII.getBranchDestBlock(MI);
1095 SmallVector<MachineBasicBlock *> Succs(MBB->successors());
1096 for (auto *S : Succs)
1097 if (S != Target)
1098 MBB->removeSuccessor(S);
1099 DebugLoc DL = MI.getDebugLoc();
1100 while (MBB->rbegin() != &MI)
1101 MBB->rbegin()->eraseFromParent();
1102 MI.eraseFromParent();
1103 BuildMI(MBB, DL, TII.get(AArch64::B)).addMBB(Target);
1104 return true;
1105 }
1106 break;
1107 case AArch64::CBNZW:
1108 case AArch64::CBNZX:
1109 case AArch64::TBNZW:
1110 case AArch64::TBNZX:
1111 // CBNZ/TBNZ with WZR/XZR -> never taken, remove branch and successor
1112 if (MI.getOperand(0).getReg() == AArch64::WZR ||
1113 MI.getOperand(0).getReg() == AArch64::XZR) {
1114 DEBUG_WITH_TYPE("optimizeTerminators",
1115 dbgs() << "Removing never taken branch: " << MI);
1116 MachineBasicBlock *Target = TII.getBranchDestBlock(MI);
1117 MI.getParent()->removeSuccessor(Target);
1118 MI.eraseFromParent();
1119 return true;
1120 }
1121 break;
1122 }
1123 }
1124 return false;
1125}
1126
1127// Find the original register that VReg is copied from.
1128static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg) {
1129 while (Register::isVirtualRegister(VReg)) {
1130 const MachineInstr *DefMI = MRI.getVRegDef(VReg);
1131 if (!DefMI || !DefMI->isFullCopy())
1132 return VReg;
1133 VReg = DefMI->getOperand(1).getReg();
1134 }
1135 return VReg;
1136}
1137
1138// Determine if VReg is defined by an instruction that can be folded into a
1139// csel instruction. If so, return the folded opcode, and the replacement
1140// register.
1141static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg,
1142 unsigned *NewReg = nullptr) {
1143 VReg = removeCopies(MRI, VReg);
1144 if (!Register::isVirtualRegister(VReg))
1145 return 0;
1146
1147 bool Is64Bit = AArch64::GPR64allRegClass.hasSubClassEq(MRI.getRegClass(VReg));
1148 const MachineInstr *DefMI = MRI.getVRegDef(VReg);
1149 if (!DefMI)
1150 return 0;
1151 unsigned Opc = 0;
1152 unsigned SrcReg = 0;
1153 switch (DefMI->getOpcode()) {
1154 case AArch64::SUBREG_TO_REG:
1155 // Check for the following way to define an 64-bit immediate:
1156 // %0:gpr32 = MOVi32imm 1
1157 // %1:gpr64 = SUBREG_TO_REG %0:gpr32, %subreg.sub_32
1158 if (!DefMI->getOperand(1).isReg())
1159 return 0;
1160 if (!DefMI->getOperand(2).isImm() ||
1161 DefMI->getOperand(2).getImm() != AArch64::sub_32)
1162 return 0;
1163 DefMI = MRI.getVRegDef(DefMI->getOperand(1).getReg());
1164 if (DefMI->getOpcode() != AArch64::MOVi32imm)
1165 return 0;
1166 if (!DefMI->getOperand(1).isImm() || DefMI->getOperand(1).getImm() != 1)
1167 return 0;
1168 assert(Is64Bit);
1169 SrcReg = AArch64::XZR;
1170 Opc = AArch64::CSINCXr;
1171 break;
1172
1173 case AArch64::MOVi32imm:
1174 case AArch64::MOVi64imm:
1175 if (!DefMI->getOperand(1).isImm() || DefMI->getOperand(1).getImm() != 1)
1176 return 0;
1177 SrcReg = Is64Bit ? AArch64::XZR : AArch64::WZR;
1178 Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr;
1179 break;
1180
1181 case AArch64::ADDSXri:
1182 case AArch64::ADDSWri:
1183 // if NZCV is used, do not fold.
1184 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr,
1185 true) == -1)
1186 return 0;
1187 // fall-through to ADDXri and ADDWri.
1188 [[fallthrough]];
1189 case AArch64::ADDXri:
1190 case AArch64::ADDWri:
1191 // add x, 1 -> csinc.
1192 if (!DefMI->getOperand(2).isImm() || DefMI->getOperand(2).getImm() != 1 ||
1193 DefMI->getOperand(3).getImm() != 0)
1194 return 0;
1195 SrcReg = DefMI->getOperand(1).getReg();
1196 Opc = Is64Bit ? AArch64::CSINCXr : AArch64::CSINCWr;
1197 break;
1198
1199 case AArch64::ORNXrr:
1200 case AArch64::ORNWrr: {
1201 // not x -> csinv, represented as orn dst, xzr, src.
1202 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
1203 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
1204 return 0;
1205 SrcReg = DefMI->getOperand(2).getReg();
1206 Opc = Is64Bit ? AArch64::CSINVXr : AArch64::CSINVWr;
1207 break;
1208 }
1209
1210 case AArch64::SUBSXrr:
1211 case AArch64::SUBSWrr:
1212 // if NZCV is used, do not fold.
1213 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr,
1214 true) == -1)
1215 return 0;
1216 // fall-through to SUBXrr and SUBWrr.
1217 [[fallthrough]];
1218 case AArch64::SUBXrr:
1219 case AArch64::SUBWrr: {
1220 // neg x -> csneg, represented as sub dst, xzr, src.
1221 unsigned ZReg = removeCopies(MRI, DefMI->getOperand(1).getReg());
1222 if (ZReg != AArch64::XZR && ZReg != AArch64::WZR)
1223 return 0;
1224 SrcReg = DefMI->getOperand(2).getReg();
1225 Opc = Is64Bit ? AArch64::CSNEGXr : AArch64::CSNEGWr;
1226 break;
1227 }
1228 default:
1229 return 0;
1230 }
1231 assert(Opc && SrcReg && "Missing parameters");
1232
1233 if (NewReg)
1234 *NewReg = SrcReg;
1235 return Opc;
1236}
1237
1240 Register DstReg, Register TrueReg,
1241 Register FalseReg, int &CondCycles,
1242 int &TrueCycles,
1243 int &FalseCycles) const {
1244 // Check register classes.
1245 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
1246 const TargetRegisterClass *RC =
1247 RI.getCommonSubClass(MRI.getRegClass(TrueReg), MRI.getRegClass(FalseReg));
1248 if (!RC)
1249 return false;
1250
1251 // Also need to check the dest regclass, in case we're trying to optimize
1252 // something like:
1253 // %1(gpr) = PHI %2(fpr), bb1, %(fpr), bb2
1254 if (!RI.getCommonSubClass(RC, MRI.getRegClass(DstReg)))
1255 return false;
1256
1257 // Expanding cbz/tbz requires an extra cycle of latency on the condition.
1258 unsigned ExtraCondLat = Cond.size() != 1;
1259
1260 // GPRs are handled by csel.
1261 // FIXME: Fold in x+1, -x, and ~x when applicable.
1262 if (AArch64::GPR64allRegClass.hasSubClassEq(RC) ||
1263 AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
1264 // Single-cycle csel, csinc, csinv, and csneg.
1265 CondCycles = 1 + ExtraCondLat;
1266 TrueCycles = FalseCycles = 1;
1267 if (canFoldIntoCSel(MRI, TrueReg))
1268 TrueCycles = 0;
1269 else if (canFoldIntoCSel(MRI, FalseReg))
1270 FalseCycles = 0;
1271 return true;
1272 }
1273
1274 // Scalar floating point is handled by fcsel.
1275 // FIXME: Form fabs, fmin, and fmax when applicable.
1276 if (AArch64::FPR64RegClass.hasSubClassEq(RC) ||
1277 AArch64::FPR32RegClass.hasSubClassEq(RC)) {
1278 CondCycles = 5 + ExtraCondLat;
1279 TrueCycles = FalseCycles = 2;
1280 return true;
1281 }
1282
1283 // No single conditional move for a 128-bit vector, but we can emit a sequence
1284 // of csetm (~1), dup (~5, cross domain), bsl (~2).
1285 if (AArch64::FPR128RegClass.hasSubClassEq(RC) &&
1286 Subtarget.isNeonAvailable() &&
1287 !MBB.getParent()->getFunction().hasMinSize()) {
1288 CondCycles = 8 + ExtraCondLat;
1289 TrueCycles = FalseCycles = 2;
1290 return true;
1291 }
1292
1293 return false;
1294}
1295
1298 const DebugLoc &DL, Register DstReg,
1300 Register TrueReg, Register FalseReg) const {
1301
1302 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
1304
1305 // A 128-bit vector has no conditional move so blend the operands with a mask
1306 // built from the flags.
1307 if (MRI.constrainRegClass(DstReg, &AArch64::FPR128RegClass)) {
1308 assert(Subtarget.isNeonAvailable() && "Expected NEON for a vector select");
1309 MRI.constrainRegClass(TrueReg, &AArch64::FPR128RegClass);
1310 MRI.constrainRegClass(FalseReg, &AArch64::FPR128RegClass);
1311 Register CondSet = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
1312 BuildMI(MBB, I, DL, get(AArch64::CSINVXr), CondSet)
1313 .addReg(AArch64::XZR)
1314 .addReg(AArch64::XZR)
1316 Register Mask = MRI.createVirtualRegister(&AArch64::FPR128RegClass);
1317 BuildMI(MBB, I, DL, get(AArch64::DUPv2i64gpr), Mask).addReg(CondSet);
1318 BuildMI(MBB, I, DL, get(AArch64::BSPv16i8), DstReg)
1319 .addReg(Mask)
1320 .addReg(TrueReg)
1321 .addReg(FalseReg);
1322 return;
1323 }
1324
1325 unsigned Opc = 0;
1326 const TargetRegisterClass *RC = nullptr;
1327 bool TryFold = false;
1328 if (MRI.constrainRegClass(DstReg, &AArch64::GPR64RegClass)) {
1329 RC = &AArch64::GPR64RegClass;
1330 Opc = AArch64::CSELXr;
1331 TryFold = true;
1332 } else if (MRI.constrainRegClass(DstReg, &AArch64::GPR32RegClass)) {
1333 RC = &AArch64::GPR32RegClass;
1334 Opc = AArch64::CSELWr;
1335 TryFold = true;
1336 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR64RegClass)) {
1337 RC = &AArch64::FPR64RegClass;
1338 Opc = AArch64::FCSELDrrr;
1339 } else if (MRI.constrainRegClass(DstReg, &AArch64::FPR32RegClass)) {
1340 RC = &AArch64::FPR32RegClass;
1341 Opc = AArch64::FCSELSrrr;
1342 }
1343 assert(RC && "Unsupported regclass");
1344
1345 // Try folding simple instructions into the csel.
1346 if (TryFold) {
1347 unsigned NewReg = 0;
1348 unsigned FoldedOpc = canFoldIntoCSel(MRI, TrueReg, &NewReg);
1349 if (FoldedOpc) {
1350 // The folded opcodes csinc, csinc and csneg apply the operation to
1351 // FalseReg, so we need to invert the condition.
1353 TrueReg = FalseReg;
1354 } else
1355 FoldedOpc = canFoldIntoCSel(MRI, FalseReg, &NewReg);
1356
1357 // Fold the operation. Leave any dead instructions for DCE to clean up.
1358 if (FoldedOpc) {
1359 FalseReg = NewReg;
1360 Opc = FoldedOpc;
1361 // Extend the live range of NewReg.
1362 MRI.clearKillFlags(NewReg);
1363 }
1364 }
1365
1366 // Pull all virtual register into the appropriate class.
1367 MRI.constrainRegClass(TrueReg, RC);
1368 // FalseReg might be WZR or XZR if the folded operand is a literal 1.
1369 assert(
1370 (FalseReg.isVirtual() || FalseReg == AArch64::WZR ||
1371 FalseReg == AArch64::XZR) &&
1372 "FalseReg was folded into a non-virtual register other than WZR or XZR");
1373 if (FalseReg.isVirtual())
1374 MRI.constrainRegClass(FalseReg, RC);
1375
1376 // Insert the csel.
1377 BuildMI(MBB, I, DL, get(Opc), DstReg)
1378 .addReg(TrueReg)
1379 .addReg(FalseReg)
1380 .addImm(CC);
1381}
1382
1383// Return true if Imm can be loaded into a register by a "cheap" sequence of
1384// instructions. For now, "cheap" means at most two instructions.
1385static bool isCheapImmediate(const MachineInstr &MI, unsigned BitSize) {
1386 if (BitSize == 32)
1387 return true;
1388
1389 assert(BitSize == 64 && "Only bit sizes of 32 or 64 allowed");
1390 uint64_t Imm = static_cast<uint64_t>(MI.getOperand(1).getImm());
1392 AArch64_IMM::expandMOVImm(Imm, BitSize, Is);
1393
1394 return Is.size() <= 2;
1395}
1396
1397// Check if a COPY instruction is cheap.
1398static bool isCheapCopy(const MachineInstr &MI, const AArch64RegisterInfo &RI) {
1399 assert(MI.isCopy() && "Expected COPY instruction");
1400 const MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
1401
1402 // Cross-bank copies (e.g., between GPR and FPR) are expensive on AArch64,
1403 // typically requiring an FMOV instruction with a 2-6 cycle latency.
1404 auto GetRegClass = [&](Register Reg) -> const TargetRegisterClass * {
1405 if (Reg.isVirtual())
1406 return MRI.getRegClass(Reg);
1407 if (Reg.isPhysical())
1408 return RI.getMinimalPhysRegClass(Reg);
1409 return nullptr;
1410 };
1411 const TargetRegisterClass *DstRC = GetRegClass(MI.getOperand(0).getReg());
1412 const TargetRegisterClass *SrcRC = GetRegClass(MI.getOperand(1).getReg());
1413 if (DstRC && SrcRC && !RI.getCommonSubClass(DstRC, SrcRC))
1414 return false;
1415
1416 return MI.isAsCheapAsAMove();
1417}
1418
1419// FIXME: this implementation should be micro-architecture dependent, so a
1420// micro-architecture target hook should be introduced here in future.
1422 if (Subtarget.hasExynosCheapAsMoveHandling()) {
1423 if (isExynosCheapAsMove(MI))
1424 return true;
1425 return MI.isAsCheapAsAMove();
1426 }
1427
1428 switch (MI.getOpcode()) {
1429 default:
1430 return MI.isAsCheapAsAMove();
1431
1432 case TargetOpcode::COPY:
1433 return isCheapCopy(MI, RI);
1434
1435 case AArch64::ADDWrs:
1436 case AArch64::ADDXrs:
1437 case AArch64::SUBWrs:
1438 case AArch64::SUBXrs:
1439 return Subtarget.hasALULSLFast() && MI.getOperand(3).getImm() <= 4;
1440
1441 // If MOVi32imm or MOVi64imm can be expanded into ORRWri or
1442 // ORRXri, it is as cheap as MOV.
1443 // Likewise if it can be expanded to MOVZ/MOVN/MOVK.
1444 case AArch64::MOVi32imm:
1445 return isCheapImmediate(MI, 32);
1446 case AArch64::MOVi64imm:
1447 return isCheapImmediate(MI, 64);
1448 }
1449}
1450
1451bool AArch64InstrInfo::isFalkorShiftExtFast(const MachineInstr &MI) {
1452 switch (MI.getOpcode()) {
1453 default:
1454 return false;
1455
1456 case AArch64::ADDWrs:
1457 case AArch64::ADDXrs:
1458 case AArch64::ADDSWrs:
1459 case AArch64::ADDSXrs: {
1460 unsigned Imm = MI.getOperand(3).getImm();
1461 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
1462 if (ShiftVal == 0)
1463 return true;
1464 return AArch64_AM::getShiftType(Imm) == AArch64_AM::LSL && ShiftVal <= 5;
1465 }
1466
1467 case AArch64::ADDWrx:
1468 case AArch64::ADDXrx:
1469 case AArch64::ADDXrx64:
1470 case AArch64::ADDSWrx:
1471 case AArch64::ADDSXrx:
1472 case AArch64::ADDSXrx64: {
1473 unsigned Imm = MI.getOperand(3).getImm();
1475 default:
1476 return false;
1477 case AArch64_AM::UXTB:
1478 case AArch64_AM::UXTH:
1479 case AArch64_AM::UXTW:
1480 case AArch64_AM::UXTX:
1482 }
1483 }
1484
1485 case AArch64::SUBWrs:
1486 case AArch64::SUBSWrs: {
1487 unsigned Imm = MI.getOperand(3).getImm();
1488 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
1489 return ShiftVal == 0 ||
1490 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 31);
1491 }
1492
1493 case AArch64::SUBXrs:
1494 case AArch64::SUBSXrs: {
1495 unsigned Imm = MI.getOperand(3).getImm();
1496 unsigned ShiftVal = AArch64_AM::getShiftValue(Imm);
1497 return ShiftVal == 0 ||
1498 (AArch64_AM::getShiftType(Imm) == AArch64_AM::ASR && ShiftVal == 63);
1499 }
1500
1501 case AArch64::SUBWrx:
1502 case AArch64::SUBXrx:
1503 case AArch64::SUBXrx64:
1504 case AArch64::SUBSWrx:
1505 case AArch64::SUBSXrx:
1506 case AArch64::SUBSXrx64: {
1507 unsigned Imm = MI.getOperand(3).getImm();
1509 default:
1510 return false;
1511 case AArch64_AM::UXTB:
1512 case AArch64_AM::UXTH:
1513 case AArch64_AM::UXTW:
1514 case AArch64_AM::UXTX:
1516 }
1517 }
1518
1519 case AArch64::LDRBBroW:
1520 case AArch64::LDRBBroX:
1521 case AArch64::LDRBroW:
1522 case AArch64::LDRBroX:
1523 case AArch64::LDRDroW:
1524 case AArch64::LDRDroX:
1525 case AArch64::LDRHHroW:
1526 case AArch64::LDRHHroX:
1527 case AArch64::LDRHroW:
1528 case AArch64::LDRHroX:
1529 case AArch64::LDRQroW:
1530 case AArch64::LDRQroX:
1531 case AArch64::LDRSBWroW:
1532 case AArch64::LDRSBWroX:
1533 case AArch64::LDRSBXroW:
1534 case AArch64::LDRSBXroX:
1535 case AArch64::LDRSHWroW:
1536 case AArch64::LDRSHWroX:
1537 case AArch64::LDRSHXroW:
1538 case AArch64::LDRSHXroX:
1539 case AArch64::LDRSWroW:
1540 case AArch64::LDRSWroX:
1541 case AArch64::LDRSroW:
1542 case AArch64::LDRSroX:
1543 case AArch64::LDRWroW:
1544 case AArch64::LDRWroX:
1545 case AArch64::LDRXroW:
1546 case AArch64::LDRXroX:
1547 case AArch64::PRFMroW:
1548 case AArch64::PRFMroX:
1549 case AArch64::STRBBroW:
1550 case AArch64::STRBBroX:
1551 case AArch64::STRBroW:
1552 case AArch64::STRBroX:
1553 case AArch64::STRDroW:
1554 case AArch64::STRDroX:
1555 case AArch64::STRHHroW:
1556 case AArch64::STRHHroX:
1557 case AArch64::STRHroW:
1558 case AArch64::STRHroX:
1559 case AArch64::STRQroW:
1560 case AArch64::STRQroX:
1561 case AArch64::STRSroW:
1562 case AArch64::STRSroX:
1563 case AArch64::STRWroW:
1564 case AArch64::STRWroX:
1565 case AArch64::STRXroW:
1566 case AArch64::STRXroX: {
1567 unsigned IsSigned = MI.getOperand(3).getImm();
1568 return !IsSigned;
1569 }
1570 }
1571}
1572
1573bool AArch64InstrInfo::isSEHInstruction(const MachineInstr &MI) {
1574 unsigned Opc = MI.getOpcode();
1575 switch (Opc) {
1576 default:
1577 return false;
1578 case AArch64::SEH_StackAlloc:
1579 case AArch64::SEH_SaveFPLR:
1580 case AArch64::SEH_SaveFPLR_X:
1581 case AArch64::SEH_SaveReg:
1582 case AArch64::SEH_SaveReg_X:
1583 case AArch64::SEH_SaveRegP:
1584 case AArch64::SEH_SaveRegP_X:
1585 case AArch64::SEH_SaveFReg:
1586 case AArch64::SEH_SaveFReg_X:
1587 case AArch64::SEH_SaveFRegP:
1588 case AArch64::SEH_SaveFRegP_X:
1589 case AArch64::SEH_SetFP:
1590 case AArch64::SEH_AddFP:
1591 case AArch64::SEH_Nop:
1592 case AArch64::SEH_PrologEnd:
1593 case AArch64::SEH_EpilogStart:
1594 case AArch64::SEH_EpilogEnd:
1595 case AArch64::SEH_PACSignLR:
1596 case AArch64::SEH_SaveAnyRegI:
1597 case AArch64::SEH_SaveAnyRegIP:
1598 case AArch64::SEH_SaveAnyRegQP:
1599 case AArch64::SEH_SaveAnyRegQPX:
1600 case AArch64::SEH_AllocZ:
1601 case AArch64::SEH_SaveZReg:
1602 case AArch64::SEH_SavePReg:
1603 return true;
1604 }
1605}
1606
1608 Register &SrcReg, Register &DstReg,
1609 unsigned &SubIdx) const {
1610 switch (MI.getOpcode()) {
1611 default:
1612 return false;
1613 case AArch64::SBFMXri: // aka sxtw
1614 case AArch64::UBFMXri: // aka uxtw
1615 // Check for the 32 -> 64 bit extension case, these instructions can do
1616 // much more.
1617 if (MI.getOperand(2).getImm() != 0 || MI.getOperand(3).getImm() != 31)
1618 return false;
1619 // This is a signed or unsigned 32 -> 64 bit extension.
1620 SrcReg = MI.getOperand(1).getReg();
1621 DstReg = MI.getOperand(0).getReg();
1622 SubIdx = AArch64::sub_32;
1623 return true;
1624 }
1625}
1626
1628 const MachineInstr &MIa, const MachineInstr &MIb) const {
1629 const MachineOperand *BaseOpA = nullptr, *BaseOpB = nullptr;
1630 int64_t OffsetA = 0, OffsetB = 0;
1631 TypeSize WidthA(0, false), WidthB(0, false);
1632 bool OffsetAIsScalable = false, OffsetBIsScalable = false;
1633
1634 assert(MIa.mayLoadOrStore() && "MIa must be a load or store.");
1635 assert(MIb.mayLoadOrStore() && "MIb must be a load or store.");
1636
1639 return false;
1640
1641 // Retrieve the base, offset from the base and width. Width
1642 // is the size of memory that is being loaded/stored (e.g. 1, 2, 4, 8). If
1643 // base are identical, and the offset of a lower memory access +
1644 // the width doesn't overlap the offset of a higher memory access,
1645 // then the memory accesses are different.
1646 // If OffsetAIsScalable and OffsetBIsScalable are both true, they
1647 // are assumed to have the same scale (vscale).
1648 if (getMemOperandWithOffsetWidth(MIa, BaseOpA, OffsetA, OffsetAIsScalable,
1649 WidthA) &&
1650 getMemOperandWithOffsetWidth(MIb, BaseOpB, OffsetB, OffsetBIsScalable,
1651 WidthB)) {
1652 if (BaseOpA->isIdenticalTo(*BaseOpB) &&
1653 OffsetAIsScalable == OffsetBIsScalable) {
1654 int LowOffset = OffsetA < OffsetB ? OffsetA : OffsetB;
1655 int HighOffset = OffsetA < OffsetB ? OffsetB : OffsetA;
1656 TypeSize LowWidth = (LowOffset == OffsetA) ? WidthA : WidthB;
1657 if (LowWidth.isScalable() == OffsetAIsScalable &&
1658 LowOffset + (int)LowWidth.getKnownMinValue() <= HighOffset)
1659 return true;
1660 }
1661 }
1662 return false;
1663}
1664
1666 const MachineBasicBlock *MBB,
1667 const MachineFunction &MF) const {
1669 return true;
1670
1671 // Do not move an instruction that can be recognized as a branch target.
1672 if (hasBTISemantics(MI))
1673 return true;
1674
1675 switch (MI.getOpcode()) {
1676 case AArch64::HINT:
1677 // CSDB hints are scheduling barriers.
1678 if (MI.getOperand(0).getImm() == 0x14)
1679 return true;
1680 break;
1681 case AArch64::DSB:
1682 case AArch64::ISB:
1683 // DSB and ISB also are scheduling barriers.
1684 return true;
1685 case AArch64::MSRpstatesvcrImm1:
1686 // SMSTART and SMSTOP are also scheduling barriers.
1687 return true;
1688 default:;
1689 }
1690 if (isSEHInstruction(MI))
1691 return true;
1692 auto Next = std::next(MI.getIterator());
1693 return Next != MBB->end() && Next->isCFIInstruction();
1694}
1695
1696/// analyzeCompare - For a comparison instruction, return the source registers
1697/// in SrcReg and SrcReg2, and the value it compares against in CmpValue.
1698/// Return true if the comparison instruction can be analyzed.
1700 Register &SrcReg2, int64_t &CmpMask,
1701 int64_t &CmpValue) const {
1702 // The first operand can be a frame index where we'd normally expect a
1703 // register.
1704 // FIXME: Pass subregisters out of analyzeCompare
1705 assert(MI.getNumOperands() >= 2 && "All AArch64 cmps should have 2 operands");
1706 if (!MI.getOperand(1).isReg() || MI.getOperand(1).getSubReg())
1707 return false;
1708
1709 switch (MI.getOpcode()) {
1710 default:
1711 break;
1712 case AArch64::PTEST_PP:
1713 case AArch64::PTEST_PP_ANY:
1714 case AArch64::PTEST_PP_FIRST:
1715 SrcReg = MI.getOperand(0).getReg();
1716 SrcReg2 = MI.getOperand(1).getReg();
1717 if (MI.getOperand(2).getSubReg())
1718 return false;
1719
1720 // Not sure about the mask and value for now...
1721 CmpMask = ~0;
1722 CmpValue = 0;
1723 return true;
1724 case AArch64::SUBSWrr:
1725 case AArch64::SUBSWrs:
1726 case AArch64::SUBSWrx:
1727 case AArch64::SUBSXrr:
1728 case AArch64::SUBSXrs:
1729 case AArch64::SUBSXrx:
1730 case AArch64::ADDSWrr:
1731 case AArch64::ADDSWrs:
1732 case AArch64::ADDSWrx:
1733 case AArch64::ADDSXrr:
1734 case AArch64::ADDSXrs:
1735 case AArch64::ADDSXrx:
1736 // Replace SUBSWrr with SUBWrr if NZCV is not used.
1737 SrcReg = MI.getOperand(1).getReg();
1738 SrcReg2 = MI.getOperand(2).getReg();
1739
1740 // FIXME: Pass subregisters out of analyzeCompare
1741 if (MI.getOperand(2).getSubReg())
1742 return false;
1743
1744 CmpMask = ~0;
1745 CmpValue = 0;
1746 return true;
1747 case AArch64::SUBSWri:
1748 case AArch64::ADDSWri:
1749 case AArch64::SUBSXri:
1750 case AArch64::ADDSXri:
1751 SrcReg = MI.getOperand(1).getReg();
1752 SrcReg2 = 0;
1753 CmpMask = ~0;
1754 CmpValue = MI.getOperand(2).getImm();
1755 return true;
1756 case AArch64::ANDSWri:
1757 case AArch64::ANDSXri:
1758 // ANDS does not use the same encoding scheme as the others xxxS
1759 // instructions.
1760 SrcReg = MI.getOperand(1).getReg();
1761 SrcReg2 = 0;
1762 CmpMask = ~0;
1764 MI.getOperand(2).getImm(),
1765 MI.getOpcode() == AArch64::ANDSWri ? 32 : 64);
1766 return true;
1767 }
1768
1769 return false;
1770}
1771
1773 MachineBasicBlock *MBB = Instr.getParent();
1774 assert(MBB && "Can't get MachineBasicBlock here");
1775 MachineFunction *MF = MBB->getParent();
1776 assert(MF && "Can't get MachineFunction here");
1779 MachineRegisterInfo *MRI = &MF->getRegInfo();
1780
1781 for (unsigned OpIdx = 0, EndIdx = Instr.getNumOperands(); OpIdx < EndIdx;
1782 ++OpIdx) {
1783 MachineOperand &MO = Instr.getOperand(OpIdx);
1784 const TargetRegisterClass *OpRegCstraints =
1785 Instr.getRegClassConstraint(OpIdx, TII, TRI);
1786
1787 // If there's no constraint, there's nothing to do.
1788 if (!OpRegCstraints)
1789 continue;
1790 // If the operand is a frame index, there's nothing to do here.
1791 // A frame index operand will resolve correctly during PEI.
1792 if (MO.isFI())
1793 continue;
1794
1795 assert(MO.isReg() &&
1796 "Operand has register constraints without being a register!");
1797
1798 Register Reg = MO.getReg();
1799 if (Reg.isPhysical()) {
1800 if (!OpRegCstraints->contains(Reg))
1801 return false;
1802 } else if (!OpRegCstraints->hasSubClassEq(MRI->getRegClass(Reg)) &&
1803 !MRI->constrainRegClass(Reg, OpRegCstraints))
1804 return false;
1805 }
1806
1807 return true;
1808}
1809
1810/// Return the opcode that does not set flags when possible - otherwise
1811/// return the original opcode. The caller is responsible to do the actual
1812/// substitution and legality checking.
1814 // Don't convert all compare instructions, because for some the zero register
1815 // encoding becomes the sp register.
1816 bool MIDefinesZeroReg = false;
1817 if (MI.definesRegister(AArch64::WZR, /*TRI=*/nullptr) ||
1818 MI.definesRegister(AArch64::XZR, /*TRI=*/nullptr))
1819 MIDefinesZeroReg = true;
1820
1821 switch (MI.getOpcode()) {
1822 default:
1823 return MI.getOpcode();
1824 case AArch64::ADDSWrr:
1825 return AArch64::ADDWrr;
1826 case AArch64::ADDSWri:
1827 return MIDefinesZeroReg ? AArch64::ADDSWri : AArch64::ADDWri;
1828 case AArch64::ADDSWrs:
1829 return MIDefinesZeroReg ? AArch64::ADDSWrs : AArch64::ADDWrs;
1830 case AArch64::ADDSWrx:
1831 return AArch64::ADDWrx;
1832 case AArch64::ADDSXrr:
1833 return AArch64::ADDXrr;
1834 case AArch64::ADDSXri:
1835 return MIDefinesZeroReg ? AArch64::ADDSXri : AArch64::ADDXri;
1836 case AArch64::ADDSXrs:
1837 return MIDefinesZeroReg ? AArch64::ADDSXrs : AArch64::ADDXrs;
1838 case AArch64::ADDSXrx:
1839 return AArch64::ADDXrx;
1840 case AArch64::SUBSWrr:
1841 return AArch64::SUBWrr;
1842 case AArch64::SUBSWri:
1843 return MIDefinesZeroReg ? AArch64::SUBSWri : AArch64::SUBWri;
1844 case AArch64::SUBSWrs:
1845 return MIDefinesZeroReg ? AArch64::SUBSWrs : AArch64::SUBWrs;
1846 case AArch64::SUBSWrx:
1847 return AArch64::SUBWrx;
1848 case AArch64::SUBSXrr:
1849 return AArch64::SUBXrr;
1850 case AArch64::SUBSXri:
1851 return MIDefinesZeroReg ? AArch64::SUBSXri : AArch64::SUBXri;
1852 case AArch64::SUBSXrs:
1853 return MIDefinesZeroReg ? AArch64::SUBSXrs : AArch64::SUBXrs;
1854 case AArch64::SUBSXrx:
1855 return AArch64::SUBXrx;
1856 }
1857}
1858
1859enum AccessKind { AK_Write = 0x01, AK_Read = 0x10, AK_All = 0x11 };
1860
1861/// True when condition flags are accessed (either by writing or reading)
1862/// on the instruction trace starting at From and ending at To.
1863///
1864/// Note: If From and To are from different blocks it's assumed CC are accessed
1865/// on the path.
1868 const TargetRegisterInfo *TRI, const AccessKind AccessToCheck = AK_All) {
1869 // Early exit if To is at the beginning of the BB.
1870 if (To == To->getParent()->begin())
1871 return true;
1872
1873 // Check whether the instructions are in the same basic block
1874 // If not, assume the condition flags might get modified somewhere.
1875 if (To->getParent() != From->getParent())
1876 return true;
1877
1878 // From must be above To.
1879 assert(std::any_of(
1880 ++To.getReverse(), To->getParent()->rend(),
1881 [From](MachineInstr &MI) { return MI.getIterator() == From; }));
1882
1883 // We iterate backward starting at \p To until we hit \p From.
1884 for (const MachineInstr &Instr :
1886 if (((AccessToCheck & AK_Write) &&
1887 Instr.modifiesRegister(AArch64::NZCV, TRI)) ||
1888 ((AccessToCheck & AK_Read) && Instr.readsRegister(AArch64::NZCV, TRI)))
1889 return true;
1890 }
1891 return false;
1892}
1893
1894std::optional<unsigned>
1895AArch64InstrInfo::canRemovePTestInstr(MachineInstr *PTest, MachineInstr *Mask,
1896 MachineInstr *Pred,
1897 const MachineRegisterInfo *MRI) const {
1898 unsigned MaskOpcode = Mask->getOpcode();
1899 unsigned PredOpcode = Pred->getOpcode();
1900 bool PredIsPTestLike = isPTestLikeOpcode(PredOpcode);
1901 bool PredIsWhileLike = isWhileOpcode(PredOpcode);
1902
1903 if (PredIsWhileLike) {
1904 // For PTEST(PG, PG), PTEST is redundant when PG is the result of a WHILEcc
1905 // instruction and the condition is "any" since WHILcc does an implicit
1906 // PTEST(ALL, PG) check and PG is always a subset of ALL.
1907 if ((Mask == Pred) && PTest->getOpcode() == AArch64::PTEST_PP_ANY)
1908 return PredOpcode;
1909
1910 // For PTEST(PTRUE_ALL, WHILE), if the element size matches, the PTEST is
1911 // redundant since WHILE performs an implicit PTEST with an all active
1912 // mask.
1913 if (isPTrueOpcode(MaskOpcode) && Mask->getOperand(1).getImm() == 31 &&
1914 getElementSizeForOpcode(MaskOpcode) ==
1915 getElementSizeForOpcode(PredOpcode))
1916 return PredOpcode;
1917
1918 // For PTEST_FIRST(PTRUE_ALL, WHILE), the PTEST_FIRST is redundant since
1919 // WHILEcc performs an implicit PTEST with an all active mask, setting
1920 // the N flag as the PTEST_FIRST would.
1921 if (PTest->getOpcode() == AArch64::PTEST_PP_FIRST &&
1922 isPTrueOpcode(MaskOpcode) && Mask->getOperand(1).getImm() == 31)
1923 return PredOpcode;
1924
1925 return {};
1926 }
1927
1928 if (PredIsPTestLike) {
1929 // For PTEST(PG, PG), PTEST is redundant when PG is the result of an
1930 // instruction that sets the flags as PTEST would and the condition is
1931 // "any" since PG is always a subset of the governing predicate of the
1932 // ptest-like instruction.
1933 if ((Mask == Pred) && PTest->getOpcode() == AArch64::PTEST_PP_ANY)
1934 return PredOpcode;
1935
1936 auto PTestLikeMask = MRI->getUniqueVRegDef(Pred->getOperand(1).getReg());
1937
1938 // If the PTEST like instruction's general predicate is not `Mask`, attempt
1939 // to look through a copy and try again. This is because some instructions
1940 // take a predicate whose register class is a subset of its result class.
1941 if (Mask != PTestLikeMask && PTestLikeMask->isFullCopy() &&
1942 PTestLikeMask->getOperand(1).getReg().isVirtual())
1943 PTestLikeMask =
1944 MRI->getUniqueVRegDef(PTestLikeMask->getOperand(1).getReg());
1945
1946 // For PTEST(PTRUE_ALL, PTEST_LIKE), the PTEST is redundant if the
1947 // the element size matches and either the PTEST_LIKE instruction uses
1948 // the same all active mask or the condition is "any".
1949 if (isPTrueOpcode(MaskOpcode) && Mask->getOperand(1).getImm() == 31 &&
1950 getElementSizeForOpcode(MaskOpcode) ==
1951 getElementSizeForOpcode(PredOpcode)) {
1952 if (Mask == PTestLikeMask || PTest->getOpcode() == AArch64::PTEST_PP_ANY)
1953 return PredOpcode;
1954 }
1955
1956 // For PTEST(PG, PTEST_LIKE(PG, ...)), the PTEST is redundant since the
1957 // flags are set based on the same mask 'PG', but PTEST_LIKE must operate
1958 // on 8-bit predicates like the PTEST. Otherwise, for instructions like
1959 // compare that also support 16/32/64-bit predicates, the implicit PTEST
1960 // performed by the compare could consider fewer lanes for these element
1961 // sizes.
1962 //
1963 // For example, consider
1964 //
1965 // ptrue p0.b ; P0=1111-1111-1111-1111
1966 // index z0.s, #0, #1 ; Z0=<0,1,2,3>
1967 // index z1.s, #1, #1 ; Z1=<1,2,3,4>
1968 // cmphi p1.s, p0/z, z1.s, z0.s ; P1=0001-0001-0001-0001
1969 // ; ^ last active
1970 // ptest p0, p1.b ; P1=0001-0001-0001-0001
1971 // ; ^ last active
1972 //
1973 // where the compare generates a canonical all active 32-bit predicate
1974 // (equivalent to 'ptrue p1.s, all'). The implicit PTEST sets the last
1975 // active flag, whereas the PTEST instruction with the same mask doesn't.
1976 // For PTEST_ANY this doesn't apply as the flags in this case would be
1977 // identical regardless of element size.
1978 uint64_t PredElementSize = getElementSizeForOpcode(PredOpcode);
1979 if (Mask == PTestLikeMask && (PredElementSize == AArch64::ElementSizeB ||
1980 PTest->getOpcode() == AArch64::PTEST_PP_ANY))
1981 return PredOpcode;
1982
1983 return {};
1984 }
1985
1986 // If OP in PTEST(PG, OP(PG, ...)) has a flag-setting variant change the
1987 // opcode so the PTEST becomes redundant.
1988 switch (PredOpcode) {
1989 case AArch64::AND_PPzPP:
1990 case AArch64::BIC_PPzPP:
1991 case AArch64::EOR_PPzPP:
1992 case AArch64::NAND_PPzPP:
1993 case AArch64::NOR_PPzPP:
1994 case AArch64::ORN_PPzPP:
1995 case AArch64::ORR_PPzPP:
1996 case AArch64::BRKA_PPzP:
1997 case AArch64::BRKPA_PPzPP:
1998 case AArch64::BRKB_PPzP:
1999 case AArch64::BRKPB_PPzPP:
2000 case AArch64::RDFFR_PPz: {
2001 // Check to see if our mask is the same. If not the resulting flag bits
2002 // may be different and we can't remove the ptest.
2003 auto *PredMask = MRI->getUniqueVRegDef(Pred->getOperand(1).getReg());
2004 if (Mask != PredMask)
2005 return {};
2006 break;
2007 }
2008 case AArch64::BRKN_PPzP: {
2009 // BRKN uses an all active implicit mask to set flags unlike the other
2010 // flag-setting instructions.
2011 // PTEST(PTRUE_B(31), BRKN(PG, A, B)) -> BRKNS(PG, A, B).
2012 if ((MaskOpcode != AArch64::PTRUE_B) ||
2013 (Mask->getOperand(1).getImm() != 31))
2014 return {};
2015 break;
2016 }
2017 case AArch64::PTRUE_B:
2018 // PTEST(OP=PTRUE_B(A), OP) -> PTRUES_B(A)
2019 break;
2020 default:
2021 // Bail out if we don't recognize the input
2022 return {};
2023 }
2024
2025 return convertToFlagSettingOpc(PredOpcode);
2026}
2027
2028/// optimizePTestInstr - Attempt to remove a ptest of a predicate-generating
2029/// operation which could set the flags in an identical manner
2030bool AArch64InstrInfo::optimizePTestInstr(
2031 MachineInstr *PTest, unsigned MaskReg, unsigned PredReg,
2032 const MachineRegisterInfo *MRI) const {
2033 auto *Mask = MRI->getUniqueVRegDef(MaskReg);
2034 auto *Pred = MRI->getUniqueVRegDef(PredReg);
2035
2036 if (Pred->isCopy() && PTest->getOpcode() == AArch64::PTEST_PP_FIRST) {
2037 // Instructions which return a multi-vector (e.g. WHILECC_x2) require copies
2038 // before the branch to extract each subregister.
2039 auto Op = Pred->getOperand(1);
2040 if (Op.isReg() && Op.getReg().isVirtual() &&
2041 Op.getSubReg() == AArch64::psub0)
2042 Pred = MRI->getUniqueVRegDef(Op.getReg());
2043 }
2044
2045 unsigned PredOpcode = Pred->getOpcode();
2046 auto NewOp = canRemovePTestInstr(PTest, Mask, Pred, MRI);
2047 if (!NewOp)
2048 return false;
2049
2050 const TargetRegisterInfo *TRI = &getRegisterInfo();
2051
2052 // If another instruction between Pred and PTest accesses flags, don't remove
2053 // the ptest or update the earlier instruction to modify them.
2054 if (areCFlagsAccessedBetweenInstrs(Pred, PTest, TRI))
2055 return false;
2056
2057 // If we pass all the checks, it's safe to remove the PTEST and use the flags
2058 // as they are prior to PTEST. Sometimes this requires the tested PTEST
2059 // operand to be replaced with an equivalent instruction that also sets the
2060 // flags.
2061 PTest->eraseFromParent();
2062 if (*NewOp != PredOpcode) {
2063 Pred->setDesc(get(*NewOp));
2064 bool succeeded = UpdateOperandRegClass(*Pred);
2065 (void)succeeded;
2066 assert(succeeded && "Operands have incompatible register classes!");
2067 Pred->addRegisterDefined(AArch64::NZCV, TRI);
2068 }
2069
2070 // Ensure that the flags def is live.
2071 if (Pred->registerDefIsDead(AArch64::NZCV, TRI)) {
2072 unsigned i = 0, e = Pred->getNumOperands();
2073 for (; i != e; ++i) {
2074 MachineOperand &MO = Pred->getOperand(i);
2075 if (MO.isReg() && MO.isDef() && MO.getReg() == AArch64::NZCV) {
2076 MO.setIsDead(false);
2077 break;
2078 }
2079 }
2080 }
2081 return true;
2082}
2083
2084/// Try to optimize a compare instruction. A compare instruction is an
2085/// instruction which produces AArch64::NZCV. It can be truly compare
2086/// instruction
2087/// when there are no uses of its destination register.
2088///
2089/// The following steps are tried in order:
2090/// 1. Convert CmpInstr into an unconditional version.
2091/// 2. Remove CmpInstr if above there is an instruction producing a needed
2092/// condition code or an instruction which can be converted into such an
2093/// instruction.
2094/// Only comparison with zero is supported.
2096 MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask,
2097 int64_t CmpValue, const MachineRegisterInfo *MRI) const {
2098 assert(CmpInstr.getParent());
2099 assert(MRI);
2100
2101 // Replace SUBSWrr with SUBWrr if NZCV is not used.
2102 int DeadNZCVIdx =
2103 CmpInstr.findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr, true);
2104 if (DeadNZCVIdx != -1) {
2105 if (CmpInstr.definesRegister(AArch64::WZR, /*TRI=*/nullptr) ||
2106 CmpInstr.definesRegister(AArch64::XZR, /*TRI=*/nullptr)) {
2107 CmpInstr.eraseFromParent();
2108 return true;
2109 }
2110 unsigned Opc = CmpInstr.getOpcode();
2111 unsigned NewOpc = convertToNonFlagSettingOpc(CmpInstr);
2112 if (NewOpc == Opc)
2113 return false;
2114 const MCInstrDesc &MCID = get(NewOpc);
2115 CmpInstr.setDesc(MCID);
2116 CmpInstr.removeOperand(DeadNZCVIdx);
2117 bool succeeded = UpdateOperandRegClass(CmpInstr);
2118 (void)succeeded;
2119 assert(succeeded && "Some operands reg class are incompatible!");
2120 return true;
2121 }
2122
2123 if (CmpInstr.getOpcode() == AArch64::PTEST_PP ||
2124 CmpInstr.getOpcode() == AArch64::PTEST_PP_ANY ||
2125 CmpInstr.getOpcode() == AArch64::PTEST_PP_FIRST)
2126 return optimizePTestInstr(&CmpInstr, SrcReg, SrcReg2, MRI);
2127
2128 if (SrcReg2 != 0)
2129 return false;
2130
2131 // CmpInstr is a Compare instruction if destination register is not used.
2132 if (!MRI->use_nodbg_empty(CmpInstr.getOperand(0).getReg()))
2133 return false;
2134
2135 if (CmpValue == 0 && substituteCmpToZero(CmpInstr, SrcReg, *MRI))
2136 return true;
2137 return (CmpValue == 0 || CmpValue == 1) &&
2138 removeCmpToZeroOrOne(CmpInstr, SrcReg, CmpValue, *MRI);
2139}
2140
2141/// Get opcode of S version of Instr.
2142/// If Instr is S version its opcode is returned.
2143/// AArch64::INSTRUCTION_LIST_END is returned if Instr does not have S version
2144/// or we are not interested in it.
2145static unsigned sForm(MachineInstr &Instr) {
2146 switch (Instr.getOpcode()) {
2147 default:
2148 return AArch64::INSTRUCTION_LIST_END;
2149
2150 case AArch64::ADDSWrr:
2151 case AArch64::ADDSWri:
2152 case AArch64::ADDSXrr:
2153 case AArch64::ADDSXri:
2154 case AArch64::ADDSWrx:
2155 case AArch64::ADDSXrx:
2156 case AArch64::ADDSWrs:
2157 case AArch64::ADDSXrs:
2158 case AArch64::SUBSWrr:
2159 case AArch64::SUBSWri:
2160 case AArch64::SUBSWrx:
2161 case AArch64::SUBSWrs:
2162 case AArch64::SUBSXrr:
2163 case AArch64::SUBSXri:
2164 case AArch64::SUBSXrx:
2165 case AArch64::SUBSXrs:
2166 case AArch64::ANDSWri:
2167 case AArch64::ANDSWrr:
2168 case AArch64::ANDSWrs:
2169 case AArch64::ANDSXri:
2170 case AArch64::ANDSXrr:
2171 case AArch64::ANDSXrs:
2172 case AArch64::BICSWrr:
2173 case AArch64::BICSXrr:
2174 case AArch64::BICSWrs:
2175 case AArch64::BICSXrs:
2176 case AArch64::ADCSWr:
2177 case AArch64::ADCSXr:
2178 case AArch64::SBCSWr:
2179 case AArch64::SBCSXr:
2180 return Instr.getOpcode();
2181
2182 case AArch64::ADDWrr:
2183 return AArch64::ADDSWrr;
2184 case AArch64::ADDWri:
2185 return AArch64::ADDSWri;
2186 case AArch64::ADDXrr:
2187 return AArch64::ADDSXrr;
2188 case AArch64::ADDXri:
2189 return AArch64::ADDSXri;
2190 case AArch64::ADDWrx:
2191 return AArch64::ADDSWrx;
2192 case AArch64::ADDXrx:
2193 return AArch64::ADDSXrx;
2194 case AArch64::ADDWrs:
2195 return AArch64::ADDSWrs;
2196 case AArch64::ADDXrs:
2197 return AArch64::ADDSXrs;
2198 case AArch64::ADCWr:
2199 return AArch64::ADCSWr;
2200 case AArch64::ADCXr:
2201 return AArch64::ADCSXr;
2202 case AArch64::SUBWrr:
2203 return AArch64::SUBSWrr;
2204 case AArch64::SUBWri:
2205 return AArch64::SUBSWri;
2206 case AArch64::SUBXrr:
2207 return AArch64::SUBSXrr;
2208 case AArch64::SUBXri:
2209 return AArch64::SUBSXri;
2210 case AArch64::SUBWrx:
2211 return AArch64::SUBSWrx;
2212 case AArch64::SUBXrx:
2213 return AArch64::SUBSXrx;
2214 case AArch64::SUBWrs:
2215 return AArch64::SUBSWrs;
2216 case AArch64::SUBXrs:
2217 return AArch64::SUBSXrs;
2218 case AArch64::SBCWr:
2219 return AArch64::SBCSWr;
2220 case AArch64::SBCXr:
2221 return AArch64::SBCSXr;
2222 case AArch64::ANDWri:
2223 return AArch64::ANDSWri;
2224 case AArch64::ANDXri:
2225 return AArch64::ANDSXri;
2226 case AArch64::ANDWrr:
2227 return AArch64::ANDSWrr;
2228 case AArch64::ANDWrs:
2229 return AArch64::ANDSWrs;
2230 case AArch64::ANDXrr:
2231 return AArch64::ANDSXrr;
2232 case AArch64::ANDXrs:
2233 return AArch64::ANDSXrs;
2234 case AArch64::BICWrr:
2235 return AArch64::BICSWrr;
2236 case AArch64::BICXrr:
2237 return AArch64::BICSXrr;
2238 case AArch64::BICWrs:
2239 return AArch64::BICSWrs;
2240 case AArch64::BICXrs:
2241 return AArch64::BICSXrs;
2242 }
2243}
2244
2245/// Check if AArch64::NZCV should be alive in successors of MBB.
2247 for (auto *BB : MBB->successors())
2248 if (BB->isLiveIn(AArch64::NZCV))
2249 return true;
2250 return false;
2251}
2252
2253/// \returns The condition code operand index for \p Instr if it is a branch
2254/// or select and -1 otherwise.
2255int AArch64InstrInfo::findCondCodeUseOperandIdxForBranchOrSelect(
2256 const MachineInstr &Instr) {
2257 switch (Instr.getOpcode()) {
2258 default:
2259 return -1;
2260
2261 case AArch64::Bcc: {
2262 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV, /*TRI=*/nullptr);
2263 assert(Idx >= 2);
2264 return Idx - 2;
2265 }
2266
2267 case AArch64::CSINVWr:
2268 case AArch64::CSINVXr:
2269 case AArch64::CSINCWr:
2270 case AArch64::CSINCXr:
2271 case AArch64::CSELWr:
2272 case AArch64::CSELXr:
2273 case AArch64::CSNEGWr:
2274 case AArch64::CSNEGXr:
2275 case AArch64::FCSELSrrr:
2276 case AArch64::FCSELDrrr: {
2277 int Idx = Instr.findRegisterUseOperandIdx(AArch64::NZCV, /*TRI=*/nullptr);
2278 assert(Idx >= 1);
2279 return Idx - 1;
2280 }
2281 }
2282}
2283
2284/// Find a condition code used by the instruction.
2285/// Returns AArch64CC::Invalid if either the instruction does not use condition
2286/// codes or we don't optimize CmpInstr in the presence of such instructions.
2288 int CCIdx =
2289 AArch64InstrInfo::findCondCodeUseOperandIdxForBranchOrSelect(Instr);
2290 return CCIdx >= 0 ? static_cast<AArch64CC::CondCode>(
2291 Instr.getOperand(CCIdx).getImm())
2293}
2294
2297 UsedNZCV UsedFlags;
2298 switch (CC) {
2299 default:
2300 break;
2301
2302 case AArch64CC::EQ: // Z set
2303 case AArch64CC::NE: // Z clear
2304 UsedFlags.Z = true;
2305 break;
2306
2307 case AArch64CC::HI: // Z clear and C set
2308 case AArch64CC::LS: // Z set or C clear
2309 UsedFlags.Z = true;
2310 [[fallthrough]];
2311 case AArch64CC::HS: // C set
2312 case AArch64CC::LO: // C clear
2313 UsedFlags.C = true;
2314 break;
2315
2316 case AArch64CC::MI: // N set
2317 case AArch64CC::PL: // N clear
2318 UsedFlags.N = true;
2319 break;
2320
2321 case AArch64CC::VS: // V set
2322 case AArch64CC::VC: // V clear
2323 UsedFlags.V = true;
2324 break;
2325
2326 case AArch64CC::GT: // Z clear, N and V the same
2327 case AArch64CC::LE: // Z set, N and V differ
2328 UsedFlags.Z = true;
2329 [[fallthrough]];
2330 case AArch64CC::GE: // N and V the same
2331 case AArch64CC::LT: // N and V differ
2332 UsedFlags.N = true;
2333 UsedFlags.V = true;
2334 break;
2335 }
2336 return UsedFlags;
2337}
2338
2339/// \returns Conditions flags used after \p CmpInstr in its MachineBB if NZCV
2340/// flags are not alive in successors of the same \p CmpInstr and \p MI parent.
2341/// \returns std::nullopt otherwise.
2342///
2343/// Collect instructions using that flags in \p CCUseInstrs if provided.
2344std::optional<UsedNZCV>
2346 const TargetRegisterInfo &TRI,
2347 SmallVectorImpl<MachineInstr *> *CCUseInstrs) {
2348 MachineBasicBlock *CmpParent = CmpInstr.getParent();
2349 if (MI.getParent() != CmpParent)
2350 return std::nullopt;
2351
2352 if (areCFlagsAliveInSuccessors(CmpParent))
2353 return std::nullopt;
2354
2355 UsedNZCV NZCVUsedAfterCmp;
2357 std::next(CmpInstr.getIterator()), CmpParent->instr_end())) {
2358 if (Instr.readsRegister(AArch64::NZCV, &TRI)) {
2360 if (CC == AArch64CC::Invalid) // Unsupported conditional instruction
2361 return std::nullopt;
2362 NZCVUsedAfterCmp |= getUsedNZCV(CC);
2363 if (CCUseInstrs)
2364 CCUseInstrs->push_back(&Instr);
2365 }
2366 if (Instr.modifiesRegister(AArch64::NZCV, &TRI))
2367 break;
2368 }
2369 return NZCVUsedAfterCmp;
2370}
2371
2372static bool isADDSRegImm(unsigned Opcode) {
2373 return Opcode == AArch64::ADDSWri || Opcode == AArch64::ADDSXri;
2374}
2375
2376static bool isSUBSRegImm(unsigned Opcode) {
2377 return Opcode == AArch64::SUBSWri || Opcode == AArch64::SUBSXri;
2378}
2379
2381 unsigned Opc = sForm(MI);
2382 switch (Opc) {
2383 case AArch64::ANDSWri:
2384 case AArch64::ANDSWrr:
2385 case AArch64::ANDSWrs:
2386 case AArch64::ANDSXri:
2387 case AArch64::ANDSXrr:
2388 case AArch64::ANDSXrs:
2389 case AArch64::BICSWrr:
2390 case AArch64::BICSXrr:
2391 case AArch64::BICSWrs:
2392 case AArch64::BICSXrs:
2393 return true;
2394 default:
2395 return false;
2396 }
2397}
2398
2399/// Check if CmpInstr can be substituted by MI.
2400///
2401/// CmpInstr can be substituted:
2402/// - CmpInstr is either 'ADDS %vreg, 0' or 'SUBS %vreg, 0'
2403/// - and, MI and CmpInstr are from the same MachineBB
2404/// - and, condition flags are not alive in successors of the CmpInstr parent
2405/// - and, if MI opcode is the S form there must be no defs of flags between
2406/// MI and CmpInstr
2407/// or if MI opcode is not the S form there must be neither defs of flags
2408/// nor uses of flags between MI and CmpInstr.
2409/// - and, C is not used after CmpInstr; CmpInstr's C is from adds/subs #0 on
2410/// SrcReg and can differ from MI (e.g. carry out of ADCS/SBCS).
2411/// - and, V is not used after CmpInstr unless MI is AND/BIC (V cleared) or MI
2412/// has NoSWrap (overflow is poison and the fold is still safe).
2414 const TargetRegisterInfo &TRI) {
2415 // MI is an opcode sForm maps (add/sub/adc/sbc/and/bic and their S forms).
2416 assert(sForm(MI) != AArch64::INSTRUCTION_LIST_END);
2417
2418 const unsigned CmpOpcode = CmpInstr.getOpcode();
2419 if (!isADDSRegImm(CmpOpcode) && !isSUBSRegImm(CmpOpcode))
2420 return false;
2421
2422 assert((CmpInstr.getOperand(2).isImm() &&
2423 CmpInstr.getOperand(2).getImm() == 0) &&
2424 "Caller guarantees that CmpInstr compares with constant 0");
2425
2426 std::optional<UsedNZCV> NZVCUsed = examineCFlagsUse(MI, CmpInstr, TRI);
2427 if (!NZVCUsed || NZVCUsed->C)
2428 return false;
2429
2430 // CmpInstr is ADDS/SUBS with immediate 0 on SrcReg (compare SrcReg to zero).
2431 // After the fold, users see NZCV from MI (or its S form), not from CmpInstr.
2432 // N/Z match CmpInstr for the value in SrcReg; C/V need not match in general
2433 // (e.g. ADCS vs adds #0), so we require C unused after CmpInstr and gate V
2434 // as below. NoSWrap makes signed overflow poison; AND/BIC clear V.
2435 if (NZVCUsed->V && !MI.getFlag(MachineInstr::NoSWrap) && !isANDOpcode(MI))
2436 return false;
2437
2438 AccessKind AccessToCheck = AK_Write;
2439 if (sForm(MI) != MI.getOpcode())
2440 AccessToCheck = AK_All;
2441 return !areCFlagsAccessedBetweenInstrs(&MI, &CmpInstr, &TRI, AccessToCheck);
2442}
2443
2444/// Substitute an instruction comparing to zero with another instruction
2445/// which produces needed condition flags.
2446///
2447/// Return true on success.
2448bool AArch64InstrInfo::substituteCmpToZero(
2449 MachineInstr &CmpInstr, unsigned SrcReg,
2450 const MachineRegisterInfo &MRI) const {
2451 // Get the unique definition of SrcReg.
2452 MachineInstr *MI = MRI.getUniqueVRegDef(SrcReg);
2453 if (!MI)
2454 return false;
2455
2456 const TargetRegisterInfo &TRI = getRegisterInfo();
2457
2458 unsigned NewOpc = sForm(*MI);
2459 if (NewOpc == AArch64::INSTRUCTION_LIST_END)
2460 return false;
2461
2462 if (!canInstrSubstituteCmpInstr(*MI, CmpInstr, TRI))
2463 return false;
2464
2465 // Update the instruction to set NZCV.
2466 MI->setDesc(get(NewOpc));
2467 CmpInstr.eraseFromParent();
2469 (void)succeeded;
2470 assert(succeeded && "Some operands reg class are incompatible!");
2471 MI->addRegisterDefined(AArch64::NZCV, &TRI);
2472 return true;
2473}
2474
2475/// \returns True if \p CmpInstr can be removed.
2476///
2477/// \p IsInvertCC is true if, after removing \p CmpInstr, condition
2478/// codes used in \p CCUseInstrs must be inverted.
2480 int CmpValue, const TargetRegisterInfo &TRI,
2482 bool &IsInvertCC) {
2483 assert((CmpValue == 0 || CmpValue == 1) &&
2484 "Only comparisons to 0 or 1 considered for removal!");
2485
2486 // MI is 'CSINCWr %vreg, wzr, wzr, <cc>' or 'CSINCXr %vreg, xzr, xzr, <cc>'
2487 unsigned MIOpc = MI.getOpcode();
2488 if (MIOpc == AArch64::CSINCWr) {
2489 if (MI.getOperand(1).getReg() != AArch64::WZR ||
2490 MI.getOperand(2).getReg() != AArch64::WZR)
2491 return false;
2492 } else if (MIOpc == AArch64::CSINCXr) {
2493 if (MI.getOperand(1).getReg() != AArch64::XZR ||
2494 MI.getOperand(2).getReg() != AArch64::XZR)
2495 return false;
2496 } else {
2497 return false;
2498 }
2500 if (MICC == AArch64CC::Invalid)
2501 return false;
2502
2503 // NZCV needs to be defined
2504 if (MI.findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr, true) != -1)
2505 return false;
2506
2507 // CmpInstr is 'ADDS %vreg, 0' or 'SUBS %vreg, 0' or 'SUBS %vreg, 1'
2508 const unsigned CmpOpcode = CmpInstr.getOpcode();
2509 bool IsSubsRegImm = isSUBSRegImm(CmpOpcode);
2510 if (CmpValue && !IsSubsRegImm)
2511 return false;
2512 if (!CmpValue && !IsSubsRegImm && !isADDSRegImm(CmpOpcode))
2513 return false;
2514
2515 // MI conditions allowed: eq, ne, mi, pl
2516 UsedNZCV MIUsedNZCV = getUsedNZCV(MICC);
2517 if (MIUsedNZCV.C || MIUsedNZCV.V)
2518 return false;
2519
2520 std::optional<UsedNZCV> NZCVUsedAfterCmp =
2521 examineCFlagsUse(MI, CmpInstr, TRI, &CCUseInstrs);
2522 // Condition flags are not used in CmpInstr basic block successors and only
2523 // Z or N flags allowed to be used after CmpInstr within its basic block
2524 if (!NZCVUsedAfterCmp || NZCVUsedAfterCmp->C || NZCVUsedAfterCmp->V)
2525 return false;
2526 // Z or N flag used after CmpInstr must correspond to the flag used in MI
2527 if ((MIUsedNZCV.Z && NZCVUsedAfterCmp->N) ||
2528 (MIUsedNZCV.N && NZCVUsedAfterCmp->Z))
2529 return false;
2530 // If CmpInstr is comparison to zero MI conditions are limited to eq, ne
2531 if (MIUsedNZCV.N && !CmpValue)
2532 return false;
2533
2534 // There must be no defs of flags between MI and CmpInstr
2535 if (areCFlagsAccessedBetweenInstrs(&MI, &CmpInstr, &TRI, AK_Write))
2536 return false;
2537
2538 // Condition code is inverted in the following cases:
2539 // 1. MI condition is ne; CmpInstr is 'ADDS %vreg, 0' or 'SUBS %vreg, 0'
2540 // 2. MI condition is eq, pl; CmpInstr is 'SUBS %vreg, 1'
2541 IsInvertCC = (CmpValue && (MICC == AArch64CC::EQ || MICC == AArch64CC::PL)) ||
2542 (!CmpValue && MICC == AArch64CC::NE);
2543 return true;
2544}
2545
2546/// Remove comparison in csinc-cmp sequence
2547///
2548/// Examples:
2549/// 1. \code
2550/// csinc w9, wzr, wzr, ne
2551/// cmp w9, #0
2552/// b.eq
2553/// \endcode
2554/// to
2555/// \code
2556/// csinc w9, wzr, wzr, ne
2557/// b.ne
2558/// \endcode
2559///
2560/// 2. \code
2561/// csinc x2, xzr, xzr, mi
2562/// cmp x2, #1
2563/// b.pl
2564/// \endcode
2565/// to
2566/// \code
2567/// csinc x2, xzr, xzr, mi
2568/// b.pl
2569/// \endcode
2570///
2571/// \param CmpInstr comparison instruction
2572/// \return True when comparison removed
2573bool AArch64InstrInfo::removeCmpToZeroOrOne(
2574 MachineInstr &CmpInstr, unsigned SrcReg, int CmpValue,
2575 const MachineRegisterInfo &MRI) const {
2576 MachineInstr *MI = MRI.getUniqueVRegDef(SrcReg);
2577 if (!MI)
2578 return false;
2579 const TargetRegisterInfo &TRI = getRegisterInfo();
2580 SmallVector<MachineInstr *, 4> CCUseInstrs;
2581 bool IsInvertCC = false;
2582 if (!canCmpInstrBeRemoved(*MI, CmpInstr, CmpValue, TRI, CCUseInstrs,
2583 IsInvertCC))
2584 return false;
2585 // Make transformation
2586 CmpInstr.eraseFromParent();
2587 if (IsInvertCC) {
2588 // Invert condition codes in CmpInstr CC users
2589 for (MachineInstr *CCUseInstr : CCUseInstrs) {
2590 int Idx = findCondCodeUseOperandIdxForBranchOrSelect(*CCUseInstr);
2591 assert(Idx >= 0 && "Unexpected instruction using CC.");
2592 MachineOperand &CCOperand = CCUseInstr->getOperand(Idx);
2594 static_cast<AArch64CC::CondCode>(CCOperand.getImm()));
2595 CCOperand.setImm(CCUse);
2596 }
2597 }
2598 return true;
2599}
2600
2601bool AArch64InstrInfo::expandPostRAPseudo(MachineInstr &MI) const {
2602 if (MI.getOpcode() != TargetOpcode::LOAD_STACK_GUARD &&
2603 MI.getOpcode() != AArch64::CATCHRET &&
2604 MI.getOpcode() != AArch64::STACK_GUARD_UNMIX)
2605 return false;
2606
2607 MachineBasicBlock &MBB = *MI.getParent();
2608 auto &Subtarget = MBB.getParent()->getSubtarget<AArch64Subtarget>();
2609 auto TRI = Subtarget.getRegisterInfo();
2610 DebugLoc DL = MI.getDebugLoc();
2611
2612 if (MI.getOpcode() == AArch64::STACK_GUARD_UNMIX) {
2613 // Expand STACK_GUARD_UNMIX to: sub Rd, fp, Rs
2614 // This computes FP - stored_mixed_value to unmix the cookie
2615 Register DstReg = MI.getOperand(0).getReg();
2616 Register SrcReg = MI.getOperand(1).getReg();
2617
2618 BuildMI(MBB, MI, DL, get(AArch64::SUBXrr), DstReg)
2619 .addReg(AArch64::FP)
2620 .addReg(SrcReg);
2621
2622 MBB.erase(MI);
2623 return true;
2624 }
2625
2626 if (MI.getOpcode() == AArch64::CATCHRET) {
2627 // Skip to the first instruction before the epilog.
2628 const TargetInstrInfo *TII =
2630 MachineBasicBlock *TargetMBB = MI.getOperand(0).getMBB();
2632 MachineBasicBlock::iterator FirstEpilogSEH = std::prev(MBBI);
2633 while (FirstEpilogSEH->getFlag(MachineInstr::FrameDestroy) &&
2634 FirstEpilogSEH != MBB.begin())
2635 FirstEpilogSEH = std::prev(FirstEpilogSEH);
2636 if (FirstEpilogSEH != MBB.begin())
2637 FirstEpilogSEH = std::next(FirstEpilogSEH);
2638 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADRP))
2639 .addReg(AArch64::X0, RegState::Define)
2640 .addMBB(TargetMBB, AArch64II::MO_PAGE);
2641 BuildMI(MBB, FirstEpilogSEH, DL, TII->get(AArch64::ADDXri))
2642 .addReg(AArch64::X0, RegState::Define)
2643 .addReg(AArch64::X0)
2645 .addImm(0);
2646 TargetMBB->setMachineBlockAddressTaken();
2647 return true;
2648 }
2649
2650 Register Reg = MI.getOperand(0).getReg();
2652 if (M.getStackProtectorGuard() == "sysreg") {
2653 const AArch64SysReg::SysReg *SrcReg =
2654 AArch64SysReg::lookupSysRegByName(M.getStackProtectorGuardReg());
2655 if (!SrcReg)
2656 report_fatal_error("Unknown SysReg for Stack Protector Guard Register");
2657
2658 // mrs xN, sysreg
2659 BuildMI(MBB, MI, DL, get(AArch64::MRS))
2661 .addImm(SrcReg->Encoding);
2662 int Offset = M.getStackProtectorGuardOffset();
2663 if (Offset >= 0 && Offset <= 32760 && Offset % 8 == 0) {
2664 // ldr xN, [xN, #offset]
2665 BuildMI(MBB, MI, DL, get(AArch64::LDRXui))
2666 .addDef(Reg)
2668 .addImm(Offset / 8);
2669 } else if (Offset >= -256 && Offset <= 255) {
2670 // ldur xN, [xN, #offset]
2671 BuildMI(MBB, MI, DL, get(AArch64::LDURXi))
2672 .addDef(Reg)
2674 .addImm(Offset);
2675 } else if (Offset >= -4095 && Offset <= 4095) {
2676 if (Offset > 0) {
2677 // add xN, xN, #offset
2678 BuildMI(MBB, MI, DL, get(AArch64::ADDXri))
2679 .addDef(Reg)
2681 .addImm(Offset)
2682 .addImm(0);
2683 } else {
2684 // sub xN, xN, #offset
2685 BuildMI(MBB, MI, DL, get(AArch64::SUBXri))
2686 .addDef(Reg)
2688 .addImm(-Offset)
2689 .addImm(0);
2690 }
2691 // ldr xN, [xN]
2692 BuildMI(MBB, MI, DL, get(AArch64::LDRXui))
2693 .addDef(Reg)
2695 .addImm(0);
2696 } else {
2697 // Cases that are larger than +/- 4095 and not a multiple of 8, or larger
2698 // than 23760.
2699 // It might be nice to use AArch64::MOVi32imm here, which would get
2700 // expanded in PreSched2 after PostRA, but our lone scratch Reg already
2701 // contains the MRS result. findScratchNonCalleeSaveRegister() in
2702 // AArch64FrameLowering might help us find such a scratch register
2703 // though. If we failed to find a scratch register, we could emit a
2704 // stream of add instructions to build up the immediate. Or, we could try
2705 // to insert a AArch64::MOVi32imm before register allocation so that we
2706 // didn't need to scavenge for a scratch register.
2707 report_fatal_error("Unable to encode Stack Protector Guard Offset");
2708 }
2709 MBB.erase(MI);
2710 return true;
2711 }
2712
2713 const GlobalValue *GV =
2714 cast<GlobalValue>((*MI.memoperands_begin())->getValue());
2715 const TargetMachine &TM = MBB.getParent()->getTarget();
2716 unsigned OpFlags = Subtarget.ClassifyGlobalReference(GV, TM);
2717 const unsigned char MO_NC = AArch64II::MO_NC;
2718
2719 unsigned GuardWidth = M.getStackProtectorGuardValueWidth().value_or(
2720 Subtarget.isTargetILP32() ? 4 : 8);
2721 if (GuardWidth != 4 && GuardWidth != 8)
2722 report_fatal_error("Unsupported stack protector value width");
2723 if ((OpFlags & AArch64II::MO_GOT) != 0) {
2724 BuildMI(MBB, MI, DL, get(AArch64::LOADgot), Reg)
2725 .addGlobalAddress(GV, 0, OpFlags);
2726 if (GuardWidth == 4) {
2727 unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32);
2728 BuildMI(MBB, MI, DL, get(AArch64::LDRWui))
2729 .addDef(Reg32, RegState::Dead)
2731 .addImm(0)
2732 .addMemOperand(*MI.memoperands_begin())
2734 } else {
2735 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
2737 .addImm(0)
2738 .addMemOperand(*MI.memoperands_begin());
2739 }
2740 } else if (TM.getCodeModel() == CodeModel::Large) {
2741 BuildMI(MBB, MI, DL, get(AArch64::MOVZXi), Reg)
2742 .addGlobalAddress(GV, 0, AArch64II::MO_G0 | MO_NC)
2743 .addImm(0);
2744 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
2746 .addGlobalAddress(GV, 0, AArch64II::MO_G1 | MO_NC)
2747 .addImm(16);
2748 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
2750 .addGlobalAddress(GV, 0, AArch64II::MO_G2 | MO_NC)
2751 .addImm(32);
2752 BuildMI(MBB, MI, DL, get(AArch64::MOVKXi), Reg)
2755 .addImm(48);
2756 if (GuardWidth == 4) {
2757 unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32);
2758 BuildMI(MBB, MI, DL, get(AArch64::LDRWui))
2759 .addDef(Reg32, RegState::Dead)
2761 .addImm(0)
2762 .addMemOperand(*MI.memoperands_begin())
2764 } else {
2765 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
2767 .addImm(0)
2768 .addMemOperand(*MI.memoperands_begin());
2769 }
2770 } else {
2771 BuildMI(MBB, MI, DL, get(AArch64::ADRP), Reg)
2772 .addGlobalAddress(GV, 0, OpFlags | AArch64II::MO_PAGE);
2773 unsigned char LoFlags = OpFlags | AArch64II::MO_PAGEOFF | MO_NC;
2774 if (GuardWidth == 4) {
2775 unsigned Reg32 = TRI->getSubReg(Reg, AArch64::sub_32);
2776 BuildMI(MBB, MI, DL, get(AArch64::LDRWui))
2777 .addDef(Reg32, RegState::Dead)
2779 .addGlobalAddress(GV, 0, LoFlags)
2780 .addMemOperand(*MI.memoperands_begin())
2782 } else {
2783 BuildMI(MBB, MI, DL, get(AArch64::LDRXui), Reg)
2785 .addGlobalAddress(GV, 0, LoFlags)
2786 .addMemOperand(*MI.memoperands_begin());
2787 }
2788 }
2789 // To match MSVC. Unlike x86_64 which uses xor instruction to mix the cookie,
2790 // we use sub instruction to mix the cookie on aarch64.
2791 // The mixing happens here in expandPostRAPseudo (after RA) to ensure we use
2792 // the final frame pointer value.
2793 if (Subtarget.getTargetTriple().isOSMSVCRT())
2794 BuildMI(MBB, MI, DL, get(AArch64::SUBXrr), Reg)
2795 .addReg(AArch64::FP)
2797
2798 MBB.erase(MI);
2799
2800 return true;
2801}
2802
2803// Return true if this instruction simply sets its single destination register
2804// to zero. This is equivalent to a register rename of the zero-register.
2806 switch (MI.getOpcode()) {
2807 default:
2808 break;
2809 case AArch64::MOVZWi:
2810 case AArch64::MOVZXi: // movz Rd, #0 (LSL #0)
2811 if (MI.getOperand(1).isImm() && MI.getOperand(1).getImm() == 0) {
2812 assert(MI.getDesc().getNumOperands() == 3 &&
2813 MI.getOperand(2).getImm() == 0 && "invalid MOVZi operands");
2814 return true;
2815 }
2816 break;
2817 case AArch64::ANDWri: // and Rd, Rzr, #imm
2818 return MI.getOperand(1).getReg() == AArch64::WZR;
2819 case AArch64::ANDXri:
2820 return MI.getOperand(1).getReg() == AArch64::XZR;
2821 case TargetOpcode::COPY:
2822 return MI.getOperand(1).getReg() == AArch64::WZR;
2823 }
2824 return false;
2825}
2826
2827// Return true if this instruction simply renames a general register without
2828// modifying bits.
2830 switch (MI.getOpcode()) {
2831 default:
2832 break;
2833 case TargetOpcode::COPY: {
2834 // GPR32 copies will by lowered to ORRXrs
2835 Register DstReg = MI.getOperand(0).getReg();
2836 return (AArch64::GPR32RegClass.contains(DstReg) ||
2837 AArch64::GPR64RegClass.contains(DstReg));
2838 }
2839 case AArch64::ORRXrs: // orr Xd, Xzr, Xm (LSL #0)
2840 if (MI.getOperand(1).getReg() == AArch64::XZR) {
2841 assert(MI.getDesc().getNumOperands() == 4 &&
2842 MI.getOperand(3).getImm() == 0 && "invalid ORRrs operands");
2843 return true;
2844 }
2845 break;
2846 case AArch64::ADDXri: // add Xd, Xn, #0 (LSL #0)
2847 if (MI.getOperand(2).getImm() == 0) {
2848 assert(MI.getDesc().getNumOperands() == 4 &&
2849 MI.getOperand(3).getImm() == 0 && "invalid ADDXri operands");
2850 return true;
2851 }
2852 break;
2853 }
2854 return false;
2855}
2856
2857// Return true if this instruction simply renames a general register without
2858// modifying bits.
2860 switch (MI.getOpcode()) {
2861 default:
2862 break;
2863 case TargetOpcode::COPY: {
2864 Register DstReg = MI.getOperand(0).getReg();
2865 return AArch64::FPR128RegClass.contains(DstReg);
2866 }
2867 case AArch64::ORRv16i8:
2868 if (MI.getOperand(1).getReg() == MI.getOperand(2).getReg()) {
2869 assert(MI.getDesc().getNumOperands() == 3 && MI.getOperand(0).isReg() &&
2870 "invalid ORRv16i8 operands");
2871 return true;
2872 }
2873 break;
2874 }
2875 return false;
2876}
2877
2878static bool isFrameLoadOpcode(int Opcode) {
2879 switch (Opcode) {
2880 default:
2881 return false;
2882 case AArch64::LDRWui:
2883 case AArch64::LDRXui:
2884 case AArch64::LDRBui:
2885 case AArch64::LDRHui:
2886 case AArch64::LDRSui:
2887 case AArch64::LDRDui:
2888 case AArch64::LDRQui:
2889 case AArch64::LDR_PXI:
2890 return true;
2891 }
2892}
2893
2895 int &FrameIndex) const {
2896 if (!isFrameLoadOpcode(MI.getOpcode()))
2897 return Register();
2898
2899 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
2900 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
2901 FrameIndex = MI.getOperand(1).getIndex();
2902 return MI.getOperand(0).getReg();
2903 }
2904 return Register();
2905}
2906
2907static bool isFrameStoreOpcode(int Opcode) {
2908 switch (Opcode) {
2909 default:
2910 return false;
2911 case AArch64::STRWui:
2912 case AArch64::STRXui:
2913 case AArch64::STRBui:
2914 case AArch64::STRHui:
2915 case AArch64::STRSui:
2916 case AArch64::STRDui:
2917 case AArch64::STRQui:
2918 case AArch64::STR_PXI:
2919 return true;
2920 }
2921}
2922
2924 int &FrameIndex) const {
2925 if (!isFrameStoreOpcode(MI.getOpcode()))
2926 return Register();
2927
2928 if (MI.getOperand(0).getSubReg() == 0 && MI.getOperand(1).isFI() &&
2929 MI.getOperand(2).isImm() && MI.getOperand(2).getImm() == 0) {
2930 FrameIndex = MI.getOperand(1).getIndex();
2931 return MI.getOperand(0).getReg();
2932 }
2933 return Register();
2934}
2935
2937 int &FrameIndex) const {
2938 if (!isFrameStoreOpcode(MI.getOpcode()))
2939 return Register();
2940
2941 if (Register Reg = isStoreToStackSlot(MI, FrameIndex))
2942 return Reg;
2943
2945 if (hasStoreToStackSlot(MI, Accesses)) {
2946 if (Accesses.size() > 1)
2947 return Register();
2948
2949 FrameIndex =
2950 cast<FixedStackPseudoSourceValue>(Accesses.front()->getPseudoValue())
2951 ->getFrameIndex();
2952 return MI.getOperand(0).getReg();
2953 }
2954 return Register();
2955}
2956
2958 int &FrameIndex) const {
2959 if (!isFrameLoadOpcode(MI.getOpcode()))
2960 return Register();
2961
2962 if (Register Reg = isLoadFromStackSlot(MI, FrameIndex))
2963 return Reg;
2964
2966 if (hasLoadFromStackSlot(MI, Accesses)) {
2967 if (Accesses.size() > 1)
2968 return Register();
2969
2970 FrameIndex =
2971 cast<FixedStackPseudoSourceValue>(Accesses.front()->getPseudoValue())
2972 ->getFrameIndex();
2973 return MI.getOperand(0).getReg();
2974 }
2975 return Register();
2976}
2977
2978/// Check all MachineMemOperands for a hint to suppress pairing.
2980 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
2981 return MMO->getFlags() & MOSuppressPair;
2982 });
2983}
2984
2985/// Set a flag on the first MachineMemOperand to suppress pairing.
2987 if (MI.memoperands_empty())
2988 return;
2989 (*MI.memoperands_begin())->setFlags(MOSuppressPair);
2990}
2991
2992/// Check all MachineMemOperands for a hint that the load/store is strided.
2994 return llvm::any_of(MI.memoperands(), [](MachineMemOperand *MMO) {
2995 return MMO->getFlags() & MOStridedAccess;
2996 });
2997}
2998
3000 switch (Opc) {
3001 default:
3002 return false;
3003 case AArch64::STURSi:
3004 case AArch64::STRSpre:
3005 case AArch64::STURDi:
3006 case AArch64::STRDpre:
3007 case AArch64::STURQi:
3008 case AArch64::STRQpre:
3009 case AArch64::STURBBi:
3010 case AArch64::STURHHi:
3011 case AArch64::STURWi:
3012 case AArch64::STRWpre:
3013 case AArch64::STURXi:
3014 case AArch64::STRXpre:
3015 case AArch64::LDURSi:
3016 case AArch64::LDRSpre:
3017 case AArch64::LDURDi:
3018 case AArch64::LDRDpre:
3019 case AArch64::LDURQi:
3020 case AArch64::LDRQpre:
3021 case AArch64::LDURWi:
3022 case AArch64::LDRWpre:
3023 case AArch64::LDURXi:
3024 case AArch64::LDRXpre:
3025 case AArch64::LDRSWpre:
3026 case AArch64::LDURSWi:
3027 case AArch64::LDURHHi:
3028 case AArch64::LDURBBi:
3029 case AArch64::LDURSBWi:
3030 case AArch64::LDURSHWi:
3031 return true;
3032 }
3033}
3034
3035std::optional<unsigned> AArch64InstrInfo::getUnscaledLdSt(unsigned Opc) {
3036 switch (Opc) {
3037 default: return {};
3038 case AArch64::PRFMui: return AArch64::PRFUMi;
3039 case AArch64::LDRXui: return AArch64::LDURXi;
3040 case AArch64::LDRWui: return AArch64::LDURWi;
3041 case AArch64::LDRBui: return AArch64::LDURBi;
3042 case AArch64::LDRHui: return AArch64::LDURHi;
3043 case AArch64::LDRSui: return AArch64::LDURSi;
3044 case AArch64::LDRDui: return AArch64::LDURDi;
3045 case AArch64::LDRQui: return AArch64::LDURQi;
3046 case AArch64::LDRBBui: return AArch64::LDURBBi;
3047 case AArch64::LDRHHui: return AArch64::LDURHHi;
3048 case AArch64::LDRSBXui: return AArch64::LDURSBXi;
3049 case AArch64::LDRSBWui: return AArch64::LDURSBWi;
3050 case AArch64::LDRSHXui: return AArch64::LDURSHXi;
3051 case AArch64::LDRSHWui: return AArch64::LDURSHWi;
3052 case AArch64::LDRSWui: return AArch64::LDURSWi;
3053 case AArch64::STRXui: return AArch64::STURXi;
3054 case AArch64::STRWui: return AArch64::STURWi;
3055 case AArch64::STRBui: return AArch64::STURBi;
3056 case AArch64::STRHui: return AArch64::STURHi;
3057 case AArch64::STRSui: return AArch64::STURSi;
3058 case AArch64::STRDui: return AArch64::STURDi;
3059 case AArch64::STRQui: return AArch64::STURQi;
3060 case AArch64::STRBBui: return AArch64::STURBBi;
3061 case AArch64::STRHHui: return AArch64::STURHHi;
3062 }
3063}
3064
3066 switch (Opc) {
3067 default:
3068 llvm_unreachable("Unhandled Opcode in getLoadStoreImmIdx");
3069 case AArch64::ADDG:
3070 case AArch64::LDAPURBi:
3071 case AArch64::LDAPURHi:
3072 case AArch64::LDAPURi:
3073 case AArch64::LDAPURSBWi:
3074 case AArch64::LDAPURSBXi:
3075 case AArch64::LDAPURSHWi:
3076 case AArch64::LDAPURSHXi:
3077 case AArch64::LDAPURSWi:
3078 case AArch64::LDAPURXi:
3079 case AArch64::LDR_PPXI:
3080 case AArch64::LDR_PXI:
3081 case AArch64::LDR_ZXI:
3082 case AArch64::LDR_ZZXI:
3083 case AArch64::LDR_ZZXI_STRIDED_CONTIGUOUS:
3084 case AArch64::LDR_ZZZXI:
3085 case AArch64::LDR_ZZZZXI:
3086 case AArch64::LDR_ZZZZXI_STRIDED_CONTIGUOUS:
3087 case AArch64::LDRBBui:
3088 case AArch64::LDRBui:
3089 case AArch64::LDRDui:
3090 case AArch64::LDRHHui:
3091 case AArch64::LDRHui:
3092 case AArch64::LDRQui:
3093 case AArch64::LDRSBWui:
3094 case AArch64::LDRSBXui:
3095 case AArch64::LDRSHWui:
3096 case AArch64::LDRSHXui:
3097 case AArch64::LDRSui:
3098 case AArch64::LDRSWui:
3099 case AArch64::LDRWui:
3100 case AArch64::LDRXui:
3101 case AArch64::LDURBBi:
3102 case AArch64::LDURBi:
3103 case AArch64::LDURDi:
3104 case AArch64::LDURHHi:
3105 case AArch64::LDURHi:
3106 case AArch64::LDURQi:
3107 case AArch64::LDURSBWi:
3108 case AArch64::LDURSBXi:
3109 case AArch64::LDURSHWi:
3110 case AArch64::LDURSHXi:
3111 case AArch64::LDURSi:
3112 case AArch64::LDURSWi:
3113 case AArch64::LDURWi:
3114 case AArch64::LDURXi:
3115 case AArch64::PRFMui:
3116 case AArch64::PRFUMi:
3117 case AArch64::ST2Gi:
3118 case AArch64::STGi:
3119 case AArch64::STLURBi:
3120 case AArch64::STLURHi:
3121 case AArch64::STLURWi:
3122 case AArch64::STLURXi:
3123 case AArch64::StoreSwiftAsyncContext:
3124 case AArch64::STR_PPXI:
3125 case AArch64::STR_PXI:
3126 case AArch64::STR_ZXI:
3127 case AArch64::STR_ZZXI:
3128 case AArch64::STR_ZZXI_STRIDED_CONTIGUOUS:
3129 case AArch64::STR_ZZZXI:
3130 case AArch64::STR_ZZZZXI:
3131 case AArch64::STR_ZZZZXI_STRIDED_CONTIGUOUS:
3132 case AArch64::STRBBui:
3133 case AArch64::STRBui:
3134 case AArch64::STRDui:
3135 case AArch64::STRHHui:
3136 case AArch64::STRHui:
3137 case AArch64::STRQui:
3138 case AArch64::STRSui:
3139 case AArch64::STRWui:
3140 case AArch64::STRXui:
3141 case AArch64::STURBBi:
3142 case AArch64::STURBi:
3143 case AArch64::STURDi:
3144 case AArch64::STURHHi:
3145 case AArch64::STURHi:
3146 case AArch64::STURQi:
3147 case AArch64::STURSi:
3148 case AArch64::STURWi:
3149 case AArch64::STURXi:
3150 case AArch64::STZ2Gi:
3151 case AArch64::STZGi:
3152 case AArch64::TAGPstack:
3153 case AArch64::ATOMIC_STORE_HINT_Bi:
3154 case AArch64::ATOMIC_STORE_HINT_Hi:
3155 case AArch64::ATOMIC_STORE_HINT_Wi:
3156 case AArch64::ATOMIC_STORE_HINT_Si:
3157 case AArch64::ATOMIC_STORE_HINT_Xi:
3158 case AArch64::ATOMIC_STORE_HINT_Di:
3159 case AArch64::ATOMIC_STORE_HINT_Bui:
3160 case AArch64::ATOMIC_STORE_HINT_Hui:
3161 case AArch64::ATOMIC_STORE_HINT_Wui:
3162 case AArch64::ATOMIC_STORE_HINT_Sui:
3163 case AArch64::ATOMIC_STORE_HINT_Xui:
3164 case AArch64::ATOMIC_STORE_HINT_Dui:
3165 return 2;
3166 case AArch64::LD1B_D_IMM:
3167 case AArch64::LD1B_H_IMM:
3168 case AArch64::LD1B_IMM:
3169 case AArch64::LD1B_S_IMM:
3170 case AArch64::LD1D_IMM:
3171 case AArch64::LD1H_D_IMM:
3172 case AArch64::LD1H_IMM:
3173 case AArch64::LD1H_S_IMM:
3174 case AArch64::LD1RB_D_IMM:
3175 case AArch64::LD1RB_H_IMM:
3176 case AArch64::LD1RB_IMM:
3177 case AArch64::LD1RB_S_IMM:
3178 case AArch64::LD1RD_IMM:
3179 case AArch64::LD1RH_D_IMM:
3180 case AArch64::LD1RH_IMM:
3181 case AArch64::LD1RH_S_IMM:
3182 case AArch64::LD1RSB_D_IMM:
3183 case AArch64::LD1RSB_H_IMM:
3184 case AArch64::LD1RSB_S_IMM:
3185 case AArch64::LD1RSH_D_IMM:
3186 case AArch64::LD1RSH_S_IMM:
3187 case AArch64::LD1RSW_IMM:
3188 case AArch64::LD1RW_D_IMM:
3189 case AArch64::LD1RW_IMM:
3190 case AArch64::LD1SB_D_IMM:
3191 case AArch64::LD1SB_H_IMM:
3192 case AArch64::LD1SB_S_IMM:
3193 case AArch64::LD1SH_D_IMM:
3194 case AArch64::LD1SH_S_IMM:
3195 case AArch64::LD1SW_D_IMM:
3196 case AArch64::LD1W_D_IMM:
3197 case AArch64::LD1W_IMM:
3198 case AArch64::LD2B_IMM:
3199 case AArch64::LD2D_IMM:
3200 case AArch64::LD2H_IMM:
3201 case AArch64::LD2W_IMM:
3202 case AArch64::LD3B_IMM:
3203 case AArch64::LD3D_IMM:
3204 case AArch64::LD3H_IMM:
3205 case AArch64::LD3W_IMM:
3206 case AArch64::LD4B_IMM:
3207 case AArch64::LD4D_IMM:
3208 case AArch64::LD4H_IMM:
3209 case AArch64::LD4W_IMM:
3210 case AArch64::LDG:
3211 case AArch64::LDNF1B_D_IMM:
3212 case AArch64::LDNF1B_H_IMM:
3213 case AArch64::LDNF1B_IMM:
3214 case AArch64::LDNF1B_S_IMM:
3215 case AArch64::LDNF1D_IMM:
3216 case AArch64::LDNF1H_D_IMM:
3217 case AArch64::LDNF1H_IMM:
3218 case AArch64::LDNF1H_S_IMM:
3219 case AArch64::LDNF1SB_D_IMM:
3220 case AArch64::LDNF1SB_H_IMM:
3221 case AArch64::LDNF1SB_S_IMM:
3222 case AArch64::LDNF1SH_D_IMM:
3223 case AArch64::LDNF1SH_S_IMM:
3224 case AArch64::LDNF1SW_D_IMM:
3225 case AArch64::LDNF1W_D_IMM:
3226 case AArch64::LDNF1W_IMM:
3227 case AArch64::LDNPDi:
3228 case AArch64::LDNPQi:
3229 case AArch64::LDNPSi:
3230 case AArch64::LDNPWi:
3231 case AArch64::LDNPXi:
3232 case AArch64::LDNT1B_ZRI:
3233 case AArch64::LDNT1D_ZRI:
3234 case AArch64::LDNT1H_ZRI:
3235 case AArch64::LDNT1W_ZRI:
3236 case AArch64::LDPDi:
3237 case AArch64::LDPQi:
3238 case AArch64::LDPSi:
3239 case AArch64::LDPWi:
3240 case AArch64::LDPXi:
3241 case AArch64::LDRBBpost:
3242 case AArch64::LDRBBpre:
3243 case AArch64::LDRBpost:
3244 case AArch64::LDRBpre:
3245 case AArch64::LDRDpost:
3246 case AArch64::LDRDpre:
3247 case AArch64::LDRHHpost:
3248 case AArch64::LDRHHpre:
3249 case AArch64::LDRHpost:
3250 case AArch64::LDRHpre:
3251 case AArch64::LDRQpost:
3252 case AArch64::LDRQpre:
3253 case AArch64::LDRSpost:
3254 case AArch64::LDRSpre:
3255 case AArch64::LDRWpost:
3256 case AArch64::LDRWpre:
3257 case AArch64::LDRXpost:
3258 case AArch64::LDRXpre:
3259 case AArch64::ST1B_D_IMM:
3260 case AArch64::ST1B_H_IMM:
3261 case AArch64::ST1B_IMM:
3262 case AArch64::ST1B_S_IMM:
3263 case AArch64::ST1D_IMM:
3264 case AArch64::ST1H_D_IMM:
3265 case AArch64::ST1H_IMM:
3266 case AArch64::ST1H_S_IMM:
3267 case AArch64::ST1W_D_IMM:
3268 case AArch64::ST1W_IMM:
3269 case AArch64::ST2B_IMM:
3270 case AArch64::ST2D_IMM:
3271 case AArch64::ST2H_IMM:
3272 case AArch64::ST2W_IMM:
3273 case AArch64::ST3B_IMM:
3274 case AArch64::ST3D_IMM:
3275 case AArch64::ST3H_IMM:
3276 case AArch64::ST3W_IMM:
3277 case AArch64::ST4B_IMM:
3278 case AArch64::ST4D_IMM:
3279 case AArch64::ST4H_IMM:
3280 case AArch64::ST4W_IMM:
3281 case AArch64::STGPi:
3282 case AArch64::STGPreIndex:
3283 case AArch64::STZGPreIndex:
3284 case AArch64::ST2GPreIndex:
3285 case AArch64::STZ2GPreIndex:
3286 case AArch64::STGPostIndex:
3287 case AArch64::STZGPostIndex:
3288 case AArch64::ST2GPostIndex:
3289 case AArch64::STZ2GPostIndex:
3290 case AArch64::STNPDi:
3291 case AArch64::STNPQi:
3292 case AArch64::STNPSi:
3293 case AArch64::STNPWi:
3294 case AArch64::STNPXi:
3295 case AArch64::STNT1B_ZRI:
3296 case AArch64::STNT1D_ZRI:
3297 case AArch64::STNT1H_ZRI:
3298 case AArch64::STNT1W_ZRI:
3299 case AArch64::STPDi:
3300 case AArch64::STPQi:
3301 case AArch64::STPSi:
3302 case AArch64::STPWi:
3303 case AArch64::STPXi:
3304 case AArch64::STRBBpost:
3305 case AArch64::STRBBpre:
3306 case AArch64::STRBpost:
3307 case AArch64::STRBpre:
3308 case AArch64::STRDpost:
3309 case AArch64::STRDpre:
3310 case AArch64::STRHHpost:
3311 case AArch64::STRHHpre:
3312 case AArch64::STRHpost:
3313 case AArch64::STRHpre:
3314 case AArch64::STRQpost:
3315 case AArch64::STRQpre:
3316 case AArch64::STRSpost:
3317 case AArch64::STRSpre:
3318 case AArch64::STRWpost:
3319 case AArch64::STRWpre:
3320 case AArch64::STRXpost:
3321 case AArch64::STRXpre:
3322 case AArch64::LD1B_2Z_IMM:
3323 case AArch64::LD1B_2Z_STRIDED_IMM:
3324 case AArch64::LD1H_2Z_IMM:
3325 case AArch64::LD1H_2Z_STRIDED_IMM:
3326 case AArch64::LD1W_2Z_IMM:
3327 case AArch64::LD1W_2Z_STRIDED_IMM:
3328 case AArch64::LD1D_2Z_IMM:
3329 case AArch64::LD1D_2Z_STRIDED_IMM:
3330 case AArch64::LD1B_4Z_IMM:
3331 case AArch64::LD1B_4Z_STRIDED_IMM:
3332 case AArch64::LD1H_4Z_IMM:
3333 case AArch64::LD1H_4Z_STRIDED_IMM:
3334 case AArch64::LD1W_4Z_IMM:
3335 case AArch64::LD1W_4Z_STRIDED_IMM:
3336 case AArch64::LD1D_4Z_IMM:
3337 case AArch64::LD1D_4Z_STRIDED_IMM:
3338 case AArch64::LD1B_2Z_IMM_PSEUDO:
3339 case AArch64::LD1H_2Z_IMM_PSEUDO:
3340 case AArch64::LD1W_2Z_IMM_PSEUDO:
3341 case AArch64::LD1D_2Z_IMM_PSEUDO:
3342 case AArch64::LD1B_4Z_IMM_PSEUDO:
3343 case AArch64::LD1H_4Z_IMM_PSEUDO:
3344 case AArch64::LD1W_4Z_IMM_PSEUDO:
3345 case AArch64::LD1D_4Z_IMM_PSEUDO:
3346 case AArch64::ST1B_2Z_IMM:
3347 case AArch64::ST1B_2Z_STRIDED_IMM:
3348 case AArch64::ST1H_2Z_IMM:
3349 case AArch64::ST1H_2Z_STRIDED_IMM:
3350 case AArch64::ST1W_2Z_IMM:
3351 case AArch64::ST1W_2Z_STRIDED_IMM:
3352 case AArch64::ST1D_2Z_IMM:
3353 case AArch64::ST1D_2Z_STRIDED_IMM:
3354 case AArch64::LDNT1B_2Z_IMM_PSEUDO:
3355 case AArch64::LDNT1B_2Z_IMM:
3356 case AArch64::LDNT1B_2Z_STRIDED_IMM:
3357 case AArch64::LDNT1H_2Z_IMM_PSEUDO:
3358 case AArch64::LDNT1H_2Z_IMM:
3359 case AArch64::LDNT1H_2Z_STRIDED_IMM:
3360 case AArch64::LDNT1W_2Z_IMM_PSEUDO:
3361 case AArch64::LDNT1W_2Z_IMM:
3362 case AArch64::LDNT1W_2Z_STRIDED_IMM:
3363 case AArch64::LDNT1D_2Z_IMM_PSEUDO:
3364 case AArch64::LDNT1D_2Z_IMM:
3365 case AArch64::LDNT1D_2Z_STRIDED_IMM:
3366 case AArch64::STNT1B_2Z_IMM:
3367 case AArch64::STNT1B_2Z_STRIDED_IMM:
3368 case AArch64::STNT1H_2Z_IMM:
3369 case AArch64::STNT1H_2Z_STRIDED_IMM:
3370 case AArch64::STNT1W_2Z_IMM:
3371 case AArch64::STNT1W_2Z_STRIDED_IMM:
3372 case AArch64::STNT1D_2Z_IMM:
3373 case AArch64::STNT1D_2Z_STRIDED_IMM:
3374 case AArch64::ST1B_2Z_IMM_PSEUDO:
3375 case AArch64::ST1H_2Z_IMM_PSEUDO:
3376 case AArch64::ST1W_2Z_IMM_PSEUDO:
3377 case AArch64::ST1D_2Z_IMM_PSEUDO:
3378 case AArch64::STNT1B_2Z_IMM_PSEUDO:
3379 case AArch64::STNT1H_2Z_IMM_PSEUDO:
3380 case AArch64::STNT1W_2Z_IMM_PSEUDO:
3381 case AArch64::STNT1D_2Z_IMM_PSEUDO:
3382 case AArch64::ST1B_4Z_IMM:
3383 case AArch64::ST1B_4Z_STRIDED_IMM:
3384 case AArch64::ST1H_4Z_IMM:
3385 case AArch64::ST1H_4Z_STRIDED_IMM:
3386 case AArch64::ST1W_4Z_IMM:
3387 case AArch64::ST1W_4Z_STRIDED_IMM:
3388 case AArch64::ST1D_4Z_IMM:
3389 case AArch64::ST1D_4Z_STRIDED_IMM:
3390 case AArch64::LDNT1B_4Z_IMM_PSEUDO:
3391 case AArch64::LDNT1B_4Z_IMM:
3392 case AArch64::LDNT1B_4Z_STRIDED_IMM:
3393 case AArch64::LDNT1H_4Z_IMM_PSEUDO:
3394 case AArch64::LDNT1H_4Z_IMM:
3395 case AArch64::LDNT1H_4Z_STRIDED_IMM:
3396 case AArch64::LDNT1W_4Z_IMM_PSEUDO:
3397 case AArch64::LDNT1W_4Z_IMM:
3398 case AArch64::LDNT1W_4Z_STRIDED_IMM:
3399 case AArch64::LDNT1D_4Z_IMM_PSEUDO:
3400 case AArch64::LDNT1D_4Z_IMM:
3401 case AArch64::LDNT1D_4Z_STRIDED_IMM:
3402 case AArch64::STNT1B_4Z_IMM:
3403 case AArch64::STNT1B_4Z_STRIDED_IMM:
3404 case AArch64::STNT1H_4Z_IMM:
3405 case AArch64::STNT1H_4Z_STRIDED_IMM:
3406 case AArch64::STNT1W_4Z_IMM:
3407 case AArch64::STNT1W_4Z_STRIDED_IMM:
3408 case AArch64::STNT1D_4Z_IMM:
3409 case AArch64::STNT1D_4Z_STRIDED_IMM:
3410 case AArch64::ST1B_4Z_IMM_PSEUDO:
3411 case AArch64::ST1H_4Z_IMM_PSEUDO:
3412 case AArch64::ST1W_4Z_IMM_PSEUDO:
3413 case AArch64::ST1D_4Z_IMM_PSEUDO:
3414 case AArch64::STNT1B_4Z_IMM_PSEUDO:
3415 case AArch64::STNT1H_4Z_IMM_PSEUDO:
3416 case AArch64::STNT1W_4Z_IMM_PSEUDO:
3417 case AArch64::STNT1D_4Z_IMM_PSEUDO:
3418 return 3;
3419 case AArch64::LDPDpost:
3420 case AArch64::LDPDpre:
3421 case AArch64::LDPQpost:
3422 case AArch64::LDPQpre:
3423 case AArch64::LDPSpost:
3424 case AArch64::LDPSpre:
3425 case AArch64::LDPWpost:
3426 case AArch64::LDPWpre:
3427 case AArch64::LDPXpost:
3428 case AArch64::LDPXpre:
3429 case AArch64::STGPpre:
3430 case AArch64::STGPpost:
3431 case AArch64::STPDpost:
3432 case AArch64::STPDpre:
3433 case AArch64::STPQpost:
3434 case AArch64::STPQpre:
3435 case AArch64::STPSpost:
3436 case AArch64::STPSpre:
3437 case AArch64::STPWpost:
3438 case AArch64::STPWpre:
3439 case AArch64::STPXpost:
3440 case AArch64::STPXpre:
3441 return 4;
3442 }
3443}
3444
3446 switch (MI.getOpcode()) {
3447 default:
3448 return false;
3449 // Scaled instructions.
3450 case AArch64::STRSui:
3451 case AArch64::STRDui:
3452 case AArch64::STRQui:
3453 case AArch64::STRXui:
3454 case AArch64::STRWui:
3455 case AArch64::LDRSui:
3456 case AArch64::LDRDui:
3457 case AArch64::LDRQui:
3458 case AArch64::LDRXui:
3459 case AArch64::LDRWui:
3460 case AArch64::LDRSWui:
3461 // Unscaled instructions.
3462 case AArch64::STURSi:
3463 case AArch64::STRSpre:
3464 case AArch64::STURDi:
3465 case AArch64::STRDpre:
3466 case AArch64::STURQi:
3467 case AArch64::STRQpre:
3468 case AArch64::STURWi:
3469 case AArch64::STRWpre:
3470 case AArch64::STURXi:
3471 case AArch64::STRXpre:
3472 case AArch64::LDURSi:
3473 case AArch64::LDRSpre:
3474 case AArch64::LDURDi:
3475 case AArch64::LDRDpre:
3476 case AArch64::LDURQi:
3477 case AArch64::LDRQpre:
3478 case AArch64::LDURWi:
3479 case AArch64::LDRWpre:
3480 case AArch64::LDURXi:
3481 case AArch64::LDRXpre:
3482 case AArch64::LDURSWi:
3483 case AArch64::LDRSWpre:
3484 // SVE instructions.
3485 case AArch64::LDR_ZXI:
3486 case AArch64::STR_ZXI:
3487 return true;
3488 }
3489}
3490
3492 switch (MI.getOpcode()) {
3493 default:
3494 assert((!MI.isCall() || !MI.isReturn()) &&
3495 "Unexpected instruction - was a new tail call opcode introduced?");
3496 return false;
3497 case AArch64::TCRETURNdi:
3498 case AArch64::TCRETURNri:
3499 case AArch64::TCRETURNrix16x17:
3500 case AArch64::TCRETURNrix17:
3501 case AArch64::TCRETURNrinotx16:
3502 case AArch64::TCRETURNriALL:
3503 case AArch64::AUTH_TCRETURN:
3504 case AArch64::AUTH_TCRETURN_BTI:
3505 return true;
3506 }
3507}
3508
3510 switch (Opc) {
3511 default:
3512 llvm_unreachable("Opcode has no flag setting equivalent!");
3513 // 32-bit cases:
3514 case AArch64::ADDWri:
3515 return AArch64::ADDSWri;
3516 case AArch64::ADDWrr:
3517 return AArch64::ADDSWrr;
3518 case AArch64::ADDWrs:
3519 return AArch64::ADDSWrs;
3520 case AArch64::ADDWrx:
3521 return AArch64::ADDSWrx;
3522 case AArch64::ANDWri:
3523 return AArch64::ANDSWri;
3524 case AArch64::ANDWrr:
3525 return AArch64::ANDSWrr;
3526 case AArch64::ANDWrs:
3527 return AArch64::ANDSWrs;
3528 case AArch64::BICWrr:
3529 return AArch64::BICSWrr;
3530 case AArch64::BICWrs:
3531 return AArch64::BICSWrs;
3532 case AArch64::SUBWri:
3533 return AArch64::SUBSWri;
3534 case AArch64::SUBWrr:
3535 return AArch64::SUBSWrr;
3536 case AArch64::SUBWrs:
3537 return AArch64::SUBSWrs;
3538 case AArch64::SUBWrx:
3539 return AArch64::SUBSWrx;
3540 // 64-bit cases:
3541 case AArch64::ADDXri:
3542 return AArch64::ADDSXri;
3543 case AArch64::ADDXrr:
3544 return AArch64::ADDSXrr;
3545 case AArch64::ADDXrs:
3546 return AArch64::ADDSXrs;
3547 case AArch64::ADDXrx:
3548 return AArch64::ADDSXrx;
3549 case AArch64::ANDXri:
3550 return AArch64::ANDSXri;
3551 case AArch64::ANDXrr:
3552 return AArch64::ANDSXrr;
3553 case AArch64::ANDXrs:
3554 return AArch64::ANDSXrs;
3555 case AArch64::BICXrr:
3556 return AArch64::BICSXrr;
3557 case AArch64::BICXrs:
3558 return AArch64::BICSXrs;
3559 case AArch64::SUBXri:
3560 return AArch64::SUBSXri;
3561 case AArch64::SUBXrr:
3562 return AArch64::SUBSXrr;
3563 case AArch64::SUBXrs:
3564 return AArch64::SUBSXrs;
3565 case AArch64::SUBXrx:
3566 return AArch64::SUBSXrx;
3567 // SVE instructions:
3568 case AArch64::AND_PPzPP:
3569 return AArch64::ANDS_PPzPP;
3570 case AArch64::BIC_PPzPP:
3571 return AArch64::BICS_PPzPP;
3572 case AArch64::EOR_PPzPP:
3573 return AArch64::EORS_PPzPP;
3574 case AArch64::NAND_PPzPP:
3575 return AArch64::NANDS_PPzPP;
3576 case AArch64::NOR_PPzPP:
3577 return AArch64::NORS_PPzPP;
3578 case AArch64::ORN_PPzPP:
3579 return AArch64::ORNS_PPzPP;
3580 case AArch64::ORR_PPzPP:
3581 return AArch64::ORRS_PPzPP;
3582 case AArch64::BRKA_PPzP:
3583 return AArch64::BRKAS_PPzP;
3584 case AArch64::BRKPA_PPzPP:
3585 return AArch64::BRKPAS_PPzPP;
3586 case AArch64::BRKB_PPzP:
3587 return AArch64::BRKBS_PPzP;
3588 case AArch64::BRKPB_PPzPP:
3589 return AArch64::BRKPBS_PPzPP;
3590 case AArch64::BRKN_PPzP:
3591 return AArch64::BRKNS_PPzP;
3592 case AArch64::RDFFR_PPz:
3593 return AArch64::RDFFRS_PPz;
3594 case AArch64::PTRUE_B:
3595 return AArch64::PTRUES_B;
3596 }
3597}
3598
3599// Is this a candidate for ld/st merging or pairing? For example, we don't
3600// touch volatiles or load/stores that have a hint to avoid pair formation.
3602
3603 bool IsPreLdSt = isPreLdSt(MI);
3604
3605 // If this is a volatile load/store, don't mess with it.
3606 if (MI.hasOrderedMemoryRef())
3607 return false;
3608
3609 // Make sure this is a reg/fi+imm (as opposed to an address reloc).
3610 // For Pre-inc LD/ST, the operand is shifted by one.
3611 assert((MI.getOperand(IsPreLdSt ? 2 : 1).isReg() ||
3612 MI.getOperand(IsPreLdSt ? 2 : 1).isFI()) &&
3613 "Expected a reg or frame index operand.");
3614
3615 // For Pre-indexed addressing quadword instructions, the third operand is the
3616 // immediate value.
3617 bool IsImmPreLdSt = IsPreLdSt && MI.getOperand(3).isImm();
3618
3619 if (!MI.getOperand(2).isImm() && !IsImmPreLdSt)
3620 return false;
3621
3622 // Can't merge/pair if the instruction modifies the base register.
3623 // e.g., ldr x0, [x0]
3624 // This case will never occur with an FI base.
3625 // However, if the instruction is an LDR<S,D,Q,W,X,SW>pre or
3626 // STR<S,D,Q,W,X>pre, it can be merged.
3627 // For example:
3628 // ldr q0, [x11, #32]!
3629 // ldr q1, [x11, #16]
3630 // to
3631 // ldp q0, q1, [x11, #32]!
3632 if (MI.getOperand(1).isReg() && !IsPreLdSt) {
3633 Register BaseReg = MI.getOperand(1).getReg();
3635 if (MI.modifiesRegister(BaseReg, TRI))
3636 return false;
3637 }
3638
3639 // Pairing SVE fills/spills is only valid for little-endian targets that
3640 // implement VLS 128.
3641 switch (MI.getOpcode()) {
3642 default:
3643 break;
3644 case AArch64::LDR_ZXI:
3645 case AArch64::STR_ZXI:
3646 if (!Subtarget.isLittleEndian() ||
3647 Subtarget.getSVEVectorSizeInBits() != 128)
3648 return false;
3649 }
3650
3651 // Check if this load/store has a hint to avoid pair formation.
3652 // MachineMemOperands hints are set by the AArch64StorePairSuppress pass.
3654 return false;
3655
3656 // Do not pair any callee-save store/reload instructions in the
3657 // prologue/epilogue if the CFI information encoded the operations as separate
3658 // instructions, as that will cause the size of the actual prologue to mismatch
3659 // with the prologue size recorded in the Windows CFI.
3660 const MCAsmInfo &MAI = MI.getMF()->getTarget().getMCAsmInfo();
3661 bool NeedsWinCFI =
3662 MAI.usesWindowsCFI() && MI.getMF()->getFunction().needsUnwindTableEntry();
3663 if (NeedsWinCFI && (MI.getFlag(MachineInstr::FrameSetup) ||
3665 return false;
3666
3667 // On some CPUs quad load/store pairs are slower than two single load/stores.
3668 if (Subtarget.isPaired128Slow()) {
3669 switch (MI.getOpcode()) {
3670 default:
3671 break;
3672 case AArch64::LDURQi:
3673 case AArch64::STURQi:
3674 case AArch64::LDRQui:
3675 case AArch64::STRQui:
3676 return false;
3677 }
3678 }
3679
3680 return true;
3681}
3682
3685 int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const {
3686 if (!LdSt.mayLoadOrStore())
3687 return false;
3688
3689 const MachineOperand *BaseOp;
3690 TypeSize WidthN(0, false);
3691 if (!getMemOperandWithOffsetWidth(LdSt, BaseOp, Offset, OffsetIsScalable,
3692 WidthN))
3693 return false;
3694 // The maximum vscale is 16 under AArch64, return the maximal extent for the
3695 // vector.
3696 Width = LocationSize::precise(WidthN);
3697 BaseOps.push_back(BaseOp);
3698 return true;
3699}
3700
3701std::optional<ExtAddrMode>
3703 const MachineOperand *Base; // Filled with the base operand of MI.
3704 int64_t Offset; // Filled with the offset of MI.
3705 bool OffsetIsScalable;
3706 if (!getMemOperandWithOffset(MemI, Base, Offset, OffsetIsScalable))
3707 return std::nullopt;
3708
3709 if (!Base->isReg())
3710 return std::nullopt;
3711 ExtAddrMode AM;
3712 AM.BaseReg = Base->getReg();
3713 AM.Displacement = Offset;
3714 AM.ScaledReg = 0;
3715 AM.Scale = 0;
3716 return AM;
3717}
3718
3720 Register Reg,
3721 const MachineInstr &AddrI,
3722 ExtAddrMode &AM) const {
3723 // Filter out instructions into which we cannot fold.
3724 unsigned NumBytes;
3725 int64_t OffsetScale = 1;
3726 switch (MemI.getOpcode()) {
3727 default:
3728 return false;
3729
3730 case AArch64::LDURQi:
3731 case AArch64::STURQi:
3732 NumBytes = 16;
3733 break;
3734
3735 case AArch64::LDURDi:
3736 case AArch64::STURDi:
3737 case AArch64::LDURXi:
3738 case AArch64::STURXi:
3739 NumBytes = 8;
3740 break;
3741
3742 case AArch64::LDURWi:
3743 case AArch64::LDURSWi:
3744 case AArch64::STURWi:
3745 NumBytes = 4;
3746 break;
3747
3748 case AArch64::LDURHi:
3749 case AArch64::STURHi:
3750 case AArch64::LDURHHi:
3751 case AArch64::STURHHi:
3752 case AArch64::LDURSHXi:
3753 case AArch64::LDURSHWi:
3754 NumBytes = 2;
3755 break;
3756
3757 case AArch64::LDRBroX:
3758 case AArch64::LDRBBroX:
3759 case AArch64::LDRSBXroX:
3760 case AArch64::LDRSBWroX:
3761 case AArch64::STRBroX:
3762 case AArch64::STRBBroX:
3763 case AArch64::LDURBi:
3764 case AArch64::LDURBBi:
3765 case AArch64::LDURSBXi:
3766 case AArch64::LDURSBWi:
3767 case AArch64::STURBi:
3768 case AArch64::STURBBi:
3769 case AArch64::LDRBui:
3770 case AArch64::LDRBBui:
3771 case AArch64::LDRSBXui:
3772 case AArch64::LDRSBWui:
3773 case AArch64::STRBui:
3774 case AArch64::STRBBui:
3775 NumBytes = 1;
3776 break;
3777
3778 case AArch64::LDRQroX:
3779 case AArch64::STRQroX:
3780 case AArch64::LDRQui:
3781 case AArch64::STRQui:
3782 NumBytes = 16;
3783 OffsetScale = 16;
3784 break;
3785
3786 case AArch64::LDRDroX:
3787 case AArch64::STRDroX:
3788 case AArch64::LDRXroX:
3789 case AArch64::STRXroX:
3790 case AArch64::LDRDui:
3791 case AArch64::STRDui:
3792 case AArch64::LDRXui:
3793 case AArch64::STRXui:
3794 NumBytes = 8;
3795 OffsetScale = 8;
3796 break;
3797
3798 case AArch64::LDRWroX:
3799 case AArch64::LDRSWroX:
3800 case AArch64::STRWroX:
3801 case AArch64::LDRWui:
3802 case AArch64::LDRSWui:
3803 case AArch64::STRWui:
3804 NumBytes = 4;
3805 OffsetScale = 4;
3806 break;
3807
3808 case AArch64::LDRHroX:
3809 case AArch64::STRHroX:
3810 case AArch64::LDRHHroX:
3811 case AArch64::STRHHroX:
3812 case AArch64::LDRSHXroX:
3813 case AArch64::LDRSHWroX:
3814 case AArch64::LDRHui:
3815 case AArch64::STRHui:
3816 case AArch64::LDRHHui:
3817 case AArch64::STRHHui:
3818 case AArch64::LDRSHXui:
3819 case AArch64::LDRSHWui:
3820 NumBytes = 2;
3821 OffsetScale = 2;
3822 break;
3823 }
3824
3825 // Check the fold operand is not the loaded/stored value.
3826 const MachineOperand &BaseRegOp = MemI.getOperand(0);
3827 if (BaseRegOp.isReg() && BaseRegOp.getReg() == Reg)
3828 return false;
3829
3830 // Handle memory instructions with a [Reg, Reg] addressing mode.
3831 if (MemI.getOperand(2).isReg()) {
3832 // Bail if the addressing mode already includes extension of the offset
3833 // register.
3834 if (MemI.getOperand(3).getImm())
3835 return false;
3836
3837 // Check if we actually have a scaled offset.
3838 if (MemI.getOperand(4).getImm() == 0)
3839 OffsetScale = 1;
3840
3841 // If the address instructions is folded into the base register, then the
3842 // addressing mode must not have a scale. Then we can swap the base and the
3843 // scaled registers.
3844 if (MemI.getOperand(1).getReg() == Reg && OffsetScale != 1)
3845 return false;
3846
3847 switch (AddrI.getOpcode()) {
3848 default:
3849 return false;
3850
3851 case AArch64::SBFMXri:
3852 // sxtw Xa, Wm
3853 // ldr Xd, [Xn, Xa, lsl #N]
3854 // ->
3855 // ldr Xd, [Xn, Wm, sxtw #N]
3856 if (AddrI.getOperand(2).getImm() != 0 ||
3857 AddrI.getOperand(3).getImm() != 31)
3858 return false;
3859
3860 AM.BaseReg = MemI.getOperand(1).getReg();
3861 if (AM.BaseReg == Reg)
3862 AM.BaseReg = MemI.getOperand(2).getReg();
3863 AM.ScaledReg = AddrI.getOperand(1).getReg();
3864 AM.Scale = OffsetScale;
3865 AM.Displacement = 0;
3867 return true;
3868
3869 case TargetOpcode::SUBREG_TO_REG: {
3870 // mov Wa, Wm
3871 // ldr Xd, [Xn, Xa, lsl #N]
3872 // ->
3873 // ldr Xd, [Xn, Wm, uxtw #N]
3874
3875 // Zero-extension looks like an ORRWrs followed by a SUBREG_TO_REG.
3876 if (AddrI.getOperand(2).getImm() != AArch64::sub_32)
3877 return false;
3878
3879 const MachineRegisterInfo &MRI = AddrI.getMF()->getRegInfo();
3880 Register OffsetReg = AddrI.getOperand(1).getReg();
3881 if (!OffsetReg.isVirtual() || !MRI.hasOneNonDBGUse(OffsetReg))
3882 return false;
3883
3884 const MachineInstr &DefMI = *MRI.getVRegDef(OffsetReg);
3885 if (DefMI.getOpcode() != AArch64::ORRWrs ||
3886 DefMI.getOperand(1).getReg() != AArch64::WZR ||
3887 DefMI.getOperand(3).getImm() != 0)
3888 return false;
3889
3890 AM.BaseReg = MemI.getOperand(1).getReg();
3891 if (AM.BaseReg == Reg)
3892 AM.BaseReg = MemI.getOperand(2).getReg();
3893 AM.ScaledReg = DefMI.getOperand(2).getReg();
3894 AM.Scale = OffsetScale;
3895 AM.Displacement = 0;
3897 return true;
3898 }
3899 }
3900 }
3901
3902 // Handle memory instructions with a [Reg, #Imm] addressing mode.
3903
3904 // Check we are not breaking a potential conversion to an LDP.
3905 auto validateOffsetForLDP = [](unsigned NumBytes, int64_t OldOffset,
3906 int64_t NewOffset) -> bool {
3907 int64_t MinOffset, MaxOffset;
3908 switch (NumBytes) {
3909 default:
3910 return true;
3911 case 4:
3912 MinOffset = -256;
3913 MaxOffset = 252;
3914 break;
3915 case 8:
3916 MinOffset = -512;
3917 MaxOffset = 504;
3918 break;
3919 case 16:
3920 MinOffset = -1024;
3921 MaxOffset = 1008;
3922 break;
3923 }
3924 return OldOffset < MinOffset || OldOffset > MaxOffset ||
3925 (NewOffset >= MinOffset && NewOffset <= MaxOffset);
3926 };
3927 auto canFoldAddSubImmIntoAddrMode = [&](int64_t Disp) -> bool {
3928 int64_t OldOffset = MemI.getOperand(2).getImm() * OffsetScale;
3929 int64_t NewOffset = OldOffset + Disp;
3930 if (!isLegalAddressingMode(NumBytes, NewOffset, /* Scale */ 0))
3931 return false;
3932 // If the old offset would fit into an LDP, but the new offset wouldn't,
3933 // bail out.
3934 if (!validateOffsetForLDP(NumBytes, OldOffset, NewOffset))
3935 return false;
3936 AM.BaseReg = AddrI.getOperand(1).getReg();
3937 AM.ScaledReg = 0;
3938 AM.Scale = 0;
3939 AM.Displacement = NewOffset;
3941 return true;
3942 };
3943
3944 auto canFoldAddRegIntoAddrMode =
3945 [&](int64_t Scale,
3947 if (MemI.getOperand(2).getImm() != 0)
3948 return false;
3949 if ((unsigned)Scale != Scale)
3950 return false;
3951 if (!isLegalAddressingMode(NumBytes, /* Offset */ 0, Scale))
3952 return false;
3953 AM.BaseReg = AddrI.getOperand(1).getReg();
3954 AM.ScaledReg = AddrI.getOperand(2).getReg();
3955 AM.Scale = Scale;
3956 AM.Displacement = 0;
3957 AM.Form = Form;
3958 return true;
3959 };
3960
3961 auto avoidSlowSTRQ = [&](const MachineInstr &MemI) {
3962 unsigned Opcode = MemI.getOpcode();
3963 return (Opcode == AArch64::STURQi || Opcode == AArch64::STRQui) &&
3964 Subtarget.isSTRQroSlow();
3965 };
3966
3967 int64_t Disp = 0;
3968 const bool OptSize = MemI.getMF()->getFunction().hasOptSize();
3969 switch (AddrI.getOpcode()) {
3970 default:
3971 return false;
3972
3973 case AArch64::ADDXri:
3974 // add Xa, Xn, #N
3975 // ldr Xd, [Xa, #M]
3976 // ->
3977 // ldr Xd, [Xn, #N'+M]
3978 Disp = AddrI.getOperand(2).getImm() << AddrI.getOperand(3).getImm();
3979 return canFoldAddSubImmIntoAddrMode(Disp);
3980
3981 case AArch64::SUBXri:
3982 // sub Xa, Xn, #N
3983 // ldr Xd, [Xa, #M]
3984 // ->
3985 // ldr Xd, [Xn, #N'+M]
3986 Disp = AddrI.getOperand(2).getImm() << AddrI.getOperand(3).getImm();
3987 return canFoldAddSubImmIntoAddrMode(-Disp);
3988
3989 case AArch64::ADDXrs: {
3990 // add Xa, Xn, Xm, lsl #N
3991 // ldr Xd, [Xa]
3992 // ->
3993 // ldr Xd, [Xn, Xm, lsl #N]
3994
3995 // Don't fold the add if the result would be slower, unless optimising for
3996 // size.
3997 unsigned Shift = static_cast<unsigned>(AddrI.getOperand(3).getImm());
3999 return false;
4000 Shift = AArch64_AM::getShiftValue(Shift);
4001 if (!OptSize) {
4002 if (Shift != 2 && Shift != 3 && Subtarget.hasAddrLSLSlow14())
4003 return false;
4004 if (avoidSlowSTRQ(MemI))
4005 return false;
4006 }
4007 return canFoldAddRegIntoAddrMode(1ULL << Shift);
4008 }
4009
4010 case AArch64::ADDXrr:
4011 // add Xa, Xn, Xm
4012 // ldr Xd, [Xa]
4013 // ->
4014 // ldr Xd, [Xn, Xm, lsl #0]
4015
4016 // Don't fold the add if the result would be slower, unless optimising for
4017 // size.
4018 if (!OptSize && avoidSlowSTRQ(MemI))
4019 return false;
4020 return canFoldAddRegIntoAddrMode(1);
4021
4022 case AArch64::ADDXrx:
4023 // add Xa, Xn, Wm, {s,u}xtw #N
4024 // ldr Xd, [Xa]
4025 // ->
4026 // ldr Xd, [Xn, Wm, {s,u}xtw #N]
4027
4028 // Don't fold the add if the result would be slower, unless optimising for
4029 // size.
4030 if (!OptSize && avoidSlowSTRQ(MemI))
4031 return false;
4032
4033 // Can fold only sign-/zero-extend of a word.
4034 unsigned Imm = static_cast<unsigned>(AddrI.getOperand(3).getImm());
4036 if (Extend != AArch64_AM::UXTW && Extend != AArch64_AM::SXTW)
4037 return false;
4038
4039 return canFoldAddRegIntoAddrMode(
4043 }
4044}
4045
4046// Given an opcode for an instruction with a [Reg, #Imm] addressing mode,
4047// return the opcode of an instruction performing the same operation, but using
4048// the [Reg, Reg] addressing mode.
4049static unsigned regOffsetOpcode(unsigned Opcode) {
4050 switch (Opcode) {
4051 default:
4052 llvm_unreachable("Address folding not implemented for instruction");
4053
4054 case AArch64::LDURQi:
4055 case AArch64::LDRQui:
4056 return AArch64::LDRQroX;
4057 case AArch64::STURQi:
4058 case AArch64::STRQui:
4059 return AArch64::STRQroX;
4060 case AArch64::LDURDi:
4061 case AArch64::LDRDui:
4062 return AArch64::LDRDroX;
4063 case AArch64::STURDi:
4064 case AArch64::STRDui:
4065 return AArch64::STRDroX;
4066 case AArch64::LDURXi:
4067 case AArch64::LDRXui:
4068 return AArch64::LDRXroX;
4069 case AArch64::STURXi:
4070 case AArch64::STRXui:
4071 return AArch64::STRXroX;
4072 case AArch64::LDURWi:
4073 case AArch64::LDRWui:
4074 return AArch64::LDRWroX;
4075 case AArch64::LDURSWi:
4076 case AArch64::LDRSWui:
4077 return AArch64::LDRSWroX;
4078 case AArch64::STURWi:
4079 case AArch64::STRWui:
4080 return AArch64::STRWroX;
4081 case AArch64::LDURHi:
4082 case AArch64::LDRHui:
4083 return AArch64::LDRHroX;
4084 case AArch64::STURHi:
4085 case AArch64::STRHui:
4086 return AArch64::STRHroX;
4087 case AArch64::LDURHHi:
4088 case AArch64::LDRHHui:
4089 return AArch64::LDRHHroX;
4090 case AArch64::STURHHi:
4091 case AArch64::STRHHui:
4092 return AArch64::STRHHroX;
4093 case AArch64::LDURSHXi:
4094 case AArch64::LDRSHXui:
4095 return AArch64::LDRSHXroX;
4096 case AArch64::LDURSHWi:
4097 case AArch64::LDRSHWui:
4098 return AArch64::LDRSHWroX;
4099 case AArch64::LDURBi:
4100 case AArch64::LDRBui:
4101 return AArch64::LDRBroX;
4102 case AArch64::LDURBBi:
4103 case AArch64::LDRBBui:
4104 return AArch64::LDRBBroX;
4105 case AArch64::LDURSBXi:
4106 case AArch64::LDRSBXui:
4107 return AArch64::LDRSBXroX;
4108 case AArch64::LDURSBWi:
4109 case AArch64::LDRSBWui:
4110 return AArch64::LDRSBWroX;
4111 case AArch64::STURBi:
4112 case AArch64::STRBui:
4113 return AArch64::STRBroX;
4114 case AArch64::STURBBi:
4115 case AArch64::STRBBui:
4116 return AArch64::STRBBroX;
4117 }
4118}
4119
4120// Given an opcode for an instruction with a [Reg, #Imm] addressing mode, return
4121// the opcode of an instruction performing the same operation, but using the
4122// [Reg, #Imm] addressing mode with scaled offset.
4123unsigned scaledOffsetOpcode(unsigned Opcode, unsigned &Scale) {
4124 switch (Opcode) {
4125 default:
4126 llvm_unreachable("Address folding not implemented for instruction");
4127
4128 case AArch64::LDURQi:
4129 Scale = 16;
4130 return AArch64::LDRQui;
4131 case AArch64::STURQi:
4132 Scale = 16;
4133 return AArch64::STRQui;
4134 case AArch64::LDURDi:
4135 Scale = 8;
4136 return AArch64::LDRDui;
4137 case AArch64::STURDi:
4138 Scale = 8;
4139 return AArch64::STRDui;
4140 case AArch64::LDURXi:
4141 Scale = 8;
4142 return AArch64::LDRXui;
4143 case AArch64::STURXi:
4144 Scale = 8;
4145 return AArch64::STRXui;
4146 case AArch64::LDURWi:
4147 Scale = 4;
4148 return AArch64::LDRWui;
4149 case AArch64::LDURSWi:
4150 Scale = 4;
4151 return AArch64::LDRSWui;
4152 case AArch64::STURWi:
4153 Scale = 4;
4154 return AArch64::STRWui;
4155 case AArch64::LDURHi:
4156 Scale = 2;
4157 return AArch64::LDRHui;
4158 case AArch64::STURHi:
4159 Scale = 2;
4160 return AArch64::STRHui;
4161 case AArch64::LDURHHi:
4162 Scale = 2;
4163 return AArch64::LDRHHui;
4164 case AArch64::STURHHi:
4165 Scale = 2;
4166 return AArch64::STRHHui;
4167 case AArch64::LDURSHXi:
4168 Scale = 2;
4169 return AArch64::LDRSHXui;
4170 case AArch64::LDURSHWi:
4171 Scale = 2;
4172 return AArch64::LDRSHWui;
4173 case AArch64::LDURBi:
4174 Scale = 1;
4175 return AArch64::LDRBui;
4176 case AArch64::LDURBBi:
4177 Scale = 1;
4178 return AArch64::LDRBBui;
4179 case AArch64::LDURSBXi:
4180 Scale = 1;
4181 return AArch64::LDRSBXui;
4182 case AArch64::LDURSBWi:
4183 Scale = 1;
4184 return AArch64::LDRSBWui;
4185 case AArch64::STURBi:
4186 Scale = 1;
4187 return AArch64::STRBui;
4188 case AArch64::STURBBi:
4189 Scale = 1;
4190 return AArch64::STRBBui;
4191 case AArch64::LDRQui:
4192 case AArch64::STRQui:
4193 Scale = 16;
4194 return Opcode;
4195 case AArch64::LDRDui:
4196 case AArch64::STRDui:
4197 case AArch64::LDRXui:
4198 case AArch64::STRXui:
4199 Scale = 8;
4200 return Opcode;
4201 case AArch64::LDRWui:
4202 case AArch64::LDRSWui:
4203 case AArch64::STRWui:
4204 Scale = 4;
4205 return Opcode;
4206 case AArch64::LDRHui:
4207 case AArch64::STRHui:
4208 case AArch64::LDRHHui:
4209 case AArch64::STRHHui:
4210 case AArch64::LDRSHXui:
4211 case AArch64::LDRSHWui:
4212 Scale = 2;
4213 return Opcode;
4214 case AArch64::LDRBui:
4215 case AArch64::LDRBBui:
4216 case AArch64::LDRSBXui:
4217 case AArch64::LDRSBWui:
4218 case AArch64::STRBui:
4219 case AArch64::STRBBui:
4220 Scale = 1;
4221 return Opcode;
4222 }
4223}
4224
4225// Given an opcode for an instruction with a [Reg, #Imm] addressing mode, return
4226// the opcode of an instruction performing the same operation, but using the
4227// [Reg, #Imm] addressing mode with unscaled offset.
4228unsigned unscaledOffsetOpcode(unsigned Opcode) {
4229 switch (Opcode) {
4230 default:
4231 llvm_unreachable("Address folding not implemented for instruction");
4232
4233 case AArch64::LDURQi:
4234 case AArch64::STURQi:
4235 case AArch64::LDURDi:
4236 case AArch64::STURDi:
4237 case AArch64::LDURXi:
4238 case AArch64::STURXi:
4239 case AArch64::LDURWi:
4240 case AArch64::LDURSWi:
4241 case AArch64::STURWi:
4242 case AArch64::LDURHi:
4243 case AArch64::STURHi:
4244 case AArch64::LDURHHi:
4245 case AArch64::STURHHi:
4246 case AArch64::LDURSHXi:
4247 case AArch64::LDURSHWi:
4248 case AArch64::LDURBi:
4249 case AArch64::STURBi:
4250 case AArch64::LDURBBi:
4251 case AArch64::STURBBi:
4252 case AArch64::LDURSBWi:
4253 case AArch64::LDURSBXi:
4254 return Opcode;
4255 case AArch64::LDRQui:
4256 return AArch64::LDURQi;
4257 case AArch64::STRQui:
4258 return AArch64::STURQi;
4259 case AArch64::LDRDui:
4260 return AArch64::LDURDi;
4261 case AArch64::STRDui:
4262 return AArch64::STURDi;
4263 case AArch64::LDRXui:
4264 return AArch64::LDURXi;
4265 case AArch64::STRXui:
4266 return AArch64::STURXi;
4267 case AArch64::LDRWui:
4268 return AArch64::LDURWi;
4269 case AArch64::LDRSWui:
4270 return AArch64::LDURSWi;
4271 case AArch64::STRWui:
4272 return AArch64::STURWi;
4273 case AArch64::LDRHui:
4274 return AArch64::LDURHi;
4275 case AArch64::STRHui:
4276 return AArch64::STURHi;
4277 case AArch64::LDRHHui:
4278 return AArch64::LDURHHi;
4279 case AArch64::STRHHui:
4280 return AArch64::STURHHi;
4281 case AArch64::LDRSHXui:
4282 return AArch64::LDURSHXi;
4283 case AArch64::LDRSHWui:
4284 return AArch64::LDURSHWi;
4285 case AArch64::LDRBBui:
4286 return AArch64::LDURBBi;
4287 case AArch64::LDRBui:
4288 return AArch64::LDURBi;
4289 case AArch64::STRBBui:
4290 return AArch64::STURBBi;
4291 case AArch64::STRBui:
4292 return AArch64::STURBi;
4293 case AArch64::LDRSBWui:
4294 return AArch64::LDURSBWi;
4295 case AArch64::LDRSBXui:
4296 return AArch64::LDURSBXi;
4297 }
4298}
4299
4300// Given the opcode of a memory load/store instruction, return the opcode of an
4301// instruction performing the same operation, but using
4302// the [Reg, Reg, {s,u}xtw #N] addressing mode with sign-/zero-extend of the
4303// offset register.
4304static unsigned offsetExtendOpcode(unsigned Opcode) {
4305 switch (Opcode) {
4306 default:
4307 llvm_unreachable("Address folding not implemented for instruction");
4308
4309 case AArch64::LDRQroX:
4310 case AArch64::LDURQi:
4311 case AArch64::LDRQui:
4312 return AArch64::LDRQroW;
4313 case AArch64::STRQroX:
4314 case AArch64::STURQi:
4315 case AArch64::STRQui:
4316 return AArch64::STRQroW;
4317 case AArch64::LDRDroX:
4318 case AArch64::LDURDi:
4319 case AArch64::LDRDui:
4320 return AArch64::LDRDroW;
4321 case AArch64::STRDroX:
4322 case AArch64::STURDi:
4323 case AArch64::STRDui:
4324 return AArch64::STRDroW;
4325 case AArch64::LDRXroX:
4326 case AArch64::LDURXi:
4327 case AArch64::LDRXui:
4328 return AArch64::LDRXroW;
4329 case AArch64::STRXroX:
4330 case AArch64::STURXi:
4331 case AArch64::STRXui:
4332 return AArch64::STRXroW;
4333 case AArch64::LDRWroX:
4334 case AArch64::LDURWi:
4335 case AArch64::LDRWui:
4336 return AArch64::LDRWroW;
4337 case AArch64::LDRSWroX:
4338 case AArch64::LDURSWi:
4339 case AArch64::LDRSWui:
4340 return AArch64::LDRSWroW;
4341 case AArch64::STRWroX:
4342 case AArch64::STURWi:
4343 case AArch64::STRWui:
4344 return AArch64::STRWroW;
4345 case AArch64::LDRHroX:
4346 case AArch64::LDURHi:
4347 case AArch64::LDRHui:
4348 return AArch64::LDRHroW;
4349 case AArch64::STRHroX:
4350 case AArch64::STURHi:
4351 case AArch64::STRHui:
4352 return AArch64::STRHroW;
4353 case AArch64::LDRHHroX:
4354 case AArch64::LDURHHi:
4355 case AArch64::LDRHHui:
4356 return AArch64::LDRHHroW;
4357 case AArch64::STRHHroX:
4358 case AArch64::STURHHi:
4359 case AArch64::STRHHui:
4360 return AArch64::STRHHroW;
4361 case AArch64::LDRSHXroX:
4362 case AArch64::LDURSHXi:
4363 case AArch64::LDRSHXui:
4364 return AArch64::LDRSHXroW;
4365 case AArch64::LDRSHWroX:
4366 case AArch64::LDURSHWi:
4367 case AArch64::LDRSHWui:
4368 return AArch64::LDRSHWroW;
4369 case AArch64::LDRBroX:
4370 case AArch64::LDURBi:
4371 case AArch64::LDRBui:
4372 return AArch64::LDRBroW;
4373 case AArch64::LDRBBroX:
4374 case AArch64::LDURBBi:
4375 case AArch64::LDRBBui:
4376 return AArch64::LDRBBroW;
4377 case AArch64::LDRSBXroX:
4378 case AArch64::LDURSBXi:
4379 case AArch64::LDRSBXui:
4380 return AArch64::LDRSBXroW;
4381 case AArch64::LDRSBWroX:
4382 case AArch64::LDURSBWi:
4383 case AArch64::LDRSBWui:
4384 return AArch64::LDRSBWroW;
4385 case AArch64::STRBroX:
4386 case AArch64::STURBi:
4387 case AArch64::STRBui:
4388 return AArch64::STRBroW;
4389 case AArch64::STRBBroX:
4390 case AArch64::STURBBi:
4391 case AArch64::STRBBui:
4392 return AArch64::STRBBroW;
4393 }
4394}
4395
4397 const ExtAddrMode &AM) const {
4398
4399 const DebugLoc &DL = MemI.getDebugLoc();
4400 MachineBasicBlock &MBB = *MemI.getParent();
4401 MachineRegisterInfo &MRI = MemI.getMF()->getRegInfo();
4402
4404 if (AM.ScaledReg) {
4405 // The new instruction will be in the form `ldr Rt, [Xn, Xm, lsl #imm]`.
4406 unsigned Opcode = regOffsetOpcode(MemI.getOpcode());
4407 MRI.constrainRegClass(AM.BaseReg, &AArch64::GPR64spRegClass);
4408 auto B = BuildMI(MBB, MemI, DL, get(Opcode))
4409 .addReg(MemI.getOperand(0).getReg(),
4410 getDefRegState(MemI.mayLoad()))
4411 .addReg(AM.BaseReg)
4412 .addReg(AM.ScaledReg)
4413 .addImm(0)
4414 .addImm(AM.Scale > 1)
4415 .setMemRefs(MemI.memoperands())
4416 .setMIFlags(MemI.getFlags());
4417 return B.getInstr();
4418 }
4419
4420 assert(AM.ScaledReg == 0 && AM.Scale == 0 &&
4421 "Addressing mode not supported for folding");
4422
4423 // The new instruction will be in the form `ld[u]r Rt, [Xn, #imm]`.
4424 unsigned Scale = 1;
4425 unsigned Opcode = MemI.getOpcode();
4426 if (isInt<9>(AM.Displacement))
4427 Opcode = unscaledOffsetOpcode(Opcode);
4428 else
4429 Opcode = scaledOffsetOpcode(Opcode, Scale);
4430
4431 auto B =
4432 BuildMI(MBB, MemI, DL, get(Opcode))
4433 .addReg(MemI.getOperand(0).getReg(), getDefRegState(MemI.mayLoad()))
4434 .addReg(AM.BaseReg)
4435 .addImm(AM.Displacement / Scale)
4436 .setMemRefs(MemI.memoperands())
4437 .setMIFlags(MemI.getFlags());
4438 return B.getInstr();
4439 }
4440
4443 // The new instruction will be in the form `ldr Rt, [Xn, Wm, {s,u}xtw #N]`.
4444 assert(AM.ScaledReg && !AM.Displacement &&
4445 "Address offset can be a register or an immediate, but not both");
4446 unsigned Opcode = offsetExtendOpcode(MemI.getOpcode());
4447 MRI.constrainRegClass(AM.BaseReg, &AArch64::GPR64spRegClass);
4448 // Make sure the offset register is in the correct register class.
4449 Register OffsetReg = AM.ScaledReg;
4450 const TargetRegisterClass *RC = MRI.getRegClass(OffsetReg);
4451 if (RC->hasSuperClassEq(&AArch64::GPR64RegClass)) {
4452 OffsetReg = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
4453 BuildMI(MBB, MemI, DL, get(TargetOpcode::COPY), OffsetReg)
4454 .addReg(AM.ScaledReg, {}, AArch64::sub_32);
4455 }
4456 auto B =
4457 BuildMI(MBB, MemI, DL, get(Opcode))
4458 .addReg(MemI.getOperand(0).getReg(), getDefRegState(MemI.mayLoad()))
4459 .addReg(AM.BaseReg)
4460 .addReg(OffsetReg)
4462 .addImm(AM.Scale != 1)
4463 .setMemRefs(MemI.memoperands())
4464 .setMIFlags(MemI.getFlags());
4465
4466 return B.getInstr();
4467 }
4468
4470 "Function must not be called with an addressing mode it can't handle");
4471}
4472
4473/// Return true if the opcode is a post-index ld/st instruction, which really
4474/// loads from base+0.
4475static bool isPostIndexLdStOpcode(unsigned Opcode) {
4476 switch (Opcode) {
4477 default:
4478 return false;
4479 case AArch64::LD1Fourv16b_POST:
4480 case AArch64::LD1Fourv1d_POST:
4481 case AArch64::LD1Fourv2d_POST:
4482 case AArch64::LD1Fourv2s_POST:
4483 case AArch64::LD1Fourv4h_POST:
4484 case AArch64::LD1Fourv4s_POST:
4485 case AArch64::LD1Fourv8b_POST:
4486 case AArch64::LD1Fourv8h_POST:
4487 case AArch64::LD1Onev16b_POST:
4488 case AArch64::LD1Onev1d_POST:
4489 case AArch64::LD1Onev2d_POST:
4490 case AArch64::LD1Onev2s_POST:
4491 case AArch64::LD1Onev4h_POST:
4492 case AArch64::LD1Onev4s_POST:
4493 case AArch64::LD1Onev8b_POST:
4494 case AArch64::LD1Onev8h_POST:
4495 case AArch64::LD1Rv16b_POST:
4496 case AArch64::LD1Rv1d_POST:
4497 case AArch64::LD1Rv2d_POST:
4498 case AArch64::LD1Rv2s_POST:
4499 case AArch64::LD1Rv4h_POST:
4500 case AArch64::LD1Rv4s_POST:
4501 case AArch64::LD1Rv8b_POST:
4502 case AArch64::LD1Rv8h_POST:
4503 case AArch64::LD1Threev16b_POST:
4504 case AArch64::LD1Threev1d_POST:
4505 case AArch64::LD1Threev2d_POST:
4506 case AArch64::LD1Threev2s_POST:
4507 case AArch64::LD1Threev4h_POST:
4508 case AArch64::LD1Threev4s_POST:
4509 case AArch64::LD1Threev8b_POST:
4510 case AArch64::LD1Threev8h_POST:
4511 case AArch64::LD1Twov16b_POST:
4512 case AArch64::LD1Twov1d_POST:
4513 case AArch64::LD1Twov2d_POST:
4514 case AArch64::LD1Twov2s_POST:
4515 case AArch64::LD1Twov4h_POST:
4516 case AArch64::LD1Twov4s_POST:
4517 case AArch64::LD1Twov8b_POST:
4518 case AArch64::LD1Twov8h_POST:
4519 case AArch64::LD1i16_POST:
4520 case AArch64::LD1i32_POST:
4521 case AArch64::LD1i64_POST:
4522 case AArch64::LD1i8_POST:
4523 case AArch64::LD2Rv16b_POST:
4524 case AArch64::LD2Rv1d_POST:
4525 case AArch64::LD2Rv2d_POST:
4526 case AArch64::LD2Rv2s_POST:
4527 case AArch64::LD2Rv4h_POST:
4528 case AArch64::LD2Rv4s_POST:
4529 case AArch64::LD2Rv8b_POST:
4530 case AArch64::LD2Rv8h_POST:
4531 case AArch64::LD2Twov16b_POST:
4532 case AArch64::LD2Twov2d_POST:
4533 case AArch64::LD2Twov2s_POST:
4534 case AArch64::LD2Twov4h_POST:
4535 case AArch64::LD2Twov4s_POST:
4536 case AArch64::LD2Twov8b_POST:
4537 case AArch64::LD2Twov8h_POST:
4538 case AArch64::LD2i16_POST:
4539 case AArch64::LD2i32_POST:
4540 case AArch64::LD2i64_POST:
4541 case AArch64::LD2i8_POST:
4542 case AArch64::LD3Rv16b_POST:
4543 case AArch64::LD3Rv1d_POST:
4544 case AArch64::LD3Rv2d_POST:
4545 case AArch64::LD3Rv2s_POST:
4546 case AArch64::LD3Rv4h_POST:
4547 case AArch64::LD3Rv4s_POST:
4548 case AArch64::LD3Rv8b_POST:
4549 case AArch64::LD3Rv8h_POST:
4550 case AArch64::LD3Threev16b_POST:
4551 case AArch64::LD3Threev2d_POST:
4552 case AArch64::LD3Threev2s_POST:
4553 case AArch64::LD3Threev4h_POST:
4554 case AArch64::LD3Threev4s_POST:
4555 case AArch64::LD3Threev8b_POST:
4556 case AArch64::LD3Threev8h_POST:
4557 case AArch64::LD3i16_POST:
4558 case AArch64::LD3i32_POST:
4559 case AArch64::LD3i64_POST:
4560 case AArch64::LD3i8_POST:
4561 case AArch64::LD4Fourv16b_POST:
4562 case AArch64::LD4Fourv2d_POST:
4563 case AArch64::LD4Fourv2s_POST:
4564 case AArch64::LD4Fourv4h_POST:
4565 case AArch64::LD4Fourv4s_POST:
4566 case AArch64::LD4Fourv8b_POST:
4567 case AArch64::LD4Fourv8h_POST:
4568 case AArch64::LD4Rv16b_POST:
4569 case AArch64::LD4Rv1d_POST:
4570 case AArch64::LD4Rv2d_POST:
4571 case AArch64::LD4Rv2s_POST:
4572 case AArch64::LD4Rv4h_POST:
4573 case AArch64::LD4Rv4s_POST:
4574 case AArch64::LD4Rv8b_POST:
4575 case AArch64::LD4Rv8h_POST:
4576 case AArch64::LD4i16_POST:
4577 case AArch64::LD4i32_POST:
4578 case AArch64::LD4i64_POST:
4579 case AArch64::LD4i8_POST:
4580 case AArch64::LDAPRWpost:
4581 case AArch64::LDAPRXpost:
4582 case AArch64::LDIAPPWpost:
4583 case AArch64::LDIAPPXpost:
4584 case AArch64::LDPDpost:
4585 case AArch64::LDPQpost:
4586 case AArch64::LDPSWpost:
4587 case AArch64::LDPSpost:
4588 case AArch64::LDPWpost:
4589 case AArch64::LDPXpost:
4590 case AArch64::LDRBBpost:
4591 case AArch64::LDRBpost:
4592 case AArch64::LDRDpost:
4593 case AArch64::LDRHHpost:
4594 case AArch64::LDRHpost:
4595 case AArch64::LDRQpost:
4596 case AArch64::LDRSBWpost:
4597 case AArch64::LDRSBXpost:
4598 case AArch64::LDRSHWpost:
4599 case AArch64::LDRSHXpost:
4600 case AArch64::LDRSWpost:
4601 case AArch64::LDRSpost:
4602 case AArch64::LDRWpost:
4603 case AArch64::LDRXpost:
4604 case AArch64::ST1Fourv16b_POST:
4605 case AArch64::ST1Fourv1d_POST:
4606 case AArch64::ST1Fourv2d_POST:
4607 case AArch64::ST1Fourv2s_POST:
4608 case AArch64::ST1Fourv4h_POST:
4609 case AArch64::ST1Fourv4s_POST:
4610 case AArch64::ST1Fourv8b_POST:
4611 case AArch64::ST1Fourv8h_POST:
4612 case AArch64::ST1Onev16b_POST:
4613 case AArch64::ST1Onev1d_POST:
4614 case AArch64::ST1Onev2d_POST:
4615 case AArch64::ST1Onev2s_POST:
4616 case AArch64::ST1Onev4h_POST:
4617 case AArch64::ST1Onev4s_POST:
4618 case AArch64::ST1Onev8b_POST:
4619 case AArch64::ST1Onev8h_POST:
4620 case AArch64::ST1Threev16b_POST:
4621 case AArch64::ST1Threev1d_POST:
4622 case AArch64::ST1Threev2d_POST:
4623 case AArch64::ST1Threev2s_POST:
4624 case AArch64::ST1Threev4h_POST:
4625 case AArch64::ST1Threev4s_POST:
4626 case AArch64::ST1Threev8b_POST:
4627 case AArch64::ST1Threev8h_POST:
4628 case AArch64::ST1Twov16b_POST:
4629 case AArch64::ST1Twov1d_POST:
4630 case AArch64::ST1Twov2d_POST:
4631 case AArch64::ST1Twov2s_POST:
4632 case AArch64::ST1Twov4h_POST:
4633 case AArch64::ST1Twov4s_POST:
4634 case AArch64::ST1Twov8b_POST:
4635 case AArch64::ST1Twov8h_POST:
4636 case AArch64::ST1i16_POST:
4637 case AArch64::ST1i32_POST:
4638 case AArch64::ST1i64_POST:
4639 case AArch64::ST1i8_POST:
4640 case AArch64::ST2GPostIndex:
4641 case AArch64::ST2Twov16b_POST:
4642 case AArch64::ST2Twov2d_POST:
4643 case AArch64::ST2Twov2s_POST:
4644 case AArch64::ST2Twov4h_POST:
4645 case AArch64::ST2Twov4s_POST:
4646 case AArch64::ST2Twov8b_POST:
4647 case AArch64::ST2Twov8h_POST:
4648 case AArch64::ST2i16_POST:
4649 case AArch64::ST2i32_POST:
4650 case AArch64::ST2i64_POST:
4651 case AArch64::ST2i8_POST:
4652 case AArch64::ST3Threev16b_POST:
4653 case AArch64::ST3Threev2d_POST:
4654 case AArch64::ST3Threev2s_POST:
4655 case AArch64::ST3Threev4h_POST:
4656 case AArch64::ST3Threev4s_POST:
4657 case AArch64::ST3Threev8b_POST:
4658 case AArch64::ST3Threev8h_POST:
4659 case AArch64::ST3i16_POST:
4660 case AArch64::ST3i32_POST:
4661 case AArch64::ST3i64_POST:
4662 case AArch64::ST3i8_POST:
4663 case AArch64::ST4Fourv16b_POST:
4664 case AArch64::ST4Fourv2d_POST:
4665 case AArch64::ST4Fourv2s_POST:
4666 case AArch64::ST4Fourv4h_POST:
4667 case AArch64::ST4Fourv4s_POST:
4668 case AArch64::ST4Fourv8b_POST:
4669 case AArch64::ST4Fourv8h_POST:
4670 case AArch64::ST4i16_POST:
4671 case AArch64::ST4i32_POST:
4672 case AArch64::ST4i64_POST:
4673 case AArch64::ST4i8_POST:
4674 case AArch64::STGPostIndex:
4675 case AArch64::STGPpost:
4676 case AArch64::STPDpost:
4677 case AArch64::STPQpost:
4678 case AArch64::STPSpost:
4679 case AArch64::STPWpost:
4680 case AArch64::STPXpost:
4681 case AArch64::STRBBpost:
4682 case AArch64::STRBpost:
4683 case AArch64::STRDpost:
4684 case AArch64::STRHHpost:
4685 case AArch64::STRHpost:
4686 case AArch64::STRQpost:
4687 case AArch64::STRSpost:
4688 case AArch64::STRWpost:
4689 case AArch64::STRXpost:
4690 case AArch64::STZ2GPostIndex:
4691 case AArch64::STZGPostIndex:
4692 return true;
4693 }
4694}
4695
4697 const MachineInstr &LdSt, const MachineOperand *&BaseOp, int64_t &Offset,
4698 bool &OffsetIsScalable, TypeSize &Width) const {
4699 assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
4700 // Handle only loads/stores with base register followed by immediate offset.
4701 if (LdSt.getNumExplicitOperands() == 3) {
4702 // Non-paired instruction (e.g., ldr x1, [x0, #8]).
4703 if ((!LdSt.getOperand(1).isReg() && !LdSt.getOperand(1).isFI()) ||
4704 !LdSt.getOperand(2).isImm())
4705 return false;
4706 } else if (LdSt.getNumExplicitOperands() == 4) {
4707 // Paired instruction (e.g., ldp x1, x2, [x0, #8]).
4708 if (!LdSt.getOperand(1).isReg() ||
4709 (!LdSt.getOperand(2).isReg() && !LdSt.getOperand(2).isFI()) ||
4710 !LdSt.getOperand(3).isImm())
4711 return false;
4712 } else
4713 return false;
4714
4715 // Get the scaling factor for the instruction and set the width for the
4716 // instruction.
4717 TypeSize Scale(0U, false);
4718 int64_t Dummy1, Dummy2;
4719
4720 // If this returns false, then it's an instruction we don't want to handle.
4721 if (!getMemOpInfo(LdSt.getOpcode(), Scale, Width, Dummy1, Dummy2))
4722 return false;
4723
4724 // Compute the offset. Offset is calculated as the immediate operand
4725 // multiplied by the scaling factor. Unscaled instructions have scaling factor
4726 // set to 1. Postindex are a special case which have an offset of 0.
4727 if (isPostIndexLdStOpcode(LdSt.getOpcode())) {
4728 BaseOp = &LdSt.getOperand(2);
4729 Offset = 0;
4730 } else if (LdSt.getNumExplicitOperands() == 3) {
4731 BaseOp = &LdSt.getOperand(1);
4732 Offset = LdSt.getOperand(2).getImm() * Scale.getKnownMinValue();
4733 } else {
4734 assert(LdSt.getNumExplicitOperands() == 4 && "invalid number of operands");
4735 BaseOp = &LdSt.getOperand(2);
4736 Offset = LdSt.getOperand(3).getImm() * Scale.getKnownMinValue();
4737 }
4738 OffsetIsScalable = Scale.isScalable();
4739
4740 return BaseOp->isReg() || BaseOp->isFI();
4741}
4742
4745 assert(LdSt.mayLoadOrStore() && "Expected a memory operation.");
4746 MachineOperand &OfsOp = LdSt.getOperand(LdSt.getNumExplicitOperands() - 1);
4747 assert(OfsOp.isImm() && "Offset operand wasn't immediate.");
4748 return OfsOp;
4749}
4750
4751bool AArch64InstrInfo::getMemOpInfo(unsigned Opcode, TypeSize &Scale,
4752 TypeSize &Width, int64_t &MinOffset,
4753 int64_t &MaxOffset) {
4754 switch (Opcode) {
4755 // Not a memory operation or something we want to handle.
4756 default:
4757 Scale = Width = TypeSize::getFixed(0);
4758 MinOffset = MaxOffset = 0;
4759 return false;
4760 // LDR / STR
4761 case AArch64::LDRQui:
4762 case AArch64::STRQui:
4763 Scale = Width = TypeSize::getFixed(16);
4764 MinOffset = 0;
4765 MaxOffset = 4095;
4766 break;
4767 case AArch64::LDRXui:
4768 case AArch64::LDRDui:
4769 case AArch64::STRXui:
4770 case AArch64::STRDui:
4771 case AArch64::PRFMui:
4772 case AArch64::ATOMIC_STORE_HINT_Xui:
4773 case AArch64::ATOMIC_STORE_HINT_Dui:
4774 Scale = Width = TypeSize::getFixed(8);
4775 MinOffset = 0;
4776 MaxOffset = 4095;
4777 break;
4778 case AArch64::LDRWui:
4779 case AArch64::LDRSui:
4780 case AArch64::LDRSWui:
4781 case AArch64::STRWui:
4782 case AArch64::STRSui:
4783 case AArch64::ATOMIC_STORE_HINT_Wui:
4784 case AArch64::ATOMIC_STORE_HINT_Sui:
4785 Scale = Width = TypeSize::getFixed(4);
4786 MinOffset = 0;
4787 MaxOffset = 4095;
4788 break;
4789 case AArch64::LDRHui:
4790 case AArch64::LDRHHui:
4791 case AArch64::LDRSHWui:
4792 case AArch64::LDRSHXui:
4793 case AArch64::STRHui:
4794 case AArch64::STRHHui:
4795 case AArch64::ATOMIC_STORE_HINT_Hui:
4796 Scale = Width = TypeSize::getFixed(2);
4797 MinOffset = 0;
4798 MaxOffset = 4095;
4799 break;
4800 case AArch64::LDRBui:
4801 case AArch64::LDRBBui:
4802 case AArch64::LDRSBWui:
4803 case AArch64::LDRSBXui:
4804 case AArch64::STRBui:
4805 case AArch64::STRBBui:
4806 case AArch64::ATOMIC_STORE_HINT_Bui:
4807 Scale = Width = TypeSize::getFixed(1);
4808 MinOffset = 0;
4809 MaxOffset = 4095;
4810 break;
4811 // post/pre inc
4812 case AArch64::STRQpre:
4813 case AArch64::LDRQpost:
4814 Scale = TypeSize::getFixed(1);
4815 Width = TypeSize::getFixed(16);
4816 MinOffset = -256;
4817 MaxOffset = 255;
4818 break;
4819 case AArch64::LDRDpost:
4820 case AArch64::LDRDpre:
4821 case AArch64::LDRXpost:
4822 case AArch64::LDRXpre:
4823 case AArch64::STRDpost:
4824 case AArch64::STRDpre:
4825 case AArch64::STRXpost:
4826 case AArch64::STRXpre:
4827 Scale = TypeSize::getFixed(1);
4828 Width = TypeSize::getFixed(8);
4829 MinOffset = -256;
4830 MaxOffset = 255;
4831 break;
4832 case AArch64::STRWpost:
4833 case AArch64::STRWpre:
4834 case AArch64::LDRWpost:
4835 case AArch64::LDRWpre:
4836 case AArch64::STRSpost:
4837 case AArch64::STRSpre:
4838 case AArch64::LDRSpost:
4839 case AArch64::LDRSpre:
4840 Scale = TypeSize::getFixed(1);
4841 Width = TypeSize::getFixed(4);
4842 MinOffset = -256;
4843 MaxOffset = 255;
4844 break;
4845 case AArch64::LDRHpost:
4846 case AArch64::LDRHpre:
4847 case AArch64::STRHpost:
4848 case AArch64::STRHpre:
4849 case AArch64::LDRHHpost:
4850 case AArch64::LDRHHpre:
4851 case AArch64::STRHHpost:
4852 case AArch64::STRHHpre:
4853 Scale = TypeSize::getFixed(1);
4854 Width = TypeSize::getFixed(2);
4855 MinOffset = -256;
4856 MaxOffset = 255;
4857 break;
4858 case AArch64::LDRBpost:
4859 case AArch64::LDRBpre:
4860 case AArch64::STRBpost:
4861 case AArch64::STRBpre:
4862 case AArch64::LDRBBpost:
4863 case AArch64::LDRBBpre:
4864 case AArch64::STRBBpost:
4865 case AArch64::STRBBpre:
4866 Scale = Width = TypeSize::getFixed(1);
4867 MinOffset = -256;
4868 MaxOffset = 255;
4869 break;
4870 // Unscaled
4871 case AArch64::LDURQi:
4872 case AArch64::STURQi:
4873 Scale = TypeSize::getFixed(1);
4874 Width = TypeSize::getFixed(16);
4875 MinOffset = -256;
4876 MaxOffset = 255;
4877 break;
4878 case AArch64::LDURXi:
4879 case AArch64::LDURDi:
4880 case AArch64::LDAPURXi:
4881 case AArch64::STURXi:
4882 case AArch64::STURDi:
4883 case AArch64::STLURXi:
4884 case AArch64::PRFUMi:
4885 case AArch64::ATOMIC_STORE_HINT_Xi:
4886 case AArch64::ATOMIC_STORE_HINT_Di:
4887 Scale = TypeSize::getFixed(1);
4888 Width = TypeSize::getFixed(8);
4889 MinOffset = -256;
4890 MaxOffset = 255;
4891 break;
4892 case AArch64::LDURWi:
4893 case AArch64::LDURSi:
4894 case AArch64::LDURSWi:
4895 case AArch64::LDAPURi:
4896 case AArch64::LDAPURSWi:
4897 case AArch64::STURWi:
4898 case AArch64::STURSi:
4899 case AArch64::STLURWi:
4900 case AArch64::ATOMIC_STORE_HINT_Wi:
4901 case AArch64::ATOMIC_STORE_HINT_Si:
4902 Scale = TypeSize::getFixed(1);
4903 Width = TypeSize::getFixed(4);
4904 MinOffset = -256;
4905 MaxOffset = 255;
4906 break;
4907 case AArch64::LDURHi:
4908 case AArch64::LDURHHi:
4909 case AArch64::LDURSHXi:
4910 case AArch64::LDURSHWi:
4911 case AArch64::LDAPURHi:
4912 case AArch64::LDAPURSHWi:
4913 case AArch64::LDAPURSHXi:
4914 case AArch64::STURHi:
4915 case AArch64::STURHHi:
4916 case AArch64::STLURHi:
4917 case AArch64::ATOMIC_STORE_HINT_Hi:
4918 Scale = TypeSize::getFixed(1);
4919 Width = TypeSize::getFixed(2);
4920 MinOffset = -256;
4921 MaxOffset = 255;
4922 break;
4923 case AArch64::LDURBi:
4924 case AArch64::LDURBBi:
4925 case AArch64::LDURSBXi:
4926 case AArch64::LDURSBWi:
4927 case AArch64::LDAPURBi:
4928 case AArch64::LDAPURSBWi:
4929 case AArch64::LDAPURSBXi:
4930 case AArch64::STURBi:
4931 case AArch64::STURBBi:
4932 case AArch64::STLURBi:
4933 case AArch64::ATOMIC_STORE_HINT_Bi:
4934 Scale = Width = TypeSize::getFixed(1);
4935 MinOffset = -256;
4936 MaxOffset = 255;
4937 break;
4938 // LDP / STP (including pre/post inc)
4939 case AArch64::LDPQi:
4940 case AArch64::LDNPQi:
4941 case AArch64::STPQi:
4942 case AArch64::STNPQi:
4943 case AArch64::LDPQpost:
4944 case AArch64::LDPQpre:
4945 case AArch64::STPQpost:
4946 case AArch64::STPQpre:
4947 Scale = TypeSize::getFixed(16);
4948 Width = TypeSize::getFixed(16 * 2);
4949 MinOffset = -64;
4950 MaxOffset = 63;
4951 break;
4952 case AArch64::LDPXi:
4953 case AArch64::LDPDi:
4954 case AArch64::LDNPXi:
4955 case AArch64::LDNPDi:
4956 case AArch64::STPXi:
4957 case AArch64::STPDi:
4958 case AArch64::STNPXi:
4959 case AArch64::STNPDi:
4960 case AArch64::LDPDpost:
4961 case AArch64::LDPDpre:
4962 case AArch64::LDPXpost:
4963 case AArch64::LDPXpre:
4964 case AArch64::STPDpost:
4965 case AArch64::STPDpre:
4966 case AArch64::STPXpost:
4967 case AArch64::STPXpre:
4968 Scale = TypeSize::getFixed(8);
4969 Width = TypeSize::getFixed(8 * 2);
4970 MinOffset = -64;
4971 MaxOffset = 63;
4972 break;
4973 case AArch64::LDPWi:
4974 case AArch64::LDPSi:
4975 case AArch64::LDNPWi:
4976 case AArch64::LDNPSi:
4977 case AArch64::STPWi:
4978 case AArch64::STPSi:
4979 case AArch64::STNPWi:
4980 case AArch64::STNPSi:
4981 case AArch64::LDPSpost:
4982 case AArch64::LDPSpre:
4983 case AArch64::LDPWpost:
4984 case AArch64::LDPWpre:
4985 case AArch64::STPSpost:
4986 case AArch64::STPSpre:
4987 case AArch64::STPWpost:
4988 case AArch64::STPWpre:
4989 Scale = TypeSize::getFixed(4);
4990 Width = TypeSize::getFixed(4 * 2);
4991 MinOffset = -64;
4992 MaxOffset = 63;
4993 break;
4994 case AArch64::StoreSwiftAsyncContext:
4995 // Store is an STRXui, but there might be an ADDXri in the expansion too.
4996 Scale = TypeSize::getFixed(1);
4997 Width = TypeSize::getFixed(8);
4998 MinOffset = 0;
4999 MaxOffset = 4095;
5000 break;
5001 case AArch64::ADDG:
5002 Scale = TypeSize::getFixed(16);
5003 Width = TypeSize::getFixed(0);
5004 MinOffset = 0;
5005 MaxOffset = 63;
5006 break;
5007 case AArch64::TAGPstack:
5008 Scale = TypeSize::getFixed(16);
5009 Width = TypeSize::getFixed(0);
5010 // TAGP with a negative offset turns into SUBP, which has a maximum offset
5011 // of 63 (not 64!).
5012 MinOffset = -63;
5013 MaxOffset = 63;
5014 break;
5015 case AArch64::LDG:
5016 case AArch64::STGi:
5017 case AArch64::STGPreIndex:
5018 case AArch64::STGPostIndex:
5019 case AArch64::STZGi:
5020 case AArch64::STZGPreIndex:
5021 case AArch64::STZGPostIndex:
5022 Scale = Width = TypeSize::getFixed(16);
5023 MinOffset = -256;
5024 MaxOffset = 255;
5025 break;
5026 // SVE
5027 case AArch64::STR_ZZZZXI:
5028 case AArch64::STR_ZZZZXI_STRIDED_CONTIGUOUS:
5029 case AArch64::LDR_ZZZZXI:
5030 case AArch64::LDR_ZZZZXI_STRIDED_CONTIGUOUS:
5031 Scale = TypeSize::getScalable(16);
5032 Width = TypeSize::getScalable(16 * 4);
5033 MinOffset = -256;
5034 MaxOffset = 252;
5035 break;
5036 case AArch64::STR_ZZZXI:
5037 case AArch64::LDR_ZZZXI:
5038 Scale = TypeSize::getScalable(16);
5039 Width = TypeSize::getScalable(16 * 3);
5040 MinOffset = -256;
5041 MaxOffset = 253;
5042 break;
5043 case AArch64::STR_ZZXI:
5044 case AArch64::STR_ZZXI_STRIDED_CONTIGUOUS:
5045 case AArch64::LDR_ZZXI:
5046 case AArch64::LDR_ZZXI_STRIDED_CONTIGUOUS:
5047 Scale = TypeSize::getScalable(16);
5048 Width = TypeSize::getScalable(16 * 2);
5049 MinOffset = -256;
5050 MaxOffset = 254;
5051 break;
5052 case AArch64::LDR_PXI:
5053 case AArch64::STR_PXI:
5054 Scale = Width = TypeSize::getScalable(2);
5055 MinOffset = -256;
5056 MaxOffset = 255;
5057 break;
5058 case AArch64::LDR_PPXI:
5059 case AArch64::STR_PPXI:
5060 Scale = TypeSize::getScalable(2);
5061 Width = TypeSize::getScalable(2 * 2);
5062 MinOffset = -256;
5063 MaxOffset = 254;
5064 break;
5065 case AArch64::LDR_ZXI:
5066 case AArch64::STR_ZXI:
5067 Scale = Width = TypeSize::getScalable(16);
5068 MinOffset = -256;
5069 MaxOffset = 255;
5070 break;
5071 case AArch64::LD1B_IMM:
5072 case AArch64::LD1H_IMM:
5073 case AArch64::LD1W_IMM:
5074 case AArch64::LD1D_IMM:
5075 case AArch64::LDNT1B_ZRI:
5076 case AArch64::LDNT1H_ZRI:
5077 case AArch64::LDNT1W_ZRI:
5078 case AArch64::LDNT1D_ZRI:
5079 case AArch64::ST1B_IMM:
5080 case AArch64::ST1H_IMM:
5081 case AArch64::ST1W_IMM:
5082 case AArch64::ST1D_IMM:
5083 case AArch64::STNT1B_ZRI:
5084 case AArch64::STNT1H_ZRI:
5085 case AArch64::STNT1W_ZRI:
5086 case AArch64::STNT1D_ZRI:
5087 case AArch64::LDNF1B_IMM:
5088 case AArch64::LDNF1H_IMM:
5089 case AArch64::LDNF1W_IMM:
5090 case AArch64::LDNF1D_IMM:
5091 // A full vectors worth of data
5092 // Width = mbytes * elements
5093 Scale = Width = TypeSize::getScalable(16);
5094 MinOffset = -8;
5095 MaxOffset = 7;
5096 break;
5097 case AArch64::LD2B_IMM:
5098 case AArch64::LD2H_IMM:
5099 case AArch64::LD2W_IMM:
5100 case AArch64::LD2D_IMM:
5101 case AArch64::ST2B_IMM:
5102 case AArch64::ST2H_IMM:
5103 case AArch64::ST2W_IMM:
5104 case AArch64::ST2D_IMM:
5105 case AArch64::LD1B_2Z_IMM:
5106 case AArch64::LD1B_2Z_STRIDED_IMM:
5107 case AArch64::LD1H_2Z_IMM:
5108 case AArch64::LD1H_2Z_STRIDED_IMM:
5109 case AArch64::LD1W_2Z_IMM:
5110 case AArch64::LD1W_2Z_STRIDED_IMM:
5111 case AArch64::LD1D_2Z_IMM:
5112 case AArch64::LD1D_2Z_STRIDED_IMM:
5113 case AArch64::LD1B_2Z_IMM_PSEUDO:
5114 case AArch64::LD1H_2Z_IMM_PSEUDO:
5115 case AArch64::LD1W_2Z_IMM_PSEUDO:
5116 case AArch64::LD1D_2Z_IMM_PSEUDO:
5117 case AArch64::ST1B_2Z_IMM:
5118 case AArch64::ST1B_2Z_STRIDED_IMM:
5119 case AArch64::ST1H_2Z_IMM:
5120 case AArch64::ST1H_2Z_STRIDED_IMM:
5121 case AArch64::ST1W_2Z_IMM:
5122 case AArch64::ST1W_2Z_STRIDED_IMM:
5123 case AArch64::ST1D_2Z_IMM:
5124 case AArch64::ST1D_2Z_STRIDED_IMM:
5125 case AArch64::LDNT1B_2Z_IMM_PSEUDO:
5126 case AArch64::LDNT1B_2Z_IMM:
5127 case AArch64::LDNT1B_2Z_STRIDED_IMM:
5128 case AArch64::LDNT1H_2Z_IMM_PSEUDO:
5129 case AArch64::LDNT1H_2Z_IMM:
5130 case AArch64::LDNT1H_2Z_STRIDED_IMM:
5131 case AArch64::LDNT1W_2Z_IMM_PSEUDO:
5132 case AArch64::LDNT1W_2Z_IMM:
5133 case AArch64::LDNT1W_2Z_STRIDED_IMM:
5134 case AArch64::LDNT1D_2Z_IMM_PSEUDO:
5135 case AArch64::LDNT1D_2Z_IMM:
5136 case AArch64::LDNT1D_2Z_STRIDED_IMM:
5137 case AArch64::STNT1B_2Z_IMM:
5138 case AArch64::STNT1B_2Z_STRIDED_IMM:
5139 case AArch64::STNT1H_2Z_IMM:
5140 case AArch64::STNT1H_2Z_STRIDED_IMM:
5141 case AArch64::STNT1W_2Z_IMM:
5142 case AArch64::STNT1W_2Z_STRIDED_IMM:
5143 case AArch64::STNT1D_2Z_IMM:
5144 case AArch64::STNT1D_2Z_STRIDED_IMM:
5145 case AArch64::ST1B_2Z_IMM_PSEUDO:
5146 case AArch64::ST1H_2Z_IMM_PSEUDO:
5147 case AArch64::ST1W_2Z_IMM_PSEUDO:
5148 case AArch64::ST1D_2Z_IMM_PSEUDO:
5149 case AArch64::STNT1B_2Z_IMM_PSEUDO:
5150 case AArch64::STNT1H_2Z_IMM_PSEUDO:
5151 case AArch64::STNT1W_2Z_IMM_PSEUDO:
5152 case AArch64::STNT1D_2Z_IMM_PSEUDO:
5153 Scale = Width = TypeSize::getScalable(16 * 2);
5154 MinOffset = -8;
5155 MaxOffset = 7;
5156 break;
5157 case AArch64::LD3B_IMM:
5158 case AArch64::LD3H_IMM:
5159 case AArch64::LD3W_IMM:
5160 case AArch64::LD3D_IMM:
5161 case AArch64::ST3B_IMM:
5162 case AArch64::ST3H_IMM:
5163 case AArch64::ST3W_IMM:
5164 case AArch64::ST3D_IMM:
5165 Scale = Width = TypeSize::getScalable(16 * 3);
5166 MinOffset = -8;
5167 MaxOffset = 7;
5168 break;
5169 case AArch64::LD4B_IMM:
5170 case AArch64::LD4H_IMM:
5171 case AArch64::LD4W_IMM:
5172 case AArch64::LD4D_IMM:
5173 case AArch64::ST4B_IMM:
5174 case AArch64::ST4H_IMM:
5175 case AArch64::ST4W_IMM:
5176 case AArch64::ST4D_IMM:
5177 case AArch64::LD1B_4Z_IMM:
5178 case AArch64::LD1B_4Z_STRIDED_IMM:
5179 case AArch64::LD1H_4Z_IMM:
5180 case AArch64::LD1H_4Z_STRIDED_IMM:
5181 case AArch64::LD1W_4Z_IMM:
5182 case AArch64::LD1W_4Z_STRIDED_IMM:
5183 case AArch64::LD1D_4Z_IMM:
5184 case AArch64::LD1D_4Z_STRIDED_IMM:
5185 case AArch64::LD1B_4Z_IMM_PSEUDO:
5186 case AArch64::LD1H_4Z_IMM_PSEUDO:
5187 case AArch64::LD1W_4Z_IMM_PSEUDO:
5188 case AArch64::LD1D_4Z_IMM_PSEUDO:
5189 case AArch64::ST1B_4Z_IMM:
5190 case AArch64::ST1B_4Z_STRIDED_IMM:
5191 case AArch64::ST1H_4Z_IMM:
5192 case AArch64::ST1H_4Z_STRIDED_IMM:
5193 case AArch64::ST1W_4Z_IMM:
5194 case AArch64::ST1W_4Z_STRIDED_IMM:
5195 case AArch64::ST1D_4Z_IMM:
5196 case AArch64::ST1D_4Z_STRIDED_IMM:
5197 case AArch64::LDNT1B_4Z_IMM_PSEUDO:
5198 case AArch64::LDNT1B_4Z_IMM:
5199 case AArch64::LDNT1B_4Z_STRIDED_IMM:
5200 case AArch64::LDNT1H_4Z_IMM_PSEUDO:
5201 case AArch64::LDNT1H_4Z_IMM:
5202 case AArch64::LDNT1H_4Z_STRIDED_IMM:
5203 case AArch64::LDNT1W_4Z_IMM_PSEUDO:
5204 case AArch64::LDNT1W_4Z_IMM:
5205 case AArch64::LDNT1W_4Z_STRIDED_IMM:
5206 case AArch64::LDNT1D_4Z_IMM_PSEUDO:
5207 case AArch64::LDNT1D_4Z_IMM:
5208 case AArch64::LDNT1D_4Z_STRIDED_IMM:
5209 case AArch64::STNT1B_4Z_IMM:
5210 case AArch64::STNT1B_4Z_STRIDED_IMM:
5211 case AArch64::STNT1H_4Z_IMM:
5212 case AArch64::STNT1H_4Z_STRIDED_IMM:
5213 case AArch64::STNT1W_4Z_IMM:
5214 case AArch64::STNT1W_4Z_STRIDED_IMM:
5215 case AArch64::STNT1D_4Z_IMM:
5216 case AArch64::STNT1D_4Z_STRIDED_IMM:
5217 case AArch64::ST1B_4Z_IMM_PSEUDO:
5218 case AArch64::ST1H_4Z_IMM_PSEUDO:
5219 case AArch64::ST1W_4Z_IMM_PSEUDO:
5220 case AArch64::ST1D_4Z_IMM_PSEUDO:
5221 case AArch64::STNT1B_4Z_IMM_PSEUDO:
5222 case AArch64::STNT1H_4Z_IMM_PSEUDO:
5223 case AArch64::STNT1W_4Z_IMM_PSEUDO:
5224 case AArch64::STNT1D_4Z_IMM_PSEUDO:
5225 Scale = Width = TypeSize::getScalable(16 * 4);
5226 MinOffset = -8;
5227 MaxOffset = 7;
5228 break;
5229 case AArch64::LD1B_H_IMM:
5230 case AArch64::LD1SB_H_IMM:
5231 case AArch64::LD1H_S_IMM:
5232 case AArch64::LD1SH_S_IMM:
5233 case AArch64::LD1W_D_IMM:
5234 case AArch64::LD1SW_D_IMM:
5235 case AArch64::ST1B_H_IMM:
5236 case AArch64::ST1H_S_IMM:
5237 case AArch64::ST1W_D_IMM:
5238 case AArch64::LDNF1B_H_IMM:
5239 case AArch64::LDNF1SB_H_IMM:
5240 case AArch64::LDNF1H_S_IMM:
5241 case AArch64::LDNF1SH_S_IMM:
5242 case AArch64::LDNF1W_D_IMM:
5243 case AArch64::LDNF1SW_D_IMM:
5244 // A half vector worth of data
5245 // Width = mbytes * elements
5246 Scale = Width = TypeSize::getScalable(8);
5247 MinOffset = -8;
5248 MaxOffset = 7;
5249 break;
5250 case AArch64::LD1B_S_IMM:
5251 case AArch64::LD1SB_S_IMM:
5252 case AArch64::LD1H_D_IMM:
5253 case AArch64::LD1SH_D_IMM:
5254 case AArch64::ST1B_S_IMM:
5255 case AArch64::ST1H_D_IMM:
5256 case AArch64::LDNF1B_S_IMM:
5257 case AArch64::LDNF1SB_S_IMM:
5258 case AArch64::LDNF1H_D_IMM:
5259 case AArch64::LDNF1SH_D_IMM:
5260 // A quarter vector worth of data
5261 // Width = mbytes * elements
5262 Scale = Width = TypeSize::getScalable(4);
5263 MinOffset = -8;
5264 MaxOffset = 7;
5265 break;
5266 case AArch64::LD1B_D_IMM:
5267 case AArch64::LD1SB_D_IMM:
5268 case AArch64::ST1B_D_IMM:
5269 case AArch64::LDNF1B_D_IMM:
5270 case AArch64::LDNF1SB_D_IMM:
5271 // A eighth vector worth of data
5272 // Width = mbytes * elements
5273 Scale = Width = TypeSize::getScalable(2);
5274 MinOffset = -8;
5275 MaxOffset = 7;
5276 break;
5277 case AArch64::ST2Gi:
5278 case AArch64::ST2GPreIndex:
5279 case AArch64::ST2GPostIndex:
5280 case AArch64::STZ2Gi:
5281 case AArch64::STZ2GPreIndex:
5282 case AArch64::STZ2GPostIndex:
5283 Scale = TypeSize::getFixed(16);
5284 Width = TypeSize::getFixed(32);
5285 MinOffset = -256;
5286 MaxOffset = 255;
5287 break;
5288 case AArch64::STGPi:
5289 case AArch64::STGPpost:
5290 case AArch64::STGPpre:
5291 Scale = Width = TypeSize::getFixed(16);
5292 MinOffset = -64;
5293 MaxOffset = 63;
5294 break;
5295 case AArch64::LD1RB_IMM:
5296 case AArch64::LD1RB_H_IMM:
5297 case AArch64::LD1RB_S_IMM:
5298 case AArch64::LD1RB_D_IMM:
5299 case AArch64::LD1RSB_H_IMM:
5300 case AArch64::LD1RSB_S_IMM:
5301 case AArch64::LD1RSB_D_IMM:
5302 Scale = Width = TypeSize::getFixed(1);
5303 MinOffset = 0;
5304 MaxOffset = 63;
5305 break;
5306 case AArch64::LD1RH_IMM:
5307 case AArch64::LD1RH_S_IMM:
5308 case AArch64::LD1RH_D_IMM:
5309 case AArch64::LD1RSH_S_IMM:
5310 case AArch64::LD1RSH_D_IMM:
5311 Scale = Width = TypeSize::getFixed(2);
5312 MinOffset = 0;
5313 MaxOffset = 63;
5314 break;
5315 case AArch64::LD1RW_IMM:
5316 case AArch64::LD1RW_D_IMM:
5317 case AArch64::LD1RSW_IMM:
5318 Scale = Width = TypeSize::getFixed(4);
5319 MinOffset = 0;
5320 MaxOffset = 63;
5321 break;
5322 case AArch64::LD1RD_IMM:
5323 Scale = Width = TypeSize::getFixed(8);
5324 MinOffset = 0;
5325 MaxOffset = 63;
5326 break;
5327 }
5328
5329 return true;
5330}
5331
5332// Scaling factor for unscaled load or store.
5334 switch (Opc) {
5335 default:
5336 llvm_unreachable("Opcode has unknown scale!");
5337 case AArch64::LDRBui:
5338 case AArch64::LDRBBui:
5339 case AArch64::LDURBBi:
5340 case AArch64::LDRSBWui:
5341 case AArch64::LDURSBWi:
5342 case AArch64::STRBui:
5343 case AArch64::STRBBui:
5344 case AArch64::STURBBi:
5345 return 1;
5346 case AArch64::LDRHui:
5347 case AArch64::LDRHHui:
5348 case AArch64::LDURHHi:
5349 case AArch64::LDRSHWui:
5350 case AArch64::LDURSHWi:
5351 case AArch64::STRHui:
5352 case AArch64::STRHHui:
5353 case AArch64::STURHHi:
5354 return 2;
5355 case AArch64::LDRSui:
5356 case AArch64::LDURSi:
5357 case AArch64::LDRSpre:
5358 case AArch64::LDRSWui:
5359 case AArch64::LDURSWi:
5360 case AArch64::LDRSWpre:
5361 case AArch64::LDRWpre:
5362 case AArch64::LDRWui:
5363 case AArch64::LDURWi:
5364 case AArch64::STRSui:
5365 case AArch64::STURSi:
5366 case AArch64::STRSpre:
5367 case AArch64::STRWui:
5368 case AArch64::STURWi:
5369 case AArch64::STRWpre:
5370 case AArch64::LDPSi:
5371 case AArch64::LDPSWi:
5372 case AArch64::LDPWi:
5373 case AArch64::STPSi:
5374 case AArch64::STPWi:
5375 return 4;
5376 case AArch64::LDRDui:
5377 case AArch64::LDURDi:
5378 case AArch64::LDRDpre:
5379 case AArch64::LDRXui:
5380 case AArch64::LDURXi:
5381 case AArch64::LDRXpre:
5382 case AArch64::STRDui:
5383 case AArch64::STURDi:
5384 case AArch64::STRDpre:
5385 case AArch64::STRXui:
5386 case AArch64::STURXi:
5387 case AArch64::STRXpre:
5388 case AArch64::LDPDi:
5389 case AArch64::LDPXi:
5390 case AArch64::STPDi:
5391 case AArch64::STPXi:
5392 return 8;
5393 case AArch64::LDRQui:
5394 case AArch64::LDURQi:
5395 case AArch64::STRQui:
5396 case AArch64::STURQi:
5397 case AArch64::STRQpre:
5398 case AArch64::LDPQi:
5399 case AArch64::LDRQpre:
5400 case AArch64::STPQi:
5401 case AArch64::STGi:
5402 case AArch64::STZGi:
5403 case AArch64::ST2Gi:
5404 case AArch64::STZ2Gi:
5405 case AArch64::STGPi:
5406 return 16;
5407 }
5408}
5409
5411 switch (MI.getOpcode()) {
5412 default:
5413 return false;
5414 case AArch64::LDRWpre:
5415 case AArch64::LDRXpre:
5416 case AArch64::LDRSWpre:
5417 case AArch64::LDRSpre:
5418 case AArch64::LDRDpre:
5419 case AArch64::LDRQpre:
5420 return true;
5421 }
5422}
5423
5425 switch (MI.getOpcode()) {
5426 default:
5427 return false;
5428 case AArch64::STRWpre:
5429 case AArch64::STRXpre:
5430 case AArch64::STRSpre:
5431 case AArch64::STRDpre:
5432 case AArch64::STRQpre:
5433 return true;
5434 }
5435}
5436
5438 return isPreLd(MI) || isPreSt(MI);
5439}
5440
5442 switch (MI.getOpcode()) {
5443 default:
5444 return false;
5445 case AArch64::LDURBBi:
5446 case AArch64::LDURHHi:
5447 case AArch64::LDURWi:
5448 case AArch64::LDRBBui:
5449 case AArch64::LDRHHui:
5450 case AArch64::LDRWui:
5451 case AArch64::LDRBBroX:
5452 case AArch64::LDRHHroX:
5453 case AArch64::LDRWroX:
5454 case AArch64::LDRBBroW:
5455 case AArch64::LDRHHroW:
5456 case AArch64::LDRWroW:
5457 return true;
5458 }
5459}
5460
5462 switch (MI.getOpcode()) {
5463 default:
5464 return false;
5465 case AArch64::LDURSBWi:
5466 case AArch64::LDURSHWi:
5467 case AArch64::LDURSBXi:
5468 case AArch64::LDURSHXi:
5469 case AArch64::LDURSWi:
5470 case AArch64::LDRSBWui:
5471 case AArch64::LDRSHWui:
5472 case AArch64::LDRSBXui:
5473 case AArch64::LDRSHXui:
5474 case AArch64::LDRSWui:
5475 case AArch64::LDRSBWroX:
5476 case AArch64::LDRSHWroX:
5477 case AArch64::LDRSBXroX:
5478 case AArch64::LDRSHXroX:
5479 case AArch64::LDRSWroX:
5480 case AArch64::LDRSBWroW:
5481 case AArch64::LDRSHWroW:
5482 case AArch64::LDRSBXroW:
5483 case AArch64::LDRSHXroW:
5484 case AArch64::LDRSWroW:
5485 return true;
5486 }
5487}
5488
5490 switch (MI.getOpcode()) {
5491 default:
5492 return false;
5493 case AArch64::LDPSi:
5494 case AArch64::LDPSWi:
5495 case AArch64::LDPDi:
5496 case AArch64::LDPQi:
5497 case AArch64::LDPWi:
5498 case AArch64::LDPXi:
5499 case AArch64::STPSi:
5500 case AArch64::STPDi:
5501 case AArch64::STPQi:
5502 case AArch64::STPWi:
5503 case AArch64::STPXi:
5504 case AArch64::STGPi:
5505 return true;
5506 }
5507}
5508
5510 assert(MI.mayLoadOrStore() && "Load or store instruction expected");
5511 unsigned Idx =
5513 : 1;
5514 return MI.getOperand(Idx);
5515}
5516
5517const MachineOperand &
5519 assert(MI.mayLoadOrStore() && "Load or store instruction expected");
5520 unsigned Idx =
5522 : 2;
5523 return MI.getOperand(Idx);
5524}
5525
5526const MachineOperand &
5528 switch (MI.getOpcode()) {
5529 default:
5530 llvm_unreachable("Unexpected opcode");
5531 case AArch64::LDRBroX:
5532 case AArch64::LDRBBroX:
5533 case AArch64::LDRSBXroX:
5534 case AArch64::LDRSBWroX:
5535 case AArch64::LDRHroX:
5536 case AArch64::LDRHHroX:
5537 case AArch64::LDRSHXroX:
5538 case AArch64::LDRSHWroX:
5539 case AArch64::LDRWroX:
5540 case AArch64::LDRSroX:
5541 case AArch64::LDRSWroX:
5542 case AArch64::LDRDroX:
5543 case AArch64::LDRXroX:
5544 case AArch64::LDRQroX:
5545 return MI.getOperand(4);
5546 }
5547}
5548
5550 Register Reg) {
5551 if (MI.getParent() == nullptr)
5552 return nullptr;
5553 const MachineFunction *MF = MI.getParent()->getParent();
5554 return MF ? MF->getRegInfo().getRegClassOrNull(Reg) : nullptr;
5555}
5556
5558 auto IsHFPR = [&](const MachineOperand &Op) {
5559 if (!Op.isReg())
5560 return false;
5561 auto Reg = Op.getReg();
5562 if (Reg.isPhysical())
5563 return AArch64::FPR16RegClass.contains(Reg);
5564 const TargetRegisterClass *TRC = ::getRegClass(MI, Reg);
5565 return TRC == &AArch64::FPR16RegClass ||
5566 TRC == &AArch64::FPR16_loRegClass;
5567 };
5568 return llvm::any_of(MI.operands(), IsHFPR);
5569}
5570
5572 auto IsQFPR = [&](const MachineOperand &Op) {
5573 if (!Op.isReg())
5574 return false;
5575 auto Reg = Op.getReg();
5576 if (Reg.isPhysical())
5577 return AArch64::FPR128RegClass.contains(Reg);
5578 const TargetRegisterClass *TRC = ::getRegClass(MI, Reg);
5579 return TRC == &AArch64::FPR128RegClass ||
5580 TRC == &AArch64::FPR128_loRegClass;
5581 };
5582 return llvm::any_of(MI.operands(), IsQFPR);
5583}
5584
5586 switch (MI.getOpcode()) {
5587 case AArch64::BRK:
5588 case AArch64::HLT:
5589 case AArch64::PACIASP:
5590 case AArch64::PACIBSP:
5591 // Implicit BTI behavior.
5592 return true;
5593 case AArch64::PAUTH_PROLOGUE:
5594 // PAUTH_PROLOGUE expands to PACI(A|B)SP.
5595 return true;
5596 case AArch64::HINT: {
5597 unsigned Imm = MI.getOperand(0).getImm();
5598 // Explicit BTI instruction.
5599 if (Imm == 32 || Imm == 34 || Imm == 36 || Imm == 38)
5600 return true;
5601 // PACI(A|B)SP instructions.
5602 if (Imm == 25 || Imm == 27)
5603 return true;
5604 return false;
5605 }
5606 default:
5607 return false;
5608 }
5609}
5610
5612 if (Reg == 0)
5613 return false;
5614 assert(Reg.isPhysical() && "Expected physical register in isFpOrNEON");
5615 return AArch64::FPR128RegClass.contains(Reg) ||
5616 AArch64::FPR64RegClass.contains(Reg) ||
5617 AArch64::FPR32RegClass.contains(Reg) ||
5618 AArch64::FPR16RegClass.contains(Reg) ||
5619 AArch64::FPR8RegClass.contains(Reg);
5620}
5621
5623 auto IsFPR = [&](const MachineOperand &Op) {
5624 if (!Op.isReg())
5625 return false;
5626 auto Reg = Op.getReg();
5627 if (Reg.isPhysical())
5628 return isFpOrNEON(Reg);
5629
5630 const TargetRegisterClass *TRC = ::getRegClass(MI, Reg);
5631 return TRC == &AArch64::FPR128RegClass ||
5632 TRC == &AArch64::FPR128_loRegClass ||
5633 TRC == &AArch64::FPR64RegClass ||
5634 TRC == &AArch64::FPR64_loRegClass ||
5635 TRC == &AArch64::FPR32RegClass || TRC == &AArch64::FPR16RegClass ||
5636 TRC == &AArch64::FPR8RegClass;
5637 };
5638 return llvm::any_of(MI.operands(), IsFPR);
5639}
5640
5641// Scale the unscaled offsets. Returns false if the unscaled offset can't be
5642// scaled.
5643static bool scaleOffset(unsigned Opc, int64_t &Offset) {
5645
5646 // If the byte-offset isn't a multiple of the stride, we can't scale this
5647 // offset.
5648 if (Offset % Scale != 0)
5649 return false;
5650
5651 // Convert the byte-offset used by unscaled into an "element" offset used
5652 // by the scaled pair load/store instructions.
5653 Offset /= Scale;
5654 return true;
5655}
5656
5657static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc) {
5658 if (FirstOpc == SecondOpc)
5659 return true;
5660 // We can also pair sign-ext and zero-ext instructions.
5661 switch (FirstOpc) {
5662 default:
5663 return false;
5664 case AArch64::STRSui:
5665 case AArch64::STURSi:
5666 return SecondOpc == AArch64::STRSui || SecondOpc == AArch64::STURSi;
5667 case AArch64::STRDui:
5668 case AArch64::STURDi:
5669 return SecondOpc == AArch64::STRDui || SecondOpc == AArch64::STURDi;
5670 case AArch64::STRQui:
5671 case AArch64::STURQi:
5672 return SecondOpc == AArch64::STRQui || SecondOpc == AArch64::STURQi;
5673 case AArch64::STRWui:
5674 case AArch64::STURWi:
5675 return SecondOpc == AArch64::STRWui || SecondOpc == AArch64::STURWi;
5676 case AArch64::STRXui:
5677 case AArch64::STURXi:
5678 return SecondOpc == AArch64::STRXui || SecondOpc == AArch64::STURXi;
5679 case AArch64::LDRSui:
5680 case AArch64::LDURSi:
5681 return SecondOpc == AArch64::LDRSui || SecondOpc == AArch64::LDURSi;
5682 case AArch64::LDRDui:
5683 case AArch64::LDURDi:
5684 return SecondOpc == AArch64::LDRDui || SecondOpc == AArch64::LDURDi;
5685 case AArch64::LDRQui:
5686 case AArch64::LDURQi:
5687 return SecondOpc == AArch64::LDRQui || SecondOpc == AArch64::LDURQi;
5688 case AArch64::LDRWui:
5689 case AArch64::LDURWi:
5690 return SecondOpc == AArch64::LDRSWui || SecondOpc == AArch64::LDURSWi;
5691 case AArch64::LDRSWui:
5692 case AArch64::LDURSWi:
5693 return SecondOpc == AArch64::LDRWui || SecondOpc == AArch64::LDURWi;
5694 case AArch64::LDRXui:
5695 case AArch64::LDURXi:
5696 return SecondOpc == AArch64::LDRXui || SecondOpc == AArch64::LDURXi;
5697 }
5698 // These instructions can't be paired based on their opcodes.
5699 return false;
5700}
5701
5702static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1,
5703 int64_t Offset1, unsigned Opcode1, int FI2,
5704 int64_t Offset2, unsigned Opcode2) {
5705 // Accesses through fixed stack object frame indices may access a different
5706 // fixed stack slot. Check that the object offsets + offsets match.
5707 if (MFI.isFixedObjectIndex(FI1) && MFI.isFixedObjectIndex(FI2)) {
5708 int64_t ObjectOffset1 = MFI.getObjectOffset(FI1);
5709 int64_t ObjectOffset2 = MFI.getObjectOffset(FI2);
5710 assert(ObjectOffset1 <= ObjectOffset2 && "Object offsets are not ordered.");
5711 // Convert to scaled object offsets.
5712 int Scale1 = AArch64InstrInfo::getMemScale(Opcode1);
5713 if (ObjectOffset1 % Scale1 != 0)
5714 return false;
5715 ObjectOffset1 /= Scale1;
5716 int Scale2 = AArch64InstrInfo::getMemScale(Opcode2);
5717 if (ObjectOffset2 % Scale2 != 0)
5718 return false;
5719 ObjectOffset2 /= Scale2;
5720 ObjectOffset1 += Offset1;
5721 ObjectOffset2 += Offset2;
5722 return ObjectOffset1 + 1 == ObjectOffset2;
5723 }
5724
5725 return FI1 == FI2;
5726}
5727
5728/// Detect opportunities for ldp/stp formation.
5729///
5730/// Only called for LdSt for which getMemOperandWithOffset returns true.
5732 ArrayRef<const MachineOperand *> BaseOps1, int64_t OpOffset1,
5733 bool OffsetIsScalable1, ArrayRef<const MachineOperand *> BaseOps2,
5734 int64_t OpOffset2, bool OffsetIsScalable2, unsigned ClusterSize,
5735 unsigned NumBytes) const {
5736 assert(BaseOps1.size() == 1 && BaseOps2.size() == 1);
5737 const MachineOperand &BaseOp1 = *BaseOps1.front();
5738 const MachineOperand &BaseOp2 = *BaseOps2.front();
5739 const MachineInstr &FirstLdSt = *BaseOp1.getParent();
5740 const MachineInstr &SecondLdSt = *BaseOp2.getParent();
5741 if (BaseOp1.getType() != BaseOp2.getType())
5742 return false;
5743
5744 assert((BaseOp1.isReg() || BaseOp1.isFI()) &&
5745 "Only base registers and frame indices are supported.");
5746
5747 // Check for both base regs and base FI.
5748 if (BaseOp1.isReg() && BaseOp1.getReg() != BaseOp2.getReg())
5749 return false;
5750
5751 // Only cluster up to a single pair.
5752 if (ClusterSize > 2)
5753 return false;
5754
5755 if (!isPairableLdStInst(FirstLdSt) || !isPairableLdStInst(SecondLdSt))
5756 return false;
5757
5758 // Can we pair these instructions based on their opcodes?
5759 unsigned FirstOpc = FirstLdSt.getOpcode();
5760 unsigned SecondOpc = SecondLdSt.getOpcode();
5761 if (!canPairLdStOpc(FirstOpc, SecondOpc))
5762 return false;
5763
5764 // Can't merge volatiles or load/stores that have a hint to avoid pair
5765 // formation, for example.
5766 if (!isCandidateToMergeOrPair(FirstLdSt) ||
5767 !isCandidateToMergeOrPair(SecondLdSt))
5768 return false;
5769
5770 // isCandidateToMergeOrPair guarantees that operand 2 is an immediate.
5771 int64_t Offset1 = FirstLdSt.getOperand(2).getImm();
5772 if (hasUnscaledLdStOffset(FirstOpc) && !scaleOffset(FirstOpc, Offset1))
5773 return false;
5774
5775 int64_t Offset2 = SecondLdSt.getOperand(2).getImm();
5776 if (hasUnscaledLdStOffset(SecondOpc) && !scaleOffset(SecondOpc, Offset2))
5777 return false;
5778
5779 // Pairwise instructions have a 7-bit signed offset field.
5780 if (Offset1 > 63 || Offset1 < -64)
5781 return false;
5782
5783 // The caller should already have ordered First/SecondLdSt by offset.
5784 // Note: except for non-equal frame index bases
5785 if (BaseOp1.isFI()) {
5786 assert((!BaseOp1.isIdenticalTo(BaseOp2) || Offset1 <= Offset2) &&
5787 "Caller should have ordered offsets.");
5788
5789 const MachineFrameInfo &MFI =
5790 FirstLdSt.getParent()->getParent()->getFrameInfo();
5791 return shouldClusterFI(MFI, BaseOp1.getIndex(), Offset1, FirstOpc,
5792 BaseOp2.getIndex(), Offset2, SecondOpc);
5793 }
5794
5795 assert(Offset1 <= Offset2 && "Caller should have ordered offsets.");
5796
5797 return Offset1 + 1 == Offset2;
5798}
5799
5801 MCRegister Reg, unsigned SubIdx,
5802 RegState State,
5803 const TargetRegisterInfo *TRI) {
5804 if (!SubIdx)
5805 return MIB.addReg(Reg, State);
5806
5807 if (Reg.isPhysical())
5808 return MIB.addReg(TRI->getSubReg(Reg, SubIdx), State);
5809 return MIB.addReg(Reg, State, SubIdx);
5810}
5811
5814 const DebugLoc &DL, MCRegister DestReg,
5815 MCRegister SrcReg, bool KillSrc,
5816 ArrayRef<unsigned> Indices) const {
5817 assert(Subtarget.hasNEON() && "Unexpected register copy without NEON");
5819 uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
5820 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
5821 unsigned NumRegs = Indices.size();
5822 MCRegister DestSubReg = TRI->getSubReg(DestReg, Indices[0]);
5823 assert(!AArch64::PNRRegClass.contains(DestSubReg) &&
5824 "Unexpected predicate tuple copy");
5825 unsigned MaxRegs = AArch64::PPRRegClass.contains(DestSubReg) ? 15 : 31;
5826
5827 int SubReg = 0, End = NumRegs, Incr = 1;
5828 // Copy in reverse if a forward copy will clobber the tuple
5829 if (((DestEncoding - SrcEncoding) & MaxRegs) < NumRegs) {
5830 SubReg = NumRegs - 1;
5831 End = -1;
5832 Incr = -1;
5833 }
5834
5835 for (; SubReg != End; SubReg += Incr) {
5836 DestSubReg = TRI->getSubReg(DestReg, Indices[SubReg]);
5837 MCRegister SrcSubReg = TRI->getSubReg(SrcReg, Indices[SubReg]);
5838 copyPhysRegImpl(MBB, I, DL, DestSubReg, SrcSubReg, KillSrc);
5839 }
5840}
5841
5844 const DebugLoc &DL, MCRegister DestReg,
5845 MCRegister SrcReg, bool KillSrc,
5846 unsigned Opcode, unsigned ZeroReg,
5847 llvm::ArrayRef<unsigned> Indices) const {
5849 unsigned NumRegs = Indices.size();
5850
5851#ifndef NDEBUG
5852 uint16_t DestEncoding = TRI->getEncodingValue(DestReg);
5853 uint16_t SrcEncoding = TRI->getEncodingValue(SrcReg);
5854 assert(DestEncoding % NumRegs == 0 && SrcEncoding % NumRegs == 0 &&
5855 "GPR reg sequences should not be able to overlap");
5856#endif
5857
5858 for (unsigned SubReg = 0; SubReg != NumRegs; ++SubReg) {
5859 const MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opcode));
5860 AddSubReg(MIB, DestReg, Indices[SubReg], RegState::Define, TRI);
5861 MIB.addReg(ZeroReg);
5862 AddSubReg(MIB, SrcReg, Indices[SubReg], getKillRegState(KillSrc), TRI);
5863 MIB.addImm(0);
5864 }
5865}
5866
5867/// Returns true if the instruction at I is in a streaming call site region,
5868/// within a single basic block.
5869/// A "call site streaming region" starts after smstart and ends at smstop
5870/// around a call to a streaming function. This walks backward from I.
5873 MachineFunction &MF = *MBB.getParent();
5875 if (!AFI->hasStreamingModeChanges())
5876 return false;
5877 // Walk backwards to find smstart/smstop
5878 for (MachineInstr &MI : reverse(make_range(MBB.begin(), I))) {
5879 unsigned Opc = MI.getOpcode();
5880 if (Opc == AArch64::MSRpstatesvcrImm1 || Opc == AArch64::MSRpstatePseudo) {
5881 // Check if this is SM change (not ZA)
5882 int64_t PState = MI.getOperand(0).getImm();
5883 if (PState == AArch64SVCR::SVCRSM || PState == AArch64SVCR::SVCRSMZA) {
5884 // Operand 1 is 1 for start, 0 for stop
5885 return MI.getOperand(1).getImm() == 1;
5886 }
5887 }
5888 }
5889 return false;
5890}
5891
5892/// Returns true if in a streaming call site region without SME-FA64.
5893static bool mustAvoidNeonAtMBBI(const AArch64Subtarget &Subtarget,
5896 return !Subtarget.hasSMEFA64() && isInStreamingCallSiteRegion(MBB, I);
5897}
5898
5901 const DebugLoc &DL, Register DestReg,
5902 Register SrcReg, bool KillSrc,
5903 bool RenamableDest,
5904 bool RenamableSrc) const {
5905 if (AArch64::GPR32spRegClass.contains(DestReg) &&
5906 AArch64::GPR32spRegClass.contains(SrcReg)) {
5907 if (DestReg == AArch64::WSP || SrcReg == AArch64::WSP) {
5908 // If either operand is WSP, expand to ADD #0.
5909 if (Subtarget.hasZeroCycleRegMoveGPR64() &&
5910 !Subtarget.hasZeroCycleRegMoveGPR32()) {
5911 // Cyclone recognizes "ADD Xd, Xn, #0" as a zero-cycle register move.
5912 MCRegister DestRegX = RI.getMatchingSuperReg(DestReg, AArch64::sub_32,
5913 &AArch64::GPR64spRegClass);
5914 MCRegister SrcRegX = RI.getMatchingSuperReg(SrcReg, AArch64::sub_32,
5915 &AArch64::GPR64spRegClass);
5916 // This instruction is reading and writing X registers. This may upset
5917 // the register scavenger and machine verifier, so we need to indicate
5918 // that we are reading an undefined value from SrcRegX, but a proper
5919 // value from SrcReg.
5920 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestRegX)
5921 .addReg(SrcRegX, RegState::Undef)
5922 .addImm(0)
5924 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
5925 ++NumZCRegMoveInstrsGPR;
5926 } else {
5927 BuildMI(MBB, I, DL, get(AArch64::ADDWri), DestReg)
5928 .addReg(SrcReg, getKillRegState(KillSrc))
5929 .addImm(0)
5931 if (Subtarget.hasZeroCycleRegMoveGPR32())
5932 ++NumZCRegMoveInstrsGPR;
5933 }
5934 } else if (Subtarget.hasZeroCycleRegMoveGPR64() &&
5935 !Subtarget.hasZeroCycleRegMoveGPR32()) {
5936 // Cyclone recognizes "ORR Xd, XZR, Xm" as a zero-cycle register move.
5937 MCRegister DestRegX = RI.getMatchingSuperReg(DestReg, AArch64::sub_32,
5938 &AArch64::GPR64spRegClass);
5939 assert(DestRegX.isValid() && "Destination super-reg not valid");
5940 MCRegister SrcRegX = RI.getMatchingSuperReg(SrcReg, AArch64::sub_32,
5941 &AArch64::GPR64spRegClass);
5942 assert(SrcRegX.isValid() && "Source super-reg not valid");
5943 // This instruction is reading and writing X registers. This may upset
5944 // the register scavenger and machine verifier, so we need to indicate
5945 // that we are reading an undefined value from SrcRegX, but a proper
5946 // value from SrcReg.
5947 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestRegX)
5948 .addReg(AArch64::XZR)
5949 .addReg(SrcRegX, RegState::Undef)
5950 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
5951 ++NumZCRegMoveInstrsGPR;
5952 } else {
5953 // Otherwise, expand to ORR WZR.
5954 BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg)
5955 .addReg(AArch64::WZR)
5956 .addReg(SrcReg, getKillRegState(KillSrc));
5957 if (Subtarget.hasZeroCycleRegMoveGPR32())
5958 ++NumZCRegMoveInstrsGPR;
5959 }
5960 return;
5961 }
5962
5963 // GPR32 zeroing
5964 if (AArch64::GPR32spRegClass.contains(DestReg) && SrcReg == AArch64::WZR) {
5965 if (Subtarget.hasZeroCycleZeroingGPR64() &&
5966 !Subtarget.hasZeroCycleZeroingGPR32()) {
5967 MCRegister DestRegX = RI.getMatchingSuperReg(DestReg, AArch64::sub_32,
5968 &AArch64::GPR64spRegClass);
5969 assert(DestRegX.isValid() && "Destination super-reg not valid");
5970 BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestRegX)
5971 .addImm(0)
5973 ++NumZCZeroingInstrsGPR;
5974 } else if (Subtarget.hasZeroCycleZeroingGPR32()) {
5975 BuildMI(MBB, I, DL, get(AArch64::MOVZWi), DestReg)
5976 .addImm(0)
5978 ++NumZCZeroingInstrsGPR;
5979 } else {
5980 BuildMI(MBB, I, DL, get(AArch64::ORRWrr), DestReg)
5981 .addReg(AArch64::WZR)
5982 .addReg(AArch64::WZR);
5983 }
5984 return;
5985 }
5986
5987 if (AArch64::GPR64spRegClass.contains(DestReg) &&
5988 AArch64::GPR64spRegClass.contains(SrcReg)) {
5989 if (DestReg == AArch64::SP || SrcReg == AArch64::SP) {
5990 // If either operand is SP, expand to ADD #0.
5991 BuildMI(MBB, I, DL, get(AArch64::ADDXri), DestReg)
5992 .addReg(SrcReg, getKillRegState(KillSrc))
5993 .addImm(0)
5995 if (Subtarget.hasZeroCycleRegMoveGPR64())
5996 ++NumZCRegMoveInstrsGPR;
5997 } else {
5998 // Otherwise, expand to ORR XZR.
5999 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg)
6000 .addReg(AArch64::XZR)
6001 .addReg(SrcReg, getKillRegState(KillSrc));
6002 if (Subtarget.hasZeroCycleRegMoveGPR64())
6003 ++NumZCRegMoveInstrsGPR;
6004 }
6005 return;
6006 }
6007
6008 // GPR64 zeroing
6009 if (AArch64::GPR64spRegClass.contains(DestReg) && SrcReg == AArch64::XZR) {
6010 if (Subtarget.hasZeroCycleZeroingGPR64()) {
6011 BuildMI(MBB, I, DL, get(AArch64::MOVZXi), DestReg)
6012 .addImm(0)
6014 ++NumZCZeroingInstrsGPR;
6015 } else {
6016 BuildMI(MBB, I, DL, get(AArch64::ORRXrr), DestReg)
6017 .addReg(AArch64::XZR)
6018 .addReg(AArch64::XZR);
6019 }
6020 return;
6021 }
6022
6023 // Copy a Predicate register by ORRing with itself.
6024 if (AArch64::PPRRegClass.contains(DestReg) &&
6025 AArch64::PPRRegClass.contains(SrcReg)) {
6026 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6027 "Unexpected SVE register.");
6028 BuildMI(MBB, I, DL, get(AArch64::ORR_PPzPP), DestReg)
6029 .addReg(SrcReg) // Pg
6030 .addReg(SrcReg)
6031 .addReg(SrcReg, getKillRegState(KillSrc));
6032 return;
6033 }
6034
6035 // Copy a predicate-as-counter register by ORRing with itself as if it
6036 // were a regular predicate (mask) register.
6037 bool DestIsPNR = AArch64::PNRRegClass.contains(DestReg);
6038 bool SrcIsPNR = AArch64::PNRRegClass.contains(SrcReg);
6039 if (DestIsPNR || SrcIsPNR) {
6040 auto ToPPR = [](MCRegister R) -> MCRegister {
6041 return (R - AArch64::PN0) + AArch64::P0;
6042 };
6043 MCRegister PPRSrcReg = SrcIsPNR ? ToPPR(SrcReg) : SrcReg.asMCReg();
6044 MCRegister PPRDestReg = DestIsPNR ? ToPPR(DestReg) : DestReg.asMCReg();
6045
6046 if (PPRSrcReg != PPRDestReg) {
6047 auto NewMI = BuildMI(MBB, I, DL, get(AArch64::ORR_PPzPP), PPRDestReg)
6048 .addReg(PPRSrcReg) // Pg
6049 .addReg(PPRSrcReg)
6050 .addReg(PPRSrcReg, getKillRegState(KillSrc));
6051 if (DestIsPNR)
6052 NewMI.addDef(DestReg, RegState::Implicit);
6053 }
6054 return;
6055 }
6056
6057 // Copy a predicate register pair by copying the individual sub-registers.
6058 if (AArch64::PPR2RegClass.contains(DestReg) &&
6059 AArch64::PPR2RegClass.contains(SrcReg)) {
6060 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6061 "Unexpected SVE predicate register.");
6062 static const unsigned Indices[] = {AArch64::psub0, AArch64::psub1};
6063 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6064 return;
6065 }
6066
6067 // Copy a Z register by ORRing with itself.
6068 if (AArch64::ZPRRegClass.contains(DestReg) &&
6069 AArch64::ZPRRegClass.contains(SrcReg)) {
6070 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6071 "Unexpected SVE register.");
6072 BuildMI(MBB, I, DL, get(AArch64::ORR_ZZZ), DestReg)
6073 .addReg(SrcReg)
6074 .addReg(SrcReg, getKillRegState(KillSrc));
6075 return;
6076 }
6077
6078 // Copy a Z register pair by copying the individual sub-registers.
6079 if ((AArch64::ZPR2RegClass.contains(DestReg) ||
6080 AArch64::ZPR2StridedOrContiguousRegClass.contains(DestReg)) &&
6081 (AArch64::ZPR2RegClass.contains(SrcReg) ||
6082 AArch64::ZPR2StridedOrContiguousRegClass.contains(SrcReg))) {
6083 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6084 "Unexpected SVE register.");
6085 static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1};
6086 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6087 return;
6088 }
6089
6090 // Copy a Z register triple by copying the individual sub-registers.
6091 if (AArch64::ZPR3RegClass.contains(DestReg) &&
6092 AArch64::ZPR3RegClass.contains(SrcReg)) {
6093 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6094 "Unexpected SVE register.");
6095 static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
6096 AArch64::zsub2};
6097 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6098 return;
6099 }
6100
6101 // Copy a Z register quad by copying the individual sub-registers.
6102 if ((AArch64::ZPR4RegClass.contains(DestReg) ||
6103 AArch64::ZPR4StridedOrContiguousRegClass.contains(DestReg)) &&
6104 (AArch64::ZPR4RegClass.contains(SrcReg) ||
6105 AArch64::ZPR4StridedOrContiguousRegClass.contains(SrcReg))) {
6106 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6107 "Unexpected SVE register.");
6108 static const unsigned Indices[] = {AArch64::zsub0, AArch64::zsub1,
6109 AArch64::zsub2, AArch64::zsub3};
6110 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6111 return;
6112 }
6113
6114 // Copy a DDDD register quad by copying the individual sub-registers.
6115 if (AArch64::DDDDRegClass.contains(DestReg) &&
6116 AArch64::DDDDRegClass.contains(SrcReg)) {
6117 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
6118 AArch64::dsub2, AArch64::dsub3};
6119 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6120 return;
6121 }
6122
6123 // Copy a DDD register triple by copying the individual sub-registers.
6124 if (AArch64::DDDRegClass.contains(DestReg) &&
6125 AArch64::DDDRegClass.contains(SrcReg)) {
6126 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1,
6127 AArch64::dsub2};
6128 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6129 return;
6130 }
6131
6132 // Copy a DD register pair by copying the individual sub-registers.
6133 if (AArch64::DDRegClass.contains(DestReg) &&
6134 AArch64::DDRegClass.contains(SrcReg)) {
6135 static const unsigned Indices[] = {AArch64::dsub0, AArch64::dsub1};
6136 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6137 return;
6138 }
6139
6140 // Copy a QQQQ register quad by copying the individual sub-registers.
6141 if (AArch64::QQQQRegClass.contains(DestReg) &&
6142 AArch64::QQQQRegClass.contains(SrcReg)) {
6143 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
6144 AArch64::qsub2, AArch64::qsub3};
6145 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6146 return;
6147 }
6148
6149 // Copy a QQQ register triple by copying the individual sub-registers.
6150 if (AArch64::QQQRegClass.contains(DestReg) &&
6151 AArch64::QQQRegClass.contains(SrcReg)) {
6152 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1,
6153 AArch64::qsub2};
6154 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6155 return;
6156 }
6157
6158 // Copy a QQ register pair by copying the individual sub-registers.
6159 if (AArch64::QQRegClass.contains(DestReg) &&
6160 AArch64::QQRegClass.contains(SrcReg)) {
6161 static const unsigned Indices[] = {AArch64::qsub0, AArch64::qsub1};
6162 copyPhysRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, Indices);
6163 return;
6164 }
6165
6166 if (AArch64::XSeqPairsClassRegClass.contains(DestReg) &&
6167 AArch64::XSeqPairsClassRegClass.contains(SrcReg)) {
6168 static const unsigned Indices[] = {AArch64::sube64, AArch64::subo64};
6169 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRXrs,
6170 AArch64::XZR, Indices);
6171 return;
6172 }
6173
6174 if (AArch64::WSeqPairsClassRegClass.contains(DestReg) &&
6175 AArch64::WSeqPairsClassRegClass.contains(SrcReg)) {
6176 static const unsigned Indices[] = {AArch64::sube32, AArch64::subo32};
6177 copyGPRRegTuple(MBB, I, DL, DestReg, SrcReg, KillSrc, AArch64::ORRWrs,
6178 AArch64::WZR, Indices);
6179 return;
6180 }
6181
6182 if (AArch64::FPR128RegClass.contains(DestReg) &&
6183 AArch64::FPR128RegClass.contains(SrcReg)) {
6184 // In streaming regions, NEON is illegal but streaming-SVE is available.
6185 // Use SVE for copies if we're in a streaming region and SME is available.
6186 // With +sme-fa64, NEON is legal in streaming mode so we can use it.
6187 if ((Subtarget.isSVEorStreamingSVEAvailable() &&
6188 !Subtarget.isNeonAvailable()) ||
6189 mustAvoidNeonAtMBBI(Subtarget, MBB, I)) {
6190 BuildMI(MBB, I, DL, get(AArch64::ORR_ZZZ))
6191 .addReg(AArch64::Z0 + (DestReg - AArch64::Q0), RegState::Define)
6192 .addReg(AArch64::Z0 + (SrcReg - AArch64::Q0))
6193 .addReg(AArch64::Z0 + (SrcReg - AArch64::Q0));
6194 } else if (Subtarget.isNeonAvailable()) {
6195 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestReg)
6196 .addReg(SrcReg)
6197 .addReg(SrcReg, getKillRegState(KillSrc));
6198 if (Subtarget.hasZeroCycleRegMoveFPR128())
6199 ++NumZCRegMoveInstrsFPR;
6200 } else {
6201 BuildMI(MBB, I, DL, get(AArch64::STRQpre))
6202 .addReg(AArch64::SP, RegState::Define)
6203 .addReg(SrcReg, getKillRegState(KillSrc))
6204 .addReg(AArch64::SP)
6205 .addImm(-16);
6206 BuildMI(MBB, I, DL, get(AArch64::LDRQpost))
6207 .addReg(AArch64::SP, RegState::Define)
6208 .addReg(DestReg, RegState::Define)
6209 .addReg(AArch64::SP)
6210 .addImm(16);
6211 }
6212 return;
6213 }
6214
6215 if (AArch64::FPR64RegClass.contains(DestReg) &&
6216 AArch64::FPR64RegClass.contains(SrcReg)) {
6217 if (Subtarget.hasZeroCycleRegMoveFPR128() &&
6218 !Subtarget.hasZeroCycleRegMoveFPR64() &&
6219 !Subtarget.hasZeroCycleRegMoveFPR32() && Subtarget.isNeonAvailable() &&
6220 !mustAvoidNeonAtMBBI(Subtarget, MBB, I)) {
6221 MCRegister DestRegQ = RI.getMatchingSuperReg(DestReg, AArch64::dsub,
6222 &AArch64::FPR128RegClass);
6223 MCRegister SrcRegQ = RI.getMatchingSuperReg(SrcReg, AArch64::dsub,
6224 &AArch64::FPR128RegClass);
6225 // This instruction is reading and writing Q registers. This may upset
6226 // the register scavenger and machine verifier, so we need to indicate
6227 // that we are reading an undefined value from SrcRegQ, but a proper
6228 // value from SrcReg.
6229 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestRegQ)
6230 .addReg(SrcRegQ, RegState::Undef)
6231 .addReg(SrcRegQ, RegState::Undef)
6232 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6233 ++NumZCRegMoveInstrsFPR;
6234 } else {
6235 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestReg)
6236 .addReg(SrcReg, getKillRegState(KillSrc));
6237 if (Subtarget.hasZeroCycleRegMoveFPR64())
6238 ++NumZCRegMoveInstrsFPR;
6239 }
6240 return;
6241 }
6242
6243 if (AArch64::FPR32RegClass.contains(DestReg) &&
6244 AArch64::FPR32RegClass.contains(SrcReg)) {
6245 if (Subtarget.hasZeroCycleRegMoveFPR128() &&
6246 !Subtarget.hasZeroCycleRegMoveFPR64() &&
6247 !Subtarget.hasZeroCycleRegMoveFPR32() && Subtarget.isNeonAvailable() &&
6248 !mustAvoidNeonAtMBBI(Subtarget, MBB, I)) {
6249 MCRegister DestRegQ = RI.getMatchingSuperReg(DestReg, AArch64::ssub,
6250 &AArch64::FPR128RegClass);
6251 MCRegister SrcRegQ = RI.getMatchingSuperReg(SrcReg, AArch64::ssub,
6252 &AArch64::FPR128RegClass);
6253 // This instruction is reading and writing Q registers. This may upset
6254 // the register scavenger and machine verifier, so we need to indicate
6255 // that we are reading an undefined value from SrcRegQ, but a proper
6256 // value from SrcReg.
6257 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestRegQ)
6258 .addReg(SrcRegQ, RegState::Undef)
6259 .addReg(SrcRegQ, RegState::Undef)
6260 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6261 ++NumZCRegMoveInstrsFPR;
6262 } else if (Subtarget.hasZeroCycleRegMoveFPR64() &&
6263 !Subtarget.hasZeroCycleRegMoveFPR32()) {
6264 MCRegister DestRegD = RI.getMatchingSuperReg(DestReg, AArch64::ssub,
6265 &AArch64::FPR64RegClass);
6266 MCRegister SrcRegD = RI.getMatchingSuperReg(SrcReg, AArch64::ssub,
6267 &AArch64::FPR64RegClass);
6268 // This instruction is reading and writing D registers. This may upset
6269 // the register scavenger and machine verifier, so we need to indicate
6270 // that we are reading an undefined value from SrcRegD, but a proper
6271 // value from SrcReg.
6272 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestRegD)
6273 .addReg(SrcRegD, RegState::Undef)
6274 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6275 ++NumZCRegMoveInstrsFPR;
6276 } else {
6277 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
6278 .addReg(SrcReg, getKillRegState(KillSrc));
6279 if (Subtarget.hasZeroCycleRegMoveFPR32())
6280 ++NumZCRegMoveInstrsFPR;
6281 }
6282 return;
6283 }
6284
6285 if (AArch64::FPR16RegClass.contains(DestReg) &&
6286 AArch64::FPR16RegClass.contains(SrcReg)) {
6287 if (Subtarget.hasZeroCycleRegMoveFPR128() &&
6288 !Subtarget.hasZeroCycleRegMoveFPR64() &&
6289 !Subtarget.hasZeroCycleRegMoveFPR32() && Subtarget.isNeonAvailable() &&
6290 !mustAvoidNeonAtMBBI(Subtarget, MBB, I)) {
6291 MCRegister DestRegQ = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
6292 &AArch64::FPR128RegClass);
6293 MCRegister SrcRegQ = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
6294 &AArch64::FPR128RegClass);
6295 // This instruction is reading and writing Q registers. This may upset
6296 // the register scavenger and machine verifier, so we need to indicate
6297 // that we are reading an undefined value from SrcRegQ, but a proper
6298 // value from SrcReg.
6299 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestRegQ)
6300 .addReg(SrcRegQ, RegState::Undef)
6301 .addReg(SrcRegQ, RegState::Undef)
6302 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6303 } else if (Subtarget.hasZeroCycleRegMoveFPR64() &&
6304 !Subtarget.hasZeroCycleRegMoveFPR32()) {
6305 MCRegister DestRegD = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
6306 &AArch64::FPR64RegClass);
6307 MCRegister SrcRegD = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
6308 &AArch64::FPR64RegClass);
6309 // This instruction is reading and writing D registers. This may upset
6310 // the register scavenger and machine verifier, so we need to indicate
6311 // that we are reading an undefined value from SrcRegD, but a proper
6312 // value from SrcReg.
6313 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestRegD)
6314 .addReg(SrcRegD, RegState::Undef)
6315 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6316 } else {
6317 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::hsub,
6318 &AArch64::FPR32RegClass);
6319 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::hsub,
6320 &AArch64::FPR32RegClass);
6321 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
6322 .addReg(SrcReg, getKillRegState(KillSrc));
6323 }
6324 return;
6325 }
6326
6327 if (AArch64::FPR8RegClass.contains(DestReg) &&
6328 AArch64::FPR8RegClass.contains(SrcReg)) {
6329 if (Subtarget.hasZeroCycleRegMoveFPR128() &&
6330 !Subtarget.hasZeroCycleRegMoveFPR64() &&
6331 !Subtarget.hasZeroCycleRegMoveFPR32() && Subtarget.isNeonAvailable() &&
6332 !mustAvoidNeonAtMBBI(Subtarget, MBB, I)) {
6333 MCRegister DestRegQ = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
6334 &AArch64::FPR128RegClass);
6335 MCRegister SrcRegQ = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
6336 &AArch64::FPR128RegClass);
6337 // This instruction is reading and writing Q registers. This may upset
6338 // the register scavenger and machine verifier, so we need to indicate
6339 // that we are reading an undefined value from SrcRegQ, but a proper
6340 // value from SrcReg.
6341 BuildMI(MBB, I, DL, get(AArch64::ORRv16i8), DestRegQ)
6342 .addReg(SrcRegQ, RegState::Undef)
6343 .addReg(SrcRegQ, RegState::Undef)
6344 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6345 } else if (Subtarget.hasZeroCycleRegMoveFPR64() &&
6346 !Subtarget.hasZeroCycleRegMoveFPR32()) {
6347 MCRegister DestRegD = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
6348 &AArch64::FPR64RegClass);
6349 MCRegister SrcRegD = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
6350 &AArch64::FPR64RegClass);
6351 // This instruction is reading and writing D registers. This may upset
6352 // the register scavenger and machine verifier, so we need to indicate
6353 // that we are reading an undefined value from SrcRegD, but a proper
6354 // value from SrcReg.
6355 BuildMI(MBB, I, DL, get(AArch64::FMOVDr), DestRegD)
6356 .addReg(SrcRegD, RegState::Undef)
6357 .addReg(SrcReg, RegState::Implicit | getKillRegState(KillSrc));
6358 } else {
6359 DestReg = RI.getMatchingSuperReg(DestReg, AArch64::bsub,
6360 &AArch64::FPR32RegClass);
6361 SrcReg = RI.getMatchingSuperReg(SrcReg, AArch64::bsub,
6362 &AArch64::FPR32RegClass);
6363 BuildMI(MBB, I, DL, get(AArch64::FMOVSr), DestReg)
6364 .addReg(SrcReg, getKillRegState(KillSrc));
6365 }
6366 return;
6367 }
6368
6369 // Copies between GPR64 and FPR64.
6370 if (AArch64::FPR64RegClass.contains(DestReg) &&
6371 AArch64::GPR64RegClass.contains(SrcReg)) {
6372 if (AArch64::XZR == SrcReg) {
6373 BuildMI(MBB, I, DL, get(AArch64::FMOVD0), DestReg);
6374 } else {
6375 BuildMI(MBB, I, DL, get(AArch64::FMOVXDr), DestReg)
6376 .addReg(SrcReg, getKillRegState(KillSrc));
6377 }
6378 return;
6379 }
6380 if (AArch64::GPR64RegClass.contains(DestReg) &&
6381 AArch64::FPR64RegClass.contains(SrcReg)) {
6382 BuildMI(MBB, I, DL, get(AArch64::FMOVDXr), DestReg)
6383 .addReg(SrcReg, getKillRegState(KillSrc));
6384 return;
6385 }
6386 // Copies between GPR32 and FPR32.
6387 if (AArch64::FPR32RegClass.contains(DestReg) &&
6388 AArch64::GPR32RegClass.contains(SrcReg)) {
6389 if (AArch64::WZR == SrcReg) {
6390 BuildMI(MBB, I, DL, get(AArch64::FMOVS0), DestReg);
6391 } else {
6392 BuildMI(MBB, I, DL, get(AArch64::FMOVWSr), DestReg)
6393 .addReg(SrcReg, getKillRegState(KillSrc));
6394 }
6395 return;
6396 }
6397 if (AArch64::GPR32RegClass.contains(DestReg) &&
6398 AArch64::FPR32RegClass.contains(SrcReg)) {
6399 BuildMI(MBB, I, DL, get(AArch64::FMOVSWr), DestReg)
6400 .addReg(SrcReg, getKillRegState(KillSrc));
6401 return;
6402 }
6403
6404 if (DestReg == AArch64::NZCV) {
6405 assert(AArch64::GPR64RegClass.contains(SrcReg) && "Invalid NZCV copy");
6406 BuildMI(MBB, I, DL, get(AArch64::MSR))
6407 .addImm(AArch64SysReg::NZCV)
6408 .addReg(SrcReg, getKillRegState(KillSrc))
6409 .addReg(AArch64::NZCV, RegState::Implicit | RegState::Define);
6410 return;
6411 }
6412
6413 if (SrcReg == AArch64::NZCV) {
6414 assert(AArch64::GPR64RegClass.contains(DestReg) && "Invalid NZCV copy");
6415 BuildMI(MBB, I, DL, get(AArch64::MRS), DestReg)
6416 .addImm(AArch64SysReg::NZCV)
6417 .addReg(AArch64::NZCV, RegState::Implicit | getKillRegState(KillSrc));
6418 return;
6419 }
6420
6421#ifndef NDEBUG
6422 errs() << RI.getRegAsmName(DestReg) << " = COPY " << RI.getRegAsmName(SrcReg)
6423 << "\n";
6424#endif
6425 llvm_unreachable("unimplemented reg-to-reg copy");
6426}
6427
6430 const DebugLoc &DL, Register DestReg,
6431 Register SrcReg, bool KillSrc,
6432 bool RenamableDest,
6433 bool RenamableSrc) const {
6434 ++NumCopyInstrs;
6435 copyPhysRegImpl(MBB, I, DL, DestReg, SrcReg, KillSrc, RenamableDest,
6436 RenamableSrc);
6437 return;
6438}
6439
6442 MachineBasicBlock::iterator InsertBefore,
6443 const MCInstrDesc &MCID,
6444 Register SrcReg, bool IsKill,
6445 unsigned SubIdx0, unsigned SubIdx1, int FI,
6446 MachineMemOperand *MMO) {
6447 Register SrcReg0 = SrcReg;
6448 Register SrcReg1 = SrcReg;
6449 if (SrcReg.isPhysical()) {
6450 SrcReg0 = TRI.getSubReg(SrcReg, SubIdx0);
6451 SubIdx0 = 0;
6452 SrcReg1 = TRI.getSubReg(SrcReg, SubIdx1);
6453 SubIdx1 = 0;
6454 }
6455 BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
6456 .addReg(SrcReg0, getKillRegState(IsKill), SubIdx0)
6457 .addReg(SrcReg1, getKillRegState(IsKill), SubIdx1)
6458 .addFrameIndex(FI)
6459 .addImm(0)
6460 .addMemOperand(MMO);
6461}
6462
6465 Register SrcReg, bool isKill, int FI,
6466 const TargetRegisterClass *RC,
6467 Register VReg,
6468 MachineInstr::MIFlag Flags) const {
6469 MachineFunction &MF = *MBB.getParent();
6470 MachineFrameInfo &MFI = MF.getFrameInfo();
6471
6473 MachineMemOperand *MMO =
6475 MFI.getObjectSize(FI), MFI.getObjectAlign(FI));
6476 unsigned Opc = 0;
6477 bool Offset = true;
6479 unsigned StackID = TargetStackID::Default;
6480 switch (RI.getSpillSize(*RC)) {
6481 case 1:
6482 if (AArch64::FPR8RegClass.hasSubClassEq(RC))
6483 Opc = AArch64::STRBui;
6484 break;
6485 case 2: {
6486 if (AArch64::FPR16RegClass.hasSubClassEq(RC))
6487 Opc = AArch64::STRHui;
6488 else if (AArch64::PNRRegClass.hasSubClassEq(RC) ||
6489 AArch64::PPRRegClass.hasSubClassEq(RC)) {
6490 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6491 "Unexpected register store without SVE store instructions");
6492 Opc = AArch64::STR_PXI;
6494 }
6495 break;
6496 }
6497 case 4:
6498 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
6499 Opc = AArch64::STRWui;
6500 if (SrcReg.isVirtual())
6501 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR32RegClass);
6502 else
6503 assert(SrcReg != AArch64::WSP);
6504 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
6505 Opc = AArch64::STRSui;
6506 else if (AArch64::PPR2RegClass.hasSubClassEq(RC)) {
6507 Opc = AArch64::STR_PPXI;
6509 }
6510 break;
6511 case 8:
6512 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
6513 Opc = AArch64::STRXui;
6514 if (SrcReg.isVirtual())
6515 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
6516 else
6517 assert(SrcReg != AArch64::SP);
6518 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
6519 Opc = AArch64::STRDui;
6520 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
6522 get(AArch64::STPWi), SrcReg, isKill,
6523 AArch64::sube32, AArch64::subo32, FI, MMO);
6524 return;
6525 }
6526 break;
6527 case 16:
6528 if (AArch64::FPR128RegClass.hasSubClassEq(RC))
6529 Opc = AArch64::STRQui;
6530 else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
6531 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6532 Opc = AArch64::ST1Twov1d;
6533 Offset = false;
6534 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
6536 get(AArch64::STPXi), SrcReg, isKill,
6537 AArch64::sube64, AArch64::subo64, FI, MMO);
6538 return;
6539 } else if (AArch64::ZPRRegClass.hasSubClassEq(RC)) {
6540 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6541 "Unexpected register store without SVE store instructions");
6542 Opc = AArch64::STR_ZXI;
6544 }
6545 break;
6546 case 24:
6547 if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
6548 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6549 Opc = AArch64::ST1Threev1d;
6550 Offset = false;
6551 }
6552 break;
6553 case 32:
6554 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
6555 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6556 Opc = AArch64::ST1Fourv1d;
6557 Offset = false;
6558 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
6559 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6560 Opc = AArch64::ST1Twov2d;
6561 Offset = false;
6562 } else if (AArch64::ZPR2StridedOrContiguousRegClass.hasSubClassEq(RC)) {
6563 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6564 "Unexpected register store without SVE store instructions");
6565 Opc = AArch64::STR_ZZXI_STRIDED_CONTIGUOUS;
6567 } else if (AArch64::ZPR2RegClass.hasSubClassEq(RC)) {
6568 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6569 "Unexpected register store without SVE store instructions");
6570 Opc = AArch64::STR_ZZXI;
6572 }
6573 break;
6574 case 48:
6575 if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
6576 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6577 Opc = AArch64::ST1Threev2d;
6578 Offset = false;
6579 } else if (AArch64::ZPR3RegClass.hasSubClassEq(RC)) {
6580 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6581 "Unexpected register store without SVE store instructions");
6582 Opc = AArch64::STR_ZZZXI;
6584 }
6585 break;
6586 case 64:
6587 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
6588 assert(Subtarget.hasNEON() && "Unexpected register store without NEON");
6589 Opc = AArch64::ST1Fourv2d;
6590 Offset = false;
6591 } else if (AArch64::ZPR4StridedOrContiguousRegClass.hasSubClassEq(RC)) {
6592 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6593 "Unexpected register store without SVE store instructions");
6594 Opc = AArch64::STR_ZZZZXI_STRIDED_CONTIGUOUS;
6596 } else if (AArch64::ZPR4RegClass.hasSubClassEq(RC)) {
6597 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6598 "Unexpected register store without SVE store instructions");
6599 Opc = AArch64::STR_ZZZZXI;
6601 }
6602 break;
6603 }
6604 assert(Opc && "Unknown register class");
6605 MFI.setStackID(FI, StackID);
6606
6608 .addReg(SrcReg, getKillRegState(isKill))
6609 .addFrameIndex(FI);
6610
6611 if (Offset)
6612 MI.addImm(0);
6613 if (PNRReg.isValid())
6614 MI.addDef(PNRReg, RegState::Implicit);
6615 MI.addMemOperand(MMO);
6616}
6617
6620 MachineBasicBlock::iterator InsertBefore,
6621 const MCInstrDesc &MCID,
6622 Register DestReg, unsigned SubIdx0,
6623 unsigned SubIdx1, int FI,
6624 MachineMemOperand *MMO) {
6625 Register DestReg0 = DestReg;
6626 Register DestReg1 = DestReg;
6627 bool IsUndef = true;
6628 if (DestReg.isPhysical()) {
6629 DestReg0 = TRI.getSubReg(DestReg, SubIdx0);
6630 SubIdx0 = 0;
6631 DestReg1 = TRI.getSubReg(DestReg, SubIdx1);
6632 SubIdx1 = 0;
6633 IsUndef = false;
6634 }
6635 BuildMI(MBB, InsertBefore, DebugLoc(), MCID)
6636 .addReg(DestReg0, RegState::Define | getUndefRegState(IsUndef), SubIdx0)
6637 .addReg(DestReg1, RegState::Define | getUndefRegState(IsUndef), SubIdx1)
6638 .addFrameIndex(FI)
6639 .addImm(0)
6640 .addMemOperand(MMO);
6641}
6642
6645 Register DestReg, int FI,
6646 const TargetRegisterClass *RC,
6647 Register VReg, unsigned SubReg,
6648 MachineInstr::MIFlag Flags) const {
6649 MachineFunction &MF = *MBB.getParent();
6650 MachineFrameInfo &MFI = MF.getFrameInfo();
6652 MachineMemOperand *MMO =
6654 MFI.getObjectSize(FI), MFI.getObjectAlign(FI));
6655
6656 unsigned Opc = 0;
6657 bool Offset = true;
6658 unsigned StackID = TargetStackID::Default;
6659 Register PNRReg;
6660 switch (TRI.getSpillSize(*RC)) {
6661 case 1:
6662 if (AArch64::FPR8RegClass.hasSubClassEq(RC))
6663 Opc = AArch64::LDRBui;
6664 break;
6665 case 2: {
6666 bool IsPNR = AArch64::PNRRegClass.hasSubClassEq(RC);
6667 if (AArch64::FPR16RegClass.hasSubClassEq(RC))
6668 Opc = AArch64::LDRHui;
6669 else if (IsPNR || AArch64::PPRRegClass.hasSubClassEq(RC)) {
6670 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6671 "Unexpected register load without SVE load instructions");
6672 if (IsPNR)
6673 PNRReg = DestReg;
6674 Opc = AArch64::LDR_PXI;
6676 }
6677 break;
6678 }
6679 case 4:
6680 if (AArch64::GPR32allRegClass.hasSubClassEq(RC)) {
6681 Opc = AArch64::LDRWui;
6682 if (DestReg.isVirtual())
6683 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR32RegClass);
6684 else
6685 assert(DestReg != AArch64::WSP);
6686 } else if (AArch64::FPR32RegClass.hasSubClassEq(RC))
6687 Opc = AArch64::LDRSui;
6688 else if (AArch64::PPR2RegClass.hasSubClassEq(RC)) {
6689 Opc = AArch64::LDR_PPXI;
6691 }
6692 break;
6693 case 8:
6694 if (AArch64::GPR64allRegClass.hasSubClassEq(RC)) {
6695 Opc = AArch64::LDRXui;
6696 if (DestReg.isVirtual())
6697 MF.getRegInfo().constrainRegClass(DestReg, &AArch64::GPR64RegClass);
6698 else
6699 assert(DestReg != AArch64::SP);
6700 } else if (AArch64::FPR64RegClass.hasSubClassEq(RC)) {
6701 Opc = AArch64::LDRDui;
6702 } else if (AArch64::WSeqPairsClassRegClass.hasSubClassEq(RC)) {
6704 get(AArch64::LDPWi), DestReg, AArch64::sube32,
6705 AArch64::subo32, FI, MMO);
6706 return;
6707 }
6708 break;
6709 case 16:
6710 if (AArch64::FPR128RegClass.hasSubClassEq(RC))
6711 Opc = AArch64::LDRQui;
6712 else if (AArch64::DDRegClass.hasSubClassEq(RC)) {
6713 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6714 Opc = AArch64::LD1Twov1d;
6715 Offset = false;
6716 } else if (AArch64::XSeqPairsClassRegClass.hasSubClassEq(RC)) {
6718 get(AArch64::LDPXi), DestReg, AArch64::sube64,
6719 AArch64::subo64, FI, MMO);
6720 return;
6721 } else if (AArch64::ZPRRegClass.hasSubClassEq(RC)) {
6722 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6723 "Unexpected register load without SVE load instructions");
6724 Opc = AArch64::LDR_ZXI;
6726 }
6727 break;
6728 case 24:
6729 if (AArch64::DDDRegClass.hasSubClassEq(RC)) {
6730 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6731 Opc = AArch64::LD1Threev1d;
6732 Offset = false;
6733 }
6734 break;
6735 case 32:
6736 if (AArch64::DDDDRegClass.hasSubClassEq(RC)) {
6737 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6738 Opc = AArch64::LD1Fourv1d;
6739 Offset = false;
6740 } else if (AArch64::QQRegClass.hasSubClassEq(RC)) {
6741 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6742 Opc = AArch64::LD1Twov2d;
6743 Offset = false;
6744 } else if (AArch64::ZPR2StridedOrContiguousRegClass.hasSubClassEq(RC)) {
6745 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6746 "Unexpected register load without SVE load instructions");
6747 Opc = AArch64::LDR_ZZXI_STRIDED_CONTIGUOUS;
6749 } else if (AArch64::ZPR2RegClass.hasSubClassEq(RC)) {
6750 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6751 "Unexpected register load without SVE load instructions");
6752 Opc = AArch64::LDR_ZZXI;
6754 }
6755 break;
6756 case 48:
6757 if (AArch64::QQQRegClass.hasSubClassEq(RC)) {
6758 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6759 Opc = AArch64::LD1Threev2d;
6760 Offset = false;
6761 } else if (AArch64::ZPR3RegClass.hasSubClassEq(RC)) {
6762 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6763 "Unexpected register load without SVE load instructions");
6764 Opc = AArch64::LDR_ZZZXI;
6766 }
6767 break;
6768 case 64:
6769 if (AArch64::QQQQRegClass.hasSubClassEq(RC)) {
6770 assert(Subtarget.hasNEON() && "Unexpected register load without NEON");
6771 Opc = AArch64::LD1Fourv2d;
6772 Offset = false;
6773 } else if (AArch64::ZPR4StridedOrContiguousRegClass.hasSubClassEq(RC)) {
6774 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6775 "Unexpected register load without SVE load instructions");
6776 Opc = AArch64::LDR_ZZZZXI_STRIDED_CONTIGUOUS;
6778 } else if (AArch64::ZPR4RegClass.hasSubClassEq(RC)) {
6779 assert(Subtarget.isSVEorStreamingSVEAvailable() &&
6780 "Unexpected register load without SVE load instructions");
6781 Opc = AArch64::LDR_ZZZZXI;
6783 }
6784 break;
6785 }
6786
6787 assert(Opc && "Unknown register class");
6788 MFI.setStackID(FI, StackID);
6789
6791 .addReg(DestReg, getDefRegState(true))
6792 .addFrameIndex(FI);
6793 if (Offset)
6794 MI.addImm(0);
6795 if (PNRReg.isValid() && !PNRReg.isVirtual())
6796 MI.addDef(PNRReg, RegState::Implicit);
6797 MI.addMemOperand(MMO);
6798}
6799
6801 const MachineInstr &UseMI,
6802 const TargetRegisterInfo *TRI) {
6803 return any_of(instructionsWithoutDebug(std::next(DefMI.getIterator()),
6804 UseMI.getIterator()),
6805 [TRI](const MachineInstr &I) {
6806 return I.modifiesRegister(AArch64::NZCV, TRI) ||
6807 I.readsRegister(AArch64::NZCV, TRI);
6808 });
6809}
6810
6811void AArch64InstrInfo::decomposeStackOffsetForDwarfOffsets(
6812 const StackOffset &Offset, int64_t &ByteSized, int64_t &VGSized) {
6813 // The smallest scalable element supported by scaled SVE addressing
6814 // modes are predicates, which are 2 scalable bytes in size. So the scalable
6815 // byte offset must always be a multiple of 2.
6816 assert(Offset.getScalable() % 2 == 0 && "Invalid frame offset");
6817
6818 // VGSized offsets are divided by '2', because the VG register is the
6819 // the number of 64bit granules as opposed to 128bit vector chunks,
6820 // which is how the 'n' in e.g. MVT::nxv1i8 is modelled.
6821 // So, for a stack offset of 16 MVT::nxv1i8's, the size is n x 16 bytes.
6822 // VG = n * 2 and the dwarf offset must be VG * 8 bytes.
6823 ByteSized = Offset.getFixed();
6824 VGSized = Offset.getScalable() / 2;
6825}
6826
6827/// Returns the offset in parts to which this frame offset can be
6828/// decomposed for the purpose of describing a frame offset.
6829/// For non-scalable offsets this is simply its byte size.
6830void AArch64InstrInfo::decomposeStackOffsetForFrameOffsets(
6831 const StackOffset &Offset, int64_t &NumBytes, int64_t &NumPredicateVectors,
6832 int64_t &NumDataVectors) {
6833 // The smallest scalable element supported by scaled SVE addressing
6834 // modes are predicates, which are 2 scalable bytes in size. So the scalable
6835 // byte offset must always be a multiple of 2.
6836 assert(Offset.getScalable() % 2 == 0 && "Invalid frame offset");
6837
6838 NumBytes = Offset.getFixed();
6839 NumDataVectors = 0;
6840 NumPredicateVectors = Offset.getScalable() / 2;
6841 // This method is used to get the offsets to adjust the frame offset.
6842 // If the function requires ADDPL to be used and needs more than two ADDPL
6843 // instructions, part of the offset is folded into NumDataVectors so that it
6844 // uses ADDVL for part of it, reducing the number of ADDPL instructions.
6845 if (NumPredicateVectors % 8 == 0 || NumPredicateVectors < -64 ||
6846 NumPredicateVectors > 62) {
6847 NumDataVectors = NumPredicateVectors / 8;
6848 NumPredicateVectors -= NumDataVectors * 8;
6849 }
6850}
6851
6852// Convenience function to create a DWARF expression for: Constant `Operation`.
6853// This helper emits compact sequences for common cases. For example, for`-15
6854// DW_OP_plus`, this helper would create DW_OP_lit15 DW_OP_minus.
6857 if (Operation == dwarf::DW_OP_plus && Constant < 0 && -Constant <= 31) {
6858 // -Constant (1 to 31)
6859 Expr.push_back(dwarf::DW_OP_lit0 - Constant);
6860 Operation = dwarf::DW_OP_minus;
6861 } else if (Constant >= 0 && Constant <= 31) {
6862 // Literal value 0 to 31
6863 Expr.push_back(dwarf::DW_OP_lit0 + Constant);
6864 } else {
6865 // Signed constant
6866 Expr.push_back(dwarf::DW_OP_consts);
6868 }
6869 return Expr.push_back(Operation);
6870}
6871
6872// Convenience function to create a DWARF expression for a register.
6873static void appendReadRegExpr(SmallVectorImpl<char> &Expr, unsigned RegNum) {
6874 Expr.push_back((char)dwarf::DW_OP_bregx);
6876 Expr.push_back(0);
6877}
6878
6879// Convenience function to create a DWARF expression for loading a register from
6880// a CFA offset.
6882 int64_t OffsetFromDefCFA) {
6883 // This assumes the top of the DWARF stack contains the CFA.
6884 Expr.push_back(dwarf::DW_OP_dup);
6885 // Add the offset to the register.
6886 appendConstantExpr(Expr, OffsetFromDefCFA, dwarf::DW_OP_plus);
6887 // Dereference the address (loads a 64 bit value)..
6888 Expr.push_back(dwarf::DW_OP_deref);
6889}
6890
6891// Convenience function to create a comment for
6892// (+/-) NumBytes (* RegScale)?
6893static void appendOffsetComment(int NumBytes, llvm::raw_string_ostream &Comment,
6894 StringRef RegScale = {}) {
6895 if (NumBytes) {
6896 Comment << (NumBytes < 0 ? " - " : " + ") << std::abs(NumBytes);
6897 if (!RegScale.empty())
6898 Comment << ' ' << RegScale;
6899 }
6900}
6901
6902// Creates an MCCFIInstruction:
6903// { DW_CFA_def_cfa_expression, ULEB128 (sizeof expr), expr }
6905 unsigned Reg,
6906 const StackOffset &Offset) {
6907 int64_t NumBytes, NumVGScaledBytes;
6908 AArch64InstrInfo::decomposeStackOffsetForDwarfOffsets(Offset, NumBytes,
6909 NumVGScaledBytes);
6910 std::string CommentBuffer;
6911 llvm::raw_string_ostream Comment(CommentBuffer);
6912
6913 if (Reg == AArch64::SP)
6914 Comment << "sp";
6915 else if (Reg == AArch64::FP)
6916 Comment << "fp";
6917 else
6918 Comment << printReg(Reg, &TRI);
6919
6920 // Build up the expression (Reg + NumBytes + VG * NumVGScaledBytes)
6921 SmallString<64> Expr;
6922 unsigned DwarfReg = TRI.getDwarfRegNum(Reg, true);
6923 assert(DwarfReg <= 31 && "DwarfReg out of bounds (0..31)");
6924 // Reg + NumBytes
6925 Expr.push_back(dwarf::DW_OP_breg0 + DwarfReg);
6926 appendLEB128<LEB128Sign::Signed>(Expr, NumBytes);
6927 appendOffsetComment(NumBytes, Comment);
6928 if (NumVGScaledBytes) {
6929 // + VG * NumVGScaledBytes
6930 appendOffsetComment(NumVGScaledBytes, Comment, "* VG");
6931 appendReadRegExpr(Expr, TRI.getDwarfRegNum(AArch64::VG, true));
6932 appendConstantExpr(Expr, NumVGScaledBytes, dwarf::DW_OP_mul);
6933 Expr.push_back(dwarf::DW_OP_plus);
6934 }
6935
6936 // Wrap this into DW_CFA_def_cfa.
6937 SmallString<64> DefCfaExpr;
6938 DefCfaExpr.push_back(dwarf::DW_CFA_def_cfa_expression);
6939 appendLEB128<LEB128Sign::Unsigned>(DefCfaExpr, Expr.size());
6940 DefCfaExpr.append(Expr.str());
6941 return MCCFIInstruction::createEscape(nullptr, DefCfaExpr.str(), SMLoc(),
6942 Comment.str());
6943}
6944
6946 unsigned FrameReg, unsigned Reg,
6947 const StackOffset &Offset,
6948 bool LastAdjustmentWasScalable) {
6949 if (Offset.getScalable())
6950 return createDefCFAExpression(TRI, Reg, Offset);
6951
6952 if (FrameReg == Reg && !LastAdjustmentWasScalable)
6953 return MCCFIInstruction::cfiDefCfaOffset(nullptr, int(Offset.getFixed()));
6954
6955 unsigned DwarfReg = TRI.getDwarfRegNum(Reg, true);
6956 return MCCFIInstruction::cfiDefCfa(nullptr, DwarfReg, (int)Offset.getFixed());
6957}
6958
6961 const StackOffset &OffsetFromDefCFA,
6962 std::optional<int64_t> IncomingVGOffsetFromDefCFA) {
6963 int64_t NumBytes, NumVGScaledBytes;
6964 AArch64InstrInfo::decomposeStackOffsetForDwarfOffsets(
6965 OffsetFromDefCFA, NumBytes, NumVGScaledBytes);
6966
6967 unsigned DwarfReg = TRI.getDwarfRegNum(Reg, true);
6968
6969 // Non-scalable offsets can use DW_CFA_offset directly.
6970 if (!NumVGScaledBytes)
6971 return MCCFIInstruction::createOffset(nullptr, DwarfReg, NumBytes);
6972
6973 std::string CommentBuffer;
6974 llvm::raw_string_ostream Comment(CommentBuffer);
6975 Comment << printReg(Reg, &TRI) << " @ cfa";
6976
6977 // Build up expression (CFA + VG * NumVGScaledBytes + NumBytes)
6978 assert(NumVGScaledBytes && "Expected scalable offset");
6979 SmallString<64> OffsetExpr;
6980 // + VG * NumVGScaledBytes
6981 StringRef VGRegScale;
6982 if (IncomingVGOffsetFromDefCFA) {
6983 appendLoadRegExpr(OffsetExpr, *IncomingVGOffsetFromDefCFA);
6984 VGRegScale = "* IncomingVG";
6985 } else {
6986 appendReadRegExpr(OffsetExpr, TRI.getDwarfRegNum(AArch64::VG, true));
6987 VGRegScale = "* VG";
6988 }
6989 appendConstantExpr(OffsetExpr, NumVGScaledBytes, dwarf::DW_OP_mul);
6990 appendOffsetComment(NumVGScaledBytes, Comment, VGRegScale);
6991 OffsetExpr.push_back(dwarf::DW_OP_plus);
6992 if (NumBytes) {
6993 // + NumBytes
6994 appendOffsetComment(NumBytes, Comment);
6995 appendConstantExpr(OffsetExpr, NumBytes, dwarf::DW_OP_plus);
6996 }
6997
6998 // Wrap this into DW_CFA_expression
6999 SmallString<64> CfaExpr;
7000 CfaExpr.push_back(dwarf::DW_CFA_expression);
7001 appendLEB128<LEB128Sign::Unsigned>(CfaExpr, DwarfReg);
7002 appendLEB128<LEB128Sign::Unsigned>(CfaExpr, OffsetExpr.size());
7003 CfaExpr.append(OffsetExpr.str());
7004
7005 return MCCFIInstruction::createEscape(nullptr, CfaExpr.str(), SMLoc(),
7006 Comment.str());
7007}
7008
7009// Helper function to emit a frame offset adjustment from a given
7010// pointer (SrcReg), stored into DestReg. This function is explicit
7011// in that it requires the opcode.
7014 const DebugLoc &DL, unsigned DestReg,
7015 unsigned SrcReg, int64_t Offset, unsigned Opc,
7016 const TargetInstrInfo *TII,
7017 MachineInstr::MIFlag Flag, bool NeedsWinCFI,
7018 bool *HasWinCFI, bool EmitCFAOffset,
7019 StackOffset CFAOffset, unsigned FrameReg) {
7020 int Sign = 1;
7021 unsigned MaxEncoding, ShiftSize;
7022 switch (Opc) {
7023 case AArch64::ADDXri:
7024 case AArch64::ADDSXri:
7025 case AArch64::SUBXri:
7026 case AArch64::SUBSXri:
7027 MaxEncoding = 0xfff;
7028 ShiftSize = 12;
7029 break;
7030 case AArch64::ADDVL_XXI:
7031 case AArch64::ADDPL_XXI:
7032 case AArch64::ADDSVL_XXI:
7033 case AArch64::ADDSPL_XXI:
7034 MaxEncoding = 31;
7035 ShiftSize = 0;
7036 if (Offset < 0) {
7037 MaxEncoding = 32;
7038 Sign = -1;
7039 Offset = -Offset;
7040 }
7041 break;
7042 default:
7043 llvm_unreachable("Unsupported opcode");
7044 }
7045
7046 // `Offset` can be in bytes or in "scalable bytes".
7047 int VScale = 1;
7048 if (Opc == AArch64::ADDVL_XXI || Opc == AArch64::ADDSVL_XXI)
7049 VScale = 16;
7050 else if (Opc == AArch64::ADDPL_XXI || Opc == AArch64::ADDSPL_XXI)
7051 VScale = 2;
7052
7053 // FIXME: If the offset won't fit in 24-bits, compute the offset into a
7054 // scratch register. If DestReg is a virtual register, use it as the
7055 // scratch register; otherwise, create a new virtual register (to be
7056 // replaced by the scavenger at the end of PEI). That case can be optimized
7057 // slightly if DestReg is SP which is always 16-byte aligned, so the scratch
7058 // register can be loaded with offset%8 and the add/sub can use an extending
7059 // instruction with LSL#3.
7060 // Currently the function handles any offsets but generates a poor sequence
7061 // of code.
7062 // assert(Offset < (1 << 24) && "unimplemented reg plus immediate");
7063
7064 const unsigned MaxEncodableValue = MaxEncoding << ShiftSize;
7065 Register TmpReg = DestReg;
7066 if (TmpReg == AArch64::XZR)
7067 TmpReg = MBB.getParent()->getRegInfo().createVirtualRegister(
7068 &AArch64::GPR64RegClass);
7069 do {
7070 uint64_t ThisVal = std::min<uint64_t>(Offset, MaxEncodableValue);
7071 unsigned LocalShiftSize = 0;
7072 if (ThisVal > MaxEncoding) {
7073 ThisVal = ThisVal >> ShiftSize;
7074 LocalShiftSize = ShiftSize;
7075 }
7076 assert((ThisVal >> ShiftSize) <= MaxEncoding &&
7077 "Encoding cannot handle value that big");
7078
7079 Offset -= ThisVal << LocalShiftSize;
7080 if (Offset == 0)
7081 TmpReg = DestReg;
7082 auto MBI = BuildMI(MBB, MBBI, DL, TII->get(Opc), TmpReg)
7083 .addReg(SrcReg)
7084 .addImm(Sign * (int)ThisVal);
7085 if (ShiftSize)
7086 MBI = MBI.addImm(
7088 MBI = MBI.setMIFlag(Flag);
7089
7090 auto Change =
7091 VScale == 1
7092 ? StackOffset::getFixed(ThisVal << LocalShiftSize)
7093 : StackOffset::getScalable(VScale * (ThisVal << LocalShiftSize));
7094 if (Sign == -1 || Opc == AArch64::SUBXri || Opc == AArch64::SUBSXri)
7095 CFAOffset += Change;
7096 else
7097 CFAOffset -= Change;
7098 if (EmitCFAOffset && DestReg == TmpReg) {
7099 MachineFunction &MF = *MBB.getParent();
7100 const TargetSubtargetInfo &STI = MF.getSubtarget();
7101 const TargetRegisterInfo &TRI = *STI.getRegisterInfo();
7102
7103 unsigned CFIIndex = MF.addFrameInst(
7104 createDefCFA(TRI, FrameReg, DestReg, CFAOffset, VScale != 1));
7105 BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION))
7106 .addCFIIndex(CFIIndex)
7107 .setMIFlags(Flag);
7108 }
7109
7110 if (NeedsWinCFI) {
7111 int Imm = (int)(ThisVal << LocalShiftSize);
7112 if (VScale != 1 && DestReg == AArch64::SP) {
7113 if (HasWinCFI)
7114 *HasWinCFI = true;
7115 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AllocZ))
7116 .addImm(ThisVal)
7117 .setMIFlag(Flag);
7118 } else if ((DestReg == AArch64::FP && SrcReg == AArch64::SP) ||
7119 (SrcReg == AArch64::FP && DestReg == AArch64::SP)) {
7120 assert(VScale == 1 && "Expected non-scalable operation");
7121 if (HasWinCFI)
7122 *HasWinCFI = true;
7123 if (Imm == 0)
7124 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_SetFP)).setMIFlag(Flag);
7125 else
7126 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_AddFP))
7127 .addImm(Imm)
7128 .setMIFlag(Flag);
7129 assert(Offset == 0 && "Expected remaining offset to be zero to "
7130 "emit a single SEH directive");
7131 } else if (DestReg == AArch64::SP) {
7132 assert(VScale == 1 && "Expected non-scalable operation");
7133 if (HasWinCFI)
7134 *HasWinCFI = true;
7135 assert(SrcReg == AArch64::SP && "Unexpected SrcReg for SEH_StackAlloc");
7136 BuildMI(MBB, MBBI, DL, TII->get(AArch64::SEH_StackAlloc))
7137 .addImm(Imm)
7138 .setMIFlag(Flag);
7139 }
7140 }
7141
7142 SrcReg = TmpReg;
7143 } while (Offset);
7144}
7145
7148 unsigned DestReg, unsigned SrcReg,
7150 MachineInstr::MIFlag Flag, bool SetNZCV,
7151 bool NeedsWinCFI, bool *HasWinCFI,
7152 bool EmitCFAOffset, StackOffset CFAOffset,
7153 unsigned FrameReg) {
7154 // If a function is marked as arm_locally_streaming, then the runtime value of
7155 // vscale in the prologue/epilogue is different the runtime value of vscale
7156 // in the function's body. To avoid having to consider multiple vscales,
7157 // we can use `addsvl` to allocate any scalable stack-slots, which under
7158 // most circumstances will be only locals, not callee-save slots.
7159 const Function &F = MBB.getParent()->getFunction();
7160 bool UseSVL = F.hasFnAttribute("aarch64_pstate_sm_body");
7161
7162 int64_t Bytes, NumPredicateVectors, NumDataVectors;
7163 AArch64InstrInfo::decomposeStackOffsetForFrameOffsets(
7164 Offset, Bytes, NumPredicateVectors, NumDataVectors);
7165
7166 // Insert ADDSXri for scalable offset at the end.
7167 bool NeedsFinalDefNZCV = SetNZCV && (NumPredicateVectors || NumDataVectors);
7168 if (NeedsFinalDefNZCV)
7169 SetNZCV = false;
7170
7171 // First emit non-scalable frame offsets, or a simple 'mov'.
7172 if (Bytes || (!Offset && SrcReg != DestReg)) {
7173 assert((DestReg != AArch64::SP || Bytes % 8 == 0) &&
7174 "SP increment/decrement not 8-byte aligned");
7175 unsigned Opc = SetNZCV ? AArch64::ADDSXri : AArch64::ADDXri;
7176 if (Bytes < 0) {
7177 Bytes = -Bytes;
7178 Opc = SetNZCV ? AArch64::SUBSXri : AArch64::SUBXri;
7179 }
7180 emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, Bytes, Opc, TII, Flag,
7181 NeedsWinCFI, HasWinCFI, EmitCFAOffset, CFAOffset,
7182 FrameReg);
7183 CFAOffset += (Opc == AArch64::ADDXri || Opc == AArch64::ADDSXri)
7184 ? StackOffset::getFixed(-Bytes)
7185 : StackOffset::getFixed(Bytes);
7186 SrcReg = DestReg;
7187 FrameReg = DestReg;
7188 }
7189
7190 assert(!(NeedsWinCFI && NumPredicateVectors) &&
7191 "WinCFI can't allocate fractions of an SVE data vector");
7192
7193 if (NumDataVectors) {
7194 emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, NumDataVectors,
7195 UseSVL ? AArch64::ADDSVL_XXI : AArch64::ADDVL_XXI, TII,
7196 Flag, NeedsWinCFI, HasWinCFI, EmitCFAOffset, CFAOffset,
7197 FrameReg);
7198 CFAOffset += StackOffset::getScalable(-NumDataVectors * 16);
7199 SrcReg = DestReg;
7200 }
7201
7202 if (NumPredicateVectors) {
7203 assert(DestReg != AArch64::SP && "Unaligned access to SP");
7204 emitFrameOffsetAdj(MBB, MBBI, DL, DestReg, SrcReg, NumPredicateVectors,
7205 UseSVL ? AArch64::ADDSPL_XXI : AArch64::ADDPL_XXI, TII,
7206 Flag, NeedsWinCFI, HasWinCFI, EmitCFAOffset, CFAOffset,
7207 FrameReg);
7208 }
7209
7210 if (NeedsFinalDefNZCV)
7211 BuildMI(MBB, MBBI, DL, TII->get(AArch64::ADDSXri), DestReg)
7212 .addReg(DestReg)
7213 .addImm(0)
7214 .addImm(0);
7215}
7216
7219 int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS,
7220 VirtRegMap *VRM) const {
7222 // This is a bit of a hack. Consider this instruction:
7223 //
7224 // %0 = COPY %sp; GPR64all:%0
7225 //
7226 // We explicitly chose GPR64all for the virtual register so such a copy might
7227 // be eliminated by RegisterCoalescer. However, that may not be possible, and
7228 // %0 may even spill. We can't spill %sp, and since it is in the GPR64all
7229 // register class, TargetInstrInfo::foldMemoryOperand() is going to try.
7230 //
7231 // To prevent that, we are going to constrain the %0 register class here.
7232 if (MI.isFullCopy()) {
7233 Register DstReg = MI.getOperand(0).getReg();
7234 Register SrcReg = MI.getOperand(1).getReg();
7235 if (SrcReg == AArch64::SP && DstReg.isVirtual()) {
7236 MF.getRegInfo().constrainRegClass(DstReg, &AArch64::GPR64RegClass);
7237 return nullptr;
7238 }
7239 if (DstReg == AArch64::SP && SrcReg.isVirtual()) {
7240 MF.getRegInfo().constrainRegClass(SrcReg, &AArch64::GPR64RegClass);
7241 return nullptr;
7242 }
7243 // Nothing can folded with copy from/to NZCV.
7244 if (SrcReg == AArch64::NZCV || DstReg == AArch64::NZCV)
7245 return nullptr;
7246 }
7247
7248 // Handle the case where a copy is being spilled or filled but the source
7249 // and destination register class don't match. For example:
7250 //
7251 // %0 = COPY %xzr; GPR64common:%0
7252 //
7253 // In this case we can still safely fold away the COPY and generate the
7254 // following spill code:
7255 //
7256 // STRXui %xzr, %stack.0
7257 //
7258 // This also eliminates spilled cross register class COPYs (e.g. between x and
7259 // d regs) of the same size. For example:
7260 //
7261 // %0 = COPY %1; GPR64:%0, FPR64:%1
7262 //
7263 // will be filled as
7264 //
7265 // LDRDui %0, fi<#0>
7266 //
7267 // instead of
7268 //
7269 // LDRXui %Temp, fi<#0>
7270 // %0 = FMOV %Temp
7271 //
7272 if (MI.isCopy() && Ops.size() == 1 &&
7273 // Make sure we're only folding the explicit COPY defs/uses.
7274 (Ops[0] == 0 || Ops[0] == 1)) {
7275 bool IsSpill = Ops[0] == 0;
7276 bool IsFill = !IsSpill;
7278 const MachineRegisterInfo &MRI = MF.getRegInfo();
7279 MachineBasicBlock &MBB = *MI.getParent();
7280 const MachineOperand &DstMO = MI.getOperand(0);
7281 const MachineOperand &SrcMO = MI.getOperand(1);
7282 Register DstReg = DstMO.getReg();
7283 Register SrcReg = SrcMO.getReg();
7284 // This is slightly expensive to compute for physical regs since
7285 // getMinimalPhysRegClass is slow.
7286 auto getRegClass = [&](unsigned Reg) {
7287 return Register::isVirtualRegister(Reg) ? MRI.getRegClass(Reg)
7288 : TRI.getMinimalPhysRegClass(Reg);
7289 };
7290
7291 if (DstMO.getSubReg() == 0 && SrcMO.getSubReg() == 0) {
7292 assert(TRI.getRegSizeInBits(*getRegClass(DstReg)) ==
7293 TRI.getRegSizeInBits(*getRegClass(SrcReg)) &&
7294 "Mismatched register size in non subreg COPY");
7295 if (IsSpill)
7296 storeRegToStackSlot(MBB, InsertPt, SrcReg, SrcMO.isKill(), FrameIndex,
7297 getRegClass(SrcReg), Register());
7298 else
7299 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex,
7300 getRegClass(DstReg), Register());
7301 return &*--InsertPt;
7302 }
7303
7304 // Handle cases like spilling def of:
7305 //
7306 // %0:sub_32<def,read-undef> = COPY %wzr; GPR64common:%0
7307 //
7308 // where the physical register source can be widened and stored to the full
7309 // virtual reg destination stack slot, in this case producing:
7310 //
7311 // STRXui %xzr, %stack.0
7312 //
7313 if (IsSpill && DstMO.isUndef() && SrcReg == AArch64::WZR &&
7314 TRI.getRegSizeInBits(*getRegClass(DstReg)) == 64) {
7315 assert(SrcMO.getSubReg() == 0 &&
7316 "Unexpected subreg on physical register");
7317 storeRegToStackSlot(MBB, InsertPt, AArch64::XZR, SrcMO.isKill(),
7318 FrameIndex, &AArch64::GPR64RegClass, Register());
7319 return &*--InsertPt;
7320 }
7321
7322 // Handle cases like filling use of:
7323 //
7324 // %0:sub_32<def,read-undef> = COPY %1; GPR64:%0, GPR32:%1
7325 //
7326 // where we can load the full virtual reg source stack slot, into the subreg
7327 // destination, in this case producing:
7328 //
7329 // LDRWui %0:sub_32<def,read-undef>, %stack.0
7330 //
7331 if (IsFill && SrcMO.getSubReg() == 0 && DstMO.isUndef()) {
7332 const TargetRegisterClass *FillRC = nullptr;
7333 switch (DstMO.getSubReg()) {
7334 default:
7335 break;
7336 case AArch64::sub_32:
7337 if (AArch64::GPR64RegClass.hasSubClassEq(getRegClass(DstReg)))
7338 FillRC = &AArch64::GPR32RegClass;
7339 break;
7340 case AArch64::ssub:
7341 FillRC = &AArch64::FPR32RegClass;
7342 break;
7343 case AArch64::dsub:
7344 FillRC = &AArch64::FPR64RegClass;
7345 break;
7346 }
7347
7348 if (FillRC) {
7349 assert(TRI.getRegSizeInBits(*getRegClass(SrcReg)) ==
7350 TRI.getRegSizeInBits(*FillRC) &&
7351 "Mismatched regclass size on folded subreg COPY");
7352 loadRegFromStackSlot(MBB, InsertPt, DstReg, FrameIndex, FillRC,
7353 Register());
7354 MachineInstr &LoadMI = *--InsertPt;
7355 MachineOperand &LoadDst = LoadMI.getOperand(0);
7356 assert(LoadDst.getSubReg() == 0 && "unexpected subreg on fill load");
7357 LoadDst.setSubReg(DstMO.getSubReg());
7358 LoadDst.setIsUndef();
7359 return &LoadMI;
7360 }
7361 }
7362 }
7363
7364 // Cannot fold.
7365 return nullptr;
7366}
7367
7369 StackOffset &SOffset,
7370 bool *OutUseUnscaledOp,
7371 unsigned *OutUnscaledOp,
7372 int64_t *EmittableOffset) {
7373 // Set output values in case of early exit.
7374 if (EmittableOffset)
7375 *EmittableOffset = 0;
7376 if (OutUseUnscaledOp)
7377 *OutUseUnscaledOp = false;
7378 if (OutUnscaledOp)
7379 *OutUnscaledOp = 0;
7380
7381 // Exit early for structured vector spills/fills as they can't take an
7382 // immediate offset.
7383 switch (MI.getOpcode()) {
7384 default:
7385 break;
7386 case AArch64::LD1Rv1d:
7387 case AArch64::LD1Rv2s:
7388 case AArch64::LD1Rv2d:
7389 case AArch64::LD1Rv4h:
7390 case AArch64::LD1Rv4s:
7391 case AArch64::LD1Rv8b:
7392 case AArch64::LD1Rv8h:
7393 case AArch64::LD1Rv16b:
7394 case AArch64::LD1Twov2d:
7395 case AArch64::LD1Threev2d:
7396 case AArch64::LD1Fourv2d:
7397 case AArch64::LD1Twov1d:
7398 case AArch64::LD1Threev1d:
7399 case AArch64::LD1Fourv1d:
7400 case AArch64::ST1Twov2d:
7401 case AArch64::ST1Threev2d:
7402 case AArch64::ST1Fourv2d:
7403 case AArch64::ST1Twov1d:
7404 case AArch64::ST1Threev1d:
7405 case AArch64::ST1Fourv1d:
7406 case AArch64::ST1i8:
7407 case AArch64::ST1i16:
7408 case AArch64::ST1i32:
7409 case AArch64::ST1i64:
7410 case AArch64::IRG:
7411 case AArch64::IRGstack:
7412 case AArch64::STGloop:
7413 case AArch64::STZGloop:
7415 }
7416
7417 // Get the min/max offset and the scale.
7418 TypeSize ScaleValue(0U, false), Width(0U, false);
7419 int64_t MinOff, MaxOff;
7420 if (!AArch64InstrInfo::getMemOpInfo(MI.getOpcode(), ScaleValue, Width, MinOff,
7421 MaxOff))
7422 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
7423
7424 // Construct the complete offset.
7425 bool IsMulVL = ScaleValue.isScalable();
7426 unsigned Scale = ScaleValue.getKnownMinValue();
7427 int64_t Offset = IsMulVL ? SOffset.getScalable() : SOffset.getFixed();
7428
7429 const MachineOperand &ImmOpnd =
7430 MI.getOperand(AArch64InstrInfo::getLoadStoreImmIdx(MI.getOpcode()));
7431 Offset += ImmOpnd.getImm() * Scale;
7432
7433 // If the offset doesn't match the scale, we rewrite the instruction to
7434 // use the unscaled instruction instead. Likewise, if we have a negative
7435 // offset and there is an unscaled op to use.
7436 std::optional<unsigned> UnscaledOp =
7438 bool useUnscaledOp = UnscaledOp && (Offset % Scale || Offset < 0);
7439 if (useUnscaledOp &&
7440 !AArch64InstrInfo::getMemOpInfo(*UnscaledOp, ScaleValue, Width, MinOff,
7441 MaxOff))
7442 llvm_unreachable("unhandled opcode in isAArch64FrameOffsetLegal");
7443
7444 Scale = ScaleValue.getKnownMinValue();
7445 assert(IsMulVL == ScaleValue.isScalable() &&
7446 "Unscaled opcode has different value for scalable");
7447
7448 int64_t Remainder = Offset % Scale;
7449 assert(!(Remainder && useUnscaledOp) &&
7450 "Cannot have remainder when using unscaled op");
7451
7452 assert(MinOff < MaxOff && "Unexpected Min/Max offsets");
7453 int64_t NewOffset = Offset / Scale;
7454 if (MinOff <= NewOffset && NewOffset <= MaxOff)
7455 Offset = Remainder;
7456 else {
7457 // Try to minimise the number of instructions required to materialise the
7458 // offset calculation. Specifically, for fixed offsets, if masking out the
7459 // low 12 bits leaves a legal add immediate, we can realise the offset
7460 // calculation with a single add instruction. Whenever this is possible,
7461 // prefer this split.
7462 int64_t HighPart = Offset & ~0xFFF;
7463 int64_t LowPart = Offset & 0xFFF;
7464 int64_t LowScaled = LowPart / Scale;
7465 if (!IsMulVL && NewOffset >= 0 && LowPart % Scale == 0 &&
7466 MinOff <= LowScaled && LowScaled <= MaxOff &&
7468 NewOffset = LowScaled;
7469 Offset = HighPart;
7470 } else {
7471 // Default to a greedy split: take the memop immediate to be maximum /
7472 // minimum expressible offset and materialise the remainder.
7473 NewOffset = NewOffset < 0 ? MinOff : MaxOff;
7474 Offset = Offset - (NewOffset * Scale);
7475 }
7476 }
7477
7478 if (EmittableOffset)
7479 *EmittableOffset = NewOffset;
7480 if (OutUseUnscaledOp)
7481 *OutUseUnscaledOp = useUnscaledOp;
7482 if (OutUnscaledOp && UnscaledOp)
7483 *OutUnscaledOp = *UnscaledOp;
7484
7485 if (IsMulVL)
7486 SOffset = StackOffset::get(SOffset.getFixed(), Offset);
7487 else
7488 SOffset = StackOffset::get(Offset, SOffset.getScalable());
7490 (SOffset ? 0 : AArch64FrameOffsetIsLegal);
7491}
7492
7494 unsigned FrameReg, StackOffset &Offset,
7495 const AArch64InstrInfo *TII) {
7496 unsigned Opcode = MI.getOpcode();
7497 unsigned ImmIdx = FrameRegIdx + 1;
7498
7499 if (Opcode == AArch64::ADDSXri || Opcode == AArch64::ADDXri) {
7500 Offset += StackOffset::getFixed(MI.getOperand(ImmIdx).getImm());
7501 emitFrameOffset(*MI.getParent(), MI, MI.getDebugLoc(),
7502 MI.getOperand(0).getReg(), FrameReg, Offset, TII,
7503 MachineInstr::NoFlags, (Opcode == AArch64::ADDSXri));
7504 MI.eraseFromParent();
7505 Offset = StackOffset();
7506 return true;
7507 }
7508
7509 int64_t NewOffset;
7510 unsigned UnscaledOp;
7511 bool UseUnscaledOp;
7512 int Status = isAArch64FrameOffsetLegal(MI, Offset, &UseUnscaledOp,
7513 &UnscaledOp, &NewOffset);
7516 // Replace the FrameIndex with FrameReg.
7517 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
7518 if (UseUnscaledOp)
7519 MI.setDesc(TII->get(UnscaledOp));
7520
7521 MI.getOperand(ImmIdx).ChangeToImmediate(NewOffset);
7522 return !Offset;
7523 }
7524
7525 return false;
7526}
7527
7533
7534MCInst AArch64InstrInfo::getNop() const { return MCInstBuilder(AArch64::NOP); }
7535
7536// AArch64 supports MachineCombiner.
7537bool AArch64InstrInfo::useMachineCombiner() const { return true; }
7538
7539// True when Opc sets flag
7540static bool isCombineInstrSettingFlag(unsigned Opc) {
7541 switch (Opc) {
7542 case AArch64::ADDSWrr:
7543 case AArch64::ADDSWri:
7544 case AArch64::ADDSXrr:
7545 case AArch64::ADDSXri:
7546 case AArch64::SUBSWrr:
7547 case AArch64::SUBSXrr:
7548 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
7549 case AArch64::SUBSWri:
7550 case AArch64::SUBSXri:
7551 return true;
7552 default:
7553 break;
7554 }
7555 return false;
7556}
7557
7558// 32b Opcodes that can be combined with a MUL
7559static bool isCombineInstrCandidate32(unsigned Opc) {
7560 switch (Opc) {
7561 case AArch64::ADDWrr:
7562 case AArch64::ADDWri:
7563 case AArch64::SUBWrr:
7564 case AArch64::ADDSWrr:
7565 case AArch64::ADDSWri:
7566 case AArch64::SUBSWrr:
7567 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
7568 case AArch64::SUBWri:
7569 case AArch64::SUBSWri:
7570 return true;
7571 default:
7572 break;
7573 }
7574 return false;
7575}
7576
7577// 64b Opcodes that can be combined with a MUL
7578static bool isCombineInstrCandidate64(unsigned Opc) {
7579 switch (Opc) {
7580 case AArch64::ADDXrr:
7581 case AArch64::ADDXri:
7582 case AArch64::SUBXrr:
7583 case AArch64::ADDSXrr:
7584 case AArch64::ADDSXri:
7585 case AArch64::SUBSXrr:
7586 // Note: MSUB Wd,Wn,Wm,Wi -> Wd = Wi - WnxWm, not Wd=WnxWm - Wi.
7587 case AArch64::SUBXri:
7588 case AArch64::SUBSXri:
7589 case AArch64::ADDv8i8:
7590 case AArch64::ADDv16i8:
7591 case AArch64::ADDv4i16:
7592 case AArch64::ADDv8i16:
7593 case AArch64::ADDv2i32:
7594 case AArch64::ADDv4i32:
7595 case AArch64::SUBv8i8:
7596 case AArch64::SUBv16i8:
7597 case AArch64::SUBv4i16:
7598 case AArch64::SUBv8i16:
7599 case AArch64::SUBv2i32:
7600 case AArch64::SUBv4i32:
7601 return true;
7602 default:
7603 break;
7604 }
7605 return false;
7606}
7607
7608// FP Opcodes that can be combined with a FMUL.
7609static bool isCombineInstrCandidateFP(const MachineInstr &Inst) {
7610 switch (Inst.getOpcode()) {
7611 default:
7612 break;
7613 case AArch64::FADDHrr:
7614 case AArch64::FADDSrr:
7615 case AArch64::FADDDrr:
7616 case AArch64::FADDv4f16:
7617 case AArch64::FADDv8f16:
7618 case AArch64::FADDv2f32:
7619 case AArch64::FADDv2f64:
7620 case AArch64::FADDv4f32:
7621 case AArch64::FSUBHrr:
7622 case AArch64::FSUBSrr:
7623 case AArch64::FSUBDrr:
7624 case AArch64::FSUBv4f16:
7625 case AArch64::FSUBv8f16:
7626 case AArch64::FSUBv2f32:
7627 case AArch64::FSUBv2f64:
7628 case AArch64::FSUBv4f32:
7629 // We can fuse FADD/FSUB with FMUL, if FADD/FSUB has the contract fast-math
7630 // flag.
7631 return Inst.getFlag(MachineInstr::FmContract);
7632 }
7633 return false;
7634}
7635
7636// Opcodes that can be combined with a MUL
7640
7641//
7642// Utility routine that checks if \param MO is defined by an
7643// \param CombineOpc instruction in the basic block \param MBB
7645 unsigned CombineOpc, unsigned ZeroReg = 0,
7646 bool CheckZeroReg = false) {
7647 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
7648 MachineInstr *MI = nullptr;
7649
7650 if (MO.isReg() && MO.getReg().isVirtual())
7651 MI = MRI.getUniqueVRegDef(MO.getReg());
7652 // And it needs to be in the trace (otherwise, it won't have a depth).
7653 if (!MI || MI->getParent() != &MBB || MI->getOpcode() != CombineOpc)
7654 return false;
7655 // Must only used by the user we combine with.
7656 if (!MRI.hasOneNonDBGUse(MI->getOperand(0).getReg()))
7657 return false;
7658
7659 if (CheckZeroReg) {
7660 assert(MI->getNumOperands() >= 4 && MI->getOperand(0).isReg() &&
7661 MI->getOperand(1).isReg() && MI->getOperand(2).isReg() &&
7662 MI->getOperand(3).isReg() && "MAdd/MSub must have a least 4 regs");
7663 // The third input reg must be zero.
7664 if (MI->getOperand(3).getReg() != ZeroReg)
7665 return false;
7666 }
7667
7668 if (isCombineInstrSettingFlag(CombineOpc) &&
7669 MI->findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr, true) == -1)
7670 return false;
7671
7672 return true;
7673}
7674
7675//
7676// Is \param MO defined by an integer multiply and can be combined?
7678 unsigned MulOpc, unsigned ZeroReg) {
7679 return canCombine(MBB, MO, MulOpc, ZeroReg, true);
7680}
7681
7682//
7683// Is \param MO defined by a floating-point multiply and can be combined?
7685 unsigned MulOpc) {
7686 return canCombine(MBB, MO, MulOpc);
7687}
7688
7689// TODO: There are many more machine instruction opcodes to match:
7690// 1. Other data types (integer, vectors)
7691// 2. Other math / logic operations (xor, or)
7692// 3. Other forms of the same operation (intrinsics and other variants)
7693bool AArch64InstrInfo::isAssociativeAndCommutative(const MachineInstr &Inst,
7694 bool Invert) const {
7695 if (Invert)
7696 return false;
7697 switch (Inst.getOpcode()) {
7698 // == Floating-point types ==
7699 // -- Floating-point instructions --
7700 case AArch64::FADDHrr:
7701 case AArch64::FADDSrr:
7702 case AArch64::FADDDrr:
7703 case AArch64::FMULHrr:
7704 case AArch64::FMULSrr:
7705 case AArch64::FMULDrr:
7706 case AArch64::FMULX16:
7707 case AArch64::FMULX32:
7708 case AArch64::FMULX64:
7709 // -- Advanced SIMD instructions --
7710 case AArch64::FADDv4f16:
7711 case AArch64::FADDv8f16:
7712 case AArch64::FADDv2f32:
7713 case AArch64::FADDv4f32:
7714 case AArch64::FADDv2f64:
7715 case AArch64::FMULv4f16:
7716 case AArch64::FMULv8f16:
7717 case AArch64::FMULv2f32:
7718 case AArch64::FMULv4f32:
7719 case AArch64::FMULv2f64:
7720 case AArch64::FMULXv4f16:
7721 case AArch64::FMULXv8f16:
7722 case AArch64::FMULXv2f32:
7723 case AArch64::FMULXv4f32:
7724 case AArch64::FMULXv2f64:
7725 // -- SVE instructions --
7726 // Opcodes FMULX_ZZZ_? don't exist because there is no unpredicated FMULX
7727 // in the SVE instruction set (though there are predicated ones).
7728 case AArch64::FADD_ZZZ_H:
7729 case AArch64::FADD_ZZZ_S:
7730 case AArch64::FADD_ZZZ_D:
7731 case AArch64::FMUL_ZZZ_H:
7732 case AArch64::FMUL_ZZZ_S:
7733 case AArch64::FMUL_ZZZ_D:
7736
7737 // == Integer types ==
7738 // -- Base instructions --
7739 // Opcodes MULWrr and MULXrr don't exist because
7740 // `MUL <Wd>, <Wn>, <Wm>` and `MUL <Xd>, <Xn>, <Xm>` are aliases of
7741 // `MADD <Wd>, <Wn>, <Wm>, WZR` and `MADD <Xd>, <Xn>, <Xm>, XZR` respectively.
7742 // The machine-combiner does not support three-source-operands machine
7743 // instruction. So we cannot reassociate MULs.
7744 case AArch64::ADDWrr:
7745 case AArch64::ADDXrr:
7746 case AArch64::ANDWrr:
7747 case AArch64::ANDXrr:
7748 case AArch64::ORRWrr:
7749 case AArch64::ORRXrr:
7750 case AArch64::EORWrr:
7751 case AArch64::EORXrr:
7752 case AArch64::EONWrr:
7753 case AArch64::EONXrr:
7754 // -- Advanced SIMD instructions --
7755 // Opcodes MULv1i64 and MULv2i64 don't exist because there is no 64-bit MUL
7756 // in the Advanced SIMD instruction set.
7757 case AArch64::ADDv8i8:
7758 case AArch64::ADDv16i8:
7759 case AArch64::ADDv4i16:
7760 case AArch64::ADDv8i16:
7761 case AArch64::ADDv2i32:
7762 case AArch64::ADDv4i32:
7763 case AArch64::ADDv1i64:
7764 case AArch64::ADDv2i64:
7765 case AArch64::MULv8i8:
7766 case AArch64::MULv16i8:
7767 case AArch64::MULv4i16:
7768 case AArch64::MULv8i16:
7769 case AArch64::MULv2i32:
7770 case AArch64::MULv4i32:
7771 case AArch64::ANDv8i8:
7772 case AArch64::ANDv16i8:
7773 case AArch64::ORRv8i8:
7774 case AArch64::ORRv16i8:
7775 case AArch64::EORv8i8:
7776 case AArch64::EORv16i8:
7777 // -- SVE instructions --
7778 case AArch64::ADD_ZZZ_B:
7779 case AArch64::ADD_ZZZ_H:
7780 case AArch64::ADD_ZZZ_S:
7781 case AArch64::ADD_ZZZ_D:
7782 case AArch64::MUL_ZZZ_B:
7783 case AArch64::MUL_ZZZ_H:
7784 case AArch64::MUL_ZZZ_S:
7785 case AArch64::MUL_ZZZ_D:
7786 case AArch64::AND_ZZZ:
7787 case AArch64::ORR_ZZZ:
7788 case AArch64::EOR_ZZZ:
7789 return true;
7790
7791 default:
7792 return false;
7793 }
7794}
7795
7796/// Find instructions that can be turned into madd.
7798 SmallVectorImpl<unsigned> &Patterns) {
7799 unsigned Opc = Root.getOpcode();
7800 MachineBasicBlock &MBB = *Root.getParent();
7801 bool Found = false;
7802
7804 return false;
7806 int Cmp_NZCV =
7807 Root.findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr, true);
7808 // When NZCV is live bail out.
7809 if (Cmp_NZCV == -1)
7810 return false;
7811 unsigned NewOpc = convertToNonFlagSettingOpc(Root);
7812 // When opcode can't change bail out.
7813 // CHECKME: do we miss any cases for opcode conversion?
7814 if (NewOpc == Opc)
7815 return false;
7816 Opc = NewOpc;
7817 }
7818
7819 auto setFound = [&](int Opcode, int Operand, unsigned ZeroReg,
7820 unsigned Pattern) {
7821 if (canCombineWithMUL(MBB, Root.getOperand(Operand), Opcode, ZeroReg)) {
7822 Patterns.push_back(Pattern);
7823 Found = true;
7824 }
7825 };
7826
7827 auto setVFound = [&](int Opcode, int Operand, unsigned Pattern) {
7828 if (canCombine(MBB, Root.getOperand(Operand), Opcode)) {
7829 Patterns.push_back(Pattern);
7830 Found = true;
7831 }
7832 };
7833
7835
7836 switch (Opc) {
7837 default:
7838 break;
7839 case AArch64::ADDWrr:
7840 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
7841 "ADDWrr does not have register operands");
7842 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDW_OP1);
7843 setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULADDW_OP2);
7844 break;
7845 case AArch64::ADDXrr:
7846 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDX_OP1);
7847 setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULADDX_OP2);
7848 break;
7849 case AArch64::SUBWrr:
7850 setFound(AArch64::MADDWrrr, 2, AArch64::WZR, MCP::MULSUBW_OP2);
7851 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBW_OP1);
7852 break;
7853 case AArch64::SUBXrr:
7854 setFound(AArch64::MADDXrrr, 2, AArch64::XZR, MCP::MULSUBX_OP2);
7855 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBX_OP1);
7856 break;
7857 case AArch64::ADDWri:
7858 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULADDWI_OP1);
7859 break;
7860 case AArch64::ADDXri:
7861 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULADDXI_OP1);
7862 break;
7863 case AArch64::SUBWri:
7864 setFound(AArch64::MADDWrrr, 1, AArch64::WZR, MCP::MULSUBWI_OP1);
7865 break;
7866 case AArch64::SUBXri:
7867 setFound(AArch64::MADDXrrr, 1, AArch64::XZR, MCP::MULSUBXI_OP1);
7868 break;
7869 case AArch64::ADDv8i8:
7870 setVFound(AArch64::MULv8i8, 1, MCP::MULADDv8i8_OP1);
7871 setVFound(AArch64::MULv8i8, 2, MCP::MULADDv8i8_OP2);
7872 break;
7873 case AArch64::ADDv16i8:
7874 setVFound(AArch64::MULv16i8, 1, MCP::MULADDv16i8_OP1);
7875 setVFound(AArch64::MULv16i8, 2, MCP::MULADDv16i8_OP2);
7876 break;
7877 case AArch64::ADDv4i16:
7878 setVFound(AArch64::MULv4i16, 1, MCP::MULADDv4i16_OP1);
7879 setVFound(AArch64::MULv4i16, 2, MCP::MULADDv4i16_OP2);
7880 setVFound(AArch64::MULv4i16_indexed, 1, MCP::MULADDv4i16_indexed_OP1);
7881 setVFound(AArch64::MULv4i16_indexed, 2, MCP::MULADDv4i16_indexed_OP2);
7882 break;
7883 case AArch64::ADDv8i16:
7884 setVFound(AArch64::MULv8i16, 1, MCP::MULADDv8i16_OP1);
7885 setVFound(AArch64::MULv8i16, 2, MCP::MULADDv8i16_OP2);
7886 setVFound(AArch64::MULv8i16_indexed, 1, MCP::MULADDv8i16_indexed_OP1);
7887 setVFound(AArch64::MULv8i16_indexed, 2, MCP::MULADDv8i16_indexed_OP2);
7888 break;
7889 case AArch64::ADDv2i32:
7890 setVFound(AArch64::MULv2i32, 1, MCP::MULADDv2i32_OP1);
7891 setVFound(AArch64::MULv2i32, 2, MCP::MULADDv2i32_OP2);
7892 setVFound(AArch64::MULv2i32_indexed, 1, MCP::MULADDv2i32_indexed_OP1);
7893 setVFound(AArch64::MULv2i32_indexed, 2, MCP::MULADDv2i32_indexed_OP2);
7894 break;
7895 case AArch64::ADDv4i32:
7896 setVFound(AArch64::MULv4i32, 1, MCP::MULADDv4i32_OP1);
7897 setVFound(AArch64::MULv4i32, 2, MCP::MULADDv4i32_OP2);
7898 setVFound(AArch64::MULv4i32_indexed, 1, MCP::MULADDv4i32_indexed_OP1);
7899 setVFound(AArch64::MULv4i32_indexed, 2, MCP::MULADDv4i32_indexed_OP2);
7900 break;
7901 case AArch64::SUBv8i8:
7902 setVFound(AArch64::MULv8i8, 1, MCP::MULSUBv8i8_OP1);
7903 setVFound(AArch64::MULv8i8, 2, MCP::MULSUBv8i8_OP2);
7904 break;
7905 case AArch64::SUBv16i8:
7906 setVFound(AArch64::MULv16i8, 1, MCP::MULSUBv16i8_OP1);
7907 setVFound(AArch64::MULv16i8, 2, MCP::MULSUBv16i8_OP2);
7908 break;
7909 case AArch64::SUBv4i16:
7910 setVFound(AArch64::MULv4i16, 1, MCP::MULSUBv4i16_OP1);
7911 setVFound(AArch64::MULv4i16, 2, MCP::MULSUBv4i16_OP2);
7912 setVFound(AArch64::MULv4i16_indexed, 1, MCP::MULSUBv4i16_indexed_OP1);
7913 setVFound(AArch64::MULv4i16_indexed, 2, MCP::MULSUBv4i16_indexed_OP2);
7914 break;
7915 case AArch64::SUBv8i16:
7916 setVFound(AArch64::MULv8i16, 1, MCP::MULSUBv8i16_OP1);
7917 setVFound(AArch64::MULv8i16, 2, MCP::MULSUBv8i16_OP2);
7918 setVFound(AArch64::MULv8i16_indexed, 1, MCP::MULSUBv8i16_indexed_OP1);
7919 setVFound(AArch64::MULv8i16_indexed, 2, MCP::MULSUBv8i16_indexed_OP2);
7920 break;
7921 case AArch64::SUBv2i32:
7922 setVFound(AArch64::MULv2i32, 1, MCP::MULSUBv2i32_OP1);
7923 setVFound(AArch64::MULv2i32, 2, MCP::MULSUBv2i32_OP2);
7924 setVFound(AArch64::MULv2i32_indexed, 1, MCP::MULSUBv2i32_indexed_OP1);
7925 setVFound(AArch64::MULv2i32_indexed, 2, MCP::MULSUBv2i32_indexed_OP2);
7926 break;
7927 case AArch64::SUBv4i32:
7928 setVFound(AArch64::MULv4i32, 1, MCP::MULSUBv4i32_OP1);
7929 setVFound(AArch64::MULv4i32, 2, MCP::MULSUBv4i32_OP2);
7930 setVFound(AArch64::MULv4i32_indexed, 1, MCP::MULSUBv4i32_indexed_OP1);
7931 setVFound(AArch64::MULv4i32_indexed, 2, MCP::MULSUBv4i32_indexed_OP2);
7932 break;
7933 }
7934 return Found;
7935}
7936
7937bool AArch64InstrInfo::isAccumulationOpcode(unsigned Opcode) const {
7938 switch (Opcode) {
7939 default:
7940 break;
7941 case AArch64::UABALB_ZZZ_D:
7942 case AArch64::UABALB_ZZZ_H:
7943 case AArch64::UABALB_ZZZ_S:
7944 case AArch64::UABALT_ZZZ_D:
7945 case AArch64::UABALT_ZZZ_H:
7946 case AArch64::UABALT_ZZZ_S:
7947 case AArch64::SABALB_ZZZ_D:
7948 case AArch64::SABALB_ZZZ_S:
7949 case AArch64::SABALB_ZZZ_H:
7950 case AArch64::SABALT_ZZZ_D:
7951 case AArch64::SABALT_ZZZ_S:
7952 case AArch64::SABALT_ZZZ_H:
7953 case AArch64::UABALv16i8_v8i16:
7954 case AArch64::UABALv2i32_v2i64:
7955 case AArch64::UABALv4i16_v4i32:
7956 case AArch64::UABALv4i32_v2i64:
7957 case AArch64::UABALv8i16_v4i32:
7958 case AArch64::UABALv8i8_v8i16:
7959 case AArch64::UABAv16i8:
7960 case AArch64::UABAv2i32:
7961 case AArch64::UABAv4i16:
7962 case AArch64::UABAv4i32:
7963 case AArch64::UABAv8i16:
7964 case AArch64::UABAv8i8:
7965 case AArch64::SABALv16i8_v8i16:
7966 case AArch64::SABALv2i32_v2i64:
7967 case AArch64::SABALv4i16_v4i32:
7968 case AArch64::SABALv4i32_v2i64:
7969 case AArch64::SABALv8i16_v4i32:
7970 case AArch64::SABALv8i8_v8i16:
7971 case AArch64::SABAv16i8:
7972 case AArch64::SABAv2i32:
7973 case AArch64::SABAv4i16:
7974 case AArch64::SABAv4i32:
7975 case AArch64::SABAv8i16:
7976 case AArch64::SABAv8i8:
7977 return true;
7978 }
7979
7980 return false;
7981}
7982
7983unsigned AArch64InstrInfo::getAccumulationStartOpcode(
7984 unsigned AccumulationOpcode) const {
7985 switch (AccumulationOpcode) {
7986 default:
7987 llvm_unreachable("Unsupported accumulation Opcode!");
7988 case AArch64::UABALB_ZZZ_D:
7989 return AArch64::UABDLB_ZZZ_D;
7990 case AArch64::UABALB_ZZZ_H:
7991 return AArch64::UABDLB_ZZZ_H;
7992 case AArch64::UABALB_ZZZ_S:
7993 return AArch64::UABDLB_ZZZ_S;
7994 case AArch64::UABALT_ZZZ_D:
7995 return AArch64::UABDLT_ZZZ_D;
7996 case AArch64::UABALT_ZZZ_H:
7997 return AArch64::UABDLT_ZZZ_H;
7998 case AArch64::UABALT_ZZZ_S:
7999 return AArch64::UABDLT_ZZZ_S;
8000 case AArch64::UABALv16i8_v8i16:
8001 return AArch64::UABDLv16i8_v8i16;
8002 case AArch64::UABALv2i32_v2i64:
8003 return AArch64::UABDLv2i32_v2i64;
8004 case AArch64::UABALv4i16_v4i32:
8005 return AArch64::UABDLv4i16_v4i32;
8006 case AArch64::UABALv4i32_v2i64:
8007 return AArch64::UABDLv4i32_v2i64;
8008 case AArch64::UABALv8i16_v4i32:
8009 return AArch64::UABDLv8i16_v4i32;
8010 case AArch64::UABALv8i8_v8i16:
8011 return AArch64::UABDLv8i8_v8i16;
8012 case AArch64::UABAv16i8:
8013 return AArch64::UABDv16i8;
8014 case AArch64::UABAv2i32:
8015 return AArch64::UABDv2i32;
8016 case AArch64::UABAv4i16:
8017 return AArch64::UABDv4i16;
8018 case AArch64::UABAv4i32:
8019 return AArch64::UABDv4i32;
8020 case AArch64::UABAv8i16:
8021 return AArch64::UABDv8i16;
8022 case AArch64::UABAv8i8:
8023 return AArch64::UABDv8i8;
8024 case AArch64::SABALB_ZZZ_D:
8025 return AArch64::SABDLB_ZZZ_D;
8026 case AArch64::SABALB_ZZZ_S:
8027 return AArch64::SABDLB_ZZZ_S;
8028 case AArch64::SABALB_ZZZ_H:
8029 return AArch64::SABDLB_ZZZ_H;
8030 case AArch64::SABALT_ZZZ_D:
8031 return AArch64::SABDLT_ZZZ_D;
8032 case AArch64::SABALT_ZZZ_S:
8033 return AArch64::SABDLT_ZZZ_S;
8034 case AArch64::SABALT_ZZZ_H:
8035 return AArch64::SABDLT_ZZZ_H;
8036 case AArch64::SABALv16i8_v8i16:
8037 return AArch64::SABDLv16i8_v8i16;
8038 case AArch64::SABALv2i32_v2i64:
8039 return AArch64::SABDLv2i32_v2i64;
8040 case AArch64::SABALv4i16_v4i32:
8041 return AArch64::SABDLv4i16_v4i32;
8042 case AArch64::SABALv4i32_v2i64:
8043 return AArch64::SABDLv4i32_v2i64;
8044 case AArch64::SABALv8i16_v4i32:
8045 return AArch64::SABDLv8i16_v4i32;
8046 case AArch64::SABALv8i8_v8i16:
8047 return AArch64::SABDLv8i8_v8i16;
8048 case AArch64::SABAv16i8:
8049 return AArch64::SABDv16i8;
8050 case AArch64::SABAv2i32:
8051 return AArch64::SABDv2i32;
8052 case AArch64::SABAv4i16:
8053 return AArch64::SABDv4i16;
8054 case AArch64::SABAv4i32:
8055 return AArch64::SABDv4i32;
8056 case AArch64::SABAv8i16:
8057 return AArch64::SABDv8i16;
8058 case AArch64::SABAv8i8:
8059 return AArch64::SABDv8i8;
8060 }
8061}
8062
8063/// Floating-Point Support
8064
8065/// Find instructions that can be turned into madd.
8067 SmallVectorImpl<unsigned> &Patterns) {
8068
8069 if (!isCombineInstrCandidateFP(Root))
8070 return false;
8071
8072 MachineBasicBlock &MBB = *Root.getParent();
8073 bool Found = false;
8074
8075 auto Match = [&](int Opcode, int Operand, unsigned Pattern) -> bool {
8076 if (canCombineWithFMUL(MBB, Root.getOperand(Operand), Opcode)) {
8077 Patterns.push_back(Pattern);
8078 return true;
8079 }
8080 return false;
8081 };
8082
8084
8085 switch (Root.getOpcode()) {
8086 default:
8087 assert(false && "Unsupported FP instruction in combiner\n");
8088 break;
8089 case AArch64::FADDHrr:
8090 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
8091 "FADDHrr does not have register operands");
8092
8093 Found = Match(AArch64::FMULHrr, 1, MCP::FMULADDH_OP1);
8094 Found |= Match(AArch64::FMULHrr, 2, MCP::FMULADDH_OP2);
8095 break;
8096 case AArch64::FADDSrr:
8097 assert(Root.getOperand(1).isReg() && Root.getOperand(2).isReg() &&
8098 "FADDSrr does not have register operands");
8099
8100 Found |= Match(AArch64::FMULSrr, 1, MCP::FMULADDS_OP1) ||
8101 Match(AArch64::FMULv1i32_indexed, 1, MCP::FMLAv1i32_indexed_OP1);
8102
8103 Found |= Match(AArch64::FMULSrr, 2, MCP::FMULADDS_OP2) ||
8104 Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLAv1i32_indexed_OP2);
8105 break;
8106 case AArch64::FADDDrr:
8107 Found |= Match(AArch64::FMULDrr, 1, MCP::FMULADDD_OP1) ||
8108 Match(AArch64::FMULv1i64_indexed, 1, MCP::FMLAv1i64_indexed_OP1);
8109
8110 Found |= Match(AArch64::FMULDrr, 2, MCP::FMULADDD_OP2) ||
8111 Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLAv1i64_indexed_OP2);
8112 break;
8113 case AArch64::FADDv4f16:
8114 Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLAv4i16_indexed_OP1) ||
8115 Match(AArch64::FMULv4f16, 1, MCP::FMLAv4f16_OP1);
8116
8117 Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLAv4i16_indexed_OP2) ||
8118 Match(AArch64::FMULv4f16, 2, MCP::FMLAv4f16_OP2);
8119 break;
8120 case AArch64::FADDv8f16:
8121 Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLAv8i16_indexed_OP1) ||
8122 Match(AArch64::FMULv8f16, 1, MCP::FMLAv8f16_OP1);
8123
8124 Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLAv8i16_indexed_OP2) ||
8125 Match(AArch64::FMULv8f16, 2, MCP::FMLAv8f16_OP2);
8126 break;
8127 case AArch64::FADDv2f32:
8128 Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLAv2i32_indexed_OP1) ||
8129 Match(AArch64::FMULv2f32, 1, MCP::FMLAv2f32_OP1);
8130
8131 Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLAv2i32_indexed_OP2) ||
8132 Match(AArch64::FMULv2f32, 2, MCP::FMLAv2f32_OP2);
8133 break;
8134 case AArch64::FADDv2f64:
8135 Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLAv2i64_indexed_OP1) ||
8136 Match(AArch64::FMULv2f64, 1, MCP::FMLAv2f64_OP1);
8137
8138 Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLAv2i64_indexed_OP2) ||
8139 Match(AArch64::FMULv2f64, 2, MCP::FMLAv2f64_OP2);
8140 break;
8141 case AArch64::FADDv4f32:
8142 Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLAv4i32_indexed_OP1) ||
8143 Match(AArch64::FMULv4f32, 1, MCP::FMLAv4f32_OP1);
8144
8145 Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLAv4i32_indexed_OP2) ||
8146 Match(AArch64::FMULv4f32, 2, MCP::FMLAv4f32_OP2);
8147 break;
8148 case AArch64::FSUBHrr:
8149 Found = Match(AArch64::FMULHrr, 1, MCP::FMULSUBH_OP1);
8150 Found |= Match(AArch64::FMULHrr, 2, MCP::FMULSUBH_OP2);
8151 Found |= Match(AArch64::FNMULHrr, 1, MCP::FNMULSUBH_OP1);
8152 break;
8153 case AArch64::FSUBSrr:
8154 Found = Match(AArch64::FMULSrr, 1, MCP::FMULSUBS_OP1);
8155
8156 Found |= Match(AArch64::FMULSrr, 2, MCP::FMULSUBS_OP2) ||
8157 Match(AArch64::FMULv1i32_indexed, 2, MCP::FMLSv1i32_indexed_OP2);
8158
8159 Found |= Match(AArch64::FNMULSrr, 1, MCP::FNMULSUBS_OP1);
8160 break;
8161 case AArch64::FSUBDrr:
8162 Found = Match(AArch64::FMULDrr, 1, MCP::FMULSUBD_OP1);
8163
8164 Found |= Match(AArch64::FMULDrr, 2, MCP::FMULSUBD_OP2) ||
8165 Match(AArch64::FMULv1i64_indexed, 2, MCP::FMLSv1i64_indexed_OP2);
8166
8167 Found |= Match(AArch64::FNMULDrr, 1, MCP::FNMULSUBD_OP1);
8168 break;
8169 case AArch64::FSUBv4f16:
8170 Found |= Match(AArch64::FMULv4i16_indexed, 2, MCP::FMLSv4i16_indexed_OP2) ||
8171 Match(AArch64::FMULv4f16, 2, MCP::FMLSv4f16_OP2);
8172
8173 Found |= Match(AArch64::FMULv4i16_indexed, 1, MCP::FMLSv4i16_indexed_OP1) ||
8174 Match(AArch64::FMULv4f16, 1, MCP::FMLSv4f16_OP1);
8175 break;
8176 case AArch64::FSUBv8f16:
8177 Found |= Match(AArch64::FMULv8i16_indexed, 2, MCP::FMLSv8i16_indexed_OP2) ||
8178 Match(AArch64::FMULv8f16, 2, MCP::FMLSv8f16_OP2);
8179
8180 Found |= Match(AArch64::FMULv8i16_indexed, 1, MCP::FMLSv8i16_indexed_OP1) ||
8181 Match(AArch64::FMULv8f16, 1, MCP::FMLSv8f16_OP1);
8182 break;
8183 case AArch64::FSUBv2f32:
8184 Found |= Match(AArch64::FMULv2i32_indexed, 2, MCP::FMLSv2i32_indexed_OP2) ||
8185 Match(AArch64::FMULv2f32, 2, MCP::FMLSv2f32_OP2);
8186
8187 Found |= Match(AArch64::FMULv2i32_indexed, 1, MCP::FMLSv2i32_indexed_OP1) ||
8188 Match(AArch64::FMULv2f32, 1, MCP::FMLSv2f32_OP1);
8189 break;
8190 case AArch64::FSUBv2f64:
8191 Found |= Match(AArch64::FMULv2i64_indexed, 2, MCP::FMLSv2i64_indexed_OP2) ||
8192 Match(AArch64::FMULv2f64, 2, MCP::FMLSv2f64_OP2);
8193
8194 Found |= Match(AArch64::FMULv2i64_indexed, 1, MCP::FMLSv2i64_indexed_OP1) ||
8195 Match(AArch64::FMULv2f64, 1, MCP::FMLSv2f64_OP1);
8196 break;
8197 case AArch64::FSUBv4f32:
8198 Found |= Match(AArch64::FMULv4i32_indexed, 2, MCP::FMLSv4i32_indexed_OP2) ||
8199 Match(AArch64::FMULv4f32, 2, MCP::FMLSv4f32_OP2);
8200
8201 Found |= Match(AArch64::FMULv4i32_indexed, 1, MCP::FMLSv4i32_indexed_OP1) ||
8202 Match(AArch64::FMULv4f32, 1, MCP::FMLSv4f32_OP1);
8203 break;
8204 }
8205 return Found;
8206}
8207
8209 SmallVectorImpl<unsigned> &Patterns) {
8210 MachineBasicBlock &MBB = *Root.getParent();
8211 bool Found = false;
8212
8213 auto Match = [&](unsigned Opcode, int Operand, unsigned Pattern) -> bool {
8214 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
8215 MachineOperand &MO = Root.getOperand(Operand);
8216 MachineInstr *MI = nullptr;
8217 if (MO.isReg() && MO.getReg().isVirtual())
8218 MI = MRI.getUniqueVRegDef(MO.getReg());
8219 // Ignore No-op COPYs in FMUL(COPY(DUP(..)))
8220 if (MI && MI->getOpcode() == TargetOpcode::COPY &&
8221 MI->getOperand(1).getReg().isVirtual())
8222 MI = MRI.getUniqueVRegDef(MI->getOperand(1).getReg());
8223 if (MI && MI->getOpcode() == Opcode) {
8224 Patterns.push_back(Pattern);
8225 return true;
8226 }
8227 return false;
8228 };
8229
8231
8232 switch (Root.getOpcode()) {
8233 default:
8234 return false;
8235 case AArch64::FMULv2f32:
8236 Found = Match(AArch64::DUPv2i32lane, 1, MCP::FMULv2i32_indexed_OP1);
8237 Found |= Match(AArch64::DUPv2i32lane, 2, MCP::FMULv2i32_indexed_OP2);
8238 break;
8239 case AArch64::FMULv2f64:
8240 Found = Match(AArch64::DUPv2i64lane, 1, MCP::FMULv2i64_indexed_OP1);
8241 Found |= Match(AArch64::DUPv2i64lane, 2, MCP::FMULv2i64_indexed_OP2);
8242 break;
8243 case AArch64::FMULv4f16:
8244 Found = Match(AArch64::DUPv4i16lane, 1, MCP::FMULv4i16_indexed_OP1);
8245 Found |= Match(AArch64::DUPv4i16lane, 2, MCP::FMULv4i16_indexed_OP2);
8246 break;
8247 case AArch64::FMULv4f32:
8248 Found = Match(AArch64::DUPv4i32lane, 1, MCP::FMULv4i32_indexed_OP1);
8249 Found |= Match(AArch64::DUPv4i32lane, 2, MCP::FMULv4i32_indexed_OP2);
8250 break;
8251 case AArch64::FMULv8f16:
8252 Found = Match(AArch64::DUPv8i16lane, 1, MCP::FMULv8i16_indexed_OP1);
8253 Found |= Match(AArch64::DUPv8i16lane, 2, MCP::FMULv8i16_indexed_OP2);
8254 break;
8255 }
8256
8257 return Found;
8258}
8259
8261 SmallVectorImpl<unsigned> &Patterns) {
8262 unsigned Opc = Root.getOpcode();
8263 MachineBasicBlock &MBB = *Root.getParent();
8264 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
8265
8266 auto Match = [&](unsigned Opcode, unsigned Pattern) -> bool {
8267 MachineOperand &MO = Root.getOperand(1);
8269 if (MI != nullptr && (MI->getOpcode() == Opcode) &&
8270 MRI.hasOneNonDBGUse(MI->getOperand(0).getReg()) &&
8274 MI->getFlag(MachineInstr::MIFlag::FmNsz)) {
8275 Patterns.push_back(Pattern);
8276 return true;
8277 }
8278 return false;
8279 };
8280
8281 switch (Opc) {
8282 default:
8283 break;
8284 case AArch64::FNEGDr:
8285 return Match(AArch64::FMADDDrrr, AArch64MachineCombinerPattern::FNMADD);
8286 case AArch64::FNEGSr:
8287 return Match(AArch64::FMADDSrrr, AArch64MachineCombinerPattern::FNMADD);
8288 }
8289
8290 return false;
8291}
8292
8293/// Return true when a code sequence can improve throughput. It
8294/// should be called only for instructions in loops.
8295/// \param Pattern - combiner pattern
8297 switch (Pattern) {
8298 default:
8299 break;
8405 return true;
8406 } // end switch (Pattern)
8407 return false;
8408}
8409
8410/// Find other MI combine patterns.
8412 SmallVectorImpl<unsigned> &Patterns) {
8413 // A - (B + C) ==> (A - B) - C or (A - C) - B
8414 unsigned Opc = Root.getOpcode();
8415 MachineBasicBlock &MBB = *Root.getParent();
8416
8417 switch (Opc) {
8418 case AArch64::SUBWrr:
8419 case AArch64::SUBSWrr:
8420 case AArch64::SUBXrr:
8421 case AArch64::SUBSXrr:
8422 // Found candidate root.
8423 break;
8424 default:
8425 return false;
8426 }
8427
8429 Root.findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr, true) ==
8430 -1)
8431 return false;
8432
8433 if (canCombine(MBB, Root.getOperand(2), AArch64::ADDWrr) ||
8434 canCombine(MBB, Root.getOperand(2), AArch64::ADDSWrr) ||
8435 canCombine(MBB, Root.getOperand(2), AArch64::ADDXrr) ||
8436 canCombine(MBB, Root.getOperand(2), AArch64::ADDSXrr)) {
8439 return true;
8440 }
8441
8442 return false;
8443}
8444
8445/// Check if the given instruction forms a gather load pattern that can be
8446/// optimized for better Memory-Level Parallelism (MLP). This function
8447/// identifies chains of NEON lane load instructions that load data from
8448/// different memory addresses into individual lanes of a 128-bit vector
8449/// register, then attempts to split the pattern into parallel loads to break
8450/// the serial dependency between instructions.
8451///
8452/// Pattern Matched:
8453/// Initial scalar load -> SUBREG_TO_REG (lane 0) -> LD1i* (lane 1) ->
8454/// LD1i* (lane 2) -> ... -> LD1i* (lane N-1, Root)
8455///
8456/// Transformed Into:
8457/// Two parallel vector loads using fewer lanes each, followed by ZIP1v2i64
8458/// to combine the results, enabling better memory-level parallelism.
8459///
8460/// Supported Element Types:
8461/// - 32-bit elements (LD1i32, 4 lanes total)
8462/// - 16-bit elements (LD1i16, 8 lanes total)
8463/// - 8-bit elements (LD1i8, 16 lanes total)
8465 SmallVectorImpl<unsigned> &Patterns,
8466 unsigned LoadLaneOpCode, unsigned NumLanes) {
8467 const MachineFunction *MF = Root.getMF();
8468
8469 // Early exit if optimizing for size.
8470 if (MF->getFunction().hasMinSize())
8471 return false;
8472
8473 const MachineRegisterInfo &MRI = MF->getRegInfo();
8475
8476 // The root of the pattern must load into the last lane of the vector.
8477 if (Root.getOperand(2).getImm() != NumLanes - 1)
8478 return false;
8479
8480 // Check that we have load into all lanes except lane 0.
8481 // For each load we also want to check that:
8482 // 1. It has a single non-debug use (since we will be replacing the virtual
8483 // register)
8484 // 2. That the addressing mode only uses a single pointer operand
8485 auto *CurrInstr = MRI.getUniqueVRegDef(Root.getOperand(1).getReg());
8486 auto Range = llvm::seq<unsigned>(1, NumLanes - 1);
8487 SmallSet<unsigned, 16> RemainingLanes(Range.begin(), Range.end());
8489 while (!RemainingLanes.empty() && CurrInstr &&
8490 CurrInstr->getOpcode() == LoadLaneOpCode &&
8491 MRI.hasOneNonDBGUse(CurrInstr->getOperand(0).getReg()) &&
8492 CurrInstr->getNumOperands() == 4) {
8493 RemainingLanes.erase(CurrInstr->getOperand(2).getImm());
8494 LoadInstrs.push_back(CurrInstr);
8495 CurrInstr = MRI.getUniqueVRegDef(CurrInstr->getOperand(1).getReg());
8496 }
8497
8498 // Check that we have found a match for lanes N-1.. 1.
8499 if (!RemainingLanes.empty())
8500 return false;
8501
8502 // Match the SUBREG_TO_REG sequence.
8503 if (CurrInstr->getOpcode() != TargetOpcode::SUBREG_TO_REG)
8504 return false;
8505
8506 // Verify that the subreg to reg loads an integer into the first lane.
8507 auto Lane0LoadReg = CurrInstr->getOperand(1).getReg();
8508 unsigned SingleLaneSizeInBits = 128 / NumLanes;
8509 if (TRI->getRegSizeInBits(Lane0LoadReg, MRI) != SingleLaneSizeInBits)
8510 return false;
8511
8512 // Verify that it also has a single non debug use.
8513 if (!MRI.hasOneNonDBGUse(Lane0LoadReg))
8514 return false;
8515
8516 LoadInstrs.push_back(MRI.getUniqueVRegDef(Lane0LoadReg));
8517
8518 // If there is any chance of aliasing, do not apply the pattern.
8519 // Walk backward through the MBB starting from Root.
8520 // Exit early if we've encountered all load instructions or hit the search
8521 // limit.
8522 auto MBBItr = Root.getIterator();
8523 unsigned RemainingSteps = GatherOptSearchLimit;
8524 SmallPtrSet<const MachineInstr *, 16> RemainingLoadInstrs;
8525 RemainingLoadInstrs.insert(LoadInstrs.begin(), LoadInstrs.end());
8526 const MachineBasicBlock *MBB = Root.getParent();
8527
8528 for (; MBBItr != MBB->begin() && RemainingSteps > 0 &&
8529 !RemainingLoadInstrs.empty();
8530 --MBBItr, --RemainingSteps) {
8531 const MachineInstr &CurrInstr = *MBBItr;
8532
8533 // Remove this instruction from remaining loads if it's one we're tracking.
8534 RemainingLoadInstrs.erase(&CurrInstr);
8535
8536 // Check for potential aliasing with any of the load instructions to
8537 // optimize.
8538 if (CurrInstr.isLoadFoldBarrier())
8539 return false;
8540 }
8541
8542 // If we hit the search limit without finding all load instructions,
8543 // don't match the pattern.
8544 if (RemainingSteps == 0 && !RemainingLoadInstrs.empty())
8545 return false;
8546
8547 switch (NumLanes) {
8548 case 4:
8550 break;
8551 case 8:
8553 break;
8554 case 16:
8556 break;
8557 default:
8558 llvm_unreachable("Got bad number of lanes for gather pattern.");
8559 }
8560
8561 return true;
8562}
8563
8564/// Search for patterns of LD instructions we can optimize.
8566 SmallVectorImpl<unsigned> &Patterns) {
8567
8568 // The pattern searches for loads into single lanes.
8569 switch (Root.getOpcode()) {
8570 case AArch64::LD1i32:
8571 return getGatherLanePattern(Root, Patterns, Root.getOpcode(), 4);
8572 case AArch64::LD1i16:
8573 return getGatherLanePattern(Root, Patterns, Root.getOpcode(), 8);
8574 case AArch64::LD1i8:
8575 return getGatherLanePattern(Root, Patterns, Root.getOpcode(), 16);
8576 default:
8577 return false;
8578 }
8579}
8580
8581/// Generate optimized instruction sequence for gather load patterns to improve
8582/// Memory-Level Parallelism (MLP). This function transforms a chain of
8583/// sequential NEON lane loads into parallel vector loads that can execute
8584/// concurrently.
8585static void
8589 DenseMap<Register, unsigned> &InstrIdxForVirtReg,
8590 unsigned Pattern, unsigned NumLanes) {
8591 MachineFunction &MF = *Root.getParent()->getParent();
8592 MachineRegisterInfo &MRI = MF.getRegInfo();
8594
8595 // Gather the initial load instructions to build the pattern.
8596 SmallVector<MachineInstr *, 16> LoadToLaneInstrs;
8597 MachineInstr *CurrInstr = &Root;
8598 for (unsigned i = 0; i < NumLanes - 1; ++i) {
8599 LoadToLaneInstrs.push_back(CurrInstr);
8600 CurrInstr = MRI.getUniqueVRegDef(CurrInstr->getOperand(1).getReg());
8601 }
8602
8603 // Sort the load instructions according to the lane.
8604 llvm::sort(LoadToLaneInstrs,
8605 [](const MachineInstr *A, const MachineInstr *B) {
8606 return A->getOperand(2).getImm() > B->getOperand(2).getImm();
8607 });
8608
8609 MachineInstr *SubregToReg = CurrInstr;
8610 LoadToLaneInstrs.push_back(
8611 MRI.getUniqueVRegDef(SubregToReg->getOperand(1).getReg()));
8612 auto LoadToLaneInstrsAscending = llvm::reverse(LoadToLaneInstrs);
8613
8614 const TargetRegisterClass *FPR128RegClass =
8615 MRI.getRegClass(Root.getOperand(0).getReg());
8616
8617 // Helper lambda to create a LD1 instruction.
8618 auto CreateLD1Instruction = [&](MachineInstr *OriginalInstr,
8619 Register SrcRegister, unsigned Lane,
8620 Register OffsetRegister,
8621 bool OffsetRegisterKillState) {
8622 auto NewRegister = MRI.createVirtualRegister(FPR128RegClass);
8623 MachineInstrBuilder LoadIndexIntoRegister =
8624 BuildMI(MF, MIMetadata(*OriginalInstr), TII->get(Root.getOpcode()),
8625 NewRegister)
8626 .addReg(SrcRegister)
8627 .addImm(Lane)
8628 .addReg(OffsetRegister, getKillRegState(OffsetRegisterKillState))
8629 .setMemRefs(OriginalInstr->memoperands());
8630 InstrIdxForVirtReg.insert(std::make_pair(NewRegister, InsInstrs.size()));
8631 InsInstrs.push_back(LoadIndexIntoRegister);
8632 return NewRegister;
8633 };
8634
8635 // Helper to create load instruction based on the NumLanes in the NEON
8636 // register we are rewriting.
8637 auto CreateLDRInstruction =
8638 [&](unsigned NumLanes, Register DestReg, Register OffsetReg,
8640 unsigned Opcode;
8641 switch (NumLanes) {
8642 case 4:
8643 Opcode = AArch64::LDRSui;
8644 break;
8645 case 8:
8646 Opcode = AArch64::LDRHui;
8647 break;
8648 case 16:
8649 Opcode = AArch64::LDRBui;
8650 break;
8651 default:
8653 "Got unsupported number of lanes in machine-combiner gather pattern");
8654 }
8655 // Immediate offset load
8656 return BuildMI(MF, MIMetadata(Root), TII->get(Opcode), DestReg)
8657 .addReg(OffsetReg)
8658 .addImm(0)
8659 .setMemRefs(MMOs);
8660 };
8661
8662 // Load the remaining lanes into register 0.
8663 auto LanesToLoadToReg0 =
8664 llvm::make_range(LoadToLaneInstrsAscending.begin() + 1,
8665 LoadToLaneInstrsAscending.begin() + NumLanes / 2);
8666 Register PrevReg = SubregToReg->getOperand(0).getReg();
8667 for (auto [Index, LoadInstr] : llvm::enumerate(LanesToLoadToReg0)) {
8668 const MachineOperand &OffsetRegOperand = LoadInstr->getOperand(3);
8669 PrevReg = CreateLD1Instruction(LoadInstr, PrevReg, Index + 1,
8670 OffsetRegOperand.getReg(),
8671 OffsetRegOperand.isKill());
8672 DelInstrs.push_back(LoadInstr);
8673 }
8674 Register LastLoadReg0 = PrevReg;
8675
8676 // First load into register 1. Perform an integer load to zero out the upper
8677 // lanes in a single instruction.
8678 MachineInstr *Lane0Load = *LoadToLaneInstrsAscending.begin();
8679 MachineInstr *OriginalSplitLoad =
8680 *std::next(LoadToLaneInstrsAscending.begin(), NumLanes / 2);
8681 Register DestRegForMiddleIndex = MRI.createVirtualRegister(
8682 MRI.getRegClass(Lane0Load->getOperand(0).getReg()));
8683
8684 const MachineOperand &OriginalSplitToLoadOffsetOperand =
8685 OriginalSplitLoad->getOperand(3);
8686 MachineInstrBuilder MiddleIndexLoadInstr =
8687 CreateLDRInstruction(NumLanes, DestRegForMiddleIndex,
8688 OriginalSplitToLoadOffsetOperand.getReg(),
8689 OriginalSplitLoad->memoperands());
8690
8691 InstrIdxForVirtReg.insert(
8692 std::make_pair(DestRegForMiddleIndex, InsInstrs.size()));
8693 InsInstrs.push_back(MiddleIndexLoadInstr);
8694 DelInstrs.push_back(OriginalSplitLoad);
8695
8696 // Subreg To Reg instruction for register 1.
8697 Register DestRegForSubregToReg = MRI.createVirtualRegister(FPR128RegClass);
8698 unsigned SubregType;
8699 switch (NumLanes) {
8700 case 4:
8701 SubregType = AArch64::ssub;
8702 break;
8703 case 8:
8704 SubregType = AArch64::hsub;
8705 break;
8706 case 16:
8707 SubregType = AArch64::bsub;
8708 break;
8709 default:
8711 "Got invalid NumLanes for machine-combiner gather pattern");
8712 }
8713
8714 auto SubRegToRegInstr =
8715 BuildMI(MF, MIMetadata(Root), TII->get(SubregToReg->getOpcode()),
8716 DestRegForSubregToReg)
8717 .addReg(DestRegForMiddleIndex, getKillRegState(true))
8718 .addImm(SubregType);
8719 InstrIdxForVirtReg.insert(
8720 std::make_pair(DestRegForSubregToReg, InsInstrs.size()));
8721 InsInstrs.push_back(SubRegToRegInstr);
8722
8723 // Load remaining lanes into register 1.
8724 auto LanesToLoadToReg1 =
8725 llvm::make_range(LoadToLaneInstrsAscending.begin() + NumLanes / 2 + 1,
8726 LoadToLaneInstrsAscending.end());
8727 PrevReg = SubRegToRegInstr->getOperand(0).getReg();
8728 for (auto [Index, LoadInstr] : llvm::enumerate(LanesToLoadToReg1)) {
8729 const MachineOperand &OffsetRegOperand = LoadInstr->getOperand(3);
8730 PrevReg = CreateLD1Instruction(LoadInstr, PrevReg, Index + 1,
8731 OffsetRegOperand.getReg(),
8732 OffsetRegOperand.isKill());
8733
8734 // Do not add the last reg to DelInstrs - it will be removed later.
8735 if (Index == NumLanes / 2 - 2) {
8736 break;
8737 }
8738 DelInstrs.push_back(LoadInstr);
8739 }
8740 Register LastLoadReg1 = PrevReg;
8741
8742 // Create the final zip instruction to combine the results.
8743 MachineInstrBuilder ZipInstr =
8744 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::ZIP1v2i64),
8745 Root.getOperand(0).getReg())
8746 .addReg(LastLoadReg0)
8747 .addReg(LastLoadReg1);
8748 InsInstrs.push_back(ZipInstr);
8749}
8750
8764
8765/// Return true when there is potentially a faster code sequence for an
8766/// instruction chain ending in \p Root. All potential patterns are listed in
8767/// the \p Pattern vector. Pattern should be sorted in priority order since the
8768/// pattern evaluator stops checking as soon as it finds a faster sequence.
8769
8770bool AArch64InstrInfo::getMachineCombinerPatterns(
8771 MachineInstr &Root, SmallVectorImpl<unsigned> &Patterns,
8772 bool DoRegPressureReduce) const {
8773 // Integer patterns
8774 if (getMaddPatterns(Root, Patterns))
8775 return true;
8776 // Floating point patterns
8777 if (getFMULPatterns(Root, Patterns))
8778 return true;
8779 if (getFMAPatterns(Root, Patterns))
8780 return true;
8781 if (getFNEGPatterns(Root, Patterns))
8782 return true;
8783
8784 // Other patterns
8785 if (getMiscPatterns(Root, Patterns))
8786 return true;
8787
8788 // Load patterns
8789 if (getLoadPatterns(Root, Patterns))
8790 return true;
8791
8792 return TargetInstrInfo::getMachineCombinerPatterns(Root, Patterns,
8793 DoRegPressureReduce);
8794}
8795
8797/// genFusedMultiply - Generate fused multiply instructions.
8798/// This function supports both integer and floating point instructions.
8799/// A typical example:
8800/// F|MUL I=A,B,0
8801/// F|ADD R,I,C
8802/// ==> F|MADD R,A,B,C
8803/// \param MF Containing MachineFunction
8804/// \param MRI Register information
8805/// \param TII Target information
8806/// \param Root is the F|ADD instruction
8807/// \param [out] InsInstrs is a vector of machine instructions and will
8808/// contain the generated madd instruction
8809/// \param IdxMulOpd is index of operand in Root that is the result of
8810/// the F|MUL. In the example above IdxMulOpd is 1.
8811/// \param MaddOpc the opcode fo the f|madd instruction
8812/// \param RC Register class of operands
8813/// \param kind of fma instruction (addressing mode) to be generated
8814/// \param ReplacedAddend is the result register from the instruction
8815/// replacing the non-combined operand, if any.
8816static MachineInstr *
8818 const TargetInstrInfo *TII, MachineInstr &Root,
8819 SmallVectorImpl<MachineInstr *> &InsInstrs, unsigned IdxMulOpd,
8820 unsigned MaddOpc, const TargetRegisterClass *RC,
8822 const Register *ReplacedAddend = nullptr) {
8823 assert(IdxMulOpd == 1 || IdxMulOpd == 2);
8824
8825 unsigned IdxOtherOpd = IdxMulOpd == 1 ? 2 : 1;
8826 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
8827 Register ResultReg = Root.getOperand(0).getReg();
8828 Register SrcReg0 = MUL->getOperand(1).getReg();
8829 bool Src0IsKill = MUL->getOperand(1).isKill();
8830 Register SrcReg1 = MUL->getOperand(2).getReg();
8831 bool Src1IsKill = MUL->getOperand(2).isKill();
8832
8833 Register SrcReg2;
8834 bool Src2IsKill;
8835 if (ReplacedAddend) {
8836 // If we just generated a new addend, we must be it's only use.
8837 SrcReg2 = *ReplacedAddend;
8838 Src2IsKill = true;
8839 } else {
8840 SrcReg2 = Root.getOperand(IdxOtherOpd).getReg();
8841 Src2IsKill = Root.getOperand(IdxOtherOpd).isKill();
8842 }
8843
8844 if (ResultReg.isVirtual())
8845 MRI.constrainRegClass(ResultReg, RC);
8846 if (SrcReg0.isVirtual())
8847 MRI.constrainRegClass(SrcReg0, RC);
8848 if (SrcReg1.isVirtual())
8849 MRI.constrainRegClass(SrcReg1, RC);
8850 if (SrcReg2.isVirtual())
8851 MRI.constrainRegClass(SrcReg2, RC);
8852
8854 if (kind == FMAInstKind::Default)
8855 MIB = BuildMI(MF, MIMetadata(Root), TII->get(MaddOpc), ResultReg)
8856 .addReg(SrcReg0, getKillRegState(Src0IsKill))
8857 .addReg(SrcReg1, getKillRegState(Src1IsKill))
8858 .addReg(SrcReg2, getKillRegState(Src2IsKill));
8859 else if (kind == FMAInstKind::Indexed)
8860 MIB = BuildMI(MF, MIMetadata(Root), TII->get(MaddOpc), ResultReg)
8861 .addReg(SrcReg2, getKillRegState(Src2IsKill))
8862 .addReg(SrcReg0, getKillRegState(Src0IsKill))
8863 .addReg(SrcReg1, getKillRegState(Src1IsKill))
8864 .addImm(MUL->getOperand(3).getImm());
8865 else if (kind == FMAInstKind::Accumulator)
8866 MIB = BuildMI(MF, MIMetadata(Root), TII->get(MaddOpc), ResultReg)
8867 .addReg(SrcReg2, getKillRegState(Src2IsKill))
8868 .addReg(SrcReg0, getKillRegState(Src0IsKill))
8869 .addReg(SrcReg1, getKillRegState(Src1IsKill));
8870 else
8871 assert(false && "Invalid FMA instruction kind \n");
8872 // Insert the MADD (MADD, FMA, FMS, FMLA, FMSL)
8873 InsInstrs.push_back(MIB);
8874 return MUL;
8875}
8876
8877static MachineInstr *
8879 const TargetInstrInfo *TII, MachineInstr &Root,
8881 MachineInstr *MAD = MRI.getUniqueVRegDef(Root.getOperand(1).getReg());
8882
8883 unsigned Opc = 0;
8884 const TargetRegisterClass *RC = MRI.getRegClass(MAD->getOperand(0).getReg());
8885 if (AArch64::FPR32RegClass.hasSubClassEq(RC))
8886 Opc = AArch64::FNMADDSrrr;
8887 else if (AArch64::FPR64RegClass.hasSubClassEq(RC))
8888 Opc = AArch64::FNMADDDrrr;
8889 else
8890 return nullptr;
8891
8892 Register ResultReg = Root.getOperand(0).getReg();
8893 Register SrcReg0 = MAD->getOperand(1).getReg();
8894 Register SrcReg1 = MAD->getOperand(2).getReg();
8895 Register SrcReg2 = MAD->getOperand(3).getReg();
8896 bool Src0IsKill = MAD->getOperand(1).isKill();
8897 bool Src1IsKill = MAD->getOperand(2).isKill();
8898 bool Src2IsKill = MAD->getOperand(3).isKill();
8899 if (ResultReg.isVirtual())
8900 MRI.constrainRegClass(ResultReg, RC);
8901 if (SrcReg0.isVirtual())
8902 MRI.constrainRegClass(SrcReg0, RC);
8903 if (SrcReg1.isVirtual())
8904 MRI.constrainRegClass(SrcReg1, RC);
8905 if (SrcReg2.isVirtual())
8906 MRI.constrainRegClass(SrcReg2, RC);
8907
8909 BuildMI(MF, MIMetadata(Root), TII->get(Opc), ResultReg)
8910 .addReg(SrcReg0, getKillRegState(Src0IsKill))
8911 .addReg(SrcReg1, getKillRegState(Src1IsKill))
8912 .addReg(SrcReg2, getKillRegState(Src2IsKill));
8913 InsInstrs.push_back(MIB);
8914
8915 return MAD;
8916}
8917
8918/// Fold (FMUL x (DUP y lane)) into (FMUL_indexed x y lane)
8919static MachineInstr *
8922 unsigned IdxDupOp, unsigned MulOpc,
8923 const TargetRegisterClass *RC, MachineRegisterInfo &MRI) {
8924 assert(((IdxDupOp == 1) || (IdxDupOp == 2)) &&
8925 "Invalid index of FMUL operand");
8926
8927 MachineFunction &MF = *Root.getMF();
8929
8930 MachineInstr *Dup =
8931 MF.getRegInfo().getUniqueVRegDef(Root.getOperand(IdxDupOp).getReg());
8932
8933 if (Dup->getOpcode() == TargetOpcode::COPY)
8934 Dup = MRI.getUniqueVRegDef(Dup->getOperand(1).getReg());
8935
8936 Register DupSrcReg = Dup->getOperand(1).getReg();
8937 MRI.clearKillFlags(DupSrcReg);
8938 MRI.constrainRegClass(DupSrcReg, RC);
8939
8940 unsigned DupSrcLane = Dup->getOperand(2).getImm();
8941
8942 unsigned IdxMulOp = IdxDupOp == 1 ? 2 : 1;
8943 MachineOperand &MulOp = Root.getOperand(IdxMulOp);
8944
8945 Register ResultReg = Root.getOperand(0).getReg();
8946
8948 MIB = BuildMI(MF, MIMetadata(Root), TII->get(MulOpc), ResultReg)
8949 .add(MulOp)
8950 .addReg(DupSrcReg)
8951 .addImm(DupSrcLane);
8952
8953 InsInstrs.push_back(MIB);
8954 return &Root;
8955}
8956
8957/// genFusedMultiplyAcc - Helper to generate fused multiply accumulate
8958/// instructions.
8959///
8960/// \see genFusedMultiply
8964 unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC) {
8965 return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
8967}
8968
8969/// genNeg - Helper to generate an intermediate negation of the second operand
8970/// of Root
8972 const TargetInstrInfo *TII, MachineInstr &Root,
8974 DenseMap<Register, unsigned> &InstrIdxForVirtReg,
8975 unsigned MnegOpc, const TargetRegisterClass *RC) {
8976 Register NewVR = MRI.createVirtualRegister(RC);
8978 BuildMI(MF, MIMetadata(Root), TII->get(MnegOpc), NewVR)
8979 .add(Root.getOperand(2));
8980 InsInstrs.push_back(MIB);
8981
8982 assert(InstrIdxForVirtReg.empty());
8983 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
8984
8985 return NewVR;
8986}
8987
8988/// genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate
8989/// instructions with an additional negation of the accumulator
8993 DenseMap<Register, unsigned> &InstrIdxForVirtReg, unsigned IdxMulOpd,
8994 unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC) {
8995 assert(IdxMulOpd == 1);
8996
8997 Register NewVR =
8998 genNeg(MF, MRI, TII, Root, InsInstrs, InstrIdxForVirtReg, MnegOpc, RC);
8999 return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
9000 FMAInstKind::Accumulator, &NewVR);
9001}
9002
9003/// genFusedMultiplyIdx - Helper to generate fused multiply accumulate
9004/// instructions.
9005///
9006/// \see genFusedMultiply
9010 unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC) {
9011 return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
9013}
9014
9015/// genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate
9016/// instructions with an additional negation of the accumulator
9020 DenseMap<Register, unsigned> &InstrIdxForVirtReg, unsigned IdxMulOpd,
9021 unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC) {
9022 assert(IdxMulOpd == 1);
9023
9024 Register NewVR =
9025 genNeg(MF, MRI, TII, Root, InsInstrs, InstrIdxForVirtReg, MnegOpc, RC);
9026
9027 return genFusedMultiply(MF, MRI, TII, Root, InsInstrs, IdxMulOpd, MaddOpc, RC,
9028 FMAInstKind::Indexed, &NewVR);
9029}
9030
9031/// genMaddR - Generate madd instruction and combine mul and add using
9032/// an extra virtual register
9033/// Example - an ADD intermediate needs to be stored in a register:
9034/// MUL I=A,B,0
9035/// ADD R,I,Imm
9036/// ==> ORR V, ZR, Imm
9037/// ==> MADD R,A,B,V
9038/// \param MF Containing MachineFunction
9039/// \param MRI Register information
9040/// \param TII Target information
9041/// \param Root is the ADD instruction
9042/// \param [out] InsInstrs is a vector of machine instructions and will
9043/// contain the generated madd instruction
9044/// \param IdxMulOpd is index of operand in Root that is the result of
9045/// the MUL. In the example above IdxMulOpd is 1.
9046/// \param MaddOpc the opcode fo the madd instruction
9047/// \param VR is a virtual register that holds the value of an ADD operand
9048/// (V in the example above).
9049/// \param RC Register class of operands
9051 const TargetInstrInfo *TII, MachineInstr &Root,
9053 unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR,
9054 const TargetRegisterClass *RC) {
9055 assert(IdxMulOpd == 1 || IdxMulOpd == 2);
9056
9057 MachineInstr *MUL = MRI.getUniqueVRegDef(Root.getOperand(IdxMulOpd).getReg());
9058 Register ResultReg = Root.getOperand(0).getReg();
9059 Register SrcReg0 = MUL->getOperand(1).getReg();
9060 bool Src0IsKill = MUL->getOperand(1).isKill();
9061 Register SrcReg1 = MUL->getOperand(2).getReg();
9062 bool Src1IsKill = MUL->getOperand(2).isKill();
9063
9064 if (ResultReg.isVirtual())
9065 MRI.constrainRegClass(ResultReg, RC);
9066 if (SrcReg0.isVirtual())
9067 MRI.constrainRegClass(SrcReg0, RC);
9068 if (SrcReg1.isVirtual())
9069 MRI.constrainRegClass(SrcReg1, RC);
9071 MRI.constrainRegClass(VR, RC);
9072
9074 BuildMI(MF, MIMetadata(Root), TII->get(MaddOpc), ResultReg)
9075 .addReg(SrcReg0, getKillRegState(Src0IsKill))
9076 .addReg(SrcReg1, getKillRegState(Src1IsKill))
9077 .addReg(VR);
9078 // Insert the MADD
9079 InsInstrs.push_back(MIB);
9080 return MUL;
9081}
9082
9083/// Do the following transformation
9084/// A - (B + C) ==> (A - B) - C
9085/// A - (B + C) ==> (A - C) - B
9087 const TargetInstrInfo *TII, MachineInstr &Root,
9090 unsigned IdxOpd1,
9091 DenseMap<Register, unsigned> &InstrIdxForVirtReg) {
9092 assert(IdxOpd1 == 1 || IdxOpd1 == 2);
9093 unsigned IdxOtherOpd = IdxOpd1 == 1 ? 2 : 1;
9094 MachineInstr *AddMI = MRI.getUniqueVRegDef(Root.getOperand(2).getReg());
9095
9096 Register ResultReg = Root.getOperand(0).getReg();
9097 Register RegA = Root.getOperand(1).getReg();
9098 bool RegAIsKill = Root.getOperand(1).isKill();
9099 Register RegB = AddMI->getOperand(IdxOpd1).getReg();
9100 bool RegBIsKill = AddMI->getOperand(IdxOpd1).isKill();
9101 Register RegC = AddMI->getOperand(IdxOtherOpd).getReg();
9102 bool RegCIsKill = AddMI->getOperand(IdxOtherOpd).isKill();
9103 Register NewVR =
9105
9106 unsigned Opcode = Root.getOpcode();
9107 if (Opcode == AArch64::SUBSWrr)
9108 Opcode = AArch64::SUBWrr;
9109 else if (Opcode == AArch64::SUBSXrr)
9110 Opcode = AArch64::SUBXrr;
9111 else
9112 assert((Opcode == AArch64::SUBWrr || Opcode == AArch64::SUBXrr) &&
9113 "Unexpected instruction opcode.");
9114
9115 uint32_t Flags = Root.mergeFlagsWith(*AddMI);
9116 Flags &= ~MachineInstr::NoSWrap;
9117 Flags &= ~MachineInstr::NoUWrap;
9118
9119 MachineInstrBuilder MIB1 =
9120 BuildMI(MF, MIMetadata(Root), TII->get(Opcode), NewVR)
9121 .addReg(RegA, getKillRegState(RegAIsKill))
9122 .addReg(RegB, getKillRegState(RegBIsKill))
9123 .setMIFlags(Flags);
9124 MachineInstrBuilder MIB2 =
9125 BuildMI(MF, MIMetadata(Root), TII->get(Opcode), ResultReg)
9126 .addReg(NewVR, getKillRegState(true))
9127 .addReg(RegC, getKillRegState(RegCIsKill))
9128 .setMIFlags(Flags);
9129
9130 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9131 InsInstrs.push_back(MIB1);
9132 InsInstrs.push_back(MIB2);
9133 DelInstrs.push_back(AddMI);
9134 DelInstrs.push_back(&Root);
9135}
9136
9137unsigned AArch64InstrInfo::getReduceOpcodeForAccumulator(
9138 unsigned int AccumulatorOpCode) const {
9139 switch (AccumulatorOpCode) {
9140 case AArch64::UABALB_ZZZ_D:
9141 case AArch64::SABALB_ZZZ_D:
9142 case AArch64::UABALT_ZZZ_D:
9143 case AArch64::SABALT_ZZZ_D:
9144 return AArch64::ADD_ZZZ_D;
9145 case AArch64::UABALB_ZZZ_H:
9146 case AArch64::SABALB_ZZZ_H:
9147 case AArch64::UABALT_ZZZ_H:
9148 case AArch64::SABALT_ZZZ_H:
9149 return AArch64::ADD_ZZZ_H;
9150 case AArch64::UABALB_ZZZ_S:
9151 case AArch64::SABALB_ZZZ_S:
9152 case AArch64::UABALT_ZZZ_S:
9153 case AArch64::SABALT_ZZZ_S:
9154 return AArch64::ADD_ZZZ_S;
9155 case AArch64::UABALv16i8_v8i16:
9156 case AArch64::SABALv8i8_v8i16:
9157 case AArch64::SABAv8i16:
9158 case AArch64::UABAv8i16:
9159 return AArch64::ADDv8i16;
9160 case AArch64::SABALv2i32_v2i64:
9161 case AArch64::UABALv2i32_v2i64:
9162 case AArch64::SABALv4i32_v2i64:
9163 return AArch64::ADDv2i64;
9164 case AArch64::UABALv4i16_v4i32:
9165 case AArch64::SABALv4i16_v4i32:
9166 case AArch64::SABALv8i16_v4i32:
9167 case AArch64::SABAv4i32:
9168 case AArch64::UABAv4i32:
9169 return AArch64::ADDv4i32;
9170 case AArch64::UABALv4i32_v2i64:
9171 return AArch64::ADDv2i64;
9172 case AArch64::UABALv8i16_v4i32:
9173 return AArch64::ADDv4i32;
9174 case AArch64::UABALv8i8_v8i16:
9175 case AArch64::SABALv16i8_v8i16:
9176 return AArch64::ADDv8i16;
9177 case AArch64::UABAv16i8:
9178 case AArch64::SABAv16i8:
9179 return AArch64::ADDv16i8;
9180 case AArch64::UABAv4i16:
9181 case AArch64::SABAv4i16:
9182 return AArch64::ADDv4i16;
9183 case AArch64::UABAv2i32:
9184 case AArch64::SABAv2i32:
9185 return AArch64::ADDv2i32;
9186 case AArch64::UABAv8i8:
9187 case AArch64::SABAv8i8:
9188 return AArch64::ADDv8i8;
9189 default:
9190 llvm_unreachable("Unknown accumulator opcode");
9191 }
9192}
9193
9194/// When getMachineCombinerPatterns() finds potential patterns,
9195/// this function generates the instructions that could replace the
9196/// original code sequence
9197void AArch64InstrInfo::genAlternativeCodeSequence(
9198 MachineInstr &Root, unsigned Pattern,
9201 DenseMap<Register, unsigned> &InstrIdxForVirtReg) const {
9202 MachineBasicBlock &MBB = *Root.getParent();
9203 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
9204 MachineFunction &MF = *MBB.getParent();
9205 const TargetInstrInfo *TII = MF.getSubtarget().getInstrInfo();
9206
9207 MachineInstr *MUL = nullptr;
9208 const TargetRegisterClass *RC;
9209 unsigned Opc;
9210 switch (Pattern) {
9211 default:
9212 // Reassociate instructions.
9213 TargetInstrInfo::genAlternativeCodeSequence(Root, Pattern, InsInstrs,
9214 DelInstrs, InstrIdxForVirtReg);
9215 return;
9217 // A - (B + C)
9218 // ==> (A - B) - C
9219 genSubAdd2SubSub(MF, MRI, TII, Root, InsInstrs, DelInstrs, 1,
9220 InstrIdxForVirtReg);
9221 return;
9223 // A - (B + C)
9224 // ==> (A - C) - B
9225 genSubAdd2SubSub(MF, MRI, TII, Root, InsInstrs, DelInstrs, 2,
9226 InstrIdxForVirtReg);
9227 return;
9230 // MUL I=A,B,0
9231 // ADD R,I,C
9232 // ==> MADD R,A,B,C
9233 // --- Create(MADD);
9235 Opc = AArch64::MADDWrrr;
9236 RC = &AArch64::GPR32RegClass;
9237 } else {
9238 Opc = AArch64::MADDXrrr;
9239 RC = &AArch64::GPR64RegClass;
9240 }
9241 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9242 break;
9245 // MUL I=A,B,0
9246 // ADD R,C,I
9247 // ==> MADD R,A,B,C
9248 // --- Create(MADD);
9250 Opc = AArch64::MADDWrrr;
9251 RC = &AArch64::GPR32RegClass;
9252 } else {
9253 Opc = AArch64::MADDXrrr;
9254 RC = &AArch64::GPR64RegClass;
9255 }
9256 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9257 break;
9262 // MUL I=A,B,0
9263 // ADD/SUB R,I,Imm
9264 // ==> MOV V, Imm/-Imm
9265 // ==> MADD R,A,B,V
9266 // --- Create(MADD);
9267 const TargetRegisterClass *RC;
9268 unsigned BitSize, MovImm;
9271 MovImm = AArch64::MOVi32imm;
9272 RC = &AArch64::GPR32spRegClass;
9273 BitSize = 32;
9274 Opc = AArch64::MADDWrrr;
9275 RC = &AArch64::GPR32RegClass;
9276 } else {
9277 MovImm = AArch64::MOVi64imm;
9278 RC = &AArch64::GPR64spRegClass;
9279 BitSize = 64;
9280 Opc = AArch64::MADDXrrr;
9281 RC = &AArch64::GPR64RegClass;
9282 }
9283 Register NewVR = MRI.createVirtualRegister(RC);
9284 uint64_t Imm = Root.getOperand(2).getImm();
9285
9286 if (Root.getOperand(3).isImm()) {
9287 unsigned Val = Root.getOperand(3).getImm();
9288 Imm = Imm << Val;
9289 }
9290 bool IsSub = Pattern == AArch64MachineCombinerPattern::MULSUBWI_OP1 ||
9292 uint64_t UImm = SignExtend64(IsSub ? -Imm : Imm, BitSize);
9293 // Check that the immediate can be composed via a single instruction.
9295 AArch64_IMM::expandMOVImm(UImm, BitSize, Insn);
9296 if (Insn.size() != 1)
9297 return;
9298 MachineInstrBuilder MIB1 =
9299 BuildMI(MF, MIMetadata(Root), TII->get(MovImm), NewVR)
9300 .addImm(IsSub ? -Imm : Imm);
9301 InsInstrs.push_back(MIB1);
9302 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9303 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
9304 break;
9305 }
9308 // MUL I=A,B,0
9309 // SUB R,I, C
9310 // ==> SUB V, 0, C
9311 // ==> MADD R,A,B,V // = -C + A*B
9312 // --- Create(MADD);
9313 const TargetRegisterClass *SubRC;
9314 unsigned SubOpc, ZeroReg;
9316 SubOpc = AArch64::SUBWrr;
9317 SubRC = &AArch64::GPR32spRegClass;
9318 ZeroReg = AArch64::WZR;
9319 Opc = AArch64::MADDWrrr;
9320 RC = &AArch64::GPR32RegClass;
9321 } else {
9322 SubOpc = AArch64::SUBXrr;
9323 SubRC = &AArch64::GPR64spRegClass;
9324 ZeroReg = AArch64::XZR;
9325 Opc = AArch64::MADDXrrr;
9326 RC = &AArch64::GPR64RegClass;
9327 }
9328 Register NewVR = MRI.createVirtualRegister(SubRC);
9329 // SUB NewVR, 0, C
9330 MachineInstrBuilder MIB1 =
9331 BuildMI(MF, MIMetadata(Root), TII->get(SubOpc), NewVR)
9332 .addReg(ZeroReg)
9333 .add(Root.getOperand(2));
9334 InsInstrs.push_back(MIB1);
9335 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9336 MUL = genMaddR(MF, MRI, TII, Root, InsInstrs, 1, Opc, NewVR, RC);
9337 break;
9338 }
9341 // MUL I=A,B,0
9342 // SUB R,C,I
9343 // ==> MSUB R,A,B,C (computes C - A*B)
9344 // --- Create(MSUB);
9346 Opc = AArch64::MSUBWrrr;
9347 RC = &AArch64::GPR32RegClass;
9348 } else {
9349 Opc = AArch64::MSUBXrrr;
9350 RC = &AArch64::GPR64RegClass;
9351 }
9352 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9353 break;
9355 Opc = AArch64::MLAv8i8;
9356 RC = &AArch64::FPR64RegClass;
9357 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9358 break;
9360 Opc = AArch64::MLAv8i8;
9361 RC = &AArch64::FPR64RegClass;
9362 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9363 break;
9365 Opc = AArch64::MLAv16i8;
9366 RC = &AArch64::FPR128RegClass;
9367 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9368 break;
9370 Opc = AArch64::MLAv16i8;
9371 RC = &AArch64::FPR128RegClass;
9372 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9373 break;
9375 Opc = AArch64::MLAv4i16;
9376 RC = &AArch64::FPR64RegClass;
9377 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9378 break;
9380 Opc = AArch64::MLAv4i16;
9381 RC = &AArch64::FPR64RegClass;
9382 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9383 break;
9385 Opc = AArch64::MLAv8i16;
9386 RC = &AArch64::FPR128RegClass;
9387 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9388 break;
9390 Opc = AArch64::MLAv8i16;
9391 RC = &AArch64::FPR128RegClass;
9392 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9393 break;
9395 Opc = AArch64::MLAv2i32;
9396 RC = &AArch64::FPR64RegClass;
9397 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9398 break;
9400 Opc = AArch64::MLAv2i32;
9401 RC = &AArch64::FPR64RegClass;
9402 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9403 break;
9405 Opc = AArch64::MLAv4i32;
9406 RC = &AArch64::FPR128RegClass;
9407 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9408 break;
9410 Opc = AArch64::MLAv4i32;
9411 RC = &AArch64::FPR128RegClass;
9412 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9413 break;
9414
9416 Opc = AArch64::MLAv8i8;
9417 RC = &AArch64::FPR64RegClass;
9418 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9419 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i8,
9420 RC);
9421 break;
9423 Opc = AArch64::MLSv8i8;
9424 RC = &AArch64::FPR64RegClass;
9425 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9426 break;
9428 Opc = AArch64::MLAv16i8;
9429 RC = &AArch64::FPR128RegClass;
9430 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9431 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv16i8,
9432 RC);
9433 break;
9435 Opc = AArch64::MLSv16i8;
9436 RC = &AArch64::FPR128RegClass;
9437 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9438 break;
9440 Opc = AArch64::MLAv4i16;
9441 RC = &AArch64::FPR64RegClass;
9442 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9443 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i16,
9444 RC);
9445 break;
9447 Opc = AArch64::MLSv4i16;
9448 RC = &AArch64::FPR64RegClass;
9449 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9450 break;
9452 Opc = AArch64::MLAv8i16;
9453 RC = &AArch64::FPR128RegClass;
9454 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9455 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i16,
9456 RC);
9457 break;
9459 Opc = AArch64::MLSv8i16;
9460 RC = &AArch64::FPR128RegClass;
9461 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9462 break;
9464 Opc = AArch64::MLAv2i32;
9465 RC = &AArch64::FPR64RegClass;
9466 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9467 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv2i32,
9468 RC);
9469 break;
9471 Opc = AArch64::MLSv2i32;
9472 RC = &AArch64::FPR64RegClass;
9473 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9474 break;
9476 Opc = AArch64::MLAv4i32;
9477 RC = &AArch64::FPR128RegClass;
9478 MUL = genFusedMultiplyAccNeg(MF, MRI, TII, Root, InsInstrs,
9479 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i32,
9480 RC);
9481 break;
9483 Opc = AArch64::MLSv4i32;
9484 RC = &AArch64::FPR128RegClass;
9485 MUL = genFusedMultiplyAcc(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9486 break;
9487
9489 Opc = AArch64::MLAv4i16_indexed;
9490 RC = &AArch64::FPR64RegClass;
9491 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9492 break;
9494 Opc = AArch64::MLAv4i16_indexed;
9495 RC = &AArch64::FPR64RegClass;
9496 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9497 break;
9499 Opc = AArch64::MLAv8i16_indexed;
9500 RC = &AArch64::FPR128RegClass;
9501 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9502 break;
9504 Opc = AArch64::MLAv8i16_indexed;
9505 RC = &AArch64::FPR128RegClass;
9506 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9507 break;
9509 Opc = AArch64::MLAv2i32_indexed;
9510 RC = &AArch64::FPR64RegClass;
9511 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9512 break;
9514 Opc = AArch64::MLAv2i32_indexed;
9515 RC = &AArch64::FPR64RegClass;
9516 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9517 break;
9519 Opc = AArch64::MLAv4i32_indexed;
9520 RC = &AArch64::FPR128RegClass;
9521 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9522 break;
9524 Opc = AArch64::MLAv4i32_indexed;
9525 RC = &AArch64::FPR128RegClass;
9526 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9527 break;
9528
9530 Opc = AArch64::MLAv4i16_indexed;
9531 RC = &AArch64::FPR64RegClass;
9532 MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
9533 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i16,
9534 RC);
9535 break;
9537 Opc = AArch64::MLSv4i16_indexed;
9538 RC = &AArch64::FPR64RegClass;
9539 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9540 break;
9542 Opc = AArch64::MLAv8i16_indexed;
9543 RC = &AArch64::FPR128RegClass;
9544 MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
9545 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv8i16,
9546 RC);
9547 break;
9549 Opc = AArch64::MLSv8i16_indexed;
9550 RC = &AArch64::FPR128RegClass;
9551 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9552 break;
9554 Opc = AArch64::MLAv2i32_indexed;
9555 RC = &AArch64::FPR64RegClass;
9556 MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
9557 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv2i32,
9558 RC);
9559 break;
9561 Opc = AArch64::MLSv2i32_indexed;
9562 RC = &AArch64::FPR64RegClass;
9563 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9564 break;
9566 Opc = AArch64::MLAv4i32_indexed;
9567 RC = &AArch64::FPR128RegClass;
9568 MUL = genFusedMultiplyIdxNeg(MF, MRI, TII, Root, InsInstrs,
9569 InstrIdxForVirtReg, 1, Opc, AArch64::NEGv4i32,
9570 RC);
9571 break;
9573 Opc = AArch64::MLSv4i32_indexed;
9574 RC = &AArch64::FPR128RegClass;
9575 MUL = genFusedMultiplyIdx(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9576 break;
9577
9578 // Floating Point Support
9580 Opc = AArch64::FMADDHrrr;
9581 RC = &AArch64::FPR16RegClass;
9582 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9583 break;
9585 Opc = AArch64::FMADDSrrr;
9586 RC = &AArch64::FPR32RegClass;
9587 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9588 break;
9590 Opc = AArch64::FMADDDrrr;
9591 RC = &AArch64::FPR64RegClass;
9592 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9593 break;
9594
9596 Opc = AArch64::FMADDHrrr;
9597 RC = &AArch64::FPR16RegClass;
9598 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9599 break;
9601 Opc = AArch64::FMADDSrrr;
9602 RC = &AArch64::FPR32RegClass;
9603 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9604 break;
9606 Opc = AArch64::FMADDDrrr;
9607 RC = &AArch64::FPR64RegClass;
9608 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9609 break;
9610
9612 Opc = AArch64::FMLAv1i32_indexed;
9613 RC = &AArch64::FPR32RegClass;
9614 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9616 break;
9618 Opc = AArch64::FMLAv1i32_indexed;
9619 RC = &AArch64::FPR32RegClass;
9620 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9622 break;
9623
9625 Opc = AArch64::FMLAv1i64_indexed;
9626 RC = &AArch64::FPR64RegClass;
9627 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9629 break;
9631 Opc = AArch64::FMLAv1i64_indexed;
9632 RC = &AArch64::FPR64RegClass;
9633 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9635 break;
9636
9638 RC = &AArch64::FPR64RegClass;
9639 Opc = AArch64::FMLAv4i16_indexed;
9640 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9642 break;
9644 RC = &AArch64::FPR64RegClass;
9645 Opc = AArch64::FMLAv4f16;
9646 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9648 break;
9650 RC = &AArch64::FPR64RegClass;
9651 Opc = AArch64::FMLAv4i16_indexed;
9652 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9654 break;
9656 RC = &AArch64::FPR64RegClass;
9657 Opc = AArch64::FMLAv4f16;
9658 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9660 break;
9661
9664 RC = &AArch64::FPR64RegClass;
9666 Opc = AArch64::FMLAv2i32_indexed;
9667 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9669 } else {
9670 Opc = AArch64::FMLAv2f32;
9671 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9673 }
9674 break;
9677 RC = &AArch64::FPR64RegClass;
9679 Opc = AArch64::FMLAv2i32_indexed;
9680 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9682 } else {
9683 Opc = AArch64::FMLAv2f32;
9684 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9686 }
9687 break;
9688
9690 RC = &AArch64::FPR128RegClass;
9691 Opc = AArch64::FMLAv8i16_indexed;
9692 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9694 break;
9696 RC = &AArch64::FPR128RegClass;
9697 Opc = AArch64::FMLAv8f16;
9698 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9700 break;
9702 RC = &AArch64::FPR128RegClass;
9703 Opc = AArch64::FMLAv8i16_indexed;
9704 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9706 break;
9708 RC = &AArch64::FPR128RegClass;
9709 Opc = AArch64::FMLAv8f16;
9710 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9712 break;
9713
9716 RC = &AArch64::FPR128RegClass;
9718 Opc = AArch64::FMLAv2i64_indexed;
9719 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9721 } else {
9722 Opc = AArch64::FMLAv2f64;
9723 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9725 }
9726 break;
9729 RC = &AArch64::FPR128RegClass;
9731 Opc = AArch64::FMLAv2i64_indexed;
9732 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9734 } else {
9735 Opc = AArch64::FMLAv2f64;
9736 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9738 }
9739 break;
9740
9743 RC = &AArch64::FPR128RegClass;
9745 Opc = AArch64::FMLAv4i32_indexed;
9746 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9748 } else {
9749 Opc = AArch64::FMLAv4f32;
9750 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9752 }
9753 break;
9754
9757 RC = &AArch64::FPR128RegClass;
9759 Opc = AArch64::FMLAv4i32_indexed;
9760 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9762 } else {
9763 Opc = AArch64::FMLAv4f32;
9764 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9766 }
9767 break;
9768
9770 Opc = AArch64::FNMSUBHrrr;
9771 RC = &AArch64::FPR16RegClass;
9772 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9773 break;
9775 Opc = AArch64::FNMSUBSrrr;
9776 RC = &AArch64::FPR32RegClass;
9777 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9778 break;
9780 Opc = AArch64::FNMSUBDrrr;
9781 RC = &AArch64::FPR64RegClass;
9782 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9783 break;
9784
9786 Opc = AArch64::FNMADDHrrr;
9787 RC = &AArch64::FPR16RegClass;
9788 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9789 break;
9791 Opc = AArch64::FNMADDSrrr;
9792 RC = &AArch64::FPR32RegClass;
9793 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9794 break;
9796 Opc = AArch64::FNMADDDrrr;
9797 RC = &AArch64::FPR64RegClass;
9798 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC);
9799 break;
9800
9802 Opc = AArch64::FMSUBHrrr;
9803 RC = &AArch64::FPR16RegClass;
9804 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9805 break;
9807 Opc = AArch64::FMSUBSrrr;
9808 RC = &AArch64::FPR32RegClass;
9809 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9810 break;
9812 Opc = AArch64::FMSUBDrrr;
9813 RC = &AArch64::FPR64RegClass;
9814 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC);
9815 break;
9816
9818 Opc = AArch64::FMLSv1i32_indexed;
9819 RC = &AArch64::FPR32RegClass;
9820 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9822 break;
9823
9825 Opc = AArch64::FMLSv1i64_indexed;
9826 RC = &AArch64::FPR64RegClass;
9827 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9829 break;
9830
9833 RC = &AArch64::FPR64RegClass;
9834 Register NewVR = MRI.createVirtualRegister(RC);
9835 MachineInstrBuilder MIB1 =
9836 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::FNEGv4f16), NewVR)
9837 .add(Root.getOperand(2));
9838 InsInstrs.push_back(MIB1);
9839 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9841 Opc = AArch64::FMLAv4f16;
9842 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9843 FMAInstKind::Accumulator, &NewVR);
9844 } else {
9845 Opc = AArch64::FMLAv4i16_indexed;
9846 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9847 FMAInstKind::Indexed, &NewVR);
9848 }
9849 break;
9850 }
9852 RC = &AArch64::FPR64RegClass;
9853 Opc = AArch64::FMLSv4f16;
9854 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9856 break;
9858 RC = &AArch64::FPR64RegClass;
9859 Opc = AArch64::FMLSv4i16_indexed;
9860 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9862 break;
9863
9866 RC = &AArch64::FPR64RegClass;
9868 Opc = AArch64::FMLSv2i32_indexed;
9869 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9871 } else {
9872 Opc = AArch64::FMLSv2f32;
9873 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9875 }
9876 break;
9877
9880 RC = &AArch64::FPR128RegClass;
9881 Register NewVR = MRI.createVirtualRegister(RC);
9882 MachineInstrBuilder MIB1 =
9883 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::FNEGv8f16), NewVR)
9884 .add(Root.getOperand(2));
9885 InsInstrs.push_back(MIB1);
9886 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9888 Opc = AArch64::FMLAv8f16;
9889 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9890 FMAInstKind::Accumulator, &NewVR);
9891 } else {
9892 Opc = AArch64::FMLAv8i16_indexed;
9893 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9894 FMAInstKind::Indexed, &NewVR);
9895 }
9896 break;
9897 }
9899 RC = &AArch64::FPR128RegClass;
9900 Opc = AArch64::FMLSv8f16;
9901 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9903 break;
9905 RC = &AArch64::FPR128RegClass;
9906 Opc = AArch64::FMLSv8i16_indexed;
9907 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9909 break;
9910
9913 RC = &AArch64::FPR128RegClass;
9915 Opc = AArch64::FMLSv2i64_indexed;
9916 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9918 } else {
9919 Opc = AArch64::FMLSv2f64;
9920 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9922 }
9923 break;
9924
9927 RC = &AArch64::FPR128RegClass;
9929 Opc = AArch64::FMLSv4i32_indexed;
9930 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9932 } else {
9933 Opc = AArch64::FMLSv4f32;
9934 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 2, Opc, RC,
9936 }
9937 break;
9940 RC = &AArch64::FPR64RegClass;
9941 Register NewVR = MRI.createVirtualRegister(RC);
9942 MachineInstrBuilder MIB1 =
9943 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::FNEGv2f32), NewVR)
9944 .add(Root.getOperand(2));
9945 InsInstrs.push_back(MIB1);
9946 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9948 Opc = AArch64::FMLAv2i32_indexed;
9949 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9950 FMAInstKind::Indexed, &NewVR);
9951 } else {
9952 Opc = AArch64::FMLAv2f32;
9953 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9954 FMAInstKind::Accumulator, &NewVR);
9955 }
9956 break;
9957 }
9960 RC = &AArch64::FPR128RegClass;
9961 Register NewVR = MRI.createVirtualRegister(RC);
9962 MachineInstrBuilder MIB1 =
9963 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::FNEGv4f32), NewVR)
9964 .add(Root.getOperand(2));
9965 InsInstrs.push_back(MIB1);
9966 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9968 Opc = AArch64::FMLAv4i32_indexed;
9969 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9970 FMAInstKind::Indexed, &NewVR);
9971 } else {
9972 Opc = AArch64::FMLAv4f32;
9973 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9974 FMAInstKind::Accumulator, &NewVR);
9975 }
9976 break;
9977 }
9980 RC = &AArch64::FPR128RegClass;
9981 Register NewVR = MRI.createVirtualRegister(RC);
9982 MachineInstrBuilder MIB1 =
9983 BuildMI(MF, MIMetadata(Root), TII->get(AArch64::FNEGv2f64), NewVR)
9984 .add(Root.getOperand(2));
9985 InsInstrs.push_back(MIB1);
9986 InstrIdxForVirtReg.insert(std::make_pair(NewVR, 0));
9988 Opc = AArch64::FMLAv2i64_indexed;
9989 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9990 FMAInstKind::Indexed, &NewVR);
9991 } else {
9992 Opc = AArch64::FMLAv2f64;
9993 MUL = genFusedMultiply(MF, MRI, TII, Root, InsInstrs, 1, Opc, RC,
9994 FMAInstKind::Accumulator, &NewVR);
9995 }
9996 break;
9997 }
10000 unsigned IdxDupOp =
10002 : 2;
10003 genIndexedMultiply(Root, InsInstrs, IdxDupOp, AArch64::FMULv2i32_indexed,
10004 &AArch64::FPR128RegClass, MRI);
10005 break;
10006 }
10009 unsigned IdxDupOp =
10011 : 2;
10012 genIndexedMultiply(Root, InsInstrs, IdxDupOp, AArch64::FMULv2i64_indexed,
10013 &AArch64::FPR128RegClass, MRI);
10014 break;
10015 }
10018 unsigned IdxDupOp =
10020 : 2;
10021 genIndexedMultiply(Root, InsInstrs, IdxDupOp, AArch64::FMULv4i16_indexed,
10022 &AArch64::FPR128_loRegClass, MRI);
10023 break;
10024 }
10027 unsigned IdxDupOp =
10029 : 2;
10030 genIndexedMultiply(Root, InsInstrs, IdxDupOp, AArch64::FMULv4i32_indexed,
10031 &AArch64::FPR128RegClass, MRI);
10032 break;
10033 }
10036 unsigned IdxDupOp =
10038 : 2;
10039 genIndexedMultiply(Root, InsInstrs, IdxDupOp, AArch64::FMULv8i16_indexed,
10040 &AArch64::FPR128_loRegClass, MRI);
10041 break;
10042 }
10044 MUL = genFNegatedMAD(MF, MRI, TII, Root, InsInstrs);
10045 break;
10046 }
10048 generateGatherLanePattern(Root, InsInstrs, DelInstrs, InstrIdxForVirtReg,
10049 Pattern, 4);
10050 break;
10051 }
10053 generateGatherLanePattern(Root, InsInstrs, DelInstrs, InstrIdxForVirtReg,
10054 Pattern, 8);
10055 break;
10056 }
10058 generateGatherLanePattern(Root, InsInstrs, DelInstrs, InstrIdxForVirtReg,
10059 Pattern, 16);
10060 break;
10061 }
10062
10063 } // end switch (Pattern)
10064 // Record MUL and ADD/SUB for deletion
10065 if (MUL)
10066 DelInstrs.push_back(MUL);
10067 DelInstrs.push_back(&Root);
10068
10069 // Set the flags on the inserted instructions to be the merged flags of the
10070 // instructions that we have combined.
10071 uint32_t Flags = Root.getFlags();
10072 if (MUL)
10073 Flags = Root.mergeFlagsWith(*MUL);
10074 for (auto *MI : InsInstrs)
10075 MI->setFlags(Flags);
10076}
10077
10078/// Replace csincr-branch sequence by simple conditional branch
10079///
10080/// Examples:
10081/// 1. \code
10082/// csinc w9, wzr, wzr, <condition code>
10083/// tbnz w9, #0, 0x44
10084/// \endcode
10085/// to
10086/// \code
10087/// b.<inverted condition code>
10088/// \endcode
10089///
10090/// 2. \code
10091/// csinc w9, wzr, wzr, <condition code>
10092/// tbz w9, #0, 0x44
10093/// \endcode
10094/// to
10095/// \code
10096/// b.<condition code>
10097/// \endcode
10098///
10099/// Replace compare and branch sequence by TBZ/TBNZ instruction when the
10100/// compare's constant operand is power of 2.
10101///
10102/// Examples:
10103/// \code
10104/// and w8, w8, #0x400
10105/// cbnz w8, L1
10106/// \endcode
10107/// to
10108/// \code
10109/// tbnz w8, #10, L1
10110/// \endcode
10111///
10112/// \param MI Conditional Branch
10113/// \return True when the simple conditional branch is generated
10114///
10116 bool IsNegativeBranch = false;
10117 bool IsTestAndBranch = false;
10118 unsigned TargetBBInMI = 0;
10119 switch (MI.getOpcode()) {
10120 default:
10121 llvm_unreachable("Unknown branch instruction?");
10122 case AArch64::Bcc:
10123 case AArch64::CBWPri:
10124 case AArch64::CBXPri:
10125 case AArch64::CBBAssertExt:
10126 case AArch64::CBHAssertExt:
10127 case AArch64::CBWPrr:
10128 case AArch64::CBXPrr:
10129 return false;
10130 case AArch64::CBZW:
10131 case AArch64::CBZX:
10132 TargetBBInMI = 1;
10133 break;
10134 case AArch64::CBNZW:
10135 case AArch64::CBNZX:
10136 TargetBBInMI = 1;
10137 IsNegativeBranch = true;
10138 break;
10139 case AArch64::TBZW:
10140 case AArch64::TBZX:
10141 TargetBBInMI = 2;
10142 IsTestAndBranch = true;
10143 break;
10144 case AArch64::TBNZW:
10145 case AArch64::TBNZX:
10146 TargetBBInMI = 2;
10147 IsNegativeBranch = true;
10148 IsTestAndBranch = true;
10149 break;
10150 }
10151 // So we increment a zero register and test for bits other
10152 // than bit 0? Conservatively bail out in case the verifier
10153 // missed this case.
10154 if (IsTestAndBranch && MI.getOperand(1).getImm())
10155 return false;
10156
10157 // Find Definition.
10158 assert(MI.getParent() && "Incomplete machine instruction\n");
10159 MachineBasicBlock *MBB = MI.getParent();
10160 MachineFunction *MF = MBB->getParent();
10161 MachineRegisterInfo *MRI = &MF->getRegInfo();
10162 Register VReg = MI.getOperand(0).getReg();
10163 if (!VReg.isVirtual())
10164 return false;
10165
10166 MachineInstr *DefMI = MRI->getVRegDef(VReg);
10167 if (!DefMI)
10168 return false;
10169
10170 // Look through COPY instructions to find definition.
10171 while (DefMI->isCopy()) {
10172 Register CopyVReg = DefMI->getOperand(1).getReg();
10173 if (!CopyVReg.isVirtual())
10174 return false;
10175 if (!MRI->hasOneNonDBGUse(CopyVReg))
10176 return false;
10177 DefMI = MRI->getVRegDef(CopyVReg);
10178 if (!DefMI)
10179 return false;
10180 }
10181
10182 switch (DefMI->getOpcode()) {
10183 default:
10184 return false;
10185 // Fold AND into a TBZ/TBNZ if constant operand is power of 2.
10186 case AArch64::ANDWri:
10187 case AArch64::ANDXri: {
10188 if (IsTestAndBranch)
10189 return false;
10190 if (DefMI->getParent() != MBB)
10191 return false;
10192 if (!MRI->hasOneNonDBGUse(VReg))
10193 return false;
10194
10195 bool Is32Bit = (DefMI->getOpcode() == AArch64::ANDWri);
10196 uint64_t Mask = AArch64_AM::decodeLogicalImmediate(
10197 DefMI->getOperand(2).getImm(), Is32Bit ? 32 : 64);
10198 if (!isPowerOf2_64(Mask))
10199 return false;
10200
10201 MachineOperand &MO = DefMI->getOperand(1);
10202 Register NewReg = MO.getReg();
10203 if (!NewReg.isVirtual())
10204 return false;
10205
10206 if (!MRI->getVRegDef(NewReg))
10207 return false;
10208
10209 MachineBasicBlock &RefToMBB = *MBB;
10210 MachineBasicBlock *TBB = MI.getOperand(1).getMBB();
10211 DebugLoc DL = MI.getDebugLoc();
10212 unsigned Imm = Log2_64(Mask);
10213 unsigned Opc = (Imm < 32)
10214 ? (IsNegativeBranch ? AArch64::TBNZW : AArch64::TBZW)
10215 : (IsNegativeBranch ? AArch64::TBNZX : AArch64::TBZX);
10216 MachineInstr *NewMI = BuildMI(RefToMBB, MI, DL, get(Opc))
10217 .addReg(NewReg)
10218 .addImm(Imm)
10219 .addMBB(TBB);
10220 // Register lives on to the CBZ now.
10221 MO.setIsKill(false);
10222
10223 // For immediate smaller than 32, we need to use the 32-bit
10224 // variant (W) in all cases. Indeed the 64-bit variant does not
10225 // allow to encode them.
10226 // Therefore, if the input register is 64-bit, we need to take the
10227 // 32-bit sub-part.
10228 if (!Is32Bit && Imm < 32)
10229 NewMI->getOperand(0).setSubReg(AArch64::sub_32);
10230 MI.eraseFromParent();
10231 return true;
10232 }
10233 // Look for CSINC
10234 case AArch64::CSINCWr:
10235 case AArch64::CSINCXr: {
10236 if (!(DefMI->getOperand(1).getReg() == AArch64::WZR &&
10237 DefMI->getOperand(2).getReg() == AArch64::WZR) &&
10238 !(DefMI->getOperand(1).getReg() == AArch64::XZR &&
10239 DefMI->getOperand(2).getReg() == AArch64::XZR))
10240 return false;
10241
10242 if (DefMI->findRegisterDefOperandIdx(AArch64::NZCV, /*TRI=*/nullptr,
10243 true) != -1)
10244 return false;
10245
10246 AArch64CC::CondCode CC = (AArch64CC::CondCode)DefMI->getOperand(3).getImm();
10247 // Convert only when the condition code is not modified between
10248 // the CSINC and the branch. The CC may be used by other
10249 // instructions in between.
10251 return false;
10252 MachineBasicBlock &RefToMBB = *MBB;
10253 MachineBasicBlock *TBB = MI.getOperand(TargetBBInMI).getMBB();
10254 DebugLoc DL = MI.getDebugLoc();
10255 if (IsNegativeBranch)
10257 BuildMI(RefToMBB, MI, DL, get(AArch64::Bcc)).addImm(CC).addMBB(TBB);
10258 MI.eraseFromParent();
10259 return true;
10260 }
10261 }
10262}
10263
10264std::pair<unsigned, unsigned>
10265AArch64InstrInfo::decomposeMachineOperandsTargetFlags(unsigned TF) const {
10266 const unsigned Mask = AArch64II::MO_FRAGMENT;
10267 return std::make_pair(TF & Mask, TF & ~Mask);
10268}
10269
10271AArch64InstrInfo::getSerializableDirectMachineOperandTargetFlags() const {
10272 using namespace AArch64II;
10273
10274 static const std::pair<unsigned, const char *> TargetFlags[] = {
10275 {MO_PAGE, "aarch64-page"}, {MO_PAGEOFF, "aarch64-pageoff"},
10276 {MO_G3, "aarch64-g3"}, {MO_G2, "aarch64-g2"},
10277 {MO_G1, "aarch64-g1"}, {MO_G0, "aarch64-g0"},
10278 {MO_HI12, "aarch64-hi12"}};
10279 return ArrayRef(TargetFlags);
10280}
10281
10283AArch64InstrInfo::getSerializableBitmaskMachineOperandTargetFlags() const {
10284 using namespace AArch64II;
10285
10286 static const std::pair<unsigned, const char *> TargetFlags[] = {
10287 {MO_COFFSTUB, "aarch64-coffstub"},
10288 {MO_GOT, "aarch64-got"},
10289 {MO_NC, "aarch64-nc"},
10290 {MO_S, "aarch64-s"},
10291 {MO_TLS, "aarch64-tls"},
10292 {MO_DLLIMPORT, "aarch64-dllimport"},
10293 {MO_PREL, "aarch64-prel"},
10294 {MO_TAGGED, "aarch64-tagged"},
10295 {MO_ARM64EC_CALLMANGLE, "aarch64-arm64ec-callmangle"},
10296 };
10297 return ArrayRef(TargetFlags);
10298}
10299
10301AArch64InstrInfo::getSerializableMachineMemOperandTargetFlags() const {
10302 static const std::pair<MachineMemOperand::Flags, const char *> TargetFlags[] =
10303 {{MOSuppressPair, "aarch64-suppress-pair"},
10304 {MOStridedAccess, "aarch64-strided-access"}};
10305 return ArrayRef(TargetFlags);
10306}
10307
10308/// Constants defining how certain sequences should be outlined.
10309/// This encompasses how an outlined function should be called, and what kind of
10310/// frame should be emitted for that outlined function.
10311///
10312/// \p MachineOutlinerDefault implies that the function should be called with
10313/// a save and restore of LR to the stack.
10314///
10315/// That is,
10316///
10317/// I1 Save LR OUTLINED_FUNCTION:
10318/// I2 --> BL OUTLINED_FUNCTION I1
10319/// I3 Restore LR I2
10320/// I3
10321/// RET
10322///
10323/// * Call construction overhead: 3 (save + BL + restore)
10324/// * Frame construction overhead: 1 (ret)
10325/// * Requires stack fixups? Yes
10326///
10327/// \p MachineOutlinerTailCall implies that the function is being created from
10328/// a sequence of instructions ending in a return.
10329///
10330/// That is,
10331///
10332/// I1 OUTLINED_FUNCTION:
10333/// I2 --> B OUTLINED_FUNCTION I1
10334/// RET I2
10335/// RET
10336///
10337/// * Call construction overhead: 1 (B)
10338/// * Frame construction overhead: 0 (Return included in sequence)
10339/// * Requires stack fixups? No
10340///
10341/// \p MachineOutlinerNoLRSave implies that the function should be called using
10342/// a BL instruction, but doesn't require LR to be saved and restored. This
10343/// happens when LR is known to be dead.
10344///
10345/// That is,
10346///
10347/// I1 OUTLINED_FUNCTION:
10348/// I2 --> BL OUTLINED_FUNCTION I1
10349/// I3 I2
10350/// I3
10351/// RET
10352///
10353/// * Call construction overhead: 1 (BL)
10354/// * Frame construction overhead: 1 (RET)
10355/// * Requires stack fixups? No
10356///
10357/// \p MachineOutlinerThunk implies that the function is being created from
10358/// a sequence of instructions ending in a call. The outlined function is
10359/// called with a BL instruction, and the outlined function tail-calls the
10360/// original call destination.
10361///
10362/// That is,
10363///
10364/// I1 OUTLINED_FUNCTION:
10365/// I2 --> BL OUTLINED_FUNCTION I1
10366/// BL f I2
10367/// B f
10368/// * Call construction overhead: 1 (BL)
10369/// * Frame construction overhead: 0
10370/// * Requires stack fixups? No
10371///
10372/// \p MachineOutlinerRegSave implies that the function should be called with a
10373/// save and restore of LR to an available register. This allows us to avoid
10374/// stack fixups. Note that this outlining variant is compatible with the
10375/// NoLRSave case.
10376///
10377/// That is,
10378///
10379/// I1 Save LR OUTLINED_FUNCTION:
10380/// I2 --> BL OUTLINED_FUNCTION I1
10381/// I3 Restore LR I2
10382/// I3
10383/// RET
10384///
10385/// * Call construction overhead: 3 (save + BL + restore)
10386/// * Frame construction overhead: 1 (ret)
10387/// * Requires stack fixups? No
10389 MachineOutlinerDefault, /// Emit a save, restore, call, and return.
10390 MachineOutlinerTailCall, /// Only emit a branch.
10391 MachineOutlinerNoLRSave, /// Emit a call and return.
10392 MachineOutlinerThunk, /// Emit a call and tail-call.
10393 MachineOutlinerRegSave /// Same as default, but save to a register.
10394};
10395
10401
10402/// Return true if the frame-record form of the outlined prologue is enabled for
10403/// the target of \p MF.
10404///
10405/// A non-leaf outlined function must save LR. On MachO, saving LR alone
10406/// (str x30) has no compact unwind encoding, so we get a large DWARF FDE
10407/// instead. Saving FP and LR as a frame record (stp x29, x30 ; mov x29, sp)
10408/// gets the small FRAME encoding, and costs one extra instruction.
10413
10414/// Return true if the outlined function in \p MBB should save FP and LR as a
10415/// frame record instead of saving LR alone.
10417 const MachineBasicBlock &MBB) {
10418 const MachineFunction &MF = *MBB.getParent();
10419
10420 // Only worth it if the function has unwind info to shrink.
10423 return false;
10424
10425 // Only safe if the outlined code never touches FP, since we overwrite it.
10427 for (const MachineInstr &MI : MBB.instrs())
10428 LRU.accumulate(MI);
10429 return LRU.available(AArch64::FP);
10430}
10431
10432/// Predict what the above will answer, for use while costing candidates. The
10433/// outlined function does not exist yet, so answer from \p RepeatedSequenceLocs
10434/// instead. This is only an estimate; buildOutlinedFrame() makes the call.
10436 std::vector<outliner::Candidate> &RepeatedSequenceLocs,
10437 const TargetRegisterInfo &TRI) {
10438 if (!isCompactUnwindFrameRecordEnabled(*RepeatedSequenceLocs.front().getMF()))
10439 return false;
10440
10441 // The outlined function is nounwind only if every candidate is, so it has
10442 // unwind info if any candidate does.
10443 if (llvm::none_of(RepeatedSequenceLocs, [](outliner::Candidate &C) {
10444 const MachineFunction &MF = *C.getMF();
10445 return MF.getInfo<AArch64FunctionInfo>()->needsDwarfUnwindInfo(MF);
10446 }))
10447 return false;
10448
10449 // FP is free in the outlined function only if it is free in every candidate.
10450 return llvm::all_of(RepeatedSequenceLocs, [&TRI](outliner::Candidate &C) {
10451 return C.isAvailableInsideSeq(AArch64::FP, TRI);
10452 });
10453}
10454
10456AArch64InstrInfo::findRegisterToSaveLRTo(outliner::Candidate &C) const {
10457 MachineFunction *MF = C.getMF();
10458 const TargetRegisterInfo &TRI = *MF->getSubtarget().getRegisterInfo();
10459 const AArch64RegisterInfo *ARI =
10460 static_cast<const AArch64RegisterInfo *>(&TRI);
10461 // Check if there is an available register across the sequence that we can
10462 // use.
10463 for (unsigned Reg : AArch64::GPR64RegClass) {
10464 if (!ARI->isReservedReg(*MF, Reg) &&
10465 Reg != AArch64::LR && // LR is not reserved, but don't use it.
10466 Reg != AArch64::X16 && // X16 is not guaranteed to be preserved.
10467 Reg != AArch64::X17 && // Ditto for X17.
10468 C.isAvailableAcrossAndOutOfSeq(Reg, TRI) &&
10469 C.isAvailableInsideSeq(Reg, TRI))
10470 return Reg;
10471 }
10472 return Register();
10473}
10474
10475static bool
10477 const outliner::Candidate &b) {
10478 const auto &MFIa = a.getMF()->getInfo<AArch64FunctionInfo>();
10479 const auto &MFIb = b.getMF()->getInfo<AArch64FunctionInfo>();
10480
10481 return MFIa->getSignReturnAddressCondition() ==
10483}
10484
10485static bool
10487 const outliner::Candidate &b) {
10488 const auto &MFIa = a.getMF()->getInfo<AArch64FunctionInfo>();
10489 const auto &MFIb = b.getMF()->getInfo<AArch64FunctionInfo>();
10490
10491 return MFIa->shouldSignWithBKey() == MFIb->shouldSignWithBKey();
10492}
10493
10495 const outliner::Candidate &b) {
10496 const AArch64Subtarget &SubtargetA =
10498 const AArch64Subtarget &SubtargetB =
10499 b.getMF()->getSubtarget<AArch64Subtarget>();
10500 return SubtargetA.hasV8_3aOps() == SubtargetB.hasV8_3aOps();
10501}
10502
10503std::optional<std::unique_ptr<outliner::OutlinedFunction>>
10504AArch64InstrInfo::getOutliningCandidateInfo(
10505 const MachineModuleInfo &MMI,
10506 std::vector<outliner::Candidate> &RepeatedSequenceLocs,
10507 unsigned MinRepeats) const {
10508 unsigned SequenceSize = 0;
10509 for (auto &MI : RepeatedSequenceLocs[0])
10510 SequenceSize += getInstSizeInBytes(MI);
10511
10512 unsigned NumBytesToCreateFrame = 0;
10513
10514 // Avoid splitting ADRP ADD/LDR pair into outlined functions.
10515 // These instructions are fused together by the scheduler.
10516 // Any candidate where ADRP is the last instruction should be rejected
10517 // as that will lead to splitting ADRP pair.
10518 MachineInstr &LastMI = RepeatedSequenceLocs[0].back();
10519 MachineInstr &FirstMI = RepeatedSequenceLocs[0].front();
10520 if (LastMI.getOpcode() == AArch64::ADRP &&
10521 (LastMI.getOperand(1).getTargetFlags() & AArch64II::MO_PAGE) != 0 &&
10522 (LastMI.getOperand(1).getTargetFlags() & AArch64II::MO_GOT) != 0) {
10523 return std::nullopt;
10524 }
10525
10526 // Similarly any candidate where the first instruction is ADD/LDR with a
10527 // page offset should be rejected to avoid ADRP splitting.
10528 if ((FirstMI.getOpcode() == AArch64::ADDXri ||
10529 FirstMI.getOpcode() == AArch64::LDRXui) &&
10530 (FirstMI.getOperand(2).getTargetFlags() & AArch64II::MO_PAGEOFF) != 0 &&
10531 (FirstMI.getOperand(2).getTargetFlags() & AArch64II::MO_GOT) != 0) {
10532 return std::nullopt;
10533 }
10534
10535 // We only allow outlining for functions having exactly matching return
10536 // address signing attributes, i.e., all share the same value for the
10537 // attribute "sign-return-address" and all share the same type of key they
10538 // are signed with.
10539 // Additionally we require all functions to simultaneously either support
10540 // v8.3a features or not. Otherwise an outlined function could get signed
10541 // using dedicated v8.3 instructions and a call from a function that doesn't
10542 // support v8.3 instructions would therefore be invalid.
10543 if (std::adjacent_find(
10544 RepeatedSequenceLocs.begin(), RepeatedSequenceLocs.end(),
10545 [](const outliner::Candidate &a, const outliner::Candidate &b) {
10546 // Return true if a and b are non-equal w.r.t. return address
10547 // signing or support of v8.3a features
10548 if (outliningCandidatesSigningScopeConsensus(a, b) &&
10549 outliningCandidatesSigningKeyConsensus(a, b) &&
10550 outliningCandidatesV8_3OpsConsensus(a, b)) {
10551 return false;
10552 }
10553 return true;
10554 }) != RepeatedSequenceLocs.end()) {
10555 return std::nullopt;
10556 }
10557
10558 // Since at this point all candidates agree on their return address signing
10559 // picking just one is fine. If the candidate functions potentially sign their
10560 // return addresses, the outlined function should do the same. Note that in
10561 // the case of "sign-return-address"="non-leaf" this is an assumption: It is
10562 // not certainly true that the outlined function will have to sign its return
10563 // address but this decision is made later, when the decision to outline
10564 // has already been made.
10565 // The same holds for the number of additional instructions we need: On
10566 // v8.3a RET can be replaced by RETAA/RETAB and no AUT instruction is
10567 // necessary. However, at this point we don't know if the outlined function
10568 // will have a RET instruction so we assume the worst.
10569 const TargetRegisterInfo &TRI = getRegisterInfo();
10570 // Performing a tail call may require extra checks when PAuth is enabled.
10571 // If PAuth is disabled, set it to zero for uniformity.
10572 unsigned NumBytesToCheckLRInTCEpilogue = 0;
10573 const auto RASignCondition = RepeatedSequenceLocs[0]
10574 .getMF()
10575 ->getInfo<AArch64FunctionInfo>()
10576 ->getSignReturnAddressCondition();
10577 if (RASignCondition != SignReturnAddress::None) {
10578 // One PAC and one AUT instructions
10579 NumBytesToCreateFrame += 8;
10580
10581 // PAuth is enabled - set extra tail call cost, if any.
10582 auto LRCheckMethod = Subtarget.getAuthenticatedLRCheckMethod(
10583 *RepeatedSequenceLocs[0].getMF());
10584 NumBytesToCheckLRInTCEpilogue =
10586 // Checking the authenticated LR value may significantly impact
10587 // SequenceSize, so account for it for more precise results.
10588 if (isTailCallReturnInst(RepeatedSequenceLocs[0].back()))
10589 SequenceSize += NumBytesToCheckLRInTCEpilogue;
10590
10591 // We have to check if sp modifying instructions would get outlined.
10592 // If so we only allow outlining if sp is unchanged overall, so matching
10593 // sub and add instructions are okay to outline, all other sp modifications
10594 // are not
10595 auto hasIllegalSPModification = [&TRI](outliner::Candidate &C) {
10596 int SPValue = 0;
10597 for (auto &MI : C) {
10598 if (MI.modifiesRegister(AArch64::SP, &TRI)) {
10599 switch (MI.getOpcode()) {
10600 case AArch64::ADDXri:
10601 case AArch64::ADDWri:
10602 assert(MI.getNumOperands() == 4 && "Wrong number of operands");
10603 assert(MI.getOperand(2).isImm() &&
10604 "Expected operand to be immediate");
10605 assert(MI.getOperand(1).isReg() &&
10606 "Expected operand to be a register");
10607 // Check if the add just increments sp. If so, we search for
10608 // matching sub instructions that decrement sp. If not, the
10609 // modification is illegal
10610 if (MI.getOperand(1).getReg() == AArch64::SP)
10611 SPValue += MI.getOperand(2).getImm();
10612 else
10613 return true;
10614 break;
10615 case AArch64::SUBXri:
10616 case AArch64::SUBWri:
10617 assert(MI.getNumOperands() == 4 && "Wrong number of operands");
10618 assert(MI.getOperand(2).isImm() &&
10619 "Expected operand to be immediate");
10620 assert(MI.getOperand(1).isReg() &&
10621 "Expected operand to be a register");
10622 // Check if the sub just decrements sp. If so, we search for
10623 // matching add instructions that increment sp. If not, the
10624 // modification is illegal
10625 if (MI.getOperand(1).getReg() == AArch64::SP)
10626 SPValue -= MI.getOperand(2).getImm();
10627 else
10628 return true;
10629 break;
10630 default:
10631 return true;
10632 }
10633 }
10634 }
10635 if (SPValue)
10636 return true;
10637 return false;
10638 };
10639 // Remove candidates with illegal stack modifying instructions
10640 llvm::erase_if(RepeatedSequenceLocs, hasIllegalSPModification);
10641
10642 // If the sequence doesn't have enough candidates left, then we're done.
10643 if (RepeatedSequenceLocs.size() < MinRepeats)
10644 return std::nullopt;
10645 }
10646
10647 // Properties about candidate MBBs that hold for all of them.
10648 unsigned FlagsSetInAll = 0xF;
10649
10650 // Compute liveness information for each candidate, and set FlagsSetInAll.
10651 for (outliner::Candidate &C : RepeatedSequenceLocs)
10652 FlagsSetInAll &= C.Flags;
10653
10654 unsigned LastInstrOpcode = RepeatedSequenceLocs[0].back().getOpcode();
10655
10656 // Helper lambda which sets call information for every candidate.
10657 auto SetCandidateCallInfo =
10658 [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) {
10659 for (outliner::Candidate &C : RepeatedSequenceLocs)
10660 C.setCallInfo(CallID, NumBytesForCall);
10661 };
10662
10663 unsigned FrameID = MachineOutlinerDefault;
10664 NumBytesToCreateFrame += 4;
10665
10666 bool HasBTI = any_of(RepeatedSequenceLocs, [](outliner::Candidate &C) {
10667 return C.getMF()->getInfo<AArch64FunctionInfo>()->branchTargetEnforcement();
10668 });
10669
10670 // We check to see if CFI Instructions are present, and if they are
10671 // we find the number of CFI Instructions in the candidates.
10672 unsigned CFICount = 0;
10673 for (auto &I : RepeatedSequenceLocs[0]) {
10674 if (I.isCFIInstruction())
10675 CFICount++;
10676 }
10677
10678 // We compare the number of found CFI Instructions to the number of CFI
10679 // instructions in the parent function for each candidate. We must check this
10680 // since if we outline one of the CFI instructions in a function, we have to
10681 // outline them all for correctness. If we do not, the address offsets will be
10682 // incorrect between the two sections of the program.
10683 for (outliner::Candidate &C : RepeatedSequenceLocs) {
10684 std::vector<MCCFIInstruction> CFIInstructions =
10685 C.getMF()->getFrameInstructions();
10686
10687 if (CFICount > 0 && CFICount != CFIInstructions.size())
10688 return std::nullopt;
10689 }
10690
10691 // Returns true if an instructions is safe to fix up, false otherwise.
10692 auto IsSafeToFixup = [this, &TRI](MachineInstr &MI) {
10693 if (MI.isCall())
10694 return true;
10695
10696 if (!MI.modifiesRegister(AArch64::SP, &TRI) &&
10697 !MI.readsRegister(AArch64::SP, &TRI))
10698 return true;
10699
10700 // Any modification of SP will break our code to save/restore LR.
10701 // FIXME: We could handle some instructions which add a constant
10702 // offset to SP, with a bit more work.
10703 if (MI.modifiesRegister(AArch64::SP, &TRI))
10704 return false;
10705
10706 // At this point, we have a stack instruction that we might need to
10707 // fix up. We'll handle it if it's a load or store.
10708 if (MI.mayLoadOrStore()) {
10709 const MachineOperand *Base; // Filled with the base operand of MI.
10710 int64_t Offset; // Filled with the offset of MI.
10711 bool OffsetIsScalable;
10712
10713 // Does it allow us to offset the base operand and is the base the
10714 // register SP?
10715 if (!getMemOperandWithOffset(MI, Base, Offset, OffsetIsScalable) ||
10716 !Base->isReg() || Base->getReg() != AArch64::SP)
10717 return false;
10718
10719 // Fixe-up code below assumes bytes.
10720 if (OffsetIsScalable)
10721 return false;
10722
10723 // Find the minimum/maximum offset for this instruction and check
10724 // if fixing it up would be in range.
10725 int64_t MinOffset,
10726 MaxOffset; // Unscaled offsets for the instruction.
10727 // The scale to multiply the offsets by.
10728 TypeSize Scale(0U, false), DummyWidth(0U, false);
10729 getMemOpInfo(MI.getOpcode(), Scale, DummyWidth, MinOffset, MaxOffset);
10730
10731 Offset += 16; // Update the offset to what it would be if we outlined.
10732 if (Offset < MinOffset * (int64_t)Scale.getFixedValue() ||
10733 Offset > MaxOffset * (int64_t)Scale.getFixedValue())
10734 return false;
10735
10736 // It's in range, so we can outline it.
10737 return true;
10738 }
10739
10740 // FIXME: Add handling for instructions like "add x0, sp, #8".
10741
10742 // We can't fix it up, so don't outline it.
10743 return false;
10744 };
10745
10746 // True if it's possible to fix up each stack instruction in this sequence.
10747 // Important for frames/call variants that modify the stack.
10748 bool AllStackInstrsSafe =
10749 llvm::all_of(RepeatedSequenceLocs[0], IsSafeToFixup);
10750
10751 // If the last instruction in any candidate is a terminator, then we should
10752 // tail call all of the candidates.
10753 if (RepeatedSequenceLocs[0].back().isTerminator()) {
10754 FrameID = MachineOutlinerTailCall;
10755 NumBytesToCreateFrame = 0;
10756 unsigned NumBytesForCall = 4 + NumBytesToCheckLRInTCEpilogue;
10757 SetCandidateCallInfo(MachineOutlinerTailCall, NumBytesForCall);
10758 }
10759
10760 else if (LastInstrOpcode == AArch64::BL ||
10761 ((LastInstrOpcode == AArch64::BLR ||
10762 LastInstrOpcode == AArch64::BLRNoIP) &&
10763 !HasBTI)) {
10764 // FIXME: Do we need to check if the code after this uses the value of LR?
10765 FrameID = MachineOutlinerThunk;
10766 NumBytesToCreateFrame = NumBytesToCheckLRInTCEpilogue;
10767 SetCandidateCallInfo(MachineOutlinerThunk, 4);
10768 }
10769
10770 else {
10771 // We need to decide how to emit calls + frames. We can always emit the same
10772 // frame if we don't need to save to the stack. If we have to save to the
10773 // stack, then we need a different frame.
10774 unsigned NumBytesNoStackCalls = 0;
10775 std::vector<outliner::Candidate> CandidatesWithoutStackFixups;
10776
10777 // Check if we have to save LR.
10778 for (outliner::Candidate &C : RepeatedSequenceLocs) {
10779 bool LRAvailable =
10781 ? C.isAvailableAcrossAndOutOfSeq(AArch64::LR, TRI)
10782 : true;
10783 // If we have a noreturn caller, then we're going to be conservative and
10784 // say that we have to save LR. If we don't have a ret at the end of the
10785 // block, then we can't reason about liveness accurately.
10786 //
10787 // FIXME: We can probably do better than always disabling this in
10788 // noreturn functions by fixing up the liveness info.
10789 bool IsNoReturn =
10790 C.getMF()->getFunction().hasFnAttribute(Attribute::NoReturn);
10791
10792 // Is LR available? If so, we don't need a save.
10793 if (LRAvailable && !IsNoReturn) {
10794 NumBytesNoStackCalls += 4;
10795 C.setCallInfo(MachineOutlinerNoLRSave, 4);
10796 CandidatesWithoutStackFixups.push_back(C);
10797 }
10798
10799 // Is an unused register available? If so, we won't modify the stack, so
10800 // we can outline with the same frame type as those that don't save LR.
10801 else if (findRegisterToSaveLRTo(C)) {
10802 NumBytesNoStackCalls += 12;
10803 C.setCallInfo(MachineOutlinerRegSave, 12);
10804 CandidatesWithoutStackFixups.push_back(C);
10805 }
10806
10807 // Is SP used in the sequence at all? If not, we don't have to modify
10808 // the stack, so we are guaranteed to get the same frame.
10809 else if (C.isAvailableInsideSeq(AArch64::SP, TRI)) {
10810 NumBytesNoStackCalls += 12;
10811 C.setCallInfo(MachineOutlinerDefault, 12);
10812 CandidatesWithoutStackFixups.push_back(C);
10813 }
10814
10815 // If we outline this, we need to modify the stack. Pretend we don't
10816 // outline this by saving all of its bytes.
10817 else {
10818 NumBytesNoStackCalls += SequenceSize;
10819 }
10820 }
10821
10822 // If there are no places where we have to save LR, then note that we
10823 // don't have to update the stack. Otherwise, give every candidate the
10824 // default call type, as long as it's safe to do so.
10825 if (!AllStackInstrsSafe ||
10826 NumBytesNoStackCalls <= RepeatedSequenceLocs.size() * 12) {
10827 RepeatedSequenceLocs = CandidatesWithoutStackFixups;
10828 FrameID = MachineOutlinerNoLRSave;
10829 if (RepeatedSequenceLocs.size() < MinRepeats)
10830 return std::nullopt;
10831 } else {
10832 SetCandidateCallInfo(MachineOutlinerDefault, 12);
10833
10834 // Bugzilla ID: 46767
10835 // TODO: Check if fixing up the stack more than once is safe so we can
10836 // outline these.
10837 //
10838 // An outline resulting in a caller that requires stack fixups at the
10839 // callsite to a callee that also requires stack fixups can happen when
10840 // there are no available registers at the candidate callsite for a
10841 // candidate that itself also has calls.
10842 //
10843 // In other words if function_containing_sequence in the following pseudo
10844 // assembly requires that we save LR at the point of the call, but there
10845 // are no available registers: in this case we save using SP and as a
10846 // result the SP offsets requires stack fixups by multiples of 16.
10847 //
10848 // function_containing_sequence:
10849 // ...
10850 // save LR to SP <- Requires stack instr fixups in OUTLINED_FUNCTION_N
10851 // call OUTLINED_FUNCTION_N
10852 // restore LR from SP
10853 // ...
10854 //
10855 // OUTLINED_FUNCTION_N:
10856 // save LR to SP <- Requires stack instr fixups in OUTLINED_FUNCTION_N
10857 // ...
10858 // bl foo
10859 // restore LR from SP
10860 // ret
10861 //
10862 // Because the code to handle more than one stack fixup does not
10863 // currently have the proper checks for legality, these cases will assert
10864 // in the AArch64 MachineOutliner. This is because the code to do this
10865 // needs more hardening, testing, better checks that generated code is
10866 // legal, etc and because it is only verified to handle a single pass of
10867 // stack fixup.
10868 //
10869 // The assert happens in AArch64InstrInfo::buildOutlinedFrame to catch
10870 // these cases until they are known to be handled. Bugzilla 46767 is
10871 // referenced in comments at the assert site.
10872 //
10873 // To avoid asserting (or generating non-legal code on noassert builds)
10874 // we remove all candidates which would need more than one stack fixup by
10875 // pruning the cases where the candidate has calls while also having no
10876 // available LR and having no available general purpose registers to copy
10877 // LR to (ie one extra stack save/restore).
10878 //
10879 if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
10880 erase_if(RepeatedSequenceLocs, [this, &TRI](outliner::Candidate &C) {
10881 auto IsCall = [](const MachineInstr &MI) { return MI.isCall(); };
10882 return (llvm::any_of(C, IsCall)) &&
10883 (!C.isAvailableAcrossAndOutOfSeq(AArch64::LR, TRI) ||
10884 !findRegisterToSaveLRTo(C));
10885 });
10886 }
10887 }
10888
10889 // If we dropped all of the candidates, bail out here.
10890 if (RepeatedSequenceLocs.size() < MinRepeats)
10891 return std::nullopt;
10892 }
10893
10894 // Does every candidate's MBB contain a call? If so, then we might have a call
10895 // in the range.
10896 if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
10897 // Check if the range contains a call. These require a save + restore of the
10898 // link register.
10899 outliner::Candidate &FirstCand = RepeatedSequenceLocs[0];
10900 bool ModStackToSaveLR = false;
10901 if (any_of(drop_end(FirstCand),
10902 [](const MachineInstr &MI) { return MI.isCall(); }))
10903 ModStackToSaveLR = true;
10904
10905 // Handle the last instruction separately. If this is a tail call, then the
10906 // last instruction is a call. We don't want to save + restore in this case.
10907 // However, it could be possible that the last instruction is a call without
10908 // it being valid to tail call this sequence. We should consider this as
10909 // well.
10910 else if (FrameID != MachineOutlinerThunk &&
10911 FrameID != MachineOutlinerTailCall && FirstCand.back().isCall())
10912 ModStackToSaveLR = true;
10913
10914 if (ModStackToSaveLR) {
10915 // We can't fix up the stack. Bail out.
10916 if (!AllStackInstrsSafe)
10917 return std::nullopt;
10918
10919 // Save + restore LR.
10920 NumBytesToCreateFrame += 8;
10921
10922 // Add the extra mov if we will save a frame record instead of just LR.
10924 RepeatedSequenceLocs, TRI))
10925 NumBytesToCreateFrame += 4;
10926 }
10927 }
10928
10929 // If we have CFI instructions, we can only outline if the outlined section
10930 // can be a tail call
10931 if (FrameID != MachineOutlinerTailCall && CFICount > 0)
10932 return std::nullopt;
10933
10934 return std::make_unique<outliner::OutlinedFunction>(
10935 RepeatedSequenceLocs, SequenceSize, NumBytesToCreateFrame, FrameID);
10936}
10937
10938void AArch64InstrInfo::mergeOutliningCandidateAttributes(
10939 Function &F, std::vector<outliner::Candidate> &Candidates) const {
10940 // If a bunch of candidates reach this point they must agree on their return
10941 // address signing. It is therefore enough to just consider the signing
10942 // behaviour of one of them
10943 const auto &CFn = Candidates.front().getMF()->getFunction();
10944
10945 if (CFn.hasFnAttribute("ptrauth-returns"))
10946 F.addFnAttr(CFn.getFnAttribute("ptrauth-returns"));
10947 if (CFn.hasFnAttribute("ptrauth-auth-traps"))
10948 F.addFnAttr(CFn.getFnAttribute("ptrauth-auth-traps"));
10949 // Since all candidates belong to the same module, just copy the
10950 // function-level attributes of an arbitrary function.
10951 if (CFn.hasFnAttribute("sign-return-address"))
10952 F.addFnAttr(CFn.getFnAttribute("sign-return-address"));
10953 if (CFn.hasFnAttribute("sign-return-address-key"))
10954 F.addFnAttr(CFn.getFnAttribute("sign-return-address-key"));
10955
10956 AArch64GenInstrInfo::mergeOutliningCandidateAttributes(F, Candidates);
10957}
10958
10959bool AArch64InstrInfo::isFunctionSafeToOutlineFrom(
10960 MachineFunction &MF, bool OutlineFromLinkOnceODRs) const {
10961 const Function &F = MF.getFunction();
10962
10963 // Can F be deduplicated by the linker? If it can, don't outline from it.
10964 if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage())
10965 return false;
10966
10967 // Don't outline from functions with section markings; the program could
10968 // expect that all the code is in the named section.
10969 // FIXME: Allow outlining from multiple functions with the same section
10970 // marking.
10971 if (F.hasSection())
10972 return false;
10973
10974 // Outlining from functions with redzones is unsafe since the outliner may
10975 // modify the stack. Check if hasRedZone is true or unknown; if yes, don't
10976 // outline from it.
10977 AArch64FunctionInfo *AFI = MF.getInfo<AArch64FunctionInfo>();
10978 if (!AFI || AFI->hasRedZone().value_or(true))
10979 return false;
10980
10981 // FIXME: Determine whether it is safe to outline from functions which contain
10982 // streaming-mode changes. We may need to ensure any smstart/smstop pairs are
10983 // outlined together and ensure it is safe to outline with async unwind info,
10984 // required for saving & restoring VG around calls.
10985 if (AFI->hasStreamingModeChanges())
10986 return false;
10987
10988 // FIXME: Teach the outliner to generate/handle Windows unwind info.
10990 return false;
10991
10992 // It's safe to outline from MF.
10993 return true;
10994}
10995
10997AArch64InstrInfo::getOutlinableRanges(MachineBasicBlock &MBB,
10998 unsigned &Flags) const {
11000 "Must track liveness!");
11002 std::pair<MachineBasicBlock::iterator, MachineBasicBlock::iterator>>
11003 Ranges;
11004 // According to the AArch64 Procedure Call Standard, the following are
11005 // undefined on entry/exit from a function call:
11006 //
11007 // * Registers x16, x17, (and thus w16, w17)
11008 // * Condition codes (and thus the NZCV register)
11009 //
11010 // If any of these registers are used inside or live across an outlined
11011 // function, then they may be modified later, either by the compiler or
11012 // some other tool (like the linker).
11013 //
11014 // To avoid outlining in these situations, partition each block into ranges
11015 // where these registers are dead. We will only outline from those ranges.
11016 LiveRegUnits LRU(getRegisterInfo());
11017 auto AreAllUnsafeRegsDead = [&LRU]() {
11018 return LRU.available(AArch64::W16) && LRU.available(AArch64::W17) &&
11019 LRU.available(AArch64::NZCV);
11020 };
11021
11022 // We need to know if LR is live across an outlining boundary later on in
11023 // order to decide how we'll create the outlined call, frame, etc.
11024 //
11025 // It's pretty expensive to check this for *every candidate* within a block.
11026 // That's some potentially n^2 behaviour, since in the worst case, we'd need
11027 // to compute liveness from the end of the block for O(n) candidates within
11028 // the block.
11029 //
11030 // So, to improve the average case, let's keep track of liveness from the end
11031 // of the block to the beginning of *every outlinable range*. If we know that
11032 // LR is available in every range we could outline from, then we know that
11033 // we don't need to check liveness for any candidate within that range.
11034 bool LRAvailableEverywhere = true;
11035 // Compute liveness bottom-up.
11036 LRU.addLiveOuts(MBB);
11037 // Update flags that require info about the entire MBB.
11038 auto UpdateWholeMBBFlags = [&Flags](const MachineInstr &MI) {
11039 if (MI.isCall() && !MI.isTerminator())
11041 };
11042 // Range: [RangeBegin, RangeEnd)
11043 MachineBasicBlock::instr_iterator RangeBegin, RangeEnd;
11044 unsigned RangeLen;
11045 auto CreateNewRangeStartingAt =
11046 [&RangeBegin, &RangeEnd,
11047 &RangeLen](MachineBasicBlock::instr_iterator NewBegin) {
11048 RangeBegin = NewBegin;
11049 RangeEnd = std::next(RangeBegin);
11050 RangeLen = 0;
11051 };
11052 auto SaveRangeIfNonEmpty = [&RangeLen, &Ranges, &RangeBegin, &RangeEnd]() {
11053 // At least one unsafe register is not dead. We do not want to outline at
11054 // this point. If it is long enough to outline from and does not cross a
11055 // bundle boundary, save the range [RangeBegin, RangeEnd).
11056 if (RangeLen <= 1)
11057 return;
11058 if (!RangeBegin.isEnd() && RangeBegin->isBundledWithPred())
11059 return;
11060 if (!RangeEnd.isEnd() && RangeEnd->isBundledWithPred())
11061 return;
11062 Ranges.emplace_back(RangeBegin, RangeEnd);
11063 };
11064 // Find the first point where all unsafe registers are dead.
11065 // FIND: <safe instr> <-- end of first potential range
11066 // SKIP: <unsafe def>
11067 // SKIP: ... everything between ...
11068 // SKIP: <unsafe use>
11069 auto FirstPossibleEndPt = MBB.instr_rbegin();
11070 for (; FirstPossibleEndPt != MBB.instr_rend(); ++FirstPossibleEndPt) {
11071 if (!FirstPossibleEndPt->isDebugInstr())
11072 LRU.stepBackward(*FirstPossibleEndPt);
11073 // Update flags that impact how we outline across the entire block,
11074 // regardless of safety.
11075 UpdateWholeMBBFlags(*FirstPossibleEndPt);
11076 if (AreAllUnsafeRegsDead())
11077 break;
11078 }
11079 // If we exhausted the entire block, we have no safe ranges to outline.
11080 if (FirstPossibleEndPt == MBB.instr_rend())
11081 return Ranges;
11082 // Current range.
11083 CreateNewRangeStartingAt(FirstPossibleEndPt->getIterator());
11084 // StartPt points to the first place where all unsafe registers
11085 // are dead (if there is any such point). Begin partitioning the MBB into
11086 // ranges.
11087 for (auto &MI : make_range(FirstPossibleEndPt, MBB.instr_rend())) {
11088 if (!MI.isDebugInstr())
11089 LRU.stepBackward(MI);
11090 UpdateWholeMBBFlags(MI);
11091 if (!AreAllUnsafeRegsDead()) {
11092 SaveRangeIfNonEmpty();
11093 CreateNewRangeStartingAt(MI.getIterator());
11094 continue;
11095 }
11096 LRAvailableEverywhere &= LRU.available(AArch64::LR);
11097 // RangeBegin may point at a debug instruction because the mapper ignores
11098 // debug instructions wherever they appear. Only count non-debug
11099 // instructions so debug info cannot make a short range outlinable.
11100 RangeBegin = MI.getIterator();
11101 if (!MI.isDebugInstr())
11102 ++RangeLen;
11103 }
11104 // Above loop misses the last (or only) range. If we are still safe, then
11105 // let's save the range.
11106 if (AreAllUnsafeRegsDead())
11107 SaveRangeIfNonEmpty();
11108 if (Ranges.empty())
11109 return Ranges;
11110 // We found the ranges bottom-up. Mapping expects the top-down. Reverse
11111 // the order.
11112 std::reverse(Ranges.begin(), Ranges.end());
11113 // If there is at least one outlinable range where LR is unavailable
11114 // somewhere, remember that.
11115 if (!LRAvailableEverywhere)
11117 return Ranges;
11118}
11119
11121AArch64InstrInfo::getOutliningTypeImpl(const MachineModuleInfo &MMI,
11123 unsigned Flags) const {
11124 MachineInstr &MI = *MIT;
11125
11126 // Don't outline anything used for return address signing. The outlined
11127 // function will get signed later if needed
11128 switch (MI.getOpcode()) {
11129 case AArch64::PACM:
11130 case AArch64::PACIASP:
11131 case AArch64::PACIBSP:
11132 case AArch64::PACIASPPC:
11133 case AArch64::PACIBSPPC:
11134 case AArch64::AUTIASP:
11135 case AArch64::AUTIBSP:
11136 case AArch64::AUTIASPPCi:
11137 case AArch64::AUTIASPPCr:
11138 case AArch64::AUTIBSPPCi:
11139 case AArch64::AUTIBSPPCr:
11140 case AArch64::RETAA:
11141 case AArch64::RETAB:
11142 case AArch64::RETAASPPCi:
11143 case AArch64::RETAASPPCr:
11144 case AArch64::RETABSPPCi:
11145 case AArch64::RETABSPPCr:
11146 case AArch64::EMITBKEY:
11147 case AArch64::PAUTH_PROLOGUE:
11148 case AArch64::PAUTH_EPILOGUE:
11150 }
11151
11152 // We can only outline these if we will tail call the outlined function, or
11153 // fix up the CFI offsets. Currently, CFI instructions are outlined only if
11154 // in a tail call.
11155 //
11156 // FIXME: If the proper fixups for the offset are implemented, this should be
11157 // possible.
11158 if (MI.isCFIInstruction())
11160
11161 // Is this a terminator for a basic block?
11162 if (MI.isTerminator())
11163 // TargetInstrInfo::getOutliningType has already filtered out anything
11164 // that would break this, so we can allow it here.
11166
11167 // Make sure none of the operands are un-outlinable.
11168 for (const MachineOperand &MOP : MI.operands()) {
11169 // A check preventing CFI indices was here before, but only CFI
11170 // instructions should have those.
11171 assert(!MOP.isCFIIndex());
11172
11173 // If it uses LR or W30 explicitly, then don't touch it.
11174 if (MOP.isReg() && !MOP.isImplicit() &&
11175 (MOP.getReg() == AArch64::LR || MOP.getReg() == AArch64::W30))
11177 }
11178
11179 // Special cases for instructions that can always be outlined, but will fail
11180 // the later tests. e.g, ADRPs, which are PC-relative use LR, but can always
11181 // be outlined because they don't require a *specific* value to be in LR.
11182 if (MI.getOpcode() == AArch64::ADRP)
11184
11185 // If MI is a call we might be able to outline it. We don't want to outline
11186 // any calls that rely on the position of items on the stack. When we outline
11187 // something containing a call, we have to emit a save and restore of LR in
11188 // the outlined function. Currently, this always happens by saving LR to the
11189 // stack. Thus, if we outline, say, half the parameters for a function call
11190 // plus the call, then we'll break the callee's expectations for the layout
11191 // of the stack.
11192 //
11193 // FIXME: Allow calls to functions which construct a stack frame, as long
11194 // as they don't access arguments on the stack.
11195 // FIXME: Figure out some way to analyze functions defined in other modules.
11196 // We should be able to compute the memory usage based on the IR calling
11197 // convention, even if we can't see the definition.
11198 if (MI.isCall()) {
11199 // Get the function associated with the call. Look at each operand and find
11200 // the one that represents the callee and get its name.
11201 const Function *Callee = nullptr;
11202 for (const MachineOperand &MOP : MI.operands()) {
11203 if (MOP.isGlobal()) {
11204 Callee = dyn_cast<Function>(MOP.getGlobal());
11205 break;
11206 }
11207 }
11208
11209 // Never outline calls to mcount. There isn't any rule that would require
11210 // this, but the Linux kernel's "ftrace" feature depends on it.
11211 if (Callee && Callee->getName() == "\01_mcount")
11213
11214 // If we don't know anything about the callee, assume it depends on the
11215 // stack layout of the caller. In that case, it's only legal to outline
11216 // as a tail-call. Explicitly list the call instructions we know about so we
11217 // don't get unexpected results with call pseudo-instructions.
11218 auto UnknownCallOutlineType = outliner::InstrType::Illegal;
11219 if (MI.getOpcode() == AArch64::BLR ||
11220 MI.getOpcode() == AArch64::BLRNoIP || MI.getOpcode() == AArch64::BL)
11221 UnknownCallOutlineType = outliner::InstrType::LegalTerminator;
11222
11223 if (!Callee)
11224 return UnknownCallOutlineType;
11225
11226 // We have a function we have information about. Check it if it's something
11227 // can safely outline.
11228 MachineFunction *CalleeMF = MMI.getMachineFunction(*Callee);
11229
11230 // We don't know what's going on with the callee at all. Don't touch it.
11231 if (!CalleeMF)
11232 return UnknownCallOutlineType;
11233
11234 // Check if we know anything about the callee saves on the function. If we
11235 // don't, then don't touch it, since that implies that we haven't
11236 // computed anything about its stack frame yet.
11237 MachineFrameInfo &MFI = CalleeMF->getFrameInfo();
11238 if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 ||
11239 MFI.getNumObjects() > 0)
11240 return UnknownCallOutlineType;
11241
11242 // At this point, we can say that CalleeMF ought to not pass anything on the
11243 // stack. Therefore, we can outline it.
11245 }
11246
11247 // Don't touch the link register or W30.
11248 if (MI.readsRegister(AArch64::W30, &getRegisterInfo()) ||
11249 MI.modifiesRegister(AArch64::W30, &getRegisterInfo()))
11251
11252 // Don't outline BTI instructions, because that will prevent the outlining
11253 // site from being indirectly callable.
11254 if (hasBTISemantics(MI))
11256
11258}
11259
11260void AArch64InstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const {
11261 for (MachineInstr &MI : MBB) {
11262 const MachineOperand *Base;
11263 TypeSize Width(0, false);
11264 int64_t Offset;
11265 bool OffsetIsScalable;
11266
11267 // Is this a load or store with an immediate offset with SP as the base?
11268 if (!MI.mayLoadOrStore() ||
11269 !getMemOperandWithOffsetWidth(MI, Base, Offset, OffsetIsScalable,
11270 Width) ||
11271 (Base->isReg() && Base->getReg() != AArch64::SP))
11272 continue;
11273
11274 // It is, so we have to fix it up.
11275 TypeSize Scale(0U, false);
11276 int64_t Dummy1, Dummy2;
11277
11278 MachineOperand &StackOffsetOperand = getMemOpBaseRegImmOfsOffsetOperand(MI);
11279 assert(StackOffsetOperand.isImm() && "Stack offset wasn't immediate!");
11280 getMemOpInfo(MI.getOpcode(), Scale, Width, Dummy1, Dummy2);
11281 assert(Scale != 0 && "Unexpected opcode!");
11282 assert(!OffsetIsScalable && "Expected offset to be a byte offset");
11283
11284 // We've pushed the return address to the stack, so add 16 to the offset.
11285 // This is safe, since we already checked if it would overflow when we
11286 // checked if this instruction was legal to outline.
11287 int64_t NewImm = (Offset + 16) / (int64_t)Scale.getFixedValue();
11288 StackOffsetOperand.setImm(NewImm);
11289 }
11290}
11291
11293 const AArch64InstrInfo *TII,
11294 bool ShouldSignReturnAddr) {
11295 if (!ShouldSignReturnAddr)
11296 return;
11297
11298 BuildMI(MBB, MBB.begin(), DebugLoc(), TII->get(AArch64::PAUTH_PROLOGUE))
11300 TII->createPauthEpilogueInstr(MBB, DebugLoc());
11301}
11302
11303void AArch64InstrInfo::buildOutlinedFrame(
11305 const outliner::OutlinedFunction &OF) const {
11306
11307 AArch64FunctionInfo *FI = MF.getInfo<AArch64FunctionInfo>();
11308
11309 if (OF.FrameConstructionID == MachineOutlinerTailCall)
11310 FI->setOutliningStyle("Tail Call");
11311 else if (OF.FrameConstructionID == MachineOutlinerThunk) {
11312 // For thunk outlining, rewrite the last instruction from a call to a
11313 // tail-call.
11314 MachineInstr *Call = &*--MBB.instr_end();
11315 unsigned TailOpcode;
11316 if (Call->getOpcode() == AArch64::BL) {
11317 TailOpcode = AArch64::TCRETURNdi;
11318 } else {
11319 assert(Call->getOpcode() == AArch64::BLR ||
11320 Call->getOpcode() == AArch64::BLRNoIP);
11321 TailOpcode = AArch64::TCRETURNriALL;
11322 }
11323 MachineInstr *TC = BuildMI(MF, DebugLoc(), get(TailOpcode))
11324 .add(Call->getOperand(0))
11325 .addImm(0);
11326 MBB.insert(MBB.end(), TC);
11328
11329 FI->setOutliningStyle("Thunk");
11330 }
11331
11332 bool IsLeafFunction = true;
11333
11334 // Is there a call in the outlined range?
11335 auto IsNonTailCall = [](const MachineInstr &MI) {
11336 return MI.isCall() && !MI.isReturn();
11337 };
11338
11339 if (llvm::any_of(MBB.instrs(), IsNonTailCall)) {
11340 // Fix up the instructions in the range, since we're going to modify the
11341 // stack.
11342
11343 // Bugzilla ID: 46767
11344 // TODO: Check if fixing up twice is safe so we can outline these.
11345 assert(OF.FrameConstructionID != MachineOutlinerDefault &&
11346 "Can only fix up stack references once");
11347 fixupPostOutline(MBB);
11348
11349 IsLeafFunction = false;
11350
11351 // LR has to be a live in so that we can save it.
11352 if (!MBB.isLiveIn(AArch64::LR))
11353 MBB.addLiveIn(AArch64::LR);
11354
11357
11358 if (OF.FrameConstructionID == MachineOutlinerTailCall ||
11359 OF.FrameConstructionID == MachineOutlinerThunk)
11360 Et = std::prev(MBB.end());
11361
11362 // There is a call in the range, so we must save LR. Save it as part of a
11363 // frame record when that gives us a smaller compact unwind encoding.
11365 // FP is saved here, so it must be live-in.
11366 if (!MBB.isLiveIn(AArch64::FP))
11367 MBB.addLiveIn(AArch64::FP);
11368
11369 // stp x29, x30, [sp, #-16]! (the pre-index imm is scaled by 8: -2 * 8)
11370 MachineInstr *STPXpre = BuildMI(MF, DebugLoc(), get(AArch64::STPXpre))
11371 .addReg(AArch64::SP, RegState::Define)
11372 .addReg(AArch64::FP)
11373 .addReg(AArch64::LR)
11374 .addReg(AArch64::SP)
11375 .addImm(-2);
11376 It = MBB.insert(It, STPXpre);
11377
11378 // mov x29, sp (add x29, sp, #0), so x29 points at the frame record.
11379 MachineInstr *SetFP = BuildMI(MF, DebugLoc(), get(AArch64::ADDXri))
11380 .addReg(AArch64::FP, RegState::Define)
11381 .addReg(AArch64::SP)
11382 .addImm(0)
11383 .addImm(0);
11384 MBB.insertAfter(It, SetFP);
11385
11386 // Describe the frame record with FP as the CFA. The encoder needs all
11387 // three to pick FRAME. No need to check for unwind info here: we only
11388 // get here if the function has it.
11389 CFIInstBuilder CFIBuilder(MBB, std::next(SetFP->getIterator()),
11391 CFIBuilder.buildDefCFA(AArch64::FP, 16);
11392 CFIBuilder.buildOffset(AArch64::LR, -8);
11393 CFIBuilder.buildOffset(AArch64::FP, -16);
11394
11395 // ldp x29, x30, [sp], #16
11396 MachineInstr *LDPXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDPXpost))
11397 .addReg(AArch64::SP, RegState::Define)
11398 .addReg(AArch64::FP, RegState::Define)
11399 .addReg(AArch64::LR, RegState::Define)
11400 .addReg(AArch64::SP)
11401 .addImm(2);
11402 Et = MBB.insert(Et, LDPXpost);
11403 } else {
11404 // Insert a save before the outlined region
11405 MachineInstr *STRXpre = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
11406 .addReg(AArch64::SP, RegState::Define)
11407 .addReg(AArch64::LR)
11408 .addReg(AArch64::SP)
11409 .addImm(-16);
11410 It = MBB.insert(It, STRXpre);
11411
11412 if (MF.getInfo<AArch64FunctionInfo>()->needsDwarfUnwindInfo(MF)) {
11413 CFIInstBuilder CFIBuilder(MBB, It, MachineInstr::FrameSetup);
11414
11415 // Add a CFI saying the stack was moved 16 B down.
11416 CFIBuilder.buildDefCFAOffset(16);
11417
11418 // Add a CFI saying that the LR that we want to find is now 16 B higher
11419 // than before.
11420 CFIBuilder.buildOffset(AArch64::LR, -16);
11421 }
11422
11423 // Insert a restore before the terminator for the function.
11424 MachineInstr *LDRXpost = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
11425 .addReg(AArch64::SP, RegState::Define)
11426 .addReg(AArch64::LR, RegState::Define)
11427 .addReg(AArch64::SP)
11428 .addImm(16);
11429 Et = MBB.insert(Et, LDRXpost);
11430 }
11431 }
11432
11433 auto RASignCondition = FI->getSignReturnAddressCondition();
11434 bool ShouldSignReturnAddr = AArch64FunctionInfo::shouldSignReturnAddress(
11435 RASignCondition, !IsLeafFunction);
11436
11437 // If this is a tail call outlined function, then there's already a return.
11438 if (OF.FrameConstructionID == MachineOutlinerTailCall ||
11439 OF.FrameConstructionID == MachineOutlinerThunk) {
11440 signOutlinedFunction(MF, MBB, this, ShouldSignReturnAddr);
11441 return;
11442 }
11443
11444 // It's not a tail call, so we have to insert the return ourselves.
11445
11446 // LR has to be a live in so that we can return to it.
11447 if (!MBB.isLiveIn(AArch64::LR))
11448 MBB.addLiveIn(AArch64::LR);
11449
11450 MachineInstr *ret = BuildMI(MF, DebugLoc(), get(AArch64::RET))
11451 .addReg(AArch64::LR);
11452 MBB.insert(MBB.end(), ret);
11453
11454 signOutlinedFunction(MF, MBB, this, ShouldSignReturnAddr);
11455
11456 FI->setOutliningStyle("Function");
11457
11458 // Did we have to modify the stack by saving the link register?
11459 if (OF.FrameConstructionID != MachineOutlinerDefault)
11460 return;
11461
11462 // We modified the stack.
11463 // Walk over the basic block and fix up all the stack accesses.
11464 fixupPostOutline(MBB);
11465}
11466
11467MachineBasicBlock::iterator AArch64InstrInfo::insertOutlinedCall(
11470
11471 // Are we tail calling?
11472 if (C.CallConstructionID == MachineOutlinerTailCall) {
11473 // If yes, then we can just branch to the label.
11474 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::TCRETURNdi))
11475 .addGlobalAddress(M.getNamedValue(MF.getName()))
11476 .addImm(0));
11477 return It;
11478 }
11479
11480 // Are we saving the link register?
11481 if (C.CallConstructionID == MachineOutlinerNoLRSave ||
11482 C.CallConstructionID == MachineOutlinerThunk) {
11483 // No, so just insert the call.
11484 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
11485 .addGlobalAddress(M.getNamedValue(MF.getName())));
11486 return It;
11487 }
11488
11489 // We want to return the spot where we inserted the call.
11491
11492 // Instructions for saving and restoring LR around the call instruction we're
11493 // going to insert.
11494 MachineInstr *Save;
11495 MachineInstr *Restore;
11496 // Can we save to a register?
11497 if (C.CallConstructionID == MachineOutlinerRegSave) {
11498 // FIXME: This logic should be sunk into a target-specific interface so that
11499 // we don't have to recompute the register.
11500 Register Reg = findRegisterToSaveLRTo(C);
11501 assert(Reg && "No callee-saved register available?");
11502
11503 // LR has to be a live in so that we can save it.
11504 if (!MBB.isLiveIn(AArch64::LR))
11505 MBB.addLiveIn(AArch64::LR);
11506
11507 // Save and restore LR from Reg.
11508 Save = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), Reg)
11509 .addReg(AArch64::XZR)
11510 .addReg(AArch64::LR)
11511 .addImm(0);
11512 Restore = BuildMI(MF, DebugLoc(), get(AArch64::ORRXrs), AArch64::LR)
11513 .addReg(AArch64::XZR)
11514 .addReg(Reg)
11515 .addImm(0);
11516 } else {
11517 // We have the default case. Save and restore from SP.
11518 Save = BuildMI(MF, DebugLoc(), get(AArch64::STRXpre))
11519 .addReg(AArch64::SP, RegState::Define)
11520 .addReg(AArch64::LR)
11521 .addReg(AArch64::SP)
11522 .addImm(-16);
11523 Restore = BuildMI(MF, DebugLoc(), get(AArch64::LDRXpost))
11524 .addReg(AArch64::SP, RegState::Define)
11525 .addReg(AArch64::LR, RegState::Define)
11526 .addReg(AArch64::SP)
11527 .addImm(16);
11528 }
11529
11530 It = MBB.insert(It, Save);
11531 It++;
11532
11533 // Insert the call.
11534 It = MBB.insert(It, BuildMI(MF, DebugLoc(), get(AArch64::BL))
11535 .addGlobalAddress(M.getNamedValue(MF.getName())));
11536 CallPt = It;
11537 It++;
11538
11539 It = MBB.insert(It, Restore);
11540 return CallPt;
11541}
11542
11543bool AArch64InstrInfo::shouldOutlineFromFunctionByDefault(
11544 MachineFunction &MF) const {
11545 return MF.getFunction().hasMinSize();
11546}
11547
11548void AArch64InstrInfo::buildClearRegister(Register Reg, MachineBasicBlock &MBB,
11550 DebugLoc &DL,
11551 bool AllowSideEffects) const {
11552 const MachineFunction &MF = *MBB.getParent();
11553 const AArch64Subtarget &STI = MF.getSubtarget<AArch64Subtarget>();
11554 const AArch64RegisterInfo &TRI = *STI.getRegisterInfo();
11555
11556 if (TRI.isGeneralPurposeRegister(MF, Reg)) {
11557 BuildMI(MBB, Iter, DL, get(AArch64::MOVZXi), Reg).addImm(0).addImm(0);
11558 } else if (STI.isSVEorStreamingSVEAvailable()) {
11559 BuildMI(MBB, Iter, DL, get(AArch64::DUP_ZI_D), Reg)
11560 .addImm(0)
11561 .addImm(0);
11562 } else if (STI.isNeonAvailable()) {
11563 BuildMI(MBB, Iter, DL, get(AArch64::MOVIv2d_ns), Reg)
11564 .addImm(0);
11565 } else {
11566 // No Advanced SIMD (streaming-compatible without SVE, or +nosimd), so use
11567 // `fmov d...` instead of `movi v...`; writing `d` also clears the upper
11568 // 64 bits.
11569 assert(STI.hasFPARMv8() && "Expected FP to be available.");
11570 Register Reg64 = TRI.getSubReg(Reg, AArch64::dsub);
11571 BuildMI(MBB, Iter, DL, get(AArch64::FMOVD0), Reg64);
11572 }
11573}
11574
11575std::optional<DestSourcePair>
11577
11578 // AArch64::ORRWrs and AArch64::ORRXrs with WZR/XZR reg
11579 // and zero immediate operands used as an alias for mov instruction.
11580 if ((MI.getOpcode() == AArch64::ORRWrs &&
11581 MI.getOperand(1).getReg() == AArch64::WZR &&
11582 MI.getOperand(3).getImm() == 0x0) ||
11583 (MI.getOpcode() == AArch64::ORRWrr &&
11584 MI.getOperand(1).getReg() == AArch64::WZR)) {
11585 // Check that the w->w move is not a zero-extending w->x mov.
11586 if ((MI.getOperand(0).getReg().isPhysical() &&
11587 MI.findRegisterDefOperandIdx(
11588 getXRegFromWReg(MI.getOperand(0).getReg()),
11589 /*TRI=*/nullptr) == -1) ||
11590 (MI.getOperand(0).getReg().isVirtual() &&
11591 !MI.getOperand(0).getSubReg()))
11592 return DestSourcePair{MI.getOperand(0), MI.getOperand(2)};
11593 }
11594
11595 if (MI.getOpcode() == AArch64::ORRXrs &&
11596 MI.getOperand(1).getReg() == AArch64::XZR &&
11597 MI.getOperand(3).getImm() == 0x0)
11598 return DestSourcePair{MI.getOperand(0), MI.getOperand(2)};
11599
11600 return std::nullopt;
11601}
11602
11603std::optional<DestSourcePair>
11605 if ((MI.getOpcode() == AArch64::ORRWrs &&
11606 MI.getOperand(1).getReg() == AArch64::WZR &&
11607 MI.getOperand(3).getImm() == 0x0) ||
11608 (MI.getOpcode() == AArch64::ORRWrr &&
11609 MI.getOperand(1).getReg() == AArch64::WZR))
11610 return DestSourcePair{MI.getOperand(0), MI.getOperand(2)};
11611 return std::nullopt;
11612}
11613
11614std::optional<RegImmPair>
11615AArch64InstrInfo::isAddImmediate(const MachineInstr &MI, Register Reg) const {
11616 int Sign = 1;
11617 int64_t Offset = 0;
11618
11619 // TODO: Handle cases where Reg is a super- or sub-register of the
11620 // destination register.
11621 const MachineOperand &Op0 = MI.getOperand(0);
11622 if (!Op0.isReg() || Reg != Op0.getReg())
11623 return std::nullopt;
11624
11625 switch (MI.getOpcode()) {
11626 default:
11627 return std::nullopt;
11628 case AArch64::SUBWri:
11629 case AArch64::SUBXri:
11630 case AArch64::SUBSWri:
11631 case AArch64::SUBSXri:
11632 Sign *= -1;
11633 [[fallthrough]];
11634 case AArch64::ADDSWri:
11635 case AArch64::ADDSXri:
11636 case AArch64::ADDWri:
11637 case AArch64::ADDXri: {
11638 // TODO: Third operand can be global address (usually some string).
11639 if (!MI.getOperand(0).isReg() || !MI.getOperand(1).isReg() ||
11640 !MI.getOperand(2).isImm())
11641 return std::nullopt;
11642 int Shift = MI.getOperand(3).getImm();
11643 assert((Shift == 0 || Shift == 12) && "Shift can be either 0 or 12");
11644 Offset = Sign * (MI.getOperand(2).getImm() << Shift);
11645 }
11646 }
11647 return RegImmPair{MI.getOperand(1).getReg(), Offset};
11648}
11649
11650/// If the given ORR instruction is a copy, and \p DescribedReg overlaps with
11651/// the destination register then, if possible, describe the value in terms of
11652/// the source register.
11653static std::optional<ParamLoadedValue>
11655 const TargetInstrInfo *TII,
11656 const TargetRegisterInfo *TRI) {
11657 auto DestSrc = TII->isCopyLikeInstr(MI);
11658 if (!DestSrc)
11659 return std::nullopt;
11660
11661 Register DestReg = DestSrc->Destination->getReg();
11662 Register SrcReg = DestSrc->Source->getReg();
11663
11664 if (!DestReg.isValid() || !SrcReg.isValid())
11665 return std::nullopt;
11666
11667 auto Expr = DIExpression::get(MI.getMF()->getFunction().getContext(), {});
11668
11669 // If the described register is the destination, just return the source.
11670 if (DestReg == DescribedReg)
11671 return ParamLoadedValue(MachineOperand::CreateReg(SrcReg, false), Expr);
11672
11673 // ORRWrs zero-extends to 64-bits, so we need to consider such cases.
11674 if (MI.getOpcode() == AArch64::ORRWrs &&
11675 TRI->isSuperRegister(DestReg, DescribedReg))
11676 return ParamLoadedValue(MachineOperand::CreateReg(SrcReg, false), Expr);
11677
11678 // We may need to describe the lower part of a ORRXrs move.
11679 if (MI.getOpcode() == AArch64::ORRXrs &&
11680 TRI->isSubRegister(DestReg, DescribedReg)) {
11681 Register SrcSubReg = TRI->getSubReg(SrcReg, AArch64::sub_32);
11682 return ParamLoadedValue(MachineOperand::CreateReg(SrcSubReg, false), Expr);
11683 }
11684
11685 assert(!TRI->isSuperOrSubRegisterEq(DestReg, DescribedReg) &&
11686 "Unhandled ORR[XW]rs copy case");
11687
11688 return std::nullopt;
11689}
11690
11691bool AArch64InstrInfo::isFunctionSafeToSplit(const MachineFunction &MF) const {
11692 // Functions cannot be split to different sections on AArch64 if they have
11693 // a red zone. This is because relaxing a cross-section branch may require
11694 // incrementing the stack pointer to spill a register, which would overwrite
11695 // the red zone.
11696 if (MF.getInfo<AArch64FunctionInfo>()->hasRedZone().value_or(true))
11697 return false;
11698
11700}
11701
11702bool AArch64InstrInfo::isMBBSafeToSplitToCold(
11703 const MachineBasicBlock &MBB) const {
11704 // Asm Goto blocks can contain conditional branches to goto labels, which can
11705 // get moved out of range of the branch instruction.
11706 auto isAsmGoto = [](const MachineInstr &MI) {
11707 return MI.getOpcode() == AArch64::INLINEASM_BR;
11708 };
11709 if (llvm::any_of(MBB, isAsmGoto) || MBB.isInlineAsmBrIndirectTarget())
11710 return false;
11711
11712 // Because jump tables are label-relative instead of table-relative, they all
11713 // must be in the same section or relocation fixup handling will fail.
11714
11715 // Check if MBB is a jump table target
11716 const MachineJumpTableInfo *MJTI = MBB.getParent()->getJumpTableInfo();
11717 auto containsMBB = [&MBB](const MachineJumpTableEntry &JTE) {
11718 return llvm::is_contained(JTE.MBBs, &MBB);
11719 };
11720 if (MJTI != nullptr && llvm::any_of(MJTI->getJumpTables(), containsMBB))
11721 return false;
11722
11723 // Check if MBB contains a jump table lookup
11724 for (const MachineInstr &MI : MBB) {
11725 switch (MI.getOpcode()) {
11726 case TargetOpcode::G_BRJT:
11727 case AArch64::JumpTableDest32:
11728 case AArch64::JumpTableDest16:
11729 case AArch64::JumpTableDest8:
11730 return false;
11731 default:
11732 continue;
11733 }
11734 }
11735
11736 // MBB isn't a special case, so it's safe to be split to the cold section.
11737 return true;
11738}
11739
11740std::optional<ParamLoadedValue>
11741AArch64InstrInfo::describeLoadedValue(const MachineInstr &MI,
11742 Register Reg) const {
11743 const MachineFunction *MF = MI.getMF();
11744 const TargetRegisterInfo *TRI = MF->getSubtarget().getRegisterInfo();
11745 switch (MI.getOpcode()) {
11746 case AArch64::MOVZWi:
11747 case AArch64::MOVZXi: {
11748 // MOVZWi may be used for producing zero-extended 32-bit immediates in
11749 // 64-bit parameters, so we need to consider super-registers.
11750 if (!TRI->isSuperRegisterEq(MI.getOperand(0).getReg(), Reg))
11751 return std::nullopt;
11752
11753 if (!MI.getOperand(1).isImm())
11754 return std::nullopt;
11755 int64_t Immediate = MI.getOperand(1).getImm();
11756 int Shift = MI.getOperand(2).getImm();
11757 return ParamLoadedValue(MachineOperand::CreateImm(Immediate << Shift),
11758 nullptr);
11759 }
11760 case AArch64::ORRWrs:
11761 case AArch64::ORRXrs:
11762 return describeORRLoadedValue(MI, Reg, this, TRI);
11763 }
11764
11766}
11767
11768bool AArch64InstrInfo::isExtendLikelyToBeFolded(
11769 MachineInstr &ExtMI, MachineRegisterInfo &MRI) const {
11770 assert(ExtMI.getOpcode() == TargetOpcode::G_SEXT ||
11771 ExtMI.getOpcode() == TargetOpcode::G_ZEXT ||
11772 ExtMI.getOpcode() == TargetOpcode::G_ANYEXT);
11773
11774 // Anyexts are nops.
11775 if (ExtMI.getOpcode() == TargetOpcode::G_ANYEXT)
11776 return true;
11777
11778 Register DefReg = ExtMI.getOperand(0).getReg();
11779 if (!MRI.hasOneNonDBGUse(DefReg))
11780 return false;
11781
11782 // It's likely that a sext/zext as a G_PTR_ADD offset will be folded into an
11783 // addressing mode.
11784 auto *UserMI = &*MRI.use_instr_nodbg_begin(DefReg);
11785 return UserMI->getOpcode() == TargetOpcode::G_PTR_ADD;
11786}
11787
11788uint64_t AArch64InstrInfo::getElementSizeForOpcode(unsigned Opc) const {
11789 return get(Opc).TSFlags & AArch64::ElementSizeMask;
11790}
11791
11792bool AArch64InstrInfo::isPTestLikeOpcode(unsigned Opc) const {
11793 return get(Opc).TSFlags & AArch64::InstrFlagIsPTestLike;
11794}
11795
11796bool AArch64InstrInfo::isWhileOpcode(unsigned Opc) const {
11797 return get(Opc).TSFlags & AArch64::InstrFlagIsWhile;
11798}
11799
11800unsigned int
11801AArch64InstrInfo::getTailDuplicateSize(CodeGenOptLevel OptLevel) const {
11802 return OptLevel >= CodeGenOptLevel::Aggressive ? 6 : 2;
11803}
11804
11805bool AArch64InstrInfo::isLegalAddressingMode(unsigned NumBytes, int64_t Offset,
11806 unsigned Scale) const {
11807 if (Offset && Scale)
11808 return false;
11809
11810 // Check Reg + Imm
11811 if (!Scale) {
11812 // 9-bit signed offset
11813 if (isInt<9>(Offset))
11814 return true;
11815
11816 // 12-bit unsigned offset
11817 unsigned Shift = Log2_64(NumBytes);
11818 if (NumBytes && Offset > 0 && (Offset / NumBytes) <= (1LL << 12) - 1 &&
11819 // Must be a multiple of NumBytes (NumBytes is a power of 2)
11820 (Offset >> Shift) << Shift == Offset)
11821 return true;
11822 return false;
11823 }
11824
11825 // Check reg1 + SIZE_IN_BYTES * reg2 and reg1 + reg2
11826 return Scale == 1 || (Scale > 0 && Scale == NumBytes);
11827}
11828
11830 if (MF.getSubtarget<AArch64Subtarget>().hardenSlsBlr())
11831 return AArch64::BLRNoIP;
11832 else
11833 return AArch64::BLR;
11834}
11835
11837 DebugLoc DL) const {
11838 MachineBasicBlock::iterator InsertPt = MBB.getFirstTerminator();
11839 auto Builder = BuildMI(MBB, InsertPt, DL, get(AArch64::PAUTH_EPILOGUE))
11841
11842 MachineFunction &MF = *MBB.getParent();
11843 const auto *AFI = MF.getInfo<AArch64FunctionInfo>();
11844 auto &AFL = *static_cast<const AArch64FrameLowering *>(
11845 MF.getSubtarget().getFrameLowering());
11846 if (AFL.getArgumentStackToRestore(MF, MBB)) {
11847 Builder.addReg(AArch64::X17, RegState::ImplicitDefine);
11848 Builder.addReg(AArch64::X16, RegState::ImplicitDefine);
11849 if (AFI->branchProtectionPAuthLR())
11850 Builder.addReg(AArch64::X15, RegState::ImplicitDefine);
11851 return;
11852 }
11853
11854 if (AFI->branchProtectionPAuthLR() && !Subtarget.hasPAuthLR())
11855 Builder.addReg(AArch64::X16, RegState::ImplicitDefine);
11856}
11857
11859AArch64InstrInfo::probedStackAlloc(MachineBasicBlock::iterator MBBI,
11860 Register TargetReg, bool FrameSetup) const {
11861 assert(TargetReg != AArch64::SP && "New top of stack cannot already be in SP");
11862
11863 MachineBasicBlock &MBB = *MBBI->getParent();
11864 MachineFunction &MF = *MBB.getParent();
11865 const AArch64InstrInfo *TII =
11866 MF.getSubtarget<AArch64Subtarget>().getInstrInfo();
11867 int64_t ProbeSize = MF.getInfo<AArch64FunctionInfo>()->getStackProbeSize();
11868 DebugLoc DL = MBB.findDebugLoc(MBBI);
11869
11870 MachineFunction::iterator MBBInsertPoint = std::next(MBB.getIterator());
11871 MachineBasicBlock *LoopTestMBB =
11872 MF.CreateMachineBasicBlock(MBB.getBasicBlock());
11873 MF.insert(MBBInsertPoint, LoopTestMBB);
11874 MachineBasicBlock *LoopBodyMBB =
11875 MF.CreateMachineBasicBlock(MBB.getBasicBlock());
11876 MF.insert(MBBInsertPoint, LoopBodyMBB);
11877 MachineBasicBlock *ExitMBB = MF.CreateMachineBasicBlock(MBB.getBasicBlock());
11878 MF.insert(MBBInsertPoint, ExitMBB);
11879 MachineInstr::MIFlag Flags =
11881
11882 // LoopTest:
11883 // SUB SP, SP, #ProbeSize
11884 emitFrameOffset(*LoopTestMBB, LoopTestMBB->end(), DL, AArch64::SP,
11885 AArch64::SP, StackOffset::getFixed(-ProbeSize), TII, Flags);
11886
11887 // CMP SP, TargetReg
11888 BuildMI(*LoopTestMBB, LoopTestMBB->end(), DL, TII->get(AArch64::SUBSXrx64),
11889 AArch64::XZR)
11890 .addReg(AArch64::SP)
11891 .addReg(TargetReg)
11893 .setMIFlags(Flags);
11894
11895 // B.<Cond> LoopExit
11896 BuildMI(*LoopTestMBB, LoopTestMBB->end(), DL, TII->get(AArch64::Bcc))
11898 .addMBB(ExitMBB)
11899 .setMIFlags(Flags);
11900
11901 // LDR XZR, [SP]
11902 BuildMI(*LoopBodyMBB, LoopBodyMBB->end(), DL, TII->get(AArch64::LDRXui))
11903 .addDef(AArch64::XZR)
11904 .addReg(AArch64::SP)
11905 .addImm(0)
11909 Align(8)))
11910 .setMIFlags(Flags);
11911
11912 // B loop
11913 BuildMI(*LoopBodyMBB, LoopBodyMBB->end(), DL, TII->get(AArch64::B))
11914 .addMBB(LoopTestMBB)
11915 .setMIFlags(Flags);
11916
11917 // LoopExit:
11918 // MOV SP, TargetReg
11919 BuildMI(*ExitMBB, ExitMBB->end(), DL, TII->get(AArch64::ADDXri), AArch64::SP)
11920 .addReg(TargetReg)
11921 .addImm(0)
11923 .setMIFlags(Flags);
11924
11925 // LDR XZR, [SP]
11926 BuildMI(*ExitMBB, ExitMBB->end(), DL, TII->get(AArch64::LDRXui))
11927 .addReg(AArch64::XZR, RegState::Define)
11928 .addReg(AArch64::SP)
11929 .addImm(0)
11930 .setMIFlags(Flags);
11931
11932 ExitMBB->splice(ExitMBB->end(), &MBB, std::next(MBBI), MBB.end());
11934
11935 LoopTestMBB->addSuccessor(ExitMBB);
11936 LoopTestMBB->addSuccessor(LoopBodyMBB);
11937 LoopBodyMBB->addSuccessor(LoopTestMBB);
11938 MBB.addSuccessor(LoopTestMBB);
11939
11940 // Update liveins.
11941 if (MF.getRegInfo().reservedRegsFrozen())
11942 fullyRecomputeLiveIns({ExitMBB, LoopBodyMBB, LoopTestMBB});
11943
11944 return ExitMBB->begin();
11945}
11946
11947namespace {
11948class AArch64PipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
11949 MachineFunction *MF;
11950 const TargetInstrInfo *TII;
11951 const TargetRegisterInfo *TRI;
11952 MachineRegisterInfo &MRI;
11953
11954 /// The block of the loop
11955 MachineBasicBlock *LoopBB;
11956 /// The conditional branch of the loop
11957 MachineInstr *CondBranch;
11958 /// The compare instruction for loop control
11959 MachineInstr *Comp;
11960 /// The number of the operand of the loop counter value in Comp
11961 unsigned CompCounterOprNum;
11962 /// The instruction that updates the loop counter value
11963 MachineInstr *Update;
11964 /// The number of the operand of the loop counter value in Update
11965 unsigned UpdateCounterOprNum;
11966 /// The initial value of the loop counter
11967 Register Init;
11968 /// True iff Update is a predecessor of Comp
11969 bool IsUpdatePriorComp;
11970
11971 /// The normalized condition used by createTripCountGreaterCondition()
11972 SmallVector<MachineOperand, 4> Cond;
11973
11974public:
11975 AArch64PipelinerLoopInfo(MachineBasicBlock *LoopBB, MachineInstr *CondBranch,
11976 MachineInstr *Comp, unsigned CompCounterOprNum,
11977 MachineInstr *Update, unsigned UpdateCounterOprNum,
11978 Register Init, bool IsUpdatePriorComp,
11979 const SmallVectorImpl<MachineOperand> &Cond)
11980 : MF(Comp->getParent()->getParent()),
11981 TII(MF->getSubtarget().getInstrInfo()),
11982 TRI(MF->getSubtarget().getRegisterInfo()), MRI(MF->getRegInfo()),
11983 LoopBB(LoopBB), CondBranch(CondBranch), Comp(Comp),
11984 CompCounterOprNum(CompCounterOprNum), Update(Update),
11985 UpdateCounterOprNum(UpdateCounterOprNum), Init(Init),
11986 IsUpdatePriorComp(IsUpdatePriorComp), Cond(Cond.begin(), Cond.end()) {}
11987
11988 bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
11989 // Make the instructions for loop control be placed in stage 0.
11990 // The predecessors of Comp are considered by the caller.
11991 return MI == Comp;
11992 }
11993
11994 std::optional<bool> createTripCountGreaterCondition(
11995 int TC, MachineBasicBlock &MBB,
11996 SmallVectorImpl<MachineOperand> &CondParam) override {
11997 // A branch instruction will be inserted as "if (Cond) goto epilogue".
11998 // Cond is normalized for such use.
11999 // The predecessors of the branch are assumed to have already been inserted.
12000 CondParam = Cond;
12001 return {};
12002 }
12003
12004 void createRemainingIterationsGreaterCondition(
12005 int TC, MachineBasicBlock &MBB, SmallVectorImpl<MachineOperand> &Cond,
12006 DenseMap<MachineInstr *, MachineInstr *> &LastStage0Insts) override;
12007
12008 void setPreheader(MachineBasicBlock *NewPreheader) override {}
12009
12010 void adjustTripCount(int TripCountAdjust) override {}
12011
12012 bool isMVEExpanderSupported() override { return true; }
12013};
12014} // namespace
12015
12016/// Clone an instruction from MI. The register of ReplaceOprNum-th operand
12017/// is replaced by ReplaceReg. The output register is newly created.
12018/// The other operands are unchanged from MI.
12019static Register cloneInstr(const MachineInstr *MI, unsigned ReplaceOprNum,
12020 Register ReplaceReg, MachineBasicBlock &MBB,
12021 MachineBasicBlock::iterator InsertTo) {
12022 MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
12023 const TargetInstrInfo *TII = MBB.getParent()->getSubtarget().getInstrInfo();
12024 MachineInstr *NewMI = MBB.getParent()->CloneMachineInstr(MI);
12025 Register Result = 0;
12026 for (unsigned I = 0; I < NewMI->getNumOperands(); ++I) {
12027 if (I == 0 && NewMI->getOperand(0).getReg().isVirtual()) {
12028 Result = MRI.createVirtualRegister(
12029 MRI.getRegClass(NewMI->getOperand(0).getReg()));
12030 NewMI->getOperand(I).setReg(Result);
12031 } else if (I == ReplaceOprNum) {
12032 MRI.constrainRegClass(ReplaceReg, TII->getRegClass(NewMI->getDesc(), I));
12033 NewMI->getOperand(I).setReg(ReplaceReg);
12034 }
12035 }
12036 MBB.insert(InsertTo, NewMI);
12037 return Result;
12038}
12039
12040void AArch64PipelinerLoopInfo::createRemainingIterationsGreaterCondition(
12043 // Create and accumulate conditions for next TC iterations.
12044 // Example:
12045 // SUBSXrr N, counter, implicit-def $nzcv # compare instruction for the last
12046 // # iteration of the kernel
12047 //
12048 // # insert the following instructions
12049 // cond = CSINCXr 0, 0, C, implicit $nzcv
12050 // counter = ADDXri counter, 1 # clone from this->Update
12051 // SUBSXrr n, counter, implicit-def $nzcv # clone from this->Comp
12052 // cond = CSINCXr cond, cond, C, implicit $nzcv
12053 // ... (repeat TC times)
12054 // SUBSXri cond, 0, implicit-def $nzcv
12055
12056 assert(CondBranch->getOpcode() == AArch64::Bcc);
12057 // CondCode to exit the loop
12059 (AArch64CC::CondCode)CondBranch->getOperand(0).getImm();
12060 if (CondBranch->getOperand(1).getMBB() == LoopBB)
12062
12063 // Accumulate conditions to exit the loop
12064 Register AccCond = AArch64::XZR;
12065
12066 // If CC holds, CurCond+1 is returned; otherwise CurCond is returned.
12067 auto AccumulateCond = [&](Register CurCond,
12069 Register NewCond = MRI.createVirtualRegister(&AArch64::GPR64commonRegClass);
12070 BuildMI(MBB, MBB.end(), Comp->getDebugLoc(), TII->get(AArch64::CSINCXr))
12071 .addReg(NewCond, RegState::Define)
12072 .addReg(CurCond)
12073 .addReg(CurCond)
12075 return NewCond;
12076 };
12077
12078 if (!LastStage0Insts.empty() && LastStage0Insts[Comp]->getParent() == &MBB) {
12079 // Update and Comp for I==0 are already exists in MBB
12080 // (MBB is an unrolled kernel)
12081 Register Counter;
12082 for (int I = 0; I <= TC; ++I) {
12083 Register NextCounter;
12084 if (I != 0)
12085 NextCounter =
12086 cloneInstr(Comp, CompCounterOprNum, Counter, MBB, MBB.end());
12087
12088 AccCond = AccumulateCond(AccCond, CC);
12089
12090 if (I != TC) {
12091 if (I == 0) {
12092 if (Update != Comp && IsUpdatePriorComp) {
12093 Counter =
12094 LastStage0Insts[Comp]->getOperand(CompCounterOprNum).getReg();
12095 NextCounter = cloneInstr(Update, UpdateCounterOprNum, Counter, MBB,
12096 MBB.end());
12097 } else {
12098 // can use already calculated value
12099 NextCounter = LastStage0Insts[Update]->getOperand(0).getReg();
12100 }
12101 } else if (Update != Comp) {
12102 NextCounter =
12103 cloneInstr(Update, UpdateCounterOprNum, Counter, MBB, MBB.end());
12104 }
12105 }
12106 Counter = NextCounter;
12107 }
12108 } else {
12109 Register Counter;
12110 if (LastStage0Insts.empty()) {
12111 // use initial counter value (testing if the trip count is sufficient to
12112 // be executed by pipelined code)
12113 Counter = Init;
12114 if (IsUpdatePriorComp)
12115 Counter =
12116 cloneInstr(Update, UpdateCounterOprNum, Counter, MBB, MBB.end());
12117 } else {
12118 // MBB is an epilogue block. LastStage0Insts[Comp] is in the kernel block.
12119 Counter = LastStage0Insts[Comp]->getOperand(CompCounterOprNum).getReg();
12120 }
12121
12122 for (int I = 0; I <= TC; ++I) {
12123 Register NextCounter;
12124 NextCounter =
12125 cloneInstr(Comp, CompCounterOprNum, Counter, MBB, MBB.end());
12126 AccCond = AccumulateCond(AccCond, CC);
12127 if (I != TC && Update != Comp)
12128 NextCounter =
12129 cloneInstr(Update, UpdateCounterOprNum, Counter, MBB, MBB.end());
12130 Counter = NextCounter;
12131 }
12132 }
12133
12134 // If AccCond == 0, the remainder is greater than TC.
12135 BuildMI(MBB, MBB.end(), Comp->getDebugLoc(), TII->get(AArch64::SUBSXri))
12136 .addReg(AArch64::XZR, RegState::Define | RegState::Dead)
12137 .addReg(AccCond)
12138 .addImm(0)
12139 .addImm(0);
12140 Cond.clear();
12142}
12143
12144static void extractPhiReg(const MachineInstr &Phi, const MachineBasicBlock *MBB,
12145 Register &RegMBB, Register &RegOther) {
12146 assert(Phi.getNumOperands() == 5);
12147 if (Phi.getOperand(2).getMBB() == MBB) {
12148 RegMBB = Phi.getOperand(1).getReg();
12149 RegOther = Phi.getOperand(3).getReg();
12150 } else {
12151 assert(Phi.getOperand(4).getMBB() == MBB);
12152 RegMBB = Phi.getOperand(3).getReg();
12153 RegOther = Phi.getOperand(1).getReg();
12154 }
12155}
12156
12158 if (!Reg.isVirtual())
12159 return false;
12160 const MachineRegisterInfo &MRI = BB->getParent()->getRegInfo();
12161 return MRI.getDefBlock(Reg) != BB;
12162}
12163
12164/// If Reg is an induction variable, return true and set some parameters
12165static bool getIndVarInfo(Register Reg, const MachineBasicBlock *LoopBB,
12166 MachineInstr *&UpdateInst,
12167 unsigned &UpdateCounterOprNum, Register &InitReg,
12168 bool &IsUpdatePriorComp) {
12169 // Example:
12170 //
12171 // Preheader:
12172 // InitReg = ...
12173 // LoopBB:
12174 // Reg0 = PHI (InitReg, Preheader), (Reg1, LoopBB)
12175 // Reg = COPY Reg0 ; COPY is ignored.
12176 // Reg1 = ADD Reg, #1; UpdateInst. Incremented by a loop invariant value.
12177 // ; Reg is the value calculated in the previous
12178 // ; iteration, so IsUpdatePriorComp == false.
12179
12180 if (LoopBB->pred_size() != 2)
12181 return false;
12182 if (!Reg.isVirtual())
12183 return false;
12184 const MachineRegisterInfo &MRI = LoopBB->getParent()->getRegInfo();
12185 UpdateInst = nullptr;
12186 UpdateCounterOprNum = 0;
12187 InitReg = 0;
12188 IsUpdatePriorComp = true;
12189 Register CurReg = Reg;
12190 while (true) {
12191 MachineInstr *Def = MRI.getVRegDef(CurReg);
12192 if (Def->getParent() != LoopBB)
12193 return false;
12194 if (Def->isCopy()) {
12195 // Ignore copy instructions unless they contain subregisters
12196 if (Def->getOperand(0).getSubReg() || Def->getOperand(1).getSubReg())
12197 return false;
12198 CurReg = Def->getOperand(1).getReg();
12199 } else if (Def->isPHI()) {
12200 if (InitReg != 0)
12201 return false;
12202 if (!UpdateInst)
12203 IsUpdatePriorComp = false;
12204 extractPhiReg(*Def, LoopBB, CurReg, InitReg);
12205 } else {
12206 if (UpdateInst)
12207 return false;
12208 switch (Def->getOpcode()) {
12209 case AArch64::ADDSXri:
12210 case AArch64::ADDSWri:
12211 case AArch64::SUBSXri:
12212 case AArch64::SUBSWri:
12213 case AArch64::ADDXri:
12214 case AArch64::ADDWri:
12215 case AArch64::SUBXri:
12216 case AArch64::SUBWri:
12217 UpdateInst = Def;
12218 UpdateCounterOprNum = 1;
12219 break;
12220 case AArch64::ADDSXrr:
12221 case AArch64::ADDSWrr:
12222 case AArch64::SUBSXrr:
12223 case AArch64::SUBSWrr:
12224 case AArch64::ADDXrr:
12225 case AArch64::ADDWrr:
12226 case AArch64::SUBXrr:
12227 case AArch64::SUBWrr:
12228 UpdateInst = Def;
12229 if (isDefinedOutside(Def->getOperand(2).getReg(), LoopBB))
12230 UpdateCounterOprNum = 1;
12231 else if (isDefinedOutside(Def->getOperand(1).getReg(), LoopBB))
12232 UpdateCounterOprNum = 2;
12233 else
12234 return false;
12235 break;
12236 default:
12237 return false;
12238 }
12239 CurReg = Def->getOperand(UpdateCounterOprNum).getReg();
12240 }
12241
12242 if (!CurReg.isVirtual())
12243 return false;
12244 if (Reg == CurReg)
12245 break;
12246 }
12247
12248 if (!UpdateInst)
12249 return false;
12250
12251 return true;
12252}
12253
12254std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
12256 // Accept loops that meet the following conditions
12257 // * The conditional branch is BCC
12258 // * The compare instruction is ADDS/SUBS/WHILEXX
12259 // * One operand of the compare is an induction variable and the other is a
12260 // loop invariant value
12261 // * The induction variable is incremented/decremented by a single instruction
12262 // * Does not contain CALL or instructions which have unmodeled side effects
12263
12264 for (MachineInstr &MI : *LoopBB)
12265 if (MI.isCall() || MI.hasUnmodeledSideEffects())
12266 // This instruction may use NZCV, which interferes with the instruction to
12267 // be inserted for loop control.
12268 return nullptr;
12269
12270 MachineBasicBlock *TBB = nullptr, *FBB = nullptr;
12272 if (analyzeBranch(*LoopBB, TBB, FBB, Cond))
12273 return nullptr;
12274
12275 // Infinite loops are not supported
12276 if (TBB == LoopBB && FBB == LoopBB)
12277 return nullptr;
12278
12279 // Must be conditional branch
12280 if (TBB != LoopBB && FBB == nullptr)
12281 return nullptr;
12282
12283 assert((TBB == LoopBB || FBB == LoopBB) &&
12284 "The Loop must be a single-basic-block loop");
12285
12286 MachineInstr *CondBranch = &*LoopBB->getFirstTerminator();
12288
12289 if (CondBranch->getOpcode() != AArch64::Bcc)
12290 return nullptr;
12291
12292 // Normalization for createTripCountGreaterCondition()
12293 if (TBB == LoopBB)
12295
12296 MachineInstr *Comp = nullptr;
12297 unsigned CompCounterOprNum = 0;
12298 for (MachineInstr &MI : reverse(*LoopBB)) {
12299 if (MI.modifiesRegister(AArch64::NZCV, &TRI)) {
12300 // Guarantee that the compare is SUBS/ADDS/WHILEXX and that one of the
12301 // operands is a loop invariant value
12302
12303 switch (MI.getOpcode()) {
12304 case AArch64::SUBSXri:
12305 case AArch64::SUBSWri:
12306 case AArch64::ADDSXri:
12307 case AArch64::ADDSWri:
12308 Comp = &MI;
12309 CompCounterOprNum = 1;
12310 break;
12311 case AArch64::ADDSWrr:
12312 case AArch64::ADDSXrr:
12313 case AArch64::SUBSWrr:
12314 case AArch64::SUBSXrr:
12315 Comp = &MI;
12316 break;
12317 default:
12318 if (isWhileOpcode(MI.getOpcode())) {
12319 Comp = &MI;
12320 break;
12321 }
12322 return nullptr;
12323 }
12324
12325 if (CompCounterOprNum == 0) {
12326 if (isDefinedOutside(Comp->getOperand(1).getReg(), LoopBB))
12327 CompCounterOprNum = 2;
12328 else if (isDefinedOutside(Comp->getOperand(2).getReg(), LoopBB))
12329 CompCounterOprNum = 1;
12330 else
12331 return nullptr;
12332 }
12333 break;
12334 }
12335 }
12336 if (!Comp)
12337 return nullptr;
12338
12339 MachineInstr *Update = nullptr;
12340 Register Init;
12341 bool IsUpdatePriorComp;
12342 unsigned UpdateCounterOprNum;
12343 if (!getIndVarInfo(Comp->getOperand(CompCounterOprNum).getReg(), LoopBB,
12344 Update, UpdateCounterOprNum, Init, IsUpdatePriorComp))
12345 return nullptr;
12346
12347 return std::make_unique<AArch64PipelinerLoopInfo>(
12348 LoopBB, CondBranch, Comp, CompCounterOprNum, Update, UpdateCounterOprNum,
12349 Init, IsUpdatePriorComp, Cond);
12350}
12351
12352/// verifyInstruction - Perform target specific instruction verification.
12353bool AArch64InstrInfo::verifyInstruction(const MachineInstr &MI,
12354 StringRef &ErrInfo) const {
12355 // Verify that immediate offsets on load/store instructions are within range.
12356 // Stack objects with an FI operand are excluded as they can be fixed up
12357 // during PEI.
12358 TypeSize Scale(0U, false), Width(0U, false);
12359 int64_t MinOffset, MaxOffset;
12360 if (getMemOpInfo(MI.getOpcode(), Scale, Width, MinOffset, MaxOffset)) {
12361 unsigned ImmIdx = getLoadStoreImmIdx(MI.getOpcode());
12362 if (MI.getOperand(ImmIdx).isImm() && !MI.getOperand(ImmIdx - 1).isFI()) {
12363 int64_t Imm = MI.getOperand(ImmIdx).getImm();
12364 if (Imm < MinOffset || Imm > MaxOffset) {
12365 ErrInfo = "Unexpected immediate on load/store instruction";
12366 return false;
12367 }
12368 }
12369 }
12370
12371 const MCInstrDesc &MCID = MI.getDesc();
12372 for (unsigned Op = 0; Op < MCID.getNumOperands(); Op++) {
12373 const MachineOperand &MO = MI.getOperand(Op);
12374 switch (MCID.operands()[Op].OperandType) {
12376 if (!MO.isImm() || MO.getImm() != 0) {
12377 ErrInfo = "OPERAND_IMPLICIT_IMM_0 should be 0";
12378 return false;
12379 }
12380 break;
12382 if (!MO.isImm() ||
12384 (AArch64_AM::getShiftValue(MO.getImm()) != 8 &&
12385 AArch64_AM::getShiftValue(MO.getImm()) != 16)) {
12386 ErrInfo = "OPERAND_SHIFT_MSL should be msl shift of 8 or 16";
12387 return false;
12388 }
12389 break;
12391 if (!MO.isImm() || (MO.getImm() != 0 && MO.getImm() != 1)) {
12392 ErrInfo = "OPERAND_IMM_UINT1 should be 0 or 1";
12393 return false;
12394 }
12395 break;
12397 if (!MO.isImm() || MO.getImm() <= 0 || MO.getImm() > 16) {
12398 ErrInfo = "OPERAND_IMM_UINT4plus1 should be in the range 1 to 16";
12399 return false;
12400 }
12401 break;
12403 if (!MO.isImm() || !isUInt<5>(MO.getImm())) {
12404 ErrInfo = "OPERAND_IMM_UINT5 should be in the range 0 to 31";
12405 return false;
12406 }
12407 break;
12409 if (!MO.isImm() || !isUInt<8>(MO.getImm())) {
12410 ErrInfo = "OPERAND_IMM_UINT8 should be in the range 0 to 255";
12411 return false;
12412 }
12413 break;
12414 default:
12415 break;
12416 }
12417 }
12418 return true;
12419}
12420
12421#define GET_INSTRINFO_HELPERS
12422#define GET_INSTRMAP_INFO
12423#include "AArch64GenInstrInfo.inc"
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
static cl::opt< unsigned > BCCDisplacementBits("aarch64-bcc-offset-bits", cl::Hidden, cl::init(19), cl::desc("Restrict range of Bcc instructions (DEBUG)"))
static Register genNeg(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, DenseMap< Register, unsigned > &InstrIdxForVirtReg, unsigned MnegOpc, const TargetRegisterClass *RC)
genNeg - Helper to generate an intermediate negation of the second operand of Root
static bool isFrameStoreOpcode(int Opcode)
static cl::opt< unsigned > GatherOptSearchLimit("aarch64-search-limit", cl::Hidden, cl::init(2048), cl::desc("Restrict range of instructions to search for the " "machine-combiner gather pattern optimization"))
static bool getMaddPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
Find instructions that can be turned into madd.
static AArch64CC::CondCode findCondCodeUsedByInstr(const MachineInstr &Instr)
Find a condition code used by the instruction.
static MachineInstr * genFusedMultiplyAcc(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC)
genFusedMultiplyAcc - Helper to generate fused multiply accumulate instructions.
static MachineInstr * genFusedMultiplyAccNeg(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, DenseMap< Register, unsigned > &InstrIdxForVirtReg, unsigned IdxMulOpd, unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC)
genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate instructions with an additional...
static bool isCombineInstrCandidate64(unsigned Opc)
static bool isFrameLoadOpcode(int Opcode)
static unsigned removeCopies(const MachineRegisterInfo &MRI, unsigned VReg)
static bool areCFlagsAccessedBetweenInstrs(MachineBasicBlock::iterator From, MachineBasicBlock::iterator To, const TargetRegisterInfo *TRI, const AccessKind AccessToCheck=AK_All)
True when condition flags are accessed (either by writing or reading) on the instruction trace starti...
static bool getFMAPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
Floating-Point Support.
static bool isADDSRegImm(unsigned Opcode)
static bool isCheapCopy(const MachineInstr &MI, const AArch64RegisterInfo &RI)
static bool isANDOpcode(MachineInstr &MI)
static bool predictCompactUnwindFrameRecordForOutlinedFunction(std::vector< outliner::Candidate > &RepeatedSequenceLocs, const TargetRegisterInfo &TRI)
Predict what the above will answer, for use while costing candidates.
static void appendOffsetComment(int NumBytes, llvm::raw_string_ostream &Comment, StringRef RegScale={})
static unsigned sForm(MachineInstr &Instr)
Get opcode of S version of Instr.
static bool isCombineInstrSettingFlag(unsigned Opc)
static bool getFNEGPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
static bool getIndVarInfo(Register Reg, const MachineBasicBlock *LoopBB, MachineInstr *&UpdateInst, unsigned &UpdateCounterOprNum, Register &InitReg, bool &IsUpdatePriorComp)
If Reg is an induction variable, return true and set some parameters.
static bool canPairLdStOpc(unsigned FirstOpc, unsigned SecondOpc)
static bool mustAvoidNeonAtMBBI(const AArch64Subtarget &Subtarget, MachineBasicBlock &MBB, MachineBasicBlock::iterator I)
Returns true if in a streaming call site region without SME-FA64.
static bool isPostIndexLdStOpcode(unsigned Opcode)
Return true if the opcode is a post-index ld/st instruction, which really loads from base+0.
static std::optional< unsigned > getLFIInstSizeInBytes(const MachineInstr &MI)
Return the maximum number of bytes of code the specified instruction may be after LFI rewriting.
static unsigned getBranchDisplacementBits(unsigned Opc)
static cl::opt< unsigned > CBDisplacementBits("aarch64-cb-offset-bits", cl::Hidden, cl::init(9), cl::desc("Restrict range of CB instructions (DEBUG)"))
static std::optional< ParamLoadedValue > describeORRLoadedValue(const MachineInstr &MI, Register DescribedReg, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI)
If the given ORR instruction is a copy, and DescribedReg overlaps with the destination register then,...
static bool getFMULPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
static void appendReadRegExpr(SmallVectorImpl< char > &Expr, unsigned RegNum)
static MachineInstr * genMaddR(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, unsigned IdxMulOpd, unsigned MaddOpc, unsigned VR, const TargetRegisterClass *RC)
genMaddR - Generate madd instruction and combine mul and add using an extra virtual register Example ...
static Register cloneInstr(const MachineInstr *MI, unsigned ReplaceOprNum, Register ReplaceReg, MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertTo)
Clone an instruction from MI.
static bool scaleOffset(unsigned Opc, int64_t &Offset)
static bool canCombineWithFMUL(MachineBasicBlock &MBB, MachineOperand &MO, unsigned MulOpc)
unsigned scaledOffsetOpcode(unsigned Opcode, unsigned &Scale)
static MachineInstr * genFusedMultiplyIdx(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC)
genFusedMultiplyIdx - Helper to generate fused multiply accumulate instructions.
static MachineInstr * genIndexedMultiply(MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, unsigned IdxDupOp, unsigned MulOpc, const TargetRegisterClass *RC, MachineRegisterInfo &MRI)
Fold (FMUL x (DUP y lane)) into (FMUL_indexed x y lane)
static cl::opt< bool > UseCompactUnwindFrameRecordForOutlinedFunctions("aarch64-outliner-compact-unwind-frame", cl::Hidden, cl::init(true), cl::desc("Use a frame record for Mach-O non-leaf outlined functions"))
static bool shouldUseCompactUnwindFrameRecordForOutlinedFunction(const MachineBasicBlock &MBB)
Return true if the outlined function in MBB should save FP and LR as a frame record instead of saving...
static bool isSUBSRegImm(unsigned Opcode)
static bool UpdateOperandRegClass(MachineInstr &Instr)
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
static bool isInStreamingCallSiteRegion(MachineBasicBlock &MBB, MachineBasicBlock::iterator I)
Returns true if the instruction at I is in a streaming call site region, within a single basic block.
static bool canCmpInstrBeRemoved(MachineInstr &MI, MachineInstr &CmpInstr, int CmpValue, const TargetRegisterInfo &TRI, SmallVectorImpl< MachineInstr * > &CCUseInstrs, bool &IsInvertCC)
unsigned unscaledOffsetOpcode(unsigned Opcode)
static bool getLoadPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
Search for patterns of LD instructions we can optimize.
static bool canInstrSubstituteCmpInstr(MachineInstr &MI, MachineInstr &CmpInstr, const TargetRegisterInfo &TRI)
Check if CmpInstr can be substituted by MI.
static UsedNZCV getUsedNZCV(AArch64CC::CondCode CC)
static bool isCombineInstrCandidateFP(const MachineInstr &Inst)
static bool isCompactUnwindFrameRecordEnabled(const MachineFunction &MF)
Return true if the frame-record form of the outlined prologue is enabled for the target of MF.
static void appendLoadRegExpr(SmallVectorImpl< char > &Expr, int64_t OffsetFromDefCFA)
static void appendConstantExpr(SmallVectorImpl< char > &Expr, int64_t Constant, dwarf::LocationAtom Operation)
static unsigned convertToNonFlagSettingOpc(const MachineInstr &MI)
Return the opcode that does not set flags when possible - otherwise return the original opcode.
static bool outliningCandidatesV8_3OpsConsensus(const outliner::Candidate &a, const outliner::Candidate &b)
static bool isCombineInstrCandidate32(unsigned Opc)
static void parseCondBranch(MachineInstr *LastInst, MachineBasicBlock *&Target, SmallVectorImpl< MachineOperand > &Cond)
static unsigned offsetExtendOpcode(unsigned Opcode)
MachineOutlinerMBBFlags
@ LRUnavailableSomewhere
@ UnsafeRegsDead
static void loadRegPairFromStackSlot(const TargetRegisterInfo &TRI, MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore, const MCInstrDesc &MCID, Register DestReg, unsigned SubIdx0, unsigned SubIdx1, int FI, MachineMemOperand *MMO)
static void generateGatherLanePattern(MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, SmallVectorImpl< MachineInstr * > &DelInstrs, DenseMap< Register, unsigned > &InstrIdxForVirtReg, unsigned Pattern, unsigned NumLanes)
Generate optimized instruction sequence for gather load patterns to improve Memory-Level Parallelism ...
static bool getMiscPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns)
Find other MI combine patterns.
static bool outliningCandidatesSigningKeyConsensus(const outliner::Candidate &a, const outliner::Candidate &b)
static const MachineInstrBuilder & AddSubReg(const MachineInstrBuilder &MIB, MCRegister Reg, unsigned SubIdx, RegState State, const TargetRegisterInfo *TRI)
static bool outliningCandidatesSigningScopeConsensus(const outliner::Candidate &a, const outliner::Candidate &b)
static bool shouldClusterFI(const MachineFrameInfo &MFI, int FI1, int64_t Offset1, unsigned Opcode1, int FI2, int64_t Offset2, unsigned Opcode2)
static bool isValidCBExtend(int64_t Opc, AArch64_AM::ShiftExtendType Ext)
static cl::opt< unsigned > TBZDisplacementBits("aarch64-tbz-offset-bits", cl::Hidden, cl::init(14), cl::desc("Restrict range of TB[N]Z instructions (DEBUG)"))
static void extractPhiReg(const MachineInstr &Phi, const MachineBasicBlock *MBB, Register &RegMBB, Register &RegOther)
static MCCFIInstruction createDefCFAExpression(const TargetRegisterInfo &TRI, unsigned Reg, const StackOffset &Offset)
static bool isDefinedOutside(Register Reg, const MachineBasicBlock *BB)
static MachineInstr * genFusedMultiply(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, unsigned IdxMulOpd, unsigned MaddOpc, const TargetRegisterClass *RC, FMAInstKind kind=FMAInstKind::Default, const Register *ReplacedAddend=nullptr)
genFusedMultiply - Generate fused multiply instructions.
static bool getGatherLanePattern(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns, unsigned LoadLaneOpCode, unsigned NumLanes)
Check if the given instruction forms a gather load pattern that can be optimized for better Memory-Le...
static MachineInstr * genFusedMultiplyIdxNeg(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, DenseMap< Register, unsigned > &InstrIdxForVirtReg, unsigned IdxMulOpd, unsigned MaddOpc, unsigned MnegOpc, const TargetRegisterClass *RC)
genFusedMultiplyAccNeg - Helper to generate fused multiply accumulate instructions with an additional...
static bool isCombineInstrCandidate(unsigned Opc)
static unsigned regOffsetOpcode(unsigned Opcode)
MachineOutlinerClass
Constants defining how certain sequences should be outlined.
@ MachineOutlinerTailCall
Emit a save, restore, call, and return.
@ MachineOutlinerRegSave
Emit a call and tail-call.
@ MachineOutlinerNoLRSave
Only emit a branch.
@ MachineOutlinerThunk
Emit a call and return.
@ MachineOutlinerDefault
static cl::opt< unsigned > BDisplacementBits("aarch64-b-offset-bits", cl::Hidden, cl::init(26), cl::desc("Restrict range of B instructions (DEBUG)"))
static bool areCFlagsAliveInSuccessors(const MachineBasicBlock *MBB)
Check if AArch64::NZCV should be alive in successors of MBB.
static void emitFrameOffsetAdj(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, unsigned DestReg, unsigned SrcReg, int64_t Offset, unsigned Opc, const TargetInstrInfo *TII, MachineInstr::MIFlag Flag, bool NeedsWinCFI, bool *HasWinCFI, bool EmitCFAOffset, StackOffset CFAOffset, unsigned FrameReg)
static bool isCheapImmediate(const MachineInstr &MI, unsigned BitSize)
static cl::opt< unsigned > CBZDisplacementBits("aarch64-cbz-offset-bits", cl::Hidden, cl::init(19), cl::desc("Restrict range of CB[N]Z instructions (DEBUG)"))
static void genSubAdd2SubSub(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs, SmallVectorImpl< MachineInstr * > &DelInstrs, unsigned IdxOpd1, DenseMap< Register, unsigned > &InstrIdxForVirtReg)
Do the following transformation A - (B + C) ==> (A - B) - C A - (B + C) ==> (A - C) - B.
static unsigned canFoldIntoCSel(const MachineRegisterInfo &MRI, unsigned VReg, unsigned *NewReg=nullptr)
static void signOutlinedFunction(MachineFunction &MF, MachineBasicBlock &MBB, const AArch64InstrInfo *TII, bool ShouldSignReturnAddr)
static MachineInstr * genFNegatedMAD(MachineFunction &MF, MachineRegisterInfo &MRI, const TargetInstrInfo *TII, MachineInstr &Root, SmallVectorImpl< MachineInstr * > &InsInstrs)
static bool canCombineWithMUL(MachineBasicBlock &MBB, MachineOperand &MO, unsigned MulOpc, unsigned ZeroReg)
static void storeRegPairToStackSlot(const TargetRegisterInfo &TRI, MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore, const MCInstrDesc &MCID, Register SrcReg, bool IsKill, unsigned SubIdx0, unsigned SubIdx1, int FI, MachineMemOperand *MMO)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static const Function * getParent(const Value *V)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
DXIL Forward Handle Accesses
@ Default
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file implements the LivePhysRegs utility for tracking liveness of physical registers.
A set of register units.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
PowerPC Reduce CR logical Operation
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file declares the machine register scavenger class.
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the SmallSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static bool canCombine(MachineBasicBlock &MBB, MachineOperand &MO, unsigned CombineOpc=0)
AArch64FunctionInfo - This class is derived from MachineFunctionInfo and contains private AArch64-spe...
SignReturnAddress getSignReturnAddressCondition() const
void setOutliningStyle(const std::string &Style)
bool needsDwarfUnwindInfo(const MachineFunction &MF) const
std::optional< bool > hasRedZone() const
static bool shouldSignReturnAddress(SignReturnAddress Condition, bool IsLRSpilled)
static bool isHForm(const MachineInstr &MI)
Returns whether the instruction is in H form (16 bit operands)
void insertSelect(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, Register DstReg, ArrayRef< MachineOperand > Cond, Register TrueReg, Register FalseReg) const override
static bool hasBTISemantics(const MachineInstr &MI)
Returns whether the instruction can be compatible with non-zero BTYPE.
static bool isQForm(const MachineInstr &MI)
Returns whether the instruction is in Q form (128 bit operands)
static bool getMemOpInfo(unsigned Opcode, TypeSize &Scale, TypeSize &Width, int64_t &MinOffset, int64_t &MaxOffset)
Returns true if opcode Opc is a memory operation.
static bool isTailCallReturnInst(const MachineInstr &MI)
Returns true if MI is one of the TCRETURN* instructions.
static bool isFPRCopy(const MachineInstr &MI)
Does this instruction rename an FPR without modifying bits?
MachineInstr * emitLdStWithAddr(MachineInstr &MemI, const ExtAddrMode &AM) const override
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is an instruction that moves/copies value from one register to an...
MachineBasicBlock * getBranchDestBlock(const MachineInstr &MI) const override
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
GetInstSize - Return the number of bytes of code the specified instruction may be.
static bool isZExtLoad(const MachineInstr &MI)
Returns whether the instruction is a zero-extending load.
bool areMemAccessesTriviallyDisjoint(const MachineInstr &MIa, const MachineInstr &MIb) const override
void copyPhysRegImpl(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
static bool isGPRCopy(const MachineInstr &MI)
Does this instruction rename a GPR without modifying bits?
static unsigned convertToFlagSettingOpc(unsigned Opc)
Return the opcode that set flags when possible.
void createPauthEpilogueInstr(MachineBasicBlock &MBB, DebugLoc DL) const
Return true when there is potentially a faster code sequence for an instruction chain ending in Root.
bool isBranchOffsetInRange(unsigned BranchOpc, int64_t BrOffset) const override
bool canInsertSelect(const MachineBasicBlock &, ArrayRef< MachineOperand > Cond, Register, Register, Register, int &, int &, int &) const override
Register isLoadFromStackSlotPostFE(const MachineInstr &MI, int &FrameIndex) const override
Check for post-frame ptr elimination stack locations as well.
static const MachineOperand & getLdStOffsetOp(const MachineInstr &MI)
Returns the immediate offset operator of a load/store.
bool isCoalescableExtInstr(const MachineInstr &MI, Register &SrcReg, Register &DstReg, unsigned &SubIdx) const override
static std::optional< unsigned > getUnscaledLdSt(unsigned Opc)
Returns the unscaled load/store for the scaled load/store opcode, if there is a corresponding unscale...
static bool hasUnscaledLdStOffset(unsigned Opc)
Return true if it has an unscaled load/store offset.
static const MachineOperand & getLdStAmountOp(const MachineInstr &MI)
Returns the shift amount operator of a load/store.
static bool isPreLdSt(const MachineInstr &MI)
Returns whether the instruction is a pre-indexed load/store.
bool analyzeBranchPredicate(MachineBasicBlock &MBB, MachineBranchPredicate &MBP, bool AllowModify) const override
void insertIndirectBranch(MachineBasicBlock &MBB, MachineBasicBlock &NewDestBB, MachineBasicBlock &RestoreBB, const DebugLoc &DL, int64_t BrOffset, RegScavenger *RS) const override
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isPairableLdStInst(const MachineInstr &MI)
Return true if pairing the given load or store may be paired with another.
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
static bool isSExtLoad(const MachineInstr &MI)
Returns whether the instruction is a sign-extending load.
const AArch64RegisterInfo & getRegisterInfo() const
getRegisterInfo - TargetInstrInfo is a superset of MRegister info.
static bool isPreSt(const MachineInstr &MI)
Returns whether the instruction is a pre-indexed store.
void insertNoop(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
AArch64InstrInfo(const AArch64Subtarget &STI)
static bool isPairedLdSt(const MachineInstr &MI)
Returns whether the instruction is a paired load/store.
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
bool getMemOperandsWithOffsetWidth(const MachineInstr &MI, SmallVectorImpl< const MachineOperand * > &BaseOps, int64_t &Offset, bool &OffsetIsScalable, LocationSize &Width) const override
static bool isStridedAccess(const MachineInstr &MI)
Return true if the given load or store is a strided memory access.
bool shouldClusterMemOps(ArrayRef< const MachineOperand * > BaseOps1, int64_t Offset1, bool OffsetIsScalable1, ArrayRef< const MachineOperand * > BaseOps2, int64_t Offset2, bool OffsetIsScalable2, unsigned ClusterSize, unsigned NumBytes) const override
Detect opportunities for ldp/stp formation.
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
bool isThroughputPattern(unsigned Pattern) const override
Return true when a code sequence can improve throughput.
MachineOperand & getMemOpBaseRegImmOfsOffsetOperand(MachineInstr &LdSt) const
Return the immediate offset of the base register in a load/store LdSt.
std::optional< ExtAddrMode > getAddrModeFromMemoryOp(const MachineInstr &MemI) const override
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
bool canFoldIntoAddrMode(const MachineInstr &MemI, Register Reg, const MachineInstr &AddrI, ExtAddrMode &AM) const override
static bool isLdStPairSuppressed(const MachineInstr &MI)
Return true if pairing the given load or store is hinted to be unprofitable.
Register isStoreToStackSlotPostFE(const MachineInstr &MI, int &FrameIndex) const override
Check for post-frame ptr elimination stack locations as well.
std::unique_ptr< TargetInstrInfo::PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
void copyPhysRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, llvm::ArrayRef< unsigned > Indices) const
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
AArch64CC::CondCode insertCmpForCondBr(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, ArrayRef< MachineOperand > Cond) const
Inserts the compare instruction needed to un-fuse a fused conditional branch instruction and returns ...
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
optimizeCompareInstr - Convert the instruction supplying the argument to the comparison into one that...
static unsigned getLoadStoreImmIdx(unsigned Opc)
Returns the index for the immediate for a given instruction.
static bool isGPRZero(const MachineInstr &MI)
Does this instruction set its full destination register to zero?
void copyGPRRegTuple(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, MCRegister DestReg, MCRegister SrcReg, bool KillSrc, unsigned Opcode, unsigned ZeroReg, llvm::ArrayRef< unsigned > Indices) const
bool getMemOperandWithOffsetWidth(const MachineInstr &MI, const MachineOperand *&BaseOp, int64_t &Offset, bool &OffsetIsScalable, TypeSize &Width) const
If OffsetIsScalable is set to 'true', the offset is scaled by vscale.
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
analyzeCompare - For a comparison instruction, return the source registers in SrcReg and SrcReg2,...
CombinerObjective getCombinerObjective(unsigned Pattern) const override
static bool isFpOrNEON(Register Reg)
Returns whether the physical register is FP or NEON.
bool isAsCheapAsAMove(const MachineInstr &MI) const override
std::optional< DestSourcePair > isCopyLikeInstrImpl(const MachineInstr &MI) const override
static void suppressLdStPair(MachineInstr &MI)
Hint that pairing the given load or store is unprofitable.
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
static bool isPreLd(const MachineInstr &MI)
Returns whether the instruction is a pre-indexed load.
bool optimizeCondBranch(MachineInstr &MI) const override
Replace csincr-branch sequence by simple conditional branch.
static int getMemScale(unsigned Opc)
Scaling factor for (scaled or unscaled) load or store.
bool isCandidateToMergeOrPair(const MachineInstr &MI) const
Return true if this is a load/store that can be potentially paired/merged.
MCInst getNop() const override
static const MachineOperand & getLdStBaseOp(const MachineInstr &MI)
Returns the base register operator of a load/store.
bool isReservedReg(const MachineFunction &MF, MCRegister Reg) const
const AArch64RegisterInfo * getRegisterInfo() const override
bool isNeonAvailable() const
Returns true if the target has NEON and the function at runtime is known to have NEON enabled (e....
bool isSVEorStreamingSVEAvailable() const
Returns true if the target has access to either the full range of SVE instructions,...
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
This is an important base class in LLVM.
Definition Constant.h:43
A debug info location.
Definition DebugLoc.h:126
bool empty() const
Definition DenseMap.h:717
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:828
bool hasOptSize() const
Optimize this function for size (-Os) or minimum size (-Oz).
Definition Function.h:699
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:696
Module * getParent()
Get the module that this global value is contained inside of...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
LLVM_ABI void accumulate(const MachineInstr &MI)
Adds all register units used, defined or clobbered in MI.
static LocationSize precise(uint64_t Value)
This class is intended to be used as a base class for asm properties and features specific to the tar...
Definition MCAsmInfo.h:67
bool usesWindowsCFI() const
Definition MCAsmInfo.h:675
static MCCFIInstruction cfiDefCfa(MCSymbol *L, unsigned Register, int64_t Offset, SMLoc Loc={})
.cfi_def_cfa defines a rule for computing CFA as: take address from Register and add Offset to it.
Definition MCDwarf.h:628
static MCCFIInstruction createOffset(MCSymbol *L, unsigned Register, int64_t Offset, SMLoc Loc={})
.cfi_offset Previous value of Register is saved at offset Offset from CFA.
Definition MCDwarf.h:670
static MCCFIInstruction cfiDefCfaOffset(MCSymbol *L, int64_t Offset, SMLoc Loc={})
.cfi_def_cfa_offset modifies a rule for computing CFA.
Definition MCDwarf.h:643
static MCCFIInstruction createEscape(MCSymbol *L, StringRef Vals, SMLoc Loc={}, StringRef Comment="")
.cfi_escape Allows the user to add arbitrary bytes to the unwind info.
Definition MCDwarf.h:756
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr bool isValid() const
Definition MCRegister.h:84
static constexpr unsigned NoRegister
Definition MCRegister.h:60
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
Set of metadata that should be preserved when using BuildMI().
bool isInlineAsmBrIndirectTarget() const
Returns true if this is the indirect dest of an INLINEASM_BR.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
reverse_instr_iterator instr_rbegin()
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
reverse_instr_iterator instr_rend()
Instructions::iterator instr_iterator
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
iterator insertAfter(iterator I, MachineInstr *MI)
Insert MI into the instruction list after I.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
void setMachineBlockAddressTaken()
Set this block to indicate that its address is used as something other than the target of a terminato...
LLVM_ABI bool isLiveIn(MCRegister Reg, LaneBitmask LaneMask=LaneBitmask::getAll()) const
Return true if the specified register is in the live in set.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
uint64_t getStackSize() const
Return the number of bytes that must be allocated to hold all of the fixed size frame objects.
void setStackID(int ObjectIdx, uint8_t ID)
bool isCalleeSavedInfoValid() const
Has the callee saved info been calculated yet?
Align getObjectAlign(int ObjectIdx) const
Return the alignment of the specified stack object.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
unsigned getNumObjects() const
Return the number of objects.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
unsigned addFrameInst(const MCCFIInstruction &Inst)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineJumpTableInfo * getJumpTableInfo() const
getJumpTableInfo - Return the jump table info object for the current function.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & addCFIIndex(unsigned CFIIndex) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & setMIFlag(MachineInstr::MIFlag Flag) const
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
reverse_iterator getReverse() const
Get a reverse iterator to the same node.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
bool isCopy() const
const MachineBasicBlock * getParent() const
bool isCall(QueryType Type=AnyInBundle) const
bool getFlag(MIFlag Flag) const
Return whether an MI flag is set.
LLVM_ABI uint32_t mergeFlagsWith(const MachineInstr &Other) const
Return the MIFlags which represent both MachineInstrs.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
LLVM_ABI bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by mayLoad / mayStore,...
bool registerDefIsDead(Register Reg, const TargetRegisterInfo *TRI) const
Returns true if the register is dead in this machine instruction.
bool definesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr fully defines the specified register.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
LLVM_ABI bool hasOrderedMemoryRef() const
Return true if this instruction may have an ordered or volatile memory reference, or if the informati...
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
ArrayRef< MachineMemOperand * > memoperands() const
Access to memory operands of the instruction.
LLVM_ABI bool isLoadFoldBarrier() const
Returns true if it is illegal to fold a load across this instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
LLVM_ABI void addRegisterDefined(Register Reg, const TargetRegisterInfo *RegInfo=nullptr)
We have determined MI defines a register.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI int findRegisterDefOperandIdx(Register Reg, const TargetRegisterInfo *TRI, bool isDead=false, bool Overlap=false) const
Returns the operand index that is a def of the specified register or -1 if it is not found.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
const std::vector< MachineJumpTableEntry > & getJumpTables() const
A description of a memory reference used in the backend.
@ MOVolatile
The memory access is volatile.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
This class contains meta information specific to a module.
LLVM_ABI MachineFunction * getMachineFunction(const Function &F) const
Returns the MachineFunction associated to IR function F if there is one, otherwise nullptr.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
MachineBasicBlock * getMBB() const
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
unsigned getTargetFlags() const
static MachineOperand CreateImm(int64_t Val)
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
bool tracksLiveness() const
tracksLiveness - Returns true when tracking register liveness accurately.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
MachineBasicBlock * getDefBlock(Register Reg) const
Return the machine basic block in which the specified virtual register is defined,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
bool reservedRegsFrozen() const
reservedRegsFrozen - Returns true after freezeReservedRegs() was called to ensure the set of reserved...
use_instr_nodbg_iterator use_instr_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClassOrNull(Register Reg) const
Return the register class of Reg, or null if Reg has not been assigned a register class yet.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
const Triple & getTargetTriple() const
Get the target triple which is a string describing the target host.
Definition Module.h:328
MI-level patchpoint operands.
Definition StackMaps.h:77
uint32_t getNumPatchBytes() const
Return the number of patchable bytes the given patchpoint should emit.
Definition StackMaps.h:105
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isValid() const
Definition Register.h:112
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
static constexpr bool isVirtualRegister(unsigned Reg)
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:66
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
Represents a location in source code.
Definition SMLoc.h:22
bool erase(PtrType Ptr)
Remove pointer from the set.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
bool empty() const
Definition SmallSet.h:169
bool erase(const T &V)
Definition SmallSet.h:200
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
void append(StringRef RHS)
Append from a StringRef.
Definition SmallString.h:68
StringRef str() const
Explicit conversion to StringRef.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
MI-level stackmap operands.
Definition StackMaps.h:36
uint32_t getNumPatchBytes() const
Return the number of patchable bytes the given stackmap should emit.
Definition StackMaps.h:51
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
int64_t getFixed() const
Returns the fixed component of the stack.
Definition TypeSize.h:46
int64_t getScalable() const
Returns the scalable component of the stack.
Definition TypeSize.h:49
static StackOffset get(int64_t Fixed, int64_t Scalable)
Definition TypeSize.h:41
static StackOffset getScalable(int64_t Scalable)
Definition TypeSize.h:40
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
MI-level Statepoint operands.
Definition StackMaps.h:159
uint32_t getNumPatchBytes() const
Return the number of patchable bytes the given statepoint should emit.
Definition StackMaps.h:208
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Object returned by analyzeLoopForPipelining.
TargetInstrInfo - Interface to description of machine instruction set.
virtual void genAlternativeCodeSequence(MachineInstr &Root, unsigned Pattern, SmallVectorImpl< MachineInstr * > &InsInstrs, SmallVectorImpl< MachineInstr * > &DelInstrs, DenseMap< Register, unsigned > &InstIdxForVirtReg) const
When getMachineCombinerPatterns() finds patterns, this function generates the instructions that could...
virtual std::optional< ParamLoadedValue > describeLoadedValue(const MachineInstr &MI, Register Reg) const
Produce the expression describing the MI loading a value into the physical register Reg.
virtual bool getMachineCombinerPatterns(MachineInstr &Root, SmallVectorImpl< unsigned > &Patterns, bool DoRegPressureReduce) const
Return true when there is potentially a faster code sequence for an instruction chain ending in Root.
virtual bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const
Test if the given instruction should be considered a scheduling boundary.
virtual CombinerObjective getCombinerObjective(unsigned Pattern) const
Return the objective of a combiner pattern.
virtual bool isFunctionSafeToSplit(const MachineFunction &MF) const
Return true if the function is a viable candidate for machine function splitting.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
CodeModel::Model getCodeModel() const
Returns the code model.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
TargetSubtargetInfo - Generic base class for all target subtargets.
virtual const TargetInstrInfo * getInstrInfo() const
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
Target - Wrapper for Target specific information.
bool isOSBinFormatMachO() const
Tests whether the environment is MachO.
Definition Triple.h:875
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
Value * getOperand(unsigned i) const
Definition User.h:207
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
self_iterator getIterator()
Definition ilist_node.h:123
A raw_ostream that writes to an std::string.
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static CondCode getInvertedCondCode(CondCode Code)
@ MO_DLLIMPORT
MO_DLLIMPORT - On a symbol operand, this represents that the reference to the symbol is for an import...
@ MO_NC
MO_NC - Indicates whether the linker is expected to check the symbol reference for overflow.
@ MO_G1
MO_G1 - A symbol operand with this flag (granule 1) represents the bits 16-31 of a 64-bit address,...
@ MO_S
MO_S - Indicates that the bits of the symbol operand represented by MO_G0 etc are signed.
@ MO_PAGEOFF
MO_PAGEOFF - A symbol operand with this flag represents the offset of that symbol within a 4K page.
@ MO_GOT
MO_GOT - This flag indicates that a symbol operand represents the address of the GOT entry for the sy...
@ MO_PREL
MO_PREL - Indicates that the bits of the symbol operand represented by MO_G0 etc are PC relative.
@ MO_G0
MO_G0 - A symbol operand with this flag (granule 0) represents the bits 0-15 of a 64-bit address,...
@ MO_ARM64EC_CALLMANGLE
MO_ARM64EC_CALLMANGLE - Operand refers to the Arm64EC-mangled version of a symbol,...
@ MO_PAGE
MO_PAGE - A symbol operand with this flag represents the pc-relative offset of the 4K page containing...
@ MO_HI12
MO_HI12 - This flag indicates that a symbol operand represents the bits 13-24 of a 64-bit address,...
@ MO_TLS
MO_TLS - Indicates that the operand being accessed is some kind of thread-local symbol.
@ MO_G2
MO_G2 - A symbol operand with this flag (granule 2) represents the bits 32-47 of a 64-bit address,...
@ MO_TAGGED
MO_TAGGED - With MO_PAGE, indicates that the page includes a memory tag in bits 56-63.
@ MO_G3
MO_G3 - A symbol operand with this flag (granule 3) represents the high 16-bits of a 64-bit address,...
@ MO_COFFSTUB
MO_COFFSTUB - On a symbol operand "FOO", this indicates that the reference is actually to the "....
unsigned getCheckerSizeInBytes(AuthCheckMethod Method)
Returns the number of bytes added by checkAuthenticatedRegister.
static uint64_t decodeLogicalImmediate(uint64_t val, unsigned regSize)
decodeLogicalImmediate - Decode a logical immediate value in the form "N:immr:imms" (where the immr a...
static unsigned getShiftValue(unsigned Imm)
getShiftValue - Extract the shift value.
static unsigned getArithExtendImm(AArch64_AM::ShiftExtendType ET, unsigned Imm)
getArithExtendImm - Encode the extend type and shift amount for an arithmetic instruction: imm: 3-bit...
constexpr bool isLegalArithImmed(const uint64_t C)
isLegalArithImmed -
static unsigned getArithShiftValue(unsigned Imm)
getArithShiftValue - get the arithmetic shift value.
static uint64_t encodeLogicalImmediate(uint64_t imm, unsigned regSize)
encodeLogicalImmediate - Return the encoded immediate value for a logical immediate instruction of th...
static AArch64_AM::ShiftExtendType getExtendType(unsigned Imm)
getExtendType - Extract the extend type for operands of arithmetic ops.
static AArch64_AM::ShiftExtendType getArithExtendType(unsigned Imm)
static AArch64_AM::ShiftExtendType getShiftType(unsigned Imm)
getShiftType - Extract the shift type.
static unsigned getShifterImm(AArch64_AM::ShiftExtendType ST, unsigned Imm)
getShifterImm - Encode the shift type and amount: imm: 6-bit shift amount shifter: 000 ==> lsl 001 ==...
void expandMOVAddr(unsigned Opcode, unsigned TargetFlags, bool IsTargetMachO, SmallVectorImpl< AddrInsnModel > &Insn)
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
static const uint64_t InstrFlagIsWhile
static const uint64_t InstrFlagIsPTestLike
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
initializer< Ty > init(const Ty &Val)
constexpr double e
InstrType
Represents how an instruction should be mapped by the outliner.
NodeAddr< InstrNode * > Instr
Definition RDFGraph.h:389
iterator end() const
Definition BasicBlock.h:89
LLVM_ABI Instruction & back() const
LLVM_ABI iterator begin() const
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
static bool isCondBranchOpcode(int Opc)
MCCFIInstruction createDefCFA(const TargetRegisterInfo &TRI, unsigned FrameReg, unsigned Reg, const StackOffset &Offset, bool LastAdjustmentWasScalable=true)
static bool isPTrueOpcode(unsigned Opc)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool succeeded(LogicalResult Result)
Utility function that returns true if the provided LogicalResult corresponds to a success value.
int isAArch64FrameOffsetLegal(const MachineInstr &MI, StackOffset &Offset, bool *OutUseUnscaledOp=nullptr, unsigned *OutUnscaledOp=nullptr, int64_t *EmittableOffset=nullptr)
Check if the Offset is a valid frame offset for MI.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
@ Renamable
Register that may be renamed.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
static bool isIndirectBranchOpcode(int Opc)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
unsigned getBLRCallOpcode(const MachineFunction &MF)
Return opcode to be used for indirect calls.
@ AArch64FrameOffsetIsLegal
Offset is legal.
@ AArch64FrameOffsetCanUpdate
Offset can apply, at least partly.
@ AArch64FrameOffsetCannotUpdate
Offset cannot apply.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
Op::Description Desc
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
static bool isSEHInstruction(const MachineInstr &MI)
bool isLFIPrePostMemAccess(unsigned Opcode)
Returns true if Opcode is a pre- or post-indexed memory access that the LFI rewriter expands with a b...
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
AArch64MachineCombinerPattern
@ MULSUBv8i16_OP2
@ FMULv4i16_indexed_OP1
@ FMLSv1i32_indexed_OP2
@ MULSUBv2i32_indexed_OP1
@ FMLAv2i32_indexed_OP2
@ MULADDv4i16_indexed_OP2
@ FMLAv1i64_indexed_OP1
@ MULSUBv16i8_OP1
@ FMLAv8i16_indexed_OP2
@ FMULv2i32_indexed_OP1
@ MULSUBv8i16_indexed_OP2
@ FMLAv1i64_indexed_OP2
@ MULSUBv4i16_indexed_OP2
@ FMLAv1i32_indexed_OP1
@ FMLAv2i64_indexed_OP2
@ FMLSv8i16_indexed_OP1
@ MULSUBv2i32_OP1
@ FMULv4i16_indexed_OP2
@ MULSUBv4i32_indexed_OP2
@ FMULv2i64_indexed_OP2
@ FMLAv4i32_indexed_OP1
@ MULADDv4i16_OP2
@ FMULv8i16_indexed_OP2
@ MULSUBv4i16_OP1
@ MULADDv4i32_OP2
@ MULADDv2i32_OP2
@ MULADDv16i8_OP2
@ FMLSv4i16_indexed_OP1
@ MULADDv16i8_OP1
@ FMLAv2i64_indexed_OP1
@ FMLAv1i32_indexed_OP2
@ FMLSv2i64_indexed_OP2
@ MULADDv2i32_OP1
@ MULADDv4i32_OP1
@ MULADDv2i32_indexed_OP1
@ MULSUBv16i8_OP2
@ MULADDv4i32_indexed_OP1
@ MULADDv2i32_indexed_OP2
@ FMLAv4i16_indexed_OP2
@ MULSUBv8i16_OP1
@ FMULv2i32_indexed_OP2
@ FMLSv2i32_indexed_OP2
@ FMLSv4i32_indexed_OP1
@ FMULv2i64_indexed_OP1
@ MULSUBv4i16_OP2
@ FMLSv4i16_indexed_OP2
@ FMLAv2i32_indexed_OP1
@ FMLSv2i32_indexed_OP1
@ FMLAv8i16_indexed_OP1
@ MULSUBv4i16_indexed_OP1
@ FMLSv4i32_indexed_OP2
@ MULADDv4i32_indexed_OP2
@ MULSUBv4i32_OP2
@ MULSUBv8i16_indexed_OP1
@ MULADDv8i16_OP2
@ MULSUBv2i32_indexed_OP2
@ FMULv4i32_indexed_OP2
@ FMLSv2i64_indexed_OP1
@ MULADDv4i16_OP1
@ FMLAv4i32_indexed_OP2
@ MULADDv8i16_indexed_OP1
@ FMULv4i32_indexed_OP1
@ FMLAv4i16_indexed_OP1
@ FMULv8i16_indexed_OP1
@ MULADDv8i16_OP1
@ MULSUBv4i32_indexed_OP1
@ MULSUBv4i32_OP1
@ FMLSv8i16_indexed_OP2
@ MULADDv8i16_indexed_OP2
@ MULSUBv2i32_OP2
@ FMLSv1i64_indexed_OP2
@ MULADDv4i16_indexed_OP1
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
void emitFrameOffset(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, unsigned DestReg, unsigned SrcReg, StackOffset Offset, const TargetInstrInfo *TII, MachineInstr::MIFlag=MachineInstr::NoFlags, bool SetNZCV=false, bool NeedsWinCFI=false, bool *HasWinCFI=nullptr, bool EmitCFAOffset=false, StackOffset InitialOffset={}, unsigned FrameReg=AArch64::SP)
emitFrameOffset - Emit instructions as needed to set DestReg to SrcReg plus Offset.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr RegState getDefRegState(bool B)
CombinerObjective
The combiner's goal may differ based on which pattern it is attempting to optimize.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
std::optional< UsedNZCV > examineCFlagsUse(MachineInstr &MI, MachineInstr &CmpInstr, const TargetRegisterInfo &TRI, SmallVectorImpl< MachineInstr * > *CCUseInstrs=nullptr)
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
auto instructionsWithoutDebug(IterT It, IterT End, bool SkipPseudoOp=true)
Construct a range iterator which begins at It and moves forwards until End is reached,...
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
static MCRegister getXRegFromWReg(MCRegister Reg)
MCCFIInstruction createCFAOffset(const TargetRegisterInfo &MRI, unsigned Reg, const StackOffset &OffsetFromDefCFA, std::optional< int64_t > IncomingVGOffsetFromDefCFA)
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
static bool isUncondBranchOpcode(int Opc)
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
bool rewriteAArch64FrameIndex(MachineInstr &MI, unsigned FrameRegIdx, unsigned FrameReg, StackOffset &Offset, const AArch64InstrInfo *TII)
rewriteAArch64FrameIndex - Rewrite MI to access 'Offset' bytes from the FP.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
static const MachineMemOperand::Flags MOSuppressPair
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
void appendLEB128(SmallVectorImpl< U > &Buffer, T Value)
Definition LEB128.h:280
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
bool optimizeTerminators(MachineBasicBlock *MBB, const TargetInstrInfo &TII)
std::pair< MachineOperand, DIExpression * > ParamLoadedValue
bool isNZCVTouchedInInstructionRange(const MachineInstr &DefMI, const MachineInstr &UseMI, const TargetRegisterInfo *TRI)
Return true if there is an instruction /after/ DefMI and before UseMI which either reads or clobbers ...
static const MachineMemOperand::Flags MOStridedAccess
constexpr RegState getUndefRegState(bool B)
void fullyRecomputeLiveIns(ArrayRef< MachineBasicBlock * > MBBs)
Convenience function for recomputing live-in's for a set of MBBs until the computation converges.
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Used to describe addressing mode similar to ExtAddrMode in CodeGenPrepare.
LLVM_ABI static const MBBSectionID ColdSectionID
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getUnknownStack(MachineFunction &MF)
Stack memory without other information.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
An individual sequence of instructions to be replaced with a call to an outlined function.
MachineFunction * getMF() const
The information necessary to create an outlined function for some class of candidate.