LLVM 24.0.0git
X86FixupLEAs.cpp
Go to the documentation of this file.
1//===-- X86FixupLEAs.cpp - use or replace LEA instructions -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines the pass that finds instructions that can be
10// re-written as LEA instructions in order to reduce pipeline delays.
11// It replaces LEAs with ADD/INC/DEC when that is better for size/speed.
12//
13//===----------------------------------------------------------------------===//
14
15#include "X86.h"
16#include "X86InstrInfo.h"
17#include "X86Subtarget.h"
18#include "llvm/ADT/Statistic.h"
24#include "llvm/CodeGen/Passes.h"
27#include "llvm/Support/Debug.h"
29using namespace llvm;
30
31#define FIXUPLEA_DESC "X86 LEA Fixup"
32#define FIXUPLEA_NAME "x86-fixup-leas"
33
34#define DEBUG_TYPE FIXUPLEA_NAME
35
36STATISTIC(NumLEAs, "Number of LEA instructions created");
37
39 "x86-fixup-leas-search-distance-threshold", cl::Hidden,
40 cl::desc("Maximum instruction distance when searching for an ADD or SUB "
41 "after a LEA"),
42 cl::init(5));
43
44namespace {
45class FixupLEAsImpl {
46 enum RegUsageState { RU_NotUsed, RU_Write, RU_Read };
47
48 /// Given a machine register, look for the instruction
49 /// which writes it in the current basic block. If found,
50 /// try to replace it with an equivalent LEA instruction.
51 /// If replacement succeeds, then also process the newly created
52 /// instruction.
53 void seekLEAFixup(MachineOperand &p, MachineBasicBlock::iterator &I,
54 MachineBasicBlock &MBB);
55
56 /// Given a memory access or LEA instruction
57 /// whose address mode uses a base and/or index register, look for
58 /// an opportunity to replace the instruction which sets the base or index
59 /// register with an equivalent LEA instruction.
60 void processInstruction(MachineBasicBlock::iterator &I,
61 MachineBasicBlock &MBB);
62
63 /// Given a LEA instruction which is unprofitable
64 /// on SlowLEA targets try to replace it with an equivalent ADD instruction.
65 void processInstructionForSlowLEA(MachineBasicBlock::iterator &I,
66 MachineBasicBlock &MBB);
67
68 /// Given a LEA instruction which is unprofitable
69 /// on SNB+ try to replace it with other instructions.
70 /// According to Intel's Optimization Reference Manual:
71 /// " For LEA instructions with three source operands and some specific
72 /// situations, instruction latency has increased to 3 cycles, and must
73 /// dispatch via port 1:
74 /// - LEA that has all three source operands: base, index, and offset
75 /// - LEA that uses base and index registers where the base is EBP, RBP,
76 /// or R13
77 /// - LEA that uses RIP relative addressing mode
78 /// - LEA that uses 16-bit addressing mode "
79 /// This function currently handles the first 2 cases only.
80 void processInstrForSlow3OpLEA(MachineBasicBlock::iterator &I,
81 MachineBasicBlock &MBB, bool OptIncDec);
82
83 /// Look for LEAs that are really two address LEAs that we might be able to
84 /// turn into regular ADD instructions.
85 bool optTwoAddrLEA(MachineBasicBlock::iterator &I,
86 MachineBasicBlock &MBB, bool OptIncDec,
87 bool UseLEAForSP) const;
88
89 /// Look for and transform the sequence
90 /// lea (reg1, reg2), reg3
91 /// sub reg3, reg4
92 /// to
93 /// sub reg1, reg4
94 /// sub reg2, reg4
95 /// It can also optimize the sequence lea/add similarly.
96 bool optLEAALU(MachineBasicBlock::iterator &I, MachineBasicBlock &MBB) const;
97
98 /// Step forwards in MBB, looking for an ADD/SUB instruction which uses
99 /// the dest register of LEA instruction I.
101 MachineBasicBlock &MBB) const;
102
103 /// Check instructions between LeaI and AluI (exclusively).
104 /// Set BaseIndexDef to true if base or index register from LeaI is defined.
105 /// Set AluDestRef to true if the dest register of AluI is used or defined.
106 /// *KilledBase is set to the killed base register usage.
107 /// *KilledIndex is set to the killed index register usage.
108 void checkRegUsage(MachineBasicBlock::iterator &LeaI,
109 MachineBasicBlock::iterator &AluI, bool &BaseIndexDef,
110 bool &AluDestRef, MachineOperand **KilledBase,
111 MachineOperand **KilledIndex) const;
112
113 /// Determine if an instruction references a machine register
114 /// and, if so, whether it reads or writes the register.
115 RegUsageState usesRegister(MachineOperand &p, MachineBasicBlock::iterator I);
116
117 /// Step backwards through a basic block, looking
118 /// for an instruction which writes a register within
119 /// a maximum of INSTR_DISTANCE_THRESHOLD instruction latency cycles.
120 MachineBasicBlock::iterator searchBackwards(MachineOperand &p,
122 MachineBasicBlock &MBB);
123
124 /// if an instruction can be converted to an
125 /// equivalent LEA, insert the new instruction into the basic block
126 /// and return a pointer to it. Otherwise, return zero.
127 MachineInstr *postRAConvertToLEA(MachineBasicBlock &MBB,
129
130public:
131 FixupLEAsImpl(ProfileSummaryInfo *PSI, MachineBlockFrequencyInfo *MBFI)
132 : PSI(PSI), MBFI(MBFI) {}
133
134 /// Loop over all of the basic blocks,
135 /// replacing instructions by equivalent LEA instructions
136 /// if needed and when possible.
137 bool runOnMachineFunction(MachineFunction &MF);
138
139private:
140 TargetSchedModel TSM;
141 const X86InstrInfo *TII = nullptr;
142 const X86RegisterInfo *TRI = nullptr;
143 ProfileSummaryInfo *PSI;
144 MachineBlockFrequencyInfo *MBFI;
145};
146
147class FixupLEAsLegacy : public MachineFunctionPass {
148public:
149 static char ID;
150
151 StringRef getPassName() const override { return FIXUPLEA_DESC; }
152
153 FixupLEAsLegacy() : MachineFunctionPass(ID) {}
154
155 bool runOnMachineFunction(MachineFunction &MF) override;
156
157 // This pass runs after regalloc and doesn't support VReg operands.
158 MachineFunctionProperties getRequiredProperties() const override {
159 return MachineFunctionProperties().setNoVRegs();
160 }
161
162 void getAnalysisUsage(AnalysisUsage &AU) const override {
163 AU.addRequired<ProfileSummaryInfoWrapperPass>();
164 AU.addRequired<LazyMachineBlockFrequencyInfoPass>();
166 }
167};
168}
169
170char FixupLEAsLegacy::ID = 0;
171
172INITIALIZE_PASS(FixupLEAsLegacy, FIXUPLEA_NAME, FIXUPLEA_DESC, false, false)
173
175FixupLEAsImpl::postRAConvertToLEA(MachineBasicBlock &MBB,
177 MachineInstr &MI = *MBBI;
178 switch (MI.getOpcode()) {
179 case X86::MOV32rr:
180 case X86::MOV64rr: {
181 const MachineOperand &Src = MI.getOperand(1);
182 const MachineOperand &Dest = MI.getOperand(0);
183 MachineInstr *NewMI =
184 BuildMI(MBB, MBBI, MI.getDebugLoc(),
185 TII->get(MI.getOpcode() == X86::MOV32rr ? X86::LEA32r
186 : X86::LEA64r))
187 .add(Dest)
188 .add(Src)
189 .addImm(1)
190 .addReg(0)
191 .addImm(0)
192 .addReg(0);
193 return NewMI;
194 }
195 }
196
197 if (!MI.isConvertibleTo3Addr())
198 return nullptr;
199
200 switch (MI.getOpcode()) {
201 default:
202 // Only convert instructions that we've verified are safe.
203 return nullptr;
204 case X86::ADD64ri32:
205 case X86::ADD64ri32_DB:
206 case X86::ADD32ri:
207 case X86::ADD32ri_DB:
208 if (!MI.getOperand(2).isImm()) {
209 // convertToThreeAddress will call getImm()
210 // which requires isImm() to be true
211 return nullptr;
212 }
213 break;
214 case X86::SHL64ri:
215 case X86::SHL32ri:
216 case X86::INC64r:
217 case X86::INC32r:
218 case X86::DEC64r:
219 case X86::DEC32r:
220 case X86::ADD64rr:
221 case X86::ADD64rr_DB:
222 case X86::ADD32rr:
223 case X86::ADD32rr_DB:
224 // These instructions are all fine to convert.
225 break;
226 }
227 return TII->convertToThreeAddress(MI, /*LIS=*/nullptr);
228}
229
231 return new FixupLEAsLegacy();
232}
233
234static bool isLEA(unsigned Opcode) {
235 return Opcode == X86::LEA32r || Opcode == X86::LEA64r ||
236 Opcode == X86::LEA64_32r;
237}
238
239bool FixupLEAsImpl::runOnMachineFunction(MachineFunction &MF) {
240 const X86Subtarget &ST = MF.getSubtarget<X86Subtarget>();
241 bool IsSlowLEA = ST.slowLEA();
242 bool IsSlow3OpsLEA = ST.slow3OpsLEA();
243 bool LEAUsesAG = ST.leaUsesAG();
244
245 bool OptIncDec = !ST.slowIncDec() || MF.getFunction().hasOptSize();
246 bool UseLEAForSP = ST.useLeaForSP();
247
248 TSM.init(&ST);
249 TII = ST.getInstrInfo();
250 TRI = ST.getRegisterInfo();
251
252 LLVM_DEBUG(dbgs() << "Start X86FixupLEAs\n";);
253 for (MachineBasicBlock &MBB : MF) {
254 // First pass. Try to remove or optimize existing LEAs.
255 bool OptIncDecPerBB =
256 OptIncDec || llvm::shouldOptimizeForSize(&MBB, PSI, MBFI);
257 for (MachineBasicBlock::iterator I = MBB.begin(); I != MBB.end(); ++I) {
258 if (!isLEA(I->getOpcode()))
259 continue;
260
261 if (optTwoAddrLEA(I, MBB, OptIncDecPerBB, UseLEAForSP))
262 continue;
263
264 if (IsSlowLEA)
265 processInstructionForSlowLEA(I, MBB);
266 else if (IsSlow3OpsLEA)
267 processInstrForSlow3OpLEA(I, MBB, OptIncDecPerBB);
268 }
269
270 // Second pass for creating LEAs. This may reverse some of the
271 // transformations above.
272 if (LEAUsesAG) {
273 for (MachineBasicBlock::iterator I = MBB.begin(); I != MBB.end(); ++I)
274 processInstruction(I, MBB);
275 }
276 }
277
278 LLVM_DEBUG(dbgs() << "End X86FixupLEAs\n";);
279
280 return true;
281}
282
283FixupLEAsImpl::RegUsageState
284FixupLEAsImpl::usesRegister(MachineOperand &p, MachineBasicBlock::iterator I) {
285 RegUsageState RegUsage = RU_NotUsed;
286 MachineInstr &MI = *I;
287
288 for (const MachineOperand &MO : MI.operands()) {
289 if (MO.isReg() && MO.getReg() == p.getReg()) {
290 if (MO.isDef())
291 return RU_Write;
292 RegUsage = RU_Read;
293 }
294 }
295 return RegUsage;
296}
297
298/// getPreviousInstr - Given a reference to an instruction in a basic
299/// block, return a reference to the previous instruction in the block,
300/// wrapping around to the last instruction of the block if the block
301/// branches to itself.
304 if (I == MBB.begin()) {
305 if (MBB.isPredecessor(&MBB)) {
306 I = --MBB.end();
307 return true;
308 } else
309 return false;
310 }
311 --I;
312 return true;
313}
314
315MachineBasicBlock::iterator FixupLEAsImpl::searchBackwards(
316 MachineOperand &p, MachineBasicBlock::iterator &I, MachineBasicBlock &MBB) {
317 int InstrDistance = 1;
319 static const int INSTR_DISTANCE_THRESHOLD = 5;
320
321 CurInst = I;
322 bool Found;
323 Found = getPreviousInstr(CurInst, MBB);
324 while (Found && I != CurInst) {
325 if (CurInst->isCall() || CurInst->isInlineAsm())
326 break;
327 if (InstrDistance > INSTR_DISTANCE_THRESHOLD)
328 break; // too far back to make a difference
329 if (usesRegister(p, CurInst) == RU_Write) {
330 return CurInst;
331 }
332 InstrDistance += TSM.computeInstrLatency(&*CurInst);
333 Found = getPreviousInstr(CurInst, MBB);
334 }
336}
337
338static inline bool isInefficientLEAReg(Register Reg) {
339 return Reg == X86::EBP || Reg == X86::RBP ||
340 Reg == X86::R13D || Reg == X86::R13;
341}
342
343/// Returns true if this LEA uses base and index registers, and the base
344/// register is known to be inefficient for the subtarget.
345// TODO: use a variant scheduling class to model the latency profile
346// of LEA instructions, and implement this logic as a scheduling predicate.
348 const MachineOperand &Index) {
349 return Base.isReg() && isInefficientLEAReg(Base.getReg()) && Index.isReg() &&
350 Index.getReg().isValid();
351}
352
353// Returns true if this operand may have a non-zero offset.
354static inline bool mayHaveOffset(const MachineOperand &Offset) {
355 return !(Offset.isImm() && Offset.getImm() == 0);
356}
357
358static inline unsigned getADDrrFromLEA(unsigned LEAOpcode) {
359 switch (LEAOpcode) {
360 default:
361 llvm_unreachable("Unexpected LEA instruction");
362 case X86::LEA32r:
363 case X86::LEA64_32r:
364 return X86::ADD32rr;
365 case X86::LEA64r:
366 return X86::ADD64rr;
367 }
368}
369
370static inline unsigned getSUBrrFromLEA(unsigned LEAOpcode) {
371 switch (LEAOpcode) {
372 default:
373 llvm_unreachable("Unexpected LEA instruction");
374 case X86::LEA32r:
375 case X86::LEA64_32r:
376 return X86::SUB32rr;
377 case X86::LEA64r:
378 return X86::SUB64rr;
379 }
380}
381
382static inline unsigned getADDriFromLEA(unsigned LEAOpcode,
383 const MachineOperand &Offset) {
384 switch (LEAOpcode) {
385 default:
386 llvm_unreachable("Unexpected LEA instruction");
387 case X86::LEA32r:
388 case X86::LEA64_32r:
389 return X86::ADD32ri;
390 case X86::LEA64r:
391 return X86::ADD64ri32;
392 }
393}
394
395static inline unsigned getSUBriFromLEA(unsigned LEAOpcode) {
396 switch (LEAOpcode) {
397 default:
398 llvm_unreachable("Unexpected LEA instruction");
399 case X86::LEA32r:
400 case X86::LEA64_32r:
401 return X86::SUB32ri;
402 case X86::LEA64r:
403 return X86::SUB64ri32;
404 }
405}
406
407static inline unsigned getINCDECFromLEA(unsigned LEAOpcode, bool IsINC) {
408 switch (LEAOpcode) {
409 default:
410 llvm_unreachable("Unexpected LEA instruction");
411 case X86::LEA32r:
412 case X86::LEA64_32r:
413 return IsINC ? X86::INC32r : X86::DEC32r;
414 case X86::LEA64r:
415 return IsINC ? X86::INC64r : X86::DEC64r;
416 }
417}
418
420FixupLEAsImpl::searchALUInst(MachineBasicBlock::iterator &I,
421 MachineBasicBlock &MBB) const {
422 unsigned InstrDistance = 1;
423
424 unsigned LEAOpcode = I->getOpcode();
425 unsigned AddOpcode = getADDrrFromLEA(LEAOpcode);
426 unsigned SubOpcode = getSUBrrFromLEA(LEAOpcode);
427 Register DestReg = I->getOperand(0).getReg();
428
429 for (MachineInstr &CurInst : instructionsWithoutDebug(
430 std::next(I), MBB.end(), /*SkipPseudoOp=*/false)) {
431 if (CurInst.isCall() || CurInst.isInlineAsm())
432 break;
433 if (InstrDistance > SearchALUInstrDistanceThreshold)
434 break;
435
436 // Check if the lea dest register is used in an add/sub instruction only.
437 for (unsigned I = 0, E = CurInst.getNumOperands(); I != E; ++I) {
438 MachineOperand &Opnd = CurInst.getOperand(I);
439 if (Opnd.isReg()) {
440 if (Opnd.getReg() == DestReg) {
441 if (Opnd.isDef() || !Opnd.isKill())
443
444 unsigned AluOpcode = CurInst.getOpcode();
445 if (AluOpcode != AddOpcode && AluOpcode != SubOpcode)
447
448 MachineOperand &Opnd2 = CurInst.getOperand(3 - I);
449 MachineOperand AluDest = CurInst.getOperand(0);
450 if (Opnd2.getReg() != AluDest.getReg())
452
453 // X - (Y + Z) may generate different flags than (X - Y) - Z when
454 // there is overflow. So we can't change the alu instruction if the
455 // flags register is live.
456 if (!CurInst.registerDefIsDead(X86::EFLAGS, TRI))
458
459 return CurInst.getIterator();
460 }
461 if (TRI->regsOverlap(DestReg, Opnd.getReg()))
463 }
464 }
465
466 InstrDistance++;
467 }
469}
470
471void FixupLEAsImpl::checkRegUsage(MachineBasicBlock::iterator &LeaI,
473 bool &BaseIndexDef, bool &AluDestRef,
474 MachineOperand **KilledBase,
475 MachineOperand **KilledIndex) const {
476 BaseIndexDef = AluDestRef = false;
477 *KilledBase = *KilledIndex = nullptr;
478 Register BaseReg = LeaI->getOperand(1 + X86::AddrBaseReg).getReg();
479 Register IndexReg = LeaI->getOperand(1 + X86::AddrIndexReg).getReg();
480 Register AluDestReg = AluI->getOperand(0).getReg();
481
482 for (MachineInstr &CurInst : llvm::make_range(std::next(LeaI), AluI)) {
483 for (MachineOperand &Opnd : CurInst.operands()) {
484 if (!Opnd.isReg())
485 continue;
486 Register Reg = Opnd.getReg();
487 if (TRI->regsOverlap(Reg, AluDestReg))
488 AluDestRef = true;
489 if (TRI->regsOverlap(Reg, BaseReg)) {
490 if (Opnd.isDef())
491 BaseIndexDef = true;
492 else if (Opnd.isKill())
493 *KilledBase = &Opnd;
494 }
495 if (TRI->regsOverlap(Reg, IndexReg)) {
496 if (Opnd.isDef())
497 BaseIndexDef = true;
498 else if (Opnd.isKill())
499 *KilledIndex = &Opnd;
500 }
501 }
502 }
503}
504
505bool FixupLEAsImpl::optLEAALU(MachineBasicBlock::iterator &I,
506 MachineBasicBlock &MBB) const {
507 // Look for an add/sub instruction which uses the result of lea.
508 MachineBasicBlock::iterator AluI = searchALUInst(I, MBB);
509 if (AluI == MachineBasicBlock::iterator())
510 return false;
511
512 // Check if there are any related register usage between lea and alu.
513 bool BaseIndexDef, AluDestRef;
514 MachineOperand *KilledBase, *KilledIndex;
515 checkRegUsage(I, AluI, BaseIndexDef, AluDestRef, &KilledBase, &KilledIndex);
516
517 MachineBasicBlock::iterator InsertPos = AluI;
518 if (BaseIndexDef) {
519 if (AluDestRef)
520 return false;
521 InsertPos = I;
522 KilledBase = KilledIndex = nullptr;
523 }
524
525 // Check if there are same registers.
526 Register AluDestReg = AluI->getOperand(0).getReg();
527 Register BaseReg = I->getOperand(1 + X86::AddrBaseReg).getReg();
528 Register IndexReg = I->getOperand(1 + X86::AddrIndexReg).getReg();
529 if (I->getOpcode() == X86::LEA64_32r) {
530 BaseReg = TRI->getSubReg(BaseReg, X86::sub_32bit);
531 IndexReg = TRI->getSubReg(IndexReg, X86::sub_32bit);
532 }
533 if (AluDestReg == IndexReg) {
534 if (BaseReg == IndexReg)
535 return false;
536 std::swap(BaseReg, IndexReg);
537 std::swap(KilledBase, KilledIndex);
538 }
539 if (BaseReg == IndexReg)
540 KilledBase = nullptr;
541
542 // Now it's safe to change instructions.
543 MachineInstr *NewMI1, *NewMI2;
544 unsigned NewOpcode = AluI->getOpcode();
545 NewMI1 = BuildMI(MBB, InsertPos, AluI->getDebugLoc(), TII->get(NewOpcode),
546 AluDestReg)
547 .addReg(AluDestReg, RegState::Kill)
548 .addReg(BaseReg, getKillRegState(KilledBase));
549 NewMI1->addRegisterDead(X86::EFLAGS, TRI);
550 NewMI2 = BuildMI(MBB, InsertPos, AluI->getDebugLoc(), TII->get(NewOpcode),
551 AluDestReg)
552 .addReg(AluDestReg, RegState::Kill)
553 .addReg(IndexReg, getKillRegState(KilledIndex));
554 NewMI2->addRegisterDead(X86::EFLAGS, TRI);
555
556 // Clear the old Kill flags.
557 if (KilledBase)
558 KilledBase->setIsKill(false);
559 if (KilledIndex)
560 KilledIndex->setIsKill(false);
561
562 MBB.getParent()->substituteDebugValuesForInst(*AluI, *NewMI2, 1);
563 MBB.erase(I);
564 MBB.erase(AluI);
565 I = NewMI1;
566 return true;
567}
568
569bool FixupLEAsImpl::optTwoAddrLEA(MachineBasicBlock::iterator &I,
570 MachineBasicBlock &MBB, bool OptIncDec,
571 bool UseLEAForSP) const {
572 MachineInstr &MI = *I;
573
574 const MachineOperand &Base = MI.getOperand(1 + X86::AddrBaseReg);
575 const MachineOperand &Scale = MI.getOperand(1 + X86::AddrScaleAmt);
576 const MachineOperand &Index = MI.getOperand(1 + X86::AddrIndexReg);
577 const MachineOperand &Disp = MI.getOperand(1 + X86::AddrDisp);
578 const MachineOperand &Segment = MI.getOperand(1 + X86::AddrSegmentReg);
579
580 if (Segment.getReg().isValid() || !Disp.isImm() || Scale.getImm() > 1 ||
581 MBB.computeRegisterLiveness(TRI, X86::EFLAGS, I) !=
583 return false;
584
585 Register DestReg = MI.getOperand(0).getReg();
586 Register BaseReg = Base.getReg();
587 Register IndexReg = Index.getReg();
588
589 // Don't change stack adjustment LEAs.
590 if (UseLEAForSP && (DestReg == X86::ESP || DestReg == X86::RSP))
591 return false;
592
593 // LEA64_32 has 64-bit operands but 32-bit result.
594 if (MI.getOpcode() == X86::LEA64_32r) {
595 if (BaseReg)
596 BaseReg = TRI->getSubReg(BaseReg, X86::sub_32bit);
597 if (IndexReg)
598 IndexReg = TRI->getSubReg(IndexReg, X86::sub_32bit);
599 }
600
601 MachineInstr *NewMI = nullptr;
602
603 // Case 1.
604 // Look for lea(%reg1, %reg2), %reg1 or lea(%reg2, %reg1), %reg1
605 // which can be turned into add %reg2, %reg1
606 if (BaseReg.isValid() && IndexReg.isValid() && Disp.getImm() == 0 &&
607 (DestReg == BaseReg || DestReg == IndexReg)) {
608 unsigned NewOpcode = getADDrrFromLEA(MI.getOpcode());
609 if (DestReg != BaseReg)
610 std::swap(BaseReg, IndexReg);
611
612 if (MI.getOpcode() == X86::LEA64_32r) {
613 // TODO: Do we need the super register implicit use?
614 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
615 .addReg(BaseReg).addReg(IndexReg)
616 .addReg(Base.getReg(), RegState::Implicit)
617 .addReg(Index.getReg(), RegState::Implicit);
618 } else {
619 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
620 .addReg(BaseReg).addReg(IndexReg);
621 }
622 } else if (DestReg == BaseReg && !IndexReg) {
623 // Case 2.
624 // This is an LEA with only a base register and a displacement,
625 // We can use ADDri or INC/DEC.
626
627 // Does this LEA have one these forms:
628 // lea %reg, 1(%reg)
629 // lea %reg, -1(%reg)
630 if (OptIncDec && (Disp.getImm() == 1 || Disp.getImm() == -1)) {
631 bool IsINC = Disp.getImm() == 1;
632 unsigned NewOpcode = getINCDECFromLEA(MI.getOpcode(), IsINC);
633
634 if (MI.getOpcode() == X86::LEA64_32r) {
635 // TODO: Do we need the super register implicit use?
636 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
637 .addReg(BaseReg).addReg(Base.getReg(), RegState::Implicit);
638 } else {
639 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
640 .addReg(BaseReg);
641 }
642 } else {
643 unsigned NewOpcode = getADDriFromLEA(MI.getOpcode(), Disp);
644 if (MI.getOpcode() == X86::LEA64_32r) {
645 // TODO: Do we need the super register implicit use?
646 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
647 .addReg(BaseReg).addImm(Disp.getImm())
648 .addReg(Base.getReg(), RegState::Implicit);
649 } else {
650 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpcode), DestReg)
651 .addReg(BaseReg).addImm(Disp.getImm());
652 }
653 }
654 } else if (BaseReg.isValid() && IndexReg.isValid() && Disp.getImm() == 0) {
655 // Case 3.
656 // Look for and transform the sequence
657 // lea (reg1, reg2), reg3
658 // sub reg3, reg4
659 return optLEAALU(I, MBB);
660 } else
661 return false;
662
664 MBB.erase(I);
665 I = NewMI;
666 return true;
667}
668
669void FixupLEAsImpl::processInstruction(MachineBasicBlock::iterator &I,
670 MachineBasicBlock &MBB) {
671 // Process a load, store, or LEA instruction.
672 MachineInstr &MI = *I;
673 int AddrOffset = X86II::getMemoryOperandIdx(MI.getDesc());
674 if (AddrOffset >= 0) {
675 MachineOperand &p = MI.getOperand(AddrOffset + X86::AddrBaseReg);
676 if (p.isReg() && p.getReg() != X86::ESP) {
677 seekLEAFixup(p, I, MBB);
678 }
679 MachineOperand &q = MI.getOperand(AddrOffset + X86::AddrIndexReg);
680 if (q.isReg() && q.getReg() != X86::ESP) {
681 seekLEAFixup(q, I, MBB);
682 }
683 }
684}
685
686void FixupLEAsImpl::seekLEAFixup(MachineOperand &p,
688 MachineBasicBlock &MBB) {
689 MachineBasicBlock::iterator MBI = searchBackwards(p, I, MBB);
690 if (MBI != MachineBasicBlock::iterator()) {
691 MachineInstr *NewMI = postRAConvertToLEA(MBB, MBI);
692 if (NewMI) {
693 ++NumLEAs;
694 LLVM_DEBUG(dbgs() << "FixLEA: Candidate to replace:"; MBI->dump(););
695 // now to replace with an equivalent LEA...
696 LLVM_DEBUG(dbgs() << "FixLEA: Replaced by: "; NewMI->dump(););
697 MBB.getParent()->substituteDebugValuesForInst(*MBI, *NewMI, 1);
698 MBB.erase(MBI);
700 static_cast<MachineBasicBlock::iterator>(NewMI);
701 processInstruction(J, MBB);
702 }
703 }
704}
705
706void FixupLEAsImpl::processInstructionForSlowLEA(MachineBasicBlock::iterator &I,
707 MachineBasicBlock &MBB) {
708 MachineInstr &MI = *I;
709 const unsigned Opcode = MI.getOpcode();
710
711 const MachineOperand &Dst = MI.getOperand(0);
712 const MachineOperand &Base = MI.getOperand(1 + X86::AddrBaseReg);
713 const MachineOperand &Scale = MI.getOperand(1 + X86::AddrScaleAmt);
714 const MachineOperand &Index = MI.getOperand(1 + X86::AddrIndexReg);
715 const MachineOperand &Offset = MI.getOperand(1 + X86::AddrDisp);
716 const MachineOperand &Segment = MI.getOperand(1 + X86::AddrSegmentReg);
717
718 if (Segment.getReg().isValid() || !Offset.isImm() ||
719 MBB.computeRegisterLiveness(TRI, X86::EFLAGS, I, 4) !=
721 return;
722 const Register DstR = Dst.getReg();
723 const Register SrcR1 = Base.getReg();
724 const Register SrcR2 = Index.getReg();
725 if ((!SrcR1 || SrcR1 != DstR) && (!SrcR2 || SrcR2 != DstR))
726 return;
727 if (Scale.getImm() > 1)
728 return;
729 LLVM_DEBUG(dbgs() << "FixLEA: Candidate to replace:"; I->dump(););
730 LLVM_DEBUG(dbgs() << "FixLEA: Replaced by: ";);
731 MachineInstr *NewMI = nullptr;
732 // Make ADD instruction for two registers writing to LEA's destination
733 if (SrcR1 && SrcR2) {
734 const MCInstrDesc &ADDrr = TII->get(getADDrrFromLEA(Opcode));
735 const MachineOperand &Src = SrcR1 == DstR ? Index : Base;
736 NewMI =
737 BuildMI(MBB, I, MI.getDebugLoc(), ADDrr, DstR).addReg(DstR).add(Src);
738 LLVM_DEBUG(NewMI->dump(););
739 }
740 // Make ADD instruction for immediate
741 if (Offset.getImm() != 0) {
742 const MCInstrDesc &ADDri =
743 TII->get(getADDriFromLEA(Opcode, Offset));
744 const MachineOperand &SrcR = SrcR1 == DstR ? Base : Index;
745 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), ADDri, DstR)
746 .add(SrcR)
747 .addImm(Offset.getImm());
748 LLVM_DEBUG(NewMI->dump(););
749 }
750 if (NewMI) {
752 MBB.erase(I);
753 I = NewMI;
754 }
755}
756
757void FixupLEAsImpl::processInstrForSlow3OpLEA(MachineBasicBlock::iterator &I,
758 MachineBasicBlock &MBB,
759 bool OptIncDec) {
760 MachineInstr &MI = *I;
761 const unsigned LEAOpcode = MI.getOpcode();
762
763 const MachineOperand &Dest = MI.getOperand(0);
764 const MachineOperand &Base = MI.getOperand(1 + X86::AddrBaseReg);
765 const MachineOperand &Scale = MI.getOperand(1 + X86::AddrScaleAmt);
766 const MachineOperand &Index = MI.getOperand(1 + X86::AddrIndexReg);
767 const MachineOperand &Offset = MI.getOperand(1 + X86::AddrDisp);
768 const MachineOperand &Segment = MI.getOperand(1 + X86::AddrSegmentReg);
769
770 if (!(TII->isThreeOperandsLEA(MI) || hasInefficientLEABaseReg(Base, Index)) ||
771 MBB.computeRegisterLiveness(TRI, X86::EFLAGS, I, 4) !=
773 Segment.getReg().isValid())
774 return;
775
776 Register DestReg = Dest.getReg();
777 Register BaseReg = Base.getReg();
778 Register IndexReg = Index.getReg();
779
780 if (MI.getOpcode() == X86::LEA64_32r) {
781 if (BaseReg)
782 BaseReg = TRI->getSubReg(BaseReg, X86::sub_32bit);
783 if (IndexReg)
784 IndexReg = TRI->getSubReg(IndexReg, X86::sub_32bit);
785 }
786
787 bool IsScale1 = Scale.getImm() == 1;
788 bool IsInefficientBase = isInefficientLEAReg(BaseReg);
789 bool IsInefficientIndex = isInefficientLEAReg(IndexReg);
790
791 // Skip these cases since it takes more than 2 instructions
792 // to replace the LEA instruction.
793 if (IsInefficientBase && DestReg == BaseReg && !IsScale1)
794 return;
795
796 LLVM_DEBUG(dbgs() << "FixLEA: Candidate to replace:"; MI.dump(););
797 LLVM_DEBUG(dbgs() << "FixLEA: Replaced by: ";);
798
799 MachineInstr *NewMI = nullptr;
800 bool BaseOrIndexIsDst = DestReg == BaseReg || DestReg == IndexReg;
801 // First try and remove the base while sticking with LEA iff base == index and
802 // scale == 1. We can handle:
803 // 1. lea D(%base,%index,1) -> lea D(,%index,2)
804 // 2. lea D(%r13/%rbp,%index) -> lea D(,%index,2)
805 // Only do this if the LEA would otherwise be split into 2-instruction
806 // (either it has a an Offset or neither base nor index are dst)
807 if (IsScale1 && BaseReg == IndexReg &&
808 (mayHaveOffset(Offset) || (IsInefficientBase && !BaseOrIndexIsDst))) {
809 NewMI = BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(LEAOpcode))
810 .add(Dest)
811 .addReg(0)
812 .addImm(2)
813 .add(Index)
814 .add(Offset)
815 .add(Segment);
816 LLVM_DEBUG(NewMI->dump(););
817
819 MBB.erase(I);
820 I = NewMI;
821 return;
822 } else if (IsScale1 && BaseOrIndexIsDst) {
823 // Try to replace LEA with one or two (for the 3-op LEA case)
824 // add instructions:
825 // 1.lea (%base,%index,1), %base => add %index,%base
826 // 2.lea (%base,%index,1), %index => add %base,%index
827
828 unsigned NewOpc = getADDrrFromLEA(MI.getOpcode());
829 if (DestReg != BaseReg)
830 std::swap(BaseReg, IndexReg);
831
832 if (MI.getOpcode() == X86::LEA64_32r) {
833 // TODO: Do we need the super register implicit use?
834 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
835 .addReg(BaseReg)
836 .addReg(IndexReg)
837 .addReg(Base.getReg(), RegState::Implicit)
838 .addReg(Index.getReg(), RegState::Implicit);
839 } else {
840 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
841 .addReg(BaseReg)
842 .addReg(IndexReg);
843 }
844 } else if (!IsInefficientBase || (!IsInefficientIndex && IsScale1)) {
845 // If the base is inefficient try switching the index and base operands,
846 // otherwise just break the 3-Ops LEA inst into 2-Ops LEA + ADD instruction:
847 // lea offset(%base,%index,scale),%dst =>
848 // lea (%base,%index,scale); add offset,%dst
849 NewMI = BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(LEAOpcode))
850 .add(Dest)
851 .add(IsInefficientBase ? Index : Base)
852 .add(Scale)
853 .add(IsInefficientBase ? Base : Index)
854 .addImm(0)
855 .add(Segment);
856 LLVM_DEBUG(NewMI->dump(););
857 }
858
859 // If either replacement succeeded above, add the offset if needed, then
860 // replace the instruction.
861 if (NewMI) {
862 // Create ADD instruction for the Offset in case of 3-Ops LEA.
863 if (mayHaveOffset(Offset)) {
864 if (OptIncDec && Offset.isImm() &&
865 (Offset.getImm() == 1 || Offset.getImm() == -1)) {
866 unsigned NewOpc =
867 getINCDECFromLEA(MI.getOpcode(), Offset.getImm() == 1);
868 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
869 .addReg(DestReg);
870 LLVM_DEBUG(NewMI->dump(););
871 } else if (Offset.isImm() && Offset.getImm() == 128) {
872 // ADD of +128 needs a 32-bit immediate, while SUB of -128 fits the
873 // sign-extended 8-bit form, three bytes shorter. EFLAGS was proved
874 // dead above, so the different flag results don't matter.
875 unsigned NewOpc = getSUBriFromLEA(MI.getOpcode());
876 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
877 .addReg(DestReg)
878 .addImm(-128);
879 LLVM_DEBUG(NewMI->dump(););
880 } else {
881 unsigned NewOpc = getADDriFromLEA(MI.getOpcode(), Offset);
882 NewMI = BuildMI(MBB, I, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
883 .addReg(DestReg)
884 .add(Offset);
885 LLVM_DEBUG(NewMI->dump(););
886 }
887 }
888
890 MBB.erase(I);
891 I = NewMI;
892 return;
893 }
894
895 // Handle the rest of the cases with inefficient base register:
896 assert(DestReg != BaseReg && "DestReg == BaseReg should be handled already!");
897 assert(IsInefficientBase && "efficient base should be handled already!");
898
899 // FIXME: Handle LEA64_32r.
900 if (LEAOpcode == X86::LEA64_32r)
901 return;
902
903 // lea (%base,%index,1), %dst => mov %base,%dst; add %index,%dst
904 if (IsScale1 && !mayHaveOffset(Offset)) {
905 bool BIK = Base.isKill() && BaseReg != IndexReg;
906 TII->copyPhysReg(MBB, MI, MI.getDebugLoc(), DestReg, BaseReg, BIK);
907 LLVM_DEBUG(MI.getPrevNode()->dump(););
908
909 unsigned NewOpc = getADDrrFromLEA(MI.getOpcode());
910 NewMI = BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
911 .addReg(DestReg)
912 .add(Index);
913 LLVM_DEBUG(NewMI->dump(););
914
916 MBB.erase(I);
917 I = NewMI;
918 return;
919 }
920
921 // lea offset(%base,%index,scale), %dst =>
922 // lea offset( ,%index,scale), %dst; add %base,%dst
923 NewMI = BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(LEAOpcode))
924 .add(Dest)
925 .addReg(0)
926 .add(Scale)
927 .add(Index)
928 .add(Offset)
929 .add(Segment);
930 LLVM_DEBUG(NewMI->dump(););
931
932 unsigned NewOpc = getADDrrFromLEA(MI.getOpcode());
933 NewMI = BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(NewOpc), DestReg)
934 .addReg(DestReg)
935 .add(Base);
936 LLVM_DEBUG(NewMI->dump(););
937
939 MBB.erase(I);
940 I = NewMI;
941}
942
943bool FixupLEAsLegacy::runOnMachineFunction(MachineFunction &MF) {
944 if (skipFunction(MF.getFunction()))
945 return false;
946
947 auto *PSI = &getAnalysis<ProfileSummaryInfoWrapperPass>().getPSI();
948 auto *MBFI = (PSI && PSI->hasProfileSummary())
949 ? &getAnalysis<LazyMachineBlockFrequencyInfoPass>().getBFI()
950 : nullptr;
951 FixupLEAsImpl PassImpl(PSI, MBFI);
952 return PassImpl.runOnMachineFunction(MF);
953}
954
957 ProfileSummaryInfo *PSI =
959 .getCachedResult<ProfileSummaryAnalysis>(
960 *MF.getFunction().getParent());
961 if (!PSI)
962 report_fatal_error("x86-fixup-leas requires ProfileSummaryAnalysis", false);
965
966 FixupLEAsImpl PassImpl(PSI, MBFI);
967 bool Changed = PassImpl.runOnMachineFunction(MF);
968 if (!Changed)
969 return PreservedAnalyses::all();
972 return PA;
973}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
===- LazyMachineBlockFrequencyInfo.h - Lazy Block Frequency -*- C++ -*–===//
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
static unsigned getINCDECFromLEA(unsigned LEAOpcode, bool IsINC)
static bool isLEA(unsigned Opcode)
static unsigned getSUBriFromLEA(unsigned LEAOpcode)
static bool isInefficientLEAReg(Register Reg)
static bool hasInefficientLEABaseReg(const MachineOperand &Base, const MachineOperand &Index)
Returns true if this LEA uses base and index registers, and the base register is known to be ineffici...
static unsigned getADDriFromLEA(unsigned LEAOpcode, const MachineOperand &Offset)
static bool getPreviousInstr(MachineBasicBlock::iterator &I, MachineBasicBlock &MBB)
getPreviousInstr - Given a reference to an instruction in a basic block, return a reference to the pr...
#define FIXUPLEA_DESC
static unsigned getADDrrFromLEA(unsigned LEAOpcode)
#define FIXUPLEA_NAME
static unsigned getSUBrrFromLEA(unsigned LEAOpcode)
static bool mayHaveOffset(const MachineOperand &Offset)
static cl::opt< unsigned > SearchALUInstrDistanceThreshold("x86-fixup-leas-search-distance-threshold", cl::Hidden, cl::desc("Maximum instruction distance when searching for an ADD or SUB " "after a LEA"), cl::init(5))
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
bool hasOptSize() const
Optimize this function for size (-Os) or minimum size (-Oz).
Definition Function.h:699
Module * getParent()
Get the module that this global value is contained inside of...
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
Emit instructions to copy a pair of physical registers.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
MachineBlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate machine basic b...
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
void substituteDebugValuesForInst(const MachineInstr &Old, MachineInstr &New, unsigned MaxOperand=UINT_MAX)
Create substitutions for any tracked values in Old, to point at New.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
Representation of each machine instruction.
LLVM_ABI void dump() const
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI bool addRegisterDead(Register Reg, const TargetRegisterInfo *RegInfo, bool AddIfNotFound=false)
We have determined MI defined a register without a use.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
Register getReg() const
getReg - Returns the register number.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
Analysis providing profile information.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isValid() const
Definition Register.h:112
LLVM_ABI void init(const TargetSubtargetInfo *TSInfo, bool EnableSModel=true, bool EnableSItins=true)
Initialize the machine model for instruction scheduling.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
int getMemoryOperandIdx(const MCInstrDesc &Desc)
initializer< Ty > init(const Ty &Val)
BaseReg
Stack frame base register. Bit 0 of FREInfo.Info.
Definition SFrame.h:77
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
OuterAnalysisManagerProxy< ModuleAnalysisManager, MachineFunction > ModuleAnalysisManagerMachineFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr RegState getKillRegState(bool B)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI bool shouldOptimizeForSize(const MachineFunction *MF, ProfileSummaryInfo *PSI, const MachineBlockFrequencyInfo *BFI, PGSOQueryType QueryType=PGSOQueryType::Other)
Returns true if machine function MF is suggested to be size-optimized based on the profile.
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
FunctionPass * createX86FixupLEAsLegacyPass()
auto instructionsWithoutDebug(IterT It, IterT End, bool SkipPseudoOp=true)
Construct a range iterator which begins at It and moves forwards until End is reached,...
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880