LLVM 24.0.0git
X86AvoidStoreForwardingBlocks.cpp
Go to the documentation of this file.
1//===- X86AvoidStoreForwardingBlocks.cpp - Avoid HW Store Forward Block ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// If a load follows a store and reloads data that the store has written to
10// memory, Intel microarchitectures can in many cases forward the data directly
11// from the store to the load, This "store forwarding" saves cycles by enabling
12// the load to directly obtain the data instead of accessing the data from
13// cache or memory.
14// A "store forward block" occurs in cases that a store cannot be forwarded to
15// the load. The most typical case of store forward block on Intel Core
16// microarchitecture that a small store cannot be forwarded to a large load.
17// The estimated penalty for a store forward block is ~13 cycles.
18//
19// This pass tries to recognize and handle cases where "store forward block"
20// is created by the compiler when lowering memcpy calls to a sequence
21// of a load and a store.
22//
23// The pass currently only handles cases where memcpy is lowered to
24// XMM/YMM registers, it tries to break the memcpy into smaller copies.
25// breaking the memcpy should be possible since there is no atomicity
26// guarantee for loads and stores to XMM/YMM.
27//
28// It could be better for performance to solve the problem by loading
29// to XMM/YMM then inserting the partial store before storing back from XMM/YMM
30// to memory, but this will result in a more conservative optimization since it
31// requires we prove that all memory accesses between the blocking store and the
32// load must alias/don't alias before we can move the store, whereas the
33// transformation done here is correct regardless to other memory accesses.
34//===----------------------------------------------------------------------===//
35
36#include "X86.h"
37#include "X86InstrInfo.h"
38#include "X86Subtarget.h"
48#include "llvm/IR/DebugLoc.h"
49#include "llvm/IR/Function.h"
51#include "llvm/MC/MCInstrDesc.h"
52
53using namespace llvm;
54
55#define DEBUG_TYPE "x86-avoid-sfb"
56
57namespace {
58
59using DisplacementSizeMap = std::map<int64_t, unsigned>;
60
61class X86AvoidSFBImpl {
62public:
63 X86AvoidSFBImpl(AliasAnalysis *AA) : AA(AA) {};
64 bool runOnMachineFunction(MachineFunction &MF);
65
66private:
67 MachineRegisterInfo *MRI = nullptr;
68 const X86InstrInfo *TII = nullptr;
69 const X86RegisterInfo *TRI = nullptr;
71 BlockedLoadsStoresPairs;
73 AliasAnalysis *AA = nullptr;
74
75 /// Returns couples of Load then Store to memory which look
76 /// like a memcpy.
77 void findPotentiallylBlockedCopies(MachineFunction &MF);
78 /// Break the memcpy's load and store into smaller copies
79 /// such that each memory load that was blocked by a smaller store
80 /// would now be copied separately.
81 void breakBlockedCopies(MachineInstr *LoadInst, MachineInstr *StoreInst,
82 const DisplacementSizeMap &BlockingStoresDispSizeMap);
83 /// Break a copy of size Size to smaller copies.
84 void buildCopies(int Size, MachineInstr *LoadInst, int64_t LdDispImm,
85 MachineInstr *StoreInst, int64_t StDispImm, int64_t Offset);
86
87 void buildCopy(MachineInstr *LoadInst, unsigned NLoadOpcode, int64_t LoadDisp,
88 MachineInstr *StoreInst, unsigned NStoreOpcode,
89 int64_t StoreDisp, unsigned Size, int64_t Offset);
90
91 bool alias(const MachineMemOperand &Op1, const MachineMemOperand &Op2) const;
92
93 unsigned getRegSizeInBytes(MachineInstr *Inst);
94};
95
96class X86AvoidSFBLegacy : public MachineFunctionPass {
97public:
98 static char ID;
99 X86AvoidSFBLegacy() : MachineFunctionPass(ID) {}
100
101 StringRef getPassName() const override {
102 return "X86 Avoid Store Forwarding Blocks";
103 }
104
105 bool runOnMachineFunction(MachineFunction &MF) override;
106
107 void getAnalysisUsage(AnalysisUsage &AU) const override {
111 }
112};
113
114} // end anonymous namespace
115
116char X86AvoidSFBLegacy::ID = 0;
117
118INITIALIZE_PASS_BEGIN(X86AvoidSFBLegacy, DEBUG_TYPE, "Machine code sinking",
119 false, false)
121INITIALIZE_PASS_END(X86AvoidSFBLegacy, DEBUG_TYPE, "Machine code sinking",
123
125 return new X86AvoidSFBLegacy();
126}
127
128static bool isXMMLoadOpcode(unsigned Opcode) {
129 return Opcode == X86::MOVUPSrm || Opcode == X86::MOVAPSrm ||
130 Opcode == X86::VMOVUPSrm || Opcode == X86::VMOVAPSrm ||
131 Opcode == X86::VMOVUPDrm || Opcode == X86::VMOVAPDrm ||
132 Opcode == X86::VMOVDQUrm || Opcode == X86::VMOVDQArm ||
133 Opcode == X86::VMOVUPSZ128rm || Opcode == X86::VMOVAPSZ128rm ||
134 Opcode == X86::VMOVUPDZ128rm || Opcode == X86::VMOVAPDZ128rm ||
135 Opcode == X86::VMOVDQU64Z128rm || Opcode == X86::VMOVDQA64Z128rm ||
136 Opcode == X86::VMOVDQU32Z128rm || Opcode == X86::VMOVDQA32Z128rm;
137}
138static bool isYMMLoadOpcode(unsigned Opcode) {
139 return Opcode == X86::VMOVUPSYrm || Opcode == X86::VMOVAPSYrm ||
140 Opcode == X86::VMOVUPDYrm || Opcode == X86::VMOVAPDYrm ||
141 Opcode == X86::VMOVDQUYrm || Opcode == X86::VMOVDQAYrm ||
142 Opcode == X86::VMOVUPSZ256rm || Opcode == X86::VMOVAPSZ256rm ||
143 Opcode == X86::VMOVUPDZ256rm || Opcode == X86::VMOVAPDZ256rm ||
144 Opcode == X86::VMOVDQU64Z256rm || Opcode == X86::VMOVDQA64Z256rm ||
145 Opcode == X86::VMOVDQU32Z256rm || Opcode == X86::VMOVDQA32Z256rm;
146}
147
148static bool isPotentialBlockedMemCpyLd(unsigned Opcode) {
149 return isXMMLoadOpcode(Opcode) || isYMMLoadOpcode(Opcode);
150}
151
152static bool isPotentialBlockedMemCpyPair(unsigned LdOpcode, unsigned StOpcode) {
153 switch (LdOpcode) {
154 case X86::MOVUPSrm:
155 case X86::MOVAPSrm:
156 return StOpcode == X86::MOVUPSmr || StOpcode == X86::MOVAPSmr;
157 case X86::VMOVUPSrm:
158 case X86::VMOVAPSrm:
159 return StOpcode == X86::VMOVUPSmr || StOpcode == X86::VMOVAPSmr;
160 case X86::VMOVUPDrm:
161 case X86::VMOVAPDrm:
162 return StOpcode == X86::VMOVUPDmr || StOpcode == X86::VMOVAPDmr;
163 case X86::VMOVDQUrm:
164 case X86::VMOVDQArm:
165 return StOpcode == X86::VMOVDQUmr || StOpcode == X86::VMOVDQAmr;
166 case X86::VMOVUPSZ128rm:
167 case X86::VMOVAPSZ128rm:
168 return StOpcode == X86::VMOVUPSZ128mr || StOpcode == X86::VMOVAPSZ128mr;
169 case X86::VMOVUPDZ128rm:
170 case X86::VMOVAPDZ128rm:
171 return StOpcode == X86::VMOVUPDZ128mr || StOpcode == X86::VMOVAPDZ128mr;
172 case X86::VMOVUPSYrm:
173 case X86::VMOVAPSYrm:
174 return StOpcode == X86::VMOVUPSYmr || StOpcode == X86::VMOVAPSYmr;
175 case X86::VMOVUPDYrm:
176 case X86::VMOVAPDYrm:
177 return StOpcode == X86::VMOVUPDYmr || StOpcode == X86::VMOVAPDYmr;
178 case X86::VMOVDQUYrm:
179 case X86::VMOVDQAYrm:
180 return StOpcode == X86::VMOVDQUYmr || StOpcode == X86::VMOVDQAYmr;
181 case X86::VMOVUPSZ256rm:
182 case X86::VMOVAPSZ256rm:
183 return StOpcode == X86::VMOVUPSZ256mr || StOpcode == X86::VMOVAPSZ256mr;
184 case X86::VMOVUPDZ256rm:
185 case X86::VMOVAPDZ256rm:
186 return StOpcode == X86::VMOVUPDZ256mr || StOpcode == X86::VMOVAPDZ256mr;
187 case X86::VMOVDQU64Z128rm:
188 case X86::VMOVDQA64Z128rm:
189 return StOpcode == X86::VMOVDQU64Z128mr || StOpcode == X86::VMOVDQA64Z128mr;
190 case X86::VMOVDQU32Z128rm:
191 case X86::VMOVDQA32Z128rm:
192 return StOpcode == X86::VMOVDQU32Z128mr || StOpcode == X86::VMOVDQA32Z128mr;
193 case X86::VMOVDQU64Z256rm:
194 case X86::VMOVDQA64Z256rm:
195 return StOpcode == X86::VMOVDQU64Z256mr || StOpcode == X86::VMOVDQA64Z256mr;
196 case X86::VMOVDQU32Z256rm:
197 case X86::VMOVDQA32Z256rm:
198 return StOpcode == X86::VMOVDQU32Z256mr || StOpcode == X86::VMOVDQA32Z256mr;
199 default:
200 return false;
201 }
202}
203
204static bool isPotentialBlockingStoreInst(unsigned Opcode, unsigned LoadOpcode) {
205 bool PBlock = false;
206 PBlock |= Opcode == X86::MOV64mr || Opcode == X86::MOV64mi32 ||
207 Opcode == X86::MOV32mr || Opcode == X86::MOV32mi ||
208 Opcode == X86::MOV16mr || Opcode == X86::MOV16mi ||
209 Opcode == X86::MOV8mr || Opcode == X86::MOV8mi;
210 if (isYMMLoadOpcode(LoadOpcode))
211 PBlock |= Opcode == X86::VMOVUPSmr || Opcode == X86::VMOVAPSmr ||
212 Opcode == X86::VMOVUPDmr || Opcode == X86::VMOVAPDmr ||
213 Opcode == X86::VMOVDQUmr || Opcode == X86::VMOVDQAmr ||
214 Opcode == X86::VMOVUPSZ128mr || Opcode == X86::VMOVAPSZ128mr ||
215 Opcode == X86::VMOVUPDZ128mr || Opcode == X86::VMOVAPDZ128mr ||
216 Opcode == X86::VMOVDQU64Z128mr ||
217 Opcode == X86::VMOVDQA64Z128mr ||
218 Opcode == X86::VMOVDQU32Z128mr || Opcode == X86::VMOVDQA32Z128mr;
219 return PBlock;
220}
221
222static const int MOV128SZ = 16;
223static const int MOV64SZ = 8;
224static const int MOV32SZ = 4;
225static const int MOV16SZ = 2;
226static const int MOV8SZ = 1;
227
228static unsigned getYMMtoXMMLoadOpcode(unsigned LoadOpcode) {
229 switch (LoadOpcode) {
230 case X86::VMOVUPSYrm:
231 case X86::VMOVAPSYrm:
232 return X86::VMOVUPSrm;
233 case X86::VMOVUPDYrm:
234 case X86::VMOVAPDYrm:
235 return X86::VMOVUPDrm;
236 case X86::VMOVDQUYrm:
237 case X86::VMOVDQAYrm:
238 return X86::VMOVDQUrm;
239 case X86::VMOVUPSZ256rm:
240 case X86::VMOVAPSZ256rm:
241 return X86::VMOVUPSZ128rm;
242 case X86::VMOVUPDZ256rm:
243 case X86::VMOVAPDZ256rm:
244 return X86::VMOVUPDZ128rm;
245 case X86::VMOVDQU64Z256rm:
246 case X86::VMOVDQA64Z256rm:
247 return X86::VMOVDQU64Z128rm;
248 case X86::VMOVDQU32Z256rm:
249 case X86::VMOVDQA32Z256rm:
250 return X86::VMOVDQU32Z128rm;
251 default:
252 llvm_unreachable("Unexpected Load Instruction Opcode");
253 }
254 return 0;
255}
256
257static unsigned getYMMtoXMMStoreOpcode(unsigned StoreOpcode) {
258 switch (StoreOpcode) {
259 case X86::VMOVUPSYmr:
260 case X86::VMOVAPSYmr:
261 return X86::VMOVUPSmr;
262 case X86::VMOVUPDYmr:
263 case X86::VMOVAPDYmr:
264 return X86::VMOVUPDmr;
265 case X86::VMOVDQUYmr:
266 case X86::VMOVDQAYmr:
267 return X86::VMOVDQUmr;
268 case X86::VMOVUPSZ256mr:
269 case X86::VMOVAPSZ256mr:
270 return X86::VMOVUPSZ128mr;
271 case X86::VMOVUPDZ256mr:
272 case X86::VMOVAPDZ256mr:
273 return X86::VMOVUPDZ128mr;
274 case X86::VMOVDQU64Z256mr:
275 case X86::VMOVDQA64Z256mr:
276 return X86::VMOVDQU64Z128mr;
277 case X86::VMOVDQU32Z256mr:
278 case X86::VMOVDQA32Z256mr:
279 return X86::VMOVDQU32Z128mr;
280 default:
281 llvm_unreachable("Unexpected Load Instruction Opcode");
282 }
283 return 0;
284}
285
286static int getAddrOffset(const MachineInstr *MI) {
287 int AddrOffset = X86II::getMemoryOperandIdx(MI->getDesc());
288 assert(AddrOffset >= 0 && "Expected a memory operand");
289 return AddrOffset;
290}
291
293 int AddrOffset = getAddrOffset(MI);
294 return MI->getOperand(AddrOffset + X86::AddrBaseReg);
295}
296
298 int AddrOffset = getAddrOffset(MI);
299 return MI->getOperand(AddrOffset + X86::AddrDisp);
300}
301
302// Relevant addressing modes contain only base register and immediate
303// displacement or frameindex and immediate displacement.
304// TODO: Consider expanding to other addressing modes in the future
306 int AddrOffset = getAddrOffset(MI);
308 const MachineOperand &Disp = getDispOperand(MI);
309 const MachineOperand &Scale = MI->getOperand(AddrOffset + X86::AddrScaleAmt);
310 const MachineOperand &Index = MI->getOperand(AddrOffset + X86::AddrIndexReg);
311 const MachineOperand &Segment = MI->getOperand(AddrOffset + X86::AddrSegmentReg);
312
313 if (!((Base.isReg() && Base.getReg() != X86::NoRegister) || Base.isFI()))
314 return false;
315 if (!Disp.isImm())
316 return false;
317 if (Scale.getImm() != 1)
318 return false;
319 if (!(Index.isReg() && Index.getReg() == X86::NoRegister))
320 return false;
321 if (!(Segment.isReg() && Segment.getReg() == X86::NoRegister))
322 return false;
323 return true;
324}
325
326// Collect potentially blocking stores.
327// Limit the number of instructions backwards we want to inspect
328// since the effect of store block won't be visible if the store
329// and load instructions have enough instructions in between to
330// keep the core busy.
332findPotentialBlockers(MachineInstr *LoadInst, unsigned InspectionLimit) {
333 SmallVector<MachineInstr *, 2> PotentialBlockers;
334 unsigned BlockCount = 0;
335 for (auto PBInst = std::next(MachineBasicBlock::reverse_iterator(LoadInst)),
336 E = LoadInst->getParent()->rend();
337 PBInst != E; ++PBInst) {
338 if (PBInst->isMetaInstruction())
339 continue;
340 BlockCount++;
341 if (BlockCount >= InspectionLimit)
342 break;
343 MachineInstr &MI = *PBInst;
344 if (MI.getDesc().isCall())
345 return PotentialBlockers;
346 PotentialBlockers.push_back(&MI);
347 }
348 // If we didn't get to the instructions limit try predecessing blocks.
349 // Ideally we should traverse the predecessor blocks in depth with some
350 // coloring algorithm, but for now let's just look at the first order
351 // predecessors.
352 if (BlockCount < InspectionLimit) {
354 int LimitLeft = InspectionLimit - BlockCount;
355 for (MachineBasicBlock *PMBB : MBB->predecessors()) {
356 int PredCount = 0;
357 for (MachineInstr &PBInst : llvm::reverse(*PMBB)) {
358 if (PBInst.isMetaInstruction())
359 continue;
360 PredCount++;
361 if (PredCount >= LimitLeft)
362 break;
363 if (PBInst.getDesc().isCall())
364 break;
365 PotentialBlockers.push_back(&PBInst);
366 }
367 }
368 }
369 return PotentialBlockers;
370}
371
372void X86AvoidSFBImpl::buildCopy(MachineInstr *LoadInst, unsigned NLoadOpcode,
373 int64_t LoadDisp, MachineInstr *StoreInst,
374 unsigned NStoreOpcode, int64_t StoreDisp,
375 unsigned Size, int64_t Offset) {
376 MachineOperand &LoadBase = getBaseOperand(LoadInst);
377 MachineOperand &StoreBase = getBaseOperand(StoreInst);
378 MachineBasicBlock *MBB = LoadInst->getParent();
379 MachineMemOperand *LMMO = *LoadInst->memoperands_begin();
380 MachineMemOperand *SMMO = *StoreInst->memoperands_begin();
381
382 Register Reg1 =
383 MRI->createVirtualRegister(TII->getRegClass(TII->get(NLoadOpcode), 0));
384 MachineInstr *NewLoad =
385 BuildMI(*MBB, LoadInst, LoadInst->getDebugLoc(), TII->get(NLoadOpcode),
386 Reg1)
387 .add(LoadBase)
388 .addImm(1)
389 .addReg(X86::NoRegister)
390 .addImm(LoadDisp)
391 .addReg(X86::NoRegister)
394 if (LoadBase.isReg())
395 getBaseOperand(NewLoad).setIsKill(false);
396 LLVM_DEBUG(NewLoad->dump());
397 // If the load and store are consecutive, use the loadInst location to
398 // reduce register pressure.
399 MachineInstr *StInst = StoreInst;
400 auto PrevInstrIt = prev_nodbg(MachineBasicBlock::instr_iterator(StoreInst),
401 MBB->instr_begin());
402 if (PrevInstrIt.getNodePtr() == LoadInst)
403 StInst = LoadInst;
404 MachineInstr *NewStore =
405 BuildMI(*MBB, StInst, StInst->getDebugLoc(), TII->get(NStoreOpcode))
406 .add(StoreBase)
407 .addImm(1)
408 .addReg(X86::NoRegister)
409 .addImm(StoreDisp)
410 .addReg(X86::NoRegister)
411 .addReg(Reg1)
414 if (StoreBase.isReg())
415 getBaseOperand(NewStore).setIsKill(false);
416 MachineOperand &StoreSrcVReg = StoreInst->getOperand(X86::AddrNumOperands);
417 assert(StoreSrcVReg.isReg() && "Expected virtual register");
418 NewStore->getOperand(X86::AddrNumOperands).setIsKill(StoreSrcVReg.isKill());
419 LLVM_DEBUG(NewStore->dump());
420}
421
422void X86AvoidSFBImpl::buildCopies(int Size, MachineInstr *LoadInst,
423 int64_t LdDispImm, MachineInstr *StoreInst,
424 int64_t StDispImm, int64_t Offset) {
425 int LdDisp = LdDispImm;
426 int StDisp = StDispImm;
427 while (Size > 0) {
428 if ((Size - MOV128SZ >= 0) && isYMMLoadOpcode(LoadInst->getOpcode())) {
429 Size = Size - MOV128SZ;
430 buildCopy(LoadInst, getYMMtoXMMLoadOpcode(LoadInst->getOpcode()), LdDisp,
431 StoreInst, getYMMtoXMMStoreOpcode(StoreInst->getOpcode()),
432 StDisp, MOV128SZ, Offset);
433 LdDisp += MOV128SZ;
434 StDisp += MOV128SZ;
435 Offset += MOV128SZ;
436 continue;
437 }
438 if (Size - MOV64SZ >= 0) {
439 Size = Size - MOV64SZ;
440 buildCopy(LoadInst, X86::MOV64rm, LdDisp, StoreInst, X86::MOV64mr, StDisp,
441 MOV64SZ, Offset);
442 LdDisp += MOV64SZ;
443 StDisp += MOV64SZ;
444 Offset += MOV64SZ;
445 continue;
446 }
447 if (Size - MOV32SZ >= 0) {
448 Size = Size - MOV32SZ;
449 buildCopy(LoadInst, X86::MOV32rm, LdDisp, StoreInst, X86::MOV32mr, StDisp,
450 MOV32SZ, Offset);
451 LdDisp += MOV32SZ;
452 StDisp += MOV32SZ;
453 Offset += MOV32SZ;
454 continue;
455 }
456 if (Size - MOV16SZ >= 0) {
457 Size = Size - MOV16SZ;
458 buildCopy(LoadInst, X86::MOV16rm, LdDisp, StoreInst, X86::MOV16mr, StDisp,
459 MOV16SZ, Offset);
460 LdDisp += MOV16SZ;
461 StDisp += MOV16SZ;
462 Offset += MOV16SZ;
463 continue;
464 }
465 if (Size - MOV8SZ >= 0) {
466 Size = Size - MOV8SZ;
467 buildCopy(LoadInst, X86::MOV8rm, LdDisp, StoreInst, X86::MOV8mr, StDisp,
468 MOV8SZ, Offset);
469 LdDisp += MOV8SZ;
470 StDisp += MOV8SZ;
471 Offset += MOV8SZ;
472 continue;
473 }
474 }
475 assert(Size == 0 && "Wrong size division");
476}
477
481 auto *StorePrevNonDbgInstr =
483 LoadInst->getParent()->instr_begin())
484 .getNodePtr();
485 if (LoadBase.isReg()) {
486 MachineInstr *LastLoad = LoadInst->getPrevNode();
487 // If the original load and store to xmm/ymm were consecutive
488 // then the partial copies were also created in
489 // a consecutive order to reduce register pressure,
490 // and the location of the last load is before the last store.
491 if (StorePrevNonDbgInstr == LoadInst)
492 LastLoad = LoadInst->getPrevNode()->getPrevNode();
493 getBaseOperand(LastLoad).setIsKill(LoadBase.isKill());
494 }
495 if (StoreBase.isReg()) {
496 MachineInstr *StInst = StoreInst;
497 if (StorePrevNonDbgInstr == LoadInst)
498 StInst = LoadInst;
499 getBaseOperand(StInst->getPrevNode()).setIsKill(StoreBase.isKill());
500 }
501}
502
503bool X86AvoidSFBImpl::alias(const MachineMemOperand &Op1,
504 const MachineMemOperand &Op2) const {
505 if (!Op1.getValue() || !Op2.getValue())
506 return true;
507
508 int64_t MinOffset = std::min(Op1.getOffset(), Op2.getOffset());
509 int64_t Overlapa = Op1.getSize().getValue() + Op1.getOffset() - MinOffset;
510 int64_t Overlapb = Op2.getSize().getValue() + Op2.getOffset() - MinOffset;
511
512 return !AA->isNoAlias(
513 MemoryLocation(Op1.getValue(), Overlapa, Op1.getAAInfo()),
514 MemoryLocation(Op2.getValue(), Overlapb, Op2.getAAInfo()));
515}
516
517void X86AvoidSFBImpl::findPotentiallylBlockedCopies(MachineFunction &MF) {
518 for (auto &MBB : MF)
519 for (auto &MI : MBB) {
520 if (!isPotentialBlockedMemCpyLd(MI.getOpcode()))
521 continue;
522 Register DefVR = MI.getOperand(0).getReg();
523 if (!MRI->hasOneNonDBGUse(DefVR))
524 continue;
525 for (MachineOperand &StoreMO :
527 MachineInstr &StoreMI = *StoreMO.getParent();
528 // Skip cases where the memcpy may overlap.
529 if (StoreMI.getParent() == MI.getParent() &&
530 isPotentialBlockedMemCpyPair(MI.getOpcode(), StoreMI.getOpcode()) &&
532 isRelevantAddressingMode(&StoreMI) &&
533 MI.hasOneMemOperand() && StoreMI.hasOneMemOperand()) {
534 // Don't split volatile or atomic accesses.
535 const MachineMemOperand *LMMO = *MI.memoperands_begin();
536 const MachineMemOperand *SMMO = *StoreMI.memoperands_begin();
537 if (LMMO->isVolatile() || LMMO->isAtomic() || SMMO->isVolatile() ||
538 SMMO->isAtomic())
539 continue;
540 if (!alias(*LMMO, *SMMO))
541 BlockedLoadsStoresPairs.push_back(std::make_pair(&MI, &StoreMI));
542 }
543 }
544 }
545}
546
547unsigned X86AvoidSFBImpl::getRegSizeInBytes(MachineInstr *LoadInst) {
548 const auto *TRC = TII->getRegClass(TII->get(LoadInst->getOpcode()), 0);
549 return TRI->getRegSizeInBits(*TRC) / 8;
550}
551
552void X86AvoidSFBImpl::breakBlockedCopies(
553 MachineInstr *LoadInst, MachineInstr *StoreInst,
554 const DisplacementSizeMap &BlockingStoresDispSizeMap) {
555 int64_t LdDispImm = getDispOperand(LoadInst).getImm();
556 int64_t StDispImm = getDispOperand(StoreInst).getImm();
557 int64_t Offset = 0;
558
559 int64_t LdDisp1 = LdDispImm;
560 int64_t LdDisp2 = 0;
561 int64_t StDisp1 = StDispImm;
562 int64_t StDisp2 = 0;
563 unsigned Size1 = 0;
564 unsigned Size2 = 0;
565 int64_t LdStDelta = StDispImm - LdDispImm;
566
567 for (auto DispSizePair : BlockingStoresDispSizeMap) {
568 LdDisp2 = DispSizePair.first;
569 StDisp2 = DispSizePair.first + LdStDelta;
570 Size2 = DispSizePair.second;
571 // Avoid copying overlapping areas.
572 if (LdDisp2 < LdDisp1) {
573 int OverlapDelta = LdDisp1 - LdDisp2;
574 LdDisp2 += OverlapDelta;
575 StDisp2 += OverlapDelta;
576 Size2 -= OverlapDelta;
577 }
578 Size1 = LdDisp2 - LdDisp1;
579
580 // Build a copy for the point until the current blocking store's
581 // displacement.
582 buildCopies(Size1, LoadInst, LdDisp1, StoreInst, StDisp1, Offset);
583 // Build a copy for the current blocking store.
584 buildCopies(Size2, LoadInst, LdDisp2, StoreInst, StDisp2, Offset + Size1);
585 LdDisp1 = LdDisp2 + Size2;
586 StDisp1 = StDisp2 + Size2;
587 Offset += Size1 + Size2;
588 }
589 unsigned Size3 = (LdDispImm + getRegSizeInBytes(LoadInst)) - LdDisp1;
590 buildCopies(Size3, LoadInst, LdDisp1, StoreInst, StDisp1, Offset);
591}
592
595 const MachineOperand &LoadBase = getBaseOperand(LoadInst);
596 const MachineOperand &StoreBase = getBaseOperand(StoreInst);
597 if (LoadBase.isReg() != StoreBase.isReg())
598 return false;
599 if (LoadBase.isReg())
600 return LoadBase.getReg() == StoreBase.getReg();
601 return LoadBase.getIndex() == StoreBase.getIndex();
602}
603
604static bool isBlockingStore(int64_t LoadDispImm, unsigned LoadSize,
605 int64_t StoreDispImm, unsigned StoreSize) {
606 return ((StoreDispImm >= LoadDispImm) &&
607 (StoreDispImm <= LoadDispImm + (LoadSize - StoreSize)));
608}
609
610// Keep track of all stores blocking a load
611static void
612updateBlockingStoresDispSizeMap(DisplacementSizeMap &BlockingStoresDispSizeMap,
613 int64_t DispImm, unsigned Size) {
614 auto [It, Inserted] = BlockingStoresDispSizeMap.try_emplace(DispImm, Size);
615 // Choose the smallest blocking store starting at this displacement.
616 if (!Inserted && It->second > Size)
617 It->second = Size;
618}
619
620// Remove blocking stores contained in each other.
621static void
622removeRedundantBlockingStores(DisplacementSizeMap &BlockingStoresDispSizeMap) {
623 if (BlockingStoresDispSizeMap.size() <= 1)
624 return;
625
627 for (auto DispSizePair : BlockingStoresDispSizeMap) {
628 int64_t CurrDisp = DispSizePair.first;
629 unsigned CurrSize = DispSizePair.second;
630 while (DispSizeStack.size()) {
631 int64_t PrevDisp = DispSizeStack.back().first;
632 unsigned PrevSize = DispSizeStack.back().second;
633 if (CurrDisp + CurrSize > PrevDisp + PrevSize)
634 break;
635 DispSizeStack.pop_back();
636 }
637 DispSizeStack.push_back(DispSizePair);
638 }
639 BlockingStoresDispSizeMap.clear();
640 for (auto Disp : DispSizeStack)
641 BlockingStoresDispSizeMap.insert(Disp);
642}
643
644bool X86AvoidSFBImpl::runOnMachineFunction(MachineFunction &MF) {
645 bool Changed = false;
646
647 const X86Subtarget &ST = MF.getSubtarget<X86Subtarget>();
648 if (ST.getCLOpts().disable_avoid_SFB || !ST.is64Bit())
649 return false;
650
651 MRI = &MF.getRegInfo();
652 assert(MRI->isSSA() && "Expected MIR to be in SSA form");
653 TII = MF.getSubtarget<X86Subtarget>().getInstrInfo();
654 TRI = MF.getSubtarget<X86Subtarget>().getRegisterInfo();
655 LLVM_DEBUG(dbgs() << "Start X86AvoidStoreForwardBlocks\n";);
656 // Look for a load then a store to XMM/YMM which look like a memcpy
657 findPotentiallylBlockedCopies(MF);
658
659 for (auto LoadStoreInstPair : BlockedLoadsStoresPairs) {
660 MachineInstr *LoadInst = LoadStoreInstPair.first;
661 int64_t LdDispImm = getDispOperand(LoadInst).getImm();
662 DisplacementSizeMap BlockingStoresDispSizeMap;
663
664 SmallVector<MachineInstr *, 2> PotentialBlockers =
665 findPotentialBlockers(LoadInst, ST.getCLOpts().sfb_inspection_limit);
666 for (auto *PBInst : PotentialBlockers) {
667 if (!isPotentialBlockingStoreInst(PBInst->getOpcode(),
668 LoadInst->getOpcode()) ||
669 !isRelevantAddressingMode(PBInst) || !PBInst->hasOneMemOperand())
670 continue;
671 int64_t PBstDispImm = getDispOperand(PBInst).getImm();
672 unsigned PBstSize = (*PBInst->memoperands_begin())->getSize().getValue();
673 // This check doesn't cover all cases, but it will suffice for now.
674 // TODO: take branch probability into consideration, if the blocking
675 // store is in an unreached block, breaking the memcopy could lose
676 // performance.
677 if (hasSameBaseOpValue(LoadInst, PBInst) &&
678 isBlockingStore(LdDispImm, getRegSizeInBytes(LoadInst), PBstDispImm,
679 PBstSize))
680 updateBlockingStoresDispSizeMap(BlockingStoresDispSizeMap, PBstDispImm,
681 PBstSize);
682 }
683
684 if (BlockingStoresDispSizeMap.empty())
685 continue;
686
687 // We found a store forward block, break the memcpy's load and store
688 // into smaller copies such that each smaller store that was causing
689 // a store block would now be copied separately.
690 MachineInstr *StoreInst = LoadStoreInstPair.second;
691 LLVM_DEBUG(dbgs() << "Blocked load and store instructions: \n");
692 LLVM_DEBUG(LoadInst->dump());
693 LLVM_DEBUG(StoreInst->dump());
694 LLVM_DEBUG(dbgs() << "Replaced with:\n");
695 removeRedundantBlockingStores(BlockingStoresDispSizeMap);
696 breakBlockedCopies(LoadInst, StoreInst, BlockingStoresDispSizeMap);
697 updateKillStatus(LoadInst, StoreInst);
698 ForRemoval.push_back(LoadInst);
699 ForRemoval.push_back(StoreInst);
700 }
701 for (auto *RemovedInst : ForRemoval) {
702 RemovedInst->eraseFromParent();
703 }
704 ForRemoval.clear();
705 BlockedLoadsStoresPairs.clear();
706 LLVM_DEBUG(dbgs() << "End X86AvoidStoreForwardBlocks\n";);
707
708 return Changed;
709}
710
711bool X86AvoidSFBLegacy::runOnMachineFunction(MachineFunction &MF) {
712 if (skipFunction(MF.getFunction()))
713 return false;
714 AliasAnalysis *AA = &getAnalysis<AAResultsWrapperPass>().getAAResults();
715 X86AvoidSFBImpl Impl(AA);
716 return Impl.runOnMachineFunction(MF);
717}
718
719PreservedAnalyses
724 .getManager()
725 .getResult<AAManager>(MF.getFunction());
726 X86AvoidSFBImpl Impl(AA);
727 bool Changed = Impl.runOnMachineFunction(MF);
730}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock & MBB
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
#define DEBUG_TYPE
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SmallVector< MachineInstr *, 2 > findPotentialBlockers(MachineInstr *LoadInst, unsigned InspectionLimit)
static unsigned getYMMtoXMMLoadOpcode(unsigned LoadOpcode)
static bool isPotentialBlockedMemCpyLd(unsigned Opcode)
static bool isPotentialBlockedMemCpyPair(unsigned LdOpcode, unsigned StOpcode)
static bool isPotentialBlockingStoreInst(unsigned Opcode, unsigned LoadOpcode)
static const int MOV64SZ
static const int MOV8SZ
static bool isXMMLoadOpcode(unsigned Opcode)
static int getAddrOffset(const MachineInstr *MI)
static bool isBlockingStore(int64_t LoadDispImm, unsigned LoadSize, int64_t StoreDispImm, unsigned StoreSize)
static bool isRelevantAddressingMode(MachineInstr *MI)
static void removeRedundantBlockingStores(DisplacementSizeMap &BlockingStoresDispSizeMap)
static const int MOV16SZ
static bool hasSameBaseOpValue(MachineInstr *LoadInst, MachineInstr *StoreInst)
static void updateBlockingStoresDispSizeMap(DisplacementSizeMap &BlockingStoresDispSizeMap, int64_t DispImm, unsigned Size)
static MachineOperand & getBaseOperand(MachineInstr *MI)
static unsigned getYMMtoXMMStoreOpcode(unsigned StoreOpcode)
static void updateKillStatus(MachineInstr *LoadInst, MachineInstr *StoreInst)
static const int MOV32SZ
static MachineOperand & getDispOperand(MachineInstr *MI)
static bool isYMMLoadOpcode(unsigned Opcode)
static const int MOV128SZ
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
bool isNoAlias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A trivial helper function to check to see if the specified pointers are no-alias.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
An instruction for reading from memory.
TypeSize getValue() const
Instructions::iterator instr_iterator
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void dump() const
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
LocationSize getSize() const
Return the size in bytes of the memory reference.
bool isAtomic() const
Returns true if this operation has an atomic ordering requirement of unordered or higher,...
AAMDNodes getAAInfo() const
Return the AA tags for the memory reference.
const Value * getValue() const
Return the base address of the memory access.
int64_t getOffset() const
For normal values, this is a byte offset added to the base address.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
const ParentTy * getParent() const
Definition ilist_node.h:34
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
Definition Attributor.h:165
int getMemoryOperandIdx(const MCInstrDesc &Desc)
@ AddrNumOperands
Definition X86BaseInfo.h:37
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
FunctionPass * createX86AvoidStoreForwardingBlocksLegacyPass()
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
IterT prev_nodbg(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It, then continue decrementing it while it points to a debug instruction.