55#define DEBUG_TYPE "x86-avoid-sfb"
59using DisplacementSizeMap = std::map<int64_t, unsigned>;
61class X86AvoidSFBImpl {
71 BlockedLoadsStoresPairs;
82 const DisplacementSizeMap &BlockingStoresDispSizeMap);
89 int64_t StoreDisp,
unsigned Size, int64_t
Offset);
102 return "X86 Avoid Store Forwarding Blocks";
116char X86AvoidSFBLegacy::ID = 0;
125 return new X86AvoidSFBLegacy();
129 return Opcode == X86::MOVUPSrm || Opcode == X86::MOVAPSrm ||
130 Opcode == X86::VMOVUPSrm || Opcode == X86::VMOVAPSrm ||
131 Opcode == X86::VMOVUPDrm || Opcode == X86::VMOVAPDrm ||
132 Opcode == X86::VMOVDQUrm || Opcode == X86::VMOVDQArm ||
133 Opcode == X86::VMOVUPSZ128rm || Opcode == X86::VMOVAPSZ128rm ||
134 Opcode == X86::VMOVUPDZ128rm || Opcode == X86::VMOVAPDZ128rm ||
135 Opcode == X86::VMOVDQU64Z128rm || Opcode == X86::VMOVDQA64Z128rm ||
136 Opcode == X86::VMOVDQU32Z128rm || Opcode == X86::VMOVDQA32Z128rm;
139 return Opcode == X86::VMOVUPSYrm || Opcode == X86::VMOVAPSYrm ||
140 Opcode == X86::VMOVUPDYrm || Opcode == X86::VMOVAPDYrm ||
141 Opcode == X86::VMOVDQUYrm || Opcode == X86::VMOVDQAYrm ||
142 Opcode == X86::VMOVUPSZ256rm || Opcode == X86::VMOVAPSZ256rm ||
143 Opcode == X86::VMOVUPDZ256rm || Opcode == X86::VMOVAPDZ256rm ||
144 Opcode == X86::VMOVDQU64Z256rm || Opcode == X86::VMOVDQA64Z256rm ||
145 Opcode == X86::VMOVDQU32Z256rm || Opcode == X86::VMOVDQA32Z256rm;
156 return StOpcode == X86::MOVUPSmr || StOpcode == X86::MOVAPSmr;
159 return StOpcode == X86::VMOVUPSmr || StOpcode == X86::VMOVAPSmr;
162 return StOpcode == X86::VMOVUPDmr || StOpcode == X86::VMOVAPDmr;
165 return StOpcode == X86::VMOVDQUmr || StOpcode == X86::VMOVDQAmr;
166 case X86::VMOVUPSZ128rm:
167 case X86::VMOVAPSZ128rm:
168 return StOpcode == X86::VMOVUPSZ128mr || StOpcode == X86::VMOVAPSZ128mr;
169 case X86::VMOVUPDZ128rm:
170 case X86::VMOVAPDZ128rm:
171 return StOpcode == X86::VMOVUPDZ128mr || StOpcode == X86::VMOVAPDZ128mr;
172 case X86::VMOVUPSYrm:
173 case X86::VMOVAPSYrm:
174 return StOpcode == X86::VMOVUPSYmr || StOpcode == X86::VMOVAPSYmr;
175 case X86::VMOVUPDYrm:
176 case X86::VMOVAPDYrm:
177 return StOpcode == X86::VMOVUPDYmr || StOpcode == X86::VMOVAPDYmr;
178 case X86::VMOVDQUYrm:
179 case X86::VMOVDQAYrm:
180 return StOpcode == X86::VMOVDQUYmr || StOpcode == X86::VMOVDQAYmr;
181 case X86::VMOVUPSZ256rm:
182 case X86::VMOVAPSZ256rm:
183 return StOpcode == X86::VMOVUPSZ256mr || StOpcode == X86::VMOVAPSZ256mr;
184 case X86::VMOVUPDZ256rm:
185 case X86::VMOVAPDZ256rm:
186 return StOpcode == X86::VMOVUPDZ256mr || StOpcode == X86::VMOVAPDZ256mr;
187 case X86::VMOVDQU64Z128rm:
188 case X86::VMOVDQA64Z128rm:
189 return StOpcode == X86::VMOVDQU64Z128mr || StOpcode == X86::VMOVDQA64Z128mr;
190 case X86::VMOVDQU32Z128rm:
191 case X86::VMOVDQA32Z128rm:
192 return StOpcode == X86::VMOVDQU32Z128mr || StOpcode == X86::VMOVDQA32Z128mr;
193 case X86::VMOVDQU64Z256rm:
194 case X86::VMOVDQA64Z256rm:
195 return StOpcode == X86::VMOVDQU64Z256mr || StOpcode == X86::VMOVDQA64Z256mr;
196 case X86::VMOVDQU32Z256rm:
197 case X86::VMOVDQA32Z256rm:
198 return StOpcode == X86::VMOVDQU32Z256mr || StOpcode == X86::VMOVDQA32Z256mr;
206 PBlock |= Opcode == X86::MOV64mr || Opcode == X86::MOV64mi32 ||
207 Opcode == X86::MOV32mr || Opcode == X86::MOV32mi ||
208 Opcode == X86::MOV16mr || Opcode == X86::MOV16mi ||
209 Opcode == X86::MOV8mr || Opcode == X86::MOV8mi;
211 PBlock |= Opcode == X86::VMOVUPSmr || Opcode == X86::VMOVAPSmr ||
212 Opcode == X86::VMOVUPDmr || Opcode == X86::VMOVAPDmr ||
213 Opcode == X86::VMOVDQUmr || Opcode == X86::VMOVDQAmr ||
214 Opcode == X86::VMOVUPSZ128mr || Opcode == X86::VMOVAPSZ128mr ||
215 Opcode == X86::VMOVUPDZ128mr || Opcode == X86::VMOVAPDZ128mr ||
216 Opcode == X86::VMOVDQU64Z128mr ||
217 Opcode == X86::VMOVDQA64Z128mr ||
218 Opcode == X86::VMOVDQU32Z128mr || Opcode == X86::VMOVDQA32Z128mr;
229 switch (LoadOpcode) {
230 case X86::VMOVUPSYrm:
231 case X86::VMOVAPSYrm:
232 return X86::VMOVUPSrm;
233 case X86::VMOVUPDYrm:
234 case X86::VMOVAPDYrm:
235 return X86::VMOVUPDrm;
236 case X86::VMOVDQUYrm:
237 case X86::VMOVDQAYrm:
238 return X86::VMOVDQUrm;
239 case X86::VMOVUPSZ256rm:
240 case X86::VMOVAPSZ256rm:
241 return X86::VMOVUPSZ128rm;
242 case X86::VMOVUPDZ256rm:
243 case X86::VMOVAPDZ256rm:
244 return X86::VMOVUPDZ128rm;
245 case X86::VMOVDQU64Z256rm:
246 case X86::VMOVDQA64Z256rm:
247 return X86::VMOVDQU64Z128rm;
248 case X86::VMOVDQU32Z256rm:
249 case X86::VMOVDQA32Z256rm:
250 return X86::VMOVDQU32Z128rm;
258 switch (StoreOpcode) {
259 case X86::VMOVUPSYmr:
260 case X86::VMOVAPSYmr:
261 return X86::VMOVUPSmr;
262 case X86::VMOVUPDYmr:
263 case X86::VMOVAPDYmr:
264 return X86::VMOVUPDmr;
265 case X86::VMOVDQUYmr:
266 case X86::VMOVDQAYmr:
267 return X86::VMOVDQUmr;
268 case X86::VMOVUPSZ256mr:
269 case X86::VMOVAPSZ256mr:
270 return X86::VMOVUPSZ128mr;
271 case X86::VMOVUPDZ256mr:
272 case X86::VMOVAPDZ256mr:
273 return X86::VMOVUPDZ128mr;
274 case X86::VMOVDQU64Z256mr:
275 case X86::VMOVDQA64Z256mr:
276 return X86::VMOVDQU64Z128mr;
277 case X86::VMOVDQU32Z256mr:
278 case X86::VMOVDQA32Z256mr:
279 return X86::VMOVDQU32Z128mr;
288 assert(AddrOffset >= 0 &&
"Expected a memory operand");
313 if (!((
Base.isReg() &&
Base.getReg() != X86::NoRegister) ||
Base.isFI()))
319 if (!(Index.isReg() && Index.getReg() == X86::NoRegister))
321 if (!(Segment.
isReg() && Segment.
getReg() == X86::NoRegister))
334 unsigned BlockCount = 0;
337 PBInst !=
E; ++PBInst) {
338 if (PBInst->isMetaInstruction())
341 if (BlockCount >= InspectionLimit)
344 if (
MI.getDesc().isCall())
345 return PotentialBlockers;
352 if (BlockCount < InspectionLimit) {
354 int LimitLeft = InspectionLimit - BlockCount;
358 if (PBInst.isMetaInstruction())
361 if (PredCount >= LimitLeft)
363 if (PBInst.getDesc().isCall())
369 return PotentialBlockers;
374 unsigned NStoreOpcode, int64_t StoreDisp,
384 MachineInstr *NewLoad =
394 if (LoadBase.
isReg())
399 MachineInstr *StInst = StoreInst;
402 if (PrevInstrIt.getNodePtr() == LoadInst)
404 MachineInstr *NewStore =
414 if (StoreBase.
isReg())
417 assert(StoreSrcVReg.
isReg() &&
"Expected virtual register");
422void X86AvoidSFBImpl::buildCopies(
int Size, MachineInstr *LoadInst,
423 int64_t LdDispImm, MachineInstr *StoreInst,
424 int64_t StDispImm, int64_t
Offset) {
425 int LdDisp = LdDispImm;
426 int StDisp = StDispImm;
440 buildCopy(LoadInst, X86::MOV64rm, LdDisp, StoreInst, X86::MOV64mr, StDisp,
449 buildCopy(LoadInst, X86::MOV32rm, LdDisp, StoreInst, X86::MOV32mr, StDisp,
458 buildCopy(LoadInst, X86::MOV16rm, LdDisp, StoreInst, X86::MOV16mr, StDisp,
467 buildCopy(LoadInst, X86::MOV8rm, LdDisp, StoreInst, X86::MOV8mr, StDisp,
481 auto *StorePrevNonDbgInstr =
485 if (LoadBase.
isReg()) {
491 if (StorePrevNonDbgInstr ==
LoadInst)
495 if (StoreBase.
isReg()) {
497 if (StorePrevNonDbgInstr ==
LoadInst)
503bool X86AvoidSFBImpl::alias(
const MachineMemOperand &Op1,
504 const MachineMemOperand &Op2)
const {
517void X86AvoidSFBImpl::findPotentiallylBlockedCopies(
MachineFunction &MF) {
519 for (
auto &
MI :
MBB) {
525 for (MachineOperand &StoreMO :
527 MachineInstr &StoreMI = *StoreMO.getParent();
535 const MachineMemOperand *LMMO = *
MI.memoperands_begin();
540 if (!alias(*LMMO, *SMMO))
541 BlockedLoadsStoresPairs.push_back(std::make_pair(&
MI, &StoreMI));
547unsigned X86AvoidSFBImpl::getRegSizeInBytes(MachineInstr *LoadInst) {
548 const auto *TRC =
TII->getRegClass(
TII->get(LoadInst->
getOpcode()), 0);
549 return TRI->getRegSizeInBits(*TRC) / 8;
552void X86AvoidSFBImpl::breakBlockedCopies(
553 MachineInstr *LoadInst, MachineInstr *StoreInst,
554 const DisplacementSizeMap &BlockingStoresDispSizeMap) {
559 int64_t LdDisp1 = LdDispImm;
561 int64_t StDisp1 = StDispImm;
565 int64_t LdStDelta = StDispImm - LdDispImm;
567 for (
auto DispSizePair : BlockingStoresDispSizeMap) {
568 LdDisp2 = DispSizePair.first;
569 StDisp2 = DispSizePair.first + LdStDelta;
570 Size2 = DispSizePair.second;
572 if (LdDisp2 < LdDisp1) {
573 int OverlapDelta = LdDisp1 - LdDisp2;
574 LdDisp2 += OverlapDelta;
575 StDisp2 += OverlapDelta;
576 Size2 -= OverlapDelta;
578 Size1 = LdDisp2 - LdDisp1;
582 buildCopies(Size1, LoadInst, LdDisp1, StoreInst, StDisp1,
Offset);
584 buildCopies(Size2, LoadInst, LdDisp2, StoreInst, StDisp2,
Offset + Size1);
585 LdDisp1 = LdDisp2 + Size2;
586 StDisp1 = StDisp2 + Size2;
589 unsigned Size3 = (LdDispImm + getRegSizeInBytes(LoadInst)) - LdDisp1;
590 buildCopies(Size3, LoadInst, LdDisp1, StoreInst, StDisp1,
Offset);
599 if (LoadBase.
isReg())
605 int64_t StoreDispImm,
unsigned StoreSize) {
606 return ((StoreDispImm >= LoadDispImm) &&
607 (StoreDispImm <= LoadDispImm + (LoadSize - StoreSize)));
613 int64_t DispImm,
unsigned Size) {
614 auto [It, Inserted] = BlockingStoresDispSizeMap.try_emplace(DispImm,
Size);
616 if (!Inserted && It->second >
Size)
623 if (BlockingStoresDispSizeMap.size() <= 1)
627 for (
auto DispSizePair : BlockingStoresDispSizeMap) {
628 int64_t CurrDisp = DispSizePair.first;
629 unsigned CurrSize = DispSizePair.second;
630 while (DispSizeStack.
size()) {
631 int64_t PrevDisp = DispSizeStack.
back().first;
632 unsigned PrevSize = DispSizeStack.
back().second;
633 if (CurrDisp + CurrSize > PrevDisp + PrevSize)
639 BlockingStoresDispSizeMap.
clear();
640 for (
auto Disp : DispSizeStack)
641 BlockingStoresDispSizeMap.insert(Disp);
648 if (
ST.getCLOpts().disable_avoid_SFB || !
ST.is64Bit())
652 assert(MRI->
isSSA() &&
"Expected MIR to be in SSA form");
657 findPotentiallylBlockedCopies(MF);
659 for (
auto LoadStoreInstPair : BlockedLoadsStoresPairs) {
660 MachineInstr *LoadInst = LoadStoreInstPair.first;
662 DisplacementSizeMap BlockingStoresDispSizeMap;
664 SmallVector<MachineInstr *, 2> PotentialBlockers =
666 for (
auto *PBInst : PotentialBlockers) {
672 unsigned PBstSize = (*PBInst->memoperands_begin())->getSize().getValue();
684 if (BlockingStoresDispSizeMap.empty())
690 MachineInstr *StoreInst = LoadStoreInstPair.second;
696 breakBlockedCopies(LoadInst, StoreInst, BlockingStoresDispSizeMap);
701 for (
auto *RemovedInst : ForRemoval) {
702 RemovedInst->eraseFromParent();
705 BlockedLoadsStoresPairs.clear();
714 AliasAnalysis *AA = &getAnalysis<AAResultsWrapperPass>().getAAResults();
715 X86AvoidSFBImpl Impl(AA);
716 return Impl.runOnMachineFunction(MF);
726 X86AvoidSFBImpl Impl(
AA);
727 bool Changed = Impl.runOnMachineFunction(MF);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
const HexagonInstrInfo * TII
Register const TargetRegisterInfo * TRI
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static SmallVector< MachineInstr *, 2 > findPotentialBlockers(MachineInstr *LoadInst, unsigned InspectionLimit)
static unsigned getYMMtoXMMLoadOpcode(unsigned LoadOpcode)
static bool isPotentialBlockedMemCpyLd(unsigned Opcode)
static bool isPotentialBlockedMemCpyPair(unsigned LdOpcode, unsigned StOpcode)
static bool isPotentialBlockingStoreInst(unsigned Opcode, unsigned LoadOpcode)
static bool isXMMLoadOpcode(unsigned Opcode)
static int getAddrOffset(const MachineInstr *MI)
static bool isBlockingStore(int64_t LoadDispImm, unsigned LoadSize, int64_t StoreDispImm, unsigned StoreSize)
static bool isRelevantAddressingMode(MachineInstr *MI)
static void removeRedundantBlockingStores(DisplacementSizeMap &BlockingStoresDispSizeMap)
static bool hasSameBaseOpValue(MachineInstr *LoadInst, MachineInstr *StoreInst)
static void updateBlockingStoresDispSizeMap(DisplacementSizeMap &BlockingStoresDispSizeMap, int64_t DispImm, unsigned Size)
static MachineOperand & getBaseOperand(MachineInstr *MI)
static unsigned getYMMtoXMMStoreOpcode(unsigned StoreOpcode)
static void updateKillStatus(MachineInstr *LoadInst, MachineInstr *StoreInst)
static MachineOperand & getDispOperand(MachineInstr *MI)
static bool isYMMLoadOpcode(unsigned Opcode)
static const int MOV128SZ
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
bool isNoAlias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A trivial helper function to check to see if the specified pointers are no-alias.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
FunctionPass class - This class is used to implement most global optimizations.
An instruction for reading from memory.
TypeSize getValue() const
instr_iterator instr_begin()
Instructions::iterator instr_iterator
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void dump() const
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
LocationSize getSize() const
Return the size in bytes of the memory reference.
bool isAtomic() const
Returns true if this operation has an atomic ordering requirement of unordered or higher,...
AAMDNodes getAAInfo() const
Return the AA tags for the memory reference.
const Value * getValue() const
Return the base address of the memory access.
int64_t getOffset() const
For normal values, this is a byte offset added to the base address.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
int getMemoryOperandIdx(const MCInstrDesc &Desc)
This is an optimization pass for GlobalISel generic memory operations.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
FunctionPass * createX86AvoidStoreForwardingBlocksLegacyPass()
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
auto reverse(ContainerTy &&C)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
IterT prev_nodbg(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It, then continue decrementing it while it points to a debug instruction.