LLVM 24.0.0git
SILoadStoreOptimizer.cpp
Go to the documentation of this file.
1//===- SILoadStoreOptimizer.cpp -------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This pass tries to fuse DS instructions with close by immediate offsets.
10// This will fuse operations such as
11// ds_read_b32 v0, v2 offset:16
12// ds_read_b32 v1, v2 offset:32
13// ==>
14// ds_read2_b32 v[0:1], v2, offset0:4 offset1:8
15//
16// The same is done for certain SMEM and VMEM opcodes, e.g.:
17// s_buffer_load_dword s4, s[0:3], 4
18// s_buffer_load_dword s5, s[0:3], 8
19// ==>
20// s_buffer_load_dwordx2 s[4:5], s[0:3], 4
21//
22// This pass also tries to promote constant offset to the immediate by
23// adjusting the base. It tries to use a base from the nearby instructions that
24// allows it to have a 13bit constant offset and then promotes the 13bit offset
25// to the immediate.
26// E.g.
27// s_movk_i32 s0, 0x1800
28// v_add_co_u32_e32 v0, vcc, s0, v2
29// v_addc_co_u32_e32 v1, vcc, 0, v6, vcc
30//
31// s_movk_i32 s0, 0x1000
32// v_add_co_u32_e32 v5, vcc, s0, v2
33// v_addc_co_u32_e32 v6, vcc, 0, v6, vcc
34// global_load_dwordx2 v[5:6], v[5:6], off
35// global_load_dwordx2 v[0:1], v[0:1], off
36// =>
37// s_movk_i32 s0, 0x1000
38// v_add_co_u32_e32 v5, vcc, s0, v2
39// v_addc_co_u32_e32 v6, vcc, 0, v6, vcc
40// global_load_dwordx2 v[5:6], v[5:6], off
41// global_load_dwordx2 v[0:1], v[5:6], off offset:2048
42//
43// Future improvements:
44//
45// - This is currently missing stores of constants because loading
46// the constant into the data register is placed between the stores, although
47// this is arguably a scheduling problem.
48//
49// - Live interval recomputing seems inefficient. This currently only matches
50// one pair, and recomputes live intervals and moves on to the next pair. It
51// would be better to compute a list of all merges that need to occur.
52//
53// - With a list of instructions to process, we can also merge more. If a
54// cluster of loads have offsets that are too large to fit in the 8-bit
55// offsets, but are close enough to fit in the 8 bits, we can add to the base
56// pointer and use the new reduced offsets.
57//
58//===----------------------------------------------------------------------===//
59
61#include "AMDGPU.h"
62#include "GCNSubtarget.h"
63#include "SIDefines.h"
67
68using namespace llvm;
69
70#define DEBUG_TYPE "si-load-store-opt"
71
72namespace {
73enum InstClassEnum {
74 UNKNOWN,
75 DS_READ,
76 DS_WRITE,
77 S_BUFFER_LOAD_IMM,
78 S_BUFFER_LOAD_SGPR_IMM,
79 S_LOAD_IMM,
80 BUFFER_LOAD,
81 BUFFER_STORE,
82 MIMG,
83 TBUFFER_LOAD,
84 TBUFFER_STORE,
85 GLOBAL_LOAD_SADDR,
86 GLOBAL_STORE_SADDR,
87 FLAT_LOAD,
88 FLAT_STORE,
89 FLAT_LOAD_SADDR,
90 FLAT_STORE_SADDR,
91 GLOBAL_LOAD, // GLOBAL_LOAD/GLOBAL_STORE are never used as the InstClass of
92 GLOBAL_STORE // any CombineInfo, they are only ever returned by
93 // getCommonInstClass.
94};
95
96struct AddressRegs {
97 unsigned char NumVAddrs = 0;
98 bool SBase = false;
99 bool SRsrc = false;
100 bool SOffset = false;
101 bool SAddr = false;
102 bool VAddr = false;
103 bool Addr = false;
104 bool SSamp = false;
105};
106
107// GFX10 image_sample instructions can have 12 vaddrs + srsrc + ssamp.
108const unsigned MaxAddressRegs = 12 + 1 + 1;
109
110class SILoadStoreOptimizer {
111 struct CombineInfo {
113 unsigned EltSize;
114 unsigned Offset;
115 unsigned Width;
116 unsigned Format;
117 unsigned BaseOff;
118 unsigned DMask;
119 InstClassEnum InstClass;
120 unsigned CPol = 0;
121 bool GDS = false;
122 const TargetRegisterClass *DataRC;
123 bool UseST64;
124 int AddrIdx[MaxAddressRegs];
125 const MachineOperand *AddrReg[MaxAddressRegs];
126 unsigned NumAddresses;
127 unsigned Order;
128
129 bool hasSameBaseAddress(const CombineInfo &CI) {
130 if (NumAddresses != CI.NumAddresses)
131 return false;
132
133 const MachineInstr &MI = *CI.I;
134 for (unsigned i = 0; i < NumAddresses; i++) {
135 const MachineOperand &AddrRegNext = MI.getOperand(AddrIdx[i]);
136
137 if (AddrReg[i]->isImm() || AddrRegNext.isImm()) {
138 if (AddrReg[i]->isImm() != AddrRegNext.isImm() ||
139 AddrReg[i]->getImm() != AddrRegNext.getImm()) {
140 return false;
141 }
142 continue;
143 }
144
145 // Check same base pointer. Be careful of subregisters, which can occur
146 // with vectors of pointers.
147 if (AddrReg[i]->getReg() != AddrRegNext.getReg() ||
148 AddrReg[i]->getSubReg() != AddrRegNext.getSubReg()) {
149 return false;
150 }
151 }
152 return true;
153 }
154
155 bool hasMergeableAddress(const MachineRegisterInfo &MRI) {
156 for (unsigned i = 0; i < NumAddresses; ++i) {
157 const MachineOperand *AddrOp = AddrReg[i];
158 // Immediates are always OK.
159 if (AddrOp->isImm())
160 continue;
161
162 // Don't try to merge addresses that aren't either immediates or registers.
163 // TODO: Should be possible to merge FrameIndexes and maybe some other
164 // non-register
165 if (!AddrOp->isReg())
166 return false;
167
168 // TODO: We should be able to merge instructions with other physical reg
169 // addresses too.
170 if (AddrOp->getReg().isPhysical() &&
171 AddrOp->getReg() != AMDGPU::SGPR_NULL)
172 return false;
173
174 // If an address has only one use then there will be no other
175 // instructions with the same address, so we can't merge this one.
176 if (MRI.hasOneNonDBGUse(AddrOp->getReg()))
177 return false;
178 }
179 return true;
180 }
181
182 void setMI(MachineBasicBlock::iterator MI, const SILoadStoreOptimizer &LSO);
183
184 // Compare by pointer order.
185 bool operator<(const CombineInfo& Other) const {
186 return (InstClass == MIMG) ? DMask < Other.DMask : Offset < Other.Offset;
187 }
188 };
189
190 struct BaseRegisters {
191 Register LoReg;
192 Register HiReg;
193
194 unsigned LoSubReg = 0;
195 unsigned HiSubReg = 0;
196 // True when using V_ADD_U64_e64 pattern
197 bool UseV64Pattern = false;
198 };
199
200 struct MemAddress {
201 BaseRegisters Base;
202 int64_t Offset = 0;
203 };
204
205 using MemInfoMap = DenseMap<MachineInstr *, MemAddress>;
206
207private:
208 MachineFunction *MF = nullptr;
209 const GCNSubtarget *STM = nullptr;
210 const SIInstrInfo *TII = nullptr;
211 const SIRegisterInfo *TRI = nullptr;
212 MachineRegisterInfo *MRI = nullptr;
213 AliasAnalysis *AA = nullptr;
214 bool OptimizeAgain;
215
216 bool canSwapInstructions(const DenseSet<Register> &ARegDefs,
217 const DenseSet<Register> &ARegUses,
218 const MachineInstr &A, const MachineInstr &B) const;
219 static bool dmasksCanBeCombined(const CombineInfo &CI,
220 const SIInstrInfo &TII,
221 const CombineInfo &Paired);
222 static bool offsetsCanBeCombined(CombineInfo &CI, const GCNSubtarget &STI,
223 CombineInfo &Paired, bool Modify = false);
224 static bool widthsFit(const GCNSubtarget &STI, const CombineInfo &CI,
225 const CombineInfo &Paired);
226 unsigned getNewOpcode(const CombineInfo &CI, const CombineInfo &Paired);
227 static std::pair<unsigned, unsigned> getSubRegIdxs(const CombineInfo &CI,
228 const CombineInfo &Paired);
229 const TargetRegisterClass *
230 getTargetRegisterClass(const CombineInfo &CI,
231 const CombineInfo &Paired) const;
232 const TargetRegisterClass *getDataRegClass(const MachineInstr &MI) const;
233
234 CombineInfo *checkAndPrepareMerge(CombineInfo &CI, CombineInfo &Paired);
235
236 void copyToDestRegs(CombineInfo &CI, CombineInfo &Paired,
237 MachineBasicBlock::iterator InsertBefore,
238 const DebugLoc &DL, AMDGPU::OpName OpName,
239 Register DestReg) const;
240 Register copyFromSrcRegs(CombineInfo &CI, CombineInfo &Paired,
241 MachineBasicBlock::iterator InsertBefore,
242 const DebugLoc &DL, AMDGPU::OpName OpName) const;
243
244 unsigned read2Opcode(unsigned EltSize, bool GDS) const;
245 unsigned read2ST64Opcode(unsigned EltSize, bool GDS) const;
247 mergeRead2Pair(CombineInfo &CI, CombineInfo &Paired,
248 MachineBasicBlock::iterator InsertBefore);
249
250 unsigned write2Opcode(unsigned EltSize, bool GDS) const;
251 unsigned write2ST64Opcode(unsigned EltSize, bool GDS) const;
252 unsigned getWrite2Opcode(const CombineInfo &CI) const;
253
255 mergeWrite2Pair(CombineInfo &CI, CombineInfo &Paired,
256 MachineBasicBlock::iterator InsertBefore);
258 mergeImagePair(CombineInfo &CI, CombineInfo &Paired,
259 MachineBasicBlock::iterator InsertBefore);
261 mergeSMemLoadImmPair(CombineInfo &CI, CombineInfo &Paired,
262 MachineBasicBlock::iterator InsertBefore);
264 mergeBufferLoadPair(CombineInfo &CI, CombineInfo &Paired,
265 MachineBasicBlock::iterator InsertBefore);
267 mergeBufferStorePair(CombineInfo &CI, CombineInfo &Paired,
268 MachineBasicBlock::iterator InsertBefore);
270 mergeTBufferLoadPair(CombineInfo &CI, CombineInfo &Paired,
271 MachineBasicBlock::iterator InsertBefore);
273 mergeTBufferStorePair(CombineInfo &CI, CombineInfo &Paired,
274 MachineBasicBlock::iterator InsertBefore);
276 mergeFlatLoadPair(CombineInfo &CI, CombineInfo &Paired,
277 MachineBasicBlock::iterator InsertBefore);
279 mergeFlatStorePair(CombineInfo &CI, CombineInfo &Paired,
280 MachineBasicBlock::iterator InsertBefore);
281
282 void updateBaseAndOffset(MachineInstr &I, Register NewBase,
283 int32_t NewOffset) const;
284 void updateAsyncLDSAddress(MachineInstr &MI, int32_t OffsetDiff) const;
285 Register computeBase(MachineInstr &MI, const MemAddress &Addr) const;
286 MachineOperand createRegOrImm(int32_t Val, MachineInstr &MI) const;
287 bool processBaseWithConstOffset64(MachineInstr *AddDef,
288 const MachineOperand &Base,
289 MemAddress &Addr) const;
290 void processBaseWithConstOffset(const MachineOperand &Base, MemAddress &Addr) const;
291 /// Promotes constant offset to the immediate by adjusting the base. It
292 /// tries to use a base from the nearby instructions that allows it to have
293 /// a 13bit constant offset which gets promoted to the immediate.
294 bool promoteConstantOffsetToImm(MachineInstr &CI,
295 MemInfoMap &Visited,
296 SmallPtrSet<MachineInstr *, 4> &Promoted) const;
297 void addInstToMergeableList(const CombineInfo &CI,
298 std::list<std::list<CombineInfo> > &MergeableInsts) const;
299
300 std::pair<MachineBasicBlock::iterator, bool> collectMergeableInsts(
302 MemInfoMap &Visited, SmallPtrSet<MachineInstr *, 4> &AnchorList,
303 std::list<std::list<CombineInfo>> &MergeableInsts) const;
304
305 static MachineMemOperand *combineKnownAdjacentMMOs(const CombineInfo &CI,
306 const CombineInfo &Paired);
307
308 static InstClassEnum getCommonInstClass(const CombineInfo &CI,
309 const CombineInfo &Paired);
310
311 bool optimizeInstsWithSameBaseAddr(std::list<CombineInfo> &MergeList,
312 bool &OptimizeListAgain);
313 bool optimizeBlock(std::list<std::list<CombineInfo> > &MergeableInsts);
314
315public:
316 SILoadStoreOptimizer(AliasAnalysis *AA) : AA(AA) {}
317 bool run(MachineFunction &MF);
318};
319
320class SILoadStoreOptimizerLegacy : public MachineFunctionPass {
321public:
322 static char ID;
323
324 SILoadStoreOptimizerLegacy() : MachineFunctionPass(ID) {}
325
326 bool runOnMachineFunction(MachineFunction &MF) override;
327
328 StringRef getPassName() const override { return "SI Load Store Optimizer"; }
329
330 void getAnalysisUsage(AnalysisUsage &AU) const override {
331 AU.setPreservesCFG();
333
335 }
336
337 MachineFunctionProperties getRequiredProperties() const override {
338 return MachineFunctionProperties().setIsSSA();
339 }
340};
341
342static unsigned getOpcodeWidth(const MachineInstr &MI, const SIInstrInfo &TII) {
343 const unsigned Opc = MI.getOpcode();
344
345 if (TII.isMUBUF(Opc)) {
346 // FIXME: Handle d16 correctly
348 }
349 if (TII.isImage(MI)) {
350 uint64_t DMaskImm =
351 TII.getNamedOperand(MI, AMDGPU::OpName::dmask)->getImm();
352 return llvm::popcount(DMaskImm);
353 }
354 if (TII.isMTBUF(Opc)) {
356 }
357
358 switch (Opc) {
359 case AMDGPU::S_BUFFER_LOAD_DWORD_IMM:
360 case AMDGPU::S_BUFFER_LOAD_DWORD_SGPR_IMM:
361 case AMDGPU::S_LOAD_DWORD_IMM:
362 case AMDGPU::GLOBAL_LOAD_DWORD:
363 case AMDGPU::GLOBAL_LOAD_DWORD_SADDR:
364 case AMDGPU::GLOBAL_STORE_DWORD:
365 case AMDGPU::GLOBAL_STORE_DWORD_SADDR:
366 case AMDGPU::FLAT_LOAD_DWORD:
367 case AMDGPU::FLAT_STORE_DWORD:
368 case AMDGPU::FLAT_LOAD_DWORD_SADDR:
369 case AMDGPU::FLAT_STORE_DWORD_SADDR:
370 return 1;
371 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM:
372 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM:
373 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM_ec:
374 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM_ec:
375 case AMDGPU::S_LOAD_DWORDX2_IMM:
376 case AMDGPU::S_LOAD_DWORDX2_IMM_ec:
377 case AMDGPU::GLOBAL_LOAD_DWORDX2:
378 case AMDGPU::GLOBAL_LOAD_DWORDX2_SADDR:
379 case AMDGPU::GLOBAL_STORE_DWORDX2:
380 case AMDGPU::GLOBAL_STORE_DWORDX2_SADDR:
381 case AMDGPU::FLAT_LOAD_DWORDX2:
382 case AMDGPU::FLAT_STORE_DWORDX2:
383 case AMDGPU::FLAT_LOAD_DWORDX2_SADDR:
384 case AMDGPU::FLAT_STORE_DWORDX2_SADDR:
385 return 2;
386 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM:
387 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM:
388 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM_ec:
389 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM_ec:
390 case AMDGPU::S_LOAD_DWORDX3_IMM:
391 case AMDGPU::S_LOAD_DWORDX3_IMM_ec:
392 case AMDGPU::GLOBAL_LOAD_DWORDX3:
393 case AMDGPU::GLOBAL_LOAD_DWORDX3_SADDR:
394 case AMDGPU::GLOBAL_STORE_DWORDX3:
395 case AMDGPU::GLOBAL_STORE_DWORDX3_SADDR:
396 case AMDGPU::FLAT_LOAD_DWORDX3:
397 case AMDGPU::FLAT_STORE_DWORDX3:
398 case AMDGPU::FLAT_LOAD_DWORDX3_SADDR:
399 case AMDGPU::FLAT_STORE_DWORDX3_SADDR:
400 return 3;
401 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM:
402 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM:
403 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM_ec:
404 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM_ec:
405 case AMDGPU::S_LOAD_DWORDX4_IMM:
406 case AMDGPU::S_LOAD_DWORDX4_IMM_ec:
407 case AMDGPU::GLOBAL_LOAD_DWORDX4:
408 case AMDGPU::GLOBAL_LOAD_DWORDX4_SADDR:
409 case AMDGPU::GLOBAL_STORE_DWORDX4:
410 case AMDGPU::GLOBAL_STORE_DWORDX4_SADDR:
411 case AMDGPU::FLAT_LOAD_DWORDX4:
412 case AMDGPU::FLAT_STORE_DWORDX4:
413 case AMDGPU::FLAT_LOAD_DWORDX4_SADDR:
414 case AMDGPU::FLAT_STORE_DWORDX4_SADDR:
415 return 4;
416 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM:
417 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM:
418 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM_ec:
419 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM_ec:
420 case AMDGPU::S_LOAD_DWORDX8_IMM:
421 case AMDGPU::S_LOAD_DWORDX8_IMM_ec:
422 return 8;
423 case AMDGPU::DS_READ_B32:
424 case AMDGPU::DS_READ_B32_gfx9:
425 case AMDGPU::DS_WRITE_B32:
426 case AMDGPU::DS_WRITE_B32_gfx9:
427 return 1;
428 case AMDGPU::DS_READ_B64:
429 case AMDGPU::DS_READ_B64_gfx9:
430 case AMDGPU::DS_WRITE_B64:
431 case AMDGPU::DS_WRITE_B64_gfx9:
432 return 2;
433 default:
434 return 0;
435 }
436}
437
438/// Maps instruction opcode to enum InstClassEnum.
439static InstClassEnum getInstClass(unsigned Opc, const SIInstrInfo &TII) {
440 switch (Opc) {
441 default:
442 if (TII.isMUBUF(Opc)) {
444 default:
445 return UNKNOWN;
446 case AMDGPU::BUFFER_LOAD_DWORD_BOTHEN:
447 case AMDGPU::BUFFER_LOAD_DWORD_BOTHEN_exact:
448 case AMDGPU::BUFFER_LOAD_DWORD_IDXEN:
449 case AMDGPU::BUFFER_LOAD_DWORD_IDXEN_exact:
450 case AMDGPU::BUFFER_LOAD_DWORD_OFFEN:
451 case AMDGPU::BUFFER_LOAD_DWORD_OFFEN_exact:
452 case AMDGPU::BUFFER_LOAD_DWORD_OFFSET:
453 case AMDGPU::BUFFER_LOAD_DWORD_OFFSET_exact:
454 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_BOTHEN:
455 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_BOTHEN_exact:
456 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_IDXEN:
457 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_IDXEN_exact:
458 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_OFFEN:
459 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_OFFEN_exact:
460 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_OFFSET:
461 case AMDGPU::BUFFER_LOAD_DWORD_VBUFFER_OFFSET_exact:
462 return BUFFER_LOAD;
463 case AMDGPU::BUFFER_STORE_DWORD_BOTHEN:
464 case AMDGPU::BUFFER_STORE_DWORD_BOTHEN_exact:
465 case AMDGPU::BUFFER_STORE_DWORD_IDXEN:
466 case AMDGPU::BUFFER_STORE_DWORD_IDXEN_exact:
467 case AMDGPU::BUFFER_STORE_DWORD_OFFEN:
468 case AMDGPU::BUFFER_STORE_DWORD_OFFEN_exact:
469 case AMDGPU::BUFFER_STORE_DWORD_OFFSET:
470 case AMDGPU::BUFFER_STORE_DWORD_OFFSET_exact:
471 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_BOTHEN:
472 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_BOTHEN_exact:
473 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_IDXEN:
474 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_IDXEN_exact:
475 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_OFFEN:
476 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_OFFEN_exact:
477 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_OFFSET:
478 case AMDGPU::BUFFER_STORE_DWORD_VBUFFER_OFFSET_exact:
479 return BUFFER_STORE;
480 }
481 }
482 if (TII.isImage(Opc)) {
483 // Ignore instructions encoded without vaddr.
484 if (!AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vaddr) &&
485 !AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vaddr0))
486 return UNKNOWN;
487 // Ignore BVH instructions
489 return UNKNOWN;
490 // TODO: Support IMAGE_GET_RESINFO and IMAGE_GET_LOD.
491 if (TII.get(Opc).mayStore() || !TII.get(Opc).mayLoad() ||
492 TII.isGather4(Opc))
493 return UNKNOWN;
494 return MIMG;
495 }
496 if (TII.isMTBUF(Opc)) {
498 default:
499 return UNKNOWN;
500 case AMDGPU::TBUFFER_LOAD_FORMAT_X_BOTHEN:
501 case AMDGPU::TBUFFER_LOAD_FORMAT_X_BOTHEN_exact:
502 case AMDGPU::TBUFFER_LOAD_FORMAT_X_IDXEN:
503 case AMDGPU::TBUFFER_LOAD_FORMAT_X_IDXEN_exact:
504 case AMDGPU::TBUFFER_LOAD_FORMAT_X_OFFEN:
505 case AMDGPU::TBUFFER_LOAD_FORMAT_X_OFFEN_exact:
506 case AMDGPU::TBUFFER_LOAD_FORMAT_X_OFFSET:
507 case AMDGPU::TBUFFER_LOAD_FORMAT_X_OFFSET_exact:
508 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_BOTHEN:
509 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_BOTHEN_exact:
510 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_IDXEN:
511 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_IDXEN_exact:
512 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_OFFEN:
513 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_OFFEN_exact:
514 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_OFFSET:
515 case AMDGPU::TBUFFER_LOAD_FORMAT_X_VBUFFER_OFFSET_exact:
516 return TBUFFER_LOAD;
517 case AMDGPU::TBUFFER_STORE_FORMAT_X_OFFEN:
518 case AMDGPU::TBUFFER_STORE_FORMAT_X_OFFEN_exact:
519 case AMDGPU::TBUFFER_STORE_FORMAT_X_OFFSET:
520 case AMDGPU::TBUFFER_STORE_FORMAT_X_OFFSET_exact:
521 case AMDGPU::TBUFFER_STORE_FORMAT_X_VBUFFER_OFFEN:
522 case AMDGPU::TBUFFER_STORE_FORMAT_X_VBUFFER_OFFEN_exact:
523 case AMDGPU::TBUFFER_STORE_FORMAT_X_VBUFFER_OFFSET:
524 case AMDGPU::TBUFFER_STORE_FORMAT_X_VBUFFER_OFFSET_exact:
525 return TBUFFER_STORE;
526 }
527 }
528 return UNKNOWN;
529 case AMDGPU::S_BUFFER_LOAD_DWORD_IMM:
530 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM:
531 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM:
532 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM:
533 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM:
534 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM_ec:
535 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM_ec:
536 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM_ec:
537 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM_ec:
538 return S_BUFFER_LOAD_IMM;
539 case AMDGPU::S_BUFFER_LOAD_DWORD_SGPR_IMM:
540 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM:
541 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM:
542 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM:
543 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM:
544 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM_ec:
545 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM_ec:
546 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM_ec:
547 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM_ec:
548 return S_BUFFER_LOAD_SGPR_IMM;
549 case AMDGPU::S_LOAD_DWORD_IMM:
550 case AMDGPU::S_LOAD_DWORDX2_IMM:
551 case AMDGPU::S_LOAD_DWORDX3_IMM:
552 case AMDGPU::S_LOAD_DWORDX4_IMM:
553 case AMDGPU::S_LOAD_DWORDX8_IMM:
554 case AMDGPU::S_LOAD_DWORDX2_IMM_ec:
555 case AMDGPU::S_LOAD_DWORDX3_IMM_ec:
556 case AMDGPU::S_LOAD_DWORDX4_IMM_ec:
557 case AMDGPU::S_LOAD_DWORDX8_IMM_ec:
558 return S_LOAD_IMM;
559 case AMDGPU::DS_READ_B32:
560 case AMDGPU::DS_READ_B32_gfx9:
561 case AMDGPU::DS_READ_B64:
562 case AMDGPU::DS_READ_B64_gfx9:
563 return DS_READ;
564 case AMDGPU::DS_WRITE_B32:
565 case AMDGPU::DS_WRITE_B32_gfx9:
566 case AMDGPU::DS_WRITE_B64:
567 case AMDGPU::DS_WRITE_B64_gfx9:
568 return DS_WRITE;
569 case AMDGPU::GLOBAL_LOAD_DWORD:
570 case AMDGPU::GLOBAL_LOAD_DWORDX2:
571 case AMDGPU::GLOBAL_LOAD_DWORDX3:
572 case AMDGPU::GLOBAL_LOAD_DWORDX4:
573 case AMDGPU::FLAT_LOAD_DWORD:
574 case AMDGPU::FLAT_LOAD_DWORDX2:
575 case AMDGPU::FLAT_LOAD_DWORDX3:
576 case AMDGPU::FLAT_LOAD_DWORDX4:
577 return FLAT_LOAD;
578 case AMDGPU::GLOBAL_LOAD_DWORD_SADDR:
579 case AMDGPU::GLOBAL_LOAD_DWORDX2_SADDR:
580 case AMDGPU::GLOBAL_LOAD_DWORDX3_SADDR:
581 case AMDGPU::GLOBAL_LOAD_DWORDX4_SADDR:
582 return GLOBAL_LOAD_SADDR;
583 case AMDGPU::GLOBAL_STORE_DWORD:
584 case AMDGPU::GLOBAL_STORE_DWORDX2:
585 case AMDGPU::GLOBAL_STORE_DWORDX3:
586 case AMDGPU::GLOBAL_STORE_DWORDX4:
587 case AMDGPU::FLAT_STORE_DWORD:
588 case AMDGPU::FLAT_STORE_DWORDX2:
589 case AMDGPU::FLAT_STORE_DWORDX3:
590 case AMDGPU::FLAT_STORE_DWORDX4:
591 return FLAT_STORE;
592 case AMDGPU::GLOBAL_STORE_DWORD_SADDR:
593 case AMDGPU::GLOBAL_STORE_DWORDX2_SADDR:
594 case AMDGPU::GLOBAL_STORE_DWORDX3_SADDR:
595 case AMDGPU::GLOBAL_STORE_DWORDX4_SADDR:
596 return GLOBAL_STORE_SADDR;
597 case AMDGPU::FLAT_LOAD_DWORD_SADDR:
598 case AMDGPU::FLAT_LOAD_DWORDX2_SADDR:
599 case AMDGPU::FLAT_LOAD_DWORDX3_SADDR:
600 case AMDGPU::FLAT_LOAD_DWORDX4_SADDR:
601 return FLAT_LOAD_SADDR;
602 case AMDGPU::FLAT_STORE_DWORD_SADDR:
603 case AMDGPU::FLAT_STORE_DWORDX2_SADDR:
604 case AMDGPU::FLAT_STORE_DWORDX3_SADDR:
605 case AMDGPU::FLAT_STORE_DWORDX4_SADDR:
606 return FLAT_STORE_SADDR;
607 }
608}
609
610/// Determines instruction subclass from opcode. Only instructions
611/// of the same subclass can be merged together. The merged instruction may have
612/// a different subclass but must have the same class.
613static unsigned getInstSubclass(unsigned Opc, const SIInstrInfo &TII) {
614 switch (Opc) {
615 default:
616 if (TII.isMUBUF(Opc))
618 if (TII.isImage(Opc)) {
620 assert(Info);
621 return Info->BaseOpcode;
622 }
623 if (TII.isMTBUF(Opc))
625 return -1;
626 case AMDGPU::DS_READ_B32:
627 case AMDGPU::DS_READ_B32_gfx9:
628 case AMDGPU::DS_READ_B64:
629 case AMDGPU::DS_READ_B64_gfx9:
630 case AMDGPU::DS_WRITE_B32:
631 case AMDGPU::DS_WRITE_B32_gfx9:
632 case AMDGPU::DS_WRITE_B64:
633 case AMDGPU::DS_WRITE_B64_gfx9:
634 return Opc;
635 case AMDGPU::S_BUFFER_LOAD_DWORD_IMM:
636 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM:
637 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM:
638 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM:
639 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM:
640 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM_ec:
641 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM_ec:
642 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM_ec:
643 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM_ec:
644 return AMDGPU::S_BUFFER_LOAD_DWORD_IMM;
645 case AMDGPU::S_BUFFER_LOAD_DWORD_SGPR_IMM:
646 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM:
647 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM:
648 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM:
649 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM:
650 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM_ec:
651 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM_ec:
652 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM_ec:
653 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM_ec:
654 return AMDGPU::S_BUFFER_LOAD_DWORD_SGPR_IMM;
655 case AMDGPU::S_LOAD_DWORD_IMM:
656 case AMDGPU::S_LOAD_DWORDX2_IMM:
657 case AMDGPU::S_LOAD_DWORDX3_IMM:
658 case AMDGPU::S_LOAD_DWORDX4_IMM:
659 case AMDGPU::S_LOAD_DWORDX8_IMM:
660 case AMDGPU::S_LOAD_DWORDX2_IMM_ec:
661 case AMDGPU::S_LOAD_DWORDX3_IMM_ec:
662 case AMDGPU::S_LOAD_DWORDX4_IMM_ec:
663 case AMDGPU::S_LOAD_DWORDX8_IMM_ec:
664 return AMDGPU::S_LOAD_DWORD_IMM;
665 case AMDGPU::GLOBAL_LOAD_DWORD:
666 case AMDGPU::GLOBAL_LOAD_DWORDX2:
667 case AMDGPU::GLOBAL_LOAD_DWORDX3:
668 case AMDGPU::GLOBAL_LOAD_DWORDX4:
669 case AMDGPU::FLAT_LOAD_DWORD:
670 case AMDGPU::FLAT_LOAD_DWORDX2:
671 case AMDGPU::FLAT_LOAD_DWORDX3:
672 case AMDGPU::FLAT_LOAD_DWORDX4:
673 return AMDGPU::FLAT_LOAD_DWORD;
674 case AMDGPU::GLOBAL_LOAD_DWORD_SADDR:
675 case AMDGPU::GLOBAL_LOAD_DWORDX2_SADDR:
676 case AMDGPU::GLOBAL_LOAD_DWORDX3_SADDR:
677 case AMDGPU::GLOBAL_LOAD_DWORDX4_SADDR:
678 return AMDGPU::GLOBAL_LOAD_DWORD_SADDR;
679 case AMDGPU::GLOBAL_STORE_DWORD:
680 case AMDGPU::GLOBAL_STORE_DWORDX2:
681 case AMDGPU::GLOBAL_STORE_DWORDX3:
682 case AMDGPU::GLOBAL_STORE_DWORDX4:
683 case AMDGPU::FLAT_STORE_DWORD:
684 case AMDGPU::FLAT_STORE_DWORDX2:
685 case AMDGPU::FLAT_STORE_DWORDX3:
686 case AMDGPU::FLAT_STORE_DWORDX4:
687 return AMDGPU::FLAT_STORE_DWORD;
688 case AMDGPU::GLOBAL_STORE_DWORD_SADDR:
689 case AMDGPU::GLOBAL_STORE_DWORDX2_SADDR:
690 case AMDGPU::GLOBAL_STORE_DWORDX3_SADDR:
691 case AMDGPU::GLOBAL_STORE_DWORDX4_SADDR:
692 return AMDGPU::GLOBAL_STORE_DWORD_SADDR;
693 case AMDGPU::FLAT_LOAD_DWORD_SADDR:
694 case AMDGPU::FLAT_LOAD_DWORDX2_SADDR:
695 case AMDGPU::FLAT_LOAD_DWORDX3_SADDR:
696 case AMDGPU::FLAT_LOAD_DWORDX4_SADDR:
697 return AMDGPU::FLAT_LOAD_DWORD_SADDR;
698 case AMDGPU::FLAT_STORE_DWORD_SADDR:
699 case AMDGPU::FLAT_STORE_DWORDX2_SADDR:
700 case AMDGPU::FLAT_STORE_DWORDX3_SADDR:
701 case AMDGPU::FLAT_STORE_DWORDX4_SADDR:
702 return AMDGPU::FLAT_STORE_DWORD_SADDR;
703 }
704}
705
706// GLOBAL loads and stores are classified as FLAT initially. If both combined
707// instructions are FLAT GLOBAL adjust the class to GLOBAL_LOAD or GLOBAL_STORE.
708// If either or both instructions are non segment specific FLAT the resulting
709// combined operation will be FLAT, potentially promoting one of the GLOBAL
710// operations to FLAT.
711// For other instructions return the original unmodified class.
712InstClassEnum
713SILoadStoreOptimizer::getCommonInstClass(const CombineInfo &CI,
714 const CombineInfo &Paired) {
715 assert(CI.InstClass == Paired.InstClass);
716
717 if ((CI.InstClass == FLAT_LOAD || CI.InstClass == FLAT_STORE) &&
719 return (CI.InstClass == FLAT_STORE) ? GLOBAL_STORE : GLOBAL_LOAD;
720
721 return CI.InstClass;
722}
723
724static AddressRegs getRegs(unsigned Opc, const SIInstrInfo &TII) {
725 AddressRegs Result;
726
727 if (TII.isMUBUF(Opc)) {
729 Result.VAddr = true;
731 Result.SRsrc = true;
733 Result.SOffset = true;
734
735 return Result;
736 }
737
738 if (TII.isImage(Opc)) {
739 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
740 if (VAddr0Idx >= 0) {
741 AMDGPU::OpName RsrcName =
742 TII.isMIMG(Opc) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
743 int RsrcIdx = AMDGPU::getNamedOperandIdx(Opc, RsrcName);
744 Result.NumVAddrs = RsrcIdx - VAddr0Idx;
745 } else {
746 Result.VAddr = true;
747 }
748 Result.SRsrc = true;
750 if (Info && AMDGPU::getMIMGBaseOpcodeInfo(Info->BaseOpcode)->Sampler)
751 Result.SSamp = true;
752
753 return Result;
754 }
755 if (TII.isMTBUF(Opc)) {
757 Result.VAddr = true;
759 Result.SRsrc = true;
761 Result.SOffset = true;
762
763 return Result;
764 }
765
766 switch (Opc) {
767 default:
768 return Result;
769 case AMDGPU::S_BUFFER_LOAD_DWORD_SGPR_IMM:
770 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM:
771 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM:
772 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM:
773 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM:
774 case AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM_ec:
775 case AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM_ec:
776 case AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM_ec:
777 case AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM_ec:
778 Result.SOffset = true;
779 [[fallthrough]];
780 case AMDGPU::S_BUFFER_LOAD_DWORD_IMM:
781 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM:
782 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM:
783 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM:
784 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM:
785 case AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM_ec:
786 case AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM_ec:
787 case AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM_ec:
788 case AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM_ec:
789 case AMDGPU::S_LOAD_DWORD_IMM:
790 case AMDGPU::S_LOAD_DWORDX2_IMM:
791 case AMDGPU::S_LOAD_DWORDX3_IMM:
792 case AMDGPU::S_LOAD_DWORDX4_IMM:
793 case AMDGPU::S_LOAD_DWORDX8_IMM:
794 case AMDGPU::S_LOAD_DWORDX2_IMM_ec:
795 case AMDGPU::S_LOAD_DWORDX3_IMM_ec:
796 case AMDGPU::S_LOAD_DWORDX4_IMM_ec:
797 case AMDGPU::S_LOAD_DWORDX8_IMM_ec:
798 Result.SBase = true;
799 return Result;
800 case AMDGPU::DS_READ_B32:
801 case AMDGPU::DS_READ_B64:
802 case AMDGPU::DS_READ_B32_gfx9:
803 case AMDGPU::DS_READ_B64_gfx9:
804 case AMDGPU::DS_WRITE_B32:
805 case AMDGPU::DS_WRITE_B64:
806 case AMDGPU::DS_WRITE_B32_gfx9:
807 case AMDGPU::DS_WRITE_B64_gfx9:
808 Result.Addr = true;
809 return Result;
810 case AMDGPU::GLOBAL_LOAD_DWORD_SADDR:
811 case AMDGPU::GLOBAL_LOAD_DWORDX2_SADDR:
812 case AMDGPU::GLOBAL_LOAD_DWORDX3_SADDR:
813 case AMDGPU::GLOBAL_LOAD_DWORDX4_SADDR:
814 case AMDGPU::GLOBAL_STORE_DWORD_SADDR:
815 case AMDGPU::GLOBAL_STORE_DWORDX2_SADDR:
816 case AMDGPU::GLOBAL_STORE_DWORDX3_SADDR:
817 case AMDGPU::GLOBAL_STORE_DWORDX4_SADDR:
818 case AMDGPU::FLAT_LOAD_DWORD_SADDR:
819 case AMDGPU::FLAT_LOAD_DWORDX2_SADDR:
820 case AMDGPU::FLAT_LOAD_DWORDX3_SADDR:
821 case AMDGPU::FLAT_LOAD_DWORDX4_SADDR:
822 case AMDGPU::FLAT_STORE_DWORD_SADDR:
823 case AMDGPU::FLAT_STORE_DWORDX2_SADDR:
824 case AMDGPU::FLAT_STORE_DWORDX3_SADDR:
825 case AMDGPU::FLAT_STORE_DWORDX4_SADDR:
826 Result.SAddr = true;
827 [[fallthrough]];
828 case AMDGPU::GLOBAL_LOAD_DWORD:
829 case AMDGPU::GLOBAL_LOAD_DWORDX2:
830 case AMDGPU::GLOBAL_LOAD_DWORDX3:
831 case AMDGPU::GLOBAL_LOAD_DWORDX4:
832 case AMDGPU::GLOBAL_STORE_DWORD:
833 case AMDGPU::GLOBAL_STORE_DWORDX2:
834 case AMDGPU::GLOBAL_STORE_DWORDX3:
835 case AMDGPU::GLOBAL_STORE_DWORDX4:
836 case AMDGPU::FLAT_LOAD_DWORD:
837 case AMDGPU::FLAT_LOAD_DWORDX2:
838 case AMDGPU::FLAT_LOAD_DWORDX3:
839 case AMDGPU::FLAT_LOAD_DWORDX4:
840 case AMDGPU::FLAT_STORE_DWORD:
841 case AMDGPU::FLAT_STORE_DWORDX2:
842 case AMDGPU::FLAT_STORE_DWORDX3:
843 case AMDGPU::FLAT_STORE_DWORDX4:
844 Result.VAddr = true;
845 return Result;
846 }
847}
848
849void SILoadStoreOptimizer::CombineInfo::setMI(MachineBasicBlock::iterator MI,
850 const SILoadStoreOptimizer &LSO) {
851 I = MI;
852 unsigned Opc = MI->getOpcode();
853 InstClass = getInstClass(Opc, *LSO.TII);
854
855 if (InstClass == UNKNOWN)
856 return;
857
858 DataRC = LSO.getDataRegClass(*MI);
859
860 switch (InstClass) {
861 case DS_READ:
862 EltSize =
863 (Opc == AMDGPU::DS_READ_B64 || Opc == AMDGPU::DS_READ_B64_gfx9) ? 8
864 : 4;
865 break;
866 case DS_WRITE:
867 EltSize =
868 (Opc == AMDGPU::DS_WRITE_B64 || Opc == AMDGPU::DS_WRITE_B64_gfx9) ? 8
869 : 4;
870 break;
871 case S_BUFFER_LOAD_IMM:
872 case S_BUFFER_LOAD_SGPR_IMM:
873 case S_LOAD_IMM:
874 EltSize = AMDGPU::convertSMRDOffsetUnits(*LSO.STM, 4);
875 break;
876 default:
877 EltSize = 4;
878 break;
879 }
880
881 if (InstClass == MIMG) {
882 DMask = LSO.TII->getNamedOperand(*I, AMDGPU::OpName::dmask)->getImm();
883 // Offset is not considered for MIMG instructions.
884 Offset = 0;
885 } else {
886 int OffsetIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::offset);
887 Offset = I->getOperand(OffsetIdx).getImm();
888 }
889
890 if (InstClass == TBUFFER_LOAD || InstClass == TBUFFER_STORE) {
891 Format = LSO.TII->getNamedOperand(*I, AMDGPU::OpName::format)->getImm();
892 const AMDGPU::GcnBufferFormatInfo *Info =
893 AMDGPU::getGcnBufferFormatInfo(Format, *LSO.STM);
894 EltSize = Info->BitsPerComp / 8;
895 }
896
897 Width = getOpcodeWidth(*I, *LSO.TII);
898
899 if ((InstClass == DS_READ) || (InstClass == DS_WRITE)) {
900 Offset &= 0xffff;
901 GDS = LSO.TII->getNamedOperand(*I, AMDGPU::OpName::gds)->getImm();
902 } else if (InstClass != MIMG) {
903 CPol = LSO.TII->getNamedOperand(*I, AMDGPU::OpName::cpol)->getImm();
904 }
905
906 AddressRegs Regs = getRegs(Opc, *LSO.TII);
907 bool isVIMAGEorVSAMPLE = LSO.TII->isVIMAGE(*I) || LSO.TII->isVSAMPLE(*I);
908
909 NumAddresses = 0;
910 for (unsigned J = 0; J < Regs.NumVAddrs; J++)
911 AddrIdx[NumAddresses++] =
912 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0) + J;
913 if (Regs.Addr)
914 AddrIdx[NumAddresses++] =
915 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::addr);
916 if (Regs.SBase)
917 AddrIdx[NumAddresses++] =
918 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::sbase);
919 if (Regs.SRsrc)
920 AddrIdx[NumAddresses++] = AMDGPU::getNamedOperandIdx(
921 Opc, isVIMAGEorVSAMPLE ? AMDGPU::OpName::rsrc : AMDGPU::OpName::srsrc);
922 if (Regs.SOffset)
923 AddrIdx[NumAddresses++] =
924 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::soffset);
925 if (Regs.SAddr)
926 AddrIdx[NumAddresses++] =
927 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::saddr);
928 if (Regs.VAddr)
929 AddrIdx[NumAddresses++] =
930 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
931 if (Regs.SSamp)
932 AddrIdx[NumAddresses++] = AMDGPU::getNamedOperandIdx(
933 Opc, isVIMAGEorVSAMPLE ? AMDGPU::OpName::samp : AMDGPU::OpName::ssamp);
934 assert(NumAddresses <= MaxAddressRegs);
935
936 for (unsigned J = 0; J < NumAddresses; J++)
937 AddrReg[J] = &I->getOperand(AddrIdx[J]);
938}
939
940} // end anonymous namespace.
941
942INITIALIZE_PASS_BEGIN(SILoadStoreOptimizerLegacy, DEBUG_TYPE,
943 "SI Load Store Optimizer", false, false)
945INITIALIZE_PASS_END(SILoadStoreOptimizerLegacy, DEBUG_TYPE,
946 "SI Load Store Optimizer", false, false)
947
948char SILoadStoreOptimizerLegacy::ID = 0;
949
950char &llvm::SILoadStoreOptimizerLegacyID = SILoadStoreOptimizerLegacy::ID;
951
953 DenseSet<Register> &RegDefs,
954 DenseSet<Register> &RegUses) {
955 for (const auto &Op : MI.operands()) {
956 if (!Op.isReg())
957 continue;
958 if (Op.isDef())
959 RegDefs.insert(Op.getReg());
960 if (Op.readsReg())
961 RegUses.insert(Op.getReg());
962 }
963}
964
965bool SILoadStoreOptimizer::canSwapInstructions(
966 const DenseSet<Register> &ARegDefs, const DenseSet<Register> &ARegUses,
967 const MachineInstr &A, const MachineInstr &B) const {
968 if (A.mayLoadOrStore() && B.mayLoadOrStore() &&
969 (A.mayStore() || B.mayStore()) && A.mayAlias(AA, B, true))
970 return false;
971 for (const auto &BOp : B.operands()) {
972 if (!BOp.isReg())
973 continue;
974 if ((BOp.isDef() || BOp.readsReg()) && ARegDefs.contains(BOp.getReg()))
975 return false;
976 if (BOp.isDef() && ARegUses.contains(BOp.getReg()))
977 return false;
978 }
979 return true;
980}
981
982// Given that \p CI and \p Paired are adjacent memory operations produce a new
983// MMO for the combined operation with a new access size.
984MachineMemOperand *
985SILoadStoreOptimizer::combineKnownAdjacentMMOs(const CombineInfo &CI,
986 const CombineInfo &Paired) {
987 const MachineMemOperand *MMOa = *CI.I->memoperands_begin();
988 const MachineMemOperand *MMOb = *Paired.I->memoperands_begin();
989
990 unsigned Size = MMOa->getSize().getValue() + MMOb->getSize().getValue();
991
992 // A base pointer for the combined operation is the same as the leading
993 // operation's pointer.
994 if (Paired < CI)
995 std::swap(MMOa, MMOb);
996
997 MachinePointerInfo PtrInfo(MMOa->getPointerInfo());
998 // If merging FLAT and GLOBAL set address space to FLAT.
1000 PtrInfo.AddrSpace = AMDGPUAS::FLAT_ADDRESS;
1001
1002 MachineFunction *MF = CI.I->getMF();
1003 return MF->getMachineMemOperand(MMOa, PtrInfo, Size);
1004}
1005
1006bool SILoadStoreOptimizer::dmasksCanBeCombined(const CombineInfo &CI,
1007 const SIInstrInfo &TII,
1008 const CombineInfo &Paired) {
1009 assert(CI.InstClass == MIMG);
1010
1011 // Ignore instructions with tfe/lwe set.
1012 const auto *TFEOp = TII.getNamedOperand(*CI.I, AMDGPU::OpName::tfe);
1013 const auto *LWEOp = TII.getNamedOperand(*CI.I, AMDGPU::OpName::lwe);
1014
1015 if ((TFEOp && TFEOp->getImm()) || (LWEOp && LWEOp->getImm()))
1016 return false;
1017
1018 // Check other optional immediate operands for equality.
1019 AMDGPU::OpName OperandsToMatch[] = {
1020 AMDGPU::OpName::cpol, AMDGPU::OpName::d16, AMDGPU::OpName::unorm,
1021 AMDGPU::OpName::da, AMDGPU::OpName::r128, AMDGPU::OpName::a16,
1022 AMDGPU::OpName::dim};
1023
1024 for (AMDGPU::OpName op : OperandsToMatch) {
1025 int Idx = AMDGPU::getNamedOperandIdx(CI.I->getOpcode(), op);
1026 if (AMDGPU::getNamedOperandIdx(Paired.I->getOpcode(), op) != Idx)
1027 return false;
1028 if (Idx != -1 &&
1029 CI.I->getOperand(Idx).getImm() != Paired.I->getOperand(Idx).getImm())
1030 return false;
1031 }
1032
1033 // Check DMask for overlaps.
1034 unsigned MaxMask = std::max(CI.DMask, Paired.DMask);
1035 unsigned MinMask = std::min(CI.DMask, Paired.DMask);
1036
1037 if (!MaxMask)
1038 return false;
1039
1040 unsigned AllowedBitsForMin = llvm::countr_zero(MaxMask);
1041 if ((1u << AllowedBitsForMin) <= MinMask)
1042 return false;
1043
1044 return true;
1045}
1046
1047static unsigned getBufferFormatWithCompCount(unsigned OldFormat,
1048 unsigned ComponentCount,
1049 const GCNSubtarget &STI) {
1050 if (ComponentCount > 4)
1051 return 0;
1052
1053 const llvm::AMDGPU::GcnBufferFormatInfo *OldFormatInfo =
1055 if (!OldFormatInfo)
1056 return 0;
1057
1058 const llvm::AMDGPU::GcnBufferFormatInfo *NewFormatInfo =
1060 ComponentCount,
1061 OldFormatInfo->NumFormat, STI);
1062
1063 if (!NewFormatInfo)
1064 return 0;
1065
1066 assert(NewFormatInfo->NumFormat == OldFormatInfo->NumFormat &&
1067 NewFormatInfo->BitsPerComp == OldFormatInfo->BitsPerComp);
1068
1069 return NewFormatInfo->Format;
1070}
1071
1072// Return the value in the inclusive range [Lo,Hi] that is aligned to the
1073// highest power of two. Note that the result is well defined for all inputs
1074// including corner cases like:
1075// - if Lo == Hi, return that value
1076// - if Lo == 0, return 0 (even though the "- 1" below underflows
1077// - if Lo > Hi, return 0 (as if the range wrapped around)
1081
1082bool SILoadStoreOptimizer::offsetsCanBeCombined(CombineInfo &CI,
1083 const GCNSubtarget &STI,
1084 CombineInfo &Paired,
1085 bool Modify) {
1086 assert(CI.InstClass != MIMG);
1087
1088 // XXX - Would the same offset be OK? Is there any reason this would happen or
1089 // be useful?
1090 if (CI.Offset == Paired.Offset)
1091 return false;
1092
1093 if (CI.GDS != Paired.GDS)
1094 return false;
1095
1096 // This won't be valid if the offset isn't aligned.
1097 if ((CI.Offset % CI.EltSize != 0) || (Paired.Offset % CI.EltSize != 0))
1098 return false;
1099
1100 if (CI.InstClass == TBUFFER_LOAD || CI.InstClass == TBUFFER_STORE) {
1101
1102 const llvm::AMDGPU::GcnBufferFormatInfo *Info0 =
1104 const llvm::AMDGPU::GcnBufferFormatInfo *Info1 =
1105 llvm::AMDGPU::getGcnBufferFormatInfo(Paired.Format, STI);
1106
1107 if (Info0->BitsPerComp != Info1->BitsPerComp ||
1108 Info0->NumFormat != Info1->NumFormat)
1109 return false;
1110
1111 // For 8-bit or 16-bit formats there is no 3-component variant.
1112 // If NumCombinedComponents is 3, try the 4-component format and use XYZ.
1113 // Example:
1114 // tbuffer_load_format_x + tbuffer_load_format_x + tbuffer_load_format_x
1115 // ==> tbuffer_load_format_xyz with format:[BUF_FMT_16_16_16_16_SNORM]
1116 unsigned NumCombinedComponents = CI.Width + Paired.Width;
1117 if (NumCombinedComponents == 3 && CI.EltSize <= 2)
1118 NumCombinedComponents = 4;
1119
1120 if (getBufferFormatWithCompCount(CI.Format, NumCombinedComponents, STI) ==
1121 0)
1122 return false;
1123
1124 // Merge only when the two access ranges are strictly back-to-back,
1125 // any gap or overlap can over-write data or leave holes.
1126 unsigned ElemIndex0 = CI.Offset / CI.EltSize;
1127 unsigned ElemIndex1 = Paired.Offset / Paired.EltSize;
1128 if (ElemIndex0 + CI.Width != ElemIndex1 &&
1129 ElemIndex1 + Paired.Width != ElemIndex0)
1130 return false;
1131
1132 // 1-byte formats require 1-byte alignment.
1133 // 2-byte formats require 2-byte alignment.
1134 // 4-byte and larger formats require 4-byte alignment.
1135 unsigned MergedBytes = CI.EltSize * NumCombinedComponents;
1136 unsigned RequiredAlign = std::min(MergedBytes, 4u);
1137 unsigned MinOff = std::min(CI.Offset, Paired.Offset);
1138 if (MinOff % RequiredAlign != 0)
1139 return false;
1140
1141 return true;
1142 }
1143
1144 uint32_t EltOffset0 = CI.Offset / CI.EltSize;
1145 uint32_t EltOffset1 = Paired.Offset / CI.EltSize;
1146 CI.UseST64 = false;
1147 CI.BaseOff = 0;
1148
1149 // Handle all non-DS instructions.
1150 if ((CI.InstClass != DS_READ) && (CI.InstClass != DS_WRITE)) {
1151 if (EltOffset0 + CI.Width != EltOffset1 &&
1152 EltOffset1 + Paired.Width != EltOffset0)
1153 return false;
1154 // Instructions with scale_offset modifier cannot be combined unless we
1155 // also generate a code to scale the offset and reset that bit.
1156 if (CI.CPol != Paired.CPol || (CI.CPol & AMDGPU::CPol::SCAL))
1157 return false;
1158 if (CI.InstClass == S_LOAD_IMM || CI.InstClass == S_BUFFER_LOAD_IMM ||
1159 CI.InstClass == S_BUFFER_LOAD_SGPR_IMM) {
1160 // Reject cases like:
1161 // dword + dwordx2 -> dwordx3
1162 // dword + dwordx3 -> dwordx4
1163 // If we tried to combine these cases, we would fail to extract a subreg
1164 // for the result of the second load due to SGPR alignment requirements.
1165 if (CI.Width != Paired.Width &&
1166 (CI.Width < Paired.Width) == (CI.Offset < Paired.Offset))
1167 return false;
1168 }
1169 return true;
1170 }
1171
1172 // If the offset in elements doesn't fit in 8-bits, we might be able to use
1173 // the stride 64 versions.
1174 if ((EltOffset0 % 64 == 0) && (EltOffset1 % 64) == 0 &&
1175 isUInt<8>(EltOffset0 / 64) && isUInt<8>(EltOffset1 / 64)) {
1176 if (Modify) {
1177 CI.Offset = EltOffset0 / 64;
1178 Paired.Offset = EltOffset1 / 64;
1179 CI.UseST64 = true;
1180 }
1181 return true;
1182 }
1183
1184 // Check if the new offsets fit in the reduced 8-bit range.
1185 if (isUInt<8>(EltOffset0) && isUInt<8>(EltOffset1)) {
1186 if (Modify) {
1187 CI.Offset = EltOffset0;
1188 Paired.Offset = EltOffset1;
1189 }
1190 return true;
1191 }
1192
1193 // Try to shift base address to decrease offsets.
1194 uint32_t Min = std::min(EltOffset0, EltOffset1);
1195 uint32_t Max = std::max(EltOffset0, EltOffset1);
1196
1197 const uint32_t Mask = maskTrailingOnes<uint32_t>(8) * 64;
1198 if (((Max - Min) & ~Mask) == 0) {
1199 if (Modify) {
1200 // From the range of values we could use for BaseOff, choose the one that
1201 // is aligned to the highest power of two, to maximise the chance that
1202 // the same offset can be reused for other load/store pairs.
1203 uint32_t BaseOff = mostAlignedValueInRange(Max - 0xff * 64, Min);
1204 // Copy the low bits of the offsets, so that when we adjust them by
1205 // subtracting BaseOff they will be multiples of 64.
1206 BaseOff |= Min & maskTrailingOnes<uint32_t>(6);
1207 CI.BaseOff = BaseOff * CI.EltSize;
1208 CI.Offset = (EltOffset0 - BaseOff) / 64;
1209 Paired.Offset = (EltOffset1 - BaseOff) / 64;
1210 CI.UseST64 = true;
1211 }
1212 return true;
1213 }
1214
1215 if (isUInt<8>(Max - Min)) {
1216 if (Modify) {
1217 // From the range of values we could use for BaseOff, choose the one that
1218 // is aligned to the highest power of two, to maximise the chance that
1219 // the same offset can be reused for other load/store pairs.
1220 uint32_t BaseOff = mostAlignedValueInRange(Max - 0xff, Min);
1221 CI.BaseOff = BaseOff * CI.EltSize;
1222 CI.Offset = EltOffset0 - BaseOff;
1223 Paired.Offset = EltOffset1 - BaseOff;
1224 }
1225 return true;
1226 }
1227
1228 return false;
1229}
1230
1231bool SILoadStoreOptimizer::widthsFit(const GCNSubtarget &STM,
1232 const CombineInfo &CI,
1233 const CombineInfo &Paired) {
1234 const unsigned Width = (CI.Width + Paired.Width);
1235 switch (CI.InstClass) {
1236 default:
1237 return (Width <= 4) && (STM.hasDwordx3LoadStores() || (Width != 3));
1238 case S_BUFFER_LOAD_IMM:
1239 case S_BUFFER_LOAD_SGPR_IMM:
1240 case S_LOAD_IMM:
1241 switch (Width) {
1242 default:
1243 return false;
1244 case 2:
1245 case 4:
1246 case 8:
1247 return true;
1248 case 3:
1249 return STM.hasScalarDwordx3Loads();
1250 }
1251 }
1252}
1253
1254const TargetRegisterClass *
1255SILoadStoreOptimizer::getDataRegClass(const MachineInstr &MI) const {
1256 if (const auto *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst)) {
1257 return TRI->getRegClassForReg(*MRI, Dst->getReg());
1258 }
1259 if (const auto *Src = TII->getNamedOperand(MI, AMDGPU::OpName::vdata)) {
1260 return TRI->getRegClassForReg(*MRI, Src->getReg());
1261 }
1262 if (const auto *Src = TII->getNamedOperand(MI, AMDGPU::OpName::data0)) {
1263 return TRI->getRegClassForReg(*MRI, Src->getReg());
1264 }
1265 if (const auto *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::sdst)) {
1266 return TRI->getRegClassForReg(*MRI, Dst->getReg());
1267 }
1268 if (const auto *Src = TII->getNamedOperand(MI, AMDGPU::OpName::sdata)) {
1269 return TRI->getRegClassForReg(*MRI, Src->getReg());
1270 }
1271 return nullptr;
1272}
1273
1274/// This function assumes that CI comes before Paired in a basic block. Return
1275/// an insertion point for the merged instruction or nullptr on failure.
1276SILoadStoreOptimizer::CombineInfo *
1277SILoadStoreOptimizer::checkAndPrepareMerge(CombineInfo &CI,
1278 CombineInfo &Paired) {
1279 // If another instruction has already been merged into CI, it may now be a
1280 // type that we can't do any further merging into.
1281 if (CI.InstClass == UNKNOWN || Paired.InstClass == UNKNOWN)
1282 return nullptr;
1283 assert(CI.InstClass == Paired.InstClass);
1284
1285 if (getInstSubclass(CI.I->getOpcode(), *TII) !=
1286 getInstSubclass(Paired.I->getOpcode(), *TII))
1287 return nullptr;
1288
1289 // Check both offsets (or masks for MIMG) can be combined and fit in the
1290 // reduced range.
1291 if (CI.InstClass == MIMG) {
1292 if (!dmasksCanBeCombined(CI, *TII, Paired))
1293 return nullptr;
1294 } else {
1295 if (!widthsFit(*STM, CI, Paired) || !offsetsCanBeCombined(CI, *STM, Paired))
1296 return nullptr;
1297 }
1298
1299 DenseSet<Register> RegDefs;
1300 DenseSet<Register> RegUses;
1301 CombineInfo *Where;
1302 if (CI.I->mayLoad()) {
1303 // Try to hoist Paired up to CI.
1304 addDefsUsesToList(*Paired.I, RegDefs, RegUses);
1305 for (MachineBasicBlock::iterator MBBI = Paired.I; --MBBI != CI.I;) {
1306 if (!canSwapInstructions(RegDefs, RegUses, *Paired.I, *MBBI))
1307 return nullptr;
1308 }
1309 Where = &CI;
1310 } else {
1311 // Try to sink CI down to Paired.
1312 addDefsUsesToList(*CI.I, RegDefs, RegUses);
1313 for (MachineBasicBlock::iterator MBBI = CI.I; ++MBBI != Paired.I;) {
1314 if (!canSwapInstructions(RegDefs, RegUses, *CI.I, *MBBI))
1315 return nullptr;
1316 }
1317 Where = &Paired;
1318 }
1319
1320 // Call offsetsCanBeCombined with modify = true so that the offsets are
1321 // correct for the new instruction. This should return true, because
1322 // this function should only be called on CombineInfo objects that
1323 // have already been confirmed to be mergeable.
1324 if (CI.InstClass == DS_READ || CI.InstClass == DS_WRITE) {
1325 if (STM->hasNeedsAligned2addrDS() &&
1326 (CI.I->memoperands_empty() ||
1327 (*CI.I->memoperands_begin())->getAlign().value() < CI.Width * 4))
1328 return nullptr;
1329 offsetsCanBeCombined(CI, *STM, Paired, true);
1330 }
1331
1332 if (CI.InstClass == DS_WRITE) {
1333 // Both data operands must be AGPR or VGPR, so the data registers needs to
1334 // be constrained to one or the other. We expect to only emit the VGPR form
1335 // here for now.
1336 //
1337 // FIXME: There is currently a hack in getRegClass to report that the write2
1338 // operands are VGPRs. In the future we should have separate agpr
1339 // instruction definitions.
1340 const MachineOperand *Data0 =
1341 TII->getNamedOperand(*CI.I, AMDGPU::OpName::data0);
1342 const MachineOperand *Data1 =
1343 TII->getNamedOperand(*Paired.I, AMDGPU::OpName::data0);
1344
1345 const MCInstrDesc &Write2Opc = TII->get(getWrite2Opcode(CI));
1346 int Data0Idx = AMDGPU::getNamedOperandIdx(Write2Opc.getOpcode(),
1347 AMDGPU::OpName::data0);
1348 int Data1Idx = AMDGPU::getNamedOperandIdx(Write2Opc.getOpcode(),
1349 AMDGPU::OpName::data1);
1350
1351 const TargetRegisterClass *DataRC0 = TII->getRegClass(Write2Opc, Data0Idx);
1352
1353 const TargetRegisterClass *DataRC1 = TII->getRegClass(Write2Opc, Data1Idx);
1354
1355 if (unsigned SubReg = Data0->getSubReg()) {
1356 DataRC0 = TRI->getMatchingSuperRegClass(MRI->getRegClass(Data0->getReg()),
1357 DataRC0, SubReg);
1358 }
1359
1360 if (unsigned SubReg = Data1->getSubReg()) {
1361 DataRC1 = TRI->getMatchingSuperRegClass(MRI->getRegClass(Data1->getReg()),
1362 DataRC1, SubReg);
1363 }
1364
1365 if (!MRI->constrainRegClass(Data0->getReg(), DataRC0) ||
1366 !MRI->constrainRegClass(Data1->getReg(), DataRC1))
1367 return nullptr;
1368
1369 // TODO: If one register can be constrained, and not the other, insert a
1370 // copy.
1371 }
1372
1373 return Where;
1374}
1375
1376// Copy the merged load result from DestReg to the original dest regs of CI and
1377// Paired.
1378void SILoadStoreOptimizer::copyToDestRegs(
1379 CombineInfo &CI, CombineInfo &Paired,
1380 MachineBasicBlock::iterator InsertBefore, const DebugLoc &DL,
1381 AMDGPU::OpName OpName, Register DestReg) const {
1382 MachineBasicBlock *MBB = CI.I->getParent();
1383
1384 auto [SubRegIdx0, SubRegIdx1] = getSubRegIdxs(CI, Paired);
1385
1386 // Copy to the old destination registers.
1387 const MCInstrDesc &CopyDesc = TII->get(TargetOpcode::COPY);
1388 auto *Dest0 = TII->getNamedOperand(*CI.I, OpName);
1389 auto *Dest1 = TII->getNamedOperand(*Paired.I, OpName);
1390
1391 // The constrained sload instructions in S_LOAD_IMM class will have
1392 // `early-clobber` flag in the dst operand. Remove the flag before using the
1393 // MOs in copies.
1394 Dest0->setIsEarlyClobber(false);
1395 Dest1->setIsEarlyClobber(false);
1396
1397 BuildMI(*MBB, InsertBefore, DL, CopyDesc)
1398 .add(*Dest0) // Copy to same destination including flags and sub reg.
1399 .addReg(DestReg, {}, SubRegIdx0);
1400 BuildMI(*MBB, InsertBefore, DL, CopyDesc)
1401 .add(*Dest1)
1402 .addReg(DestReg, RegState::Kill, SubRegIdx1);
1403}
1404
1405// Return a register for the source of the merged store after copying the
1406// original source regs of CI and Paired into it.
1408SILoadStoreOptimizer::copyFromSrcRegs(CombineInfo &CI, CombineInfo &Paired,
1409 MachineBasicBlock::iterator InsertBefore,
1410 const DebugLoc &DL,
1411 AMDGPU::OpName OpName) const {
1412 MachineBasicBlock *MBB = CI.I->getParent();
1413
1414 auto [SubRegIdx0, SubRegIdx1] = getSubRegIdxs(CI, Paired);
1415
1416 // Copy to the new source register.
1417 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1418 Register SrcReg = MRI->createVirtualRegister(SuperRC);
1419
1420 const auto *Src0 = TII->getNamedOperand(*CI.I, OpName);
1421 const auto *Src1 = TII->getNamedOperand(*Paired.I, OpName);
1422
1423 BuildMI(*MBB, InsertBefore, DL, TII->get(AMDGPU::REG_SEQUENCE), SrcReg)
1424 .add(*Src0)
1425 .addImm(SubRegIdx0)
1426 .add(*Src1)
1427 .addImm(SubRegIdx1);
1428
1429 return SrcReg;
1430}
1431
1432unsigned SILoadStoreOptimizer::read2Opcode(unsigned EltSize, bool GDS) const {
1433 if (GDS || STM->ldsRequiresM0Init())
1434 return (EltSize == 4) ? AMDGPU::DS_READ2_B32 : AMDGPU::DS_READ2_B64;
1435 return (EltSize == 4) ? AMDGPU::DS_READ2_B32_gfx9 : AMDGPU::DS_READ2_B64_gfx9;
1436}
1437
1438unsigned SILoadStoreOptimizer::read2ST64Opcode(unsigned EltSize,
1439 bool GDS) const {
1440 if (GDS || STM->ldsRequiresM0Init())
1441 return (EltSize == 4) ? AMDGPU::DS_READ2ST64_B32 : AMDGPU::DS_READ2ST64_B64;
1442
1443 return (EltSize == 4) ? AMDGPU::DS_READ2ST64_B32_gfx9
1444 : AMDGPU::DS_READ2ST64_B64_gfx9;
1445}
1446
1448SILoadStoreOptimizer::mergeRead2Pair(CombineInfo &CI, CombineInfo &Paired,
1449 MachineBasicBlock::iterator InsertBefore) {
1450 MachineBasicBlock *MBB = CI.I->getParent();
1451
1452 // Be careful, since the addresses could be subregisters themselves in weird
1453 // cases, like vectors of pointers.
1454 const auto *AddrReg = TII->getNamedOperand(*CI.I, AMDGPU::OpName::addr);
1455
1456 unsigned NewOffset0 = std::min(CI.Offset, Paired.Offset);
1457 unsigned NewOffset1 = std::max(CI.Offset, Paired.Offset);
1458 unsigned Opc = CI.UseST64 ? read2ST64Opcode(CI.EltSize, CI.GDS)
1459 : read2Opcode(CI.EltSize, CI.GDS);
1460
1461 assert((isUInt<8>(NewOffset0) && isUInt<8>(NewOffset1)) &&
1462 (NewOffset0 != NewOffset1) && "Computed offset doesn't fit");
1463
1464 const MCInstrDesc &Read2Desc = TII->get(Opc);
1465
1466 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1467 Register DestReg = MRI->createVirtualRegister(SuperRC);
1468
1469 DebugLoc DL =
1470 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1471
1472 Register BaseReg = AddrReg->getReg();
1473 unsigned BaseSubReg = AddrReg->getSubReg();
1474 RegState BaseRegFlags = {};
1475 if (CI.BaseOff) {
1476 Register ImmReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1477 BuildMI(*MBB, InsertBefore, DL, TII->get(AMDGPU::S_MOV_B32), ImmReg)
1478 .addImm(CI.BaseOff);
1479
1480 BaseReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1481 BaseRegFlags = RegState::Kill;
1482
1483 TII->getAddNoCarry(*MBB, InsertBefore, DL, BaseReg)
1484 .addReg(ImmReg)
1485 .addReg(AddrReg->getReg(), {}, BaseSubReg)
1486 .addImm(0); // clamp bit
1487 BaseSubReg = 0;
1488 }
1489
1490 MachineInstrBuilder Read2 =
1491 BuildMI(*MBB, InsertBefore, DL, Read2Desc, DestReg)
1492 .addReg(BaseReg, BaseRegFlags, BaseSubReg) // addr
1493 .addImm(NewOffset0) // offset0
1494 .addImm(NewOffset1) // offset1
1495 .addImm(CI.GDS) // gds
1496 .cloneMergedMemRefs({&*CI.I, &*Paired.I});
1497
1498 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdst, DestReg);
1499
1500 CI.I->eraseFromParent();
1501 Paired.I->eraseFromParent();
1502
1503 LLVM_DEBUG(dbgs() << "Inserted read2: " << *Read2 << '\n');
1504 return Read2;
1505}
1506
1507unsigned SILoadStoreOptimizer::write2Opcode(unsigned EltSize, bool GDS) const {
1508 if (GDS || STM->ldsRequiresM0Init())
1509 return (EltSize == 4) ? AMDGPU::DS_WRITE2_B32 : AMDGPU::DS_WRITE2_B64;
1510 return (EltSize == 4) ? AMDGPU::DS_WRITE2_B32_gfx9
1511 : AMDGPU::DS_WRITE2_B64_gfx9;
1512}
1513
1514unsigned SILoadStoreOptimizer::write2ST64Opcode(unsigned EltSize,
1515 bool GDS) const {
1516 if (GDS || STM->ldsRequiresM0Init())
1517 return (EltSize == 4) ? AMDGPU::DS_WRITE2ST64_B32
1518 : AMDGPU::DS_WRITE2ST64_B64;
1519
1520 return (EltSize == 4) ? AMDGPU::DS_WRITE2ST64_B32_gfx9
1521 : AMDGPU::DS_WRITE2ST64_B64_gfx9;
1522}
1523
1524unsigned SILoadStoreOptimizer::getWrite2Opcode(const CombineInfo &CI) const {
1525 return CI.UseST64 ? write2ST64Opcode(CI.EltSize, CI.GDS)
1526 : write2Opcode(CI.EltSize, CI.GDS);
1527}
1528
1529MachineBasicBlock::iterator SILoadStoreOptimizer::mergeWrite2Pair(
1530 CombineInfo &CI, CombineInfo &Paired,
1531 MachineBasicBlock::iterator InsertBefore) {
1532 MachineBasicBlock *MBB = CI.I->getParent();
1533
1534 // Be sure to use .addOperand(), and not .addReg() with these. We want to be
1535 // sure we preserve the subregister index and any register flags set on them.
1536 const MachineOperand *AddrReg =
1537 TII->getNamedOperand(*CI.I, AMDGPU::OpName::addr);
1538 const MachineOperand *Data0 =
1539 TII->getNamedOperand(*CI.I, AMDGPU::OpName::data0);
1540 const MachineOperand *Data1 =
1541 TII->getNamedOperand(*Paired.I, AMDGPU::OpName::data0);
1542
1543 unsigned NewOffset0 = CI.Offset;
1544 unsigned NewOffset1 = Paired.Offset;
1545 unsigned Opc = getWrite2Opcode(CI);
1546
1547 if (NewOffset0 > NewOffset1) {
1548 // Canonicalize the merged instruction so the smaller offset comes first.
1549 std::swap(NewOffset0, NewOffset1);
1550 std::swap(Data0, Data1);
1551 }
1552
1553 assert((isUInt<8>(NewOffset0) && isUInt<8>(NewOffset1)) &&
1554 (NewOffset0 != NewOffset1) && "Computed offset doesn't fit");
1555
1556 const MCInstrDesc &Write2Desc = TII->get(Opc);
1557 DebugLoc DL =
1558 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1559
1560 Register BaseReg = AddrReg->getReg();
1561 unsigned BaseSubReg = AddrReg->getSubReg();
1562 RegState BaseRegFlags = {};
1563 if (CI.BaseOff) {
1564 Register ImmReg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
1565 BuildMI(*MBB, InsertBefore, DL, TII->get(AMDGPU::S_MOV_B32), ImmReg)
1566 .addImm(CI.BaseOff);
1567
1568 BaseReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1569 BaseRegFlags = RegState::Kill;
1570
1571 TII->getAddNoCarry(*MBB, InsertBefore, DL, BaseReg)
1572 .addReg(ImmReg)
1573 .addReg(AddrReg->getReg(), {}, BaseSubReg)
1574 .addImm(0); // clamp bit
1575 BaseSubReg = 0;
1576 }
1577
1578 MachineInstrBuilder Write2 =
1579 BuildMI(*MBB, InsertBefore, DL, Write2Desc)
1580 .addReg(BaseReg, BaseRegFlags, BaseSubReg) // addr
1581 .add(*Data0) // data0
1582 .add(*Data1) // data1
1583 .addImm(NewOffset0) // offset0
1584 .addImm(NewOffset1) // offset1
1585 .addImm(CI.GDS) // gds
1586 .cloneMergedMemRefs({&*CI.I, &*Paired.I});
1587
1588 CI.I->eraseFromParent();
1589 Paired.I->eraseFromParent();
1590
1591 LLVM_DEBUG(dbgs() << "Inserted write2 inst: " << *Write2 << '\n');
1592 return Write2;
1593}
1594
1596SILoadStoreOptimizer::mergeImagePair(CombineInfo &CI, CombineInfo &Paired,
1597 MachineBasicBlock::iterator InsertBefore) {
1598 MachineBasicBlock *MBB = CI.I->getParent();
1599 DebugLoc DL =
1600 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1601
1602 const unsigned Opcode = getNewOpcode(CI, Paired);
1603
1604 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1605
1606 Register DestReg = MRI->createVirtualRegister(SuperRC);
1607 unsigned MergedDMask = CI.DMask | Paired.DMask;
1608 unsigned DMaskIdx =
1609 AMDGPU::getNamedOperandIdx(CI.I->getOpcode(), AMDGPU::OpName::dmask);
1610
1611 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode), DestReg);
1612 for (unsigned I = 1, E = (*CI.I).getNumOperands(); I != E; ++I) {
1613 if (I == DMaskIdx)
1614 MIB.addImm(MergedDMask);
1615 else
1616 MIB.add((*CI.I).getOperand(I));
1617 }
1618
1619 // It shouldn't be possible to get this far if the two instructions
1620 // don't have a single memoperand, because MachineInstr::mayAlias()
1621 // will return true if this is the case.
1622 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
1623
1624 MachineInstr *New = MIB.addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1625
1626 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata, DestReg);
1627
1628 CI.I->eraseFromParent();
1629 Paired.I->eraseFromParent();
1630 return New;
1631}
1632
1633MachineBasicBlock::iterator SILoadStoreOptimizer::mergeSMemLoadImmPair(
1634 CombineInfo &CI, CombineInfo &Paired,
1635 MachineBasicBlock::iterator InsertBefore) {
1636 MachineBasicBlock *MBB = CI.I->getParent();
1637 DebugLoc DL =
1638 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1639
1640 const unsigned Opcode = getNewOpcode(CI, Paired);
1641
1642 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1643
1644 Register DestReg = MRI->createVirtualRegister(SuperRC);
1645 unsigned MergedOffset = std::min(CI.Offset, Paired.Offset);
1646
1647 // It shouldn't be possible to get this far if the two instructions
1648 // don't have a single memoperand, because MachineInstr::mayAlias()
1649 // will return true if this is the case.
1650 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
1651
1652 MachineInstrBuilder New =
1653 BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode), DestReg)
1654 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::sbase));
1655 if (CI.InstClass == S_BUFFER_LOAD_SGPR_IMM)
1656 New.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::soffset));
1657 New.addImm(MergedOffset);
1658 New.addImm(CI.CPol).addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1659
1660 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::sdst, DestReg);
1661
1662 CI.I->eraseFromParent();
1663 Paired.I->eraseFromParent();
1664 return New;
1665}
1666
1667MachineBasicBlock::iterator SILoadStoreOptimizer::mergeBufferLoadPair(
1668 CombineInfo &CI, CombineInfo &Paired,
1669 MachineBasicBlock::iterator InsertBefore) {
1670 MachineBasicBlock *MBB = CI.I->getParent();
1671
1672 DebugLoc DL =
1673 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1674
1675 const unsigned Opcode = getNewOpcode(CI, Paired);
1676
1677 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1678
1679 // Copy to the new source register.
1680 Register DestReg = MRI->createVirtualRegister(SuperRC);
1681 unsigned MergedOffset = std::min(CI.Offset, Paired.Offset);
1682
1683 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode), DestReg);
1684
1685 AddressRegs Regs = getRegs(Opcode, *TII);
1686
1687 if (Regs.VAddr)
1688 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr));
1689
1690 // It shouldn't be possible to get this far if the two instructions
1691 // don't have a single memoperand, because MachineInstr::mayAlias()
1692 // will return true if this is the case.
1693 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
1694
1695 MachineInstr *New =
1696 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::srsrc))
1697 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::soffset))
1698 .addImm(MergedOffset) // offset
1699 .addImm(CI.CPol) // cpol
1700 .addImm(0) // swz
1701 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1702
1703 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata, DestReg);
1704
1705 CI.I->eraseFromParent();
1706 Paired.I->eraseFromParent();
1707 return New;
1708}
1709
1710MachineBasicBlock::iterator SILoadStoreOptimizer::mergeTBufferLoadPair(
1711 CombineInfo &CI, CombineInfo &Paired,
1712 MachineBasicBlock::iterator InsertBefore) {
1713 MachineBasicBlock *MBB = CI.I->getParent();
1714
1715 DebugLoc DL =
1716 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1717
1718 const unsigned Opcode = getNewOpcode(CI, Paired);
1719
1720 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1721
1722 // Copy to the new source register.
1723 Register DestReg = MRI->createVirtualRegister(SuperRC);
1724 unsigned MergedOffset = std::min(CI.Offset, Paired.Offset);
1725
1726 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode), DestReg);
1727
1728 AddressRegs Regs = getRegs(Opcode, *TII);
1729
1730 if (Regs.VAddr)
1731 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr));
1732
1733 // For 8-bit or 16-bit tbuffer formats there is no 3-component encoding.
1734 // If the combined count is 3 (e.g. X+X+X or XY+X), promote to 4 components
1735 // and use XYZ of XYZW to enable the merge.
1736 unsigned NumCombinedComponents = CI.Width + Paired.Width;
1737 if (NumCombinedComponents == 3 && CI.EltSize <= 2)
1738 NumCombinedComponents = 4;
1739 unsigned JoinedFormat =
1740 getBufferFormatWithCompCount(CI.Format, NumCombinedComponents, *STM);
1741
1742 // It shouldn't be possible to get this far if the two instructions
1743 // don't have a single memoperand, because MachineInstr::mayAlias()
1744 // will return true if this is the case.
1745 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
1746
1747 MachineInstr *New =
1748 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::srsrc))
1749 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::soffset))
1750 .addImm(MergedOffset) // offset
1751 .addImm(JoinedFormat) // format
1752 .addImm(CI.CPol) // cpol
1753 .addImm(0) // swz
1754 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1755
1756 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata, DestReg);
1757
1758 CI.I->eraseFromParent();
1759 Paired.I->eraseFromParent();
1760 return New;
1761}
1762
1763MachineBasicBlock::iterator SILoadStoreOptimizer::mergeTBufferStorePair(
1764 CombineInfo &CI, CombineInfo &Paired,
1765 MachineBasicBlock::iterator InsertBefore) {
1766 MachineBasicBlock *MBB = CI.I->getParent();
1767 DebugLoc DL =
1768 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1769
1770 const unsigned Opcode = getNewOpcode(CI, Paired);
1771
1772 Register SrcReg =
1773 copyFromSrcRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata);
1774
1775 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode))
1776 .addReg(SrcReg, RegState::Kill);
1777
1778 AddressRegs Regs = getRegs(Opcode, *TII);
1779
1780 if (Regs.VAddr)
1781 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr));
1782
1783 // For 8-bit or 16-bit tbuffer formats there is no 3-component encoding.
1784 // If the combined count is 3 (e.g. X+X+X or XY+X), promote to 4 components
1785 // and use XYZ of XYZW to enable the merge.
1786 unsigned NumCombinedComponents = CI.Width + Paired.Width;
1787 if (NumCombinedComponents == 3 && CI.EltSize <= 2)
1788 NumCombinedComponents = 4;
1789 unsigned JoinedFormat =
1790 getBufferFormatWithCompCount(CI.Format, NumCombinedComponents, *STM);
1791
1792 // It shouldn't be possible to get this far if the two instructions
1793 // don't have a single memoperand, because MachineInstr::mayAlias()
1794 // will return true if this is the case.
1795 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
1796
1797 MachineInstr *New =
1798 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::srsrc))
1799 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::soffset))
1800 .addImm(std::min(CI.Offset, Paired.Offset)) // offset
1801 .addImm(JoinedFormat) // format
1802 .addImm(CI.CPol) // cpol
1803 .addImm(0) // swz
1804 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1805
1806 CI.I->eraseFromParent();
1807 Paired.I->eraseFromParent();
1808 return New;
1809}
1810
1811MachineBasicBlock::iterator SILoadStoreOptimizer::mergeFlatLoadPair(
1812 CombineInfo &CI, CombineInfo &Paired,
1813 MachineBasicBlock::iterator InsertBefore) {
1814 MachineBasicBlock *MBB = CI.I->getParent();
1815
1816 DebugLoc DL =
1817 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1818
1819 const unsigned Opcode = getNewOpcode(CI, Paired);
1820
1821 const TargetRegisterClass *SuperRC = getTargetRegisterClass(CI, Paired);
1822 Register DestReg = MRI->createVirtualRegister(SuperRC);
1823
1824 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode), DestReg);
1825
1826 if (auto *SAddr = TII->getNamedOperand(*CI.I, AMDGPU::OpName::saddr))
1827 MIB.add(*SAddr);
1828
1829 MachineInstr *New =
1830 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr))
1831 .addImm(std::min(CI.Offset, Paired.Offset))
1832 .addImm(CI.CPol)
1833 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1834
1835 copyToDestRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdst, DestReg);
1836
1837 CI.I->eraseFromParent();
1838 Paired.I->eraseFromParent();
1839 return New;
1840}
1841
1842MachineBasicBlock::iterator SILoadStoreOptimizer::mergeFlatStorePair(
1843 CombineInfo &CI, CombineInfo &Paired,
1844 MachineBasicBlock::iterator InsertBefore) {
1845 MachineBasicBlock *MBB = CI.I->getParent();
1846
1847 DebugLoc DL =
1848 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
1849
1850 const unsigned Opcode = getNewOpcode(CI, Paired);
1851
1852 Register SrcReg =
1853 copyFromSrcRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata);
1854
1855 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode))
1856 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr))
1857 .addReg(SrcReg, RegState::Kill);
1858
1859 if (auto *SAddr = TII->getNamedOperand(*CI.I, AMDGPU::OpName::saddr))
1860 MIB.add(*SAddr);
1861
1862 MachineInstr *New =
1863 MIB.addImm(std::min(CI.Offset, Paired.Offset))
1864 .addImm(CI.CPol)
1865 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
1866
1867 CI.I->eraseFromParent();
1868 Paired.I->eraseFromParent();
1869 return New;
1870}
1871
1874 unsigned Width) {
1875 // Conservatively returns true if not found the MMO.
1876 return STM.isXNACKEnabled() &&
1877 (MMOs.size() != 1 || MMOs[0]->getAlign().value() < Width * 4);
1878}
1879
1880unsigned SILoadStoreOptimizer::getNewOpcode(const CombineInfo &CI,
1881 const CombineInfo &Paired) {
1882 const unsigned Width = CI.Width + Paired.Width;
1883 const CombineInfo &Leading = Paired < CI ? Paired : CI;
1884 // If XNACK is enabled, use the constrained opcodes when the first load is
1885 // under-aligned.
1886 const bool NeedsConstrainedOpc =
1887 needsConstrainedOpcode(*STM, Leading.I->memoperands(), Width);
1888
1889 switch (getCommonInstClass(CI, Paired)) {
1890 default:
1891 assert(CI.InstClass == BUFFER_LOAD || CI.InstClass == BUFFER_STORE);
1892 // FIXME: Handle d16 correctly
1893 return AMDGPU::getMUBUFOpcode(AMDGPU::getMUBUFBaseOpcode(CI.I->getOpcode()),
1894 Width);
1895 case TBUFFER_LOAD:
1896 case TBUFFER_STORE:
1897 return AMDGPU::getMTBUFOpcode(AMDGPU::getMTBUFBaseOpcode(CI.I->getOpcode()),
1898 Width);
1899
1900 case UNKNOWN:
1901 llvm_unreachable("Unknown instruction class");
1902 case S_BUFFER_LOAD_IMM: {
1903 switch (Width) {
1904 default:
1905 return 0;
1906 case 2:
1907 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM_ec
1908 : AMDGPU::S_BUFFER_LOAD_DWORDX2_IMM;
1909 case 3:
1910 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM_ec
1911 : AMDGPU::S_BUFFER_LOAD_DWORDX3_IMM;
1912 case 4:
1913 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM_ec
1914 : AMDGPU::S_BUFFER_LOAD_DWORDX4_IMM;
1915 case 8:
1916 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM_ec
1917 : AMDGPU::S_BUFFER_LOAD_DWORDX8_IMM;
1918 }
1919 }
1920 case S_BUFFER_LOAD_SGPR_IMM: {
1921 switch (Width) {
1922 default:
1923 return 0;
1924 case 2:
1925 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM_ec
1926 : AMDGPU::S_BUFFER_LOAD_DWORDX2_SGPR_IMM;
1927 case 3:
1928 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM_ec
1929 : AMDGPU::S_BUFFER_LOAD_DWORDX3_SGPR_IMM;
1930 case 4:
1931 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM_ec
1932 : AMDGPU::S_BUFFER_LOAD_DWORDX4_SGPR_IMM;
1933 case 8:
1934 return NeedsConstrainedOpc ? AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM_ec
1935 : AMDGPU::S_BUFFER_LOAD_DWORDX8_SGPR_IMM;
1936 }
1937 }
1938 case S_LOAD_IMM: {
1939 switch (Width) {
1940 default:
1941 return 0;
1942 case 2:
1943 return NeedsConstrainedOpc ? AMDGPU::S_LOAD_DWORDX2_IMM_ec
1944 : AMDGPU::S_LOAD_DWORDX2_IMM;
1945 case 3:
1946 return NeedsConstrainedOpc ? AMDGPU::S_LOAD_DWORDX3_IMM_ec
1947 : AMDGPU::S_LOAD_DWORDX3_IMM;
1948 case 4:
1949 return NeedsConstrainedOpc ? AMDGPU::S_LOAD_DWORDX4_IMM_ec
1950 : AMDGPU::S_LOAD_DWORDX4_IMM;
1951 case 8:
1952 return NeedsConstrainedOpc ? AMDGPU::S_LOAD_DWORDX8_IMM_ec
1953 : AMDGPU::S_LOAD_DWORDX8_IMM;
1954 }
1955 }
1956 case GLOBAL_LOAD:
1957 switch (Width) {
1958 default:
1959 return 0;
1960 case 2:
1961 return AMDGPU::GLOBAL_LOAD_DWORDX2;
1962 case 3:
1963 return AMDGPU::GLOBAL_LOAD_DWORDX3;
1964 case 4:
1965 return AMDGPU::GLOBAL_LOAD_DWORDX4;
1966 }
1967 case GLOBAL_LOAD_SADDR:
1968 switch (Width) {
1969 default:
1970 return 0;
1971 case 2:
1972 return AMDGPU::GLOBAL_LOAD_DWORDX2_SADDR;
1973 case 3:
1974 return AMDGPU::GLOBAL_LOAD_DWORDX3_SADDR;
1975 case 4:
1976 return AMDGPU::GLOBAL_LOAD_DWORDX4_SADDR;
1977 }
1978 case GLOBAL_STORE:
1979 switch (Width) {
1980 default:
1981 return 0;
1982 case 2:
1983 return AMDGPU::GLOBAL_STORE_DWORDX2;
1984 case 3:
1985 return AMDGPU::GLOBAL_STORE_DWORDX3;
1986 case 4:
1987 return AMDGPU::GLOBAL_STORE_DWORDX4;
1988 }
1989 case GLOBAL_STORE_SADDR:
1990 switch (Width) {
1991 default:
1992 return 0;
1993 case 2:
1994 return AMDGPU::GLOBAL_STORE_DWORDX2_SADDR;
1995 case 3:
1996 return AMDGPU::GLOBAL_STORE_DWORDX3_SADDR;
1997 case 4:
1998 return AMDGPU::GLOBAL_STORE_DWORDX4_SADDR;
1999 }
2000 case FLAT_LOAD:
2001 switch (Width) {
2002 default:
2003 return 0;
2004 case 2:
2005 return AMDGPU::FLAT_LOAD_DWORDX2;
2006 case 3:
2007 return AMDGPU::FLAT_LOAD_DWORDX3;
2008 case 4:
2009 return AMDGPU::FLAT_LOAD_DWORDX4;
2010 }
2011 case FLAT_STORE:
2012 switch (Width) {
2013 default:
2014 return 0;
2015 case 2:
2016 return AMDGPU::FLAT_STORE_DWORDX2;
2017 case 3:
2018 return AMDGPU::FLAT_STORE_DWORDX3;
2019 case 4:
2020 return AMDGPU::FLAT_STORE_DWORDX4;
2021 }
2022 case FLAT_LOAD_SADDR:
2023 switch (Width) {
2024 default:
2025 return 0;
2026 case 2:
2027 return AMDGPU::FLAT_LOAD_DWORDX2_SADDR;
2028 case 3:
2029 return AMDGPU::FLAT_LOAD_DWORDX3_SADDR;
2030 case 4:
2031 return AMDGPU::FLAT_LOAD_DWORDX4_SADDR;
2032 }
2033 case FLAT_STORE_SADDR:
2034 switch (Width) {
2035 default:
2036 return 0;
2037 case 2:
2038 return AMDGPU::FLAT_STORE_DWORDX2_SADDR;
2039 case 3:
2040 return AMDGPU::FLAT_STORE_DWORDX3_SADDR;
2041 case 4:
2042 return AMDGPU::FLAT_STORE_DWORDX4_SADDR;
2043 }
2044 case MIMG:
2045 assert(((unsigned)llvm::popcount(CI.DMask | Paired.DMask) == Width) &&
2046 "No overlaps");
2047 return AMDGPU::getMaskedMIMGOp(CI.I->getOpcode(), Width);
2048 }
2049}
2050
2051std::pair<unsigned, unsigned>
2052SILoadStoreOptimizer::getSubRegIdxs(const CombineInfo &CI,
2053 const CombineInfo &Paired) {
2054 assert((CI.InstClass != MIMG ||
2055 ((unsigned)llvm::popcount(CI.DMask | Paired.DMask) ==
2056 CI.Width + Paired.Width)) &&
2057 "No overlaps");
2058
2059 unsigned Idx0;
2060 unsigned Idx1;
2061
2062 static const unsigned Idxs[5][4] = {
2063 {AMDGPU::sub0, AMDGPU::sub0_sub1, AMDGPU::sub0_sub1_sub2, AMDGPU::sub0_sub1_sub2_sub3},
2064 {AMDGPU::sub1, AMDGPU::sub1_sub2, AMDGPU::sub1_sub2_sub3, AMDGPU::sub1_sub2_sub3_sub4},
2065 {AMDGPU::sub2, AMDGPU::sub2_sub3, AMDGPU::sub2_sub3_sub4, AMDGPU::sub2_sub3_sub4_sub5},
2066 {AMDGPU::sub3, AMDGPU::sub3_sub4, AMDGPU::sub3_sub4_sub5, AMDGPU::sub3_sub4_sub5_sub6},
2067 {AMDGPU::sub4, AMDGPU::sub4_sub5, AMDGPU::sub4_sub5_sub6, AMDGPU::sub4_sub5_sub6_sub7},
2068 };
2069
2070 assert(CI.Width >= 1 && CI.Width <= 4);
2071 assert(Paired.Width >= 1 && Paired.Width <= 4);
2072
2073 if (Paired < CI) {
2074 Idx1 = Idxs[0][Paired.Width - 1];
2075 Idx0 = Idxs[Paired.Width][CI.Width - 1];
2076 } else {
2077 Idx0 = Idxs[0][CI.Width - 1];
2078 Idx1 = Idxs[CI.Width][Paired.Width - 1];
2079 }
2080
2081 return {Idx0, Idx1};
2082}
2083
2084const TargetRegisterClass *
2085SILoadStoreOptimizer::getTargetRegisterClass(const CombineInfo &CI,
2086 const CombineInfo &Paired) const {
2087 if (CI.InstClass == S_BUFFER_LOAD_IMM ||
2088 CI.InstClass == S_BUFFER_LOAD_SGPR_IMM || CI.InstClass == S_LOAD_IMM) {
2089 switch (CI.Width + Paired.Width) {
2090 default:
2091 return nullptr;
2092 case 2:
2093 return &AMDGPU::SReg_64_XEXECRegClass;
2094 case 3:
2095 return &AMDGPU::SGPR_96RegClass;
2096 case 4:
2097 return &AMDGPU::SGPR_128RegClass;
2098 case 8:
2099 return &AMDGPU::SGPR_256RegClass;
2100 case 16:
2101 return &AMDGPU::SGPR_512RegClass;
2102 }
2103 }
2104
2105 // FIXME: This should compute the instruction to use, and then use the result
2106 // of TII->getRegClass.
2107 unsigned BitWidth = 32 * (CI.Width + Paired.Width);
2108 return TRI->isAGPRClass(getDataRegClass(*CI.I))
2109 ? TRI->getAGPRClassForBitWidth(BitWidth)
2110 : TRI->getVGPRClassForBitWidth(BitWidth);
2111}
2112
2113MachineBasicBlock::iterator SILoadStoreOptimizer::mergeBufferStorePair(
2114 CombineInfo &CI, CombineInfo &Paired,
2115 MachineBasicBlock::iterator InsertBefore) {
2116 MachineBasicBlock *MBB = CI.I->getParent();
2117 DebugLoc DL =
2118 DebugLoc::getMergedLocation(CI.I->getDebugLoc(), Paired.I->getDebugLoc());
2119
2120 const unsigned Opcode = getNewOpcode(CI, Paired);
2121
2122 Register SrcReg =
2123 copyFromSrcRegs(CI, Paired, InsertBefore, DL, AMDGPU::OpName::vdata);
2124
2125 auto MIB = BuildMI(*MBB, InsertBefore, DL, TII->get(Opcode))
2126 .addReg(SrcReg, RegState::Kill);
2127
2128 AddressRegs Regs = getRegs(Opcode, *TII);
2129
2130 if (Regs.VAddr)
2131 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::vaddr));
2132
2133
2134 // It shouldn't be possible to get this far if the two instructions
2135 // don't have a single memoperand, because MachineInstr::mayAlias()
2136 // will return true if this is the case.
2137 assert(CI.I->hasOneMemOperand() && Paired.I->hasOneMemOperand());
2138
2139 MachineInstr *New =
2140 MIB.add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::srsrc))
2141 .add(*TII->getNamedOperand(*CI.I, AMDGPU::OpName::soffset))
2142 .addImm(std::min(CI.Offset, Paired.Offset)) // offset
2143 .addImm(CI.CPol) // cpol
2144 .addImm(0) // swz
2145 .addMemOperand(combineKnownAdjacentMMOs(CI, Paired));
2146
2147 CI.I->eraseFromParent();
2148 Paired.I->eraseFromParent();
2149 return New;
2150}
2151
2152MachineOperand
2153SILoadStoreOptimizer::createRegOrImm(int32_t Val, MachineInstr &MI) const {
2154 APInt V(32, Val, true);
2155 if (TII->isInlineConstant(V))
2156 return MachineOperand::CreateImm(Val);
2157
2158 Register Reg = MRI->createVirtualRegister(&AMDGPU::SReg_32RegClass);
2159 MachineInstr *Mov =
2160 BuildMI(*MI.getParent(), MI.getIterator(), MI.getDebugLoc(),
2161 TII->get(AMDGPU::S_MOV_B32), Reg)
2162 .addImm(Val);
2163 (void)Mov;
2164 LLVM_DEBUG(dbgs() << " "; Mov->dump());
2165 return MachineOperand::CreateReg(Reg, false);
2166}
2167
2168// Compute base address using Addr and return the final register.
2169Register SILoadStoreOptimizer::computeBase(MachineInstr &MI,
2170 const MemAddress &Addr) const {
2171 MachineBasicBlock *MBB = MI.getParent();
2173 const DebugLoc &DL = MI.getDebugLoc();
2174
2175 LLVM_DEBUG(dbgs() << " Re-Computed Anchor-Base:\n");
2176
2177 // Use V_ADD_U64_e64 when the original pattern used it (gfx1250+)
2178 if (Addr.Base.UseV64Pattern) {
2179 Register FullDestReg = MRI->createVirtualRegister(
2180 TII->getRegClass(TII->get(AMDGPU::V_ADD_U64_e64), 0));
2181
2182 // Load the 64-bit offset into an SGPR pair if needed
2183 Register OffsetReg = MRI->createVirtualRegister(&AMDGPU::SReg_64RegClass);
2184 MachineInstr *MovOffset =
2185 BuildMI(*MBB, MBBI, DL, TII->get(AMDGPU::S_MOV_B64_IMM_PSEUDO),
2186 OffsetReg)
2187 .addImm(Addr.Offset);
2188 MachineInstr *Add64 =
2189 BuildMI(*MBB, MBBI, DL, TII->get(AMDGPU::V_ADD_U64_e64), FullDestReg)
2190 .addReg(Addr.Base.LoReg)
2191 .addReg(OffsetReg, RegState::Kill)
2192 .addImm(0);
2193 (void)MovOffset;
2194 (void)Add64;
2195 LLVM_DEBUG(dbgs() << " " << *MovOffset << "\n";
2196 dbgs() << " " << *Add64 << "\n\n";);
2197
2198 return FullDestReg;
2199 }
2200
2201 // Original carry-chain pattern (V_ADD_CO_U32 + V_ADDC_U32)
2202 assert((TRI->getRegSizeInBits(Addr.Base.LoReg, *MRI) == 32 ||
2203 Addr.Base.LoSubReg) &&
2204 "Expected 32-bit Base-Register-Low!!");
2205
2206 assert((TRI->getRegSizeInBits(Addr.Base.HiReg, *MRI) == 32 ||
2207 Addr.Base.HiSubReg) &&
2208 "Expected 32-bit Base-Register-Hi!!");
2209
2210 MachineOperand OffsetLo = createRegOrImm(static_cast<int32_t>(Addr.Offset), MI);
2211 MachineOperand OffsetHi =
2212 createRegOrImm(static_cast<int32_t>(Addr.Offset >> 32), MI);
2213
2214 const auto *CarryRC = TRI->getWaveMaskRegClass();
2215 Register CarryReg = MRI->createVirtualRegister(CarryRC);
2216 Register DeadCarryReg = MRI->createVirtualRegister(CarryRC);
2217
2218 Register DestSub0 = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2219 Register DestSub1 = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
2220 MachineInstr *LoHalf =
2221 BuildMI(*MBB, MBBI, DL, TII->get(AMDGPU::V_ADD_CO_U32_e64), DestSub0)
2222 .addReg(CarryReg, RegState::Define)
2223 .addReg(Addr.Base.LoReg, {}, Addr.Base.LoSubReg)
2224 .add(OffsetLo)
2225 .addImm(0); // clamp bit
2226
2227 MachineInstr *HiHalf =
2228 BuildMI(*MBB, MBBI, DL, TII->get(AMDGPU::V_ADDC_U32_e64), DestSub1)
2229 .addReg(DeadCarryReg, RegState::Define | RegState::Dead)
2230 .addReg(Addr.Base.HiReg, {}, Addr.Base.HiSubReg)
2231 .add(OffsetHi)
2232 .addReg(CarryReg, RegState::Kill)
2233 .addImm(0); // clamp bit
2234
2235 Register FullDestReg = MRI->createVirtualRegister(TRI->getVGPR64Class());
2236 MachineInstr *FullBase =
2237 BuildMI(*MBB, MBBI, DL, TII->get(TargetOpcode::REG_SEQUENCE), FullDestReg)
2238 .addReg(DestSub0)
2239 .addImm(AMDGPU::sub0)
2240 .addReg(DestSub1)
2241 .addImm(AMDGPU::sub1);
2242
2243 (void)LoHalf;
2244 (void)HiHalf;
2245 (void)FullBase;
2246 LLVM_DEBUG(dbgs() << " " << *LoHalf << "\n";
2247 dbgs() << " " << *HiHalf << "\n";
2248 dbgs() << " " << *FullBase << "\n\n";);
2249
2250 return FullDestReg;
2251}
2252
2253// Update base and offset with the NewBase and NewOffset in MI.
2254void SILoadStoreOptimizer::updateBaseAndOffset(MachineInstr &MI,
2255 Register NewBase,
2256 int32_t NewOffset) const {
2257 auto *Base = TII->getNamedOperand(MI, AMDGPU::OpName::vaddr);
2258 Base->setReg(NewBase);
2259 Base->setIsKill(false);
2260 TII->getNamedOperand(MI, AMDGPU::OpName::offset)->setImm(NewOffset);
2261}
2262
2263// Helper to extract a 64-bit constant offset from a V_ADD_U64_e64 instruction.
2264// Returns true if successful, populating Addr with base register info and
2265// offset.
2266bool SILoadStoreOptimizer::processBaseWithConstOffset64(
2267 MachineInstr *AddDef, const MachineOperand &Base, MemAddress &Addr) const {
2268 if (!Base.isReg())
2269 return false;
2270
2271 MachineOperand *Src0 = TII->getNamedOperand(*AddDef, AMDGPU::OpName::src0);
2272 MachineOperand *Src1 = TII->getNamedOperand(*AddDef, AMDGPU::OpName::src1);
2273
2274 const MachineOperand *BaseOp = nullptr;
2275
2276 auto Offset = TII->getImmOrMaterializedImm(*MRI, *Src1);
2277
2278 if (Offset) {
2279 BaseOp = Src0;
2280 Addr.Offset = *Offset;
2281 } else {
2282 // Both or neither are constants - can't handle this pattern
2283 return false;
2284 }
2285
2286 // Now extract the base register (which should be a 64-bit VGPR).
2287 Addr.Base.LoReg = BaseOp->getReg();
2288 Addr.Base.UseV64Pattern = true;
2289 return true;
2290}
2291
2292// Analyze Base and extracts:
2293// - 32bit base registers, subregisters
2294// - 64bit constant offset
2295// Expecting base computation as:
2296// %OFFSET0:sgpr_32 = S_MOV_B32 8000
2297// %LO:vgpr_32, %c:sreg_64_xexec =
2298// V_ADD_CO_U32_e64 %BASE_LO:vgpr_32, %103:sgpr_32,
2299// %HI:vgpr_32, = V_ADDC_U32_e64 %BASE_HI:vgpr_32, 0, killed %c:sreg_64_xexec
2300// %Base:vreg_64 =
2301// REG_SEQUENCE %LO:vgpr_32, %subreg.sub0, %HI:vgpr_32, %subreg.sub1
2302//
2303// Also handles V_ADD_U64_e64 pattern (gfx1250+):
2304// %OFFSET:sreg_64 = S_MOV_B64_IMM_PSEUDO 256
2305// %Base:vreg_64 = V_ADD_U64_e64 %BASE:vreg_64, %OFFSET:sreg_64, 0
2306void SILoadStoreOptimizer::processBaseWithConstOffset(const MachineOperand &Base,
2307 MemAddress &Addr) const {
2308 if (!Base.isReg())
2309 return;
2310
2311 MachineInstr *Def = MRI->getUniqueVRegDef(Base.getReg());
2312 if (!Def)
2313 return;
2314
2315 // Try V_ADD_U64_e64 pattern first (simpler, used on gfx1250+)
2316 if (Def->getOpcode() == AMDGPU::V_ADD_U64_e64) {
2317 if (processBaseWithConstOffset64(Def, Base, Addr))
2318 return;
2319 }
2320
2321 // Fall through to REG_SEQUENCE + V_ADD_CO_U32 + V_ADDC_U32 pattern
2322 if (Def->getOpcode() != AMDGPU::REG_SEQUENCE || Def->getNumOperands() != 5)
2323 return;
2324
2325 MachineOperand BaseLo = Def->getOperand(1);
2326 MachineOperand BaseHi = Def->getOperand(3);
2327 if (!BaseLo.isReg() || !BaseHi.isReg())
2328 return;
2329
2330 MachineInstr *BaseLoDef = MRI->getUniqueVRegDef(BaseLo.getReg());
2331 MachineInstr *BaseHiDef = MRI->getUniqueVRegDef(BaseHi.getReg());
2332
2333 if (!BaseLoDef || BaseLoDef->getOpcode() != AMDGPU::V_ADD_CO_U32_e64 ||
2334 !BaseHiDef || BaseHiDef->getOpcode() != AMDGPU::V_ADDC_U32_e64)
2335 return;
2336
2337 MachineOperand *Src0 = TII->getNamedOperand(*BaseLoDef, AMDGPU::OpName::src0);
2338 MachineOperand *Src1 = TII->getNamedOperand(*BaseLoDef, AMDGPU::OpName::src1);
2339
2340 auto Offset0P = TII->getImmOrMaterializedImm(*MRI, *Src0);
2341 if (Offset0P)
2342 BaseLo = *Src1;
2343 else {
2344 if (!(Offset0P = TII->getImmOrMaterializedImm(*MRI, *Src1)))
2345 return;
2346 BaseLo = *Src0;
2347 }
2348
2349 if (!BaseLo.isReg())
2350 return;
2351
2352 Src0 = TII->getNamedOperand(*BaseHiDef, AMDGPU::OpName::src0);
2353 Src1 = TII->getNamedOperand(*BaseHiDef, AMDGPU::OpName::src1);
2354
2355 if (Src0->isImm())
2356 std::swap(Src0, Src1);
2357
2358 if (!Src1->isImm() || Src0->isImm())
2359 return;
2360
2361 uint64_t Offset1 = Src1->getImm();
2362 BaseHi = *Src0;
2363
2364 if (!BaseHi.isReg())
2365 return;
2366
2367 Addr.Base.LoReg = BaseLo.getReg();
2368 Addr.Base.HiReg = BaseHi.getReg();
2369 Addr.Base.LoSubReg = BaseLo.getSubReg();
2370 Addr.Base.HiSubReg = BaseHi.getSubReg();
2371 Addr.Offset = (*Offset0P & 0x00000000ffffffff) | (Offset1 << 32);
2372}
2373
2374// Maintain the correct LDS address for async loads and stores.
2375// It becomes incorrect when promoteConstantOffsetToImm adds an offset only
2376// meant for the global address operand. For async loads the LDS address is in
2377// vdst. For async stores, the LDS address is in vdata.
2378void SILoadStoreOptimizer::updateAsyncLDSAddress(MachineInstr &MI,
2379 int32_t OffsetDiff) const {
2380 if (!TII->usesASYNC_CNT(MI) || OffsetDiff == 0)
2381 return;
2382
2383 MachineOperand *LDSAddr = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
2384 if (!LDSAddr)
2385 LDSAddr = TII->getNamedOperand(MI, AMDGPU::OpName::vdata);
2386 assert(LDSAddr);
2387
2388 Register OldReg = LDSAddr->getReg();
2389 Register NewReg = MRI->createVirtualRegister(MRI->getRegClass(OldReg));
2390 MachineBasicBlock &MBB = *MI.getParent();
2391 const DebugLoc &DL = MI.getDebugLoc();
2392 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_ADD_U32_e64), NewReg)
2393 .addReg(OldReg)
2394 .addImm(-OffsetDiff)
2395 .addImm(0);
2396
2397 LDSAddr->setReg(NewReg);
2398}
2399
2400bool SILoadStoreOptimizer::promoteConstantOffsetToImm(
2401 MachineInstr &MI,
2402 MemInfoMap &Visited,
2403 SmallPtrSet<MachineInstr *, 4> &AnchorList) const {
2404
2405 if (!STM->hasFlatInstOffsets() || !SIInstrInfo::isFLAT(MI))
2406 return false;
2407
2408 // TODO: Support FLAT_SCRATCH. Currently code expects 64-bit pointers.
2410 return false;
2411
2414
2416 ? AMDGPU::FlatAddrSpace::FlatGlobal
2417 : AMDGPU::FlatAddrSpace::FLAT;
2418 bool AllowNegativeOffset =
2419 TII->allowNegativeFlatOffset(FlatVariant) && !TII->usesASYNC_CNT(MI);
2420 // The async global instructions use i24 offset for global address but u16
2421 // offset for LDS address. In this case, we just only promote when the offset
2422 // is u16.
2423 bool IsOffsetU16 = TII->usesASYNC_CNT(MI);
2424
2425 if (AnchorList.count(&MI))
2426 return false;
2427
2428 LLVM_DEBUG(dbgs() << "\nTryToPromoteConstantOffsetToImmFor "; MI.dump());
2429
2430 if (TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm()) {
2431 LLVM_DEBUG(dbgs() << " Const-offset is already promoted.\n";);
2432 return false;
2433 }
2434
2435 // Step1: Find the base-registers and a 64bit constant offset.
2436 MachineOperand &Base = *TII->getNamedOperand(MI, AMDGPU::OpName::vaddr);
2437 auto [It, Inserted] = Visited.try_emplace(&MI);
2438 MemAddress MAddr;
2439 if (Inserted) {
2440 processBaseWithConstOffset(Base, MAddr);
2441 It->second = MAddr;
2442 } else
2443 MAddr = It->second;
2444
2445 if (MAddr.Offset == 0) {
2446 LLVM_DEBUG(dbgs() << " Failed to extract constant-offset or there are no"
2447 " constant offsets that can be promoted.\n";);
2448 return false;
2449 }
2450
2451 LLVM_DEBUG(dbgs() << " BASE: {" << printReg(MAddr.Base.HiReg, TRI) << ", "
2452 << printReg(MAddr.Base.LoReg, TRI)
2453 << "} Offset: " << MAddr.Offset << "\n\n";);
2454
2455 // Step2: Traverse through MI's basic block and find an anchor(that has the
2456 // same base-registers) with the highest 13bit distance from MI's offset.
2457 // E.g. (64bit loads)
2458 // bb:
2459 // addr1 = &a + 4096; load1 = load(addr1, 0)
2460 // addr2 = &a + 6144; load2 = load(addr2, 0)
2461 // addr3 = &a + 8192; load3 = load(addr3, 0)
2462 // addr4 = &a + 10240; load4 = load(addr4, 0)
2463 // addr5 = &a + 12288; load5 = load(addr5, 0)
2464 //
2465 // Starting from the first load, the optimization will try to find a new base
2466 // from which (&a + 4096) has 13 bit distance. Both &a + 6144 and &a + 8192
2467 // has 13bit distance from &a + 4096. The heuristic considers &a + 8192
2468 // as the new-base(anchor) because of the maximum distance which can
2469 // accommodate more intermediate bases presumably.
2470 //
2471 // Step3: move (&a + 8192) above load1. Compute and promote offsets from
2472 // (&a + 8192) for load1, load2, load4.
2473 // addr = &a + 8192
2474 // load1 = load(addr, -4096)
2475 // load2 = load(addr, -2048)
2476 // load3 = load(addr, 0)
2477 // load4 = load(addr, 2048)
2478 // addr5 = &a + 12288; load5 = load(addr5, 0)
2479 //
2480 MachineInstr *AnchorInst = nullptr;
2481 MemAddress AnchorAddr;
2482 uint32_t MaxDist = std::numeric_limits<uint32_t>::min();
2484 bool MIIsAnchor = false;
2485
2486 MachineBasicBlock *MBB = MI.getParent();
2489 ++MBBI;
2490 const SITargetLowering *TLI = STM->getTargetLowering();
2491
2492 for ( ; MBBI != E; ++MBBI) {
2493 MachineInstr &MINext = *MBBI;
2494 // TODO: Support finding an anchor(with same base) from store addresses or
2495 // any other load addresses where the opcodes are different.
2496 if (MINext.getOpcode() != MI.getOpcode() ||
2497 TII->getNamedOperand(MINext, AMDGPU::OpName::offset)->getImm())
2498 continue;
2499
2500 const MachineOperand &BaseNext =
2501 *TII->getNamedOperand(MINext, AMDGPU::OpName::vaddr);
2502 MemAddress MAddrNext;
2503 auto [It, Inserted] = Visited.try_emplace(&MINext);
2504 if (Inserted) {
2505 processBaseWithConstOffset(BaseNext, MAddrNext);
2506 It->second = MAddrNext;
2507 } else
2508 MAddrNext = It->second;
2509
2510 if (MAddrNext.Base.LoReg != MAddr.Base.LoReg ||
2511 MAddrNext.Base.HiReg != MAddr.Base.HiReg ||
2512 MAddrNext.Base.LoSubReg != MAddr.Base.LoSubReg ||
2513 MAddrNext.Base.HiSubReg != MAddr.Base.HiSubReg)
2514 continue;
2515
2516 InstsWCommonBase.emplace_back(&MINext, MAddrNext.Offset);
2517
2518 if (AllowNegativeOffset) {
2519 int64_t Dist = MAddr.Offset - MAddrNext.Offset;
2520 TargetLoweringBase::AddrMode AM;
2521 AM.HasBaseReg = true;
2522 AM.BaseOffs = Dist;
2523 if (TLI->isLegalFlatAddressingMode(AM, AS) &&
2524 (uint32_t)std::abs(Dist) > MaxDist) {
2525 MaxDist = std::abs(Dist);
2526
2527 AnchorAddr = MAddrNext;
2528 AnchorInst = &MINext;
2529 }
2530 }
2531 }
2532
2533 // When negative offsets are not allowed, pick the candidate with the smallest
2534 // offset as anchor so all promoted offsets are non-negative. If MI itself has
2535 // the smallest offset, MI becomes the reference point (MIIsAnchor).
2536 if (!AllowNegativeOffset && !InstsWCommonBase.empty()) {
2537 for (auto &[Inst, Offset] : InstsWCommonBase) {
2538 int64_t Dist = MAddr.Offset - Offset;
2539 TargetLoweringBase::AddrMode AM;
2540 AM.HasBaseReg = true;
2541 AM.BaseOffs = Dist;
2542 if (Dist >= 0 && TLI->isLegalFlatAddressingMode(AM, AS) &&
2543 (!IsOffsetU16 || isUInt<16>(Dist)) &&
2544 (!AnchorInst || Offset < AnchorAddr.Offset)) {
2545 AnchorAddr = Visited[Inst];
2546 AnchorInst = Inst;
2547 }
2548 }
2549 if (!AnchorInst)
2550 MIIsAnchor = true;
2551 }
2552
2553 if (AnchorInst) {
2554 LLVM_DEBUG(dbgs() << " Anchor-Inst(with max-distance from Offset): ";
2555 AnchorInst->dump());
2556 LLVM_DEBUG(dbgs() << " Anchor-Offset from BASE: "
2557 << AnchorAddr.Offset << "\n\n");
2558
2559 // Instead of moving up, just re-compute anchor-instruction's base address.
2560 Register Base = computeBase(MI, AnchorAddr);
2561
2562 int32_t OffsetDiff = MAddr.Offset - AnchorAddr.Offset;
2563 updateBaseAndOffset(MI, Base, OffsetDiff);
2564 updateAsyncLDSAddress(MI, OffsetDiff);
2565 LLVM_DEBUG(dbgs() << " After promotion: "; MI.dump(););
2566
2567 for (auto [OtherMI, OtherOffset] : InstsWCommonBase) {
2568 TargetLoweringBase::AddrMode AM;
2569 AM.HasBaseReg = true;
2570 AM.BaseOffs = OtherOffset - AnchorAddr.Offset;
2571
2572 if (TLI->isLegalFlatAddressingMode(AM, AS) &&
2573 (AllowNegativeOffset || AM.BaseOffs >= 0) &&
2574 (!IsOffsetU16 || isUInt<16>(AM.BaseOffs))) {
2575 LLVM_DEBUG(dbgs() << " Promote Offset(" << OtherOffset; dbgs() << ")";
2576 OtherMI->dump());
2577 int32_t OtherOffsetDiff = OtherOffset - AnchorAddr.Offset;
2578 updateBaseAndOffset(*OtherMI, Base, OtherOffsetDiff);
2579 updateAsyncLDSAddress(*OtherMI, OtherOffsetDiff);
2580 LLVM_DEBUG(dbgs() << " After promotion: "; OtherMI->dump());
2581 }
2582 }
2583 AnchorList.insert(AnchorInst);
2584 return true;
2585 }
2586
2587 if (MIIsAnchor) {
2588 LLVM_DEBUG(dbgs() << " MI is anchor (smallest offset); promoting "
2589 "candidates relative to MI's base.\n");
2590
2591 Register Base = TII->getNamedOperand(MI, AMDGPU::OpName::vaddr)->getReg();
2592 bool AnyPromoted = false;
2593
2594 for (auto [OtherMI, OtherOffset] : InstsWCommonBase) {
2595 int64_t Dist = OtherOffset - MAddr.Offset;
2596 TargetLoweringBase::AddrMode AM;
2597 AM.HasBaseReg = true;
2598 AM.BaseOffs = Dist;
2599 if (Dist >= 0 && TLI->isLegalFlatAddressingMode(AM, AS) &&
2600 (!IsOffsetU16 || isUInt<16>(Dist))) {
2601 LLVM_DEBUG(dbgs() << " Promote Offset(" << OtherOffset << ")";
2602 OtherMI->dump());
2603 updateBaseAndOffset(*OtherMI, Base, Dist);
2604 updateAsyncLDSAddress(*OtherMI, Dist);
2605 LLVM_DEBUG(dbgs() << " After promotion: "; OtherMI->dump());
2606 AnyPromoted = true;
2607 }
2608 }
2609
2610 if (AnyPromoted) {
2611 TII->getNamedOperand(MI, AMDGPU::OpName::vaddr)->setIsKill(false);
2612 AnchorList.insert(&MI);
2613 return true;
2614 }
2615 }
2616
2617 return false;
2618}
2619
2620void SILoadStoreOptimizer::addInstToMergeableList(const CombineInfo &CI,
2621 std::list<std::list<CombineInfo> > &MergeableInsts) const {
2622 for (std::list<CombineInfo> &AddrList : MergeableInsts) {
2623 if (AddrList.front().InstClass == CI.InstClass &&
2624 AddrList.front().hasSameBaseAddress(CI)) {
2625 AddrList.emplace_back(CI);
2626 return;
2627 }
2628 }
2629
2630 // Base address not found, so add a new list.
2631 MergeableInsts.emplace_back(1, CI);
2632}
2633
2634std::pair<MachineBasicBlock::iterator, bool>
2635SILoadStoreOptimizer::collectMergeableInsts(
2637 MemInfoMap &Visited, SmallPtrSet<MachineInstr *, 4> &AnchorList,
2638 std::list<std::list<CombineInfo>> &MergeableInsts) const {
2639 bool Modified = false;
2640
2641 // Sort potential mergeable instructions into lists. One list per base address.
2642 unsigned Order = 0;
2643 MachineBasicBlock::iterator BlockI = Begin;
2644 for (; BlockI != End; ++BlockI) {
2645 MachineInstr &MI = *BlockI;
2646
2647 // We run this before checking if an address is mergeable, because it can produce
2648 // better code even if the instructions aren't mergeable.
2649 if (promoteConstantOffsetToImm(MI, Visited, AnchorList))
2650 Modified = true;
2651
2652 // Treat volatile accesses, ordered accesses and unmodeled side effects as
2653 // barriers. We can look after this barrier for separate merges.
2654 if (MI.hasOrderedMemoryRef() || MI.hasUnmodeledSideEffects()) {
2655 LLVM_DEBUG(dbgs() << "Breaking search on barrier: " << MI);
2656
2657 // Search will resume after this instruction in a separate merge list.
2658 ++BlockI;
2659 break;
2660 }
2661
2662 const InstClassEnum InstClass = getInstClass(MI.getOpcode(), *TII);
2663 if (InstClass == UNKNOWN)
2664 continue;
2665
2666 // Do not merge VMEM buffer instructions with "swizzled" bit set.
2667 int Swizzled =
2668 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::swz);
2669 if (Swizzled != -1 && MI.getOperand(Swizzled).getImm())
2670 continue;
2671
2672 if (InstClass == TBUFFER_LOAD || InstClass == TBUFFER_STORE) {
2673 if (!STM->hasRelaxedTBufferOOBMode()) {
2674 LLVM_DEBUG(
2675 dbgs() << "Skip tbuffer combine: relaxed OOB mode not enabled\n");
2676 continue;
2677 }
2678
2679 const MachineOperand *Fmt =
2680 TII->getNamedOperand(MI, AMDGPU::OpName::format);
2681 if (!AMDGPU::getGcnBufferFormatInfo(Fmt->getImm(), *STM)) {
2682 LLVM_DEBUG(dbgs() << "Skip tbuffer with unknown format: " << MI);
2683 continue;
2684 }
2685 } else if (InstClass == MIMG) {
2686 // Do not merge MIMG instructions with tfe or lwe enabled.
2687 // TFE/LWE add a status result that the image merge path does not model.
2688 const auto *TFEOp = TII->getNamedOperand(MI, AMDGPU::OpName::tfe);
2689 if (TFEOp && TFEOp->getImm())
2690 continue;
2691
2692 const auto *LWEOp = TII->getNamedOperand(MI, AMDGPU::OpName::lwe);
2693 if (LWEOp && LWEOp->getImm())
2694 continue;
2695 }
2696
2697 CombineInfo CI;
2698 CI.setMI(MI, *this);
2699 CI.Order = Order++;
2700
2701 if (!CI.hasMergeableAddress(*MRI))
2702 continue;
2703
2704 LLVM_DEBUG(dbgs() << "Mergeable: " << MI);
2705
2706 addInstToMergeableList(CI, MergeableInsts);
2707 }
2708
2709 // At this point we have lists of Mergeable instructions.
2710 //
2711 // Part 2: Sort lists by offset and then for each CombineInfo object in the
2712 // list try to find an instruction that can be merged with I. If an instruction
2713 // is found, it is stored in the Paired field. If no instructions are found, then
2714 // the CombineInfo object is deleted from the list.
2715
2716 for (std::list<std::list<CombineInfo>>::iterator I = MergeableInsts.begin(),
2717 E = MergeableInsts.end(); I != E;) {
2718
2719 std::list<CombineInfo> &MergeList = *I;
2720 if (MergeList.size() <= 1) {
2721 // This means we have found only one instruction with a given address
2722 // that can be merged, and we need at least 2 instructions to do a merge,
2723 // so this list can be discarded.
2724 I = MergeableInsts.erase(I);
2725 continue;
2726 }
2727
2728 // Sort the lists by offsets, this way mergeable instructions will be
2729 // adjacent to each other in the list, which will make it easier to find
2730 // matches.
2731 MergeList.sort(
2732 [] (const CombineInfo &A, const CombineInfo &B) {
2733 return A.Offset < B.Offset;
2734 });
2735 ++I;
2736 }
2737
2738 return {BlockI, Modified};
2739}
2740
2741// Scan through looking for adjacent LDS operations with constant offsets from
2742// the same base register. We rely on the scheduler to do the hard work of
2743// clustering nearby loads, and assume these are all adjacent.
2744bool SILoadStoreOptimizer::optimizeBlock(
2745 std::list<std::list<CombineInfo> > &MergeableInsts) {
2746 bool Modified = false;
2747
2748 for (std::list<std::list<CombineInfo>>::iterator I = MergeableInsts.begin(),
2749 E = MergeableInsts.end(); I != E;) {
2750 std::list<CombineInfo> &MergeList = *I;
2751
2752 bool OptimizeListAgain = false;
2753 if (!optimizeInstsWithSameBaseAddr(MergeList, OptimizeListAgain)) {
2754 // We weren't able to make any changes, so delete the list so we don't
2755 // process the same instructions the next time we try to optimize this
2756 // block.
2757 I = MergeableInsts.erase(I);
2758 continue;
2759 }
2760
2761 Modified = true;
2762
2763 // We made changes, but also determined that there were no more optimization
2764 // opportunities, so we don't need to reprocess the list
2765 if (!OptimizeListAgain) {
2766 I = MergeableInsts.erase(I);
2767 continue;
2768 }
2769 OptimizeAgain = true;
2770 }
2771 return Modified;
2772}
2773
2774bool
2775SILoadStoreOptimizer::optimizeInstsWithSameBaseAddr(
2776 std::list<CombineInfo> &MergeList,
2777 bool &OptimizeListAgain) {
2778 if (MergeList.empty())
2779 return false;
2780
2781 bool Modified = false;
2782
2783 for (auto I = MergeList.begin(), Next = std::next(I); Next != MergeList.end();
2784 Next = std::next(I)) {
2785
2786 auto First = I;
2787 auto Second = Next;
2788
2789 if ((*First).Order > (*Second).Order)
2790 std::swap(First, Second);
2791 CombineInfo &CI = *First;
2792 CombineInfo &Paired = *Second;
2793
2794 CombineInfo *Where = checkAndPrepareMerge(CI, Paired);
2795 if (!Where) {
2796 ++I;
2797 continue;
2798 }
2799
2800 Modified = true;
2801
2802 LLVM_DEBUG(dbgs() << "Merging: " << *CI.I << " with: " << *Paired.I);
2803
2805 switch (CI.InstClass) {
2806 default:
2807 llvm_unreachable("unknown InstClass");
2808 break;
2809 case DS_READ:
2810 NewMI = mergeRead2Pair(CI, Paired, Where->I);
2811 break;
2812 case DS_WRITE:
2813 NewMI = mergeWrite2Pair(CI, Paired, Where->I);
2814 break;
2815 case S_BUFFER_LOAD_IMM:
2816 case S_BUFFER_LOAD_SGPR_IMM:
2817 case S_LOAD_IMM:
2818 NewMI = mergeSMemLoadImmPair(CI, Paired, Where->I);
2819 OptimizeListAgain |= CI.Width + Paired.Width < 8;
2820 break;
2821 case BUFFER_LOAD:
2822 NewMI = mergeBufferLoadPair(CI, Paired, Where->I);
2823 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2824 break;
2825 case BUFFER_STORE:
2826 NewMI = mergeBufferStorePair(CI, Paired, Where->I);
2827 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2828 break;
2829 case MIMG:
2830 NewMI = mergeImagePair(CI, Paired, Where->I);
2831 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2832 break;
2833 case TBUFFER_LOAD:
2834 NewMI = mergeTBufferLoadPair(CI, Paired, Where->I);
2835 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2836 break;
2837 case TBUFFER_STORE:
2838 NewMI = mergeTBufferStorePair(CI, Paired, Where->I);
2839 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2840 break;
2841 case FLAT_LOAD:
2842 case FLAT_LOAD_SADDR:
2843 case GLOBAL_LOAD:
2844 case GLOBAL_LOAD_SADDR:
2845 NewMI = mergeFlatLoadPair(CI, Paired, Where->I);
2846 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2847 break;
2848 case FLAT_STORE:
2849 case FLAT_STORE_SADDR:
2850 case GLOBAL_STORE:
2851 case GLOBAL_STORE_SADDR:
2852 NewMI = mergeFlatStorePair(CI, Paired, Where->I);
2853 OptimizeListAgain |= CI.Width + Paired.Width < 4;
2854 break;
2855 }
2856 CI.setMI(NewMI, *this);
2857 CI.Order = Where->Order;
2858 if (I == Second)
2859 I = Next;
2860
2861 MergeList.erase(Second);
2862 }
2863
2864 return Modified;
2865}
2866
2867bool SILoadStoreOptimizerLegacy::runOnMachineFunction(MachineFunction &MF) {
2868 if (skipFunction(MF.getFunction()))
2869 return false;
2870 return SILoadStoreOptimizer(
2871 &getAnalysis<AAResultsWrapperPass>().getAAResults())
2872 .run(MF);
2873}
2874
2875bool SILoadStoreOptimizer::run(MachineFunction &MF) {
2876 this->MF = &MF;
2877 STM = &MF.getSubtarget<GCNSubtarget>();
2878 if (!STM->loadStoreOptEnabled())
2879 return false;
2880
2881 TII = STM->getInstrInfo();
2882 TRI = &TII->getRegisterInfo();
2883
2884 MRI = &MF.getRegInfo();
2885
2886 LLVM_DEBUG(dbgs() << "Running SILoadStoreOptimizer\n");
2887
2888 bool Modified = false;
2889
2890 // Contains the list of instructions for which constant offsets are being
2891 // promoted to the IMM. This is tracked for an entire block at time.
2892 SmallPtrSet<MachineInstr *, 4> AnchorList;
2893 MemInfoMap Visited;
2894
2895 for (MachineBasicBlock &MBB : MF) {
2896 MachineBasicBlock::iterator SectionEnd;
2897 for (MachineBasicBlock::iterator I = MBB.begin(), E = MBB.end(); I != E;
2898 I = SectionEnd) {
2899 bool CollectModified;
2900 std::list<std::list<CombineInfo>> MergeableInsts;
2901
2902 // First pass: Collect list of all instructions we know how to merge in a
2903 // subset of the block.
2904 std::tie(SectionEnd, CollectModified) =
2905 collectMergeableInsts(I, E, Visited, AnchorList, MergeableInsts);
2906
2907 Modified |= CollectModified;
2908
2909 do {
2910 OptimizeAgain = false;
2911 Modified |= optimizeBlock(MergeableInsts);
2912 } while (OptimizeAgain);
2913 }
2914
2915 Visited.clear();
2916 AnchorList.clear();
2917 }
2918
2919 return Modified;
2920}
2921
2922PreservedAnalyses
2925 MFPropsModifier _(*this, MF);
2926
2927 if (MF.getFunction().hasOptNone())
2928 return PreservedAnalyses::all();
2929
2931 .getManager();
2932 AAResults &AA = FAM.getResult<AAManager>(MF.getFunction());
2933
2934 bool Changed = SILoadStoreOptimizer(&AA).run(MF);
2935 if (!Changed)
2936 return PreservedAnalyses::all();
2937
2940 return PA;
2941}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned uint64_t
INITIALIZE_PASS(AMDGPUImageIntrinsicOptimizer, DEBUG_TYPE, "AMDGPU Image Intrinsic Optimizer", false, false) char AMDGPUImageIntrinsicOptimizer void addInstToMergeableList(IntrinsicInst *II, SmallVector< SmallVector< IntrinsicInst *, 4 > > &MergeableInsts, const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr)
BasicBlock::iterator collectMergeableInsts(BasicBlock::iterator I, BasicBlock::iterator E, SmallVector< SmallVector< IntrinsicInst *, 4 > > &MergeableInsts)
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
#define DEBUG_TYPE
#define op(i)
const HexagonInstrInfo * TII
#define _
static MaybeAlign getAlign(Value *Ptr)
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
static uint32_t mostAlignedValueInRange(uint32_t Lo, uint32_t Hi)
static bool needsConstrainedOpcode(const GCNSubtarget &STM, ArrayRef< MachineMemOperand * > MMOs, unsigned Width)
static void addDefsUsesToList(const MachineInstr &MI, DenseSet< Register > &RegDefs, DenseSet< Register > &RegUses)
static unsigned getBufferFormatWithCompCount(unsigned OldFormat, unsigned ComponentCount, const GCNSubtarget &STI)
static bool optimizeBlock(BasicBlock &BB, bool &ModifiedDT, const TargetTransformInfo &TTI, const DataLayout &DL, bool HasBranchDivergence, DomTreeUpdater *DTU)
#define LLVM_DEBUG(...)
Definition Debug.h:119
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
A debug info location.
Definition DebugLoc.h:126
static LLVM_ABI DebugLoc getMergedLocation(DebugLoc LocA, DebugLoc LocB)
When two instructions are combined into a single instruction we also need to combine the original loc...
Definition DebugLoc.cpp:186
Implements a dense probed hash-table based set.
Definition DenseSet.h:281
bool hasOptNone() const
Do not optimize this function (-O0).
Definition Function.h:686
bool loadStoreOptEnabled() const
const SIInstrInfo * getInstrInfo() const override
bool hasDwordx3LoadStores() const
const SITargetLowering * getTargetLowering() const override
bool hasRelaxedTBufferOOBMode() const
bool ldsRequiresM0Init() const
Return if most LDS instructions have an m0 use that require m0 to be initialized.
bool isXNACKEnabled() const
const HexagonRegisterInfo & getRegisterInfo() const
TypeSize getValue() const
unsigned getOpcode() const
Return the opcode number for this descriptor.
An RAII based helper class to modify MachineFunctionProperties when running pass.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineInstrBuilder & cloneMergedMemRefs(ArrayRef< const MachineInstr * > OtherMIs) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void dump() const
A description of a memory reference used in the backend.
LocationSize getSize() const
Return the size in bytes of the memory reference.
unsigned getAddrSpace() const
const MachinePointerInfo & getPointerInfo() const
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
static MachineOperand CreateImm(int64_t Val)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
static bool isFLATScratch(const MachineInstr &MI)
static bool isVIMAGE(const MachineInstr &MI)
static bool isFLATGlobal(const MachineInstr &MI)
static bool isVSAMPLE(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
LLVM_READONLY MachineOperand * getNamedOperand(MachineInstr &MI, AMDGPU::OpName OperandName) const
Returns the operand named Op.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
bool isLegalFlatAddressingMode(const AddrMode &AM, unsigned AddrSpace) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
reference emplace_back(ArgTypes &&... Args)
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
Definition DenseSet.h:182
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
Definition Attributor.h:165
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset)
Convert ByteOffset to dwords if the subtarget uses dword SMRD immediate offsets.
bool getMTBUFHasSrsrc(unsigned Opc)
int getMTBUFElements(unsigned Opc)
bool getMTBUFHasSoffset(unsigned Opc)
int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements)
int getMUBUFBaseOpcode(unsigned Opc)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
int getMTBUFBaseOpcode(unsigned Opc)
bool getMUBUFHasVAddr(unsigned Opc)
int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool getMUBUFHasSoffset(unsigned Opc)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
bool getMTBUFHasVAddr(unsigned Opc)
int getMUBUFElements(unsigned Opc)
const GcnBufferFormatInfo * getGcnBufferFormatInfo(uint8_t BitsPerComp, uint8_t NumComponents, uint8_t NumFormat, const MCSubtargetInfo &STI)
bool getMUBUFHasSrsrc(unsigned Opc)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
BaseReg
Stack frame base register. Bit 0 of FREInfo.Info.
Definition SFrame.h:77
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
bool operator<(int64_t V1, const APSInt &V2)
Definition APSInt.h:360
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
RegState
Flags to represent properties of register accesses.
constexpr T maskLeadingOnes(unsigned N)
Create a bitmask with the N left-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:89
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
char & SILoadStoreOptimizerLegacyID
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
@ Other
Any other memory.
Definition ModRef.h:68
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
DWARFExpression::Operation Op
std::vector< std::pair< LineLocation, FunctionId > > AnchorList
constexpr unsigned BitWidth
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880