LLVM 24.0.0git
SIPeepholeSDWA.cpp
Go to the documentation of this file.
1//===- SIPeepholeSDWA.cpp - Peephole optimization for SDWA instructions ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file This pass tries to apply several peephole SDWA patterns.
10///
11/// E.g. original:
12/// V_LSHRREV_B32_e32 %0, 16, %1
13/// V_ADD_CO_U32_e32 %2, %0, %3
14/// V_LSHLREV_B32_e32 %4, 16, %2
15///
16/// Replace:
17/// V_ADD_CO_U32_sdwa %4, %1, %3
18/// dst_sel:WORD_1 dst_unused:UNUSED_PAD src0_sel:WORD_1 src1_sel:DWORD
19///
20//===----------------------------------------------------------------------===//
21
22#include "SIPeepholeSDWA.h"
23#include "AMDGPU.h"
24#include "GCNSubtarget.h"
25#include "llvm/ADT/Statistic.h"
27#include <optional>
28
29using namespace llvm;
30
31#define DEBUG_TYPE "si-peephole-sdwa"
32
33STATISTIC(NumSDWAPatternsFound, "Number of SDWA patterns found.");
34STATISTIC(NumSDWAInstructionsPeepholed,
35 "Number of instruction converted to SDWA.");
36
37namespace {
38
39bool isConvertibleToSDWA(MachineInstr &MI, const GCNSubtarget &ST,
40 const SIInstrInfo *TII);
41class SDWAOperand;
42class SDWADstOperand;
43
44using SDWAOperandsVector = SmallVector<SDWAOperand *, 4>;
46
47class SIPeepholeSDWA {
48private:
50 const SIRegisterInfo *TRI;
51 const SIInstrInfo *TII;
52
54 SDWAOperandsMap PotentialMatches;
55 SmallVector<MachineInstr *, 8> ConvertedInstructions;
56
57 std::optional<int64_t> foldToImm(const MachineOperand &Op) const;
58
59 // If MI is a v_and_b32 with a 0xffff or 0xff immediate, return the masked
60 // value operand and the matching SDWA selector (WORD_0 / BYTE_0).
61 std::optional<std::pair<MachineOperand *, AMDGPU::SDWA::SdwaSel>>
62 matchAndMask(MachineInstr &MI) const;
63
64 // VOPC SDWA instructions carry the SDWA TSFlag but have no dst_sel operand.
65 bool isSDWAWithDstSel(const MachineInstr &Inst) const;
66
67 void matchSDWAOperands(MachineBasicBlock &MBB);
68 std::unique_ptr<SDWAOperand> matchSDWAOperand(MachineInstr &MI);
69 void pseudoOpConvertToVOP2(MachineInstr &MI,
70 const GCNSubtarget &ST) const;
71 void convertVcndmaskToVOP2(MachineInstr &MI, const GCNSubtarget &ST) const;
72 MachineInstr *createSDWAVersion(MachineInstr &MI);
73 bool convertToSDWA(MachineInstr &MI, const SDWAOperandsVector &SDWAOperands);
74 void legalizeScalarOperands(MachineInstr &MI, const GCNSubtarget &ST) const;
75 bool splitLshlOrForSDWA(MachineBasicBlock &MBB);
76
77public:
78 bool run(MachineFunction &MF);
79};
80
81class SIPeepholeSDWALegacy : public MachineFunctionPass {
82public:
83 static char ID;
84
85 SIPeepholeSDWALegacy() : MachineFunctionPass(ID) {}
86
87 StringRef getPassName() const override { return "SI Peephole SDWA"; }
88
89 bool runOnMachineFunction(MachineFunction &MF) override;
90
91 void getAnalysisUsage(AnalysisUsage &AU) const override {
92 AU.setPreservesCFG();
94 }
95};
96
97using namespace AMDGPU::SDWA;
98
99class SDWAOperand {
100private:
101 MachineOperand *Target; // Operand that would be used in converted instruction
102 MachineOperand *Replaced; // Operand that would be replace by Target
103
104 /// Returns true iff the SDWA selection of this SDWAOperand can be combined
105 /// with the SDWA selections of its uses in \p MI.
106 virtual bool canCombineSelections(const MachineInstr &MI,
107 const SIInstrInfo *TII) = 0;
108
109public:
110 SDWAOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp)
111 : Target(TargetOp), Replaced(ReplacedOp) {
112 assert(Target->isReg());
113 assert(Replaced->isReg());
114 }
115
116 virtual ~SDWAOperand() = default;
117
118 virtual MachineInstr *potentialToConvert(const SIInstrInfo *TII,
119 const GCNSubtarget &ST,
120 SDWAOperandsMap *PotentialMatches = nullptr) = 0;
121 virtual bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) = 0;
122
123 MachineOperand *getTargetOperand() const { return Target; }
124 MachineOperand *getReplacedOperand() const { return Replaced; }
125 MachineInstr *getParentInst() const { return Target->getParent(); }
126
127 MachineRegisterInfo *getMRI() const {
128 return &getParentInst()->getMF()->getRegInfo();
129 }
130
131#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
132 virtual void print(raw_ostream& OS) const = 0;
133 void dump() const { print(dbgs()); }
134#endif
135};
136
137class SDWASrcOperand : public SDWAOperand {
138private:
139 SdwaSel SrcSel;
140 bool Abs;
141 bool Neg;
142 bool Sext;
143
144public:
145 SDWASrcOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
146 SdwaSel SrcSel_ = DWORD, bool Abs_ = false, bool Neg_ = false,
147 bool Sext_ = false)
148 : SDWAOperand(TargetOp, ReplacedOp), SrcSel(SrcSel_), Abs(Abs_),
149 Neg(Neg_), Sext(Sext_) {}
150
151 MachineInstr *potentialToConvert(const SIInstrInfo *TII,
152 const GCNSubtarget &ST,
153 SDWAOperandsMap *PotentialMatches = nullptr) override;
154 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
155 bool canCombineSelections(const MachineInstr &MI,
156 const SIInstrInfo *TII) override;
157
158 SdwaSel getSrcSel() const { return SrcSel; }
159 bool getAbs() const { return Abs; }
160 bool getNeg() const { return Neg; }
161 bool getSext() const { return Sext; }
162
163 uint64_t getSrcMods(const SIInstrInfo *TII,
164 const MachineOperand *SrcOp) const;
165
166#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
167 void print(raw_ostream& OS) const override;
168#endif
169};
170
171class SDWADstOperand : public SDWAOperand {
172private:
173 SdwaSel DstSel;
174 DstUnused DstUn;
175
176public:
177 SDWADstOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
178 SdwaSel DstSel_ = DWORD, DstUnused DstUn_ = UNUSED_PAD)
179 : SDWAOperand(TargetOp, ReplacedOp), DstSel(DstSel_), DstUn(DstUn_) {}
180
181 MachineInstr *potentialToConvert(const SIInstrInfo *TII,
182 const GCNSubtarget &ST,
183 SDWAOperandsMap *PotentialMatches = nullptr) override;
184 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
185 bool canCombineSelections(const MachineInstr &MI,
186 const SIInstrInfo *TII) override;
187
188 SdwaSel getDstSel() const { return DstSel; }
189 DstUnused getDstUnused() const { return DstUn; }
190
191#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
192 void print(raw_ostream& OS) const override;
193#endif
194};
195
196class SDWADstPreserveOperand : public SDWADstOperand {
197private:
198 MachineOperand *Preserve;
199
200public:
201 SDWADstPreserveOperand(MachineOperand *TargetOp, MachineOperand *ReplacedOp,
202 MachineOperand *PreserveOp, SdwaSel DstSel_ = DWORD)
203 : SDWADstOperand(TargetOp, ReplacedOp, DstSel_, UNUSED_PRESERVE),
204 Preserve(PreserveOp) {}
205
206 bool convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) override;
207 bool canCombineSelections(const MachineInstr &MI,
208 const SIInstrInfo *TII) override;
209
210 MachineOperand *getPreservedOperand() const { return Preserve; }
211
212#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
213 void print(raw_ostream& OS) const override;
214#endif
215};
216
217} // end anonymous namespace
218
219INITIALIZE_PASS(SIPeepholeSDWALegacy, DEBUG_TYPE, "SI Peephole SDWA", false,
220 false)
221
222char SIPeepholeSDWALegacy::ID = 0;
223
224char &llvm::SIPeepholeSDWALegacyID = SIPeepholeSDWALegacy::ID;
225
226#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
228 switch(Sel) {
229 case BYTE_0: OS << "BYTE_0"; break;
230 case BYTE_1: OS << "BYTE_1"; break;
231 case BYTE_2: OS << "BYTE_2"; break;
232 case BYTE_3: OS << "BYTE_3"; break;
233 case WORD_0: OS << "WORD_0"; break;
234 case WORD_1: OS << "WORD_1"; break;
235 case DWORD: OS << "DWORD"; break;
236 }
237 return OS;
238}
239
241 switch(Un) {
242 case UNUSED_PAD: OS << "UNUSED_PAD"; break;
243 case UNUSED_SEXT: OS << "UNUSED_SEXT"; break;
244 case UNUSED_PRESERVE: OS << "UNUSED_PRESERVE"; break;
245 }
246 return OS;
247}
248
250void SDWASrcOperand::print(raw_ostream& OS) const {
251 OS << "SDWA src: " << *getTargetOperand()
252 << " src_sel:" << getSrcSel()
253 << " abs:" << getAbs() << " neg:" << getNeg()
254 << " sext:" << getSext() << '\n';
255}
256
258void SDWADstOperand::print(raw_ostream& OS) const {
259 OS << "SDWA dst: " << *getTargetOperand()
260 << " dst_sel:" << getDstSel()
261 << " dst_unused:" << getDstUnused() << '\n';
262}
263
265void SDWADstPreserveOperand::print(raw_ostream& OS) const {
266 OS << "SDWA preserve dst: " << *getTargetOperand()
267 << " dst_sel:" << getDstSel()
268 << " preserve:" << *getPreservedOperand() << '\n';
269}
270
271#endif
272
273static void copyRegOperand(MachineOperand &To, const MachineOperand &From) {
274 assert(To.isReg() && From.isReg());
275 To.setReg(From.getReg());
276 To.setSubReg(From.getSubReg());
277 To.setIsUndef(From.isUndef());
278 if (To.isUse()) {
279 To.setIsKill(From.isKill());
280 } else {
281 To.setIsDead(From.isDead());
282 }
283}
284
285static bool isSameReg(const MachineOperand &LHS, const MachineOperand &RHS) {
286 return LHS.isReg() &&
287 RHS.isReg() &&
288 LHS.getReg() == RHS.getReg() &&
289 LHS.getSubReg() == RHS.getSubReg();
290}
291
293 const MachineRegisterInfo *MRI) {
294 if (!Reg->isReg() || !Reg->isDef())
295 return nullptr;
296
297 return MRI->getOneNonDBGUse(Reg->getReg());
298}
299
301 const MachineRegisterInfo *MRI) {
302 if (!Reg->isReg())
303 return nullptr;
304
305 return MRI->getOneDef(Reg->getReg());
306}
307
308/// Combine an SDWA instruction's existing SDWA selection \p Sel with
309/// the SDWA selection \p OperandSel of its operand. If the selections
310/// are compatible, return the combined selection, otherwise return a
311/// nullopt.
312/// For example, if we have Sel = BYTE_0 Sel and OperandSel = WORD_1:
313/// BYTE_0 Sel (WORD_1 Sel (%X)) -> BYTE_2 Sel (%X)
314static std::optional<SdwaSel> combineSdwaSel(SdwaSel Sel, SdwaSel OperandSel) {
315 if (Sel == SdwaSel::DWORD)
316 return OperandSel;
317
318 if (Sel == OperandSel || OperandSel == SdwaSel::DWORD)
319 return Sel;
320
321 if (Sel == SdwaSel::WORD_1 || Sel == SdwaSel::BYTE_2 ||
322 Sel == SdwaSel::BYTE_3)
323 return {};
324
325 if (OperandSel == SdwaSel::WORD_0)
326 return Sel;
327
328 if (OperandSel == SdwaSel::WORD_1) {
329 if (Sel == SdwaSel::BYTE_0)
330 return SdwaSel::BYTE_2;
331 if (Sel == SdwaSel::BYTE_1)
332 return SdwaSel::BYTE_3;
333 if (Sel == SdwaSel::WORD_0)
334 return SdwaSel::WORD_1;
335 }
336
337 return {};
338}
339
340uint64_t SDWASrcOperand::getSrcMods(const SIInstrInfo *TII,
341 const MachineOperand *SrcOp) const {
342 uint64_t Mods = 0;
343 const auto *MI = SrcOp->getParent();
344 if (TII->getNamedOperand(*MI, AMDGPU::OpName::src0) == SrcOp) {
345 if (auto *Mod = TII->getNamedOperand(*MI, AMDGPU::OpName::src0_modifiers)) {
346 Mods = Mod->getImm();
347 }
348 } else if (TII->getNamedOperand(*MI, AMDGPU::OpName::src1) == SrcOp) {
349 if (auto *Mod = TII->getNamedOperand(*MI, AMDGPU::OpName::src1_modifiers)) {
350 Mods = Mod->getImm();
351 }
352 }
353 if (Abs || Neg) {
354 assert(!Sext &&
355 "Float and integer src modifiers can't be set simultaneously");
356 Mods |= Abs ? SISrcMods::ABS : 0u;
357 Mods ^= Neg ? SISrcMods::NEG : 0u;
358 } else if (Sext) {
359 Mods |= SISrcMods::SEXT;
360 }
361
362 return Mods;
363}
364
365MachineInstr *SDWASrcOperand::potentialToConvert(const SIInstrInfo *TII,
366 const GCNSubtarget &ST,
367 SDWAOperandsMap *PotentialMatches) {
368 if (PotentialMatches != nullptr) {
369 // Fill out the map for all uses if all can be converted
370 MachineOperand *Reg = getReplacedOperand();
371 if (!Reg->isReg() || !Reg->isDef())
372 return nullptr;
373
374 for (MachineInstr &UseMI : getMRI()->use_nodbg_instructions(Reg->getReg()))
375 // Check that all instructions that use Reg can be converted
376 if (!isConvertibleToSDWA(UseMI, ST, TII) ||
377 !canCombineSelections(UseMI, TII))
378 return nullptr;
379
380 // Now that it's guaranteed all uses are legal, iterate over the uses again
381 // to add them for later conversion.
382 for (MachineOperand &UseMO : getMRI()->use_nodbg_operands(Reg->getReg())) {
383 // Should not get a subregister here
384 assert(isSameReg(UseMO, *Reg));
385
386 SDWAOperandsMap &potentialMatchesMap = *PotentialMatches;
387 MachineInstr *UseMI = UseMO.getParent();
388 potentialMatchesMap[UseMI].push_back(this);
389 }
390 return nullptr;
391 }
392
393 // For SDWA src operand potential instruction is one that use register
394 // defined by parent instruction
395 MachineOperand *PotentialMO = findSingleRegUse(getReplacedOperand(), getMRI());
396 if (!PotentialMO)
397 return nullptr;
398
399 MachineInstr *Parent = PotentialMO->getParent();
400
401 return canCombineSelections(*Parent, TII) ? Parent : nullptr;
402}
403
404bool SDWASrcOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) {
405 assert((!Sext || !TII->getSubtarget().zeroesHigh16BitsOfDest(
406 getParentInst()->getOpcode())) &&
407 "Cannot use sign-extension with instruction that zeroes high bits");
408 switch (MI.getOpcode()) {
409 case AMDGPU::V_CVT_F32_FP8_sdwa:
410 case AMDGPU::V_CVT_F32_BF8_sdwa:
411 case AMDGPU::V_CVT_PK_F32_FP8_sdwa:
412 case AMDGPU::V_CVT_PK_F32_BF8_sdwa:
413 // Does not support input modifiers: noabs, noneg, nosext.
414 return false;
415 case AMDGPU::V_CNDMASK_B32_sdwa:
416 // SISrcMods uses the same bitmask for SEXT and NEG modifiers and
417 // hence the compiler can only support one type of modifier for
418 // each SDWA instruction. For V_CNDMASK_B32_sdwa, this is NEG
419 // since its operands get printed using
420 // AMDGPUInstPrinter::printOperandAndFPInputMods which produces
421 // the output intended for NEG if SEXT is set.
422 //
423 // The ISA does actually support both modifiers on most SDWA
424 // instructions.
425 //
426 // FIXME Accept SEXT here after fixing this issue.
427 if (Sext)
428 return false;
429 break;
430 }
431
432 // Find operand in instruction that matches source operand and replace it with
433 // target operand. Set corresponding src_sel
434 bool IsPreserveSrc = false;
435 MachineOperand *Src = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
436 MachineOperand *SrcSel = TII->getNamedOperand(MI, AMDGPU::OpName::src0_sel);
437 MachineOperand *SrcMods =
438 TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers);
439 assert(Src && (Src->isReg() || Src->isImm()));
440 if (!isSameReg(*Src, *getReplacedOperand())) {
441 // If this is not src0 then it could be src1
442 Src = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
443 SrcSel = TII->getNamedOperand(MI, AMDGPU::OpName::src1_sel);
444 SrcMods = TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers);
445
446 if (!Src ||
447 !isSameReg(*Src, *getReplacedOperand())) {
448 // It's possible this Src is a tied operand for
449 // UNUSED_PRESERVE, in which case we can either
450 // abandon the peephole attempt, or if legal we can
451 // copy the target operand into the tied slot
452 // if the preserve operation will effectively cause the same
453 // result by overwriting the rest of the dst.
454 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
455 MachineOperand *DstUnused =
456 TII->getNamedOperand(MI, AMDGPU::OpName::dst_unused);
457
458 if (Dst &&
459 DstUnused->getImm() == AMDGPU::SDWA::DstUnused::UNUSED_PRESERVE) {
460 // This will work if the tied src is accessing WORD_0, and the dst is
461 // writing WORD_1. Modifiers don't matter because all the bits that
462 // would be impacted are being overwritten by the dst.
463 // Any other case will not work.
464 SdwaSel DstSel = static_cast<SdwaSel>(
465 TII->getNamedImmOperand(MI, AMDGPU::OpName::dst_sel));
466 if (DstSel == AMDGPU::SDWA::SdwaSel::WORD_1 &&
467 getSrcSel() == AMDGPU::SDWA::SdwaSel::WORD_0) {
468 IsPreserveSrc = true;
469 auto DstIdx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
470 AMDGPU::OpName::vdst);
471 auto TiedIdx = MI.findTiedOperandIdx(DstIdx);
472 Src = &MI.getOperand(TiedIdx);
473 SrcSel = nullptr;
474 SrcMods = nullptr;
475 } else {
476 // Not legal to convert this src
477 return false;
478 }
479 }
480 }
481 assert(Src && Src->isReg());
482
483 if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
484 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
485 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
486 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
487 !isSameReg(*Src, *getReplacedOperand())) {
488 // In case of v_mac_f16/32_sdwa this pass can try to apply src operand to
489 // src2. This is not allowed.
490 return false;
491 }
492
493 assert(isSameReg(*Src, *getReplacedOperand()) &&
494 (IsPreserveSrc || (SrcSel && SrcMods)));
495 }
496 copyRegOperand(*Src, *getTargetOperand());
497 if (!IsPreserveSrc) {
498 SdwaSel ExistingSel = static_cast<SdwaSel>(SrcSel->getImm());
499 SrcSel->setImm(*combineSdwaSel(ExistingSel, getSrcSel()));
500 SrcMods->setImm(getSrcMods(TII, Src));
501 }
502 getTargetOperand()->setIsKill(false);
503 return true;
504}
505
506/// Verify that the SDWA selection operand \p SrcSelOpName of the SDWA
507/// instruction \p MI can be combined with the selection \p OpSel.
508static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII,
509 AMDGPU::OpName SrcSelOpName, SdwaSel OpSel) {
510 assert(TII->isSDWA(MI.getOpcode()));
511
512 const MachineOperand *SrcSelOp = TII->getNamedOperand(MI, SrcSelOpName);
513 SdwaSel SrcSel = static_cast<SdwaSel>(SrcSelOp->getImm());
514
515 return combineSdwaSel(SrcSel, OpSel).has_value();
516}
517
518/// Verify that \p Op is the same register as the operand of the SDWA
519/// instruction \p MI named by \p SrcOpName and that the SDWA
520/// selection \p SrcSelOpName can be combined with the \p OpSel.
521static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII,
522 AMDGPU::OpName SrcOpName,
523 AMDGPU::OpName SrcSelOpName, MachineOperand *Op,
524 SdwaSel OpSel) {
525 assert(TII->isSDWA(MI.getOpcode()));
526
527 const MachineOperand *Src = TII->getNamedOperand(MI, SrcOpName);
528 if (!Src || !isSameReg(*Src, *Op))
529 return true;
530
531 return canCombineOpSel(MI, TII, SrcSelOpName, OpSel);
532}
533
534bool SDWASrcOperand::canCombineSelections(const MachineInstr &MI,
535 const SIInstrInfo *TII) {
536 if (!TII->isSDWA(MI.getOpcode()))
537 return true;
538
539 using namespace AMDGPU;
540
541 return canCombineOpSel(MI, TII, OpName::src0, OpName::src0_sel,
542 getReplacedOperand(), getSrcSel()) &&
543 canCombineOpSel(MI, TII, OpName::src1, OpName::src1_sel,
544 getReplacedOperand(), getSrcSel());
545}
546
547MachineInstr *SDWADstOperand::potentialToConvert(const SIInstrInfo *TII,
548 const GCNSubtarget &ST,
549 SDWAOperandsMap *PotentialMatches) {
550 // For SDWA dst operand potential instruction is one that defines register
551 // that this operand uses
552 MachineRegisterInfo *MRI = getMRI();
553 MachineInstr *ParentMI = getParentInst();
554
555 MachineOperand *PotentialMO = findSingleRegDef(getReplacedOperand(), MRI);
556 if (!PotentialMO)
557 return nullptr;
558
559 // Check that ParentMI is the only instruction that uses replaced register
560 for (MachineInstr &UseInst : MRI->use_nodbg_instructions(PotentialMO->getReg())) {
561 if (&UseInst != ParentMI)
562 return nullptr;
563 }
564
565 MachineInstr *Parent = PotentialMO->getParent();
566 return canCombineSelections(*Parent, TII) ? Parent : nullptr;
567}
568
569bool SDWADstOperand::convertToSDWA(MachineInstr &MI, const SIInstrInfo *TII) {
570 // Replace vdst operand in MI with target operand. Set dst_sel and dst_unused
571
572 if ((MI.getOpcode() == AMDGPU::V_FMAC_F16_sdwa ||
573 MI.getOpcode() == AMDGPU::V_FMAC_F32_sdwa ||
574 MI.getOpcode() == AMDGPU::V_MAC_F16_sdwa ||
575 MI.getOpcode() == AMDGPU::V_MAC_F32_sdwa) &&
576 getDstSel() != AMDGPU::SDWA::DWORD) {
577 // v_mac_f16/32_sdwa allow dst_sel to be equal only to DWORD
578 return false;
579 }
580
581 MachineOperand *Operand = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
582 assert(Operand &&
583 Operand->isReg() &&
584 isSameReg(*Operand, *getReplacedOperand()));
585 copyRegOperand(*Operand, *getTargetOperand());
586 MachineOperand *DstSel= TII->getNamedOperand(MI, AMDGPU::OpName::dst_sel);
587 assert(DstSel);
588
589 SdwaSel ExistingSel = static_cast<SdwaSel>(DstSel->getImm());
590 DstSel->setImm(combineSdwaSel(ExistingSel, getDstSel()).value());
591
592 MachineOperand *DstUnused= TII->getNamedOperand(MI, AMDGPU::OpName::dst_unused);
594 DstUnused->setImm(getDstUnused());
595
596 // Remove original instruction because it would conflict with our new
597 // instruction by register definition
598 getParentInst()->eraseFromParent();
599 return true;
600}
601
602bool SDWADstOperand::canCombineSelections(const MachineInstr &MI,
603 const SIInstrInfo *TII) {
604 if (!TII->isSDWA(MI.getOpcode()))
605 return true;
606
607 return canCombineOpSel(MI, TII, AMDGPU::OpName::dst_sel, getDstSel());
608}
609
610bool SDWADstPreserveOperand::convertToSDWA(MachineInstr &MI,
611 const SIInstrInfo *TII) {
612 // MI should be moved right before v_or_b32.
613 // For this we should clear all kill flags on uses of MI src-operands or else
614 // we can encounter problem with use of killed operand.
615 for (MachineOperand &MO : MI.uses()) {
616 if (!MO.isReg())
617 continue;
618 getMRI()->clearKillFlags(MO.getReg());
619 }
620
621 // Move MI before v_or_b32
622 MI.getParent()->remove(&MI);
623 getParentInst()->getParent()->insert(getParentInst(), &MI);
624
625 // Add Implicit use of preserved register
626 MachineInstrBuilder MIB(*MI.getMF(), MI);
627 MIB.addReg(getPreservedOperand()->getReg(),
628 RegState::ImplicitKill,
629 getPreservedOperand()->getSubReg());
630
631 // Tie dst to implicit use
632 MI.tieOperands(AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdst),
633 MI.getNumOperands() - 1);
634
635 // Convert MI as any other SDWADstOperand and remove v_or_b32
636 return SDWADstOperand::convertToSDWA(MI, TII);
637}
638
639bool SDWADstPreserveOperand::canCombineSelections(const MachineInstr &MI,
640 const SIInstrInfo *TII) {
641 return SDWADstOperand::canCombineSelections(MI, TII);
642}
643
644std::optional<int64_t>
645SIPeepholeSDWA::foldToImm(const MachineOperand &Op) const {
646 if (Op.isImm()) {
647 return Op.getImm();
648 }
649
650 // If this is not immediate then it can be copy of immediate value, e.g.:
651 // %1 = S_MOV_B32 255;
652 if (Op.isReg()) {
653 for (const MachineOperand &Def : MRI->def_operands(Op.getReg())) {
654 if (!isSameReg(Op, Def))
655 continue;
656
657 const MachineInstr *DefInst = Def.getParent();
658 if (!TII->isFoldableCopy(*DefInst))
659 return std::nullopt;
660
661 const MachineOperand &Copied = DefInst->getOperand(1);
662 if (!Copied.isImm())
663 return std::nullopt;
664
665 return Copied.getImm();
666 }
667 }
668
669 return std::nullopt;
670}
671
672std::optional<std::pair<MachineOperand *, SdwaSel>>
673SIPeepholeSDWA::matchAndMask(MachineInstr &MI) const {
674 if (MI.getOpcode() != AMDGPU::V_AND_B32_e32 &&
675 MI.getOpcode() != AMDGPU::V_AND_B32_e64)
676 return std::nullopt;
677
678 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
679 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
680 MachineOperand *ValSrc = Src1;
681 std::optional<int64_t> Imm = foldToImm(*Src0);
682 if (!Imm) {
683 Imm = foldToImm(*Src1);
684 ValSrc = Src0;
685 }
686 if (!Imm || (*Imm != 0x0000ffff && *Imm != 0x000000ff))
687 return std::nullopt;
688
689 return std::make_pair(ValSrc, *Imm == 0x0000ffff ? WORD_0 : BYTE_0);
690}
691
692bool SIPeepholeSDWA::isSDWAWithDstSel(const MachineInstr &Inst) const {
693 return TII->isSDWA(Inst) &&
694 AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::dst_sel);
695}
696
697std::unique_ptr<SDWAOperand>
698SIPeepholeSDWA::matchSDWAOperand(MachineInstr &MI) {
699 unsigned Opcode = MI.getOpcode();
700 switch (Opcode) {
701 case AMDGPU::V_LSHRREV_B32_e32:
702 case AMDGPU::V_ASHRREV_I32_e32:
703 case AMDGPU::V_LSHLREV_B32_e32:
704 case AMDGPU::V_LSHRREV_B32_e64:
705 case AMDGPU::V_ASHRREV_I32_e64:
706 case AMDGPU::V_LSHLREV_B32_e64: {
707 // from: v_lshrrev_b32_e32 v1, 16/24, v0
708 // to SDWA src:v0 src_sel:WORD_1/BYTE_3
709
710 // from: v_ashrrev_i32_e32 v1, 16/24, v0
711 // to SDWA src:v0 src_sel:WORD_1/BYTE_3 sext:1
712
713 // from: v_lshlrev_b32_e32 v1, 16/24, v0
714 // to SDWA dst:v1 dst_sel:WORD_1/BYTE_3 dst_unused:UNUSED_PAD
715 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
716 auto Imm = foldToImm(*Src0);
717 if (!Imm)
718 break;
719
720 if (*Imm != 16 && *Imm != 24)
721 break;
722
723 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
724 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
725 if (!Src1->isReg() || Src1->getReg().isPhysical() ||
726 Dst->getReg().isPhysical())
727 break;
728
729 if (Opcode == AMDGPU::V_LSHLREV_B32_e32 ||
730 Opcode == AMDGPU::V_LSHLREV_B32_e64) {
731 return std::make_unique<SDWADstOperand>(
732 Dst, Src1, *Imm == 16 ? WORD_1 : BYTE_3, UNUSED_PAD);
733 }
734 return std::make_unique<SDWASrcOperand>(
735 Src1, Dst, *Imm == 16 ? WORD_1 : BYTE_3, false, false,
736 Opcode != AMDGPU::V_LSHRREV_B32_e32 &&
737 Opcode != AMDGPU::V_LSHRREV_B32_e64);
738 break;
739 }
740
741 case AMDGPU::V_LSHRREV_B16_e32:
742 case AMDGPU::V_LSHLREV_B16_e32:
743 case AMDGPU::V_LSHRREV_B16_e64:
744 case AMDGPU::V_LSHRREV_B16_opsel_e64:
745 case AMDGPU::V_LSHLREV_B16_opsel_e64:
746 case AMDGPU::V_LSHLREV_B16_e64: {
747 // V_ASHRREV_I16_e32 and V_ASHRREV_I16_e64 are
748 // not included here because they zero-fill the high 16-bits.
749
750 // from: v_lshrrev_b16_e32 v1, 8, v0
751 // to SDWA src:v0 src_sel:BYTE_1
752
753 // from: v_lshlrev_b16_e32 v1, 8, v0
754 // to SDWA dst:v1 dst_sel:BYTE_1 dst_unused:UNUSED_PAD
755 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
756 auto Imm = foldToImm(*Src0);
757 if (!Imm || *Imm != 8)
758 break;
759
760 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
761 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
762
763 if (!Src1->isReg() || Src1->getReg().isPhysical() ||
764 Dst->getReg().isPhysical())
765 break;
766
767 if (Opcode == AMDGPU::V_LSHLREV_B16_e32 ||
768 Opcode == AMDGPU::V_LSHLREV_B16_opsel_e64 ||
769 Opcode == AMDGPU::V_LSHLREV_B16_e64)
770 return std::make_unique<SDWADstOperand>(Dst, Src1, BYTE_1, UNUSED_PAD);
771 return std::make_unique<SDWASrcOperand>(Src1, Dst, BYTE_1, false, false,
772 false);
773 break;
774 }
775
776 case AMDGPU::V_BFE_I32_e64:
777 case AMDGPU::V_BFE_U32_e64: {
778 // e.g.:
779 // from: v_bfe_u32 v1, v0, 8, 8
780 // to SDWA src:v0 src_sel:BYTE_1
781
782 // offset | width | src_sel
783 // ------------------------
784 // 0 | 8 | BYTE_0
785 // 0 | 16 | WORD_0
786 // 0 | 32 | DWORD ?
787 // 8 | 8 | BYTE_1
788 // 16 | 8 | BYTE_2
789 // 16 | 16 | WORD_1
790 // 24 | 8 | BYTE_3
791
792 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
793 auto Offset = foldToImm(*Src1);
794 if (!Offset)
795 break;
796
797 MachineOperand *Src2 = TII->getNamedOperand(MI, AMDGPU::OpName::src2);
798 auto Width = foldToImm(*Src2);
799 if (!Width)
800 break;
801
802 SdwaSel SrcSel = DWORD;
803
804 if (*Offset == 0 && *Width == 8)
805 SrcSel = BYTE_0;
806 else if (*Offset == 0 && *Width == 16)
807 SrcSel = WORD_0;
808 else if (*Offset == 0 && *Width == 32)
809 SrcSel = DWORD;
810 else if (*Offset == 8 && *Width == 8)
811 SrcSel = BYTE_1;
812 else if (*Offset == 16 && *Width == 8)
813 SrcSel = BYTE_2;
814 else if (*Offset == 16 && *Width == 16)
815 SrcSel = WORD_1;
816 else if (*Offset == 24 && *Width == 8)
817 SrcSel = BYTE_3;
818 else
819 break;
820
821 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
822 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
823
824 if (!Src0->isReg() || Src0->getReg().isPhysical() ||
825 Dst->getReg().isPhysical())
826 break;
827
828 return std::make_unique<SDWASrcOperand>(
829 Src0, Dst, SrcSel, false, false, Opcode != AMDGPU::V_BFE_U32_e64);
830 }
831
832 case AMDGPU::V_AND_B32_e32:
833 case AMDGPU::V_AND_B32_e64: {
834 // e.g.:
835 // from: v_and_b32_e32 v1, 0x0000ffff/0x000000ff, v0
836 // to SDWA src:v0 src_sel:WORD_0/BYTE_0
837 auto Mask = matchAndMask(MI);
838 if (!Mask)
839 break;
840 MachineOperand *ValSrc = Mask->first;
841
842 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
843
844 if (!ValSrc->isReg() || ValSrc->getReg().isPhysical() ||
845 Dst->getReg().isPhysical())
846 break;
847
848 return std::make_unique<SDWASrcOperand>(ValSrc, Dst, Mask->second);
849 }
850
851 case AMDGPU::V_OR_B32_e32:
852 case AMDGPU::V_OR_B32_e64: {
853 // Patterns for dst_unused:UNUSED_PRESERVE.
854 // e.g., from:
855 // v_add_f16_sdwa v0, v1, v2 dst_sel:WORD_1 dst_unused:UNUSED_PAD
856 // src1_sel:WORD_1 src2_sel:WORD1
857 // v_add_f16_e32 v3, v1, v2
858 // v_or_b32_e32 v4, v0, v3
859 // to SDWA preserve dst:v4 dst_sel:WORD_1 dst_unused:UNUSED_PRESERVE preserve:v3
860
861 // Check if one of operands of v_or_b32 is SDWA instruction
862 using CheckRetType =
863 std::optional<std::pair<MachineOperand *, MachineOperand *>>;
864 auto CheckOROperandsForSDWA =
865 [&](const MachineOperand *Op1, const MachineOperand *Op2) -> CheckRetType {
866 if (!Op1 || !Op1->isReg() || !Op2 || !Op2->isReg())
867 return CheckRetType(std::nullopt);
868
869 MachineOperand *Op1Def = findSingleRegDef(Op1, MRI);
870 if (!Op1Def)
871 return CheckRetType(std::nullopt);
872
873 MachineInstr *Op1Inst = Op1Def->getParent();
874 if (!isSDWAWithDstSel(*Op1Inst))
875 return CheckRetType(std::nullopt);
876
877 MachineOperand *Op2Def = findSingleRegDef(Op2, MRI);
878 if (!Op2Def)
879 return CheckRetType(std::nullopt);
880
881 return CheckRetType(std::pair(Op1Def, Op2Def));
882 };
883
884 MachineOperand *OrSDWA = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
885 MachineOperand *OrOther = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
886 assert(OrSDWA && OrOther);
887 auto Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
888 if (!Res) {
889 OrSDWA = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
890 OrOther = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
891 assert(OrSDWA && OrOther);
892 Res = CheckOROperandsForSDWA(OrSDWA, OrOther);
893 if (!Res)
894 break;
895 }
896
897 MachineOperand *OrSDWADef = Res->first;
898 MachineOperand *OrOtherDef = Res->second;
899 assert(OrSDWADef && OrOtherDef);
900
901 MachineInstr *SDWAInst = OrSDWADef->getParent();
902 MachineInstr *OtherInst = OrOtherDef->getParent();
903
904 // Check that OtherInstr is actually bitwise compatible with SDWAInst = their
905 // destination patterns don't overlap. Compatible instruction can be either
906 // regular instruction with compatible bitness or SDWA instruction with
907 // correct dst_sel
908 // SDWAInst | OtherInst bitness / OtherInst dst_sel
909 // -----------------------------------------------------
910 // DWORD | no / no
911 // WORD_0 | no / BYTE_2/3, WORD_1
912 // WORD_1 | 8/16-bit instructions / BYTE_0/1, WORD_0
913 // BYTE_0 | no / BYTE_1/2/3, WORD_1
914 // BYTE_1 | 8-bit / BYTE_0/2/3, WORD_1
915 // BYTE_2 | 8/16-bit / BYTE_0/1/3. WORD_0
916 // BYTE_3 | 8/16/24-bit / BYTE_0/1/2, WORD_0
917 // E.g. if SDWAInst is v_add_f16_sdwa dst_sel:WORD_1 then v_add_f16 is OK
918 // but v_add_f32 is not.
919
920 // TODO: add support for non-SDWA instructions as OtherInst.
921 // For now this only works with SDWA instructions. For regular instructions
922 // there is no way to determine if the instruction writes only 8/16/24-bit
923 // out of full register size and all registers are at min 32-bit wide.
924 if (!isSDWAWithDstSel(*OtherInst))
925 break;
926
927 SdwaSel DstSel = static_cast<SdwaSel>(
928 TII->getNamedImmOperand(*SDWAInst, AMDGPU::OpName::dst_sel));
929 SdwaSel OtherDstSel = static_cast<SdwaSel>(
930 TII->getNamedImmOperand(*OtherInst, AMDGPU::OpName::dst_sel));
931
932 bool DstSelAgree = false;
933 switch (DstSel) {
934 case WORD_0: DstSelAgree = ((OtherDstSel == BYTE_2) ||
935 (OtherDstSel == BYTE_3) ||
936 (OtherDstSel == WORD_1));
937 break;
938 case WORD_1: DstSelAgree = ((OtherDstSel == BYTE_0) ||
939 (OtherDstSel == BYTE_1) ||
940 (OtherDstSel == WORD_0));
941 break;
942 case BYTE_0: DstSelAgree = ((OtherDstSel == BYTE_1) ||
943 (OtherDstSel == BYTE_2) ||
944 (OtherDstSel == BYTE_3) ||
945 (OtherDstSel == WORD_1));
946 break;
947 case BYTE_1: DstSelAgree = ((OtherDstSel == BYTE_0) ||
948 (OtherDstSel == BYTE_2) ||
949 (OtherDstSel == BYTE_3) ||
950 (OtherDstSel == WORD_1));
951 break;
952 case BYTE_2: DstSelAgree = ((OtherDstSel == BYTE_0) ||
953 (OtherDstSel == BYTE_1) ||
954 (OtherDstSel == BYTE_3) ||
955 (OtherDstSel == WORD_0));
956 break;
957 case BYTE_3: DstSelAgree = ((OtherDstSel == BYTE_0) ||
958 (OtherDstSel == BYTE_1) ||
959 (OtherDstSel == BYTE_2) ||
960 (OtherDstSel == WORD_0));
961 break;
962 default: DstSelAgree = false;
963 }
964
965 if (!DstSelAgree)
966 break;
967
968 // Also OtherInst dst_unused should be UNUSED_PAD
969 DstUnused OtherDstUnused = static_cast<DstUnused>(
970 TII->getNamedImmOperand(*OtherInst, AMDGPU::OpName::dst_unused));
971 if (OtherDstUnused != DstUnused::UNUSED_PAD)
972 break;
973
974 // Create DstPreserveOperand
975 MachineOperand *OrDst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
976 assert(OrDst && OrDst->isReg());
977
978 return std::make_unique<SDWADstPreserveOperand>(
979 OrDst, OrSDWADef, OrOtherDef, DstSel);
980
981 }
982 }
983
984 return std::unique_ptr<SDWAOperand>(nullptr);
985}
986
987#if !defined(NDEBUG)
988static raw_ostream& operator<<(raw_ostream &OS, const SDWAOperand &Operand) {
989 Operand.print(OS);
990 return OS;
991}
992#endif
993
994void SIPeepholeSDWA::matchSDWAOperands(MachineBasicBlock &MBB) {
995 for (MachineInstr &MI : MBB) {
996 if (auto Operand = matchSDWAOperand(MI)) {
997 LLVM_DEBUG(dbgs() << "Match: " << MI << "To: " << *Operand << '\n');
998 SDWAOperands[&MI] = std::move(Operand);
999 ++NumSDWAPatternsFound;
1000 }
1001 }
1002}
1003
1004// Convert the V_ADD_CO_U32_e64 into V_ADD_CO_U32_e32. This allows
1005// isConvertibleToSDWA to perform its transformation on V_ADD_CO_U32_e32 into
1006// V_ADD_CO_U32_sdwa.
1007//
1008// We are transforming from a VOP3 into a VOP2 form of the instruction.
1009// %19:vgpr_32 = V_AND_B32_e32 255,
1010// killed %16:vgpr_32, implicit $exec
1011// %47:vgpr_32, %49:sreg_64_xexec = V_ADD_CO_U32_e64
1012// %26.sub0:vreg_64, %19:vgpr_32, implicit $exec
1013// %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64
1014// %26.sub1:vreg_64, %54:vgpr_32, killed %49:sreg_64_xexec, implicit $exec
1015//
1016// becomes
1017// %47:vgpr_32 = V_ADD_CO_U32_sdwa
1018// 0, %26.sub0:vreg_64, 0, killed %16:vgpr_32, 0, 6, 0, 6, 0,
1019// implicit-def $vcc, implicit $exec
1020// %48:vgpr_32, dead %50:sreg_64_xexec = V_ADDC_U32_e64
1021// %26.sub1:vreg_64, %54:vgpr_32, killed $vcc, implicit $exec
1022void SIPeepholeSDWA::pseudoOpConvertToVOP2(MachineInstr &MI,
1023 const GCNSubtarget &ST) const {
1024 int Opc = MI.getOpcode();
1025 assert((Opc == AMDGPU::V_ADD_CO_U32_e64 || Opc == AMDGPU::V_SUB_CO_U32_e64) &&
1026 "Currently only handles V_ADD_CO_U32_e64 or V_SUB_CO_U32_e64");
1027
1028 // Can the candidate MI be shrunk?
1029 if (!TII->canShrink(MI, *MRI))
1030 return;
1032 // Find the related ADD instruction.
1033 const MachineOperand *Sdst = TII->getNamedOperand(MI, AMDGPU::OpName::sdst);
1034 if (!Sdst)
1035 return;
1036 MachineOperand *NextOp = findSingleRegUse(Sdst, MRI);
1037 if (!NextOp)
1038 return;
1039 MachineInstr &MISucc = *NextOp->getParent();
1040
1041 // Make sure the carry in/out are subsequently unused.
1042 MachineOperand *CarryIn = TII->getNamedOperand(MISucc, AMDGPU::OpName::src2);
1043 if (!CarryIn)
1044 return;
1045 MachineOperand *CarryOut = TII->getNamedOperand(MISucc, AMDGPU::OpName::sdst);
1046 if (!CarryOut)
1047 return;
1048 if (!MRI->hasOneNonDBGUse(CarryIn->getReg()) ||
1049 !MRI->use_nodbg_empty(CarryOut->getReg()))
1050 return;
1051 // Make sure VCC or its subregs are dead before MI.
1052 MachineBasicBlock &MBB = *MI.getParent();
1053 if (MISucc.getParent() != &MBB)
1054 return; // Loop depends on MI and MISucc in same MBB.
1056 MBB.computeRegisterLiveness(TRI, AMDGPU::VCC, MI, 25);
1057 if (Liveness != MachineBasicBlock::LQR_Dead)
1058 return;
1059 // Check if VCC is referenced in range of (MI,MISucc].
1060 for (auto I = std::next(MI.getIterator()), E = MISucc.getIterator();
1061 I != E; ++I) {
1062 if (I->modifiesRegister(AMDGPU::VCC, TRI))
1063 return;
1064 }
1065
1066 // Replace MI with V_{SUB|ADD}_I32_e32
1067 BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(Opc))
1068 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::vdst))
1069 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::src0))
1070 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::src1))
1071 .setMIFlags(MI.getFlags());
1072
1073 MI.eraseFromParent();
1074
1075 // Since the carry output of MI is now VCC, update its use in MISucc.
1076
1077 MISucc.substituteRegister(CarryIn->getReg(), TRI->getVCC(), 0, *TRI);
1078}
1079
1080/// Try to convert an \p MI in VOP3 which takes an src2 carry-in
1081/// operand into the corresponding VOP2 form which expects the
1082/// argument in VCC. To this end, add an copy from the carry-in to
1083/// VCC. The conversion will only be applied if \p MI can be shrunk
1084/// to VOP2 and if VCC can be proven to be dead before \p MI.
1085void SIPeepholeSDWA::convertVcndmaskToVOP2(MachineInstr &MI,
1086 const GCNSubtarget &ST) const {
1087 assert(MI.getOpcode() == AMDGPU::V_CNDMASK_B32_e64);
1088
1089 LLVM_DEBUG(dbgs() << "Attempting VOP2 conversion: " << MI);
1090 if (!TII->canShrink(MI, *MRI)) {
1091 LLVM_DEBUG(dbgs() << "Cannot shrink instruction\n");
1092 return;
1093 }
1094
1095 const MachineOperand &CarryIn =
1096 *TII->getNamedOperand(MI, AMDGPU::OpName::src2);
1097 Register CarryReg = CarryIn.getReg();
1098 MachineInstr *CarryDef = MRI->getVRegDef(CarryReg);
1099 if (!CarryDef) {
1100 LLVM_DEBUG(dbgs() << "Missing carry-in operand definition\n");
1101 return;
1102 }
1103
1104 // Make sure VCC or its subregs are dead before MI.
1105 MCRegister Vcc = TRI->getVCC();
1106 MachineBasicBlock &MBB = *MI.getParent();
1109 if (Liveness != MachineBasicBlock::LQR_Dead) {
1110 LLVM_DEBUG(dbgs() << "VCC not known to be dead before instruction\n");
1111 return;
1112 }
1113
1114 BuildMI(MBB, MI, MI.getDebugLoc(), TII->get(AMDGPU::COPY), Vcc).add(CarryIn);
1115
1116 auto Converted = BuildMI(MBB, MI, MI.getDebugLoc(),
1117 TII->get(AMDGPU::getVOPe32(MI.getOpcode())))
1118 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::vdst))
1119 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::src0))
1120 .add(*TII->getNamedOperand(MI, AMDGPU::OpName::src1))
1121 .setMIFlags(MI.getFlags());
1122 TII->fixImplicitOperands(*Converted);
1123 LLVM_DEBUG(dbgs() << "Converted to VOP2: " << *Converted);
1124 (void)Converted;
1125 MI.eraseFromParent();
1126}
1127
1128namespace {
1129bool isConvertibleToSDWA(MachineInstr &MI,
1130 const GCNSubtarget &ST,
1131 const SIInstrInfo* TII) {
1132 // Check if this is already an SDWA instruction
1133 unsigned Opc = MI.getOpcode();
1134 if (TII->isSDWA(Opc))
1135 return true;
1136
1137 // Can only be handled after ealier conversion to
1138 // AMDGPU::V_CNDMASK_B32_e32 which is not always possible.
1139 if (Opc == AMDGPU::V_CNDMASK_B32_e64)
1140 return false;
1141
1142 // Check if this instruction has opcode that supports SDWA
1143 if (AMDGPU::getSDWAOp(Opc) == -1)
1145
1146 if (AMDGPU::getSDWAOp(Opc) == -1)
1147 return false;
1148
1149 if (!ST.hasSDWAOmod() && TII->hasModifiersSet(MI, AMDGPU::OpName::omod))
1150 return false;
1151
1152 if (TII->isVOPC(Opc)) {
1153 if (!ST.hasSDWASdst()) {
1154 const MachineOperand *SDst = TII->getNamedOperand(MI, AMDGPU::OpName::sdst);
1155 if (SDst && (SDst->getReg() != AMDGPU::VCC &&
1156 SDst->getReg() != AMDGPU::VCC_LO))
1157 return false;
1158 }
1159
1160 if (!ST.hasSDWAOutModsVOPC() &&
1161 (TII->hasModifiersSet(MI, AMDGPU::OpName::clamp) ||
1162 TII->hasModifiersSet(MI, AMDGPU::OpName::omod)))
1163 return false;
1164
1165 } else if (TII->getNamedOperand(MI, AMDGPU::OpName::sdst) ||
1166 !TII->getNamedOperand(MI, AMDGPU::OpName::vdst)) {
1167 return false;
1168 }
1169
1170 if (!ST.hasSDWAMac() && (Opc == AMDGPU::V_FMAC_F16_e32 ||
1171 Opc == AMDGPU::V_FMAC_F32_e32 ||
1172 Opc == AMDGPU::V_MAC_F16_e32 ||
1173 Opc == AMDGPU::V_MAC_F32_e32))
1174 return false;
1175
1176 // Check if target supports this SDWA opcode
1177 if (TII->pseudoToMCOpcode(Opc) == -1 ||
1178 TII->pseudoToMCOpcode(AMDGPU::getSDWAOp(Opc)) == -1)
1179 return false;
1180
1181 if (MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0)) {
1182 if (!Src0->isReg() && !Src0->isImm())
1183 return false;
1184 }
1185
1186 if (MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1)) {
1187 if (!Src1->isReg() && !Src1->isImm())
1188 return false;
1189 }
1190
1191 return true;
1192}
1193} // namespace
1194
1195MachineInstr *SIPeepholeSDWA::createSDWAVersion(MachineInstr &MI) {
1196 unsigned Opcode = MI.getOpcode();
1197 assert(!TII->isSDWA(Opcode));
1198
1199 int SDWAOpcode = AMDGPU::getSDWAOp(Opcode);
1200 if (SDWAOpcode == -1)
1201 SDWAOpcode = AMDGPU::getSDWAOp(AMDGPU::getVOPe32(Opcode));
1202 assert(SDWAOpcode != -1);
1203
1204 const MCInstrDesc &SDWADesc = TII->get(SDWAOpcode);
1205
1206 // Create SDWA version of instruction MI and initialize its operands
1207 MachineInstrBuilder SDWAInst =
1208 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), SDWADesc)
1209 .setMIFlags(MI.getFlags());
1210
1211 // Copy dst, if it is present in original then should also be present in SDWA
1212 MachineOperand *Dst = TII->getNamedOperand(MI, AMDGPU::OpName::vdst);
1213 if (Dst) {
1214 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::vdst));
1215 SDWAInst.add(*Dst);
1216 } else if ((Dst = TII->getNamedOperand(MI, AMDGPU::OpName::sdst))) {
1217 assert(Dst && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst));
1218 SDWAInst.add(*Dst);
1219 } else {
1220 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::sdst));
1221 SDWAInst.addReg(TRI->getVCC(), RegState::Define);
1222 }
1223
1224 // Copy src0, initialize src0_modifiers. All sdwa instructions has src0 and
1225 // src0_modifiers (except for v_nop_sdwa, but it can't get here)
1226 MachineOperand *Src0 = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
1227 assert(Src0 && AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0) &&
1228 AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_modifiers));
1229 if (auto *Mod = TII->getNamedOperand(MI, AMDGPU::OpName::src0_modifiers))
1230 SDWAInst.addImm(Mod->getImm());
1231 else
1232 SDWAInst.addImm(0);
1233 SDWAInst.add(*Src0);
1234
1235 // Copy src1 if present, initialize src1_modifiers.
1236 MachineOperand *Src1 = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
1237 if (Src1) {
1238 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1) &&
1239 AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_modifiers));
1240 if (auto *Mod = TII->getNamedOperand(MI, AMDGPU::OpName::src1_modifiers))
1241 SDWAInst.addImm(Mod->getImm());
1242 else
1243 SDWAInst.addImm(0);
1244 SDWAInst.add(*Src1);
1245 }
1246
1247 if (SDWAOpcode == AMDGPU::V_FMAC_F16_sdwa ||
1248 SDWAOpcode == AMDGPU::V_FMAC_F32_sdwa ||
1249 SDWAOpcode == AMDGPU::V_MAC_F16_sdwa ||
1250 SDWAOpcode == AMDGPU::V_MAC_F32_sdwa) {
1251 // v_mac_f16/32 has additional src2 operand tied to vdst
1252 MachineOperand *Src2 = TII->getNamedOperand(MI, AMDGPU::OpName::src2);
1253 assert(Src2);
1254 SDWAInst.add(*Src2);
1255 }
1256
1257 // Copy clamp if present, initialize otherwise
1258 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::clamp));
1259 MachineOperand *Clamp = TII->getNamedOperand(MI, AMDGPU::OpName::clamp);
1260 if (Clamp) {
1261 SDWAInst.add(*Clamp);
1262 } else {
1263 SDWAInst.addImm(0);
1264 }
1265
1266 // Copy omod if present, initialize otherwise if needed
1267 if (AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::omod)) {
1268 MachineOperand *OMod = TII->getNamedOperand(MI, AMDGPU::OpName::omod);
1269 if (OMod) {
1270 SDWAInst.add(*OMod);
1271 } else {
1272 SDWAInst.addImm(0);
1273 }
1274 }
1275
1276 // Initialize SDWA specific operands
1277 if (AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::dst_sel))
1278 SDWAInst.addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1279
1280 if (AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::dst_unused))
1281 SDWAInst.addImm(AMDGPU::SDWA::DstUnused::UNUSED_PAD);
1282
1283 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src0_sel));
1284 SDWAInst.addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1285
1286 if (Src1) {
1287 assert(AMDGPU::hasNamedOperand(SDWAOpcode, AMDGPU::OpName::src1_sel));
1288 SDWAInst.addImm(AMDGPU::SDWA::SdwaSel::DWORD);
1289 }
1290
1291 // Check for a preserved register that needs to be copied.
1292 MachineInstr *Ret = SDWAInst.getInstr();
1293 TII->fixImplicitOperands(*Ret);
1294 return Ret;
1295}
1296
1297bool SIPeepholeSDWA::convertToSDWA(MachineInstr &MI,
1298 const SDWAOperandsVector &SDWAOperands) {
1299 LLVM_DEBUG(dbgs() << "Convert instruction:" << MI);
1300
1301 MachineInstr *SDWAInst;
1302 if (TII->isSDWA(MI.getOpcode())) {
1303 // Clone the instruction to allow revoking changes
1304 // made to MI during the processing of the operands
1305 // if the conversion fails.
1306 SDWAInst = MI.getMF()->CloneMachineInstr(&MI);
1307 MI.getParent()->insert(MI.getIterator(), SDWAInst);
1308 } else {
1309 SDWAInst = createSDWAVersion(MI);
1310 }
1311
1312 // Apply all sdwa operand patterns.
1313 bool Converted = false;
1314 for (auto &Operand : SDWAOperands) {
1315 LLVM_DEBUG(dbgs() << *SDWAInst << "\nOperand: " << *Operand);
1316 // There should be no intersection between SDWA operands and potential MIs
1317 // e.g.:
1318 // v_and_b32 v0, 0xff, v1 -> src:v1 sel:BYTE_0
1319 // v_and_b32 v2, 0xff, v0 -> src:v0 sel:BYTE_0
1320 // v_add_u32 v3, v4, v2
1321 //
1322 // In that example it is possible that we would fold 2nd instruction into
1323 // 3rd (v_add_u32_sdwa) and then try to fold 1st instruction into 2nd (that
1324 // was already destroyed). So if SDWAOperand is also a potential MI then do
1325 // not apply it.
1326 if (PotentialMatches.count(Operand->getParentInst()) == 0)
1327 Converted |= Operand->convertToSDWA(*SDWAInst, TII);
1328 }
1329
1330 if (!Converted) {
1331 SDWAInst->eraseFromParent();
1332 return false;
1333 }
1334
1335 ConvertedInstructions.push_back(SDWAInst);
1336 for (MachineOperand &MO : SDWAInst->uses()) {
1337 if (!MO.isReg())
1338 continue;
1339
1340 MRI->clearKillFlags(MO.getReg());
1341 }
1342 LLVM_DEBUG(dbgs() << "\nInto:" << *SDWAInst << '\n');
1343 ++NumSDWAInstructionsPeepholed;
1344
1345 MI.eraseFromParent();
1346 return true;
1347}
1348
1349// If an instruction was converted to SDWA it should not have immediates or SGPR
1350// operands (allowed one SGPR on GFX9). Copy its scalar operands into VGPRs.
1351void SIPeepholeSDWA::legalizeScalarOperands(MachineInstr &MI,
1352 const GCNSubtarget &ST) const {
1353 const MCInstrDesc &Desc = TII->get(MI.getOpcode());
1354 unsigned ConstantBusCount = 0;
1355 for (MachineOperand &Op : MI.explicit_uses()) {
1356 if (Op.isReg()) {
1357 if (TRI->isVGPR(*MRI, Op.getReg()))
1358 continue;
1359
1360 if (ST.hasSDWAScalar() && ConstantBusCount == 0) {
1361 ++ConstantBusCount;
1362 continue;
1363 }
1364 } else if (!Op.isImm())
1365 continue;
1366
1367 unsigned I = Op.getOperandNo();
1368 const TargetRegisterClass *OpRC = TII->getRegClass(Desc, I);
1369 if (!OpRC || !TRI->isVSSuperClass(OpRC))
1370 continue;
1371
1372 Register VGPR = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1373 auto Copy = BuildMI(*MI.getParent(), MI.getIterator(), MI.getDebugLoc(),
1374 TII->get(AMDGPU::V_MOV_B32_e32), VGPR);
1375 if (Op.isImm())
1376 Copy.addImm(Op.getImm());
1377 else if (Op.isReg())
1378 Copy.addReg(Op.getReg(), getKillRegState(Op.isKill()), Op.getSubReg());
1379 Op.ChangeToRegister(VGPR, false);
1380 }
1381}
1382
1383// Re-fold the masked high-half pack (hi << 16) | (z & 0xffff) into a single
1384// v_or_b32_sdwa src1_sel:WORD_0, which ISel's fused v_lshl_or_b32 blocks.
1385bool SIPeepholeSDWA::splitLshlOrForSDWA(MachineBasicBlock &MBB) {
1386 struct Candidate {
1387 MachineInstr *LshlOr;
1388 MachineInstr *AndMI;
1389 MachineOperand *Hi;
1390 MachineOperand *ValSrc;
1391 };
1392 SmallVector<Candidate, 4> Candidates;
1393
1394 for (MachineInstr &MI : MBB) {
1395 if (MI.getOpcode() != AMDGPU::V_LSHL_OR_B32_e64)
1396 continue;
1397
1398 MachineOperand *Shift = TII->getNamedOperand(MI, AMDGPU::OpName::src1);
1399 std::optional<int64_t> ShiftImm = foldToImm(*Shift);
1400 if (!ShiftImm || *ShiftImm != 16)
1401 continue;
1402
1403 MachineOperand *Hi = TII->getNamedOperand(MI, AMDGPU::OpName::src0);
1404 MachineOperand *Src2 = TII->getNamedOperand(MI, AMDGPU::OpName::src2);
1405 // Src2 must be a virtual reg so getVRegDef below is valid.
1406 if (!Hi->isReg() || !Src2->isReg() || !Src2->getReg().isVirtual())
1407 continue;
1408
1409 // The 0xffff mask must come from a single-use v_and so it can be dropped.
1410 if (!MRI->hasOneNonDBGUse(Src2->getReg()))
1411 continue;
1412 MachineInstr *AndMI = MRI->getVRegDef(Src2->getReg());
1413 if (!AndMI)
1414 continue;
1415 std::optional<std::pair<MachineOperand *, SdwaSel>> Mask =
1416 matchAndMask(*AndMI);
1417 if (!Mask || Mask->second != WORD_0)
1418 continue;
1419 MachineOperand *ValSrc = Mask->first;
1420 if (!ValSrc->isReg() || !TRI->isVGPR(*MRI, ValSrc->getReg()))
1421 continue;
1422
1423 Candidates.push_back({&MI, AndMI, Hi, ValSrc});
1424 }
1425
1426 for (const Candidate &C : Candidates) {
1427 MachineOperand *Dst = TII->getNamedOperand(*C.LshlOr, AMDGPU::OpName::vdst);
1428
1429 Register ShiftReg = MRI->createVirtualRegister(&AMDGPU::VGPR_32RegClass);
1430 BuildMI(*C.LshlOr->getParent(), *C.LshlOr, C.LshlOr->getDebugLoc(),
1431 TII->get(AMDGPU::V_LSHLREV_B32_e64), ShiftReg)
1432 .addImm(16)
1433 .add(*C.Hi);
1434
1435 // vdst, src0_mods, src0, src1_mods, src1, clamp, dst_sel, dst_unused,
1436 // src0_sel, src1_sel.
1437 BuildMI(*C.LshlOr->getParent(), *C.LshlOr, C.LshlOr->getDebugLoc(),
1438 TII->get(AMDGPU::V_OR_B32_sdwa))
1439 .add(*Dst)
1440 .addImm(0)
1441 .addReg(ShiftReg)
1442 .addImm(0)
1443 .add(*C.ValSrc)
1444 .addImm(0)
1445 .addImm(DWORD)
1447 .addImm(DWORD)
1448 .addImm(WORD_0);
1449
1450 MRI->clearKillFlags(C.ValSrc->getReg());
1451 C.LshlOr->eraseFromParent();
1452 C.AndMI->eraseFromParent();
1453 }
1454
1455 return !Candidates.empty();
1456}
1457
1458bool SIPeepholeSDWALegacy::runOnMachineFunction(MachineFunction &MF) {
1459 if (skipFunction(MF.getFunction()))
1460 return false;
1461
1462 return SIPeepholeSDWA().run(MF);
1463}
1464
1465bool SIPeepholeSDWA::run(MachineFunction &MF) {
1466 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1467
1468 if (!ST.hasSDWA())
1469 return false;
1470
1471 MRI = &MF.getRegInfo();
1472 TRI = ST.getRegisterInfo();
1473 TII = ST.getInstrInfo();
1474
1475 // Find all SDWA operands in MF.
1476 bool Ret = false;
1477 for (MachineBasicBlock &MBB : MF) {
1478 bool Changed = false;
1479 do {
1480 Ret |= splitLshlOrForSDWA(MBB);
1481
1482 // Preprocess the ADD/SUB pairs so they could be SDWA'ed.
1483 // Look for a possible ADD or SUB that resulted from a previously lowered
1484 // V_{ADD|SUB}_U64_PSEUDO. The function pseudoOpConvertToVOP2
1485 // lowers the pair of instructions into e32 form.
1486 matchSDWAOperands(MBB);
1487 for (const auto &OperandPair : SDWAOperands) {
1488 const auto &Operand = OperandPair.second;
1489 MachineInstr *PotentialMI = Operand->potentialToConvert(TII, ST);
1490 if (!PotentialMI)
1491 continue;
1492
1493 switch (PotentialMI->getOpcode()) {
1494 case AMDGPU::V_ADD_CO_U32_e64:
1495 case AMDGPU::V_SUB_CO_U32_e64:
1496 pseudoOpConvertToVOP2(*PotentialMI, ST);
1497 break;
1498 case AMDGPU::V_CNDMASK_B32_e64:
1499 convertVcndmaskToVOP2(*PotentialMI, ST);
1500 break;
1501 };
1502 }
1503 SDWAOperands.clear();
1504
1505 // Generate potential match list.
1506 matchSDWAOperands(MBB);
1507
1508 for (const auto &OperandPair : SDWAOperands) {
1509 const auto &Operand = OperandPair.second;
1510 MachineInstr *PotentialMI =
1511 Operand->potentialToConvert(TII, ST, &PotentialMatches);
1512
1513 if (PotentialMI && isConvertibleToSDWA(*PotentialMI, ST, TII))
1514 PotentialMatches[PotentialMI].push_back(Operand.get());
1515 }
1516
1517 for (auto &PotentialPair : PotentialMatches) {
1518 MachineInstr &PotentialMI = *PotentialPair.first;
1519 convertToSDWA(PotentialMI, PotentialPair.second);
1520 }
1521
1522 PotentialMatches.clear();
1523 SDWAOperands.clear();
1524
1525 Changed = !ConvertedInstructions.empty();
1526
1527 if (Changed)
1528 Ret = true;
1529 while (!ConvertedInstructions.empty())
1530 legalizeScalarOperands(*ConvertedInstructions.pop_back_val(), ST);
1531 } while (Changed);
1532 }
1533
1534 return Ret;
1535}
1536
MachineInstrBuilder & UseMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock & MBB
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:686
AMD GCN specific subclass of TargetSubtarget.
#define DEBUG_TYPE
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
static MachineOperand * findSingleRegDef(const MachineOperand *Reg, const MachineRegisterInfo *MRI)
static void copyRegOperand(MachineOperand &To, const MachineOperand &From)
static MachineOperand * findSingleRegUse(const MachineOperand *Reg, const MachineRegisterInfo *MRI)
static std::optional< SdwaSel > combineSdwaSel(SdwaSel Sel, SdwaSel OperandSel)
Combine an SDWA instruction's existing SDWA selection Sel with the SDWA selection OperandSel of its o...
static bool isSameReg(const MachineOperand &LHS, const MachineOperand &RHS)
static bool canCombineOpSel(const MachineInstr &MI, const SIInstrInfo *TII, AMDGPU::OpName SrcSelOpName, SdwaSel OpSel)
Verify that the SDWA selection operand SrcSelOpName of the SDWA instruction MI can be combined with t...
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
Value * RHS
Value * LHS
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
bool hasOptNone() const
Do not optimize this function (-O0).
Definition Function.h:686
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
LLVM_ABI void substituteRegister(Register FromReg, Register ToReg, unsigned SubIdx, const TargetRegisterInfo &RegInfo)
Replace all occurrences of FromReg with ToReg:SubIdx, properly composing subreg indices where necessa...
mop_range uses()
Returns all operands which may be register uses.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
void setImm(int64_t immVal)
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
void setIsKill(bool Val=true)
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI MachineOperand * getOneNonDBGUse(Register RegNo) const
If the register has a single non-Debug use, returns it; otherwise returns nullptr.
MachineOperand * getOneDef(Register Reg) const
Returns the defining operand if there is exactly one operand defining the specified register,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
iterator_range< def_iterator > def_operands(Register Reg) const
This class implements a map that also provides access to all stored values in a deterministic order.
Definition MapVector.h:38
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
Changed
@ VGPR
Address space for VGPRs.
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
LLVM_READONLY int32_t getSDWAOp(uint32_t Opcode)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
@ Offset
Definition DWP.cpp:577
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr RegState getKillRegState(bool B)
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
Op::Description Desc
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
char & SIPeepholeSDWALegacyID
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58